akm-cli 0.9.24 → 0.9.25-alpha.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/CHANGELOG.md +173 -0
  2. package/dist/cli.js +1 -1
  3. package/dist/commands/health/checks.js +10 -11
  4. package/dist/commands/improve/consolidate/pair-pass.js +4 -2
  5. package/dist/commands/improve/consolidate.js +10 -4
  6. package/dist/commands/improve/execution.js +4 -11
  7. package/dist/commands/improve/extract-prompt.js +4 -4
  8. package/dist/commands/improve/extract.js +11 -13
  9. package/dist/commands/improve/improve-cli.js +65 -34
  10. package/dist/commands/improve/improve-strategies.js +49 -43
  11. package/dist/commands/improve/improve-usage-report.js +8 -17
  12. package/dist/commands/improve/loop-stages.js +3 -0
  13. package/dist/commands/improve/preparation.js +3 -1
  14. package/dist/commands/improve/reflect-noise.js +125 -0
  15. package/dist/commands/improve/reflect.js +105 -172
  16. package/dist/commands/improve/retrieval-gate.js +7 -2
  17. package/dist/commands/improve/stage.js +67 -24
  18. package/dist/commands/proposal/drain.js +11 -2
  19. package/dist/commands/proposal/proposal-cli.js +1 -5
  20. package/dist/commands/proposal/propose-cli.js +2 -2
  21. package/dist/commands/proposal/propose.js +72 -84
  22. package/dist/commands/proposal/validators/proposal-quality-validators.js +4 -2
  23. package/dist/commands/read/search-cli.js +0 -38
  24. package/dist/commands/remember.js +3 -3
  25. package/dist/commands/sources/schema-repair.js +1 -1
  26. package/dist/core/config/config-schema.js +37 -60
  27. package/dist/core/config/engine-semantics.js +15 -11
  28. package/dist/core/config/schema/improve-processes.js +18 -2
  29. package/dist/core/improve-result.js +3 -3
  30. package/dist/core/redaction.js +4 -0
  31. package/dist/core/spawn-env.js +25 -0
  32. package/dist/core/structured.js +11 -1
  33. package/dist/execution/source.js +10 -0
  34. package/dist/indexer/passes/memory-inference.js +2 -1
  35. package/dist/integrations/agent/builder-shared.js +15 -0
  36. package/dist/integrations/agent/config.js +1 -1
  37. package/dist/integrations/agent/engine-resolution.js +13 -31
  38. package/dist/integrations/agent/execution.js +48 -22
  39. package/dist/integrations/agent/index.js +1 -1
  40. package/dist/integrations/agent/profiles.js +2 -2
  41. package/dist/integrations/agent/prompts.js +55 -114
  42. package/dist/integrations/agent/request-lowering.js +21 -8
  43. package/dist/integrations/agent/runner-dispatch.js +96 -3
  44. package/dist/integrations/agent/runner.js +8 -2
  45. package/dist/integrations/harnesses/aider/agent-builder.js +10 -23
  46. package/dist/integrations/harnesses/aider/index.js +0 -5
  47. package/dist/integrations/harnesses/amazonq/agent-builder.js +9 -21
  48. package/dist/integrations/harnesses/amazonq/index.js +0 -5
  49. package/dist/integrations/harnesses/claude/agent-builder.js +47 -36
  50. package/dist/integrations/harnesses/claude/index.js +0 -14
  51. package/dist/integrations/harnesses/codex/agent-builder.js +3 -4
  52. package/dist/integrations/harnesses/codex/index.js +0 -4
  53. package/dist/integrations/harnesses/copilot/agent-builder.js +4 -13
  54. package/dist/integrations/harnesses/copilot/index.js +2 -7
  55. package/dist/integrations/harnesses/gemini/agent-builder.js +4 -13
  56. package/dist/integrations/harnesses/gemini/index.js +0 -5
  57. package/dist/integrations/harnesses/ids.js +12 -10
  58. package/dist/integrations/harnesses/opencode/agent-builder.js +41 -9
  59. package/dist/integrations/harnesses/opencode/index.js +0 -8
  60. package/dist/integrations/harnesses/opencode/model-config.js +33 -0
  61. package/dist/integrations/harnesses/opencode/model-work-agent.js +115 -0
  62. package/dist/integrations/harnesses/opencode-sdk/harness.js +2 -7
  63. package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +114 -28
  64. package/dist/integrations/harnesses/openhands/agent-builder.js +7 -21
  65. package/dist/integrations/harnesses/openhands/index.js +0 -5
  66. package/dist/integrations/harnesses/pi/agent-builder.js +7 -21
  67. package/dist/integrations/harnesses/pi/index.js +0 -5
  68. package/dist/llm/client.js +5 -0
  69. package/dist/llm/feature-gate.js +12 -9
  70. package/dist/llm/index-passes.js +2 -5
  71. package/dist/llm/memory-infer.js +6 -5
  72. package/dist/llm/structured-call.js +33 -11
  73. package/dist/output/shapes/passthrough.js +1 -0
  74. package/dist/scripts/akm-migrate-node.js +298 -228
  75. package/dist/scripts/akm-migrate.js +298 -228
  76. package/dist/workflows/exec/step-work.js +6 -5
  77. package/dist/workflows/exec/unit-dispatch.js +4 -13
  78. package/dist/workflows/freeze/step-values.js +1 -1
  79. package/docs/reference/cli.md +41 -18
  80. package/docs/reference/configuration.md +165 -12
  81. package/docs/reference/data-and-telemetry.md +2 -3
  82. package/docs/reference/workflow-schema.md +10 -9
  83. package/package.json +1 -1
  84. package/schemas/akm-config.json +108 -0
@@ -11,7 +11,6 @@
11
11
  * pre-dispatch refusals still emit both).
12
12
  */
13
13
  import fs from "node:fs";
14
- import os from "node:os";
15
14
  import path from "node:path";
16
15
  import { assembleAssetFromString, serializeFrontmatter } from "../../core/asset/asset-serialize.js";
17
16
  import { parseFrontmatter } from "../../core/asset/frontmatter.js";
@@ -31,20 +30,20 @@ import { lookup } from "../../indexer/indexer.js";
31
30
  import { DEFAULT_LLM_TIMEOUT_MS } from "../../integrations/agent/config.js";
32
31
  import { fallbackAnnouncement, NO_ENGINE_MESSAGE_SUFFIX, NO_ENGINE_REMEDY, withEngineFallback, } from "../../integrations/agent/engine-fallback.js";
33
32
  import { buildExecution, resolveExecution } from "../../integrations/agent/execution.js";
34
- import { buildReflectOutputRepairPrompt, buildReflectPrompt, extractDraftConfidence, parseAgentProposalPayload, REFLECT_CONTENT_CAP, REFLECT_TRUNCATION_MARKER, } from "../../integrations/agent/prompts.js";
35
- import { runnerIsLlm, runnerSupportsFileWrite } from "../../integrations/agent/runner.js";
36
- import { assertRunnerCredentials, collectDispatchSensitiveValues, runExecution, } from "../../integrations/agent/runner-dispatch.js";
33
+ import { buildReflectOutputRepairPrompt, buildReflectPrompt, parseAgentProposalPayload, REFLECT_CONTENT_CAP, REFLECT_TRUNCATION_MARKER, } from "../../integrations/agent/prompts.js";
34
+ import { runnerIsLlm } from "../../integrations/agent/runner.js";
35
+ import { assertRunnerCredentials, collectDispatchSensitiveValues, } from "../../integrations/agent/runner-dispatch.js";
37
36
  import { isJsonSchemaKnownUnsupported, LlmCallError } from "../../llm/client.js";
38
37
  import { baseFailureFields, enoentHintMessage, isEnoentFailure } from "../agent/agent-support.js";
39
38
  import { checkReflectSize, isValidDescription } from "../proposal/validators/proposal-quality-validators.js";
40
39
  import { CHARS_PER_TOKEN, DEFAULT_CONTEXT_LENGTH_TOKENS } from "./consolidate/chunking.js";
41
40
  import { deriveLessonRef } from "./distill.js";
42
41
  import { findAssetFilePath } from "./eligibility.js";
43
- import { resolveImproveLlmExecution } from "./execution.js";
42
+ import { resolveImproveExecution } from "./execution.js";
44
43
  import { recordLedgerAttempt } from "./ledger.js";
45
- import { classifyReflectChange, splitFrontmatter } from "./reflect-noise.js";
44
+ import { classifyReflectChange, findReflectDefect, splitFrontmatter } from "./reflect-noise.js";
46
45
  import { loadRetrievalQueries, runRetrievalRegressionGate } from "./retrieval-gate.js";
47
- import { callStage, mintProposal, noticeSet, rejectedProposalContext, resolveQualityGateJudge, runReflectQualityJudge, } from "./stage.js";
46
+ import { callStageOnce, errMessage, mintProposal, noticeSet, rejectedProposalContext, resolveQualityGateJudge, runReflectQualityJudge, } from "./stage.js";
48
47
  const MAX_FEEDBACK_LINES = 10;
49
48
  const MAX_GLOBAL_FEEDBACK_LINES = 20;
50
49
  function readOnlyEventsContext(ctx) {
@@ -83,16 +82,6 @@ export const REFLECT_ALLOWED_TYPES = new Set([
83
82
  const REFLECT_REFUSED_TYPES = new Set(["secret"]);
84
83
  /** Identity fields the model may never change (a renamed `name` breaks ref resolution). */
85
84
  const PROTECTED_FRONTMATTER_FIELDS = new Set(["name", "ref", "id", "slug", "type"]);
86
- /**
87
- * A fresh tmp path per iteration for the agent/SDK file-write contract (long
88
- * bodies are written to a file instead of fenced JSON on stdout). The direct
89
- * LLM runner has no filesystem and never gets one.
90
- */
91
- function synthesizeReflectDraftPath(ref) {
92
- const safeRef = (ref ?? "no-ref").replace(/[^a-z0-9_-]/gi, "_");
93
- const rand = Math.random().toString(36).slice(2, 8);
94
- return path.join(os.tmpdir(), `akm-reflect-${safeRef}-${Date.now()}-${rand}.md`);
95
- }
96
85
  /** Lesson lint findings for the prompt: a concrete starting point for the revision. */
97
86
  function buildSchemaHints(type, content) {
98
87
  if (!content || type !== "lesson")
@@ -328,7 +317,12 @@ export function sanitizeReflectPayload(payload, sourceContent, targetRef) {
328
317
  ...(truncationMarkerLeaked ? { truncationMarkerLeaked } : {}),
329
318
  };
330
319
  }
331
- // ── Direct-LLM output contract ───────────────────────────────────────────────
320
+ // ── Output contract ──────────────────────────────────────────────────────────
321
+ //
322
+ // Every engine kind is asked for the same reply: the JSON object of
323
+ // REFLECT_JSON_SCHEMA, or the framed-markdown frame for an LLM engine that
324
+ // rejects JSON Schema. The reply is validated by the parse functions below and
325
+ // repaired once, whatever engine produced it.
332
326
  const REFLECT_FRONTMATTER_PATCH_JSON_SCHEMA = {
333
327
  type: "object",
334
328
  required: ["description", "when_to_use"],
@@ -366,7 +360,7 @@ const REFLECT_UNSCOPED_JSON_SCHEMA = {
366
360
  },
367
361
  };
368
362
  /**
369
- * Frame for JSON Schema unless the connection disabled it or already proved
363
+ * Frame for JSON Schema unless the LLM connection disabled it or already proved
370
364
  * this process that it rejects it (the transport retries plain text on a 4xx).
371
365
  */
372
366
  function wantsJsonSchemaOutput(connection) {
@@ -379,7 +373,7 @@ function parsedRecord(result) {
379
373
  ? result.parsed
380
374
  : undefined;
381
375
  }
382
- function reflectLlmTelemetry(result) {
376
+ function reflectTelemetry(result) {
383
377
  const parsed = parsedRecord(result);
384
378
  if (!parsed || (parsed.outputMode !== "json_schema" && parsed.outputMode !== "framed_markdown"))
385
379
  return undefined;
@@ -484,13 +478,17 @@ function parseFramedReflectOutput(raw, targetRef) {
484
478
  return { ref, content, confidence, ...(frontmatter ? { frontmatter } : {}) };
485
479
  }
486
480
  /**
487
- * One reflect iteration through the direct LLM runner, as an agent-shaped
488
- * result (errors captured, never thrown except configuration). An unparseable
489
- * response gets one repair turn within the original deadline.
481
+ * One reflect iteration on any engine, as an agent-shaped result (errors
482
+ * captured, never thrown except configuration). Every engine kind is sent the
483
+ * same request, with the reply's schema as the output schema, and its reply is
484
+ * held to the same contract: an unparseable reply gets one repair turn within
485
+ * the original deadline, then fails. A failed agent or SDK dispatch is reported
486
+ * as it ran, with its exit code and stderr; an LLM call has only its message.
490
487
  */
491
- export async function runReflectViaLlm(opts) {
488
+ export async function runReflectIteration(opts) {
492
489
  const start = Date.now();
493
490
  let repairAttempts = 0;
491
+ // Model work is bounded: with no timeout from the caller or the runner, the default.
494
492
  const configuredTimeout = Object.hasOwn(opts, "timeoutMs")
495
493
  ? (opts.timeoutMs ?? null)
496
494
  : Object.hasOwn(opts.runner, "timeoutMs")
@@ -505,7 +503,7 @@ export async function runReflectViaLlm(opts) {
505
503
  ? parseSchemaReflectOutput(raw, opts.targetRef)
506
504
  : parseFramedReflectOutput(raw, opts.targetRef);
507
505
  const failure = (err, reason, stdout = "", exitCode = 1) => {
508
- const msg = err instanceof Error ? err.message : String(err);
506
+ const msg = errMessage(err);
509
507
  return {
510
508
  ok: false,
511
509
  stdout,
@@ -517,8 +515,13 @@ export async function runReflectViaLlm(opts) {
517
515
  parsed: { outputMode: opts.outputMode, repairAttempts },
518
516
  };
519
517
  };
518
+ // A reply that broke the contract; its message is the parser's, on every engine kind.
519
+ const invalidReply = (err, reply) => failure(err, "parse_error", reply, 0);
520
+ // The result of the dispatch that failed, kept for an agent or SDK engine.
521
+ let dispatched;
522
+ // Reflect parses and repairs its own reply (the repair turn below), so one dispatch, unvalidated.
520
523
  const call = async (callMessages, repairTimeoutMs) => {
521
- const outcome = await callStage({
524
+ const outcome = await callStageOnce({
522
525
  feature: "reflect_proposal",
523
526
  runner: opts.runner,
524
527
  prompt: callMessages.at(-1)?.content ?? "",
@@ -535,13 +538,18 @@ export async function runReflectViaLlm(opts) {
535
538
  // Visible chain-of-thought can exhaust the output before the envelope.
536
539
  enableThinking: false,
537
540
  ...(opts.chat ? { chat: opts.chat } : {}),
541
+ ...(opts.runSdk ? { runSdk: opts.runSdk } : {}),
542
+ ...(opts.runOptions ? { runOptions: opts.runOptions } : {}),
538
543
  },
544
+ ...(opts.environment ? { current: { environment: opts.environment } } : {}),
539
545
  ...(opts.onNotices ? { onNotices: opts.onNotices } : {}),
540
546
  });
541
547
  if (!outcome.ok) {
548
+ if (!runnerIsLlm(opts.runner))
549
+ dispatched = outcome.result;
542
550
  throw outcome.reason === "timeout"
543
551
  ? new LlmCallError(outcome.error ?? "timeout", "timeout")
544
- : new Error(outcome.error ?? "LLM call failed");
552
+ : new Error(outcome.error ?? "dispatch failed");
545
553
  }
546
554
  return outcome.raw;
547
555
  };
@@ -556,7 +564,7 @@ export async function runReflectViaLlm(opts) {
556
564
  }
557
565
  catch (err) {
558
566
  if (opts.allowRepair === false)
559
- return failure(err, "parse_error", stdout, 0);
567
+ return invalidReply(err, stdout);
560
568
  if (opts.signal?.aborted)
561
569
  return failure(new Error("Reflect request aborted"), "aborted", stdout);
562
570
  const remaining = deadline === undefined ? undefined : deadline - Date.now();
@@ -570,7 +578,7 @@ export async function runReflectViaLlm(opts) {
570
578
  payload = parse(acceptedOutput);
571
579
  }
572
580
  catch (repairErr) {
573
- return failure(repairErr, "parse_error", acceptedOutput, 0);
581
+ return invalidReply(repairErr, acceptedOutput);
574
582
  }
575
583
  }
576
584
  return {
@@ -585,6 +593,8 @@ export async function runReflectViaLlm(opts) {
585
593
  catch (err) {
586
594
  if (err instanceof ConfigError)
587
595
  throw err;
596
+ if (dispatched)
597
+ return { ...dispatched, parsed: { outputMode: opts.outputMode, repairAttempts } };
588
598
  const reason = opts.signal?.aborted
589
599
  ? "aborted"
590
600
  : err instanceof LlmCallError && err.code === "timeout"
@@ -685,8 +695,9 @@ async function resolveReflectSource(options, stash, emitFailed) {
685
695
  }
686
696
  /**
687
697
  * The single engine for this invocation: `--engine`, the improve strategy's
688
- * LLM-only reflect process, or `defaults.engine` (announced when it falls back
689
- * to the SDK binary). Unattended improve refuses a tool-capable engine.
698
+ * reflect process, or `defaults.engine` (announced when it falls back to the
699
+ * SDK binary). Whatever its kind, reflect runs it under the model-work tool
700
+ * policy.
690
701
  */
691
702
  function resolveReflectRunner(options) {
692
703
  const config = options.config ?? loadConfig();
@@ -700,14 +711,14 @@ function resolveReflectRunner(options) {
700
711
  lowered = lower({ content: "reflect engine selection", config, current: { engine: options.engine } });
701
712
  }
702
713
  else if (options.improveProfile) {
703
- const resolved = resolveImproveLlmExecution({
714
+ const resolved = resolveImproveExecution({
704
715
  config,
705
716
  profile: activeStrategy,
706
717
  process: activeStrategy?.processes?.reflect,
707
718
  processName: "reflect",
708
719
  });
709
720
  if (!resolved) {
710
- throw new ConfigError("Reflect requires an LLM engine for the active improve strategy.", "LLM_NOT_CONFIGURED", "Set defaults.llmEngine or improve.strategies.<name>.processes.reflect.engine.");
721
+ throw new ConfigError("Reflect requires an engine for the active improve strategy.", "LLM_NOT_CONFIGURED", "Set defaults.llmEngine or improve.strategies.<name>.processes.reflect.engine.");
711
722
  }
712
723
  lowered = resolved;
713
724
  }
@@ -723,19 +734,20 @@ function resolveReflectRunner(options) {
723
734
  lowered = lower({ content: "reflect engine selection", config });
724
735
  }
725
736
  const runnerSpec = lowered.runner;
726
- if (options.eventSource === "improve" && !runnerIsLlm(runnerSpec)) {
727
- throw new ConfigError(`Unattended improve requires an LLM engine for reflect; engine "${runnerSpec.engine ?? options.engine ?? "unknown"}" is tool-capable.`, "INVALID_CONFIG_FILE", "Set defaults.llmEngine or improve.strategies.<name>.processes.reflect.engine to an LLM engine.");
728
- }
729
737
  const engineName = runnerSpec.engine ?? options.engine;
730
738
  if (!engineName)
731
739
  throw new ConfigError("Reflect requires a named engine.", "INVALID_CONFIG_FILE");
732
740
  return { config, activeStrategy, runnerSpec, engineName, notices: lowered.notices };
733
741
  }
734
- /** Lower a runner and check its credentials, so a bad transport fails before any work. */
742
+ /**
743
+ * Lower a runner under the model-work tool policy and check its credentials,
744
+ * so a bad transport, or one that cannot confine the policy, fails before any work.
745
+ */
735
746
  function preflightReflectDispatch(runnerSpec, onNotices) {
736
747
  const prepared = resolveExecution({
737
748
  content: "Validate reflect operation transport before dispatch.",
738
749
  runner: runnerSpec,
750
+ modelWork: true,
739
751
  });
740
752
  const lowered = buildExecution(prepared.request, prepared.runner);
741
753
  onNotices(lowered.notices);
@@ -767,13 +779,10 @@ async function gatherReflectPromptSources(options, stash, parsedRef, assetConten
767
779
  }
768
780
  /** The exact prompt reflect sends, shared by dispatch and `--show-prompt`. */
769
781
  function buildReflectPromptText(args) {
770
- const { options, parsedRef, assetContent, sources, runnerSpec, draftFilePath, priorDraft } = args;
782
+ const { options, parsedRef, assetContent, sources, runnerSpec, priorDraft } = args;
771
783
  const { feedback, schemaHints, relatedLessons, rejectedProposals, standardsContext } = sources;
772
- const outputMode = runnerIsLlm(runnerSpec)
773
- ? wantsJsonSchemaOutput(runnerSpec.connection)
774
- ? "json_schema"
775
- : "framed_markdown"
776
- : undefined;
784
+ // An LLM engine that rejects JSON Schema gets the framed contract; every other engine gets the JSON object.
785
+ const outputMode = runnerIsLlm(runnerSpec) && !wantsJsonSchemaOutput(runnerSpec.connection) ? "framed_markdown" : "json_schema";
777
786
  const input = {
778
787
  ...(options.ref ? { ref: options.ref } : {}),
779
788
  ...(parsedRef?.type ? { type: parsedRef.type } : {}),
@@ -787,89 +796,65 @@ function buildReflectPromptText(args) {
787
796
  ...(options.avoidPatterns && options.avoidPatterns.length > 0 ? { avoidPatterns: options.avoidPatterns } : {}),
788
797
  ...(rejectedProposals.length > 0 ? { rejectedProposals } : {}),
789
798
  ...(priorDraft !== undefined ? { priorDraft } : {}),
790
- ...(draftFilePath ? { draftFilePath } : {}),
791
- ...(outputMode ? { outputMode } : {}),
799
+ outputMode,
792
800
  };
793
801
  const contentBudgetChars = computeReflectContentBudgetChars(input, runnerSpec);
794
802
  const { prompt } = buildReflectPrompt({
795
803
  ...input,
796
804
  ...(contentBudgetChars !== undefined ? { contentBudgetChars } : {}),
797
805
  });
798
- return { prompt, ...(outputMode ? { outputMode } : {}) };
806
+ return { prompt, outputMode };
799
807
  }
800
808
  /**
801
809
  * Dispatch with the optional self-refine loop: up to `maxRefineIters` passes,
802
- * each critiquing the prior draft, stopping early on an unchanged draft. The
803
- * direct-LLM repair budget is shared across passes.
810
+ * each critiquing the prior draft, stopping early on an unchanged draft. Every
811
+ * engine kind runs the same iteration, and the repair budget is shared across
812
+ * passes. Under the model-work tool policy an agent can edit only its own
813
+ * scratch directory, which is gone once it returns, so it returns the proposal
814
+ * as its reply, as an LLM does.
804
815
  */
805
816
  async function runReflectRefineIterations(args) {
806
- const { run, parsedRef, assetContent, sources, agentEnv, draftPaths } = args;
817
+ const { run, parsedRef, assetContent, sources, agentEnv } = args;
807
818
  const { options, runnerSpec } = run;
808
819
  const maxRefineIters = Math.max(1, options.maxRefineIters ?? 1);
809
- const canWriteFile = runnerSupportsFileWrite(runnerSpec);
810
820
  let result = {};
811
821
  let priorDraft;
812
- let lastDraftPath;
813
822
  let repairAttempts = 0;
823
+ // Only an agent or SDK engine runs a child process, with an environment and spawn or SDK seams.
824
+ const childProcess = runnerIsLlm(runnerSpec)
825
+ ? {}
826
+ : {
827
+ ...(Object.keys(agentEnv).length > 0 ? { environment: agentEnv } : {}),
828
+ ...(options.runSdk ? { runSdk: options.runSdk } : {}),
829
+ ...(options.runAgentOptions ? { runOptions: options.runAgentOptions } : {}),
830
+ };
814
831
  for (let iter = 0; iter < maxRefineIters; iter++) {
815
- const draftFilePath = canWriteFile ? synthesizeReflectDraftPath(options.ref) : undefined;
816
- if (draftFilePath) {
817
- draftPaths.push(draftFilePath);
818
- lastDraftPath = draftFilePath;
819
- }
820
832
  const { prompt, outputMode } = buildReflectPromptText({
821
833
  options,
822
834
  parsedRef,
823
835
  assetContent,
824
836
  sources,
825
837
  runnerSpec,
826
- draftFilePath,
827
838
  priorDraft,
828
839
  });
829
- let iterResult;
830
- if (runnerIsLlm(runnerSpec)) {
831
- iterResult = await runReflectViaLlm({
832
- prompt,
833
- runner: runnerSpec,
834
- ...(Object.hasOwn(options, "timeoutMs") ? { timeoutMs: options.timeoutMs } : {}),
835
- ...(options.signal ? { signal: options.signal } : {}),
836
- priorDraft,
837
- iteration: iter,
838
- ...(outputMode === "json_schema"
839
- ? { responseSchema: options.ref ? REFLECT_JSON_SCHEMA : REFLECT_UNSCOPED_JSON_SCHEMA }
840
- : {}),
841
- outputMode: outputMode ?? "framed_markdown",
842
- ...(options.ref ? { targetRef: options.ref } : {}),
843
- allowRepair: repairAttempts === 0,
844
- ...(options.chat ? { chat: options.chat } : {}),
845
- onNotices: run.notices.add,
846
- });
847
- }
848
- else {
849
- const conversation = priorDraft !== undefined && iter > 0
850
- ? [
851
- { role: "user", content: prompt },
852
- { role: "assistant", content: priorDraft },
853
- ]
854
- : undefined;
855
- const current = {
856
- ...(Object.hasOwn(options, "timeoutMs") ? { timeout: options.timeoutMs } : {}),
857
- ...(Object.keys(agentEnv).length > 0 ? { environment: agentEnv } : {}),
858
- };
859
- const prepared = resolveExecution({
860
- content: conversation ? REFLECT_CRITIQUE_PROMPT : prompt,
861
- ...(conversation ? { conversation } : {}),
862
- runner: runnerSpec,
863
- ...(Object.keys(current).length > 0 ? { current } : {}),
864
- });
865
- const lowered = buildExecution(prepared.request, prepared.runner);
866
- run.notices.add(lowered.notices);
867
- iterResult = await runExecution(lowered, {
868
- ...(options.runSdk ? { runSdk: options.runSdk } : {}),
869
- runOptions: { ...(options.signal ? { signal: options.signal } : {}), ...(options.runAgentOptions ?? {}) },
870
- });
871
- }
872
- const telemetry = reflectLlmTelemetry(iterResult);
840
+ const iterResult = await runReflectIteration({
841
+ prompt,
842
+ runner: runnerSpec,
843
+ ...(Object.hasOwn(options, "timeoutMs") ? { timeoutMs: options.timeoutMs } : {}),
844
+ ...(options.signal ? { signal: options.signal } : {}),
845
+ priorDraft,
846
+ iteration: iter,
847
+ ...(outputMode === "json_schema"
848
+ ? { responseSchema: options.ref ? REFLECT_JSON_SCHEMA : REFLECT_UNSCOPED_JSON_SCHEMA }
849
+ : {}),
850
+ outputMode,
851
+ ...(options.ref ? { targetRef: options.ref } : {}),
852
+ allowRepair: repairAttempts === 0,
853
+ ...(options.chat ? { chat: options.chat } : {}),
854
+ ...childProcess,
855
+ onNotices: run.notices.add,
856
+ });
857
+ const telemetry = reflectTelemetry(iterResult);
873
858
  if (telemetry)
874
859
  repairAttempts += telemetry.repairAttempts;
875
860
  result = telemetry
@@ -878,47 +863,25 @@ async function runReflectRefineIterations(args) {
878
863
  if (!result.ok)
879
864
  break;
880
865
  if (iter < maxRefineIters - 1) {
881
- const priorFromLlm = parsedRecord(result)?.priorDraft;
882
- const nextDraft = typeof priorFromLlm === "string" ? priorFromLlm : (result.stdout ?? "");
866
+ const priorFromReply = parsedRecord(result)?.priorDraft;
867
+ const nextDraft = typeof priorFromReply === "string" ? priorFromReply : (result.stdout ?? "");
883
868
  if (priorDraft !== undefined && nextDraft === priorDraft)
884
869
  break;
885
870
  priorDraft = nextDraft;
886
871
  }
887
872
  }
888
- return { result, lastDraftPath };
873
+ return result;
889
874
  }
890
- /**
891
- * The proposal payload from a successful run: the agent's draft file
892
- * (file-write contract, `DRAFT_WRITTEN confidence=<n>` on stdout) or the JSON
893
- * payload on stdout.
894
- */
895
- function resolveReflectPayload(run, result, lastDraftPath, sensitiveValues) {
875
+ /** The proposal payload from a successful run: the JSON payload on stdout. */
876
+ function resolveReflectPayload(run, result) {
896
877
  const { options } = run;
897
- const draftFileExists = lastDraftPath !== undefined && fs.existsSync(lastDraftPath) && fs.statSync(lastDraftPath).size > 0;
898
- const draftSignaled = /\bDRAFT_WRITTEN\b/.test(result.stdout ?? "");
899
- if (draftSignaled && lastDraftPath && !draftFileExists) {
900
- run.emitFailed("parse_error", "draft_missing", options.ref, exitCodeMeta(result));
901
- return {
902
- failure: reflectFailure(run, result, "parse_error", `Agent emitted DRAFT_WRITTEN but draft file is missing or empty (${lastDraftPath}). The file-write contract failed; either the agent's file tools are broken or the path was unwritable.`, true),
903
- };
904
- }
905
- if (draftFileExists && lastDraftPath) {
906
- const draftConfidence = extractDraftConfidence(result.stdout);
907
- return {
908
- payload: {
909
- ref: options.ref ?? "",
910
- content: redactSensitiveText(fs.readFileSync(lastDraftPath, "utf8"), sensitiveValues),
911
- ...(draftConfidence !== undefined ? { confidence: draftConfidence } : {}),
912
- },
913
- };
914
- }
915
878
  try {
916
879
  return { payload: parseAgentProposalPayload(result.stdout ?? "") };
917
880
  }
918
881
  catch (err) {
919
882
  run.emitFailed("parse_error", "parse_error", options.ref, {
920
883
  ...exitCodeMeta(result),
921
- ...(reflectLlmTelemetry(result) ?? {}),
884
+ ...(reflectTelemetry(result) ?? {}),
922
885
  });
923
886
  return {
924
887
  failure: reflectFailure(run, result, "parse_error", err instanceof Error ? err.message : String(err), true),
@@ -935,12 +898,13 @@ const NOISE_SUBREASONS = {
935
898
  * exact content that would be persisted, then mint. A judge pass is staged only
936
899
  * when the body is unchanged: a body edit the judge passes, or one made with the
937
900
  * gate off, waits for review. Size-flagged or truncation-leaking content skips
938
- * the judge and waits for review.
901
+ * the judge and waits for review. A revision with a deterministic defect is
902
+ * refused before the judge runs, whether or not the gate is on.
939
903
  */
940
904
  async function finalizeReflectProposal(args) {
941
905
  const { run, assetContent, result, judge, feedback } = args;
942
906
  const { options } = run;
943
- const telemetry = reflectLlmTelemetry(result) ?? {};
907
+ const telemetry = reflectTelemetry(result) ?? {};
944
908
  const sanitized = sanitizeReflectPayload({ content: args.payload.content, ...(args.payload.frontmatter ? { frontmatter: args.payload.frontmatter } : {}) }, assetContent, args.payload.ref);
945
909
  const payload = {
946
910
  ...args.payload,
@@ -981,11 +945,16 @@ async function finalizeReflectProposal(args) {
981
945
  }, options.eventsCtx);
982
946
  return reflectFailure(run, result, "quality_rejected", message, false);
983
947
  };
948
+ // A defect no judge needs to weigh is refused before any judge call, whether or not the gate is on.
949
+ const defect = assetContent === undefined ? undefined : findReflectDefect(assetContent, payload.content, options.defectFilter);
950
+ if (defect)
951
+ return refuse(defect, { reflectDefect: defect }, `Reflect proposal refused before the judge: ${defect}`);
984
952
  let verdict;
985
953
  let judgeFailed = false;
986
954
  if (judged) {
987
955
  verdict = await runReflectQualityJudge(run.config, payload.content, assetContent ?? "", feedback, options.chat, {
988
956
  runnerSelectionFrozen: true,
957
+ ref: payload.ref,
989
958
  ...(judge.runner ? { llmRunner: judge.runner } : {}),
990
959
  ...(Object.hasOwn(options, "timeoutMs") ? { timeoutMs: options.timeoutMs } : {}),
991
960
  ...(options.signal ? { signal: options.signal } : {}),
@@ -1114,8 +1083,6 @@ export async function renderReflectPromptPreview(options) {
1114
1083
  assetContent: source.assetContent,
1115
1084
  sources,
1116
1085
  runnerSpec,
1117
- // The same tmp-path shape a dispatch would use; never written.
1118
- draftFilePath: runnerSupportsFileWrite(runnerSpec) ? synthesizeReflectDraftPath(ref) : undefined,
1119
1086
  priorDraft: undefined,
1120
1087
  });
1121
1088
  return { ref, prompt, engine: engineName, engineKind: runnerSpec.kind };
@@ -1143,7 +1110,7 @@ export async function akmReflect(options = {}) {
1143
1110
  judgeRunner = runnerSpec;
1144
1111
  }
1145
1112
  else {
1146
- const resolved = resolveImproveLlmExecution({ config, processName: "reflect_proposal_quality-judge" });
1113
+ const resolved = resolveImproveExecution({ config, processName: "reflect_proposal_quality-judge" });
1147
1114
  if (resolved)
1148
1115
  notices.add(resolved.notices);
1149
1116
  judgeRunner = resolved?.runner;
@@ -1151,7 +1118,7 @@ export async function akmReflect(options = {}) {
1151
1118
  }
1152
1119
  const skippedNoJudge = judgeWanted && !judgeRunner;
1153
1120
  if (skippedNoJudge) {
1154
- warnOnce("reflect-quality-gate-no-judge", "Reflect proposal quality gate has no LLM configured to judge proposals (set defaults.llmEngine). Skipping the gate for this run; the proposal is queued for human review instead.");
1121
+ warnOnce("reflect-quality-gate-no-judge", "Reflect proposal quality gate has no engine configured to judge proposals (set defaults.llmEngine). Skipping the gate for this run; the proposal is queued for human review instead.");
1155
1122
  }
1156
1123
  preflightReflectDispatch(runnerSpec, notices.add);
1157
1124
  if (judgeRunner && judgeRunner !== runnerSpec)
@@ -1162,13 +1129,11 @@ export async function akmReflect(options = {}) {
1162
1129
  ...(Object.keys(agentEnv).length > 0 ? { env: agentEnv } : {}),
1163
1130
  ...(options.runAgentOptions ?? {}),
1164
1131
  });
1165
- const draftPaths = [];
1166
1132
  let result;
1167
1133
  let payload;
1168
1134
  try {
1169
- const iterated = await runReflectRefineIterations({ run, parsedRef, assetContent, sources, agentEnv, draftPaths });
1135
+ result = await runReflectRefineIterations({ run, parsedRef, assetContent, sources, agentEnv });
1170
1136
  emitInvoked();
1171
- result = iterated.result;
1172
1137
  if (!result.ok) {
1173
1138
  if (isEnoentFailure(result)) {
1174
1139
  emitFailed("spawn_failed", "enoent", options.ref, {
@@ -1191,11 +1156,11 @@ export async function akmReflect(options = {}) {
1191
1156
  };
1192
1157
  emitFailed(envelope.reason, envelope.reason === "parse_error" ? "parse_error" : "agent_crash", options.ref, {
1193
1158
  ...(envelope.exitCode !== null ? { exitCode: envelope.exitCode } : {}),
1194
- ...(reflectLlmTelemetry(result) ?? {}),
1159
+ ...(reflectTelemetry(result) ?? {}),
1195
1160
  });
1196
1161
  return { ...envelope, ...notices.fields() };
1197
1162
  }
1198
- const resolved = resolveReflectPayload(run, result, iterated.lastDraftPath, sensitiveValues);
1163
+ const resolved = resolveReflectPayload(run, result);
1199
1164
  if ("failure" in resolved)
1200
1165
  return resolved.failure;
1201
1166
  payload = resolved.payload;
@@ -1205,43 +1170,11 @@ export async function akmReflect(options = {}) {
1205
1170
  emitInvoked();
1206
1171
  throw error;
1207
1172
  }
1208
- finally {
1209
- for (const draftPath of draftPaths) {
1210
- try {
1211
- if (fs.existsSync(draftPath))
1212
- fs.unlinkSync(draftPath);
1213
- }
1214
- catch {
1215
- // best-effort
1216
- }
1217
- }
1218
- }
1219
1173
  const unsafeContent = generatedContentRejection(payload.content, redactSensitiveText(payload.content, sensitiveValues));
1220
1174
  if (unsafeContent) {
1221
1175
  emitFailed("parse_error", "parse_error", options.ref, exitCodeMeta(result));
1222
1176
  return reflectFailure(run, result, "parse_error", unsafeContent, false);
1223
1177
  }
1224
- // A retargeted proposal is refused (malformed refs are left to proposal validation).
1225
- if (options.ref) {
1226
- let retargeted = false;
1227
- try {
1228
- const expected = parseRefInput(options.ref);
1229
- const actual = parseRefInput(payload.ref);
1230
- retargeted = expected.type !== actual.type || expected.name !== actual.name;
1231
- }
1232
- catch {
1233
- retargeted = false;
1234
- }
1235
- if (retargeted) {
1236
- emitFailed("parse_error", "ref_mismatch", options.ref, {
1237
- expectedRef: options.ref,
1238
- actualRef: payload.ref,
1239
- ...exitCodeMeta(result),
1240
- ...(reflectLlmTelemetry(result) ?? {}),
1241
- });
1242
- return reflectFailure(run, result, "parse_error", `Agent retargeted proposal: expected ref "${options.ref}" but got "${payload.ref}". Proposal rejected to prevent silent ref hallucination.`, true);
1243
- }
1244
- }
1245
1178
  return finalizeReflectProposal({
1246
1179
  run,
1247
1180
  payload,
@@ -32,6 +32,10 @@ const GRADE_SCHEMA = {
32
32
  additionalProperties: false,
33
33
  properties: { grade: { type: "integer", minimum: 0, maximum: 3 }, reason: { type: "string" } },
34
34
  };
35
+ function parseGrade(raw) {
36
+ const grade = parseEmbeddedJsonResponse(raw)?.grade;
37
+ return typeof grade === "number" && Number.isInteger(grade) && grade >= 0 && grade <= 3 ? grade : undefined;
38
+ }
35
39
  /** Up to five distinct task queries, in the given order, whitespace collapsed. */
36
40
  export function usableRetrievalQueries(raw) {
37
41
  const out = [];
@@ -100,10 +104,11 @@ export async function runRetrievalRegressionGate(args) {
100
104
  ...(args.signal ? { signal: args.signal } : {}),
101
105
  ...(args.chat ? { chat: args.chat } : {}),
102
106
  },
107
+ parse: parseGrade,
103
108
  ...(args.onNotices ? { onNotices: args.onNotices } : {}),
104
109
  });
105
- const grade = outcome.ok ? parseEmbeddedJsonResponse(outcome.raw)?.grade : undefined;
106
- if (typeof grade !== "number" || !Number.isInteger(grade) || grade < 0 || grade > 3) {
110
+ const grade = outcome.ok ? parseGrade(outcome.raw) : undefined;
111
+ if (grade === undefined) {
107
112
  return {
108
113
  pass: false,
109
114
  queries: args.queries.length,