akm-cli 0.9.24 → 0.9.25-alpha.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +173 -0
- package/dist/cli.js +1 -1
- package/dist/commands/health/checks.js +10 -11
- package/dist/commands/improve/consolidate/pair-pass.js +4 -2
- package/dist/commands/improve/consolidate.js +10 -4
- package/dist/commands/improve/execution.js +4 -11
- package/dist/commands/improve/extract-prompt.js +4 -4
- package/dist/commands/improve/extract.js +11 -13
- package/dist/commands/improve/improve-cli.js +65 -34
- package/dist/commands/improve/improve-strategies.js +49 -43
- package/dist/commands/improve/improve-usage-report.js +8 -17
- package/dist/commands/improve/loop-stages.js +3 -0
- package/dist/commands/improve/preparation.js +3 -1
- package/dist/commands/improve/reflect-noise.js +125 -0
- package/dist/commands/improve/reflect.js +105 -172
- package/dist/commands/improve/retrieval-gate.js +7 -2
- package/dist/commands/improve/stage.js +67 -24
- package/dist/commands/proposal/drain.js +11 -2
- package/dist/commands/proposal/proposal-cli.js +1 -5
- package/dist/commands/proposal/propose-cli.js +2 -2
- package/dist/commands/proposal/propose.js +72 -84
- package/dist/commands/proposal/validators/proposal-quality-validators.js +4 -2
- package/dist/commands/read/search-cli.js +0 -38
- package/dist/commands/remember.js +3 -3
- package/dist/commands/sources/schema-repair.js +1 -1
- package/dist/core/config/config-schema.js +37 -60
- package/dist/core/config/engine-semantics.js +15 -11
- package/dist/core/config/schema/improve-processes.js +18 -2
- package/dist/core/improve-result.js +3 -3
- package/dist/core/redaction.js +4 -0
- package/dist/core/spawn-env.js +25 -0
- package/dist/core/structured.js +11 -1
- package/dist/execution/source.js +10 -0
- package/dist/indexer/passes/memory-inference.js +2 -1
- package/dist/integrations/agent/builder-shared.js +15 -0
- package/dist/integrations/agent/config.js +1 -1
- package/dist/integrations/agent/engine-resolution.js +13 -31
- package/dist/integrations/agent/execution.js +48 -22
- package/dist/integrations/agent/index.js +1 -1
- package/dist/integrations/agent/profiles.js +2 -2
- package/dist/integrations/agent/prompts.js +55 -114
- package/dist/integrations/agent/request-lowering.js +21 -8
- package/dist/integrations/agent/runner-dispatch.js +96 -3
- package/dist/integrations/agent/runner.js +8 -2
- package/dist/integrations/harnesses/aider/agent-builder.js +10 -23
- package/dist/integrations/harnesses/aider/index.js +0 -5
- package/dist/integrations/harnesses/amazonq/agent-builder.js +9 -21
- package/dist/integrations/harnesses/amazonq/index.js +0 -5
- package/dist/integrations/harnesses/claude/agent-builder.js +47 -36
- package/dist/integrations/harnesses/claude/index.js +0 -14
- package/dist/integrations/harnesses/codex/agent-builder.js +3 -4
- package/dist/integrations/harnesses/codex/index.js +0 -4
- package/dist/integrations/harnesses/copilot/agent-builder.js +4 -13
- package/dist/integrations/harnesses/copilot/index.js +2 -7
- package/dist/integrations/harnesses/gemini/agent-builder.js +4 -13
- package/dist/integrations/harnesses/gemini/index.js +0 -5
- package/dist/integrations/harnesses/ids.js +12 -10
- package/dist/integrations/harnesses/opencode/agent-builder.js +41 -9
- package/dist/integrations/harnesses/opencode/index.js +0 -8
- package/dist/integrations/harnesses/opencode/model-config.js +33 -0
- package/dist/integrations/harnesses/opencode/model-work-agent.js +115 -0
- package/dist/integrations/harnesses/opencode-sdk/harness.js +2 -7
- package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +114 -28
- package/dist/integrations/harnesses/openhands/agent-builder.js +7 -21
- package/dist/integrations/harnesses/openhands/index.js +0 -5
- package/dist/integrations/harnesses/pi/agent-builder.js +7 -21
- package/dist/integrations/harnesses/pi/index.js +0 -5
- package/dist/llm/client.js +5 -0
- package/dist/llm/feature-gate.js +12 -9
- package/dist/llm/index-passes.js +2 -5
- package/dist/llm/memory-infer.js +6 -5
- package/dist/llm/structured-call.js +33 -11
- package/dist/output/shapes/passthrough.js +1 -0
- package/dist/scripts/akm-migrate-node.js +298 -228
- package/dist/scripts/akm-migrate.js +298 -228
- package/dist/workflows/exec/step-work.js +6 -5
- package/dist/workflows/exec/unit-dispatch.js +4 -13
- package/dist/workflows/freeze/step-values.js +1 -1
- package/docs/reference/cli.md +41 -18
- package/docs/reference/configuration.md +165 -12
- package/docs/reference/data-and-telemetry.md +2 -3
- package/docs/reference/workflow-schema.md +10 -9
- package/package.json +1 -1
- package/schemas/akm-config.json +108 -0
|
@@ -11,7 +11,6 @@
|
|
|
11
11
|
* pre-dispatch refusals still emit both).
|
|
12
12
|
*/
|
|
13
13
|
import fs from "node:fs";
|
|
14
|
-
import os from "node:os";
|
|
15
14
|
import path from "node:path";
|
|
16
15
|
import { assembleAssetFromString, serializeFrontmatter } from "../../core/asset/asset-serialize.js";
|
|
17
16
|
import { parseFrontmatter } from "../../core/asset/frontmatter.js";
|
|
@@ -31,20 +30,20 @@ import { lookup } from "../../indexer/indexer.js";
|
|
|
31
30
|
import { DEFAULT_LLM_TIMEOUT_MS } from "../../integrations/agent/config.js";
|
|
32
31
|
import { fallbackAnnouncement, NO_ENGINE_MESSAGE_SUFFIX, NO_ENGINE_REMEDY, withEngineFallback, } from "../../integrations/agent/engine-fallback.js";
|
|
33
32
|
import { buildExecution, resolveExecution } from "../../integrations/agent/execution.js";
|
|
34
|
-
import { buildReflectOutputRepairPrompt, buildReflectPrompt,
|
|
35
|
-
import { runnerIsLlm
|
|
36
|
-
import { assertRunnerCredentials, collectDispatchSensitiveValues,
|
|
33
|
+
import { buildReflectOutputRepairPrompt, buildReflectPrompt, parseAgentProposalPayload, REFLECT_CONTENT_CAP, REFLECT_TRUNCATION_MARKER, } from "../../integrations/agent/prompts.js";
|
|
34
|
+
import { runnerIsLlm } from "../../integrations/agent/runner.js";
|
|
35
|
+
import { assertRunnerCredentials, collectDispatchSensitiveValues, } from "../../integrations/agent/runner-dispatch.js";
|
|
37
36
|
import { isJsonSchemaKnownUnsupported, LlmCallError } from "../../llm/client.js";
|
|
38
37
|
import { baseFailureFields, enoentHintMessage, isEnoentFailure } from "../agent/agent-support.js";
|
|
39
38
|
import { checkReflectSize, isValidDescription } from "../proposal/validators/proposal-quality-validators.js";
|
|
40
39
|
import { CHARS_PER_TOKEN, DEFAULT_CONTEXT_LENGTH_TOKENS } from "./consolidate/chunking.js";
|
|
41
40
|
import { deriveLessonRef } from "./distill.js";
|
|
42
41
|
import { findAssetFilePath } from "./eligibility.js";
|
|
43
|
-
import {
|
|
42
|
+
import { resolveImproveExecution } from "./execution.js";
|
|
44
43
|
import { recordLedgerAttempt } from "./ledger.js";
|
|
45
|
-
import { classifyReflectChange, splitFrontmatter } from "./reflect-noise.js";
|
|
44
|
+
import { classifyReflectChange, findReflectDefect, splitFrontmatter } from "./reflect-noise.js";
|
|
46
45
|
import { loadRetrievalQueries, runRetrievalRegressionGate } from "./retrieval-gate.js";
|
|
47
|
-
import {
|
|
46
|
+
import { callStageOnce, errMessage, mintProposal, noticeSet, rejectedProposalContext, resolveQualityGateJudge, runReflectQualityJudge, } from "./stage.js";
|
|
48
47
|
const MAX_FEEDBACK_LINES = 10;
|
|
49
48
|
const MAX_GLOBAL_FEEDBACK_LINES = 20;
|
|
50
49
|
function readOnlyEventsContext(ctx) {
|
|
@@ -83,16 +82,6 @@ export const REFLECT_ALLOWED_TYPES = new Set([
|
|
|
83
82
|
const REFLECT_REFUSED_TYPES = new Set(["secret"]);
|
|
84
83
|
/** Identity fields the model may never change (a renamed `name` breaks ref resolution). */
|
|
85
84
|
const PROTECTED_FRONTMATTER_FIELDS = new Set(["name", "ref", "id", "slug", "type"]);
|
|
86
|
-
/**
|
|
87
|
-
* A fresh tmp path per iteration for the agent/SDK file-write contract (long
|
|
88
|
-
* bodies are written to a file instead of fenced JSON on stdout). The direct
|
|
89
|
-
* LLM runner has no filesystem and never gets one.
|
|
90
|
-
*/
|
|
91
|
-
function synthesizeReflectDraftPath(ref) {
|
|
92
|
-
const safeRef = (ref ?? "no-ref").replace(/[^a-z0-9_-]/gi, "_");
|
|
93
|
-
const rand = Math.random().toString(36).slice(2, 8);
|
|
94
|
-
return path.join(os.tmpdir(), `akm-reflect-${safeRef}-${Date.now()}-${rand}.md`);
|
|
95
|
-
}
|
|
96
85
|
/** Lesson lint findings for the prompt: a concrete starting point for the revision. */
|
|
97
86
|
function buildSchemaHints(type, content) {
|
|
98
87
|
if (!content || type !== "lesson")
|
|
@@ -328,7 +317,12 @@ export function sanitizeReflectPayload(payload, sourceContent, targetRef) {
|
|
|
328
317
|
...(truncationMarkerLeaked ? { truncationMarkerLeaked } : {}),
|
|
329
318
|
};
|
|
330
319
|
}
|
|
331
|
-
// ──
|
|
320
|
+
// ── Output contract ──────────────────────────────────────────────────────────
|
|
321
|
+
//
|
|
322
|
+
// Every engine kind is asked for the same reply: the JSON object of
|
|
323
|
+
// REFLECT_JSON_SCHEMA, or the framed-markdown frame for an LLM engine that
|
|
324
|
+
// rejects JSON Schema. The reply is validated by the parse functions below and
|
|
325
|
+
// repaired once, whatever engine produced it.
|
|
332
326
|
const REFLECT_FRONTMATTER_PATCH_JSON_SCHEMA = {
|
|
333
327
|
type: "object",
|
|
334
328
|
required: ["description", "when_to_use"],
|
|
@@ -366,7 +360,7 @@ const REFLECT_UNSCOPED_JSON_SCHEMA = {
|
|
|
366
360
|
},
|
|
367
361
|
};
|
|
368
362
|
/**
|
|
369
|
-
* Frame for JSON Schema unless the connection disabled it or already proved
|
|
363
|
+
* Frame for JSON Schema unless the LLM connection disabled it or already proved
|
|
370
364
|
* this process that it rejects it (the transport retries plain text on a 4xx).
|
|
371
365
|
*/
|
|
372
366
|
function wantsJsonSchemaOutput(connection) {
|
|
@@ -379,7 +373,7 @@ function parsedRecord(result) {
|
|
|
379
373
|
? result.parsed
|
|
380
374
|
: undefined;
|
|
381
375
|
}
|
|
382
|
-
function
|
|
376
|
+
function reflectTelemetry(result) {
|
|
383
377
|
const parsed = parsedRecord(result);
|
|
384
378
|
if (!parsed || (parsed.outputMode !== "json_schema" && parsed.outputMode !== "framed_markdown"))
|
|
385
379
|
return undefined;
|
|
@@ -484,13 +478,17 @@ function parseFramedReflectOutput(raw, targetRef) {
|
|
|
484
478
|
return { ref, content, confidence, ...(frontmatter ? { frontmatter } : {}) };
|
|
485
479
|
}
|
|
486
480
|
/**
|
|
487
|
-
* One reflect iteration
|
|
488
|
-
*
|
|
489
|
-
*
|
|
481
|
+
* One reflect iteration on any engine, as an agent-shaped result (errors
|
|
482
|
+
* captured, never thrown except configuration). Every engine kind is sent the
|
|
483
|
+
* same request, with the reply's schema as the output schema, and its reply is
|
|
484
|
+
* held to the same contract: an unparseable reply gets one repair turn within
|
|
485
|
+
* the original deadline, then fails. A failed agent or SDK dispatch is reported
|
|
486
|
+
* as it ran, with its exit code and stderr; an LLM call has only its message.
|
|
490
487
|
*/
|
|
491
|
-
export async function
|
|
488
|
+
export async function runReflectIteration(opts) {
|
|
492
489
|
const start = Date.now();
|
|
493
490
|
let repairAttempts = 0;
|
|
491
|
+
// Model work is bounded: with no timeout from the caller or the runner, the default.
|
|
494
492
|
const configuredTimeout = Object.hasOwn(opts, "timeoutMs")
|
|
495
493
|
? (opts.timeoutMs ?? null)
|
|
496
494
|
: Object.hasOwn(opts.runner, "timeoutMs")
|
|
@@ -505,7 +503,7 @@ export async function runReflectViaLlm(opts) {
|
|
|
505
503
|
? parseSchemaReflectOutput(raw, opts.targetRef)
|
|
506
504
|
: parseFramedReflectOutput(raw, opts.targetRef);
|
|
507
505
|
const failure = (err, reason, stdout = "", exitCode = 1) => {
|
|
508
|
-
const msg =
|
|
506
|
+
const msg = errMessage(err);
|
|
509
507
|
return {
|
|
510
508
|
ok: false,
|
|
511
509
|
stdout,
|
|
@@ -517,8 +515,13 @@ export async function runReflectViaLlm(opts) {
|
|
|
517
515
|
parsed: { outputMode: opts.outputMode, repairAttempts },
|
|
518
516
|
};
|
|
519
517
|
};
|
|
518
|
+
// A reply that broke the contract; its message is the parser's, on every engine kind.
|
|
519
|
+
const invalidReply = (err, reply) => failure(err, "parse_error", reply, 0);
|
|
520
|
+
// The result of the dispatch that failed, kept for an agent or SDK engine.
|
|
521
|
+
let dispatched;
|
|
522
|
+
// Reflect parses and repairs its own reply (the repair turn below), so one dispatch, unvalidated.
|
|
520
523
|
const call = async (callMessages, repairTimeoutMs) => {
|
|
521
|
-
const outcome = await
|
|
524
|
+
const outcome = await callStageOnce({
|
|
522
525
|
feature: "reflect_proposal",
|
|
523
526
|
runner: opts.runner,
|
|
524
527
|
prompt: callMessages.at(-1)?.content ?? "",
|
|
@@ -535,13 +538,18 @@ export async function runReflectViaLlm(opts) {
|
|
|
535
538
|
// Visible chain-of-thought can exhaust the output before the envelope.
|
|
536
539
|
enableThinking: false,
|
|
537
540
|
...(opts.chat ? { chat: opts.chat } : {}),
|
|
541
|
+
...(opts.runSdk ? { runSdk: opts.runSdk } : {}),
|
|
542
|
+
...(opts.runOptions ? { runOptions: opts.runOptions } : {}),
|
|
538
543
|
},
|
|
544
|
+
...(opts.environment ? { current: { environment: opts.environment } } : {}),
|
|
539
545
|
...(opts.onNotices ? { onNotices: opts.onNotices } : {}),
|
|
540
546
|
});
|
|
541
547
|
if (!outcome.ok) {
|
|
548
|
+
if (!runnerIsLlm(opts.runner))
|
|
549
|
+
dispatched = outcome.result;
|
|
542
550
|
throw outcome.reason === "timeout"
|
|
543
551
|
? new LlmCallError(outcome.error ?? "timeout", "timeout")
|
|
544
|
-
: new Error(outcome.error ?? "
|
|
552
|
+
: new Error(outcome.error ?? "dispatch failed");
|
|
545
553
|
}
|
|
546
554
|
return outcome.raw;
|
|
547
555
|
};
|
|
@@ -556,7 +564,7 @@ export async function runReflectViaLlm(opts) {
|
|
|
556
564
|
}
|
|
557
565
|
catch (err) {
|
|
558
566
|
if (opts.allowRepair === false)
|
|
559
|
-
return
|
|
567
|
+
return invalidReply(err, stdout);
|
|
560
568
|
if (opts.signal?.aborted)
|
|
561
569
|
return failure(new Error("Reflect request aborted"), "aborted", stdout);
|
|
562
570
|
const remaining = deadline === undefined ? undefined : deadline - Date.now();
|
|
@@ -570,7 +578,7 @@ export async function runReflectViaLlm(opts) {
|
|
|
570
578
|
payload = parse(acceptedOutput);
|
|
571
579
|
}
|
|
572
580
|
catch (repairErr) {
|
|
573
|
-
return
|
|
581
|
+
return invalidReply(repairErr, acceptedOutput);
|
|
574
582
|
}
|
|
575
583
|
}
|
|
576
584
|
return {
|
|
@@ -585,6 +593,8 @@ export async function runReflectViaLlm(opts) {
|
|
|
585
593
|
catch (err) {
|
|
586
594
|
if (err instanceof ConfigError)
|
|
587
595
|
throw err;
|
|
596
|
+
if (dispatched)
|
|
597
|
+
return { ...dispatched, parsed: { outputMode: opts.outputMode, repairAttempts } };
|
|
588
598
|
const reason = opts.signal?.aborted
|
|
589
599
|
? "aborted"
|
|
590
600
|
: err instanceof LlmCallError && err.code === "timeout"
|
|
@@ -685,8 +695,9 @@ async function resolveReflectSource(options, stash, emitFailed) {
|
|
|
685
695
|
}
|
|
686
696
|
/**
|
|
687
697
|
* The single engine for this invocation: `--engine`, the improve strategy's
|
|
688
|
-
*
|
|
689
|
-
*
|
|
698
|
+
* reflect process, or `defaults.engine` (announced when it falls back to the
|
|
699
|
+
* SDK binary). Whatever its kind, reflect runs it under the model-work tool
|
|
700
|
+
* policy.
|
|
690
701
|
*/
|
|
691
702
|
function resolveReflectRunner(options) {
|
|
692
703
|
const config = options.config ?? loadConfig();
|
|
@@ -700,14 +711,14 @@ function resolveReflectRunner(options) {
|
|
|
700
711
|
lowered = lower({ content: "reflect engine selection", config, current: { engine: options.engine } });
|
|
701
712
|
}
|
|
702
713
|
else if (options.improveProfile) {
|
|
703
|
-
const resolved =
|
|
714
|
+
const resolved = resolveImproveExecution({
|
|
704
715
|
config,
|
|
705
716
|
profile: activeStrategy,
|
|
706
717
|
process: activeStrategy?.processes?.reflect,
|
|
707
718
|
processName: "reflect",
|
|
708
719
|
});
|
|
709
720
|
if (!resolved) {
|
|
710
|
-
throw new ConfigError("Reflect requires an
|
|
721
|
+
throw new ConfigError("Reflect requires an engine for the active improve strategy.", "LLM_NOT_CONFIGURED", "Set defaults.llmEngine or improve.strategies.<name>.processes.reflect.engine.");
|
|
711
722
|
}
|
|
712
723
|
lowered = resolved;
|
|
713
724
|
}
|
|
@@ -723,19 +734,20 @@ function resolveReflectRunner(options) {
|
|
|
723
734
|
lowered = lower({ content: "reflect engine selection", config });
|
|
724
735
|
}
|
|
725
736
|
const runnerSpec = lowered.runner;
|
|
726
|
-
if (options.eventSource === "improve" && !runnerIsLlm(runnerSpec)) {
|
|
727
|
-
throw new ConfigError(`Unattended improve requires an LLM engine for reflect; engine "${runnerSpec.engine ?? options.engine ?? "unknown"}" is tool-capable.`, "INVALID_CONFIG_FILE", "Set defaults.llmEngine or improve.strategies.<name>.processes.reflect.engine to an LLM engine.");
|
|
728
|
-
}
|
|
729
737
|
const engineName = runnerSpec.engine ?? options.engine;
|
|
730
738
|
if (!engineName)
|
|
731
739
|
throw new ConfigError("Reflect requires a named engine.", "INVALID_CONFIG_FILE");
|
|
732
740
|
return { config, activeStrategy, runnerSpec, engineName, notices: lowered.notices };
|
|
733
741
|
}
|
|
734
|
-
/**
|
|
742
|
+
/**
|
|
743
|
+
* Lower a runner under the model-work tool policy and check its credentials,
|
|
744
|
+
* so a bad transport, or one that cannot confine the policy, fails before any work.
|
|
745
|
+
*/
|
|
735
746
|
function preflightReflectDispatch(runnerSpec, onNotices) {
|
|
736
747
|
const prepared = resolveExecution({
|
|
737
748
|
content: "Validate reflect operation transport before dispatch.",
|
|
738
749
|
runner: runnerSpec,
|
|
750
|
+
modelWork: true,
|
|
739
751
|
});
|
|
740
752
|
const lowered = buildExecution(prepared.request, prepared.runner);
|
|
741
753
|
onNotices(lowered.notices);
|
|
@@ -767,13 +779,10 @@ async function gatherReflectPromptSources(options, stash, parsedRef, assetConten
|
|
|
767
779
|
}
|
|
768
780
|
/** The exact prompt reflect sends, shared by dispatch and `--show-prompt`. */
|
|
769
781
|
function buildReflectPromptText(args) {
|
|
770
|
-
const { options, parsedRef, assetContent, sources, runnerSpec,
|
|
782
|
+
const { options, parsedRef, assetContent, sources, runnerSpec, priorDraft } = args;
|
|
771
783
|
const { feedback, schemaHints, relatedLessons, rejectedProposals, standardsContext } = sources;
|
|
772
|
-
|
|
773
|
-
|
|
774
|
-
? "json_schema"
|
|
775
|
-
: "framed_markdown"
|
|
776
|
-
: undefined;
|
|
784
|
+
// An LLM engine that rejects JSON Schema gets the framed contract; every other engine gets the JSON object.
|
|
785
|
+
const outputMode = runnerIsLlm(runnerSpec) && !wantsJsonSchemaOutput(runnerSpec.connection) ? "framed_markdown" : "json_schema";
|
|
777
786
|
const input = {
|
|
778
787
|
...(options.ref ? { ref: options.ref } : {}),
|
|
779
788
|
...(parsedRef?.type ? { type: parsedRef.type } : {}),
|
|
@@ -787,89 +796,65 @@ function buildReflectPromptText(args) {
|
|
|
787
796
|
...(options.avoidPatterns && options.avoidPatterns.length > 0 ? { avoidPatterns: options.avoidPatterns } : {}),
|
|
788
797
|
...(rejectedProposals.length > 0 ? { rejectedProposals } : {}),
|
|
789
798
|
...(priorDraft !== undefined ? { priorDraft } : {}),
|
|
790
|
-
|
|
791
|
-
...(outputMode ? { outputMode } : {}),
|
|
799
|
+
outputMode,
|
|
792
800
|
};
|
|
793
801
|
const contentBudgetChars = computeReflectContentBudgetChars(input, runnerSpec);
|
|
794
802
|
const { prompt } = buildReflectPrompt({
|
|
795
803
|
...input,
|
|
796
804
|
...(contentBudgetChars !== undefined ? { contentBudgetChars } : {}),
|
|
797
805
|
});
|
|
798
|
-
return { prompt,
|
|
806
|
+
return { prompt, outputMode };
|
|
799
807
|
}
|
|
800
808
|
/**
|
|
801
809
|
* Dispatch with the optional self-refine loop: up to `maxRefineIters` passes,
|
|
802
|
-
* each critiquing the prior draft, stopping early on an unchanged draft.
|
|
803
|
-
*
|
|
810
|
+
* each critiquing the prior draft, stopping early on an unchanged draft. Every
|
|
811
|
+
* engine kind runs the same iteration, and the repair budget is shared across
|
|
812
|
+
* passes. Under the model-work tool policy an agent can edit only its own
|
|
813
|
+
* scratch directory, which is gone once it returns, so it returns the proposal
|
|
814
|
+
* as its reply, as an LLM does.
|
|
804
815
|
*/
|
|
805
816
|
async function runReflectRefineIterations(args) {
|
|
806
|
-
const { run, parsedRef, assetContent, sources, agentEnv
|
|
817
|
+
const { run, parsedRef, assetContent, sources, agentEnv } = args;
|
|
807
818
|
const { options, runnerSpec } = run;
|
|
808
819
|
const maxRefineIters = Math.max(1, options.maxRefineIters ?? 1);
|
|
809
|
-
const canWriteFile = runnerSupportsFileWrite(runnerSpec);
|
|
810
820
|
let result = {};
|
|
811
821
|
let priorDraft;
|
|
812
|
-
let lastDraftPath;
|
|
813
822
|
let repairAttempts = 0;
|
|
823
|
+
// Only an agent or SDK engine runs a child process, with an environment and spawn or SDK seams.
|
|
824
|
+
const childProcess = runnerIsLlm(runnerSpec)
|
|
825
|
+
? {}
|
|
826
|
+
: {
|
|
827
|
+
...(Object.keys(agentEnv).length > 0 ? { environment: agentEnv } : {}),
|
|
828
|
+
...(options.runSdk ? { runSdk: options.runSdk } : {}),
|
|
829
|
+
...(options.runAgentOptions ? { runOptions: options.runAgentOptions } : {}),
|
|
830
|
+
};
|
|
814
831
|
for (let iter = 0; iter < maxRefineIters; iter++) {
|
|
815
|
-
const draftFilePath = canWriteFile ? synthesizeReflectDraftPath(options.ref) : undefined;
|
|
816
|
-
if (draftFilePath) {
|
|
817
|
-
draftPaths.push(draftFilePath);
|
|
818
|
-
lastDraftPath = draftFilePath;
|
|
819
|
-
}
|
|
820
832
|
const { prompt, outputMode } = buildReflectPromptText({
|
|
821
833
|
options,
|
|
822
834
|
parsedRef,
|
|
823
835
|
assetContent,
|
|
824
836
|
sources,
|
|
825
837
|
runnerSpec,
|
|
826
|
-
draftFilePath,
|
|
827
838
|
priorDraft,
|
|
828
839
|
});
|
|
829
|
-
|
|
830
|
-
|
|
831
|
-
|
|
832
|
-
|
|
833
|
-
|
|
834
|
-
|
|
835
|
-
|
|
836
|
-
|
|
837
|
-
|
|
838
|
-
|
|
839
|
-
|
|
840
|
-
|
|
841
|
-
|
|
842
|
-
|
|
843
|
-
|
|
844
|
-
|
|
845
|
-
|
|
846
|
-
|
|
847
|
-
}
|
|
848
|
-
else {
|
|
849
|
-
const conversation = priorDraft !== undefined && iter > 0
|
|
850
|
-
? [
|
|
851
|
-
{ role: "user", content: prompt },
|
|
852
|
-
{ role: "assistant", content: priorDraft },
|
|
853
|
-
]
|
|
854
|
-
: undefined;
|
|
855
|
-
const current = {
|
|
856
|
-
...(Object.hasOwn(options, "timeoutMs") ? { timeout: options.timeoutMs } : {}),
|
|
857
|
-
...(Object.keys(agentEnv).length > 0 ? { environment: agentEnv } : {}),
|
|
858
|
-
};
|
|
859
|
-
const prepared = resolveExecution({
|
|
860
|
-
content: conversation ? REFLECT_CRITIQUE_PROMPT : prompt,
|
|
861
|
-
...(conversation ? { conversation } : {}),
|
|
862
|
-
runner: runnerSpec,
|
|
863
|
-
...(Object.keys(current).length > 0 ? { current } : {}),
|
|
864
|
-
});
|
|
865
|
-
const lowered = buildExecution(prepared.request, prepared.runner);
|
|
866
|
-
run.notices.add(lowered.notices);
|
|
867
|
-
iterResult = await runExecution(lowered, {
|
|
868
|
-
...(options.runSdk ? { runSdk: options.runSdk } : {}),
|
|
869
|
-
runOptions: { ...(options.signal ? { signal: options.signal } : {}), ...(options.runAgentOptions ?? {}) },
|
|
870
|
-
});
|
|
871
|
-
}
|
|
872
|
-
const telemetry = reflectLlmTelemetry(iterResult);
|
|
840
|
+
const iterResult = await runReflectIteration({
|
|
841
|
+
prompt,
|
|
842
|
+
runner: runnerSpec,
|
|
843
|
+
...(Object.hasOwn(options, "timeoutMs") ? { timeoutMs: options.timeoutMs } : {}),
|
|
844
|
+
...(options.signal ? { signal: options.signal } : {}),
|
|
845
|
+
priorDraft,
|
|
846
|
+
iteration: iter,
|
|
847
|
+
...(outputMode === "json_schema"
|
|
848
|
+
? { responseSchema: options.ref ? REFLECT_JSON_SCHEMA : REFLECT_UNSCOPED_JSON_SCHEMA }
|
|
849
|
+
: {}),
|
|
850
|
+
outputMode,
|
|
851
|
+
...(options.ref ? { targetRef: options.ref } : {}),
|
|
852
|
+
allowRepair: repairAttempts === 0,
|
|
853
|
+
...(options.chat ? { chat: options.chat } : {}),
|
|
854
|
+
...childProcess,
|
|
855
|
+
onNotices: run.notices.add,
|
|
856
|
+
});
|
|
857
|
+
const telemetry = reflectTelemetry(iterResult);
|
|
873
858
|
if (telemetry)
|
|
874
859
|
repairAttempts += telemetry.repairAttempts;
|
|
875
860
|
result = telemetry
|
|
@@ -878,47 +863,25 @@ async function runReflectRefineIterations(args) {
|
|
|
878
863
|
if (!result.ok)
|
|
879
864
|
break;
|
|
880
865
|
if (iter < maxRefineIters - 1) {
|
|
881
|
-
const
|
|
882
|
-
const nextDraft = typeof
|
|
866
|
+
const priorFromReply = parsedRecord(result)?.priorDraft;
|
|
867
|
+
const nextDraft = typeof priorFromReply === "string" ? priorFromReply : (result.stdout ?? "");
|
|
883
868
|
if (priorDraft !== undefined && nextDraft === priorDraft)
|
|
884
869
|
break;
|
|
885
870
|
priorDraft = nextDraft;
|
|
886
871
|
}
|
|
887
872
|
}
|
|
888
|
-
return
|
|
873
|
+
return result;
|
|
889
874
|
}
|
|
890
|
-
/**
|
|
891
|
-
|
|
892
|
-
* (file-write contract, `DRAFT_WRITTEN confidence=<n>` on stdout) or the JSON
|
|
893
|
-
* payload on stdout.
|
|
894
|
-
*/
|
|
895
|
-
function resolveReflectPayload(run, result, lastDraftPath, sensitiveValues) {
|
|
875
|
+
/** The proposal payload from a successful run: the JSON payload on stdout. */
|
|
876
|
+
function resolveReflectPayload(run, result) {
|
|
896
877
|
const { options } = run;
|
|
897
|
-
const draftFileExists = lastDraftPath !== undefined && fs.existsSync(lastDraftPath) && fs.statSync(lastDraftPath).size > 0;
|
|
898
|
-
const draftSignaled = /\bDRAFT_WRITTEN\b/.test(result.stdout ?? "");
|
|
899
|
-
if (draftSignaled && lastDraftPath && !draftFileExists) {
|
|
900
|
-
run.emitFailed("parse_error", "draft_missing", options.ref, exitCodeMeta(result));
|
|
901
|
-
return {
|
|
902
|
-
failure: reflectFailure(run, result, "parse_error", `Agent emitted DRAFT_WRITTEN but draft file is missing or empty (${lastDraftPath}). The file-write contract failed; either the agent's file tools are broken or the path was unwritable.`, true),
|
|
903
|
-
};
|
|
904
|
-
}
|
|
905
|
-
if (draftFileExists && lastDraftPath) {
|
|
906
|
-
const draftConfidence = extractDraftConfidence(result.stdout);
|
|
907
|
-
return {
|
|
908
|
-
payload: {
|
|
909
|
-
ref: options.ref ?? "",
|
|
910
|
-
content: redactSensitiveText(fs.readFileSync(lastDraftPath, "utf8"), sensitiveValues),
|
|
911
|
-
...(draftConfidence !== undefined ? { confidence: draftConfidence } : {}),
|
|
912
|
-
},
|
|
913
|
-
};
|
|
914
|
-
}
|
|
915
878
|
try {
|
|
916
879
|
return { payload: parseAgentProposalPayload(result.stdout ?? "") };
|
|
917
880
|
}
|
|
918
881
|
catch (err) {
|
|
919
882
|
run.emitFailed("parse_error", "parse_error", options.ref, {
|
|
920
883
|
...exitCodeMeta(result),
|
|
921
|
-
...(
|
|
884
|
+
...(reflectTelemetry(result) ?? {}),
|
|
922
885
|
});
|
|
923
886
|
return {
|
|
924
887
|
failure: reflectFailure(run, result, "parse_error", err instanceof Error ? err.message : String(err), true),
|
|
@@ -935,12 +898,13 @@ const NOISE_SUBREASONS = {
|
|
|
935
898
|
* exact content that would be persisted, then mint. A judge pass is staged only
|
|
936
899
|
* when the body is unchanged: a body edit the judge passes, or one made with the
|
|
937
900
|
* gate off, waits for review. Size-flagged or truncation-leaking content skips
|
|
938
|
-
* the judge and waits for review.
|
|
901
|
+
* the judge and waits for review. A revision with a deterministic defect is
|
|
902
|
+
* refused before the judge runs, whether or not the gate is on.
|
|
939
903
|
*/
|
|
940
904
|
async function finalizeReflectProposal(args) {
|
|
941
905
|
const { run, assetContent, result, judge, feedback } = args;
|
|
942
906
|
const { options } = run;
|
|
943
|
-
const telemetry =
|
|
907
|
+
const telemetry = reflectTelemetry(result) ?? {};
|
|
944
908
|
const sanitized = sanitizeReflectPayload({ content: args.payload.content, ...(args.payload.frontmatter ? { frontmatter: args.payload.frontmatter } : {}) }, assetContent, args.payload.ref);
|
|
945
909
|
const payload = {
|
|
946
910
|
...args.payload,
|
|
@@ -981,11 +945,16 @@ async function finalizeReflectProposal(args) {
|
|
|
981
945
|
}, options.eventsCtx);
|
|
982
946
|
return reflectFailure(run, result, "quality_rejected", message, false);
|
|
983
947
|
};
|
|
948
|
+
// A defect no judge needs to weigh is refused before any judge call, whether or not the gate is on.
|
|
949
|
+
const defect = assetContent === undefined ? undefined : findReflectDefect(assetContent, payload.content, options.defectFilter);
|
|
950
|
+
if (defect)
|
|
951
|
+
return refuse(defect, { reflectDefect: defect }, `Reflect proposal refused before the judge: ${defect}`);
|
|
984
952
|
let verdict;
|
|
985
953
|
let judgeFailed = false;
|
|
986
954
|
if (judged) {
|
|
987
955
|
verdict = await runReflectQualityJudge(run.config, payload.content, assetContent ?? "", feedback, options.chat, {
|
|
988
956
|
runnerSelectionFrozen: true,
|
|
957
|
+
ref: payload.ref,
|
|
989
958
|
...(judge.runner ? { llmRunner: judge.runner } : {}),
|
|
990
959
|
...(Object.hasOwn(options, "timeoutMs") ? { timeoutMs: options.timeoutMs } : {}),
|
|
991
960
|
...(options.signal ? { signal: options.signal } : {}),
|
|
@@ -1114,8 +1083,6 @@ export async function renderReflectPromptPreview(options) {
|
|
|
1114
1083
|
assetContent: source.assetContent,
|
|
1115
1084
|
sources,
|
|
1116
1085
|
runnerSpec,
|
|
1117
|
-
// The same tmp-path shape a dispatch would use; never written.
|
|
1118
|
-
draftFilePath: runnerSupportsFileWrite(runnerSpec) ? synthesizeReflectDraftPath(ref) : undefined,
|
|
1119
1086
|
priorDraft: undefined,
|
|
1120
1087
|
});
|
|
1121
1088
|
return { ref, prompt, engine: engineName, engineKind: runnerSpec.kind };
|
|
@@ -1143,7 +1110,7 @@ export async function akmReflect(options = {}) {
|
|
|
1143
1110
|
judgeRunner = runnerSpec;
|
|
1144
1111
|
}
|
|
1145
1112
|
else {
|
|
1146
|
-
const resolved =
|
|
1113
|
+
const resolved = resolveImproveExecution({ config, processName: "reflect_proposal_quality-judge" });
|
|
1147
1114
|
if (resolved)
|
|
1148
1115
|
notices.add(resolved.notices);
|
|
1149
1116
|
judgeRunner = resolved?.runner;
|
|
@@ -1151,7 +1118,7 @@ export async function akmReflect(options = {}) {
|
|
|
1151
1118
|
}
|
|
1152
1119
|
const skippedNoJudge = judgeWanted && !judgeRunner;
|
|
1153
1120
|
if (skippedNoJudge) {
|
|
1154
|
-
warnOnce("reflect-quality-gate-no-judge", "Reflect proposal quality gate has no
|
|
1121
|
+
warnOnce("reflect-quality-gate-no-judge", "Reflect proposal quality gate has no engine configured to judge proposals (set defaults.llmEngine). Skipping the gate for this run; the proposal is queued for human review instead.");
|
|
1155
1122
|
}
|
|
1156
1123
|
preflightReflectDispatch(runnerSpec, notices.add);
|
|
1157
1124
|
if (judgeRunner && judgeRunner !== runnerSpec)
|
|
@@ -1162,13 +1129,11 @@ export async function akmReflect(options = {}) {
|
|
|
1162
1129
|
...(Object.keys(agentEnv).length > 0 ? { env: agentEnv } : {}),
|
|
1163
1130
|
...(options.runAgentOptions ?? {}),
|
|
1164
1131
|
});
|
|
1165
|
-
const draftPaths = [];
|
|
1166
1132
|
let result;
|
|
1167
1133
|
let payload;
|
|
1168
1134
|
try {
|
|
1169
|
-
|
|
1135
|
+
result = await runReflectRefineIterations({ run, parsedRef, assetContent, sources, agentEnv });
|
|
1170
1136
|
emitInvoked();
|
|
1171
|
-
result = iterated.result;
|
|
1172
1137
|
if (!result.ok) {
|
|
1173
1138
|
if (isEnoentFailure(result)) {
|
|
1174
1139
|
emitFailed("spawn_failed", "enoent", options.ref, {
|
|
@@ -1191,11 +1156,11 @@ export async function akmReflect(options = {}) {
|
|
|
1191
1156
|
};
|
|
1192
1157
|
emitFailed(envelope.reason, envelope.reason === "parse_error" ? "parse_error" : "agent_crash", options.ref, {
|
|
1193
1158
|
...(envelope.exitCode !== null ? { exitCode: envelope.exitCode } : {}),
|
|
1194
|
-
...(
|
|
1159
|
+
...(reflectTelemetry(result) ?? {}),
|
|
1195
1160
|
});
|
|
1196
1161
|
return { ...envelope, ...notices.fields() };
|
|
1197
1162
|
}
|
|
1198
|
-
const resolved = resolveReflectPayload(run, result
|
|
1163
|
+
const resolved = resolveReflectPayload(run, result);
|
|
1199
1164
|
if ("failure" in resolved)
|
|
1200
1165
|
return resolved.failure;
|
|
1201
1166
|
payload = resolved.payload;
|
|
@@ -1205,43 +1170,11 @@ export async function akmReflect(options = {}) {
|
|
|
1205
1170
|
emitInvoked();
|
|
1206
1171
|
throw error;
|
|
1207
1172
|
}
|
|
1208
|
-
finally {
|
|
1209
|
-
for (const draftPath of draftPaths) {
|
|
1210
|
-
try {
|
|
1211
|
-
if (fs.existsSync(draftPath))
|
|
1212
|
-
fs.unlinkSync(draftPath);
|
|
1213
|
-
}
|
|
1214
|
-
catch {
|
|
1215
|
-
// best-effort
|
|
1216
|
-
}
|
|
1217
|
-
}
|
|
1218
|
-
}
|
|
1219
1173
|
const unsafeContent = generatedContentRejection(payload.content, redactSensitiveText(payload.content, sensitiveValues));
|
|
1220
1174
|
if (unsafeContent) {
|
|
1221
1175
|
emitFailed("parse_error", "parse_error", options.ref, exitCodeMeta(result));
|
|
1222
1176
|
return reflectFailure(run, result, "parse_error", unsafeContent, false);
|
|
1223
1177
|
}
|
|
1224
|
-
// A retargeted proposal is refused (malformed refs are left to proposal validation).
|
|
1225
|
-
if (options.ref) {
|
|
1226
|
-
let retargeted = false;
|
|
1227
|
-
try {
|
|
1228
|
-
const expected = parseRefInput(options.ref);
|
|
1229
|
-
const actual = parseRefInput(payload.ref);
|
|
1230
|
-
retargeted = expected.type !== actual.type || expected.name !== actual.name;
|
|
1231
|
-
}
|
|
1232
|
-
catch {
|
|
1233
|
-
retargeted = false;
|
|
1234
|
-
}
|
|
1235
|
-
if (retargeted) {
|
|
1236
|
-
emitFailed("parse_error", "ref_mismatch", options.ref, {
|
|
1237
|
-
expectedRef: options.ref,
|
|
1238
|
-
actualRef: payload.ref,
|
|
1239
|
-
...exitCodeMeta(result),
|
|
1240
|
-
...(reflectLlmTelemetry(result) ?? {}),
|
|
1241
|
-
});
|
|
1242
|
-
return reflectFailure(run, result, "parse_error", `Agent retargeted proposal: expected ref "${options.ref}" but got "${payload.ref}". Proposal rejected to prevent silent ref hallucination.`, true);
|
|
1243
|
-
}
|
|
1244
|
-
}
|
|
1245
1178
|
return finalizeReflectProposal({
|
|
1246
1179
|
run,
|
|
1247
1180
|
payload,
|
|
@@ -32,6 +32,10 @@ const GRADE_SCHEMA = {
|
|
|
32
32
|
additionalProperties: false,
|
|
33
33
|
properties: { grade: { type: "integer", minimum: 0, maximum: 3 }, reason: { type: "string" } },
|
|
34
34
|
};
|
|
35
|
+
function parseGrade(raw) {
|
|
36
|
+
const grade = parseEmbeddedJsonResponse(raw)?.grade;
|
|
37
|
+
return typeof grade === "number" && Number.isInteger(grade) && grade >= 0 && grade <= 3 ? grade : undefined;
|
|
38
|
+
}
|
|
35
39
|
/** Up to five distinct task queries, in the given order, whitespace collapsed. */
|
|
36
40
|
export function usableRetrievalQueries(raw) {
|
|
37
41
|
const out = [];
|
|
@@ -100,10 +104,11 @@ export async function runRetrievalRegressionGate(args) {
|
|
|
100
104
|
...(args.signal ? { signal: args.signal } : {}),
|
|
101
105
|
...(args.chat ? { chat: args.chat } : {}),
|
|
102
106
|
},
|
|
107
|
+
parse: parseGrade,
|
|
103
108
|
...(args.onNotices ? { onNotices: args.onNotices } : {}),
|
|
104
109
|
});
|
|
105
|
-
const grade = outcome.ok ?
|
|
106
|
-
if (
|
|
110
|
+
const grade = outcome.ok ? parseGrade(outcome.raw) : undefined;
|
|
111
|
+
if (grade === undefined) {
|
|
107
112
|
return {
|
|
108
113
|
pass: false,
|
|
109
114
|
queries: args.queries.length,
|