akm-cli 0.9.23 → 0.9.25-alpha.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +339 -0
- package/dist/commands/health/checks.js +10 -11
- package/dist/commands/improve/consolidate/pair-pass.js +3 -2
- package/dist/commands/improve/consolidate.js +3 -2
- package/dist/commands/improve/execution.js +4 -10
- package/dist/commands/improve/extract-prompt.js +4 -4
- package/dist/commands/improve/extract.js +10 -13
- package/dist/commands/improve/improve-cli.js +32 -33
- package/dist/commands/improve/improve-strategies.js +49 -43
- package/dist/commands/improve/improve-usage-report.js +8 -17
- package/dist/commands/improve/preparation.js +3 -1
- package/dist/commands/improve/reflect.js +132 -177
- package/dist/commands/improve/stage.js +69 -18
- package/dist/commands/proposal/drain.js +19 -7
- package/dist/commands/proposal/proposal-cli.js +1 -5
- package/dist/commands/proposal/propose-cli.js +2 -2
- package/dist/commands/proposal/propose.js +80 -83
- package/dist/commands/remember.js +3 -3
- package/dist/commands/sources/schema-repair.js +1 -1
- package/dist/core/config/config-schema.js +37 -60
- package/dist/core/config/engine-semantics.js +15 -11
- package/dist/core/config/schema/engines.js +33 -15
- package/dist/core/config/schema/improve-processes.js +2 -2
- package/dist/core/improve-result.js +3 -3
- package/dist/core/structured.js +10 -0
- package/dist/execution/source.js +14 -0
- package/dist/indexer/passes/memory-inference.js +2 -1
- package/dist/integrations/agent/builder-shared.js +15 -0
- package/dist/integrations/agent/config.js +2 -0
- package/dist/integrations/agent/engine-resolution.js +16 -31
- package/dist/integrations/agent/execution.js +46 -21
- package/dist/integrations/agent/index.js +1 -1
- package/dist/integrations/agent/model-map.js +16 -15
- package/dist/integrations/agent/prompts.js +53 -76
- package/dist/integrations/agent/request-lowering.js +20 -9
- package/dist/integrations/agent/runner-dispatch.js +103 -4
- package/dist/integrations/agent/runner.js +8 -2
- package/dist/integrations/harnesses/aider/agent-builder.js +11 -23
- package/dist/integrations/harnesses/aider/index.js +0 -5
- package/dist/integrations/harnesses/amazonq/agent-builder.js +10 -21
- package/dist/integrations/harnesses/amazonq/index.js +0 -5
- package/dist/integrations/harnesses/claude/agent-builder.js +61 -38
- package/dist/integrations/harnesses/claude/index.js +0 -14
- package/dist/integrations/harnesses/codex/agent-builder.js +4 -4
- package/dist/integrations/harnesses/codex/index.js +0 -4
- package/dist/integrations/harnesses/copilot/agent-builder.js +4 -13
- package/dist/integrations/harnesses/copilot/index.js +2 -7
- package/dist/integrations/harnesses/gemini/agent-builder.js +4 -13
- package/dist/integrations/harnesses/gemini/index.js +0 -5
- package/dist/integrations/harnesses/ids.js +18 -10
- package/dist/integrations/harnesses/opencode/agent-builder.js +52 -10
- package/dist/integrations/harnesses/opencode/index.js +0 -8
- package/dist/integrations/harnesses/opencode/model-config.js +80 -0
- package/dist/integrations/harnesses/opencode/model-work-agent.js +80 -0
- package/dist/integrations/harnesses/opencode-sdk/harness.js +6 -7
- package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +168 -24
- package/dist/integrations/harnesses/openhands/agent-builder.js +8 -21
- package/dist/integrations/harnesses/openhands/index.js +0 -5
- package/dist/integrations/harnesses/pi/agent-builder.js +8 -21
- package/dist/integrations/harnesses/pi/index.js +0 -5
- package/dist/llm/client.js +5 -0
- package/dist/llm/feature-gate.js +10 -4
- package/dist/llm/index-passes.js +3 -6
- package/dist/llm/memory-infer.js +6 -5
- package/dist/llm/structured-call.js +33 -11
- package/dist/scripts/akm-migrate-node.js +298 -204
- package/dist/scripts/akm-migrate.js +298 -204
- package/dist/workflows/exec/step-work.js +6 -5
- package/dist/workflows/freeze/step-values.js +1 -1
- package/docs/reference/cli.md +34 -14
- package/docs/reference/configuration.md +171 -12
- package/docs/reference/workflow-schema.md +13 -9
- package/package.json +1 -1
- package/schemas/akm-config.json +36 -0
|
@@ -11,7 +11,6 @@
|
|
|
11
11
|
* pre-dispatch refusals still emit both).
|
|
12
12
|
*/
|
|
13
13
|
import fs from "node:fs";
|
|
14
|
-
import os from "node:os";
|
|
15
14
|
import path from "node:path";
|
|
16
15
|
import { assembleAssetFromString, serializeFrontmatter } from "../../core/asset/asset-serialize.js";
|
|
17
16
|
import { parseFrontmatter } from "../../core/asset/frontmatter.js";
|
|
@@ -27,24 +26,25 @@ import { parseEmbeddedJsonResponse } from "../../core/parse.js";
|
|
|
27
26
|
import { redactSensitiveText } from "../../core/redaction.js";
|
|
28
27
|
import { resolveStandardsContext } from "../../core/standards/resolve-standards-context.js";
|
|
29
28
|
import { warn, warnOnce } from "../../core/warn.js";
|
|
29
|
+
import { MODEL_WORK_TOOLS } from "../../execution/source.js";
|
|
30
30
|
import { lookup } from "../../indexer/indexer.js";
|
|
31
|
-
import {
|
|
31
|
+
import { DEFAULT_MODEL_WORK_TIMEOUT_MS } from "../../integrations/agent/config.js";
|
|
32
32
|
import { fallbackAnnouncement, NO_ENGINE_MESSAGE_SUFFIX, NO_ENGINE_REMEDY, withEngineFallback, } from "../../integrations/agent/engine-fallback.js";
|
|
33
33
|
import { buildExecution, resolveExecution } from "../../integrations/agent/execution.js";
|
|
34
|
-
import { buildReflectOutputRepairPrompt, buildReflectPrompt,
|
|
35
|
-
import { runnerIsLlm
|
|
36
|
-
import { assertRunnerCredentials, collectDispatchSensitiveValues,
|
|
34
|
+
import { buildReflectOutputRepairPrompt, buildReflectPrompt, parseAgentProposalPayload, REFLECT_CONTENT_CAP, REFLECT_TRUNCATION_MARKER, } from "../../integrations/agent/prompts.js";
|
|
35
|
+
import { runnerIsLlm } from "../../integrations/agent/runner.js";
|
|
36
|
+
import { assertRunnerCredentials, collectDispatchSensitiveValues, } from "../../integrations/agent/runner-dispatch.js";
|
|
37
37
|
import { isJsonSchemaKnownUnsupported, LlmCallError } from "../../llm/client.js";
|
|
38
38
|
import { baseFailureFields, enoentHintMessage, isEnoentFailure } from "../agent/agent-support.js";
|
|
39
39
|
import { checkReflectSize, isValidDescription } from "../proposal/validators/proposal-quality-validators.js";
|
|
40
40
|
import { CHARS_PER_TOKEN, DEFAULT_CONTEXT_LENGTH_TOKENS } from "./consolidate/chunking.js";
|
|
41
41
|
import { deriveLessonRef } from "./distill.js";
|
|
42
42
|
import { findAssetFilePath } from "./eligibility.js";
|
|
43
|
-
import {
|
|
43
|
+
import { resolveImproveExecution } from "./execution.js";
|
|
44
44
|
import { recordLedgerAttempt } from "./ledger.js";
|
|
45
45
|
import { classifyReflectChange, splitFrontmatter } from "./reflect-noise.js";
|
|
46
46
|
import { loadRetrievalQueries, runRetrievalRegressionGate } from "./retrieval-gate.js";
|
|
47
|
-
import {
|
|
47
|
+
import { callStageOnce, errMessage, mintProposal, noticeSet, rejectedProposalContext, resolveQualityGateJudge, runReflectQualityJudge, } from "./stage.js";
|
|
48
48
|
const MAX_FEEDBACK_LINES = 10;
|
|
49
49
|
const MAX_GLOBAL_FEEDBACK_LINES = 20;
|
|
50
50
|
function readOnlyEventsContext(ctx) {
|
|
@@ -83,16 +83,6 @@ export const REFLECT_ALLOWED_TYPES = new Set([
|
|
|
83
83
|
const REFLECT_REFUSED_TYPES = new Set(["secret"]);
|
|
84
84
|
/** Identity fields the model may never change (a renamed `name` breaks ref resolution). */
|
|
85
85
|
const PROTECTED_FRONTMATTER_FIELDS = new Set(["name", "ref", "id", "slug", "type"]);
|
|
86
|
-
/**
|
|
87
|
-
* A fresh tmp path per iteration for the agent/SDK file-write contract (long
|
|
88
|
-
* bodies are written to a file instead of fenced JSON on stdout). The direct
|
|
89
|
-
* LLM runner has no filesystem and never gets one.
|
|
90
|
-
*/
|
|
91
|
-
function synthesizeReflectDraftPath(ref) {
|
|
92
|
-
const safeRef = (ref ?? "no-ref").replace(/[^a-z0-9_-]/gi, "_");
|
|
93
|
-
const rand = Math.random().toString(36).slice(2, 8);
|
|
94
|
-
return path.join(os.tmpdir(), `akm-reflect-${safeRef}-${Date.now()}-${rand}.md`);
|
|
95
|
-
}
|
|
96
86
|
/** Lesson lint findings for the prompt: a concrete starting point for the revision. */
|
|
97
87
|
function buildSchemaHints(type, content) {
|
|
98
88
|
if (!content || type !== "lesson")
|
|
@@ -328,7 +318,12 @@ export function sanitizeReflectPayload(payload, sourceContent, targetRef) {
|
|
|
328
318
|
...(truncationMarkerLeaked ? { truncationMarkerLeaked } : {}),
|
|
329
319
|
};
|
|
330
320
|
}
|
|
331
|
-
// ──
|
|
321
|
+
// ── Output contract ──────────────────────────────────────────────────────────
|
|
322
|
+
//
|
|
323
|
+
// Every engine kind is asked for the same reply: the JSON object of
|
|
324
|
+
// REFLECT_JSON_SCHEMA, or the framed-markdown frame for an LLM engine that
|
|
325
|
+
// rejects JSON Schema. The reply is validated by the parse functions below and
|
|
326
|
+
// repaired once, whatever engine produced it.
|
|
332
327
|
const REFLECT_FRONTMATTER_PATCH_JSON_SCHEMA = {
|
|
333
328
|
type: "object",
|
|
334
329
|
required: ["description", "when_to_use"],
|
|
@@ -366,7 +361,7 @@ const REFLECT_UNSCOPED_JSON_SCHEMA = {
|
|
|
366
361
|
},
|
|
367
362
|
};
|
|
368
363
|
/**
|
|
369
|
-
* Frame for JSON Schema unless the connection disabled it or already proved
|
|
364
|
+
* Frame for JSON Schema unless the LLM connection disabled it or already proved
|
|
370
365
|
* this process that it rejects it (the transport retries plain text on a 4xx).
|
|
371
366
|
*/
|
|
372
367
|
function wantsJsonSchemaOutput(connection) {
|
|
@@ -379,7 +374,7 @@ function parsedRecord(result) {
|
|
|
379
374
|
? result.parsed
|
|
380
375
|
: undefined;
|
|
381
376
|
}
|
|
382
|
-
function
|
|
377
|
+
function reflectTelemetry(result) {
|
|
383
378
|
const parsed = parsedRecord(result);
|
|
384
379
|
if (!parsed || (parsed.outputMode !== "json_schema" && parsed.outputMode !== "framed_markdown"))
|
|
385
380
|
return undefined;
|
|
@@ -484,18 +479,22 @@ function parseFramedReflectOutput(raw, targetRef) {
|
|
|
484
479
|
return { ref, content, confidence, ...(frontmatter ? { frontmatter } : {}) };
|
|
485
480
|
}
|
|
486
481
|
/**
|
|
487
|
-
* One reflect iteration
|
|
488
|
-
*
|
|
489
|
-
*
|
|
482
|
+
* One reflect iteration on any engine, as an agent-shaped result (errors
|
|
483
|
+
* captured, never thrown except configuration). Every engine kind is sent the
|
|
484
|
+
* same request, with the reply's schema as the output schema, and its reply is
|
|
485
|
+
* held to the same contract: an unparseable reply gets one repair turn within
|
|
486
|
+
* the original deadline, then fails. A failed agent or SDK dispatch is reported
|
|
487
|
+
* as it ran, with its exit code and stderr; an LLM call has only its message.
|
|
490
488
|
*/
|
|
491
|
-
export async function
|
|
489
|
+
export async function runReflectIteration(opts) {
|
|
492
490
|
const start = Date.now();
|
|
493
491
|
let repairAttempts = 0;
|
|
492
|
+
// Model work is bounded: with no timeout from the caller or the runner, the default.
|
|
494
493
|
const configuredTimeout = Object.hasOwn(opts, "timeoutMs")
|
|
495
494
|
? (opts.timeoutMs ?? null)
|
|
496
495
|
: Object.hasOwn(opts.runner, "timeoutMs")
|
|
497
496
|
? (opts.runner.timeoutMs ?? null)
|
|
498
|
-
:
|
|
497
|
+
: DEFAULT_MODEL_WORK_TIMEOUT_MS;
|
|
499
498
|
const deadline = typeof configuredTimeout === "number" ? start + configuredTimeout : undefined;
|
|
500
499
|
const messages = [{ role: "user", content: opts.prompt ?? "" }];
|
|
501
500
|
if (opts.priorDraft !== undefined && opts.iteration > 0) {
|
|
@@ -505,7 +504,7 @@ export async function runReflectViaLlm(opts) {
|
|
|
505
504
|
? parseSchemaReflectOutput(raw, opts.targetRef)
|
|
506
505
|
: parseFramedReflectOutput(raw, opts.targetRef);
|
|
507
506
|
const failure = (err, reason, stdout = "", exitCode = 1) => {
|
|
508
|
-
const msg =
|
|
507
|
+
const msg = errMessage(err);
|
|
509
508
|
return {
|
|
510
509
|
ok: false,
|
|
511
510
|
stdout,
|
|
@@ -517,8 +516,21 @@ export async function runReflectViaLlm(opts) {
|
|
|
517
516
|
parsed: { outputMode: opts.outputMode, repairAttempts },
|
|
518
517
|
};
|
|
519
518
|
};
|
|
519
|
+
// A reply that broke the contract. An agent or SDK engine's failure names the engine; an LLM
|
|
520
|
+
// reply keeps its message, because improve feeds a failed reflect's `error` into the next
|
|
521
|
+
// prompts as a pattern to avoid, and rewording it would change those requests.
|
|
522
|
+
const invalidReply = (err, reply) => {
|
|
523
|
+
if (runnerIsLlm(opts.runner))
|
|
524
|
+
return failure(err, "parse_error", reply, 0);
|
|
525
|
+
const attempts = repairAttempts + 1;
|
|
526
|
+
const message = `Engine "${opts.runner.engine}" reply was not a valid reflect proposal after ${attempts} attempt${attempts === 1 ? "" : "s"}: ${errMessage(err)}`;
|
|
527
|
+
return failure(new Error(message), "parse_error", reply, 0);
|
|
528
|
+
};
|
|
529
|
+
// The result of the dispatch that failed, kept for an agent or SDK engine.
|
|
530
|
+
let dispatched;
|
|
531
|
+
// Reflect parses and repairs its own reply (the repair turn below), so one dispatch, unvalidated.
|
|
520
532
|
const call = async (callMessages, repairTimeoutMs) => {
|
|
521
|
-
const outcome = await
|
|
533
|
+
const outcome = await callStageOnce({
|
|
522
534
|
feature: "reflect_proposal",
|
|
523
535
|
runner: opts.runner,
|
|
524
536
|
prompt: callMessages.at(-1)?.content ?? "",
|
|
@@ -535,13 +547,18 @@ export async function runReflectViaLlm(opts) {
|
|
|
535
547
|
// Visible chain-of-thought can exhaust the output before the envelope.
|
|
536
548
|
enableThinking: false,
|
|
537
549
|
...(opts.chat ? { chat: opts.chat } : {}),
|
|
550
|
+
...(opts.runSdk ? { runSdk: opts.runSdk } : {}),
|
|
551
|
+
...(opts.runOptions ? { runOptions: opts.runOptions } : {}),
|
|
538
552
|
},
|
|
553
|
+
...(opts.environment ? { current: { environment: opts.environment } } : {}),
|
|
539
554
|
...(opts.onNotices ? { onNotices: opts.onNotices } : {}),
|
|
540
555
|
});
|
|
541
556
|
if (!outcome.ok) {
|
|
557
|
+
if (!runnerIsLlm(opts.runner))
|
|
558
|
+
dispatched = outcome.result;
|
|
542
559
|
throw outcome.reason === "timeout"
|
|
543
560
|
? new LlmCallError(outcome.error ?? "timeout", "timeout")
|
|
544
|
-
: new Error(outcome.error ?? "
|
|
561
|
+
: new Error(outcome.error ?? "dispatch failed");
|
|
545
562
|
}
|
|
546
563
|
return outcome.raw;
|
|
547
564
|
};
|
|
@@ -556,7 +573,7 @@ export async function runReflectViaLlm(opts) {
|
|
|
556
573
|
}
|
|
557
574
|
catch (err) {
|
|
558
575
|
if (opts.allowRepair === false)
|
|
559
|
-
return
|
|
576
|
+
return invalidReply(err, stdout);
|
|
560
577
|
if (opts.signal?.aborted)
|
|
561
578
|
return failure(new Error("Reflect request aborted"), "aborted", stdout);
|
|
562
579
|
const remaining = deadline === undefined ? undefined : deadline - Date.now();
|
|
@@ -570,7 +587,7 @@ export async function runReflectViaLlm(opts) {
|
|
|
570
587
|
payload = parse(acceptedOutput);
|
|
571
588
|
}
|
|
572
589
|
catch (repairErr) {
|
|
573
|
-
return
|
|
590
|
+
return invalidReply(repairErr, acceptedOutput);
|
|
574
591
|
}
|
|
575
592
|
}
|
|
576
593
|
return {
|
|
@@ -585,6 +602,8 @@ export async function runReflectViaLlm(opts) {
|
|
|
585
602
|
catch (err) {
|
|
586
603
|
if (err instanceof ConfigError)
|
|
587
604
|
throw err;
|
|
605
|
+
if (dispatched)
|
|
606
|
+
return { ...dispatched, parsed: { outputMode: opts.outputMode, repairAttempts } };
|
|
588
607
|
const reason = opts.signal?.aborted
|
|
589
608
|
? "aborted"
|
|
590
609
|
: err instanceof LlmCallError && err.code === "timeout"
|
|
@@ -685,8 +704,9 @@ async function resolveReflectSource(options, stash, emitFailed) {
|
|
|
685
704
|
}
|
|
686
705
|
/**
|
|
687
706
|
* The single engine for this invocation: `--engine`, the improve strategy's
|
|
688
|
-
*
|
|
689
|
-
*
|
|
707
|
+
* reflect process, or `defaults.engine` (announced when it falls back to the
|
|
708
|
+
* SDK binary). Whatever its kind, reflect runs it under the model-work tool
|
|
709
|
+
* policy.
|
|
690
710
|
*/
|
|
691
711
|
function resolveReflectRunner(options) {
|
|
692
712
|
const config = options.config ?? loadConfig();
|
|
@@ -700,14 +720,14 @@ function resolveReflectRunner(options) {
|
|
|
700
720
|
lowered = lower({ content: "reflect engine selection", config, current: { engine: options.engine } });
|
|
701
721
|
}
|
|
702
722
|
else if (options.improveProfile) {
|
|
703
|
-
const resolved =
|
|
723
|
+
const resolved = resolveImproveExecution({
|
|
704
724
|
config,
|
|
705
725
|
profile: activeStrategy,
|
|
706
726
|
process: activeStrategy?.processes?.reflect,
|
|
707
727
|
processName: "reflect",
|
|
708
728
|
});
|
|
709
729
|
if (!resolved) {
|
|
710
|
-
throw new ConfigError("Reflect requires an
|
|
730
|
+
throw new ConfigError("Reflect requires an engine for the active improve strategy.", "LLM_NOT_CONFIGURED", "Set defaults.llmEngine or improve.strategies.<name>.processes.reflect.engine.");
|
|
711
731
|
}
|
|
712
732
|
lowered = resolved;
|
|
713
733
|
}
|
|
@@ -723,19 +743,20 @@ function resolveReflectRunner(options) {
|
|
|
723
743
|
lowered = lower({ content: "reflect engine selection", config });
|
|
724
744
|
}
|
|
725
745
|
const runnerSpec = lowered.runner;
|
|
726
|
-
if (options.eventSource === "improve" && !runnerIsLlm(runnerSpec)) {
|
|
727
|
-
throw new ConfigError(`Unattended improve requires an LLM engine for reflect; engine "${runnerSpec.engine ?? options.engine ?? "unknown"}" is tool-capable.`, "INVALID_CONFIG_FILE", "Set defaults.llmEngine or improve.strategies.<name>.processes.reflect.engine to an LLM engine.");
|
|
728
|
-
}
|
|
729
746
|
const engineName = runnerSpec.engine ?? options.engine;
|
|
730
747
|
if (!engineName)
|
|
731
748
|
throw new ConfigError("Reflect requires a named engine.", "INVALID_CONFIG_FILE");
|
|
732
749
|
return { config, activeStrategy, runnerSpec, engineName, notices: lowered.notices };
|
|
733
750
|
}
|
|
734
|
-
/**
|
|
751
|
+
/**
|
|
752
|
+
* Lower a runner under the model-work tool policy and check its credentials,
|
|
753
|
+
* so a bad transport, or one that cannot confine the policy, fails before any work.
|
|
754
|
+
*/
|
|
735
755
|
function preflightReflectDispatch(runnerSpec, onNotices) {
|
|
736
756
|
const prepared = resolveExecution({
|
|
737
757
|
content: "Validate reflect operation transport before dispatch.",
|
|
738
758
|
runner: runnerSpec,
|
|
759
|
+
current: { tools: MODEL_WORK_TOOLS },
|
|
739
760
|
});
|
|
740
761
|
const lowered = buildExecution(prepared.request, prepared.runner);
|
|
741
762
|
onNotices(lowered.notices);
|
|
@@ -767,13 +788,10 @@ async function gatherReflectPromptSources(options, stash, parsedRef, assetConten
|
|
|
767
788
|
}
|
|
768
789
|
/** The exact prompt reflect sends, shared by dispatch and `--show-prompt`. */
|
|
769
790
|
function buildReflectPromptText(args) {
|
|
770
|
-
const { options, parsedRef, assetContent, sources, runnerSpec,
|
|
791
|
+
const { options, parsedRef, assetContent, sources, runnerSpec, priorDraft } = args;
|
|
771
792
|
const { feedback, schemaHints, relatedLessons, rejectedProposals, standardsContext } = sources;
|
|
772
|
-
|
|
773
|
-
|
|
774
|
-
? "json_schema"
|
|
775
|
-
: "framed_markdown"
|
|
776
|
-
: undefined;
|
|
793
|
+
// An LLM engine that rejects JSON Schema gets the framed contract; every other engine gets the JSON object.
|
|
794
|
+
const outputMode = runnerIsLlm(runnerSpec) && !wantsJsonSchemaOutput(runnerSpec.connection) ? "framed_markdown" : "json_schema";
|
|
777
795
|
const input = {
|
|
778
796
|
...(options.ref ? { ref: options.ref } : {}),
|
|
779
797
|
...(parsedRef?.type ? { type: parsedRef.type } : {}),
|
|
@@ -787,89 +805,65 @@ function buildReflectPromptText(args) {
|
|
|
787
805
|
...(options.avoidPatterns && options.avoidPatterns.length > 0 ? { avoidPatterns: options.avoidPatterns } : {}),
|
|
788
806
|
...(rejectedProposals.length > 0 ? { rejectedProposals } : {}),
|
|
789
807
|
...(priorDraft !== undefined ? { priorDraft } : {}),
|
|
790
|
-
|
|
791
|
-
...(outputMode ? { outputMode } : {}),
|
|
808
|
+
outputMode,
|
|
792
809
|
};
|
|
793
810
|
const contentBudgetChars = computeReflectContentBudgetChars(input, runnerSpec);
|
|
794
811
|
const { prompt } = buildReflectPrompt({
|
|
795
812
|
...input,
|
|
796
813
|
...(contentBudgetChars !== undefined ? { contentBudgetChars } : {}),
|
|
797
814
|
});
|
|
798
|
-
return { prompt,
|
|
815
|
+
return { prompt, outputMode };
|
|
799
816
|
}
|
|
800
817
|
/**
|
|
801
818
|
* Dispatch with the optional self-refine loop: up to `maxRefineIters` passes,
|
|
802
|
-
* each critiquing the prior draft, stopping early on an unchanged draft.
|
|
803
|
-
*
|
|
819
|
+
* each critiquing the prior draft, stopping early on an unchanged draft. Every
|
|
820
|
+
* engine kind runs the same iteration, and the repair budget is shared across
|
|
821
|
+
* passes. Under the model-work tool policy an agent can edit only its own
|
|
822
|
+
* scratch directory, which is gone once it returns, so it returns the proposal
|
|
823
|
+
* as its reply, as an LLM does.
|
|
804
824
|
*/
|
|
805
825
|
async function runReflectRefineIterations(args) {
|
|
806
|
-
const { run, parsedRef, assetContent, sources, agentEnv
|
|
826
|
+
const { run, parsedRef, assetContent, sources, agentEnv } = args;
|
|
807
827
|
const { options, runnerSpec } = run;
|
|
808
828
|
const maxRefineIters = Math.max(1, options.maxRefineIters ?? 1);
|
|
809
|
-
const canWriteFile = runnerSupportsFileWrite(runnerSpec);
|
|
810
829
|
let result = {};
|
|
811
830
|
let priorDraft;
|
|
812
|
-
let lastDraftPath;
|
|
813
831
|
let repairAttempts = 0;
|
|
832
|
+
// Only an agent or SDK engine runs a child process, with an environment and spawn or SDK seams.
|
|
833
|
+
const childProcess = runnerIsLlm(runnerSpec)
|
|
834
|
+
? {}
|
|
835
|
+
: {
|
|
836
|
+
...(Object.keys(agentEnv).length > 0 ? { environment: agentEnv } : {}),
|
|
837
|
+
...(options.runSdk ? { runSdk: options.runSdk } : {}),
|
|
838
|
+
...(options.runAgentOptions ? { runOptions: options.runAgentOptions } : {}),
|
|
839
|
+
};
|
|
814
840
|
for (let iter = 0; iter < maxRefineIters; iter++) {
|
|
815
|
-
const draftFilePath = canWriteFile ? synthesizeReflectDraftPath(options.ref) : undefined;
|
|
816
|
-
if (draftFilePath) {
|
|
817
|
-
draftPaths.push(draftFilePath);
|
|
818
|
-
lastDraftPath = draftFilePath;
|
|
819
|
-
}
|
|
820
841
|
const { prompt, outputMode } = buildReflectPromptText({
|
|
821
842
|
options,
|
|
822
843
|
parsedRef,
|
|
823
844
|
assetContent,
|
|
824
845
|
sources,
|
|
825
846
|
runnerSpec,
|
|
826
|
-
draftFilePath,
|
|
827
847
|
priorDraft,
|
|
828
848
|
});
|
|
829
|
-
|
|
830
|
-
|
|
831
|
-
|
|
832
|
-
|
|
833
|
-
|
|
834
|
-
|
|
835
|
-
|
|
836
|
-
|
|
837
|
-
|
|
838
|
-
|
|
839
|
-
|
|
840
|
-
|
|
841
|
-
|
|
842
|
-
|
|
843
|
-
|
|
844
|
-
|
|
845
|
-
|
|
846
|
-
|
|
847
|
-
}
|
|
848
|
-
else {
|
|
849
|
-
const conversation = priorDraft !== undefined && iter > 0
|
|
850
|
-
? [
|
|
851
|
-
{ role: "user", content: prompt },
|
|
852
|
-
{ role: "assistant", content: priorDraft },
|
|
853
|
-
]
|
|
854
|
-
: undefined;
|
|
855
|
-
const current = {
|
|
856
|
-
...(Object.hasOwn(options, "timeoutMs") ? { timeout: options.timeoutMs } : {}),
|
|
857
|
-
...(Object.keys(agentEnv).length > 0 ? { environment: agentEnv } : {}),
|
|
858
|
-
};
|
|
859
|
-
const prepared = resolveExecution({
|
|
860
|
-
content: conversation ? REFLECT_CRITIQUE_PROMPT : prompt,
|
|
861
|
-
...(conversation ? { conversation } : {}),
|
|
862
|
-
runner: runnerSpec,
|
|
863
|
-
...(Object.keys(current).length > 0 ? { current } : {}),
|
|
864
|
-
});
|
|
865
|
-
const lowered = buildExecution(prepared.request, prepared.runner);
|
|
866
|
-
run.notices.add(lowered.notices);
|
|
867
|
-
iterResult = await runExecution(lowered, {
|
|
868
|
-
...(options.runSdk ? { runSdk: options.runSdk } : {}),
|
|
869
|
-
runOptions: { ...(options.signal ? { signal: options.signal } : {}), ...(options.runAgentOptions ?? {}) },
|
|
870
|
-
});
|
|
871
|
-
}
|
|
872
|
-
const telemetry = reflectLlmTelemetry(iterResult);
|
|
849
|
+
const iterResult = await runReflectIteration({
|
|
850
|
+
prompt,
|
|
851
|
+
runner: runnerSpec,
|
|
852
|
+
...(Object.hasOwn(options, "timeoutMs") ? { timeoutMs: options.timeoutMs } : {}),
|
|
853
|
+
...(options.signal ? { signal: options.signal } : {}),
|
|
854
|
+
priorDraft,
|
|
855
|
+
iteration: iter,
|
|
856
|
+
...(outputMode === "json_schema"
|
|
857
|
+
? { responseSchema: options.ref ? REFLECT_JSON_SCHEMA : REFLECT_UNSCOPED_JSON_SCHEMA }
|
|
858
|
+
: {}),
|
|
859
|
+
outputMode,
|
|
860
|
+
...(options.ref ? { targetRef: options.ref } : {}),
|
|
861
|
+
allowRepair: repairAttempts === 0,
|
|
862
|
+
...(options.chat ? { chat: options.chat } : {}),
|
|
863
|
+
...childProcess,
|
|
864
|
+
onNotices: run.notices.add,
|
|
865
|
+
});
|
|
866
|
+
const telemetry = reflectTelemetry(iterResult);
|
|
873
867
|
if (telemetry)
|
|
874
868
|
repairAttempts += telemetry.repairAttempts;
|
|
875
869
|
result = telemetry
|
|
@@ -878,47 +872,25 @@ async function runReflectRefineIterations(args) {
|
|
|
878
872
|
if (!result.ok)
|
|
879
873
|
break;
|
|
880
874
|
if (iter < maxRefineIters - 1) {
|
|
881
|
-
const
|
|
882
|
-
const nextDraft = typeof
|
|
875
|
+
const priorFromReply = parsedRecord(result)?.priorDraft;
|
|
876
|
+
const nextDraft = typeof priorFromReply === "string" ? priorFromReply : (result.stdout ?? "");
|
|
883
877
|
if (priorDraft !== undefined && nextDraft === priorDraft)
|
|
884
878
|
break;
|
|
885
879
|
priorDraft = nextDraft;
|
|
886
880
|
}
|
|
887
881
|
}
|
|
888
|
-
return
|
|
882
|
+
return result;
|
|
889
883
|
}
|
|
890
|
-
/**
|
|
891
|
-
|
|
892
|
-
* (file-write contract, `DRAFT_WRITTEN confidence=<n>` on stdout) or the JSON
|
|
893
|
-
* payload on stdout.
|
|
894
|
-
*/
|
|
895
|
-
function resolveReflectPayload(run, result, lastDraftPath, sensitiveValues) {
|
|
884
|
+
/** The proposal payload from a successful run: the JSON payload on stdout. */
|
|
885
|
+
function resolveReflectPayload(run, result) {
|
|
896
886
|
const { options } = run;
|
|
897
|
-
const draftFileExists = lastDraftPath !== undefined && fs.existsSync(lastDraftPath) && fs.statSync(lastDraftPath).size > 0;
|
|
898
|
-
const draftSignaled = /\bDRAFT_WRITTEN\b/.test(result.stdout ?? "");
|
|
899
|
-
if (draftSignaled && lastDraftPath && !draftFileExists) {
|
|
900
|
-
run.emitFailed("parse_error", "draft_missing", options.ref, exitCodeMeta(result));
|
|
901
|
-
return {
|
|
902
|
-
failure: reflectFailure(run, result, "parse_error", `Agent emitted DRAFT_WRITTEN but draft file is missing or empty (${lastDraftPath}). The file-write contract failed; either the agent's file tools are broken or the path was unwritable.`, true),
|
|
903
|
-
};
|
|
904
|
-
}
|
|
905
|
-
if (draftFileExists && lastDraftPath) {
|
|
906
|
-
const draftConfidence = extractDraftConfidence(result.stdout);
|
|
907
|
-
return {
|
|
908
|
-
payload: {
|
|
909
|
-
ref: options.ref ?? "",
|
|
910
|
-
content: redactSensitiveText(fs.readFileSync(lastDraftPath, "utf8"), sensitiveValues),
|
|
911
|
-
...(draftConfidence !== undefined ? { confidence: draftConfidence } : {}),
|
|
912
|
-
},
|
|
913
|
-
};
|
|
914
|
-
}
|
|
915
887
|
try {
|
|
916
888
|
return { payload: parseAgentProposalPayload(result.stdout ?? "") };
|
|
917
889
|
}
|
|
918
890
|
catch (err) {
|
|
919
891
|
run.emitFailed("parse_error", "parse_error", options.ref, {
|
|
920
892
|
...exitCodeMeta(result),
|
|
921
|
-
...(
|
|
893
|
+
...(reflectTelemetry(result) ?? {}),
|
|
922
894
|
});
|
|
923
895
|
return {
|
|
924
896
|
failure: reflectFailure(run, result, "parse_error", err instanceof Error ? err.message : String(err), true),
|
|
@@ -932,13 +904,15 @@ const NOISE_SUBREASONS = {
|
|
|
932
904
|
};
|
|
933
905
|
/**
|
|
934
906
|
* Sanitize, drop a no-op/cosmetic (and optionally low-value) change, judge the
|
|
935
|
-
* exact content that would be persisted, then mint.
|
|
936
|
-
*
|
|
907
|
+
* exact content that would be persisted, then mint. A judge pass is staged only
|
|
908
|
+
* when the body is unchanged: a body edit the judge passes, or one made with the
|
|
909
|
+
* gate off, waits for review. Size-flagged or truncation-leaking content skips
|
|
910
|
+
* the judge and waits for review.
|
|
937
911
|
*/
|
|
938
912
|
async function finalizeReflectProposal(args) {
|
|
939
913
|
const { run, assetContent, result, judge, feedback } = args;
|
|
940
914
|
const { options } = run;
|
|
941
|
-
const telemetry =
|
|
915
|
+
const telemetry = reflectTelemetry(result) ?? {};
|
|
942
916
|
const sanitized = sanitizeReflectPayload({ content: args.payload.content, ...(args.payload.frontmatter ? { frontmatter: args.payload.frontmatter } : {}) }, assetContent, args.payload.ref);
|
|
943
917
|
const payload = {
|
|
944
918
|
...args.payload,
|
|
@@ -980,6 +954,7 @@ async function finalizeReflectProposal(args) {
|
|
|
980
954
|
return reflectFailure(run, result, "quality_rejected", message, false);
|
|
981
955
|
};
|
|
982
956
|
let verdict;
|
|
957
|
+
let judgeFailed = false;
|
|
983
958
|
if (judged) {
|
|
984
959
|
verdict = await runReflectQualityJudge(run.config, payload.content, assetContent ?? "", feedback, options.chat, {
|
|
985
960
|
runnerSelectionFrozen: true,
|
|
@@ -988,7 +963,9 @@ async function finalizeReflectProposal(args) {
|
|
|
988
963
|
...(options.signal ? { signal: options.signal } : {}),
|
|
989
964
|
onNotices: run.notices.add,
|
|
990
965
|
});
|
|
991
|
-
|
|
966
|
+
// A judge that timed out, errored or replied unparseably gave no verdict: a person reviews the revision.
|
|
967
|
+
judgeFailed = verdict.reviewNeeded === true && verdict.score === -1;
|
|
968
|
+
if (!verdict.pass && !judgeFailed) {
|
|
992
969
|
return refuse(verdict.reason, {
|
|
993
970
|
qualityScore: verdict.score,
|
|
994
971
|
qualityReason: verdict.reason,
|
|
@@ -997,7 +974,7 @@ async function finalizeReflectProposal(args) {
|
|
|
997
974
|
}
|
|
998
975
|
}
|
|
999
976
|
// #722: a rewrite of an existing asset must not grade lower on its own retrieval queries.
|
|
1000
|
-
if (
|
|
977
|
+
if (verdict?.pass && judge.runner && assetContent !== undefined) {
|
|
1001
978
|
const retrieval = await runRetrievalRegressionGate({
|
|
1002
979
|
ref: payload.ref,
|
|
1003
980
|
before: assetContent,
|
|
@@ -1026,9 +1003,18 @@ async function finalizeReflectProposal(args) {
|
|
|
1026
1003
|
};
|
|
1027
1004
|
const reviewReasons = [
|
|
1028
1005
|
...(judge.skippedNoJudge ? ["no-judge-configured"] : []),
|
|
1006
|
+
...(judgeFailed ? ["judge-error"] : []),
|
|
1029
1007
|
...(sanitized.sizeGuardRatio ? ["reflect-size-ratio"] : []),
|
|
1030
1008
|
...(sanitized.truncationMarkerLeaked ? ["reflect-truncation-leak"] : []),
|
|
1031
1009
|
];
|
|
1010
|
+
// A revision that changes the body is never auto-accepted: on labelled edits, the judge's
|
|
1011
|
+
// passes on body edits were good 12 times in 37, and on frontmatter-only edits 13 in 13.
|
|
1012
|
+
// One that nothing above holds for review (the judge passed it, or the gate is off) waits
|
|
1013
|
+
// for a person, as does a revision with no source to compare.
|
|
1014
|
+
const bodyOf = (content) => splitFrontmatter(content).body.replace(/\s+/g, " ").trim();
|
|
1015
|
+
const bodyEdit = reviewReasons.length === 0 && (assetContent === undefined || bodyOf(assetContent) !== bodyOf(payload.content));
|
|
1016
|
+
if (bodyEdit)
|
|
1017
|
+
reviewReasons.push("body-edit");
|
|
1032
1018
|
const proposal = mintProposal(run.stash, options.ctx, {
|
|
1033
1019
|
ref: payload.ref,
|
|
1034
1020
|
...(options.target ? { target: options.target } : {}),
|
|
@@ -1042,8 +1028,12 @@ async function finalizeReflectProposal(args) {
|
|
|
1042
1028
|
? {
|
|
1043
1029
|
review: {
|
|
1044
1030
|
reason: reviewReasons.join("+"),
|
|
1045
|
-
gate:
|
|
1031
|
+
// The quality gate's hand-off to a person, as distill's: the triage drain leaves it alone.
|
|
1032
|
+
gate: judgeFailed ? "quality-gate" : "reflect",
|
|
1046
1033
|
...(sanitized.sizeGuardRatio ? { measured: Math.round(sanitized.sizeGuardRatio.ratio * 100) } : {}),
|
|
1034
|
+
// The reviewer sees why the judge passed it (with the gate off, nothing judged it).
|
|
1035
|
+
...(bodyEdit && verdict?.criteria ? { scores: verdict.criteria } : {}),
|
|
1036
|
+
...(bodyEdit && verdict ? { judgeReason: verdict.reason } : {}),
|
|
1047
1037
|
},
|
|
1048
1038
|
}
|
|
1049
1039
|
: { judged: verdict });
|
|
@@ -1055,6 +1045,7 @@ async function finalizeReflectProposal(args) {
|
|
|
1055
1045
|
source: "reflect",
|
|
1056
1046
|
engine: run.engineName,
|
|
1057
1047
|
...(judge.skippedNoJudge ? { qualityGateSkippedNoJudge: true } : {}),
|
|
1048
|
+
...(judgeFailed ? { qualityReason: verdict?.reason } : {}),
|
|
1058
1049
|
...(sanitized.sizeGuardRatio
|
|
1059
1050
|
? { sizeGuardRatio: sanitized.sizeGuardRatio.code, sizeGuardRatioValue: sanitized.sizeGuardRatio.ratio }
|
|
1060
1051
|
: {}),
|
|
@@ -1095,8 +1086,6 @@ export async function renderReflectPromptPreview(options) {
|
|
|
1095
1086
|
assetContent: source.assetContent,
|
|
1096
1087
|
sources,
|
|
1097
1088
|
runnerSpec,
|
|
1098
|
-
// The same tmp-path shape a dispatch would use; never written.
|
|
1099
|
-
draftFilePath: runnerSupportsFileWrite(runnerSpec) ? synthesizeReflectDraftPath(ref) : undefined,
|
|
1100
1089
|
priorDraft: undefined,
|
|
1101
1090
|
});
|
|
1102
1091
|
return { ref, prompt, engine: engineName, engineKind: runnerSpec.kind };
|
|
@@ -1124,7 +1113,7 @@ export async function akmReflect(options = {}) {
|
|
|
1124
1113
|
judgeRunner = runnerSpec;
|
|
1125
1114
|
}
|
|
1126
1115
|
else {
|
|
1127
|
-
const resolved =
|
|
1116
|
+
const resolved = resolveImproveExecution({ config, processName: "reflect_proposal_quality-judge" });
|
|
1128
1117
|
if (resolved)
|
|
1129
1118
|
notices.add(resolved.notices);
|
|
1130
1119
|
judgeRunner = resolved?.runner;
|
|
@@ -1132,7 +1121,7 @@ export async function akmReflect(options = {}) {
|
|
|
1132
1121
|
}
|
|
1133
1122
|
const skippedNoJudge = judgeWanted && !judgeRunner;
|
|
1134
1123
|
if (skippedNoJudge) {
|
|
1135
|
-
warnOnce("reflect-quality-gate-no-judge", "Reflect proposal quality gate has no
|
|
1124
|
+
warnOnce("reflect-quality-gate-no-judge", "Reflect proposal quality gate has no engine configured to judge proposals (set defaults.llmEngine). Skipping the gate for this run; the proposal is queued for human review instead.");
|
|
1136
1125
|
}
|
|
1137
1126
|
preflightReflectDispatch(runnerSpec, notices.add);
|
|
1138
1127
|
if (judgeRunner && judgeRunner !== runnerSpec)
|
|
@@ -1143,13 +1132,11 @@ export async function akmReflect(options = {}) {
|
|
|
1143
1132
|
...(Object.keys(agentEnv).length > 0 ? { env: agentEnv } : {}),
|
|
1144
1133
|
...(options.runAgentOptions ?? {}),
|
|
1145
1134
|
});
|
|
1146
|
-
const draftPaths = [];
|
|
1147
1135
|
let result;
|
|
1148
1136
|
let payload;
|
|
1149
1137
|
try {
|
|
1150
|
-
|
|
1138
|
+
result = await runReflectRefineIterations({ run, parsedRef, assetContent, sources, agentEnv });
|
|
1151
1139
|
emitInvoked();
|
|
1152
|
-
result = iterated.result;
|
|
1153
1140
|
if (!result.ok) {
|
|
1154
1141
|
if (isEnoentFailure(result)) {
|
|
1155
1142
|
emitFailed("spawn_failed", "enoent", options.ref, {
|
|
@@ -1172,11 +1159,11 @@ export async function akmReflect(options = {}) {
|
|
|
1172
1159
|
};
|
|
1173
1160
|
emitFailed(envelope.reason, envelope.reason === "parse_error" ? "parse_error" : "agent_crash", options.ref, {
|
|
1174
1161
|
...(envelope.exitCode !== null ? { exitCode: envelope.exitCode } : {}),
|
|
1175
|
-
...(
|
|
1162
|
+
...(reflectTelemetry(result) ?? {}),
|
|
1176
1163
|
});
|
|
1177
1164
|
return { ...envelope, ...notices.fields() };
|
|
1178
1165
|
}
|
|
1179
|
-
const resolved = resolveReflectPayload(run, result
|
|
1166
|
+
const resolved = resolveReflectPayload(run, result);
|
|
1180
1167
|
if ("failure" in resolved)
|
|
1181
1168
|
return resolved.failure;
|
|
1182
1169
|
payload = resolved.payload;
|
|
@@ -1186,43 +1173,11 @@ export async function akmReflect(options = {}) {
|
|
|
1186
1173
|
emitInvoked();
|
|
1187
1174
|
throw error;
|
|
1188
1175
|
}
|
|
1189
|
-
finally {
|
|
1190
|
-
for (const draftPath of draftPaths) {
|
|
1191
|
-
try {
|
|
1192
|
-
if (fs.existsSync(draftPath))
|
|
1193
|
-
fs.unlinkSync(draftPath);
|
|
1194
|
-
}
|
|
1195
|
-
catch {
|
|
1196
|
-
// best-effort
|
|
1197
|
-
}
|
|
1198
|
-
}
|
|
1199
|
-
}
|
|
1200
1176
|
const unsafeContent = generatedContentRejection(payload.content, redactSensitiveText(payload.content, sensitiveValues));
|
|
1201
1177
|
if (unsafeContent) {
|
|
1202
1178
|
emitFailed("parse_error", "parse_error", options.ref, exitCodeMeta(result));
|
|
1203
1179
|
return reflectFailure(run, result, "parse_error", unsafeContent, false);
|
|
1204
1180
|
}
|
|
1205
|
-
// A retargeted proposal is refused (malformed refs are left to proposal validation).
|
|
1206
|
-
if (options.ref) {
|
|
1207
|
-
let retargeted = false;
|
|
1208
|
-
try {
|
|
1209
|
-
const expected = parseRefInput(options.ref);
|
|
1210
|
-
const actual = parseRefInput(payload.ref);
|
|
1211
|
-
retargeted = expected.type !== actual.type || expected.name !== actual.name;
|
|
1212
|
-
}
|
|
1213
|
-
catch {
|
|
1214
|
-
retargeted = false;
|
|
1215
|
-
}
|
|
1216
|
-
if (retargeted) {
|
|
1217
|
-
emitFailed("parse_error", "ref_mismatch", options.ref, {
|
|
1218
|
-
expectedRef: options.ref,
|
|
1219
|
-
actualRef: payload.ref,
|
|
1220
|
-
...exitCodeMeta(result),
|
|
1221
|
-
...(reflectLlmTelemetry(result) ?? {}),
|
|
1222
|
-
});
|
|
1223
|
-
return reflectFailure(run, result, "parse_error", `Agent retargeted proposal: expected ref "${options.ref}" but got "${payload.ref}". Proposal rejected to prevent silent ref hallucination.`, true);
|
|
1224
|
-
}
|
|
1225
|
-
}
|
|
1226
1181
|
return finalizeReflectProposal({
|
|
1227
1182
|
run,
|
|
1228
1183
|
payload,
|