taskflow-core 0.1.4 → 0.1.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -2
- package/dist/detached-runner.js +17 -0
- package/dist/detached-runner.js.map +1 -1
- package/dist/index.d.ts +3 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +3 -0
- package/dist/index.js.map +1 -1
- package/dist/interpolate.d.ts +5 -0
- package/dist/interpolate.d.ts.map +1 -1
- package/dist/interpolate.js +14 -3
- package/dist/interpolate.js.map +1 -1
- package/dist/reflexion.d.ts +46 -0
- package/dist/reflexion.d.ts.map +1 -0
- package/dist/reflexion.js +100 -0
- package/dist/reflexion.js.map +1 -0
- package/dist/runtime.d.ts.map +1 -1
- package/dist/runtime.js +352 -15
- package/dist/runtime.js.map +1 -1
- package/dist/schema.d.ts +11 -5
- package/dist/schema.d.ts.map +1 -1
- package/dist/schema.js +113 -2
- package/dist/schema.js.map +1 -1
- package/dist/scorer-runtime.d.ts +30 -0
- package/dist/scorer-runtime.d.ts.map +1 -0
- package/dist/scorer-runtime.js +134 -0
- package/dist/scorer-runtime.js.map +1 -0
- package/dist/scorers.d.ts +120 -0
- package/dist/scorers.d.ts.map +1 -0
- package/dist/scorers.js +313 -0
- package/dist/scorers.js.map +1 -0
- package/dist/store.d.ts +24 -1
- package/dist/store.d.ts.map +1 -1
- package/dist/store.js.map +1 -1
- package/package.json +1 -1
package/dist/runtime.js
CHANGED
|
@@ -13,7 +13,7 @@ import * as path from "node:path";
|
|
|
13
13
|
import * as fs from "node:fs";
|
|
14
14
|
import { coerceArray, evaluateCondition, interpolate, safeParse, tryEvaluateCondition } from "./interpolate.js";
|
|
15
15
|
import { contractViolations } from "./contract.js";
|
|
16
|
-
import { isFailed, isTransientError, mapWithConcurrencyLimit } from "./runner-core.js";
|
|
16
|
+
import { isFailed, isTransientError, mapWithConcurrencyLimit, sanitizeErrorMessage } from "./runner-core.js";
|
|
17
17
|
/** Default runner used when no host injected one: fail loudly rather than
|
|
18
18
|
* silently spawn anything (core is host-neutral and cannot spawn pi/codex). */
|
|
19
19
|
const noRunnerInjected = async (_cwd, _agents, agentName, task) => ({
|
|
@@ -29,6 +29,9 @@ const noRunnerInjected = async (_cwd, _agents, agentName, task) => ({
|
|
|
29
29
|
import { aggregateUsage, emptyUsage } from "./usage.js";
|
|
30
30
|
import { dependenciesOf, finalPhase, LOOP_DEFAULT_MAX_ITERATIONS, LOOP_HARD_MAX_ITERATIONS, MAX_DYNAMIC_MAP_ITEMS, MAX_DYNAMIC_NESTING, parseTtlMs, resolveArgs, topoLayers, TOURNAMENT_DEFAULT_VARIANTS, TOURNAMENT_HARD_MAX_VARIANTS, validateTaskflow } from "./schema.js";
|
|
31
31
|
import { verifyTaskflow } from "./verify.js";
|
|
32
|
+
import { combineScores, combineWithJudge, evaluatePureScorer, formatScorerReport, parseJudgeOutput, SCORE_DEFAULT_THRESHOLD, scoreResultJSON, scorerShapeErrors } from "./scorers.js";
|
|
33
|
+
import { runCodeCompilesScorer } from "./scorer-runtime.js";
|
|
34
|
+
import { buildReflexionSummary, isContractViolation, REFLEXION_SENTINEL } from "./reflexion.js";
|
|
32
35
|
import { hashInput, newRunId, runsDir } from "./store.js";
|
|
33
36
|
import { CacheStore, resolveFingerprint } from "./cache.js";
|
|
34
37
|
import { compileTaskflowToIR, phaseFingerprint } from "./flowir/index.js";
|
|
@@ -65,7 +68,7 @@ export function summarizeReuse(state) {
|
|
|
65
68
|
savedUSD,
|
|
66
69
|
};
|
|
67
70
|
}
|
|
68
|
-
function buildInterpolationContext(state, previousOutput, locals, onRead) {
|
|
71
|
+
function buildInterpolationContext(state, previousOutput, locals, onRead, reflexion) {
|
|
69
72
|
const steps = {};
|
|
70
73
|
for (const [id, ps] of Object.entries(state.phases)) {
|
|
71
74
|
// Include both done AND failed phases so downstream phases can see
|
|
@@ -82,7 +85,7 @@ function buildInterpolationContext(state, previousOutput, locals, onRead) {
|
|
|
82
85
|
}
|
|
83
86
|
}
|
|
84
87
|
}
|
|
85
|
-
return { args: state.args, steps, previousOutput, locals, onRead };
|
|
88
|
+
return { args: state.args, steps, previousOutput, locals, onRead, reflexion };
|
|
86
89
|
}
|
|
87
90
|
function resultToPhaseState(id, r, inputHash, parseJson) {
|
|
88
91
|
const failed = isFailed(r);
|
|
@@ -553,9 +556,26 @@ async function runSpawnedChildren(assignments, ctxDir, parentNodeId, phase, deps
|
|
|
553
556
|
return { reports: `\n\n<!-- ctx_spawn: ${lines.length} child report(s) -->\n${lines.join("\n\n")}`, usage };
|
|
554
557
|
}
|
|
555
558
|
async function executePhase(phase, state, deps, prior, emitProgress, _retryDepth = 0, opts) {
|
|
559
|
+
// Side-effect classification: stamp the marker at the single exit point so
|
|
560
|
+
// every type branch inside executePhaseInner is covered. A skipped phase ran
|
|
561
|
+
// nothing — no side effect to record.
|
|
562
|
+
const stamp = (ps) => {
|
|
563
|
+
if (phase.idempotent === false && ps.status !== "skipped") {
|
|
564
|
+
ps.sideEffect = true;
|
|
565
|
+
// Resume double-fire warning (issue #20): a non-idempotent phase is never
|
|
566
|
+
// cached, so on resume it RE-EXECUTES even though a prior attempt already
|
|
567
|
+
// completed — re-firing its side effect. Surface it so operators aren't
|
|
568
|
+
// surprised by a second webhook/deploy. Only when a prior DONE state
|
|
569
|
+
// existed (the resume signal) and this run actually re-ran (status done).
|
|
570
|
+
if (prior?.status === "done" && ps.status === "done" && !ps.cacheHit) {
|
|
571
|
+
ps.warnings = [...(ps.warnings ?? []), "idempotent:false phase re-executed on resume (a prior attempt had completed) — its side effect fired again"];
|
|
572
|
+
}
|
|
573
|
+
}
|
|
574
|
+
return ps;
|
|
575
|
+
};
|
|
556
576
|
// Non-keyword cwd (or none): no workspace lifecycle — run directly.
|
|
557
577
|
if (!isWorkspaceKeyword(phase.cwd)) {
|
|
558
|
-
return executePhaseInner(phase, state, deps, prior, emitProgress, _retryDepth, opts);
|
|
578
|
+
return stamp(await executePhaseInner(phase, state, deps, prior, emitProgress, _retryDepth, opts));
|
|
559
579
|
}
|
|
560
580
|
let ws;
|
|
561
581
|
try {
|
|
@@ -577,7 +597,7 @@ async function executePhase(phase, state, deps, prior, emitProgress, _retryDepth
|
|
|
577
597
|
const msg = ws.note ? `${tag} — ${ws.note}` : `${tag} at ${ws.dir}`;
|
|
578
598
|
ps.warnings = [...(ps.warnings ?? []), msg];
|
|
579
599
|
}
|
|
580
|
-
return ps;
|
|
600
|
+
return stamp(ps);
|
|
581
601
|
}
|
|
582
602
|
finally {
|
|
583
603
|
try {
|
|
@@ -658,6 +678,14 @@ async function executePhaseInner(phase, state, deps, prior, emitProgress, _retry
|
|
|
658
678
|
if (state.flowDefHash === "failed" && cacheScope === "cross-run") {
|
|
659
679
|
cacheScope = "run-only";
|
|
660
680
|
}
|
|
681
|
+
// Side-effect classification: a non-idempotent phase is NEVER cached — not
|
|
682
|
+
// served from within-run resume, not served from or written to the cross-run
|
|
683
|
+
// store (cachedPhase and recordCache both gate on scope). One assignment
|
|
684
|
+
// covers every cache path, including the map per-item path (perItemCacheable
|
|
685
|
+
// requires scope === "cross-run").
|
|
686
|
+
if (phase.idempotent === false) {
|
|
687
|
+
cacheScope = "off";
|
|
688
|
+
}
|
|
661
689
|
const cc = {
|
|
662
690
|
scope: cacheScope,
|
|
663
691
|
ttlMs: phase.cache?.ttl ? (parseTtlMs(phase.cache.ttl) ?? undefined) : undefined,
|
|
@@ -785,7 +813,13 @@ async function executePhaseInner(phase, state, deps, prior, emitProgress, _retry
|
|
|
785
813
|
// policy stops immediately (no point burning attempts on a hard error).
|
|
786
814
|
const withinExplicit = attempt < explicitMax - 1;
|
|
787
815
|
const transient = isTransientError(last);
|
|
788
|
-
|
|
816
|
+
// Side-effect classification: the implicit transient retry is a runtime
|
|
817
|
+
// safety net the author did not ask for — repeating a side-effecting
|
|
818
|
+
// call behind the author's back is exactly the hazard idempotent:false
|
|
819
|
+
// exists to prevent. Explicit retry{} (withinExplicit) stays honored:
|
|
820
|
+
// it is the author's declaration that a repeat is acceptable.
|
|
821
|
+
const allowTransient = phase.idempotent !== false;
|
|
822
|
+
const withinTransient = transient && allowTransient && attempt < DEFAULT_TRANSIENT_RETRIES;
|
|
789
823
|
if (!withinExplicit && !withinTransient)
|
|
790
824
|
break;
|
|
791
825
|
// Backoff: prefer the explicit policy's curve when the phase defines one
|
|
@@ -1025,6 +1059,216 @@ async function executePhaseInner(phase, state, deps, prior, emitProgress, _retry
|
|
|
1025
1059
|
return ps;
|
|
1026
1060
|
}
|
|
1027
1061
|
}
|
|
1062
|
+
// Scoring gate (`score`): deterministic scorers → zero-token auto-pass /
|
|
1063
|
+
// LLM judge / task fallback. Self-contained (including its own onBlock:retry
|
|
1064
|
+
// mirror) so the non-score gate path below stays byte-identical.
|
|
1065
|
+
const scoreRaw = type === "gate" ? phase.score : undefined;
|
|
1066
|
+
if (scoreRaw !== undefined) {
|
|
1067
|
+
const shapeErrs = scorerShapeErrors(scoreRaw);
|
|
1068
|
+
if (shapeErrs.length === 0) {
|
|
1069
|
+
const sc = scoreRaw;
|
|
1070
|
+
const combine = sc.combine ?? "all";
|
|
1071
|
+
const threshold = combine === "weighted" ? (sc.threshold ?? SCORE_DEFAULT_THRESHOLD) : undefined;
|
|
1072
|
+
const scoreId = JSON.stringify(scoreRaw);
|
|
1073
|
+
// One full score evaluation (deterministics → auto-pass | judge | task |
|
|
1074
|
+
// deterministic BLOCK). Re-invoked by the onBlock:retry loop, so it
|
|
1075
|
+
// rebuilds its interpolation context from the CURRENT state each call.
|
|
1076
|
+
const evaluateScore = async () => {
|
|
1077
|
+
const freshPrev = lastCompletedOutput(state, phase);
|
|
1078
|
+
const freshCtx = buildInterpolationContext(state, freshPrev, undefined, onRead);
|
|
1079
|
+
const tInterp = interpolate(sc.target ?? "{previous.output}", freshCtx);
|
|
1080
|
+
const targetResolved = tInterp.missing.length === 0;
|
|
1081
|
+
const target = tInterp.text;
|
|
1082
|
+
// Deterministic scorers — pure ones inline, code-compiles via the
|
|
1083
|
+
// impure runtime module. Skipped entirely when the target ref did not
|
|
1084
|
+
// resolve (scoring a literal placeholder would be noise).
|
|
1085
|
+
const results = [];
|
|
1086
|
+
if (targetResolved) {
|
|
1087
|
+
for (let i = 0; i < sc.scorers.length; i++) {
|
|
1088
|
+
const s = sc.scorers[i];
|
|
1089
|
+
results.push(s.type === "code-compiles"
|
|
1090
|
+
? await runCodeCompilesScorer(s, i, target)
|
|
1091
|
+
: evaluatePureScorer(s, i, target));
|
|
1092
|
+
}
|
|
1093
|
+
}
|
|
1094
|
+
// Weighted + judge: the judge's weight enlarges the denominator so the
|
|
1095
|
+
// deterministic combination is a LOWER BOUND — clearing the threshold
|
|
1096
|
+
// without the judge means the judge could not change the outcome.
|
|
1097
|
+
const judgeWeight = combine === "weighted" && sc.judge ? (sc.weights?.[sc.scorers.length] ?? 1) : 0;
|
|
1098
|
+
const det = combineScores(results, combine, sc.weights, threshold ?? SCORE_DEFAULT_THRESHOLD, judgeWeight);
|
|
1099
|
+
// Auto-pass is only sound when the judge could not veto it:
|
|
1100
|
+
// - no judge configured → the deterministics ARE the decision;
|
|
1101
|
+
// - weighted + judge → det.passed used the judge-inflated denominator
|
|
1102
|
+
// (lower bound), so the judge's score cannot drop it below threshold.
|
|
1103
|
+
// all/any WITH a judge must NOT auto-skip: there the judge's verdict is
|
|
1104
|
+
// authoritative (it may check what scorers cannot — e.g. factuality) and
|
|
1105
|
+
// skipping it would silently bypass a configured quality check.
|
|
1106
|
+
const judgeCannotVeto = !sc.judge || combine === "weighted";
|
|
1107
|
+
if (targetResolved && det.passed && judgeCannotVeto) {
|
|
1108
|
+
// AUTO-PASS — zero LLM tokens (mirrors the eval-skip fast-path).
|
|
1109
|
+
const inputHash = cacheKeys(cc, [phase.id, "score-skip", scoreId, target]).key;
|
|
1110
|
+
const scores = { results, combined: det.combined, threshold };
|
|
1111
|
+
const ps = {
|
|
1112
|
+
id: phase.id,
|
|
1113
|
+
status: "done",
|
|
1114
|
+
output: "PASS (scorers passed — no LLM call)",
|
|
1115
|
+
gate: { verdict: "pass", scores },
|
|
1116
|
+
json: scoreResultJSON(results, det.combined, "pass", threshold),
|
|
1117
|
+
usage: emptyUsage(),
|
|
1118
|
+
inputHash,
|
|
1119
|
+
endedAt: Date.now(),
|
|
1120
|
+
};
|
|
1121
|
+
if (readRefs.length)
|
|
1122
|
+
ps.reads = readRefsToReads(readRefs, state);
|
|
1123
|
+
return ps;
|
|
1124
|
+
}
|
|
1125
|
+
const report = targetResolved
|
|
1126
|
+
? formatScorerReport(results, det.combined, threshold)
|
|
1127
|
+
: `## Deterministic scorer report\n(scorers skipped — score.target did not resolve: ${tInterp.missing.join(", ")})`;
|
|
1128
|
+
// Judge fallback — the LLM-as-judge decides, with the target and the
|
|
1129
|
+
// deterministic report in evidence. Fail-open on unparseable output.
|
|
1130
|
+
if (sc.judge) {
|
|
1131
|
+
const judgeAgent = resolveAgent(sc.judge.agent ?? phase.agent, deps, state);
|
|
1132
|
+
const judgeText = interpolate(sc.judge.task, freshCtx).text;
|
|
1133
|
+
// Neutralize fences in the (model-produced) target so it cannot
|
|
1134
|
+
// close the evidence block and inject instructions at prompt level.
|
|
1135
|
+
const safeTarget = target.replace(/```/g, "`\u200b``");
|
|
1136
|
+
const fullJudgeTask = `${preRead}${judgeText}\n\n---\n\n## Target under evaluation\n\`\`\`\n${safeTarget}\n\`\`\`\n\n${report}\n\n` +
|
|
1137
|
+
`Return JSON {"score": 0.0-1.0, "verdict": "pass"|"block", "reason": "..."} (or end with VERDICT: PASS|BLOCK).`;
|
|
1138
|
+
const ckJ = cacheKeys(cc, [phase.id, judgeAgent, phase.model ?? "", fullJudgeTask, scoreId]);
|
|
1139
|
+
const inputHash = ckJ.key;
|
|
1140
|
+
const cachedJ = cachedPhase(cc, ckJ);
|
|
1141
|
+
if (cachedJ)
|
|
1142
|
+
return cachedJ;
|
|
1143
|
+
const r = await runOne(judgeAgent, fullJudgeTask, liveSink(state, phase.id, emitProgress), nodeIdFor("judge"));
|
|
1144
|
+
const ps = resultToPhaseState(phase.id, r, inputHash, false);
|
|
1145
|
+
if (ps.status === "done") {
|
|
1146
|
+
const judged = parseJudgeOutput(r.output);
|
|
1147
|
+
// Weighted: the judge's score folds into the combination and the
|
|
1148
|
+
// threshold decides. all/any: the judge's verdict is authoritative.
|
|
1149
|
+
const final = combine === "weighted"
|
|
1150
|
+
? combineWithJudge(results, sc.weights, threshold ?? SCORE_DEFAULT_THRESHOLD, judged.score)
|
|
1151
|
+
: { combined: judged.score, passed: judged.verdict === "pass" };
|
|
1152
|
+
const verdict = final.passed ? "pass" : "block";
|
|
1153
|
+
ps.gate = { verdict, reason: judged.reason, scores: { results, combined: final.combined, threshold } };
|
|
1154
|
+
ps.json = scoreResultJSON(results, final.combined, verdict, threshold, { score: judged.score, reason: judged.reason });
|
|
1155
|
+
}
|
|
1156
|
+
if (readRefs.length)
|
|
1157
|
+
ps.reads = readRefsToReads(readRefs, state);
|
|
1158
|
+
return ps;
|
|
1159
|
+
}
|
|
1160
|
+
// Task fallback — the gate's own LLM task runs with the scorer report
|
|
1161
|
+
// appended, verdict parsed as usual.
|
|
1162
|
+
if (phase.task) {
|
|
1163
|
+
const agentName = resolveAgent(phase.agent, deps, state);
|
|
1164
|
+
const text = interpolate(phase.task, freshCtx).text;
|
|
1165
|
+
const fullTask = `${preRead}${text}\n\n---\n\n${report}`;
|
|
1166
|
+
const ckT = cacheKeys(cc, [phase.id, agentName, phase.model ?? "", fullTask, scoreId]);
|
|
1167
|
+
const inputHash = ckT.key;
|
|
1168
|
+
const cachedT = cachedPhase(cc, ckT);
|
|
1169
|
+
if (cachedT)
|
|
1170
|
+
return cachedT;
|
|
1171
|
+
const r = await runOne(agentName, fullTask, liveSink(state, phase.id, emitProgress), nodeIdFor(), contractCheck);
|
|
1172
|
+
const ps = resultToPhaseState(phase.id, r, inputHash, parseJson);
|
|
1173
|
+
if (ps.status === "done") {
|
|
1174
|
+
const v = parseGateVerdict(r.output);
|
|
1175
|
+
ps.gate = { ...v, scores: { results, combined: det.combined, threshold } };
|
|
1176
|
+
ps.json = scoreResultJSON(results, det.combined, v.verdict, threshold);
|
|
1177
|
+
}
|
|
1178
|
+
if (readRefs.length)
|
|
1179
|
+
ps.reads = readRefsToReads(readRefs, state);
|
|
1180
|
+
return ps;
|
|
1181
|
+
}
|
|
1182
|
+
// No LLM fallback configured.
|
|
1183
|
+
if (!targetResolved) {
|
|
1184
|
+
// Unresolved target with no judge/task is AMBIGUITY, not an explicit
|
|
1185
|
+
// failure — fail-open PASS with a warning (project invariant).
|
|
1186
|
+
const inputHash = cacheKeys(cc, [phase.id, "score-unresolved", scoreId]).key;
|
|
1187
|
+
return {
|
|
1188
|
+
id: phase.id,
|
|
1189
|
+
status: "done",
|
|
1190
|
+
output: "PASS (score.target did not resolve — fail-open)",
|
|
1191
|
+
gate: { verdict: "pass", reason: `score.target unresolved: ${tInterp.missing.join(", ")}` },
|
|
1192
|
+
json: scoreResultJSON([], 0, "pass", threshold),
|
|
1193
|
+
warnings: [`gate score.target did not resolve (${tInterp.missing.join(", ")}) — fail-open PASS`],
|
|
1194
|
+
usage: emptyUsage(),
|
|
1195
|
+
inputHash,
|
|
1196
|
+
endedAt: Date.now(),
|
|
1197
|
+
};
|
|
1198
|
+
}
|
|
1199
|
+
// Deterministic explicit failure is NOT ambiguity: BLOCK.
|
|
1200
|
+
const inputHash = cacheKeys(cc, [phase.id, "score-block", scoreId, target]).key;
|
|
1201
|
+
const scores = { results, combined: det.combined, threshold };
|
|
1202
|
+
const ps = {
|
|
1203
|
+
id: phase.id,
|
|
1204
|
+
status: "done",
|
|
1205
|
+
output: `BLOCK (deterministic scorers failed — no LLM fallback)\n\n${report}`,
|
|
1206
|
+
gate: { verdict: "block", reason: "deterministic scorers below threshold", scores },
|
|
1207
|
+
json: scoreResultJSON(results, det.combined, "block", threshold),
|
|
1208
|
+
usage: emptyUsage(),
|
|
1209
|
+
inputHash,
|
|
1210
|
+
endedAt: Date.now(),
|
|
1211
|
+
};
|
|
1212
|
+
if (readRefs.length)
|
|
1213
|
+
ps.reads = readRefsToReads(readRefs, state);
|
|
1214
|
+
return ps;
|
|
1215
|
+
};
|
|
1216
|
+
let ps = await evaluateScore();
|
|
1217
|
+
// onBlock:retry — mirrors the non-score gate loop below (re-run upstream
|
|
1218
|
+
// deps, then RE-SCORE), sharing its depth cap and budget/abort guards.
|
|
1219
|
+
if (ps.gate?.verdict === "block") {
|
|
1220
|
+
const onBlockV = phase.onBlock ?? "halt";
|
|
1221
|
+
const MAX_RETRY_DEPTH = 3;
|
|
1222
|
+
let attempt = 0;
|
|
1223
|
+
while (onBlockV === "retry" && attempt < (phase.retry?.max ?? 1)) {
|
|
1224
|
+
if (deps.signal?.aborted || overBudget(state).over)
|
|
1225
|
+
break;
|
|
1226
|
+
attempt++;
|
|
1227
|
+
if (_retryDepth < MAX_RETRY_DEPTH) {
|
|
1228
|
+
const { _cwdOverride: _dropGateWs, ...depsForUpstream } = deps;
|
|
1229
|
+
for (const depId of phase.dependsOn ?? []) {
|
|
1230
|
+
const d = state.def.phases.find((p) => p.id === depId);
|
|
1231
|
+
if (!d)
|
|
1232
|
+
continue;
|
|
1233
|
+
const dPs = await executePhase(d, state, depsForUpstream, prior, emitProgress, _retryDepth + 1, undefined);
|
|
1234
|
+
state.phases[depId] = dPs;
|
|
1235
|
+
}
|
|
1236
|
+
}
|
|
1237
|
+
const prevAttempts = ps.attempts ?? 0;
|
|
1238
|
+
ps = await evaluateScore();
|
|
1239
|
+
ps.attempts = prevAttempts + (ps.attempts ?? 0);
|
|
1240
|
+
if (ps.gate?.verdict !== "block" || overBudget(state).over)
|
|
1241
|
+
break;
|
|
1242
|
+
}
|
|
1243
|
+
if (attempt > 0)
|
|
1244
|
+
ps.attempts = Math.max(ps.attempts ?? 0, attempt);
|
|
1245
|
+
}
|
|
1246
|
+
recordCache(cc, ps);
|
|
1247
|
+
return ps;
|
|
1248
|
+
}
|
|
1249
|
+
// Malformed score (validation reports it) — fail-open: fall through to
|
|
1250
|
+
// the plain LLM gate with a warning so an authoring slip degrades to the
|
|
1251
|
+
// historical behavior instead of crashing the phase.
|
|
1252
|
+
const scoreWarning = `gate 'score' is malformed and was ignored: ${shapeErrs[0]}`;
|
|
1253
|
+
const interpM = interpolate(phase.task ?? "", ctx);
|
|
1254
|
+
const textM = interpM.text;
|
|
1255
|
+
const refWarningM = warnUnresolvedRefs(phase.id, interpM.missing);
|
|
1256
|
+
const fullTaskM = preRead + textM;
|
|
1257
|
+
const agentNameM = resolveAgent(phase.agent, deps, state);
|
|
1258
|
+
const ckM = cacheKeys(cc, [phase.id, agentNameM, phase.model ?? "", fullTaskM]);
|
|
1259
|
+
const cachedM = cachedPhase(cc, ckM);
|
|
1260
|
+
if (cachedM)
|
|
1261
|
+
return cachedM;
|
|
1262
|
+
const rM = await runOne(agentNameM, fullTaskM, liveSink(state, phase.id, emitProgress), nodeIdFor(), contractCheck);
|
|
1263
|
+
const psM = resultToPhaseState(phase.id, rM, ckM.key, parseJson);
|
|
1264
|
+
if (readRefs.length)
|
|
1265
|
+
psM.reads = readRefsToReads(readRefs, state);
|
|
1266
|
+
psM.warnings = [...(psM.warnings ?? []), scoreWarning, ...(refWarningM ? [refWarningM] : [])];
|
|
1267
|
+
if (psM.status === "done")
|
|
1268
|
+
psM.gate = parseGateVerdict(rM.output);
|
|
1269
|
+
recordCache(cc, psM);
|
|
1270
|
+
return psM;
|
|
1271
|
+
}
|
|
1028
1272
|
const interp = interpolate(phase.task ?? "", ctx);
|
|
1029
1273
|
const text = interp.text;
|
|
1030
1274
|
const refWarning = warnUnresolvedRefs(phase.id, interp.missing);
|
|
@@ -1617,14 +1861,17 @@ async function executePhaseInner(phase, state, deps, prior, emitProgress, _retry
|
|
|
1617
1861
|
const rawMax = phase.maxIterations ?? LOOP_DEFAULT_MAX_ITERATIONS;
|
|
1618
1862
|
const maxIters = Math.max(1, Math.min(LOOP_HARD_MAX_ITERATIONS, Math.floor(rawMax)));
|
|
1619
1863
|
const convergence = phase.convergence ?? true;
|
|
1864
|
+
const reflexionOn = phase.reflexion === true;
|
|
1620
1865
|
// Canonical first-iteration body for the cache key. It must fold in the
|
|
1621
1866
|
// interpolated task/upstream refs so that a changed upstream changes the
|
|
1622
1867
|
// key and recompute no longer silently reuses a stale loop (critic finding).
|
|
1868
|
+
// Reflexion loops resolve {reflexion} to the SENTINEL here so the key
|
|
1869
|
+
// reflects the true first prompt (not a literal placeholder).
|
|
1623
1870
|
const firstBodyCtx = buildInterpolationContext(state, previousOutput, {
|
|
1624
1871
|
loop: { iteration: 1, lastOutput: "", maxIterations: maxIters },
|
|
1625
|
-
}, (ref) => readRefs.push(ref));
|
|
1872
|
+
}, (ref) => readRefs.push(ref), reflexionOn ? REFLEXION_SENTINEL : undefined);
|
|
1626
1873
|
const firstBody = preRead + interpolate(phase.task ?? "", firstBodyCtx).text;
|
|
1627
|
-
const inputHash = hashInput(phase.id, "loop", phase.until ?? "", firstBody, String(maxIters));
|
|
1874
|
+
const inputHash = hashInput(phase.id, "loop", phase.until ?? "", firstBody, String(maxIters), reflexionOn ? "reflexion" : "");
|
|
1628
1875
|
const usages = [];
|
|
1629
1876
|
const loopWarnings = [];
|
|
1630
1877
|
let lastOutput = "";
|
|
@@ -1632,24 +1879,103 @@ async function executePhaseInner(phase, state, deps, prior, emitProgress, _retry
|
|
|
1632
1879
|
let iterations = 0;
|
|
1633
1880
|
let stop = "maxIterations";
|
|
1634
1881
|
let failedResult;
|
|
1882
|
+
// Bounded history of failed iterations (issue #17): when a reflexion loop
|
|
1883
|
+
// continues past failures, only the terminal one survives in `error` — keep
|
|
1884
|
+
// the rest for post-hoc debugging. Capped to avoid unbounded state growth.
|
|
1885
|
+
const LOOP_FAILURE_HISTORY_CAP = 20;
|
|
1886
|
+
const loopFailures = [];
|
|
1887
|
+
const recordLoopFailure = (iteration, r) => {
|
|
1888
|
+
const err = isContractViolation(r.errorMessage)
|
|
1889
|
+
? r.errorMessage
|
|
1890
|
+
: sanitizeErrorMessage(r.errorMessage || r.stderr || "") || `iteration ${iteration} failed`;
|
|
1891
|
+
loopFailures.push({ iteration, error: err });
|
|
1892
|
+
if (loopFailures.length > LOOP_FAILURE_HISTORY_CAP)
|
|
1893
|
+
loopFailures.shift();
|
|
1894
|
+
};
|
|
1895
|
+
// Reflexion state: what the NEXT iteration should be told about THIS one.
|
|
1896
|
+
let reflexionNext;
|
|
1897
|
+
let lastReflexion;
|
|
1898
|
+
let reflexionAppendWarned = false;
|
|
1899
|
+
// With reflexion on, a failed iteration continues (as feedback) instead of
|
|
1900
|
+
// terminating — track whether the LAST iteration failed so an exhausted
|
|
1901
|
+
// loop still fails (reflexion defers failure, it does not erase it).
|
|
1902
|
+
let lastIterationFailed = false;
|
|
1635
1903
|
for (let i = 1; i <= maxIters; i++) {
|
|
1636
1904
|
if (deps.signal?.aborted) {
|
|
1637
1905
|
stop = "aborted";
|
|
1638
1906
|
break;
|
|
1639
1907
|
}
|
|
1640
1908
|
iterations = i;
|
|
1909
|
+
// Assemble the reflexion summary for this iteration (fail-open: a
|
|
1910
|
+
// reflexion assembly bug must never sink the phase).
|
|
1911
|
+
let reflexionStr;
|
|
1912
|
+
if (reflexionOn) {
|
|
1913
|
+
if (i === 1 || !reflexionNext) {
|
|
1914
|
+
reflexionStr = REFLEXION_SENTINEL;
|
|
1915
|
+
}
|
|
1916
|
+
else {
|
|
1917
|
+
try {
|
|
1918
|
+
reflexionStr = buildReflexionSummary(reflexionNext);
|
|
1919
|
+
}
|
|
1920
|
+
catch (e) {
|
|
1921
|
+
reflexionStr = REFLEXION_SENTINEL;
|
|
1922
|
+
loopWarnings.push(`reflexion summary failed to assemble (iteration ${i}): ${e instanceof Error ? e.message : String(e)}`);
|
|
1923
|
+
}
|
|
1924
|
+
lastReflexion = reflexionStr;
|
|
1925
|
+
}
|
|
1926
|
+
}
|
|
1641
1927
|
// The body sees its iteration number and the prior iteration's output.
|
|
1642
1928
|
const bodyCtx = buildInterpolationContext(state, previousOutput, {
|
|
1643
1929
|
loop: { iteration: i, lastOutput, maxIterations: maxIters },
|
|
1644
|
-
}, (ref) => readRefs.push(ref));
|
|
1645
|
-
|
|
1930
|
+
}, (ref) => readRefs.push(ref), reflexionStr);
|
|
1931
|
+
let body = preRead + interpolate(phase.task ?? "", bodyCtx).text;
|
|
1932
|
+
// Auto-append: reflexion is on but the task never mentions {reflexion} —
|
|
1933
|
+
// inject the summary anyway (opt-in via the flag IS the author's intent)
|
|
1934
|
+
// and tell them once how to control placement.
|
|
1935
|
+
if (reflexionOn && reflexionStr && reflexionStr !== REFLEXION_SENTINEL && !(phase.task ?? "").includes("{reflexion}")) {
|
|
1936
|
+
body = `${body}\n\n---\n\n${reflexionStr}`;
|
|
1937
|
+
if (!reflexionAppendWarned) {
|
|
1938
|
+
reflexionAppendWarned = true;
|
|
1939
|
+
loopWarnings.push("reflexion: true but the task has no {reflexion} placeholder — the summary was auto-appended; add {reflexion} to control placement");
|
|
1940
|
+
}
|
|
1941
|
+
}
|
|
1646
1942
|
const r = await runOne(agentName, body, liveSink(state, phase.id, emitProgress), undefined, contractCheck);
|
|
1647
1943
|
usages.push(r.usage);
|
|
1944
|
+
// Fold cumulative loop spend into the live phase state so the run-level
|
|
1945
|
+
// budget guard (overBudget reads state.phases[*].usage) sees the loop's
|
|
1946
|
+
// accrual mid-phase — otherwise each iteration would overwrite the last
|
|
1947
|
+
// and a reflexion loop could spend past the ceiling unnoticed.
|
|
1948
|
+
const livePs = state.phases[phase.id];
|
|
1949
|
+
if (livePs)
|
|
1950
|
+
livePs.usage = aggregateUsage(usages);
|
|
1648
1951
|
if (isFailed(r)) {
|
|
1952
|
+
// Reflexion mode: a body failure becomes feedback for the next
|
|
1953
|
+
// iteration instead of terminating the loop. Timeout, abort, and an
|
|
1954
|
+
// exhausted budget still hard-stop (consistent with "timedOut is never
|
|
1955
|
+
// retried"; continuing past the budget would spend past the ceiling).
|
|
1956
|
+
const hardStop = !reflexionOn || r.phaseTimeout === true || deps.signal?.aborted === true || overBudget(state).over;
|
|
1957
|
+
if (hardStop) {
|
|
1958
|
+
failedResult = r;
|
|
1959
|
+
stop = "failed";
|
|
1960
|
+
recordLoopFailure(i, r);
|
|
1961
|
+
break;
|
|
1962
|
+
}
|
|
1649
1963
|
failedResult = r;
|
|
1650
|
-
|
|
1651
|
-
|
|
1964
|
+
lastIterationFailed = true;
|
|
1965
|
+
recordLoopFailure(i, r);
|
|
1966
|
+
// Sanitize before injecting into the next prompt: raw provider errors
|
|
1967
|
+
// can carry HTML/transport noise (same policy as the transcript path).
|
|
1968
|
+
const rawErr = r.errorMessage || r.stderr || undefined;
|
|
1969
|
+
reflexionNext = {
|
|
1970
|
+
iteration: i,
|
|
1971
|
+
outcome: isContractViolation(r.errorMessage) ? "contract-violation" : "subagent-error",
|
|
1972
|
+
output: r.output,
|
|
1973
|
+
errorMessage: isContractViolation(r.errorMessage) ? r.errorMessage : rawErr ? sanitizeErrorMessage(rawErr) : undefined,
|
|
1974
|
+
};
|
|
1975
|
+
continue;
|
|
1652
1976
|
}
|
|
1977
|
+
lastIterationFailed = false;
|
|
1978
|
+
failedResult = undefined;
|
|
1653
1979
|
prevOutput = lastOutput;
|
|
1654
1980
|
lastOutput = r.output;
|
|
1655
1981
|
// Expose this iteration's output as {steps.<thisId>.output|json} so the
|
|
@@ -1676,9 +2002,20 @@ async function executePhaseInner(phase, state, deps, prior, emitProgress, _retry
|
|
|
1676
2002
|
stop = "converged";
|
|
1677
2003
|
break;
|
|
1678
2004
|
}
|
|
2005
|
+
// Succeeded but not done — the next iteration reflects on the unmet
|
|
2006
|
+
// stop condition (until-not-met is a signal too, not just failures).
|
|
2007
|
+
if (reflexionOn) {
|
|
2008
|
+
reflexionNext = { iteration: i, outcome: "until-not-met", output: lastOutput, until: phase.until };
|
|
2009
|
+
}
|
|
1679
2010
|
}
|
|
1680
2011
|
const aggUsage = usages.length ? aggregateUsage(usages) : emptyUsage();
|
|
1681
|
-
|
|
2012
|
+
// Reflexion: an exhausted loop whose LAST iteration failed is a failure —
|
|
2013
|
+
// reflexion defers termination for feedback, it does not convert a failing
|
|
2014
|
+
// loop into a success.
|
|
2015
|
+
if (reflexionOn && stop === "maxIterations" && lastIterationFailed) {
|
|
2016
|
+
stop = "failed";
|
|
2017
|
+
}
|
|
2018
|
+
if (stop === "failed" || stop === "aborted") {
|
|
1682
2019
|
return {
|
|
1683
2020
|
id: phase.id,
|
|
1684
2021
|
status: "failed",
|
|
@@ -1686,7 +2023,7 @@ async function executePhaseInner(phase, state, deps, prior, emitProgress, _retry
|
|
|
1686
2023
|
usage: aggUsage,
|
|
1687
2024
|
timedOut: failedResult?.phaseTimeout || undefined,
|
|
1688
2025
|
error: failedResult?.errorMessage || failedResult?.stderr || (stop === "aborted" ? "Aborted" : `loop '${phase.id}' iteration ${iterations} failed`),
|
|
1689
|
-
loop: { iterations, stop },
|
|
2026
|
+
loop: { iterations, stop, ...(lastReflexion ? { reflexion: lastReflexion } : {}), ...(loopFailures.length ? { failures: [...loopFailures] } : {}) },
|
|
1690
2027
|
warnings: loopWarnings.length ? loopWarnings : undefined,
|
|
1691
2028
|
inputHash,
|
|
1692
2029
|
reads: readRefsToReads(readRefs, state),
|
|
@@ -1699,7 +2036,7 @@ async function executePhaseInner(phase, state, deps, prior, emitProgress, _retry
|
|
|
1699
2036
|
output: lastOutput,
|
|
1700
2037
|
json: parseJson ? safeParse(lastOutput) : undefined,
|
|
1701
2038
|
usage: aggUsage,
|
|
1702
|
-
loop: { iterations, stop },
|
|
2039
|
+
loop: { iterations, stop, ...(lastReflexion ? { reflexion: lastReflexion } : {}), ...(loopFailures.length ? { failures: [...loopFailures] } : {}) },
|
|
1703
2040
|
warnings: loopWarnings.length ? loopWarnings : undefined,
|
|
1704
2041
|
inputHash,
|
|
1705
2042
|
reads: readRefsToReads(readRefs, state),
|