@tea-agent/loop-agent 0.42.0-next.15 → 0.42.0-next.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. package/CHANGELOG.md +12 -0
  2. package/dist/application/task-lifecycle/advance.js +9 -4
  3. package/dist/build-stamp.json +3 -3
  4. package/dist/commands/task-source-prepare.js +3 -1
  5. package/dist/executors/dag-pi-executor.js +825 -65
  6. package/dist/executors/shell-executor.js +103 -35
  7. package/dist/shared/dag-failure-category.js +6 -0
  8. package/dist/task/contract/apply.js +36 -2
  9. package/dist/task/source-prepare/parse-intent.js +7 -0
  10. package/dist/workflows/dag/dag-retry-schema.js +3 -0
  11. package/dist/workflows/dag/frontend-design-policy.js +1 -1
  12. package/dist/workflows/dag/frontend-risk.js +2 -0
  13. package/dist/workflows/dag/frontend-shadow-dual-write.js +16 -1
  14. package/dist/workflows/dag/frontend-shape.js +16 -6
  15. package/dist/workflows/dag/frontend-test-execution-evidence.js +48 -0
  16. package/dist/workflows/dag/frontend-verification-trace.js +32 -0
  17. package/dist/workflows/dag/init-hybrid.js +9 -10
  18. package/dist/workflows/dag/node-execution.js +124 -72
  19. package/dist/workflows/dag/rerun-feedback.js +1 -0
  20. package/dist/workflows/dag/rerun-plan.js +90 -4
  21. package/dist/workflows/dag/rerun-run.js +7 -0
  22. package/dist/workflows/dag/retry-policy.js +16 -10
  23. package/dist/workflows/dag/runner-exit-diagnostics.js +125 -0
  24. package/dist/workflows/dag/runner.js +67 -7
  25. package/dist/workflows/dag/structured-output-repair.js +4 -1
  26. package/dist/workflows/dag/validate.js +10 -8
  27. package/dist/workflows/dag/workspace-checkpoint.js +66 -0
  28. package/docs/templates/agent-dag.schema.json +2 -2
  29. package/package.json +1 -1
  30. package/skills/frontend-plan/SKILL.md +6 -1
  31. package/skills/frontend-plan/references/decision-contract.md +69 -18
  32. package/skills/frontend-plan/references/design-decisions.md +32 -0
@@ -13,7 +13,7 @@ import { buildDagNodePromptEnvelope, formatConvergenceFeedbackBlock, } from "./p
13
13
  import { persistLongNodeOutputArtifacts } from "./upstream-artifacts.js";
14
14
  import { materializeDeclaredArtifactFacts } from "./artifact-bindings.js";
15
15
  import { buildOutputLimitRecoverySection, loadBackendTestWriterProgressForRetry, } from "./backend-test-writer-completeness.js";
16
- import { computeBackoffDelayMs, isCanonicalFinalVerifyShellRetryCandidate, isRetryablePiFailureCategory, isSafeReadOnlyPiRetryCandidate, isTargetTemplateTransientRetryNode, isWriterEmptyDiffRetryCandidate, isWriterTransportRetryCandidate, projectFrontendNodeProtocolFailureReason, resolveFrontendPlanBackupRoute, resolveFrontendPlanRetryStep, VERIFY_SHELL_RETRY_POLICY, } from "./retry-policy.js";
16
+ import { computeBackoffDelayMs, isCanonicalFinalVerifyShellRetryCandidate, isRetryablePiFailureCategory, isSafeReadOnlyPiRetryCandidate, isTargetTemplateTransientRetryNode, isWriterEmptyDiffRetryCandidate, isWriterTransportRetryCandidate, OUTPUT_LIMIT_RETRY_CATEGORY, projectFrontendNodeProtocolFailureReason, resolveFrontendPlanBackupRoute, resolveFrontendPlanRetryStep, VERIFY_SHELL_RETRY_POLICY, } from "./retry-policy.js";
17
17
  import { applyNodeActivity, evaluateNodeLiveness, resolveLivenessPolicy, } from "./liveness-policy.js";
18
18
  import { buildProtocolRetryInstruction, normalizeReviewVerdictAfterRetries, parseJsonReviewVerdict, validateOutputProtocol, } from "./output-protocol.js";
19
19
  import { getStructuredContractValidator } from "./contract-output-registry.js";
@@ -257,6 +257,9 @@ function frontendPlanValidationRetryGuidance(reason) {
257
257
  if (/verificationTargets.*uiStates|uiStates.*verificationTargets/i.test(reason)) {
258
258
  guidance.push("For this retry, provide verificationTargets[].uiStates as an array; use [] when the target has no named UI state.");
259
259
  }
260
+ if (/duplicate|already recorded/i.test(reason)) {
261
+ guidance.push("For this retry, re-commit the corrected entry with the same id and replace=true; the ledger compiles the latest submission as the full replacement. A duplicate receipt is a correction invitation, not a prohibition, and backfilling a cited fact's reverse reference in-node does not re-derive committed facts.");
262
+ }
260
263
  return guidance;
261
264
  }
262
265
  function compactRetryText(text, maxChars) {
@@ -774,6 +777,87 @@ export function buildAttemptPrompt(task, basePrompt, attemptNumber, previousFail
774
777
  "</retry_instruction>",
775
778
  ].join("\n");
776
779
  }
780
+ // Repair-category guidance must outrank the retry ladder position: the
781
+ // ladder advances monotonically on transport failures (e.g. length →
782
+ // compact-terminal-first), and its "keep the committed ledger intact"
783
+ // instruction directly contradicts the repair action for invalid-output /
784
+ // truncated ledger facts (re-commit corrected record_* facts). When both
785
+ // apply, the model receives the repair instruction, not the rung script.
786
+ if (previousFailureCategory === "invalid-output" &&
787
+ task.structuredContractOutput &&
788
+ previousProtocolReason) {
789
+ // The frontend plan node's compile authority is the committed typed
790
+ // ledger, not a fenced JSON text artifact: its retry guidance must
791
+ // direct the model to re-commit corrected record_* facts and
792
+ // finalize_plan. The legacy full-contract JSON guidance below applies
793
+ // only to nodes whose authority is still a text contract artifact.
794
+ if (task.structuredContractOutput.schemaId ===
795
+ "frontend-implementation-contract-plan-patch-v1") {
796
+ const splitGuidance = /write-set-too-large|split the task/.test(previousProtocolReason ?? "")
797
+ ? [
798
+ "",
799
+ "The writeSet is too large for one implement node. Preserve the complete requirement and file coverage. This needs a Contract-level task split, not a Plan formatting repair. Report the write-set-too-large finding and the affected paths for the controller to resume Contract/split orchestration. Plan cannot change protected targets.files or call Contract-only tools; do not invent a split tool or discard required files.",
800
+ ]
801
+ : [];
802
+ return [
803
+ basePrompt,
804
+ "",
805
+ "<retry_instruction>",
806
+ "Previous plan ledger facts failed canonical contract validation:",
807
+ previousProtocolReason,
808
+ "Fix the reported violations by re-committing corrected record_* facts and calling finalize_plan exactly once. The committed typed ledger is the only compile authority.",
809
+ "Do not emit a fenced JSON contract, protected skeleton fields, or an openspec-citations block; narrative JSON is not validated and wastes the output budget.",
810
+ ...frontendPlanValidationRetryGuidance(previousProtocolReason),
811
+ ...splitGuidance,
812
+ "</retry_instruction>",
813
+ ].join("\n");
814
+ }
815
+ return [
816
+ basePrompt,
817
+ "",
818
+ "<retry_instruction>",
819
+ "Previous attempt produced an invalid frontend implementation contract:",
820
+ previousProtocolReason,
821
+ ...frontendStructuredArtifactRetryGuidance(task.structuredContractOutput.schemaId),
822
+ "Fix every reported field violation: do not emit null for optional fields, do not misspell field names, and match the required types exactly.",
823
+ "</retry_instruction>",
824
+ ].join("\n");
825
+ }
826
+ if (previousFailureCategory === "invalid-output" &&
827
+ task.id === "frontend-scout-pi") {
828
+ return [
829
+ basePrompt,
830
+ "",
831
+ "<retry_instruction>",
832
+ "The previous Scout attempt did not commit a complete, runtime-evidenced target surface.",
833
+ "Search the repository only as needed to establish the real entrypoint, implementation ownership, and applicable test path. Commit record_target_surface with completeness=complete and unresolvedPaths=[] only after at least one named target has fresh runtime evidence, unless the task source explicitly declares a greenfield target: in that case every future path must be source-declared by the runtime-enriched fact. If ownership truly cannot be established, record completeness=blocked with each unresolved path; do not make Plan discover it.",
834
+ "</retry_instruction>",
835
+ ].join("\n");
836
+ }
837
+ if (previousFailureCategory === "structured-output-truncated" &&
838
+ task.structuredContractOutput) {
839
+ if (task.structuredContractOutput.schemaId ===
840
+ "frontend-implementation-contract-plan-patch-v1") {
841
+ return [
842
+ basePrompt,
843
+ "",
844
+ "<retry_instruction>",
845
+ "Previous attempt was truncated by the provider (stopReason=length) before the plan facts were fully committed.",
846
+ "Re-commit the missing record_* facts and call finalize_plan exactly once; the committed typed ledger is the only compile authority. Do NOT re-read contract/scout outputs or source files — use the facts already in context. Commit record_* facts one tool call per message, then finalize_plan immediately.",
847
+ "Do not emit a fenced JSON contract, protected skeleton fields, or an openspec-citations block; narrative JSON is not validated and wastes the output budget.",
848
+ "</retry_instruction>",
849
+ ].join("\n");
850
+ }
851
+ return [
852
+ basePrompt,
853
+ "",
854
+ "<retry_instruction>",
855
+ "Previous attempt was truncated by the provider (stopReason=length) before the JSON contract was completed.",
856
+ ...frontendStructuredArtifactRetryGuidance(task.structuredContractOutput.schemaId),
857
+ "The JSON artifact must be complete; omit evidence excerpts and duplicated upstream context.",
858
+ "</retry_instruction>",
859
+ ].join("\n");
860
+ }
777
861
  if (frontendPlanRetryStep === "compact-terminal-first") {
778
862
  return [
779
863
  basePrompt,
@@ -845,83 +929,25 @@ export function buildAttemptPrompt(task, basePrompt, attemptNumber, previousFail
845
929
  "</retry_instruction>",
846
930
  ].join("\n");
847
931
  }
848
- if (previousFailureCategory === "invalid-output" &&
849
- task.structuredContractOutput &&
850
- previousProtocolReason) {
851
- // The frontend plan node's compile authority is the committed typed
852
- // ledger, not a fenced JSON text artifact: its retry guidance must
853
- // direct the model to re-commit corrected record_* facts and
854
- // finalize_plan. The legacy full-contract JSON guidance below applies
855
- // only to nodes whose authority is still a text contract artifact.
856
- if (task.structuredContractOutput.schemaId ===
857
- "frontend-implementation-contract-plan-patch-v1") {
858
- const splitGuidance = /write-set-too-large|split the task/.test(previousProtocolReason ?? "")
859
- ? [
860
- "",
861
- "The writeSet is too large for one implement node. Do NOT delete implementation files to squeeze under the limit — that drops required work. Split the task via record_split_proposal (or narrow targets.files to a genuine subset) so each implement node stays bounded; the full file set must remain covered across the split.",
862
- ]
863
- : [];
864
- return [
865
- basePrompt,
866
- "",
867
- "<retry_instruction>",
868
- "Previous plan ledger facts failed canonical contract validation:",
869
- previousProtocolReason,
870
- "Fix the reported violations by re-committing corrected record_* facts and calling finalize_plan exactly once. The committed typed ledger is the only compile authority.",
871
- "Do not emit a fenced JSON contract, protected skeleton fields, or an openspec-citations block; narrative JSON is not validated and wastes the output budget.",
872
- ...frontendPlanValidationRetryGuidance(previousProtocolReason),
873
- ...splitGuidance,
874
- "</retry_instruction>",
875
- ].join("\n");
876
- }
932
+ // Generic output-limit fallback. Every branch above this one carries a
933
+ // more precise instruction for the same capacity signal (contract repair
934
+ // reasons, ladder rungs — compact-terminal-first is the reason-mandated
935
+ // rung for length-before-terminal —, protocol/review/read-burst repair),
936
+ // so output-limit must not shadow them.
937
+ if (previousFailureCategory === OUTPUT_LIMIT_RETRY_CATEGORY) {
877
938
  return [
878
939
  basePrompt,
879
940
  "",
880
941
  "<retry_instruction>",
881
- "Previous attempt produced an invalid frontend implementation contract:",
882
- previousProtocolReason,
883
- ...frontendStructuredArtifactRetryGuidance(task.structuredContractOutput.schemaId),
884
- "Fix every reported field violation: do not emit null for optional fields, do not misspell field names, and match the required types exactly.",
885
- "</retry_instruction>",
886
- ].join("\n");
887
- }
888
- if (previousFailureCategory === "invalid-output" &&
889
- task.id === "frontend-scout-pi") {
890
- return [
891
- basePrompt,
892
- "",
893
- "<retry_instruction>",
894
- "The previous Scout attempt did not commit a complete, runtime-evidenced target surface.",
895
- "Search the repository only as needed to establish the real entrypoint, implementation ownership, and applicable test path. Commit record_target_surface with completeness=complete and unresolvedPaths=[] only after at least one named target has fresh runtime evidence, unless the task source explicitly declares a greenfield target: in that case every future path must be source-declared by the runtime-enriched fact. If ownership truly cannot be established, record completeness=blocked with each unresolved path; do not make Plan discover it.",
896
- "</retry_instruction>",
897
- ].join("\n");
898
- }
899
- if (previousFailureCategory === "structured-output-truncated" &&
900
- task.structuredContractOutput) {
901
- if (task.structuredContractOutput.schemaId ===
902
- "frontend-implementation-contract-plan-patch-v1") {
903
- return [
904
- basePrompt,
905
- "",
906
- "<retry_instruction>",
907
- "Previous attempt was truncated by the provider (stopReason=length) before the plan facts were fully committed.",
908
- "Re-commit the missing record_* facts and call finalize_plan exactly once; the committed typed ledger is the only compile authority. Do NOT re-read contract/scout outputs or source files — use the facts already in context. Commit record_* facts one tool call per message, then finalize_plan immediately.",
909
- "Do not emit a fenced JSON contract, protected skeleton fields, or an openspec-citations block; narrative JSON is not validated and wastes the output budget.",
910
- "</retry_instruction>",
911
- ].join("\n");
912
- }
913
- return [
914
- basePrompt,
915
- "",
916
- "<retry_instruction>",
917
- "Previous attempt was truncated by the provider (stopReason=length) before the JSON contract was completed.",
918
- ...frontendStructuredArtifactRetryGuidance(task.structuredContractOutput.schemaId),
919
- "The JSON artifact must be complete; omit evidence excerpts and duplicated upstream context.",
942
+ "The previous turn ended with stopReason=length before the required output was complete. This is output-capacity truncation, not empty output.",
943
+ "Continue incrementally from already committed typed facts and current in-scope workspace files. Do not repeat completed discovery, decisions, facts, or writes.",
944
+ "Generate the smallest unfinished unit next, validate its required format immediately, repair any format error in this node, and only then continue to the next unfinished unit or terminal tool.",
945
+ "Keep prose minimal and finish the required terminal/output protocol as soon as the remaining work is valid.",
920
946
  "</retry_instruction>",
921
947
  ].join("\n");
922
948
  }
923
949
  if (previousFailureCategory === "writer-empty-diff") {
924
- const maxAttempts = task.retryPolicy?.maxAttempts ?? 3;
950
+ const maxAttempts = task.retryPolicy?.maxAttempts ?? 5;
925
951
  // When a completeness progress exists for this writer, fold the concrete
926
952
  // target paths into the empty-diff retry so the model does not guess and
927
953
  // does not need to read a forbidden `.harness/**` evidence file.
@@ -949,7 +975,7 @@ export function buildAttemptPrompt(task, basePrompt, attemptNumber, previousFail
949
975
  ].join("\n");
950
976
  }
951
977
  if (previousFailureCategory === "incomplete-write-set") {
952
- const maxAttempts = task.retryPolicy?.maxAttempts ?? 3;
978
+ const maxAttempts = task.retryPolicy?.maxAttempts ?? 5;
953
979
  const bindingOnly = task.id?.startsWith("generate-backend-md-case-") === true &&
954
980
  (recoveryTargetPaths?.length ?? 0) === 1 &&
955
981
  Object.values(recoveryDiagnostics ?? {}).flat().some((detail) => /(?:unclassified Test Points|duplicate Test Point bindings)/i.test(detail));
@@ -1765,6 +1791,32 @@ export async function executeDagNode(input) {
1765
1791
  durationMs: 0,
1766
1792
  };
1767
1793
  }
1794
+ // stopReason=length on top of a bare empty-output verdict is a provider
1795
+ // capacity signal, not a true empty response: reroute it through the
1796
+ // dedicated output-limit retry path while keeping the raw category for
1797
+ // diagnostics. Bare empty-output and transport aliases (network,
1798
+ // nonzero-exit, unknown) are rerouted — categories that already carry a
1799
+ // precise repair instruction (output-too-large, invalid-output,
1800
+ // protocol-invalid, structured-output-truncated via the validators below,
1801
+ // writer categories) keep their classification so their exact retry
1802
+ // guidance still reaches the model.
1803
+ if (task.executor === "pi" &&
1804
+ !result.ok &&
1805
+ result.stopReason === "length" &&
1806
+ (result.failureCategory === undefined ||
1807
+ ["empty-output", "network", "nonzero-exit", "unknown"].includes(result.failureCategory))) {
1808
+ result = {
1809
+ ...result,
1810
+ rawFailureCategory: result.rawFailureCategory ?? result.failureCategory,
1811
+ failureCategory: OUTPUT_LIMIT_RETRY_CATEGORY,
1812
+ stderr: [
1813
+ result.stderr,
1814
+ "output-limit: stopReason=length; preserve completed work and retry only the unfinished output",
1815
+ ]
1816
+ .filter(Boolean)
1817
+ .join("\n"),
1818
+ };
1819
+ }
1768
1820
  if (task.id === "generate-backend-md-plan-pi" &&
1769
1821
  !result.ok &&
1770
1822
  (result.failureCategory === "output-too-large" ||
@@ -1898,7 +1950,7 @@ export async function executeDagNode(input) {
1898
1950
  !result.ok &&
1899
1951
  retryPolicy !== undefined) {
1900
1952
  const submissions = await countContractRecordSubmissions(runDir, nodeId);
1901
- if (submissions === 0) {
1953
+ if (submissions === 0 && result.stopReason !== "length") {
1902
1954
  result = {
1903
1955
  ...result,
1904
1956
  failureCategory: "empty-output",
@@ -408,6 +408,7 @@ export async function deriveDagRerunFeedback(input) {
408
408
  return (node?.status === "ERROR" &&
409
409
  [
410
410
  "empty-output",
411
+ "output-limit",
411
412
  "invalid-output",
412
413
  "writer-thinking-exhausted",
413
414
  "writer-budget-exhausted",
@@ -2,6 +2,7 @@ import { createHash } from "node:crypto";
2
2
  import { readFile } from "node:fs/promises";
3
3
  import { z } from "zod";
4
4
  import { assertTaskContractBindingFresh } from "./task-contract-binding.js";
5
+ import { workspaceDriftedPaths, } from "./workspace-checkpoint.js";
5
6
  import { isSafeReadOnlyPiRetryCandidate, isTransportCrashedExclusiveWriter } from "./retry-policy.js";
6
7
  import { resolveDagTaskSourcePath } from "../../task/dag-source-paths.js";
7
8
  import { composeDagPromptOverridePrompt, MAX_DAG_RERUN_PROMPT_OVERRIDE_CHARS, } from "../../shared/dag-prompt-override.js";
@@ -1110,6 +1111,46 @@ function deriveSuggestedAction(input) {
1110
1111
  }
1111
1112
  return "manual";
1112
1113
  }
1114
+ /**
1115
+ * Gitignore-style-lite matcher for readSet/writeSet path patterns in scoped
1116
+ * workspace-drift evaluation: literal paths match exactly; `**` spans path
1117
+ * segments (a trailing `/**` also matches when nothing follows), `*` stays
1118
+ * within one segment, `?` is one non-separator char.
1119
+ */
1120
+ function pathMatchesWorkspacePattern(pathValue, pattern) {
1121
+ const normalizedPath = pathValue.replace(/\\/g, "/").replace(/^\.\//, "");
1122
+ const trimmed = pattern.trim().replace(/\\/g, "/").replace(/^\.\//, "");
1123
+ if (!trimmed || trimmed === "/")
1124
+ return false;
1125
+ if (trimmed === normalizedPath)
1126
+ return true;
1127
+ let source = "";
1128
+ for (let index = 0; index < trimmed.length; index += 1) {
1129
+ const char = trimmed[index];
1130
+ if (char === "*") {
1131
+ if (trimmed[index + 1] === "*") {
1132
+ if (trimmed[index + 2] === "/") {
1133
+ source += "(?:.*/)?";
1134
+ index += 2;
1135
+ }
1136
+ else {
1137
+ source += ".*";
1138
+ index += 1;
1139
+ }
1140
+ }
1141
+ else {
1142
+ source += "[^/]*";
1143
+ }
1144
+ }
1145
+ else if (char === "?") {
1146
+ source += "[^/]";
1147
+ }
1148
+ else {
1149
+ source += char.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
1150
+ }
1151
+ }
1152
+ return new RegExp(`^${source}$`).test(normalizedPath);
1153
+ }
1113
1154
  function deriveRisk(reasonCodes, resetCount) {
1114
1155
  if (reasonCodes.includes("restart-subgraph-contains-writer") ||
1115
1156
  reasonCodes.includes("restart-subgraph-contains-unsafe-shell") ||
@@ -1213,7 +1254,14 @@ export async function evaluateDagRerunPlan(input) {
1213
1254
  blockedReasons.push("parent-lifecycle-ineligible");
1214
1255
  }
1215
1256
  else if (input.parentState.status !== "failed" &&
1216
- input.parentState.status !== "partial_failed") {
1257
+ input.parentState.status !== "partial_failed" &&
1258
+ // A superseded parent is a terminal operator judgment about the run as
1259
+ // a whole, not about its deterministic read-only compute: its FINISHED
1260
+ // upstream nodes are still valid fact donors for the imported set, and
1261
+ // every unsafe case (writers in the reset subgraph, decision gates,
1262
+ // stale checkpoints, unresolved ERROR nodes outside the closure) keeps
1263
+ // its own dedicated block below.
1264
+ input.parentState.status !== "superseded") {
1217
1265
  reasonCodes.push("parent-lifecycle-ineligible");
1218
1266
  blockedReasons.push("parent-lifecycle-ineligible");
1219
1267
  }
@@ -1241,8 +1289,36 @@ export async function evaluateDagRerunPlan(input) {
1241
1289
  else if (input.currentWorkspace?.fingerprint &&
1242
1290
  input.parentTerminalWorkspace.fingerprint !==
1243
1291
  input.currentWorkspace.fingerprint) {
1244
- reasonCodes.push("workspace-drift");
1245
- blockedReasons.push("workspace-drift");
1292
+ // Whole-fingerprint drift is refined to content-level paths: the rerun
1293
+ // re-derives (readSet) and rewrites (writeSet) the reset subgraph's own
1294
+ // paths, so drift confined to those paths is recomputed by the rerun
1295
+ // itself and must not force a full standalone rerun. Drift anywhere
1296
+ // else — especially inputs of imported fact-reuse nodes — keeps the
1297
+ // strict block. Indeterminate checkpoints fail closed.
1298
+ const driftedPaths = workspaceDriftedPaths(input.parentTerminalWorkspace, input.currentWorkspace);
1299
+ const importedReadPaths = importedNodeIds.flatMap((nodeId) => {
1300
+ const task = tasksByIdForPlan.get(nodeId);
1301
+ return task?.readSet ?? [];
1302
+ });
1303
+ const scopedDrift = driftedPaths !== undefined &&
1304
+ driftedPaths.length > 0 &&
1305
+ driftedPaths.every((driftPath) => !importedReadPaths.some((pattern) => pathMatchesWorkspacePattern(driftPath, pattern)) &&
1306
+ resetNodeIds.some((nodeId) => {
1307
+ const task = tasksByIdForPlan.get(nodeId);
1308
+ if (!task)
1309
+ return false;
1310
+ return [
1311
+ ...(task.writeSet ?? []),
1312
+ ...(task.readSet ?? []),
1313
+ ].some((pattern) => pathMatchesWorkspacePattern(driftPath, pattern));
1314
+ }));
1315
+ if (scopedDrift) {
1316
+ reasonCodes.push("workspace-drift-scoped");
1317
+ }
1318
+ else {
1319
+ reasonCodes.push("workspace-drift");
1320
+ blockedReasons.push("workspace-drift");
1321
+ }
1246
1322
  }
1247
1323
  else if (!input.currentWorkspace?.fingerprint) {
1248
1324
  reasonCodes.push("workspace-checkpoint-missing");
@@ -1293,8 +1369,18 @@ export async function evaluateDagRerunPlan(input) {
1293
1369
  }
1294
1370
  for (const nodeId of importedNodeIds) {
1295
1371
  const record = input.parentState.nodes[nodeId];
1296
- if (!record)
1372
+ if (!record) {
1373
+ reasonCodes.push("parent-facts-invalid");
1374
+ blockedReasons.push("parent-facts-invalid");
1375
+ blockingNodes.push({ nodeId, reasonCode: "parent-facts-invalid" });
1297
1376
  continue;
1377
+ }
1378
+ if (record.status === "PENDING" || record.status === "RUNNING") {
1379
+ reasonCodes.push("parent-facts-invalid");
1380
+ blockedReasons.push("parent-facts-invalid");
1381
+ blockingNodes.push({ nodeId, reasonCode: "parent-facts-invalid" });
1382
+ continue;
1383
+ }
1298
1384
  if (record.status === "ERROR") {
1299
1385
  reasonCodes.push("parent-facts-invalid");
1300
1386
  blockedReasons.push("parent-facts-invalid");
@@ -784,6 +784,13 @@ async function stageContinuationRun(input) {
784
784
  if (!parentRecord) {
785
785
  throw new Error(`parent node record missing for imported node: ${nodeId}`);
786
786
  }
787
+ if (parentRecord.status !== "FINISHED" &&
788
+ !(parentRecord.status === "SKIPPED" &&
789
+ (parentRecord.skippedReason?.includes("condition") ||
790
+ parentRecord.skippedReason?.includes("runIf") ||
791
+ parentRecord.skippedReason?.includes("run-if")))) {
792
+ throw new Error(`parent node is not a settled fact donor for imported node: ${nodeId}`);
793
+ }
787
794
  const { manifestNode, importedRecord } = await importNodeFacts({
788
795
  parentRunDir: input.parentRunDir,
789
796
  newRunDir: runDir,
@@ -7,8 +7,8 @@ import { TYPED_EVENT_FACT_KINDS } from "./frontend-typed-event-store.js";
7
7
  * is re-exported here unchanged so retry-policy stays the single import
8
8
  * surface for retry policies and helpers.
9
9
  */
10
- export { ALL_DAG_RETRY_CATEGORIES, CONTEXT_OVERFLOW_RETRY_CATEGORY, DEFAULT_DAG_RETRY_CATEGORIES, INCOMPLETE_WRITE_SET_RETRY_CATEGORY, PROTOCOL_AWARE_DAG_RETRY_CATEGORIES, PROTOCOL_INVALID_RETRY_CATEGORY, READ_BURST_RETRY_CATEGORY, REVIEW_TERMINAL_MISSING_RETRY_CATEGORY, STRUCTURED_ARTIFACT_INVALID_RETRY_CATEGORY, STRUCTURED_OUTPUT_RETRY_CATEGORY, STRUCTURED_OUTPUT_TRUNCATED_RETRY_CATEGORY, STRUCTURED_REQUIRED_DAG_RETRY_CATEGORIES, TARGET_TEMPLATE_TRANSIENT_RETRY_PROFILE, WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY, WRITER_EMPTY_DIFF_RETRY_CATEGORY, dagRetryBackoffSchema, dagRetryCategorySchema, dagRetryPolicySchema, } from "./dag-retry-schema.js";
11
- import { CONTEXT_OVERFLOW_RETRY_CATEGORY, DEFAULT_DAG_RETRY_CATEGORIES, INCOMPLETE_WRITE_SET_RETRY_CATEGORY, PROTOCOL_AWARE_DAG_RETRY_CATEGORIES, READ_BURST_RETRY_CATEGORY, REVIEW_TERMINAL_MISSING_RETRY_CATEGORY, STRUCTURED_ARTIFACT_INVALID_RETRY_CATEGORY, STRUCTURED_OUTPUT_RETRY_CATEGORY, STRUCTURED_REQUIRED_DAG_RETRY_CATEGORIES, TARGET_TEMPLATE_TRANSIENT_RETRY_PROFILE, WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY, WRITER_EMPTY_DIFF_RETRY_CATEGORY, } from "./dag-retry-schema.js";
10
+ export { ALL_DAG_RETRY_CATEGORIES, CONTEXT_OVERFLOW_RETRY_CATEGORY, DEFAULT_DAG_RETRY_CATEGORIES, INCOMPLETE_WRITE_SET_RETRY_CATEGORY, OUTPUT_LIMIT_RETRY_CATEGORY, PROTOCOL_AWARE_DAG_RETRY_CATEGORIES, PROTOCOL_INVALID_RETRY_CATEGORY, READ_BURST_RETRY_CATEGORY, REVIEW_TERMINAL_MISSING_RETRY_CATEGORY, STRUCTURED_ARTIFACT_INVALID_RETRY_CATEGORY, STRUCTURED_OUTPUT_RETRY_CATEGORY, STRUCTURED_OUTPUT_TRUNCATED_RETRY_CATEGORY, STRUCTURED_REQUIRED_DAG_RETRY_CATEGORIES, TARGET_TEMPLATE_TRANSIENT_RETRY_PROFILE, WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY, WRITER_EMPTY_DIFF_RETRY_CATEGORY, dagRetryBackoffSchema, dagRetryCategorySchema, dagRetryPolicySchema, } from "./dag-retry-schema.js";
11
+ import { CONTEXT_OVERFLOW_RETRY_CATEGORY, DEFAULT_DAG_RETRY_CATEGORIES, INCOMPLETE_WRITE_SET_RETRY_CATEGORY, OUTPUT_LIMIT_RETRY_CATEGORY, PROTOCOL_AWARE_DAG_RETRY_CATEGORIES, READ_BURST_RETRY_CATEGORY, REVIEW_TERMINAL_MISSING_RETRY_CATEGORY, STRUCTURED_ARTIFACT_INVALID_RETRY_CATEGORY, STRUCTURED_OUTPUT_RETRY_CATEGORY, STRUCTURED_REQUIRED_DAG_RETRY_CATEGORIES, TARGET_TEMPLATE_TRANSIENT_RETRY_PROFILE, WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY, WRITER_EMPTY_DIFF_RETRY_CATEGORY, } from "./dag-retry-schema.js";
12
12
  const RETRY_SAFE_PI_ROLES = new Set([
13
13
  "planner",
14
14
  "scout",
@@ -19,7 +19,7 @@ const RETRY_SAFE_PI_ROLES = new Set([
19
19
  ]);
20
20
  /**
21
21
  * The default retry policy applied to safe generated read-only Pi nodes.
22
- * Total attempts: 3, exponential backoff with cap.
22
+ * Total attempts: 5, exponential backoff with cap.
23
23
  *
24
24
  * Includes `context-overflow`: a read-only node that blew the context window
25
25
  * (400 request too large, e.g. a review node accumulating too many reads) is a
@@ -29,7 +29,7 @@ const RETRY_SAFE_PI_ROLES = new Set([
29
29
  * design-review/review fell straight to ERROR without a retry attempt.
30
30
  */
31
31
  export const DEFAULT_READ_ONLY_PI_RETRY_POLICY = {
32
- maxAttempts: 3,
32
+ maxAttempts: 5,
33
33
  backoff: "exponential",
34
34
  initialDelayMs: 2000,
35
35
  maxDelayMs: 30000,
@@ -125,7 +125,10 @@ export const WRITER_EMPTY_DIFF_RETRY_POLICY = {
125
125
  backoff: "exponential",
126
126
  initialDelayMs: 2000,
127
127
  maxDelayMs: 30000,
128
- retryCategories: [WRITER_EMPTY_DIFF_RETRY_CATEGORY],
128
+ retryCategories: [
129
+ WRITER_EMPTY_DIFF_RETRY_CATEGORY,
130
+ OUTPUT_LIMIT_RETRY_CATEGORY,
131
+ ],
129
132
  };
130
133
  /**
131
134
  * Bounded transport retry for standard exclusive implementers when a provider
@@ -140,6 +143,7 @@ export const WRITER_TRANSPORT_RETRY_POLICY = {
140
143
  retryCategories: [
141
144
  WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY,
142
145
  CONTEXT_OVERFLOW_RETRY_CATEGORY,
146
+ OUTPUT_LIMIT_RETRY_CATEGORY,
143
147
  ],
144
148
  };
145
149
  /**
@@ -175,6 +179,7 @@ export const TARGET_TEMPLATE_WRITER_TRANSPORT_RETRY_POLICY = {
175
179
  retryCategories: [
176
180
  WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY,
177
181
  CONTEXT_OVERFLOW_RETRY_CATEGORY,
182
+ OUTPUT_LIMIT_RETRY_CATEGORY,
178
183
  ],
179
184
  };
180
185
  /**
@@ -182,13 +187,14 @@ export const TARGET_TEMPLATE_WRITER_TRANSPORT_RETRY_POLICY = {
182
187
  * recovery attempts driven by Completeness Gate (missing/broken target files).
183
188
  */
184
189
  export const BACKEND_TEST_WRITER_COMPLETENESS_RETRY_POLICY = {
185
- maxAttempts: 3,
190
+ maxAttempts: 5,
186
191
  backoff: "exponential",
187
192
  initialDelayMs: 2000,
188
193
  maxDelayMs: 30000,
189
194
  retryCategories: [
190
195
  WRITER_EMPTY_DIFF_RETRY_CATEGORY,
191
196
  INCOMPLETE_WRITE_SET_RETRY_CATEGORY,
197
+ OUTPUT_LIMIT_RETRY_CATEGORY,
192
198
  ],
193
199
  };
194
200
  /** Markdown shard writers use one full attempt plus at most one bounded binding repair. */
@@ -472,9 +478,9 @@ function reasonMandatedFrontendPlanRetryStep(reason) {
472
478
  }
473
479
  }
474
480
  /**
475
- * Opt-in retry policy mounted on the frontend-plan-pi node. maxAttempts=3
476
- * bounds plan retry cost: the four-rung ladder is only partially traversable
477
- * before the run fails non-converging (rungs: normal → bounded-tool-only →
481
+ * Opt-in retry policy mounted on the frontend-plan-pi node. maxAttempts=5
482
+ * lets the producer validate and repair each generated increment while still
483
+ * bounding the four-rung ladder (normal → bounded-tool-only →
478
484
  * compact-terminal-first → backup-model). `invalid-output` (typed-fact schema
479
485
  * violations, e.g. verification targets referencing undeclared UI states)
480
486
  * retries at the `normal` rung with the contractCheck reason injected via
@@ -485,7 +491,7 @@ export const FRONTEND_PLAN_LADDER_RETRY_POLICY = {
485
491
  // incremental record_* calls; dense call bursts hit provider rate limits
486
492
  // whose windows exceed the old 30s cap. Longer backoff gives the limit
487
493
  // window time to expire before the next attempt.
488
- maxAttempts: 3,
494
+ maxAttempts: 5,
489
495
  backoff: "exponential",
490
496
  initialDelayMs: 5000,
491
497
  maxDelayMs: 60000,
@@ -0,0 +1,125 @@
1
+ import { appendFileSync, mkdirSync, unlinkSync } from "node:fs";
2
+ import { hostname } from "node:os";
3
+ import path from "node:path";
4
+ /**
5
+ * Runner exit-time diagnostics (silent-death forensics).
6
+ *
7
+ * A DAG runner can disappear without a crash report and without stderr (five
8
+ * reproductions during frontend recovery/long-plan runs, 2026-09-06): when a
9
+ * supervisor hard-kills the process or the event loop is wedged, no JS runs,
10
+ * so nothing is persisted and the run is only discoverable later as an
11
+ * orphaned RUNNING state with a frozen heartbeat. That leaves no evidence of
12
+ * WHERE the runner was when it died or whether the exit was even observable
13
+ * by Node.
14
+ *
15
+ * This module gives every abnormal death a durable trace next to state.json:
16
+ * - an `armed` line is written when the checkpoint starts executing;
17
+ * - a `process-exit` line is appended from the process `exit` hook with the
18
+ * exit code and a live state snapshot (status, terminalReason, RUNNING
19
+ * nodes, heartbeat staleness);
20
+ * - `uncaught-exception` / `unhandled-rejection` lines record the error
21
+ * before the process is allowed to crash exactly as it would have
22
+ * via uncaughtExceptionMonitor without changing the host's crash policy.
23
+ *
24
+ * On a clean terminal return the runner calls stop(), which removes every
25
+ * listener and deletes the journal, so completed runs carry no noise. A file
26
+ * left behind therefore means: diagnostics were armed, then the process went
27
+ * away through a path Node could not observe (`armed` line only → external
28
+ * SIGKILL/OOM/event-loop wedge) or through an observed abnormal exit (`armed`
29
+ * + one or more event lines → the reason is recorded in the file).
30
+ *
31
+ * Journal writes are strictly synchronous and best-effort: an append failure
32
+ * must never alter the runner's own crash/exit behavior.
33
+ */
34
+ export const RUNNER_EXIT_DIAGNOSTICS_FILE = "runner-exit-diagnostics.jsonl";
35
+ function describeError(error) {
36
+ if (error instanceof Error) {
37
+ return { name: error.name, message: error.message };
38
+ }
39
+ try {
40
+ return { message: String(error) };
41
+ }
42
+ catch {
43
+ return { message: "<unstringifiable error>" };
44
+ }
45
+ }
46
+ function appendJournalLineSync(journalPath, record) {
47
+ try {
48
+ mkdirSync(path.dirname(journalPath), { recursive: true });
49
+ appendFileSync(journalPath, `${JSON.stringify(record)}\n`, "utf8");
50
+ }
51
+ catch {
52
+ // Best-effort: journaling must never crash or alter exit behavior.
53
+ }
54
+ }
55
+ function removeJournalSync(journalPath) {
56
+ try {
57
+ unlinkSync(journalPath);
58
+ }
59
+ catch {
60
+ // Tolerate a missing journal (e.g. nothing was ever written).
61
+ }
62
+ }
63
+ export function installRunnerExitDiagnostics(options) {
64
+ const journalPath = path.join(options.runDir, RUNNER_EXIT_DIAGNOSTICS_FILE);
65
+ const target = options.target ?? process;
66
+ const armedAt = Date.now();
67
+ let stopped = false;
68
+ const base = {
69
+ runId: options.runId,
70
+ title: options.title,
71
+ pid: process.pid,
72
+ hostname: hostname(),
73
+ };
74
+ const safeSnapshot = () => {
75
+ try {
76
+ return options.snapshot();
77
+ }
78
+ catch {
79
+ return { status: "<snapshot-error>", running: [] };
80
+ }
81
+ };
82
+ const record = (kind, extra) => ({
83
+ ts: new Date().toISOString(),
84
+ kind,
85
+ ...base,
86
+ ...(extra ?? {}),
87
+ state: safeSnapshot(),
88
+ });
89
+ appendJournalLineSync(journalPath, {
90
+ ...record("armed"),
91
+ armedAt: new Date(armedAt).toISOString(),
92
+ });
93
+ const onProcessExit = (code) => {
94
+ if (stopped)
95
+ return;
96
+ appendJournalLineSync(journalPath, record("process-exit", {
97
+ code: typeof code === "number" ? code : process.exitCode ?? 0,
98
+ elapsedMs: Date.now() - armedAt,
99
+ }));
100
+ };
101
+ const onUncaughtException = (error, origin) => {
102
+ if (stopped)
103
+ return;
104
+ appendJournalLineSync(journalPath, record(origin === "unhandledRejection" ? "unhandled-rejection" : "uncaught-exception", {
105
+ error: describeError(error),
106
+ elapsedMs: Date.now() - armedAt,
107
+ }));
108
+ // Monitor listeners neither suppress fatal errors nor force a crash
109
+ // when the host has its own handlers or a nonfatal rejection policy.
110
+ };
111
+ target.on("exit", onProcessExit);
112
+ target.on("uncaughtExceptionMonitor", onUncaughtException);
113
+ return {
114
+ journalPath,
115
+ stop: () => {
116
+ if (stopped)
117
+ return;
118
+ stopped = true;
119
+ target.removeListener("exit", onProcessExit);
120
+ target.removeListener("uncaughtExceptionMonitor", onUncaughtException);
121
+ // Clean completion: leave completed runs free of diagnostics noise.
122
+ removeJournalSync(journalPath);
123
+ },
124
+ };
125
+ }