@tea-agent/loop-agent 0.42.0-next.15 → 0.42.0-next.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/dist/application/task-lifecycle/advance.js +9 -4
- package/dist/build-stamp.json +3 -3
- package/dist/commands/task-source-prepare.js +3 -1
- package/dist/executors/dag-pi-executor.js +825 -65
- package/dist/executors/shell-executor.js +103 -35
- package/dist/shared/dag-failure-category.js +6 -0
- package/dist/task/contract/apply.js +36 -2
- package/dist/task/source-prepare/parse-intent.js +7 -0
- package/dist/workflows/dag/dag-retry-schema.js +3 -0
- package/dist/workflows/dag/frontend-design-policy.js +1 -1
- package/dist/workflows/dag/frontend-risk.js +2 -0
- package/dist/workflows/dag/frontend-shadow-dual-write.js +16 -1
- package/dist/workflows/dag/frontend-shape.js +16 -6
- package/dist/workflows/dag/frontend-test-execution-evidence.js +48 -0
- package/dist/workflows/dag/frontend-verification-trace.js +32 -0
- package/dist/workflows/dag/init-hybrid.js +9 -10
- package/dist/workflows/dag/node-execution.js +124 -72
- package/dist/workflows/dag/rerun-feedback.js +1 -0
- package/dist/workflows/dag/rerun-plan.js +90 -4
- package/dist/workflows/dag/rerun-run.js +7 -0
- package/dist/workflows/dag/retry-policy.js +16 -10
- package/dist/workflows/dag/runner-exit-diagnostics.js +125 -0
- package/dist/workflows/dag/runner.js +67 -7
- package/dist/workflows/dag/structured-output-repair.js +4 -1
- package/dist/workflows/dag/validate.js +10 -8
- package/dist/workflows/dag/workspace-checkpoint.js +66 -0
- package/docs/templates/agent-dag.schema.json +2 -2
- package/package.json +1 -1
- package/skills/frontend-plan/SKILL.md +6 -1
- package/skills/frontend-plan/references/decision-contract.md +69 -18
- package/skills/frontend-plan/references/design-decisions.md +32 -0
|
@@ -13,7 +13,7 @@ import { buildDagNodePromptEnvelope, formatConvergenceFeedbackBlock, } from "./p
|
|
|
13
13
|
import { persistLongNodeOutputArtifacts } from "./upstream-artifacts.js";
|
|
14
14
|
import { materializeDeclaredArtifactFacts } from "./artifact-bindings.js";
|
|
15
15
|
import { buildOutputLimitRecoverySection, loadBackendTestWriterProgressForRetry, } from "./backend-test-writer-completeness.js";
|
|
16
|
-
import { computeBackoffDelayMs, isCanonicalFinalVerifyShellRetryCandidate, isRetryablePiFailureCategory, isSafeReadOnlyPiRetryCandidate, isTargetTemplateTransientRetryNode, isWriterEmptyDiffRetryCandidate, isWriterTransportRetryCandidate, projectFrontendNodeProtocolFailureReason, resolveFrontendPlanBackupRoute, resolveFrontendPlanRetryStep, VERIFY_SHELL_RETRY_POLICY, } from "./retry-policy.js";
|
|
16
|
+
import { computeBackoffDelayMs, isCanonicalFinalVerifyShellRetryCandidate, isRetryablePiFailureCategory, isSafeReadOnlyPiRetryCandidate, isTargetTemplateTransientRetryNode, isWriterEmptyDiffRetryCandidate, isWriterTransportRetryCandidate, OUTPUT_LIMIT_RETRY_CATEGORY, projectFrontendNodeProtocolFailureReason, resolveFrontendPlanBackupRoute, resolveFrontendPlanRetryStep, VERIFY_SHELL_RETRY_POLICY, } from "./retry-policy.js";
|
|
17
17
|
import { applyNodeActivity, evaluateNodeLiveness, resolveLivenessPolicy, } from "./liveness-policy.js";
|
|
18
18
|
import { buildProtocolRetryInstruction, normalizeReviewVerdictAfterRetries, parseJsonReviewVerdict, validateOutputProtocol, } from "./output-protocol.js";
|
|
19
19
|
import { getStructuredContractValidator } from "./contract-output-registry.js";
|
|
@@ -257,6 +257,9 @@ function frontendPlanValidationRetryGuidance(reason) {
|
|
|
257
257
|
if (/verificationTargets.*uiStates|uiStates.*verificationTargets/i.test(reason)) {
|
|
258
258
|
guidance.push("For this retry, provide verificationTargets[].uiStates as an array; use [] when the target has no named UI state.");
|
|
259
259
|
}
|
|
260
|
+
if (/duplicate|already recorded/i.test(reason)) {
|
|
261
|
+
guidance.push("For this retry, re-commit the corrected entry with the same id and replace=true; the ledger compiles the latest submission as the full replacement. A duplicate receipt is a correction invitation, not a prohibition, and backfilling a cited fact's reverse reference in-node does not re-derive committed facts.");
|
|
262
|
+
}
|
|
260
263
|
return guidance;
|
|
261
264
|
}
|
|
262
265
|
function compactRetryText(text, maxChars) {
|
|
@@ -774,6 +777,87 @@ export function buildAttemptPrompt(task, basePrompt, attemptNumber, previousFail
|
|
|
774
777
|
"</retry_instruction>",
|
|
775
778
|
].join("\n");
|
|
776
779
|
}
|
|
780
|
+
// Repair-category guidance must outrank the retry ladder position: the
|
|
781
|
+
// ladder advances monotonically on transport failures (e.g. length →
|
|
782
|
+
// compact-terminal-first), and its "keep the committed ledger intact"
|
|
783
|
+
// instruction directly contradicts the repair action for invalid-output /
|
|
784
|
+
// truncated ledger facts (re-commit corrected record_* facts). When both
|
|
785
|
+
// apply, the model receives the repair instruction, not the rung script.
|
|
786
|
+
if (previousFailureCategory === "invalid-output" &&
|
|
787
|
+
task.structuredContractOutput &&
|
|
788
|
+
previousProtocolReason) {
|
|
789
|
+
// The frontend plan node's compile authority is the committed typed
|
|
790
|
+
// ledger, not a fenced JSON text artifact: its retry guidance must
|
|
791
|
+
// direct the model to re-commit corrected record_* facts and
|
|
792
|
+
// finalize_plan. The legacy full-contract JSON guidance below applies
|
|
793
|
+
// only to nodes whose authority is still a text contract artifact.
|
|
794
|
+
if (task.structuredContractOutput.schemaId ===
|
|
795
|
+
"frontend-implementation-contract-plan-patch-v1") {
|
|
796
|
+
const splitGuidance = /write-set-too-large|split the task/.test(previousProtocolReason ?? "")
|
|
797
|
+
? [
|
|
798
|
+
"",
|
|
799
|
+
"The writeSet is too large for one implement node. Preserve the complete requirement and file coverage. This needs a Contract-level task split, not a Plan formatting repair. Report the write-set-too-large finding and the affected paths for the controller to resume Contract/split orchestration. Plan cannot change protected targets.files or call Contract-only tools; do not invent a split tool or discard required files.",
|
|
800
|
+
]
|
|
801
|
+
: [];
|
|
802
|
+
return [
|
|
803
|
+
basePrompt,
|
|
804
|
+
"",
|
|
805
|
+
"<retry_instruction>",
|
|
806
|
+
"Previous plan ledger facts failed canonical contract validation:",
|
|
807
|
+
previousProtocolReason,
|
|
808
|
+
"Fix the reported violations by re-committing corrected record_* facts and calling finalize_plan exactly once. The committed typed ledger is the only compile authority.",
|
|
809
|
+
"Do not emit a fenced JSON contract, protected skeleton fields, or an openspec-citations block; narrative JSON is not validated and wastes the output budget.",
|
|
810
|
+
...frontendPlanValidationRetryGuidance(previousProtocolReason),
|
|
811
|
+
...splitGuidance,
|
|
812
|
+
"</retry_instruction>",
|
|
813
|
+
].join("\n");
|
|
814
|
+
}
|
|
815
|
+
return [
|
|
816
|
+
basePrompt,
|
|
817
|
+
"",
|
|
818
|
+
"<retry_instruction>",
|
|
819
|
+
"Previous attempt produced an invalid frontend implementation contract:",
|
|
820
|
+
previousProtocolReason,
|
|
821
|
+
...frontendStructuredArtifactRetryGuidance(task.structuredContractOutput.schemaId),
|
|
822
|
+
"Fix every reported field violation: do not emit null for optional fields, do not misspell field names, and match the required types exactly.",
|
|
823
|
+
"</retry_instruction>",
|
|
824
|
+
].join("\n");
|
|
825
|
+
}
|
|
826
|
+
if (previousFailureCategory === "invalid-output" &&
|
|
827
|
+
task.id === "frontend-scout-pi") {
|
|
828
|
+
return [
|
|
829
|
+
basePrompt,
|
|
830
|
+
"",
|
|
831
|
+
"<retry_instruction>",
|
|
832
|
+
"The previous Scout attempt did not commit a complete, runtime-evidenced target surface.",
|
|
833
|
+
"Search the repository only as needed to establish the real entrypoint, implementation ownership, and applicable test path. Commit record_target_surface with completeness=complete and unresolvedPaths=[] only after at least one named target has fresh runtime evidence, unless the task source explicitly declares a greenfield target: in that case every future path must be source-declared by the runtime-enriched fact. If ownership truly cannot be established, record completeness=blocked with each unresolved path; do not make Plan discover it.",
|
|
834
|
+
"</retry_instruction>",
|
|
835
|
+
].join("\n");
|
|
836
|
+
}
|
|
837
|
+
if (previousFailureCategory === "structured-output-truncated" &&
|
|
838
|
+
task.structuredContractOutput) {
|
|
839
|
+
if (task.structuredContractOutput.schemaId ===
|
|
840
|
+
"frontend-implementation-contract-plan-patch-v1") {
|
|
841
|
+
return [
|
|
842
|
+
basePrompt,
|
|
843
|
+
"",
|
|
844
|
+
"<retry_instruction>",
|
|
845
|
+
"Previous attempt was truncated by the provider (stopReason=length) before the plan facts were fully committed.",
|
|
846
|
+
"Re-commit the missing record_* facts and call finalize_plan exactly once; the committed typed ledger is the only compile authority. Do NOT re-read contract/scout outputs or source files — use the facts already in context. Commit record_* facts one tool call per message, then finalize_plan immediately.",
|
|
847
|
+
"Do not emit a fenced JSON contract, protected skeleton fields, or an openspec-citations block; narrative JSON is not validated and wastes the output budget.",
|
|
848
|
+
"</retry_instruction>",
|
|
849
|
+
].join("\n");
|
|
850
|
+
}
|
|
851
|
+
return [
|
|
852
|
+
basePrompt,
|
|
853
|
+
"",
|
|
854
|
+
"<retry_instruction>",
|
|
855
|
+
"Previous attempt was truncated by the provider (stopReason=length) before the JSON contract was completed.",
|
|
856
|
+
...frontendStructuredArtifactRetryGuidance(task.structuredContractOutput.schemaId),
|
|
857
|
+
"The JSON artifact must be complete; omit evidence excerpts and duplicated upstream context.",
|
|
858
|
+
"</retry_instruction>",
|
|
859
|
+
].join("\n");
|
|
860
|
+
}
|
|
777
861
|
if (frontendPlanRetryStep === "compact-terminal-first") {
|
|
778
862
|
return [
|
|
779
863
|
basePrompt,
|
|
@@ -845,83 +929,25 @@ export function buildAttemptPrompt(task, basePrompt, attemptNumber, previousFail
|
|
|
845
929
|
"</retry_instruction>",
|
|
846
930
|
].join("\n");
|
|
847
931
|
}
|
|
848
|
-
|
|
849
|
-
|
|
850
|
-
|
|
851
|
-
|
|
852
|
-
|
|
853
|
-
|
|
854
|
-
// finalize_plan. The legacy full-contract JSON guidance below applies
|
|
855
|
-
// only to nodes whose authority is still a text contract artifact.
|
|
856
|
-
if (task.structuredContractOutput.schemaId ===
|
|
857
|
-
"frontend-implementation-contract-plan-patch-v1") {
|
|
858
|
-
const splitGuidance = /write-set-too-large|split the task/.test(previousProtocolReason ?? "")
|
|
859
|
-
? [
|
|
860
|
-
"",
|
|
861
|
-
"The writeSet is too large for one implement node. Do NOT delete implementation files to squeeze under the limit — that drops required work. Split the task via record_split_proposal (or narrow targets.files to a genuine subset) so each implement node stays bounded; the full file set must remain covered across the split.",
|
|
862
|
-
]
|
|
863
|
-
: [];
|
|
864
|
-
return [
|
|
865
|
-
basePrompt,
|
|
866
|
-
"",
|
|
867
|
-
"<retry_instruction>",
|
|
868
|
-
"Previous plan ledger facts failed canonical contract validation:",
|
|
869
|
-
previousProtocolReason,
|
|
870
|
-
"Fix the reported violations by re-committing corrected record_* facts and calling finalize_plan exactly once. The committed typed ledger is the only compile authority.",
|
|
871
|
-
"Do not emit a fenced JSON contract, protected skeleton fields, or an openspec-citations block; narrative JSON is not validated and wastes the output budget.",
|
|
872
|
-
...frontendPlanValidationRetryGuidance(previousProtocolReason),
|
|
873
|
-
...splitGuidance,
|
|
874
|
-
"</retry_instruction>",
|
|
875
|
-
].join("\n");
|
|
876
|
-
}
|
|
932
|
+
// Generic output-limit fallback. Every branch above this one carries a
|
|
933
|
+
// more precise instruction for the same capacity signal (contract repair
|
|
934
|
+
// reasons, ladder rungs — compact-terminal-first is the reason-mandated
|
|
935
|
+
// rung for length-before-terminal —, protocol/review/read-burst repair),
|
|
936
|
+
// so output-limit must not shadow them.
|
|
937
|
+
if (previousFailureCategory === OUTPUT_LIMIT_RETRY_CATEGORY) {
|
|
877
938
|
return [
|
|
878
939
|
basePrompt,
|
|
879
940
|
"",
|
|
880
941
|
"<retry_instruction>",
|
|
881
|
-
"
|
|
882
|
-
|
|
883
|
-
|
|
884
|
-
"
|
|
885
|
-
"</retry_instruction>",
|
|
886
|
-
].join("\n");
|
|
887
|
-
}
|
|
888
|
-
if (previousFailureCategory === "invalid-output" &&
|
|
889
|
-
task.id === "frontend-scout-pi") {
|
|
890
|
-
return [
|
|
891
|
-
basePrompt,
|
|
892
|
-
"",
|
|
893
|
-
"<retry_instruction>",
|
|
894
|
-
"The previous Scout attempt did not commit a complete, runtime-evidenced target surface.",
|
|
895
|
-
"Search the repository only as needed to establish the real entrypoint, implementation ownership, and applicable test path. Commit record_target_surface with completeness=complete and unresolvedPaths=[] only after at least one named target has fresh runtime evidence, unless the task source explicitly declares a greenfield target: in that case every future path must be source-declared by the runtime-enriched fact. If ownership truly cannot be established, record completeness=blocked with each unresolved path; do not make Plan discover it.",
|
|
896
|
-
"</retry_instruction>",
|
|
897
|
-
].join("\n");
|
|
898
|
-
}
|
|
899
|
-
if (previousFailureCategory === "structured-output-truncated" &&
|
|
900
|
-
task.structuredContractOutput) {
|
|
901
|
-
if (task.structuredContractOutput.schemaId ===
|
|
902
|
-
"frontend-implementation-contract-plan-patch-v1") {
|
|
903
|
-
return [
|
|
904
|
-
basePrompt,
|
|
905
|
-
"",
|
|
906
|
-
"<retry_instruction>",
|
|
907
|
-
"Previous attempt was truncated by the provider (stopReason=length) before the plan facts were fully committed.",
|
|
908
|
-
"Re-commit the missing record_* facts and call finalize_plan exactly once; the committed typed ledger is the only compile authority. Do NOT re-read contract/scout outputs or source files — use the facts already in context. Commit record_* facts one tool call per message, then finalize_plan immediately.",
|
|
909
|
-
"Do not emit a fenced JSON contract, protected skeleton fields, or an openspec-citations block; narrative JSON is not validated and wastes the output budget.",
|
|
910
|
-
"</retry_instruction>",
|
|
911
|
-
].join("\n");
|
|
912
|
-
}
|
|
913
|
-
return [
|
|
914
|
-
basePrompt,
|
|
915
|
-
"",
|
|
916
|
-
"<retry_instruction>",
|
|
917
|
-
"Previous attempt was truncated by the provider (stopReason=length) before the JSON contract was completed.",
|
|
918
|
-
...frontendStructuredArtifactRetryGuidance(task.structuredContractOutput.schemaId),
|
|
919
|
-
"The JSON artifact must be complete; omit evidence excerpts and duplicated upstream context.",
|
|
942
|
+
"The previous turn ended with stopReason=length before the required output was complete. This is output-capacity truncation, not empty output.",
|
|
943
|
+
"Continue incrementally from already committed typed facts and current in-scope workspace files. Do not repeat completed discovery, decisions, facts, or writes.",
|
|
944
|
+
"Generate the smallest unfinished unit next, validate its required format immediately, repair any format error in this node, and only then continue to the next unfinished unit or terminal tool.",
|
|
945
|
+
"Keep prose minimal and finish the required terminal/output protocol as soon as the remaining work is valid.",
|
|
920
946
|
"</retry_instruction>",
|
|
921
947
|
].join("\n");
|
|
922
948
|
}
|
|
923
949
|
if (previousFailureCategory === "writer-empty-diff") {
|
|
924
|
-
const maxAttempts = task.retryPolicy?.maxAttempts ??
|
|
950
|
+
const maxAttempts = task.retryPolicy?.maxAttempts ?? 5;
|
|
925
951
|
// When a completeness progress exists for this writer, fold the concrete
|
|
926
952
|
// target paths into the empty-diff retry so the model does not guess and
|
|
927
953
|
// does not need to read a forbidden `.harness/**` evidence file.
|
|
@@ -949,7 +975,7 @@ export function buildAttemptPrompt(task, basePrompt, attemptNumber, previousFail
|
|
|
949
975
|
].join("\n");
|
|
950
976
|
}
|
|
951
977
|
if (previousFailureCategory === "incomplete-write-set") {
|
|
952
|
-
const maxAttempts = task.retryPolicy?.maxAttempts ??
|
|
978
|
+
const maxAttempts = task.retryPolicy?.maxAttempts ?? 5;
|
|
953
979
|
const bindingOnly = task.id?.startsWith("generate-backend-md-case-") === true &&
|
|
954
980
|
(recoveryTargetPaths?.length ?? 0) === 1 &&
|
|
955
981
|
Object.values(recoveryDiagnostics ?? {}).flat().some((detail) => /(?:unclassified Test Points|duplicate Test Point bindings)/i.test(detail));
|
|
@@ -1765,6 +1791,32 @@ export async function executeDagNode(input) {
|
|
|
1765
1791
|
durationMs: 0,
|
|
1766
1792
|
};
|
|
1767
1793
|
}
|
|
1794
|
+
// stopReason=length on top of a bare empty-output verdict is a provider
|
|
1795
|
+
// capacity signal, not a true empty response: reroute it through the
|
|
1796
|
+
// dedicated output-limit retry path while keeping the raw category for
|
|
1797
|
+
// diagnostics. Bare empty-output and transport aliases (network,
|
|
1798
|
+
// nonzero-exit, unknown) are rerouted — categories that already carry a
|
|
1799
|
+
// precise repair instruction (output-too-large, invalid-output,
|
|
1800
|
+
// protocol-invalid, structured-output-truncated via the validators below,
|
|
1801
|
+
// writer categories) keep their classification so their exact retry
|
|
1802
|
+
// guidance still reaches the model.
|
|
1803
|
+
if (task.executor === "pi" &&
|
|
1804
|
+
!result.ok &&
|
|
1805
|
+
result.stopReason === "length" &&
|
|
1806
|
+
(result.failureCategory === undefined ||
|
|
1807
|
+
["empty-output", "network", "nonzero-exit", "unknown"].includes(result.failureCategory))) {
|
|
1808
|
+
result = {
|
|
1809
|
+
...result,
|
|
1810
|
+
rawFailureCategory: result.rawFailureCategory ?? result.failureCategory,
|
|
1811
|
+
failureCategory: OUTPUT_LIMIT_RETRY_CATEGORY,
|
|
1812
|
+
stderr: [
|
|
1813
|
+
result.stderr,
|
|
1814
|
+
"output-limit: stopReason=length; preserve completed work and retry only the unfinished output",
|
|
1815
|
+
]
|
|
1816
|
+
.filter(Boolean)
|
|
1817
|
+
.join("\n"),
|
|
1818
|
+
};
|
|
1819
|
+
}
|
|
1768
1820
|
if (task.id === "generate-backend-md-plan-pi" &&
|
|
1769
1821
|
!result.ok &&
|
|
1770
1822
|
(result.failureCategory === "output-too-large" ||
|
|
@@ -1898,7 +1950,7 @@ export async function executeDagNode(input) {
|
|
|
1898
1950
|
!result.ok &&
|
|
1899
1951
|
retryPolicy !== undefined) {
|
|
1900
1952
|
const submissions = await countContractRecordSubmissions(runDir, nodeId);
|
|
1901
|
-
if (submissions === 0) {
|
|
1953
|
+
if (submissions === 0 && result.stopReason !== "length") {
|
|
1902
1954
|
result = {
|
|
1903
1955
|
...result,
|
|
1904
1956
|
failureCategory: "empty-output",
|
|
@@ -2,6 +2,7 @@ import { createHash } from "node:crypto";
|
|
|
2
2
|
import { readFile } from "node:fs/promises";
|
|
3
3
|
import { z } from "zod";
|
|
4
4
|
import { assertTaskContractBindingFresh } from "./task-contract-binding.js";
|
|
5
|
+
import { workspaceDriftedPaths, } from "./workspace-checkpoint.js";
|
|
5
6
|
import { isSafeReadOnlyPiRetryCandidate, isTransportCrashedExclusiveWriter } from "./retry-policy.js";
|
|
6
7
|
import { resolveDagTaskSourcePath } from "../../task/dag-source-paths.js";
|
|
7
8
|
import { composeDagPromptOverridePrompt, MAX_DAG_RERUN_PROMPT_OVERRIDE_CHARS, } from "../../shared/dag-prompt-override.js";
|
|
@@ -1110,6 +1111,46 @@ function deriveSuggestedAction(input) {
|
|
|
1110
1111
|
}
|
|
1111
1112
|
return "manual";
|
|
1112
1113
|
}
|
|
1114
|
+
/**
|
|
1115
|
+
* Gitignore-style-lite matcher for readSet/writeSet path patterns in scoped
|
|
1116
|
+
* workspace-drift evaluation: literal paths match exactly; `**` spans path
|
|
1117
|
+
* segments (a trailing `/**` also matches when nothing follows), `*` stays
|
|
1118
|
+
* within one segment, `?` is one non-separator char.
|
|
1119
|
+
*/
|
|
1120
|
+
function pathMatchesWorkspacePattern(pathValue, pattern) {
|
|
1121
|
+
const normalizedPath = pathValue.replace(/\\/g, "/").replace(/^\.\//, "");
|
|
1122
|
+
const trimmed = pattern.trim().replace(/\\/g, "/").replace(/^\.\//, "");
|
|
1123
|
+
if (!trimmed || trimmed === "/")
|
|
1124
|
+
return false;
|
|
1125
|
+
if (trimmed === normalizedPath)
|
|
1126
|
+
return true;
|
|
1127
|
+
let source = "";
|
|
1128
|
+
for (let index = 0; index < trimmed.length; index += 1) {
|
|
1129
|
+
const char = trimmed[index];
|
|
1130
|
+
if (char === "*") {
|
|
1131
|
+
if (trimmed[index + 1] === "*") {
|
|
1132
|
+
if (trimmed[index + 2] === "/") {
|
|
1133
|
+
source += "(?:.*/)?";
|
|
1134
|
+
index += 2;
|
|
1135
|
+
}
|
|
1136
|
+
else {
|
|
1137
|
+
source += ".*";
|
|
1138
|
+
index += 1;
|
|
1139
|
+
}
|
|
1140
|
+
}
|
|
1141
|
+
else {
|
|
1142
|
+
source += "[^/]*";
|
|
1143
|
+
}
|
|
1144
|
+
}
|
|
1145
|
+
else if (char === "?") {
|
|
1146
|
+
source += "[^/]";
|
|
1147
|
+
}
|
|
1148
|
+
else {
|
|
1149
|
+
source += char.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
1150
|
+
}
|
|
1151
|
+
}
|
|
1152
|
+
return new RegExp(`^${source}$`).test(normalizedPath);
|
|
1153
|
+
}
|
|
1113
1154
|
function deriveRisk(reasonCodes, resetCount) {
|
|
1114
1155
|
if (reasonCodes.includes("restart-subgraph-contains-writer") ||
|
|
1115
1156
|
reasonCodes.includes("restart-subgraph-contains-unsafe-shell") ||
|
|
@@ -1213,7 +1254,14 @@ export async function evaluateDagRerunPlan(input) {
|
|
|
1213
1254
|
blockedReasons.push("parent-lifecycle-ineligible");
|
|
1214
1255
|
}
|
|
1215
1256
|
else if (input.parentState.status !== "failed" &&
|
|
1216
|
-
input.parentState.status !== "partial_failed"
|
|
1257
|
+
input.parentState.status !== "partial_failed" &&
|
|
1258
|
+
// A superseded parent is a terminal operator judgment about the run as
|
|
1259
|
+
// a whole, not about its deterministic read-only compute: its FINISHED
|
|
1260
|
+
// upstream nodes are still valid fact donors for the imported set, and
|
|
1261
|
+
// every unsafe case (writers in the reset subgraph, decision gates,
|
|
1262
|
+
// stale checkpoints, unresolved ERROR nodes outside the closure) keeps
|
|
1263
|
+
// its own dedicated block below.
|
|
1264
|
+
input.parentState.status !== "superseded") {
|
|
1217
1265
|
reasonCodes.push("parent-lifecycle-ineligible");
|
|
1218
1266
|
blockedReasons.push("parent-lifecycle-ineligible");
|
|
1219
1267
|
}
|
|
@@ -1241,8 +1289,36 @@ export async function evaluateDagRerunPlan(input) {
|
|
|
1241
1289
|
else if (input.currentWorkspace?.fingerprint &&
|
|
1242
1290
|
input.parentTerminalWorkspace.fingerprint !==
|
|
1243
1291
|
input.currentWorkspace.fingerprint) {
|
|
1244
|
-
|
|
1245
|
-
|
|
1292
|
+
// Whole-fingerprint drift is refined to content-level paths: the rerun
|
|
1293
|
+
// re-derives (readSet) and rewrites (writeSet) the reset subgraph's own
|
|
1294
|
+
// paths, so drift confined to those paths is recomputed by the rerun
|
|
1295
|
+
// itself and must not force a full standalone rerun. Drift anywhere
|
|
1296
|
+
// else — especially inputs of imported fact-reuse nodes — keeps the
|
|
1297
|
+
// strict block. Indeterminate checkpoints fail closed.
|
|
1298
|
+
const driftedPaths = workspaceDriftedPaths(input.parentTerminalWorkspace, input.currentWorkspace);
|
|
1299
|
+
const importedReadPaths = importedNodeIds.flatMap((nodeId) => {
|
|
1300
|
+
const task = tasksByIdForPlan.get(nodeId);
|
|
1301
|
+
return task?.readSet ?? [];
|
|
1302
|
+
});
|
|
1303
|
+
const scopedDrift = driftedPaths !== undefined &&
|
|
1304
|
+
driftedPaths.length > 0 &&
|
|
1305
|
+
driftedPaths.every((driftPath) => !importedReadPaths.some((pattern) => pathMatchesWorkspacePattern(driftPath, pattern)) &&
|
|
1306
|
+
resetNodeIds.some((nodeId) => {
|
|
1307
|
+
const task = tasksByIdForPlan.get(nodeId);
|
|
1308
|
+
if (!task)
|
|
1309
|
+
return false;
|
|
1310
|
+
return [
|
|
1311
|
+
...(task.writeSet ?? []),
|
|
1312
|
+
...(task.readSet ?? []),
|
|
1313
|
+
].some((pattern) => pathMatchesWorkspacePattern(driftPath, pattern));
|
|
1314
|
+
}));
|
|
1315
|
+
if (scopedDrift) {
|
|
1316
|
+
reasonCodes.push("workspace-drift-scoped");
|
|
1317
|
+
}
|
|
1318
|
+
else {
|
|
1319
|
+
reasonCodes.push("workspace-drift");
|
|
1320
|
+
blockedReasons.push("workspace-drift");
|
|
1321
|
+
}
|
|
1246
1322
|
}
|
|
1247
1323
|
else if (!input.currentWorkspace?.fingerprint) {
|
|
1248
1324
|
reasonCodes.push("workspace-checkpoint-missing");
|
|
@@ -1293,8 +1369,18 @@ export async function evaluateDagRerunPlan(input) {
|
|
|
1293
1369
|
}
|
|
1294
1370
|
for (const nodeId of importedNodeIds) {
|
|
1295
1371
|
const record = input.parentState.nodes[nodeId];
|
|
1296
|
-
if (!record)
|
|
1372
|
+
if (!record) {
|
|
1373
|
+
reasonCodes.push("parent-facts-invalid");
|
|
1374
|
+
blockedReasons.push("parent-facts-invalid");
|
|
1375
|
+
blockingNodes.push({ nodeId, reasonCode: "parent-facts-invalid" });
|
|
1297
1376
|
continue;
|
|
1377
|
+
}
|
|
1378
|
+
if (record.status === "PENDING" || record.status === "RUNNING") {
|
|
1379
|
+
reasonCodes.push("parent-facts-invalid");
|
|
1380
|
+
blockedReasons.push("parent-facts-invalid");
|
|
1381
|
+
blockingNodes.push({ nodeId, reasonCode: "parent-facts-invalid" });
|
|
1382
|
+
continue;
|
|
1383
|
+
}
|
|
1298
1384
|
if (record.status === "ERROR") {
|
|
1299
1385
|
reasonCodes.push("parent-facts-invalid");
|
|
1300
1386
|
blockedReasons.push("parent-facts-invalid");
|
|
@@ -784,6 +784,13 @@ async function stageContinuationRun(input) {
|
|
|
784
784
|
if (!parentRecord) {
|
|
785
785
|
throw new Error(`parent node record missing for imported node: ${nodeId}`);
|
|
786
786
|
}
|
|
787
|
+
if (parentRecord.status !== "FINISHED" &&
|
|
788
|
+
!(parentRecord.status === "SKIPPED" &&
|
|
789
|
+
(parentRecord.skippedReason?.includes("condition") ||
|
|
790
|
+
parentRecord.skippedReason?.includes("runIf") ||
|
|
791
|
+
parentRecord.skippedReason?.includes("run-if")))) {
|
|
792
|
+
throw new Error(`parent node is not a settled fact donor for imported node: ${nodeId}`);
|
|
793
|
+
}
|
|
787
794
|
const { manifestNode, importedRecord } = await importNodeFacts({
|
|
788
795
|
parentRunDir: input.parentRunDir,
|
|
789
796
|
newRunDir: runDir,
|
|
@@ -7,8 +7,8 @@ import { TYPED_EVENT_FACT_KINDS } from "./frontend-typed-event-store.js";
|
|
|
7
7
|
* is re-exported here unchanged so retry-policy stays the single import
|
|
8
8
|
* surface for retry policies and helpers.
|
|
9
9
|
*/
|
|
10
|
-
export { ALL_DAG_RETRY_CATEGORIES, CONTEXT_OVERFLOW_RETRY_CATEGORY, DEFAULT_DAG_RETRY_CATEGORIES, INCOMPLETE_WRITE_SET_RETRY_CATEGORY, PROTOCOL_AWARE_DAG_RETRY_CATEGORIES, PROTOCOL_INVALID_RETRY_CATEGORY, READ_BURST_RETRY_CATEGORY, REVIEW_TERMINAL_MISSING_RETRY_CATEGORY, STRUCTURED_ARTIFACT_INVALID_RETRY_CATEGORY, STRUCTURED_OUTPUT_RETRY_CATEGORY, STRUCTURED_OUTPUT_TRUNCATED_RETRY_CATEGORY, STRUCTURED_REQUIRED_DAG_RETRY_CATEGORIES, TARGET_TEMPLATE_TRANSIENT_RETRY_PROFILE, WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY, WRITER_EMPTY_DIFF_RETRY_CATEGORY, dagRetryBackoffSchema, dagRetryCategorySchema, dagRetryPolicySchema, } from "./dag-retry-schema.js";
|
|
11
|
-
import { CONTEXT_OVERFLOW_RETRY_CATEGORY, DEFAULT_DAG_RETRY_CATEGORIES, INCOMPLETE_WRITE_SET_RETRY_CATEGORY, PROTOCOL_AWARE_DAG_RETRY_CATEGORIES, READ_BURST_RETRY_CATEGORY, REVIEW_TERMINAL_MISSING_RETRY_CATEGORY, STRUCTURED_ARTIFACT_INVALID_RETRY_CATEGORY, STRUCTURED_OUTPUT_RETRY_CATEGORY, STRUCTURED_REQUIRED_DAG_RETRY_CATEGORIES, TARGET_TEMPLATE_TRANSIENT_RETRY_PROFILE, WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY, WRITER_EMPTY_DIFF_RETRY_CATEGORY, } from "./dag-retry-schema.js";
|
|
10
|
+
export { ALL_DAG_RETRY_CATEGORIES, CONTEXT_OVERFLOW_RETRY_CATEGORY, DEFAULT_DAG_RETRY_CATEGORIES, INCOMPLETE_WRITE_SET_RETRY_CATEGORY, OUTPUT_LIMIT_RETRY_CATEGORY, PROTOCOL_AWARE_DAG_RETRY_CATEGORIES, PROTOCOL_INVALID_RETRY_CATEGORY, READ_BURST_RETRY_CATEGORY, REVIEW_TERMINAL_MISSING_RETRY_CATEGORY, STRUCTURED_ARTIFACT_INVALID_RETRY_CATEGORY, STRUCTURED_OUTPUT_RETRY_CATEGORY, STRUCTURED_OUTPUT_TRUNCATED_RETRY_CATEGORY, STRUCTURED_REQUIRED_DAG_RETRY_CATEGORIES, TARGET_TEMPLATE_TRANSIENT_RETRY_PROFILE, WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY, WRITER_EMPTY_DIFF_RETRY_CATEGORY, dagRetryBackoffSchema, dagRetryCategorySchema, dagRetryPolicySchema, } from "./dag-retry-schema.js";
|
|
11
|
+
import { CONTEXT_OVERFLOW_RETRY_CATEGORY, DEFAULT_DAG_RETRY_CATEGORIES, INCOMPLETE_WRITE_SET_RETRY_CATEGORY, OUTPUT_LIMIT_RETRY_CATEGORY, PROTOCOL_AWARE_DAG_RETRY_CATEGORIES, READ_BURST_RETRY_CATEGORY, REVIEW_TERMINAL_MISSING_RETRY_CATEGORY, STRUCTURED_ARTIFACT_INVALID_RETRY_CATEGORY, STRUCTURED_OUTPUT_RETRY_CATEGORY, STRUCTURED_REQUIRED_DAG_RETRY_CATEGORIES, TARGET_TEMPLATE_TRANSIENT_RETRY_PROFILE, WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY, WRITER_EMPTY_DIFF_RETRY_CATEGORY, } from "./dag-retry-schema.js";
|
|
12
12
|
const RETRY_SAFE_PI_ROLES = new Set([
|
|
13
13
|
"planner",
|
|
14
14
|
"scout",
|
|
@@ -19,7 +19,7 @@ const RETRY_SAFE_PI_ROLES = new Set([
|
|
|
19
19
|
]);
|
|
20
20
|
/**
|
|
21
21
|
* The default retry policy applied to safe generated read-only Pi nodes.
|
|
22
|
-
* Total attempts:
|
|
22
|
+
* Total attempts: 5, exponential backoff with cap.
|
|
23
23
|
*
|
|
24
24
|
* Includes `context-overflow`: a read-only node that blew the context window
|
|
25
25
|
* (400 request too large, e.g. a review node accumulating too many reads) is a
|
|
@@ -29,7 +29,7 @@ const RETRY_SAFE_PI_ROLES = new Set([
|
|
|
29
29
|
* design-review/review fell straight to ERROR without a retry attempt.
|
|
30
30
|
*/
|
|
31
31
|
export const DEFAULT_READ_ONLY_PI_RETRY_POLICY = {
|
|
32
|
-
maxAttempts:
|
|
32
|
+
maxAttempts: 5,
|
|
33
33
|
backoff: "exponential",
|
|
34
34
|
initialDelayMs: 2000,
|
|
35
35
|
maxDelayMs: 30000,
|
|
@@ -125,7 +125,10 @@ export const WRITER_EMPTY_DIFF_RETRY_POLICY = {
|
|
|
125
125
|
backoff: "exponential",
|
|
126
126
|
initialDelayMs: 2000,
|
|
127
127
|
maxDelayMs: 30000,
|
|
128
|
-
retryCategories: [
|
|
128
|
+
retryCategories: [
|
|
129
|
+
WRITER_EMPTY_DIFF_RETRY_CATEGORY,
|
|
130
|
+
OUTPUT_LIMIT_RETRY_CATEGORY,
|
|
131
|
+
],
|
|
129
132
|
};
|
|
130
133
|
/**
|
|
131
134
|
* Bounded transport retry for standard exclusive implementers when a provider
|
|
@@ -140,6 +143,7 @@ export const WRITER_TRANSPORT_RETRY_POLICY = {
|
|
|
140
143
|
retryCategories: [
|
|
141
144
|
WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY,
|
|
142
145
|
CONTEXT_OVERFLOW_RETRY_CATEGORY,
|
|
146
|
+
OUTPUT_LIMIT_RETRY_CATEGORY,
|
|
143
147
|
],
|
|
144
148
|
};
|
|
145
149
|
/**
|
|
@@ -175,6 +179,7 @@ export const TARGET_TEMPLATE_WRITER_TRANSPORT_RETRY_POLICY = {
|
|
|
175
179
|
retryCategories: [
|
|
176
180
|
WRITER_CLEAN_TIMEOUT_RETRY_CATEGORY,
|
|
177
181
|
CONTEXT_OVERFLOW_RETRY_CATEGORY,
|
|
182
|
+
OUTPUT_LIMIT_RETRY_CATEGORY,
|
|
178
183
|
],
|
|
179
184
|
};
|
|
180
185
|
/**
|
|
@@ -182,13 +187,14 @@ export const TARGET_TEMPLATE_WRITER_TRANSPORT_RETRY_POLICY = {
|
|
|
182
187
|
* recovery attempts driven by Completeness Gate (missing/broken target files).
|
|
183
188
|
*/
|
|
184
189
|
export const BACKEND_TEST_WRITER_COMPLETENESS_RETRY_POLICY = {
|
|
185
|
-
maxAttempts:
|
|
190
|
+
maxAttempts: 5,
|
|
186
191
|
backoff: "exponential",
|
|
187
192
|
initialDelayMs: 2000,
|
|
188
193
|
maxDelayMs: 30000,
|
|
189
194
|
retryCategories: [
|
|
190
195
|
WRITER_EMPTY_DIFF_RETRY_CATEGORY,
|
|
191
196
|
INCOMPLETE_WRITE_SET_RETRY_CATEGORY,
|
|
197
|
+
OUTPUT_LIMIT_RETRY_CATEGORY,
|
|
192
198
|
],
|
|
193
199
|
};
|
|
194
200
|
/** Markdown shard writers use one full attempt plus at most one bounded binding repair. */
|
|
@@ -472,9 +478,9 @@ function reasonMandatedFrontendPlanRetryStep(reason) {
|
|
|
472
478
|
}
|
|
473
479
|
}
|
|
474
480
|
/**
|
|
475
|
-
* Opt-in retry policy mounted on the frontend-plan-pi node. maxAttempts=
|
|
476
|
-
*
|
|
477
|
-
*
|
|
481
|
+
* Opt-in retry policy mounted on the frontend-plan-pi node. maxAttempts=5
|
|
482
|
+
* lets the producer validate and repair each generated increment while still
|
|
483
|
+
* bounding the four-rung ladder (normal → bounded-tool-only →
|
|
478
484
|
* compact-terminal-first → backup-model). `invalid-output` (typed-fact schema
|
|
479
485
|
* violations, e.g. verification targets referencing undeclared UI states)
|
|
480
486
|
* retries at the `normal` rung with the contractCheck reason injected via
|
|
@@ -485,7 +491,7 @@ export const FRONTEND_PLAN_LADDER_RETRY_POLICY = {
|
|
|
485
491
|
// incremental record_* calls; dense call bursts hit provider rate limits
|
|
486
492
|
// whose windows exceed the old 30s cap. Longer backoff gives the limit
|
|
487
493
|
// window time to expire before the next attempt.
|
|
488
|
-
maxAttempts:
|
|
494
|
+
maxAttempts: 5,
|
|
489
495
|
backoff: "exponential",
|
|
490
496
|
initialDelayMs: 5000,
|
|
491
497
|
maxDelayMs: 60000,
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
import { appendFileSync, mkdirSync, unlinkSync } from "node:fs";
|
|
2
|
+
import { hostname } from "node:os";
|
|
3
|
+
import path from "node:path";
|
|
4
|
+
/**
|
|
5
|
+
* Runner exit-time diagnostics (silent-death forensics).
|
|
6
|
+
*
|
|
7
|
+
* A DAG runner can disappear without a crash report and without stderr (five
|
|
8
|
+
* reproductions during frontend recovery/long-plan runs, 2026-09-06): when a
|
|
9
|
+
* supervisor hard-kills the process or the event loop is wedged, no JS runs,
|
|
10
|
+
* so nothing is persisted and the run is only discoverable later as an
|
|
11
|
+
* orphaned RUNNING state with a frozen heartbeat. That leaves no evidence of
|
|
12
|
+
* WHERE the runner was when it died or whether the exit was even observable
|
|
13
|
+
* by Node.
|
|
14
|
+
*
|
|
15
|
+
* This module gives every abnormal death a durable trace next to state.json:
|
|
16
|
+
* - an `armed` line is written when the checkpoint starts executing;
|
|
17
|
+
* - a `process-exit` line is appended from the process `exit` hook with the
|
|
18
|
+
* exit code and a live state snapshot (status, terminalReason, RUNNING
|
|
19
|
+
* nodes, heartbeat staleness);
|
|
20
|
+
* - `uncaught-exception` / `unhandled-rejection` lines record the error
|
|
21
|
+
* before the process is allowed to crash exactly as it would have
|
|
22
|
+
* via uncaughtExceptionMonitor without changing the host's crash policy.
|
|
23
|
+
*
|
|
24
|
+
* On a clean terminal return the runner calls stop(), which removes every
|
|
25
|
+
* listener and deletes the journal, so completed runs carry no noise. A file
|
|
26
|
+
* left behind therefore means: diagnostics were armed, then the process went
|
|
27
|
+
* away through a path Node could not observe (`armed` line only → external
|
|
28
|
+
* SIGKILL/OOM/event-loop wedge) or through an observed abnormal exit (`armed`
|
|
29
|
+
* + one or more event lines → the reason is recorded in the file).
|
|
30
|
+
*
|
|
31
|
+
* Journal writes are strictly synchronous and best-effort: an append failure
|
|
32
|
+
* must never alter the runner's own crash/exit behavior.
|
|
33
|
+
*/
|
|
34
|
+
export const RUNNER_EXIT_DIAGNOSTICS_FILE = "runner-exit-diagnostics.jsonl";
|
|
35
|
+
function describeError(error) {
|
|
36
|
+
if (error instanceof Error) {
|
|
37
|
+
return { name: error.name, message: error.message };
|
|
38
|
+
}
|
|
39
|
+
try {
|
|
40
|
+
return { message: String(error) };
|
|
41
|
+
}
|
|
42
|
+
catch {
|
|
43
|
+
return { message: "<unstringifiable error>" };
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
function appendJournalLineSync(journalPath, record) {
|
|
47
|
+
try {
|
|
48
|
+
mkdirSync(path.dirname(journalPath), { recursive: true });
|
|
49
|
+
appendFileSync(journalPath, `${JSON.stringify(record)}\n`, "utf8");
|
|
50
|
+
}
|
|
51
|
+
catch {
|
|
52
|
+
// Best-effort: journaling must never crash or alter exit behavior.
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
function removeJournalSync(journalPath) {
|
|
56
|
+
try {
|
|
57
|
+
unlinkSync(journalPath);
|
|
58
|
+
}
|
|
59
|
+
catch {
|
|
60
|
+
// Tolerate a missing journal (e.g. nothing was ever written).
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
export function installRunnerExitDiagnostics(options) {
|
|
64
|
+
const journalPath = path.join(options.runDir, RUNNER_EXIT_DIAGNOSTICS_FILE);
|
|
65
|
+
const target = options.target ?? process;
|
|
66
|
+
const armedAt = Date.now();
|
|
67
|
+
let stopped = false;
|
|
68
|
+
const base = {
|
|
69
|
+
runId: options.runId,
|
|
70
|
+
title: options.title,
|
|
71
|
+
pid: process.pid,
|
|
72
|
+
hostname: hostname(),
|
|
73
|
+
};
|
|
74
|
+
const safeSnapshot = () => {
|
|
75
|
+
try {
|
|
76
|
+
return options.snapshot();
|
|
77
|
+
}
|
|
78
|
+
catch {
|
|
79
|
+
return { status: "<snapshot-error>", running: [] };
|
|
80
|
+
}
|
|
81
|
+
};
|
|
82
|
+
const record = (kind, extra) => ({
|
|
83
|
+
ts: new Date().toISOString(),
|
|
84
|
+
kind,
|
|
85
|
+
...base,
|
|
86
|
+
...(extra ?? {}),
|
|
87
|
+
state: safeSnapshot(),
|
|
88
|
+
});
|
|
89
|
+
appendJournalLineSync(journalPath, {
|
|
90
|
+
...record("armed"),
|
|
91
|
+
armedAt: new Date(armedAt).toISOString(),
|
|
92
|
+
});
|
|
93
|
+
const onProcessExit = (code) => {
|
|
94
|
+
if (stopped)
|
|
95
|
+
return;
|
|
96
|
+
appendJournalLineSync(journalPath, record("process-exit", {
|
|
97
|
+
code: typeof code === "number" ? code : process.exitCode ?? 0,
|
|
98
|
+
elapsedMs: Date.now() - armedAt,
|
|
99
|
+
}));
|
|
100
|
+
};
|
|
101
|
+
const onUncaughtException = (error, origin) => {
|
|
102
|
+
if (stopped)
|
|
103
|
+
return;
|
|
104
|
+
appendJournalLineSync(journalPath, record(origin === "unhandledRejection" ? "unhandled-rejection" : "uncaught-exception", {
|
|
105
|
+
error: describeError(error),
|
|
106
|
+
elapsedMs: Date.now() - armedAt,
|
|
107
|
+
}));
|
|
108
|
+
// Monitor listeners neither suppress fatal errors nor force a crash
|
|
109
|
+
// when the host has its own handlers or a nonfatal rejection policy.
|
|
110
|
+
};
|
|
111
|
+
target.on("exit", onProcessExit);
|
|
112
|
+
target.on("uncaughtExceptionMonitor", onUncaughtException);
|
|
113
|
+
return {
|
|
114
|
+
journalPath,
|
|
115
|
+
stop: () => {
|
|
116
|
+
if (stopped)
|
|
117
|
+
return;
|
|
118
|
+
stopped = true;
|
|
119
|
+
target.removeListener("exit", onProcessExit);
|
|
120
|
+
target.removeListener("uncaughtExceptionMonitor", onUncaughtException);
|
|
121
|
+
// Clean completion: leave completed runs free of diagnostics noise.
|
|
122
|
+
removeJournalSync(journalPath);
|
|
123
|
+
},
|
|
124
|
+
};
|
|
125
|
+
}
|