@tea-agent/loop-agent 0.44.0-next.8 → 0.44.0-next.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. package/CHANGELOG.md +97 -0
  2. package/dist/application/evaluation/budget.js +21 -0
  3. package/dist/application/evaluation/corpus-hash.js +10 -15
  4. package/dist/application/evaluation/corpus.js +2 -1
  5. package/dist/application/evaluation/frontend-browser-acceptance.js +77 -0
  6. package/dist/application/evaluation/frontend-corpus-browser-acceptance.js +175 -0
  7. package/dist/application/evaluation/frontend-gateway-evidence.js +226 -0
  8. package/dist/application/evaluation/frontend-pair-registration.js +143 -0
  9. package/dist/application/evaluation/frontend-paired-summary.js +150 -0
  10. package/dist/application/evaluation/frontend-run-observation.js +144 -0
  11. package/dist/application/evaluation/frontend-shared-evidence.js +105 -0
  12. package/dist/application/evaluation/types.js +45 -12
  13. package/dist/application/task-lifecycle/observe.js +26 -1
  14. package/dist/build-stamp.json +3 -3
  15. package/dist/cli/command-definitions.js +11 -0
  16. package/dist/cli/program.js +4 -0
  17. package/dist/commands/task-advance.js +3 -5
  18. package/dist/commands/task-source-prepare.js +13 -1
  19. package/dist/executors/dag-pi/tools/design-terminal-tools.js +17 -7
  20. package/dist/executors/dag-pi/tools/review-terminal-tools.js +17 -7
  21. package/dist/executors/dag-pi-executor.js +2048 -125
  22. package/dist/executors/pi-executor.js +6 -1
  23. package/dist/executors/pi-sdk-executor.js +76 -11
  24. package/dist/executors/shell-executor.js +88 -12
  25. package/dist/infrastructure/console/operation-store.js +2 -0
  26. package/dist/infrastructure/evaluation/corpus-store.js +37 -12
  27. package/dist/shared/operator/capabilities.js +10 -5
  28. package/dist/task/config-types.js +22 -9
  29. package/dist/task/contract/project.js +13 -0
  30. package/dist/task/contract/schema.js +2 -8
  31. package/dist/task/source-prepare/parse-intent.js +124 -5
  32. package/dist/task/source-prepare/prepare.js +6 -1
  33. package/dist/task/source-prepare/semantic-intake.js +17 -2
  34. package/dist/task/source-prepare/source-execution.js +65 -0
  35. package/dist/task/source-prepare/source-fidelity-pi.js +21 -8
  36. package/dist/task/source-prepare/source-provider-budget.js +290 -0
  37. package/dist/worker/console/operator-actions.js +0 -1
  38. package/dist/worker/console/operator-user-error.js +1 -1
  39. package/dist/worker/console/prd-intake-bridge.js +5 -15
  40. package/dist/workflows/dag/budget-enforcement.js +31 -2
  41. package/dist/workflows/dag/frontend-closeout.js +3 -1
  42. package/dist/workflows/dag/frontend-contract-facts.js +11 -8
  43. package/dist/workflows/dag/frontend-design-policy.js +6 -6
  44. package/dist/workflows/dag/frontend-durable-tools.js +90 -70
  45. package/dist/workflows/dag/frontend-implementation-contract.js +146 -59
  46. package/dist/workflows/dag/frontend-plan-canary.js +66 -16
  47. package/dist/workflows/dag/frontend-plan-decision-contract.js +13 -21
  48. package/dist/workflows/dag/frontend-plan-progress.js +92 -0
  49. package/dist/workflows/dag/frontend-recovery-plan.js +11 -10
  50. package/dist/workflows/dag/frontend-recovery-run.js +51 -46
  51. package/dist/workflows/dag/frontend-repair-assertions.js +36 -0
  52. package/dist/workflows/dag/frontend-review-context.js +29 -1
  53. package/dist/workflows/dag/frontend-review-scopes.js +92 -2
  54. package/dist/workflows/dag/frontend-risk.js +8 -3
  55. package/dist/workflows/dag/frontend-root-observation.js +261 -0
  56. package/dist/workflows/dag/frontend-session-budget.js +52 -39
  57. package/dist/workflows/dag/frontend-session-context.js +10 -0
  58. package/dist/workflows/dag/frontend-test-execution-evidence.js +206 -48
  59. package/dist/workflows/dag/frontend-typed-event-store.js +11 -6
  60. package/dist/workflows/dag/frontend-verification-trace.js +6 -0
  61. package/dist/workflows/dag/frontend-writer-admission.js +2 -1
  62. package/dist/workflows/dag/init-hybrid.js +108 -37
  63. package/dist/workflows/dag/node-execution.js +11 -11
  64. package/dist/workflows/dag/rerun-feedback.js +35 -4
  65. package/dist/workflows/dag/rerun-task.js +62 -67
  66. package/dist/workflows/dag/runner.js +58 -12
  67. package/dist/workflows/dag/types.js +15 -8
  68. package/docs/architecture/runtime-boundaries.md +1 -1
  69. package/docs/templates/backend-test-dag.json +4 -4
  70. package/docs/templates/frontend-implementation-contract.schema.json +31 -11
  71. package/package.json +1 -1
  72. package/skills/frontend-design-review/SKILL.md +8 -11
  73. package/skills/frontend-design-review/references/review-checklist.md +3 -4
  74. package/skills/frontend-review/SKILL.md +5 -1
  75. package/skills/loop-agent/references/command-reference.md +8 -0
@@ -1,3 +1,7 @@
1
+ import { sourceBudgetContractIdentity } from "../../application/evaluation/budget.js";
2
+ import { sha256OfCanonicalJson } from "../../task/contract/hash.js";
3
+ import { readTaskSourceProviderBudget, freezeTaskSourceUsage } from "../../task/source-prepare/source-provider-budget.js";
4
+ import { inspectFrontendTestCommand } from "./frontend-test-execution-evidence.js";
1
5
  import { createHash } from "node:crypto";
2
6
  import { frontendExecutionPolicySchema } from "../../shared/frontend-execution-policy.js";
3
7
  import { access, readdir, readFile, realpath } from "node:fs/promises";
@@ -30,7 +34,7 @@ import { extractRequirementFactsFromMarkdown } from "../../task/source-prepare/p
30
34
  import { computeLedgerInputDigest, parseLedgerJson, recoverLedgerInputContract, validateRequirementLedger, } from "../../task/source-prepare/ledger.js";
31
35
  import { REQUIREMENT_LEDGER_FILE_NAME } from "../../task/contract/constants.js";
32
36
  import { dagHasWriterExecution } from "./task-contract-binding.js";
33
- import { BACKEND_TEST_MODULE_INDEX_HEADER, BACKEND_TEST_MODULE_SPLIT_REASONS, backendTestModuleIndexHeaderMarkdown, canonicalBackendTestModuleMarkdownPath, } from "./backend-test-plan-protocol.js";
37
+ import { BACKEND_TEST_MODULE_INDEX_HEADER, BACKEND_TEST_MODULE_SPLIT_REASONS, } from "./backend-test-plan-protocol.js";
34
38
  import { DEFAULT_VERIFY_TIMEOUT_MS, resolveVerifyPreset, } from "../../executors/shell-verification.js";
35
39
  import { resolveExecutorModelMatrices } from "../../executors/model-routing.js";
36
40
  import { normalizeTaskRequirementText, resolveTaskDagTemplateSelection, } from "./task-demand-routing.js";
@@ -995,7 +999,7 @@ function extractFrontendVerifyCommandsFromMarkdown(input) {
995
999
  .trim()
996
1000
  .replace(/^[-*]\s*(?:\[[ xX]\]\s*)?/, "")
997
1001
  .trim();
998
- const codeSpanCommands = Array.from(bulletless.matchAll(/`([^`]+)`/g), (match) => match[1].trim());
1002
+ const codeSpanCommands = Array.from(bulletless.matchAll(/`([^`]+)`/g), (match) => match[1]?.trim() ?? "").filter(Boolean);
999
1003
  const candidates = codeSpanCommands.length > 0 ? codeSpanCommands : [bulletless];
1000
1004
  for (const candidate of candidates) {
1001
1005
  if (isSupportedMarkdownVerifyCommand(candidate)) {
@@ -1037,8 +1041,8 @@ function extractFrontendMockVerifyCommandsFromMarkdown(input) {
1037
1041
  if (!/(?:mock|模拟服务|接口桩)/i.test(line))
1038
1042
  continue;
1039
1043
  for (const match of line.matchAll(/`([^`]+)`/g)) {
1040
- const commandText = match[1].trim();
1041
- if (!isSupportedMarkdownVerifyCommand(commandText))
1044
+ const commandText = match[1]?.trim();
1045
+ if (!commandText || !isSupportedMarkdownVerifyCommand(commandText))
1042
1046
  continue;
1043
1047
  const command = markdownVerifyCommand(input.repoRoot, commandText);
1044
1048
  if (command)
@@ -1064,12 +1068,23 @@ function chooseFrontendVerifyCommands(input) {
1064
1068
  /**
1065
1069
  * Generation-time lane split for frontend verify commands. The mode itself
1066
1070
  * is owned by `classifyFrontendVerifyCommandText` (contract module); this
1067
- * wrapper only preserves the local call shape.
1071
+ * wrapper adds the generation-time policy for unrecognized commands.
1072
+ *
1073
+ * An unrecognized command has no inferable lane. Defaulting it to `behavior`
1074
+ * would place it in the test-observed lane and block writer admission with an
1075
+ * error no model-side change can clear, so generation fails loudly instead and
1076
+ * asks the operator to declare the lane.
1068
1077
  */
1069
- function classifyFrontendVerifyCommand(command) {
1070
- return classifyFrontendVerifyCommandText(command);
1078
+ function classifyFrontendVerifyCommand(command, label) {
1079
+ const mode = classifyFrontendVerifyCommandText(command);
1080
+ if (!mode) {
1081
+ throw new Error(`verifyCommands[${label}]: unrecognized verification command ${JSON.stringify(command)}; declare its lane explicitly with mode: "static" or mode: "behavior", or use a recognized test runner such as \`npx vitest run\``);
1082
+ }
1083
+ return mode;
1071
1084
  }
1072
- function buildExplicitFrontendVerifyCommands(taskConfig, repoRoot) {
1085
+ /** Exported for direct unit testing of lane assignment and the
1086
+ * unrecognized-command guard; not part of the module's public runtime API. */
1087
+ export function buildExplicitFrontendVerifyCommands(taskConfig, repoRoot) {
1073
1088
  const staticCommands = [];
1074
1089
  const behaviorCommands = [];
1075
1090
  if (!repoRoot)
@@ -1081,7 +1096,22 @@ function buildExplicitFrontendVerifyCommands(taskConfig, repoRoot) {
1081
1096
  label: command.label,
1082
1097
  timeoutMs: command.timeoutMs,
1083
1098
  };
1084
- if (classifyFrontendVerifyCommand(command.command) === "static") {
1099
+ // An explicit `mode` declaration wins over automatic classification. The
1100
+ // operator knows the command's nature; the classifier only guesses from
1101
+ // text. This matters because the behavior lane requires test-observation
1102
+ // capability at writer admission, so a governance/lint script that text
1103
+ // classification calls "behavior" blocks admission deterministically and
1104
+ // no model-side fix can clear it. A declared command still runs and its
1105
+ // exit code is still checked in whichever lane it lands.
1106
+ // An undeclared command is classified by text, and
1107
+ // `classifyFrontendVerifyCommand` fails at generation when that text
1108
+ // carries no recognizable verification semantics. Failing there keeps the
1109
+ // decision with the operator, where it is cheap to fix, instead of
1110
+ // silently entering the test-observed behavior lane and surfacing much
1111
+ // later as an unactionable `FRONTEND_TEST_OBSERVATION_INVALID`.
1112
+ const mode = command.mode ??
1113
+ classifyFrontendVerifyCommand(command.command, command.label);
1114
+ if (mode === "static") {
1085
1115
  staticCommands.push(verifyCommand);
1086
1116
  }
1087
1117
  else {
@@ -1102,10 +1132,13 @@ function verifyCommandKey(command) {
1102
1132
  return [cwdKey, normalizedArgs.join("\0"), envKey].join("\u0001");
1103
1133
  }
1104
1134
  function normalizeVerifyCommandArgs(args) {
1135
+ const [shell, flag, payload] = args;
1105
1136
  if (args.length === 3 &&
1106
- /^(?:bash|sh)(?:\.exe)?$/i.test(path.basename(args[0])) &&
1107
- args[1] === "-lc") {
1108
- const parsed = tokenizeSimpleShellCommand(args[2]);
1137
+ shell !== undefined &&
1138
+ payload !== undefined &&
1139
+ /^(?:bash|sh)(?:\.exe)?$/i.test(path.basename(shell)) &&
1140
+ flag === "-lc") {
1141
+ const parsed = tokenizeSimpleShellCommand(payload);
1109
1142
  if (parsed)
1110
1143
  return normalizeVerifyCommandArgs(parsed);
1111
1144
  }
@@ -1483,6 +1516,7 @@ function deriveParallelScoutPaths(taskConfig) {
1483
1516
  : allowed,
1484
1517
  };
1485
1518
  }
1519
+ const FRONTEND_NO_STATIC_VERIFICATION_MARKER = `node -e "console.log('${FRONTEND_NO_VERIFICATION_MARKER_TEXT}; static/behavior verification not-run')"`;
1486
1520
  /**
1487
1521
  * Resolve frontend verification fallbacks from the target project's own
1488
1522
  * package scripts. The DAG builder is also used by unit fixtures without a
@@ -1492,7 +1526,6 @@ function deriveParallelScoutPaths(taskConfig) {
1492
1526
  * at verify time, so such projects fall through to the tsc probe and may end
1493
1527
  * up with no fallback commands plus a generation-time advisory.
1494
1528
  */
1495
- const FRONTEND_NO_STATIC_VERIFICATION_MARKER = `node -e "console.log('${FRONTEND_NO_VERIFICATION_MARKER_TEXT}; static/behavior verification not-run')"`;
1496
1529
  async function discoverFrontendFallbackVerifyCommands(repoRoot) {
1497
1530
  const genericFallback = {
1498
1531
  staticCommands: ["npm run typecheck", "npm run build"],
@@ -2559,26 +2592,12 @@ function resolveFrontendMockContextBlock(sources) {
2559
2592
  parts.push("Mock-backed frontend verification is required. Prefer the detected native service; otherwise the assessment may select an existing browser interception harness or reversible request adapter. Any handler, fixture, adapter, and UI changes stay in the single frontend-implement-pi writeSet.");
2560
2593
  }
2561
2594
  if (mode === "not-required") {
2562
- // The assessment sentence is deliberately split: every branch keeps the
2563
- // "select not-needed when Mock is intentionally skipped" guidance, because
2564
- // that is the correct action in both cases, and only the permissive
2565
- // "or select a safe Mock strategy if project evidence supports one" tail is
2566
- // withheld when the allow-list is frozen.
2567
- //
2568
- // Emitting both was a direct instruction conflict, and it cost a whole
2569
- // frontend run (dogfood R2): the model resolved it the permissive way, the
2570
- // typed tool accepted ten `native` records with ok:true, and the
2571
- // design-policy shell rejected the contract ~90s later, killing 8
2572
- // downstream nodes. The fix belongs here - stop contradicting ourselves -
2573
- // rather than in extra enforcement that would leave the model guessing.
2574
- const mockAssessment = "Generation-time evidence does not require Mock. The assessment must still use contract/scout evidence: select not-needed when Mock is intentionally skipped";
2575
2595
  if (frontendMockStrategyMustBeNotNeeded(sources)) {
2576
- parts.push(`${mockAssessment}.`);
2577
- parts.push('Auto mode has no confirmed project Mock capability or no deterministic Mock verification command. The structured contract must set mockApi.strategy to "not-needed". Keep the real request path as the default, record any unproved backend behavior as Real Integration Gap, and do not add Mock files or dependencies within this run.');
2596
+ parts.push('Generation-time evidence does not require Mock. The assessment must still use contract/scout evidence: select not-needed when Mock is intentionally skipped. Auto mode has no confirmed project Mock capability or no deterministic Mock verification command. The structured contract must set mockApi.strategy to "not-needed". Keep the real request path as the default, record any unproved backend behavior as Real Integration Gap, and do not add Mock files or dependencies within this run.');
2578
2597
  parts.push('HARD CONSTRAINT (frozen at generation time): this DAG allows only mockApi.strategy "not-needed"; the prewrite gate rejects any other strategy. If project governance (openspec / ai_workspace / decision records, e.g. a DEC rule requiring native) demands Mock-backed verification, that is a generation-time contract gap, not a plan-revision defect: declare frontendMock.verifyCommands (or policy: "required") in task.json and regenerate the DAG.');
2579
2598
  }
2580
2599
  else {
2581
- parts.push(`${mockAssessment}, or select a safe Mock strategy if project evidence supports one.`);
2600
+ parts.push("Generation-time evidence does not require Mock. The assessment must still use contract/scout evidence: select not-needed when Mock is intentionally skipped, or select a safe Mock strategy if project evidence supports one.");
2582
2601
  }
2583
2602
  }
2584
2603
  if (mode === "blocked") {
@@ -3063,7 +3082,7 @@ async function buildFrontendHybridDagFromTask(sources) {
3063
3082
  "- interactions[]: name, trigger, expectedBehavior, implementationTargets, verificationTargetIds",
3064
3083
  "- targets: routes, publicApiChanges (files are runtime-owned)",
3065
3084
  "- mockApi: strategy, productionDefaultOff, activation, endpoints[]",
3066
- "- verificationTargets[]: id (a stable id you choose, e.g. VT-001; it is not derived from test titles), commandId (frozen directory key), mode + commandLabel (runtime-resolved), file, requirementIds, uiStates, scope? (display-only)",
3085
+ "- verificationTargets[]: id (stable target identity; execution binds commandId + file, never test titles), commandId (frozen directory key), mode + commandLabel (runtime-resolved), file, requirementIds, uiStates, scope? (display-only)",
3067
3086
  "- designEvidence: source, paths, conflicts; evidenceGaps[] (optional)",
3068
3087
  "- optional: stylingStrategy, uiComponentChoices[], dependencyPolicy, residualRisks[], realIntegrationGap",
3069
3088
  "- uiComponentChoices[]: purpose, component, decision (specified|reuse-existing|new), specReference { path, section, line } | null, rationale",
@@ -3146,14 +3165,52 @@ async function buildFrontendHybridDagFromTask(sources) {
3146
3165
  ...explicitFrontendVerifyCommands.behaviorCommands,
3147
3166
  ].map(verifyCommandKey));
3148
3167
  const adapterVerifyCommands = (sources.verifyCommands?.final ?? []).filter((command) => !explicitCommandKeys.has(verifyCommandKey(command)));
3168
+ // Split adapter commands by lane BEFORE handing them to either lane. The
3169
+ // same array used to be passed to both `chooseFrontendVerifyCommands` calls,
3170
+ // so every adapter command entered static and behavior at once. The behavior
3171
+ // lane is test-observed at writer admission, so a governance script from
3172
+ // harness.json (`bash scripts/check-repo.sh`) demanded test-report
3173
+ // capability it can never have and blocked admission deterministically.
3174
+ // Adapter commands carry no declared lane, so classify them from their text;
3175
+ // an explicit declaration still wins because it is chosen earlier.
3176
+ //
3177
+ // A declared task verifier owns this task's verification boundary and the
3178
+ // adapter array is discarded for both lanes, so an unclassifiable adapter
3179
+ // entry is harmless then and must not fail generation.
3149
3180
  const hasDeclaredFrontendVerification = explicitFrontendVerifyCommands.staticCommands.length > 0 ||
3150
3181
  explicitFrontendVerifyCommands.behaviorCommands.length > 0 ||
3151
3182
  parsedFrontendVerifyCommands.staticCommands.length > 0 ||
3152
3183
  parsedFrontendVerifyCommands.behaviorCommands.length > 0;
3184
+ const adapterStaticCommands = [];
3185
+ const adapterBehaviorCommands = [];
3186
+ for (const command of adapterVerifyCommands) {
3187
+ // Classify on the bare argv only. The frozen shell form prepends an env
3188
+ // prefix and a `cd …` prologue, which would obscure the runner name.
3189
+ const text = command.args.join(" ");
3190
+ const mode = classifyFrontendVerifyCommandText(text);
3191
+ if (mode === "static") {
3192
+ adapterStaticCommands.push(command);
3193
+ continue;
3194
+ }
3195
+ if (mode === "behavior") {
3196
+ adapterBehaviorCommands.push(command);
3197
+ continue;
3198
+ }
3199
+ // Unrecognized adapter command. When the task declares any verification
3200
+ // of its own, it owns the verification boundary and the adapter array is
3201
+ // discarded for both lanes, so an unclassifiable entry is harmless and
3202
+ // must not fail generation. Only an adapter command that would actually
3203
+ // be USED needs a lane, and defaulting it to `behavior` would make writer
3204
+ // admission demand test-report capability it cannot provide. Fail loudly
3205
+ // there and name the operator's fix.
3206
+ if (hasDeclaredFrontendVerification)
3207
+ continue;
3208
+ throw new Error(`adapter verification command ${JSON.stringify(text)} (label ${JSON.stringify(command.label)}) has no recognizable lane; declare it explicitly with mode: "static" or mode: "behavior" in task verifyCommands, or remove it from the harness manifest`);
3209
+ }
3153
3210
  const staticVerifyCommands = chooseFrontendVerifyCommands({
3154
3211
  explicitCommands: explicitFrontendVerifyCommands.staticCommands,
3155
3212
  parsedCommands: parsedFrontendVerifyCommands.staticCommands,
3156
- adapterCommands: adapterVerifyCommands,
3213
+ adapterCommands: adapterStaticCommands,
3157
3214
  });
3158
3215
  const partitionedStaticVerifyCommands = partitionFrontendStaticVerifyCommands({
3159
3216
  repoRoot: sources.repoRoot,
@@ -3163,7 +3220,7 @@ async function buildFrontendHybridDagFromTask(sources) {
3163
3220
  const behaviorVerifyCommands = chooseFrontendVerifyCommands({
3164
3221
  explicitCommands: explicitFrontendVerifyCommands.behaviorCommands,
3165
3222
  parsedCommands: parsedFrontendVerifyCommands.behaviorCommands,
3166
- adapterCommands: adapterVerifyCommands,
3223
+ adapterCommands: adapterBehaviorCommands,
3167
3224
  // A declared task verifier owns this task's verification boundary. A
3168
3225
  // static-only task must not inherit unrelated root-level test commands.
3169
3226
  allowAdapter: !hasDeclaredFrontendVerification,
@@ -3238,6 +3295,16 @@ async function buildFrontendHybridDagFromTask(sources) {
3238
3295
  behaviorCommandTexts: behaviorVerifyEvidence.commandTexts,
3239
3296
  mockCommandTexts: mockVerifyEvidence?.commandTexts ?? [],
3240
3297
  });
3298
+ // Freeze capability only after final selection; discarded adapter commands
3299
+ // cannot block this task. Unsupported inputs remain explicit preflight
3300
+ // failures and writer admission refuses them before any implementation.
3301
+ behaviorVerifyEvidence.preflight = await Promise.all(behaviorShellCommands.map(async (command, index) => {
3302
+ const label = behaviorVerifyEvidence.commandLabels[index] ?? command;
3303
+ const prior = sources.verificationPreflight?.find(entry => entry.label === label);
3304
+ const inspected = await inspectFrontendTestCommand({ command, cwd: sources.repoRoot ?? process.cwd(), workspaceRoot: sources.repoRoot ?? process.cwd(), allowedPaths: implementPaths.allowedPaths, forbiddenPaths });
3305
+ const commandId = frontendVerifyDirectory.find(entry => entry.mode === "behavior" && entry.label === label)?.commandId;
3306
+ return { label, status: prior?.status ?? "ok", testObservation: { ...inspected.observation, commandId } };
3307
+ }));
3241
3308
  const fixedVerificationContext = [
3242
3309
  "## Fixed frontend verification entrypoints",
3243
3310
  "These shell entrypoints are fixed at DAG generation and are the only commands the static and behavior shell nodes execute. A strategy or plan may add tests behind an existing entrypoint inside writeSet, but must not invent or replace commands or assume subtask_prompt executes a command.",
@@ -3429,7 +3496,7 @@ async function buildFrontendHybridDagFromTask(sources) {
3429
3496
  "Plan only the delta between the frozen frontend-contract-pi facts and frontend-scout-pi target surface. Do not reinterpret the task, repeat requirements, search the repository, or choose implementation order.",
3430
3497
  "Record only: requirement-to-file/verification coverage; component/styling choices; applicable UI state and interaction behavior; data/Mock strategy; and a dependency policy or genuine evidence gap. Reuse Scout paths. If scope is missing, record a blocking gap instead of inventing a path.",
3431
3498
  "Use the typed tool schemas as the field contract. Runtime owns schemaVersion, sourceBinding, riskLevel, targets.files, mockApi.productionDefaultOff, aliases, command allowlisting, path containment, and final validation; do not restate those rules or emit a full JSON contract.",
3432
- `Cover each frozen requirement ID exactly once: ${requirementIds.join(", ") || "(none)"}. Bind every verification target to a frozen commandId from the directory above plus a Scout-confirmed file. Behavior commands prove observable behavior: one target may cover multiple related requirementIds when one test behavior proves them together; do not mechanically create one target per requirement. A behavior target id is the stable identifier of that contract entry and its file must be a test file. Static commands are project-wide checks traced by file and command only.`,
3499
+ `Cover each frozen requirement ID exactly once: ${requirementIds.join(", ") || "(none)"}. Bind every verification target to a frozen commandId from the directory above plus a Scout-confirmed file. Behavior commands prove observable behavior: one target may cover multiple related requirementIds when one test behavior proves them together; do not mechanically create one target per requirement. A behavior target id is the stable machine trace token and its file must be a test file. Static commands are project-wide checks traced by file and command only.`,
3433
3500
  "UX vocabulary protocol: record_state_registry FIRST with the full global vocabulary — one stable kebab-case behavior-domain name per UI state/interaction (e.g. planner-task-edit, focus-queue-move), never one name per AC number and never a rename of an already-recorded concept. Details consume complete execution-group scopes and reuse the same global names across scopes. Constraints/exclusions must not manufacture UI. Then record_state_flow entries whose names all come from that registry; uiState names must use the contract's declared authoritative ids (declaredUiStates in the plan input) when present. Retry attempts see committedUx in this input — reuse those exact names. Components: one choice may cover many state/interaction ids via covers; reuse-existing requires evidencePath naming an existing repo file (greenfield must be decision=new).",
3434
3501
  ...(requiresOpenspecClassification ? ["When a component choice uses an OpenSpec selection, cite that selection; otherwise do not classify unrelated candidates."] : []),
3435
3502
  "Call finalize_plan; correct rejected facts until one successful terminal after the necessary typed facts. Return no Markdown narrative.",
@@ -3611,7 +3678,6 @@ async function buildFrontendHybridDagFromTask(sources) {
3611
3678
  "Execute in fixed stages and report each in the delivery summary: (1) Contract confirm, (2) Tests sync, (3) Component/UI state implementation, (4) API/Mock wiring per contract.mockApi, (5) Focused checks behind frozen entrypoints only, (6) Diff cleanup.",
3612
3679
  "Map every requirement id, expectedOutcome, interaction trigger/expectedBehavior, and applicable UI state from the contract to concrete files. Do not invent shell verification commands; only frozen static/behavior entrypoints will run.",
3613
3680
  "A behavior verification target's target.id is only the contract's identifier for that entry; it does not need to appear in test titles. Never add tests, rename describe/it/test titles, or restructure files just to carry generated ids — reuse affected existing test files and their names. The requirement ↔ verification-target association lives in the contract (requirementIds / verificationTargetIds), not in title strings.",
3614
- "Tests must genuinely prove the behavior each target maps to; a passing test-file execution does not by itself prove every mapped behavior is covered — keep assertions aligned with the contract's expectedBehavior, and state any residual gap honestly in Tests Changed.",
3615
3681
  "Begin implementation after the contract and its target files are confirmed. Do not spend the turn collecting optional context. If the canonical contract lacks behavior needed to edit safely, stop and state the blocking reason in the summary instead of reopening broad discovery.",
3616
3682
  "Your implementation status is derived by the executor from mechanical facts (persisted write-tool events, run delta, write guard, requirement coverage, focused-check failures), never from any IMPLEMENTATION_OUTCOME first line. Do not emit an IMPLEMENTATION_OUTCOME first line.",
3617
3683
  "The node runs a bounded micro-loop: after each write attempt the executor re-runs frozen focused checks and records a per-round diff checkpoint; the write guard stays active every round. Only repair local issues attributable to the current diff (syntax/type/import/format/unit-assert/obvious omission). Never change requirements, design, writeSet, or verification strictness inside the loop.",
@@ -5158,7 +5224,7 @@ async function buildBackendTestHybridDag(sources) {
5158
5224
  `STRICT_BACKEND_TEST_MODULE_LAYOUT=${JSON.stringify(taskConfig.backendTest.moduleLayout)}`,
5159
5225
  ]
5160
5226
  : []),
5161
- `Include exactly one \`## Module Index\` table with this exact header: \`${backendTestModuleIndexHeaderMarkdown()}\`. The Markdown Path cell must contain exactly one resolved repository-relative path such as \`${canonicalBackendTestModuleMarkdownPath(layout.markdownDir, "health")}\`; do not emit a Markdown link or repeat the path. Split Reason is exactly one of \`${BACKEND_TEST_MODULE_SPLIT_REASONS.join("\`, \`")}\`. Group by stable business resource/domain, not by CRUD operation, AC, parameter/field axis, scenario type or regression purpose: one resource's list/detail/create/update/delete and its filters/response assertions/regression floor belong in one module. Multiple modules owning the same exact \`METHOD /path\` are forbidden unless every such row is \`explicit-user-layout\` from primary-requirement path pairs or has a documented \`output-budget\` proof. Keep the total module count at the smallest safe value and never exceed 8 modules. Name model-derived modules with stable lowercase business stems such as \`health\` or \`resource_notes\`; explicit-user-layout preserves the primary requirement filename stem even when it is more specific. Do not use priority-only stems \`p0\`, \`p1\` or \`p2\`; Priority belongs only in the Coverage Matrix. Pure hexadecimal/hash-like opaque stems and test-purpose-only stems are forbidden. Do not use Case-ID-like module filenames. Markdown Path, Pytest Path and downstream automation mapping must be one-to-one and exact; for model-derived modules the default pair remains \`${layout.markdownDir}/<module>.md\` and \`${layout.scriptDir}/test_<module>.py\`, while explicit-user-layout preserves the primary requirement paths. Do not hand-write a conflicting module count in prose; the Module Index row count is the only count truth.`,
5227
+ `Include exactly one \`## Module Index\` table with this exact header: \`| Module Stem | Business Resource | Owned Operations | Owned Rule Keys | Case IDs | Split Reason | Markdown Path | Pytest Path |\`. The Markdown Path cell must contain exactly one resolved repository-relative path such as \`${layout.markdownDir}/health.md\`; do not emit a Markdown link or repeat the path. Split Reason is exactly one of \`explicit-user-layout\`, \`primary-business-resource\`, \`independent-business-resource\`, \`output-budget\`. Group by stable business resource/domain, not by CRUD operation, AC, parameter/field axis, scenario type or regression purpose: one resource's list/detail/create/update/delete and its filters/response assertions/regression floor belong in one module. Multiple modules owning the same exact \`METHOD /path\` are forbidden unless every such row is \`explicit-user-layout\` from primary-requirement path pairs or has a documented \`output-budget\` proof. Keep the total module count at the smallest safe value and never exceed 8 modules. Name model-derived modules with stable lowercase business stems such as \`health\` or \`resource_notes\`; explicit-user-layout preserves the primary requirement filename stem even when it is more specific. Do not use priority-only stems \`p0\`, \`p1\` or \`p2\`; Priority belongs only in the Coverage Matrix. Pure hexadecimal/hash-like opaque stems and test-purpose-only stems are forbidden. Do not use Case-ID-like module filenames. Markdown Path, Pytest Path and downstream automation mapping must be one-to-one and exact; for model-derived modules the default pair remains \`${layout.markdownDir}/<module>.md\` and \`${layout.scriptDir}/test_<module>.py\`, while explicit-user-layout preserves the primary requirement paths. Do not hand-write a conflicting module count in prose; the Module Index row count is the only count truth.`,
5162
5228
  "Scenario Partitions (query/filter axes): inspect every affected GET/list operation for query/path parameters whose bound source documents a finite enum or classification domain. If at least one such axis exists, add exactly one machine-readable `## Scenario Partitions` section after the Coverage Matrix using exactly `| Partition ID | Operation | Axis | Domain | Required Slots | Expected by Slot | Bind Rule |` with the separator row and one row per eligible axis. If no affected axis has a source-backed finite domain, omit the entire `## Scenario Partitions` heading and section; do not emit an explanatory prose-only section. Partition ID is a stable `SP-<OPERATION>-<AXIS>` token; Domain must copy the legal values verbatim from the bound OpenAPI enum or requirement sentence (never guess), using bare semicolon-separated identifier values inside the single table cell (for example `ACTIVE; ARCHIVED`, with no Markdown backticks or prose); Required Slots must contain `each-value` and exactly one `not-in-set`, plus `omitted` only when the parameter is optional. Before returning, expand every declared partition into its complete deterministic exact slot ID set: one `TP-<Partition ID>-<VALUE-TOKEN>` per Domain value, `TP-<Partition ID>-OMITTED` only for an optional axis, and exactly one `TP-<Partition ID>-NOT-IN-SET`. Every expanded slot ID must appear verbatim in the binding Rule's `Required Test Points` cell and be assigned to concrete Case IDs in that same Coverage Matrix row; ordinary alias/family Test Points do not replace this inventory. Scheme A: Case count may be smaller than the enum count, but every exact slot still needs an independent variant Test Point and pytest.param id; never use SINGLE/MULTIPLE aliases as coverage. Expected by Slot states the documented expectation per slot kind (`domain-value`, `default-behavior`, `empty-result`/`excluded-result` when documented, or `GAP` when the source does not document the complement expectation — never invent 空列表/400). POST/PUT body field-validation enums stay in the Coverage Matrix as `TP-<FIELD>-ENUM-*` and MUST NOT get a Scenario Partition row. Do not create partitions for axes without a documented legal-value domain. Only GET/list query or path parameters whose bound source documents a finite enum or classification set may become a Scenario Partition. Do not create partitions for free-form strings, primary keys, required-or-optional-only parameters, or boundary/format-only axes. If an axis has no finite legal-value domain, do not declare a Partition row and do not invent NOT-IN-SET cases. Cross-axis combinations stay as ONE nominal Case; never declare a cross-axis cartesian partition.",
5163
5229
  "Before finalizing README, calculate the predicted collected-item count as `sum(max(1, number of variant Test Points in each Case))`. If the task declares an item budget, the prediction must not exceed it. Reduce excess only by removing duplicate execution and converting same-request checkpoints to assertions; never drop required rules, boundaries, enums, operation-specific inputs, or business states. Record the prediction in README. Use only environment-supported fixtures/targets/isolation, record evidence gaps in Chinese, and do not emit JSON, pytest, or execute commands.",
5164
5230
  ...(sharedSetupPrompt ? [sharedSetupPrompt] : []),
@@ -5398,7 +5464,7 @@ async function buildBackendTestHybridDag(sources) {
5398
5464
  "Output budget protocol (hard, max output <=16K per turn): Write exactly the frozen `{{item.pytestPath}}`. Never paste full Python modules into assistant chat. Do not merge or split modules. Do not reduce params/assertions/skips to fit. If OUTPUT_LIMIT_RECOVERY is injected, continue only listed missing/broken scripts.",
5399
5465
  "Align every variant pytest.param payload with the Markdown scenario intent (empty/missing/null/length/pattern/enum/wrong-type/nominal). Prefer literal payloads over Faker for intent-critical fields so pre-execution scenario-param checks can verify them. Hard contract: intent=enum-invalid MUST pass a concrete invalid value literal (string/number/boolean), never `_OMIT`/None/missing key; intent=missing/empty may use `_OMIT` or delete the key; intent=custom-literal:trim|whitespace-padded requires a leading/trailing whitespace string with non-empty trimmed content (all-whitespace belongs to empty/whitespace-only, not trim); intent=custom-literal:ACTIVE|ARCHIVED requires the exact enum string, never descriptive tokens like filter-active; intent=max/min/max+1 should pass a repeated-string length expression, a bare length number N, or a helper named _*_LEN{N} / _*_MAX_LENGTH / _*_OVER_LENGTH — never a bare 1 for oversize. Hard contract: request payload dicts may only contain DTO field keys from Payload Allowed Paths; never put expect/expected/echo_* helper keys inside the JSON body dict. Path/query/header identifiers and scenario-control metadata (including `id`, expected codes, and selector labels) must stay in separate pytest parameters and helper arguments; never merge them into a DTO patch or JSON body unless that exact path is allowed by the Markdown payload contract. Normalize the configured API base URL with `rstrip(\"/\")` (or equivalently join exactly one slash) before appending endpoint paths; generated requests must never contain a `//api/...` path. When the bound source documents a concrete non-secret local API URL, generated clients must use it as the fallback in `os.environ.get(\"API_BASE_URL\", \"<documented-url>\")`; do not require an otherwise-uninjected environment variable or fail setup solely because it is absent. Missing-field helpers must remove keys idempotently with `payload.pop(field, None)`, never `del payload[field]`, because optional fields may already be absent.",
5400
5466
  "For every response contract that requires an object or pagination envelope, first assert that each envelope/data value is a dict and that required keys exist, then index fields and assert values. Never let an incidental KeyError or list/string TypeError stand in for the explicit response-shape contract failure.",
5401
- 'Ensure every automatable final Markdown Case ID in this module appears in exactly one primary pytest test function or pytest test class method region, using the exact `primary symbol` declared by Markdown. Skip evidence-only meta Cases that declare `脚本/primary symbol=无` with empty variants; do not invent a business pytest symbol for them. The symbol must start with `test_BE_<MODULE>_<NNN>_` so every parameterized collected item remains associated with its Case. Module-level functions and class-based pytest methods are both supported. Only `变体测试点` may use stable `pytest.param(..., id="TP-...")` IDs, and every atomic variant ID must appear exactly once with a genuine input/state/outcome change. A Case with exactly one variant Test Point still needs one literal `pytest.param(..., id="TP-...")` row; never leave a single-variant Case as a bare function with the TP only in the docstring. Use a literal direct `pytest.param(..., id=...)` expression for every row; never hide or wrap it behind `_post_case`, `_put_case`, row-factory functions, comprehensions, generators, or dynamically returned parameter lists; do not use decorator-level `ids=[...]`, generated suffixes, or IDs that extend/shorten the exact Markdown TP. Do not parameterize `场景断言测试点` or `横切证据测试点`; execute all assertion checkpoints within the same business journey/item and use shared helpers for cross-cutting evidence. Governance-only cross-cutting bindings such as writeSet compliance, execution count, report existence, or orchestration state are metadata-only in business pytest: preserve their IDs in `Cross-Cutting-Test-Points`, but never assert `__file__`, filesystem placement, pytest invocation count, Harness state, or report artifacts inside the business test. Harness-owned evidence verifies those bindings. The first statement inside every primary symbol must be a triple-quoted docstring containing exact lines `Case-ID: BE-...`, `Assertion-Test-Points: TP-...;TP-...` and `Cross-Cutting-Test-Points: TP-...;TP-...` (use `none` when empty), for example `def test_BE_X_001():\n """\n Case-ID: BE-X-001\n Assertion-Test-Points: TP-X-ASSERT\n Cross-Cutting-Test-Points: none\n """`. Module/class docstrings, comments before `def`, and singular `Assertion-Test-Point:` comments never bind a Test Point. Implement request dictionaries so their direct and nested key paths and enum literals exactly satisfy the Case `Payload Required Paths`, `Payload Allowed Paths`, and `Payload Enum`; for `Payload Contract: none`, do not invent a JSON/body DTO. GET/list filters still declare query fields in those payload labels when the Case varies `params=`/`query=` keys. Python `True`/`False` may implement JSON/OpenAPI `true`/`false` query or body booleans. GET/DELETE setup journeys may create resources, but their setup DTO must not change the target operation\'s no-body payload contract. No Test Point may be invented, renamed, omitted or bound in two modes. The generated pytest collection shape must equal the Markdown prediction `sum(max(1, variant count per Case))`; keep it at or below the task\'s explicit budget by removing duplicate execution, never by collapsing multiple parameter rows under a coarse family TP. Assertions come only from 预期结果 and setup comes only from 前置条件/测试数据/自动化映射.',
5467
+ 'Ensure every automatable final Markdown Case ID in this module appears in exactly one primary pytest test function or pytest test class method region, using the exact `primary symbol` declared by Markdown. Skip evidence-only meta Cases that declare `脚本/primary symbol=无` with empty variants; do not invent a business pytest symbol for them. The symbol must start with `test_BE_<MODULE>_<NNN>_` so every parameterized collected item remains associated with its Case. Module-level functions and class-based pytest methods are both supported. Only `变体测试点` may use stable `pytest.param(..., id="TP-...")` IDs, and every atomic variant ID must appear exactly once with a genuine input/state/outcome change. A Case with exactly one variant Test Point still needs one literal `pytest.param(..., id="TP-...")` row; never leave a single-variant Case as a bare function with the TP only in the docstring. Use a literal direct `pytest.param(..., id=...)` expression for every row; never hide or wrap it behind `_post_case`, `_put_case`, row-factory functions, comprehensions, generators, or dynamically returned parameter lists; do not use decorator-level `ids=[...]`, generated suffixes, or IDs that extend/shorten the exact Markdown TP. Do not parameterize `场景断言测试点` or `横切证据测试点`; execute all assertion checkpoints within the same business journey/item and use shared helpers for cross-cutting evidence. Governance-only cross-cutting bindings such as writeSet compliance, execution count, report existence, or orchestration state are metadata-only in business pytest: preserve their IDs in `Cross-Cutting-Test-Points`, but never assert `__file__`, filesystem placement, pytest invocation count, Harness state, or report artifacts inside the business test. Harness-owned evidence verifies those bindings. The primary symbol docstring must contain exact metadata lines `Case-ID: BE-...`, `Assertion-Test-Points: TP-...;TP-...` and `Cross-Cutting-Test-Points: TP-...;TP-...` (use `none` when empty). Implement request dictionaries so their direct and nested key paths and enum literals exactly satisfy the Case `Payload Required Paths`, `Payload Allowed Paths`, and `Payload Enum`; for `Payload Contract: none`, do not invent a JSON/body DTO. GET/list filters still declare query fields in those payload labels when the Case varies `params=`/`query=` keys. Python `True`/`False` may implement JSON/OpenAPI `true`/`false` query or body booleans. GET/DELETE setup journeys may create resources, but their setup DTO must not change the target operation\'s no-body payload contract. No Test Point may be invented, renamed, omitted or bound in two modes. The generated pytest collection shape must equal the Markdown prediction `sum(max(1, variant count per Case))`; keep it at or below the task\'s explicit budget by removing duplicate execution, never by collapsing multiple parameter rows under a coarse family TP. Assertions come only from 预期结果 and setup comes only from 前置条件/测试数据/自动化映射.',
5402
5468
  "Name the generated pytest file so it corresponds one-to-one with its source Markdown module file: this module stem `{{item.stem}}` maps to exactly the frozen `{{item.pytestPath}}`. The <module> stem is the Markdown filename without the `.md` extension, lowercased and with non-alphanumeric characters replaced by underscores. For example, `resource_notes` → `testcase/test_resource_notes.py`, `health` → `testcase/test_health.py`. If Markdown automation mapping names a different path than this module stem path, still write the frozen manifest pytest path and do not invent prefixes. Never merge multiple Markdown modules into one pytest file, never split one module across several files, and never invent pytest filenames unrelated to the Markdown modules.",
5403
5469
  "Scenario Partition slots: every `TP-<Partition ID>-...` variant Test Point declared by this module's Markdown MUST become exactly one literal direct `pytest.param(..., id=\"TP-<Partition ID>-...\")` row with the exact slot ID; the not-in-set slot passes a concrete literal absent from the documented Domain (e.g. `UNKNOWN_TYPE`) — never `_OMIT`, never a descriptive token. Never split one slot into multiple params or merge several slots under a family TP id. Slot filtering requests hit the documented list endpoint with the slot value as the query/path filter.",
5404
5470
  "Keep this module self-contained: define module-local fixtures and helpers directly in `{{item.pytestPath}}`, so pytest discovers every fixture dependency without external plugin registration. The request log must include method, URL/path, and request parameters (query plus JSON/body/payload summary). The response log must include status code and response result (JSON/text/body summary), and both records must be visible in pytest stdout/stderr without changing assertions. Recursively redact sensitive values and apply bounded truncation before logging.",
@@ -5456,7 +5522,7 @@ async function buildBackendTestHybridDag(sources) {
5456
5522
  outputContract: "First non-empty line is IMPLEMENTATION_OUTCOME: changed|blocked, followed by a concise repair summary. This node runs only for REPAIRABLE initial facts, so already-satisfied is invalid and a successful outcome requires a non-empty bounded diff. Modify only generated pytest scripts/helpers/factories and preserve every Markdown Case, Test Point, primary symbol and assertion meaning.",
5457
5523
  subtask_prompt: [
5458
5524
  "Repair the generated backend pytest asset as one bounded program using the direct upstream collection assessment. This is the only repair attempt and happens before any business test body execution. The direct upstream JSON includes authoritative `repairPaths` and bounded `repairFindings`; treat both as the complete mandatory checklist without searching for a run directory or report file. Treat any upstream line such as `Repair paths: testcase/test_x.py` as equivalent authoritative repairPaths evidence. Directly read and edit that testcase path; do not search for separate root-level `contracts/**`, guess a DAG run directory, or require another report artifact. If the read tool successfully returns the testcase file, the path exists—continue the bounded repair and never later claim that file is absent.",
5459
- "Initial status REPAIRABLE means at least one listed finding remains: `already-satisfied` is forbidden, and you must produce a non-empty bounded diff on repairPaths before returning `IMPLEMENTATION_OUTCOME: changed`. Fix only readiness-proven generated testcase-local defects on initial facts repairPaths: create exact safe missing mapped test_*.py paths, repair syntax/import/symbol/decorator/parameterization, close generated fixture dependencies/plugin registration, and repair initial Markdown-to-pytest correspondence findings. Use this deterministic repair map instead of reading analyzer implementation: findings about `Case-ID`, `Assertion-Test-Points`, or `Cross-Cutting-Test-Points` are fixed by making the first statement in the declared primary symbol a triple-quoted docstring with the exact plural metadata lines; these are the primary symbol docstring metadata lines. Module/class docstrings, comments before `def`, and singular `Assertion-Test-Point:` comments are invalid. Variant binding findings are fixed in the literal direct `pytest.param(..., id=\"TP-...\")` row; primary-symbol cardinality/name findings are fixed in the function name or duplicate primary symbols; script mismatch is fixed only on the authoritative assessment repairPaths; payload findings are fixed in request payload construction. Do not read controller `src/**` or inspect JS/TS analyzer code. Do not search for `testcase/**/README.md`. Never invent a business pytest symbol for evidence-only Markdown Cases that declare `脚本/primary symbol=无` with empty variants. For fixture defects inspect both provider and importer listed by repairPaths; fix ScopeMismatch by aligning fixture scopes or inlining request-scoped values so module fixtures never depend on function fixtures; when a shared fixture depends on sibling fixtures, register the whole provider module through an exact pytest_plugins declaration rather than importing only the outer fixture. Do not create unrelated pytest scripts.",
5525
+ "Initial status REPAIRABLE means at least one listed finding remains: `already-satisfied` is forbidden, and you must produce a non-empty bounded diff on repairPaths before returning `IMPLEMENTATION_OUTCOME: changed`. Fix only readiness-proven generated testcase-local defects on initial facts repairPaths: create exact safe missing mapped test_*.py paths, repair syntax/import/symbol/decorator/parameterization, close generated fixture dependencies/plugin registration, and repair initial Markdown-to-pytest correspondence findings. Use this deterministic repair map instead of reading analyzer implementation: findings about `Case-ID`, `Assertion-Test-Points`, or `Cross-Cutting-Test-Points` are fixed by editing the declared primary symbol docstring metadata lines; variant binding findings are fixed in the literal direct `pytest.param(..., id=\"TP-...\")` row; primary-symbol cardinality/name findings are fixed in the function name or duplicate primary symbols; script mismatch is fixed only on the authoritative assessment repairPaths; payload findings are fixed in request payload construction. Do not read controller `src/**` or inspect JS/TS analyzer code. Do not search for `testcase/**/README.md`. Never invent a business pytest symbol for evidence-only Markdown Cases that declare `脚本/primary symbol=无` with empty variants. For fixture defects inspect both provider and importer listed by repairPaths; fix ScopeMismatch by aligning fixture scopes or inlining request-scoped values so module fixtures never depend on function fixtures; when a shared fixture depends on sibling fixtures, register the whole provider module through an exact pytest_plugins declaration rather than importing only the outer fixture. Do not create unrelated pytest scripts.",
5460
5526
  "This is the single pytest incremental synchronization round. The `Findings` in `reports/backend-test-pytest-collection-initial.md` are the mandatory repair checklist: resolve every repairable listed finding on every authoritative `Repair paths` file before considering any other advisory evidence, and never substitute an unrelated scenario-param cleanup for a listed correspondence/collection defect. For every assessment-listed path, compare the effective Markdown Case/Test Points/test data and its `Payload Contract`/`Payload Required Paths`/`Payload Allowed Paths`/`Payload Enum` labels with the generated module. Incrementally add or repair only missing symbols, params, assertions and payload builders. Repair every assessment-listed missing nested path, unexpected key and enum mismatch; preserve exact DTO keys, nested shapes, enum/boundary literals, operation transport and business preconditions; remove guessed replacement keys only when the effective Markdown proves the exact contract. Keep path/query/header identifiers and scenario-control metadata separate from DTO patches and JSON bodies; an `id` used for a path target must be passed to the request path/helper, never inserted into a body patch unless `id` is explicitly listed in Payload Allowed Paths. Flatten every variant into a literal direct `pytest.param(..., id=\"TP-...\")` row; replace `_post_case`/`_put_case` or other parameter-row factories because correspondence and scenario readiness require the actual row values and IDs to be statically visible. Also repair helper call sites to match their defined return signatures; do not tuple-unpack a helper that returns one scalar value.",
5461
5527
  "Preserve final testcase/md/** semantics, every Case ID, Rule/Test Point binding, primary symbol, parameter ID, expected status/body/schema assertion, HTTP logging, redaction and truncation behavior.",
5462
5528
  "Use local edit only on assessment-listed paths; keep summaries short; never rewrite unrelated modules.",
@@ -7397,6 +7463,11 @@ async function buildHybridDagForTemplate(sources, template, options = {}) {
7397
7463
  spec.runtimeContract = GENERATED_DAG_RUNTIME_CONTRACT;
7398
7464
  }
7399
7465
  spec.sourceBinding = buildDagSourceBinding(sources, template === "frontend-implementation" ? "frontend-implementation" : undefined);
7466
+ if (template === "frontend-implementation" && sources.repoRoot) {
7467
+ const ledger = await readTaskSourceProviderBudget({ repoRoot: sources.repoRoot, taskId: sources.taskId, maxProviderRequests: spec.budget.limits.maxProviderRequests });
7468
+ if (ledger)
7469
+ spec.sourceProviderBudget = { ledger, contract: sourceBudgetContractIdentity(taskContractBinding), sourceBindingSha256: sha256OfCanonicalJson(spec.sourceBinding), evidence: await freezeTaskSourceUsage({ repoRoot: sources.repoRoot, taskId: sources.taskId, maxProviderRequests: ledger.maxProviderRequests }, ledger) };
7470
+ }
7400
7471
  spec.taskContractBinding = taskContractBinding;
7401
7472
  assertNoGovernanceFlagOnDisallowedTemplate(spec, template);
7402
7473
  if (template === "standard-dag" ||
@@ -1,3 +1,5 @@
1
+ import { reserveTaskSourceProviderRequest } from "../../task/source-prepare/source-provider-budget.js";
2
+ import { frontendPlanSemanticDigest } from "./frontend-plan-progress.js";
1
3
  import { collectFrontendExecutionGroups } from "./frontend-execution-groups.js";
2
4
  import { reserveDagProviderRequest } from "./budget-enforcement.js";
3
5
  import { createHash } from "node:crypto";
@@ -374,15 +376,11 @@ async function countContractRecordSubmissions(runDir, nodeId) {
374
376
  * `hasTerminalFact` reflects a committed `finalize_plan` terminal fact.
375
377
  */
376
378
  async function readFrontendPlanCommittedSnapshot(runDir, nodeId) {
377
- const records = await readTypedEventStoreFromJsonl(path.join(runDir, nodeId, "plan-typed-facts.jsonl"));
378
- const committed = records.filter((record) => record.phase === "committed");
379
- const digest = createHash("sha256")
380
- .update(committed
381
- .map((record) => record.payloadSha256)
382
- .sort()
383
- .join("\n"))
384
- .digest("hex");
385
- const hasTerminalFact = committed.some((record) => record.fact.kind === "finalize_plan");
379
+ const decisions = await readTypedEventStoreFromJsonl(path.join(runDir, nodeId, "plan-decision-facts.jsonl"));
380
+ const records = decisions.some(r => r.phase === "committed") ? decisions : await readTypedEventStoreFromJsonl(path.join(runDir, nodeId, "plan-typed-facts.jsonl"));
381
+ const committed = records.filter(record => record.phase === "committed");
382
+ const digest = frontendPlanSemanticDigest(committed);
383
+ const hasTerminalFact = committed.some((record) => ["finalize_plan", "finalize_decision"].includes(String(record.fact.kind)));
386
384
  return { digest, hasTerminalFact };
387
385
  }
388
386
  function planInputRecord(value) {
@@ -1564,6 +1562,7 @@ export async function executeDagNode(input) {
1564
1562
  // evidence so a late best-effort write cannot recreate an archived run dir.
1565
1563
  void input.persistState().catch(() => { });
1566
1564
  };
1565
+ const scoutCompletedShards = task.id === "frontend-scout-pi" && isSafeReadOnlyPiRetryCandidate(task) ? new Map() : undefined;
1567
1566
  let terminalResult;
1568
1567
  let previousFailureCategory;
1569
1568
  let previousProtocolReason;
@@ -1665,7 +1664,8 @@ export async function executeDagNode(input) {
1665
1664
  attemptPrompt,
1666
1665
  });
1667
1666
  result = await executeNode({
1668
- ...(state.budgetLedger?.mode === "hard" && state.budgetLedger.limits.maxProviderRequests !== undefined ? { reserveProviderRequest: () => reserveDagProviderRequest({ state, nodeId, attempt: attemptNumber, persist: input.persistState }) } : {}),
1667
+ ...(scoutCompletedShards ? { scoutCompletedShards } : {}),
1668
+ ...(state.budgetLedger?.mode === "hard" && state.budgetLedger.limits.maxProviderRequests !== undefined ? { reserveProviderRequest: () => reserveDagProviderRequest({ state, nodeId, attempt: attemptNumber, persist: input.persistState, ...(spec.sourceProviderBudget ? { reserveTaskRequest: () => reserveTaskSourceProviderRequest({ repoRoot: cwd, taskId: spec.sourceProviderBudget.ledger.taskId, maxProviderRequests: spec.sourceProviderBudget.ledger.maxProviderRequests, expectedBudgetId: spec.sourceProviderBudget.ledger.budgetId, minimumReservedRequests: spec.sourceProviderBudget.ledger.reservedRequests }) } : {}) }) } : {}),
1669
1669
  task,
1670
1670
  cwd,
1671
1671
  model,
@@ -1906,7 +1906,7 @@ export async function executeDagNode(input) {
1906
1906
  // reason, detect new committed facts, and escalate instead of re-running.
1907
1907
  if (frontendPlanLadderEnabled && !result.ok) {
1908
1908
  const afterSnapshot = await readFrontendPlanCommittedSnapshot(runDir, nodeId);
1909
- const hasNewCommittedFact = beforeCommittedDigest !== undefined &&
1909
+ const hasNewCommittedFact = !result.stderr?.includes("FRONTEND_PLAN_NO_PROGRESS") && beforeCommittedDigest !== undefined &&
1910
1910
  afterSnapshot.digest !== beforeCommittedDigest;
1911
1911
  const protocolReason = projectFrontendNodeProtocolFailureReason({
1912
1912
  failureCategory: result.failureCategory,
@@ -1,9 +1,10 @@
1
+ import { z } from "zod";
1
2
  import { createHash } from "node:crypto";
2
3
  import { readFile } from "node:fs/promises";
3
4
  import path from "node:path";
4
5
  import { parseJsonReviewVerdict } from "./output-protocol.js";
5
6
  import { FRONTEND_DESIGN_REVIEW_NODE_ID, FRONTEND_REVIEW_NODE_ID, readCommittedDesignRequestFact, readCommittedReviewRequestFact, } from "./frontend-review-findings.js";
6
- import { readTypedEventStoreFromJsonl } from "./frontend-typed-event-store.js";
7
+ import { readTypedEventStoreFromJsonl, reviewFindingSchema } from "./frontend-typed-event-store.js";
7
8
  export const MAX_RERUN_FEEDBACK_CHARS = 12_000;
8
9
  const RERUN_FEEDBACK_TRUNCATION_MARKER = "\n\n...[上一轮反馈已按长度上限截断]...\n\n";
9
10
  const MAX_EVIDENCE_REFS = 8;
@@ -236,6 +237,35 @@ async function readCommittedTerminalKind(input) {
236
237
  }
237
238
  return undefined;
238
239
  }
240
+ /** Current rejecting terminal, with final-review approval as a barrier to
241
+ * older design findings. Binds snapshot consumers to the actual ledger bytes. */
242
+ export async function readCurrentFrontendRepairFindings(runDir) {
243
+ for (const [sourceNodeId, file, approveKind, rejectKind] of [
244
+ [FRONTEND_REVIEW_NODE_ID, "review-typed-facts.jsonl", "approve_review", "request_review_changes"],
245
+ [FRONTEND_DESIGN_REVIEW_NODE_ID, "design-typed-facts.jsonl", "approve_design", "request_design_changes"],
246
+ ]) {
247
+ const ledger = path.join(runDir, sourceNodeId, file);
248
+ let raw;
249
+ try {
250
+ raw = await readFile(ledger, "utf8");
251
+ }
252
+ catch (error) {
253
+ if (error.code === "ENOENT")
254
+ continue;
255
+ throw error;
256
+ }
257
+ const terminal = (await readTypedEventStoreFromJsonl(ledger)).filter(record => record.phase === "committed" && [approveKind, rejectKind].includes(String(record.fact.kind))).at(-1);
258
+ if (!terminal)
259
+ continue;
260
+ const { sha256OfCanonicalJson } = await import("../../task/contract/hash.js");
261
+ if (terminal.payloadSha256 !== sha256OfCanonicalJson(terminal.fact))
262
+ throw new Error("frontend repair terminal payload hash mismatch");
263
+ const verdict = terminal.fact.kind === approveKind ? "approve" : "request";
264
+ const findings = verdict === "request" ? z.array(reviewFindingSchema).parse(terminal.fact.findings) : [];
265
+ return { sourceNodeId, ledgerSha256: sha256(raw), terminalEventId: terminal.eventId, verdict, findings, fact: terminal.fact };
266
+ }
267
+ return undefined;
268
+ }
239
269
  async function readReviewTerminalKind(runDir) {
240
270
  return readCommittedTerminalKind({
241
271
  runDir,
@@ -354,9 +384,10 @@ export async function deriveDagRerunFeedback(input) {
354
384
  // reviewer re-flags them as "unresolved previous-round feedback" (r18).
355
385
  // A committed approval is a resolution barrier and prevents older
356
386
  // findings from being resurrected.
357
- const typedFact = (await readTypedReviewRequestFact(input.runDir)) ??
358
- (await readTypedDesignRequestFact(input.runDir)) ??
359
- (await readAncestorTypedRequestFact(input.runDir));
387
+ const current = await readCurrentFrontendRepairFindings(input.runDir);
388
+ if (current?.verdict === "approve")
389
+ return undefined;
390
+ const typedFact = current ? { nodeId: current.sourceNodeId, fact: current.fact } : await readAncestorTypedRequestFact(input.runDir);
360
391
  if (typedFact) {
361
392
  const parentSourceBindingHash = sourceBindingHash(input.spec);
362
393
  return withDigest({