@tea-agent/loop-agent 0.39.0-beta.12 → 0.39.0-beta.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (114) hide show
  1. package/CHANGELOG.md +27 -0
  2. package/dist/application/dag/generate-task-dag.js +6 -2
  3. package/dist/application/task-lifecycle/advance.js +20 -4
  4. package/dist/application/task-lifecycle/observe.js +171 -17
  5. package/dist/application/task-lifecycle/plan-transitions.js +42 -7
  6. package/dist/build-stamp.json +3 -3
  7. package/dist/commands/client-recovery.js +8 -36
  8. package/dist/executors/dag-pi-executor.js +636 -109
  9. package/dist/executors/pi-executor.js +8 -4
  10. package/dist/executors/pi-sdk-executor.js +33 -5
  11. package/dist/executors/shell-executor.js +102 -32
  12. package/dist/governance/checks.js +1 -0
  13. package/dist/shared/pi-context-pressure/checkpoint.js +116 -0
  14. package/dist/shared/pi-context-pressure/compaction-policy.js +151 -0
  15. package/dist/shared/pi-context-pressure/env.js +58 -0
  16. package/dist/shared/pi-context-pressure/extension.js +100 -0
  17. package/dist/shared/pi-context-pressure/index.js +7 -0
  18. package/dist/shared/pi-context-pressure/overflow.js +252 -0
  19. package/dist/shared/pi-context-pressure/sift-bridge.js +386 -0
  20. package/dist/shared/pi-context-pressure/telemetry.js +51 -0
  21. package/dist/task/frontend-project-capability.js +3 -1
  22. package/dist/task/source-prepare/fragment-inventory.js +4 -1
  23. package/dist/worker/console/chat/pi-runtime.js +146 -5
  24. package/dist/worker/console/chat/provider-error.js +2 -1
  25. package/dist/worker/console/chat/routes.js +3 -0
  26. package/dist/worker/console/chat/sift-bridge.js +1 -0
  27. package/dist/worker/console/dag-execution-receipt.js +20 -2
  28. package/dist/worker/console/operator-actions.js +4 -3
  29. package/dist/worker/console/static/assets/{abnfDiagram-N423BO3Z-CXj_GnSb.js → abnfDiagram-N423BO3Z-DC863mud.js} +1 -1
  30. package/dist/worker/console/static/assets/{arc-BZp6JAp7.js → arc-CftC38G9.js} +1 -1
  31. package/dist/worker/console/static/assets/{architectureDiagram-T3A2C74G-DfpcEuYU.js → architectureDiagram-T3A2C74G-B3PAPQCw.js} +1 -1
  32. package/dist/worker/console/static/assets/{blockDiagram-VBNYF7ZC-_Gf0xadb.js → blockDiagram-VBNYF7ZC-C9UDJNRv.js} +1 -1
  33. package/dist/worker/console/static/assets/{c4Diagram-5PPSVZJV-Ct2QPCmv.js → c4Diagram-5PPSVZJV--BJ76fv_.js} +1 -1
  34. package/dist/worker/console/static/assets/channel-CUz-Bg86.js +1 -0
  35. package/dist/worker/console/static/assets/{chunk-2GRJ4B5K-BfklkzKl.js → chunk-2GRJ4B5K--hiIqoGp.js} +1 -1
  36. package/dist/worker/console/static/assets/{chunk-2Q5K7J3B-Dy25vZJV.js → chunk-2Q5K7J3B-DgTzpIa3.js} +1 -1
  37. package/dist/worker/console/static/assets/{chunk-5RXB4S5H-BFlCRZep.js → chunk-5RXB4S5H-DIwpJziP.js} +1 -1
  38. package/dist/worker/console/static/assets/{chunk-5VM5RSS4-op3oVxIE.js → chunk-5VM5RSS4-BJzbdUi0.js} +1 -1
  39. package/dist/worker/console/static/assets/{chunk-6Q2QTUOP-C0R7rzF2.js → chunk-6Q2QTUOP-BbGouI1Z.js} +1 -1
  40. package/dist/worker/console/static/assets/{chunk-GF5L2VYU-Bt44TCGy.js → chunk-GF5L2VYU-B9247w69.js} +1 -1
  41. package/dist/worker/console/static/assets/{chunk-JWPE2WC7-_uEE_XFx.js → chunk-JWPE2WC7-CXob_wYy.js} +1 -1
  42. package/dist/worker/console/static/assets/{chunk-KBJHAD2P-C3TOYGZ9.js → chunk-KBJHAD2P-C6FUel1g.js} +1 -1
  43. package/dist/worker/console/static/assets/{chunk-RYQCIY6F-Cz60oBPV.js → chunk-RYQCIY6F-DGKRYKHu.js} +1 -1
  44. package/dist/worker/console/static/assets/{chunk-XXDRQBXY-8Bik0qis.js → chunk-XXDRQBXY-kNBqNPfx.js} +1 -1
  45. package/dist/worker/console/static/assets/classDiagram-JCYQIIEL-BhlUacmD.js +1 -0
  46. package/dist/worker/console/static/assets/classDiagram-v2-OCEON4UE-BhlUacmD.js +1 -0
  47. package/dist/worker/console/static/assets/{cose-bilkent-JH36ORCC-C2QIOA4a.js → cose-bilkent-JH36ORCC--xmkDwfD.js} +1 -1
  48. package/dist/worker/console/static/assets/{cynefin-VYW2F7L2-CSH_yUUd.js → cynefin-VYW2F7L2-Cw5FMMuG.js} +1 -1
  49. package/dist/worker/console/static/assets/{cynefinDiagram-MW4NZA55-BDLw2XFM.js → cynefinDiagram-MW4NZA55-CuLslt2t.js} +1 -1
  50. package/dist/worker/console/static/assets/{dagre-VZM6K2ZE-CcD9ZtF4.js → dagre-VZM6K2ZE-D9ngqrs2.js} +1 -1
  51. package/dist/worker/console/static/assets/{diagram-7IWD3JNH-DqwtkBsI.js → diagram-7IWD3JNH-RDRSbHQp.js} +1 -1
  52. package/dist/worker/console/static/assets/{diagram-B4RE2ZJO-BWVEcqsC.js → diagram-B4RE2ZJO-BgQ1P9aV.js} +1 -1
  53. package/dist/worker/console/static/assets/{diagram-LBJQPF4R-M-WAbtQi.js → diagram-LBJQPF4R-D3WWyPtn.js} +1 -1
  54. package/dist/worker/console/static/assets/{diagram-Q27KOJAE-DKW7SixP.js → diagram-Q27KOJAE-L2k28wTR.js} +1 -1
  55. package/dist/worker/console/static/assets/{diagram-UB23O5K3-PHHPPtrj.js → diagram-UB23O5K3-DDNrA5Qw.js} +1 -1
  56. package/dist/worker/console/static/assets/{ebnfDiagram-BXEA7PRR-BQD-B4RX.js → ebnfDiagram-BXEA7PRR-BaRFCOdM.js} +1 -1
  57. package/dist/worker/console/static/assets/{erDiagram-JOGREHBK-BFvULb51.js → erDiagram-JOGREHBK-CM33DVVM.js} +1 -1
  58. package/dist/worker/console/static/assets/{flowDiagram-UKHOOZJN-Bw-aXTCb.js → flowDiagram-UKHOOZJN-CVBSjOxe.js} +1 -1
  59. package/dist/worker/console/static/assets/{ganttDiagram-PKOTCBZU-BGKw04Qx.js → ganttDiagram-PKOTCBZU-BdscewE_.js} +1 -1
  60. package/dist/worker/console/static/assets/{gitGraphDiagram-DS77QQ5N-RqKaPR-I.js → gitGraphDiagram-DS77QQ5N-HRuSZb7g.js} +1 -1
  61. package/dist/worker/console/static/assets/{index-B_D8rbWc.js → index-B9JJQsVK.js} +102 -72
  62. package/dist/worker/console/static/assets/index-rWaGv4jz.css +1 -0
  63. package/dist/worker/console/static/assets/{infoDiagram-6WML65LV-YO5dnzrJ.js → infoDiagram-6WML65LV-cYfHvfAR.js} +1 -1
  64. package/dist/worker/console/static/assets/{ishikawaDiagram-WSZJBQD7-Bi1VZHMq.js → ishikawaDiagram-WSZJBQD7-CZmaoOuy.js} +1 -1
  65. package/dist/worker/console/static/assets/{journeyDiagram-NVQOT4AX-Bszls1DD.js → journeyDiagram-NVQOT4AX-ntYijh7g.js} +1 -1
  66. package/dist/worker/console/static/assets/{kanban-definition-27J2QSJJ-PhzeaZ09.js → kanban-definition-27J2QSJJ-Cz3b0DeD.js} +1 -1
  67. package/dist/worker/console/static/assets/{linear-BsjbDoXi.js → linear-DvGonpsP.js} +1 -1
  68. package/dist/worker/console/static/assets/{mermaid.core-0B7NnWKk.js → mermaid.core-B-3vjfyW.js} +5 -5
  69. package/dist/worker/console/static/assets/{mindmap-definition-FAOFIHXS-BJr4Fj-q.js → mindmap-definition-FAOFIHXS-5iKxlsX3.js} +1 -1
  70. package/dist/worker/console/static/assets/{pegDiagram-VL7TDLO6-moC4fpGB.js → pegDiagram-VL7TDLO6-C8BUwapW.js} +1 -1
  71. package/dist/worker/console/static/assets/{pieDiagram-7S7Q4E2Y-BEw37-2c.js → pieDiagram-7S7Q4E2Y-B2rPoQQG.js} +1 -1
  72. package/dist/worker/console/static/assets/{quadrantDiagram-CIZ2JOQS-Cq6LyasU.js → quadrantDiagram-CIZ2JOQS-jDeVUAy4.js} +1 -1
  73. package/dist/worker/console/static/assets/{railroadDiagram-AXF67PYL-DUCMcK0D.js → railroadDiagram-AXF67PYL-DV140Dwm.js} +1 -1
  74. package/dist/worker/console/static/assets/{requirementDiagram-LRYGKXZP-C3upTZm7.js → requirementDiagram-LRYGKXZP-CnYgHtYC.js} +1 -1
  75. package/dist/worker/console/static/assets/{sankeyDiagram-W5VNT64P-BI_gMsCW.js → sankeyDiagram-W5VNT64P-SIK3CdWw.js} +1 -1
  76. package/dist/worker/console/static/assets/{sequenceDiagram-SI44F4Z6-YFOIRzfN.js → sequenceDiagram-SI44F4Z6-DW_vZix7.js} +1 -1
  77. package/dist/worker/console/static/assets/{sizeCapture-X5ZJPWSS-dOnB7UDD.js → sizeCapture-X5ZJPWSS-NBAtkg0C.js} +1 -1
  78. package/dist/worker/console/static/assets/{stateDiagram-OKZ733FA-BXUniaIh.js → stateDiagram-OKZ733FA-Bi1bQxpi.js} +1 -1
  79. package/dist/worker/console/static/assets/stateDiagram-v2-UEYNNEHI-DjWKYjQ1.js +1 -0
  80. package/dist/worker/console/static/assets/{swimlanes-SLNWSIFB-2tA4wTNu.js → swimlanes-SLNWSIFB-8PT_uP_i.js} +2 -2
  81. package/dist/worker/console/static/assets/swimlanesDiagram-ULZ7WXOC-B4cdm7bH.js +8 -0
  82. package/dist/worker/console/static/assets/{timeline-definition-Z64GVDOM-DO0HkJXC.js → timeline-definition-Z64GVDOM-CD0ZZPk0.js} +1 -1
  83. package/dist/worker/console/static/assets/{vennDiagram-T6HMQDX7-DvIOzixv.js → vennDiagram-T6HMQDX7-BbEgWdhK.js} +1 -1
  84. package/dist/worker/console/static/assets/{wardleyDiagram-T6FBY63Y-DXDZ0cTj.js → wardleyDiagram-T6FBY63Y-DCwPKdLl.js} +1 -1
  85. package/dist/worker/console/static/assets/{xychartDiagram-ELKLHX3M-B32Ark0D.js → xychartDiagram-ELKLHX3M-sfC5PcNg.js} +1 -1
  86. package/dist/worker/console/static/index.html +2 -2
  87. package/dist/worker/observe/static/styles.css +9 -0
  88. package/dist/worker/observe/static/views/dag-inspector.js +40 -0
  89. package/dist/worker/observe/static/views/session-timeline.js +135 -0
  90. package/dist/workflows/dag/backend-test-case-coverage-analysis.js +157 -7
  91. package/dist/workflows/dag/backend-test-pytest-collection.js +70 -2
  92. package/dist/workflows/dag/backend-test-result-contract.js +4 -0
  93. package/dist/workflows/dag/backend-test-scenario-param.js +339 -53
  94. package/dist/workflows/dag/backend-test-writer-completeness.js +11 -0
  95. package/dist/workflows/dag/dag-retry-schema.js +138 -0
  96. package/dist/workflows/dag/frontend-implementation-contract.js +174 -31
  97. package/dist/workflows/dag/frontend-review-context.js +12 -1
  98. package/dist/workflows/dag/frontend-shadow-dual-write.js +59 -13
  99. package/dist/workflows/dag/frontend-writer-admission.js +13 -0
  100. package/dist/workflows/dag/init-hybrid.js +8 -3
  101. package/dist/workflows/dag/node-execution.js +277 -3
  102. package/dist/workflows/dag/rerun-feedback.js +135 -3
  103. package/dist/workflows/dag/retry-policy.js +13 -122
  104. package/dist/workflows/dag/types.js +11 -4
  105. package/docs/architecture/runtime-boundaries.md +2 -1
  106. package/docs/templates/backend-test-dag.json +4 -3
  107. package/harness.json +3 -3
  108. package/package.json +4 -3
  109. package/dist/worker/console/static/assets/channel-3TxJgYaH.js +0 -1
  110. package/dist/worker/console/static/assets/classDiagram-JCYQIIEL-BHIkXpp3.js +0 -1
  111. package/dist/worker/console/static/assets/classDiagram-v2-OCEON4UE-BHIkXpp3.js +0 -1
  112. package/dist/worker/console/static/assets/index-BdNx6fj0.css +0 -1
  113. package/dist/worker/console/static/assets/stateDiagram-v2-UEYNNEHI-Cdi6UhLa.js +0 -1
  114. package/dist/worker/console/static/assets/swimlanesDiagram-ULZ7WXOC-D-RJBbb0.js +0 -8
@@ -539,6 +539,18 @@ export const frontendImplementationContractSchema = z
539
539
  message: `unknown verification target ${targetId}`,
540
540
  path: ["requirements"],
541
541
  });
542
+ // Interactions bind to VTs too: an interaction whose behavior
543
+ // verification points at an unmaterialized VT is untraceable —
544
+ // r20 review finding (dangling *-BEHAVIOR references sailed
545
+ // through plan/design/implement to the final review).
546
+ for (const interaction of value.interactions)
547
+ for (const targetId of interaction.verificationTargetIds)
548
+ if (!verificationIds.includes(targetId))
549
+ ctx.addIssue({
550
+ code: "custom",
551
+ message: `interaction "${interaction.name}" references unknown verification target ${targetId}`,
552
+ path: ["interactions"],
553
+ });
542
554
  // Source fidelity ledger binding (AC-005/AC-006): 绑定携带
543
555
  // requirementToFragments(ledger v2)时,每个 requirement 必须携带
544
556
  // 非空 sourceFragmentIds,否则 fail closed(防引用伪造/缺失)。
@@ -570,12 +582,23 @@ export const frontendImplementationContractSchema = z
570
582
  if (state.applicable &&
571
583
  (!state.expectedBehavior ||
572
584
  !state.implementationTargets?.length ||
573
- !state.verificationTargetIds?.length))
585
+ !state.verificationTargetIds?.length)) {
586
+ // Name the state and the exact missing fields: the fixer is a
587
+ // model iterating on finalize receipts — it cannot fix a
588
+ // defect it cannot locate (r18: 39 blind finalize retries).
589
+ const missing = [];
590
+ if (!state.expectedBehavior)
591
+ missing.push("expectedBehavior");
592
+ if (!state.implementationTargets?.length)
593
+ missing.push("implementationTargets");
594
+ if (!state.verificationTargetIds?.length)
595
+ missing.push("verificationTargetIds");
574
596
  ctx.addIssue({
575
597
  code: "custom",
576
- message: "applicable UI state requires behavior, implementation, and verification",
598
+ message: `applicable UI state "${state.name}" is missing: ${missing.join(", ")} — record_state_flow it again with those fields filled`,
577
599
  path: ["uiStates"],
578
600
  });
601
+ }
579
602
  if (!state.applicable && !state.notApplicableReason)
580
603
  ctx.addIssue({
581
604
  code: "custom",
@@ -1124,11 +1147,40 @@ function deriveFrontendVerificationCoverage(value, canonicalBinding) {
1124
1147
  ])];
1125
1148
  const evidenceGap = asRecord(requirement.evidenceGap);
1126
1149
  const hasProof = provenRequirementIds.has(requirementId);
1150
+ // A committed evidenceGap with a blank description means "no gap":
1151
+ // small-output models emit the slot defensively with description
1152
+ // "" and the strict gap schema (description min 1 char) would
1153
+ // fail the whole compile. Strip the incoming slot first and only
1154
+ // re-add it when usable — the requirement either has proof (no
1155
+ // gap needed) or receives the derived blocking gap below.
1156
+ const evidenceGapDescription = asString(evidenceGap?.description).trim();
1157
+ const hasUsableEvidenceGap = evidenceGap !== undefined && evidenceGapDescription !== "";
1158
+ // Source fidelity bindings are deterministic ledger data, not
1159
+ // model-authored content: when the plan requirement carries no
1160
+ // binding (r6 — the contract input block degraded, the model
1161
+ // correctly refused to invent ids), inject the ledger binding
1162
+ // for the requirement id. The strict schema fails a ledger-bound
1163
+ // contract on any requirement without sourceFragmentIds.
1164
+ const declaredSourceFragmentIds = Array.isArray(requirement.sourceFragmentIds)
1165
+ ? requirement.sourceFragmentIds
1166
+ : undefined;
1167
+ const ledgerBoundFragmentIds = Array.isArray(canonicalBinding.requirementToFragments?.[requirementId])
1168
+ ? canonicalBinding.requirementToFragments[requirementId]
1169
+ : undefined;
1170
+ const sourceFragmentIds = declaredSourceFragmentIds && declaredSourceFragmentIds.length > 0
1171
+ ? declaredSourceFragmentIds
1172
+ : ledgerBoundFragmentIds;
1173
+ const { evidenceGap: _incomingGap, sourceFragmentIds: _incomingFragmentIds, ...requirementWithoutGap } = requirement;
1174
+ void _incomingGap;
1175
+ void _incomingFragmentIds;
1127
1176
  return {
1128
- ...requirement,
1177
+ ...requirementWithoutGap,
1178
+ ...(sourceFragmentIds ? { sourceFragmentIds } : {}),
1129
1179
  ...(verificationTargetIds.length > 0 ? { verificationTargetIds } : {}),
1130
1180
  ...(hasProof
1131
- ? (evidenceGap ? { evidenceGap: { ...evidenceGap, blocking: false } } : {})
1181
+ ? (hasUsableEvidenceGap
1182
+ ? { evidenceGap: { ...evidenceGap, blocking: false } }
1183
+ : {})
1132
1184
  : {
1133
1185
  evidenceGap: {
1134
1186
  requirementId,
@@ -1140,10 +1192,16 @@ function deriveFrontendVerificationCoverage(value, canonicalBinding) {
1140
1192
  })
1141
1193
  : record.requirements;
1142
1194
  const modelEvidenceGaps = Array.isArray(record.evidenceGaps)
1143
- ? record.evidenceGaps.map((item) => {
1195
+ ? record.evidenceGaps
1196
+ .map((item) => {
1144
1197
  const gap = asRecord(item);
1145
1198
  if (!gap)
1146
1199
  return item;
1200
+ // Blank-description gaps are meaningless statements that would
1201
+ // fail the strict gap schema; drop them like the embedded
1202
+ // per-requirement empty slots.
1203
+ if (asString(gap.description).trim() === "")
1204
+ return undefined;
1147
1205
  const requirementId = canonicalizeRequirementId(asString(gap.requirementId));
1148
1206
  // Model gaps are advisory. Blocking status is reconstructed below
1149
1207
  // from the source binding and executable verification targets.
@@ -1156,6 +1214,7 @@ function deriveFrontendVerificationCoverage(value, canonicalBinding) {
1156
1214
  blocking: false,
1157
1215
  };
1158
1216
  })
1217
+ .filter((item) => item !== undefined)
1159
1218
  : [];
1160
1219
  const derivedBlockingGaps = canonicalBinding.requirementIds
1161
1220
  .filter((requirementId) => !provenRequirementIds.has(requirementId))
@@ -1790,12 +1849,25 @@ export async function analyzeFrontendImplementationContract(input) {
1790
1849
  if (parsedTargetFiles.some((file) => file.startsWith("/") || file.includes("\\")))
1791
1850
  fail("blocked", "invalid-output: frontend contract target paths must be relative POSIX paths", candidateJsonSha256);
1792
1851
  const parsedStates = asRecord(parsed)?.uiStates;
1793
- if (Array.isArray(parsedStates) && parsedStates.some((item) => {
1794
- const state = asRecord(item);
1795
- return state?.applicable === true &&
1796
- (!asString(state.expectedBehavior) || asStringArray(state.implementationTargets).length === 0 || asStringArray(state.verificationTargetIds).length === 0);
1797
- }))
1798
- fail("retryable-invalid", "invalid-output: applicable UI state requires behavior, implementation, and verification", candidateJsonSha256);
1852
+ if (Array.isArray(parsedStates)) {
1853
+ const incompleteStates = [];
1854
+ for (const item of parsedStates) {
1855
+ const state = asRecord(item);
1856
+ if (!state || state.applicable !== true)
1857
+ continue;
1858
+ const missing = [];
1859
+ if (!asString(state.expectedBehavior))
1860
+ missing.push("expectedBehavior");
1861
+ if (asStringArray(state.implementationTargets).length === 0)
1862
+ missing.push("implementationTargets");
1863
+ if (asStringArray(state.verificationTargetIds).length === 0)
1864
+ missing.push("verificationTargetIds");
1865
+ if (missing.length > 0)
1866
+ incompleteStates.push(`"${asString(state.name)}": missing ${missing.join(", ")}`);
1867
+ }
1868
+ if (incompleteStates.length > 0)
1869
+ fail("retryable-invalid", `invalid-output: applicable UI states incomplete — re-record each with record_state_flow filling the named fields: ${incompleteStates.join("; ")}`, candidateJsonSha256);
1870
+ }
1799
1871
  const parsedMockApi = asRecord(parsed)?.mockApi;
1800
1872
  if (asRecord(parsedMockApi) &&
1801
1873
  typeof asRecord(parsedMockApi)?.strategy === "string" &&
@@ -1820,6 +1892,35 @@ export async function analyzeFrontendImplementationContract(input) {
1820
1892
  const result = frontendImplementationContractSchema.safeParse(candidate);
1821
1893
  if (!result.success)
1822
1894
  fail("retryable-invalid", `invalid-output: ${result.error.issues.map((issue) => `${issue.path.join(".")}: ${issue.message}`).join("; ")}`, candidateJsonSha256);
1895
+ // targets.files is runtime-owned (the skeleton derives it from the task
1896
+ // writeSet, a glob the model must not edit). Narrow it AFTER the full
1897
+ // schema validation — never before, so model-authored unsafe/absolute
1898
+ // declarations are still rejected — to every CONCRETE path the validated
1899
+ // contract references (requirement implementationTargets,
1900
+ // verification-target files, Mock endpoint fixture/consumer paths). The
1901
+ // prewrite containment semantics keep covering everything the contract
1902
+ // references because every narrowed entry already matched the declared
1903
+ // patterns during validation. Satisfies frozen requirements like "deliver
1904
+ // exactly these files" without asking the model to edit a protected
1905
+ // field (r10/r11 design-review findings).
1906
+ const referencedConcretePaths = [
1907
+ ...new Set([
1908
+ ...result.data.requirements.flatMap((requirement) => requirement.implementationTargets),
1909
+ // Mock endpoint paths stay in the set: the zod refine validated
1910
+ // them against the declared targets.files patterns, so the
1911
+ // narrowing must keep covering them. Verification-target files
1912
+ // deliberately do NOT join: they are test artifacts, not
1913
+ // deliverables, and pulling them in would change which gate
1914
+ // fires for an out-of-writeSet VT (AC-3a semantics).
1915
+ ...(result.data.mockApi?.endpoints ?? []).flatMap((endpoint) => [endpoint.fixture, endpoint.consumer].filter((value) => typeof value === "string" && value.length > 0)),
1916
+ ].filter((value) => value.length > 0)),
1917
+ ].sort();
1918
+ if (referencedConcretePaths.length > 0) {
1919
+ result.data.targets = {
1920
+ ...result.data.targets,
1921
+ files: referencedConcretePaths,
1922
+ };
1923
+ }
1823
1924
  const blockingGaps = [
1824
1925
  ...(result.data.evidenceGaps ?? []),
1825
1926
  ...result.data.requirements.flatMap((item) => item.evidenceGap ? [item.evidenceGap] : []),
@@ -2014,7 +2115,7 @@ const VERIFICATION_SYMBOL_MAX_CHARS = 60;
2014
2115
  * loading") and `describe(...)` / `it(...)` forms stay valid — a real symbol
2015
2116
  * may be a function name, a dotted path, or a describe/it title.
2016
2117
  */
2017
- function isSuspiciousVerificationSymbol(symbol) {
2118
+ export function isSuspiciousVerificationSymbol(symbol) {
2018
2119
  const trimmed = symbol.trim();
2019
2120
  if (!trimmed)
2020
2121
  return false;
@@ -2038,6 +2139,66 @@ function assertVerificationSymbolShapes(contract) {
2038
2139
  }
2039
2140
  }
2040
2141
  }
2142
+ /**
2143
+ * Thrown by analyzeFrontendPlanPatchCandidate when the design-policy
2144
+ * pre-checks report findings. Carries the structured findings so callers
2145
+ * (finalize_plan receipt) can synthesize fix suggestions instead of making
2146
+ * the model re-derive them from prose.
2147
+ */
2148
+ export class PlanPolicyPrecheckFailure extends Error {
2149
+ findings;
2150
+ constructor(message, findings) {
2151
+ super(message);
2152
+ this.name = "PlanPolicyPrecheckFailure";
2153
+ this.findings = findings;
2154
+ }
2155
+ }
2156
+ /**
2157
+ * Shared tail of the plan patch validation: analyze the merged contract and
2158
+ * front-load the plan-attributable design-policy checks. Throws the same
2159
+ * errors the node self-check throws (FrontendContractFailure for schema
2160
+ * failures, Error for policy findings) so every caller — the node validator
2161
+ * and the finalize_plan receipt — surfaces identical diagnostics.
2162
+ */
2163
+ export async function analyzeFrontendPlanPatchCandidate(input) {
2164
+ const analysis = await analyzeFrontendImplementationContract(input);
2165
+ // A committed VT with a fabricated symbol is an immutable ledger fact —
2166
+ // the record boundary rejects duplicate ids, so the model cannot overwrite
2167
+ // it and throwing here deadlocks the receipt loop (r19: VT-AC006-BEHAVIOR).
2168
+ // Drop suspicious symbols deterministically instead: the VT stays valid and
2169
+ // the deterministic trace gate verifies file+command (and resolvability)
2170
+ // after verification. Mirrors the shell materialization's drop semantics.
2171
+ for (const target of analysis.canonical.verificationTargets) {
2172
+ if (target.symbol && isSuspiciousVerificationSymbol(target.symbol)) {
2173
+ target.symbol = undefined;
2174
+ }
2175
+ }
2176
+ // Front-load the verification-symbol shape check so fabricated symbols
2177
+ // are fixed by the plan retry ladder in-node instead of failing the
2178
+ // verify trace gate at the end of the run.
2179
+ assertVerificationSymbolShapes(analysis.canonical);
2180
+ // Front-load the plan-attributable design-policy checks so the §5.1
2181
+ // retry ladder can fix these facts in-node with diagnostics instead of
2182
+ // the run terminating at the policy shell. The policy shell remains the
2183
+ // final authority and re-runs the identical checks on committed facts.
2184
+ const policyPreFindings = [
2185
+ ...checkUiDesignCoverage(analysis.canonical),
2186
+ ...checkUiStateAttribution(analysis.canonical),
2187
+ ...checkDependencies(analysis.canonical, await deriveAllowedDependenciesFromRun(input.runDir)),
2188
+ ...checkTargetPaths(analysis.canonical, await deriveWriteSetFromRun(input.runDir)),
2189
+ ];
2190
+ if (policyPreFindings.length > 0) {
2191
+ const message = `frontend plan policy pre-check failed (fix these plan facts, then re-commit and finalize): ${policyPreFindings
2192
+ .map((finding) => `${finding.code}: ${finding.message}`)
2193
+ .join("; ")}`;
2194
+ throw new PlanPolicyPrecheckFailure(message, policyPreFindings.map((finding) => ({
2195
+ code: finding.code,
2196
+ message: finding.message,
2197
+ ...(finding.path ? { path: finding.path } : {}),
2198
+ })));
2199
+ }
2200
+ return analysis;
2201
+ }
2041
2202
  export async function validateFrontendPlanPatchNodeOutput(input) {
2042
2203
  const nodeId = input.nodeId ?? "frontend-plan-pi";
2043
2204
  const attempt = input.attempt ?? 1;
@@ -2089,29 +2250,11 @@ export async function validateFrontendPlanPatchNodeOutput(input) {
2089
2250
  runDir: input.runDir,
2090
2251
  contract: merged,
2091
2252
  });
2092
- const analysis = await analyzeFrontendImplementationContract({
2253
+ const analysis = await analyzeFrontendPlanPatchCandidate({
2093
2254
  runDir: input.runDir,
2094
2255
  rawContractText: serializeDeterministicJson(merged),
2095
2256
  sourceBinding: input.sourceBinding,
2096
2257
  });
2097
- // Front-load the verification-symbol shape check so fabricated symbols
2098
- // are fixed by the plan retry ladder in-node instead of failing the
2099
- // verify trace gate at the end of the run.
2100
- assertVerificationSymbolShapes(analysis.canonical);
2101
- // Front-load the plan-attributable design-policy checks so the §5.1
2102
- // retry ladder can fix these facts in-node with diagnostics instead of
2103
- // the run terminating at the policy shell. The policy shell remains the
2104
- // final authority and re-runs the identical checks on committed facts.
2105
- const policyPreFindings = [
2106
- ...checkUiDesignCoverage(analysis.canonical),
2107
- ...checkUiStateAttribution(analysis.canonical),
2108
- ...checkDependencies(analysis.canonical, await deriveAllowedDependenciesFromRun(input.runDir)),
2109
- ...checkTargetPaths(analysis.canonical, await deriveWriteSetFromRun(input.runDir)),
2110
- ];
2111
- if (policyPreFindings.length > 0)
2112
- throw new Error(`frontend plan policy pre-check failed (fix these plan facts, then re-commit and finalize): ${policyPreFindings
2113
- .map((finding) => `${finding.code}: ${finding.message}`)
2114
- .join("; ")}`);
2115
2258
  const normalizedArtifact = await writeDeterministicJsonArtifact(input.runDir, path.posix.join(candidateDir, `attempt-${attempt}.normalized.json`), analysis.canonical);
2116
2259
  await writeReport({
2117
2260
  classification: "accepted-normalized",
@@ -287,9 +287,20 @@ export async function runFrontendReviewContextGate(input) {
287
287
  if (!parsedContract.success) {
288
288
  throw new FrontendReviewContextFailure("review-context-invalid-contract", "frontend review context invalid implementation contract");
289
289
  }
290
+ // Authorized diff surface = deliverable files + every verification-target
291
+ // file the contract references. The writer legitimately writes VT test
292
+ // artifacts (behavior verification), and the reviewer must audit their
293
+ // diffs; scoping this to targets.files alone failed the baseline-overlap
294
+ // check on app.test.js (r19 extreme smoke).
295
+ const authorizedChangedPaths = [
296
+ ...new Set([
297
+ ...parsedContract.data.targets.files,
298
+ ...parsedContract.data.verificationTargets.map((target) => target.file),
299
+ ]),
300
+ ];
290
301
  const diff = await runFrontendWorktreeDiffGate({
291
302
  ...input,
292
- authorizedChangedPaths: parsedContract.data.targets.files,
303
+ authorizedChangedPaths,
293
304
  });
294
305
  const verificationTrace = await readRequiredJson(input.runDir, "contracts/frontend-verification-trace.json");
295
306
  const repairAssessment = await readOptionalRepairAssessment(input.runDir);
@@ -260,7 +260,12 @@ export function restorePlanPatchFromCommittedFacts(records) {
260
260
  if (isRecord(fact.patch))
261
261
  return fact.patch;
262
262
  }
263
- return undefined;
263
+ // No finalize-published snapshot: the receipt fix loop can end an attempt
264
+ // before any finalize_plan succeeds while dozens of record_* facts are
265
+ // already committed. Assemble the patch from those record facts instead of
266
+ // declaring the ledger missing (r19: "ledger missing" discarded a ledger
267
+ // with 92 committed facts).
268
+ return assemblePlanPatchFromCommittedFacts(records);
264
269
  }
265
270
  /**
266
271
  * A+B (AC-004): reverse of `planLedgerFactsFromPatch` for the incremental
@@ -280,12 +285,33 @@ export function assemblePlanPatchFromCommittedFacts(records, contractRequirement
280
285
  PLAN_LEDGER_FACT_KINDS.includes(fact.kind));
281
286
  if (facts.length === 0)
282
287
  return undefined;
283
- for (const fact of facts) {
284
- if (isRecord(fact.patch))
285
- return fact.patch;
286
- }
288
+ // Singleton record facts: the LAST committed fact wins. The model corrects
289
+ // a rejected plan by re-recording the fact (r18: 39 finalize retries never
290
+ // converged because the first state-flow fact kept shadowing the
291
+ // corrections). Requirements / verification targets / evidence gaps stay
292
+ // additive (aggregated below); component choices accumulate; singletons
293
+ // supersede.
294
+ const lastByKind = (kind) => {
295
+ let found;
296
+ for (const fact of facts) {
297
+ if (fact.kind === kind)
298
+ found = fact;
299
+ }
300
+ return found;
301
+ };
302
+ // Patch snapshots (published by earlier finalize calls) carry no routes;
303
+ // exclude them so a snapshot cannot shadow the model's route selection.
304
+ const routeSurface = (() => {
305
+ let found;
306
+ for (const fact of facts) {
307
+ if (fact.kind !== "target-surface" || isRecord(fact.patch))
308
+ continue;
309
+ found = fact;
310
+ }
311
+ return found;
312
+ })();
287
313
  const patch = {};
288
- const targetSurface = facts.find((fact) => fact.kind === "target-surface");
314
+ const targetSurface = routeSurface;
289
315
  if (targetSurface) {
290
316
  const routes = canonicalStringSet(asStringArray(targetSurface.routes));
291
317
  if (routes.length > 0)
@@ -295,14 +321,34 @@ export function assemblePlanPatchFromCommittedFacts(records, contractRequirement
295
321
  const collectedChoices = componentChoices.flatMap((fact) => Array.isArray(fact.uiComponentChoices)
296
322
  ? fact.uiComponentChoices.filter(isRecord)
297
323
  : []);
324
+ // Later corrections supersede earlier submissions per purpose: without
325
+ // this, a re-recorded choice (fixing a wrong specReference) would appear
326
+ // twice and the reviewer would still audit the stale entry (r19).
327
+ const choicesByPurpose = new Map();
328
+ const dedupedChoices = [];
329
+ for (const choice of collectedChoices) {
330
+ const purpose = asString(choice.purpose);
331
+ if (!purpose) {
332
+ dedupedChoices.push(choice);
333
+ continue;
334
+ }
335
+ if (choicesByPurpose.has(purpose)) {
336
+ const at = dedupedChoices.findIndex((existing) => asString(existing.purpose) === purpose);
337
+ dedupedChoices[at] = choice;
338
+ }
339
+ else {
340
+ choicesByPurpose.set(purpose, choice);
341
+ dedupedChoices.push(choice);
342
+ }
343
+ }
298
344
  const stylingStrategy = componentChoices
299
345
  .map((fact) => asString(fact.stylingStrategy))
300
346
  .find((value) => value.length > 0);
301
- if (collectedChoices.length > 0)
302
- patch.uiComponentChoices = collectedChoices;
347
+ if (dedupedChoices.length > 0)
348
+ patch.uiComponentChoices = dedupedChoices;
303
349
  if (stylingStrategy)
304
350
  patch.stylingStrategy = stylingStrategy;
305
- const stateFlow = facts.find((fact) => fact.kind === "state-flow");
351
+ const stateFlow = lastByKind("state-flow");
306
352
  if (stateFlow) {
307
353
  if (Array.isArray(stateFlow.uiStates)) {
308
354
  patch.uiStates = stateFlow.uiStates.filter(isRecord);
@@ -311,17 +357,17 @@ export function assemblePlanPatchFromCommittedFacts(records, contractRequirement
311
357
  patch.interactions = stateFlow.interactions.filter(isRecord);
312
358
  }
313
359
  }
314
- const mockApi = facts.find((fact) => fact.kind === "mock-api");
360
+ const mockApi = lastByKind("mock-api");
315
361
  if (mockApi && isRecord(mockApi.mockApi)) {
316
362
  patch.mockApi = mockApi.mockApi;
317
363
  }
318
- const deviation = facts.find((fact) => fact.kind === "design-deviation");
364
+ const deviation = lastByKind("design-deviation");
319
365
  if (deviation) {
320
366
  const conflicts = canonicalStringSet(asStringArray(deviation.conflicts));
321
367
  if (conflicts.length > 0)
322
368
  patch.designEvidence = { conflicts };
323
369
  }
324
- const dependency = facts.find((fact) => fact.kind === "dependency");
370
+ const dependency = lastByKind("dependency");
325
371
  const dependencyPolicy = asString(dependency?.policy);
326
372
  if (dependencyPolicy)
327
373
  patch.dependencyPolicy = dependencyPolicy;
@@ -607,7 +653,7 @@ export function readCompleteScoutTargetSurface(records) {
607
653
  if (!namedPaths.some((candidate) => freshPaths.has(candidate)))
608
654
  return {
609
655
  ok: false,
610
- reason: "frontend scout target surface lacks fresh runtime evidence for a named target path",
656
+ reason: "frontend scout target surface lacks fresh runtime evidence for a named target path; record_target_surface enriches every declared entrypoint/implementation/test path against the real workspace, so at least one named path must exist on disk (an existing directory counts; a file to be created later cannot count)",
611
657
  };
612
658
  return { ok: true, surface };
613
659
  }
@@ -78,6 +78,19 @@ export const frontendWriterAdmissionResultV1Schema = z
78
78
  path: z.string().optional(),
79
79
  })
80
80
  .strict()),
81
+ // r12 policy change: a request_design_changes verdict no longer blocks
82
+ // writer admission. The verdict and its findings ride along as
83
+ // advisory context for the implement node; blocking is owned by the
84
+ // deterministic gates (prewrite, write guard) and the final code
85
+ // review.
86
+ designReviewAdvisory: z
87
+ .object({
88
+ verdict: z.enum(["approve_design", "request_design_changes"]),
89
+ restartPhase: z.string().min(1).optional(),
90
+ findingCount: z.number().int().nonnegative(),
91
+ })
92
+ .strict()
93
+ .optional(),
81
94
  failureSource: frontendAdmissionFailureSourceSchema.optional(),
82
95
  })
83
96
  .strict();
@@ -2564,9 +2564,10 @@ async function resolveFrontendOpenspecGateConfig(sources) {
2564
2564
  const frontendComponentConformanceInstruction = [
2565
2565
  "## Component Selection conformance (uiComponentChoices; hard rule)",
2566
2566
  "每个 UI 用途必须在契约的 uiComponentChoices[] 中声明组件选型:{ purpose, component, decision, specReference, rationale }。",
2567
+ "purpose is the stable coverage key:它应精确匹配 interaction.name 或 uiState.name;职责语义由对应 interaction.expectedBehavior / uiState.expectedBehavior 与 rationale 表达。不得仅因 purpose 与 interaction 或 component 标识符相同而判缺陷。",
2567
2568
  "- decision=specified:前端规范(候选组件/主题桶 + 任务源显式引用)已定义该用途组件 → 必须使用该组件,并给精确 specReference { path, section, line }(path 必须是 openspec/ai_workspace 受支持规范路径)。",
2568
2569
  "- decision=reuse-existing:仅当该组件/惯例**确实已存在于仓库当前代码**(如复用现有 ActiveRunBadge 的 oc- class 惯例)→ specReference 可为 null,rationale 必须指明复用的具体现有组件/文件与依据。",
2569
- "- decision=new:任务源/PRD 要求**新增**该组件(仓库当前不存在该组件文件)→ decision 必须为 new,不得标 reuse-existing;调用 record_component_choice 时传 sourceRequirementIds(关联的 frozen requirement ID),runtime 会从 source-fidelity ledger 物化精确的任务源 PRD { path, section, line }。不要读取 PRD 或手填/猜测 specReference;rationale 说明新增纯展示组件、复用既有 CSS 命名与主题变量约定。",
2570
+ "- decision=new:任务源/PRD 要求**新增**该组件(仓库当前不存在该组件文件)→ decision 必须为 new,不得标 reuse-existing;调用 record_component_choice 时传 sourceRequirementIds(关联的 frozen requirement ID)与 plan checklist 列出的 sourceFragmentId,runtime 校验其隶属关系并物化精确的任务源 PRD { path, section, line }。不要读取 PRD 或手填/猜测 specReference;rationale 说明新增纯展示组件、复用既有 CSS 命名与主题变量约定。",
2570
2571
  "不得静默替换规范组件或自创组件而无偏差声明;spec 已定义该用途组件时不得改选其它组件。",
2571
2572
  "prewrite gate 确定性交叉校验:仅 decision=specified 的 specReference 必须是候选 OpenSpec 路径、在 typed decision ledger 中声明且有成功 read 事件;decision=new 的 PRD 引用走任务源可追溯性审查,不得按 OpenSpec 候选拒绝。候选桶非空且契约有 UI 可见工作而 uiComponentChoices 缺失/空 → component-choices-missing。",
2572
2573
  ].join("\n");
@@ -3144,6 +3145,7 @@ async function buildFrontendHybridDagFromTask(sources) {
3144
3145
  skills: FRONTEND_IMPLEMENTATION_SKILLS,
3145
3146
  outputContract: "Typed requirement facts plus a concise Markdown contract. Submit through the incremental typed tools record_requirement / record_constraint / record_evidence_expectation / record_handoff_intent / record_open_question / record_split_proposal / record_openspec_selection, then call finalize_contract exactly once. Requirements use stable REQ/BR/AC identifiers with source spans and a disposition (explicit | repository-resolvable | assumption | blocking); each requirement registers evidence expectations across static/behavior/Mock/real-integration (required | optional | not-applicable), and UI-visible or interactive requirements register a non-blocking frontend-test handoff intent. End finalize_contract with a single contract disposition of ready | ready-with-assumptions | blocked. Cover scope, non-goals, acceptance criteria, UI states, target runtime environment, risks, and verification expectations; do not fix target files, components, or implementation methods as requirements. When OpenSpec candidates exist, classify only the ones you actually use: call record_openspec_selection once per required/relevant path; never enumerate irrelevant candidates (unmentioned defaults to irrelevant) and never emit a fenced selection JSON. No file writes.",
3146
3147
  subtask_prompt: [
3148
+ "OUTPUT BUDGET DISCIPLINE (hard requirement, extreme-environment safe): the provider output window is small — NEVER attempt to emit the whole contract in one response; a single large JSON dump will be truncated and rejected. Incremental submission through the typed tools is the ONLY supported output mode. Start submitting with the FIRST tool call: after each read, call record_requirement for the requirements you have already confirmed, one or a few per call. Every tool-call round MUST make progress by submitting at least one record_* fact. Do not re-read the same source file that is already materialized in this session; read each file at most once.",
3147
3149
  "Read task source and produce a concise frontend implementation contract as typed requirement facts plus narrative Markdown.",
3148
3150
  "Assign each requirement the SAME id as the ledger canonical requirement it covers (sourceBinding.requirementIds, e.g. AC-001) — do NOT invent new REQ/BR prefixed ids for canonical requirements: the compiled contract must match the ledger canonical requirement ids exactly or schema validation rejects it (unknown requirement id). Requirements use the canonical id with a source span (task-source section or repository file:line). Label each requirement's disposition as explicit | repository-resolvable | assumption | blocking; a blocking requirement must name its owner (human-decision or external-state) and evidence refs.",
3149
3151
  "Source fidelity ledger: when the DAG sourceBinding carries a requirement→fragment mapping (requirementToFragments, e.g. REQ-SRC-* ids from the managed ledger), each record_requirement MUST declare the fragments that requirement is bound to: set sourceFragmentIds to the mapped fragment ids (the authoritative provenance evidence the design policy verifies). sourceRefs (fragment→path display refs) are optional — declare them only when you have the exact path from the materialized source; otherwise omit them rather than inventing paths. Declare exactly what the ledger binds — do not invent ids, do not omit them, and do not re-derive them from prose. A requirement that the ledger binds but the contract omits (or fabricates) fails writer admission.",
@@ -3208,6 +3210,7 @@ async function buildFrontendHybridDagFromTask(sources) {
3208
3210
  ...(requiresOpenspecClassification ? ["When a component choice uses an OpenSpec selection, cite that selection; otherwise do not classify unrelated candidates."] : []),
3209
3211
  "Call finalize_plan exactly once after the necessary typed facts. Return no Markdown narrative.",
3210
3212
  "TOOL-ONLY PLAN: Do not read Contract/Scout stdout, task sources, or Scout-confirmed target files. Contract and Scout already own evidence discovery; use the injected upstream facts, record a genuine evidence gap when those facts are insufficient, and start committing record_* facts immediately. For decision=new, pass sourceRequirementIds to record_component_choice; runtime derives the exact PRD citation from the frozen ledger.",
3213
+ "Output budget protocol (hard, max output <=16K per turn): never enumerate-reason the whole requirement list before your first record_* call — that reasoning burns the entire output budget and the attempt dies with zero committed facts. Process requirements in order: think about ONE requirement briefly, immediately emit its record calls (up to 5 per message), then move to the next. If your budget runs low, stop recording and call finalize_plan with what is committed — the retry ladder continues the remainder in a fresh session.",
3211
3214
  fixedVerificationContext,
3212
3215
  scopedOpenspecContext,
3213
3216
  mockContextBlock,
@@ -3299,6 +3302,7 @@ async function buildFrontendHybridDagFromTask(sources) {
3299
3302
  "Request design changes when the Mock strategy is MOCK_STRATEGY: blocked, missing, unsupported by repository evidence, inconsistent with the API contract, outside authorized paths/dependencies, unable to prove production-default-off behavior with the fixed production/default-real-path static check, or missing deterministic behavior verification for a declared behavior target or selected Mock strategy. Mock strategies require Mock-backed evidence. A static-only contract is allowed only when every verification target is static and maps to a declared static entrypoint. not-needed otherwise requires applicable real/no-remote behavior evidence unless auto mode explicitly skipped Mock because no project Mock capability exists; in that case the plan must preserve the real request path and record the Real Integration Gap.",
3300
3303
  "Also request design changes for missing applicable UI states, unsupported dependency additions, design-system drift without reason, weak interaction coverage, broad scope, inline fake data, schema drift, or missing deterministic verification commands.",
3301
3304
  "Component selection conformance is a hard blocking condition: request_design_changes when the frontend spec (component/theme/rule.components bucket) already defines a component for a purpose but the plan selects another or self-invents one without a declared deviation; when uiComponentChoices is missing/empty for UI-visible work while the frozen component/theme bucket is non-empty; when a decision=specified specReference.path is missing a ledger OpenSpec reference or successful read event; or when a decision=new component lacks a traceable task-source/PRD specReference. A PRD reference for decision=new is not an OpenSpec citation and must not be rejected merely for lacking an OpenSpec read event.",
3305
+ "For uiComponentChoices, purpose is the stable coverage key and must match interaction.name or uiState.name. Responsibility is expressed by the matched expectedBehavior plus rationale; you must not reject it merely for matching an interaction or component identifier.",
3302
3306
  "You must NOT make authoritative assertions about the execution result of frozen verification commands (typecheck/test/build/lint/etc.). Predicting that a command will necessarily pass or fail, or declaring an acceptance criterion unreachable on that basis, is out of your authority: command results are deterministically established by frontend-verify-shell. Any concern about verification feasibility must be recorded only as a non-blocking verification concern in findings (severity must not be Critical, and it must never be the sole fatal basis for request_design_changes). Only semantic design defects (component selection, state flow, interaction contract, or conflicts with the specification) may be Critical. A pure command-will-fail prediction must not be classified as contract-requirement-gap.",
3303
3307
  "Read-only: do not modify repository files.",
3304
3308
  "LARGE-FILE AUDIT (avoid full reads): style/theme audit files can be large (e.g. styles.css is often hundreds of KB). Prefer grep to locate the exact rules/variables you must verify (e.g. grep the oc- class, is-* modifier, or --oc- theme variables with their line numbers), then read only the narrow line range when surrounding context is needed. Do not read a large style/test file in full — a single full read can exhaust the read budget and fail the attempt.",
@@ -3464,6 +3468,7 @@ async function buildFrontendHybridDagFromTask(sources) {
3464
3468
  "Treat lint status exactly as passed | baseline-debt | failed | unavailable. baseline-debt may continue only with intact evidence and zero diagnostics on writer-changed files; report the tolerated debt count and never rewrite it as lint passed. Typecheck, build, and test still require successful final exits.",
3465
3469
  "Flag .skip/.only, deleted or weakened tests, unauthorized config changes, Mock-only evidence claimed as real integration, and Browser/visual claims (always not-run in this workflow).",
3466
3470
  "The contract embedded in frontend-review-context.json is the effective plan materialized by frontend-design-policy-shell. Do not re-open task sources, OpenSpec, AI workspace, design-review, writer summary, or verification node prose. Inspect only the canonical review context, its bound diff, and diff-referenced files when semantic review requires source code.",
3471
+ "For uiComponentChoices, purpose is the stable coverage key and must match interaction.name or uiState.name. Responsibility is expressed by the matched expectedBehavior plus rationale; you must not reject it merely for matching an interaction or component identifier.",
3467
3472
  "Treat a commented-out real request, default-enabled Mock, production entrypoint importing test mocks, API/fixture contract drift, unauthorized Mock dependency/path, or missing behavior evidence for the selected strategy as at least Important. Mock strategies require Mock-backed evidence. not-needed requires applicable real/no-remote behavior evidence unless auto mode explicitly skipped Mock because no project Mock capability exists; in that case verify that the real request remains the default and the Real Integration Gap is preserved.",
3468
3473
  "Inspect the frontend-verify-shell evidence in the review context directly, including the production/default-real-path static check, and require Mock activation to be off for that check.",
3469
3474
  "Distinguish Mock-backed evidence from real API integration evidence and preserve the Real Integration Gap when the backend was not exercised.",
@@ -4709,7 +4714,7 @@ async function buildBackendTestHybridDag(sources) {
4709
4714
  "Coverage priority is strict inside the declared scope: P0 product requirements/task hard constraints always remain in scope; P1 exhaustively supplements documented operations, fields, business rules, statuses and errors only for Affected Operations; P2 adds bounded protocol robustness only when it is relevant to the change and does not invent product behavior. Coverage percentages describe the declared affected scope, never whole-API completeness unless every operation is explicitly listed. Conflicts or undefined expectations must stay visible as GAP/CONFLICT with precise source pointers, never guessed.",
4710
4715
  "For uniqueness/lifecycle rules cover absent, active-existing, deleted-existing, create-delete-recreate, restore-then-recreate and documented scope/case-normalization states. For every enum cover every valid value plus bounded invalid equivalence classes (unknown, case variant, whitespace, empty, null/missing and wrong types as applicable). For every length/number rule cover min-1, min, nominal, max and max+1. For format rules cover each allowed class separately plus a valid mixed value, and representative forbidden classes including uppercase, internal/leading/trailing whitespace, tab/newline, unsupported punctuation, slash, emoji or control characters when the source contract supports that expectation.",
4711
4716
  "Mandatory module index: include a `## Module Index` table in the plan artifact that lists every planned module as a canonical relative link of the exact form `[label](./<stem>.md)` plus a `testcase/md/<stem>.md` path cell, so a downstream deterministic manifest can parse the module list. Group by stable business resource/domain, not by CRUD operation: one resource's list/detail/create/update/delete cases belong in one module such as `resource_notes`; split only when a single module would exceed the per-child 16K output protocol, keep the total module count at the smallest safe value, and never exceed 8 modules. Name each module file with a stable lowercase business stem such as `health` or `resource_notes`. Pure hexadecimal/hash-like opaque stems such as `a401606` or `deadbeef` are forbidden. Do not use priority-only stems `p0`, `p1` or `p2`; Priority belongs only in the Coverage Matrix and never defines module files. Do not use Case-ID-like module filenames such as `BE-HEALTH.md` or `BE-NOTES.md`. The relative link target MUST equal the on-disk filename stem the sharded writer will create. For every automatable case, `自动化映射` must name exactly `testcase/test_<module>.py`, where <module> is that Markdown filename without `.md`, lowercased, with non-alphanumeric characters replaced by underscores. Example: `testcase/md/health.md` → `testcase/test_health.py`; `testcase/md/resource_notes.md` → `testcase/test_resource_notes.py`. Never invent a different pytest path in Markdown than the module stem implies.",
4712
- "Scenario Partitions (query/filter axes): for every affected GET/list operation, declare one row per enum or classification axis used for filtering (query/path parameters such as type/status/category). Add a mandatory machine-readable `## Scenario Partitions` section after the Coverage Matrix using exactly `| Partition ID | Operation | Axis | Domain | Required Slots | Expected by Slot | Bind Rule |` with the separator row. Partition ID is a stable `SP-<OPERATION>-<AXIS>` token; Domain must copy the legal values verbatim from the bound OpenAPI enum or requirement sentence (never guess), using bare semicolon-separated identifier values inside the single table cell (for example `ACTIVE; ARCHIVED`, with no Markdown backticks or prose); Required Slots writes `each-value` plus `omitted` only when the parameter is optional; Expected by Slot states the documented expectation per slot kind (`domain-value`, `default-behavior`, `empty-result`/`excluded-result` when documented, or `GAP` when the source does not document the complement expectation — never invent 空列表/400). POST/PUT body field-validation enums stay in the Coverage Matrix as `TP-<FIELD>-ENUM-*` and MUST NOT get a Scenario Partition row. Do not create partitions for axes without a documented legal-value domain. Only GET/list query or path parameters whose bound source documents a finite enum or classification set may become a Scenario Partition. Do not create partitions for free-form strings, primary keys, required-or-optional-only parameters, or boundary/format-only axes. If an axis has no finite legal-value domain, do not declare a Partition row and do not invent NOT-IN-SET cases. Cross-axis combinations stay as ONE nominal Case; never declare a cross-axis cartesian partition.",
4717
+ "Scenario Partitions (query/filter axes): for every affected GET/list operation, declare one row per enum or classification axis used for filtering (query/path parameters such as type/status/category). Add a mandatory machine-readable `## Scenario Partitions` section after the Coverage Matrix using exactly `| Partition ID | Operation | Axis | Domain | Required Slots | Expected by Slot | Bind Rule |` with the separator row. Partition ID is a stable `SP-<OPERATION>-<AXIS>` token; Domain must copy the legal values verbatim from the bound OpenAPI enum or requirement sentence (never guess), using bare semicolon-separated identifier values inside the single table cell (for example `ACTIVE; ARCHIVED`, with no Markdown backticks or prose); Required Slots must contain `each-value` and exactly one `not-in-set`, plus `omitted` only when the parameter is optional; Expected by Slot states the documented expectation per slot kind (`domain-value`, `default-behavior`, `empty-result`/`excluded-result` when documented, or `GAP` when the source does not document the complement expectation — never invent 空列表/400). POST/PUT body field-validation enums stay in the Coverage Matrix as `TP-<FIELD>-ENUM-*` and MUST NOT get a Scenario Partition row. Do not create partitions for axes without a documented legal-value domain. Only GET/list query or path parameters whose bound source documents a finite enum or classification set may become a Scenario Partition. Do not create partitions for free-form strings, primary keys, required-or-optional-only parameters, or boundary/format-only axes. If an axis has no finite legal-value domain, do not declare a Partition row and do not invent NOT-IN-SET cases. Cross-axis combinations stay as ONE nominal Case; never declare a cross-axis cartesian partition.",
4713
4718
  "Before finalizing README, calculate the predicted collected-item count as `sum(max(1, number of variant Test Points in each Case))`. If the task declares an item budget, the prediction must not exceed it. Reduce excess only by removing duplicate execution and converting same-request checkpoints to assertions; never drop required rules, boundaries, enums, operation-specific inputs, or business states. Record the prediction in README. Use only environment-supported fixtures/targets/isolation, record evidence gaps in Chinese, and do not emit JSON, pytest, or execute commands.",
4714
4719
  ...(sharedSetupPrompt ? [sharedSetupPrompt] : []),
4715
4720
  intake.boundedSourceContext,
@@ -5004,7 +5009,7 @@ async function buildBackendTestHybridDag(sources) {
5004
5009
  writerOutcomePolicy: { type: "implementation-outcome-v1", requireChangedFiles: true },
5005
5010
  outputContract: "First non-empty line is IMPLEMENTATION_OUTCOME: changed|blocked, followed by a concise repair summary. This node runs only for REPAIRABLE initial facts, so already-satisfied is invalid and a successful outcome requires a non-empty bounded diff. Modify only generated pytest scripts/helpers/factories and preserve every Markdown Case, Test Point, primary symbol and assertion meaning.",
5006
5011
  subtask_prompt: [
5007
- "Repair the generated backend pytest asset as one bounded program using the direct upstream collection assessment. This is the only repair attempt and happens before any business test body execution. Treat any upstream line such as `Repair paths: testcase/test_x.py` as complete authoritative repairPaths evidence. Directly read and edit that testcase path; do not search for separate root-level `contracts/**`, guess a DAG run directory, or require another report artifact. If the read tool successfully returns the testcase file, the path exists—continue the bounded repair and never later claim that file is absent.",
5012
+ "Repair the generated backend pytest asset as one bounded program using the direct upstream collection assessment. This is the only repair attempt and happens before any business test body execution. The direct upstream JSON includes authoritative `repairPaths` and bounded `repairFindings`; treat both as the complete mandatory checklist without searching for a run directory or report file. Treat any upstream line such as `Repair paths: testcase/test_x.py` as equivalent authoritative repairPaths evidence. Directly read and edit that testcase path; do not search for separate root-level `contracts/**`, guess a DAG run directory, or require another report artifact. If the read tool successfully returns the testcase file, the path exists—continue the bounded repair and never later claim that file is absent.",
5008
5013
  "Initial status REPAIRABLE means at least one listed finding remains: `already-satisfied` is forbidden, and you must produce a non-empty bounded diff on repairPaths before returning `IMPLEMENTATION_OUTCOME: changed`. Fix only readiness-proven generated testcase-local defects on initial facts repairPaths: create exact safe missing mapped test_*.py paths, repair syntax/import/symbol/decorator/parameterization, close generated fixture dependencies/plugin registration, and repair initial Markdown-to-pytest correspondence findings. Use this deterministic repair map instead of reading analyzer implementation: findings about `Case-ID`, `Assertion-Test-Points`, or `Cross-Cutting-Test-Points` are fixed by editing the declared primary symbol docstring metadata lines; variant binding findings are fixed in the literal direct `pytest.param(..., id=\"TP-...\")` row; primary-symbol cardinality/name findings are fixed in the function name or duplicate primary symbols; script mismatch is fixed only on the authoritative assessment repairPaths; payload findings are fixed in request payload construction. Do not read controller `src/**` or inspect JS/TS analyzer code. Do not search for `testcase/**/README.md`. Never invent a business pytest symbol for evidence-only Markdown Cases that declare `脚本/primary symbol=无` with empty variants. For fixture defects inspect both provider and importer listed by repairPaths; fix ScopeMismatch by aligning fixture scopes or inlining request-scoped values so module fixtures never depend on function fixtures; when a shared fixture depends on sibling fixtures, register the whole provider module through an exact pytest_plugins declaration rather than importing only the outer fixture. Do not create unrelated pytest scripts.",
5009
5014
  "This is the single pytest incremental synchronization round. The `Findings` in `reports/backend-test-pytest-collection-initial.md` are the mandatory repair checklist: resolve every repairable listed finding on every authoritative `Repair paths` file before considering any other advisory evidence, and never substitute an unrelated scenario-param cleanup for a listed correspondence/collection defect. For every assessment-listed path, compare the effective Markdown Case/Test Points/test data and its `Payload Contract`/`Payload Required Paths`/`Payload Allowed Paths`/`Payload Enum` labels with the generated module. Incrementally add or repair only missing symbols, params, assertions and payload builders. Repair every assessment-listed missing nested path, unexpected key and enum mismatch; preserve exact DTO keys, nested shapes, enum/boundary literals, operation transport and business preconditions; remove guessed replacement keys only when the effective Markdown proves the exact contract. Keep path/query/header identifiers and scenario-control metadata separate from DTO patches and JSON bodies; an `id` used for a path target must be passed to the request path/helper, never inserted into a body patch unless `id` is explicitly listed in Payload Allowed Paths. Flatten every variant into a literal direct `pytest.param(..., id=\"TP-...\")` row; replace `_post_case`/`_put_case` or other parameter-row factories because correspondence and scenario readiness require the actual row values and IDs to be statically visible. Also repair helper call sites to match their defined return signatures; do not tuple-unpack a helper that returns one scalar value.",
5010
5015
  "Preserve final testcase/md/** semantics, every Case ID, Rule/Test Point binding, primary symbol, parameter ID, expected status/body/schema assertion, HTTP logging, redaction and truncation behavior.",