@tea-agent/loop-agent 0.44.0-next.10 → 0.44.0-next.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/CHANGELOG.md +11 -0
  2. package/dist/build-stamp.json +3 -3
  3. package/dist/executors/dag-pi/sessions/index.js +6 -0
  4. package/dist/executors/dag-pi/sessions/plan-batches.js +280 -0
  5. package/dist/executors/dag-pi/sessions/plan-prompts.js +367 -0
  6. package/dist/executors/dag-pi/sessions/scout-parallel.js +197 -0
  7. package/dist/executors/dag-pi/sessions/segmented-plan.js +1184 -0
  8. package/dist/executors/dag-pi/sessions/writer-evidence.js +60 -0
  9. package/dist/executors/dag-pi-executor.js +12 -2075
  10. package/dist/executors/shell-executor.js +39 -11
  11. package/dist/worker/console/routes.js +5 -0
  12. package/dist/worker/console/static/app-icon.svg +39 -0
  13. package/dist/worker/console/static/index.html +3 -0
  14. package/dist/worker/console/static/manifest.webmanifest +19 -0
  15. package/dist/worker/observe/static/console-theme.js +38 -0
  16. package/dist/workflows/dag/checkpoint.js +686 -0
  17. package/dist/workflows/dag/frontend-repair.js +24 -0
  18. package/dist/workflows/dag/frontend-review-scopes.js +10 -2
  19. package/dist/workflows/dag/frontend-test-execution-evidence.js +163 -67
  20. package/dist/workflows/dag/frontend-test-framework-adapters.js +309 -0
  21. package/dist/workflows/dag/hybrid/sources.js +1812 -0
  22. package/dist/workflows/dag/hybrid/templates/backend-test.js +1807 -0
  23. package/dist/workflows/dag/hybrid/templates/frontend-test.js +702 -0
  24. package/dist/workflows/dag/hybrid/templates/frontend.js +1529 -0
  25. package/dist/workflows/dag/hybrid/templates/index.js +8 -0
  26. package/dist/workflows/dag/hybrid/templates/kg-bootstrap.js +449 -0
  27. package/dist/workflows/dag/hybrid/templates/knowledge-sync.js +495 -0
  28. package/dist/workflows/dag/hybrid/templates/shared.js +101 -0
  29. package/dist/workflows/dag/hybrid/templates/standard.js +265 -0
  30. package/dist/workflows/dag/hybrid/types.js +32 -0
  31. package/dist/workflows/dag/init-hybrid.js +42 -7162
  32. package/dist/workflows/dag/runner.js +11 -1178
  33. package/dist/workflows/dag/terminal-status.js +502 -0
  34. package/docs/architecture/dag-execution.md +4 -4
  35. package/package.json +1 -1
@@ -0,0 +1,1807 @@
1
+ import { createHash } from "node:crypto";
2
+ import { deflateRawSync } from "node:zlib";
3
+ import { assertValidDagSpec } from "../../validate.js";
4
+ import { DEFAULT_DAG_EXECUTOR_MODELS, DEFAULT_DAG_OUTPUT_LANGUAGE, parseDagSpec } from "../../types.js";
5
+ import { BACKEND_TEST_MARKDOWN_BINDING_RETRY_POLICY, BACKEND_TEST_MD_PLAN_RETRY_POLICY, BACKEND_TEST_WRITER_COMPLETENESS_RETRY_POLICY } from "../../retry-policy.js";
6
+ import { BACKEND_TEST_MODULE_INDEX_HEADER, BACKEND_TEST_MODULE_SPLIT_REASONS } from "../../backend-test-plan-protocol.js";
7
+ import { DEFAULT_VERIFY_TIMEOUT_MS } from "../../../../executors/shell-verification.js";
8
+ import { BACKEND_TEST_EXECUTION_DEFAULT_TEST_ROOT, buildBackendTestExecutionPreflightShellSnippet } from "../../backend-test-execution-contract.js";
9
+ import { resolveBackendTestLayout } from "../../backend-test-layout.js";
10
+ import { buildBackendTestOutcomeGateShellSnippet } from "../../backend-test-result-contract.js";
11
+ import { buildBackendTestIntakeContext } from "../../backend-test-intake-context.js";
12
+ import { GENERATED_DAG_RUNTIME_CONTRACT, HYBRID_DEFAULTS, STANDARD_GLOBAL_CONSTRAINTS, buildBackendTestAnalysisSourceBindingContract, buildSourceContextBlock, buildVerifyEvidence, extractObjective, extractSuccessCriteria, resolveDagVerifyStrategy, toDagSourcePath } from "../sources.js";
13
+ import { applyDefaultReadOnlyRetryPolicy, commonForbiddenPaths, commonReadOnlyPaths, stampGeneratedArtifactBindings, } from "./shared.js";
14
+ export function buildAnalyzeInputsNode(sources) {
15
+ const sourceBindingContract = buildBackendTestAnalysisSourceBindingContract(sources);
16
+ return {
17
+ id: "analyze-inputs-pi",
18
+ depends_on: [],
19
+ role: "planner",
20
+ executor: "pi",
21
+ complexity: "MED",
22
+ writePolicy: "read-only",
23
+ allowedPaths: commonReadOnlyPaths(sources),
24
+ forbiddenPaths: commonForbiddenPaths(sources),
25
+ outputContract: "Pure Backend Test Analysis v2 JSON object matching docs/templates/backend-test-analysis.schema.json. No Markdown prose and no file writes.",
26
+ subtask_prompt: [
27
+ "Read the task source materials and return exactly one JSON object matching Backend Test Analysis v2.",
28
+ "Do not wrap it in explanatory prose. A single fenced json block is tolerated, but pure JSON is preferred.",
29
+ "Copy the sourceBinding object exactly from the JSON block below; do not infer, add, remove, or reclassify source paths.",
30
+ "Only kind=reference sources belong in referencePaths; kind=constraint sources MUST NOT be included in referencePaths.",
31
+ "## Exact Backend Test Analysis sourceBinding JSON",
32
+ JSON.stringify(sourceBindingContract, null, 2),
33
+ "For every endpoint, explicitly set responseBody.kind=array|object|scalar|empty|unknown and ordering=specified|unspecified|not-applicable. Add itemSchemaRef for arrays when documented.",
34
+ "For response fields, use comparison=exact|parseable-only|semantic when the source defines assertion semantics; date-time fields whose precision is unspecified should use parseable-only, not string equality.",
35
+ "Endpoint sourceRefs and field sourceRefs must cite only requirement/reference evidence actually read. Empty sourceRefs are allowed only when normalizing legacy v1 input; newly generated v2 should cite evidence.",
36
+ "For externalDependencies and risks, emit canonical items with exactly description plus optional name and sourceRef. For a dependency target, put the target value in name. Do not emit type, target, kind, required, severity, mitigation, level, impact, sourceRefs, or custom keys in newly generated v2 output.",
37
+ "Use empty arrays for categories not documented. Never include credentials, tokens, private keys, or secret values.",
38
+ "Required top-level keys: schemaVersion=2, sourceBinding, acceptanceCriteria, endpoints, dataModels, businessRules, stateTransitions, boundaryConstraints, externalDependencies, risks, evidenceGaps.",
39
+ "Read-only: do not modify code, docs, artifacts, or repository files.",
40
+ buildSourceContextBlock(sources),
41
+ ].join("\n\n"),
42
+ };
43
+ }
44
+ export function buildBackendTestAnalysisContractGateNode(sources) {
45
+ return {
46
+ id: "backend-test-analysis-contract-shell",
47
+ depends_on: ["analyze-inputs-pi"],
48
+ role: "verifier",
49
+ executor: "shell",
50
+ complexity: "LOW",
51
+ writePolicy: "read-only",
52
+ allowedPaths: commonReadOnlyPaths(sources),
53
+ forbiddenPaths: commonForbiddenPaths(sources),
54
+ outputContract: "Validated run-owned Backend Test Analysis v2 artifact pointer, schema ID, and SHA-256 (legacy v1 input is normalized to v2).",
55
+ subtask_prompt: "Materialize and validate the backend-test analysis contract under the current DAG run.",
56
+ shell: {
57
+ commands: [],
58
+ jsonArtifactGate: {
59
+ fromNodeId: "analyze-inputs-pi",
60
+ schemaId: "backend-test-analysis-v2",
61
+ artifactName: "backend-test-analysis.json",
62
+ outputDir: "contracts",
63
+ },
64
+ cwd: ".",
65
+ timeoutMs: 60000,
66
+ },
67
+ };
68
+ }
69
+ export function buildBackendTestEnvironmentScoutNode(sources) {
70
+ return {
71
+ id: "backend-test-environment-scout-pi",
72
+ depends_on: ["backend-test-analysis-contract-shell"],
73
+ role: "scout",
74
+ executor: "pi",
75
+ complexity: "MED",
76
+ writePolicy: "read-only",
77
+ allowedPaths: commonReadOnlyPaths(sources),
78
+ forbiddenPaths: commonForbiddenPaths(sources),
79
+ outputContract: "Pure Backend Test Execution Contract v1 JSON object matching docs/templates/backend-test-execution.schema.json. No Markdown prose and no file writes.",
80
+ subtask_prompt: [
81
+ "Read-only environment scout for backend-test pytest MVP.",
82
+ "Return exactly one JSON object matching Backend Test Execution Contract v1 (schema docs/templates/backend-test-execution.schema.json).",
83
+ "Prefer pure JSON; a single fenced json block is tolerated; no trailing prose.",
84
+ "Discover only non-secret evidence: pytest config files (pytest.ini / pyproject.toml / setup.cfg test paths), candidate test roots, existing fixtures/clients, documented run commands, and env *names* (not values).",
85
+ "Do NOT search the whole repo for secrets, .env values, tokens, private keys, or production credentials.",
86
+ 'framework must be "pytest". Default targetMode to "in-process" unless evidence clearly shows an external service base URL env name or documented managed start/stop with sourceRef.',
87
+ 'Do NOT select targetMode "managed-command" unless task source documents a safe start/stop command with an explicit sourceRef; otherwise leave managedCommand absent (do not invent managed mode). For external-running-service, missing managed start/stop is expected and is NOT an evidenceGap.',
88
+ "testRoot and workingDirectory must be repo-relative posix paths without .. or absolute form. Adapter default testRoot is testcase when evidence is incomplete.",
89
+ "runner must not include secret values. report.format must be junit with a relativeHint under the run (e.g. reports/backend-test-junit.xml).",
90
+ "requiredEnvNames lists env NAMES only. baseUrlEnvName is required only for external-running-service and must match ^[A-Z_][A-Z0-9_]*$.",
91
+ "evidenceGaps are optional notes only. Do NOT list greenfield/expected-later items as gaps: missing test_*.py / conftest (generate-pytest will create them), missing pytest.ini when testRoot defaults to testcase/, projected schema under ai_workspace/** instead of docs/templates/**, or optional API_BASE_URL when a documented default base URL exists.",
92
+ "Prefer evidenceGaps: [] for MVP greenfield external pytest. Use evidenceGaps only for true blockers the later generate nodes cannot fix (e.g. no viable testRoot at all). Populate evidenceRefs with repo-relative paths actually read.",
93
+ "Required top-level keys: schemaVersion, framework, runner, testRoot, workingDirectory, report, targetMode, existingFixtures, authenticationMode, requiredEnvNames, dataIsolation, evidenceGaps, evidenceRefs.",
94
+ "Read-only: do not modify code, docs, artifacts, or repository files.",
95
+ buildSourceContextBlock(sources),
96
+ ].join("\n\n"),
97
+ };
98
+ }
99
+ export function buildBackendTestExecutionContractGateNode(sources) {
100
+ return {
101
+ id: "backend-test-execution-contract-shell",
102
+ depends_on: ["backend-test-environment-scout-pi"],
103
+ role: "verifier",
104
+ executor: "shell",
105
+ complexity: "LOW",
106
+ writePolicy: "read-only",
107
+ allowedPaths: commonReadOnlyPaths(sources),
108
+ forbiddenPaths: commonForbiddenPaths(sources),
109
+ outputContract: "Validated run-owned Backend Test Execution Contract v1 artifact pointer, schema ID, and SHA-256.",
110
+ subtask_prompt: "Materialize and validate the backend-test execution contract under the current DAG run.",
111
+ shell: {
112
+ commands: [],
113
+ jsonArtifactGate: {
114
+ fromNodeId: "backend-test-environment-scout-pi",
115
+ schemaId: "backend-test-execution-v1",
116
+ artifactName: "backend-test-execution.json",
117
+ outputDir: "contracts",
118
+ },
119
+ cwd: ".",
120
+ timeoutMs: 60000,
121
+ },
122
+ };
123
+ }
124
+ export function buildGenerateBackendFunctionalCasesNode(sources) {
125
+ return {
126
+ id: "generate-backend-functional-cases-pi",
127
+ depends_on: [
128
+ "backend-test-analysis-contract-shell",
129
+ "backend-test-execution-contract-shell",
130
+ ],
131
+ role: "implementer",
132
+ executor: "pi",
133
+ toolProfile: "write",
134
+ complexity: "MED",
135
+ writePolicy: "exclusive",
136
+ writeSet: ["testcase/md/**"],
137
+ allowedPaths: ["testcase/md/**"],
138
+ forbiddenPaths: commonForbiddenPaths(sources),
139
+ // 注意:Pi 节点超时由 executor 层控制(默认 30 分钟)
140
+ // 如需调整,在 harness.json 的 executors.pi 中配置 modelConfig.timeoutMs
141
+ subtask_prompt: [
142
+ "Read both validated run-owned contracts before generating functional cases:",
143
+ "- contracts/backend-test-analysis.json: authoritative requirements, AC IDs, endpoints, fields, rules, boundaries, risks, and evidence gaps.",
144
+ "- contracts/backend-test-execution.json: pytest target mode, base URL env name, readiness, fixtures, and data-isolation constraints.",
145
+ "Generate cases from the analysis contract; use the execution contract only to keep preconditions and automation feasibility realistic.",
146
+ "Do not proceed from the execution contract alone. Do not re-read source documents or fall back to free-form analysis.",
147
+ "",
148
+ "## Output Steps (do in order):",
149
+ "1. First, output a brief summary: how many modules, how many cases planned per module",
150
+ "2. Then write each test case file under testcase/md/",
151
+ "",
152
+ "## Format Rules:",
153
+ "- Each test case ID: BE-<MODULE>-<NNN> (e.g. BE-ORDER-001) — always write the FULL id; never abbreviate as 002, 003 in matrices",
154
+ "- Each file covers one module",
155
+ "- Case structure: ID, Title, Acceptance Criteria, Business Rules, Precondition, Steps, Expected Result",
156
+ "- Every emitted case MUST declare at least one semantically applicable explicit AC-* under Acceptance Criteria; list BR-* separately under Business Rules",
157
+ "- If a BR-only scenario has no semantically valid in-scope AC, do not create a standalone case for it; record the limitation in the summary for the manifest evidenceGaps instead",
158
+ "- Never relabel a negative/boundary/BR-only behavior as AC-002 or another unrelated AC merely to make acIds non-empty",
159
+ "- Map each case to acceptance criteria (AC-xxx)",
160
+ "",
161
+ "## AC ↔ case consistency (CRITICAL — prevents review request-revision):",
162
+ "- Every AC-xxx listed on a case body MUST appear only on cases that truly exercise that AC",
163
+ "- Any AC-coverage matrix / summary table MUST list the same full BE-* case IDs that the case bodies claim — never 'all cases' / '全部用例' unless every case body maps that AC",
164
+ "- Prefer one primary BE-* case for suite-level ACs (e.g. AC-008 pytest exit 0) rather than tagging every case",
165
+ "- Out-of-scope ACs (Flyway, frontend e2e, mvn test, etc.) must NOT be claimed in MD; leave them for manifest evidenceGaps",
166
+ "",
167
+ "## Coverage Requirements:",
168
+ "- Positive paths: happy path for each acceptance criterion",
169
+ "- Negative paths: error scenarios (invalid input, not found, state violations)",
170
+ "",
171
+ "## Conditional Coverage (include ONLY if mentioned in upstream analysis):",
172
+ "- Boundary conditions: include ONLY if upstream analyze-inputs-pi mentions value ranges, length limits, numeric bounds, or format constraints",
173
+ "- State transitions: include ONLY if upstream analyze-inputs-pi mentions state machine",
174
+ "- Authentication scenarios: include ONLY if upstream analyze-inputs-pi mentions auth mechanism",
175
+ "- Timeout scenarios: include ONLY if upstream analyze-inputs-pi mentions timeout handling",
176
+ "- Concurrency scenarios: include ONLY if upstream analyze-inputs-pi mentions concurrency/idempotency rules",
177
+ "- If not mentioned, do NOT generate these test cases",
178
+ "",
179
+ "## Constraints:",
180
+ "- Stay within writeSet: testcase/md/**",
181
+ "- Do NOT re-read source documents or fall back to free-form analysis; use the two validated run-owned contracts only",
182
+ "- Do not write root artifacts/**",
183
+ ].join("\n\n"),
184
+ };
185
+ }
186
+ export function buildEmitBackendCaseManifestNode(sources, options) {
187
+ return {
188
+ id: options?.id ?? "emit-backend-case-manifest-pi",
189
+ depends_on: options?.dependsOn ?? [
190
+ "generate-backend-functional-cases-pi",
191
+ "backend-test-analysis-contract-shell",
192
+ ],
193
+ role: "scout",
194
+ executor: "pi",
195
+ complexity: "MED",
196
+ writePolicy: "read-only",
197
+ allowedPaths: commonReadOnlyPaths(sources),
198
+ forbiddenPaths: commonForbiddenPaths(sources),
199
+ outputContract: "Pure Backend Test Case Manifest v1 JSON (schema docs/templates/backend-test-case-manifest.schema.json). No file writes; model must not write .harness/**.",
200
+ subtask_prompt: [
201
+ "Emit Backend Test Case Manifest v1 as pure JSON (or one fenced json block with no trailing text).",
202
+ BACKEND_TEST_CASE_MANIFEST_OUTPUT_INSTRUCTIONS,
203
+ "Read-only: use validated contracts/backend-test-analysis.json pointer + testcase/md/** only. Do not write repository files or .harness/**.",
204
+ "No secrets or credential-shaped fields.",
205
+ ].join("\n\n"),
206
+ };
207
+ }
208
+ export function buildBackendTestCaseManifestGateNode(sources, options) {
209
+ const fromNodeId = options?.fromNodeId ?? "emit-backend-case-manifest-pi";
210
+ return {
211
+ id: options?.id ?? "backend-test-case-manifest-shell",
212
+ depends_on: options?.dependsOn ?? ["emit-backend-case-manifest-pi"],
213
+ role: "verifier",
214
+ executor: "shell",
215
+ complexity: "LOW",
216
+ writePolicy: "read-only",
217
+ allowedPaths: commonReadOnlyPaths(sources),
218
+ forbiddenPaths: commonForbiddenPaths(sources),
219
+ outputContract: "Validated run-owned Backend Test Case Manifest v1 at contracts/backend-test-case-manifest.json (schemaId backend-test-case-manifest-v1) with deterministic AC coverage.",
220
+ subtask_prompt: "Materialize and validate Backend Test Case Manifest v1; fail closed on duplicate IDs, unknown AC, missing AC coverage without gap, or skipped without gapReason.",
221
+ shell: {
222
+ commands: [],
223
+ jsonArtifactGate: {
224
+ fromNodeId,
225
+ schemaId: "backend-test-case-manifest-v1",
226
+ artifactName: "backend-test-case-manifest.json",
227
+ outputDir: "contracts",
228
+ },
229
+ cwd: ".",
230
+ timeoutMs: 60000,
231
+ },
232
+ };
233
+ }
234
+ export function buildBackendTestTraceabilityGateNode(sources, options = {}) {
235
+ return {
236
+ id: options.id ?? "backend-test-traceability-gate-shell",
237
+ depends_on: options.dependsOn ?? [
238
+ "generate-backend-pytest-pi",
239
+ // Effective Case Manifest v1 path (exclusive condition branches):
240
+ // pass → first manifest shell; request-revision → final manifest shell.
241
+ // Artifact path is always contracts/backend-test-case-manifest.json.
242
+ "backend-test-case-manifest-shell",
243
+ "backend-test-case-manifest-final-shell",
244
+ ],
245
+ dependsPolicy: "all-or-condition-skip",
246
+ role: "verifier",
247
+ executor: "shell",
248
+ complexity: "LOW",
249
+ writePolicy: "read-only",
250
+ allowedPaths: commonReadOnlyPaths(sources),
251
+ forbiddenPaths: commonForbiddenPaths(sources),
252
+ outputContract: "Deterministic traceability: generated cases have real file/symbol; skipped/unsupported have gapReason; convention symbols scanned under testcase/**/test_*.py.",
253
+ subtask_prompt: "Fail closed when generated automation claims do not resolve to workspace pytest symbols, or skip/unsupported lacks gapReason.",
254
+ shell: {
255
+ commands: ["backend-test-traceability-gate"],
256
+ cwd: ".",
257
+ timeoutMs: 60000,
258
+ },
259
+ };
260
+ }
261
+ export function buildReviewBackendCasesNode(sources, options) {
262
+ return {
263
+ id: "review-backend-cases-pi",
264
+ depends_on: options?.dependsOn ?? [
265
+ "backend-test-case-manifest-shell",
266
+ "backend-test-analysis-contract-shell",
267
+ ],
268
+ role: "reviewer",
269
+ executor: "pi",
270
+ complexity: "HIGH",
271
+ writePolicy: "read-only",
272
+ allowedPaths: commonReadOnlyPaths(sources),
273
+ forbiddenPaths: commonForbiddenPaths(sources),
274
+ outputContract: "advisory case review evidence whose first non-empty line is VERDICT: pass or VERDICT: request-revision; followed by Findings and Coverage Assessment. No file writes; this review neither authorizes nor blocks pytest generation.",
275
+ subtask_prompt: [
276
+ "Review the generated backend functional test cases under testcase/md/ and the validated Case Manifest v1.",
277
+ "",
278
+ "## Mandatory First Line:",
279
+ "First non-empty line must be exactly: VERDICT: pass or VERDICT: request-revision",
280
+ "",
281
+ "## Review Checklist:",
282
+ "- ID format: every case uses BE-<MODULE>-<NNN> (full ids only in bodies and matrices)",
283
+ "- Positive coverage: each in-scope acceptance criterion (AC-xxx) has happy-path case",
284
+ "- Negative coverage: error scenarios (invalid input, not found, state violations)",
285
+ "- Traceability: each explicit AC maps to a case ID or an evidenceGap in contracts/backend-test-case-manifest.json",
286
+ "- Case structure: ID, Title, Precondition, Steps, Expected Result",
287
+ "- No duplicate IDs across files",
288
+ "- Manifest consistency (Critical): every AC claimed in MD case bodies/matrices must match manifest caseId→acIds; never accept 'all cases cover AC-xxx' unless every case maps that AC",
289
+ "",
290
+ "## Conditional Coverage (check ONLY if mentioned in upstream analysis):",
291
+ "- Boundary coverage: check ONLY if analyze-inputs-pi mentions value ranges, length limits, numeric bounds, or format constraints",
292
+ "- State transition coverage: check ONLY if analyze-inputs-pi mentions state machine",
293
+ "- Authentication coverage: check ONLY if analyze-inputs-pi mentions auth mechanism",
294
+ "- Timeout coverage: check ONLY if analyze-inputs-pi mentions timeout handling",
295
+ "- Concurrency coverage: check ONLY if analyze-inputs-pi mentions concurrency/idempotency rules",
296
+ "- If not mentioned, do NOT flag as missing",
297
+ "",
298
+ "## Do NOT treat as Critical alone:",
299
+ "- Missing test_*.py / automation still planned (expected before generate-backend-pytest-pi)",
300
+ "- Out-of-scope ACs already listed in manifest evidenceGaps (Flyway, frontend e2e, mvn test)",
301
+ "",
302
+ "## Verdict Rules:",
303
+ "- All Critical checks pass + Important findings ≤ 2 → VERDICT: pass",
304
+ "- Any Critical fails OR Important > 2 → VERDICT: request-revision",
305
+ "- Any request-revision verdict is advisory evidence for canonical context, retrospective, and L-5; it does not authorize or block the pytest writer.",
306
+ "",
307
+ "## Output After Verdict:",
308
+ "1. Coverage Assessment table (AC → full BE-* case IDs) using manifest + MD",
309
+ "2. Findings list (Critical/Important/Informational)",
310
+ "3. Statistics (total cases, positive/negative/boundary breakdown)",
311
+ "4. Required follow-up actions (only when request-revision; no in-run writer)",
312
+ "",
313
+ "## Constraints:",
314
+ "- Read-only: do not modify files",
315
+ "- Read validated analysis + case manifest artifacts; do not recompute coverage percentages",
316
+ "- Use testcase/md/ files for case review",
317
+ ]
318
+ .filter((line) => line !== "")
319
+ .join("\n\n"),
320
+ };
321
+ }
322
+ export function buildExecuteBackendPytestNode(sources, options = {}) {
323
+ // Keep the target worktree read-only: JUnit is runner-owned evidence under
324
+ // the current DAG run and moves with active → completed/paused lifecycle.
325
+ // Adapter default testRoot is frozen at DAG generation time (auditable) and
326
+ // cross-checked against the materialized execution contract in preflight.
327
+ const frozenTestRoot = options.testRoot ?? BACKEND_TEST_EXECUTION_DEFAULT_TEST_ROOT;
328
+ const preflightCommand = buildBackendTestExecutionPreflightShellSnippet({
329
+ expectedTestRoot: frozenTestRoot,
330
+ });
331
+ // Map pytest exit 0/1 → node success ONLY when JUnit exists (assertion-fail is a
332
+ // legal result). Do not change global shell ok semantics. Persist raw exit for parse.
333
+ const nodeId = options.id ?? "execute-backend-pytest-shell";
334
+ const reportStem = options.reportStem ?? "backend-test";
335
+ const reportName = `${reportStem}-junit.xml`;
336
+ const exitName = `${reportStem}-pytest-exit.txt`;
337
+ // Preflight snippet is already fail-closed (&&). Only the pytest body may use
338
+ // ";" so STATUS capture still runs after non-zero pytest exits.
339
+ const pytestBody = [
340
+ `REPORT="\${HARNESS_DAG_RUN_DIR}/reports/${reportName}"`,
341
+ `EXIT_FILE="\${HARNESS_DAG_RUN_DIR}/reports/${exitName}"`,
342
+ 'mkdir -p "$(dirname "${REPORT}")"',
343
+ `PYTHONUTF8=1 PYTHONIOENCODING=utf-8 PYTHONDONTWRITEBYTECODE=1 python -m pytest ${frozenTestRoot}/ -v -p no:cacheprovider --junitxml="\${REPORT}"`,
344
+ "STATUS=$?",
345
+ 'printf "%s" "${STATUS}" > "${EXIT_FILE}"',
346
+ 'printf "JUnit report: %s\\n" "${REPORT}"',
347
+ 'printf "pytestExitCode=%s\\n" "${STATUS}"',
348
+ 'if { [ "${STATUS}" -eq 0 ] || [ "${STATUS}" -eq 1 ]; } && [ -s "${REPORT}" ]; then exit 0; fi',
349
+ 'exit "${STATUS}"',
350
+ ].join("; ");
351
+ const pytestCommand = `${preflightCommand} && { ${pytestBody}; }`;
352
+ return {
353
+ id: nodeId,
354
+ depends_on: options.dependsOn ?? [
355
+ "backend-test-semantic-gate-shell",
356
+ "backend-test-execution-contract-shell",
357
+ ],
358
+ role: "verifier",
359
+ executor: "shell",
360
+ complexity: "LOW",
361
+ writePolicy: "read-only",
362
+ allowedPaths: commonReadOnlyPaths(sources),
363
+ forbiddenPaths: commonForbiddenPaths(sources),
364
+ outputContract: "Archived pytest stdout/stderr; raw pytestExitCode side-channel + JUnit at $HARNESS_DAG_RUN_DIR/reports/**. Exit 0/1 with non-empty JUnit finishes the node so parse/classify/retrospect can run; assertion failures remain recorded in exit file.",
365
+ subtask_prompt: "Run pytest for the backend test suite; write JUnit + pytestExitCode evidence only under the current HARNESS_DAG_RUN_DIR/reports/.",
366
+ shell: {
367
+ commands: [pytestCommand],
368
+ envAllowlist: collectBackendTestShellEnvAllowlist(sources),
369
+ verifyEvidence: buildVerifyEvidence({
370
+ phase: "final",
371
+ quota: "full",
372
+ commandSource: "inline",
373
+ fallbackCommands: [pytestCommand],
374
+ finalFullRequired: true,
375
+ commandTimeoutMs: DEFAULT_VERIFY_TIMEOUT_MS,
376
+ }),
377
+ cwd: ".",
378
+ timeoutMs: DEFAULT_VERIFY_TIMEOUT_MS,
379
+ },
380
+ };
381
+ }
382
+ export function buildClassifyBackendTestResultNode(sources) {
383
+ return {
384
+ id: "classify-backend-test-result-pi",
385
+ depends_on: ["execute-and-parse-backend-pytest-shell"],
386
+ role: "reviewer",
387
+ executor: "pi",
388
+ complexity: "MED",
389
+ writePolicy: "read-only",
390
+ allowedPaths: commonReadOnlyPaths(sources),
391
+ forbiddenPaths: commonForbiddenPaths(sources),
392
+ outputContract: "Pure JSON classification: category in {ProductBug,TestBug,EnvFailure,ContractMismatch,FlakyTest,Unknown}, evidence[], confidence (capped), notes. No file writes.",
393
+ subtask_prompt: [
394
+ "Read-only classifier for Backend Test Result v1.",
395
+ "Return exactly one JSON object (prefer pure JSON; single fenced json block tolerated; no trailing prose).",
396
+ "The object must contain exactly category, evidence, confidence, notes. evidence must be a non-empty array of strings, confidence must be a number from 0 through 1, and notes must be a non-empty string. Do not emit schemaVersion or custom fields.",
397
+ 'Minimal shape: {"category":"Unknown","evidence":["outcome=completed-with-failures"],"confidence":0.5,"notes":"Single-run evidence is insufficient for a stronger classification."}',
398
+ "Read contracts/backend-test-result-initial.json (run-owned initial Result v1). Do NOT invent pass rates from raw logs.",
399
+ "category must be one of: ProductBug, TestBug, EnvFailure, ContractMismatch, FlakyTest, Unknown.",
400
+ "Hard constraints:",
401
+ "- Single-run failure MUST NOT use FlakyTest (use Unknown, TestBug, or ProductBug).",
402
+ "- executionStatus/outcome collection-error, command-error, or report-error MUST NOT use ProductBug.",
403
+ "- Prefer EnvFailure/Unknown/TestBug for env, import, collection, and missing-report cases.",
404
+ "- confidence must respect deterministic caps (≤0.75 for assertion failures; ≤0.6 for env/collection).",
405
+ "Include evidence[] referencing result fields (outcome, failed, failures[].name, executionStatus).",
406
+ "Read-only: do not modify code, docs, artifacts, or repository files.",
407
+ ].join("\n\n"),
408
+ };
409
+ }
410
+ export function buildTestRetrospectNode(sources) {
411
+ const canWriteReport = taskAllowsBackendTestReportWrite(sources);
412
+ return {
413
+ id: "test-retrospect-pi",
414
+ depends_on: ["select-effective-backend-test-result-shell"],
415
+ role: "closeout",
416
+ executor: "pi",
417
+ complexity: "MED",
418
+ ...(canWriteReport
419
+ ? {
420
+ toolProfile: "write",
421
+ writePolicy: "exclusive",
422
+ writeSet: ["docs/test-reports/**"],
423
+ allowedPaths: ["docs/test-reports/**"],
424
+ }
425
+ : {
426
+ writePolicy: "read-only",
427
+ allowedPaths: commonReadOnlyPaths(sources),
428
+ }),
429
+ forbiddenPaths: commonForbiddenPaths(sources),
430
+ outputContract: canWriteReport
431
+ ? "Maturity rating in assistant output plus a report written under docs/test-reports/**."
432
+ : "Read-only maturity rating and retrospective in assistant output; no repository file writes because task allowedPaths do not authorize docs/test-reports/**.",
433
+ subtask_prompt: [
434
+ "Read the complete JSON from direct upstream select-effective-backend-test-result-shell and generate a test retrospective report.",
435
+ "That JSON contains result, manifest (including coverageSummary), and classification. Treat those fields as authoritative; do not rely on pointer/hash summaries.",
436
+ "",
437
+ "## Output Steps (do in order):",
438
+ "1. First, output the maturity rating on the first line: Rating: A/B/C/D",
439
+ canWriteReport
440
+ ? "2. Then write the full report under docs/test-reports/"
441
+ : "2. Keep the full retrospective in assistant output only; do not write repository files because docs/test-reports/** is outside task allowedPaths.",
442
+ "",
443
+ "## Stats authority (deterministic only):",
444
+ "- Pass rate, failed/error/skipped counts, and failure list MUST come from contracts/backend-test-result.json only.",
445
+ "- AC coverage ratio / case counts MUST come from contracts/backend-test-case-manifest.json coverageSummary (or gate-derived fields). Do NOT invent coverage %.",
446
+ "- Automation coverage MUST use coverageSummary.generatedCount / coverageSummary.caseCount. If either field is missing, write unavailable; do not estimate.",
447
+ "- Code coverage MUST come only from the validated contracts/code-coverage-v1.json artifact generated by coverage.py/pytest-cov or JaCoCo. Show line, branch, function/method, covered, total, ratio, threshold, status, source scope, requirement IDs, tool, commit, and artifact hash.",
448
+ "- Use classify-backend-test-result-pi JSON as interpretive evidence only.",
449
+ "- NEVER rewrite a failed result as passed. Result v1 is authoritative for testOutcome; pipeline completion and L-5 readiness are separate conclusions.",
450
+ "",
451
+ "## Report Structure:",
452
+ "1. Maturity Rating with rationale",
453
+ "2. Test Coverage Summary (Result v1 pass rate, AC coverage, automation coverage, and code coverage)",
454
+ "3. Failed Test Analysis (failure/error details, category, confidence, evidence, and owner direction)",
455
+ "4. Defects (local Bug ledger in the same report directory; unavailable when absent)",
456
+ "5. Risks (Critical/High/Medium/Low, impact, controls, residual risk, treatment; Critical risks block L-5, High risks do not automatically block)",
457
+ "6. Regression Recommendations (immediate, related, periodic, deferred; every item links to failure/risk/AC/case IDs)",
458
+ "7. L-5 conclusion with blocking items",
459
+ "",
460
+ "## Rating Criteria:",
461
+ "- L-5 ready requires pass rate=100%, AC coverage=100%, automation coverage≥90%, line coverage≥80%, branch coverage≥70%, skipped=0, and no blocking Critical risk.",
462
+ "- Any required metric fail or unavailable means L-5 not-ready. Function/method coverage is displayed but not a gate. Preserve the existing A/B/C/D single-run rating separately.",
463
+ "",
464
+ "## Constraints:",
465
+ canWriteReport
466
+ ? "- Stay within writeSet: docs/test-reports/**"
467
+ : "- Read-only: do not modify repository files",
468
+ "- Do NOT re-read source documents — use upstream outputs only",
469
+ "- Do not write root artifacts/**",
470
+ ].join("\n\n"),
471
+ };
472
+ }
473
+ export function buildBackendTestOutcomeGateNode(sources) {
474
+ const gateCommand = buildBackendTestOutcomeGateShellSnippet();
475
+ return {
476
+ id: "backend-test-outcome-gate-shell",
477
+ depends_on: ["l5-metrics-pi"],
478
+ role: "verifier",
479
+ executor: "shell",
480
+ complexity: "LOW",
481
+ writePolicy: "read-only",
482
+ allowedPaths: commonReadOnlyPaths(sources),
483
+ forbiddenPaths: commonForbiddenPaths(sources),
484
+ outputContract: "Shell exit 0 only when Result v1 outcome=passed with failed=0 and error=0; non-zero otherwise. Ignores retrospective Markdown.",
485
+ subtask_prompt: "Gate the backend-test DAG on run-owned Result v1 shell facts only (not retrospective prose).",
486
+ shell: {
487
+ commands: [gateCommand],
488
+ verifyEvidence: buildVerifyEvidence({
489
+ phase: "final",
490
+ quota: "full",
491
+ commandSource: "inline",
492
+ fallbackCommands: [gateCommand],
493
+ finalFullRequired: true,
494
+ commandTimeoutMs: 60_000,
495
+ }),
496
+ cwd: ".",
497
+ timeoutMs: 60000,
498
+ },
499
+ };
500
+ }
501
+ export function buildL5MetricsNode(sources) {
502
+ return {
503
+ id: "l5-metrics-pi",
504
+ depends_on: ["test-retrospect-pi"],
505
+ role: "reviewer",
506
+ executor: "pi",
507
+ complexity: "MED",
508
+ writePolicy: "read-only",
509
+ allowedPaths: commonReadOnlyPaths(sources),
510
+ forbiddenPaths: commonForbiddenPaths(sources),
511
+ outputContract: "Exactly one JSON object with status=ready|not-ready, metrics, and blockingItems; no file writes.",
512
+ subtask_prompt: [
513
+ "You are the independent L-5 metrics node at the end of the existing backend-test DAG.",
514
+ "The direct upstream test-retrospect-pi output is the primary report to assess. Read it together with the run-owned Result v1, Case Manifest v1, and Code Coverage v1 artifacts when present.",
515
+ "Do not create a new DAG, rewrite the retrospective report, change test outcome, or modify any repository file.",
516
+ "Return exactly one JSON object and no surrounding prose.",
517
+ 'Required shape: {"status":"ready"|"not-ready","metrics":{"passRate":metric,"acCoverage":metric,"automationCoverage":metric,"lineCoverage":metric,"branchCoverage":metric,"skipped":metric,"criticalRisks":metric},"blockingItems":[string]}.',
518
+ "Each metric must contain numerator, denominator, ratio, threshold, status=pass|fail|unavailable, and reason (null only when passed).",
519
+ "Use only explicit evidence. Missing or invalid required evidence is unavailable, never zero or an estimate.",
520
+ "L-5 ready requires pass rate=100%, AC coverage=100%, automation coverage>=90%, line coverage>=80%, branch coverage>=70%, skipped=0, and zero blocking Critical risks.",
521
+ "Function/method coverage is display-only and does not gate L-5. Preserve the distinction between L-5 maturity and the Result v1 outcome gate.",
522
+ ].join("\n\n"),
523
+ };
524
+ }
525
+ export function applyBackendTestLayoutToText(text, layout) {
526
+ if (layout.isDefault)
527
+ return text;
528
+ const replacements = [
529
+ ["testcase/test_", `${layout.scriptDir}/test_`],
530
+ ["testcase/md/", `${layout.markdownDir}/`],
531
+ ["testcase/**", `${layout.testRoot}/**`],
532
+ ["testcase/", `${layout.testRoot}/`],
533
+ ];
534
+ replacements.sort((a, b) => b[0].length - a[0].length);
535
+ let output = "";
536
+ const resolvedPrefix = `${layout.testRoot}/`;
537
+ for (let i = 0; i < text.length;) {
538
+ if (text.startsWith(resolvedPrefix, i)) {
539
+ output += resolvedPrefix;
540
+ i += resolvedPrefix.length;
541
+ continue;
542
+ }
543
+ const hit = replacements.find(([token]) => text.startsWith(token, i));
544
+ if (hit) {
545
+ output += hit[1];
546
+ i += hit[0].length;
547
+ continue;
548
+ }
549
+ output += text[i];
550
+ i += 1;
551
+ }
552
+ return output;
553
+ }
554
+ export function buildBackendTestModuleManifestShellCommand(layout, moduleLayout) {
555
+ // The extractor is base64-encoded so the shell command is fully opaque to
556
+ // bash: no backticks (command substitution), no regex \/ escaping, no
557
+ // backslash-counting through TS-string -> JSON.stringify -> bash -c -> node -e.
558
+ // Backticks in the README body are stripped at runtime via
559
+ // String.fromCharCode(96), so the extractor source contains no backtick.
560
+ // mdDir/testPrefix are injected as JSON literals so the same extractor
561
+ // works for any configured backendTest layout (plan A).
562
+ const mdDirLiteral = JSON.stringify(layout.markdownDir);
563
+ const scriptDirLiteral = JSON.stringify(layout.scriptDir);
564
+ const strictLayoutCompressed = moduleLayout
565
+ ? deflateRawSync(Buffer.from(JSON.stringify(moduleLayout), "utf8")).toString("base64")
566
+ : null;
567
+ const strictLayoutExpression = strictLayoutCompressed
568
+ ? `JSON.parse(zlib.inflateRawSync(Buffer.from(${JSON.stringify(strictLayoutCompressed)},'base64')).toString('utf8'))`
569
+ : "null";
570
+ const strictLayoutShaLiteral = JSON.stringify(moduleLayout
571
+ ? createHash("sha256").update(JSON.stringify(moduleLayout)).digest("hex")
572
+ : null);
573
+ const escOpen = String.fromCharCode(92, 91); // \[
574
+ const escClose = String.fromCharCode(92, 93); // \]
575
+ const escBslash = String.fromCharCode(92, 92); // \\
576
+ const script = `const fs=require('fs'),path=require('path'),crypto=require('crypto'),zlib=require('zlib');
577
+ const mdDir=${mdDirLiteral};
578
+ const scriptDir=${scriptDirLiteral};
579
+ const strictLayout=${strictLayoutExpression};
580
+ const strictLayoutSha256=${strictLayoutShaLiteral};
581
+ const esc=s=>s.replace(/[${escOpen}${escClose}{}()*+?^$|${escBslash}]/g,'${escBslash}$&');
582
+ const rxMdPath=new RegExp(esc(mdDir)+'${escBslash}/([A-Za-z0-9_.-]+)${escBslash}.md','g');
583
+ const runDir=process.env.HARNESS_DAG_RUN_DIR||'';
584
+ if(!runDir){process.stderr.write('missing HARNESS_DAG_RUN_DIR for backend-test Markdown plan artifact\\n');process.exit(2);}
585
+ const planPath=path.join(runDir,'generate-backend-md-plan-pi','plan.md');
586
+ if(!fs.existsSync(planPath)){process.stderr.write('missing backend-test Markdown plan artifact: '+planPath+'\\n');process.exit(2);}
587
+ let readme=fs.readFileSync(planPath,'utf8');
588
+ const norm=s=>String(s).toLowerCase().replace(/[^a-z0-9]+/g,'_').replace(/^_+|_+$/g,'').replace(/_+/g,'_');
589
+ const bt=String.fromCharCode(96);
590
+ const stripBackticks=s=>s.split(bt).join('');
591
+ const invalidReason=raw=>{const st=norm(raw);if(/^p[0-2]$/.test(st))return 'priority-only-module-stem';if(/^[a-f][a-f0-9]{6,63}$/.test(st))return 'opaque-hash-module-stem';if(st==='readme')return 'reserved-module-stem';if(!/^[a-z][a-z0-9_]*$/.test(st))return 'invalid-syntax';if(/^(?:be|tp|ac|req|br)[_-]/i.test(st))return 'case-like-module-stem';return null;};
592
+ const valid=raw=>invalidReason(raw)===null;
593
+ const rxRelLink=/\\[[^\\]]+\\]\\(\\.\\/([A-Za-z0-9_.-]+)\\.md\\)/g;
594
+ const headingAlias={'\u6a21\u5757\u7d22\u5f15':'Module Index','\u8986\u76d6\u8303\u56f4':'Coverage Scope','\u8986\u76d6\u77e9\u9635':'Coverage Matrix','\u573a\u666f\u5206\u533a':'Scenario Partitions'};
595
+ const allLines=readme.replace(/\\r\\n/g,'\\n').replace(/\\r/g,'\\n').split('\\n');
596
+ let headingRepair=false;
597
+ for(let i=0;i<allLines.length;i++){const alias=/^##\\s+(模块索引|覆盖范围|覆盖矩阵|场景分区)\\s*$/.exec(allLines[i].trim());if(alias){allLines[i]='## '+headingAlias[alias[1]];headingRepair=true;}}
598
+ let partitionRepair=false;
599
+ const partHeadTrim=[];for(let i=0;i<allLines.length;i++){if(allLines[i].trim()==='## Scenario Partitions')partHeadTrim.push(i);}
600
+ if(partHeadTrim.length===1){
601
+ const pStart=partHeadTrim[0]+1;let pEnd=allLines.length;for(let i=pStart;i<allLines.length;i++){if(/^##\\s+\\S/.test(allLines[i].trim())){pEnd=i;break;}}
602
+ const pCells=line=>line.split('|').slice(1,-1).map(v=>String(v||'').trim());
603
+ const pHeader=['Partition ID','Operation','Axis','Domain','Required Slots','Expected by Slot','Bind Rule'];
604
+ for(let i=pStart;i<pEnd;i++){
605
+ if(!allLines[i].includes('|')) continue;
606
+ const row=pCells(allLines[i]);
607
+ if(row.length<=7) continue;
608
+ if(pHeader.every((v,idx)=>row[idx]===v)) continue;
609
+ const pid=String(row[0]||'').trim();
610
+ const extras=row.slice(7).join(';').split(/[;,,;]/).map(v=>v.trim().split(bt).join('')).filter(Boolean);
611
+ const prefix='TP-'+pid.toUpperCase()+'-';
612
+ if(extras.length && extras.every(t=>/^TP-[A-Z0-9-]+$/i.test(t)&&(t.toUpperCase().startsWith(prefix)||t.toUpperCase().startsWith('TP-SP-')))){
613
+ allLines[i]='| '+row.slice(0,7).join(' | ')+' |';
614
+ partitionRepair=true;
615
+ }
616
+ }
617
+ }
618
+ if(headingRepair||partitionRepair)readme=allLines.join('\\n');
619
+ const headings=[];for(let i=0;i<allLines.length;i++){if(allLines[i].trim()==='## Module Index')headings.push(i);}
620
+ if(headings.length!==1){process.stderr.write((headings.length===0?'missing-module-index':'duplicate-module-index')+'; require exactly one exact ## Module Index section\\n');process.exit(2);}
621
+ const start=headings[0]+1;let end=allLines.length;for(let i=start;i<allLines.length;i++){if(/^##\\s+\\S/.test(allLines[i].trim())){end=i;break;}}
622
+ const section=allLines.slice(start,end).join('\\n');
623
+ const raw=[];
624
+ const lines=section.split('\\n').filter(l=>l.includes('|'));
625
+ const cells=line=>line.split('|').slice(1,-1).map(value=>stripBackticks(value).trim());
626
+ const expectedHeader=${JSON.stringify(BACKEND_TEST_MODULE_INDEX_HEADER)};
627
+ const headerIndex=lines.findIndex(line=>{const row=cells(line);return expectedHeader.every((value,index)=>row[index]===value);});
628
+ if(headerIndex<0){process.stderr.write('invalid-module-index-header: require exact business ownership and path columns\\n');process.exit(2);}
629
+ const dataRows=lines.slice(headerIndex+2).map(cells).filter(row=>row.length>=8&&row[0]&&row[0]!=='Module Stem');
630
+ const allowedSplit=new Set(${JSON.stringify(BACKEND_TEST_MODULE_SPLIT_REASONS)});
631
+ const operationOwners=new Map();const canonicalSeen=new Set();const declaredModules=[];let planRepairApplied=headingRepair||partitionRepair;
632
+ for(const row of dataRows){
633
+ const rawStem=String(row[0]||'').trim(),resource=String(row[1]||'').trim(),operations=String(row[2]||'').split(';').map(value=>value.trim()).filter(Boolean),split=String(row[5]||'').trim();
634
+ const mdMatches=[...String(row[6]||'').matchAll(rxMdPath)];
635
+ if(mdMatches.length!==1){process.stderr.write('module-markdown-path-mismatch: '+(rawStem||'unknown')+'; require exactly one canonical Markdown Path\\n');process.exit(2);}
636
+ let stem=norm(mdMatches[0][1]),markdownPath=mdMatches[0][0];
637
+ if(strictLayout&&!strictLayout.modules.some(item=>item.markdownPath===markdownPath)){planRepairApplied=true;continue;}
638
+ if(!resource){process.stderr.write('missing-business-resource: '+stem+'\\n');process.exit(2);}
639
+ if(/^(?:response|resp|regression|positive|negative|boundary|error|combo|filter)(?:[_ -]|$)/i.test(resource)){process.stderr.write('test-purpose-business-resource: '+stem+' -> '+resource+'\\n');process.exit(2);}
640
+ if(!allowedSplit.has(split)){process.stderr.write('invalid-module-split-reason: '+stem+' -> '+split+'\\n');process.exit(2);}
641
+ if(operations.length===0){process.stderr.write('missing-owned-operation: '+stem+'\\n');process.exit(2);}
642
+ let pytestPath=String(row[7]||'').replace(/^\\.\\//,'').replace(/\\\\/g,'/');
643
+ if(!/^[A-Za-z0-9_./-]+\\.py$/.test(pytestPath)||pytestPath.includes('..')){process.stderr.write('invalid-module-pytest-path: '+stem+' -> '+pytestPath+'\\n');process.exit(2);}
644
+ const requiredCandidates=strictLayout?strictLayout.modules.filter(item=>item.markdownPath===markdownPath):[];
645
+ if(strictLayout&&requiredCandidates.length!==1){process.stderr.write('MODULE_LAYOUT_CONFLICT: no unique required module for Markdown Path '+markdownPath+'\\n');process.exit(2);}
646
+ const required=requiredCandidates[0];
647
+ if(!required&&norm(rawStem)!==stem)planRepairApplied=true;
648
+ if(required){
649
+ stem=required.stem;
650
+ if(norm(rawStem)!==stem)planRepairApplied=true;
651
+ if(!required.markdownPath.startsWith(mdDir+'/')||!required.pytestPath.startsWith(scriptDir+'/')||required.markdownPath.includes('..')||required.pytestPath.includes('..')){process.stderr.write('MODULE_LAYOUT_CONFLICT: strict paths outside frozen roots for '+stem+'\\n');process.exit(2);}
652
+ if(markdownPath!==required.markdownPath||pytestPath!==required.pytestPath||resource!==required.businessResource||JSON.stringify(operations)!==JSON.stringify(required.ownedOperations)||split!==required.splitReason){planRepairApplied=true;}
653
+ markdownPath=required.markdownPath;pytestPath=required.pytestPath;
654
+ if(required.splitReason==='output-budget'){
655
+ const proof=required.budgetProof;const peers=strictLayout.modules.filter(item=>item.budgetProof&&item.budgetProof.groupId===proof.groupId);
656
+ if(!proof||proof.estimatedOutputChars<=proof.maxOutputCharsPerWriter||Math.ceil(proof.estimatedOutputChars/Math.max(1,peers.length))>proof.maxOutputCharsPerWriter||peers.some(item=>item.budgetProof.inputSha256!==proof.inputSha256||item.budgetProof.estimatorVersion!==proof.estimatorVersion)){process.stderr.write('INVALID_OUTPUT_BUDGET_PROOF: '+stem+'\\n');process.exit(2);}
657
+ }
658
+ }
659
+ const item={stem,markdownPath,pytestPath,businessResource:required?required.businessResource:resource,ownedOperations:required?required.ownedOperations:operations,ownedRuleKeys:String(row[3]||'').split(';').map(value=>value.trim()).filter(Boolean),caseIds:String(row[4]||'').split(';').map(value=>value.trim()).filter(Boolean),splitReason:required?required.splitReason:split,...(required&&required.budgetProof?{budgetProof:required.budgetProof}:{})};
660
+ if(canonicalSeen.has(item.stem)){process.stderr.write('duplicate-canonical-module-stem: '+item.stem+'; module identity must be unique\\n');process.exit(2);}
661
+ canonicalSeen.add(item.stem);
662
+ declaredModules.push(item);
663
+ for(const operation of item.ownedOperations){const owners=operationOwners.get(operation)||[];owners.push({stem,split:item.splitReason});operationOwners.set(operation,owners);}
664
+ }
665
+ if(strictLayout){
666
+ const missing=strictLayout.modules.filter(item=>!declaredModules.some(actual=>actual.stem===item.stem));
667
+ if(missing.length){process.stderr.write('MODULE_LAYOUT_CONFLICT: plan missing required modules '+missing.map(item=>item.stem).join(',')+'; deterministic Plan-only repair cannot invent Rule/Case ownership\\n');process.exit(2);}
668
+ }
669
+ if(planRepairApplied){
670
+ const table=['| '+expectedHeader.join(' | ')+' |','|'+expectedHeader.map(()=>'---').join('|')+'|',...declaredModules.map(item=>'| '+[item.stem,item.businessResource,item.ownedOperations.join('; '),item.ownedRuleKeys.join('; '),item.caseIds.join('; '),item.splitReason,'['+item.stem+'](./'+item.stem+'.md) '+item.markdownPath,item.pytestPath].join(' | ')+' |')].join('\\n');
671
+ readme=[...allLines.slice(0,headings[0]+1),table,...allLines.slice(end)].join('\\n');
672
+ fs.writeFileSync(planPath,readme,'utf8');
673
+ } else if(headingRepair||partitionRepair){
674
+ fs.writeFileSync(planPath,readme,'utf8');
675
+ }
676
+ for(const [operation,owners] of operationOwners){if(owners.length>1&&!owners.every(owner=>owner.split==='explicit-user-layout'||owner.split==='output-budget')){process.stderr.write('overlapping-operation-modules: '+operation+' -> '+owners.map(owner=>owner.stem).join(',')+'; merge by business resource or use an authoritative explicit-user-layout\\n');process.exit(2);}}
677
+ if(!strictLayout){
678
+ for(const line of lines){
679
+ const bare=stripBackticks(line);
680
+ for(const m of bare.matchAll(rxMdPath)){raw.push(m[1]);}
681
+ }
682
+ for(const m of section.matchAll(rxRelLink)){raw.push(m[1]);}
683
+ }
684
+ const invalid=[];for(const r of raw){const reason=invalidReason(r);if(reason)invalid.push({stem:norm(r),reason});}
685
+ if(invalid.length){for(const item of invalid)process.stderr.write(item.reason+': '+item.stem+'; use a stable business resource/domain stem\\n');process.exit(2);}
686
+ const modules=[];
687
+ for(const item of declaredModules){if(valid(item.stem))modules.push(item);}
688
+ if(modules.length===0){process.stderr.write('empty-module-index: require at least one stable business module\\n');process.exit(2);}
689
+ if(modules.length>8){process.stderr.write('excessive-module-count: '+modules.length+' > 8; merge by the smallest stable business resource/domain set\\n');process.exit(2);}
690
+ const planReadPath=path.relative(process.cwd(),planPath).split(path.sep).join('/');
691
+ const planSha256=crypto.createHash('sha256').update(readme).digest('hex');
692
+ const slotTok=v=>String(v).trim().toUpperCase().replace(/[^A-Z0-9]+/g,'-').replace(/^-+|-+$/g,'');
693
+ const partHead=[];for(let i=0;i<allLines.length;i++){if(allLines[i].trim()==='## Scenario Partitions')partHead.push(i);}
694
+ if(partHead.length>1){process.stderr.write('duplicate-scenario-partitions; require at most one exact ## Scenario Partitions section\\n');process.exit(2);}
695
+ const slotByModule=new Map(modules.map(m=>[m.stem,[]]));const unassigned=[];
696
+ if(partHead.length===1){
697
+ const pStart=partHead[0]+1;let pEnd=allLines.length;for(let i=pStart;i<allLines.length;i++){if(/^##\\s+\\S/.test(allLines[i].trim())){pEnd=i;break;}}
698
+ const pLines=allLines.slice(pStart,pEnd).filter(l=>l.includes('|'));
699
+ const pCells=line=>line.split('|').slice(1,-1).map(v=>stripBackticks(v).trim());
700
+ const pHeader=['Partition ID','Operation','Axis','Domain','Required Slots','Expected by Slot','Bind Rule'];
701
+ const pH=pLines.findIndex(line=>pHeader.every((v,i)=>pCells(line)[i]===v));
702
+ if(pH<0){process.stderr.write('MISSING_PARTITION_TABLE: Scenario Partitions section has no canonical header row\\n');process.exit(2);}
703
+ const pRows=pLines.slice(pH+2).map(pCells).filter(r=>r.length>=7&&r[0]&&r[0]!=='Partition ID');
704
+ for(const row of pRows){
705
+ const pid=String(row[0]||'').trim(),op=String(row[1]||'').trim(),domain=String(row[3]||'').split(/[;,,;]/).map(v=>v.trim()).filter(Boolean),optional=/omit/i.test(String(row[4]||'')),bind=String(row[6]||'').trim();
706
+ const prefix=/^SP-[A-Z0-9][A-Z0-9._-]*$/i.test(pid)?('TP-'+pid.toUpperCase()):('TP-SP-'+slotTok(op)+'-'+slotTok(String(row[2]||'')));
707
+ const slotIds=[...domain.map(v=>prefix+'-'+slotTok(v)),...(optional?[prefix+'-OMITTED']:[]),prefix+'-NOT-IN-SET'];
708
+ const owners=modules.filter(m=>(m.ownedRuleKeys||[]).includes(bind)||(m.ownedOperations||[]).some(o=>String(o).replace(/\\s+/g,' ').toUpperCase()===op.replace(/\\s+/g,' ').toUpperCase()));
709
+ const uniqueOwners=[...new Set(owners.map(o=>o.stem))];
710
+ if(uniqueOwners.length===1){for(const id of slotIds)slotByModule.get(uniqueOwners[0]).push(id);}
711
+ else if(modules.length===1){for(const id of slotIds)slotByModule.get(modules[0].stem).push(id);}
712
+ else unassigned.push(...slotIds);
713
+ }
714
+ }
715
+ if(unassigned.length){process.stderr.write('unassigned-partition-slots: '+unassigned.join(',')+'\\n');process.exit(2);}
716
+ const slotInventory={schemaId:'backend-test-scenario-partition-slots-v1',planSha256,modules:modules.map(m=>({stem:m.stem,requiredVariantSlots:[...new Set(slotByModule.get(m.stem)||[])]})),unassignedSlotIds:[]};
717
+ fs.mkdirSync(path.join(runDir,'contracts'),{recursive:true});
718
+ fs.writeFileSync(path.join(runDir,'contracts','backend-test-scenario-partition-slots.json'),JSON.stringify(slotInventory,null,2));
719
+ process.stdout.write(JSON.stringify({modules:modules.map(item=>({...item,planReadPath,requiredVariantSlots:(slotInventory.modules.find(m=>m.stem===item.stem)||{requiredVariantSlots:[]}).requiredVariantSlots})),planReadPath,planSha256,moduleLayout:{schemaId:'backend-test-module-layout-facts-v1',source:strictLayout?'task-contract':'plan-derived',contractSha256:strictLayoutSha256,mode:strictLayout&&strictLayout.mode||'business-resource-layout',planRepairAttemptCount:planRepairApplied?1:0,planRepairOutcome:planRepairApplied?'changed':'not-required'},partitionSlots:slotInventory}));
720
+ `;
721
+ const encoded = deflateRawSync(Buffer.from(script, "utf8")).toString("base64");
722
+ return `node -e "eval(require('zlib').inflateRawSync(Buffer.from('${encoded}','base64')).toString('utf8'))"`;
723
+ }
724
+ export function buildBackendTestAnalysisHybridDag(sources) {
725
+ const analysis = buildAnalyzeInputsNode(sources);
726
+ const contract = buildBackendTestAnalysisContractGateNode(sources);
727
+ const review = {
728
+ id: "review-backend-test-analysis-pi",
729
+ depends_on: [contract.id],
730
+ role: "reviewer",
731
+ executor: "pi",
732
+ complexity: "LOW",
733
+ writePolicy: "read-only",
734
+ allowedPaths: commonReadOnlyPaths(sources),
735
+ forbiddenPaths: commonForbiddenPaths(sources),
736
+ outputContract: "Concise Markdown review of the validated Backend Test Analysis v2 contract, covering source binding, explicit requirement coverage, evidence gaps, risks, and downstream readiness. No file writes.",
737
+ subtask_prompt: [
738
+ "Review the validated run-owned `contracts/backend-test-analysis.json` artifact.",
739
+ "Treat that structured artifact as the authoritative analysis contract; do not substitute the producer's assistant prose.",
740
+ "Check that every explicit AC/REQ/BR is represented by acceptanceCriteria or an evidenceGap, source references are precise, and unknown behavior remains an evidence gap rather than an invented requirement.",
741
+ "Summarize source-binding integrity, coverage, risks, evidence gaps, and whether the contract is ready for a later backend-test generation workflow.",
742
+ "Read-only: do not modify code, docs, artifacts, or repository files.",
743
+ ].join("\n\n"),
744
+ };
745
+ const closeout = {
746
+ id: "backend-test-analysis-closeout-static",
747
+ depends_on: [review.id],
748
+ role: "closeout",
749
+ executor: "static",
750
+ complexity: "LOW",
751
+ writePolicy: "none",
752
+ allowedPaths: [],
753
+ forbiddenPaths: commonForbiddenPaths(sources),
754
+ outputContract: "Deterministic handoff pointing to the validated analysis contract and its independent read-only review.",
755
+ subtask_prompt: "Close the read-only analysis workflow after the structured contract and review both succeed.",
756
+ static: {
757
+ resultMarkdown: [
758
+ "# Backend-test analysis closeout",
759
+ "",
760
+ "Validated contract: `contracts/backend-test-analysis.json`.",
761
+ "Independent review: `review-backend-test-analysis-pi`.",
762
+ ].join("\n"),
763
+ },
764
+ };
765
+ const spec = {
766
+ version: 3,
767
+ title: `Backend test analysis DAG: ${sources.taskConfig.title}`,
768
+ runtimeContract: GENERATED_DAG_RUNTIME_CONTRACT,
769
+ outputLanguage: sources.outputLanguage ?? DEFAULT_DAG_OUTPUT_LANGUAGE,
770
+ objective: extractObjective(sources.requirementMarkdown, sources.taskConfig.title),
771
+ successCriteria: [
772
+ "A strict Backend Test Analysis v2 contract is materialized with immutable source binding",
773
+ "Every explicit requirement is covered or preserved as an evidence gap",
774
+ "An independent read-only review consumes the validated structured artifact",
775
+ ],
776
+ globalConstraints: [
777
+ ...sources.taskConfig.hardConstraints,
778
+ "This workflow is analysis-only: no repository writer, dynamic expansion, test generation, or test execution is allowed.",
779
+ ],
780
+ defaults: {
781
+ ...BACKEND_TEST_DEFAULTS,
782
+ contextProfile: sources.taskConfig.contextProfile,
783
+ },
784
+ skillsByRole: BACKEND_TEST_SKILLS_BY_ROLE,
785
+ executorModels: sources.executorModelMatrix ?? DEFAULT_DAG_EXECUTOR_MODELS,
786
+ verifyStrategy: resolveDagVerifyStrategy(sources.taskConfig),
787
+ tasks: [analysis, contract, review, closeout],
788
+ };
789
+ applyDefaultReadOnlyRetryPolicy(spec);
790
+ stampGeneratedArtifactBindings(spec);
791
+ parseDagSpec(spec);
792
+ assertValidDagSpec(spec);
793
+ return spec;
794
+ }
795
+ export async function buildBackendTestGapFillDag(sources) {
796
+ const { taskConfig } = sources;
797
+ const ro = commonReadOnlyPaths(sources);
798
+ const forbidden = commonForbiddenPaths(sources);
799
+ const intake = await buildBackendTestIntakeContext(sources);
800
+ const layout = resolveBackendTestLayout(taskConfig.backendTest);
801
+ const sharedSetup = intake.sharedSetup;
802
+ const gapDocPath = taskConfig.backendTest?.gapDoc?.trim();
803
+ if (!gapDocPath) {
804
+ throw new Error("backendTest.mode=gap-fill requires backendTest.gapDoc pointing at the user missing-scenario document");
805
+ }
806
+ const envAllowlist = collectBackendTestShellEnvAllowlist(sources);
807
+ // N1: reuse the full-chain environment validation unchanged.
808
+ const environment = {
809
+ id: "validate-backend-test-environment-shell",
810
+ depends_on: [],
811
+ role: "verifier",
812
+ executor: "shell",
813
+ complexity: "LOW",
814
+ writePolicy: "read-only",
815
+ allowedPaths: ro,
816
+ forbiddenPaths: forbidden,
817
+ outputContract: "Run-owned reports/backend-test-environment.md with PASS/FAIL runtime, bounded project discovery, fixture and HTML-renderer facts; no secret values.",
818
+ subtask_prompt: "Fail fast before model work when Python/pytest cannot run in the clean shell. Inspect only bounded common config, conftest, test-root and server-entry candidates; never read .env values or credentials.",
819
+ shell: {
820
+ commands: ["python --version", "python -m pytest --version", "python -m pytest --help"],
821
+ backendTestPipeline: "markdown-environment",
822
+ cwd: ".",
823
+ timeoutMs: 60000,
824
+ envAllowlist,
825
+ },
826
+ };
827
+ // N2': deterministic gap ingest (shell pipeline owns parsing + planning).
828
+ const ingestGap = {
829
+ id: "ingest-backend-test-gap-shell",
830
+ depends_on: [environment.id],
831
+ role: "verifier",
832
+ executor: "shell",
833
+ complexity: "LOW",
834
+ writePolicy: "read-only",
835
+ allowedPaths: ro,
836
+ forbiddenPaths: forbidden,
837
+ outputContract: "Run-owned contracts/backend-test-gap-plan-v1.json with targetModules, newTestPoints, reuseSetupRefs, conflicts, alreadyPresent and exact targetPaths; identity mismatch (taskId/requirement hash) fails closed before any writer runs.",
838
+ subtask_prompt: "Parse the bound gap document (backend-test-gap-v1) and union it with previous coverage facts missingSlots; emit the deterministic gap plan. No file writes to testcase assets.",
839
+ shell: {
840
+ commands: [],
841
+ backendTestPipeline: "ingest-backend-test-gap",
842
+ cwd: ".",
843
+ timeoutMs: 60000,
844
+ envAllowlist,
845
+ },
846
+ };
847
+ // N3': patch README Matrix/Partitions for planned slots only.
848
+ const patchReadme = {
849
+ id: "patch-backend-readme-matrix-pi",
850
+ depends_on: [ingestGap.id],
851
+ role: "implementer",
852
+ executor: "pi",
853
+ toolProfile: "write",
854
+ complexity: "LOW",
855
+ writePolicy: "exclusive",
856
+ writeSet: [layout.readmePath],
857
+ allowedPaths: [layout.readmePath],
858
+ forbiddenPaths: forbidden,
859
+ writerOutcomePolicy: {
860
+ type: "implementation-outcome-v1",
861
+ requireChangedFiles: false,
862
+ },
863
+ retryPolicy: BACKEND_TEST_WRITER_COMPLETENESS_RETRY_POLICY,
864
+ outputContract: "First non-empty line is IMPLEMENTATION_OUTCOME: changed|already-satisfied|blocked. Patch ONLY the Coverage Matrix rows and Scenario Partitions table entries named by contracts/backend-test-gap-plan-v1.json newTestPoints; never rewrite unrelated README sections, never renumber Cases, never add Module Index modules.",
865
+ subtask_prompt: [
866
+ "Read contracts/backend-test-gap-plan-v1.json. For each planned slot in newTestPoints, patch the owning Coverage Matrix row (Required Test Points/Case IDs) and, for TP-SP slots, the Scenario Partitions table so the slot is declared. already-satisfied is valid when every planned slot is already present; changed requires a bounded diff limited to the slot rows.",
867
+ "When every planned slot is already declared, return already-satisfied without editing. Never expand scope beyond newTestPoints; never touch module Markdown or pytest files here.",
868
+ intake.boundedSourceContext,
869
+ ].join("\n\n"),
870
+ };
871
+ // N4': patch only the planned module Markdown files (map over targetModules).
872
+ const patchMdCases = {
873
+ id: "patch-backend-md-cases-map",
874
+ depends_on: [patchReadme.id],
875
+ role: "verifier",
876
+ executor: "static",
877
+ complexity: "LOW",
878
+ writePolicy: "none",
879
+ allowedPaths: [],
880
+ forbiddenPaths: forbidden,
881
+ outputContract: "Serial aggregate of per-module Markdown patch writers; each child writes exactly one planned testcase module file with a bounded append/patch diff.",
882
+ subtask_prompt: "Expand the gap plan targetModules into one sharded Markdown patch writer child per module and run them serially. Child failures fail-close the map barrier.",
883
+ static: {
884
+ resultMarkdown: "Backend-test gap-fill Markdown patch map expansion barrier.",
885
+ },
886
+ dynamicExpansion: {
887
+ type: "map_agent",
888
+ workflowNodeId: "patch-backend-md-cases-map",
889
+ itemsFrom: "$.nodes['ingest-backend-test-gap-shell'].json.plan.targetModules",
890
+ itemName: "item",
891
+ maxItems: 8,
892
+ maxExpandedNodes: 8,
893
+ childIdPrefix: "patch-backend-md-case",
894
+ tokenBudget: { maxTotalTokens: 1000000 },
895
+ failOnTokenBudgetExhaustion: true,
896
+ childTask: {
897
+ executor: "pi",
898
+ role: "implementer",
899
+ skills: BACKEND_TEST_SKILLS_BY_ROLE.implementer,
900
+ toolProfile: "write",
901
+ complexity: "LOW",
902
+ writePolicy: "exclusive",
903
+ allowedPaths: [`${layout.markdownDir}/{{item.stem}}.md`],
904
+ forbiddenPaths: forbidden,
905
+ writeSet: [`${layout.markdownDir}/{{item.stem}}.md`],
906
+ writerOutcomePolicy: {
907
+ type: "implementation-outcome-v1",
908
+ requireChangedFiles: false,
909
+ },
910
+ retryPolicy: BACKEND_TEST_WRITER_COMPLETENESS_RETRY_POLICY,
911
+ outputContract: "Patch exactly one planned module Markdown file: add missing slot Test Points (existing Case variant lists first; new BE-<MODULE>-<max+1> Case only when no Case can own the slot). Case IDs are never renumbered; sections follow the full-chain contract.",
912
+ subtaskPromptTemplate: [
913
+ "Read contracts/backend-test-gap-plan-v1.json and patch exactly testcase/md/{{item.stem}}.md. For every planned slot owned by this module: prefer appending the variant to the Case named by the plan (or the most related existing Case); create a new Case only when no existing Case can own it, numbering BE-<MODULE>-<NNN> from the module's current maximum +1. Keep every required h3 section; cite Matrix Rule Keys exactly; TP-SP slots use the exact deterministic slot IDs from the plan.",
914
+ sharedSetup
915
+ ? `When the slot needs shared pre-steps defined by the bound user document, reference them with 引用前置: ${sharedSetup.path}#<SS-ID|SS-DEFAULT> instead of duplicating steps.`
916
+ : "Prepare any needed state locally inside this Case (本地准备); no shared setup document is bound.",
917
+ "already-satisfied is valid when every planned slot of this module is already covered. Never modify other modules, README, or pytest files.",
918
+ intake.boundedSourceContext,
919
+ ].join("\n\n"),
920
+ },
921
+ },
922
+ };
923
+ // N5': validate patched Markdown with the same N6 pipeline (partition slots fail-closed).
924
+ const validateMd = {
925
+ id: "validate-backend-md-cases-shell",
926
+ depends_on: [patchMdCases.id],
927
+ role: "verifier",
928
+ executor: "shell",
929
+ complexity: "LOW",
930
+ writePolicy: "read-only",
931
+ allowedPaths: ro,
932
+ forbiddenPaths: forbidden,
933
+ outputContract: "Run-owned reports/backend-md-case-validation.md, coverage analysis and facts v4 with scenarioPartitions; planned slots must be covered (missingSlots for planned entries fail closed).",
934
+ subtask_prompt: "Record advisory findings for Markdown structure and deterministically analyze the final README Coverage Scope, Coverage Matrix and Scenario Partitions against final Case rule/test-point bindings. Planned gap slots that remain uncovered fail this node closed.",
935
+ shell: {
936
+ commands: [],
937
+ backendTestPipeline: "markdown-cases",
938
+ cwd: ".",
939
+ timeoutMs: 60000,
940
+ envAllowlist,
941
+ },
942
+ };
943
+ // N6': patch only the planned pytest modules (append params/functions).
944
+ const patchPytest = {
945
+ id: "patch-backend-pytest-cases-map",
946
+ depends_on: [validateMd.id],
947
+ role: "verifier",
948
+ executor: "static",
949
+ complexity: "LOW",
950
+ writePolicy: "none",
951
+ allowedPaths: [],
952
+ forbiddenPaths: forbidden,
953
+ outputContract: "Serial aggregate of per-module pytest patch writers; each child appends to exactly one planned testcase/test_<stem>.py without rewriting the file.",
954
+ subtask_prompt: "Expand the gap plan targetModules into one sharded pytest patch writer child per module and run them serially. Child failures fail-close the map barrier.",
955
+ static: {
956
+ resultMarkdown: "Backend-test gap-fill pytest patch map expansion barrier.",
957
+ },
958
+ dynamicExpansion: {
959
+ type: "map_agent",
960
+ workflowNodeId: "patch-backend-pytest-cases-map",
961
+ itemsFrom: "$.nodes['ingest-backend-test-gap-shell'].json.plan.targetModules",
962
+ itemName: "item",
963
+ maxItems: 8,
964
+ maxExpandedNodes: 8,
965
+ childIdPrefix: "patch-backend-pytest-case",
966
+ tokenBudget: { maxTotalTokens: 1000000 },
967
+ failOnTokenBudgetExhaustion: true,
968
+ childTask: {
969
+ executor: "pi",
970
+ role: "implementer",
971
+ skills: BACKEND_TEST_SKILLS_BY_ROLE.implementer,
972
+ toolProfile: "write",
973
+ complexity: "LOW",
974
+ writePolicy: "exclusive",
975
+ allowedPaths: [`${layout.scriptDir}/test_{{item.stem}}.py`],
976
+ forbiddenPaths: Array.from(new Set([
977
+ ...forbidden,
978
+ `${layout.markdownDir}/**`,
979
+ "conftest.py",
980
+ "pytest.ini",
981
+ "pyproject.toml",
982
+ "setup.cfg",
983
+ ])),
984
+ writeSet: [`${layout.scriptDir}/test_{{item.stem}}.py`],
985
+ writerOutcomePolicy: {
986
+ type: "implementation-outcome-v1",
987
+ requireChangedFiles: false,
988
+ },
989
+ retryPolicy: BACKEND_TEST_WRITER_COMPLETENESS_RETRY_POLICY,
990
+ outputContract: "Append-only patch of exactly one planned pytest module: add pytest.param rows / test functions for planned slots with exact TP ids; never rewrite or delete existing functions.",
991
+ subtaskPromptTemplate: [
992
+ "Read contracts/backend-test-gap-plan-v1.json and the patched module Markdown, then patch only testcase/test_{{item.stem}}.py. For each planned slot owned by this module: append an exact literal pytest.param(..., id=\"<slot-id>\") row to the owning Case's primary symbol, or append a new test function test_BE_<MODULE>_<NNN>_<desc> when a new Case was created. Existing functions, params and assertions must remain byte-stable; append-only edits.",
993
+ sharedSetup
994
+ ? "Reuse the module-top user_shared_setup fixture for shared pre-steps; never re-create documented setup steps inside test bodies, and never create or modify conftest.py."
995
+ : "Keep any needed setup local to the new function; never create or modify conftest.py.",
996
+ "already-satisfied is valid when every planned slot already has its exact pytest.param id collected. Do not modify Markdown or other modules.",
997
+ ].join("\n\n"),
998
+ },
999
+ },
1000
+ };
1001
+ // N7'-N13': reuse the deterministic collection/repair/gate/trace/manifest chain.
1002
+ const collectionAssess = {
1003
+ id: "assess-backend-pytest-collection-shell",
1004
+ depends_on: [patchPytest.id],
1005
+ role: "verifier",
1006
+ executor: "shell",
1007
+ complexity: "LOW",
1008
+ writePolicy: "read-only",
1009
+ allowedPaths: ro,
1010
+ forbiddenPaths: forbidden,
1011
+ outputContract: "Run-owned collection-v3 facts with repairPaths limited to the gap plan target pytest files.",
1012
+ subtask_prompt: "Resolve final Markdown-mapped scripts; scenario-param assess, pytest --collect-only and no-business-body fixture preflight. Repair eligibility is limited to planned pytest files.",
1013
+ shell: {
1014
+ commands: [],
1015
+ backendTestPipeline: "markdown-collection-assess",
1016
+ cwd: ".",
1017
+ timeoutMs: 120000,
1018
+ envAllowlist,
1019
+ },
1020
+ };
1021
+ const repairPytest = {
1022
+ id: "repair-backend-pytest-collection-pi",
1023
+ depends_on: [collectionAssess.id],
1024
+ runIf: "$.nodes['assess-backend-pytest-collection-shell'].json.repairEligible == true",
1025
+ role: "implementer",
1026
+ executor: "pi",
1027
+ toolProfile: "write",
1028
+ complexity: "MED",
1029
+ writePolicy: "exclusive",
1030
+ writeSet: [`${layout.scriptDir}/test_*.py`],
1031
+ allowedPaths: Array.from(new Set([...ro, layout.scriptDir])),
1032
+ forbiddenPaths: Array.from(new Set([
1033
+ ...forbidden,
1034
+ `${layout.markdownDir}/**`,
1035
+ "conftest.py",
1036
+ "pytest.ini",
1037
+ "pyproject.toml",
1038
+ "setup.cfg",
1039
+ ])),
1040
+ writerOutcomePolicy: { type: "implementation-outcome-v1", requireChangedFiles: true },
1041
+ outputContract: "First non-empty line is IMPLEMENTATION_OUTCOME: changed|blocked. Repair only generated pytest defects on initial facts repairPaths; preserve every Markdown Case, Test Point, primary symbol and assertion meaning.",
1042
+ subtask_prompt: [
1043
+ "Repair the generated backend pytest asset as one bounded program using the direct upstream collection assessment. Treat any upstream `Repair paths:` line as complete authoritative repairPaths evidence. Only planned testcase test_*.py files may change.",
1044
+ "Do not modify Markdown, conftest, pytest config, production code or dependencies. Do not add skip/xfail, remove tests, loosen assertions or replace the real API with mocks.",
1045
+ ].join("\n\n"),
1046
+ };
1047
+ const collectionEffective = {
1048
+ id: "effective-backend-pytest-collection-gate-shell",
1049
+ depends_on: [collectionAssess.id, repairPytest.id],
1050
+ role: "verifier",
1051
+ executor: "shell",
1052
+ complexity: "LOW",
1053
+ writePolicy: "read-only",
1054
+ allowedPaths: ro,
1055
+ forbiddenPaths: forbidden,
1056
+ outputContract: "Canonical backend-test-execution-readiness.json v2 bound to final asset hashes.",
1057
+ subtask_prompt: "Materialize effective collection facts; eligibility excludes non-exact mappings and unsafe payloads as in the full chain.",
1058
+ shell: {
1059
+ commands: [],
1060
+ backendTestPipeline: "markdown-collection-effective",
1061
+ cwd: ".",
1062
+ timeoutMs: 120000,
1063
+ envAllowlist,
1064
+ },
1065
+ };
1066
+ collectionEffective.dependsPolicy = "all-or-condition-skip";
1067
+ const traceability = {
1068
+ id: "backend-test-traceability-gate-shell",
1069
+ depends_on: [collectionEffective.id],
1070
+ role: "verifier",
1071
+ executor: "shell",
1072
+ complexity: "LOW",
1073
+ writePolicy: "read-only",
1074
+ allowedPaths: ro,
1075
+ forbiddenPaths: forbidden,
1076
+ outputContract: "Run-owned traceability and correspondence facts bound after effective collection.",
1077
+ subtask_prompt: "Deterministically scan final readiness-authorized Markdown-mapped pytest scripts; correspondence findings stay advisory.",
1078
+ shell: {
1079
+ commands: [],
1080
+ backendTestPipeline: "markdown-traceability",
1081
+ cwd: ".",
1082
+ timeoutMs: 60000,
1083
+ envAllowlist,
1084
+ },
1085
+ };
1086
+ const manifest = {
1087
+ id: "backend-test-case-manifest-shell",
1088
+ depends_on: [traceability.id],
1089
+ role: "verifier",
1090
+ executor: "shell",
1091
+ complexity: "LOW",
1092
+ writePolicy: "read-only",
1093
+ allowedPaths: ro,
1094
+ forbiddenPaths: forbidden,
1095
+ outputContract: "Run-owned contracts/backend-test-case-manifest.json materialized only from facts; the single machine input for L-5 and closeout.",
1096
+ subtask_prompt: "Materialize the canonical Backend Test Case Manifest only from coverage and correspondence facts; never re-analyze sources.",
1097
+ shell: {
1098
+ commands: [],
1099
+ backendTestPipeline: "markdown-manifest",
1100
+ cwd: ".",
1101
+ timeoutMs: 60000,
1102
+ envAllowlist,
1103
+ },
1104
+ };
1105
+ const execute = {
1106
+ id: "execute-backend-pytest-and-html-report-shell",
1107
+ depends_on: [manifest.id],
1108
+ role: "verifier",
1109
+ executor: "shell",
1110
+ complexity: "LOW",
1111
+ writePolicy: "read-only",
1112
+ allowedPaths: ro,
1113
+ forbiddenPaths: forbidden,
1114
+ outputContract: "One scoped pytest execution over readiness-authorized eligible items producing pytest-html plus the Chinese HTML report and L-5 dashboard; exit 0/1 with valid evidence continues.",
1115
+ subtask_prompt: `Execute readiness-authorized eligible pytest items once${taskConfig.backendTest?.executeScope === "all"
1116
+ ? " (executeScope=all: every eligible item)"
1117
+ : " (executeScope=affected: prefer newly added planned items when readiness marks them; otherwise every eligible item)"}, render the report, and keep evidence run-owned.`,
1118
+ shell: {
1119
+ commands: [],
1120
+ backendTestPipeline: "markdown-execute-html",
1121
+ cwd: ".",
1122
+ timeoutMs: taskConfig.backendTest?.executeTimeoutMs ?? 300000,
1123
+ envAllowlist,
1124
+ },
1125
+ };
1126
+ const report = {
1127
+ id: "backend-test-report-shell",
1128
+ depends_on: [execute.id],
1129
+ role: "verifier",
1130
+ executor: "shell",
1131
+ complexity: "LOW",
1132
+ writePolicy: "read-only",
1133
+ allowedPaths: ro,
1134
+ forbiddenPaths: forbidden,
1135
+ outputContract: "Run-owned backend test Result v1 with pytestExitCode, junit and evidence hashes; failure categories keep product/automation semantics.",
1136
+ subtask_prompt: "Materialize the backend-test Result v1 from the execution evidence; never fabricate metrics.",
1137
+ shell: {
1138
+ commands: [],
1139
+ backendTestPipeline: "markdown-report",
1140
+ cwd: ".",
1141
+ timeoutMs: 60000,
1142
+ envAllowlist,
1143
+ },
1144
+ };
1145
+ const spec = {
1146
+ version: 3,
1147
+ title: `Backend test gap-fill: ${sources.taskId}`,
1148
+ runtimeContract: GENERATED_DAG_RUNTIME_CONTRACT,
1149
+ outputLanguage: sources.outputLanguage ?? DEFAULT_DAG_OUTPUT_LANGUAGE,
1150
+ objective: `Incrementally fill backend-test gap scenarios for task ${sources.taskId} from the bound gap document without renumbering existing Cases.`,
1151
+ successCriteria: [
1152
+ "Every planned gap slot is declared in README (Matrix/Scenario Partitions), covered by a Case test point, and collected as an exact pytest.param id",
1153
+ "Existing Case IDs and unrelated modules remain byte-stable; no conftest.py is created or modified",
1154
+ "A second run over a satisfied plan returns already-satisfied with no asset diff",
1155
+ ],
1156
+ globalConstraints: [
1157
+ ...sources.taskConfig.hardConstraints.filter(Boolean),
1158
+ `backendTestGapDoc=${gapDocPath}`,
1159
+ `backendTestExecuteScope=${taskConfig.backendTest?.executeScope ?? "affected"}`,
1160
+ ],
1161
+ skillsByRole: BACKEND_TEST_SKILLS_BY_ROLE,
1162
+ executorModels: sources.executorModelMatrix ?? DEFAULT_DAG_EXECUTOR_MODELS,
1163
+ verifyStrategy: resolveDagVerifyStrategy(taskConfig),
1164
+ tasks: [
1165
+ environment,
1166
+ ingestGap,
1167
+ patchReadme,
1168
+ patchMdCases,
1169
+ validateMd,
1170
+ patchPytest,
1171
+ collectionAssess,
1172
+ repairPytest,
1173
+ collectionEffective,
1174
+ traceability,
1175
+ manifest,
1176
+ execute,
1177
+ report,
1178
+ ],
1179
+ };
1180
+ spec.backendTestLayout = layout;
1181
+ applyBackendTestWorkspaceControl(spec, taskConfig.backendTest?.workspaceControl ?? "git");
1182
+ if (sharedSetup) {
1183
+ spec.backendTestSharedSetup = { ...sharedSetup };
1184
+ }
1185
+ applyDefaultReadOnlyRetryPolicy(spec);
1186
+ stampGeneratedArtifactBindings(spec);
1187
+ parseDagSpec(spec);
1188
+ assertValidDagSpec(spec);
1189
+ return spec;
1190
+ }
1191
+ export async function buildBackendTestHybridDag(sources) {
1192
+ const { taskConfig } = sources;
1193
+ if (taskConfig.backendTest?.mode === "gap-fill") {
1194
+ return buildBackendTestGapFillDag(sources);
1195
+ }
1196
+ const ro = commonReadOnlyPaths(sources);
1197
+ const forbidden = commonForbiddenPaths(sources);
1198
+ const intake = await buildBackendTestIntakeContext(sources);
1199
+ // Plan A: resolve the frozen artifact layout once; every generated path
1200
+ // (README, module Markdown, pytest script, pytest target) derives from it.
1201
+ // Default config resolves to the historical testcase/ layout.
1202
+ const layout = resolveBackendTestLayout(taskConfig.backendTest);
1203
+ const applyLayout = (text) => applyBackendTestLayoutToText(text, layout);
1204
+ // Plan B: when the user binds exactly one shared-setup document, generated
1205
+ // Markdown references it (引用前置/SS-*) and each pytest module hoists the
1206
+ // documented steps into a module-top `user_shared_setup` fixture. Absent the
1207
+ // binding, every prompt below stays byte-identical to the pre-B baseline.
1208
+ const sharedSetup = intake.sharedSetup;
1209
+ const sharedSetupPrompt = sharedSetup
1210
+ ? [
1211
+ "## User shared setup document (bound, read-only)",
1212
+ `A user-provided shared pre-step document is bound at \`${sharedSetup.path}\` (readPath \`${sharedSetup.readPath}\`, sha256 ${sharedSetup.sha256}). Treat it as an authoritative read-only source: never rewrite, split or renumber it, and never invent steps absent from it.`,
1213
+ "When a Case needs a pre-step that this document already defines, the Case MUST reference it instead of duplicating the steps: in `### 前置条件` write `引用前置: <bound-path>#<SS-ID or heading>` lines (plus a `Shared Setup Refs` list of the referenced SS IDs). Extract stable step IDs only from explicit machine IDs (e.g. `SS-1`, `SS-REGISTER-01`); when the document is prose without stable IDs, reference the whole document as `引用前置: <bound-path>#SS-DEFAULT` and do not guess splits. Preparation unique to a Case stays inline but must be labeled `本地准备:`.",
1214
+ "Payload Contract still describes ONLY the target request; setup POST/PUT steps defined by the shared document never redefine the target Case payload contract.",
1215
+ ].join("\n\n")
1216
+ : null;
1217
+ const sharedSetupPytestPrompt = sharedSetup
1218
+ ? [
1219
+ "## User shared setup hoisting (module top)",
1220
+ `The user shared setup document \`${sharedSetup.path}\` (readPath \`${sharedSetup.readPath}\`) defines the common pre-steps. At the TOP of this module file (before any test function), generate exactly one module-scoped fixture named \`user_shared_setup\` implemented by a private helper \`_user_shared_setup()\` that performs the documented shared steps in document order, using the exact resource names/fields from the document. Cases reference these steps via their \`引用前置\` lines; each \`test_BE_*\` body then performs ONLY its target operation and must not repeat documented shared create/setup steps. Generate teardown code only when the user document explicitly documents a cleanup step; never invent a DELETE.`,
1221
+ "Duplication of this prefix across module files is accepted by design: do not extract it into conftest.py, a shared helper module, pytest_plugins, or a `_shared_setup.py`; the module file stays self-contained. NEVER create or modify conftest.py.",
1222
+ ].join("\n\n")
1223
+ : null;
1224
+ const shellNode = (id, depends_on, pipeline, prompt, outputContract, commands = [], timeoutMs = 60000) => ({
1225
+ id,
1226
+ depends_on,
1227
+ role: "verifier",
1228
+ executor: "shell",
1229
+ complexity: "LOW",
1230
+ writePolicy: "read-only",
1231
+ allowedPaths: ro,
1232
+ forbiddenPaths: forbidden,
1233
+ outputContract,
1234
+ subtask_prompt: prompt,
1235
+ shell: {
1236
+ commands,
1237
+ backendTestPipeline: pipeline,
1238
+ cwd: ".",
1239
+ timeoutMs,
1240
+ ...(commands.length
1241
+ ? { envAllowlist: collectBackendTestShellEnvAllowlist(sources) }
1242
+ : {}),
1243
+ },
1244
+ });
1245
+ const environment = shellNode("validate-backend-test-environment-shell", [], "markdown-environment", "Fail fast before model work when Python/pytest cannot run in the clean shell. Inspect only bounded common config, conftest, test-root and server-entry candidates; never read .env values or credentials.", "Run-owned reports/backend-test-environment.md with PASS/FAIL runtime, bounded project discovery, fixture and HTML-renderer facts; no secret values.", [
1246
+ "python --version",
1247
+ "python -m pytest --version",
1248
+ "python -m pytest --help",
1249
+ ]);
1250
+ // N2 line (sharded): README plan → manifest shell → map_agent barrier.
1251
+ // Each module Markdown card is written by an independent Pi child session
1252
+ // with its own 16K token budget, isolating single-large-file truncation risk.
1253
+ const generateMdPlan = {
1254
+ id: "generate-backend-md-plan-pi",
1255
+ depends_on: [environment.id],
1256
+ role: "planner",
1257
+ executor: "pi",
1258
+ toolProfile: "read-only",
1259
+ complexity: "MED",
1260
+ writePolicy: "read-only",
1261
+ allowedPaths: [],
1262
+ readSet: [
1263
+ toDagSourcePath(sources, sources.requirementPath),
1264
+ ...(sources.constraintMarkdown
1265
+ ? [toDagSourcePath(sources, sources.constraintPath)]
1266
+ : []),
1267
+ ...intake.referenceIndex.map((entry) => entry.readPath),
1268
+ ".harness/dag-runs/**/reports/backend-test-environment.md",
1269
+ ],
1270
+ forbiddenPaths: forbidden,
1271
+ retryPolicy: BACKEND_TEST_MD_PLAN_RETRY_POLICY,
1272
+ outputContract: "Return a Chinese, human-readable Markdown-first plan with Coverage Scope, Coverage Matrix, Scenario Partitions when applicable, and a machine-parseable Module Index. The runtime persists it as a run-owned Harness artifact; do not write testcase/md/README.md, execute pytest, or modify project files.",
1273
+ subtask_prompt: [
1274
+ "This is a required plan-generation node. Read only the strict read set and return the complete Markdown plan in the assistant response. Start the response with the final Markdown artifact immediately; never narrate analysis, reasoning, source summaries, or plans for producing the plan. The runtime persists the response as a run-owned Harness artifact named generate-backend-md-plan-pi/plan.md. Do not write testcase/md/README.md or any project file; module case cards are written by downstream sharded nodes.",
1275
+ "Output budget protocol (hard, max output <=16K per turn): Emit the complete required section skeleton before filling long table rows, including exactly one Module Index table with at least one module row. Never paste full Matrix, case bodies, source text, analysis or reasoning into assistant chat. README holds only Scope+Matrix+module index; never inline full case bodies. If a Completeness Gate / OUTPUT_LIMIT_RECOVERY retry is injected, continue only listed target paths.",
1276
+ "Return the complete plan as plain Markdown. Do not emit JSON or code fences. The plan must contain the exact English protocol headings ## Coverage Scope, ## Coverage Matrix, ## Scenario Partitions when applicable, and ## Module Index. Never translate those headings into 覆盖范围/覆盖矩阵/场景分区/模块索引.",
1277
+ "Read the upstream environment report only through the strict read set. Generate the Markdown-first backend test plan; it will be persisted under the current DAG run's Harness artifacts, not under testcase/md/.",
1278
+ "Write human-readable content in Simplified Chinese by default. Keep English protocol literals exact: section headings, table headers, Partition IDs, TP IDs, Case IDs, HTTP methods, paths, field names, enum values, commands, filenames, code symbols and source citations. Never translate ## Module Index into ## 模块索引.",
1279
+ "Create the concise plan entry page: test objective, target/environment, isolation/cleanup, module summary and a linked case index table with Case ID, Chinese case name, scenario type, endpoint and expected status/result. Avoid repeating every case body in the plan artifact.",
1280
+ "Before the Coverage Matrix, write a mandatory machine-readable `## Coverage Scope` section in the plan artifact using exactly `| Field | Value |`, immediately followed by the separator row `|---|---|`, and these six unique rows: `Change Classification`, `Coverage Policy`, `Affected Operations`, `Affected Rule Keys`, `Regression Floor`, `Scope Evidence`. Always set `Change Classification` to `new-operation` and `Coverage Policy` to `full-contract`; do NOT reason about whether operations are new or existing. Cover all in-scope rules from the requirement document at full depth; treat the product requirement as the coverage baseline and use API contract evidence (fields/status/enum/boundary/format) to supplement scenario dimensions. Scope is limited to operations/rules the requirement document (or its referenced API contract) explicitly describes; do not expand to unrelated operations that the requirement does not mention. List affected operations exactly as `METHOD /path`, stable rule keys separated by semicolons, and precise source pointers as Scope Evidence.",
1281
+ "Coverage depth is full over the in-scope rules: fully cover every documented status, request/response field rule, requiredness, enum, boundary, format, auth and business state of each affected operation the requirement describes, but do not re-test unrelated operations the requirement does not mention. Inspect shared validator/helper/DTO/query builder evidence and expand Affected Operations when the same affected path can affect them; unresolved impact stays visible as GAP/CONFLICT.",
1282
+ "Before writing cases, build the mandatory machine-readable Coverage Matrix inside the plan artifact itself. Its section heading line must be exactly `## Coverage Matrix` with no numeric prefix/suffix; never place the canonical Matrix only in a module file. Use this exact header: `| Rule Key | Priority | Source | Endpoint/Field | Dimension | Rule | Required Test Points | Case IDs | Status |`. Every data row must contain exactly 9 pipe-delimited cells and must never omit `Dimension`; use concise dimensions such as requirement, operation, response-status, requiredness, enum, boundary, format, business-state or error. Use only P0/P1/P2 and COVERED/PARTIAL/GAP/CONFLICT. Use stable `TP-<UPPERCASE-HYPHENATED-ID>` test points separated by semicolons.",
1283
+ "Each Rule Key must appear in exactly one Matrix row. Preserve each AC/REQ/BR Rule Key as one row; if one product rule spans multiple dimensions, use a concise composite Dimension in that single row instead of duplicating the key. Derive OpenAPI Rule Keys exactly as the deterministic analyzer does: operation token is `<HTTP-METHOD>-<PATH>` with braces removed and every non-alphanumeric run replaced by a hyphen, uppercase (for example POST `/api/resource-notes` → `POST-API-RESOURCE-NOTES`); response statuses use `API-<OPERATION>-RESPONSE-STATUS`; body/parameter fields use `API-<OPERATION>-<FIELD>-REQUIRED|ENUM|MIN-LENGTH|MAX-LENGTH|MINIMUM|MAXIMUM|PATTERN|FORMAT`. Do not invent aliases such as API-CREATE-FIELDS when a deterministic key applies.",
1284
+ "Coverage priority is strict inside the declared scope: P0 product requirements/task hard constraints always remain in scope; P1 exhaustively supplements documented operations, fields, business rules, statuses and errors only for Affected Operations; P2 adds bounded protocol robustness only when it is relevant to the change and does not invent product behavior. Coverage percentages describe the declared affected scope, never whole-API completeness unless every operation is explicitly listed. Conflicts or undefined expectations must stay visible as GAP/CONFLICT with precise source pointers, never guessed.",
1285
+ "For uniqueness/lifecycle rules cover absent, active-existing, deleted-existing, create-delete-recreate, restore-then-recreate and documented scope/case-normalization states. For every enum cover every valid value plus bounded invalid equivalence classes (unknown, case variant, whitespace, empty, null/missing and wrong types as applicable). For every length/number rule cover min-1, min, nominal, max and max+1. For format rules cover each allowed class separately plus a valid mixed value, and representative forbidden classes including uppercase, internal/leading/trailing whitespace, tab/newline, unsupported punctuation, slash, emoji or control characters when the source contract supports that expectation.",
1286
+ "Mandatory module layout contract: first inspect the PRIMARY requirement for an explicit list of required Markdown/Python output path pairs. When explicit paths are present, they are authoritative `explicit-user-layout`: reproduce their exact filenames, count and one-to-one pairs in Module Index; do not rename, merge, split, omit or add a module from reference/example scripts. Only when the primary requirement has no explicit file layout may you derive the smallest `business-resource-layout`. Existing examples, historical regression functions and shared setup may add evidence/assertions to an existing required module, but never create an extra physical module by themselves. A reference-only `resp_regression`, positive/negative/boundary/error/response module is forbidden. The Module Stem cell must contain only the plain filename stem; never put Markdown link syntax or a path in that cell.",
1287
+ ...(taskConfig.backendTest?.moduleLayout
1288
+ ? [
1289
+ "A strict task-contract `backendTest.moduleLayout` is bound and is authoritative over model-derived layout. Reproduce every stem, businessResource, ownedOperations, splitReason, markdownPath and pytestPath exactly; do not add, omit, rename or reorder physical modules. The downstream preflight compares exact path sets and may perform at most one deterministic Plan-only pruning/path normalization; it cannot invent missing Rule/Case ownership.",
1290
+ `STRICT_BACKEND_TEST_MODULE_LAYOUT=${JSON.stringify(taskConfig.backendTest.moduleLayout)}`,
1291
+ ]
1292
+ : []),
1293
+ `Include exactly one \`## Module Index\` table with this exact header: \`| Module Stem | Business Resource | Owned Operations | Owned Rule Keys | Case IDs | Split Reason | Markdown Path | Pytest Path |\`. The Markdown Path cell must contain exactly one resolved repository-relative path such as \`${layout.markdownDir}/health.md\`; do not emit a Markdown link or repeat the path. Split Reason is exactly one of \`explicit-user-layout\`, \`primary-business-resource\`, \`independent-business-resource\`, \`output-budget\`. Group by stable business resource/domain, not by CRUD operation, AC, parameter/field axis, scenario type or regression purpose: one resource's list/detail/create/update/delete and its filters/response assertions/regression floor belong in one module. Multiple modules owning the same exact \`METHOD /path\` are forbidden unless every such row is \`explicit-user-layout\` from primary-requirement path pairs or has a documented \`output-budget\` proof. Keep the total module count at the smallest safe value and never exceed 8 modules. Name model-derived modules with stable lowercase business stems such as \`health\` or \`resource_notes\`; explicit-user-layout preserves the primary requirement filename stem even when it is more specific. Do not use priority-only stems \`p0\`, \`p1\` or \`p2\`; Priority belongs only in the Coverage Matrix. Pure hexadecimal/hash-like opaque stems and test-purpose-only stems are forbidden. Do not use Case-ID-like module filenames. Markdown Path, Pytest Path and downstream automation mapping must be one-to-one and exact; for model-derived modules the default pair remains \`${layout.markdownDir}/<module>.md\` and \`${layout.scriptDir}/test_<module>.py\`, while explicit-user-layout preserves the primary requirement paths. Do not hand-write a conflicting module count in prose; the Module Index row count is the only count truth.`,
1294
+ "Scenario Partitions (query/filter axes): inspect every affected GET/list operation for query/path parameters whose bound source documents a finite enum or classification domain. If at least one such axis exists, add exactly one machine-readable `## Scenario Partitions` section after the Coverage Matrix using exactly `| Partition ID | Operation | Axis | Domain | Required Slots | Expected by Slot | Bind Rule |` with the separator row and one row per eligible axis. If no affected axis has a source-backed finite domain, omit the entire `## Scenario Partitions` heading and section; do not emit an explanatory prose-only section. Partition ID is a stable `SP-<OPERATION>-<AXIS>` token; Domain must copy the legal values verbatim from the bound OpenAPI enum or requirement sentence (never guess), using bare semicolon-separated identifier values inside the single table cell (for example `ACTIVE; ARCHIVED`, with no Markdown backticks or prose); Required Slots must contain `each-value` and exactly one `not-in-set`, plus `omitted` only when the parameter is optional. Before returning, expand every declared partition into its complete deterministic exact slot ID set: one `TP-<Partition ID>-<VALUE-TOKEN>` per Domain value, `TP-<Partition ID>-OMITTED` only for an optional axis, and exactly one `TP-<Partition ID>-NOT-IN-SET`. Every expanded slot ID must appear verbatim in the binding Rule's `Required Test Points` cell and be assigned to concrete Case IDs in that same Coverage Matrix row; ordinary alias/family Test Points do not replace this inventory. Scheme A: Case count may be smaller than the enum count, but every exact slot still needs an independent variant Test Point and pytest.param id; never use SINGLE/MULTIPLE aliases as coverage. Expected by Slot states the documented expectation per slot kind (`domain-value`, `default-behavior`, `empty-result`/`excluded-result` when documented, or `GAP` when the source does not document the complement expectation — never invent 空列表/400). POST/PUT body field-validation enums stay in the Coverage Matrix as `TP-<FIELD>-ENUM-*` and MUST NOT get a Scenario Partition row. Do not create partitions for axes without a documented legal-value domain. Only GET/list query or path parameters whose bound source documents a finite enum or classification set may become a Scenario Partition. Do not create partitions for free-form strings, primary keys, required-or-optional-only parameters, or boundary/format-only axes. If an axis has no finite legal-value domain, do not declare a Partition row and do not invent NOT-IN-SET cases. Cross-axis combinations stay as ONE nominal Case; never declare a cross-axis cartesian partition.",
1295
+ "Before finalizing README, calculate the predicted collected-item count as `sum(max(1, number of variant Test Points in each Case))`. If the task declares an item budget, the prediction must not exceed it. Reduce excess only by removing duplicate execution and converting same-request checkpoints to assertions; never drop required rules, boundaries, enums, operation-specific inputs, or business states. Record the prediction in README. Use only environment-supported fixtures/targets/isolation, record evidence gaps in Chinese, and do not emit JSON, pytest, or execute commands.",
1296
+ ...(sharedSetupPrompt ? [sharedSetupPrompt] : []),
1297
+ intake.boundedSourceContext,
1298
+ "## Authoritative reference index",
1299
+ JSON.stringify(intake.referenceIndex, null, 2),
1300
+ "For each index entry, use `readPath` for Pi read-tool calls and copy `path` exactly into Markdown Source References. Bound files under .harness/tasks/<taskId>/source/** are read-only inputs: reading them is allowed even though writing .harness/** is forbidden. Never resolve `path` relative to the repository root, search for substitutes, or fall back to docs/** when a bound read fails.",
1301
+ "Read only precise indexed references needed for AC/API/field/rule evidence; references remain authoritative over derived text.",
1302
+ ].join("\n\n"),
1303
+ };
1304
+ const materializeMdManifest = {
1305
+ id: "materialize-backend-md-module-manifest-shell",
1306
+ depends_on: [generateMdPlan.id],
1307
+ role: "verifier",
1308
+ executor: "shell",
1309
+ complexity: "LOW",
1310
+ writePolicy: "read-only",
1311
+ allowedPaths: ro,
1312
+ forbiddenPaths: forbidden,
1313
+ outputContract: "Stdout JSON {modules:[{stem,planReadPath}],planReadPath,planSha256,moduleLayout} parsed from the run-owned generate-backend-md-plan-pi/plan.md artifact after strict-layout validation and at most one deterministic Plan-only repair.",
1314
+ subtask_prompt: "Parse only $HARNESS_DAG_RUN_DIR/generate-backend-md-plan-pi/plan.md, validate the optional strict module layout and output-budget proof, apply at most one deterministic Plan-only Module Index repair without project writes, then emit one JSON line. No testcase/md/README.md fallback.",
1315
+ shell: {
1316
+ commands: [buildBackendTestModuleManifestShellCommand(layout, taskConfig.backendTest?.moduleLayout)],
1317
+ cwd: ".",
1318
+ timeoutMs: 60000,
1319
+ },
1320
+ };
1321
+ const generateMdCasesMap = {
1322
+ id: "generate-backend-md-cases-map",
1323
+ depends_on: [materializeMdManifest.id],
1324
+ role: "verifier",
1325
+ executor: "static",
1326
+ complexity: "LOW",
1327
+ writePolicy: "none",
1328
+ allowedPaths: [],
1329
+ forbiddenPaths: forbidden,
1330
+ outputContract: "Serial aggregate of sharded Markdown module case-card writers. Each child writes exactly one testcase/md/<stem>.md with its own 16K Pi budget.",
1331
+ subtask_prompt: "Expand the README module manifest into one sharded Markdown writer child per module and run them serially. Child failures fail-close the map barrier.",
1332
+ static: {
1333
+ resultMarkdown: "Backend-test Markdown case-card map expansion barrier.",
1334
+ },
1335
+ dynamicExpansion: {
1336
+ type: "map_agent",
1337
+ workflowNodeId: "generate-backend-md-cases-map",
1338
+ itemsFrom: "$.nodes['materialize-backend-md-module-manifest-shell'].output.modules",
1339
+ itemName: "item",
1340
+ maxItems: 8,
1341
+ maxExpandedNodes: 8,
1342
+ childIdPrefix: "generate-backend-md-case",
1343
+ tokenBudget: { maxTotalTokens: 3000000 },
1344
+ failOnTokenBudgetExhaustion: true,
1345
+ childTask: {
1346
+ executor: "pi",
1347
+ role: "implementer",
1348
+ skills: BACKEND_TEST_SKILLS_BY_ROLE.implementer,
1349
+ toolProfile: "write",
1350
+ complexity: "MED",
1351
+ writePolicy: "exclusive",
1352
+ allowedPaths: ["{{item.markdownPath}}"],
1353
+ forbiddenPaths: forbidden,
1354
+ writeSet: ["{{item.markdownPath}}"],
1355
+ readSet: [
1356
+ "{{item.planReadPath}}",
1357
+ toDagSourcePath(sources, sources.requirementPath),
1358
+ ...(sources.constraintMarkdown
1359
+ ? [toDagSourcePath(sources, sources.constraintPath)]
1360
+ : []),
1361
+ ...intake.referenceIndex.map((entry) => entry.readPath),
1362
+ "{{item.markdownPath}}",
1363
+ ],
1364
+ writerOutcomePolicy: {
1365
+ type: "implementation-outcome-v1",
1366
+ requireChangedFiles: true,
1367
+ },
1368
+ retryPolicy: BACKEND_TEST_MARKDOWN_BINDING_RETRY_POLICY,
1369
+ outputContract: "Write exactly the frozen `{{item.markdownPath}}` Chinese module Markdown case-card file with BE-<MODULE>-<NNN> cases and the seven required h3 sections; keep machine IDs/literals exact and do not execute pytest or modify production code/config or the README.",
1370
+ subtaskPromptTemplate: [
1371
+ "This is a required file-generation node for exactly one Markdown module. Read the upstream run-owned Markdown plan artifact at `{{item.planReadPath}}` (Coverage Scope + Coverage Matrix + Module Index) and the bounded references, then immediately use write tools to create the single frozen file `{{item.markdownPath}}`. Do not read or recreate testcase/md/README.md. Do not end after analysis or planning, and do not return before a non-empty bounded diff exists. Do not modify any other module file.",
1372
+ "Output budget protocol (hard, max output <=16K per turn): Never paste full Matrix, other modules' case bodies, or source text into assistant chat. Each write/edit tool call touches at most one file (this module). Compact tables/lists are required; omitting required sections or in-scope variants is forbidden. If a Completeness Gate / OUTPUT_LIMIT_RECOVERY retry is injected, continue only listed target paths.",
1373
+ "The first non-empty response line must be exactly IMPLEMENTATION_OUTCOME: changed after the module file has been written, or IMPLEMENTATION_OUTCOME: blocked when precise missing evidence prevents safe generation. already-satisfied is not valid for this node.",
1374
+ "Write human-readable content in Simplified Chinese by default. Keep English only for machine-readable IDs and technical literals such as Case/AC/REQ/BR IDs, HTTP methods, paths, field names, enum values, commands, filenames, code symbols and exact source citations.",
1375
+ 'Write the module {{item.stem}} as readable case cards covering every in-scope rule/Test Point the README Coverage Matrix assigns to this module. Every case starts with `## BE-<MODULE>-<NNN>|<中文用例名称>`. `<NNN>` is exactly three zero-padded digits (`001`, `002`, ...), never two digits (`01`), a bare number, or an alphabetic suffix such as `011A`. Every case must include `### 覆盖规则`, `### 测试点`, `### 场景类型`, `### 前置条件`, `### 操作步骤`, `### 预期结果`, and `### 自动化映射` Do not group cases under "## 测试类 ..." (or any h2 grouping) headings that force Cases down to h3; each Case must be a direct h2 (`##`), and its seven sections must be h3 (`###`) children of that Case. If you need to convey a pytest class, state it inside the Case\'s `### 自动化映射` instead. Forbidden: `## 测试类 X` then `### BE-PD-001` and `### 覆盖规则` at the same h3 level. Required: `## BE-PD-001` then `### 覆盖规则`.; `覆盖规则` and `测试点` must reference exact Matrix Rule Keys/Test Points. Add `测试目的`, `验收标准`, `需求依据`, and `测试数据` for readable evidence. The `验收标准` section must list the exact applicable `AC-...` IDs, and every explicit task AC must appear in at least one Case. Every automatable case explicitly names its target pytest script and exactly one primary symbol so traceability scans only that script/symbol. Evidence-only meta cases that exist solely for non-executable assertion/cross-cutting process evidence may declare `脚本:无` and `primary symbol:无` with empty `变体测试点`, and must not invent a business pytest item.',
1376
+ "Name this module file with the exact frozen Module Index stem `{{item.stem}}` (filename `{{item.markdownPath}}`). Never reinterpret or rename an explicit-user-layout stem. Priority-only stems `p0`, `p1` and `p2` are forbidden and must never produce `p0.md` or `test_p0.py`. Pure hexadecimal/hash-like opaque stems such as `a401606` and `deadbeef` are also forbidden. Do not use Case-ID-like module filenames. For every automatable case, `自动化映射` must name exactly `{{item.pytestPath}}`, where the module stem is this Markdown filename without `.md`, lowercased, with non-alphanumeric characters replaced by underscores. Example: `health` → `testcase/test_health.py`; `resource_notes` → `testcase/test_resource_notes.py`. Never invent a different pytest path in Markdown than the module stem implies.",
1377
+ "AUTOMATION_BINDING_FORMAT_V1 is a literal machine contract. Under every Case's `### 自动化映射`, write these independent lines exactly: `- 脚本:<path|无>`, `- primary symbol:<symbol|无>`, `- 变体测试点:<semicolon-separated TP IDs|无>`, `- 场景断言测试点:<semicolon-separated TP IDs|无>`, `- 横切证据测试点:<semicolon-separated TP IDs|无>`. TP IDs must be on the same line after the colon. Forbidden classification forms include `TP-X(变体测试点)`, `[变体测试点] TP-X`, `【变体测试点】:TP-X`, pipe-delimited annotations, tables, or nested TP lists. Before returning, verify that the Case `### 测试点` exact set equals the pairwise-disjoint union of the three canonical binding lines; do not add, remove, rename or duplicate a TP to make the format pass.",
1378
+ "Scenario Partition slots: requiredVariantSlots for this module are `{{item.requiredVariantSlots}}`. Every listed ID MUST appear in this module's Cases as exactly one variant Test Point in both `### 测试点` and `变体测试点`. Slot IDs copy the declared Partition ID exactly; never drop the HTTP method, invent, merge, renumber or split slot IDs. Ordinary alias Test Points do not satisfy a partition slot, and aggregate aliases such as `SINGLE`/`MULTIPLE` are forbidden. Scheme A: one Case may carry many slots; do not create one Case per enum value just to match Case count. Prefer ONE Case per partition with a parameter table over duplicated Cases per value. The not-in-set slot value must be a concrete literal absent from the Domain (e.g. `UNKNOWN_TYPE`) and its expected result must come from the bound source — when Expected by Slot is GAP, the Case states the expectation as GAP evidence, never a guessed 空列表/400. Never create cross-axis combination variants beyond the single documented nominal.",
1379
+ "For every variant Test Point, write its machine-checkable `场景意图: <TP-ID>; operation=...; target=...; intent=...` line inside that same Case body/自动化映射. Never collect Scenario Intent lines in a file-level appendix, implementation-details block, or another Case; local TP ownership is mandatory.",
1380
+ "Every Case must keep at least one numbered executable line under `### 操作步骤`; a compact variant/result table may follow but must not replace the numbered action anchor. Keep numbered/bulleted independently assertable results under `### 预期结果`. The exact `### 操作步骤` and `### 预期结果` headings must remain present for every Case, including compact/table-based Cases; never compress later Cases by dropping required headings. Every result must name the observable HTTP status, response field/value, state transition or membership condition, never vague wording such as ‘符合预期’.",
1381
+ "In every `自动化映射`, use exactly these machine-readable list labels: `脚本`, `primary symbol`, `变体测试点`, `场景断言测试点`, `横切证据测试点`, plus a deterministic payload contract. For operations without a request body write `Payload Contract: none`. Otherwise write `Payload Required Paths`, `Payload Allowed Paths`, and `Payload Enum` (write `none` when there is no enum); nested fields use dot paths such as `approver.name`. Each Case describes exactly one target request payload contract: put every payload label on its own list line, never concatenate multiple operations or setup POST/PUT contracts into one label line, and never repeat a `Payload Contract:` token inside explanatory prose/details after the machine-readable line. Values must come only from bound API/DTO evidence, never guesses. Each Test Point from `### 测试点` must appear in exactly one binding list, and every Test Point named in any binding list must also be declared in that Case's `### 测试点`; write `无` for an empty list. A variant Test Point is atomic: one exact endpoint/input/precondition/outcome row equals one exact pytest item and one exact TP ID. If a parameter table has five rows, declare five distinct variant TP IDs in Markdown; never declare one family TP and append row suffixes only in pytest. Classify as `variant` only when endpoint, request input, precondition business state, or expected outcome genuinely changes and therefore needs an independent pytest parameter item. Classify CRUD checkpoints, status/body/header/schema assertions and multiple checks over the same response/journey as `assertion`; classify shared HTTP logging/redaction/truncation evidence as `cross-cutting`. Never create a Test Point merely to parameterize a checkpoint. Every non-cross-cutting TP ID is owned by exactly one Case; when the same response/schema/error assertion is needed in different Cases, use distinct Case-specific TP IDs instead of reusing one assertion TP across Cases. Keep the script path identical to the module one-to-one path and declare exactly one primary symbol named with the canonical Case prefix, for example `BE-RN-003` → `test_BE_RN_003_<description>`; non-Case-prefixed primary symbols are forbidden because parameterized item association must remain deterministic. For evidence-only meta Cases with no executable business journey, write `脚本:无` and `primary symbol:无`, keep `变体测试点:无`, and place process evidence only in assertion/cross-cutting lists. If the bound contract only says an identifier is returned/present, do not declare a concrete identifier type. If a 404 Case needs a nonexistent path identifier but its syntax/type is unspecified, define a create-delete-derived valid identifier journey instead of an arbitrary UUID/text placeholder. For redaction scenarios, list sensitive header/field key names only. Never write any header-name-and-value pair, credential placeholder, fake token, anti-example, or other secret-shaped literal in Markdown; state only that a test-only value is supplied at runtime and omitted. Put implementation-only restrictions in a concise `<details>` block rather than dominating the main case flow. Use only environment-supported fixtures/targets/isolation, record evidence gaps in Chinese, and do not emit JSON, pytest, or execute commands.",
1382
+ ...(sharedSetupPrompt ? [sharedSetupPrompt] : []),
1383
+ intake.boundedSourceContext,
1384
+ "## Authoritative reference index",
1385
+ JSON.stringify(intake.referenceIndex, null, 2),
1386
+ "For each index entry, use `readPath` for Pi read-tool calls and copy `path` exactly into Markdown Source References. Bound files under .harness/tasks/<taskId>/source/** are read-only inputs: reading them is allowed even though writing .harness/** is forbidden. Never resolve `path` relative to the repository root, search for substitutes, or fall back to docs/** when a bound read fails.",
1387
+ "Read only precise indexed references needed for AC/API/field/rule evidence; references remain authoritative over derived text.",
1388
+ ].join("\n\n"),
1389
+ },
1390
+ },
1391
+ };
1392
+ const reviewCases = {
1393
+ id: "review-and-revise-backend-md-cases-pi",
1394
+ depends_on: [generateMdCasesMap.id, materializeMdManifest.id],
1395
+ role: "implementer",
1396
+ executor: "pi",
1397
+ toolProfile: "write",
1398
+ complexity: "MED",
1399
+ writePolicy: "exclusive",
1400
+ writeSet: [`${layout.markdownDir}/**`],
1401
+ readSet: [
1402
+ ".harness/dag-runs/**/generate-backend-md-plan-pi/plan.md",
1403
+ toDagSourcePath(sources, sources.requirementPath),
1404
+ ...(sources.constraintMarkdown
1405
+ ? [toDagSourcePath(sources, sources.constraintPath)]
1406
+ : []),
1407
+ ...intake.referenceIndex.map((entry) => entry.readPath),
1408
+ `${layout.markdownDir}/*.md`,
1409
+ ],
1410
+ allowedPaths: Array.from(new Set([...ro, `${layout.markdownDir}/**`])),
1411
+ forbiddenPaths: Array.from(new Set([...forbidden, layout.readmePath])),
1412
+ writerOutcomePolicy: { type: "implementation-outcome-v1" },
1413
+ outputContract: "First non-empty line is IMPLEMENTATION_OUTCOME: changed|already-satisfied|blocked. Perform exactly one bounded incremental synchronization of testcase/md/** against all bound source references; preserve valid Cases and report a concise summary.",
1414
+ subtask_prompt: [
1415
+ "Perform one gap-targeted synchronization, not a full-suite rewrite or stylistic review. Read the immutable run-owned Markdown plan from the direct upstream manifest's planReadPath. Start from explicit bound source IDs/error codes/DTO fields/normative quoted rules and the plan Coverage Matrix; open and edit only modules that own a missing or conflicting rule. Never create or edit testcase/md/README.md and never modify the run-owned plan artifact. Preserve unrelated valid modules byte-for-byte and avoid optional wording cleanup.",
1416
+ "Output budget protocol: never dump full Matrix/case bodies into assistant chat. Inspect the immutable run-owned plan first, build a concise target list from its Matrix and Module Index, then read/write only target modules one file per tool call. Do not traverse every module when the Matrix and source token inventory show no gap; return `already-satisfied`. When adding omitted in-scope cases, keep every required section. Do not bulk-delete in-scope cases to save tokens.",
1417
+ "For every variant Test Point, ensure the Markdown scenario intent is machine-checkable and located inside that same Case body/自动化映射, never in a file-level appendix, implementation-details block, or another Case. Use an exact transport target: `场景意图: <TP-ID>; operation=<METHOD /path>; target=<body.field|query.field|path.field|header.field|request>; intent=<empty|missing|null|min-1|min|max|max+1|pattern-invalid|enum-invalid|wrong-type|nominal-operation|custom-literal:V>; bound=<n optional>; example=<optional>; expectedCode=<optional>`. Never use vague targets such as field=resource/health. Keep pytest params aligned to the exact target. For intent=missing/empty/default-omit, pytest may use `_OMIT` or delete the key; for intent=enum-invalid use a concrete invalid enum literal (for example `UNKNOWN_STATUS`), never `_OMIT`/missing-key; for trim/padded samples use `custom-literal:trim` or a real padded string, not a bare token like `filter-active` when the intent is `custom-literal:ACTIVE`.",
1418
+ "Treat the requirement document as the coverage baseline; scope is limited to operations/rules it (or its referenced API contract) describes, and API contract evidence supplements scenario dimensions. For every in-scope operation, check applicable lifecycle/uniqueness states (including deleted-existing when in scope), valid enum values, bounded invalid classes, min-1/min/nominal/max/max+1, allowed/forbidden format classes, required/null/missing/wrong-type semantics, status/error codes, auth and state transitions. Inspect shared validator/helper/DTO/query builder evidence and expand Affected Operations when the same affected path can affect them; unresolved impact stays visible as GAP/CONFLICT. Directly add in-scope omissions; reject scope expansion to operations absent from the requirement document; undefined impact remains GAP/CONFLICT rather than invented behavior.",
1419
+ "Check AC completeness/meaning, endpoint, fields/shape, status/error codes, rules, states, documented boundaries/auth, positive/negative coverage, executable steps and assertable results. Require the exact `## Coverage Scope` Field/Value table with the `|---|---|` separator row, a valid classification-policy pair, non-empty Affected Operations/Rule Keys/Scope Evidence, and the classification-specific Regression Floor. Require the exact unnumbered `## Coverage Matrix` heading in the immutable run-owned plan artifact, exact headers, exactly 9 cells in every data row (including a non-empty Dimension), deterministic OpenAPI Rule Keys for every in-scope affected operation, exactly one Matrix row per Rule Key (merge multi-dimension product rows), and bidirectional Matrix Rule/Test Point ↔ Case bindings. Never describe affected-scope coverage as whole-API completeness. Every explicit AC ID must appear in at least one Case `验收标准`; every explicit in-scope AC/REQ/BR Rule Key cited by a Case must have exactly one Coverage Matrix row, and no Case may cite a source Rule Key omitted from the Matrix. Every Matrix Case ID must share at least one of that row's Required Test Points and the Case must cite that Rule Key. Perform an explicit execution-redundancy review: merge checkpoint-only parameter rows, repeated default/read-back assertions, DELETE status/body/follow-up-read checks, response schema/Content-Type checks, PUT full-update/timestamp checks, repeated list setup and identical null/empty inputs when endpoint, input partition, precondition state and expected outcome are the same. Preserve separate POST/PUT, boundary, enum, wrong-type, role/tenant and distinct business-state variants. Directly repair malformed headings/rows/keys and binding modes rather than merely commenting on them. Reject avoidable English prose, duplicated bilingual wording, repeated boilerplate, oversized unstructured sections, a `### 操作步骤` section that contains only a table without any numbered executable line, vague results such as ‘符合预期’, Case-ID-like module filenames (for example `BE-HEALTH.md`), dropped exact `### 操作步骤`/`### 预期结果` headings, and missing or drifted script/function mapping where it can be derived.",
1420
+ "Correct testcase/md/** directly: add documented omissions, remove unsupported cases, preserve every frozen Module Index filename exactly (never rename an explicit-user-layout module; model-derived invalid stems must have been rejected before map expansion), normalize every Case ID to hyphen-separated module segments plus exactly three zero-padded digits (`BE-RESOURCE_NOTES-01` → `BE-RESOURCE-NOTES-001`; `BE-RN-011A` must be renumbered or merged) consistently across headings/index/mappings, fix automation mappings so each automatable case points at `testcase/test_<module>.py` derived from that module filename and declares exactly one primary symbol (evidence-only meta cases may keep `脚本/primary symbol=无` with empty variants), assign every Test Point exactly one of `变体测试点`/`场景断言测试点`/`横切证据测试点`, then perform an exact-set check: each Case's `### 测试点` set must equal (not merely contain) the union of those three binding lists; delete stale/legacy aliases and ensure every binding-list Test Point is present, expand every variant parameter row into its own atomic TP ID, make every non-cross-cutting TP Case-specific and owned by exactly one Case, require every primary symbol to start with the canonical Case prefix, ensure every explicit AC ID appears in an applicable Case `验收标准`, merge execution duplicates, improve navigation/tables/Chinese wording, or record gaps in Chinese. Remove every credential/header value, placeholder, fake token and anti-example from Markdown. Sensitive key names may remain only as a plain list; values must be described as runtime-only and omitted, with no colon/value pair or literal example anywhere, including details blocks and explanatory text. Keep Case IDs, AC/REQ/BR IDs, HTTP methods, paths, fields, enum values, filenames, code symbols and source citations as exact machine-readable identifiers; only normalize Case ID separator/sequence formatting as specified above. Recalculate predicted collected items as `sum(max(1, variant count per Case))`; when the task declares a budget, directly merge redundant journeys/reclassify same-request checkpoints until the prediction is within budget, while preserving all required coverage. The validator accepts Chinese and legacy English section aliases; retain or converge to the Chinese human-readable headings without losing structure.",
1421
+ "This is the single Markdown incremental synchronization round. Read every authoritative reference index entry whose role hints include acceptance-criteria, api-contract, data-contract or business-rule; do not rely on the derived PRD as a complete inventory. Preserve every explicit AC/REQ/BR ID, every documented HTTP/business error code, every DTO/JSON field, enum value, boundary, format, nested shape, transaction/state/idempotency/uniqueness/auth/tenant/cross-field rule. For each natural-language normative business rule preserved as required scope, include its exact source sentence without paraphrase together with source path and line/heading anchor so the deterministic ledger can verify quote/hash provenance. Ensure every Case declares exactly `Payload Contract: none` or the three labels `Payload Required Paths`, `Payload Allowed Paths`, and `Payload Enum`; every label must occupy its own machine-readable list line, and a Case must never concatenate target/setup operations or multiple `Payload Contract` tokens onto one line, and explanatory prose/details must not repeat any `Payload Contract:` token; never infer missing keys or enum values. A target GET/DELETE operation with no request body must remain `Payload Contract: none` even when its setup journey performs POST/PUT with a DTO; setup payloads never redefine the target Case payload contract. Add only missing Matrix rows/Test Points/Cases/assertions or repair exact drift; do not rewrite already-valid unrelated modules. Work gap-targeted: inspect source anchors and affected modules first, leave unrelated valid modules byte-stable, and return `already-satisfied` without restating the full suite when no gap exists.",
1422
+ "For affected API fields, use one valid nominal payload plus atomic required/missing/null/empty/wrong-type, every documented enum value plus bounded invalid classes, documented min-1/min/nominal/max/max+1, formats and nested object/array constraints. Do not generate a Cartesian product or invent undocumented constraints. Do not invent a concrete identifier type when the source only requires presence; for a missing-resource 404 path with unspecified identifier syntax/type, synchronize the Case to a create-delete-derived valid identifier journey rather than an arbitrary UUID/text placeholder.",
1423
+ "Scenario Partitions synchronization: when the run-owned plan declares `## Scenario Partitions`, verify each declared partition's slots are fully materialized as variant Test Points with exact `TP-<Partition ID>-...` IDs (each-value per Domain value, OMITTED only for optional axes, exactly one NOT-IN-SET with intent=enum-invalid). Before returning, derive the complete exact slot set from every legal Scenario Partitions row and compare it with both the binding Coverage Matrix Rule's Required Test Points and the final Case `### 测试点`/`变体测试点` sets; directly add every missing exact slot to the already-assigned Case IDs; ordinary alias Test Points do not satisfy a partition slot, and aggregate aliases such as `SINGLE`/`MULTIPLE` are forbidden. Scheme A: keep existing Case structure and add missing exact variant Test Points to already-assigned Cases instead of creating one Case per enum value. Directly add missing slot rows/Cases. Record an illegal plan Partition row that has no source-backed finite domain as GAP/CONFLICT and remove only its derived `TP-SP-*` slots/Cases from target modules; never modify the immutable plan artifact. Never delete a legal source-backed partition or drop its complement slot to force coverage green. When the bound source does not document the complement expectation, keep the slot with GAP expected instead of guessing. Body-field validation enums (`TP-<FIELD>-ENUM-*`) are NOT partitions — do not add partition rows for them.",
1424
+ "Before returning, verify that every explicit source AC/REQ/BR, error code and strong DTO field token appears in the run-owned plan or an applicable module Case. If a fact cannot be safely automated, retain it as GAP/CONFLICT with its exact source pointer instead of dropping it. Return already-satisfied only when no target file needs an incremental edit.",
1425
+ "Read only indexed source paths. Do not scan the repository, modify source/**, generate pytest, execute tests, or emit JSON.",
1426
+ ...(sharedSetupPrompt ? [sharedSetupPrompt] : []),
1427
+ intake.boundedSourceContext,
1428
+ "## Authoritative reference index",
1429
+ JSON.stringify(intake.referenceIndex, null, 2),
1430
+ "For each index entry, use `readPath` for Pi read-tool calls and keep `path` as the exact Markdown Source References citation. Bound files under .harness/tasks/<taskId>/source/** are read-only inputs: reading them is allowed even though writing .harness/** is forbidden. Never resolve `path` relative to the repository root, search for substitutes, or fall back to docs/** when a bound read fails.",
1431
+ ].join("\n\n"),
1432
+ };
1433
+ const validateCases = shellNode("validate-backend-md-cases-shell", [reviewCases.id], "markdown-cases", "Record advisory findings for Markdown structure and deterministically analyze the final README Coverage Scope and Coverage Matrix against final Case rule/test-point bindings. Validate the classification-policy pair, affected operations/rules, scope evidence and regression floor; require documented OpenAPI completeness only for declared affected operations, while all explicit AC/REQ/BR remain in scope. Detect missing in-scope product/API rules, enum values, invalid equivalence classes, boundaries, format classes, business lifecycle states, GAP/CONFLICT, bidirectional Matrix/Case drift, non-canonical Case IDs, unclassified Test Points, duplicate binding modes and non-cross-cutting Test Points bound by multiple Cases. Do not validate source-reference existence. Write human and machine evidence from the same facts. Keep quality findings advisory, but fail closed after writing the report when secret-shaped values are detected. Coverage FAIL stays advisory.", "Run-owned reports/backend-md-case-validation.md, reports/backend-test-case-coverage-analysis.md and contracts/backend-test-case-coverage-facts.json v4 with Coverage Scope plus PASS/FAIL/UNAVAILABLE advisory facts; downstream execution continues.");
1434
+ if (sharedSetup) {
1435
+ validateCases.subtask_prompt += ` Deterministically verify shared setup references: every 引用前置 line must cite exactly the bound path ${sharedSetup.path} with a stable SS anchor (or SS-DEFAULT); any other citation, unbound path, invented step ID or 引用前置 line without a bound document is a FAIL finding. Never rewrite the user document.`;
1436
+ }
1437
+ // N5 line (sharded): pytest shared-asset plan → manifest shell → map_agent barrier.
1438
+ // Shared helpers/factories are written once by the plan node; each module's
1439
+ // test_<stem>.py is written by an independent Pi child (own 16K budget).
1440
+ const generatePytestPlan = {
1441
+ id: "generate-backend-pytest-plan-pi",
1442
+ depends_on: [validateCases.id],
1443
+ role: "implementer",
1444
+ executor: "static",
1445
+ complexity: "LOW",
1446
+ writePolicy: "none",
1447
+ allowedPaths: [],
1448
+ forbiddenPaths: forbidden,
1449
+ outputContract: "Deterministic pytest generation handoff. Per-module map children write self-contained test_<module>.py files with bounded local fixtures, HTTP logging/redaction and payload builders; no model invocation, JSON, shared-asset writes or pytest execution.",
1450
+ subtask_prompt: [
1451
+ "Convert testcase/md/** into pytest using upstream environment and advisory validation evidence plus only bounded pytest config/conftest. This node is a deterministic handoff: NO shared helper/factory/fixture files are generated anywhere in the pipeline; each module's self-contained test_<module>.py (module-local fixtures, HTTP logging/redaction, payload builders) is written by a downstream sharded node. A FAIL advisory report does not authorize inventing missing behavior; use the final Markdown facts that are present.",
1452
+ "Output budget protocol (hard, max output <=16K per turn): downstream module writers write exactly one test_<module>.py per write/edit tool call. Never paste full Python modules into assistant chat. Do not reduce params/assertions/skips semantics to fit.",
1453
+ "Align every variant pytest.param payload with the Markdown scenario intent (empty/missing/null/length/pattern/enum/wrong-type/nominal). Prefer literal payloads over Faker for intent-critical fields so pre-execution scenario-param checks can verify them. Hard contract: intent=enum-invalid MUST pass a concrete invalid value literal (string/number/boolean), never `_OMIT`/None/missing key; intent=missing/empty may use `_OMIT` or delete the key; intent=custom-literal:trim|whitespace-padded requires a leading/trailing whitespace string with non-empty trimmed content (all-whitespace belongs to empty/whitespace-only, not trim); intent=custom-literal:ACTIVE|ARCHIVED requires the exact enum string, never descriptive tokens like filter-active; intent=max/min/max+1 should pass a repeated-string length expression, a bare length number N, or a helper named _*_LEN{N} / _*_MAX_LENGTH / _*_OVER_LENGTH — never a bare 1 for oversize. Hard contract: request payload dicts may only contain DTO field keys from Payload Allowed Paths; never put expect/expected/echo_* helper keys inside the JSON body dict. Path/query/header identifiers and scenario-control metadata (including `id`, expected codes, and selector labels) must stay in separate pytest parameters and helper arguments; never merge them into a DTO patch or JSON body unless that exact path is allowed by the Markdown payload contract. Normalize the configured API base URL with `rstrip(\"/\")` (or equivalently join exactly one slash) before appending endpoint paths; generated requests must never contain a `//api/...` path. When the bound source documents a concrete non-secret local API URL, generated clients must use it as the fallback in `os.environ.get(\"API_BASE_URL\", \"<documented-url>\")`; do not require an otherwise-uninjected environment variable or fail setup solely because it is absent. Missing-field helpers must remove keys idempotently with `payload.pop(field, None)`, never `del payload[field]`, because optional fields may already be absent.",
1454
+ "Generate a reusable HTTP logging helper (or equivalent client wrapper) and call it for every interface request. The request log must include method, URL/path, and request parameters (query plus JSON/body/payload summary). The response log must include status code and response result (JSON/text/body summary), and both records must be visible in pytest stdout/stderr without changing assertions. If shared pytest fixtures are generated, keep their dependency graph in one provider module and require each downstream test module to register that provider with an exact pytest_plugins tuple; importing only the outer fixture is insufficient and will be rejected by fixture-resolution preflight.",
1455
+ "HTTP response header names are case-insensitive. If the helper stores a lower-case normalized header map, every Content-Type or other header assertion must query the lower-case key (for example `content-type`) or use an explicitly case-insensitive accessor; never call a case-sensitive plain dict with `Content-Type` when the stored key is lower-case. Preserve the actual media-type assertion rather than dropping it.",
1456
+ "Compare timestamps and other semantically equivalent protocol values by parsed meaning, not byte-for-byte serialization. In particular, normalize valid ISO-8601 instants before equality/order assertions so differences such as omitted trailing fractional seconds do not create TestBug failures; preserve exact-string assertions only when the Markdown explicitly requires representation equality.",
1457
+ "Before logging, recursively redact sensitive keys and header values including authorization, proxy-authorization, cookie, set-cookie, token, password, secret, api key and credentials. Never print full Authorization/Cookie values. Apply bounded truncation to serialized request and response bodies (with an explicit truncation marker) so large payloads cannot flood pytest or report artifacts.",
1458
+ "Do not read source/**, add cases, reassign ACs, modify conftest/config/production code, use skip/xfail, swallow assertions, execute pytest, or emit JSON. For best-effort cleanup, catch only the narrow transport exception actually raised by the selected HTTP client (for example `requests.RequestException` or `urllib.error.URLError`); never use bare `except`, `Exception`, or `BaseException` with `pass`.",
1459
+ ].join("\n\n"),
1460
+ static: {
1461
+ resultMarkdown: "Shared pytest plan is deterministic: each downstream module writer owns one self-contained test_<module>.py and must not depend on generated shared helper/factory files.",
1462
+ },
1463
+ };
1464
+ const materializePytestManifest = {
1465
+ id: "materialize-backend-pytest-module-manifest-shell",
1466
+ depends_on: [generatePytestPlan.id],
1467
+ role: "verifier",
1468
+ executor: "shell",
1469
+ complexity: "LOW",
1470
+ writePolicy: "read-only",
1471
+ allowedPaths: ro,
1472
+ forbiddenPaths: forbidden,
1473
+ outputContract: "Stdout JSON {modules:[{stem,planReadPath}],planReadPath,planSha256,moduleLayout} parsed from the final strict-layout-validated run-owned Markdown plan artifact, matching the Markdown map manifest.",
1474
+ subtask_prompt: "Parse only the final run-owned generate-backend-md-plan-pi/plan.md artifact with the same strict module-layout and output-budget contract; do not search for or fall back to testcase/**/README.md. No project file writes.",
1475
+ shell: {
1476
+ commands: [buildBackendTestModuleManifestShellCommand(layout, taskConfig.backendTest?.moduleLayout)],
1477
+ cwd: ".",
1478
+ timeoutMs: 60000,
1479
+ },
1480
+ };
1481
+ const generatePytestCasesMap = {
1482
+ id: "generate-backend-pytest-cases-map",
1483
+ depends_on: [materializePytestManifest.id],
1484
+ role: "verifier",
1485
+ executor: "static",
1486
+ complexity: "LOW",
1487
+ writePolicy: "none",
1488
+ allowedPaths: [],
1489
+ forbiddenPaths: forbidden,
1490
+ outputContract: "Serial aggregate of sharded pytest module writers. Each child writes exactly one self-contained testcase/test_<stem>.py with its own 16K Pi budget and no generated shared-asset dependency.",
1491
+ subtask_prompt: "Expand the README module manifest into one sharded pytest writer child per module and run them serially. Child failures fail-close the map barrier.",
1492
+ static: {
1493
+ resultMarkdown: "Backend-test pytest module map expansion barrier.",
1494
+ },
1495
+ dynamicExpansion: {
1496
+ type: "map_agent",
1497
+ workflowNodeId: "generate-backend-pytest-cases-map",
1498
+ itemsFrom: "$.nodes['materialize-backend-pytest-module-manifest-shell'].output.modules",
1499
+ itemName: "item",
1500
+ maxItems: 8,
1501
+ maxExpandedNodes: 8,
1502
+ childIdPrefix: "generate-backend-pytest-case",
1503
+ tokenBudget: { maxTotalTokens: 3000000 },
1504
+ failOnTokenBudgetExhaustion: true,
1505
+ childTask: {
1506
+ executor: "pi",
1507
+ role: "implementer",
1508
+ skills: BACKEND_TEST_SKILLS_BY_ROLE.implementer,
1509
+ toolProfile: "write",
1510
+ complexity: "MED",
1511
+ writePolicy: "exclusive",
1512
+ allowedPaths: ["{{item.pytestPath}}"],
1513
+ forbiddenPaths: Array.from(new Set([
1514
+ ...forbidden,
1515
+ "testcase/md/**",
1516
+ "conftest.py",
1517
+ "pytest.ini",
1518
+ "pyproject.toml",
1519
+ "setup.cfg",
1520
+ ])),
1521
+ writeSet: ["{{item.pytestPath}}"],
1522
+ writerOutcomePolicy: {
1523
+ type: "implementation-outcome-v1",
1524
+ requireChangedFiles: true,
1525
+ },
1526
+ retryPolicy: BACKEND_TEST_WRITER_COMPLETENESS_RETRY_POLICY,
1527
+ outputContract: "Write exactly the frozen pytest module file `{{item.pytestPath}}` whose actual test function region contains the exact Case ID, preferably in the function name or docstring. the frozen `{{item.markdownPath}}` maps one-to-one to `{{item.pytestPath}}`; never merge or split modules. No JSON and no pytest execution.",
1528
+ subtaskPromptTemplate: [
1529
+ "Convert the single frozen Markdown module `{{item.markdownPath}}` into one self-contained pytest module. Before writing, also read the run-owned Markdown plan artifact at `{{item.planReadPath}}` and use its explicit API target/environment table as the authoritative fallback base URL for every module. A task/Markdown `API_BASE_URL` target takes precedence over project README dev-server URLs; never infer a backend API fallback from a frontend/Vite port such as localhost:3000. After reading the module Markdown, the run-owned plan artifact, and the bounded pytest config/conftest, immediately use write tools to create the single frozen file `{{item.pytestPath}}`. Define any bounded HTTP client fixture, request logging/redaction/truncation helper and payload builders needed by this module inside that same file; do not import generated testcase/**/helpers/** or testcase/**/factories/** assets. Do not end after analysis or planning. Do not modify Markdown, conftest, helpers/factories, or any other module's pytest script.",
1530
+ "Output budget protocol (hard, max output <=16K per turn): Write exactly the frozen `{{item.pytestPath}}`. Never paste full Python modules into assistant chat. Do not merge or split modules. Do not reduce params/assertions/skips to fit. If OUTPUT_LIMIT_RECOVERY is injected, continue only listed missing/broken scripts.",
1531
+ "Align every variant pytest.param payload with the Markdown scenario intent (empty/missing/null/length/pattern/enum/wrong-type/nominal). Prefer literal payloads over Faker for intent-critical fields so pre-execution scenario-param checks can verify them. Hard contract: intent=enum-invalid MUST pass a concrete invalid value literal (string/number/boolean), never `_OMIT`/None/missing key; intent=missing/empty may use `_OMIT` or delete the key; intent=custom-literal:trim|whitespace-padded requires a leading/trailing whitespace string with non-empty trimmed content (all-whitespace belongs to empty/whitespace-only, not trim); intent=custom-literal:ACTIVE|ARCHIVED requires the exact enum string, never descriptive tokens like filter-active; intent=max/min/max+1 should pass a repeated-string length expression, a bare length number N, or a helper named _*_LEN{N} / _*_MAX_LENGTH / _*_OVER_LENGTH — never a bare 1 for oversize. Hard contract: request payload dicts may only contain DTO field keys from Payload Allowed Paths; never put expect/expected/echo_* helper keys inside the JSON body dict. Path/query/header identifiers and scenario-control metadata (including `id`, expected codes, and selector labels) must stay in separate pytest parameters and helper arguments; never merge them into a DTO patch or JSON body unless that exact path is allowed by the Markdown payload contract. Normalize the configured API base URL with `rstrip(\"/\")` (or equivalently join exactly one slash) before appending endpoint paths; generated requests must never contain a `//api/...` path. When the bound source documents a concrete non-secret local API URL, generated clients must use it as the fallback in `os.environ.get(\"API_BASE_URL\", \"<documented-url>\")`; do not require an otherwise-uninjected environment variable or fail setup solely because it is absent. Missing-field helpers must remove keys idempotently with `payload.pop(field, None)`, never `del payload[field]`, because optional fields may already be absent.",
1532
+ "For every response contract that requires an object or pagination envelope, first assert that each envelope/data value is a dict and that required keys exist, then index fields and assert values. Never let an incidental KeyError or list/string TypeError stand in for the explicit response-shape contract failure.",
1533
+ 'Ensure every automatable final Markdown Case ID in this module appears in exactly one primary pytest test function or pytest test class method region, using the exact `primary symbol` declared by Markdown. Skip evidence-only meta Cases that declare `脚本/primary symbol=无` with empty variants; do not invent a business pytest symbol for them. The symbol must start with `test_BE_<MODULE>_<NNN>_` so every parameterized collected item remains associated with its Case. Module-level functions and class-based pytest methods are both supported. Only `变体测试点` may use stable `pytest.param(..., id="TP-...")` IDs, and every atomic variant ID must appear exactly once with a genuine input/state/outcome change. A Case with exactly one variant Test Point still needs one literal `pytest.param(..., id="TP-...")` row; never leave a single-variant Case as a bare function with the TP only in the docstring. Use a literal direct `pytest.param(..., id=...)` expression for every row; never hide or wrap it behind `_post_case`, `_put_case`, row-factory functions, comprehensions, generators, or dynamically returned parameter lists; do not use decorator-level `ids=[...]`, generated suffixes, or IDs that extend/shorten the exact Markdown TP. Do not parameterize `场景断言测试点` or `横切证据测试点`; execute all assertion checkpoints within the same business journey/item and use shared helpers for cross-cutting evidence. Governance-only cross-cutting bindings such as writeSet compliance, execution count, report existence, or orchestration state are metadata-only in business pytest: preserve their IDs in `Cross-Cutting-Test-Points`, but never assert `__file__`, filesystem placement, pytest invocation count, Harness state, or report artifacts inside the business test. Harness-owned evidence verifies those bindings. The primary symbol docstring must contain exact metadata lines `Case-ID: BE-...`, `Assertion-Test-Points: TP-...;TP-...` and `Cross-Cutting-Test-Points: TP-...;TP-...` (use `none` when empty). Implement request dictionaries so their direct and nested key paths and enum literals exactly satisfy the Case `Payload Required Paths`, `Payload Allowed Paths`, and `Payload Enum`; for `Payload Contract: none`, do not invent a JSON/body DTO. GET/list filters still declare query fields in those payload labels when the Case varies `params=`/`query=` keys. Python `True`/`False` may implement JSON/OpenAPI `true`/`false` query or body booleans. GET/DELETE setup journeys may create resources, but their setup DTO must not change the target operation\'s no-body payload contract. No Test Point may be invented, renamed, omitted or bound in two modes. The generated pytest collection shape must equal the Markdown prediction `sum(max(1, variant count per Case))`; keep it at or below the task\'s explicit budget by removing duplicate execution, never by collapsing multiple parameter rows under a coarse family TP. Assertions come only from 预期结果 and setup comes only from 前置条件/测试数据/自动化映射.',
1534
+ "Name the generated pytest file so it corresponds one-to-one with its source Markdown module file: this module stem `{{item.stem}}` maps to exactly the frozen `{{item.pytestPath}}`. The <module> stem is the Markdown filename without the `.md` extension, lowercased and with non-alphanumeric characters replaced by underscores. For example, `resource_notes` → `testcase/test_resource_notes.py`, `health` → `testcase/test_health.py`. If Markdown automation mapping names a different path than this module stem path, still write the frozen manifest pytest path and do not invent prefixes. Never merge multiple Markdown modules into one pytest file, never split one module across several files, and never invent pytest filenames unrelated to the Markdown modules.",
1535
+ "Scenario Partition slots: every `TP-<Partition ID>-...` variant Test Point declared by this module's Markdown MUST become exactly one literal direct `pytest.param(..., id=\"TP-<Partition ID>-...\")` row with the exact slot ID; the not-in-set slot passes a concrete literal absent from the documented Domain (e.g. `UNKNOWN_TYPE`) — never `_OMIT`, never a descriptive token. Never split one slot into multiple params or merge several slots under a family TP id. Slot filtering requests hit the documented list endpoint with the slot value as the query/path filter.",
1536
+ "Keep this module self-contained: define module-local fixtures and helpers directly in `{{item.pytestPath}}`, so pytest discovers every fixture dependency without external plugin registration. The request log must include method, URL/path, and request parameters (query plus JSON/body/payload summary). The response log must include status code and response result (JSON/text/body summary), and both records must be visible in pytest stdout/stderr without changing assertions. Recursively redact sensitive values and apply bounded truncation before logging.",
1537
+ "Materialize every automatable Markdown Case exactly once as one canonical primary pytest symbol. Preserve every explicit variant Test Point as a stable pytest.param id and every assertion/cross-cutting binding as declared. Build request payloads from the effective Markdown test data literally: keep all declared DTO keys, nested shapes, enum values, missing/null/boundary variants and business-state preconditions; never substitute guessed convenience fields or rename contract fields. Never assert an identifier's concrete Python/JSON type unless the Markdown or bound contract explicitly declares that type; when only presence is required, accept any non-null scalar identifier and serialize it safely into the path. For a nonexistent-resource 404 Case whose identifier syntax/type is not declared, obtain a syntactically valid identifier from a live create response and delete it before the 404 request; never invent an arbitrary UUID/text identifier that may fail path conversion with 400. Respect every local helper's actual return signature: never tuple-unpack a scalar status/id/helper result, and never treat a tuple response as a scalar.",
1538
+ "Do not read source/**, add cases, reassign ACs, modify conftest/config/production code, use skip/xfail, swallow assertions, execute pytest, or emit JSON. For best-effort cleanup, catch only the narrow transport exception actually raised by the selected HTTP client (for example `requests.RequestException` or `urllib.error.URLError`); never use bare `except`, `Exception`, or `BaseException` with `pass`.",
1539
+ ...(sharedSetupPytestPrompt ? [sharedSetupPytestPrompt] : []),
1540
+ ].join("\n\n"),
1541
+ },
1542
+ },
1543
+ };
1544
+ const collectionAssess = shellNode("assess-backend-pytest-collection-shell", [generatePytestCasesMap.id], "markdown-collection-assess", "Resolve final Markdown-mapped scripts before any business test body execution. Run scenario-param assess/at-most-one deterministic repair first, then pytest --collect-only and a no-business-body pytest --setup-plan fixture-resolution preflight. A safe missing mapped script, generated-local syntax/import defect, or generated fixture dependency/plugin-registration defect is REPAIRABLE; dependency, third-party plugin, production-module, environment, safety and unknown failures remain blocked. Materialize hash-bound collection-v3 facts with repairPaths and fixtureResolutionStatus.", "Run-owned reports/backend-test-pytest-collection-initial.md and contracts/backend-test-pytest-collection-initial.json (collection-v3) with bounded collection/fixture diagnostics, repairPaths, asset hashes, collected item IDs and deterministic repair eligibility.", [], 120000);
1545
+ // N10 owns the deterministic scenario-param rewrite that may update mapped
1546
+ // generated test modules before collection. Declare that bounded shell write
1547
+ // authority so filesystem-only workspace control does not misclassify the
1548
+ // intentional repair as an out-of-bounds mutation.
1549
+ collectionAssess.writePolicy = "exclusive";
1550
+ collectionAssess.writeSet = [applyLayout("testcase/**/test_*.py")];
1551
+ collectionAssess.allowedPaths = Array.from(new Set([...ro, ...collectionAssess.writeSet]));
1552
+ const repairPytest = {
1553
+ id: "repair-backend-pytest-collection-pi",
1554
+ depends_on: [collectionAssess.id],
1555
+ runIf: "$.nodes['assess-backend-pytest-collection-shell'].json.repairEligible == true",
1556
+ role: "implementer",
1557
+ executor: "pi",
1558
+ toolProfile: "write",
1559
+ complexity: "MED",
1560
+ writePolicy: "exclusive",
1561
+ writeSet: Array.from(new Set([
1562
+ applyLayout("testcase/**/test_*.py"),
1563
+ `${layout.testRoot}/helpers/**/*.py`,
1564
+ `${layout.testRoot}/factories/**/*.py`,
1565
+ `${layout.testRoot}/__*_helpers.py`,
1566
+ `${layout.scriptDir}/helpers/**/*.py`,
1567
+ `${layout.scriptDir}/factories/**/*.py`,
1568
+ `${layout.scriptDir}/__*_helpers.py`,
1569
+ ])),
1570
+ readSet: [
1571
+ ".harness/dag-runs/**/reports/backend-test-pytest-collection-initial.md",
1572
+ ".harness/dag-runs/**/reports/backend-test-markdown-pytest-correspondence-initial.md",
1573
+ ".harness/dag-runs/**/generate-backend-md-plan-pi/plan.md",
1574
+ `${layout.markdownDir}/**`,
1575
+ `${layout.testRoot}/**/*.py`,
1576
+ ],
1577
+ contextBudget: { maxTurns: 24, maxTokens: 80_000 },
1578
+ allowedPaths: Array.from(new Set([...ro, `${layout.testRoot}/**`])),
1579
+ forbiddenPaths: Array.from(new Set([
1580
+ ...forbidden,
1581
+ "testcase/md/**",
1582
+ "conftest.py",
1583
+ "pytest.ini",
1584
+ "pyproject.toml",
1585
+ "setup.cfg",
1586
+ ])),
1587
+ writerOutcomePolicy: { type: "implementation-outcome-v1", requireChangedFiles: true },
1588
+ outputContract: "First non-empty line is IMPLEMENTATION_OUTCOME: changed|blocked, followed by a concise repair summary. This node runs only for REPAIRABLE initial facts, so already-satisfied is invalid and a successful outcome requires a non-empty bounded diff. Modify only generated pytest scripts/helpers/factories and preserve every Markdown Case, Test Point, primary symbol and assertion meaning.",
1589
+ subtask_prompt: [
1590
+ "Repair the generated backend pytest asset as one bounded program using the direct upstream collection assessment. This is the only repair attempt and happens before any business test body execution. The direct upstream JSON includes authoritative `repairPaths` and bounded `repairFindings`; treat both as the complete mandatory checklist without searching for a run directory or report file. Treat any upstream line such as `Repair paths: testcase/test_x.py` as equivalent authoritative repairPaths evidence. Directly read and edit that testcase path; do not search for separate root-level `contracts/**`, guess a DAG run directory, or require another report artifact. If the read tool successfully returns the testcase file, the path exists—continue the bounded repair and never later claim that file is absent.",
1591
+ "Initial status REPAIRABLE means at least one listed finding remains: `already-satisfied` is forbidden, and you must produce a non-empty bounded diff on repairPaths before returning `IMPLEMENTATION_OUTCOME: changed`. Fix only readiness-proven generated testcase-local defects on initial facts repairPaths: create exact safe missing mapped test_*.py paths, repair syntax/import/symbol/decorator/parameterization, close generated fixture dependencies/plugin registration, and repair initial Markdown-to-pytest correspondence findings. Use this deterministic repair map instead of reading analyzer implementation: findings about `Case-ID`, `Assertion-Test-Points`, or `Cross-Cutting-Test-Points` are fixed by editing the declared primary symbol docstring metadata lines; variant binding findings are fixed in the literal direct `pytest.param(..., id=\"TP-...\")` row; primary-symbol cardinality/name findings are fixed in the function name or duplicate primary symbols; script mismatch is fixed only on the authoritative assessment repairPaths; payload findings are fixed in request payload construction. Do not read controller `src/**` or inspect JS/TS analyzer code. Do not search for `testcase/**/README.md`. Never invent a business pytest symbol for evidence-only Markdown Cases that declare `脚本/primary symbol=无` with empty variants. For fixture defects inspect both provider and importer listed by repairPaths; fix ScopeMismatch by aligning fixture scopes or inlining request-scoped values so module fixtures never depend on function fixtures; when a shared fixture depends on sibling fixtures, register the whole provider module through an exact pytest_plugins declaration rather than importing only the outer fixture. Do not create unrelated pytest scripts.",
1592
+ "This is the single pytest incremental synchronization round. The `Findings` in `reports/backend-test-pytest-collection-initial.md` are the mandatory repair checklist: resolve every repairable listed finding on every authoritative `Repair paths` file before considering any other advisory evidence, and never substitute an unrelated scenario-param cleanup for a listed correspondence/collection defect. For every assessment-listed path, compare the effective Markdown Case/Test Points/test data and its `Payload Contract`/`Payload Required Paths`/`Payload Allowed Paths`/`Payload Enum` labels with the generated module. Incrementally add or repair only missing symbols, params, assertions and payload builders. Repair every assessment-listed missing nested path, unexpected key and enum mismatch; preserve exact DTO keys, nested shapes, enum/boundary literals, operation transport and business preconditions; remove guessed replacement keys only when the effective Markdown proves the exact contract. Keep path/query/header identifiers and scenario-control metadata separate from DTO patches and JSON bodies; an `id` used for a path target must be passed to the request path/helper, never inserted into a body patch unless `id` is explicitly listed in Payload Allowed Paths. Flatten every variant into a literal direct `pytest.param(..., id=\"TP-...\")` row; replace `_post_case`/`_put_case` or other parameter-row factories because correspondence and scenario readiness require the actual row values and IDs to be statically visible. Also repair helper call sites to match their defined return signatures; do not tuple-unpack a helper that returns one scalar value.",
1593
+ "Preserve final testcase/md/** semantics, every Case ID, Rule/Test Point binding, primary symbol, parameter ID, expected status/body/schema assertion, HTTP logging, redaction and truncation behavior.",
1594
+ "Use local edit only on assessment-listed paths; keep summaries short; never rewrite unrelated modules.",
1595
+ "Do not reinterpret requirements beyond the effective Markdown and bounded assessment diagnostics. Do not modify Markdown, conftest, pytest config, production code or dependencies.",
1596
+ "Do not add skip/skipif/xfail, remove tests, reduce collected items, loosen assertions, swallow exceptions, use try/except ImportError fallback, mutate sys.path/PYTHONPATH, or replace the real API with mocks.",
1597
+ "Do not execute pytest; the deterministic effective collection gate owns the final collection attempt.",
1598
+ ].join("\n\n"),
1599
+ };
1600
+ const collectionEffective = shellNode("effective-backend-pytest-collection-gate-shell", [collectionAssess.id, repairPytest.id], "markdown-collection-effective", "If initial collection+fixture readiness passed, verify unchanged asset hashes and reuse it. If the single repair ran, rerun scenario-param preflight, final collection and no-business-body fixture resolution once. Recompute effective correspondence/payload shape and materialize Case/symbol/item `ELIGIBLE|NOT_ELIGIBLE` facts; Payload-contract-bound field-target MISMATCH/UNDETERMINED scenario items, non-exact mappings and payload UNSAFE/UNAVAILABLE items are excluded. Generic request-level nominal/health observations remain advisory when payload shape is SAFE and correspondence exact. BLOCKED facts, repair failure, residual fixture failure, zero eligible items or hash drift prevent business pytest execution. Materialize canonical backend-test-execution-readiness.json v2.", "Run-owned effective collection-v3 facts, reports/backend-test-execution-eligibility.md, contracts/backend-test-execution-eligibility.json and contracts/backend-test-execution-readiness.json v2 proving exact final assets are collectable, fixture-resolvable, payload-safe at item level and hash-bound; initial PASS is reused, repair path records attempt=1.", [], 120000);
1601
+ collectionEffective.dependsPolicy = "all-or-condition-skip";
1602
+ const traceability = shellNode("backend-test-traceability-gate-shell", [collectionEffective.id], "markdown-traceability", "Deterministically scan only final readiness-authorized Markdown-mapped pytest scripts. Produce bidirectional Markdown module/Case/Test Point ↔ pytest file/primary symbol correspondence, payload-shape safety, per-item execution eligibility and logging findings. scenario-param assessment/repair already ran before collection; consume and display its final PASS/PARTIAL/FAIL/UNAVAILABLE facts without modifying pytest assets after readiness was frozen. Correspondence findings remain advisory; Never block pytest solely on correspondence FAIL.", "Run-owned reports/backend-test-traceability.md, reports/backend-test-markdown-pytest-correspondence.md, contracts/backend-test-markdown-pytest-correspondence-facts.json, reports/backend-test-scenario-param-consistency.md and contracts/backend-test-scenario-param-consistency-facts.json (initial+final) with optional repair audit; PASS/FAIL/UNAVAILABLE correspondence facts bound after effective collection.");
1603
+ const manifest = shellNode("backend-test-case-manifest-shell", [traceability.id], "markdown-manifest", "Materialize the canonical Backend Test Case Manifest only from contracts/backend-test-case-coverage-facts.json and contracts/backend-test-markdown-pytest-correspondence-facts.json. Validate schema, task binding, input hashes and freshness; never re-read source semantics, re-analyze Coverage Matrix, rescan pytest symbols or recompute a second set of metrics. Missing/stale/conflicting facts produce partial/unavailable diagnostics rather than fabricated zeros.", "Run-owned contracts/backend-test-case-manifest.json with materializationStatus, sourceFactsIssues, validated coverageScope, coverageSummary, ruleCoverageSummary and correspondenceSummary; this is the single machine input for L-5 and closeout.");
1604
+ const pytestCommand = [
1605
+ 'mkdir -p "${HARNESS_DAG_RUN_DIR}/reports"',
1606
+ 'echo "pytest targets are resolved at runtime from execution-readiness v2 eligibleItemIds"',
1607
+ ].join("; ");
1608
+ const execute = shellNode("execute-backend-pytest-and-html-report-shell", [manifest.id], "markdown-execute-html", "Read canonical contracts/backend-test-execution-readiness.json v2, verify final asset and eligibility-input hashes, then execute exactly its `eligibleItemIds` pytest node IDs once. Never execute `excludedItems`; retain each exclusion reason as residual TestBug/automation evidence rather than ProductBug. Prefer the deterministic module one-to-one path when a mapped script is missing but the module stem file exists. Generate a native pytest-html self-contained report, then render the primary self-contained Chinese HTML report from the same pytest-html plus final Markdown case metadata without rerun. Keep 测试结论 and quality status; make node 6 Markdown validation + case coverage and node 13 traceability + Markdown-to-pytest correspondence expandable to their full escaped details; show each failure overview item with its original pytest message plus deterministic evidence-based reason analysis; list failure/error case cards before the remaining cases while preserving stable order. Each polished per-case result card includes concise scenario, automation test name, result, duration, and redacted bounded HTTP request parameters/response results for both passed and failed cases. Do not render a technical/execution evidence section in HTML; retain auditable paths and hashes in facts.", "One scoped pytest execution over readiness-authorized eligible pytest items producing a valid pytest-html report with per-case captured output, self-contained reports/backend-test.html, reports/backend-test.md, reports/backend-test-facts.md, a deterministic self-contained reports/backend-test-l5-dashboard.html (machine-computed L-5 metrics, no JSON), and an optional contracts/code-coverage-v1.json when jacocoCoverage is configured (JaCoCo TCP dump → jacoco.xml → parsed; failure-safe); exit 0/1 with valid evidence continues.", [pytestCommand], taskConfig.backendTest?.executeTimeoutMs ?? 300000);
1609
+ if (execute.shell) {
1610
+ execute.shell.envAllowlist = collectBackendTestShellEnvAllowlist(sources);
1611
+ }
1612
+ const report = {
1613
+ id: "backend-test-report-and-l5-pi",
1614
+ depends_on: [execute.id],
1615
+ role: "closeout",
1616
+ executor: "static",
1617
+ complexity: "LOW",
1618
+ writePolicy: "none",
1619
+ allowedPaths: [],
1620
+ forbiddenPaths: forbidden,
1621
+ outputContract: "Deterministic evidence handoff linking the node 15 Markdown facts, HTML report, failure analysis and L-5 dashboard; no model invocation, JSON or writes.",
1622
+ subtask_prompt: [
1623
+ "Generate the final Markdown report only from authoritative run-owned artifacts. Read node 1 reports/backend-test-environment.md; node 6 backend-md-case-validation.md and backend-test-case-coverage-analysis.md; nodes 10/12 backend-test-pytest-collection initial/effective reports and facts; node 13 backend-test-traceability.md, backend-test-markdown-pytest-correspondence.md and backend-test-scenario-param-consistency.md; node 14 contracts/backend-test-case-manifest.json; and node 15 backend-test-result.json, backend-test-facts.md, backend-test-failure-analysis.md, pytest-html/HTML and L-5 dashboard. Do not use node 2/3/4/7/8/9 assistant prose as facts. Do not emit JSON.",
1624
+ "Use this exact human-facing section order: 测试结论 → 执行概览 → 质量校验 → 失败分析 → 风险与建议 → 证据与 L-5. Put the decision and key numbers first, use compact tables/bullets, and keep headings concise. Do not paste entire upstream reports, duplicate per-case tables already present in facts, or repeat the same evidence in multiple sections; link to paths/hashes and quote only the findings needed for the conclusion. Prefer linking reports/backend-test-failure-analysis.md for structured failure analysis rather than inventing classifications.",
1625
+ "Output budget: list evidence paths first, then write a short fixed six-section report; never paste upstream full text into chat.",
1626
+ "The L-5 metrics and visualization are produced deterministically by node 15 at reports/backend-test-l5-dashboard.html. Link to that dashboard as the authoritative L-5 view. Pytest execution facts come from node 15; coverage/correspondence numbers and materializationStatus come from node 14; collection authorization comes from nodes 10/12; detailed coverage findings come from node 6; detailed mapping findings come from node 13. Never recompute these values. If machine manifest and human reports disagree, report evidence inconsistency rather than silently choosing.",
1627
+ "Always state the exact Coverage Scope classification, policy, affected operations, regression floor, completeness claim, PASS/FAIL/UNAVAILABLE status and findings from node 6 case validation + coverage, plus node 13 traceability + correspondence. Affected-scope or affected-operations-full coverage must never be described as whole-API completeness unless every operation is explicitly listed. Their FAIL status does not block pytest, but it must remain visible and must never be rewritten as PASS.",
1628
+ "Include environment, case quality/review, automation mapping, exact pytest facts, failure classification/analysis, risks, regression recommendations, evidence paths/hashes, coverage availability, and L-5 READY/NOT READY. Distinguish Markdown Case count, primary pytest symbol count, collected pytest item count, variant/assertion/cross-cutting Test Point counts and execution amplification; never describe pytest item count as the number of business scenarios.",
1629
+ "Never override Shell/pytest-html facts. L-5 requires pass=100%, AC=100%, automation>=90%, line>=80%, branch>=70%, skipped=0 and no blocking Critical risk.",
1630
+ "Node 15 already rendered the authoritative Markdown and HTML outputs; this node only hands off their stable paths without recomputation.",
1631
+ ].join("\n\n"),
1632
+ static: {
1633
+ resultMarkdown: [
1634
+ "# Backend-test deterministic closeout",
1635
+ "",
1636
+ "Authoritative execution facts: `reports/backend-test.md` and `contracts/backend-test-result.json`.",
1637
+ "Failure analysis: `reports/backend-test-failure-analysis.md`.",
1638
+ "HTML report: `reports/backend-test.html`.",
1639
+ "L-5 dashboard: `reports/backend-test-l5-dashboard.html`.",
1640
+ ].join("\n"),
1641
+ },
1642
+ };
1643
+ const spec = {
1644
+ version: 3,
1645
+ title: `Backend test DAG: ${taskConfig.title}`,
1646
+ runtimeContract: GENERATED_DAG_RUNTIME_CONTRACT,
1647
+ outputLanguage: sources.outputLanguage ?? DEFAULT_DAG_OUTPUT_LANGUAGE,
1648
+ objective: extractObjective(sources.requirementMarkdown, taskConfig.title),
1649
+ successCriteria: extractSuccessCriteria(sources.requirementMarkdown, sources.taskId),
1650
+ globalConstraints: [
1651
+ ...taskConfig.hardConstraints,
1652
+ ...STANDARD_GLOBAL_CONSTRAINTS,
1653
+ "backend-test-dag uses exactly 16 real top-level tasks: Markdown cases are sharded via a README plan + manifest shell + map_agent barrier (one child per module, each with its own 16K Pi budget), and pytest is likewise sharded via a shared-asset plan + manifest shell + map_agent barrier. Pytest collection runs once on the green path and at most twice only when one bounded pre-execution repair is eligible; business pytest test bodies execute exactly once over safe scripts explicitly mapped by final Markdown cases.",
1654
+ "Model nodes produce Markdown and pytest assets, never backend-test business JSON envelopes.",
1655
+ "Environment, advisory Markdown validation/coverage, collection initial/effective facts, advisory traceability/correspondence, pre-execution scenario-param consistency (with at most one deterministic param repair), canonical manifest, pytest-html, HTML, failure-analysis and execution facts are deterministic evidence. Node 6 quality findings stay advisory; nodes 10/12 form the fail-closed collection authorization; node 13 traceability/scenario-param findings stay advisory by default; node 14 partial/unavailable manifest does not block node 15 when effective collection remains fresh.",
1656
+ "Only Markdown case generation/review may read source facts; pytest generation must not read source/**.",
1657
+ "Functional case IDs use canonical BE-<MODULE>-<NNN> with exactly three digits and no alphabetic suffix. Every Test Point has exactly one variant/assertion/cross-cutting binding; only variant bindings create pytest parameter items. Production code/config, skip/xfail, execution-result repair and business pytest rerun are forbidden; pre-execution repairs are limited to one collection-proven generated-test asset repair and one scenario-param payload repair (deterministic preferred).",
1658
+ "Writer nodes must obey multi-file output-budget protocol under 16K max tokens: one file per write/edit, no chat dumps; Completeness Gate may trigger bounded incomplete-write-set recovery without lowering coverage quality.",
1659
+ ],
1660
+ defaults: {
1661
+ ...BACKEND_TEST_DEFAULTS,
1662
+ contextProfile: taskConfig.contextProfile,
1663
+ },
1664
+ skillsByRole: BACKEND_TEST_SKILLS_BY_ROLE,
1665
+ executorModels: sources.executorModelMatrix ?? DEFAULT_DAG_EXECUTOR_MODELS,
1666
+ verifyStrategy: resolveDagVerifyStrategy(taskConfig),
1667
+ tasks: [
1668
+ environment,
1669
+ generateMdPlan,
1670
+ materializeMdManifest,
1671
+ generateMdCasesMap,
1672
+ reviewCases,
1673
+ validateCases,
1674
+ generatePytestPlan,
1675
+ materializePytestManifest,
1676
+ generatePytestCasesMap,
1677
+ collectionAssess,
1678
+ repairPytest,
1679
+ collectionEffective,
1680
+ traceability,
1681
+ manifest,
1682
+ execute,
1683
+ report,
1684
+ ],
1685
+ };
1686
+ // Plan A: rewrite every backend-test node's layout-dependent strings
1687
+ // (prompts, output contracts, writeSet/allowedPaths, shell commands)
1688
+ // against the resolved layout. The default layout is identity, so the
1689
+ // packaged template stays byte-identical with the historical contract.
1690
+ applyBackendTestLayoutToDagSpec(spec, layout);
1691
+ applyBackendTestWorkspaceControl(spec, taskConfig.backendTest?.workspaceControl ?? "git");
1692
+ // Plan B: carry the bound shared-setup document on the spec so runtime
1693
+ // validators (N6 markdown case validation) see the same binding as prompts.
1694
+ if (intake.sharedSetup) {
1695
+ spec.backendTestSharedSetup = { ...intake.sharedSetup };
1696
+ }
1697
+ applyDefaultReadOnlyRetryPolicy(spec);
1698
+ stampGeneratedArtifactBindings(spec);
1699
+ parseDagSpec(spec);
1700
+ assertValidDagSpec(spec);
1701
+ return spec;
1702
+ }
1703
+ export function applyBackendTestLayoutToDagSpec(spec, layout) {
1704
+ if (layout.isDefault) {
1705
+ spec.backendTestLayout = layout;
1706
+ return;
1707
+ }
1708
+ const rewrite = (value) => applyBackendTestLayoutToText(value, layout);
1709
+ for (const task of spec.tasks) {
1710
+ task.subtask_prompt = rewrite(task.subtask_prompt);
1711
+ if (task.outputContract)
1712
+ task.outputContract = rewrite(task.outputContract);
1713
+ task.writeSet = task.writeSet?.map(rewrite);
1714
+ task.allowedPaths = task.allowedPaths?.map(rewrite);
1715
+ task.forbiddenPaths = task.forbiddenPaths?.map(rewrite);
1716
+ if (task.dynamicExpansion?.childTask) {
1717
+ const child = task.dynamicExpansion.childTask;
1718
+ if (typeof child.subtaskPromptTemplate === "string") {
1719
+ child.subtaskPromptTemplate = rewrite(child.subtaskPromptTemplate);
1720
+ }
1721
+ if (typeof child.outputContract === "string") {
1722
+ child.outputContract = rewrite(child.outputContract);
1723
+ }
1724
+ if (Array.isArray(child.allowedPaths)) {
1725
+ child.allowedPaths = child.allowedPaths.map(rewrite);
1726
+ }
1727
+ if (Array.isArray(child.writeSet)) {
1728
+ child.writeSet = child.writeSet.map(rewrite);
1729
+ }
1730
+ if (Array.isArray(child.forbiddenPaths)) {
1731
+ child.forbiddenPaths = child.forbiddenPaths.map(rewrite);
1732
+ }
1733
+ }
1734
+ if (task.shell?.commands) {
1735
+ task.shell.commands = task.shell.commands.map(rewrite);
1736
+ }
1737
+ }
1738
+ // Rewrite remaining prose-level constraints and success criteria that
1739
+ // reference the generated asset tree.
1740
+ spec.globalConstraints = spec.globalConstraints?.map(rewrite);
1741
+ spec.successCriteria = spec.successCriteria?.map(rewrite);
1742
+ spec.backendTestLayout = layout;
1743
+ }
1744
+ export const BACKEND_TEST_CASE_MANIFEST_OUTPUT_INSTRUCTIONS = [
1745
+ "The final fenced JSON block is authoritative and MUST conform exactly to Backend Test Case Manifest v1.",
1746
+ "Top-level keys MUST be exactly: schemaVersion, sourceBinding, cases, evidenceGaps. Do NOT emit coverageSummary — the shell materializer always computes it from sourceBinding/cases/evidenceGaps. Set schemaVersion to numeric 1. Do NOT emit schemaId, manifestType, taskId, modules, acCoverage, brCoverage, dataIsolation, readiness, or other custom top-level keys.",
1747
+ "Copy sourceBinding exactly from contracts/backend-test-analysis.json: taskId, requirementPath, requirementSha256, referencePaths, requirementIds. Preserve Unicode paths exactly; never replace characters in source/需求.md or other paths.",
1748
+ "Each cases[] item MUST use exactly: caseId, non-empty acIds, title, category, automationStatus; optional endpointRef, ruleRefs, file, symbol, gapReason, evidenceRef. Do NOT use id, module, brIds, endpoint, or priority.",
1749
+ "category MUST be exactly one of: positive, negative, boundary, state-transition, auth, timeout, concurrency, other.",
1750
+ "Before pytest generation, set automationStatus=planned. Use generated only with both file and symbol. Use skipped or unsupported only with gapReason.",
1751
+ "Each evidenceGaps[] item MUST use exactly: optional acId, optional caseId, required description, optional evidenceRef. Every gap requires at least acId or caseId. Do NOT use requirementId, relatedBrIds, or sourceRef.",
1752
+ "Every case must map to at least one semantically applicable explicit AC-* in acIds. If no AC applies, omit that case and bind an evidence gap to the nearest applicable acId or caseId; never emit an unbound informational gap.",
1753
+ "acIds MUST exactly match the explicit AC-* values in the written case body; do not infer ACs from Business Rules or summary matrices.",
1754
+ "Use full BE-<MODULE>-<NNN> caseId strings. Do not invent coverage percentages and do not emit coverageSummary; shell always writes the canonical summary.",
1755
+ 'Minimal shape example: {"schemaVersion":1,"sourceBinding":{"taskId":"...","requirementPath":"source/需求.md","requirementSha256":"<64 lowercase hex>","referencePaths":[],"requirementIds":["AC-001"]},"cases":[{"caseId":"BE-MODULE-001","acIds":["AC-001"],"title":"...","category":"positive","automationStatus":"planned","evidenceRef":"testcase/md/module.md"}],"evidenceGaps":[]}',
1756
+ ].join("\n\n");
1757
+ export function collectBackendTestShellEnvAllowlist(sources) {
1758
+ const names = new Set();
1759
+ const collectAssignments = (text) => {
1760
+ const assignmentPattern = /(?:^|[\s;&|])([A-Z_][A-Z0-9_]*)\s*=/g;
1761
+ for (const match of text.matchAll(assignmentPattern)) {
1762
+ if (match[1])
1763
+ names.add(match[1]);
1764
+ }
1765
+ };
1766
+ for (const constraint of sources.taskConfig.hardConstraints) {
1767
+ collectAssignments(constraint);
1768
+ }
1769
+ for (const verify of sources.taskConfig.verifyCommands) {
1770
+ collectAssignments(verify.command);
1771
+ }
1772
+ return [...names].sort();
1773
+ }
1774
+ export function taskAllowsBackendTestReportWrite(sources) {
1775
+ return sources.taskConfig.allowedPaths.some((pattern) => pattern === "docs/test-reports/**" ||
1776
+ pattern === "docs/**" ||
1777
+ pattern === "**");
1778
+ }
1779
+ export const BACKEND_TEST_DEFAULTS = {
1780
+ ...HYBRID_DEFAULTS,
1781
+ writePolicy: "read-only",
1782
+ };
1783
+ export const BACKEND_TEST_SKILLS_BY_ROLE = {
1784
+ planner: ["loop-agent"],
1785
+ scout: [],
1786
+ implementer: ["test-driven-development", "verification-before-completion"],
1787
+ reviewer: ["requesting-code-review", "code-review-core"],
1788
+ verifier: ["verification-before-completion", "systematic-debugging"],
1789
+ closeout: ["loop-agent", "verification-before-completion"],
1790
+ };
1791
+ export function applyBackendTestWorkspaceControl(spec, control) {
1792
+ if (control !== "filesystem-only") {
1793
+ delete spec.backendTestWorkspaceControl;
1794
+ return;
1795
+ }
1796
+ spec.backendTestWorkspaceControl = control;
1797
+ for (const task of spec.tasks) {
1798
+ if ((task.executor === "pi" && task.toolProfile === "write") ||
1799
+ task.shell?.backendTestPipeline) {
1800
+ task.writeGuardPolicy = "filesystem-only";
1801
+ }
1802
+ const child = task.dynamicExpansion?.childTask;
1803
+ if (child?.executor === "pi" && child.toolProfile === "write") {
1804
+ child.writeGuardPolicy = "filesystem-only";
1805
+ }
1806
+ }
1807
+ }