@tea-agent/loop-agent 0.42.0-next.8 → 0.42.0-next.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +9 -0
- package/dist/application/dag/run-dag.js +40 -0
- package/dist/build-stamp.json +3 -3
- package/dist/executors/dag-pi-executor.js +313 -56
- package/dist/executors/shell-executor.js +47 -19
- package/dist/worker/observe/routes.js +4 -0
- package/dist/workflows/dag/backend-test-case-coverage-analysis.js +18 -0
- package/dist/workflows/dag/backend-test-scenario-param.js +30 -1
- package/dist/workflows/dag/frontend-recovery-plan.js +2 -1
- package/dist/workflows/dag/frontend-recovery-run.js +33 -5
- package/dist/workflows/dag/frontend-writer-admission.js +44 -10
- package/dist/workflows/dag/init-hybrid.js +3 -2
- package/dist/workflows/dag/node-execution.js +76 -0
- package/dist/workflows/dag/rerun-feedback.js +59 -0
- package/dist/workflows/dag/rerun-task.js +8 -2
- package/dist/workflows/dag/runner.js +18 -11
- package/dist/workflows/dag/scheduler.js +21 -6
- package/docs/templates/backend-test-dag.json +1 -1
- package/package.json +1 -1
|
@@ -309,7 +309,10 @@ export async function rerunStandaloneTask(input) {
|
|
|
309
309
|
state,
|
|
310
310
|
});
|
|
311
311
|
}
|
|
312
|
-
catch {
|
|
312
|
+
catch (error) {
|
|
313
|
+
if (state.terminalReason === "design-request-changes") {
|
|
314
|
+
throw new Error(`frontend design recovery feedback unavailable: ${error instanceof Error ? error.message : String(error)}`);
|
|
315
|
+
}
|
|
313
316
|
// Feedback is additive recovery context. A corrupt/missing historical
|
|
314
317
|
// artifact must be reported honestly but must not block an otherwise
|
|
315
318
|
// eligible full rerun.
|
|
@@ -378,7 +381,10 @@ export async function rerunStandaloneTask(input) {
|
|
|
378
381
|
kind: "standalone-task-rerun",
|
|
379
382
|
})
|
|
380
383
|
: undefined;
|
|
381
|
-
|
|
384
|
+
const nestedAutoRecoveryLease = recoveryLease?.kind === "held" &&
|
|
385
|
+
recoveryLease.lease.kind === "auto-frontend-recovery" &&
|
|
386
|
+
recoveryLease.lease.requestId === input.requestId;
|
|
387
|
+
if (recoveryLease?.kind === "held" && !nestedAutoRecoveryLease) {
|
|
382
388
|
return blockedResult({
|
|
383
389
|
parentRunId: input.parentRunId,
|
|
384
390
|
reason: input.reason,
|
|
@@ -1087,7 +1087,7 @@ export async function finalizeTerminalRunStatus(state, taskCount, runDir, cwd, f
|
|
|
1087
1087
|
if (hasFrontendWriter && !skipFrontendRecovery) {
|
|
1088
1088
|
const admission = await readFrontendPrewriteResult(runDir);
|
|
1089
1089
|
if (admission.ok) {
|
|
1090
|
-
// M8: admission
|
|
1090
|
+
// M8: admission uses the accepted/blocked state machine. A blocked admission whose
|
|
1091
1091
|
// failureSource is frontend-plan-pi is the contract-invalid analog of
|
|
1092
1092
|
// the old retryable-invalid path (candidate continuation may re-plan);
|
|
1093
1093
|
// every other blocked admission is terminal.
|
|
@@ -1130,7 +1130,10 @@ export async function finalizeTerminalRunStatus(state, taskCount, runDir, cwd, f
|
|
|
1130
1130
|
if (continuationCount < frontendRecoveryQuota) {
|
|
1131
1131
|
recoveryTrigger = {
|
|
1132
1132
|
requestId: state.runId,
|
|
1133
|
-
|
|
1133
|
+
// The rejected design is a plan input defect. Reset from the
|
|
1134
|
+
// planner so the next child can incorporate the committed
|
|
1135
|
+
// findings before design review runs again.
|
|
1136
|
+
failureSource: "frontend-plan-pi",
|
|
1134
1137
|
failureOwner: classifyReviewIssueCategoryFailureOwner(designRequestFact.issueCategory),
|
|
1135
1138
|
protocolFailureReason: `frontend design review request_design_changes (issueCategory=${designRequestFact.issueCategory})`,
|
|
1136
1139
|
recoveryMode: "new-dag",
|
|
@@ -1295,10 +1298,12 @@ export async function finalizeTerminalRunStatus(state, taskCount, runDir, cwd, f
|
|
|
1295
1298
|
* A parent-run atomic lease is acquired before `OperationStore.create`, then
|
|
1296
1299
|
* its operation id is persisted into that lease before the controller may run.
|
|
1297
1300
|
* `clientRequestId` remains a useful store-level replay key, but is not a
|
|
1298
|
-
* cross-process recovery admission boundary. The
|
|
1299
|
-
*
|
|
1300
|
-
*
|
|
1301
|
-
*
|
|
1301
|
+
* cross-process recovery admission boundary. The reset partition is frozen via
|
|
1302
|
+
* `computeFrontendRecoveryPlanForSource` and serialized under the run dir's
|
|
1303
|
+
* bounded `.runtime/` namespace for audit. The operation invokes
|
|
1304
|
+
* `dag rerun-task`, which regenerates a valid child DagSpec and carries typed
|
|
1305
|
+
* parent feedback through the managed rerun path. `planHash` is the second
|
|
1306
|
+
* idempotency dimension of `actionParams`.
|
|
1302
1307
|
*/
|
|
1303
1308
|
async function createRecoveryOperation(input) {
|
|
1304
1309
|
const { cwd, runDir, spec, state, trigger } = input;
|
|
@@ -1344,11 +1349,13 @@ async function createRecoveryOperation(input) {
|
|
|
1344
1349
|
},
|
|
1345
1350
|
cliArgs: [
|
|
1346
1351
|
"dag",
|
|
1347
|
-
"
|
|
1348
|
-
"--
|
|
1349
|
-
|
|
1350
|
-
"--
|
|
1351
|
-
|
|
1352
|
+
"rerun-task",
|
|
1353
|
+
"--run-id",
|
|
1354
|
+
state.runId,
|
|
1355
|
+
"--reason",
|
|
1356
|
+
trigger.protocolFailureReason ?? "frontend automatic recovery",
|
|
1357
|
+
"--request-id",
|
|
1358
|
+
trigger.requestId,
|
|
1352
1359
|
"--json",
|
|
1353
1360
|
],
|
|
1354
1361
|
});
|
|
@@ -4,7 +4,7 @@ import { isPauseOnHumanDecisionGate } from "./decision-envelope.js";
|
|
|
4
4
|
import { evaluateConditionExpression } from "./dynamic-runtime/condition.js";
|
|
5
5
|
import { sha256Hex } from "./frontend-implementation-contract.js";
|
|
6
6
|
import { boundaryForTargetNode, computeFrontendShapeNodeDigest, evaluateFrontendShapeBoundary, } from "./frontend-shape-facts.js";
|
|
7
|
-
import { FRONTEND_WRITER_ADMISSION_RESULT_ARTIFACT, frontendWriterAdmissionResultV1Schema, } from "./frontend-writer-admission.js";
|
|
7
|
+
import { FRONTEND_WRITER_ADMISSION_RESULT_ARTIFACT, frontendWriterAdmissionResultV1Schema, normalizeConcreteWriteSetPath, } from "./frontend-writer-admission.js";
|
|
8
8
|
import { FRONTEND_RECOVERY_IMPORT_MANIFEST_REL_PATH, FRONTEND_RECOVERY_INTENT_REL_DIR, } from "./frontend-recovery-plan.js";
|
|
9
9
|
export function isConditionSkippedReason(reason) {
|
|
10
10
|
return Boolean(reason?.startsWith("condition "));
|
|
@@ -174,10 +174,7 @@ export async function readFrontendWriterAdmissionResult(runDir) {
|
|
|
174
174
|
/** Authorize only the `accepted` classification (four-state admission).
|
|
175
175
|
* `requires-human-approval`, `blocked`, and `stale` are all denied. */
|
|
176
176
|
export function isFrontendWriterAuthorized(result) {
|
|
177
|
-
return result.classification === "accepted"
|
|
178
|
-
result.classification === "admitted"
|
|
179
|
-
? "authorized"
|
|
180
|
-
: "denied";
|
|
177
|
+
return result.classification === "accepted" ? "authorized" : "denied";
|
|
181
178
|
}
|
|
182
179
|
/** Fail-closed: leave no PENDING nodes that look "still scheduled" after abort. */
|
|
183
180
|
export function markPendingNodesControllerInterrupted(state, reason = "run aborted by controller (abortSignal)") {
|
|
@@ -417,6 +414,12 @@ export async function executeDagRanksOnce(input) {
|
|
|
417
414
|
continue;
|
|
418
415
|
}
|
|
419
416
|
const decision = isFrontendWriterAuthorized(admission.result);
|
|
417
|
+
const normalizedWriteSet = admission.result.writeSet
|
|
418
|
+
.map(normalizeConcreteWriteSetPath)
|
|
419
|
+
.filter((entry) => entry !== undefined);
|
|
420
|
+
const concreteWriteSetValid = (admission.result.writeSet.length > 0 &&
|
|
421
|
+
normalizedWriteSet.length === admission.result.writeSet.length &&
|
|
422
|
+
new Set(normalizedWriteSet).size === normalizedWriteSet.length);
|
|
420
423
|
node.frontendWriterAdmission = {
|
|
421
424
|
schemaVersion: 1,
|
|
422
425
|
writerNodeId: id,
|
|
@@ -428,12 +431,24 @@ export async function executeDagRanksOnce(input) {
|
|
|
428
431
|
? `classification: ${admission.result.classification}`
|
|
429
432
|
: null,
|
|
430
433
|
};
|
|
431
|
-
if (decision === "denied") {
|
|
434
|
+
if (decision === "denied" || !concreteWriteSetValid) {
|
|
435
|
+
node.frontendWriterAdmission.reason = decision === "denied"
|
|
436
|
+
? `classification: ${admission.result.classification}`
|
|
437
|
+
: "admission writeSet is not a concrete unique path set";
|
|
432
438
|
node.status = "SKIPPED";
|
|
433
439
|
node.skippedReason = "frontend-prewrite-not-authorized";
|
|
434
440
|
node.finishedAt = checkedAt;
|
|
435
441
|
frontendAdmissionSettled.push(id);
|
|
436
442
|
}
|
|
443
|
+
else {
|
|
444
|
+
node.runtimeWriteAuthorization = {
|
|
445
|
+
schemaVersion: 1,
|
|
446
|
+
status: "validated",
|
|
447
|
+
approvalSourceNodeId: FRONTEND_WRITER_ADMISSION_RESULT_ARTIFACT,
|
|
448
|
+
approvalDigest: admission.result.admissionDigest,
|
|
449
|
+
effectiveWriteSet: normalizedWriteSet,
|
|
450
|
+
};
|
|
451
|
+
}
|
|
437
452
|
}
|
|
438
453
|
if (frontendAdmissionSettled.length > 0)
|
|
439
454
|
await input.persistState();
|
|
@@ -148,7 +148,7 @@
|
|
|
148
148
|
"validate-backend-test-environment-shell"
|
|
149
149
|
],
|
|
150
150
|
"complexity": "MED",
|
|
151
|
-
"subtask_prompt": "This is a required plan-generation node. Read only the strict read set and return the complete Markdown plan in the assistant response. The runtime persists the response as a run-owned Harness artifact named generate-backend-md-plan-pi/plan.md. Do not write testcase/md/README.md or any project file; module case cards are written by downstream sharded nodes.\n\nOutput budget protocol (hard, max output <=16K per turn): Never paste full Matrix, case bodies, or source text into assistant chat. README holds only Scope+Matrix+module index; never inline full case bodies. If a Completeness Gate / OUTPUT_LIMIT_RECOVERY retry is injected, continue only listed target paths.\n\nReturn the complete plan as plain Markdown. Do not emit JSON or code fences. The plan must contain the exact English protocol headings ## Coverage Scope, ## Coverage Matrix, ## Scenario Partitions when applicable, and ## Module Index. Never translate those headings into 覆盖范围/覆盖矩阵/场景分区/模块索引.\n\nRead the upstream environment report only through the strict read set. Generate the Markdown-first backend test plan; it will be persisted under the current DAG run's Harness artifacts, not under testcase/md/.\n\nWrite human-readable content in Simplified Chinese by default. Keep English protocol literals exact: section headings, table headers, Partition IDs, TP IDs, Case IDs, HTTP methods, paths, field names, enum values, commands, filenames, code symbols and source citations. Never translate ## Module Index into ## 模块索引.\n\nCreate the concise plan entry page: test objective, target/environment, isolation/cleanup, module summary and a linked case index table with Case ID, Chinese case name, scenario type, endpoint and expected status/result. Avoid repeating every case body in the plan artifact.\n\nBefore the Coverage Matrix, write a mandatory machine-readable `## Coverage Scope` section in the plan artifact using exactly `| Field | Value |`, immediately followed by the separator row `|---|---|`, and these six unique rows: `Change Classification`, `Coverage Policy`, `Affected Operations`, `Affected Rule Keys`, `Regression Floor`, `Scope Evidence`. Always set `Change Classification` to `new-operation` and `Coverage Policy` to `full-contract`; do NOT reason about whether operations are new or existing. Cover all in-scope rules from the requirement document at full depth; treat the product requirement as the coverage baseline and use API contract evidence (fields/status/enum/boundary/format) to supplement scenario dimensions. Scope is limited to operations/rules the requirement document (or its referenced API contract) explicitly describes; do not expand to unrelated operations that the requirement does not mention. List affected operations exactly as `METHOD /path`, stable rule keys separated by semicolons, and precise source pointers as Scope Evidence.\n\nCoverage depth is full over the in-scope rules: fully cover every documented status, request/response field rule, requiredness, enum, boundary, format, auth and business state of each affected operation the requirement describes, but do not re-test unrelated operations the requirement does not mention. Inspect shared validator/helper/DTO/query builder evidence and expand Affected Operations when the same affected path can affect them; unresolved impact stays visible as GAP/CONFLICT.\n\nBefore writing cases, build the mandatory machine-readable Coverage Matrix inside the plan artifact itself. Its section heading line must be exactly `## Coverage Matrix` with no numeric prefix/suffix; never place the canonical Matrix only in a module file. Use this exact header: `| Rule Key | Priority | Source | Endpoint/Field | Dimension | Rule | Required Test Points | Case IDs | Status |`. Every data row must contain exactly 9 pipe-delimited cells and must never omit `Dimension`; use concise dimensions such as requirement, operation, response-status, requiredness, enum, boundary, format, business-state or error. Use only P0/P1/P2 and COVERED/PARTIAL/GAP/CONFLICT. Use stable `TP-<UPPERCASE-HYPHENATED-ID>` test points separated by semicolons.\n\nEach Rule Key must appear in exactly one Matrix row. Preserve each AC/REQ/BR Rule Key as one row; if one product rule spans multiple dimensions, use a concise composite Dimension in that single row instead of duplicating the key. Derive OpenAPI Rule Keys exactly as the deterministic analyzer does: operation token is `<HTTP-METHOD>-<PATH>` with braces removed and every non-alphanumeric run replaced by a hyphen, uppercase (for example POST `/api/resource-notes` → `POST-API-RESOURCE-NOTES`); response statuses use `API-<OPERATION>-RESPONSE-STATUS`; body/parameter fields use `API-<OPERATION>-<FIELD>-REQUIRED|ENUM|MIN-LENGTH|MAX-LENGTH|MINIMUM|MAXIMUM|PATTERN|FORMAT`. Do not invent aliases such as API-CREATE-FIELDS when a deterministic key applies.\n\nCoverage priority is strict inside the declared scope: P0 product requirements/task hard constraints always remain in scope; P1 exhaustively supplements documented operations, fields, business rules, statuses and errors only for Affected Operations; P2 adds bounded protocol robustness only when it is relevant to the change and does not invent product behavior. Coverage percentages describe the declared affected scope, never whole-API completeness unless every operation is explicitly listed. Conflicts or undefined expectations must stay visible as GAP/CONFLICT with precise source pointers, never guessed.\n\nFor uniqueness/lifecycle rules cover absent, active-existing, deleted-existing, create-delete-recreate, restore-then-recreate and documented scope/case-normalization states. For every enum cover every valid value plus bounded invalid equivalence classes (unknown, case variant, whitespace, empty, null/missing and wrong types as applicable). For every length/number rule cover min-1, min, nominal, max and max+1. For format rules cover each allowed class separately plus a valid mixed value, and representative forbidden classes including uppercase, internal/leading/trailing whitespace, tab/newline, unsupported punctuation, slash, emoji or control characters when the source contract supports that expectation.\n\nMandatory module layout contract: first inspect the PRIMARY requirement for an explicit list of required Markdown/Python output path pairs. When explicit paths are present, they are authoritative `explicit-user-layout`: reproduce their exact filenames, count and one-to-one pairs in Module Index; do not rename, merge, split, omit or add a module from reference/example scripts. Only when the primary requirement has no explicit file layout may you derive the smallest `business-resource-layout`. Existing examples, historical regression functions and shared setup may add evidence/assertions to an existing required module, but never create an extra physical module by themselves. A reference-only `resp_regression`, positive/negative/boundary/error/response module is forbidden. The Module Stem cell must contain only the plain filename stem; never put Markdown link syntax or a path in that cell.\n\nInclude exactly one `## Module Index` table with this exact header: `| Module Stem | Business Resource | Owned Operations | Owned Rule Keys | Case IDs | Split Reason | Markdown Path | Pytest Path |`. Use canonical relative links `[label](./<stem>.md)` inside the Markdown Path cell followed by the resolved `${layout.markdownDir}/<stem>.md` path. Split Reason is exactly one of `explicit-user-layout`, `primary-business-resource`, `independent-business-resource`, or `output-budget`. Group by stable business resource/domain, not by CRUD operation, AC, parameter/field axis, scenario type or regression purpose: one resource's list/detail/create/update/delete and its filters/response assertions/regression floor belong in one module. Multiple modules owning the same exact `METHOD /path` are forbidden unless every such row is `explicit-user-layout` from primary-requirement path pairs or has a documented `output-budget` proof. Keep the total module count at the smallest safe value and never exceed 8 modules. Name model-derived modules with stable lowercase business stems such as `health` or `resource_notes`; explicit-user-layout preserves the primary requirement filename stem even when it is more specific. Do not use priority-only stems `p0`, `p1` or `p2`; Priority belongs only in the Coverage Matrix. Pure hexadecimal/hash-like opaque stems and test-purpose-only stems are forbidden. Do not use Case-ID-like module filenames. The relative link target, Markdown Path, Pytest Path and downstream automation mapping must be one-to-one and exact; for model-derived modules the default pair remains `testcase/md/<module>.md` and `testcase/test_<module>.py`, while explicit-user-layout preserves the primary requirement paths. Do not hand-write a conflicting module count in prose; the Module Index row count is the only count truth.\n\nScenario Partitions (query/filter axes): for every affected GET/list operation, declare one row per enum or classification axis used for filtering (query/path parameters such as type/status/category). Add a mandatory machine-readable `## Scenario Partitions` section after the Coverage Matrix using exactly `| Partition ID | Operation | Axis | Domain | Required Slots | Expected by Slot | Bind Rule |` with the separator row. Partition ID is a stable `SP-<OPERATION>-<AXIS>` token; Domain must copy the legal values verbatim from the bound OpenAPI enum or requirement sentence (never guess), using bare semicolon-separated identifier values inside the single table cell (for example `ACTIVE; ARCHIVED`, with no Markdown backticks or prose); Required Slots must contain `each-value` and exactly one `not-in-set`, plus `omitted` only when the parameter is optional. Before returning, expand every declared partition into its complete deterministic exact slot ID set: one `TP-<Partition ID>-<VALUE-TOKEN>` per Domain value, `TP-<Partition ID>-OMITTED` only for an optional axis, and exactly one `TP-<Partition ID>-NOT-IN-SET`. Every expanded slot ID must appear verbatim in the binding Rule's `Required Test Points` cell and be assigned to concrete Case IDs in that same Coverage Matrix row; ordinary alias/family Test Points do not replace this inventory. Scheme A: Case count may be smaller than the enum count, but every exact slot still needs an independent variant Test Point and pytest.param id; never use SINGLE/MULTIPLE aliases as coverage. Expected by Slot states the documented expectation per slot kind (`domain-value`, `default-behavior`, `empty-result`/`excluded-result` when documented, or `GAP` when the source does not document the complement expectation — never invent 空列表/400). POST/PUT body field-validation enums stay in the Coverage Matrix as `TP-<FIELD>-ENUM-*` and MUST NOT get a Scenario Partition row. Do not create partitions for axes without a documented legal-value domain. Only GET/list query or path parameters whose bound source documents a finite enum or classification set may become a Scenario Partition. Do not create partitions for free-form strings, primary keys, required-or-optional-only parameters, or boundary/format-only axes. If an axis has no finite legal-value domain, do not declare a Partition row and do not invent NOT-IN-SET cases. Cross-axis combinations stay as ONE nominal Case; never declare a cross-axis cartesian partition.\n\nBefore finalizing README, calculate the predicted collected-item count as `sum(max(1, number of variant Test Points in each Case))`. If the task declares an item budget, the prediction must not exceed it. Reduce excess only by removing duplicate execution and converting same-request checkpoints to assertions; never drop required rules, boundaries, enums, operation-specific inputs, or business states. Record the prediction in README. Use only environment-supported fixtures/targets/isolation, record evidence gaps in Chinese, and do not emit JSON, pytest, or execute commands.\n\n## Derived task contract: 需求.md\n\n# Backend test\n- AC-001 proof\n\n## Authoritative reference index\n\n[]\n\nFor each index entry, use `readPath` for Pi read-tool calls and copy `path` exactly into Markdown Source References. Bound files under .harness/tasks/<taskId>/source/** are read-only inputs: reading them is allowed even though writing .harness/** is forbidden. Never resolve `path` relative to the repository root, search for substitutes, or fall back to docs/** when a bound read fails.\n\nRead only precise indexed references needed for AC/API/field/rule evidence; references remain authoritative over derived text.",
|
|
151
|
+
"subtask_prompt": "This is a required plan-generation node. Read only the strict read set and return the complete Markdown plan in the assistant response. The runtime persists the response as a run-owned Harness artifact named generate-backend-md-plan-pi/plan.md. Do not write testcase/md/README.md or any project file; module case cards are written by downstream sharded nodes.\n\nOutput budget protocol (hard, max output <=16K per turn): Never paste full Matrix, case bodies, or source text into assistant chat. README holds only Scope+Matrix+module index; never inline full case bodies. If a Completeness Gate / OUTPUT_LIMIT_RECOVERY retry is injected, continue only listed target paths.\n\nReturn the complete plan as plain Markdown. Do not emit JSON or code fences. The plan must contain the exact English protocol headings ## Coverage Scope, ## Coverage Matrix, ## Scenario Partitions when applicable, and ## Module Index. Never translate those headings into 覆盖范围/覆盖矩阵/场景分区/模块索引.\n\nRead the upstream environment report only through the strict read set. Generate the Markdown-first backend test plan; it will be persisted under the current DAG run's Harness artifacts, not under testcase/md/.\n\nWrite human-readable content in Simplified Chinese by default. Keep English protocol literals exact: section headings, table headers, Partition IDs, TP IDs, Case IDs, HTTP methods, paths, field names, enum values, commands, filenames, code symbols and source citations. Never translate ## Module Index into ## 模块索引.\n\nCreate the concise plan entry page: test objective, target/environment, isolation/cleanup, module summary and a linked case index table with Case ID, Chinese case name, scenario type, endpoint and expected status/result. Avoid repeating every case body in the plan artifact.\n\nBefore the Coverage Matrix, write a mandatory machine-readable `## Coverage Scope` section in the plan artifact using exactly `| Field | Value |`, immediately followed by the separator row `|---|---|`, and these six unique rows: `Change Classification`, `Coverage Policy`, `Affected Operations`, `Affected Rule Keys`, `Regression Floor`, `Scope Evidence`. Always set `Change Classification` to `new-operation` and `Coverage Policy` to `full-contract`; do NOT reason about whether operations are new or existing. Cover all in-scope rules from the requirement document at full depth; treat the product requirement as the coverage baseline and use API contract evidence (fields/status/enum/boundary/format) to supplement scenario dimensions. Scope is limited to operations/rules the requirement document (or its referenced API contract) explicitly describes; do not expand to unrelated operations that the requirement does not mention. List affected operations exactly as `METHOD /path`, stable rule keys separated by semicolons, and precise source pointers as Scope Evidence.\n\nCoverage depth is full over the in-scope rules: fully cover every documented status, request/response field rule, requiredness, enum, boundary, format, auth and business state of each affected operation the requirement describes, but do not re-test unrelated operations the requirement does not mention. Inspect shared validator/helper/DTO/query builder evidence and expand Affected Operations when the same affected path can affect them; unresolved impact stays visible as GAP/CONFLICT.\n\nBefore writing cases, build the mandatory machine-readable Coverage Matrix inside the plan artifact itself. Its section heading line must be exactly `## Coverage Matrix` with no numeric prefix/suffix; never place the canonical Matrix only in a module file. Use this exact header: `| Rule Key | Priority | Source | Endpoint/Field | Dimension | Rule | Required Test Points | Case IDs | Status |`. Every data row must contain exactly 9 pipe-delimited cells and must never omit `Dimension`; use concise dimensions such as requirement, operation, response-status, requiredness, enum, boundary, format, business-state or error. Use only P0/P1/P2 and COVERED/PARTIAL/GAP/CONFLICT. Use stable `TP-<UPPERCASE-HYPHENATED-ID>` test points separated by semicolons.\n\nEach Rule Key must appear in exactly one Matrix row. Preserve each AC/REQ/BR Rule Key as one row; if one product rule spans multiple dimensions, use a concise composite Dimension in that single row instead of duplicating the key. Derive OpenAPI Rule Keys exactly as the deterministic analyzer does: operation token is `<HTTP-METHOD>-<PATH>` with braces removed and every non-alphanumeric run replaced by a hyphen, uppercase (for example POST `/api/resource-notes` → `POST-API-RESOURCE-NOTES`); response statuses use `API-<OPERATION>-RESPONSE-STATUS`; body/parameter fields use `API-<OPERATION>-<FIELD>-REQUIRED|ENUM|MIN-LENGTH|MAX-LENGTH|MINIMUM|MAXIMUM|PATTERN|FORMAT`. Do not invent aliases such as API-CREATE-FIELDS when a deterministic key applies.\n\nCoverage priority is strict inside the declared scope: P0 product requirements/task hard constraints always remain in scope; P1 exhaustively supplements documented operations, fields, business rules, statuses and errors only for Affected Operations; P2 adds bounded protocol robustness only when it is relevant to the change and does not invent product behavior. Coverage percentages describe the declared affected scope, never whole-API completeness unless every operation is explicitly listed. Conflicts or undefined expectations must stay visible as GAP/CONFLICT with precise source pointers, never guessed.\n\nFor uniqueness/lifecycle rules cover absent, active-existing, deleted-existing, create-delete-recreate, restore-then-recreate and documented scope/case-normalization states. For every enum cover every valid value plus bounded invalid equivalence classes (unknown, case variant, whitespace, empty, null/missing and wrong types as applicable). For every length/number rule cover min-1, min, nominal, max and max+1. For format rules cover each allowed class separately plus a valid mixed value, and representative forbidden classes including uppercase, internal/leading/trailing whitespace, tab/newline, unsupported punctuation, slash, emoji or control characters when the source contract supports that expectation.\n\nMandatory module layout contract: first inspect the PRIMARY requirement for an explicit list of required Markdown/Python output path pairs. When explicit paths are present, they are authoritative `explicit-user-layout`: reproduce their exact filenames, count and one-to-one pairs in Module Index; do not rename, merge, split, omit or add a module from reference/example scripts. Only when the primary requirement has no explicit file layout may you derive the smallest `business-resource-layout`. Existing examples, historical regression functions and shared setup may add evidence/assertions to an existing required module, but never create an extra physical module by themselves. A reference-only `resp_regression`, positive/negative/boundary/error/response module is forbidden. The Module Stem cell must contain only the plain filename stem; never put Markdown link syntax or a path in that cell.\n\nInclude exactly one `## Module Index` table with this exact header: `| Module Stem | Business Resource | Owned Operations | Owned Rule Keys | Case IDs | Split Reason | Markdown Path | Pytest Path |`. Use canonical relative links `[label](./<stem>.md)` inside the Markdown Path cell followed by the resolved `${layout.markdownDir}/<stem>.md` path. Split Reason is exactly one of `explicit-user-layout`, `primary-business-resource`, `independent-business-resource`, or `output-budget`. Group by stable business resource/domain, not by CRUD operation, AC, parameter/field axis, scenario type or regression purpose: one resource's list/detail/create/update/delete and its filters/response assertions/regression floor belong in one module. Multiple modules owning the same exact `METHOD /path` are forbidden unless every such row is `explicit-user-layout` from primary-requirement path pairs or has a documented `output-budget` proof. Keep the total module count at the smallest safe value and never exceed 8 modules. Name model-derived modules with stable lowercase business stems such as `health` or `resource_notes`; explicit-user-layout preserves the primary requirement filename stem even when it is more specific. Do not use priority-only stems `p0`, `p1` or `p2`; Priority belongs only in the Coverage Matrix. Pure hexadecimal/hash-like opaque stems and test-purpose-only stems are forbidden. Do not use Case-ID-like module filenames. The relative link target, Markdown Path, Pytest Path and downstream automation mapping must be one-to-one and exact; for model-derived modules the default pair remains `testcase/md/<module>.md` and `testcase/test_<module>.py`, while explicit-user-layout preserves the primary requirement paths. Do not hand-write a conflicting module count in prose; the Module Index row count is the only count truth.\n\nScenario Partitions (query/filter axes): inspect every affected GET/list operation for query/path parameters whose bound source documents a finite enum or classification domain. If at least one such axis exists, add exactly one machine-readable `## Scenario Partitions` section after the Coverage Matrix using exactly `| Partition ID | Operation | Axis | Domain | Required Slots | Expected by Slot | Bind Rule |` with the separator row and one row per eligible axis. If no affected axis has a source-backed finite domain, omit the entire `## Scenario Partitions` heading and section; do not emit an explanatory prose-only section. Partition ID is a stable `SP-<OPERATION>-<AXIS>` token; Domain must copy the legal values verbatim from the bound OpenAPI enum or requirement sentence (never guess), using bare semicolon-separated identifier values inside the single table cell (for example `ACTIVE; ARCHIVED`, with no Markdown backticks or prose); Required Slots must contain `each-value` and exactly one `not-in-set`, plus `omitted` only when the parameter is optional. Before returning, expand every declared partition into its complete deterministic exact slot ID set: one `TP-<Partition ID>-<VALUE-TOKEN>` per Domain value, `TP-<Partition ID>-OMITTED` only for an optional axis, and exactly one `TP-<Partition ID>-NOT-IN-SET`. Every expanded slot ID must appear verbatim in the binding Rule's `Required Test Points` cell and be assigned to concrete Case IDs in that same Coverage Matrix row; ordinary alias/family Test Points do not replace this inventory. Scheme A: Case count may be smaller than the enum count, but every exact slot still needs an independent variant Test Point and pytest.param id; never use SINGLE/MULTIPLE aliases as coverage. Expected by Slot states the documented expectation per slot kind (`domain-value`, `default-behavior`, `empty-result`/`excluded-result` when documented, or `GAP` when the source does not document the complement expectation — never invent 空列表/400). POST/PUT body field-validation enums stay in the Coverage Matrix as `TP-<FIELD>-ENUM-*` and MUST NOT get a Scenario Partition row. Do not create partitions for axes without a documented legal-value domain. Only GET/list query or path parameters whose bound source documents a finite enum or classification set may become a Scenario Partition. Do not create partitions for free-form strings, primary keys, required-or-optional-only parameters, or boundary/format-only axes. If an axis has no finite legal-value domain, do not declare a Partition row and do not invent NOT-IN-SET cases. Cross-axis combinations stay as ONE nominal Case; never declare a cross-axis cartesian partition.\n\nBefore finalizing README, calculate the predicted collected-item count as `sum(max(1, number of variant Test Points in each Case))`. If the task declares an item budget, the prediction must not exceed it. Reduce excess only by removing duplicate execution and converting same-request checkpoints to assertions; never drop required rules, boundaries, enums, operation-specific inputs, or business states. Record the prediction in README. Use only environment-supported fixtures/targets/isolation, record evidence gaps in Chinese, and do not emit JSON, pytest, or execute commands.\n\n## Derived task contract: 需求.md\n\n# Backend test\n- AC-001 proof\n\n## Authoritative reference index\n\n[]\n\nFor each index entry, use `readPath` for Pi read-tool calls and copy `path` exactly into Markdown Source References. Bound files under .harness/tasks/<taskId>/source/** are read-only inputs: reading them is allowed even though writing .harness/** is forbidden. Never resolve `path` relative to the repository root, search for substitutes, or fall back to docs/** when a bound read fails.\n\nRead only precise indexed references needed for AC/API/field/rule evidence; references remain authoritative over derived text.",
|
|
152
152
|
"executor": "pi",
|
|
153
153
|
"role": "planner",
|
|
154
154
|
"toolProfile": "read-only",
|