@tea-agent/loop-agent 0.42.0-next.8 → 0.42.0-next.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2299,19 +2299,38 @@ async function collectReviewRestartPhase(runDir, reviewNodeId) {
2299
2299
  }
2300
2300
  /** A+B (AC-009): deterministic §4.10 restart phase for a committed
2301
2301
  * `request_design_changes` fact. Mirrors the review collector. */
2302
- /** Number of findings attached to the committed request_design_changes
2303
- * terminal fact (0 when absent) — used for the admission advisory. */
2304
- async function collectDesignReviewFindingCount(runDir, designNodeId) {
2302
+ /** Read the committed design request as a bounded structured capsule. */
2303
+ async function collectDesignReviewFindings(runDir, designNodeId) {
2305
2304
  try {
2306
2305
  const records = await readTypedEventStoreFromJsonl(path.join(runDir, designNodeId, "design-typed-facts.jsonl"));
2307
2306
  const request = records.find((record) => record.phase === "committed" &&
2308
2307
  record.fact.kind ===
2309
2308
  "request_design_changes");
2310
- const findings = (request?.fact).findings;
2311
- return Array.isArray(findings) ? findings.length : 0;
2309
+ const fact = request?.fact;
2310
+ const findings = Array.isArray(fact?.findings)
2311
+ ? fact.findings.flatMap((finding) => {
2312
+ if (!finding || typeof finding !== "object")
2313
+ return [];
2314
+ const value = finding;
2315
+ if (!(["Critical", "Important", "Minor", "Info"].includes(value.severity) &&
2316
+ typeof value.issue === "string" && value.issue.trim().length > 0))
2317
+ return [];
2318
+ return [{
2319
+ severity: value.severity,
2320
+ ...(typeof value.file === "string" && value.file.trim() ? { file: value.file } : {}),
2321
+ ...(typeof value.line === "number" && Number.isInteger(value.line) && value.line > 0 ? { line: value.line } : {}),
2322
+ issue: value.issue,
2323
+ ...(typeof value.requiredChange === "string" && value.requiredChange.trim() ? { requiredChange: value.requiredChange } : {}),
2324
+ }];
2325
+ }).slice(0, 16)
2326
+ : [];
2327
+ const evidenceRefs = Array.isArray(fact?.evidenceRefs)
2328
+ ? fact.evidenceRefs.filter((ref) => typeof ref === "string" && ref.trim().length > 0).slice(0, 16)
2329
+ : [];
2330
+ return { findings, evidenceRefs };
2312
2331
  }
2313
2332
  catch {
2314
- return 0;
2333
+ return { findings: [], evidenceRefs: [] };
2315
2334
  }
2316
2335
  }
2317
2336
  async function collectDesignRestartPhase(runDir, designNodeId) {
@@ -2625,17 +2644,15 @@ async function executeFrontendWriterAdmission(input, meta) {
2625
2644
  const shell = input.task.shell;
2626
2645
  const config = shell.frontendWriterAdmission;
2627
2646
  try {
2628
- // 1. Design verdict handling (r12 policy change). The typed terminal
2647
+ // 1. Design verdict handling. The typed terminal
2629
2648
  // verdict is read from the committed typed design terminal fact, not
2630
2649
  // from the legacy first-line VERDICT text protocol (AC-003/AC-004).
2631
- // A request_design_changes verdict no longer blocks writer admission:
2632
- // it is downgraded to an advisory that rides along in the admission
2633
- // receipt (and reaches implement via the review node's upstream
2634
- // context). Blocking is owned by the deterministic gates (prewrite,
2635
- // write guard) and the final code review. A missing typed terminal
2636
- // verdict still fails closed. In small topology the design review node
2637
- // is absent and the check is vacuous.
2650
+ // A request_design_changes verdict blocks writer admission and is copied
2651
+ // into the receipt as a bounded advisory capsule for recovery. A missing
2652
+ // typed terminal verdict still fails closed. In small topology the design
2653
+ // review node is absent and the check is vacuous.
2638
2654
  let verdictPass = true;
2655
+ let designRequested = false;
2639
2656
  let designAdvisory;
2640
2657
  if (config.designReviewFromNodeId) {
2641
2658
  const designKinds = await collectDesignTerminalKinds(meta.runDir, config.designReviewFromNodeId);
@@ -2643,11 +2660,16 @@ async function executeFrontendWriterAdmission(input, meta) {
2643
2660
  verdictPass = false;
2644
2661
  }
2645
2662
  else if (designKinds[0] !== "approve_design") {
2663
+ designRequested = true;
2664
+ const designFindings = await collectDesignReviewFindings(meta.runDir, config.designReviewFromNodeId);
2646
2665
  designAdvisory = {
2647
- verdict: designKinds[0],
2666
+ verdict: "request_design_changes",
2648
2667
  restartPhase: await collectDesignRestartPhase(meta.runDir, config.designReviewFromNodeId),
2649
- findingCount: await collectDesignReviewFindingCount(meta.runDir, config.designReviewFromNodeId),
2668
+ findingCount: designFindings.findings.length,
2669
+ findings: designFindings.findings,
2670
+ evidenceRefs: designFindings.evidenceRefs,
2650
2671
  };
2672
+ verdictPass = false;
2651
2673
  }
2652
2674
  }
2653
2675
  // 2. Read the canonical contract + design-policy result.
@@ -2715,7 +2737,9 @@ async function executeFrontendWriterAdmission(input, meta) {
2715
2737
  allowedMockStrategies: config.allowedMockStrategies,
2716
2738
  workspaceFiles,
2717
2739
  });
2718
- // 6. Design verdict enforcement: request-revision is a blocked admission.
2740
+ // 6. Design verdict enforcement: request_design_changes is a blocked
2741
+ // admission. This keeps the writer fail-closed until the recovery plan
2742
+ // has incorporated the reviewer's concrete findings.
2719
2743
  const failureSource = policyRaw.classification === "blocked"
2720
2744
  ? "frontend-design-policy-shell"
2721
2745
  : "frontend-writer-admission-shell";
@@ -2724,8 +2748,12 @@ async function executeFrontendWriterAdmission(input, meta) {
2724
2748
  : [
2725
2749
  ...admission.findings,
2726
2750
  {
2727
- code: "design-approval-missing",
2728
- message: "design review typed terminal verdict missing or ambiguous (fail closed)",
2751
+ code: designRequested
2752
+ ? "design-review-requested"
2753
+ : "design-approval-missing",
2754
+ message: designRequested
2755
+ ? "design review requested changes; writer admission is blocked until recovery incorporates the committed findings"
2756
+ : "design review typed terminal verdict missing or ambiguous (fail closed)",
2729
2757
  },
2730
2758
  ];
2731
2759
  const classification = findings.length === 0 && admission.classification === "accepted"
@@ -278,6 +278,10 @@ export async function serveObserveStatic(req, res, staticDir, options = {}) {
278
278
  "Content-Length": content.byteLength,
279
279
  "X-Content-Type-Options": "nosniff",
280
280
  "X-Frame-Options": "SAMEORIGIN",
281
+ // Filenames are unhashed and Console/observe builds change in place;
282
+ // heuristic caching would let a stale operator-chrome.js (old nav,
283
+ // old capabilities) survive a Console restart/update.
284
+ "Cache-Control": "no-store",
281
285
  };
282
286
  if (contentType.startsWith("text/html") && options.htmlEnhancer) {
283
287
  const enhanced = options.htmlEnhancer(Buffer.from(content));
@@ -1482,6 +1482,13 @@ for top in ast.walk(tree):
1482
1482
  for keyword in node.keywords:
1483
1483
  if keyword.arg in {'payload','body','request_body','request_json','dto','json','data','json_body','params','query','query_params'} and isinstance(keyword.value,ast.Name) and keyword.value.id in positional:
1484
1484
  transport.append((keyword.value.id,positional.index(keyword.value.id)-call_offset))
1485
+ if keyword.arg is None and isinstance(keyword.value,ast.Name):
1486
+ alias=keyword.value.id
1487
+ assigned_value=assigned.get(alias)
1488
+ if isinstance(assigned_value,ast.Name) and assigned_value.id in positional:
1489
+ transport.append((assigned_value.id,positional.index(assigned_value.id)-call_offset))
1490
+ elif isinstance(assigned_value,ast.Call) and isinstance(assigned_value.func,ast.Name) and assigned_value.func.id == 'dict' and assigned_value.args and isinstance(assigned_value.args[0],ast.Name) and assigned_value.args[0].id in positional:
1491
+ transport.append((assigned_value.args[0].id,positional.index(assigned_value.args[0].id)-call_offset))
1485
1492
  if isinstance(node.func,ast.Attribute) and node.func.attr.lower() in {'post','put','patch'} and len(node.args) >= 2:
1486
1493
  candidate=node.args[1]
1487
1494
  if isinstance(candidate,ast.Name) and candidate.id in positional:
@@ -1493,6 +1500,17 @@ for top in ast.walk(tree):
1493
1500
  matched=node.args[nested_position] if nested_position >= 0 and nested_position < len(node.args) else None
1494
1501
  if isinstance(matched,ast.Name) and matched.id in positional:
1495
1502
  transport.append((matched.id,positional.index(matched.id)-call_offset))
1503
+ # Assignment order in ast.walk is not source order. Re-scan **kwargs
1504
+ # aliases once the complete local assignment table is available.
1505
+ for node in ast.walk(top):
1506
+ if not isinstance(node,ast.Call): continue
1507
+ for keyword in node.keywords:
1508
+ if keyword.arg is not None or not isinstance(keyword.value,ast.Name): continue
1509
+ assigned_value=assigned.get(keyword.value.id)
1510
+ if isinstance(assigned_value,ast.Name) and assigned_value.id in positional:
1511
+ transport.append((assigned_value.id,positional.index(assigned_value.id)-call_offset))
1512
+ elif isinstance(assigned_value,ast.Call) and isinstance(assigned_value.func,ast.Name) and assigned_value.func.id == 'dict' and assigned_value.args and isinstance(assigned_value.args[0],ast.Name) and assigned_value.args[0].id in positional:
1513
+ transport.append((assigned_value.args[0].id,positional.index(assigned_value.args[0].id)-call_offset))
1496
1514
  for returned_name in returned_names:
1497
1515
  entries=mutated.get(returned_name,[])
1498
1516
  if entries:
@@ -1476,6 +1476,32 @@ function hasSafeSameModulePayloadHelper(source, block, parameterNames) {
1476
1476
  return false;
1477
1477
  return Boolean(safeSameModuleDictHelper(source, call.name));
1478
1478
  }
1479
+ function observeBranchPayloadField(input) {
1480
+ const region = scenarioFunctionRegion(input.source, input.tpId);
1481
+ if (!region || !input.block)
1482
+ return undefined;
1483
+ const args = extractPytestParamTopLevelArgs(input.block);
1484
+ const escapedField = input.field.replace(/[.*+?^${}()|[\\]\\]/g, "\\$&");
1485
+ for (let index = 0; index < input.parameterNames.length; index += 1) {
1486
+ const parameter = input.parameterNames[index];
1487
+ const raw = args[index]?.trim();
1488
+ if (!raw || !/^['\"](?:[^'\"\\\\]|\\\\.)*['\"]$/.test(raw))
1489
+ continue;
1490
+ const branchValue = raw.slice(1, -1).replace(/\\\\([\\\"'])/g, "$1");
1491
+ const escapedParameter = parameter.replace(/[.*+?^${}()|[\\]\\]/g, "\\$&");
1492
+ const escapedBranchValue = branchValue.replace(/[.*+?^${}()|[\\]\\]/g, "\\$&");
1493
+ const branch = new RegExp(`(?:^|\\n)([ \\t]*)if\\s+${escapedParameter}\\s*==\\s*['\"]${escapedBranchValue}['\"]\\s*:\\s*([\\s\\S]*?)(?=\\n\\1(?:elif\\b|else\\s*:)|\\n\\1(?:if\\b|def\\b)|$)`, "m").exec(region)?.[2];
1494
+ if (!branch)
1495
+ continue;
1496
+ const value = new RegExp(`['\"]${escapedField}['\"]\\s*:\\s*([^,}\\n]+)`).exec(branch)?.[1];
1497
+ if (!value)
1498
+ continue;
1499
+ const observed = observeLiteralToken(value.trim(), input.field, input.constants);
1500
+ if (observed)
1501
+ return observed;
1502
+ }
1503
+ return undefined;
1504
+ }
1479
1505
  function observeDeterministicFunctionFieldOverride(input) {
1480
1506
  const region = scenarioFunctionRegion(input.source, input.tpId);
1481
1507
  if (!region)
@@ -1608,7 +1634,10 @@ export async function assessBackendScenarioParamConsistency(input) {
1608
1634
  constants,
1609
1635
  })
1610
1636
  : undefined;
1611
- const observed = functionFieldOverride ?? helperFieldObserved ?? observeParamFeatures(block, inferred.field, constants, parameterNames);
1637
+ const branchFieldObserved = source && inferred.field
1638
+ ? observeBranchPayloadField({ source, tpId, block, parameterNames, field: inferred.field, constants })
1639
+ : undefined;
1640
+ const observed = functionFieldOverride ?? helperFieldObserved ?? branchFieldObserved ?? observeParamFeatures(block, inferred.field, constants, parameterNames);
1612
1641
  // Request-level / health checks often have no pytest.param payload row,
1613
1642
  // or only a TP-id label positional (no field value).
1614
1643
  const requestLevelNoParam = Boolean(source) &&
@@ -26,7 +26,8 @@ export function deriveFrontendRecoveryFailureSource(result) {
26
26
  * Compute the frontend candidate-continuation reset partition:
27
27
  * - `frontend-plan-pi` failure → reset plan + design-review + policy/admission (+ descendants);
28
28
  * - `frontend-design-policy-shell` failure → reset policy/admission (+ descendants);
29
- * - admission/design-review failure → reset writer admission (+ descendants);
29
+ * - design-review rejection → reset plan + design review + admission (+ descendants);
30
+ * - other admission failures → reset writer admission (+ descendants);
30
31
  *
31
32
  * Everything upstream of the failure source is imported as verified parent facts.
32
33
  */
@@ -275,7 +275,35 @@ async function casUpdateFrontendRecoveryState(parentRunDir, expectedRevision, de
275
275
  async function materializeChildStaging(input) {
276
276
  const { cwd, parentRunDir, parentRunId, parentSpec, parentState, recovery, requestId, childRunId, stagingDir, createdAt, } = input;
277
277
  await prepareRunDir(stagingDir);
278
- await writeJsonVerified(stagingDir, "run.json", parentSpec);
278
+ // Design-review rejection must be visible to the reset planner. Recovery
279
+ // children otherwise copy the old spec and only rerun design review, which
280
+ // deterministically re-rejects the unchanged plan. Bind the parent's typed
281
+ // findings into the child spec as ordinary rerun feedback.
282
+ let childSpec = parentSpec;
283
+ if (recovery.failureSource === "frontend-plan-pi") {
284
+ try {
285
+ const { deriveDagRerunFeedback } = await import("./rerun-feedback.js");
286
+ const taskId = parentSpec.taskContractBinding?.taskId ??
287
+ parentSpec.sourceBinding?.taskId ??
288
+ parentRunId;
289
+ const feedback = await deriveDagRerunFeedback({
290
+ parentRunId,
291
+ taskId,
292
+ runDir: parentRunDir,
293
+ spec: parentSpec,
294
+ state: parentState,
295
+ });
296
+ if (feedback)
297
+ childSpec = { ...parentSpec, rerunFeedback: feedback };
298
+ }
299
+ catch (error) {
300
+ // A design rejection cannot be safely retried without its typed
301
+ // findings. Fail closed instead of deterministically re-running the
302
+ // unchanged plan and producing another rejection.
303
+ throw new Error(`frontend recovery feedback unavailable: ${error instanceof Error ? error.message : String(error)}`);
304
+ }
305
+ }
306
+ await writeJsonVerified(stagingDir, "run.json", childSpec);
279
307
  // Phase 5: writer transient partial-write recovery resets the writer subtree
280
308
  // (frontend-implement-pi) instead of the candidate producer; everything
281
309
  // upstream is imported as verified parent facts.
@@ -300,10 +328,10 @@ async function materializeChildStaging(input) {
300
328
  result: admissionRead.result,
301
329
  });
302
330
  }
303
- const { ranks } = topoSortToRanks(parentSpec);
331
+ const { ranks } = topoSortToRanks(childSpec);
304
332
  const resetSet = new Set(plan.resetNodeIds);
305
333
  const importedSet = new Set(plan.importedNodeIds);
306
- const tasksById = new Map(parentSpec.tasks.map((task) => [task.id, task]));
334
+ const tasksById = new Map(childSpec.tasks.map((task) => [task.id, task]));
307
335
  const nodes = {};
308
336
  const importedFacts = [];
309
337
  for (const nodeId of plan.importedNodeIds) {
@@ -328,7 +356,7 @@ async function materializeChildStaging(input) {
328
356
  }
329
357
  nodes[nodeId] = buildPendingNodeRecord(task);
330
358
  }
331
- for (const task of parentSpec.tasks) {
359
+ for (const task of childSpec.tasks) {
332
360
  if (!resetSet.has(task.id) && !importedSet.has(task.id)) {
333
361
  throw new Error(`node ${task.id} missing from frontend recovery plan partition`);
334
362
  }
@@ -338,7 +366,7 @@ async function materializeChildStaging(input) {
338
366
  const canonicalContract = await copyArtifactVerified({
339
367
  parentRunDir,
340
368
  newRunDir: stagingDir,
341
- relativePath: resolveCanonicalContractRelPath(parentSpec),
369
+ relativePath: resolveCanonicalContractRelPath(childSpec),
342
370
  });
343
371
  const writerAdmissionResult = await copyArtifactVerified({
344
372
  parentRunDir,
@@ -22,8 +22,9 @@ export const frontendWriterAdmissionClassificationSchema = z.enum([
22
22
  "requires-human-approval",
23
23
  "blocked",
24
24
  "stale",
25
- // Legacy fixture alias: writeSet-external test helpers still emit `admitted`.
26
- // Canonical producers always emit one of the four values above.
25
+ // Legacy artifacts remain parseable for diagnostics, but `admitted` is never
26
+ // accepted by the runtime authorization gate. Canonical producers always
27
+ // emit one of the four values above.
27
28
  "admitted",
28
29
  ]);
29
30
  /** A+B (AC-006): Micro topology uses `requirement-to-target-and-evidence`
@@ -78,16 +79,22 @@ export const frontendWriterAdmissionResultV1Schema = z
78
79
  path: z.string().optional(),
79
80
  })
80
81
  .strict()),
81
- // r12 policy change: a request_design_changes verdict no longer blocks
82
- // writer admission. The verdict and its findings ride along as
83
- // advisory context for the implement node; blocking is owned by the
84
- // deterministic gates (prewrite, write guard) and the final code
85
- // review.
82
+ // Design findings are preserved as a bounded, structured capsule. A
83
+ // request_design_changes verdict is a blocking admission decision; the
84
+ // capsule is consumed by recovery and by any explicitly authorized retry.
86
85
  designReviewAdvisory: z
87
86
  .object({
88
87
  verdict: z.enum(["approve_design", "request_design_changes"]),
89
88
  restartPhase: z.string().min(1).optional(),
90
89
  findingCount: z.number().int().nonnegative(),
90
+ findings: z.array(z.object({
91
+ severity: z.enum(["Critical", "Important", "Minor", "Info"]),
92
+ file: z.string().min(1).optional(),
93
+ line: z.number().int().positive().optional(),
94
+ issue: z.string().min(1),
95
+ requiredChange: z.string().min(1).optional(),
96
+ }).strict()),
97
+ evidenceRefs: z.array(z.string().min(1)),
91
98
  })
92
99
  .strict()
93
100
  .optional(),
@@ -115,7 +122,7 @@ export function isConcreteWriteSetPath(value) {
115
122
  const path = value.trim();
116
123
  if (path === "" || path === "." || path === "./")
117
124
  return false;
118
- if (path.startsWith("/") || path.includes("\\"))
125
+ if (path.startsWith("/") || path.includes("\\") || path.includes("*") || path.includes("?"))
119
126
  return false;
120
127
  const segments = path.split("/");
121
128
  if (segments.some((segment) => segment === ".." || segment === "**" || segment === "")) {
@@ -123,6 +130,32 @@ export function isConcreteWriteSetPath(value) {
123
130
  }
124
131
  return true;
125
132
  }
133
+ export function normalizeConcreteWriteSetPath(value) {
134
+ const normalized = value.trim();
135
+ return isConcreteWriteSetPath(normalized) ? normalized : undefined;
136
+ }
137
+ /**
138
+ * Deterministically extract repository file paths that frozen verification
139
+ * commands operate on (`--config <file>`, `node --check <file>`). The plan's
140
+ * verification targets do not always name verification infrastructure (a
141
+ * vitest config, fixtures), yet verify-shell cannot run without it and the
142
+ * writer cannot create it unless the effective writeSet authorizes the file.
143
+ */
144
+ export function collectVerificationCommandFiles(commands) {
145
+ const files = new Set();
146
+ const configRe = /--config\s+([\w@./-]+\.(?:js|mjs|cjs|ts|json))/g;
147
+ const checkRe = /node\s+--check\s+([\w@./-]+\.(?:js|mjs|cjs))/g;
148
+ for (const command of commands) {
149
+ for (const re of [configRe, checkRe]) {
150
+ for (const match of command.matchAll(re)) {
151
+ const file = match[1];
152
+ if (file && file.includes("/"))
153
+ files.add(file);
154
+ }
155
+ }
156
+ }
157
+ return [...files].sort();
158
+ }
126
159
  /** Derive the concrete writeSet from implementation targets ∪ non-static
127
160
  * verification target files. Entries are de-duplicated and lexicographically
128
161
  * sorted (frozen). */
@@ -146,7 +179,8 @@ export function deriveWriteSet(contract, options) {
146
179
  : candidates;
147
180
  const writeSet = new Set();
148
181
  for (const candidate of expanded) {
149
- if (!isConcreteWriteSetPath(candidate)) {
182
+ const normalized = normalizeConcreteWriteSetPath(candidate);
183
+ if (!normalized) {
150
184
  findings.push({
151
185
  code: "write-set-path-not-concrete",
152
186
  message: `writeSet entry is not a concrete path: ${candidate}`,
@@ -154,7 +188,7 @@ export function deriveWriteSet(contract, options) {
154
188
  });
155
189
  continue;
156
190
  }
157
- writeSet.add(candidate);
191
+ writeSet.add(normalized);
158
192
  }
159
193
  return { writeSet: [...writeSet].sort(), findings };
160
194
  }
@@ -3473,7 +3473,7 @@ async function buildFrontendHybridDagFromTask(sources) {
3473
3473
  "Audit the frontend plan before implementation. frontend-plan-pi is emitted to you as canonical full-contract JSON after the runtime applied and validated the planner's editable patch against its protected skeleton; there is no separate plan prose.",
3474
3474
  "Your authoritative terminal verdict is exactly one committed typed tool call: approve_design or request_design_changes. Call exactly one of them; after calling one, do not call the other.",
3475
3475
  "request_design_changes must carry a typed issueCategory, at least one evidenceRef, and non-empty findings.",
3476
- "Your verdict is consumed only as deterministic data input by frontend-writer-admission-shell; it no longer drives any branch or gate. request_design_changes blocks writer admission (terminal).",
3476
+ "Your verdict is consumed as deterministic data input by frontend-writer-admission-shell. approve_design permits admission; request_design_changes blocks writer admission until a recovery plan incorporates every Critical/Important finding.",
3477
3477
  "Request design changes when the Mock strategy is MOCK_STRATEGY: blocked, missing, unsupported by repository evidence, inconsistent with the API contract, outside authorized paths/dependencies, unable to prove production-default-off behavior with the fixed production/default-real-path static check, or missing deterministic behavior verification for a declared behavior target or selected Mock strategy. Mock strategies require Mock-backed evidence. A static-only contract is allowed only when every verification target is static and maps to a declared static entrypoint. not-needed otherwise requires applicable real/no-remote behavior evidence unless auto mode explicitly skipped Mock because no project Mock capability exists; in that case the plan must preserve the real request path and record the Real Integration Gap.",
3478
3478
  "Also request design changes for missing applicable UI states, unsupported dependency additions, design-system drift without reason, weak interaction coverage, broad scope, inline fake data, schema drift, or missing deterministic verification commands.",
3479
3479
  "Component selection conformance is a hard blocking condition: request_design_changes when the frontend spec (component/theme/rule.components bucket) already defines a component for a purpose but the plan selects another or self-invents one without a declared deviation; when uiComponentChoices is missing/empty for UI-visible work while the frozen component/theme bucket is non-empty; when a decision=specified specReference.path is missing a ledger OpenSpec reference or successful read event; or when a decision=new component lacks a traceable task-source/PRD specReference. A PRD reference for decision=new is not an OpenSpec citation and must not be rejected merely for lacking an OpenSpec read event.",
@@ -3554,6 +3554,7 @@ async function buildFrontendHybridDagFromTask(sources) {
3554
3554
  "Execute in fixed stages and report each in the delivery summary: (1) Contract confirm, (2) Tests sync, (3) Component/UI state implementation, (4) API/Mock wiring per contract.mockApi, (5) Focused checks behind frozen entrypoints only, (6) Diff cleanup.",
3555
3555
  "Map every requirement id, expectedOutcome, interaction trigger/expectedBehavior, and applicable UI state from the contract to concrete files. Do not invent shell verification commands; only frozen static/behavior entrypoints will run.",
3556
3556
  "For every non-static verification target, treat target.id as a stable trace token and include that exact token in a real describe/it/test literal title (for example, it('[VT-DASHBOARD-SHELL] renders the dashboard', ...)). One test title may carry multiple target ids when it proves multiple grouped behaviors; comments and ordinary strings do not count as trace evidence.",
3557
+ "Before reporting Tests Changed as done, self-audit with the executor's rule: collect ONLY the string literals passed directly to describe(/it(/test( calls in each test file and confirm every non-static target id for that file appears inside one of those literals. A token in a comment, a variable, or a non-title string does not satisfy the trace check; if any id is missing from the literal titles, edit the title strings before finishing.",
3557
3558
  "Begin implementation after the contract and its target files are confirmed. Do not spend the turn collecting optional context. If the canonical contract lacks behavior needed to edit safely, stop and state the blocking reason in the summary instead of reopening broad discovery.",
3558
3559
  "Your implementation status is derived by the executor from mechanical facts (persisted write-tool events, run delta, write guard, requirement coverage, focused-check failures), never from any IMPLEMENTATION_OUTCOME first line. Do not emit an IMPLEMENTATION_OUTCOME first line.",
3559
3560
  "The node runs a bounded micro-loop: after each write attempt the executor re-runs frozen focused checks and records a per-round diff checkpoint; the write guard stays active every round. Only repair local issues attributable to the current diff (syntax/type/import/format/unit-assert/obvious omission). Never change requirements, design, writeSet, or verification strictness inside the loop.",
@@ -5095,7 +5096,7 @@ async function buildBackendTestHybridDag(sources) {
5095
5096
  ]
5096
5097
  : []),
5097
5098
  "Include exactly one `## Module Index` table with this exact header: `| Module Stem | Business Resource | Owned Operations | Owned Rule Keys | Case IDs | Split Reason | Markdown Path | Pytest Path |`. Use canonical relative links `[label](./<stem>.md)` inside the Markdown Path cell followed by the resolved `${layout.markdownDir}/<stem>.md` path. Split Reason is exactly one of `explicit-user-layout`, `primary-business-resource`, `independent-business-resource`, or `output-budget`. Group by stable business resource/domain, not by CRUD operation, AC, parameter/field axis, scenario type or regression purpose: one resource's list/detail/create/update/delete and its filters/response assertions/regression floor belong in one module. Multiple modules owning the same exact `METHOD /path` are forbidden unless every such row is `explicit-user-layout` from primary-requirement path pairs or has a documented `output-budget` proof. Keep the total module count at the smallest safe value and never exceed 8 modules. Name model-derived modules with stable lowercase business stems such as `health` or `resource_notes`; explicit-user-layout preserves the primary requirement filename stem even when it is more specific. Do not use priority-only stems `p0`, `p1` or `p2`; Priority belongs only in the Coverage Matrix. Pure hexadecimal/hash-like opaque stems and test-purpose-only stems are forbidden. Do not use Case-ID-like module filenames. The relative link target, Markdown Path, Pytest Path and downstream automation mapping must be one-to-one and exact; for model-derived modules the default pair remains `testcase/md/<module>.md` and `testcase/test_<module>.py`, while explicit-user-layout preserves the primary requirement paths. Do not hand-write a conflicting module count in prose; the Module Index row count is the only count truth.",
5098
- "Scenario Partitions (query/filter axes): for every affected GET/list operation, declare one row per enum or classification axis used for filtering (query/path parameters such as type/status/category). Add a mandatory machine-readable `## Scenario Partitions` section after the Coverage Matrix using exactly `| Partition ID | Operation | Axis | Domain | Required Slots | Expected by Slot | Bind Rule |` with the separator row. Partition ID is a stable `SP-<OPERATION>-<AXIS>` token; Domain must copy the legal values verbatim from the bound OpenAPI enum or requirement sentence (never guess), using bare semicolon-separated identifier values inside the single table cell (for example `ACTIVE; ARCHIVED`, with no Markdown backticks or prose); Required Slots must contain `each-value` and exactly one `not-in-set`, plus `omitted` only when the parameter is optional. Before returning, expand every declared partition into its complete deterministic exact slot ID set: one `TP-<Partition ID>-<VALUE-TOKEN>` per Domain value, `TP-<Partition ID>-OMITTED` only for an optional axis, and exactly one `TP-<Partition ID>-NOT-IN-SET`. Every expanded slot ID must appear verbatim in the binding Rule's `Required Test Points` cell and be assigned to concrete Case IDs in that same Coverage Matrix row; ordinary alias/family Test Points do not replace this inventory. Scheme A: Case count may be smaller than the enum count, but every exact slot still needs an independent variant Test Point and pytest.param id; never use SINGLE/MULTIPLE aliases as coverage. Expected by Slot states the documented expectation per slot kind (`domain-value`, `default-behavior`, `empty-result`/`excluded-result` when documented, or `GAP` when the source does not document the complement expectation — never invent 空列表/400). POST/PUT body field-validation enums stay in the Coverage Matrix as `TP-<FIELD>-ENUM-*` and MUST NOT get a Scenario Partition row. Do not create partitions for axes without a documented legal-value domain. Only GET/list query or path parameters whose bound source documents a finite enum or classification set may become a Scenario Partition. Do not create partitions for free-form strings, primary keys, required-or-optional-only parameters, or boundary/format-only axes. If an axis has no finite legal-value domain, do not declare a Partition row and do not invent NOT-IN-SET cases. Cross-axis combinations stay as ONE nominal Case; never declare a cross-axis cartesian partition.",
5099
+ "Scenario Partitions (query/filter axes): inspect every affected GET/list operation for query/path parameters whose bound source documents a finite enum or classification domain. If at least one such axis exists, add exactly one machine-readable `## Scenario Partitions` section after the Coverage Matrix using exactly `| Partition ID | Operation | Axis | Domain | Required Slots | Expected by Slot | Bind Rule |` with the separator row and one row per eligible axis. If no affected axis has a source-backed finite domain, omit the entire `## Scenario Partitions` heading and section; do not emit an explanatory prose-only section. Partition ID is a stable `SP-<OPERATION>-<AXIS>` token; Domain must copy the legal values verbatim from the bound OpenAPI enum or requirement sentence (never guess), using bare semicolon-separated identifier values inside the single table cell (for example `ACTIVE; ARCHIVED`, with no Markdown backticks or prose); Required Slots must contain `each-value` and exactly one `not-in-set`, plus `omitted` only when the parameter is optional. Before returning, expand every declared partition into its complete deterministic exact slot ID set: one `TP-<Partition ID>-<VALUE-TOKEN>` per Domain value, `TP-<Partition ID>-OMITTED` only for an optional axis, and exactly one `TP-<Partition ID>-NOT-IN-SET`. Every expanded slot ID must appear verbatim in the binding Rule's `Required Test Points` cell and be assigned to concrete Case IDs in that same Coverage Matrix row; ordinary alias/family Test Points do not replace this inventory. Scheme A: Case count may be smaller than the enum count, but every exact slot still needs an independent variant Test Point and pytest.param id; never use SINGLE/MULTIPLE aliases as coverage. Expected by Slot states the documented expectation per slot kind (`domain-value`, `default-behavior`, `empty-result`/`excluded-result` when documented, or `GAP` when the source does not document the complement expectation — never invent 空列表/400). POST/PUT body field-validation enums stay in the Coverage Matrix as `TP-<FIELD>-ENUM-*` and MUST NOT get a Scenario Partition row. Do not create partitions for axes without a documented legal-value domain. Only GET/list query or path parameters whose bound source documents a finite enum or classification set may become a Scenario Partition. Do not create partitions for free-form strings, primary keys, required-or-optional-only parameters, or boundary/format-only axes. If an axis has no finite legal-value domain, do not declare a Partition row and do not invent NOT-IN-SET cases. Cross-axis combinations stay as ONE nominal Case; never declare a cross-axis cartesian partition.",
5099
5100
  "Before finalizing README, calculate the predicted collected-item count as `sum(max(1, number of variant Test Points in each Case))`. If the task declares an item budget, the prediction must not exceed it. Reduce excess only by removing duplicate execution and converting same-request checkpoints to assertions; never drop required rules, boundaries, enums, operation-specific inputs, or business states. Record the prediction in README. Use only environment-supported fixtures/targets/isolation, record evidence gaps in Chinese, and do not emit JSON, pytest, or execute commands.",
5100
5101
  ...(sharedSetupPrompt ? [sharedSetupPrompt] : []),
5101
5102
  intake.boundedSourceContext,
@@ -6,6 +6,7 @@ import { pathMatchesPattern } from "../../shared/git-progress.js";
6
6
  import { redactSecrets, truncateUtf8Preview } from "../../shared/preview.js";
7
7
  import { recordDecisionEnvelopeForNode, shouldPauseOnHumanEscalation, writeHumanEscalationArtifacts, } from "./decision-envelope.js";
8
8
  import { FRONTEND_PREWRITE_RESULT_SOURCE_ARTIFACT, FRONTEND_WRITER_NODE_IDS, isFrontendWriterAuthorized, readFrontendPrewriteResult, } from "./scheduler.js";
9
+ import { collectVerificationCommandFiles, } from "./frontend-writer-admission.js";
9
10
  import { writeNodeRecord, writeNodeSkillArtifacts } from "./run-store.js";
10
11
  import { resolveContextPolicy } from "./context-policy.js";
11
12
  import { buildDagNodePromptEnvelope, formatConvergenceFeedbackBlock, } from "./prompt.js";
@@ -1282,6 +1283,52 @@ export async function executeDagNode(input) {
1282
1283
  await skipFrontendWriter(record);
1283
1284
  return;
1284
1285
  }
1286
+ // The admission artifact is the effective authorization boundary. Never
1287
+ // leave the writer using the broad task glob after the shell has frozen a
1288
+ // concrete set: doing so makes the receipt auditable but unenforceable.
1289
+ // Files referenced by frozen verification commands are unioned in: the
1290
+ // plan's verification targets do not always name verification
1291
+ // infrastructure, yet verify-shell cannot run without it and the writer
1292
+ // must be authorized to create it.
1293
+ const admissionWriteSetEntries = admission.result.writeSet.map((entry) => entry.trim().replace(/\\/g, "/").replace(/^\.\//, ""));
1294
+ const frozenVerificationBundle = spec.tasks.find((specTask) => specTask.shell?.frontendVerificationBundle)?.shell?.frontendVerificationBundle;
1295
+ const verificationCommandFiles = frozenVerificationBundle
1296
+ ? collectVerificationCommandFiles([
1297
+ ...(frozenVerificationBundle.staticCommands ?? []),
1298
+ ...(frozenVerificationBundle.behaviorCommands ?? []),
1299
+ ...(frozenVerificationBundle.mockCommands ?? []),
1300
+ ...(frozenVerificationBundle.lintCommands ?? []),
1301
+ ])
1302
+ : [];
1303
+ const extraVerificationFiles = verificationCommandFiles.filter((file) => !admissionWriteSetEntries.includes(file) &&
1304
+ file &&
1305
+ !file.includes("*") &&
1306
+ !file.includes("?") &&
1307
+ !file.split("/").some((segment) => segment === "..") &&
1308
+ task.allowedPaths.some((allowed) => pathMatchesPattern(file, allowed)) &&
1309
+ !task.forbiddenPaths.some((forbidden) => pathMatchesPattern(file, forbidden)));
1310
+ const admittedWriteSet = [
1311
+ ...new Set([...admissionWriteSetEntries, ...extraVerificationFiles]),
1312
+ ];
1313
+ if (admittedWriteSet.length === 0 ||
1314
+ new Set(admittedWriteSet).size !== admittedWriteSet.length ||
1315
+ admittedWriteSet.some((entry) => !entry ||
1316
+ entry.includes("*") ||
1317
+ entry.includes("?") ||
1318
+ entry.split("/").some((segment) => segment === "..") ||
1319
+ !task.allowedPaths.some((allowed) => pathMatchesPattern(entry, allowed)) ||
1320
+ task.forbiddenPaths.some((forbidden) => pathMatchesPattern(entry, forbidden)))) {
1321
+ await failBeforePrompt(new Error("frontend writer admission contains an invalid effective writeSet"), "final-write-set-approval-invalid");
1322
+ return;
1323
+ }
1324
+ task = { ...task, writeSet: admittedWriteSet };
1325
+ node.runtimeWriteAuthorization = {
1326
+ schemaVersion: 1,
1327
+ status: "validated",
1328
+ approvalSourceNodeId: FRONTEND_PREWRITE_RESULT_SOURCE_ARTIFACT,
1329
+ approvalDigest: admission.result.admissionDigest,
1330
+ effectiveWriteSet: [...admittedWriteSet],
1331
+ };
1285
1332
  node.frontendWriterAdmission = record;
1286
1333
  }
1287
1334
  let projectGovernanceContext;
@@ -1457,6 +1504,26 @@ export async function executeDagNode(input) {
1457
1504
  return;
1458
1505
  }
1459
1506
  }
1507
+ if (FRONTEND_WRITER_NODE_IDS.includes(nodeId)) {
1508
+ // Design-review findings are not reliable in the provider's prose output
1509
+ // (typed terminal nodes commonly return an empty assistant message). Inject
1510
+ // the bounded admission capsule explicitly so an authorized retry has the
1511
+ // reviewer's concrete issue/evidence context.
1512
+ try {
1513
+ const admission = await readFrontendPrewriteResult(runDir);
1514
+ const advisory = admission.ok ? admission.result.designReviewAdvisory : undefined;
1515
+ if (advisory?.findings?.length || advisory?.evidenceRefs?.length) {
1516
+ prompt = `${prompt}\n\n<design_review_findings>\n${JSON.stringify({
1517
+ verdict: advisory.verdict,
1518
+ findings: advisory.findings.slice(0, 16),
1519
+ evidenceRefs: advisory.evidenceRefs.slice(0, 16),
1520
+ })}\n</design_review_findings>\nAddress every Critical/Important finding before writing.`;
1521
+ }
1522
+ }
1523
+ catch {
1524
+ // Admission is already enforced above; prompt enrichment is best effort.
1525
+ }
1526
+ }
1460
1527
  node.resolvedSkills = resolvedSkills;
1461
1528
  await writeNodeSkillArtifacts(runDir, nodeId, resolvedSkills);
1462
1529
  let model = resolveModelForTask(task, spec.executorModels);
@@ -1568,6 +1635,9 @@ export async function executeDagNode(input) {
1568
1635
  return acc;
1569
1636
  }, {});
1570
1637
  let attemptPrompt = buildAttemptPrompt(task, prompt, attemptNumber, previousFailureCategory, previousProtocolReason, recoveryTargetPaths, recoveryDiagnostics, frontendPlanRetryStep);
1638
+ if (task.id === "frontend-scout-pi" && attemptNumber > 1) {
1639
+ attemptPrompt = `${attemptPrompt}\n\nSCOUT RETRY (reuse existing evidence): preserve all committed target-surface/design-evidence facts and do not re-read files already covered by the prior attempt. Inspect only unresolvedPaths or missing target-surface fields, then commit the minimal correction. If the existing facts are complete, commit the same canonical facts without broad rediscovery.`;
1640
+ }
1571
1641
  if (attemptNumber > 1 &&
1572
1642
  // The plan node (frontend-plan-pi) is a typed-facts ladder task:
1573
1643
  // its retry is driven by the §5.1 ladder (compact-terminal-first
@@ -1793,6 +1863,9 @@ export async function executeDagNode(input) {
1793
1863
  sdkAttempted: result.sdkAttempted,
1794
1864
  tokensUsed: result.tokensUsed,
1795
1865
  parsedEvents: result.parsedEvents,
1866
+ stopReason: result.stopReason,
1867
+ thinkingObserved: result.thinkingObserved,
1868
+ writeToolCallCount: result.writeToolCallCount,
1796
1869
  artifactPath: `${nodeId}/attempt-${attemptNumber}.json`,
1797
1870
  };
1798
1871
  if (retryPolicy !== undefined) {
@@ -1874,6 +1947,9 @@ export async function executeDagNode(input) {
1874
1947
  retryPolicy === undefined
1875
1948
  ? result.parsedEvents
1876
1949
  : sumAttemptMetric(attempts, (attempt) => attempt.parsedEvents);
1950
+ node.stopReason = result.stopReason;
1951
+ node.thinkingObserved = result.thinkingObserved;
1952
+ node.writeToolCallCount = result.writeToolCallCount;
1877
1953
  node.lastActivityAt = attemptFinishedAt;
1878
1954
  if (result.failureCategory === "termination-unconfirmed") {
1879
1955
  node.needsAttentionReason = "attempt-termination-unconfirmed";
@@ -399,6 +399,65 @@ export async function deriveDagRerunFeedback(input) {
399
399
  },
400
400
  });
401
401
  }
402
+ // Preserve actionable writer failures even when the run stopped before a
403
+ // final review node could commit typed findings. This is especially
404
+ // important for provider length exhaustion: the next run must see the
405
+ // stop reason/tool statistics instead of repeating the same prompt.
406
+ const writerNodeId = ["frontend-implement-pi", "implement-pi"].find((nodeId) => {
407
+ const node = input.state.nodes[nodeId];
408
+ return (node?.status === "ERROR" &&
409
+ [
410
+ "empty-output",
411
+ "invalid-output",
412
+ "writer-thinking-exhausted",
413
+ "writer-budget-exhausted",
414
+ "incomplete-write-set",
415
+ "partial-success-with-context-overflow",
416
+ ].includes(node.failureCategory ?? ""));
417
+ });
418
+ if (writerNodeId) {
419
+ const writerNode = await readNodeRecord(input.runDir, pass, undefined, writerNodeId, input.state.nodes[writerNodeId]);
420
+ if (writerNode) {
421
+ const writerFeedback = [
422
+ `writer failure category: ${writerNode.failureCategory ?? "unknown"}`,
423
+ writerNode.stopReason ? `provider stopReason: ${writerNode.stopReason}` : "",
424
+ typeof writerNode.writeToolCallCount === "number"
425
+ ? `write tool calls: ${writerNode.writeToolCallCount}`
426
+ : "",
427
+ writerNode.stderr?.trim() ?? "",
428
+ canonicalNodeOutput(writerNode),
429
+ ]
430
+ .filter(Boolean)
431
+ .join("\n\n");
432
+ const parentSourceBindingHash = sourceBindingHash(input.spec);
433
+ return withDigest({
434
+ schemaVersion: 1,
435
+ kind: "unresolved-terminal-feedback",
436
+ parentRunId: input.parentRunId,
437
+ taskId: input.taskId,
438
+ sourceNodeId: writerNodeId,
439
+ verdict: "request-revision",
440
+ failureCategory: writerNode.failureCategory,
441
+ feedbackText: scrubAndBoundFeedbackText(writerFeedback, input.runDir, input.state.cwd),
442
+ evidenceRefs: [
443
+ {
444
+ nodeId: writerNodeId,
445
+ preservedNodeRecordPath: `${writerNodeId}.json`,
446
+ },
447
+ ],
448
+ attemptSummary: { completedPasses: 1, maxPasses: 1 },
449
+ binding: {
450
+ ...(parentSourceBindingHash
451
+ ? { sourceCanonicalHash: parentSourceBindingHash }
452
+ : {}),
453
+ ...(input.spec.taskContractBinding?.canonicalHash
454
+ ? { taskContractCanonicalHash: input.spec.taskContractBinding.canonicalHash }
455
+ : {}),
456
+ status: "unknown",
457
+ },
458
+ });
459
+ }
460
+ }
402
461
  return undefined;
403
462
  }
404
463
  const completedPasses = pass?.pass ?? Math.max(1, input.state.convergence?.currentPass ?? 1);