@tea-agent/loop-agent 0.42.0-next.15 → 0.42.0-next.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. package/CHANGELOG.md +12 -0
  2. package/dist/application/task-lifecycle/advance.js +9 -4
  3. package/dist/build-stamp.json +3 -3
  4. package/dist/commands/task-source-prepare.js +3 -1
  5. package/dist/executors/dag-pi-executor.js +825 -65
  6. package/dist/executors/shell-executor.js +103 -35
  7. package/dist/shared/dag-failure-category.js +6 -0
  8. package/dist/task/contract/apply.js +36 -2
  9. package/dist/task/source-prepare/parse-intent.js +7 -0
  10. package/dist/workflows/dag/dag-retry-schema.js +3 -0
  11. package/dist/workflows/dag/frontend-design-policy.js +1 -1
  12. package/dist/workflows/dag/frontend-risk.js +2 -0
  13. package/dist/workflows/dag/frontend-shadow-dual-write.js +16 -1
  14. package/dist/workflows/dag/frontend-shape.js +16 -6
  15. package/dist/workflows/dag/frontend-test-execution-evidence.js +48 -0
  16. package/dist/workflows/dag/frontend-verification-trace.js +32 -0
  17. package/dist/workflows/dag/init-hybrid.js +9 -10
  18. package/dist/workflows/dag/node-execution.js +124 -72
  19. package/dist/workflows/dag/rerun-feedback.js +1 -0
  20. package/dist/workflows/dag/rerun-plan.js +90 -4
  21. package/dist/workflows/dag/rerun-run.js +7 -0
  22. package/dist/workflows/dag/retry-policy.js +16 -10
  23. package/dist/workflows/dag/runner-exit-diagnostics.js +125 -0
  24. package/dist/workflows/dag/runner.js +67 -7
  25. package/dist/workflows/dag/structured-output-repair.js +4 -1
  26. package/dist/workflows/dag/validate.js +10 -8
  27. package/dist/workflows/dag/workspace-checkpoint.js +66 -0
  28. package/docs/templates/agent-dag.schema.json +2 -2
  29. package/package.json +1 -1
  30. package/skills/frontend-plan/SKILL.md +6 -1
  31. package/skills/frontend-plan/references/decision-contract.md +69 -18
  32. package/skills/frontend-plan/references/design-decisions.md +32 -0
@@ -1,5 +1,6 @@
1
1
  import { spawn } from "node:child_process";
2
- import { createHash } from "node:crypto";
2
+ import { createHash, randomUUID } from "node:crypto";
3
+ import { withFrontendTestReporter, parseFrontendTestExecutionReport } from "../workflows/dag/frontend-test-execution-evidence.js";
3
4
  import { appendFileSync, existsSync, mkdirSync, writeFileSync } from "node:fs";
4
5
  import { mkdir, readFile, stat, writeFile } from "node:fs/promises";
5
6
  import path from "node:path";
@@ -1950,6 +1951,23 @@ async function executeFrontendVerificationBundle(input, meta) {
1950
1951
  const bundle = shell.frontendVerificationBundle;
1951
1952
  const cwd = resolveShellCwd(input.cwd, shell.cwd);
1952
1953
  const results = [];
1954
+ const commandResultsByKey = new Map();
1955
+ const commandKey = (command) => `${cwd}\u0000${command.trim()}`;
1956
+ const reuseCommandResult = (result, command) => {
1957
+ const reusedAt = new Date().toISOString();
1958
+ return {
1959
+ ...result,
1960
+ command,
1961
+ durationMs: 0,
1962
+ wallDurationMs: 0,
1963
+ executionDurationMs: result.executionDurationMs ?? result.durationMs,
1964
+ startedAt: reusedAt,
1965
+ finishedAt: reusedAt,
1966
+ stdoutArtifactPath: undefined,
1967
+ stderrArtifactPath: undefined,
1968
+ reused: true,
1969
+ };
1970
+ };
1953
1971
  let beforeStatus;
1954
1972
  try {
1955
1973
  beforeStatus = await readGitStatusPorcelain(input.cwd);
@@ -1958,22 +1976,32 @@ async function executeFrontendVerificationBundle(input, meta) {
1958
1976
  beforeStatus = undefined;
1959
1977
  }
1960
1978
  const lintResults = [];
1979
+ const successfulLabels = new Map();
1980
+ const successfulCommandTexts = new Map();
1981
+ const executedTests = new Map();
1982
+ const commandExecutionEvidence = new Map();
1961
1983
  for (const command of bundle.lintCommands ?? []) {
1962
- const commandNumber = results.length + 1;
1963
- const result = await executeShellCommand({
1964
- command,
1965
- cwd,
1966
- timeoutMs: shell.timeoutMs ?? DEFAULT_SHELL_TIMEOUT_MS,
1967
- envAllowlist: shell.envAllowlist,
1968
- dagRunMeta: { runDir: meta.runDir, runId: meta.runId },
1969
- reportActivity: input.reportActivity,
1970
- outputArtifacts: {
1971
- stdoutPath: path.join(meta.runDir, input.task.id, "commands", `${commandNumber}.stdout.txt`),
1972
- stderrPath: path.join(meta.runDir, input.task.id, "commands", `${commandNumber}.stderr.txt`),
1973
- },
1974
- });
1975
- lintResults.push(result);
1984
+ const key = commandKey(command);
1985
+ const cached = commandResultsByKey.get(key);
1986
+ let result = cached ? reuseCommandResult(cached, command) : undefined;
1987
+ if (!result) {
1988
+ const commandNumber = results.length + 1;
1989
+ result = await executeShellCommand({
1990
+ command,
1991
+ cwd,
1992
+ timeoutMs: shell.timeoutMs ?? DEFAULT_SHELL_TIMEOUT_MS,
1993
+ envAllowlist: shell.envAllowlist,
1994
+ dagRunMeta: { runDir: meta.runDir, runId: meta.runId },
1995
+ reportActivity: input.reportActivity,
1996
+ outputArtifacts: {
1997
+ stdoutPath: path.join(meta.runDir, input.task.id, "commands", `${commandNumber}.stdout.txt`),
1998
+ stderrPath: path.join(meta.runDir, input.task.id, "commands", `${commandNumber}.stderr.txt`),
1999
+ },
2000
+ });
2001
+ commandResultsByKey.set(key, result);
2002
+ }
1976
2003
  results.push(result);
2004
+ lintResults.push(result);
1977
2005
  }
1978
2006
  let lintAssessment;
1979
2007
  if ((bundle.lintCommands?.length ?? 0) > 0 &&
@@ -2007,30 +2035,64 @@ async function executeFrontendVerificationBundle(input, meta) {
2007
2035
  labels: bundle.behaviorEvidence.commandLabels,
2008
2036
  },
2009
2037
  ];
2010
- const successfulLabels = new Map();
2038
+ const failedGroups = new Set();
2039
+ const groupFailures = [];
2011
2040
  for (const group of lintBlocked ? [] : groups) {
2012
2041
  for (const [index, command] of group.commands.entries()) {
2013
- const commandNumber = results.length + 1;
2014
- const result = await executeShellCommand({
2015
- command,
2016
- cwd,
2017
- timeoutMs: shell.timeoutMs ?? DEFAULT_SHELL_TIMEOUT_MS,
2018
- envAllowlist: shell.envAllowlist,
2019
- dagRunMeta: { runDir: meta.runDir, runId: meta.runId },
2020
- reportActivity: input.reportActivity,
2021
- outputArtifacts: {
2022
- stdoutPath: path.join(meta.runDir, input.task.id, "commands", `${commandNumber}.stdout.txt`),
2023
- stderrPath: path.join(meta.runDir, input.task.id, "commands", `${commandNumber}.stderr.txt`),
2024
- },
2025
- });
2042
+ if (failedGroups.has(group.name))
2043
+ break;
2044
+ const key = commandKey(command);
2045
+ const cached = commandResultsByKey.get(key);
2046
+ let result = cached ? reuseCommandResult(cached, command) : undefined;
2047
+ if (!result) {
2048
+ const commandNumber = results.length + 1;
2049
+ const reportPath = path.join(meta.runDir, input.task.id, "commands", `${commandNumber}-${randomUUID()}.vitest.json`);
2050
+ const observedCommand = await withFrontendTestReporter({ command, cwd, reportPath });
2051
+ if (observedCommand)
2052
+ await mkdir(path.dirname(reportPath), { recursive: true });
2053
+ result = await executeShellCommand({
2054
+ command: observedCommand ?? command,
2055
+ cwd,
2056
+ timeoutMs: shell.timeoutMs ?? DEFAULT_SHELL_TIMEOUT_MS,
2057
+ envAllowlist: shell.envAllowlist,
2058
+ dagRunMeta: { runDir: meta.runDir, runId: meta.runId },
2059
+ reportActivity: input.reportActivity,
2060
+ outputArtifacts: {
2061
+ stdoutPath: path.join(meta.runDir, input.task.id, "commands", `${commandNumber}.stdout.txt`),
2062
+ stderrPath: path.join(meta.runDir, input.task.id, "commands", `${commandNumber}.stderr.txt`),
2063
+ },
2064
+ });
2065
+ let reportError;
2066
+ if (observedCommand && result.ok) {
2067
+ try {
2068
+ commandExecutionEvidence.set(key, parseFrontendTestExecutionReport(JSON.parse(await readFile(reportPath, "utf8")), cwd));
2069
+ }
2070
+ catch (error) {
2071
+ reportError = `frontend test execution report unavailable: ${error instanceof Error ? error.message : String(error)}`;
2072
+ }
2073
+ }
2074
+ if (reportError) {
2075
+ result = { ...result, ok: false, failureCategory: "invalid-output", stderr: reportError };
2076
+ }
2077
+ commandResultsByKey.set(key, result);
2078
+ }
2026
2079
  results.push(result);
2027
2080
  if (result.ok && group.labels[index]) {
2081
+ const observations = commandExecutionEvidence.get(key);
2082
+ if (observations !== undefined)
2083
+ executedTests.set(group.labels[index], observations);
2028
2084
  const labels = successfulLabels.get(group.name) ?? [];
2029
2085
  labels.push(group.labels[index]);
2030
2086
  successfulLabels.set(group.name, labels);
2087
+ const texts = successfulCommandTexts.get(group.name) ?? [];
2088
+ texts.push(command);
2089
+ successfulCommandTexts.set(group.name, texts);
2031
2090
  }
2032
- if (!result.ok)
2091
+ if (!result.ok) {
2092
+ groupFailures.push(result);
2093
+ failedGroups.add(group.name);
2033
2094
  break;
2095
+ }
2034
2096
  }
2035
2097
  }
2036
2098
  if (beforeStatus !== undefined) {
@@ -2060,12 +2122,15 @@ async function executeFrontendVerificationBundle(input, meta) {
2060
2122
  exitCode: result.exitCode,
2061
2123
  failureCategory: result.failureCategory,
2062
2124
  command: result.command,
2125
+ reused: result.reused === true,
2063
2126
  }));
2127
+ // Lint baseline-debt is tolerated, so a lint-only failure must not become the
2128
+ // verification failure. A shared lint/static or lint/behavior command is
2129
+ // different: once reused by a non-lint lane, its failure belongs to that lane
2130
+ // and must remain the primary failure evidence.
2064
2131
  const firstFailure = lintBlocked
2065
2132
  ? lintResults.find((result) => !result.ok)
2066
- : results
2067
- .filter((result) => !lintResults.includes(result))
2068
- .find((result) => !result.ok);
2133
+ : groupFailures[0];
2069
2134
  const lintSyntheticFailure = lintBlocked && !firstFailure
2070
2135
  ? {
2071
2136
  failureCategory: "invalid-output",
@@ -2082,7 +2147,7 @@ async function executeFrontendVerificationBundle(input, meta) {
2082
2147
  mock: {
2083
2148
  nodeId: input.task.id,
2084
2149
  commandLabels: successfulLabels.get("mock") ?? [],
2085
- commandTexts: bundle.mockCommands.slice(0, successfulLabels.get("mock")?.length ?? 0),
2150
+ commandTexts: successfulCommandTexts.get("mock") ?? [],
2086
2151
  },
2087
2152
  static: {
2088
2153
  nodeId: input.task.id,
@@ -2091,6 +2156,7 @@ async function executeFrontendVerificationBundle(input, meta) {
2091
2156
  behavior: {
2092
2157
  nodeId: input.task.id,
2093
2158
  commandLabels: successfulLabels.get("behavior") ?? [],
2159
+ executedTests,
2094
2160
  },
2095
2161
  },
2096
2162
  });
@@ -2123,7 +2189,9 @@ async function executeFrontendVerificationBundle(input, meta) {
2123
2189
  nodeId: failedNodeId,
2124
2190
  rawFailureCategory: failureCategory,
2125
2191
  });
2126
- const failureOwner = classifiedOwner === "unknown" ? "implement-pi" : classifiedOwner;
2192
+ const failureOwner = classifiedOwner === "unknown" &&
2193
+ ["typecheck", "build", "lint", "component-test", "unit-test", "trace"].includes(failureClass)
2194
+ ? "implementation" : classifiedOwner;
2127
2195
  const restartPhase = classifyRestartPhase(failureOwner);
2128
2196
  // A+B (AC-008): verification-config routing. Produce structured
2129
2197
  // unresolved entrypoint refs from committed plan facts and the frozen
@@ -63,6 +63,12 @@ const RAW_TO_NORMALIZED = {
63
63
  // compaction / fresh context before retrying — an executor-side
64
64
  // environment condition, not a plan-quality defect.
65
65
  "context-overflow": "executor",
66
+ // stopReason=length output-capacity truncation: same executor-side
67
+ // capacity class as context-overflow. Legacy thinking-exhausted labels
68
+ // (pre-output-limit artifacts) map to the same class.
69
+ "output-limit": "executor",
70
+ "writer-thinking-exhausted": "executor",
71
+ "planner-thinking-exhausted": "executor",
66
72
  // Deterministic capability/policy refusal: missing verified launcher JS
67
73
  // entry, capability allowlist rejection, or missing bounded write tool
68
74
  // profile. These are environment/config defects the operator repairs
@@ -147,7 +147,7 @@ export async function applyTaskContract(input) {
147
147
  ref,
148
148
  changedFiles: [],
149
149
  idempotentReplay: true,
150
- txId: existing.txId,
150
+ ...(existing.txId ? { txId: existing.txId } : { noOp: true }),
151
151
  };
152
152
  }
153
153
  const lock = await acquireTaskContractLock({
@@ -175,7 +175,9 @@ export async function applyTaskContract(input) {
175
175
  ref,
176
176
  changedFiles: [],
177
177
  idempotentReplay: true,
178
- txId: existingUnderLock.txId,
178
+ ...(existingUnderLock.txId
179
+ ? { txId: existingUnderLock.txId }
180
+ : { noOp: true }),
179
181
  };
180
182
  }
181
183
  const state = await observeTaskContractState(input.repoRoot, input.taskId);
@@ -251,6 +253,38 @@ export async function applyTaskContract(input) {
251
253
  sourceHashes,
252
254
  taskConfigSha256,
253
255
  });
256
+ // Content-level no-op: when the freshly projected draft hashes
257
+ // identically to both the recorded contract and the on-disk observation,
258
+ // the journaled commit would only churn revision/updatedAt metadata and
259
+ // force downstream consumers (writeSet gate digest binds
260
+ // contractRevision+contractCanonicalHash) to re-issue approvals for
261
+ // zero content change. Return the current ref untouched instead. A
262
+ // drifted disk (observed !== ref) never reaches this short-circuit: the
263
+ // apply is then a repair projection and must run.
264
+ if (state.ref &&
265
+ canonicalHash === state.ref.canonicalHash &&
266
+ state.observedCanonicalHash === state.ref.canonicalHash) {
267
+ await assertLockHeld(lock);
268
+ await appendRequestLedgerEntry({
269
+ repoRoot: input.repoRoot,
270
+ taskId: input.taskId,
271
+ entry: {
272
+ schemaVersion: 1,
273
+ requestId: input.requestId,
274
+ requestPayloadSha256: input.requestPayloadSha256,
275
+ revision: state.ref.revision,
276
+ canonicalHash: state.ref.canonicalHash,
277
+ recordedAt: new Date().toISOString(),
278
+ resultSummary: { ok: true, operation: "apply" },
279
+ },
280
+ });
281
+ return {
282
+ ref: state.ref,
283
+ changedFiles: [],
284
+ idempotentReplay: false,
285
+ noOp: true,
286
+ };
287
+ }
254
288
  const nextRevision = currentRevision + 1;
255
289
  const nextRef = {
256
290
  schemaVersion: TASK_CONTRACT_REF_SCHEMA_VERSION,
@@ -13,6 +13,13 @@ const HEADING_ALIASES = {
13
13
  "需求范围",
14
14
  "已确认范围",
15
15
  "已确认需求",
16
+ // Common zh-CN frontend-PRD variants seen in real PRDs; without these a
17
+ // raw-PRD fallback parse (semantic intake unavailable) yields an empty
18
+ // scope and flips the import to incomplete (EMPTY scope deadlock).
19
+ "目标表面",
20
+ "交付范围",
21
+ "实现范围",
22
+ "交付内容",
16
23
  ]),
17
24
  nonGoals: new Set(["非目标", "out of scope", "non-goals", "non goals"]),
18
25
  acceptance: new Set([
@@ -32,7 +32,10 @@ export const DEFAULT_DAG_RETRY_CATEGORIES = [
32
32
  "rate-limit",
33
33
  "unavailable",
34
34
  "empty-output",
35
+ "output-limit",
35
36
  ];
37
+ /** Provider stopReason=length: output/context capacity ended the turn early. */
38
+ export const OUTPUT_LIMIT_RETRY_CATEGORY = "output-limit";
36
39
  export const STRUCTURED_OUTPUT_RETRY_CATEGORY = "output-too-large";
37
40
  export const PROTOCOL_INVALID_RETRY_CATEGORY = "protocol-invalid";
38
41
  /** Recoverable model artifact/schema formatting failure on read-only structured nodes. */
@@ -259,7 +259,7 @@ export function checkTargetPaths(contract, writeSetPatterns) {
259
259
  if (contract.targets.files.length > MAX_WRITE_SET_FILES) {
260
260
  findings.push({
261
261
  code: "write-set-too-large",
262
- message: `implementation writeSet has ${contract.targets.files.length} files (limit ${MAX_WRITE_SET_FILES}); split the task (record_split_proposal) so no single implement node exceeds the bound`,
262
+ message: `implementation writeSet has ${contract.targets.files.length} files (limit ${MAX_WRITE_SET_FILES}); split the task at the Contract boundary so no single implement node exceeds the bound; preserve all required files`,
263
263
  });
264
264
  }
265
265
  for (const file of contract.targets.files) {
@@ -70,8 +70,10 @@ function blobOf(input) {
70
70
  ].join("\n");
71
71
  }
72
72
  function pathSpreadScore(paths) {
73
+ const auxiliaryRoot = /^(?:test|tests|__tests__|e2e|cypress|spec|specs|docs|documentation)$/i;
73
74
  const tops = new Set(paths
74
75
  .map((p) => p.replace(/\\/g, "/").split("/").filter(Boolean)[0] ?? "")
76
+ .filter((root) => !auxiliaryRoot.test(root))
75
77
  .filter(Boolean));
76
78
  return tops.size;
77
79
  }
@@ -485,7 +485,22 @@ export function assemblePlanPatchFromCommittedFacts(records, contractRequirement
485
485
  }
486
486
  const verificationTargets = verificationTargetOrder
487
487
  .map((id) => verificationTargetById.get(id))
488
- .filter((entry) => Boolean(entry));
488
+ .filter((entry) => Boolean(entry))
489
+ .map((entry) => {
490
+ // Coverage precedes UX. A later state-flow fact explicitly naming
491
+ // this VT establishes the reverse binding without another model call.
492
+ // No path/name heuristics: unbound or ambiguous decisions stay subject
493
+ // to canonical validation. Recompute from live states after removals.
494
+ const boundStates = [...stateByName.values()]
495
+ .filter((state) => state.applicable === true &&
496
+ asStringArray(state.verificationTargetIds).includes(asString(entry.id)))
497
+ .map((state) => asString(state.name));
498
+ if (boundStates.length === 0)
499
+ return entry;
500
+ return { ...entry, uiStates: [...new Set([
501
+ ...asStringArray(entry.uiStates), ...boundStates,
502
+ ])].sort() };
503
+ });
489
504
  if (verificationTargets.length > 0) {
490
505
  patch.verificationTargets = verificationTargets;
491
506
  }
@@ -46,8 +46,15 @@ const EXECUTABLE_SHAPE_RANK = {
46
46
  high: 3,
47
47
  };
48
48
  function pathSpreadScore(paths) {
49
+ // Verification-only roots are routinely authorized together with the
50
+ // implementation root (for example `src/**` + `test/**`). They do not make
51
+ // a UI change cross-domain, so exclude them from the topology spread signal.
52
+ // Keep product packages and source roots intact: `apps/**` + `packages/**`
53
+ // must still count as a real spread and prevent an unsafe small topology.
54
+ const auxiliaryRoot = /^(?:test|tests|__tests__|e2e|cypress|spec|specs|docs|documentation)$/i;
49
55
  const tops = new Set(paths
50
56
  .map((p) => p.replace(/\\/g, "/").split("/").filter(Boolean)[0] ?? "")
57
+ .filter((root) => !auxiliaryRoot.test(root))
51
58
  .filter(Boolean));
52
59
  return tops.size;
53
60
  }
@@ -65,6 +72,13 @@ function collectEscalationSignals(blob, supervised) {
65
72
  }
66
73
  return [...new Set(signals)];
67
74
  }
75
+ /** Negation applies to the API mention itself, never to unrelated Mock policy. */
76
+ function hasFrontendRemoteSignal(blob) {
77
+ const remaining = blob
78
+ .replace(/\bno\s+api\b/gi, "")
79
+ .replace(/无\s*(?:api\b|接口)/gi, "");
80
+ return /\b(api|fetch|axios|graphql|endpoint)\b|远程|接口/i.test(remaining);
81
+ }
68
82
  function resolveBaselineFrontendTaskShape(input) {
69
83
  const blob = [
70
84
  input.requirementMarkdown ?? "",
@@ -84,9 +98,7 @@ function resolveBaselineFrontendTaskShape(input) {
84
98
  const paths = input.allowedPaths ?? [];
85
99
  const spread = pathSpreadScore(paths);
86
100
  const concentrated = spread <= 1 && paths.length <= 4;
87
- const hasApiSignal = /\b(api|fetch|axios|graphql|endpoint|远程|接口)\b/i.test(blob) &&
88
- !(/(?:^|[^\w])(?:no\s+api|not-needed)(?:$|[^\w])/i.test(blob) ||
89
- /无\s*api|无\s*接口/i.test(blob));
101
+ const hasApiSignal = hasFrontendRemoteSignal(blob);
90
102
  const complexity = (input.complexity ?? "").toLowerCase();
91
103
  // split-required wins over everything: it is a termination/escalation
92
104
  // signal, not an executable topology.
@@ -266,9 +278,7 @@ export function deriveFrontendRuntimeShapeFacts(input) {
266
278
  const supervised = /\bsupervised\b/i.test(blob);
267
279
  const escalationSignals = collectEscalationSignals(blob, supervised);
268
280
  const paths = input.allowedPaths ?? [];
269
- const hasApiSignal = /\b(api|fetch|axios|graphql|endpoint|远程|接口)\b/i.test(blob) &&
270
- !(/(?:^|[^\w])(?:no\s+api|not-needed)(?:$|[^\w])/i.test(blob) ||
271
- /无\s*api|无\s*接口/i.test(blob));
281
+ const hasApiSignal = hasFrontendRemoteSignal(blob);
272
282
  const requiresFull = escalationSignals.length > 0 || pathSpreadScore(paths) > 1 || hasApiSignal;
273
283
  return {
274
284
  splitRequired: input.splitRequired,
@@ -0,0 +1,48 @@
1
+ import path from "node:path";
2
+ import { readFile } from "node:fs/promises";
3
+ import { z } from "zod";
4
+ /** Observe supported frozen Vitest commands without changing their selection. */
5
+ export async function withFrontendTestReporter(input) {
6
+ // Compound commands/wrappers need their own adapter: never append flags to
7
+ // a different shell command, change selection, or override a user reporter.
8
+ if (/[;&|<>`\n\r$]/.test(input.command) || /--(?:reporter|outputFile)/.test(input.command))
9
+ return undefined;
10
+ const direct = /^(?:(?:npx|pnpm exec|yarn exec)\s+)?vitest\s+run(?:\s|$)/.test(input.command.trim());
11
+ const npm = /^npm\s+(?:run\s+([\w:-]+)|(test))(?=\s|$)(.*)$/.exec(input.command.trim());
12
+ let separator = "";
13
+ if (!direct) {
14
+ if (!npm)
15
+ return undefined;
16
+ let manifest;
17
+ try {
18
+ manifest = JSON.parse(await readFile(path.join(input.cwd, "package.json"), "utf8"));
19
+ }
20
+ catch {
21
+ return undefined;
22
+ }
23
+ const script = manifest.scripts?.[npm[1] ?? "test"] ?? "";
24
+ if (!/^vitest\s+run(?:\s|$)/.test(script) || /[;&|<>`\n\r$]|--(?:reporter|outputFile)/.test(script))
25
+ return undefined;
26
+ if (npm[3]?.trim() && !/^--(?:\s|$)/.test(npm[3].trim()))
27
+ return undefined;
28
+ separator = npm[3]?.trim() ? "" : " --";
29
+ }
30
+ const quote = (value) => `'${value.replace(/'/g, `'"'"'`)}'`;
31
+ return `${input.command}${separator} --reporter=json --outputFile=${quote(input.reportPath)}`;
32
+ }
33
+ const resultSchema = z.object({
34
+ testResults: z.array(z.object({
35
+ name: z.string().min(1),
36
+ assertionResults: z.array(z.object({
37
+ title: z.string(), fullName: z.string().optional(),
38
+ ancestorTitles: z.array(z.string()).optional(), status: z.string(),
39
+ })),
40
+ })),
41
+ });
42
+ export function parseFrontendTestExecutionReport(raw, cwd) {
43
+ return resultSchema.parse(raw).testResults.flatMap((suite) => suite.assertionResults.map((test) => ({
44
+ file: path.relative(cwd, path.resolve(cwd, suite.name)).replace(/\\/g, "/"),
45
+ title: test.fullName ?? [...(test.ancestorTitles ?? []), test.title].join(" "),
46
+ status: test.status,
47
+ })));
48
+ }
@@ -152,6 +152,7 @@ export function extractFrontendTestTitles(content) {
152
152
  const tokens = lexFrontendTestSource(content);
153
153
  const titles = [];
154
154
  const configurationModifiers = new Set(["each", "runIf", "skipIf"]);
155
+ const nonExecutingModifiers = new Set(["skip", "todo", "fails"]);
155
156
  function closingParenIndex(openIndex) {
156
157
  let depth = 0;
157
158
  for (let cursor = openIndex; cursor < tokens.length; cursor += 1) {
@@ -177,12 +178,21 @@ export function extractFrontendTestTitles(content) {
177
178
  }
178
179
  let cursor = index + 1;
179
180
  let hasConfigurationModifier = false;
181
+ let hasNonExecutingModifier = false;
182
+ let conditionalModifier;
180
183
  while (tokens[cursor]?.type === "punctuation" &&
181
184
  tokens[cursor]?.value === "." &&
182
185
  tokens[cursor + 1]?.type === "identifier") {
183
186
  if (tokens[cursor + 1]?.type === "identifier" &&
184
187
  configurationModifiers.has(tokens[cursor + 1].value)) {
185
188
  hasConfigurationModifier = true;
189
+ if (["runIf", "skipIf"].includes(tokens[cursor + 1].value)) {
190
+ conditionalModifier = tokens[cursor + 1].value;
191
+ }
192
+ }
193
+ if (tokens[cursor + 1]?.type === "identifier" &&
194
+ nonExecutingModifiers.has(tokens[cursor + 1].value)) {
195
+ hasNonExecutingModifier = true;
186
196
  }
187
197
  cursor += 2;
188
198
  }
@@ -191,6 +201,13 @@ export function extractFrontendTestTitles(content) {
191
201
  continue;
192
202
  }
193
203
  if (hasConfigurationModifier) {
204
+ if (conditionalModifier && tokens[cursor + 1]?.type === "identifier" &&
205
+ tokens[cursor + 2]?.value === ")") {
206
+ const condition = tokens[cursor + 1].value;
207
+ if ((conditionalModifier === "runIf" && condition === "false") ||
208
+ (conditionalModifier === "skipIf" && condition === "true"))
209
+ hasNonExecutingModifier = true;
210
+ }
194
211
  const close = closingParenIndex(cursor);
195
212
  if (close === null ||
196
213
  tokens[close + 1]?.type !== "punctuation" ||
@@ -199,6 +216,11 @@ export function extractFrontendTestTitles(content) {
199
216
  }
200
217
  cursor = close + 1;
201
218
  }
219
+ if (hasNonExecutingModifier) {
220
+ // A skipped suite also skips every nested test title in its body.
221
+ index = closingParenIndex(cursor) ?? tokens.length;
222
+ continue;
223
+ }
202
224
  if (tokens[cursor + 1]?.type !== "string")
203
225
  continue;
204
226
  titles.push(tokens[cursor + 1].value);
@@ -302,6 +324,15 @@ export async function evaluateVerificationTargets(input) {
302
324
  target,
303
325
  });
304
326
  issues.push(...fileTrace.issues);
327
+ const executed = input.executedTests?.get(target.commandLabel);
328
+ if (target.mode !== "static" && executed !== undefined) {
329
+ // safePath permits "./" prefixes; reporter files are cwd-relative.
330
+ const targetFile = path.posix.normalize(target.file.replace(/\\/g, "/"));
331
+ const matching = executed.filter((test) => test.file === targetFile && traceKeyMatchesTitle(target.id, test.title));
332
+ if (matching.length === 0 || matching.some((test) => test.status !== "passed")) {
333
+ issues.push(`behavior target was not fully executed and passed: ${target.id} in ${target.file}`);
334
+ }
335
+ }
305
336
  const status = issues.length ? "failed" : "ok";
306
337
  if (issues.length)
307
338
  hardIssues.push(...issues.map((i) => `${target.id}: ${i}`));
@@ -473,6 +504,7 @@ export async function runFrontendVerificationTraceGate(input) {
473
504
  contract,
474
505
  labelOwners,
475
506
  workspaceRoot: input.workspaceRoot,
507
+ executedTests: input.evidence?.behavior.executedTests,
476
508
  });
477
509
  if (hardIssues.length) {
478
510
  throw new Error(`trace: ${hardIssues.join("; ")}`);
@@ -2559,8 +2559,8 @@ function resolveFrontendMockContextBlock(sources) {
2559
2559
  if (mode === "not-required") {
2560
2560
  parts.push("Generation-time evidence does not require Mock. The assessment must still use contract/scout evidence: select not-needed when Mock is intentionally skipped, or select a safe Mock strategy if project evidence supports one.");
2561
2561
  if (frontendMockStrategyMustBeNotNeeded(sources)) {
2562
- parts.push('Auto mode has no confirmed project Mock capability or no deterministic Mock verification command. The structured contract must set mockApi.strategy to "not-needed". Do not add Mock files or dependencies; keep the real request path as default and record any unproved backend behavior as Real Integration Gap.');
2563
- parts.push('HARD CONSTRAINT (frozen at generation time): this DAG allows only mockApi.strategy "not-needed"; the prewrite gate rejects any other strategy. If project governance (openspec / ai_workspace / decision records, e.g. a DEC rule requiring native) demands Mock-backed verification, that is a generation-time contract gap, not a plan-revision defect: declare frontendMock.verifyCommands (or policy: "required") in task.json and regenerate the DAG. Do not emit any mockApi.strategy outside the allowlist and do not add Mock files or dependencies within this run.');
2562
+ parts.push('Auto mode has no confirmed project Mock capability or no deterministic Mock verification command. The structured contract must set mockApi.strategy to "not-needed". Keep the real request path as the default, record any unproved backend behavior as Real Integration Gap, and do not add Mock files or dependencies within this run.');
2563
+ parts.push('HARD CONSTRAINT (frozen at generation time): this DAG allows only mockApi.strategy "not-needed"; the prewrite gate rejects any other strategy. If project governance (openspec / ai_workspace / decision records, e.g. a DEC rule requiring native) demands Mock-backed verification, that is a generation-time contract gap, not a plan-revision defect: declare frontendMock.verifyCommands (or policy: "required") in task.json and regenerate the DAG.');
2564
2564
  }
2565
2565
  }
2566
2566
  if (mode === "blocked") {
@@ -3226,9 +3226,7 @@ async function buildFrontendHybridDagFromTask(sources) {
3226
3226
  "Reference frozen commands ONLY by commandId (record_plan_verification_target entry.commandId). The runtime resolves mode and label; never invent a mode or type.",
3227
3227
  ...frontendVerifyDirectory.map((entry) => ` - ${entry.commandId} [${entry.mode}]: ${JSON.stringify(entry.label)}`),
3228
3228
  `- Static command source: ${staticVerifyEvidence.commandSource}`,
3229
- ...staticVerifyEvidence.commandLabels.map((command) => ` - ${JSON.stringify(command)}`),
3230
3229
  `- Behavior command source: ${behaviorVerifyEvidence.commandSource}`,
3231
- ...behaviorVerifyEvidence.commandLabels.map((command) => ` - ${JSON.stringify(command)}`),
3232
3230
  ].join("\n");
3233
3231
  const advisories = [];
3234
3232
  if (!hasDeclaredFrontendVerification &&
@@ -3394,7 +3392,10 @@ async function buildFrontendHybridDagFromTask(sources) {
3394
3392
  depends_on: ["frontend-contract-pi", "frontend-scout-pi"],
3395
3393
  role: "planner",
3396
3394
  executor: "pi",
3397
- complexity: "MED",
3395
+ // Small topology has already proven a concentrated, no-remote scope;
3396
+ // keep its bounded plan on the LOW model tier. Standard/High retain
3397
+ // MED for broader contract-to-surface decisions.
3398
+ complexity: frontendTaskShape.shape === "small" ? "LOW" : "MED",
3398
3399
  writePolicy: "read-only",
3399
3400
  retryPolicy: FRONTEND_PLAN_LADDER_RETRY_POLICY,
3400
3401
  allowedPaths: readOnlyPaths,
@@ -3503,18 +3504,16 @@ async function buildFrontendHybridDagFromTask(sources) {
3503
3504
  "Your authoritative terminal verdict is exactly one committed typed tool call: approve_design or request_design_changes. Call exactly one of them; after calling one, do not call the other.",
3504
3505
  "request_design_changes must carry a typed issueCategory, at least one evidenceRef, and non-empty findings.",
3505
3506
  "Your verdict is consumed as deterministic data input by frontend-writer-admission-shell. approve_design permits admission; request_design_changes blocks writer admission until a recovery plan incorporates every Critical/Important finding.",
3506
- "Request design changes when the Mock strategy is MOCK_STRATEGY: blocked, missing, unsupported by repository evidence, inconsistent with the API contract, outside authorized paths/dependencies, unable to prove production-default-off behavior with the fixed production/default-real-path static check, or missing deterministic behavior verification for a declared behavior target or selected Mock strategy. Mock strategies require Mock-backed evidence. A static-only contract is allowed only when every verification target is static and maps to a declared static entrypoint. not-needed otherwise requires applicable real/no-remote behavior evidence unless auto mode explicitly skipped Mock because no project Mock capability exists; in that case the plan must preserve the real request path and record the Real Integration Gap.",
3507
+ "Request design changes when the Mock strategy is blocked, missing, unsupported by repository evidence, inconsistent with the API contract, outside authorized paths/dependencies, unable to prove production-default-off behavior with the fixed production/default-real-path static check, or missing deterministic behavior verification for a declared behavior target or selected Mock strategy. Mock strategies require Mock-backed evidence; a static-only contract is allowed only when every verification target is static and maps to a declared static entrypoint; not-needed requires applicable real/no-remote behavior evidence unless auto mode explicitly skipped Mock because no project Mock capability exists, in which case the plan must preserve the real request path and record the Real Integration Gap.",
3507
3508
  "Also request design changes for missing applicable UI states, unsupported dependency additions, design-system drift without reason, weak interaction coverage, broad scope, inline fake data, schema drift, or missing deterministic verification commands.",
3508
- "Component selection conformance is a hard blocking condition: request_design_changes when the frontend spec (component/theme/rule.components bucket) already defines a component for a purpose but the plan selects another or self-invents one without a declared deviation; when uiComponentChoices is missing/empty for UI-visible work while the frozen component/theme bucket is non-empty; when a decision=specified specReference.path is missing a ledger OpenSpec reference or successful read event; or when a decision=new component lacks a traceable task-source/PRD specReference. A PRD reference for decision=new is not an OpenSpec citation and must not be rejected merely for lacking an OpenSpec read event.",
3509
- "For uiComponentChoices, purpose is the stable coverage key and must match interaction.name or uiState.name. Responsibility is expressed by the matched expectedBehavior plus rationale; you must not reject it merely for matching an interaction or component identifier.",
3510
- "You must NOT make authoritative assertions about the execution result of frozen verification commands (typecheck/test/build/lint/etc.). Predicting that a command will necessarily pass or fail, or declaring an acceptance criterion unreachable on that basis, is out of your authority: command results are deterministically established by frontend-verify-shell. Any concern about verification feasibility must be recorded only as a non-blocking verification concern in findings (severity must not be Critical, and it must never be the sole fatal basis for request_design_changes). Only semantic design defects (component selection, state flow, interaction contract, or conflicts with the specification) may be Critical. A pure command-will-fail prediction must not be classified as contract-requirement-gap.",
3509
+ "Component selection conformance is a hard blocking condition: request_design_changes when the frontend spec (component/theme/rule.components bucket) already defines a component for a purpose but the plan selects another or self-invents one without a declared deviation; when uiComponentChoices is missing/empty for UI-visible work while the frozen component/theme bucket is non-empty; when a decision=specified specReference.path is missing a ledger OpenSpec reference or successful read event; or when a decision=new component lacks a traceable task-source/PRD specReference. A PRD reference for decision=new is not an OpenSpec citation and must not be rejected merely for lacking an OpenSpec read event. For uiComponentChoices, purpose is the stable coverage key and must match interaction.name or uiState.name; responsibility is expressed by the matched expectedBehavior plus rationale, and you must not reject it merely for matching an interaction or component identifier.",
3510
+ "You must NOT make authoritative assertions about the execution result of frozen verification commands: command results are deterministically established by frontend-verify-shell. Record a verification-feasibility concern only as a non-blocking finding (severity must not be Critical, and it must never be the sole fatal basis for request_design_changes). Only semantic design defects (component selection, state flow, interaction contract, or conflicts with the specification) may be Critical; a pure command-will-fail prediction must not be classified as contract-requirement-gap.",
3511
3511
  "Read-only: do not modify repository files.",
3512
3512
  "LARGE-FILE AUDIT (avoid full reads): style/theme audit files can be large (e.g. styles.css is often hundreds of KB). Prefer grep to locate the exact rules/variables you must verify (e.g. grep the oc- class, is-* modifier, or --oc- theme variables with their line numbers), then read only the narrow line range when surrounding context is needed. Do not read a large style/test file in full — a single full read can exhaust the read budget and fail the attempt.",
3513
3513
  "Canonical contract reading: frontend-design-policy-shell prints absolute paths for Contract, Contract index, and the non-blocking Capacity diagnostic. Read the capacity diagnostic first. When it recommends full-contract, read the exact Contract path. When it recommends indexed-sections, read the Contract index and its hash-bound section files instead of opening the full contract. Never resolve a bare contracts/... path against the repository root or hunt for substitutes. Implementation target files inside the writeSet are created later by the implement node: do not read them and do not treat their absence as a design defect.",
3514
3514
  fixedVerificationContext,
3515
3515
  sourceContexts.designReview,
3516
3516
  scopedOpenspecContext,
3517
- frontendContractFieldSummary,
3518
3517
  mockContextBlock,
3519
3518
  ].join("\n\n"),
3520
3519
  },