tickmarkr 2.1.6 → 2.1.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -789,6 +789,20 @@ acceptance is required on every task (a nested list of observable outcomes).
789
789
  hard value anywhere in the domain, the criterion asserts a universal that may be FALSE ABOUT THE
790
790
  WORLD — bound it or say where it stops holding, rather than demanding a value that does not exist.
791
791
 
792
+ PICK THE CRITERION FORM FROM WHO COULD BE WRONG:
793
+ - When the WORKER could be wrong because it can choose the value, use "test:" and pin the exact
794
+ literal it could otherwise choose; an example selected by its implementer proves only itself.
795
+ - When the AUTHOR could be wrong by omitting a member from a list, quantify universally over the
796
+ authoritative closed set; a hand-written enumeration can repeat the same omission as the code.
797
+ - When the REVIEWER could be wrong about a prose artefact, use "judge:" to replay a recorded incident
798
+ against the changed prose; a keyword check proves vocabulary, not that the artefact prevents a repeat.
799
+
800
+ PRE-SCOPE BY TEXT, ENUMERATE BLOCKERS BY EXECUTION:
801
+ - Before assigning files[], sweep text across the repository tree for names, callers, tests and prose.
802
+ A text sweep produces a candidate list; only running the change enumerates the real blocker set.
803
+ Keep the candidates for scope, then execute the production path and full gates before declaring the
804
+ set closed — this milestone paid a halted run to learn that the two populations are not identical.
805
+
792
806
  WHICH SIDE OF A RUN INHERITS ENVIRONMENT — AND IT DEPENDS ON THE DRIVER (OBS-542):
793
807
  - Gate commands and "command:"/"test:" oracles INHERIT THE DAEMON'S ENVIRONMENT. They are children of
794
808
  the daemon, so launching it as \`bash -c 'set -a; . .env.test; set +a; exec tickmarkr run'\` reaches
@@ -0,0 +1,35 @@
1
+ import type { Task } from "../graph/schema.js";
2
+ export type OwnershipCorroboration = {
3
+ kind: "direct-import";
4
+ source: string;
5
+ } | {
6
+ kind: "command-entry-spawn";
7
+ source: string;
8
+ entry: "src/cli/index.ts";
9
+ };
10
+ export type OwnershipFinding = {
11
+ code: "unowned-test";
12
+ test: string;
13
+ taskIds: string[];
14
+ corroboration?: OwnershipCorroboration;
15
+ detail: string;
16
+ } | {
17
+ code: "test-path-outside-allowlist";
18
+ taskId: string;
19
+ test: string;
20
+ path: string;
21
+ detail: string;
22
+ } | {
23
+ code: "unordered-context-write";
24
+ taskId: string;
25
+ ownerTaskId: string;
26
+ path: string;
27
+ detail: string;
28
+ };
29
+ /**
30
+ * Cross-task ownership evidence. Findings are data: this checker never throws or changes the graph;
31
+ * the compile seam promotes only a corroborated unowned-test finding and reports every other shape.
32
+ */
33
+ export declare function ownershipFindings(tasks: readonly Task[], repoRoot: string): OwnershipFinding[];
34
+ export declare function renderOwnershipFinding(finding: OwnershipFinding): string;
35
+ export declare function blocksCompile(finding: OwnershipFinding): boolean;
@@ -0,0 +1,213 @@
1
+ import { readFileSync, readdirSync } from "node:fs";
2
+ import { basename, extname, join, posix } from "node:path";
3
+ import { filesGlob } from "../graph/files-glob.js";
4
+ import { collateralHits } from "./collateral.js";
5
+ const normalize = (path) => path.replace(/^\.\//, "").split("\\").join("/");
6
+ function testSources(repoRoot) {
7
+ const root = join(repoRoot, "tests");
8
+ try {
9
+ return readdirSync(root, { recursive: true, encoding: "utf8" })
10
+ .filter((path) => path.endsWith(".test.ts"))
11
+ .sort()
12
+ .flatMap((path) => {
13
+ try {
14
+ return [{ path: `tests/${normalize(path)}`, text: readFileSync(join(root, path), "utf8") }];
15
+ }
16
+ catch {
17
+ return [];
18
+ }
19
+ });
20
+ }
21
+ catch {
22
+ return [];
23
+ }
24
+ }
25
+ function namedSources(test, tasks) {
26
+ const stem = basename(test).replace(/\.test\.ts$/, "");
27
+ const matches = new Map();
28
+ for (const task of tasks) {
29
+ for (const entry of task.files.map(normalize).filter((path) => path.startsWith("src/") && !/[*?{[]/.test(path))) {
30
+ const source = basename(entry, extname(entry));
31
+ if (stem === source || stem.startsWith(`${source}-`)) {
32
+ matches.set(`${task.id}:${entry}`, { taskId: task.id, source: entry });
33
+ }
34
+ }
35
+ }
36
+ return [...matches.values()];
37
+ }
38
+ const moduleKey = (path) => normalize(path).replace(/\.(?:[cm]?[jt]sx?)$/, "");
39
+ function directImportSpecifiers(text) {
40
+ // Comments cannot create an edge. Keep strings intact because they are the import target.
41
+ const source = text.replace(/\/\*[\s\S]*?\*\//g, "").replace(/^\s*\/\/.*$/gm, "");
42
+ const specifiers = new Set();
43
+ for (const match of source.matchAll(/\bimport\s+(?:type\s+)?(?:[\w$*{},\s]+?\s+from\s+)?["']([^"']+)["']/g)) {
44
+ specifiers.add(match[1]);
45
+ }
46
+ for (const match of source.matchAll(/\bimport\s*\(\s*["']([^"']+)["']\s*\)/g)) {
47
+ specifiers.add(match[1]);
48
+ }
49
+ return [...specifiers];
50
+ }
51
+ function directlyImports(test, source) {
52
+ // DIRECT is load-bearing: do not walk through imported helpers. src/run/journal.ts alone has 84
53
+ // test importers in the measured tree, so transitive closure would recreate the raw alarm flood.
54
+ const target = moduleKey(source);
55
+ return directImportSpecifiers(test.text).some((specifier) => {
56
+ const imported = specifier.startsWith(".")
57
+ ? posix.normalize(posix.join(posix.dirname(test.path), specifier))
58
+ : specifier.startsWith("src/") ? specifier : "";
59
+ return imported !== "" && moduleKey(imported) === target;
60
+ });
61
+ }
62
+ function invokesChildProcessSpawn(text) {
63
+ const source = text.replace(/\/\*[\s\S]*?\*\//g, "").replace(/^\s*\/\/.*$/gm, "");
64
+ const bindings = new Set();
65
+ for (const match of source.matchAll(/\bimport\s*{([^}]*)}\s*from\s*["'](?:node:)?child_process["']/g)) {
66
+ for (const member of match[1].split(",")) {
67
+ const binding = member.trim().match(/^spawn(?:Sync)?(?:\s+as\s+([A-Za-z_$][\w$]*))?$/);
68
+ if (binding)
69
+ bindings.add(binding[1] ?? member.trim());
70
+ }
71
+ }
72
+ for (const match of source.matchAll(/\bimport\s*\*\s*as\s*([A-Za-z_$][\w$]*)\s*from\s*["'](?:node:)?child_process["']/g)) {
73
+ if (new RegExp(`\\b${match[1]}\\.spawn(?:Sync)?\\s*\\(`).test(source))
74
+ return true;
75
+ }
76
+ return [...bindings].some((binding) => new RegExp(`\\b${binding}\\s*\\(`).test(source));
77
+ }
78
+ function mentionsCommandEntry(text) {
79
+ return /(?:^|\/)src\/cli\/index\.(?:ts|js)\b/.test(text)
80
+ || /["'`]src["'`]\s*,\s*["'`]cli["'`]\s*,\s*["'`]index\.(?:ts|js)["'`]/.test(text);
81
+ }
82
+ function corroboration(test, matches) {
83
+ // A .test.ts-shaped collateral fixture is not by itself a dedicated test. Requiring a runner leaf
84
+ // keeps import-only scan fixtures advisory while every executable subject in the measured union stays.
85
+ const executable = test.text.replace(/\/\*[\s\S]*?\*\//g, "").replace(/^\s*\/\/.*$/gm, "");
86
+ if (!/\b(?:test|it)(?:\.(?:concurrent|each|fails|only|skip|todo))*\s*\(/.test(executable))
87
+ return undefined;
88
+ for (const match of matches) {
89
+ if (directlyImports(test, match.source))
90
+ return { kind: "direct-import", source: match.source };
91
+ }
92
+ if (invokesChildProcessSpawn(test.text) && mentionsCommandEntry(test.text)) {
93
+ const command = matches.find(({ source }) => /^src\/cli\/commands\/[^/]+\.(?:[cm]?[jt]sx?)$/.test(source));
94
+ if (command)
95
+ return { kind: "command-entry-spawn", source: command.source, entry: "src/cli/index.ts" };
96
+ }
97
+ return undefined;
98
+ }
99
+ function repositoryPaths(text) {
100
+ const paths = new Set();
101
+ for (const match of text.matchAll(/["'`]((?:src|tests|fixtures|scripts|skills|docs|\.claude)\/[^"'`\s]+)["'`]/g)) {
102
+ paths.add(normalize(match[1]));
103
+ }
104
+ return [...paths].sort();
105
+ }
106
+ function dependencyOrdered(a, b, byId) {
107
+ const reaches = (from, target) => {
108
+ const seen = new Set();
109
+ const pending = [...from.deps];
110
+ while (pending.length > 0) {
111
+ const id = pending.pop();
112
+ if (id === target)
113
+ return true;
114
+ if (seen.has(id))
115
+ continue;
116
+ seen.add(id);
117
+ pending.push(...(byId.get(id)?.deps ?? []));
118
+ }
119
+ return false;
120
+ };
121
+ return reaches(a, b.id) || reaches(b, a.id);
122
+ }
123
+ /**
124
+ * Cross-task ownership evidence. Findings are data: this checker never throws or changes the graph;
125
+ * the compile seam promotes only a corroborated unowned-test finding and reports every other shape.
126
+ */
127
+ export function ownershipFindings(tasks, repoRoot) {
128
+ const sources = testSources(repoRoot);
129
+ const sourceByPath = new Map(sources.map((source) => [source.path, source]));
130
+ const byId = new Map(tasks.map((task) => [task.id, task]));
131
+ const indexed = tasks.map((task) => {
132
+ const files = task.files.map(normalize);
133
+ const context = task.context.map(normalize);
134
+ return {
135
+ task,
136
+ owns: files.length === 0 ? () => false : filesGlob(files),
137
+ allows: files.length === 0 ? () => true : filesGlob([...files, ...context]),
138
+ };
139
+ });
140
+ const owners = (path) => indexed.filter(({ owns }) => owns(path));
141
+ const predictedBy = new Map();
142
+ for (const [taskId, hits] of collateralHits(tasks, repoRoot)) {
143
+ for (const hit of hits) {
144
+ if (!sourceByPath.has(hit))
145
+ continue;
146
+ const ids = predictedBy.get(hit) ?? new Set();
147
+ ids.add(taskId);
148
+ predictedBy.set(hit, ids);
149
+ }
150
+ }
151
+ for (const source of sources) {
152
+ const ids = predictedBy.get(source.path) ?? new Set();
153
+ for (const { taskId } of namedSources(source.path, tasks))
154
+ ids.add(taskId);
155
+ if (ids.size > 0)
156
+ predictedBy.set(source.path, ids);
157
+ }
158
+ const findings = [];
159
+ for (const [test, taskIds] of predictedBy) {
160
+ if (owners(test).length === 0) {
161
+ const ids = [...taskIds].sort();
162
+ const source = sourceByPath.get(test);
163
+ const evidence = corroboration(source, namedSources(test, tasks));
164
+ findings.push({
165
+ code: "unowned-test",
166
+ test,
167
+ taskIds: ids,
168
+ ...(evidence ? { corroboration: evidence } : {}),
169
+ detail: `${test} is a dedicated test of source owned by ${ids.join(", ")} but no task owns the test`
170
+ + (evidence?.kind === "direct-import" ? `; it imports ${evidence.source} directly`
171
+ : evidence?.kind === "command-entry-spawn"
172
+ ? `; it spawns ${evidence.entry} to exercise ${evidence.source}` : ""),
173
+ });
174
+ }
175
+ }
176
+ for (const source of sources) {
177
+ for (const owner of owners(source.path)) {
178
+ for (const path of repositoryPaths(source.text)) {
179
+ if (!owner.allows(path)) {
180
+ findings.push({
181
+ code: "test-path-outside-allowlist",
182
+ taskId: owner.task.id,
183
+ test: source.path,
184
+ path,
185
+ detail: `${source.path} owned by ${owner.task.id} references ${path} outside that task's files[] and context[]`,
186
+ });
187
+ }
188
+ }
189
+ }
190
+ }
191
+ for (const reader of indexed) {
192
+ for (const entry of reader.task.context.map(normalize)) {
193
+ for (const owner of owners(entry)) {
194
+ if (owner.task.id === reader.task.id || dependencyOrdered(reader.task, owner.task, byId))
195
+ continue;
196
+ findings.push({
197
+ code: "unordered-context-write",
198
+ taskId: reader.task.id,
199
+ ownerTaskId: owner.task.id,
200
+ path: entry,
201
+ detail: `${reader.task.id} names ${entry} as context while ${owner.task.id} owns it for writing, with no dependency order between them`,
202
+ });
203
+ }
204
+ }
205
+ }
206
+ return findings.sort((a, b) => `${a.code}:${"test" in a ? a.test : a.path}:${"taskId" in a ? a.taskId : ""}`.localeCompare(`${b.code}:${"test" in b ? b.test : b.path}:${"taskId" in b ? b.taskId : ""}`));
207
+ }
208
+ export function renderOwnershipFinding(finding) {
209
+ return `tickmarkr: ownership-lint[${finding.code}]: ${finding.detail}`;
210
+ }
211
+ export function blocksCompile(finding) {
212
+ return finding.code === "unowned-test" && finding.corroboration !== undefined;
213
+ }
@@ -125,6 +125,7 @@ export declare class HerdrDriver implements ExecutorDriver {
125
125
  narrator(cwd: string, command: string, runId?: string): Promise<Slot>;
126
126
  reconcile(desired: Set<string>, runId: string, opts?: {
127
127
  spareLiveLlm?: boolean;
128
+ endedRunIds?: Set<string>;
128
129
  }): Promise<void>;
129
130
  worktree(repo: string, branch: string, baseRef: string): Promise<string>;
130
131
  }
@@ -1188,12 +1188,13 @@ export class HerdrDriver {
1188
1188
  return s;
1189
1189
  });
1190
1190
  }
1191
- // OBS-17 T2 / v1.22b T1: close every tickmarkr-owned pane that should not exist (superseded attempts,
1192
- // killed-daemon orphans, leftovers from OLDER runs) — in this run's workspace OR misplaced in any
1193
- // other one — then reap the tabs those closes emptied. Ownership is decided ONLY by parseOwnedName
1194
- // (drivers/types.ts panesToClose) — foreign names never become candidates, in any workspace; a pane
1195
- // this same run legitimately holds elsewhere is left alone (only run age marks a misplaced pane
1196
- // garbage). spareLiveLlm: same-run judge/review/consult panes have no journal row while live (their
1191
+ // OBS-17 T2 / v1.22b T1: close THIS RUN'S OWN tickmarkr-owned panes that should not exist
1192
+ // (superseded attempts, killed-daemon orphans of this run), in this run's workspace, then reap the
1193
+ // tabs those closes emptied. Ownership is decided ONLY by parseOwnedName (drivers/types.ts
1194
+ // panesToClose) — foreign names never become candidates. OBS-769/OBS-772/OBS-777: the sweep does
1195
+ // NOT reach other workspaces and reaches another runId only when the daemon's repository-scoped
1196
+ // snapshot proves that run ended. Run age is not consulted and never was. spareLiveLlm: same-run
1197
+ // judge/review/consult panes have no journal row while live (their
1197
1198
  // events land after the verdict is read), so mid-run sweeps spare them; boundary sweeps (start/
1198
1199
  // resume/end) run with nothing in flight and take them too. Cosmetic by contract: every failure —
1199
1200
  // herdr gone, pane vanished mid-sweep, unparseable listing — is swallowed; this method never throws.
@@ -56,6 +56,7 @@ export interface FleetAgent {
56
56
  }
57
57
  export declare function panesToClose(agents: FleetAgent[], desired: Set<string>, ws: string, runId: string, opts?: {
58
58
  spareLiveLlm?: boolean;
59
+ endedRunIds?: Set<string>;
59
60
  }): {
60
61
  paneId: string;
61
62
  tabId?: string;
@@ -81,5 +82,6 @@ export interface ExecutorDriver {
81
82
  narrator?: (cwd: string, command: string, runId?: string) => Promise<Slot>;
82
83
  reconcile?: (desired: Set<string>, runId: string, opts?: {
83
84
  spareLiveLlm?: boolean;
85
+ endedRunIds?: Set<string>;
84
86
  }) => Promise<void>;
85
87
  }
@@ -21,12 +21,53 @@ export function isForeignName(name) {
21
21
  return parseOwnedName(name) === null;
22
22
  }
23
23
  // v1.22b T1: workspace-aware fold over a fleet snapshot — decides which owned task panes are garbage
24
- // right now. In-workspace: the existing desired-set/spareLiveLlm sweep (OBS-17 T2). Out-of-workspace:
25
- // an owned pane from a DIFFERENT run is a misplaced leftover (bug, foreign actor, pre-VIS-10 relic)
26
- // and closes regardless of `desired`; an owned pane from THIS run elsewhere is left alone — a live
27
- // run can legitimately hold panes across workspaces, so only run age marks a misplaced pane garbage.
24
+ // right now: the desired-set/spareLiveLlm sweep (OBS-17 T2), scoped to THIS RUN'S OWN panes (by runId,
25
+ // OBS-772) in THIS RUN'S OWN WORKSPACE (OBS-769). Both conditions, and neither alone is the rule.
28
26
  // Watch panes are operator-owned after run end and are reclaimed by the next run; foreign names
29
- // (parseOwnedName fails) are never candidates, in any workspace.
27
+ // (parseOwnedName fails) are never candidates.
28
+ //
29
+ // OBS-769 — WHY THE SWEEP STOPS AT THE WORKSPACE BOUNDARY. It used to close an owned pane carrying
30
+ // any OTHER runId in any other workspace, unconditionally, as a "misplaced leftover". Two tickmarkr
31
+ // runs in two repositories are lawful (the lock forbids two runs in ONE repository, not on one
32
+ // machine) and herdr gives each its own workspace — so that branch made every pair of concurrent
33
+ // runs kill each other's LIVE workers. Measured 2026-08-28: the run in w0 closed run ...2958's
34
+ // panes at 23:42:40.351/.392, and 53s later ...2958's own task-human sweep closed w0's live codex
35
+ // worker at 23:43:34.096. ...2958 ended 0/8. The death detector cannot see it: closing the pane
36
+ // makes paneAbsent, processTree, confirmedProcessTree and worktreeDelta true by ONE cause, and a
37
+ // closed pane can never accrue the CPU that the `cpu-accruing` hold reads.
38
+ // The comment this replaces claimed "only run age marks a misplaced pane garbage" — there was no age
39
+ // check in the code, and age is the wrong predicate anyway: w0's run STARTED EARLIER than ...2958,
40
+ // so an age rule would have licensed exactly the kill that landed. Run age says nothing about
41
+ // liveness, and a sweeping daemon cannot read another repository's run state. The workspace is the
42
+ // only ownership boundary available without cross-repo I/O, so it is the one enforced.
43
+ // Cost, named: an orphan pane from a dead run stranded in a workspace no later run opens is now left
44
+ // for the operator. That is cosmetic (`reconcile` is cosmetic by contract — "visibility is never a
45
+ // gate"), and a cosmetic cleanup must never be able to kill a live worker.
46
+ // OBS-772 — WHY THE runId LINE EXISTS, AND WHY THE WORKSPACE LINE ALONE WAS NOT THE FIX. The first
47
+ // repair was workspace-scoped only, and its own comment dismissed the residue — "two runs sharing one
48
+ // workspace would still sweep each other" — as unreachable, on the reasoning that one workspace per run
49
+ // is herdr's placement. That reasoned from ONE driver to the whole product. OrcaDriver has no workspace
50
+ // dimension at all: orca.ts passes a single ORCA_SPACE as the workspaceId for EVERY checkout and as
51
+ // `ws`, so `workspaceId !== ws` is never true there and every foreign pane fell straight through. Orca
52
+ // users had zero protection while the defect read as fixed. The runId line is the real rule and it is
53
+ // driver-agnostic: reconcile exists to clean up THIS RUN's panes, and a leftover from a dead run is
54
+ // exactly what cannot be told from a live run's pane without liveness data this process does not have.
55
+ // Both lines are kept — the workspace line preserves the pre-existing sparing of this run's own panes
56
+ // in another workspace, which the runId line alone would not.
57
+ // ⚠ WHAT THE runId LINE COST BEFORE OBS-777 — SUSPENDED, NOT NARROWED, and the price was larger than
58
+ // it read. Sparing every other runId suspended OBS-17's FOUNDING use case: "a killed daemon can't
59
+ // close its slots". This sweep was built to reclaim exactly those orphans, but could not reclaim ANY
60
+ // previous run's panes. Three separate pins asserted the old behaviour (reconcile.test.ts,
61
+ // orca-placement.test.ts,
62
+ // reconcile-live.test.ts); all three were changed deliberately, and the third is why this paragraph
63
+ // exists rather than a shorter one — two flipped pins is a trade, three is a pattern.
64
+ // OBS-777 RESTORES that reclamation: the CALLER passes `opts.endedRunIds`, a Set the daemon computes
65
+ // ONCE at run start from this repository's own `run-end` journals and dead lock holders. This fold
66
+ // stays pure — it gains one optional field, not a repo root — a foreign repository's runId is never
67
+ // resolvable and so stays spared by construction, and no driver learns about workspaces.
68
+ // ponytail: two conditions, no geometry reasoning, nothing driver-specific. `reconcile` is cosmetic by
69
+ // contract, and a cosmetic cleanup must never be able to kill a live worker — which is why the
70
+ // ended-run authority is the only safe way to restore the sweep without reviving the cross-run kill.
30
71
  export function panesToClose(agents, desired, ws, runId, opts) {
31
72
  const out = [];
32
73
  for (const a of agents) {
@@ -35,15 +76,14 @@ export function panesToClose(agents, desired, ws, runId, opts) {
35
76
  const owned = parseOwnedName(a.name);
36
77
  if (!owned || owned.role === "watch")
37
78
  continue;
38
- if (a.workspaceId === ws) {
39
- if (desired.has(a.name))
40
- continue;
41
- if (opts?.spareLiveLlm && owned.runId === runId && (owned.role === "judge" || owned.role === "review" || owned.role === "consult"))
42
- continue;
43
- }
44
- else if (owned.runId === runId) {
45
- continue; // this run's own pane in another workspace — never touched
46
- }
79
+ if (owned.runId !== runId && !opts?.endedRunIds?.has(owned.runId))
80
+ continue;
81
+ if (a.workspaceId !== ws)
82
+ continue; // OBS-769: another workspace is another run's business
83
+ if (desired.has(a.name))
84
+ continue;
85
+ if (opts?.spareLiveLlm && owned.runId === runId && (owned.role === "judge" || owned.role === "review" || owned.role === "consult"))
86
+ continue;
47
87
  out.push({ paneId: a.paneId, tabId: a.tabId });
48
88
  }
49
89
  return out;
@@ -28,6 +28,19 @@ export interface VitestListedTest {
28
28
  file: string;
29
29
  projectName?: string;
30
30
  }
31
+ export type VitestListResult = {
32
+ status: "listed";
33
+ tests: VitestListedTest[];
34
+ } | {
35
+ status: "failed";
36
+ error: string;
37
+ };
38
+ export declare function listVitestTests(cwd: string): Promise<VitestListResult>;
39
+ export interface NamedTestAudit {
40
+ criterion: string;
41
+ matches: VitestListedTest[];
42
+ }
43
+ export declare function auditNamedTestOracles(items: readonly AcceptanceItem[], listedTests: readonly VitestListedTest[]): NamedTestAudit[];
31
44
  export type AcceptanceCorpusAuditResult = {
32
45
  specPath: string;
33
46
  status: "parse-failed";
@@ -98,6 +98,47 @@ export function testFiltered(testCmd, name) {
98
98
  const fwd = wrapped ? "-- " : "";
99
99
  return `${testCmd} ${fwd}-t ${shq(pattern)}`;
100
100
  }
101
+ const VitestListedTestsSchema = z.array(z.object({
102
+ name: z.string(),
103
+ file: z.string(),
104
+ projectName: z.string().optional(),
105
+ }));
106
+ export async function listVitestTests(cwd) {
107
+ const result = await sh(`${shq(join(cwd, "node_modules/.bin/vitest"))} list --json`, cwd);
108
+ if (result.code !== 0) {
109
+ return { status: "failed", error: (result.stderr || result.stdout || `exit ${result.code}`).trim() };
110
+ }
111
+ try {
112
+ const start = result.stdout.indexOf("[");
113
+ if (start < 0)
114
+ return { status: "failed", error: "runner emitted no JSON test listing" };
115
+ const parsed = VitestListedTestsSchema.safeParse(JSON.parse(result.stdout.slice(start)));
116
+ return parsed.success
117
+ ? { status: "listed", tests: parsed.data }
118
+ : { status: "failed", error: z.prettifyError(parsed.error) };
119
+ }
120
+ catch (error) {
121
+ return { status: "failed", error: error instanceof Error ? error.message : String(error) };
122
+ }
123
+ }
124
+ export function auditNamedTestOracles(items, listedTests) {
125
+ const runnerNames = listedTests.map((listed) => ({
126
+ listed,
127
+ fullName: listed.name.split(" > ").join(" "),
128
+ }));
129
+ return items.flatMap((item) => {
130
+ if (typeof item !== "object" || item.oracle !== "test")
131
+ return [];
132
+ return [{
133
+ criterion: item.test,
134
+ // OBS-511: mirror the gate's leaf-anchored suffix rule — this denominator must count
135
+ // exactly the tests the shipped -t filter would select.
136
+ matches: runnerNames
137
+ .filter(({ fullName }) => fullName === item.test || fullName.endsWith(` ${item.test}`))
138
+ .map(({ listed }) => listed),
139
+ }];
140
+ });
141
+ }
101
142
  function corpusSpecPaths(root) {
102
143
  const paths = [];
103
144
  const visit = (dir) => {
@@ -116,32 +157,18 @@ function corpusSpecPaths(root) {
116
157
  // listing. Every discovered path contributes either all parser-produced acceptance items or one named
117
158
  // parse failure; exceptions are evidence, never permission to shrink the corpus silently.
118
159
  export function auditAcceptanceCorpus(corpusRoot, listedTests) {
119
- const runnerNames = listedTests.map((listed) => ({
120
- listed,
121
- fullName: listed.name.split(" > ").join(" "),
122
- }));
123
160
  const results = [];
124
161
  for (const specPath of corpusSpecPaths(corpusRoot)) {
125
162
  try {
126
163
  const graph = compileNative(specPath);
127
164
  for (const task of graph.tasks) {
128
165
  for (const item of task.acceptance) {
166
+ const namedTest = auditNamedTestOracles([item], listedTests)[0];
129
167
  results.push({
130
168
  specPath,
131
169
  status: "parsed",
132
170
  item,
133
- ...(typeof item === "object" && item.oracle === "test"
134
- ? {
135
- namedTest: {
136
- criterion: item.test,
137
- // OBS-511: mirror the gate's leaf-anchored suffix rule — the audit's denominator
138
- // must count exactly the tests the -t filter would select, or doctor and gate disagree.
139
- matches: runnerNames
140
- .filter(({ fullName }) => fullName === item.test || fullName.endsWith(` ${item.test}`))
141
- .map(({ listed }) => listed),
142
- },
143
- }
144
- : {}),
171
+ ...(namedTest ? { namedTest } : {}),
145
172
  });
146
173
  }
147
174
  }
@@ -117,7 +117,18 @@ const VOCAB_RE = /\b(?:error|fail(?:ed|ure|ing)?)\b/i;
117
117
  // shapes are emitted by the process that was asked to run the oracle; they are deliberately kept in
118
118
  // this runner-output classifier rather than applied to any judge-authored reason text. A real test
119
119
  // failure still dominates below because one regression-shaped line makes the whole output regression.
120
- const INFRA_RE = /\bE(?:AGAIN|MFILE|NFILE|NOMEM|NOSPC)\b|JavaScript heap out of memory|Cannot allocate memory|Resource temporarily unavailable|Token not found in system keyring|Process from config\.webServer was not able to start|\[birpc\] rpc is closed, cannot call\b/i;
120
+ // OBS-791: `[vitest-worker]: Timeout calling "<method>"` is the SAME birpc mechanism one layer up.
121
+ // Vitest's bundled birpc has a fixed 60s DEFAULT_TIMEOUT with no override (fixed upstream only in
122
+ // vitest 4.x), so a suite whose tests all pass can still die at teardown when the worker->host RPC
123
+ // window closes. `.github/workflows/release.yml` has forgiven this exact fingerprint since 2026-08-11
124
+ // while this classifier called it a regression — the product charged a worker for the failure the
125
+ // release gate was written to forgive. The METHOD NAME IS DELIBERATELY NOT PINNED: the timeout is a
126
+ // property of the RPC window, not of `onTaskUpdate`, and pinning one method would forgive a run and
127
+ // charge its sibling for the same infrastructure event. This stays a CLOSED signature — the
128
+ // `[vitest-worker]: ` prefix is vitest's own RPC layer and never user assertion text — and the
129
+ // per-line vetoes below are untouched, so one real failure anywhere still makes the whole output a
130
+ // regression.
131
+ const INFRA_RE = /\bE(?:AGAIN|MFILE|NFILE|NOMEM|NOSPC)\b|JavaScript heap out of memory|Cannot allocate memory|Resource temporarily unavailable|Token not found in system keyring|Process from config\.webServer was not able to start|\[birpc\] rpc is closed, cannot call\b|\[vitest-worker\]: Timeout calling\b/i;
121
132
  // Capture invalidation is deliberately narrower than the gate's infrastructure vocabulary above:
122
133
  // keyring/config-webServer startup failures remain gate concerns, while this policy is specifically
123
134
  // for evidence that the capture ran while the machine was resource-starved.