tickmarkr 2.1.7 → 2.1.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3,6 +3,7 @@ import type { WorkerAdapter } from "../../adapters/types.js";
3
3
  import { type KimiDoctorTurnResult } from "../../adapters/kimi.js";
4
4
  import { type CatalogReadResult } from "../../adapters/catalog-remote.js";
5
5
  import { type ShResult } from "../../run/git.js";
6
+ import { type VitestListResult } from "../../gates/acceptance.js";
6
7
  /** Where a newer `table_<date>.csv` is discovered — the deployed site builds filenames by
7
8
  * concatenation and publishes no index, so the release listing is the only enumerable surface. */
8
9
  export declare const LIVEBENCH_RELEASES_URL = "https://api.github.com/repos/LiveBench/livebench.github.io/contents/public";
@@ -20,6 +21,8 @@ export type DoctorOpts = {
20
21
  orcaStatusProbe?: (cwd: string, binary: string) => Promise<ShResult>;
21
22
  /** Test seam for shell-path discovery; absence remains a normal doctor row, never an exception. */
22
23
  resolveOrcaBinary?: (cwd: string) => string | undefined;
24
+ /** Test seam for the runner-owned JSON listing used by the acceptance-oracle report row. */
25
+ listTests?: (cwd: string) => Promise<VitestListResult>;
23
26
  };
24
27
  type OrcaCapability = {
25
28
  verdict: "pass" | "fail";
@@ -8,7 +8,7 @@ import { allAdapters, binaryShadowWarnings, detectCandidateClis, flagDriftWarnin
8
8
  import { CLAUDE_ALIAS_IDENTITY_STAMPS, claudeCode, resolveClaudeAliasIdentity } from "../../adapters/claude-code.js";
9
9
  import { shq } from "../../adapters/types.js";
10
10
  import { BANNER, compactTokens, dim, fail, kvRow, legend, ok, rule, statusRow, title } from "../../brand.js";
11
- import { tickmarkrDir, stateDirName } from "../../graph/graph.js";
11
+ import { graphPath, loadGraph, tickmarkrDir, stateDirName } from "../../graph/graph.js";
12
12
  import { catalogModelAdvisory, catalogTierRanking, declaredModelWindow, hasWindowsConfig, modelLints, suggestOverlay, ttyVisual } from "../../adapters/model-lints.js";
13
13
  import { loadConfig, overlayPreferShapes } from "../../config/config.js";
14
14
  import { HerdrDriver } from "../../drivers/herdr.js";
@@ -17,6 +17,7 @@ import { kimi, probeKimiDoctorTurn } from "../../adapters/kimi.js";
17
17
  import { denyPreferCollisionLine, denyPreferCollisions, disallowedBy, excludedChannels, exclusionLine, preferRanks } from "../../route/preference.js";
18
18
  import { LIVEBENCH_TABLE_DATE, readCachedCatalog, refreshCatalogCommand } from "../../adapters/catalog-remote.js";
19
19
  import { sh } from "../../run/git.js";
20
+ import { auditNamedTestOracles, listVitestTests } from "../../gates/acceptance.js";
20
21
  /** Where a newer `table_<date>.csv` is discovered — the deployed site builds filenames by
21
22
  * concatenation and publishes no index, so the release listing is the only enumerable surface. */
22
23
  export const LIVEBENCH_RELEASES_URL = "https://api.github.com/repos/LiveBench/livebench.github.io/contents/public";
@@ -362,6 +363,27 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
362
363
  const healthy = h.installed && (a.id !== kimi.id || h.authed);
363
364
  return alignedStatusRow(healthy ? "pass" : "fail", a.id, state);
364
365
  });
366
+ if (existsSync(graphPath(cwd))) {
367
+ try {
368
+ const graph = loadGraph(cwd);
369
+ const items = graph.tasks.flatMap((task) => task.acceptance);
370
+ if (items.some((item) => typeof item === "object" && item.oracle === "test")) {
371
+ const listing = await (opts.listTests ?? listVitestTests)(cwd);
372
+ if (listing.status === "failed") {
373
+ rows.push(alignedStatusRow("fail", "acceptance-oracles", `runner listing failed — ${listing.error.split("\n")[0]}`));
374
+ }
375
+ else {
376
+ const audited = auditNamedTestOracles(items, listing.tests);
377
+ const resolved = audited.filter((row) => row.matches.length === 1).length;
378
+ const verdict = resolved === audited.length ? "pass" : "fail";
379
+ rows.push(alignedStatusRow(verdict, "acceptance-oracles", `${resolved}/${audited.length} resolved — each resolved oracle's shipped filter matches exactly one runner-listed test`));
380
+ }
381
+ }
382
+ }
383
+ catch (error) {
384
+ rows.push(alignedStatusRow("fail", "acceptance-oracles", `graph unreadable — ${error instanceof Error ? error.message : String(error)}`));
385
+ }
386
+ }
365
387
  if (catalog.warning) {
366
388
  rows.push(attentionRow(`model catalog cache unreadable — ${catalog.warning}; using vendored fallback (advisory — routing unchanged)`));
367
389
  }
@@ -1,2 +1,6 @@
1
- import type { WorkerAdapter } from "../../adapters/types.js";
2
- export declare function plan(argv: string[], cwd?: string, adapters?: WorkerAdapter[], harnessFrom?: string | undefined): Promise<string>;
1
+ import { type VitestListResult } from "../../gates/acceptance.js";
2
+ import { type WorkerAdapter } from "../../adapters/types.js";
3
+ export type PlanOpts = {
4
+ listTests?: (cwd: string) => Promise<VitestListResult>;
5
+ };
6
+ export declare function plan(argv: string[], cwd?: string, adapters?: WorkerAdapter[], harnessFrom?: string | undefined, opts?: PlanOpts): Promise<string>;
@@ -3,15 +3,20 @@ import { formatModelAuthLine, contextWindowLints, modelLints, preferEntryLints,
3
3
  import { GLYPHS, dim, rule, title, warn } from "../../brand.js";
4
4
  import { parseArgs } from "node:util";
5
5
  import { collateralLints, sourceScopeLints } from "../../compile/collateral.js";
6
+ import { classifyContextPath } from "../../compile/native.js";
6
7
  import { DEFAULT_CONFIG, overlayPreferShapes, ROUTING_MODES, TIER_RANK } from "../../config/config.js";
7
8
  import { loadGraph } from "../../graph/graph.js";
9
+ import { renderAcceptanceItem } from "../../graph/schema.js";
8
10
  import { resolveRunMode } from "../../run/daemon.js";
9
11
  import { excludedChannels, exclusionLine } from "../../route/preference.js";
10
12
  import { staffLedEvidence } from "../../route/profile.js";
11
13
  import { route, RoutingError } from "../../route/router.js";
14
+ import { auditNamedTestOracles, listVitestTests } from "../../gates/acceptance.js";
12
15
  import { modelId } from "../../gates/review.js";
13
16
  import { loadRoutingProfile } from "../../run/journal.js";
14
17
  import { harnessLine, resolveHarness } from "../harness.js";
18
+ import { shq } from "../../adapters/types.js";
19
+ import { shGit } from "../../run/git.js";
15
20
  // T4 (v1.50): TTY-only brand pass — the title helper frames the routing table, lint/unroutable
16
21
  // markers carry the attention glyph, section labels dim to chrome (the doctor/status system).
17
22
  // Gated on ttyVisual(): the non-TTY surface returns untouched (byte-pinned, machine-consumable).
@@ -35,12 +40,60 @@ const fleetCanCrossVendorReview = (channels) => {
35
40
  return true;
36
41
  return false;
37
42
  };
43
+ const NAMED_CRITERION_PATH = /(?:^|[\s("'`])((?:src|tests|fixtures|scripts)\/[A-Za-z0-9_@{}*?.,/+-]+\.(?:tsx|ts|jsx|json|js|mjs|cjs|md|txt))(?![A-Za-z0-9])/g;
44
+ async function taskInputFindings(tasks, cwd) {
45
+ const pathsByTask = tasks.map((task) => {
46
+ const paths = new Set(task.files.map((path) => path.replace(/^\.\//, "")));
47
+ for (const item of task.acceptance) {
48
+ const texts = typeof item === "string"
49
+ ? [item]
50
+ : [renderAcceptanceItem(item), ...Object.values(item).filter((value) => typeof value === "string")];
51
+ for (const text of texts) {
52
+ for (const match of text.matchAll(NAMED_CRITERION_PATH))
53
+ paths.add(match[1]);
54
+ }
55
+ }
56
+ return { task, paths: [...paths] };
57
+ });
58
+ if (pathsByTask.every(({ paths }) => paths.length === 0))
59
+ return [];
60
+ const [tree, diff] = await Promise.all([
61
+ shGit("git ls-tree --full-tree -r --name-only -z HEAD", cwd),
62
+ shGit("git diff --name-only -z HEAD --", cwd),
63
+ ]);
64
+ if (tree.code !== 0 || diff.code !== 0)
65
+ return [];
66
+ const tracked = new Set(tree.stdout.split("\0").filter(Boolean));
67
+ const changed = new Set(diff.stdout.split("\0").filter(Boolean));
68
+ const findings = [];
69
+ for (const { task, paths } of pathsByTask) {
70
+ for (const path of paths) {
71
+ const state = classifyContextPath(path, tracked, cwd);
72
+ if (state.kind === "untracked") {
73
+ findings.push({
74
+ taskId: task.id,
75
+ severity: "refuse",
76
+ detail: `task path ${JSON.stringify(path)} exists in the working tree but no commit holds it`,
77
+ });
78
+ }
79
+ else if (state.kind === "ok"
80
+ && (changed.has(path) || [...changed].some((candidate) => candidate.startsWith(`${path}/`)))) {
81
+ findings.push({
82
+ taskId: task.id,
83
+ severity: "warn",
84
+ detail: `tracked path ${JSON.stringify(path)} differs from HEAD — run git restore -- ${shq(path)} to discard the divergence, or commit it before dispatch`,
85
+ });
86
+ }
87
+ }
88
+ }
89
+ return findings;
90
+ }
38
91
  // v1.89 T4: harnessFrom is the resolver's INPUT — a caller (the byte-pinned goldens) fixes the location
39
92
  // and keeps this machine's absolute paths out of a fixture. The default is the INVOKED entrypoint,
40
93
  // `process.argv[1]`: the bin symlink a global install puts on PATH, which resolves to dist/cli/index.js.
41
94
  // It is NOT `import.meta.url` — that names dist/cli/commands/plan.js, an internal module of the harness
42
95
  // rather than the harness that was invoked, so the banner would identify the wrong file entirely.
43
- export async function plan(argv, cwd = process.cwd(), adapters = allAdapters(), harnessFrom = process.argv[1]) {
96
+ export async function plan(argv, cwd = process.cwd(), adapters = allAdapters(), harnessFrom = process.argv[1], opts = {}) {
44
97
  // ponytail: hardcoded 24h TTL — promote to config when an operator asks. mtime is the signal because
45
98
  // doctor.json has no probe timestamp and a schema field would break the existing-files compat invariant.
46
99
  const DOCTOR_STALE_MS = 24 * 60 * 60 * 1000;
@@ -51,6 +104,37 @@ export async function plan(argv, cwd = process.cwd(), adapters = allAdapters(),
51
104
  throw new Error(`--mode must be one of ${ROUTING_MODES.join(" | ")} (got ${values.mode})`);
52
105
  }
53
106
  const g = loadGraph(cwd);
107
+ const oracleRefusals = new Map();
108
+ if (g.tasks.some((task) => task.acceptance.some((item) => typeof item === "object" && item.oracle === "test"))) {
109
+ const listing = await (opts.listTests ?? listVitestTests)(cwd);
110
+ for (const task of g.tasks) {
111
+ const refusals = [];
112
+ if (listing.status === "failed") {
113
+ if (task.acceptance.some((item) => typeof item === "object" && item.oracle === "test")) {
114
+ refusals.push(`acceptance oracle unresolved — runner listing failed: ${listing.error.split("\n")[0]}`);
115
+ }
116
+ }
117
+ else {
118
+ for (const audit of auditNamedTestOracles(task.acceptance, listing.tests)) {
119
+ if (audit.matches.length === 0) {
120
+ refusals.push(`acceptance oracle ${JSON.stringify(audit.criterion)} matches zero runner-listed test names`);
121
+ }
122
+ else if (audit.matches.length > 1) {
123
+ refusals.push(`acceptance oracle ${JSON.stringify(audit.criterion)} matches ${audit.matches.length} runner-listed test names`);
124
+ }
125
+ }
126
+ }
127
+ if (refusals.length)
128
+ oracleRefusals.set(task.id, refusals);
129
+ }
130
+ }
131
+ const inputFindings = await taskInputFindings(g.tasks, cwd);
132
+ const inputRefusals = new Map();
133
+ for (const finding of inputFindings) {
134
+ if (finding.severity !== "refuse")
135
+ continue;
136
+ inputRefusals.set(finding.taskId, [...(inputRefusals.get(finding.taskId) ?? []), finding.detail]);
137
+ }
54
138
  const { cfg, mode, source } = resolveRunMode(cwd, { flag: values.mode, spec: g.mode });
55
139
  // readDoctor cache path: staleness line only fires here (probeAll fallback is fresh by construction).
56
140
  const cached = readDoctor(cwd);
@@ -152,6 +236,11 @@ export async function plan(argv, cwd = process.cwd(), adapters = allAdapters(),
152
236
  let cost = 0;
153
237
  const routed = [];
154
238
  for (const t of g.tasks) {
239
+ const refusals = [...(oracleRefusals.get(t.id) ?? []), ...(inputRefusals.get(t.id) ?? [])];
240
+ if (refusals.length) {
241
+ lines.push(` ${t.id.padEnd(6)} ${t.shape.padEnd(10)} !! pre-dispatch refusal — ${refusals.join("; ")}`);
242
+ continue;
243
+ }
155
244
  try {
156
245
  const r = route(t, cfg, channels, dispatchProfile);
157
246
  routed.push({ taskId: t.id, adapter: r.assignment.adapter, model: r.assignment.model });
@@ -192,6 +281,10 @@ export async function plan(argv, cwd = process.cwd(), adapters = allAdapters(),
192
281
  lints.push(`${t.id}: unroutable — ${msg}`);
193
282
  }
194
283
  }
284
+ const inputWarnings = inputFindings.filter((finding) => finding.severity === "warn");
285
+ if (inputWarnings.length) {
286
+ lines.push("", "input warnings:", ...inputWarnings.map((finding) => ` ! ${finding.taskId}: ${finding.detail}`));
287
+ }
195
288
  lines.push("", `est. cost (API channels only, rough): ~$${cost.toFixed(2)} + judge/review/consult calls`);
196
289
  // VIS-04: summary only when a profile is active AND something deviates. Labeled by the switch — off = preview.
197
290
  if (profile && deviations) {
@@ -1,4 +1,4 @@
1
- import { readFileSync } from "node:fs";
1
+ import { existsSync, readFileSync } from "node:fs";
2
2
  import { join } from "node:path";
3
3
  import { GLYPHS, LIVE } from "../../brand.js";
4
4
  import { DEFAULT_CONFIG, loadConfig } from "../../config/config.js";
@@ -7,8 +7,8 @@ import { formatOwnedName, parseOwnedName, } from "../../drivers/types.js";
7
7
  import { blockedTasks, graphDefinitionHash, loadGraph, stateDirName } from "../../graph/graph.js";
8
8
  import { GATE_NAMES } from "../../graph/schema.js";
9
9
  import { foldActivity } from "../../run/activity.js";
10
- import { Journal, engagementComparable, isQualityFailureParkKind, preservedRefsByTask, recordedTaskFailureKind, runHasEnded, upheldFeedbackByTask, } from "../../run/journal.js";
11
- import { isPidLive } from "../../run/lock.js";
10
+ import { Journal, engagementComparable, isQualityFailureParkKind, parseRunId, preservedRefsByTask, recordedTaskFailureKind, runHasEnded, upheldFeedbackByTask, } from "../../run/journal.js";
11
+ import { isPidLive, runLockOwner } from "../../run/lock.js";
12
12
  import { normalizeGateOutcome } from "../../run/outcome.js";
13
13
  import { desiredPanes } from "../../run/reconcile.js";
14
14
  import { normalizeStallSnapshot } from "../../run/stall.js";
@@ -850,20 +850,56 @@ const parseJournalSnapshot = (raw) => raw.split("\n").flatMap((line) => {
850
850
  return [];
851
851
  }
852
852
  });
853
+ // runLockOwner is the single lock payload/liveness reader. Its current path helper also initializes
854
+ // .tickmarkr metadata; an engine-written lock necessarily passed through that initializer already, so
855
+ // status avoids invoking it in stripped reader-purity fixtures where no engine lock can exist.
856
+ const canReadRunLockOwner = (cwd) => existsSync(join(cwd, stateDirName(cwd), ".gitignore"));
857
+ const reportableLockRunId = (owner) => {
858
+ if (!owner?.live || owner.runId === undefined)
859
+ return undefined;
860
+ try {
861
+ return parseRunId(owner.runId);
862
+ }
863
+ catch {
864
+ // Repository-wide locks can be held for non-run work such as compile; those are not a status run.
865
+ return undefined;
866
+ }
867
+ };
868
+ const readRunJournalRaw = (cwd, runId, allowMissingJournal) => {
869
+ const dir = join(cwd, stateDirName(cwd), "runs", runId);
870
+ try {
871
+ return readFileSync(join(dir, "journal.jsonl"), "utf8");
872
+ }
873
+ catch (error) {
874
+ if (error.code === "ENOENT") {
875
+ if (allowMissingJournal)
876
+ return "";
877
+ throw new Error(`no journal for ${runId} at ${dir}`);
878
+ }
879
+ throw error;
880
+ }
881
+ };
853
882
  /**
854
883
  * ONE journal read per answer, and every claim about the run folded from THOSE bytes — the board's
855
884
  * task rows, activity, phases, gates, liveness and tip verification, and the compact one-line form
856
885
  * alike. A second read is what lets two snapshots be presented as one state: a line the daemon
857
886
  * appends between them pairs a task count from one instant with a verdict from another.
858
887
  *
859
- * An explicit <runId> is a resolution, not a hint: `Journal.open` refuses an id without a readable
860
- * journal, so status fails loudly naming that id instead of rendering any other run.
888
+ * An explicit <runId> is a resolution, not a hint: it refuses an id without a readable journal, so
889
+ * status fails loudly naming that id instead of rendering any other run. The implicit form first
890
+ * asks the repository lock accessor which run is live, then falls back to the newest journalled run;
891
+ * a lock-selected run may still be in the directory-before-first-journal window, which renders as an
892
+ * empty snapshot for that run rather than borrowing the previous run's journal.
861
893
  */
862
894
  const readRunRecord = (cwd, graph, namedRunId) => {
863
- const runId = namedRunId ?? Journal.latestRunId(cwd, { withJournal: true });
895
+ const explicitRunId = namedRunId === undefined ? undefined : parseRunId(namedRunId);
896
+ const lockedRunId = explicitRunId === undefined && canReadRunLockOwner(cwd)
897
+ ? reportableLockRunId(runLockOwner(cwd))
898
+ : undefined;
899
+ const runId = explicitRunId ?? lockedRunId ?? Journal.latestRunId(cwd, { withJournal: true });
864
900
  if (!runId)
865
901
  return undefined;
866
- const raw = readFileSync(join(Journal.open(cwd, runId).dir, "journal.jsonl"), "utf8");
902
+ const raw = readRunJournalRaw(cwd, runId, lockedRunId === runId);
867
903
  const events = parseJournalSnapshot(raw);
868
904
  // The resume comparator is the fail-closed baseline; a matching graph-rehash is the daemon's
869
905
  // append-only audit that authorizes this status replay after stop-amend-resume.
@@ -1380,15 +1416,19 @@ export async function status(argv, cwd = process.cwd(), opts = {}) {
1380
1416
  let journalCursor = 0;
1381
1417
  const consumeDecisionEvents = () => {
1382
1418
  // A named run is followed, never re-resolved: --watch <runId> keeps reporting that run even as
1383
- // newer runs start. Only the no-argument form tracks latest, as before.
1384
- const runId = namedRunId ?? Journal.latestRunId(cwd, { withJournal: true });
1419
+ // newer runs start. The no-argument form tracks the live lock first, then latest journal.
1420
+ const explicitRunId = namedRunId === undefined ? undefined : parseRunId(namedRunId);
1421
+ const lockedRunId = explicitRunId === undefined && canReadRunLockOwner(cwd)
1422
+ ? reportableLockRunId(runLockOwner(cwd))
1423
+ : undefined;
1424
+ const runId = explicitRunId ?? lockedRunId ?? Journal.latestRunId(cwd, { withJournal: true });
1385
1425
  if (!runId)
1386
1426
  return [];
1387
1427
  if (decisionRunId !== runId) {
1388
1428
  decisionRunId = runId;
1389
1429
  journalCursor = 0;
1390
1430
  }
1391
- const journalEvents = Journal.open(cwd, runId).read();
1431
+ const journalEvents = parseJournalSnapshot(readRunJournalRaw(cwd, runId, lockedRunId === runId));
1392
1432
  if (journalEvents.length < journalCursor)
1393
1433
  journalCursor = 0;
1394
1434
  const fresh = decisionEventsFromJournal(journalEvents, runId, stateDirName(cwd))
@@ -2,7 +2,7 @@ import { existsSync, readFileSync, statSync } from "node:fs";
2
2
  import { join } from "node:path";
3
3
  import { validateGraph } from "../graph/schema.js";
4
4
  import { taskUnitContractErrors } from "./collateral.js";
5
- import { ownershipFindings, renderOwnershipFinding } from "./ownership.js";
5
+ import { blocksCompile, ownershipFindings, renderOwnershipFinding } from "./ownership.js";
6
6
  import { CompileError } from "./common.js";
7
7
  import { compileGsd, isGsdPhaseDir } from "./gsd.js";
8
8
  import { compileNative, TICKMARKR_NATIVE_MARKER } from "./native.js";
@@ -53,13 +53,23 @@ export function finalizePlan(plan, src, repoRoot) {
53
53
  },
54
54
  tasks: plan.tasks,
55
55
  }), src);
56
- // overseer-217: report-only in 2.1.7. The 2.1.8 removal condition is one real authored-graph run
57
- // plus a measured false-positive rate for the conventional source-name → test-name mapping.
58
- // Until then this warning MUST NOT become a compile abort: a heuristic cannot deadlock compile.
56
+ // overseer-217 removal condition, now paid: on this milestone's authored graph the conventional
57
+ // name map emitted 21 raw unowned-test findings; review found 1 real and 20 false, while intersecting
58
+ // with a direct import or command-entry spawn retained the real one and left 0 false positives. That
59
+ // is a precision measurement, NOT a recall claim. The prior halted semantic-contract class has no
60
+ // name/import/ownership relation (one of its three members has no matching literal anywhere), so this
61
+ // rule could not have caught it and does not claim to. The promoted rule already earned a true positive
62
+ // during authoring: assigning the plan command to oracle-preflight left two dedicated plan tests unowned.
59
63
  if (repoRoot) {
60
- for (const finding of ownershipFindings(graph.tasks, repoRoot)) {
64
+ const findings = ownershipFindings(graph.tasks, repoRoot);
65
+ const blocking = findings.filter(blocksCompile);
66
+ for (const finding of findings.filter((item) => !blocksCompile(item))) {
61
67
  console.warn(renderOwnershipFinding(finding));
62
68
  }
69
+ if (blocking.length > 0) {
70
+ throw new CompileError(`${src} violates cross-task test ownership (${blocking.length} error${blocking.length === 1 ? "" : "s"}):\n`
71
+ + blocking.map((finding) => ` - ${renderOwnershipFinding(finding)}`).join("\n"));
72
+ }
63
73
  }
64
74
  return graph;
65
75
  }
@@ -789,6 +789,20 @@ acceptance is required on every task (a nested list of observable outcomes).
789
789
  hard value anywhere in the domain, the criterion asserts a universal that may be FALSE ABOUT THE
790
790
  WORLD — bound it or say where it stops holding, rather than demanding a value that does not exist.
791
791
 
792
+ PICK THE CRITERION FORM FROM WHO COULD BE WRONG:
793
+ - When the WORKER could be wrong because it can choose the value, use "test:" and pin the exact
794
+ literal it could otherwise choose; an example selected by its implementer proves only itself.
795
+ - When the AUTHOR could be wrong by omitting a member from a list, quantify universally over the
796
+ authoritative closed set; a hand-written enumeration can repeat the same omission as the code.
797
+ - When the REVIEWER could be wrong about a prose artefact, use "judge:" to replay a recorded incident
798
+ against the changed prose; a keyword check proves vocabulary, not that the artefact prevents a repeat.
799
+
800
+ PRE-SCOPE BY TEXT, ENUMERATE BLOCKERS BY EXECUTION:
801
+ - Before assigning files[], sweep text across the repository tree for names, callers, tests and prose.
802
+ A text sweep produces a candidate list; only running the change enumerates the real blocker set.
803
+ Keep the candidates for scope, then execute the production path and full gates before declaring the
804
+ set closed — this milestone paid a halted run to learn that the two populations are not identical.
805
+
792
806
  WHICH SIDE OF A RUN INHERITS ENVIRONMENT — AND IT DEPENDS ON THE DRIVER (OBS-542):
793
807
  - Gate commands and "command:"/"test:" oracles INHERIT THE DAEMON'S ENVIRONMENT. They are children of
794
808
  the daemon, so launching it as \`bash -c 'set -a; . .env.test; set +a; exec tickmarkr run'\` reaches
@@ -1,8 +1,17 @@
1
1
  import type { Task } from "../graph/schema.js";
2
+ export type OwnershipCorroboration = {
3
+ kind: "direct-import";
4
+ source: string;
5
+ } | {
6
+ kind: "command-entry-spawn";
7
+ source: string;
8
+ entry: "src/cli/index.ts";
9
+ };
2
10
  export type OwnershipFinding = {
3
11
  code: "unowned-test";
4
12
  test: string;
5
13
  taskIds: string[];
14
+ corroboration?: OwnershipCorroboration;
6
15
  detail: string;
7
16
  } | {
8
17
  code: "test-path-outside-allowlist";
@@ -18,8 +27,9 @@ export type OwnershipFinding = {
18
27
  detail: string;
19
28
  };
20
29
  /**
21
- * Advisory cross-task ownership check. Findings are data: callers may report them, but this checker
22
- * never throws and never changes the graph.
30
+ * Cross-task ownership evidence. Findings are data: this checker never throws or changes the graph;
31
+ * the compile seam promotes only a corroborated unowned-test finding and reports every other shape.
23
32
  */
24
33
  export declare function ownershipFindings(tasks: readonly Task[], repoRoot: string): OwnershipFinding[];
25
34
  export declare function renderOwnershipFinding(finding: OwnershipFinding): string;
35
+ export declare function blocksCompile(finding: OwnershipFinding): boolean;
@@ -1,5 +1,5 @@
1
1
  import { readFileSync, readdirSync } from "node:fs";
2
- import { basename, extname, join } from "node:path";
2
+ import { basename, extname, join, posix } from "node:path";
3
3
  import { filesGlob } from "../graph/files-glob.js";
4
4
  import { collateralHits } from "./collateral.js";
5
5
  const normalize = (path) => path.replace(/^\.\//, "").split("\\").join("/");
@@ -22,20 +22,79 @@ function testSources(repoRoot) {
22
22
  return [];
23
23
  }
24
24
  }
25
- // Conventional, not universal: status-watch-alive.test.ts is dedicated to status.ts even though it
26
- // reaches that command through the CLI entry point. Keep this heuristic advisory until authored-graph
27
- // measurements establish its false-positive rate.
28
- function namedSourceTasks(test, tasks) {
25
+ function namedSources(test, tasks) {
29
26
  const stem = basename(test).replace(/\.test\.ts$/, "");
30
- const ids = new Set();
27
+ const matches = new Map();
31
28
  for (const task of tasks) {
32
29
  for (const entry of task.files.map(normalize).filter((path) => path.startsWith("src/") && !/[*?{[]/.test(path))) {
33
30
  const source = basename(entry, extname(entry));
34
- if (stem === source || stem.startsWith(`${source}-`))
35
- ids.add(task.id);
31
+ if (stem === source || stem.startsWith(`${source}-`)) {
32
+ matches.set(`${task.id}:${entry}`, { taskId: task.id, source: entry });
33
+ }
34
+ }
35
+ }
36
+ return [...matches.values()];
37
+ }
38
+ const moduleKey = (path) => normalize(path).replace(/\.(?:[cm]?[jt]sx?)$/, "");
39
+ function directImportSpecifiers(text) {
40
+ // Comments cannot create an edge. Keep strings intact because they are the import target.
41
+ const source = text.replace(/\/\*[\s\S]*?\*\//g, "").replace(/^\s*\/\/.*$/gm, "");
42
+ const specifiers = new Set();
43
+ for (const match of source.matchAll(/\bimport\s+(?:type\s+)?(?:[\w$*{},\s]+?\s+from\s+)?["']([^"']+)["']/g)) {
44
+ specifiers.add(match[1]);
45
+ }
46
+ for (const match of source.matchAll(/\bimport\s*\(\s*["']([^"']+)["']\s*\)/g)) {
47
+ specifiers.add(match[1]);
48
+ }
49
+ return [...specifiers];
50
+ }
51
+ function directlyImports(test, source) {
52
+ // DIRECT is load-bearing: do not walk through imported helpers. src/run/journal.ts alone has 84
53
+ // test importers in the measured tree, so transitive closure would recreate the raw alarm flood.
54
+ const target = moduleKey(source);
55
+ return directImportSpecifiers(test.text).some((specifier) => {
56
+ const imported = specifier.startsWith(".")
57
+ ? posix.normalize(posix.join(posix.dirname(test.path), specifier))
58
+ : specifier.startsWith("src/") ? specifier : "";
59
+ return imported !== "" && moduleKey(imported) === target;
60
+ });
61
+ }
62
+ function invokesChildProcessSpawn(text) {
63
+ const source = text.replace(/\/\*[\s\S]*?\*\//g, "").replace(/^\s*\/\/.*$/gm, "");
64
+ const bindings = new Set();
65
+ for (const match of source.matchAll(/\bimport\s*{([^}]*)}\s*from\s*["'](?:node:)?child_process["']/g)) {
66
+ for (const member of match[1].split(",")) {
67
+ const binding = member.trim().match(/^spawn(?:Sync)?(?:\s+as\s+([A-Za-z_$][\w$]*))?$/);
68
+ if (binding)
69
+ bindings.add(binding[1] ?? member.trim());
36
70
  }
37
71
  }
38
- return [...ids];
72
+ for (const match of source.matchAll(/\bimport\s*\*\s*as\s*([A-Za-z_$][\w$]*)\s*from\s*["'](?:node:)?child_process["']/g)) {
73
+ if (new RegExp(`\\b${match[1]}\\.spawn(?:Sync)?\\s*\\(`).test(source))
74
+ return true;
75
+ }
76
+ return [...bindings].some((binding) => new RegExp(`\\b${binding}\\s*\\(`).test(source));
77
+ }
78
+ function mentionsCommandEntry(text) {
79
+ return /(?:^|\/)src\/cli\/index\.(?:ts|js)\b/.test(text)
80
+ || /["'`]src["'`]\s*,\s*["'`]cli["'`]\s*,\s*["'`]index\.(?:ts|js)["'`]/.test(text);
81
+ }
82
+ function corroboration(test, matches) {
83
+ // A .test.ts-shaped collateral fixture is not by itself a dedicated test. Requiring a runner leaf
84
+ // keeps import-only scan fixtures advisory while every executable subject in the measured union stays.
85
+ const executable = test.text.replace(/\/\*[\s\S]*?\*\//g, "").replace(/^\s*\/\/.*$/gm, "");
86
+ if (!/\b(?:test|it)(?:\.(?:concurrent|each|fails|only|skip|todo))*\s*\(/.test(executable))
87
+ return undefined;
88
+ for (const match of matches) {
89
+ if (directlyImports(test, match.source))
90
+ return { kind: "direct-import", source: match.source };
91
+ }
92
+ if (invokesChildProcessSpawn(test.text) && mentionsCommandEntry(test.text)) {
93
+ const command = matches.find(({ source }) => /^src\/cli\/commands\/[^/]+\.(?:[cm]?[jt]sx?)$/.test(source));
94
+ if (command)
95
+ return { kind: "command-entry-spawn", source: command.source, entry: "src/cli/index.ts" };
96
+ }
97
+ return undefined;
39
98
  }
40
99
  function repositoryPaths(text) {
41
100
  const paths = new Set();
@@ -62,11 +121,12 @@ function dependencyOrdered(a, b, byId) {
62
121
  return reaches(a, b.id) || reaches(b, a.id);
63
122
  }
64
123
  /**
65
- * Advisory cross-task ownership check. Findings are data: callers may report them, but this checker
66
- * never throws and never changes the graph.
124
+ * Cross-task ownership evidence. Findings are data: this checker never throws or changes the graph;
125
+ * the compile seam promotes only a corroborated unowned-test finding and reports every other shape.
67
126
  */
68
127
  export function ownershipFindings(tasks, repoRoot) {
69
128
  const sources = testSources(repoRoot);
129
+ const sourceByPath = new Map(sources.map((source) => [source.path, source]));
70
130
  const byId = new Map(tasks.map((task) => [task.id, task]));
71
131
  const indexed = tasks.map((task) => {
72
132
  const files = task.files.map(normalize);
@@ -81,6 +141,8 @@ export function ownershipFindings(tasks, repoRoot) {
81
141
  const predictedBy = new Map();
82
142
  for (const [taskId, hits] of collateralHits(tasks, repoRoot)) {
83
143
  for (const hit of hits) {
144
+ if (!sourceByPath.has(hit))
145
+ continue;
84
146
  const ids = predictedBy.get(hit) ?? new Set();
85
147
  ids.add(taskId);
86
148
  predictedBy.set(hit, ids);
@@ -88,7 +150,7 @@ export function ownershipFindings(tasks, repoRoot) {
88
150
  }
89
151
  for (const source of sources) {
90
152
  const ids = predictedBy.get(source.path) ?? new Set();
91
- for (const taskId of namedSourceTasks(source.path, tasks))
153
+ for (const { taskId } of namedSources(source.path, tasks))
92
154
  ids.add(taskId);
93
155
  if (ids.size > 0)
94
156
  predictedBy.set(source.path, ids);
@@ -97,11 +159,17 @@ export function ownershipFindings(tasks, repoRoot) {
97
159
  for (const [test, taskIds] of predictedBy) {
98
160
  if (owners(test).length === 0) {
99
161
  const ids = [...taskIds].sort();
162
+ const source = sourceByPath.get(test);
163
+ const evidence = corroboration(source, namedSources(test, tasks));
100
164
  findings.push({
101
165
  code: "unowned-test",
102
166
  test,
103
167
  taskIds: ids,
104
- detail: `${test} is a dedicated test of source owned by ${ids.join(", ")} but no task owns the test`,
168
+ ...(evidence ? { corroboration: evidence } : {}),
169
+ detail: `${test} is a dedicated test of source owned by ${ids.join(", ")} but no task owns the test`
170
+ + (evidence?.kind === "direct-import" ? `; it imports ${evidence.source} directly`
171
+ : evidence?.kind === "command-entry-spawn"
172
+ ? `; it spawns ${evidence.entry} to exercise ${evidence.source}` : ""),
105
173
  });
106
174
  }
107
175
  }
@@ -140,3 +208,6 @@ export function ownershipFindings(tasks, repoRoot) {
140
208
  export function renderOwnershipFinding(finding) {
141
209
  return `tickmarkr: ownership-lint[${finding.code}]: ${finding.detail}`;
142
210
  }
211
+ export function blocksCompile(finding) {
212
+ return finding.code === "unowned-test" && finding.corroboration !== undefined;
213
+ }
@@ -28,6 +28,19 @@ export interface VitestListedTest {
28
28
  file: string;
29
29
  projectName?: string;
30
30
  }
31
+ export type VitestListResult = {
32
+ status: "listed";
33
+ tests: VitestListedTest[];
34
+ } | {
35
+ status: "failed";
36
+ error: string;
37
+ };
38
+ export declare function listVitestTests(cwd: string): Promise<VitestListResult>;
39
+ export interface NamedTestAudit {
40
+ criterion: string;
41
+ matches: VitestListedTest[];
42
+ }
43
+ export declare function auditNamedTestOracles(items: readonly AcceptanceItem[], listedTests: readonly VitestListedTest[]): NamedTestAudit[];
31
44
  export type AcceptanceCorpusAuditResult = {
32
45
  specPath: string;
33
46
  status: "parse-failed";
@@ -98,6 +98,47 @@ export function testFiltered(testCmd, name) {
98
98
  const fwd = wrapped ? "-- " : "";
99
99
  return `${testCmd} ${fwd}-t ${shq(pattern)}`;
100
100
  }
101
+ const VitestListedTestsSchema = z.array(z.object({
102
+ name: z.string(),
103
+ file: z.string(),
104
+ projectName: z.string().optional(),
105
+ }));
106
+ export async function listVitestTests(cwd) {
107
+ const result = await sh(`${shq(join(cwd, "node_modules/.bin/vitest"))} list --json`, cwd);
108
+ if (result.code !== 0) {
109
+ return { status: "failed", error: (result.stderr || result.stdout || `exit ${result.code}`).trim() };
110
+ }
111
+ try {
112
+ const start = result.stdout.indexOf("[");
113
+ if (start < 0)
114
+ return { status: "failed", error: "runner emitted no JSON test listing" };
115
+ const parsed = VitestListedTestsSchema.safeParse(JSON.parse(result.stdout.slice(start)));
116
+ return parsed.success
117
+ ? { status: "listed", tests: parsed.data }
118
+ : { status: "failed", error: z.prettifyError(parsed.error) };
119
+ }
120
+ catch (error) {
121
+ return { status: "failed", error: error instanceof Error ? error.message : String(error) };
122
+ }
123
+ }
124
+ export function auditNamedTestOracles(items, listedTests) {
125
+ const runnerNames = listedTests.map((listed) => ({
126
+ listed,
127
+ fullName: listed.name.split(" > ").join(" "),
128
+ }));
129
+ return items.flatMap((item) => {
130
+ if (typeof item !== "object" || item.oracle !== "test")
131
+ return [];
132
+ return [{
133
+ criterion: item.test,
134
+ // OBS-511: mirror the gate's leaf-anchored suffix rule — this denominator must count
135
+ // exactly the tests the shipped -t filter would select.
136
+ matches: runnerNames
137
+ .filter(({ fullName }) => fullName === item.test || fullName.endsWith(` ${item.test}`))
138
+ .map(({ listed }) => listed),
139
+ }];
140
+ });
141
+ }
101
142
  function corpusSpecPaths(root) {
102
143
  const paths = [];
103
144
  const visit = (dir) => {
@@ -116,32 +157,18 @@ function corpusSpecPaths(root) {
116
157
  // listing. Every discovered path contributes either all parser-produced acceptance items or one named
117
158
  // parse failure; exceptions are evidence, never permission to shrink the corpus silently.
118
159
  export function auditAcceptanceCorpus(corpusRoot, listedTests) {
119
- const runnerNames = listedTests.map((listed) => ({
120
- listed,
121
- fullName: listed.name.split(" > ").join(" "),
122
- }));
123
160
  const results = [];
124
161
  for (const specPath of corpusSpecPaths(corpusRoot)) {
125
162
  try {
126
163
  const graph = compileNative(specPath);
127
164
  for (const task of graph.tasks) {
128
165
  for (const item of task.acceptance) {
166
+ const namedTest = auditNamedTestOracles([item], listedTests)[0];
129
167
  results.push({
130
168
  specPath,
131
169
  status: "parsed",
132
170
  item,
133
- ...(typeof item === "object" && item.oracle === "test"
134
- ? {
135
- namedTest: {
136
- criterion: item.test,
137
- // OBS-511: mirror the gate's leaf-anchored suffix rule — the audit's denominator
138
- // must count exactly the tests the -t filter would select, or doctor and gate disagree.
139
- matches: runnerNames
140
- .filter(({ fullName }) => fullName === item.test || fullName.endsWith(` ${item.test}`))
141
- .map(({ listed }) => listed),
142
- },
143
- }
144
- : {}),
171
+ ...(namedTest ? { namedTest } : {}),
145
172
  });
146
173
  }
147
174
  }
@@ -176,11 +176,11 @@ async function coveringTests(worktree, baseRef) {
176
176
  }
177
177
  const covering = tests.filter((t) => reachOf(t).has(file));
178
178
  if (!covering.length)
179
- return undefined; // nothing covers this file — only the full suite can speak for it
179
+ continue; // nothing covers this file — keep every attributable selection already accumulated
180
180
  for (const t of covering)
181
181
  selected.add(t);
182
182
  }
183
- return [...selected].sort();
183
+ return selected.size ? [...selected].sort() : undefined;
184
184
  }
185
185
  /**
186
186
  * The configured test command narrowed to these files. Mirrors testFiltered's `--` rule (acceptance.ts:104):
@@ -287,7 +287,7 @@ export async function runGates(task, ctx) {
287
287
  }
288
288
  return { results: sorted, commits };
289
289
  };
290
- const toolGates = ["build", "test", "lint"].filter(enabled);
290
+ const toolGates = ["build", "lint"].filter(enabled);
291
291
  /**
292
292
  * v1.87 T5: the shell gates run their commands against the WORKING TREE, while evidence, scope,
293
293
  * the judged diff and the merge all read COMMITS. Uncommitted work is therefore visible to
@@ -337,28 +337,28 @@ export async function runGates(task, ctx) {
337
337
  + `Uncommitted at round end:\n${dirt}`,
338
338
  meta: { dirtyWorktree: true, dirtyAtRoundEnd: true },
339
339
  });
340
- // build/test/lint vs the shared baseline
341
- const runBattery = async (commands, selected) => {
342
- if (!toolGates.length)
340
+ // shell tools vs the shared baseline
341
+ const runBattery = async (commands, selected, gates = toolGates) => {
342
+ if (!gates.length)
343
343
  return;
344
344
  if (!v185) {
345
- // ponytail: compareToBaseline batches build/test/lint — their starts are emitted at iteration,
345
+ // ponytail: compareToBaseline batches adjacent tools — their starts are emitted at iteration,
346
346
  // not at true execution start. They are collectively sub-second (measured), so the debounce
347
347
  // suppresses them anyway; split compareToBaseline only if a tool gate ever gets slow.
348
- // ponytail: legacy runs build/test/lint in ONE compareToBaseline call, so there is one interval
348
+ // ponytail: legacy runs adjacent tools in ONE compareToBaseline call, so there is one interval
349
349
  // to measure and each of its gates carries it. Split it only if this branch ever stops batching.
350
350
  const batchAt = Date.now();
351
351
  const batchLoadStart = loadProvider();
352
- const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, toolGates);
352
+ const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [...gates]);
353
353
  const batch = { durationMs: Date.now() - batchAt, load1Start: batchLoadStart, load1End: loadProvider() };
354
- for (const g of toolGates)
354
+ for (const g of gates)
355
355
  spans.set(g, batch);
356
356
  // The same refusal AFTER the commands, because a green command can dirty the tree the check
357
357
  // above just proved clean. Batched, legacy cannot say WHICH command did it, so the refusal
358
358
  // lands on the last gate that had one — the round dies there either way. A red battery is
359
359
  // reported as the red it is: the round already ends, and the command output is the better lead.
360
360
  const dirt = toolResults.every((r) => r.pass) ? await dirtyWorktree() : undefined;
361
- const blame = dirt ? [...toolGates].reverse().find((g) => commands[g]) : undefined;
361
+ const blame = dirt ? [...gates].reverse().find((g) => commands[g]) : undefined;
362
362
  for (const r of toolResults) {
363
363
  await emitStart(r.gate);
364
364
  await record(r.gate === blame ? dirtyRefusal(blame, dirt, commands[blame]) : r);
@@ -366,8 +366,8 @@ export async function runGates(task, ctx) {
366
366
  return;
367
367
  }
368
368
  // T4 (OBS-265): one command at a time, stopping at the first red — a failed build no longer buys
369
- // the full vitest suite before anyone reads its verdict.
370
- for (const g of toolGates) {
369
+ // any later tool before anyone reads its verdict.
370
+ for (const g of gates) {
371
371
  await emitStart(g);
372
372
  const [r] = await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g]));
373
373
  // the screen's interval IS the test gate's first interval, so the split needs no second clock
@@ -437,9 +437,9 @@ export async function runGates(task, ctx) {
437
437
  * screen IS the round's verdict: what it produced is journaled (in the order it ran) and the round
438
438
  * ends there, so a drive-by out-of-scope edit costs <1s instead of the whole battery.
439
439
  *
440
- * A green screen changes nothing downstream. The recorded sequence stays GATE_NAMES order, so the
441
- * journal, `tickmarkr report`, the surfaces, and resume's GATE_NAMES walk over already-satisfied
442
- * gates all keep reading exactly one order.
440
+ * A green screen changes nothing downstream. The returned record stays in GATE_NAMES order while
441
+ * the event stream reports the order gates actually ran; resume's GATE_NAMES walk over satisfied
442
+ * records therefore keeps declaration order without making the live stream lie about execution.
443
443
  *
444
444
  * ponytail: the price of that is re-reading two git checks (~40ms) in their canonical positions
445
445
  * rather than teaching every consumer of the gate stream a second order. Both reads see the same
@@ -447,7 +447,7 @@ export async function runGates(task, ctx) {
447
447
  * for. Charge it only when there IS a battery command to protect.
448
448
  */
449
449
  const screenBlocks = async () => {
450
- if (!toolGates.some((g) => ctx.commands[g]))
450
+ if (!toolGates.some((g) => ctx.commands[g]) && !(enabled("test") && ctx.commands.test))
451
451
  return false;
452
452
  const screened = [];
453
453
  for (const [gate, compute] of [["evidence", evidenceResult], ["scope", scopeResult]]) {
@@ -626,12 +626,7 @@ export async function runGates(task, ctx) {
626
626
  }
627
627
  if (v185 && await screenBlocks())
628
628
  return done();
629
- // A non-final round may run only the tests covering its own diff; the merge-candidate round below
630
- // pays the full suite anyway, so a selection that misses costs a round and can never merge.
631
- const selected = v185 && ctx.selectTests && enabled("test") && ctx.commands.test
632
- ? await coveringTests(ctx.worktree, ctx.baseRef)
633
- : undefined;
634
- await runBattery(selected ? { ...ctx.commands, test: testCommandForFiles(ctx.commands.test, selected) } : ctx.commands, selected);
629
+ await runBattery(ctx.commands);
635
630
  if (failed())
636
631
  return done();
637
632
  if (enabled("evidence")) {
@@ -644,6 +639,14 @@ export async function runGates(task, ctx) {
644
639
  if (failed())
645
640
  return done();
646
641
  }
642
+ // A non-final round may run only the tests covering its own diff; the merge-candidate round below
643
+ // pays the full suite anyway, so a selection that misses costs a round and can never merge.
644
+ const selected = v185 && ctx.selectTests && enabled("test") && ctx.commands.test
645
+ ? await coveringTests(ctx.worktree, ctx.baseRef)
646
+ : undefined;
647
+ await runBattery(selected ? { ...ctx.commands, test: testCommandForFiles(ctx.commands.test, selected) } : ctx.commands, selected, enabled("test") ? ["test"] : []);
648
+ if (failed())
649
+ return done();
647
650
  if (v185 && (enabled("acceptance") || enabled("review"))) {
648
651
  // Judge and review are launched TOGETHER (96m of serialization over 5 runs). Enforcement is
649
652
  // unchanged — it is still the AND of both, both still fail closed, and neither reads the other's
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "tickmarkr",
3
- "version": "2.1.7",
3
+ "version": "2.1.8",
4
4
  "description": "Spec in, verified work out.",
5
5
  "type": "module",
6
6
  "license": "MIT",