tickmarkr 2.1.5 → 2.1.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,142 @@
1
+ import { readFileSync, readdirSync } from "node:fs";
2
+ import { basename, extname, join } from "node:path";
3
+ import { filesGlob } from "../graph/files-glob.js";
4
+ import { collateralHits } from "./collateral.js";
5
+ const normalize = (path) => path.replace(/^\.\//, "").split("\\").join("/");
6
+ function testSources(repoRoot) {
7
+ const root = join(repoRoot, "tests");
8
+ try {
9
+ return readdirSync(root, { recursive: true, encoding: "utf8" })
10
+ .filter((path) => path.endsWith(".test.ts"))
11
+ .sort()
12
+ .flatMap((path) => {
13
+ try {
14
+ return [{ path: `tests/${normalize(path)}`, text: readFileSync(join(root, path), "utf8") }];
15
+ }
16
+ catch {
17
+ return [];
18
+ }
19
+ });
20
+ }
21
+ catch {
22
+ return [];
23
+ }
24
+ }
25
+ // Conventional, not universal: status-watch-alive.test.ts is dedicated to status.ts even though it
26
+ // reaches that command through the CLI entry point. Keep this heuristic advisory until authored-graph
27
+ // measurements establish its false-positive rate.
28
+ function namedSourceTasks(test, tasks) {
29
+ const stem = basename(test).replace(/\.test\.ts$/, "");
30
+ const ids = new Set();
31
+ for (const task of tasks) {
32
+ for (const entry of task.files.map(normalize).filter((path) => path.startsWith("src/") && !/[*?{[]/.test(path))) {
33
+ const source = basename(entry, extname(entry));
34
+ if (stem === source || stem.startsWith(`${source}-`))
35
+ ids.add(task.id);
36
+ }
37
+ }
38
+ return [...ids];
39
+ }
40
+ function repositoryPaths(text) {
41
+ const paths = new Set();
42
+ for (const match of text.matchAll(/["'`]((?:src|tests|fixtures|scripts|skills|docs|\.claude)\/[^"'`\s]+)["'`]/g)) {
43
+ paths.add(normalize(match[1]));
44
+ }
45
+ return [...paths].sort();
46
+ }
47
+ function dependencyOrdered(a, b, byId) {
48
+ const reaches = (from, target) => {
49
+ const seen = new Set();
50
+ const pending = [...from.deps];
51
+ while (pending.length > 0) {
52
+ const id = pending.pop();
53
+ if (id === target)
54
+ return true;
55
+ if (seen.has(id))
56
+ continue;
57
+ seen.add(id);
58
+ pending.push(...(byId.get(id)?.deps ?? []));
59
+ }
60
+ return false;
61
+ };
62
+ return reaches(a, b.id) || reaches(b, a.id);
63
+ }
64
+ /**
65
+ * Advisory cross-task ownership check. Findings are data: callers may report them, but this checker
66
+ * never throws and never changes the graph.
67
+ */
68
+ export function ownershipFindings(tasks, repoRoot) {
69
+ const sources = testSources(repoRoot);
70
+ const byId = new Map(tasks.map((task) => [task.id, task]));
71
+ const indexed = tasks.map((task) => {
72
+ const files = task.files.map(normalize);
73
+ const context = task.context.map(normalize);
74
+ return {
75
+ task,
76
+ owns: files.length === 0 ? () => false : filesGlob(files),
77
+ allows: files.length === 0 ? () => true : filesGlob([...files, ...context]),
78
+ };
79
+ });
80
+ const owners = (path) => indexed.filter(({ owns }) => owns(path));
81
+ const predictedBy = new Map();
82
+ for (const [taskId, hits] of collateralHits(tasks, repoRoot)) {
83
+ for (const hit of hits) {
84
+ const ids = predictedBy.get(hit) ?? new Set();
85
+ ids.add(taskId);
86
+ predictedBy.set(hit, ids);
87
+ }
88
+ }
89
+ for (const source of sources) {
90
+ const ids = predictedBy.get(source.path) ?? new Set();
91
+ for (const taskId of namedSourceTasks(source.path, tasks))
92
+ ids.add(taskId);
93
+ if (ids.size > 0)
94
+ predictedBy.set(source.path, ids);
95
+ }
96
+ const findings = [];
97
+ for (const [test, taskIds] of predictedBy) {
98
+ if (owners(test).length === 0) {
99
+ const ids = [...taskIds].sort();
100
+ findings.push({
101
+ code: "unowned-test",
102
+ test,
103
+ taskIds: ids,
104
+ detail: `${test} is a dedicated test of source owned by ${ids.join(", ")} but no task owns the test`,
105
+ });
106
+ }
107
+ }
108
+ for (const source of sources) {
109
+ for (const owner of owners(source.path)) {
110
+ for (const path of repositoryPaths(source.text)) {
111
+ if (!owner.allows(path)) {
112
+ findings.push({
113
+ code: "test-path-outside-allowlist",
114
+ taskId: owner.task.id,
115
+ test: source.path,
116
+ path,
117
+ detail: `${source.path} owned by ${owner.task.id} references ${path} outside that task's files[] and context[]`,
118
+ });
119
+ }
120
+ }
121
+ }
122
+ }
123
+ for (const reader of indexed) {
124
+ for (const entry of reader.task.context.map(normalize)) {
125
+ for (const owner of owners(entry)) {
126
+ if (owner.task.id === reader.task.id || dependencyOrdered(reader.task, owner.task, byId))
127
+ continue;
128
+ findings.push({
129
+ code: "unordered-context-write",
130
+ taskId: reader.task.id,
131
+ ownerTaskId: owner.task.id,
132
+ path: entry,
133
+ detail: `${reader.task.id} names ${entry} as context while ${owner.task.id} owns it for writing, with no dependency order between them`,
134
+ });
135
+ }
136
+ }
137
+ }
138
+ return findings.sort((a, b) => `${a.code}:${"test" in a ? a.test : a.path}:${"taskId" in a ? a.taskId : ""}`.localeCompare(`${b.code}:${"test" in b ? b.test : b.path}:${"taskId" in b ? b.taskId : ""}`));
139
+ }
140
+ export function renderOwnershipFinding(finding) {
141
+ return `tickmarkr: ownership-lint[${finding.code}]: ${finding.detail}`;
142
+ }
@@ -125,6 +125,7 @@ export declare class HerdrDriver implements ExecutorDriver {
125
125
  narrator(cwd: string, command: string, runId?: string): Promise<Slot>;
126
126
  reconcile(desired: Set<string>, runId: string, opts?: {
127
127
  spareLiveLlm?: boolean;
128
+ endedRunIds?: Set<string>;
128
129
  }): Promise<void>;
129
130
  worktree(repo: string, branch: string, baseRef: string): Promise<string>;
130
131
  }
@@ -1188,12 +1188,13 @@ export class HerdrDriver {
1188
1188
  return s;
1189
1189
  });
1190
1190
  }
1191
- // OBS-17 T2 / v1.22b T1: close every tickmarkr-owned pane that should not exist (superseded attempts,
1192
- // killed-daemon orphans, leftovers from OLDER runs) — in this run's workspace OR misplaced in any
1193
- // other one — then reap the tabs those closes emptied. Ownership is decided ONLY by parseOwnedName
1194
- // (drivers/types.ts panesToClose) — foreign names never become candidates, in any workspace; a pane
1195
- // this same run legitimately holds elsewhere is left alone (only run age marks a misplaced pane
1196
- // garbage). spareLiveLlm: same-run judge/review/consult panes have no journal row while live (their
1191
+ // OBS-17 T2 / v1.22b T1: close THIS RUN'S OWN tickmarkr-owned panes that should not exist
1192
+ // (superseded attempts, killed-daemon orphans of this run), in this run's workspace, then reap the
1193
+ // tabs those closes emptied. Ownership is decided ONLY by parseOwnedName (drivers/types.ts
1194
+ // panesToClose) — foreign names never become candidates. OBS-769/OBS-772/OBS-777: the sweep does
1195
+ // NOT reach other workspaces and reaches another runId only when the daemon's repository-scoped
1196
+ // snapshot proves that run ended. Run age is not consulted and never was. spareLiveLlm: same-run
1197
+ // judge/review/consult panes have no journal row while live (their
1197
1198
  // events land after the verdict is read), so mid-run sweeps spare them; boundary sweeps (start/
1198
1199
  // resume/end) run with nothing in flight and take them too. Cosmetic by contract: every failure —
1199
1200
  // herdr gone, pane vanished mid-sweep, unparseable listing — is swallowed; this method never throws.
@@ -56,6 +56,7 @@ export interface FleetAgent {
56
56
  }
57
57
  export declare function panesToClose(agents: FleetAgent[], desired: Set<string>, ws: string, runId: string, opts?: {
58
58
  spareLiveLlm?: boolean;
59
+ endedRunIds?: Set<string>;
59
60
  }): {
60
61
  paneId: string;
61
62
  tabId?: string;
@@ -81,5 +82,6 @@ export interface ExecutorDriver {
81
82
  narrator?: (cwd: string, command: string, runId?: string) => Promise<Slot>;
82
83
  reconcile?: (desired: Set<string>, runId: string, opts?: {
83
84
  spareLiveLlm?: boolean;
85
+ endedRunIds?: Set<string>;
84
86
  }) => Promise<void>;
85
87
  }
@@ -21,12 +21,53 @@ export function isForeignName(name) {
21
21
  return parseOwnedName(name) === null;
22
22
  }
23
23
  // v1.22b T1: workspace-aware fold over a fleet snapshot — decides which owned task panes are garbage
24
- // right now. In-workspace: the existing desired-set/spareLiveLlm sweep (OBS-17 T2). Out-of-workspace:
25
- // an owned pane from a DIFFERENT run is a misplaced leftover (bug, foreign actor, pre-VIS-10 relic)
26
- // and closes regardless of `desired`; an owned pane from THIS run elsewhere is left alone — a live
27
- // run can legitimately hold panes across workspaces, so only run age marks a misplaced pane garbage.
24
+ // right now: the desired-set/spareLiveLlm sweep (OBS-17 T2), scoped to THIS RUN'S OWN panes (by runId,
25
+ // OBS-772) in THIS RUN'S OWN WORKSPACE (OBS-769). Both conditions, and neither alone is the rule.
28
26
  // Watch panes are operator-owned after run end and are reclaimed by the next run; foreign names
29
- // (parseOwnedName fails) are never candidates, in any workspace.
27
+ // (parseOwnedName fails) are never candidates.
28
+ //
29
+ // OBS-769 — WHY THE SWEEP STOPS AT THE WORKSPACE BOUNDARY. It used to close an owned pane carrying
30
+ // any OTHER runId in any other workspace, unconditionally, as a "misplaced leftover". Two tickmarkr
31
+ // runs in two repositories are lawful (the lock forbids two runs in ONE repository, not on one
32
+ // machine) and herdr gives each its own workspace — so that branch made every pair of concurrent
33
+ // runs kill each other's LIVE workers. Measured 2026-08-28: the run in w0 closed run ...2958's
34
+ // panes at 23:42:40.351/.392, and 53s later ...2958's own task-human sweep closed w0's live codex
35
+ // worker at 23:43:34.096. ...2958 ended 0/8. The death detector cannot see it: closing the pane
36
+ // makes paneAbsent, processTree, confirmedProcessTree and worktreeDelta true by ONE cause, and a
37
+ // closed pane can never accrue the CPU that the `cpu-accruing` hold reads.
38
+ // The comment this replaces claimed "only run age marks a misplaced pane garbage" — there was no age
39
+ // check in the code, and age is the wrong predicate anyway: w0's run STARTED EARLIER than ...2958,
40
+ // so an age rule would have licensed exactly the kill that landed. Run age says nothing about
41
+ // liveness, and a sweeping daemon cannot read another repository's run state. The workspace is the
42
+ // only ownership boundary available without cross-repo I/O, so it is the one enforced.
43
+ // Cost, named: an orphan pane from a dead run stranded in a workspace no later run opens is now left
44
+ // for the operator. That is cosmetic (`reconcile` is cosmetic by contract — "visibility is never a
45
+ // gate"), and a cosmetic cleanup must never be able to kill a live worker.
46
+ // OBS-772 — WHY THE runId LINE EXISTS, AND WHY THE WORKSPACE LINE ALONE WAS NOT THE FIX. The first
47
+ // repair was workspace-scoped only, and its own comment dismissed the residue — "two runs sharing one
48
+ // workspace would still sweep each other" — as unreachable, on the reasoning that one workspace per run
49
+ // is herdr's placement. That reasoned from ONE driver to the whole product. OrcaDriver has no workspace
50
+ // dimension at all: orca.ts passes a single ORCA_SPACE as the workspaceId for EVERY checkout and as
51
+ // `ws`, so `workspaceId !== ws` is never true there and every foreign pane fell straight through. Orca
52
+ // users had zero protection while the defect read as fixed. The runId line is the real rule and it is
53
+ // driver-agnostic: reconcile exists to clean up THIS RUN's panes, and a leftover from a dead run is
54
+ // exactly what cannot be told from a live run's pane without liveness data this process does not have.
55
+ // Both lines are kept — the workspace line preserves the pre-existing sparing of this run's own panes
56
+ // in another workspace, which the runId line alone would not.
57
+ // ⚠ WHAT THE runId LINE COST BEFORE OBS-777 — SUSPENDED, NOT NARROWED, and the price was larger than
58
+ // it read. Sparing every other runId suspended OBS-17's FOUNDING use case: "a killed daemon can't
59
+ // close its slots". This sweep was built to reclaim exactly those orphans, but could not reclaim ANY
60
+ // previous run's panes. Three separate pins asserted the old behaviour (reconcile.test.ts,
61
+ // orca-placement.test.ts,
62
+ // reconcile-live.test.ts); all three were changed deliberately, and the third is why this paragraph
63
+ // exists rather than a shorter one — two flipped pins is a trade, three is a pattern.
64
+ // OBS-777 RESTORES that reclamation: the CALLER passes `opts.endedRunIds`, a Set the daemon computes
65
+ // ONCE at run start from this repository's own `run-end` journals and dead lock holders. This fold
66
+ // stays pure — it gains one optional field, not a repo root — a foreign repository's runId is never
67
+ // resolvable and so stays spared by construction, and no driver learns about workspaces.
68
+ // ponytail: two conditions, no geometry reasoning, nothing driver-specific. `reconcile` is cosmetic by
69
+ // contract, and a cosmetic cleanup must never be able to kill a live worker — which is why the
70
+ // ended-run authority is the only safe way to restore the sweep without reviving the cross-run kill.
30
71
  export function panesToClose(agents, desired, ws, runId, opts) {
31
72
  const out = [];
32
73
  for (const a of agents) {
@@ -35,15 +76,14 @@ export function panesToClose(agents, desired, ws, runId, opts) {
35
76
  const owned = parseOwnedName(a.name);
36
77
  if (!owned || owned.role === "watch")
37
78
  continue;
38
- if (a.workspaceId === ws) {
39
- if (desired.has(a.name))
40
- continue;
41
- if (opts?.spareLiveLlm && owned.runId === runId && (owned.role === "judge" || owned.role === "review" || owned.role === "consult"))
42
- continue;
43
- }
44
- else if (owned.runId === runId) {
45
- continue; // this run's own pane in another workspace — never touched
46
- }
79
+ if (owned.runId !== runId && !opts?.endedRunIds?.has(owned.runId))
80
+ continue;
81
+ if (a.workspaceId !== ws)
82
+ continue; // OBS-769: another workspace is another run's business
83
+ if (desired.has(a.name))
84
+ continue;
85
+ if (opts?.spareLiveLlm && owned.runId === runId && (owned.role === "judge" || owned.role === "review" || owned.role === "consult"))
86
+ continue;
47
87
  out.push({ paneId: a.paneId, tabId: a.tabId });
48
88
  }
49
89
  return out;
@@ -117,7 +117,18 @@ const VOCAB_RE = /\b(?:error|fail(?:ed|ure|ing)?)\b/i;
117
117
  // shapes are emitted by the process that was asked to run the oracle; they are deliberately kept in
118
118
  // this runner-output classifier rather than applied to any judge-authored reason text. A real test
119
119
  // failure still dominates below because one regression-shaped line makes the whole output regression.
120
- const INFRA_RE = /\bE(?:AGAIN|MFILE|NFILE|NOMEM|NOSPC)\b|JavaScript heap out of memory|Cannot allocate memory|Resource temporarily unavailable|Token not found in system keyring|Process from config\.webServer was not able to start/i;
120
+ // OBS-791: `[vitest-worker]: Timeout calling "<method>"` is the SAME birpc mechanism one layer up.
121
+ // Vitest's bundled birpc has a fixed 60s DEFAULT_TIMEOUT with no override (fixed upstream only in
122
+ // vitest 4.x), so a suite whose tests all pass can still die at teardown when the worker->host RPC
123
+ // window closes. `.github/workflows/release.yml` has forgiven this exact fingerprint since 2026-08-11
124
+ // while this classifier called it a regression — the product charged a worker for the failure the
125
+ // release gate was written to forgive. The METHOD NAME IS DELIBERATELY NOT PINNED: the timeout is a
126
+ // property of the RPC window, not of `onTaskUpdate`, and pinning one method would forgive a run and
127
+ // charge its sibling for the same infrastructure event. This stays a CLOSED signature — the
128
+ // `[vitest-worker]: ` prefix is vitest's own RPC layer and never user assertion text — and the
129
+ // per-line vetoes below are untouched, so one real failure anywhere still makes the whole output a
130
+ // regression.
131
+ const INFRA_RE = /\bE(?:AGAIN|MFILE|NFILE|NOMEM|NOSPC)\b|JavaScript heap out of memory|Cannot allocate memory|Resource temporarily unavailable|Token not found in system keyring|Process from config\.webServer was not able to start|\[birpc\] rpc is closed, cannot call\b|\[vitest-worker\]: Timeout calling\b/i;
121
132
  // Capture invalidation is deliberately narrower than the gate's infrastructure vocabulary above:
122
133
  // keyring/config-webServer startup failures remain gate concerns, while this policy is specifically
123
134
  // for evidence that the capture ran while the machine was resource-starved.
@@ -127,6 +138,12 @@ const CAPTURE_EXHAUSTION_RE = /\bE(?:AGAIN|MFILE|NFILE|NOMEM|NOSPC)\b|JavaScript
127
138
  const ERROR_CLASS_RE = /\b[A-Za-z][A-Za-z0-9]*Error\b/;
128
139
  const isInfraLine = (l) => INFRA_RE.test(l) && !ERROR_CLASS_RE.test(l) && !namesFailure(l) && !SUMMARY_FAIL_RE.test(l);
129
140
  const namesRegression = (l) => (isFailureShaped(l) || ERROR_CLASS_RE.test(l)) && !isInfraLine(l);
141
+ // Infrastructure signatures must enter the fingerprint diff too. Otherwise a signature such as
142
+ // birpc's assertion-free RPC death is classified correctly in the raw output but collapses to the
143
+ // content-free UNRECOGNIZED_FAILURE marker before the fresh-failure path can ask the same classifier.
144
+ // This does not make the vocabulary open-ended: INFRA_RE is still the one closed list, and
145
+ // classifyFailureOutput's ERROR_CLASS_RE/namesFailure vetoes still decide mixed lines.
146
+ const isFingerprintShaped = (l) => isFailureShaped(l) || INFRA_RE.test(l);
130
147
  /**
131
148
  * What a nonzero runner exit is evidence OF. `undefined` when the output names neither — the
132
149
  * unreadable-runner case the existing fail-closed path already owns.
@@ -196,10 +213,10 @@ export function fingerprint(output) {
196
213
  // prefixed form keeps fingerprinting too (baseline-recorded package-level reds stay forgivable).
197
214
  const shaped = [];
198
215
  for (const l of lines) {
199
- if (isFailureShaped(l))
216
+ if (isFingerprintShaped(l))
200
217
  shaped.push(l);
201
218
  const stripped = stripTurboPrefix(l);
202
- if (stripped !== undefined && !PASS_LINE_RE.test(stripped) && isFailureShaped(stripped))
219
+ if (stripped !== undefined && !PASS_LINE_RE.test(stripped) && isFingerprintShaped(stripped))
203
220
  shaped.push(stripped);
204
221
  }
205
222
  if (!shaped.length)
@@ -544,21 +561,6 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
544
561
  continue;
545
562
  }
546
563
  const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
547
- // T9: classify BEFORE the baseline diff, and record it on every nonzero result. An infra-only
548
- // exit means the runner never completed a suite, so there is nothing to forgive and nothing
549
- // verified — it fails, and `meta.infra` marks it so the merge predicate cannot read it as a
550
- // satisfied gate even if some future producer reports it as a pass. Baseline forgiveness stays
551
- // exactly where it belongs: on failures the runner actually reported and the baseline already had.
552
- const classification = classifyFailureOutput(raw);
553
- if (classification === "infra") {
554
- record({
555
- gate: name,
556
- pass: false,
557
- details: `exit ${r.code} on infrastructure alone — the runner never completed a suite, so this gate verified nothing:\n${unrecognizedEvidence(raw) || raw.trim().split("\n").slice(0, 10).join("\n")}`,
558
- meta: { classification, infra: true },
559
- });
560
- continue;
561
- }
562
564
  // OBS-278: only a failure SHAPE is a verdict — everything fingerprint() keeps is one, except the
563
565
  // unrecognized-output marker, which is evidence for the operator and never grounds to reject.
564
566
  // ponytail: ceiling — a runner whose failure output holds no shape above and whose baseline is
@@ -567,6 +569,25 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
567
569
  // that runner's position rule (leading verdict + identifier, or identifier + separator + trailing
568
570
  // verdict); loosening back to vocabulary re-opens OBS-278.
569
571
  const { failing, unreadable } = freshFailures(entry, raw);
572
+ // T9: classify the FRESH diff before charging it. The complete runner output can legitimately
573
+ // contain a baseline-recorded assertion beside a newly introduced infrastructure death; letting
574
+ // that known assertion outvote the fresh birpc line turns machine failure into a worker defect.
575
+ // `classifyFailureOutput` remains the single discriminator. When there is no fresh fingerprint,
576
+ // retain the whole-output read so a repeated infra abort can never be baseline-forgiven as green.
577
+ const freshClassification = failing.length ? classifyFailureOutput(failing.join("\n")) : undefined;
578
+ const classification = freshClassification ?? (!failing.length ? classifyFailureOutput(raw) : undefined);
579
+ if (classification === "infra") {
580
+ const evidence = failing.length
581
+ ? failing.slice(0, 10).join("\n")
582
+ : unrecognizedEvidence(raw) || raw.trim().split("\n").slice(0, 10).join("\n");
583
+ record({
584
+ gate: name,
585
+ pass: false,
586
+ details: `exit ${r.code}; the fresh failures carry infrastructure evidence alone — the runner never completed a suite, so this gate verified nothing:\n${evidence}`,
587
+ meta: { classification, infra: true },
588
+ });
589
+ continue;
590
+ }
570
591
  // OBS-534 (T2): only a recorded VERDICT can be forgiven. A green baseline has no red to forgive,
571
592
  // and neither has a capture that was killed at its ceiling — it recorded a cause instead, so it
572
593
  // fails closed on the same branch rather than reading as "only pre-existing failures". Legacy