tickmarkr 2.5.1 → 2.5.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/dist/adapters/qwen.d.ts +1 -0
  2. package/dist/adapters/qwen.js +8 -0
  3. package/dist/adapters/types.d.ts +1 -0
  4. package/dist/cli/commands/approve.js +1 -1
  5. package/dist/cli/commands/doctor.d.ts +8 -0
  6. package/dist/cli/commands/doctor.js +50 -1
  7. package/dist/cli/commands/init.js +1 -1
  8. package/dist/cli/commands/run.d.ts +5 -0
  9. package/dist/cli/commands/run.js +16 -1
  10. package/dist/cli/commands/status.js +8 -14
  11. package/dist/compile/native.js +3 -0
  12. package/dist/compile/ownership.d.ts +9 -0
  13. package/dist/compile/ownership.js +65 -2
  14. package/dist/config/config.d.ts +6 -0
  15. package/dist/config/config.js +6 -0
  16. package/dist/drivers/subprocess.d.ts +2 -3
  17. package/dist/drivers/subprocess.js +2 -3
  18. package/dist/gates/baseline.d.ts +13 -0
  19. package/dist/gates/baseline.js +99 -30
  20. package/dist/gates/llm.d.ts +1 -1
  21. package/dist/gates/llm.js +35 -13
  22. package/dist/gates/review.d.ts +11 -0
  23. package/dist/gates/review.js +67 -18
  24. package/dist/gates/run-gates.js +2 -2
  25. package/dist/run/daemon.d.ts +8 -3
  26. package/dist/run/daemon.js +3198 -2934
  27. package/dist/run/git.d.ts +4 -2
  28. package/dist/run/git.js +11 -23
  29. package/dist/run/journal.js +16 -7
  30. package/dist/run/merge.d.ts +1 -1
  31. package/dist/run/merge.js +126 -18
  32. package/dist/run/operator-state.d.ts +1 -1
  33. package/dist/run/operator-state.js +5 -9
  34. package/package.json +1 -1
  35. package/skills/tickmarkr-auto/SKILL.md +1 -1
  36. package/skills/tickmarkr-loop/SKILL.md +1 -1
  37. package/skills/tickmarkr-overseer/SKILL.md +61 -4
  38. package/skills/tickmarkr-overseer/scripts/watch-context.sh +5 -2
  39. package/skills/tickmarkr-overseer/scripts/watch-launch.sh +38 -0
package/dist/run/git.d.ts CHANGED
@@ -4,6 +4,8 @@ export { ROUTING_ENV_SEAMS };
4
4
  export declare const FORK_CAP_ENV = "VITEST_MAX_FORKS";
5
5
  export declare const SUITE_PARENT_ENV = "TICKMARKR_SUITE_PARENT";
6
6
  export declare const DEFAULT_FORK_CAP = "6";
7
+ /** Verification owns the serialized suite window; workers keep their concurrency budget. */
8
+ export declare const runWithVerificationBudget: <T>(capacity: RunCapacity, fn: () => Promise<T>) => Promise<T>;
7
9
  /**
8
10
  * OBS-618: how many PROCESSES one vitest fork can hold at its peak — the fork, a daemon it spawns,
9
11
  * and that daemon's own `git` child. The previous formula counted FORKS and assumed one process
@@ -38,7 +40,7 @@ export declare const resolvedForkCap: () => string;
38
40
  /**
39
41
  * T7: the CAPACITY a suite verdict was measured under — the fork cap the command's child actually
40
42
  * received, and the core count that cap was divided from. Two verdicts are comparable only when both
41
- * numbers match: a run resumed at a different concurrency divides the same machine by a different
43
+ * numbers match: a run resumed at a different capacity divides the machine by a different
42
44
  * number, so a green measured in that other world is not evidence about this one.
43
45
  *
44
46
  * This pair is the WHOLE comparable identity, and the load averages a gate row already carries beside
@@ -77,7 +79,7 @@ export declare function sameCapacity(recorded: unknown, current: RunCapacity | u
77
79
  export declare const describeCapacity: (value: unknown) => string;
78
80
  /**
79
81
  * The capacity a child spawned on THIS async context would receive: the same precedence `shell`
80
- * applies below — an operator export of the cap wins over the run's own derived value — beside the
82
+ * applies below — the verification budget captures any operator export at run start — beside the
81
83
  * cores it was divided from. A caller holding a command's own result reads the capacity off THAT
82
84
  * result (`ShResult.capacity`, stamped where the child's environment was built); this is for the
83
85
  * decisions taken BEFORE any child exists — a cache hit, a reuse predicate.
package/dist/run/git.js CHANGED
@@ -15,24 +15,11 @@ export { ROUTING_ENV_SEAMS };
15
15
  export const FORK_CAP_ENV = "VITEST_MAX_FORKS";
16
16
  export const SUITE_PARENT_ENV = "TICKMARKR_SUITE_PARENT";
17
17
  export const DEFAULT_FORK_CAP = "6";
18
- /**
19
- * The fork budget belongs to ONE run: how many gate suites can be in flight at once is exactly that
20
- * run's resolved concurrency, so the per-suite cap must divide the machine by THAT number and no
21
- * other. Two things rule out a module-level variable. A single Node process holds more than one
22
- * runDaemon call — the suites do it, and so does a supervisor driving two repositories — so a
23
- * captured-once global hands whichever run started first its cap to every other run, and only a
24
- * reset seam production never calls could hide it. And re-deriving the number at spawn time reads
25
- * whatever `process.argv` or the config overlay says NOW, not what the run resolved: `parseArgs`
26
- * settles `--concurrency 2 --concurrency 8` on the LAST occurrence, and an overlay is a mutable
27
- * file, so a re-derived cap can disagree with the concurrency the run is actually enforcing.
28
- *
29
- * AsyncLocalStorage is the stdlib answer to both. The store is entered once, around the run body,
30
- * from the single value `runDaemon` resolved; every shell that run spawns — baseline capture, gate
31
- * batteries, tip verify, worker environments — inherits it through the async context, and a
32
- * concurrent run's shells inherit their own. There is nothing to reset: leaving the run leaves
33
- * the store, so sequential runs cannot inherit each other either.
34
- */
18
+ /** Worker concurrency is resolved once per run and isolated from other async runs. */
35
19
  const forkBudget = new AsyncLocalStorage();
20
+ const verificationBudget = new AsyncLocalStorage();
21
+ /** Verification owns the serialized suite window; workers keep their concurrency budget. */
22
+ export const runWithVerificationBudget = (capacity, fn) => verificationBudget.run(capacity, fn);
36
23
  /**
37
24
  * OBS-618: how many PROCESSES one vitest fork can hold at its peak — the fork, a daemon it spawns,
38
25
  * and that daemon's own `git` child. The previous formula counted FORKS and assumed one process
@@ -102,12 +89,12 @@ export const describeCapacity = (value) => {
102
89
  };
103
90
  /**
104
91
  * The capacity a child spawned on THIS async context would receive: the same precedence `shell`
105
- * applies below — an operator export of the cap wins over the run's own derived value — beside the
92
+ * applies below — the verification budget captures any operator export at run start — beside the
106
93
  * cores it was divided from. A caller holding a command's own result reads the capacity off THAT
107
94
  * result (`ShResult.capacity`, stamped where the child's environment was built); this is for the
108
95
  * decisions taken BEFORE any child exists — a cache hit, a reuse predicate.
109
96
  */
110
- export const resolvedCapacity = () => ({
97
+ export const resolvedCapacity = () => verificationBudget.getStore() ?? ({
111
98
  forkCap: Number(FORK_CAP_ENV in process.env ? process.env[FORK_CAP_ENV] : resolvedForkCap()),
112
99
  cores: availableParallelism(),
113
100
  });
@@ -151,9 +138,10 @@ function shell(cmd, cwd, timeoutMs, login) {
151
138
  const env = { ...process.env };
152
139
  for (const k of ROUTING_ENV_SEAMS)
153
140
  delete env[k];
154
- // OBS-110: apply the run's own fork cap only when the operator has not already set one.
155
- if (!(FORK_CAP_ENV in env))
156
- env[FORK_CAP_ENV] = resolvedForkCap();
141
+ // A run freezes the operator override and cores at startup; admission can lower the round cap.
142
+ env[FORK_CAP_ENV] = verificationBudget.getStore()
143
+ ? String(resolvedCapacity().forkCap)
144
+ : env[FORK_CAP_ENV] ?? resolvedForkCap();
157
145
  // OBS-854: descendants can leave the checkout (nested fixture suites do), so cwd alone cannot
158
146
  // attribute them. Every daemon shell exports the daemon pid as their durable parentage marker.
159
147
  env[SUITE_PARENT_ENV] = String(process.pid);
@@ -163,7 +151,7 @@ function shell(cmd, cwd, timeoutMs, login) {
163
151
  // for; re-deriving the run's own budget after the command returned would stamp a cap no child ran
164
152
  // under. `Number` of an unparseable export is NaN, which every reader treats as malformed and
165
153
  // therefore fails closed — the honest direction when the cap in play cannot be stated.
166
- const capacity = { forkCap: Number(env[FORK_CAP_ENV]), cores: availableParallelism() };
154
+ const capacity = { forkCap: Number(env[FORK_CAP_ENV]), cores: resolvedCapacity().cores };
167
155
  const attempt = () => new Promise((resolve) => {
168
156
  const startedAt = Date.now();
169
157
  // detached: bash gets its own process group so a timeout can kill the whole tree —
@@ -949,18 +949,27 @@ export function gateResultJournalData(gate, pass, details, meta = {}) {
949
949
  const signalBasis = deriveSignalBasis(gate, pass, details, meta);
950
950
  return { gate, pass, details, ...meta, signalBasis, signalQuality: signalQualityFromBasis(signalBasis) };
951
951
  }
952
- // T3 (Sol #2 / Fable F2): one canonical engagement identity, shared by status AND resume. The run-start
953
- // event records graphDefinitionHash (over compiled task definitions only — see graph.graphDefinitionHash);
954
- // this is the single field both consumers read, and the single comparator below is the single place the
955
- // journal↔graph join is decided. unbound (no recorded definition hash, e.g. a pre-v1.44 journal) and
952
+ // T3 (Sol #2 / Fable F2) + OBS-978: one canonical engagement identity, shared by status, plan, the operator
953
+ // fold AND resume. The run-start event records graphDefinitionHash (over compiled task definitions only — see
954
+ // graph.graphDefinitionHash); each audited graph-rehash row (resume --graph-changed) then moves the identity to
955
+ // its `to`, so the recorded hash is the last audited rehash, else run-start. A row is audited when its `from`
956
+ // names the identity it replaced, or the run-start one (all pre-OBS-978 daemons wrote); a row auditing neither
957
+ // binds nothing — the journal is unbound until a release from null. unbound (also a pre-v1.44 journal) and
956
958
  // mismatch are both not-comparable — status renders the notice either way; resume refuses either way and
957
959
  // distinguishes the reason only for its message and the --graph-changed release event.
958
960
  export function recordedGraphDefinitionHash(events) {
961
+ const start = events.find((e) => e.event === "run-start");
962
+ if (!start)
963
+ return undefined;
964
+ const origin = typeof start.data.graphDefinitionHash === "string" ? start.data.graphDefinitionHash : null;
965
+ let recorded = origin;
959
966
  for (const e of events) {
960
- if (e.event === "run-start" && typeof e.data.graphDefinitionHash === "string")
961
- return e.data.graphDefinitionHash;
967
+ if (e.event !== "graph-rehash")
968
+ continue;
969
+ const audited = e.data.from === recorded || e.data.from === origin;
970
+ recorded = audited && typeof e.data.to === "string" ? e.data.to : null;
962
971
  }
963
- return undefined;
972
+ return recorded ?? undefined;
964
973
  }
965
974
  // THE shared comparator (criterion: status and resume decide through one comparator). status reads
966
975
  // .comparable; resume reads .comparable plus .reason/.recorded for its refusal message and the release.
@@ -12,7 +12,7 @@ export interface TipVerifyResult {
12
12
  forgiven?: boolean;
13
13
  /**
14
14
  * OBS-534: what a nonzero exit is evidence OF, taken from the battery's own readers — `ceilingKillResult`
15
- * for a kill, `classifyFailureOutput` for everything else. `infra` means nothing was verified.
15
+ * for a kill, the shared runner classifier for everything else. `infra` means nothing was verified.
16
16
  */
17
17
  cause?: FailureClassification;
18
18
  }
package/dist/run/merge.js CHANGED
@@ -1,7 +1,7 @@
1
- import { existsSync, writeFileSync } from "node:fs";
1
+ import { existsSync, readFileSync, writeFileSync } from "node:fs";
2
2
  import { join } from "node:path";
3
3
  import { shq } from "../adapters/types.js";
4
- import { ceilingKillResult, classifyFailureOutput, effectiveCeilingMs, fingerprint, freshFailures, } from "../gates/baseline.js";
4
+ import { ceilingKillResult, classifyFreshRunnerOutput, classifyRunnerOutput, fileCountDeficit, waitForCalmWindow, effectiveCeilingMs, fingerprint, freshFailures, } from "../gates/baseline.js";
5
5
  import { tickmarkrDir } from "../graph/graph.js";
6
6
  import { describeCapacity, gitHead, linkNodeModules, resolveIntegrationBranch, sameCapacity, sh, shGit, shGitOk, WORKTREES_DIR } from "./git.js";
7
7
  export function integrationBranch(cfg, runId) {
@@ -52,16 +52,121 @@ export async function mergeTask(intWt, taskBranch, message, gatedCommit) {
52
52
  // fingerprint, or output with no recognizable failure shape, → failed.
53
53
  export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
54
54
  const results = [];
55
+ let runStartCommands;
56
+ let journalFound = false;
57
+ let hasRunEvidenceOrMalformed = false;
58
+ const journalPath = join(runDir, "journal.jsonl");
59
+ if (existsSync(journalPath)) {
60
+ journalFound = true;
61
+ try {
62
+ const raw = readFileSync(journalPath, "utf8").trim();
63
+ if (!raw) {
64
+ hasRunEvidenceOrMalformed = true;
65
+ }
66
+ else {
67
+ const lines = raw.split("\n");
68
+ for (const line of lines) {
69
+ if (!line.trim())
70
+ continue;
71
+ try {
72
+ const parsed = JSON.parse(line);
73
+ if (parsed.event === "run-start" && parsed.data?.commands) {
74
+ runStartCommands = parsed.data.commands;
75
+ break;
76
+ }
77
+ if (parsed.event && !parsed.event.startsWith("tip-verify")) {
78
+ hasRunEvidenceOrMalformed = true;
79
+ }
80
+ }
81
+ catch {
82
+ // Recover valid rows: skip malformed rows while searching for run-start
83
+ hasRunEvidenceOrMalformed = true;
84
+ }
85
+ }
86
+ }
87
+ }
88
+ catch {
89
+ hasRunEvidenceOrMalformed = true;
90
+ }
91
+ }
92
+ const gatesToRun = [];
55
93
  for (const [gate, cmd] of Object.entries(commands)) {
56
- const entry = baseline?.commands[gate];
94
+ if (gate === "tipTest")
95
+ continue;
96
+ if (gate === "test") {
97
+ gatesToRun.push(["test", commands.tipTest ?? cmd]);
98
+ }
99
+ else {
100
+ gatesToRun.push([gate, cmd]);
101
+ }
102
+ }
103
+ if (!commands.test && commands.tipTest) {
104
+ gatesToRun.push(["test", commands.tipTest]);
105
+ }
106
+ if (journalFound && hasRunEvidenceOrMalformed && runStartCommands === undefined) {
107
+ for (const [gate, cmd] of gatesToRun) {
108
+ const artifact = join(runDir, `tip-verify-${gate}.log`);
109
+ const details = `tip verify could not establish command provenance: run-start evidence missing or malformed in journal`;
110
+ writeFileSync(artifact, details + "\n");
111
+ results.push({
112
+ gate,
113
+ cmd,
114
+ pass: false,
115
+ exitCode: 1,
116
+ fingerprints: [],
117
+ details,
118
+ artifact,
119
+ });
120
+ }
121
+ return results;
122
+ }
123
+ for (const [gate, cmd] of gatesToRun) {
124
+ if (runStartCommands !== undefined) {
125
+ const expectedCmd = gate === "test"
126
+ ? (runStartCommands.tipTest ?? runStartCommands.test)
127
+ : runStartCommands[gate];
128
+ if (expectedCmd === undefined || cmd !== expectedCmd) {
129
+ const artifact = join(runDir, `tip-verify-${gate}.log`);
130
+ const details = `tip verify command "${cmd}" for gate "${gate}" was not named in run-start row`;
131
+ writeFileSync(artifact, details + "\n");
132
+ results.push({
133
+ gate,
134
+ cmd,
135
+ pass: false,
136
+ exitCode: 1,
137
+ fingerprints: [],
138
+ details,
139
+ artifact,
140
+ });
141
+ continue;
142
+ }
143
+ }
144
+ // Leg-2 T3 M1 (RULING-230-15): the entry is chosen by the RECORDED identity — the same provenance the
145
+ // command was validated against above — never the current config. After a resume whose config task
146
+ // command converged on the recorded tip command, the current-config test read "no distinct tip" and
147
+ // forgave a tip red against the TASK entry. Only standalone verify (no run-start row) has no record.
148
+ const identity = runStartCommands ?? commands;
149
+ const hasTipCommand = Boolean(identity.tipTest && identity.tipTest !== identity.test);
150
+ const entry = gate === "test" && (hasTipCommand || (!identity.test && identity.tipTest))
151
+ ? baseline?.commands.tipTest
152
+ : baseline?.commands[gate];
57
153
  // OBS-534: the ceiling is the BATTERY's, derived by effectiveCeilingMs from the same baseline entry
58
154
  // this loop already reads for forgiveness two lines down — never the flat DEFAULT_SHELL_TIMEOUT_MS
59
155
  // `sh` defaults to. A suite whose capture measured 600007ms carries a recorded 1800021ms ceiling;
60
156
  // running it under 600000ms here SIGKILLed a green tip three times while every per-task gate passed.
61
157
  const ceilingMs = effectiveCeilingMs(entry);
62
- const r = await sh(cmd, intWt, ceilingMs);
63
- const raw = r.stdout + "\n" + r.stderr;
64
- const stripped = raw.split(intWt).join("");
158
+ let r = await sh(cmd, intWt, ceilingMs);
159
+ let raw = r.stdout + "\n" + r.stderr;
160
+ let stripped = raw.split(intWt).join("");
161
+ let rerun;
162
+ if (gate === "test" && !r.timedOut && !fileCountDeficit(entry, stripped)
163
+ && classifyFreshRunnerOutput(entry, stripped, r.code) === "infra") {
164
+ const waitedMs = await waitForCalmWindow();
165
+ rerun = `runner-infra rerun after waiting ${waitedMs}ms for a calm load window`;
166
+ r = await sh(cmd, intWt, ceilingMs);
167
+ raw = r.stdout + "\n" + r.stderr;
168
+ stripped = raw.split(intWt).join("");
169
+ }
65
170
  const artifact = join(runDir, `tip-verify-${gate}.log`);
66
171
  // Battery parity on the ceiling too (baseline.ts Q24): the kill is read BEFORE the exit code is
67
172
  // interpreted at all. A SIGKILLed battery never returned a verdict, so no line of its partial
@@ -76,14 +181,16 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
76
181
  pass: false,
77
182
  exitCode: r.code,
78
183
  fingerprints: [],
79
- details: killed.details,
184
+ details: `${rerun ? `infra; ${rerun}: ` : ""}${killed.details}`,
80
185
  cause: killed.meta?.classification,
81
186
  artifact,
82
187
  });
83
188
  continue;
84
189
  }
85
190
  const { failing, unreadable } = freshFailures(entry, stripped);
86
- const cause = r.code === 0 ? undefined : classifyFailureOutput(stripped);
191
+ const deficit = fileCountDeficit(entry, stripped);
192
+ const cause = deficit ? "infra" : classifyFreshRunnerOutput(entry, stripped, r.code);
193
+ const greenTeardown = classifyRunnerOutput(stripped, r.code) === "green-teardown";
87
194
  // `?? 1` is the battery's own default (baseline.ts compareToBaseline): an exitCode-less legacy
88
195
  // entry reads as red-at-baseline there, so it must read the same here or old baselines silently
89
196
  // lose forgiveness. OBS-534 (T2): a capture killed at its ceiling now records a CAUSE and no
@@ -106,7 +213,7 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
106
213
  const comparable = sameCapacity(entry?.capacity, r.capacity);
107
214
  const forgiven = r.code !== 0 && baselineRed && failing.length === 0 && !unreadable && cause !== "infra"
108
215
  && comparable;
109
- const pass = r.code === 0 || forgiven;
216
+ const pass = !deficit && (r.code === 0 || greenTeardown || forgiven);
110
217
  if (!pass)
111
218
  writeFileSync(artifact, raw);
112
219
  results.push({
@@ -114,15 +221,16 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
114
221
  cmd,
115
222
  pass,
116
223
  exitCode: r.code,
117
- fingerprints: r.code !== 0 ? fingerprint(stripped) : [],
118
- details: r.code === 0 ? "exit 0"
119
- : forgiven ? `exit ${r.code} but only baseline-recorded failures (forgiven vs baseline)`
120
- : !comparable && baselineRed && failing.length === 0
121
- ? `exit ${r.code}; every failure is baseline-recorded, but that capture ran under `
122
- + `${describeCapacity(entry?.capacity)} and this verification ran under ${describeCapacity(r.capacity)} `
123
- + `— forgiveness across a changed capacity is not evidence`
124
- : `exit ${r.code}`,
125
- ...(forgiven ? { forgiven: true } : {}),
224
+ fingerprints: r.code !== 0 && !greenTeardown ? fingerprint(stripped) : [],
225
+ details: `${cause === "infra" ? "infra; " : ""}${rerun ? `${rerun}: ` : ""}` + (deficit?.replace(/^infra; /, "") ?? (r.code === 0 ? "exit 0"
226
+ : greenTeardown ? `exit ${r.code} after a green suite summary; only the runner's teardown fingerprint followed it`
227
+ : forgiven ? `exit ${r.code} but only baseline-recorded failures (forgiven vs baseline)`
228
+ : !comparable && baselineRed && failing.length === 0
229
+ ? `exit ${r.code}; every failure is baseline-recorded, but that capture ran under `
230
+ + `${describeCapacity(entry?.capacity)} and this verification ran under ${describeCapacity(r.capacity)} `
231
+ + `— forgiveness across a changed capacity is not evidence`
232
+ : `exit ${r.code}`)),
233
+ ...(forgiven && !deficit ? { forgiven: true } : {}),
126
234
  ...(cause ? { cause } : {}),
127
235
  ...(pass ? {} : { artifact }),
128
236
  });
@@ -57,7 +57,7 @@ export declare class OperatorStateFold {
57
57
  private tasks;
58
58
  private start?;
59
59
  private startEvent?;
60
- private latestGraphRehash?;
60
+ private rehashes;
61
61
  private end?;
62
62
  private active;
63
63
  private approved;
@@ -10,7 +10,7 @@ export class OperatorStateFold {
10
10
  tasks = new Map();
11
11
  start;
12
12
  startEvent;
13
- latestGraphRehash;
13
+ rehashes = [];
14
14
  end;
15
15
  active = false;
16
16
  approved = false;
@@ -34,7 +34,7 @@ export class OperatorStateFold {
34
34
  this.tipFailed = false;
35
35
  }
36
36
  if (e.event === "graph-rehash")
37
- this.latestGraphRehash = { ...e, data: { from: e.data.from, to: e.data.to } };
37
+ this.rehashes = [...this.rehashes, { ...e, data: { from: e.data.from, to: e.data.to } }];
38
38
  if (e.event === "tip-verify-failed" || (e.event === "tip-verify" && e.data.pass === false))
39
39
  this.tipFailed = true;
40
40
  if (e.event === "run-end") {
@@ -146,13 +146,9 @@ export class OperatorStateFold {
146
146
  comparableTo(hash) {
147
147
  if (!hash)
148
148
  return false;
149
- const events = [this.startEvent, this.latestGraphRehash].filter((e) => e !== undefined);
150
- const baseline = engagementComparable(events, hash);
151
- if (this.latestGraphRehash) {
152
- const from = baseline.comparable ? baseline.recorded : baseline.reason === "mismatch" ? baseline.recorded : null;
153
- return this.latestGraphRehash.data.to === hash && this.latestGraphRehash.data.from === from;
154
- }
155
- return baseline.comparable;
149
+ // The shared comparator audits the rehash chain; the fold keeps only the rows it reads.
150
+ const events = [this.startEvent, ...this.rehashes].filter((e) => e !== undefined);
151
+ return engagementComparable(events, hash).comparable;
156
152
  }
157
153
  }
158
154
  /** C1/C6 share this pure reader; callers supply the same observation and journal snapshot. */
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "tickmarkr",
3
- "version": "2.5.1",
3
+ "version": "2.5.3",
4
4
  "description": "Spec in, verified work out.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -92,7 +92,7 @@ After sending, **confirm delivery** by reading the target pane and verifying the
92
92
  1. **Prepare** — confirm the target list. Run the [binary preflight](#binary-preflight-before-compile-or-run). Check `git status`, confirm no tickmarkr run is active, and work from a non-main branch.
93
93
  2. **Compile** — run `tickmarkr compile <spec-or-directory>`. Fix source-spec defects instead of editing the generated graph.
94
94
  3. **Plan** — run `tickmarkr plan`. Review routes, capability-floor warnings, and human gates before execution.
95
- 4. **Run** — run `tickmarkr run`. Watch the run journal for its terminal events rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every green clause for you. Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the relevant agent session.
95
+ 4. **Run** — run `tickmarkr run`. A watch ending the seat's turn is no watch: keep a **blocking journal consumer** alive for the run's terminal events — the shipped watcher below, or a foreground `until grep` on the run's terminal events — and ensure it is re-armed at most every twenty minutes. Never rely on a `Monitor`-only wake. Watch the run journal rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every green clause for you. Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the relevant agent session.
96
96
  5. **Verify and consolidate** — continue only after a green run. A run is green when the run-end event exists in the journal, the tip verify is not "failed", and the summary's `failed`, `human`, `blocked` and `pending` buckets are all empty — a run with a parked task is partial, not green. Tickmarkr consolidates accepted work on `tickmarkr/<runId>` and never signs off to the main branch. A human controls any later release merge.
97
97
  6. **Record** — `tickmarkr report <runId> --md` prints Markdown to stdout; redirect explicitly beside the spec (for example `tickmarkr report <runId> --md > feature.record.md`) and commit the execution record when the repository tracks those records.
98
98
  7. **Continue** — move to the next requested target. If a target fails or is parked, stop with the journal evidence rather than silently skipping it.
@@ -88,7 +88,7 @@ When spawning consultants (agents gathering synthesis input for decisions like S
88
88
  1. **Prepare** — start from the requested spec. Run the [binary preflight](#binary-preflight-before-compile-or-run). Check `git status`, confirm no tickmarkr run is active, and work from a non-main branch.
89
89
  2. **Compile** — run `tickmarkr compile <spec>`. Correct compilation errors in the spec, never in the generated graph.
90
90
  3. **Plan** — run `tickmarkr plan`. Review the routing table, capability-floor warnings, and every human gate, including work that each gate blocks.
91
- 4. **Run** — run `tickmarkr run`. Watch the run journal for its terminal events rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every green clause for you. Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the agent session; do not turn them into proxy questions.
91
+ 4. **Run** — run `tickmarkr run`. A watch ending the seat's turn is no watch: keep a **blocking journal consumer** alive for the run's terminal events — the shipped watcher below, or a foreground `until grep` on the run's terminal events — and ensure it is re-armed at most every twenty minutes. Never rely on a `Monitor`-only wake. Watch the run journal rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every green clause for you. Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the agent session; do not turn them into proxy questions.
92
92
  5. **Verify and consolidate** — accept only a green run. A run is green when the run-end event exists in the journal, the tip verify is not "failed", and the summary's `failed`, `human`, `blocked` and `pending` buckets are all empty — a run with a parked task is partial, not green. Tickmarkr consolidates accepted task work on `tickmarkr/<runId>`; it never signs off to the main branch. A human may later merge that integration branch through the repository's normal release process.
93
93
  6. **Record** — `tickmarkr report <runId> --md` prints Markdown to stdout. Redirect it explicitly beside the source spec (for example `tickmarkr report <runId> --md > feature.record.md`) and commit the execution record when the repository tracks those records. Then [stand down](#stand-down-mission-end-and-retirement).
94
94
 
@@ -149,6 +149,14 @@ through brief lineage. **An executor choice nobody made is still an executor cho
149
149
  guidance belongs in the memory file or the shipped docs.
150
150
  4. Arm the watcher and your own supervision beat (Supervision). Report the hierarchy map (pane ids + names) to the user.
151
151
 
152
+ ### Seat-spawn and Leg-2 recipes
153
+
154
+ Every mission to a Claude or Grok seat is delivered only with `herdr pane run <pane> "<message>"` and
155
+ verified by reading the pane back; never use `agent prompt` for mission delivery. Launch a Grok seat with
156
+ `herdr agent start <seat> --kind grok --pane <pane> -- -m grok-4.6`. For Leg-2, a Codex reviewer under
157
+ `workspace-write` must be briefed with an in-worktree verdict path such as
158
+ `<repo>/.tickmarkr/overseer/verdicts/<task>.md`, and its verdict must be written there before it is read.
159
+
152
160
  ## Supervising tickmarkr as the executor — WHO DOES WHAT
153
161
 
154
162
  When the mission runs `/tickmarkr-auto` (tickmarkr dispatches the workers), supervision changes shape —
@@ -204,6 +212,9 @@ journal tail to decide what happens next, or sweeping orphans — you have taken
204
212
  - **The journal is the source of truth**, not panes. Watchers go on `run-end` / `task-human` /
205
213
  `task-failed` / `consult-verdict`; never sleep-poll inside an agent turn. **Never key a watcher on an
206
214
  agent's `done`** — that is turn end and fires the moment a seat finishes acknowledging you.
215
+ A watch ending the seat's turn is no watch: keep a **blocking journal consumer** alive for those terminal
216
+ events — the shipped watcher below, or a foreground `until grep` on the run's terminal events — and
217
+ ensure it is re-armed at most every twenty minutes. Never rely on a `Monitor`-only wake.
207
218
  **All four are covered by one shipped instrument** — `scripts/watch-journal.sh <runs-dir> [poll] [cap]
208
219
  [events-csv]` — which arms on a line baseline, wakes once, and grades a `run-end` against every green
209
220
  clause. `scripts/watch-parks.sh` stays the park-specific wake for THIS seat (it counts parks and speaks
@@ -476,6 +487,52 @@ number — an unmeasured budget is not a small budget.
476
487
  .claude/skills/tickmarkr-overseer/scripts/watch-context.sh overseer <overseer-agent-or-pane> 50 50 <handoff-file>
477
488
  ```
478
489
 
490
+ ### A GO has a deadline — arm `watch-launch.sh` in the same act as the GO
491
+
492
+ A GO that produces no run is a silent failure until someone notices; on 2026-09-11 an orchestrator's codex
493
+ sandbox was rooted at the main repo, the spec worktree was outside its writable roots, it stopped at the
494
+ denial without reporting, the overseer's 10-minute wake expired un-re-armed, and three hours passed.
495
+ Two rules close that hole:
496
+
497
+ - **Every orchestrator seat is sandbox-rooted at the worktree it will run in** (`cd <worktree>` before
498
+ `herdr agent start … --sandbox workspace-write`), and its brief says: *a denied path or refused command
499
+ is reported to the overseer pane within 60 s — never a silent stop.*
500
+ - **The overseer arms the launch watcher in the SAME act as the GO**, with the lock path the run will
501
+ create, and treats `LAUNCH_OVERDUE` as a first-class event (read the orchestrator pane, fix the seat,
502
+ re-issue the GO):
503
+
504
+ ```bash
505
+ .claude/skills/tickmarkr-overseer/scripts/watch-launch.sh <worktree>/.tickmarkr/graph.lock 900 <overseer-pane> &
506
+ ```
507
+
508
+ It prints `LAUNCH_OK` with the lock's contents when the run starts (exit 0) and, past the deadline, delivers
509
+ `LAUNCH OVERDUE …` to the overseer pane AND as an OS notification (exit 3). Any wake you arm yourself with
510
+ a cap (a background `until` loop) must be RE-ARMED on every expiry; an expired wake is not a watch.
511
+
512
+
513
+ ### A GO has a deadline — arm `watch-launch.sh` in the same act as the GO
514
+
515
+ A GO that produces no run is a silent failure until someone notices; on 2026-09-11 an orchestrator's codex
516
+ sandbox was rooted at the main repo, the spec worktree was outside its writable roots, it stopped at the
517
+ denial without reporting, the overseer's 10-minute wake expired un-re-armed, and three hours passed.
518
+ Two rules close that hole:
519
+
520
+ - **Every orchestrator seat is sandbox-rooted at the worktree it will run in** (`cd <worktree>` before
521
+ `herdr agent start … --sandbox workspace-write`), and its brief says: *a denied path or refused command
522
+ is reported to the overseer pane within 60 s — never a silent stop.*
523
+ - **The overseer arms the launch watcher in the SAME act as the GO**, with the lock path the run will
524
+ create, and treats `LAUNCH_OVERDUE` as a first-class event (read the orchestrator pane, fix the seat,
525
+ re-issue the GO):
526
+
527
+ ```bash
528
+ .claude/skills/tickmarkr-overseer/scripts/watch-launch.sh <worktree>/.tickmarkr/graph.lock 900 <overseer-pane> &
529
+ ```
530
+
531
+ It prints `LAUNCH_OK` with the lock's contents when the run starts (exit 0) and, past the deadline, delivers
532
+ `LAUNCH OVERDUE …` to the overseer pane AND as an OS notification (exit 3). Any wake you arm yourself with
533
+ a cap (a background `until` loop) must be RE-ARMED on every expiry; an expired wake is not a watch.
534
+
535
+
479
536
  The first argument chooses the closed per-seat tier (`orchestrator-context` or `overseer-context`),
480
537
  and every beat names the second argument as that tier's seat. The watcher beats only after reading a
481
538
  rendered percentage, keeps beating on the supervision cadence even when its requested poll is slower,
@@ -580,9 +637,9 @@ they are left implicit:
580
637
  Send only when the seat is idle and the ANSI prompt line is empty or dim-only (the Esc/SGR discriminator
581
638
  separates an autosuggest ghost from typed input), then read back activity or an ACK; presence is not
582
639
  delivery. If a stale draft must be replaced, supersede it explicitly with
583
- `agent prompt " <-- disregard … ACTUAL: …"` instead of stacking another instruction behind it.
640
+ `herdr pane run <pane> "<-- disregard … ACTUAL: …"` instead of stacking another instruction behind it.
584
641
  - **A MESSAGE TO A WORKING SEAT IS A QUEUED MESSAGE, AND THE QUEUE DRAINS ONLY AT TURN BOUNDARIES.**
585
- Delivery is not arrival: `agent prompt` to a `working` claude seat lands in its queue (`Press up to
642
+ Delivery is not arrival: a message sent to a `working` claude seat lands in its queue (`Press up to
586
643
  edit queued messages` on the seat's prompt line is the tell) and is READ only when the current turn
587
644
  ends — and with in-process teammates a turn runs 20–40 minutes, so steering latency equals subagent
588
645
  runtime. Measured 2026-08-17/18 (P98 leg 1): a FREEZE HOLD and a checker-release directive stacked
@@ -622,7 +679,7 @@ they are left implicit:
622
679
  - **AGENT NAMES ARE GLOBAL ACROSS WORKSPACES — verify a seat you spawned by PANE ID, never by name.**
623
680
  Names must be unique among live agents *everywhere*, not within your workspace, so another workspace can
624
681
  already hold `opus`, `sol`, `reviewer` or `orch`. When it does, your `agent start` **fails**, your pane
625
- is left a bare shell, and `agent list` / `agent read` / `agent prompt` for that name then resolve to the
682
+ is left a bare shell, and `agent list` / `agent read` for that name then resolve to the
626
683
  **stranger's seat**. Measured 2026-08-06 (OBS-392): a spawn of `fable` collided with a live seat in
627
684
  another workspace; `agent list` reported `fable -> blocked` and it was read as *this* seat coming up
628
685
  blocked. It was an operator research session sitting on a *"Resume full session?"* prompt. One more
@@ -644,7 +701,7 @@ they are left implicit:
644
701
  Re-arm name-keyed watchers in the same act as the rename; file-keyed artifact watchers are
645
702
  unaffected (one more reason to prefer them).
646
703
  - Stale typed input is unclearable via CLI — supersede it:
647
- `pane run "<-- disregard everything before this arrow (stale draft). ACTUAL: <message>"`.
704
+ `herdr pane run <pane> "<-- disregard everything before this arrow (stale draft). ACTUAL: <message>"`.
648
705
  **But DISCRIMINATE before you supersede or file it: text on an idle seat's prompt line has FOUR
649
706
  authors** — the seat's own draft, an operator, another agent's `agent send` (writes WITHOUT Enter),
650
707
  and claude-code's AUTOSUGGEST, which renders context-plausible ghost text BYTE-IDENTICAL to a typed
@@ -220,17 +220,20 @@ banner_pct() {
220
220
  stripped = bare; gsub(rule, "", stripped)
221
221
  if (bare != "" && stripped == "") last = NR }
222
222
  END { labelled = 0
223
- for (i = 1; i <= NR; i++) if (line[i] ~ /Context [0-9]+% used/) { print line[i]; labelled = 1 }
223
+ for (i = 1; i <= NR; i++) if (line[i] ~ /Context [0-9]+% used/ || line[i] ~ /ctx [0-9]+%/) { print line[i]; labelled = 1 }
224
224
  if (!labelled && last) for (i = last + 1; i <= NR; i++) if (line[i] ~ /[0-9]+%/) print line[i] }
225
225
  ' | tail -1)
226
226
  # No rule line in the window, or no percentage below it: say so out loud rather than guess.
227
227
  [ -n "$banner" ] || { printf 'UNREADABLE\n'; return 0; }
228
+ # OBS-964 add.6: a claude-code statusline plugin (Graft) can replace the model row with a self-labelled
229
+ # `ctx N%` row; the label IS the selector, so no model name is required on that row.
230
+ printf '%s\n' "$banner" | grep -Eq 'ctx [0-9]+%' ||
228
231
  printf '%s\n' "$banner" |
229
232
  grep -Eqi '(^|[^[:alnum:]])(claude|opus|sonnet|haiku|fable|gpt|gemini|glm|kimi|grok|composer|openai|zai)[[:alnum:]_./-]*([[:space:]]|$)' ||
230
233
  { printf 'UNREADABLE\n'; return 0; }
231
234
  # OBS-964: a codex banner carries TWO percentages ("Context 16% used · weekly 96% left"); the LAST one is the
232
235
  # quota, not the fill. Prefer the labelled context figure; fall back to the last bare % (claude-code banner).
233
- pct=$(printf '%s\n' "$banner" | grep -oE 'Context [0-9]+% used' | grep -oE '[0-9]+' | head -1)
236
+ pct=$(printf '%s\n' "$banner" | grep -oE '(Context [0-9]+% used|ctx [0-9]+%)' | grep -oE '[0-9]+' | head -1)
234
237
  # OBS-964 add.3: a codex footer while WORKING drops the labelled figure but keeps `weekly N% left`; strip the
235
238
  # quota phrase before the bare-% fallback so a quota can never be read as fill.
236
239
  [ -n "$pct" ] || pct=$(printf '%s\n' "$banner" | sed -E 's/weekly [0-9]+% left//g' | grep -oE '[0-9]+%' | tail -1 | tr -d '%')
@@ -0,0 +1,38 @@
1
+ #!/usr/bin/env bash
2
+ # watch-launch.sh — a GO that produced no run is a silent failure until someone notices. This watcher
3
+ # notices. Arm it in the SAME act as the GO (orchestrator briefed to compile → plan → run) and it waits
4
+ # for the run's lock; when the lock has not appeared by the deadline it delivers LAUNCH OVERDUE to the
5
+ # overseer's pane AND as an OS notification, so the wake reaches a seat instead of a log nobody reads.
6
+ #
7
+ # Why it exists (2026-09-11): an orchestrator's codex sandbox was rooted at the main repo, the spec
8
+ # worktree was outside its writable roots, it stopped at the denial without reporting, and the overseer's
9
+ # own 10-minute wake expired un-re-armed. Three hours passed before anyone looked. A launch has a
10
+ # deadline; silence past it is the event.
11
+ #
12
+ # usage: watch-launch.sh <lock-path> <deadline-s> <overseer-pane> [poll-s]
13
+ # <lock-path> the run's .tickmarkr/graph.lock in the worktree the run will be launched in
14
+ # <deadline-s> seconds from now by which the lock must exist (a compile+plan+launch takes minutes,
15
+ # never hours; 900 is a generous default for a 7-task spec)
16
+ # <overseer-pane> the pane that must hear about it (herdr pane id), e.g. wZ:p18S
17
+ # [poll-s] poll interval, default 15
18
+ # exit 0 LAUNCH_OK (lock seen; prints its contents) · exit 3 LAUNCH_OVERDUE (delivered) · exit 64 usage
19
+ set -u
20
+ LOCK="${1:-}"; DEADLINE="${2:-}"; PANE="${3:-}"; POLL="${4:-15}"
21
+ [ -n "$LOCK" ] && [ -n "$DEADLINE" ] && [ -n "$PANE" ] || { echo "usage: watch-launch.sh <lock-path> <deadline-s> <overseer-pane> [poll-s]" >&2; exit 64; }
22
+ start=$(date +%s)
23
+ while :; do
24
+ if [ -f "$LOCK" ]; then
25
+ printf 'LAUNCH_OK %s %s\n' "$(date -u +%H:%M:%SZ)" "$(cat "$LOCK" 2>/dev/null | tr -d '\n')"
26
+ exit 0
27
+ fi
28
+ now=$(date +%s)
29
+ if [ $((now - start)) -ge "$DEADLINE" ]; then
30
+ msg="LAUNCH OVERDUE $(date -u +%H:%M:%SZ): no lock at $LOCK after ${DEADLINE}s — read the orchestrator pane NOW (sandbox denial? preflight refusal? unsubmitted GO?)"
31
+ echo "LAUNCH_OVERDUE $msg"
32
+ # Both deliveries, always: a pane the overseer reads AND a notification the operator sees.
33
+ herdr pane run "$PANE" "$msg" >/dev/null 2>&1 || echo " (pane delivery failed — the notification is the only path)"
34
+ herdr notification show "$msg" >/dev/null 2>&1 || true
35
+ exit 3
36
+ fi
37
+ sleep "$POLL"
38
+ done