tickmarkr 2.1.0 → 2.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -105,7 +105,7 @@ Outside multi-agent environments, run the loop directly.
105
105
 
106
106
  ### Version preflight
107
107
 
108
- Before \`tickmarkr compile\` or \`tickmarkr run\`: run \`tickmarkr version\`, read \`package.json\` version, and if the binary is older on major.minor, stop and tell the operator to update. Never proceed on hope — stale binaries silently skip daemon gates. Also verify no run is live before starting one: \`pgrep -f "tickmarkr (run|resume)"\` must be empty — match the process, not one install path (\`dist/cli/index.js\` alone misses global and homebrew installs), and treat a held \`.tickmarkr/graph.lock\` as a live run until its holder pid is proven dead.
108
+ Before \`tickmarkr compile\` or \`tickmarkr run\`: run \`tickmarkr version\`, read \`package.json\` version, and stop if the versions differ anywhere (major, minor, or patch); the binary and repository must agree on the entire version, so binary \`2.1.0\` versus repository \`2.1.1\` is a stop. Tell the operator to update or link the correct binary. Never proceed on hope — stale binaries silently skip daemon gates. Also verify no run is live in this repository before starting one: lead with this repository's \`.tickmarkr/graph.lock\`, read its holder pid, and treat it as live until \`kill -0 <pid>\` proves that holder dead. Never require a machine-wide process pattern to be empty: a lawful run in another repository — or the probing shell's own argv — can match. If you use a process probe as secondary evidence, exclude the probing process, resolve every candidate's own cwd (for example with \`lsof -a -p <pid> -d cwd\`), and count only candidates whose cwd is this repository root.
109
109
 
110
110
  ### Tip-verify-before-green
111
111
 
@@ -527,10 +527,34 @@ const foldTaskEffort = (tasks, events) => {
527
527
  return [...effort.values()].sort((left, right) => right.total - left.total
528
528
  || order.get(left.taskId) - order.get(right.taskId));
529
529
  };
530
- // Attention (review) and failure (park) wear the ONE amethyst, so the bar segments and their legend
531
- // markers carry the distinction in SHAPE: a full block for dispatch, light shade for a review round,
532
- // dark shade for a human park. Same display width, so the fitted bar keeps its measured columns.
533
- const EFFORT_GLYPH = { dispatch: "\u2588", review: "\u2592", park: "\u2593" };
530
+ // ONE glyph for every segment, distinction by COLOUR. This was shade-based — full block for dispatch,
531
+ // light shade for a review round, dark shade for a human park — because attention and failure share the
532
+ // ONE amethyst and shape was their only discriminator. OPERATOR 2026-08-25, with a screenshot: the rows
533
+ // rendered at DIFFERENT HEIGHTS. A row carrying the shade glyphs pulls a fallback font whose cell box is
534
+ // taller, so the whole line grows and its full blocks stretch with it — T1 (dispatch+review+park) stood
535
+ // visibly taller than T3 (dispatch only). Same display WIDTH was verified and preserved; nobody checked
536
+ // height, and the fitted-column arithmetic cannot see it.
537
+ //
538
+ // So review moves off amethyst onto `information` (cornflower) — an EXISTING token. Deliberately not a
539
+ // new amber: brand.ts calls the five live colours operator-approved and says "no token introduces a
540
+ // sixth colour", and a bar chart is not the place to spend that.
541
+ //
542
+ // The shape scheme existed so a reader WITHOUT colour could still separate a review round from a human
543
+ // park. That reader does not exist on this panel. `renderFrame` computes `const unicode = visual()`
544
+ // (`visual()` = `isTTY === true && NO_COLOR === undefined`) and, when it is false, EARLY-RETURNS the
545
+ // ASCII machine/CI surface at the `if (!unicode)` branch — which sits ABOVE the `effortPanel(...)` call
546
+ // below, so the panel is never built on that path. Measured before it was explained: a frame rendered
547
+ // with isTTY=true and NO_COLOR=1 contains no "WHERE THE EFFORT WENT" and no block glyph at all.
548
+ // (The `--watch` loop has its own `visual()` gate; that one is NOT what suppresses this panel, and the
549
+ // non-watch path reaches renderFrame unconditionally.) So the shades were paying an accessibility cost
550
+ // on a surface that cannot render colourless, and a conditional to restore them under NO_COLOR would
551
+ // have been an unreachable branch (the v1.80 injected-clock lesson: the fix for that is deletion).
552
+ //
553
+ // One glyph, three colours. Review moves off amethyst onto `information` (cornflower) — an EXISTING
554
+ // token, deliberately not a new amber: brand.ts calls the five live colours operator-approved and says
555
+ // "no token introduces a sixth colour". Every row now draws the same block, which is what makes the
556
+ // rows equal height.
557
+ const EFFORT_GLYPH = { dispatch: "\u2588", review: "\u2588", park: "\u2588" };
534
558
  /**
535
559
  * Prototype panel, fitted by the cockpit's display-cell authority before board-wide wrapping.
536
560
  *
@@ -557,7 +581,7 @@ const effortPanel = (tasks, events, columns) => {
557
581
  const reviewEnd = Math.round(((task.dispatches + task.reviews) / maxTotal) * barColumns);
558
582
  const parkEnd = Math.round((task.total / maxTotal) * barColumns);
559
583
  const stack = ok(EFFORT_GLYPH.dispatch.repeat(dispatchEnd))
560
- + warn(EFFORT_GLYPH.review.repeat(Math.max(0, reviewEnd - dispatchEnd)))
584
+ + information(EFFORT_GLYPH.review.repeat(Math.max(0, reviewEnd - dispatchEnd)))
561
585
  + fail(EFFORT_GLYPH.park.repeat(Math.max(0, parkEnd - reviewEnd)));
562
586
  return `${prefix}${fitCells(stack, barColumns)} ${counts}`;
563
587
  });
@@ -567,7 +591,7 @@ const effortPanel = (tasks, events, columns) => {
567
591
  "",
568
592
  ...rows,
569
593
  "",
570
- ` ${ok(EFFORT_GLYPH.dispatch)} ${dim("dispatch")} ${warn(EFFORT_GLYPH.review)} ${dim("review round")} ${fail(EFFORT_GLYPH.park)} ${dim("human park")}`,
594
+ ` ${ok(EFFORT_GLYPH.dispatch)} ${dim("dispatch")} ${information(EFFORT_GLYPH.review)} ${dim("review round")} ${fail(EFFORT_GLYPH.park)} ${dim("human park")}`,
571
595
  ];
572
596
  };
573
597
  // VIS-11 (v1.13): a liveness header for renderFrame — last journal event age + whether the recorded
@@ -104,7 +104,15 @@ export async function verify(argv, cwd = process.cwd()) {
104
104
  }
105
105
  const wantAcceptance = acceptance.length > 0 && !values["no-acceptance"];
106
106
  const wantReview = !values["no-review"];
107
- const gates = GATE_NAMES.filter((g) => (g !== "acceptance" || wantAcceptance) && (g !== "review" || wantReview));
107
+ // A gate that enforced nothing must not print a green row. `files` has exactly three sources —
108
+ // explicit --files, a compiled task's own files[], or nothing — and only the third leaves the
109
+ // allowlist empty, where scopeGate passes as "no file scope declared — unrestricted"
110
+ // (scope.ts:41). An honest `details` string is no defence: the ROW is what gets quoted, and
111
+ // quoting it launders a check that never ran. So scope filters on availability exactly as
112
+ // acceptance and review do below — the report omits the gate rather than crediting one that
113
+ // gated no allowlist. (Narrowing the empty allowlist to the changed set is the same green row by
114
+ // another mechanism, and hides the same fact.)
115
+ const gates = GATE_NAMES.filter((g) => (g !== "acceptance" || wantAcceptance) && (g !== "review" || wantReview) && (g !== "scope" || files.length > 0));
108
116
  const task = {
109
117
  id: "VERIFY", title: "standalone verification", goal, shape: "implement", complexity: 5,
110
118
  deps: [], files, context: [], acceptance: acceptance.length ? acceptance : ["(deterministic verification only)"],
@@ -23,17 +23,17 @@ export declare class DeliveryCorruptedError extends Error {
23
23
  export type DriverJournal = (event: string, slotName: string, data: Record<string, unknown>) => void;
24
24
  /** First-generation join direction from measured trailer-safe floor (43-MEASUREMENT.md). */
25
25
  export declare function workerSplitDirection(paneCols: number | null, safeFloor?: number, margin?: number): "right" | "down";
26
- export declare const BOARD_HEIGHT_SHARE = 0.72;
26
+ export declare const BOARD_WIDTH_SHARE = 0.5;
27
27
  export interface BoardPlacement {
28
- /** Always down: the split is vertical, so the board can own the caller's FULL width. */
29
- direction: "down";
30
- /** herdr's split ratio is the FIRST child's share, and a down split's first child is the TOP
31
- * region — the region the board occupies once it is swapped above the caller. */
28
+ /** Always right: the board sits BESIDE the caller, so the narration keeps a column of its own. */
29
+ direction: "right";
30
+ /** herdr's split ratio is the FIRST child's share, and a right split's first child is the LEFT
31
+ * region — the caller's, i.e. the narration. The board takes the remainder, on the right.
32
+ * Measured against a live herdr 2026-08-25: a 256-col caller split right at 0.5 leaves the caller
33
+ * at x=36 w=128 and puts the NEW pane at x=164 w=128. */
32
34
  ratio: number;
33
- /** The new pane is swapped ABOVE the caller; the split alone would leave the board underneath. */
34
- swap: "above";
35
35
  }
36
- /** The single approved vertical-stack record. The caller's columns are accepted and deliberately
36
+ /** The single approved side-by-side record. The caller's columns are accepted and deliberately
37
37
  * ignored: this signature is where width used to decide the arrangement, and the parameter stays
38
38
  * so that "the plan does not depend on it" is a property a caller (and a test) can exercise. */
39
39
  export declare function boardSplitPlan(_callerCols?: number | null): BoardPlacement;
@@ -73,15 +73,27 @@ export function workerSplitDirection(paneCols, safeFloor = TRAILER_SAFE_FLOOR_CO
73
73
  // placement any more. Every width-derived variant of this placement has been wrong in the operator's
74
74
  // tab: the halving floor sent a 189-column board below the seat (2026-08-18), and the width-first
75
75
  // side split that replaced it puts the board and the narration shoulder to shoulder when the board is
76
- // the surface the operator reads and the narration is the rail beneath it. The placement is now ONE
77
- // record — the board stacked ABOVE the caller at full width, taking 72% of the height — and it is
78
- // invariant: no terminal width, measured or unmeasurable, can select a different arrangement.
79
- export const BOARD_HEIGHT_SHARE = 0.72; // board 72 / narration 28, the operator's stack
80
- /** The single approved vertical-stack record. The caller's columns are accepted and deliberately
76
+ // the surface the operator reads and the narration is the rail beneath it. The placement is ONE
77
+ // record and it is invariant: no terminal width, measured or unmeasurable, can select a different
78
+ // arrangement. THAT invariance is the hard-won part and it is unchanged here.
79
+ //
80
+ // What the record SAYS changed on operator instruction (2026-08-25): the board sits to the RIGHT of
81
+ // the caller, not stacked above it. The vertical stack gave the board 72% of the HEIGHT at full
82
+ // width, and in the operator's own tab that was mostly empty board over a squeezed narration rail —
83
+ // a task table is a handful of rows, so height was the axis it did not need and the narration did.
84
+ // Side by side spends the axis the board actually uses.
85
+ //
86
+ // ⚠ This is NOT a return to the width-derived placement that was wrong twice. Those variants let the
87
+ // MEASURED width choose the arrangement, so the same run rendered differently in different terminals
88
+ // and neither operator nor test could name one expected geometry. This record is constant: every
89
+ // caller width gets `right`, and `boardSplitPlan` still ignores the columns it is handed. The
90
+ // invariance test's width-sensitive control still fails, which is the property that mattered.
91
+ export const BOARD_WIDTH_SHARE = 0.5; // narration 50 / board 50, side by side
92
+ /** The single approved side-by-side record. The caller's columns are accepted and deliberately
81
93
  * ignored: this signature is where width used to decide the arrangement, and the parameter stays
82
94
  * so that "the plan does not depend on it" is a property a caller (and a test) can exercise. */
83
95
  export function boardSplitPlan(_callerCols) {
84
- return { direction: "down", ratio: BOARD_HEIGHT_SHARE, swap: "above" };
96
+ return { direction: "right", ratio: BOARD_WIDTH_SHARE };
85
97
  }
86
98
  /** The tab a slot belongs to: its TASK — worker, judge, review and consult panes for one task share it.
87
99
  * Returns undefined for everything else, which keeps those on the dedicated-tab path.
@@ -1081,16 +1093,17 @@ export class HerdrDriver {
1081
1093
  return p.workspace_id === this.ws && typeof p.pane_id === "string" && owned?.role === "watch" && owned.taskId === "run";
1082
1094
  }).map((p) => p.pane_id);
1083
1095
  }
1084
- // T2: the watch is a sibling of the daemon's own pane, never a separate tab — stacked ABOVE it at
1085
- // the caller's full width, always, whatever the terminal measures. Its durable owned name is how a
1096
+ // T2: the watch is a sibling of the daemon's own pane, never a separate tab — placed to the RIGHT
1097
+ // of it, always, whatever the terminal measures. Its durable owned name is how a
1086
1098
  // later daemon RECOGNIZES the board it must retire, so a run never stacks a second one.
1087
1099
  async watchSlot(cwd, name) {
1088
1100
  if (!this.ws)
1089
1101
  throw new Error("herdr watch placement requires HERDR_WORKSPACE_ID — refusing unseeded pane");
1090
1102
  if (!this.callerPane)
1091
1103
  throw new Error("herdr watch placement requires HERDR_PANE_ID — refusing untargeted split");
1092
- // One invariant placement (boardSplitPlan): split the caller down, then swap the new pane above
1093
- // it. No layout read decides this — width chose the arrangement twice and was wrong twice.
1104
+ // One invariant placement (boardSplitPlan): split the caller RIGHT. No layout read decides this —
1105
+ // width chose the arrangement twice and was wrong twice. No swap: a right split already lands the
1106
+ // new pane — the board — beside the caller, so there is no second operation to verify.
1094
1107
  const plan = boardSplitPlan();
1095
1108
  const sp = await this.herdr(`pane split ${shq(this.callerPane)} --direction ${plan.direction} --ratio ${plan.ratio} --no-focus`);
1096
1109
  if (sp.code !== 0)
@@ -1104,30 +1117,12 @@ export class HerdrDriver {
1104
1117
  }
1105
1118
  if (typeof pane !== "string" || !pane)
1106
1119
  throw new Error(`herdr watch split returned no pane id: ${sp.stdout}`);
1107
- // The split leaves the board UNDER the caller; the swap is what makes the stack the requested
1108
- // one. Verified, not assumed: a swap that failed would leave a board below the narration while
1109
- // the daemon reported the geometry it asked for. Instead the split pane is closed and the failure
1110
- // propagates — the daemon swallows it and runs boardless, which is honest about what is on screen.
1111
- // `pane swap` answers a no-op with a ZERO exit and `changed:false` (herdr socket API: a swap it
1112
- // declined is a non-error response), so an exit code alone proves nothing about the geometry —
1113
- // that is exactly the path that would leave the board below the narration while the daemon
1114
- // reported the stack. The documented `changed` flag is the verification; anything else — a
1115
- // nonzero exit, `changed:false`, an unparseable result — fails closed.
1116
- const swapped = await this.herdr(`pane swap --source-pane ${shq(pane)} --target-pane ${shq(this.callerPane)}`);
1117
- // Flag lives at `result.swap.changed` (verbatim 0.8.0); see herdr-swap-shape.test.ts.
1118
- let swapChanged;
1119
- try {
1120
- const result = JSON.parse(swapped.stdout).result;
1121
- swapChanged = result?.swap?.changed ?? result?.changed;
1122
- }
1123
- catch {
1124
- /* fail closed below */
1125
- }
1126
- if (swapped.code !== 0 || swapChanged !== true) {
1127
- await this.discardSplit(pane, `herdr watch swap ${plan.swap} failed: ${swapped.code !== 0
1128
- ? swapped.stderr || swapped.stdout
1129
- : `herdr reported no swap took place: ${swapped.stdout || swapped.stderr}`}`);
1130
- }
1120
+ // No swap step to verify any more. The stack needed one — the split landed the board UNDER the
1121
+ // caller and only the swap made the geometry the requested one, so a `pane swap` that no-opped
1122
+ // with a ZERO exit could leave a board below the narration while the daemon reported the stack;
1123
+ // the `changed` flag existed to catch exactly that. A right split places the board where it
1124
+ // belongs in ONE operation that either returns a pane id or throws above, so there is no
1125
+ // second-operation gap left to fail closed on.
1131
1126
  const renamed = await this.herdr(`pane rename ${shq(pane)} ${shq(name)}`);
1132
1127
  if (renamed.code !== 0 || await this.namedPaneId(name) !== pane) {
1133
1128
  await this.discardSplit(pane, `herdr watch rename failed: ${renamed.stderr || renamed.stdout}`);
@@ -90,6 +90,25 @@ export declare const effectiveCeilingMs: (entry?: Pick<BaselineCommand, "duratio
90
90
  * and never enters baseline forgiveness (there is no runner output to forgive).
91
91
  */
92
92
  export declare function ceilingKillResult(gate: string, r: ShResult, ceilingMs: number): GateResult | undefined;
93
+ /**
94
+ * OBS-612: the capture gets its OWN ceiling, thirty minutes, and it is not the shell default.
95
+ *
96
+ * `captureBaseline` used to call `sh(cmd, cwd)` with no ceiling argument, so it inherited
97
+ * `DEFAULT_SHELL_TIMEOUT_MS` (600s) — while every CONSUMER of the result sizes its ceiling with
98
+ * `effectiveCeilingMs`. A repo whose suite runs longer than ten minutes therefore had its capture
99
+ * SIGKILLed every single time, recording `{infra: true, fingerprints: []}`, and `freshFailures`
100
+ * correctly forgives nothing for an infra entry. The result is a baseline that forgives nothing on a
101
+ * repo that has pre-existing failures — every task gate reds on failures the diff did not cause.
102
+ *
103
+ * Measured before it was fixed: EIGHT consecutive runs of this repository, 2026-08-20 to 2026-08-25,
104
+ * every one killed within 325ms of 600000ms on a suite that needs ~700s. That tight a cluster is a
105
+ * hard timeout, not variable failure. Each of those captures dutifully computed and stored
106
+ * `ceilingMs ≈ 1800000` — the right answer — which the next run never read.
107
+ *
108
+ * A capture is a MEASUREMENT of how long the suite takes; giving it the same ceiling as an ordinary
109
+ * shell asks it to finish faster than the thing it is measuring.
110
+ */
111
+ export declare const CAPTURE_CEILING_MS = 1800000;
93
112
  export declare function captureBaseline(cwd: string, commands: Record<string, string>): Promise<Baseline>;
94
113
  export interface VacuousOracleWarning {
95
114
  kind: "vacuous-oracle";
@@ -318,10 +318,29 @@ export function ceilingKillResult(gate, r, ceilingMs) {
318
318
  meta: { classification: "infra", infra: true, kind: "ceiling-kill", durationMs, ceilingMs },
319
319
  };
320
320
  }
321
+ /**
322
+ * OBS-612: the capture gets its OWN ceiling, thirty minutes, and it is not the shell default.
323
+ *
324
+ * `captureBaseline` used to call `sh(cmd, cwd)` with no ceiling argument, so it inherited
325
+ * `DEFAULT_SHELL_TIMEOUT_MS` (600s) — while every CONSUMER of the result sizes its ceiling with
326
+ * `effectiveCeilingMs`. A repo whose suite runs longer than ten minutes therefore had its capture
327
+ * SIGKILLed every single time, recording `{infra: true, fingerprints: []}`, and `freshFailures`
328
+ * correctly forgives nothing for an infra entry. The result is a baseline that forgives nothing on a
329
+ * repo that has pre-existing failures — every task gate reds on failures the diff did not cause.
330
+ *
331
+ * Measured before it was fixed: EIGHT consecutive runs of this repository, 2026-08-20 to 2026-08-25,
332
+ * every one killed within 325ms of 600000ms on a suite that needs ~700s. That tight a cluster is a
333
+ * hard timeout, not variable failure. Each of those captures dutifully computed and stored
334
+ * `ceilingMs ≈ 1800000` — the right answer — which the next run never read.
335
+ *
336
+ * A capture is a MEASUREMENT of how long the suite takes; giving it the same ceiling as an ordinary
337
+ * shell asks it to finish faster than the thing it is measuring.
338
+ */
339
+ export const CAPTURE_CEILING_MS = 1_800_000;
321
340
  export async function captureBaseline(cwd, commands) {
322
341
  const base = { commands: {} };
323
342
  for (const [name, cmd] of Object.entries(commands)) {
324
- const r = await sh(cmd, cwd);
343
+ const r = await sh(cmd, cwd, CAPTURE_CEILING_MS);
325
344
  // ponytail: strip the executing cwd so repo-root capture and worktree compare fingerprint identically; /private-vs-/tmp symlink variance is out of scope
326
345
  // ponytail: a capture that was itself killed records the ceiling as its "measurement", which
327
346
  // scales the next ceiling up — the right direction for a suite that never finished once.
@@ -336,6 +355,14 @@ export async function captureBaseline(cwd, commands) {
336
355
  // it is the one thing the kill did establish — and still scales the next ceiling up.
337
356
  // `timedOut` is set in exactly one place (git.ts's kill timer), so no ordinary exit reaches here.
338
357
  if (r.timedOut === true) {
358
+ // OBS-612: SAY SO. An unbaselinable command is invisible on every surface — `status` reports
359
+ // gates and supervision, never "your baseline forgives nothing" — so eight runs of this repo
360
+ // passed through here in silence while every later gate paid for it. The operator reads this
361
+ // line at run start, BEFORE any task gate reds, which is the whole point: the failure is
362
+ // otherwise indistinguishable from a repo that simply has flaky tests.
363
+ console.error(`tickmarkr: baseline capture for "${name}" was killed at its ${CAPTURE_CEILING_MS}ms ceiling — `
364
+ + `it recorded NO fingerprints, so nothing is forgiven and every gate will treat a pre-existing `
365
+ + `failure as a fresh one. Raise the ceiling or shorten the command.`);
339
366
  base.commands[name] = { infra: true, fingerprints: [], durationMs, ceilingMs: effectiveCeilingMs({ durationMs }) };
340
367
  continue;
341
368
  }
package/dist/run/git.d.ts CHANGED
@@ -3,11 +3,30 @@ export { ROUTING_ENV_SEAMS };
3
3
  export declare const FORK_CAP_ENV = "VITEST_MAX_FORKS";
4
4
  export declare const DEFAULT_FORK_CAP = "6";
5
5
  /**
6
- * cap = max(1, floor(cores / concurrency)). Total test parallelism is then cap × concurrency, which
7
- * never exceeds max(cores, concurrency): at or below the core count the floor keeps the product
8
- * under `cores`, and above it the floor is 0 so the minimum of 1 pins the product to `concurrency`
9
- * itself. (The tempting `cap × concurrency ≤ cores` is unsatisfiable once concurrency > cores —
10
- * a run cannot give a suite less than one fork.)
6
+ * OBS-618: how many PROCESSES one vitest fork can hold at its peak — the fork, a daemon it spawns,
7
+ * and that daemon's own `git` child. The previous formula counted FORKS and assumed one process
8
+ * each, which is the assumption this suite violates by design: its tests spawn daemons that spawn
9
+ * git, so demand is a MULTIPLE of the cap and the fork table sees that multiple, not the cap.
10
+ *
11
+ * Measured, not guessed: on an 18-core machine at `concurrency: 1` the old formula authorised 18
12
+ * forks. Six gate batteries across two tasks with no shared file (T1, T3) and two vendors (claude,
13
+ * codex) then died on `spawn EAGAIN` — at load1 as low as 2.21, so this is fork-table exhaustion and
14
+ * not CPU saturation. 18 ÷ 3 = 6, which is exactly the value `DEFAULT_FORK_CAP` was already set to;
15
+ * the number was right and only the derivation was missing.
16
+ */
17
+ export declare const SPAWN_FANOUT = 3;
18
+ /**
19
+ * cap = max(1, floor(cores / (concurrency × SPAWN_FANOUT))). Total PROCESS demand is then
20
+ * cap × concurrency × SPAWN_FANOUT, which stays at or under `cores` wherever the floor is non-zero,
21
+ * and the `max(1, …)` pins it to one fork per run when the machine is too small to divide — a run
22
+ * cannot give a suite less than one fork.
23
+ *
24
+ * Portable by construction: `availableParallelism()` reports the CPUs actually usable by this
25
+ * process (it honours CPU affinity), so a 2-core CI runner derives 1, an 18-core workstation derives
26
+ * 6, and a 64-core host derives 21 — no machine is named anywhere and no value is pinned.
27
+ * ⚠ It is a FLOOR of 1, never 0: on a small host the cap stops dividing and the protection this
28
+ * provides runs out. That is the honest bound — a 2-core runner at concurrency 1 gets one fork and
29
+ * the fan-out is then bounded by the suite itself, not by us.
11
30
  */
12
31
  export declare const deriveForkCap: (concurrency: number, cores?: number) => number;
13
32
  /** Run `fn` with the fork budget this run's resolved concurrency implies. */
package/dist/run/git.js CHANGED
@@ -32,13 +32,32 @@ export const DEFAULT_FORK_CAP = "6";
32
32
  */
33
33
  const forkBudget = new AsyncLocalStorage();
34
34
  /**
35
- * cap = max(1, floor(cores / concurrency)). Total test parallelism is then cap × concurrency, which
36
- * never exceeds max(cores, concurrency): at or below the core count the floor keeps the product
37
- * under `cores`, and above it the floor is 0 so the minimum of 1 pins the product to `concurrency`
38
- * itself. (The tempting `cap × concurrency ≤ cores` is unsatisfiable once concurrency > cores —
39
- * a run cannot give a suite less than one fork.)
35
+ * OBS-618: how many PROCESSES one vitest fork can hold at its peak — the fork, a daemon it spawns,
36
+ * and that daemon's own `git` child. The previous formula counted FORKS and assumed one process
37
+ * each, which is the assumption this suite violates by design: its tests spawn daemons that spawn
38
+ * git, so demand is a MULTIPLE of the cap and the fork table sees that multiple, not the cap.
39
+ *
40
+ * Measured, not guessed: on an 18-core machine at `concurrency: 1` the old formula authorised 18
41
+ * forks. Six gate batteries across two tasks with no shared file (T1, T3) and two vendors (claude,
42
+ * codex) then died on `spawn EAGAIN` — at load1 as low as 2.21, so this is fork-table exhaustion and
43
+ * not CPU saturation. 18 ÷ 3 = 6, which is exactly the value `DEFAULT_FORK_CAP` was already set to;
44
+ * the number was right and only the derivation was missing.
45
+ */
46
+ export const SPAWN_FANOUT = 3;
47
+ /**
48
+ * cap = max(1, floor(cores / (concurrency × SPAWN_FANOUT))). Total PROCESS demand is then
49
+ * cap × concurrency × SPAWN_FANOUT, which stays at or under `cores` wherever the floor is non-zero,
50
+ * and the `max(1, …)` pins it to one fork per run when the machine is too small to divide — a run
51
+ * cannot give a suite less than one fork.
52
+ *
53
+ * Portable by construction: `availableParallelism()` reports the CPUs actually usable by this
54
+ * process (it honours CPU affinity), so a 2-core CI runner derives 1, an 18-core workstation derives
55
+ * 6, and a 64-core host derives 21 — no machine is named anywhere and no value is pinned.
56
+ * ⚠ It is a FLOOR of 1, never 0: on a small host the cap stops dividing and the protection this
57
+ * provides runs out. That is the honest bound — a 2-core runner at concurrency 1 gets one fork and
58
+ * the fan-out is then bounded by the suite itself, not by us.
40
59
  */
41
- export const deriveForkCap = (concurrency, cores = availableParallelism()) => Math.max(1, Math.floor(cores / Math.max(1, concurrency)));
60
+ export const deriveForkCap = (concurrency, cores = availableParallelism()) => Math.max(1, Math.floor(cores / (Math.max(1, concurrency) * SPAWN_FANOUT)));
42
61
  /** Run `fn` with the fork budget this run's resolved concurrency implies. */
43
62
  export const runWithForkBudget = (concurrency, fn) => forkBudget.run(String(deriveForkCap(concurrency)), fn);
44
63
  /** The cap owned by the run on this async context; the standalone default outside one. */
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "tickmarkr",
3
- "version": "2.1.0",
3
+ "version": "2.1.2",
4
4
  "description": "Spec in, verified work out.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -45,10 +45,27 @@ Before `tickmarkr compile` or `tickmarkr run`, compare the installed binary agai
45
45
 
46
46
  1. Run `tickmarkr version` (one line, machine-parseable).
47
47
  2. Read the `version` field from the repository's `package.json`.
48
- 3. If the binary is **older on major.minor** than the repo (e.g. binary `1.36.x` vs repo `1.38.x`), **stop immediately** and tell the operator to update the global install (`npm i -g tickmarkr@latest`) or link the repo binary. Do not compile, plan, or run on hope.
48
+ 3. If the binary and repository do not **agree on the entire version** (including the patch; e.g. binary `2.1.0` vs repo `2.1.1`), **stop immediately** and tell the operator to update the global install (`npm i -g tickmarkr@latest`) or link the repo binary. Do not compile, plan, or run on hope.
49
49
 
50
50
  A stale binary silently skips daemon gates shipped in newer releases — the v1.38 run exposed this when a global `1.36.0` binary missed the daemon tip-verify gate entirely (OBS-38). Preflight failure is always stop-and-report; never proceed-and-hope.
51
51
 
52
+ ### No run may be live in THIS repository
53
+
54
+ The version check above is only half the preflight. Before `compile` or `run`, confirm no run is already
55
+ live **in this repository**:
56
+
57
+ 1. Lead with this repository's own `.tickmarkr/graph.lock`. Read its recorded holder pid.
58
+ 2. Treat the lock as held by a LIVE run until `kill -0 <pid>` proves that holder dead.
59
+ 3. **Never require a machine-wide process pattern to be empty.** A lawful run in another repository — or
60
+ the probing shell's own argv — matches such a pattern, so an empty result is not evidence of safety
61
+ and a non-empty one is not evidence of danger.
62
+ 4. If you use a process probe as secondary evidence, exclude the probing process itself, resolve every
63
+ candidate's own working directory (for example `lsof -a -p <pid> -d cwd`), and count only candidates
64
+ whose cwd is **this repository root**.
65
+
66
+ The invariant this protects is per-repository — *never run two tickmarkr runs in the same repository
67
+ concurrently* — so a machine-wide check answers a question nobody asked and blocks work that is lawful.
68
+
52
69
  ## Verified handoffs (agent-to-agent messaging)
53
70
 
54
71
  When relaying missions between agents in a multi-agent terminal, **never use bare send-text** (`herdr agent send` / pane send-text) — it writes text without pressing Enter, so handoffs sit unsubmitted (OBS-39).
@@ -40,10 +40,27 @@ Before `tickmarkr compile` or `tickmarkr run`, compare the installed binary agai
40
40
 
41
41
  1. Run `tickmarkr version` (one line, machine-parseable).
42
42
  2. Read the `version` field from the repository's `package.json`.
43
- 3. If the binary is **older on major.minor** than the repo (e.g. binary `1.36.x` vs repo `1.38.x`), **stop immediately** and tell the operator to update the global install (`npm i -g tickmarkr@latest`) or link the repo binary. Do not compile, plan, or run on hope.
43
+ 3. If the binary and repository do not **agree on the entire version** (including the patch; e.g. binary `2.1.0` vs repo `2.1.1`), **stop immediately** and tell the operator to update the global install (`npm i -g tickmarkr@latest`) or link the repo binary. Do not compile, plan, or run on hope.
44
44
 
45
45
  A stale binary silently skips daemon gates shipped in newer releases — the v1.38 run exposed this when a global `1.36.0` binary missed the daemon tip-verify gate entirely (OBS-38). Preflight failure is always stop-and-report; never proceed-and-hope.
46
46
 
47
+ ### No run may be live in THIS repository
48
+
49
+ The version check above is only half the preflight. Before `compile` or `run`, confirm no run is already
50
+ live **in this repository**:
51
+
52
+ 1. Lead with this repository's own `.tickmarkr/graph.lock`. Read its recorded holder pid.
53
+ 2. Treat the lock as held by a LIVE run until `kill -0 <pid>` proves that holder dead.
54
+ 3. **Never require a machine-wide process pattern to be empty.** A lawful run in another repository — or
55
+ the probing shell's own argv — matches such a pattern, so an empty result is not evidence of safety
56
+ and a non-empty one is not evidence of danger.
57
+ 4. If you use a process probe as secondary evidence, exclude the probing process itself, resolve every
58
+ candidate's own working directory (for example `lsof -a -p <pid> -d cwd`), and count only candidates
59
+ whose cwd is **this repository root**.
60
+
61
+ The invariant this protects is per-repository — *never run two tickmarkr runs in the same repository
62
+ concurrently* — so a machine-wide check answers a question nobody asked and blocks work that is lawful.
63
+
47
64
  ## Verified handoffs (agent-to-agent messaging)
48
65
 
49
66
  When relaying missions between agents in a multi-agent terminal, **never use bare send-text** (`herdr agent send` / pane send-text) — it writes text without pressing Enter, so handoffs sit unsubmitted (OBS-39).
@@ -35,6 +35,16 @@ through brief lineage. **An executor choice nobody made is still an executor cho
35
35
  status, and either ADOPT the
36
36
  existing orchestrator (updated brief, re-armed watchers) or, if the old hierarchy is dead, archive the
37
37
  stale brief and build fresh.
38
+ ⚠ **VERIFY EVERY INHERITED WATCHER FROM THE PROCESS TABLE BEFORE YOU TRUST IT — re-arming your own
39
+ watchers is NOT enough, and a seat told only to re-arm its own is told the wrong thing.** An inherited
40
+ *"watcher armed"* line is a claim, not a watcher: it is a report by a seat that no longer exists, which
41
+ is strictly WEAKER than the live seat's report rule 11 already forbids trusting — and it reads as
42
+ settled fact. So at every adopt, walk the predecessor's watchers by class — **journal watchers,
43
+ artifact watchers, dialog watchers and beat loops, which is the closed set a session owns** — probe
44
+ each from the process table yourself (`pgrep -f <token>`, discriminated per rule 11), and re-arm every
45
+ one the table does not show. Earned 2026-08-25 (OBS-622): a handoff recorded *"artifact watcher armed"*
46
+ over two live consult verdicts; at adopt the only `watch-artifacts.sh` on the machine belonged to a
47
+ different repository, and nothing had been watching either file.
38
48
  **An adopted seat ANNOUNCES itself, in the same act as re-arming:** tell the adopted orchestrator the
39
49
  fresh seat is live (verified send: probe token + read-back). Through the gap its view of your tier read
40
50
  STALE, and a tier that believes it is unsupervised escalates into a file nobody is reading. Earned
@@ -69,9 +79,10 @@ through brief lineage. **An executor choice nobody made is still an executor cho
69
79
  2026-07-29, re-earned 2026-08-17):**
70
80
  - `OVERSEER` — you. Do not add a second live run surface: the daemon self-places the shipped board
71
81
  ABOVE the supervising seat that invokes the run.
72
- - `ORCH` — the orchestrator with the daemon-placed, run-id-pinned shipped board STACKED ABOVE it: the
73
- board owns the tab's full width and the top 72% of its height, and the orchestrator's own narration
74
- is the rail underneath. **Look for that `role: "watch"` pane; never hand-place or hand-roll a live
82
+ - `ORCH` — the orchestrator with the daemon-placed, run-id-pinned shipped board BESIDE it: the board
83
+ takes the RIGHT half of the tab and the orchestrator's own narration keeps the LEFT half. (It was a
84
+ full-width board above a narration rail until 2026-08-25; the operator changed it, because a task
85
+ table is a few rows and it was spending height it did not need while squeezing the narration.) **Look for that `role: "watch"` pane; never hand-place or hand-roll a live
75
86
  run surface. Nothing else, ever: a work seat NEVER splits into the ORCH tab.** Operator verbatim:
76
87
  *"in orch tab should be the orch and the watcher only."* Re-earned 2026-08-17: a planning seat split
77
88
  beside the orchestrator, and the operator caught it, again. The daemon owns this vertical stack and
@@ -140,8 +151,8 @@ journal tail to decide what happens next, or sweeping orphans — you have taken
140
151
  ### What the ORCHESTRATOR does, and what you require of it
141
152
 
142
153
  - **The live surface arrives with the run.** `tickmarkr run` is stdout-silent until run-end by design;
143
- its daemon self-places one shipped `role: "watch"` board ABOVE the supervising seat — full width, top
144
- 72% of the height, narration below — and pins that board to the daemon's run id. **Look for the matching
154
+ its daemon self-places one shipped `role: "watch"` board BESIDE the supervising seat — the right half
155
+ of the tab, narration on the left — and pins that board to the daemon's run id. **Look for the matching
145
156
  daemon-placed pane.** If it is absent, treat that as a daemon/run liveness fault and use the normal
146
157
  recovery path; never hand-place, hand-roll, or launch a replacement live surface. A board the daemon
147
158
  could not stack is not silently re-arranged: the split is closed and the run continues BOARDLESS, so an
@@ -347,6 +358,21 @@ they are left implicit:
347
358
  - Five distinct send failures in one leg — front-truncation, sitting unsubmitted, two silent losses,
348
359
  a probe that mistook its own echo for a reply — is what a prose send protocol costs under load:
349
360
  run `scripts/seat-send.sh` instead.
361
+ - **A `pane run` into a pane whose FOREGROUND is busy is a DELAYED command, not a lost one.** The
362
+ shell buffers the line and executes it the instant the foreground process exits — which, when that
363
+ process is a run daemon, means *at run-end*, unattended, possibly hours later. Measured 2026-08-24: a
364
+ resume typed into the run pane while the daemon still held it fired by itself at the next run-end;
365
+ the seat that sent it read the later activity as an unattributed injection and nearly declared the
366
+ pane compromised. Two consequences: **check the target pane's foreground before sending** (a live
367
+ daemon owns it — send from your own shell instead), and **before attributing any unexplained
368
+ activity to an intruder, ask what YOU left buffered there.** The benign explanation is the common
369
+ one, and the alarming one costs a false security incident.
370
+ - **A compound command that `cd`s POISONS every later relative path in the same call.** The Bash tool's
371
+ working directory persists, so `cd /tmp && …` followed by `cat .planning/x` silently reads the wrong
372
+ tree. Measured twice on 2026-08-24: once reading a worktree's config and nearly authoring a duplicate
373
+ fix for interims that were present all along, once failing a release export script that existed.
374
+ **State reads during supervision use ABSOLUTE paths**, and a read that contradicts what you believe
375
+ is a cue to check your cwd before you rewrite your model of the world.
350
376
  - **Guard-before-Enter** (race-safe prompt answering): chain with `&&` — pane get shows `blocked` && pane
351
377
  read shows the expected option under the cursor && only then send-keys. If no longer `blocked`, someone
352
378
  already answered; do nothing.
@@ -408,6 +434,24 @@ tier ages to `STALE` (never `ABSENT`) within six beats, which is the state that
408
434
  Stand down explicitly when you hand off, or a deliberate exit reads as a death. Same rule as rule 29
409
435
  below, now with a conventional path the other tier already reads: `tickmarkr status` shows it.
410
436
 
437
+ ⚠ **THE LOOP ABOVE BINDS TO A PROCESS, NOT TO A SEAT — and that is a defect this skill shipped.**
438
+ The beat keeps running while its *session* lives, so a loop started by a seat that has since been
439
+ cleared, re-briefed, or replaced keeps beating that tier's file forever. Measured 2026-08-24
440
+ (OBS-583): a **2d20h** orphan loop from a predecessor seat held `orchestrator ARMED` through a
441
+ **three-hour window in which no orchestrator was alive**, and it would have silently re-armed a
442
+ recorded stand-down within 10 seconds. On the same sweep the overseer tier had **three** beat loops,
443
+ one owned by an unrelated session. So:
444
+ - **At every adopt, clear, or re-brief, sweep for pre-existing loops on YOUR tier before arming one**
445
+ (`pgrep -f "tickmarkr beat <tier>"`), trace each to its parent session, and kill the **loop only**
446
+ — never the parent — then verify the parent survived.
447
+ - **`ARMED` is a claim about a process, not about a seat.** Before trusting any tier's `ARMED`, ask
448
+ whose session owns the beater; a tier can be armed and seatless, which is *worse* than ABSENT
449
+ because it reads as coverage (rule 11's outliving-its-trigger failure, in beat form).
450
+ - Stand-down must kill the loop **and** run `--stand-down`; the second without the first is undone
451
+ by the next tick.
452
+ The product fix (a seat-bound or sentinel-terminated beat, armed and stood down in one act) is
453
+ queued; until it ships, this sweep is the guard.
454
+
411
455
  Arm the bundled watcher as its OWN Bash call with `run_in_background` — chaining it after other commands
412
456
  with `&` orphans it from the wake chain. It prints one wake reason and exits; re-arm after every wake.
413
457
 
@@ -427,6 +471,21 @@ mid-work — reading files, context climbing. So a status-keyed watcher can both
427
471
  worker** and **fire on a working one**, and neither failure announces itself. The bundled watcher inherits
428
472
  this; so does any `herdr agent wait`. It is still worth arming — it catches vanished panes and real
429
473
  blocks — but **never treat its silence as evidence a worker is healthy.**
474
+ **A THIRD failure of the same proxy, and it hits the SUPERVISING seat, not a worker: a session wedged
475
+ on a provider error or a CLI auto-update reports `working` forever.** Measured 2026-08-24: an
476
+ orchestrator took an API 529 mid-turn, its frame froze with the error rendered, and `agent_status`
477
+ read `working` across ten minutes while nothing advanced — and its own last transcript line claimed a
478
+ resume it had never issued, which disk falsified (journal ended at `run-end`, no lock, no daemon).
479
+ Nothing in a watcher set aimed at the RUN can see this: a wake delivered to a wedged seat's queue
480
+ reaches nobody. Two cheap guards, both of which this seat lacked:
481
+ - **Poll the supervised seat's VIEWPORT for a persistent error frame** — the same text present in two
482
+ reads minutes apart is the tell; one read cannot distinguish a frozen frame from a live one.
483
+ - **Falsify the seat's own claims against disk at every state change it reports.** A wedged or
484
+ context-exhausted seat narrates intentions as completions. The lock, the journal's last row, and the
485
+ process table settle it in one command.
486
+ When it fires: interrupt (Esc, bounded), have the seat correct the false record in writing rather than
487
+ silently, then handoff + `/clear` + fresh brief BEFORE it takes the next boundary — a seat that just
488
+ mis-reported its own state is not the seat to hand a release decision to.
430
489
  Two keys that do not lie, in order of strength:
431
490
  - **The daemon's own waiter.** What `herdr pane wait-output` is matching on tells you the phase from the
432
491
  harness's state machine rather than from a status field: a `--match` on a readiness banner means the
@@ -501,6 +560,15 @@ that hang is byte-identical to a seat still working. The script prints each unfi
501
560
  last line on every timeout heartbeat — read it there, and when in doubt `tail -1` the artifact, never
502
561
  the transcript's claim about it.
503
562
 
563
+ **An ABSENT artifact ALONE cannot discriminate a working producer from a dead watcher.** A producer still
564
+ working and a watcher that died with its seat write byte-identical evidence — nothing — so a missing file
565
+ is one signal carrying at least three meanings (still working, watcher dead, producer dead), and it is
566
+ **never** evidence that the watcher is still waiting. Reading it that way infers an instrument's liveness
567
+ from the silence it was built to sit through. Settle it with two probes that do not share a failure:
568
+ the watcher from the process table, the producer from its pane or seat status. Measured 2026-08-25
569
+ (OBS-622): both consultants were live and had written nothing, so the artifact side could not see that
570
+ nothing was watching them.
571
+
504
572
  **Arm it in the same call as the spawn, not the next one.** A watcher armed "after I finish this step"
505
573
  leaves a gap exactly as wide as however long you stay busy, and you will be busy — you just spawned work.
506
574
  **Measured 2026-08-06 (OBS-369): two consult verdicts, 30KB and 12.8KB, sat COMPLETE with their markers
@@ -628,6 +696,50 @@ orchestrator turn boundary.
628
696
  fix helps ONE operator and leaves every other user with the defect. If an overlay is the interim, it
629
697
  says so in writing and names its removal condition.
630
698
 
699
+ 8. **A SHIPPED VERSION IS NOT DONE UNTIL THE STATE IT LEAVES BEHIND IS CLEAN.** Publishing is the loud
700
+ half; the quiet half is that the NEXT seat inherits either the truth or a confident lie. Run this
701
+ before you stand down from any release — **operator instruction, 2026-08-25: *"overseer should always
702
+ leave clean state after a version is shipped"***. Every line below is a defect that actually happened
703
+ on the release that produced this rule.
704
+
705
+ - **REWRITE THE MEMORY INDEX FIRST, and read it back.** Minutes after `2.1.1` hit npm, the index line
706
+ a fresh session loads still read *"⛔ 2.1.1 CANNOT ship from run …2011"* — true when written, and by
707
+ then the exact opposite of the truth. **The index is what everyone loads and the body is what nobody
708
+ opens** (Evidence rule 25), so a stale index is not a cosmetic lag; it is the most-read wrong
709
+ sentence in the project. State what shipped, what did NOT, and the first three things the next seat
710
+ should do.
711
+ - **VERIFY EVERY ID THE CODE NOW CITES ACTUALLY EXISTS.** A release lands source comments citing
712
+ ledger ids. One of that release's entries was written by a heredoc in a command that then timed out
713
+ — the entry survived, but nothing had checked. `grep -c '^## OBS-<id>'` for each id the diff
714
+ introduced. A citation pointing at nothing is the defect the ledger itself files (OBS-604), shipped
715
+ into `src/`.
716
+ - **KILL THE BEAT LOOP *AND* RUN `--stand-down`.** Either alone is worse than neither: the loop
717
+ without the stand-down re-arms a tier you retired within 10s, and the stand-down without the loop
718
+ is undone by the next tick. Verify `status` reads `DISARMED` — which means *handed off*, distinct
719
+ from `STALE` (armed then died) and `ABSENT` (never armed).
720
+ - **RECORD YOUR WATCHERS AS DYING WITH THIS SESSION — never as "armed".** A written stand-down or
721
+ handoff may NOT carry the bare wording *"watcher armed"* for anything this seat owns: that form
722
+ states an act and lets the successor read a fact, and it survived into a handoff exactly once before
723
+ costing two unwatched consult verdicts (OBS-622). The admissible form names the lifetime and the
724
+ work it leaves the successor — *"watchers armed by this session (journal, artifact, dialog, beat);
725
+ they die with it — re-arm on adopt"* — and, per rule 11, says which tier's watchers were NOT armed.
726
+ A detached watcher is the one exception and must be labelled as such, with its heartbeat file, since
727
+ it outlives the seat instead.
728
+ - **SWEEP THE PANES THE RUN LEFT.** A daemon killed by a signal flushes its journal and releases its
729
+ lock but **does not clean up its worker panes or its board**. Two orphaned worker panes and a dead
730
+ board pane sat in the operator's tab bar until he screenshotted them. Verify each is inert first
731
+ (no agent, nothing running in its worktree) and confirm the WORK is on its branch — then close.
732
+ Emptied tabs disappear on their own.
733
+ - **CORRECT EVERY TAB LABEL.** `ORCH · 2.1.1 T1 regate` was still on screen hours after that regate
734
+ ended. Tab labels are how the operator reads fleet state; a stale one is a false status report.
735
+ - **LEAVE THE TREE CLEAN AND SAY WHAT IS UNMERGED.** Name the branches that hold real but ungated
736
+ work, so the next seat neither discards nor trusts them. *"Zero merges, T1/T3 branches ungated, T5
737
+ never dispatched"* is a handoff; *"the run ended"* is not.
738
+
739
+ ⚠ **The half of this that is NOT operator discipline must be QUEUED, not absorbed:** a daemon that
740
+ orphans its panes on SIGTERM is a PRODUCT defect and belongs in `src/**`. Sweeping by hand every time
741
+ is the local remedy, and per rule 7 it says so in writing and names its removal condition.
742
+
631
743
  ---
632
744
 
633
745
  ## Briefing a seat to audit a security-shaped check — phrasing matters
@@ -708,6 +820,17 @@ twice.** They are mission-independent on purpose: nothing here names a task, a l
708
820
 
709
821
  **Instruments**
710
822
 
823
+ 10a. **A CONTROL NAMES THE PROPERTY IT VERIFIES, NEVER AN INSTANCE OF IT** — and when it names an
824
+ instance, the seat that wrote it must correct the CONTROL rather than let it stop lawful work.
825
+ Measured 2026-08-24: a ruling armed *"the dispatch must resolve to `codex:gpt-5.6-terra` — anything
826
+ else, STOP"*, when the property the control existed to prove was *"a lawful channel OUTSIDE the
827
+ task's tried set, admitted by the amended floor."* Marginal-cost routing correctly picked a
828
+ cheaper mid-tier channel that satisfied the property completely — never tried, proven on the same
829
+ run — and the letter of the control said halt. **An over-specified control converts a working
830
+ mechanism into a false stop, and the seat under it will obey.** Write the property; if you catch
831
+ yourself naming a channel, a model, a pid or a hash, ask what that instance is standing in for.
832
+ The converse holds too: **a control loose enough to pass on the wrong thing is worse** — the fix is
833
+ precision about the property, not about the example.
711
834
  11. **For any guard whose failure is SILENCE — detector, lint, watcher, gate, alarm branch — the acceptance
712
835
  test is a POSITIVE CONTROL, not a clean run.** A zero cannot distinguish *nothing is broken* from *the
713
836
  instrument is blind* from *the check does not exist*. Remove the condition it should catch and confirm
@@ -727,6 +850,14 @@ twice.** They are mission-independent on purpose: nothing here names a task, a l
727
850
  owns it** — "watchers alive" is the one claim a seat cannot verify about itself. Measured 2026-08-06:
728
851
  an orchestrator sat `idle` through three merges and two dispatches with no journal watcher in the
729
852
  process table, while its own last report read *"daemon, board, sweeper, watcher all alive"* (OBS-366).
853
+ **STATE THE LIFETIME, because an unstated one is read as the mission's: a session-scoped watcher DIES
854
+ WITH THE SEAT THAT ARMED IT.** Every watcher a seat arms — journal, artifact, dialog, beat loop — is
855
+ session-scoped unless it was deliberately detached (`ppid 1`, the heartbeat form below), so `/clear`,
856
+ a crash, an adopt or a stand-down ends it, and **a handoff is the one moment the arming seat stops
857
+ existing** — which is exactly when its watchers are most likely to be believed. The inverse failure is
858
+ the same root read the other way: a DETACHED loop outlives its seat and holds a tier `ARMED` with
859
+ nobody home (OBS-583). Neither direction may be assumed; the lifetime is a property of how the watcher
860
+ was launched, and it belongs in writing next to every claim that one is armed.
730
861
  **And the process-table probe has a standard idiom that DEFEATS it, so the rule above needs one more
731
862
  line to be usable.** Never probe for a watcher with `ps … | grep <token> | grep -v grep`: a poll-grep
732
863
  watcher carries the word `grep` in its own argv, so the filter whose job is removing the *probing* grep
@@ -736,7 +867,19 @@ twice.** They are mission-independent on purpose: nothing here names a task, a l
736
867
  it as a defect — and was corrected forty minutes later when the watcher fired normally, having been
737
868
  alive throughout. Two hypotheses (`ps` truncation; multi-column truncation) were formed and killed by
738
869
  measurement first, and the first falsification was itself run against the wrong `ps` form. **Use
739
- `pgrep -f <token>`, or read the lock's own pid.** The general rule: **an exclusion filter is exactly as
870
+ `pgrep -f <token>`, or read the lock's own pid.**
871
+ ⚠ **AND `pgrep -f` HAS ITS OWN INVERSE FAILURE, so the recommended fix is not free: it matches the
872
+ ARGV OF THE SHELL RUNNING IT.** A probe written as `pgrep -f "npm test"` is itself a process whose
873
+ command line contains `npm test`, so it returns its own shell — a PHANTOM that looks exactly like the
874
+ contamination you are hunting. Measured 2026-08-25: a load-ceiling wake was investigated, a second
875
+ `npm test` "outside the gate's worktree" was found, and it was the probe. It had vanished by the next
876
+ command, which is the tell — a real second suite does not exit between two reads. The escalation would
877
+ have been a false contamination alarm during a task's last attempt.
878
+ Discriminate before you believe a hit: **resolve each pid's `cwd` AND drop any whose own command
879
+ contains the probe** (`pgrep`/`bash -c`), or match a pattern the target has and the probe cannot —
880
+ the binary's real path rather than the words you typed. `grep -v grep` fails toward *not there*;
881
+ `pgrep -f` fails toward *there twice*, and this direction gets ACTED ON, which is worse.
882
+ The general rule: **an exclusion filter is exactly as
740
883
  dangerous as an over-broad inclusion filter, and it fails in the direction that reads as "not there" —
741
884
  which is the direction that gets acted on.**
742
885
  Two corollaries: **re-arm a wake-and-exit watcher as the same turn's LAST act**, not the next turn's
@@ -57,12 +57,34 @@ pane_of() {
57
57
  | sed -n 's/.*"pane_id":"\([^"]*\)".*/\1/p' | head -1
58
58
  }
59
59
 
60
- # The rendered `❯` line — the only state that discriminates submitted from sitting. head -1 returns
61
- # the first rendered line, i.e. a PREFIX of a wrapped draft, so callers compare prefixes, never
62
- # whole sentences (OBS-396: a full-sentence match returns false on a message that arrived intact).
60
+ # The rendered `❯` line — the only state that discriminates submitted from sitting. tail -1 takes
61
+ # the LAST `❯` line, which is always the LIVE INPUT BOX: a wrapped draft's continuation lines carry
62
+ # no `❯`, so the last one is still that draft's FIRST rendered line, i.e. a PREFIX of the draft —
63
+ # callers compare prefixes, never whole sentences (OBS-396: a full-sentence match returns false on a
64
+ # message that arrived intact).
65
+ #
66
+ # SUBMITTED-ECHO DISCRIMINATOR (OBS-603, captured 2026-08-25 on an overseer's first send). claude-code
67
+ # echoes a SUBMITTED message into the transcript with the same `❯` glyph, and a long one still occupies
68
+ # this 14-line window ABOVE the empty box. The old `head -1` returned that echo, `is_ours` matched it
69
+ # (the echo IS a prefix of $MSG), and one delivered message drew TWO false verdicts: SEND_UNSUBMITTED on
70
+ # the send that made the echo, then REFUSED_BOX_OCCUPIED on the NEXT send — the worse half, because it
71
+ # refuses to deliver at all and leaves the seat unreachable until the echo scrolls out. Both reactions
72
+ # this script warns against are then wrong: re-sending appends and submits both, escalating reports a
73
+ # failure that never happened. OBS-396's phantom, one layer up.
74
+ #
75
+ # ⚠ POSITION IS NOT THE DISCRIMINATOR, and the kimi control below proves it: the two TUIs render in
76
+ # OPPOSITE order. kimi puts the live box FIRST and stages BELOW it; claude-code puts the echo FIRST and
77
+ # the live box BELOW. So `head -1` is wrong for one and `tail -1` is wrong for the other — a fix that
78
+ # only flipped them would have traded this defect for the staged-queue defect OBS-552 already paid for.
79
+ # What actually identifies the ordinary input box is that it is the LAST `❯` line which is NOT a staged
80
+ # entry: staged lines carry the `↑ to edit · ctrl-s to steer` affordance and belong to `staged_line()`,
81
+ # which owns that concept. Dropping them here makes this function's contract exact — the live ORDINARY
82
+ # box — instead of positional, and it reads the affordance rather than a vendor name, so any TUI that
83
+ # grows the same queue is covered without a matcher list.
63
84
  prompt_line() {
64
85
  herdr agent read "$TARGET" --source visible --lines 14 2>/dev/null \
65
- | sed -n 's/^[[:space:]]*❯[[:space:]]*//p' | head -1 | sed 's/[[:space:]]*$//'
86
+ | sed '/↑ to edit/{/steer/d;}' \
87
+ | sed -n 's/^[[:space:]]*❯[[:space:]]*//p' | tail -1 | sed 's/[[:space:]]*$//'
66
88
  }
67
89
 
68
90
  # STAGED-QUEUE DISCRIMINATOR (OBS-552 addendum, captured 2026-08-19 on a live kimi seat). After its
@@ -1,5 +1,5 @@
1
1
  #!/bin/bash
2
- # watch-contamination.sh — wake the AUTHORITY seat when a gate VERDICT may be contaminated.
2
+ # watch-contamination.sh — wake the AUTHORITY seat when a gate VERDICT may be contaminated.
3
3
  #
4
4
  # Operator, 2026-08-07: "that is the kind of job I need overseer to be vigilant about."
5
5
  #
@@ -30,7 +30,7 @@ INFRA_RE='vitest-worker|Timeout calling|JS heap out of memory|ENOMEM|EAGAIN|spaw
30
30
 
31
31
  load1() { uptime | sed 's/.*load averages*: *//' | awk '{print $1}' | tr -d ','; }
32
32
  # Count real vitest runners only. `pgrep -f vitest` over-counts by an order: it matches the WORD
33
- # in any argv — worker prompts, VITEST_MAX_FORKS in shell strings (measured 7 vs 2 real, 2026-08-10).
33
+ # in any argv — worker prompts, VITEST_MAX_FORKS in shell strings (measured 7 vs 2 real, 2026-08-10).
34
34
  # [v] keeps the pattern from matching its own grep. Forked workers retitle to "node (vitest N)" (comm).
35
35
  vitest_n() { ps -axo comm,command 2>/dev/null | grep -Ec 'node_modules/(\.bin/)?[v]itest|[v]itest/dist/|\([v]itest [0-9]+\)'; }
36
36
 
@@ -47,7 +47,11 @@ while [ "$elapsed" -lt "$CAP" ]; do
47
47
  elapsed=$((elapsed + POLL))
48
48
 
49
49
  L=$(load1); V=$(vitest_n)
50
- Li=${L%%.*}; [ -z "$Li" ] && Li=0
50
+ # OBS-585: strip at the first NON-DIGIT, never at "." - a locale that renders the load average
51
+ # with U+066B (3٫48) or a comma leaves the period-strip a no-op, the numeric test below then
52
+ # errors, and its 2>/dev/null hides that, so the LOAD trigger dies silently at every load.
53
+ # Proven against the incident this watcher exists for: 35٫26 vs ceiling 24 never fired.
54
+ Li=$(printf %s "$L" | sed "s/[^0-9].*//"); [ -z "$Li" ] && Li=0
51
55
 
52
56
  if [ -f "$J" ]; then
53
57
  seen=$(cat "$STATE" 2>/dev/null); [ -z "$seen" ] && seen=0
@@ -63,7 +67,7 @@ while [ "$elapsed" -lt "$CAP" ]; do
63
67
  task=$(printf '%s' "$hit" | sed -n 's/.*"taskId":"\([^"]*\)".*/\1/p')
64
68
  gate=$(printf '%s' "$hit" | sed -n 's/.*"gate":"\([^"]*\)".*/\1/p')
65
69
  echo "CONTAMINATED_VERDICT task=$task gate=$gate load=$L vitest=$V"
66
- echo " an infra fingerprint appeared in a FAILED gate — this red is not evidence about the diff"
70
+ echo " an infra fingerprint appeared in a FAILED gate — this red is not evidence about the diff"
67
71
  echo " ruling owed: is the attempt chargeable? (OBS-426: infra failures are not)"
68
72
  exit 0
69
73
  fi
@@ -77,4 +81,4 @@ while [ "$elapsed" -lt "$CAP" ]; do
77
81
  fi
78
82
  done
79
83
 
80
- echo "WATCH_CAP_REACHED load=$(load1) vitest=$(vitest_n) — no contaminated verdict seen in ${CAP}s"
84
+ echo "WATCH_CAP_REACHED load=$(load1) vitest=$(vitest_n) — no contaminated verdict seen in ${CAP}s"