tickmarkr 2.0.0 → 2.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -90,6 +90,25 @@ export declare const effectiveCeilingMs: (entry?: Pick<BaselineCommand, "duratio
90
90
  * and never enters baseline forgiveness (there is no runner output to forgive).
91
91
  */
92
92
  export declare function ceilingKillResult(gate: string, r: ShResult, ceilingMs: number): GateResult | undefined;
93
+ /**
94
+ * OBS-612: the capture gets its OWN ceiling, thirty minutes, and it is not the shell default.
95
+ *
96
+ * `captureBaseline` used to call `sh(cmd, cwd)` with no ceiling argument, so it inherited
97
+ * `DEFAULT_SHELL_TIMEOUT_MS` (600s) — while every CONSUMER of the result sizes its ceiling with
98
+ * `effectiveCeilingMs`. A repo whose suite runs longer than ten minutes therefore had its capture
99
+ * SIGKILLed every single time, recording `{infra: true, fingerprints: []}`, and `freshFailures`
100
+ * correctly forgives nothing for an infra entry. The result is a baseline that forgives nothing on a
101
+ * repo that has pre-existing failures — every task gate reds on failures the diff did not cause.
102
+ *
103
+ * Measured before it was fixed: EIGHT consecutive runs of this repository, 2026-08-20 to 2026-08-25,
104
+ * every one killed within 325ms of 600000ms on a suite that needs ~700s. That tight a cluster is a
105
+ * hard timeout, not variable failure. Each of those captures dutifully computed and stored
106
+ * `ceilingMs ≈ 1800000` — the right answer — which the next run never read.
107
+ *
108
+ * A capture is a MEASUREMENT of how long the suite takes; giving it the same ceiling as an ordinary
109
+ * shell asks it to finish faster than the thing it is measuring.
110
+ */
111
+ export declare const CAPTURE_CEILING_MS = 1800000;
93
112
  export declare function captureBaseline(cwd: string, commands: Record<string, string>): Promise<Baseline>;
94
113
  export interface VacuousOracleWarning {
95
114
  kind: "vacuous-oracle";
@@ -318,10 +318,29 @@ export function ceilingKillResult(gate, r, ceilingMs) {
318
318
  meta: { classification: "infra", infra: true, kind: "ceiling-kill", durationMs, ceilingMs },
319
319
  };
320
320
  }
321
+ /**
322
+ * OBS-612: the capture gets its OWN ceiling, thirty minutes, and it is not the shell default.
323
+ *
324
+ * `captureBaseline` used to call `sh(cmd, cwd)` with no ceiling argument, so it inherited
325
+ * `DEFAULT_SHELL_TIMEOUT_MS` (600s) — while every CONSUMER of the result sizes its ceiling with
326
+ * `effectiveCeilingMs`. A repo whose suite runs longer than ten minutes therefore had its capture
327
+ * SIGKILLed every single time, recording `{infra: true, fingerprints: []}`, and `freshFailures`
328
+ * correctly forgives nothing for an infra entry. The result is a baseline that forgives nothing on a
329
+ * repo that has pre-existing failures — every task gate reds on failures the diff did not cause.
330
+ *
331
+ * Measured before it was fixed: EIGHT consecutive runs of this repository, 2026-08-20 to 2026-08-25,
332
+ * every one killed within 325ms of 600000ms on a suite that needs ~700s. That tight a cluster is a
333
+ * hard timeout, not variable failure. Each of those captures dutifully computed and stored
334
+ * `ceilingMs ≈ 1800000` — the right answer — which the next run never read.
335
+ *
336
+ * A capture is a MEASUREMENT of how long the suite takes; giving it the same ceiling as an ordinary
337
+ * shell asks it to finish faster than the thing it is measuring.
338
+ */
339
+ export const CAPTURE_CEILING_MS = 1_800_000;
321
340
  export async function captureBaseline(cwd, commands) {
322
341
  const base = { commands: {} };
323
342
  for (const [name, cmd] of Object.entries(commands)) {
324
- const r = await sh(cmd, cwd);
343
+ const r = await sh(cmd, cwd, CAPTURE_CEILING_MS);
325
344
  // ponytail: strip the executing cwd so repo-root capture and worktree compare fingerprint identically; /private-vs-/tmp symlink variance is out of scope
326
345
  // ponytail: a capture that was itself killed records the ceiling as its "measurement", which
327
346
  // scales the next ceiling up — the right direction for a suite that never finished once.
@@ -336,6 +355,14 @@ export async function captureBaseline(cwd, commands) {
336
355
  // it is the one thing the kill did establish — and still scales the next ceiling up.
337
356
  // `timedOut` is set in exactly one place (git.ts's kill timer), so no ordinary exit reaches here.
338
357
  if (r.timedOut === true) {
358
+ // OBS-612: SAY SO. An unbaselinable command is invisible on every surface — `status` reports
359
+ // gates and supervision, never "your baseline forgives nothing" — so eight runs of this repo
360
+ // passed through here in silence while every later gate paid for it. The operator reads this
361
+ // line at run start, BEFORE any task gate reds, which is the whole point: the failure is
362
+ // otherwise indistinguishable from a repo that simply has flaky tests.
363
+ console.error(`tickmarkr: baseline capture for "${name}" was killed at its ${CAPTURE_CEILING_MS}ms ceiling — `
364
+ + `it recorded NO fingerprints, so nothing is forgiven and every gate will treat a pre-existing `
365
+ + `failure as a fresh one. Raise the ceiling or shorten the command.`);
339
366
  base.commands[name] = { infra: true, fingerprints: [], durationMs, ceilingMs: effectiveCeilingMs({ durationMs }) };
340
367
  continue;
341
368
  }
package/dist/run/git.d.ts CHANGED
@@ -3,11 +3,30 @@ export { ROUTING_ENV_SEAMS };
3
3
  export declare const FORK_CAP_ENV = "VITEST_MAX_FORKS";
4
4
  export declare const DEFAULT_FORK_CAP = "6";
5
5
  /**
6
- * cap = max(1, floor(cores / concurrency)). Total test parallelism is then cap × concurrency, which
7
- * never exceeds max(cores, concurrency): at or below the core count the floor keeps the product
8
- * under `cores`, and above it the floor is 0 so the minimum of 1 pins the product to `concurrency`
9
- * itself. (The tempting `cap × concurrency ≤ cores` is unsatisfiable once concurrency > cores —
10
- * a run cannot give a suite less than one fork.)
6
+ * OBS-618: how many PROCESSES one vitest fork can hold at its peak — the fork, a daemon it spawns,
7
+ * and that daemon's own `git` child. The previous formula counted FORKS and assumed one process
8
+ * each, which is the assumption this suite violates by design: its tests spawn daemons that spawn
9
+ * git, so demand is a MULTIPLE of the cap and the fork table sees that multiple, not the cap.
10
+ *
11
+ * Measured, not guessed: on an 18-core machine at `concurrency: 1` the old formula authorised 18
12
+ * forks. Six gate batteries across two tasks with no shared file (T1, T3) and two vendors (claude,
13
+ * codex) then died on `spawn EAGAIN` — at load1 as low as 2.21, so this is fork-table exhaustion and
14
+ * not CPU saturation. 18 ÷ 3 = 6, which is exactly the value `DEFAULT_FORK_CAP` was already set to;
15
+ * the number was right and only the derivation was missing.
16
+ */
17
+ export declare const SPAWN_FANOUT = 3;
18
+ /**
19
+ * cap = max(1, floor(cores / (concurrency × SPAWN_FANOUT))). Total PROCESS demand is then
20
+ * cap × concurrency × SPAWN_FANOUT, which stays at or under `cores` wherever the floor is non-zero,
21
+ * and the `max(1, …)` pins it to one fork per run when the machine is too small to divide — a run
22
+ * cannot give a suite less than one fork.
23
+ *
24
+ * Portable by construction: `availableParallelism()` reports the CPUs actually usable by this
25
+ * process (it honours CPU affinity), so a 2-core CI runner derives 1, an 18-core workstation derives
26
+ * 6, and a 64-core host derives 21 — no machine is named anywhere and no value is pinned.
27
+ * ⚠ It is a FLOOR of 1, never 0: on a small host the cap stops dividing and the protection this
28
+ * provides runs out. That is the honest bound — a 2-core runner at concurrency 1 gets one fork and
29
+ * the fan-out is then bounded by the suite itself, not by us.
11
30
  */
12
31
  export declare const deriveForkCap: (concurrency: number, cores?: number) => number;
13
32
  /** Run `fn` with the fork budget this run's resolved concurrency implies. */
package/dist/run/git.js CHANGED
@@ -32,13 +32,32 @@ export const DEFAULT_FORK_CAP = "6";
32
32
  */
33
33
  const forkBudget = new AsyncLocalStorage();
34
34
  /**
35
- * cap = max(1, floor(cores / concurrency)). Total test parallelism is then cap × concurrency, which
36
- * never exceeds max(cores, concurrency): at or below the core count the floor keeps the product
37
- * under `cores`, and above it the floor is 0 so the minimum of 1 pins the product to `concurrency`
38
- * itself. (The tempting `cap × concurrency ≤ cores` is unsatisfiable once concurrency > cores —
39
- * a run cannot give a suite less than one fork.)
35
+ * OBS-618: how many PROCESSES one vitest fork can hold at its peak — the fork, a daemon it spawns,
36
+ * and that daemon's own `git` child. The previous formula counted FORKS and assumed one process
37
+ * each, which is the assumption this suite violates by design: its tests spawn daemons that spawn
38
+ * git, so demand is a MULTIPLE of the cap and the fork table sees that multiple, not the cap.
39
+ *
40
+ * Measured, not guessed: on an 18-core machine at `concurrency: 1` the old formula authorised 18
41
+ * forks. Six gate batteries across two tasks with no shared file (T1, T3) and two vendors (claude,
42
+ * codex) then died on `spawn EAGAIN` — at load1 as low as 2.21, so this is fork-table exhaustion and
43
+ * not CPU saturation. 18 ÷ 3 = 6, which is exactly the value `DEFAULT_FORK_CAP` was already set to;
44
+ * the number was right and only the derivation was missing.
45
+ */
46
+ export const SPAWN_FANOUT = 3;
47
+ /**
48
+ * cap = max(1, floor(cores / (concurrency × SPAWN_FANOUT))). Total PROCESS demand is then
49
+ * cap × concurrency × SPAWN_FANOUT, which stays at or under `cores` wherever the floor is non-zero,
50
+ * and the `max(1, …)` pins it to one fork per run when the machine is too small to divide — a run
51
+ * cannot give a suite less than one fork.
52
+ *
53
+ * Portable by construction: `availableParallelism()` reports the CPUs actually usable by this
54
+ * process (it honours CPU affinity), so a 2-core CI runner derives 1, an 18-core workstation derives
55
+ * 6, and a 64-core host derives 21 — no machine is named anywhere and no value is pinned.
56
+ * ⚠ It is a FLOOR of 1, never 0: on a small host the cap stops dividing and the protection this
57
+ * provides runs out. That is the honest bound — a 2-core runner at concurrency 1 gets one fork and
58
+ * the fan-out is then bounded by the suite itself, not by us.
40
59
  */
41
- export const deriveForkCap = (concurrency, cores = availableParallelism()) => Math.max(1, Math.floor(cores / Math.max(1, concurrency)));
60
+ export const deriveForkCap = (concurrency, cores = availableParallelism()) => Math.max(1, Math.floor(cores / (Math.max(1, concurrency) * SPAWN_FANOUT)));
42
61
  /** Run `fn` with the fork budget this run's resolved concurrency implies. */
43
62
  export const runWithForkBudget = (concurrency, fn) => forkBudget.run(String(deriveForkCap(concurrency)), fn);
44
63
  /** The cap owned by the run on this async context; the standalone default outside one. */
@@ -3,12 +3,21 @@ import { Box, render, Text, useApp, useInput } from "ink";
3
3
  import { useRef, useState } from "react";
4
4
  import { ToggleMark } from "./components.js";
5
5
  import { clip, INK, inkInput, inkOutput, KeyBar, padCell, Pointer } from "./frame.js";
6
- const DRIVERS = ["auto", "herdr", "subprocess"];
6
+ const DRIVERS = ["auto", "herdr", "subprocess", "orca"];
7
+ // One line per driver, shown for the value currently selected: four environments no longer fit on
8
+ // a single description row at 80 columns (the shipped three already clipped), and orca is an
9
+ // EXPLICIT choice — `auto` still means herdr-else-subprocess and never reaches for it.
10
+ const DRIVER_DESC = {
11
+ auto: "auto: herdr when HERDR_ENV=1, else subprocess — never orca",
12
+ herdr: "herdr: every worker runs in a visible pane you can watch and unblock",
13
+ subprocess: "subprocess: headless child processes — no cockpit, same fail-closed gates",
14
+ orca: "orca: visible terminals in the Orca app — an explicit choice, never auto's",
15
+ };
7
16
  const VISIBILITY = ["pane", "headless"];
8
17
  // Three fields, one toggle, one action — a descriptor array, deliberately not a forms framework.
9
- function buildRows(offerSkills) {
18
+ function buildRows(offerSkills, driver) {
10
19
  const rows = [
11
- { id: "driver", section: "Run", label: "Driver", desc: "auto: herdr when HERDR_ENV=1, else subprocess · herdr: visible panes · subprocess: headless child processes" },
20
+ { id: "driver", section: "Run", label: "Driver", desc: DRIVER_DESC[driver] },
12
21
  { id: "concurrency", section: "Run", label: "Concurrency", desc: "parallel task batteries per run — min 1; an empty or zero entry reverts on leave" },
13
22
  {
14
23
  id: "visibility",
@@ -36,7 +45,6 @@ function buildRows(offerSkills) {
36
45
  }
37
46
  export function InitWizardApp({ fields, frameColumns = 74 }) {
38
47
  const { exit } = useApp();
39
- const rows = buildRows(fields.offerSkills);
40
48
  const [cursor, setCursor] = useState(0);
41
49
  const cursorRef = useRef(0);
42
50
  const driverRef = useRef(fields.driver);
@@ -47,6 +55,8 @@ export function InitWizardApp({ fields, frameColumns = 74 }) {
47
55
  const lastValidConcurrencyRef = useRef(fields.concurrency);
48
56
  const typedRef = useRef(false);
49
57
  const doneRef = useRef(false);
58
+ // Built per render so the description row follows the driver the cursor last cycled to.
59
+ const rows = buildRows(fields.offerSkills, driverRef.current);
50
60
  const [, setRevision] = useState(0);
51
61
  const bump = () => setRevision((n) => n + 1);
52
62
  // Leaving the row (or pressing Enter on it) is the commit point: a parse ≥ 1 becomes the new
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "tickmarkr",
3
- "version": "2.0.0",
3
+ "version": "2.1.1",
4
4
  "description": "Spec in, verified work out.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -35,6 +35,10 @@ through brief lineage. **An executor choice nobody made is still an executor cho
35
35
  status, and either ADOPT the
36
36
  existing orchestrator (updated brief, re-armed watchers) or, if the old hierarchy is dead, archive the
37
37
  stale brief and build fresh.
38
+ **An adopted seat ANNOUNCES itself, in the same act as re-arming:** tell the adopted orchestrator the
39
+ fresh seat is live (verified send: probe token + read-back). Through the gap its view of your tier read
40
+ STALE, and a tier that believes it is unsupervised escalates into a file nobody is reading. Earned
41
+ 2026-08-22: a fresh seat re-armed all four watchers and announced nothing until the operator asked.
38
42
  ⚠ **Adopting a hierarchy silently adopts its EXECUTOR CHOICE.** The P92→P98 GSD drift propagated
39
43
  exactly this way: each overseer read the prior brief, reproduced "the same two-leg pattern as the
40
44
  last three phases", and the unruled bypass of the engine became load-bearing through repetition.
@@ -65,9 +69,10 @@ through brief lineage. **An executor choice nobody made is still an executor cho
65
69
  2026-07-29, re-earned 2026-08-17):**
66
70
  - `OVERSEER` — you. Do not add a second live run surface: the daemon self-places the shipped board
67
71
  ABOVE the supervising seat that invokes the run.
68
- - `ORCH` — the orchestrator with the daemon-placed, run-id-pinned shipped board STACKED ABOVE it: the
69
- board owns the tab's full width and the top 72% of its height, and the orchestrator's own narration
70
- is the rail underneath. **Look for that `role: "watch"` pane; never hand-place or hand-roll a live
72
+ - `ORCH` — the orchestrator with the daemon-placed, run-id-pinned shipped board BESIDE it: the board
73
+ takes the RIGHT half of the tab and the orchestrator's own narration keeps the LEFT half. (It was a
74
+ full-width board above a narration rail until 2026-08-25; the operator changed it, because a task
75
+ table is a few rows and it was spending height it did not need while squeezing the narration.) **Look for that `role: "watch"` pane; never hand-place or hand-roll a live
71
76
  run surface. Nothing else, ever: a work seat NEVER splits into the ORCH tab.** Operator verbatim:
72
77
  *"in orch tab should be the orch and the watcher only."* Re-earned 2026-08-17: a planning seat split
73
78
  beside the orchestrator, and the operator caught it, again. The daemon owns this vertical stack and
@@ -136,8 +141,8 @@ journal tail to decide what happens next, or sweeping orphans — you have taken
136
141
  ### What the ORCHESTRATOR does, and what you require of it
137
142
 
138
143
  - **The live surface arrives with the run.** `tickmarkr run` is stdout-silent until run-end by design;
139
- its daemon self-places one shipped `role: "watch"` board ABOVE the supervising seat — full width, top
140
- 72% of the height, narration below — and pins that board to the daemon's run id. **Look for the matching
144
+ its daemon self-places one shipped `role: "watch"` board BESIDE the supervising seat — the right half
145
+ of the tab, narration on the left — and pins that board to the daemon's run id. **Look for the matching
141
146
  daemon-placed pane.** If it is absent, treat that as a daemon/run liveness fault and use the normal
142
147
  recovery path; never hand-place, hand-roll, or launch a replacement live surface. A board the daemon
143
148
  could not stack is not silently re-arranged: the split is closed and the run continues BOARDLESS, so an
@@ -240,6 +245,9 @@ is a lossy summary nobody trusts while a clean session re-oriented from disk-ver
240
245
  **Do the same for yourself before you are forced to**: write the handoff while your judgment is still
241
246
  good, not after. If your own context cannot be read by the watcher, say so to the operator and ask for the
242
247
  number — an unmeasured budget is not a small budget.
248
+ **Every handoff's re-arm list ends with the announce step from Setup 0** — inform the surviving
249
+ orchestrator the fresh seat is live — or the next seat re-arms silently beside a tier that still
250
+ believes it is alone.
243
251
 
244
252
  ## Supervising GSD legs — when the mission dispatches `/gsd:*` instead of `tickmarkr run`
245
253
 
@@ -340,6 +348,21 @@ they are left implicit:
340
348
  - Five distinct send failures in one leg — front-truncation, sitting unsubmitted, two silent losses,
341
349
  a probe that mistook its own echo for a reply — is what a prose send protocol costs under load:
342
350
  run `scripts/seat-send.sh` instead.
351
+ - **A `pane run` into a pane whose FOREGROUND is busy is a DELAYED command, not a lost one.** The
352
+ shell buffers the line and executes it the instant the foreground process exits — which, when that
353
+ process is a run daemon, means *at run-end*, unattended, possibly hours later. Measured 2026-08-24: a
354
+ resume typed into the run pane while the daemon still held it fired by itself at the next run-end;
355
+ the seat that sent it read the later activity as an unattributed injection and nearly declared the
356
+ pane compromised. Two consequences: **check the target pane's foreground before sending** (a live
357
+ daemon owns it — send from your own shell instead), and **before attributing any unexplained
358
+ activity to an intruder, ask what YOU left buffered there.** The benign explanation is the common
359
+ one, and the alarming one costs a false security incident.
360
+ - **A compound command that `cd`s POISONS every later relative path in the same call.** The Bash tool's
361
+ working directory persists, so `cd /tmp && …` followed by `cat .planning/x` silently reads the wrong
362
+ tree. Measured twice on 2026-08-24: once reading a worktree's config and nearly authoring a duplicate
363
+ fix for interims that were present all along, once failing a release export script that existed.
364
+ **State reads during supervision use ABSOLUTE paths**, and a read that contradicts what you believe
365
+ is a cue to check your cwd before you rewrite your model of the world.
343
366
  - **Guard-before-Enter** (race-safe prompt answering): chain with `&&` — pane get shows `blocked` && pane
344
367
  read shows the expected option under the cursor && only then send-keys. If no longer `blocked`, someone
345
368
  already answered; do nothing.
@@ -401,6 +424,24 @@ tier ages to `STALE` (never `ABSENT`) within six beats, which is the state that
401
424
  Stand down explicitly when you hand off, or a deliberate exit reads as a death. Same rule as rule 29
402
425
  below, now with a conventional path the other tier already reads: `tickmarkr status` shows it.
403
426
 
427
+ ⚠ **THE LOOP ABOVE BINDS TO A PROCESS, NOT TO A SEAT — and that is a defect this skill shipped.**
428
+ The beat keeps running while its *session* lives, so a loop started by a seat that has since been
429
+ cleared, re-briefed, or replaced keeps beating that tier's file forever. Measured 2026-08-24
430
+ (OBS-583): a **2d20h** orphan loop from a predecessor seat held `orchestrator ARMED` through a
431
+ **three-hour window in which no orchestrator was alive**, and it would have silently re-armed a
432
+ recorded stand-down within 10 seconds. On the same sweep the overseer tier had **three** beat loops,
433
+ one owned by an unrelated session. So:
434
+ - **At every adopt, clear, or re-brief, sweep for pre-existing loops on YOUR tier before arming one**
435
+ (`pgrep -f "tickmarkr beat <tier>"`), trace each to its parent session, and kill the **loop only**
436
+ — never the parent — then verify the parent survived.
437
+ - **`ARMED` is a claim about a process, not about a seat.** Before trusting any tier's `ARMED`, ask
438
+ whose session owns the beater; a tier can be armed and seatless, which is *worse* than ABSENT
439
+ because it reads as coverage (rule 11's outliving-its-trigger failure, in beat form).
440
+ - Stand-down must kill the loop **and** run `--stand-down`; the second without the first is undone
441
+ by the next tick.
442
+ The product fix (a seat-bound or sentinel-terminated beat, armed and stood down in one act) is
443
+ queued; until it ships, this sweep is the guard.
444
+
404
445
  Arm the bundled watcher as its OWN Bash call with `run_in_background` — chaining it after other commands
405
446
  with `&` orphans it from the wake chain. It prints one wake reason and exits; re-arm after every wake.
406
447
 
@@ -420,6 +461,21 @@ mid-work — reading files, context climbing. So a status-keyed watcher can both
420
461
  worker** and **fire on a working one**, and neither failure announces itself. The bundled watcher inherits
421
462
  this; so does any `herdr agent wait`. It is still worth arming — it catches vanished panes and real
422
463
  blocks — but **never treat its silence as evidence a worker is healthy.**
464
+ **A THIRD failure of the same proxy, and it hits the SUPERVISING seat, not a worker: a session wedged
465
+ on a provider error or a CLI auto-update reports `working` forever.** Measured 2026-08-24: an
466
+ orchestrator took an API 529 mid-turn, its frame froze with the error rendered, and `agent_status`
467
+ read `working` across ten minutes while nothing advanced — and its own last transcript line claimed a
468
+ resume it had never issued, which disk falsified (journal ended at `run-end`, no lock, no daemon).
469
+ Nothing in a watcher set aimed at the RUN can see this: a wake delivered to a wedged seat's queue
470
+ reaches nobody. Two cheap guards, both of which this seat lacked:
471
+ - **Poll the supervised seat's VIEWPORT for a persistent error frame** — the same text present in two
472
+ reads minutes apart is the tell; one read cannot distinguish a frozen frame from a live one.
473
+ - **Falsify the seat's own claims against disk at every state change it reports.** A wedged or
474
+ context-exhausted seat narrates intentions as completions. The lock, the journal's last row, and the
475
+ process table settle it in one command.
476
+ When it fires: interrupt (Esc, bounded), have the seat correct the false record in writing rather than
477
+ silently, then handoff + `/clear` + fresh brief BEFORE it takes the next boundary — a seat that just
478
+ mis-reported its own state is not the seat to hand a release decision to.
423
479
  Two keys that do not lie, in order of strength:
424
480
  - **The daemon's own waiter.** What `herdr pane wait-output` is matching on tells you the phase from the
425
481
  harness's state machine rather than from a status field: a `--match` on a readiness banner means the
@@ -701,6 +757,17 @@ twice.** They are mission-independent on purpose: nothing here names a task, a l
701
757
 
702
758
  **Instruments**
703
759
 
760
+ 10a. **A CONTROL NAMES THE PROPERTY IT VERIFIES, NEVER AN INSTANCE OF IT** — and when it names an
761
+ instance, the seat that wrote it must correct the CONTROL rather than let it stop lawful work.
762
+ Measured 2026-08-24: a ruling armed *"the dispatch must resolve to `codex:gpt-5.6-terra` — anything
763
+ else, STOP"*, when the property the control existed to prove was *"a lawful channel OUTSIDE the
764
+ task's tried set, admitted by the amended floor."* Marginal-cost routing correctly picked a
765
+ cheaper mid-tier channel that satisfied the property completely — never tried, proven on the same
766
+ run — and the letter of the control said halt. **An over-specified control converts a working
767
+ mechanism into a false stop, and the seat under it will obey.** Write the property; if you catch
768
+ yourself naming a channel, a model, a pid or a hash, ask what that instance is standing in for.
769
+ The converse holds too: **a control loose enough to pass on the wrong thing is worse** — the fix is
770
+ precision about the property, not about the example.
704
771
  11. **For any guard whose failure is SILENCE — detector, lint, watcher, gate, alarm branch — the acceptance
705
772
  test is a POSITIVE CONTROL, not a clean run.** A zero cannot distinguish *nothing is broken* from *the
706
773
  instrument is blind* from *the check does not exist*. Remove the condition it should catch and confirm
@@ -729,7 +796,19 @@ twice.** They are mission-independent on purpose: nothing here names a task, a l
729
796
  it as a defect — and was corrected forty minutes later when the watcher fired normally, having been
730
797
  alive throughout. Two hypotheses (`ps` truncation; multi-column truncation) were formed and killed by
731
798
  measurement first, and the first falsification was itself run against the wrong `ps` form. **Use
732
- `pgrep -f <token>`, or read the lock's own pid.** The general rule: **an exclusion filter is exactly as
799
+ `pgrep -f <token>`, or read the lock's own pid.**
800
+ ⚠ **AND `pgrep -f` HAS ITS OWN INVERSE FAILURE, so the recommended fix is not free: it matches the
801
+ ARGV OF THE SHELL RUNNING IT.** A probe written as `pgrep -f "npm test"` is itself a process whose
802
+ command line contains `npm test`, so it returns its own shell — a PHANTOM that looks exactly like the
803
+ contamination you are hunting. Measured 2026-08-25: a load-ceiling wake was investigated, a second
804
+ `npm test` "outside the gate's worktree" was found, and it was the probe. It had vanished by the next
805
+ command, which is the tell — a real second suite does not exit between two reads. The escalation would
806
+ have been a false contamination alarm during a task's last attempt.
807
+ Discriminate before you believe a hit: **resolve each pid's `cwd` AND drop any whose own command
808
+ contains the probe** (`pgrep`/`bash -c`), or match a pattern the target has and the probe cannot —
809
+ the binary's real path rather than the words you typed. `grep -v grep` fails toward *not there*;
810
+ `pgrep -f` fails toward *there twice*, and this direction gets ACTED ON, which is worse.
811
+ The general rule: **an exclusion filter is exactly as
733
812
  dangerous as an over-broad inclusion filter, and it fails in the direction that reads as "not there" —
734
813
  which is the direction that gets acted on.**
735
814
  Two corollaries: **re-arm a wake-and-exit watcher as the same turn's LAST act**, not the next turn's
@@ -57,12 +57,34 @@ pane_of() {
57
57
  | sed -n 's/.*"pane_id":"\([^"]*\)".*/\1/p' | head -1
58
58
  }
59
59
 
60
- # The rendered `❯` line — the only state that discriminates submitted from sitting. head -1 returns
61
- # the first rendered line, i.e. a PREFIX of a wrapped draft, so callers compare prefixes, never
62
- # whole sentences (OBS-396: a full-sentence match returns false on a message that arrived intact).
60
+ # The rendered `❯` line — the only state that discriminates submitted from sitting. tail -1 takes
61
+ # the LAST `❯` line, which is always the LIVE INPUT BOX: a wrapped draft's continuation lines carry
62
+ # no `❯`, so the last one is still that draft's FIRST rendered line, i.e. a PREFIX of the draft —
63
+ # callers compare prefixes, never whole sentences (OBS-396: a full-sentence match returns false on a
64
+ # message that arrived intact).
65
+ #
66
+ # SUBMITTED-ECHO DISCRIMINATOR (OBS-603, captured 2026-08-25 on an overseer's first send). claude-code
67
+ # echoes a SUBMITTED message into the transcript with the same `❯` glyph, and a long one still occupies
68
+ # this 14-line window ABOVE the empty box. The old `head -1` returned that echo, `is_ours` matched it
69
+ # (the echo IS a prefix of $MSG), and one delivered message drew TWO false verdicts: SEND_UNSUBMITTED on
70
+ # the send that made the echo, then REFUSED_BOX_OCCUPIED on the NEXT send — the worse half, because it
71
+ # refuses to deliver at all and leaves the seat unreachable until the echo scrolls out. Both reactions
72
+ # this script warns against are then wrong: re-sending appends and submits both, escalating reports a
73
+ # failure that never happened. OBS-396's phantom, one layer up.
74
+ #
75
+ # ⚠ POSITION IS NOT THE DISCRIMINATOR, and the kimi control below proves it: the two TUIs render in
76
+ # OPPOSITE order. kimi puts the live box FIRST and stages BELOW it; claude-code puts the echo FIRST and
77
+ # the live box BELOW. So `head -1` is wrong for one and `tail -1` is wrong for the other — a fix that
78
+ # only flipped them would have traded this defect for the staged-queue defect OBS-552 already paid for.
79
+ # What actually identifies the ordinary input box is that it is the LAST `❯` line which is NOT a staged
80
+ # entry: staged lines carry the `↑ to edit · ctrl-s to steer` affordance and belong to `staged_line()`,
81
+ # which owns that concept. Dropping them here makes this function's contract exact — the live ORDINARY
82
+ # box — instead of positional, and it reads the affordance rather than a vendor name, so any TUI that
83
+ # grows the same queue is covered without a matcher list.
63
84
  prompt_line() {
64
85
  herdr agent read "$TARGET" --source visible --lines 14 2>/dev/null \
65
- | sed -n 's/^[[:space:]]*❯[[:space:]]*//p' | head -1 | sed 's/[[:space:]]*$//'
86
+ | sed '/↑ to edit/{/steer/d;}' \
87
+ | sed -n 's/^[[:space:]]*❯[[:space:]]*//p' | tail -1 | sed 's/[[:space:]]*$//'
66
88
  }
67
89
 
68
90
  # STAGED-QUEUE DISCRIMINATOR (OBS-552 addendum, captured 2026-08-19 on a live kimi seat). After its
@@ -1,5 +1,5 @@
1
1
  #!/bin/bash
2
- # watch-contamination.sh — wake the AUTHORITY seat when a gate VERDICT may be contaminated.
2
+ # watch-contamination.sh — wake the AUTHORITY seat when a gate VERDICT may be contaminated.
3
3
  #
4
4
  # Operator, 2026-08-07: "that is the kind of job I need overseer to be vigilant about."
5
5
  #
@@ -30,7 +30,7 @@ INFRA_RE='vitest-worker|Timeout calling|JS heap out of memory|ENOMEM|EAGAIN|spaw
30
30
 
31
31
  load1() { uptime | sed 's/.*load averages*: *//' | awk '{print $1}' | tr -d ','; }
32
32
  # Count real vitest runners only. `pgrep -f vitest` over-counts by an order: it matches the WORD
33
- # in any argv — worker prompts, VITEST_MAX_FORKS in shell strings (measured 7 vs 2 real, 2026-08-10).
33
+ # in any argv — worker prompts, VITEST_MAX_FORKS in shell strings (measured 7 vs 2 real, 2026-08-10).
34
34
  # [v] keeps the pattern from matching its own grep. Forked workers retitle to "node (vitest N)" (comm).
35
35
  vitest_n() { ps -axo comm,command 2>/dev/null | grep -Ec 'node_modules/(\.bin/)?[v]itest|[v]itest/dist/|\([v]itest [0-9]+\)'; }
36
36
 
@@ -47,7 +47,11 @@ while [ "$elapsed" -lt "$CAP" ]; do
47
47
  elapsed=$((elapsed + POLL))
48
48
 
49
49
  L=$(load1); V=$(vitest_n)
50
- Li=${L%%.*}; [ -z "$Li" ] && Li=0
50
+ # OBS-585: strip at the first NON-DIGIT, never at "." - a locale that renders the load average
51
+ # with U+066B (3٫48) or a comma leaves the period-strip a no-op, the numeric test below then
52
+ # errors, and its 2>/dev/null hides that, so the LOAD trigger dies silently at every load.
53
+ # Proven against the incident this watcher exists for: 35٫26 vs ceiling 24 never fired.
54
+ Li=$(printf %s "$L" | sed "s/[^0-9].*//"); [ -z "$Li" ] && Li=0
51
55
 
52
56
  if [ -f "$J" ]; then
53
57
  seen=$(cat "$STATE" 2>/dev/null); [ -z "$seen" ] && seen=0
@@ -63,7 +67,7 @@ while [ "$elapsed" -lt "$CAP" ]; do
63
67
  task=$(printf '%s' "$hit" | sed -n 's/.*"taskId":"\([^"]*\)".*/\1/p')
64
68
  gate=$(printf '%s' "$hit" | sed -n 's/.*"gate":"\([^"]*\)".*/\1/p')
65
69
  echo "CONTAMINATED_VERDICT task=$task gate=$gate load=$L vitest=$V"
66
- echo " an infra fingerprint appeared in a FAILED gate — this red is not evidence about the diff"
70
+ echo " an infra fingerprint appeared in a FAILED gate — this red is not evidence about the diff"
67
71
  echo " ruling owed: is the attempt chargeable? (OBS-426: infra failures are not)"
68
72
  exit 0
69
73
  fi
@@ -77,4 +81,4 @@ while [ "$elapsed" -lt "$CAP" ]; do
77
81
  fi
78
82
  done
79
83
 
80
- echo "WATCH_CAP_REACHED load=$(load1) vitest=$(vitest_n) — no contaminated verdict seen in ${CAP}s"
84
+ echo "WATCH_CAP_REACHED load=$(load1) vitest=$(vitest_n) — no contaminated verdict seen in ${CAP}s"