tickmarkr 1.84.0 → 1.86.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. package/README.md +4 -2
  2. package/dist/adapters/catalog-remote.d.ts +64 -0
  3. package/dist/adapters/catalog-remote.js +287 -0
  4. package/dist/adapters/catalog.d.ts +96 -0
  5. package/dist/adapters/catalog.js +176 -0
  6. package/dist/adapters/claude-code.d.ts +1 -0
  7. package/dist/adapters/claude-code.js +59 -1
  8. package/dist/adapters/fake.js +42 -4
  9. package/dist/adapters/model-lints.d.ts +25 -5
  10. package/dist/adapters/model-lints.js +184 -50
  11. package/dist/adapters/model-windows.d.ts +31 -0
  12. package/dist/adapters/model-windows.js +69 -0
  13. package/dist/adapters/prompt.d.ts +5 -1
  14. package/dist/adapters/prompt.js +13 -4
  15. package/dist/adapters/registry.d.ts +25 -26
  16. package/dist/adapters/registry.js +173 -110
  17. package/dist/adapters/types.d.ts +3 -0
  18. package/dist/adapters/types.js +36 -3
  19. package/dist/brand.d.ts +5 -1
  20. package/dist/brand.js +18 -2
  21. package/dist/cli/commands/doctor.d.ts +3 -0
  22. package/dist/cli/commands/doctor.js +43 -21
  23. package/dist/cli/commands/fleet.d.ts +7 -0
  24. package/dist/cli/commands/fleet.js +94 -74
  25. package/dist/cli/commands/init.js +118 -5
  26. package/dist/cli/commands/status.js +202 -46
  27. package/dist/compile/collateral.d.ts +86 -2
  28. package/dist/compile/collateral.js +294 -3
  29. package/dist/compile/gsd.d.ts +2 -1
  30. package/dist/compile/gsd.js +68 -2
  31. package/dist/compile/native.d.ts +14 -0
  32. package/dist/compile/native.js +161 -12
  33. package/dist/config/config.d.ts +82 -5
  34. package/dist/config/config.js +253 -66
  35. package/dist/config/fleet-overlay.d.ts +25 -20
  36. package/dist/config/fleet-overlay.js +195 -77
  37. package/dist/config/fleet-why.d.ts +23 -0
  38. package/dist/config/fleet-why.js +42 -0
  39. package/dist/drivers/herdr.d.ts +21 -3
  40. package/dist/drivers/herdr.js +344 -110
  41. package/dist/gates/acceptance.js +7 -2
  42. package/dist/gates/baseline.d.ts +1 -0
  43. package/dist/gates/baseline.js +91 -13
  44. package/dist/gates/llm.d.ts +0 -1
  45. package/dist/gates/llm.js +5 -30
  46. package/dist/gates/review.d.ts +9 -1
  47. package/dist/gates/review.js +105 -10
  48. package/dist/gates/run-gates.d.ts +9 -0
  49. package/dist/gates/run-gates.js +285 -41
  50. package/dist/gates/verdict-cause.d.ts +4 -0
  51. package/dist/gates/verdict-cause.js +63 -0
  52. package/dist/graph/schema.d.ts +6 -0
  53. package/dist/graph/schema.js +8 -5
  54. package/dist/route/router.d.ts +0 -5
  55. package/dist/route/router.js +16 -20
  56. package/dist/run/consult.d.ts +6 -0
  57. package/dist/run/consult.js +35 -25
  58. package/dist/run/daemon.d.ts +48 -2
  59. package/dist/run/daemon.js +1488 -330
  60. package/dist/run/journal.d.ts +56 -3
  61. package/dist/run/journal.js +358 -4
  62. package/dist/run/stall.d.ts +35 -1
  63. package/dist/run/stall.js +118 -8
  64. package/dist/tui/cockpit/capture.d.ts +12 -0
  65. package/dist/tui/cockpit/capture.js +37 -1
  66. package/dist/tui/cockpit/components.d.ts +2 -0
  67. package/dist/tui/cockpit/components.js +8 -8
  68. package/dist/tui/cockpit/derive.d.ts +29 -2
  69. package/dist/tui/cockpit/derive.js +219 -23
  70. package/dist/tui/cockpit/run-cockpit.js +128 -27
  71. package/dist/tui/cockpit/theme.d.ts +32 -26
  72. package/dist/tui/cockpit/theme.js +11 -5
  73. package/dist/tui/ink/components.d.ts +0 -15
  74. package/dist/tui/ink/components.js +0 -17
  75. package/dist/tui/ink/fleet-app.d.ts +4 -1
  76. package/dist/tui/ink/fleet-app.js +134 -13
  77. package/fixtures/sample.native.md +1 -1
  78. package/package.json +1 -1
  79. package/skills/tickmarkr-overseer/SKILL.md +354 -34
  80. package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +70 -0
  81. package/skills/tickmarkr-overseer/scripts/watch-panes.sh +1 -1
  82. package/dist/tui/ink/studio-app.d.ts +0 -59
  83. package/dist/tui/ink/studio-app.js +0 -320
  84. package/dist/tui/save.d.ts +0 -38
  85. package/dist/tui/save.js +0 -96
  86. package/dist/tui/staging.d.ts +0 -29
  87. package/dist/tui/staging.js +0 -78
@@ -22,12 +22,26 @@ Requires `HERDR_ENV=1`; if unset, say so and stop.
22
22
  status, and either ADOPT the
23
23
  existing orchestrator (updated brief, re-armed watchers) or, if the old hierarchy is dead, archive the
24
24
  stale brief and build fresh.
25
+ 0a. **READ THE PROJECT MEMORY BEFORE YOU START — it already contains discipline you are about to re-earn.**
26
+ `~/.claude/projects/<cwd-slug>/memory/` (slug = the absolute cwd with `/` → `-`). Read `MEMORY.md`, then
27
+ `ls` the topic entries and open every one whose name concerns METHOD or DISCIPLINE rather than a shipped
28
+ milestone — names like `*-discipline`, `*-drill`, `*-parity`, `*-least-permission`, `context-reset-*`,
29
+ `consults-*`, `agent-*`.
30
+ **Earned 2026-08-04, expensively.** That directory held `…-falsification-drill-discipline.md`, written
31
+ three weeks earlier: *"a gate or grep-pin is assumed WRONG until a falsification drill proves it bites…
32
+ run the drill that should redden it and SEE the red before trusting green."* That is Evidence discipline
33
+ rule 11 below, verbatim in substance. Nothing surfaced it, so an orchestrator and an overseer re-derived
34
+ it independently, twice, inside one hour — and the overseer then filed it as a NEW rule into a
35
+ mission-scoped brief. **A memory that exists and is never opened costs more than one that was never
36
+ written, because everyone assumes the lesson is somewhere.** Entries may predate a project rename; search
37
+ by concept, not by the current product name.
25
38
  1. Load the `herdr` skill. `herdr pane list` to map the workspace — the focused pane is yours. Rename your
26
39
  tab OVERSEER; create ONE tab ORCHESTRATOR.
27
40
  **Live tab labels (standing operator rule, 2026-07-12):** on every decision or state change (role
28
41
  handoff, task done/merged, run end) rename the affected tabs — and keep labels SHORT: the role as the
29
42
  main name plus at most ONE hot-state token. Vocabulary: ORCH carries the milestone and progress
30
- fraction (`ORCH · v1.19 4/5`, updated on every task-done); WORKERS carries the task token (tickmarkr
43
+ fraction (`ORCH · v1.19 4/5`, updated on every task-done); tickmarkr opens ONE TAB PER TASK, labelled
44
+ with the task id and holding that task's worker plus its judge/review/consult panes (tickmarkr
31
45
  updates it). Never long context strings or ✓-chains.
32
46
  2. **Orchestrator**: Launch the orchestrator with your agent host. Spawning on current herdr is two-step — the one-shot `agent start --cwd` form was removed in the herdr CLI redesign and now fails with `unknown option` (OBS-138): first create the pane with `herdr tab create --workspace <ws> --cwd <repo> --label "ORCH · <version>"` and parse `result.root_pane.pane_id` from its JSON, then start the agent in it. For Claude Code, use `herdr agent start orchestrator --kind claude --pane <root-pane-id> -- --permission-mode bypassPermissions` (append `--model <m>` after the `--` if the operator has a policy). For Codex, use `herdr agent start orchestrator --kind codex --pane <root-pane-id> -- --dangerously-bypass-approvals-and-sandbox` (add `--model <m>` to specify the model). The unsandboxed flag is REQUIRED: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation — do not downgrade it. Workers you never spawn — tickmarkr spawns its own visible worker panes. Auxiliary agents you do spawn (consultants, reviewers, scouts) follow the same forms: never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions`, and a read-only codex consultant may use `--sandbox read-only`.
33
47
  3. **Standing instructions travel as a brief FILE, never as pane text** — PTY input truncates at ~1024B and a
@@ -35,32 +49,98 @@ Requires `HERDR_ENV=1`; if unset, say so and stop.
35
49
  (inside the tickmarkr state dir — already self-gitignored, no exclude step needed), then send one line:
36
50
  `herdr pane run <orch> "Read .tickmarkr/overseer/ORCH-BRIEF.md and follow it exactly."` The brief MUST contain: the
37
51
  mission, the pane mechanics below, rules 1–2, and require a verbatim one-sentence acknowledgment of the
38
- human-checkpoint rule before anything is dispatched. Delete the dir at mission end.
52
+ human-checkpoint rule before anything is dispatched.
53
+ **⚠ HARVEST BEFORE YOU DELETE.** At mission end the brief dir goes — but a long mission accumulates
54
+ *method guards* in that brief (how to know a thing, not what is true of this spec), and deleting them
55
+ re-earns each one at full price on the next mission. So before removing the dir: lift every durable,
56
+ mission-independent guard into **this skill** (Evidence discipline, below) or the project's `CLAUDE.md`,
57
+ and only then delete. A guard's home must outlive the mission that earned it. The project ledger does
58
+ NOT count as that home — `CLAUDE.md` itself says planning records are read-only archives and current
59
+ guidance belongs in the memory file or the shipped docs.
39
60
  4. Arm the watcher (Supervision). Report the hierarchy map (pane ids + names) to the user.
40
61
 
41
- ## Supervising tickmarkr as the executor
62
+ ## Supervising tickmarkr as the executor — WHO DOES WHAT
42
63
 
43
- When the mission runs `/tickmarkr-auto` (tickmarkr dispatches the workers), supervision changes shape:
64
+ When the mission runs `/tickmarkr-auto` (tickmarkr dispatches the workers), supervision changes shape —
65
+ and the first thing to get right is that **almost none of it is yours**.
44
66
 
45
- - **Give the run a live surface.** `tickmarkr run` is stdout-silent until run-end by design — split a pane in the
46
- ORCHESTRATOR tab running `tickmarkr status --watch`. Narration also arrives as herdr notifications.
47
- - **Watch the journal, not the panes.** The append-only journal
48
- (`.tickmarkr/runs/<runId>/journal.jsonl`) is the
49
- source of truth. Arm a background watcher on `run-end` / `task-human` / `task-failed` / `consult-verdict`
50
- events; never sleep-poll inside an agent turn.
67
+ **THE ORCHESTRATOR OWNS THE LOOP. You rule and record. You do not drive.**
68
+
69
+ | | ORCHESTRATOR | OVERSEER |
70
+ |---|---|---|
71
+ | `compile` · `plan` · `run` · `resume` | **owns** | never |
72
+ | journal watchers, live surface, dialog watchers | **owns** | watches the ORCHESTRATOR, not the run |
73
+ | orphan sweeps, worker pane hygiene | **owns** | — |
74
+ | reading a gate failure and assembling its evidence | **owns** | reads the file it writes |
75
+ | **deciding** a gate, spend, or ship | never | **owns** |
76
+ | executing `tickmarkr approve` after a ruling | **owns** | never |
77
+ | git writes, the ledger, records, the operator | never | **owns** |
78
+
79
+ **Why this is a rule and not a preference — it has a measured cost.** On 2026-08-04 an OVERSEER ran the
80
+ loop itself: compile, plan, run, resume, approve, sweeps, even source fixes. The operator caught it —
81
+ *"you are doing the job of the orchestrator, the orchestrator is the one should be taking care of the
82
+ gates"* — and the receipt was a pair of numbers: **the ORCHESTRATOR sat at 369K tokens while the OVERSEER
83
+ burned 747K.** Collapsing two tiers into one does not just waste a seat; it **burns the context of the
84
+ seat that cannot be replaced cheaply**, because the orchestrator can `/clear` against a brief while the
85
+ overseer holds the mission's judgment. A tier collapse is therefore a context leak with a delay fuse.
86
+
87
+ **The tell, and you will not notice it from inside:** if you are typing `tickmarkr resume`, or reading a
88
+ journal tail to decide what happens next, or sweeping orphans — you have taken the loop. Hand it back.
89
+
90
+ ### What the ORCHESTRATOR does, and what you require of it
91
+
92
+ - **A live surface.** `tickmarkr run` is stdout-silent until run-end by design, and the run spawns its own
93
+ watch board (`role: "watch"`, one per run) — **look for that pane before building anything.** Do NOT use
94
+ `tickmarkr status --watch` as the surface: `status <runId>` has reported the WRONG run, so a board built
95
+ on it shows a previous milestone's numbers under the current run's id.
96
+ - **The journal is the source of truth**, not panes. Watchers go on `run-end` / `task-human` /
97
+ `task-failed` / `consult-verdict`; never sleep-poll inside an agent turn. **Never key a watcher on an
98
+ agent's `done`** — that is turn end and fires the moment a seat finishes acknowledging you.
51
99
  - **Daemon liveness ≠ journal activity.** A dead daemon emits no events, so journal watchers sleep through
52
- its death. `tickmarkr status` prints last-event age + daemon pid liveness; check it before diagnosing a stall.
53
- Recovery is `tickmarkr resume <runId>` — crash-safe by design (journal replay restores attempt counts and
54
- consult channel bans).
55
- - **Gate quiet ≠ idle.** Between `worker-result` and the batched `gate-result`s tickmarkr runs shell gates plus a
56
- headless LLM judge/review with little visible signal — check the journal timestamps before intervening.
57
- - **Classify gate failures before reacting.** The same fingerprint failing across DIFFERENT workers, or a
58
- scope/test catch-22 (attempt N edits a file → scope gate fails; attempt N+1 leaves it → test gate fails),
59
- is a PLAN defect: widen `files_modified` in the phase PLAN, recompile the phase dir after the run ends or
60
- the task parks, release (`human → pending`), resume. A cross-vendor review rejection with concrete findings
61
- is a REAL defect — let the escalation ladder work.
62
- - **Dialog watchers go stale per attempt.** Every retry/escalation may spawn a new pane; re-arm dialog
63
- watchers on each `task-dispatch` journal event.
100
+ its death. Liveness comes from the lock's OWN pid (`kill -0`), never a command-name grep. Recovery is
101
+ `tickmarkr resume <runId>` — **the orchestrator's command, not yours** — and note that resume REPLAYS the
102
+ journal's `baseRef`, so a fix landed on the base branch is unreachable by the running run.
103
+ - **Gate quiet ≠ idle.** Between `worker-result` and the batched `gate-result`s, shell gates plus a
104
+ headless judge/review run with little visible signal. Clock the CURRENT phase: a worker heartbeat is
105
+ stale by design once gates start, and clocking the wrong one makes a healthy gate read as a stalled
106
+ worker — the false alarm that gets a good run killed.
107
+ - **Classify gate failures before reacting.** The same fingerprint across DIFFERENT workers, or a
108
+ scope/test catch-22 (attempt N edits a file → scope fails; attempt N+1 leaves it → test fails), is a
109
+ PLAN defect. So is any blocker OUTSIDE the task's `files[]` — no worker can fix it and a retry is
110
+ knowingly wasted. A cross-vendor review rejection with concrete findings is a REAL defect; let the
111
+ ladder work.
112
+ - **Dialog watchers go stale per attempt.** Every retry may spawn a new pane; re-arm on each
113
+ `task-dispatch`.
114
+
115
+ ### What YOU do
116
+
117
+ Read the evidence file the orchestrator writes, rule on it against your pre-committed release criterion,
118
+ record the ruling with what it set aside, and hand the ruling back for execution. That is the whole job,
119
+ and it is the only work that cannot be delegated — which is exactly why nothing else should occupy you.
120
+
121
+ **And before you write the ruling: check that YOUR REMEDY is buildable inside the task's `files[]`.** The
122
+ orchestrator is told to classify a blocker outside a task's scope as a plan defect. Nothing tells the
123
+ OVERSEER that *its own instruction* can be that defect — so it arrives carrying your authority and is not
124
+ re-examined.
125
+
126
+ **Measured 2026-08-06.** A ruling directed a task to make a function parameter required. Verified
127
+ afterwards: two callers pass one argument, in a file **no task in the graph owned**. The worker would have
128
+ been trapped — edit it and fail `scope`, leave it and fail `build` — and the failure would have surfaced
129
+ as a worker defect at the next park. Worse, the unowned file was exactly the relation that task existed to
130
+ detect: **the ruling would have made the worker commit the violation the task was being built to catch.**
131
+
132
+ > Run the callers before you write the remedy: locate the symbol, **enumerate** every caller and asserting
133
+ > test, classify each against `files[]`, and resolve who owns the out-of-scope ones. A sweep for this class
134
+ > then found **five of six remaining tasks exposed**, so it is a shape, not an accident.
135
+
136
+ ### Context is a supervised resource, for BOTH tiers
137
+
138
+ Arm a context watcher on the orchestrator at spawn time and treat a threshold wake as a first-class event:
139
+ finish the step, write a handoff, `/clear` **plus a fresh brief — never `/compact`**, because a compaction
140
+ is a lossy summary nobody trusts while a clean session re-oriented from disk-verifiable state is reliable.
141
+ **Do the same for yourself before you are forced to**: write the handoff while your judgment is still
142
+ good, not after. If your own context cannot be read by the watcher, say so to the operator and ask for the
143
+ number — an unmeasured budget is not a small budget.
64
144
 
65
145
  ## Pane mechanics that bite
66
146
 
@@ -71,8 +151,12 @@ When the mission runs `/tickmarkr-auto` (tickmarkr dispatches the workers), supe
71
151
  - **Guard-before-Enter** (race-safe prompt answering): chain with `&&` — pane get shows `blocked` && pane
72
152
  read shows the expected option under the cursor && only then send-keys. If no longer `blocked`, someone
73
153
  already answered; do nothing.
74
- - `herdr wait agent-status` exits 1 on timeout, 0 on match — but ALSO 0 (with an error JSON) when the pane is
75
- GONE. Never chain `wait && act` without confirming the pane exists.
154
+ - **A dead pane accepts your dispatch and reports success.** `herdr wait agent-status` exits 1 on timeout, 0
155
+ on match — but ALSO 0 (with an error JSON) when the pane is GONE. So does `pane run`: sending to a vanished
156
+ pane prints `{"error":{"code":"pane_not_found"}}` and **still exits 0**, so `pane run … >/dev/null && echo
157
+ sent` reports a delivery that never happened. Never chain `wait && act` or trust a send's exit status —
158
+ confirm the pane exists and read it back. An orchestrator's pane can vanish mid-mission without any event
159
+ reaching you; the first symptom is a dispatch into nothing.
76
160
  - Stale typed input is unclearable via CLI — supersede it:
77
161
  `pane run "<-- disregard everything before this arrow (stale draft). ACTUAL: <message>"`.
78
162
 
@@ -90,21 +174,257 @@ orchestrator gets a 90s grace window to handle worker blocks first. For long par
90
174
  `herdr wait agent-status <pane> --status <s> --timeout <ms>` beats the watcher. When parking a human
91
175
  checkpoint, also fire `herdr notification show "HUMAN CHECKPOINT: <gate>" --sound request`.
92
176
 
177
+ **Every seat you spawn gets an ARTIFACT watcher armed in the SAME call that spawns it** — bundled, and
178
+ keyed on the deliverable rather than the seat:
179
+
180
+ ```bash
181
+ .claude/skills/tickmarkr-overseer/scripts/watch-artifacts.sh <MARKER> <cap-s> <poll-s> <file>...
182
+ ```
183
+
184
+ It wakes when every named file exists AND ends with its terminal marker, and on timeout it reports each
185
+ file as READY / PARTIAL / ABSENT so a quiet arm still proves the watcher was alive. Tell each seat, in its
186
+ brief, the exact marker its report must end with — you cannot watch for a marker you never demanded.
187
+
188
+ **Arm it in the same call as the spawn, not the next one.** A watcher armed "after I finish this step"
189
+ leaves a gap exactly as wide as however long you stay busy, and you will be busy — you just spawned work.
190
+ **Measured 2026-08-06 (OBS-369): two consult verdicts, 30KB and 12.8KB, sat COMPLETE with their markers
191
+ while the overseer hand-polled and reported them as still running. The operator had to ask.** Project
192
+ memory has carried this rule since before that session and the operator had already flagged it twice; it
193
+ was re-earned a third time because nothing in this skill made it a spawn-time step. It is one now.
194
+
195
+ **Never key a watcher on an agent's `done`.** `done` is TURN end, not mission end — a briefed seat flips to
196
+ `done` the moment it finishes acknowledging you, and a watcher waiting on "not working" wakes instantly and
197
+ reports a deliverable that does not exist. Key it on the artifact instead: the deliverable file existing **and
198
+ containing its terminal marker**, or `blocked`, or the pane being gone. Those are the three states that
199
+ actually require you. The same applies to a run: the journal's `run-end` event is the signal, never an
200
+ orchestrator turn boundary.
201
+
93
202
  ## Specialist pipeline rules
94
203
 
95
204
  - **Dedicated consultant tab**: Consultants (agents spawned to gather synthesis input for decisions like SCOPER analysis or architectural reviews) must run in a DEDICATED tab separate from the ORCHESTRATOR tab. When the orchestrator stands down, the consultant panes should persist so their assessments remain available for review and reference.
96
205
  - **Scoper worktree rule**: The SCOPER (or any worktree-based specialist synthesizing into the spec pipeline) must do ALL git operations in a dedicated worktree (e.g., `git worktree add /private/tmp/tkr-scoper-v155 -b spec/...`), never switching the main checkout's branch. This prevents race conditions between the specialist's branch operations and the orchestrator's shipping logic.
206
+ - **One fresh pane per consult ROUND** (operator rule, 2026-08-04): every consult round spawns a NEW pane in
207
+ the consult tab rather than re-prompting the seat that answered last round — unless there is a stated
208
+ reason to reuse. Two payoffs: each round starts with a clean context window instead of inheriting the
209
+ previous round's, which is what fills a long mission's seats and forces mid-mission `/clear`; and the
210
+ prior round's pane persists as a readable record of what that seat actually saw and said. Reuse only when
211
+ continuity of the seat's own reasoning is the point, and say so when you do.
212
+ - **CLOSE WHAT YOU SPAWNED** (operator observation, 2026-08-04: *"orch doesn't auto close the panes that he
213
+ created when no more needed"*). Panes accumulate silently — one mission reached **15 panes and 10 tabs**,
214
+ ten of them holding live agent sessions for work that had been on disk and fully consumed for hours.
215
+ Neither seat cleaned up, because neither had been told to.
216
+ - **Whoever spawns a pane owns closing it.** The orchestrator closes its workers; the overseer closes the
217
+ consultants and sweeps it spawned. The overseer sweeps whatever is left at mission end.
218
+ - **Verify the deliverable is ON DISK before closing** — a pane is the only place an unwritten finding
219
+ exists, and an agent that rendered "Done" without writing its artifact did not deliver (the
220
+ trust-disk-over-transcripts rule — cited by NAME, because a renumbered list orphans a "rule N").
221
+ **Non-empty is the FLOOR, not the test: completeness is the artifact's own TERMINAL MARKER.** One
222
+ consult verdict was 10KB on disk while its seat still read `working` — size proved it had started, and
223
+ only the closing `VERDICT:` line proved it had finished. Require the terminal marker a report is
224
+ supposed to end with, per seat, then close.
225
+ - **A pane is not an archive; the report is.** Once a worker's findings are written and synthesized, its
226
+ transcript adds nothing the report does not.
227
+ - **KEEP: the active seat, and the most recent SETTLED consult round.** That last one earns its place —
228
+ re-prompting the seat that found a defect to confirm its own fix is cheaper and stricter than briefing
229
+ a fresh one, which is the stated-reason exception above. Close consult rounds only once a later round
230
+ has re-derived their findings.
231
+ - Emptied tabs disappear on their own; do not close tabs by hand.
97
232
 
98
233
  ## Non-negotiable rules
99
234
 
100
- 1. **Takeover rule**: only act on a worker if it needs input AND the orchestrator is not `working`.
101
- 2. **Human checkpoints (absolute)**: any gate marked `autonomous: false` or asking for product/visual
102
- sign-off is NEVER auto-answered — regardless of how obviously correct the highlighted option looks. Leave
103
- it blocked and bring the user the decision WITH evidence. If the mission explicitly delegates authority,
104
- routine-class gates may be overseer-decided after polling the operator first — but spend and ship gates
105
- NEVER self-decide.
106
- 3. **Trust disk over transcripts**: verify artifacts on disk before building on them; a subagent killed
235
+ 1. **Do not drive the run.** The ORCHESTRATOR owns `compile`/`plan`/`run`/`resume`, the watchers, the
236
+ sweeps, and assembling gate evidence. You rule, record, and talk to the operator. Measured cost of
237
+ ignoring this: one OVERSEER at 747K tokens beside an idle ORCHESTRATOR at 369K, doing one tier's work
238
+ in the seat that cannot cheaply `/clear`. **The tell is your own hands** — typing `tickmarkr resume`,
239
+ tailing a journal to decide the next move, sweeping orphans. Hand it back.
240
+ 2. **Takeover rule**: only act on a worker if it needs input AND the orchestrator is not `working`.
241
+ 3. **Human checkpoints**: any gate marked `autonomous: false` or asking for product/visual sign-off is
242
+ NEVER auto-answered — regardless of how obviously correct the highlighted option looks. Leave it
243
+ blocked and bring the user the decision WITH evidence.
244
+ **When the mission delegates authority, the carve-out is the IRREVERSIBLE CREDENTIAL-BEARING ACT
245
+ ITSELF — not every decision upstream of it.** `npm publish`, `git tag`, a push to a public remote run
246
+ under the operator's name and account, and *"in charge" is not an npm token*. **Everything upstream is
247
+ yours**: whether a fix warrants a patch release, what rides which milestone, what to spend, what order
248
+ to ship in. Announce it with its cost basis and ACT.
249
+ **Earned 2026-08-06, at a measured price.** An OVERSEER holding a written delegation asked the operator
250
+ *"patch release ahead of the milestone, or fold it in?"* — a SCHEDULING question, no credential
251
+ anywhere near it, about a fix that did not yet exist. The operator was away **8.5 hours**. The run
252
+ ended, parked seven tasks and went unwatched; the context watcher fired and exited unread. Operator,
253
+ verbatim: *"why did you wait for me to decide earlier? you should have taken the decision your self ..
254
+ I delegated this to you, remember?"*
255
+ **The tell: if no credential, tag or public remote is touched by the ACTION you are about to take, it
256
+ is not the carve-out — decide it.** And a blocking ask is never the only option: route it through a
257
+ cross-vendor consult and rule, which is what a pre-committed release criterion already prescribes.
258
+ **A supervising seat that blocks is not neutral — it is unwatched.** Waiting has a running cost the
259
+ question never displays, and that cost lands on the run, not on the seat that waited.
260
+ 4. **Trust disk over transcripts**: verify artifacts on disk before building on them; a subagent killed
107
261
  mid-flight still renders "Done" without writing its artifact.
108
- 4. **Report concisely on every state change**: what happened, who handled it, what's next. Lead with the
109
- outcome. Surface product decisions; never make them.
110
- 5. **Log every abnormality** to `.planning/OBSERVATIONS.md` (or the project's ledger), even mid-run.
262
+ 5. **Report concisely on every state change**: what happened, who handled it, what's next. Lead with the
263
+ outcome. **Surface product decisions — and under a standing delegation, MAKE them and say you did.**
264
+ Without a delegation, surface and wait. With one, deciding IS the job; report the ruling and its basis
265
+ rather than the question. Say *"I decided"*, never *"you approved"* — a record implying a signature it
266
+ never received is this rule's own defect class running in the opposite direction.
267
+ 6. **Log every abnormality** to `.planning/OBSERVATIONS.md` (or the project's ledger), even mid-run.
268
+ 7. **Every fix is evaluated for shipping.** The tarball is `files: [dist, schema, skills, fixtures]` — so
269
+ `src/**` and `skills/**` reach users while `.overseer/**` and `.tickmarkr/**` reach nobody. Before
270
+ calling a fix done, ask where it lands: a local overlay or a scaffold script standing in for a source
271
+ fix helps ONE operator and leaves every other user with the defect. If an overlay is the interim, it
272
+ says so in writing and names its removal condition.
273
+
274
+ ---
275
+
276
+ ## Briefing a seat to audit a security-shaped check — phrasing matters
277
+
278
+ **Earned 2026-08-04.** A consult seat was asked to *"hunt one more forged pass"* on an authorization gate.
279
+ Its provider cut the session off mid-work — *"We take extra caution with cybersecurity requests"* — and the
280
+ report was never written. The seat had already found the real defect (a timezone-dependent clock) and that
281
+ finding was recovered only by reading its pane before closing it.
282
+
283
+ **The work is legitimate; the framing is what trips the filter.** Ask for completeness, not exploitation:
284
+
285
+ - ✗ "find a forged pass", "bypass this", "attack the gate", "how would you defeat it"
286
+ - ✓ **"enumerate every input this check depends on, and confirm each one is bound"**
287
+ - ✓ "which of these inputs can a caller still control?"
288
+ - ✓ "state what this check does NOT establish"
289
+
290
+ That phrasing produces the same findings — the timezone hole IS an unbound input — without asking a model to
291
+ generate an attack. **And read the pane before closing a seat that ended without its artifact:** a refusal
292
+ mid-work leaves real findings in the transcript and nowhere else, which is the one case where the pane, not
293
+ the report, is the deliverable.
294
+
295
+ ## Evidence discipline — the durable core
296
+
297
+ Distilled from a v1.86 spec-repair mission that produced 31 numbered rules, ~90 audit findings and nine
298
+ errors authored by the supervising seat itself. **Every line below was earned by a defect, most of them
299
+ twice.** They are mission-independent on purpose: nothing here names a task, a line number or a figure.
300
+
301
+ **Rot**
302
+
303
+ 1. **A quotation is exact bytes.** `grep` it before attributing it; if it does not hit, it is not a
304
+ quotation. A fabricated quotation is the only error that presents itself as primary evidence, so the
305
+ natural check is already answered on its face.
306
+ 2. **A verified quotation ROTS** — the source moves underneath it. Pin it (`as written at <sha/time>`) or
307
+ re-verify. Documenting a repair is the highest-risk case: the edit you describe is the edit that
308
+ falsifies your description.
309
+ 3. **A FINDING rots exactly like a quotation, and carries more authority while doing it** — a quotation
310
+ invites checking; *"the consultant found X"* invites action. Re-derive the premise before acting. A
311
+ *dissolved* finding gets marked SUPERSEDED, never silently dropped.
312
+ 4. **Before editing a line, sweep for the records that QUOTE it** — rule 2 used prospectively, which is the
313
+ only time it is cheap. **And the dual, from the mutating end: any repair to a CONDITION a document
314
+ DESCRIBES must sweep the descriptions in the same edit.** Fixing the world falsifies the prose about the
315
+ world, and that prose is nobody's assigned target — it is collateral. Four occurrences in one phase; twice
316
+ the fix was right and only the record was wrong, which is the version that survives review because the
317
+ change itself is defensible. **Keep the defect's record when you resolve it** — a passage that flagged a
318
+ risk which was then relied upon anyway is more instructive than a clean line saying "resolved".
319
+ 5. **Never freeze a moving number.** A figure describing anything under active edit is a quotation on a
320
+ timer; record the derivation command, not the value. A count over a population your own work adds to is
321
+ self-invalidating — state a floor.
322
+
323
+ **Scope of a result**
324
+
325
+ 6. **Every gate, tool and verdict states what it does NOT establish.** A green gate is a claim about form
326
+ until its negative scope says otherwise. This applies to a *seat's own verdict* as much as to a tool:
327
+ an unchecked cite in a task with no finding is unchecked, not confirmed.
328
+ 7. **Never aggregate per-axis PASSes into "it is clean."** Carrying the PASS and dropping the scope
329
+ manufactures a clean bill nobody issued.
330
+ 8. **Never exclude a path from a search whose purpose is to find a counterexample there** — and an
331
+ INCLUSION list excludes just as effectively, while being harder to see because every entry is
332
+ individually justified. Cite the line you actually verified.
333
+ 9. **A name-keyed sweep answers "is the name absent", not "is the concept absent."** Sweep the mechanism's
334
+ vocabulary, and one level further: sweep for what the mechanism *does to* its consumers, not who calls
335
+ it — a thing the harness *applies* to consumers is named by them in no vocabulary at all. Expect
336
+ over-return; discriminating hits is the cost of the method.
337
+ 10. **A hit proves BYTES, not attribution** — and N hits can be ONE origin copied N times.
338
+
339
+ **Instruments**
340
+
341
+ 11. **For any guard whose failure is SILENCE — detector, lint, watcher, gate, alarm branch — the acceptance
342
+ test is a POSITIVE CONTROL, not a clean run.** A zero cannot distinguish *nothing is broken* from *the
343
+ instrument is blind* from *the check does not exist*. Remove the condition it should catch and confirm
344
+ it FIRES; only then trust its quiet. **A comment asserting the check is enough to make its own author
345
+ believe it ran.**
346
+ **This rule is the oldest one here and the most re-earned.** Project memory has carried it since
347
+ 2026-07-14 as the *falsification drill* — *"eleven gate/pin defects in v1.7 alone, every one caught by a
348
+ drill rather than by a passing test"*, including a `grep -c` gate that exits 0 whether tests pass or
349
+ fail. Prefer a **compile-time guarantee** (a required parameter → a type error) over a grep-pin whenever
350
+ the choice exists: a grep-pin guarding a silent default is a hope. Read step 0a — this is what happens
351
+ when nobody opens the memory.
352
+ **Turn this rule on your own WATCHERS, because they are the guard you are least likely to aim it at.**
353
+ A journal watcher armed on `run-end`/`task-human`/`task-failed` is *supposed* to stay silent through a
354
+ clean merge — so a dead watcher and a correctly-quiet one emit byte-identical evidence from inside the
355
+ seat that owns it, and they diverge only at the first event the wake was actually for. **Watcher
356
+ liveness is proved by the PROCESS TABLE, never by its silence, and never by the report of the seat that
357
+ owns it** — "watchers alive" is the one claim a seat cannot verify about itself. Measured 2026-08-06:
358
+ an orchestrator sat `idle` through three merges and two dispatches with no journal watcher in the
359
+ process table, while its own last report read *"daemon, board, sweeper, watcher all alive"* (OBS-366).
360
+ Two corollaries: **re-arm a wake-and-exit watcher as the same turn's LAST act**, not the next turn's
361
+ first — the gap between them is unwatched and its width is however long the seat stays busy; and **a
362
+ handoff that re-arms one tier's watchers must say which tier's it did NOT re-arm.**
363
+ **A watcher has TWO failure modes, and the second is invisible from inside: never armed, and
364
+ OUTLIVING ITS TRIGGER.** A watcher aimed at an event that can no longer occur **reads as coverage and
365
+ is worse than none** — the process table shows it alive and the seat that armed it remembers arming
366
+ it. When a decision cancels the event a watcher waits on, stand it down **in the same act**. (Earned
367
+ 2026-08-06: an orchestrator did exactly this, unprompted, the moment a ruling cancelled the recompile
368
+ its standby watcher was waiting for.)
369
+ **And ASK the negative, explicitly — it is the question that produces gaps.** *"Which tier's watchers
370
+ did you NOT re-arm?"* A handoff reporting what IS armed produces a list; a handoff required to name
371
+ what is not armed produced, in one answer: a run's live surface that had **died mid-run** and was
372
+ found only in the post-run audit, and a worker-liveness tier that had **never been armed**, leaving
373
+ the daemon both the supervised thing and the sole watcher of its own workers.
374
+ 12. **Check which QUANTIFIER the claim uses before quoting a derivation for it.** "The path is N" needs a
375
+ maximum; *"both chains"*, *"the only consumer"*, *"exactly one owner"* need an ENUMERATION — and a
376
+ max-with-tie-break silently answers the first question when you asked the second.
377
+ 13. **Derive mechanically; hand-derived sets are wrong.** Prefer the real parser's output over a
378
+ re-implementation, and a property over an enumeration. State what the mechanism cannot establish.
379
+ **And make every instrument PRINT WHAT IT ACTED ON.** A tool that does not name its target cannot be
380
+ caught answering about the wrong thing: a dry-compile helper that silently ignored its path argument
381
+ returned the same verdict for two different files, and a seat read one answer as two results and
382
+ concluded both forms were valid. The tell is unavailable unless the tool volunteers it. Corollary —
383
+ an instrument that takes an input must be handed a DELIBERATELY BAD one before its clean runs are
384
+ worth anything (rule 11 applied to tools, not just to gates).
385
+ 14. **A unit is not a measurement.** A configured timeout is a KILL CEILING, not a duration — never compare
386
+ it to a wall clock or quote it to an operator as an estimate.
387
+ 15. **Verify through the path that LOADS, not the path you edited.** Mirrored trees and symlinks mean your
388
+ check can confirm a shadow copy; `sed -i` on a tracked symlink silently replaces it with a regular file.
389
+ 16. **Never edit a script with a live instance** — bash reads by byte offset, so even a comment-only
390
+ insertion corrupts the running process. Cancel, edit, syntax-check, re-arm, in that order.
391
+
392
+ **Propagation**
393
+
394
+ 17. **A confirmed single-site or single-axis miss is a CLASS, not an instance.** Re-run the same sweep shape
395
+ on every sibling; ask what other dimension the fixtures hold constant. **A ruling that fixes one
396
+ instance of a class it just defined is incomplete by construction** — the sweep is part of the ruling.
397
+ 18. **When a boundary moves, every clause referencing the old boundary moves with it.** Neither clause ever
398
+ looks wrong alone, so single-clause review cannot catch this class. Disambiguate any term used at two
399
+ levels.
400
+ 19. **A METHOD GUARD found by one seat must be promoted to where every seat reads it, in the same pass that
401
+ reads it.** Otherwise it is re-earned at full price — and the second earning is worse, because by then
402
+ the wrong answer carries a citation.
403
+
404
+ **Authority**
405
+
406
+ 20. **Open the file the instruction is about, even when the instruction comes from above.** A ruling reads
407
+ as settled, and that is exactly when it goes unchecked. Overseer rulings are wrong at roughly the rate
408
+ of everyone else's.
409
+ 21. **State the verification standard alongside the instruction**, or the defect appears at the seam.
410
+ 22. **An overclaimed self-criticism is the least-audited sentence you will write** — a harsh line invites no
411
+ check, so it ships unverified. Including in a section like this one.
412
+ 23. **A pre-commitment needs a TRIGGER and a SUBJECT SET. Naming only the trigger is a live hazard.**
413
+ A bound was rewritten mid-mission to make its CONDITION mechanically checkable — and that rewrite was
414
+ already the product of one near-miss. It was **still** incomplete: a third task later parked with
415
+ **both halves of the condition present**, and the bound did not apply, because that task was not in
416
+ its subject set. A seat reading only the trigger would have closed the milestone on a task the
417
+ pre-commitment was never about. **State both: what fires it, and what it is ABOUT.** A correct trigger
418
+ with an unstated subject executes on the first thing matching its shape, carrying the authority of the
419
+ decision it was written for.
420
+ 24. **A REMEDIATION is believed where a guard would be drilled.** Rule 11 says a guard whose failure is
421
+ silence needs a positive control. **Nobody applies that to a FIX**, because a fix is not an
422
+ instrument — so a shipped remediation is remembered as coverage and never re-read. One was recalled as
423
+ *"the reaper shipped in v1.78"*; its own changelog said it reaps only what a helper **tracks**,
424
+ *"without a broad migration"*, and measurement found the untracked majority behaving exactly as it had
425
+ been left. **Ask of any remembered fix: what did it actually cover, in its own words, at the time?**
426
+ 25. **The fabrication lives in the INDEX line, not the body — and the index is what everyone loads.**
427
+ In that same case the memory body said *"expect regrowth until the product fix ships."* The one-line
428
+ summary said *"reaper shipped."* **The accurate body was never opened, because the index had already
429
+ answered the question.** Audit index and summary lines against the bodies they point at; a compression
430
+ that drops a qualifier is indistinguishable from a fact.
@@ -0,0 +1,70 @@
1
+ #!/bin/bash
2
+ # Wake the overseer when spawned seats have actually DELIVERED.
3
+ #
4
+ # watch-artifacts.sh <marker> <cap-seconds> <poll-seconds> <file>...
5
+ #
6
+ # Completion is the ARTIFACT plus its TERMINAL MARKER — never an agent's `done`, which is turn end and
7
+ # fires the moment a seat finishes acknowledging you. Never file existence alone either: a seat killed
8
+ # mid-write leaves a large, plausible, truncated file. Size proves it started; the marker proves it
9
+ # finished. Measured 2026-08-06: one 10KB verdict was on disk while its seat still read `working`, and two
10
+ # 30KB/12.8KB verdicts sat COMPLETE for minutes with nothing watching, found only because the operator
11
+ # asked (OBS-369).
12
+ #
13
+ # ARM THIS IN THE SAME CALL THAT SPAWNS THE SEAT. A watcher armed later leaves an unwatched gap exactly as
14
+ # wide as however long you stay busy — and you will be busy, because you just spawned work.
15
+ #
16
+ # Prints one wake reason and EXITS. Re-arm after every wake.
17
+ set -u
18
+ # macOS ships bash 3.2, where `set -u` makes "${arr[@]}" on an EMPTY array a fatal unbound-variable
19
+ # error. Every expansion below therefore uses the ${arr[@]+"${arr[@]}"} guard. Caught by the timeout
20
+ # control, not by the completion one: the happy path was green while the path that runs on almost every
21
+ # arm crashed before printing its reason — a watcher that dies without a wake reason is the exact failure
22
+ # this script exists to prevent.
23
+
24
+ MARKER="${1:?usage: watch-artifacts.sh <marker> <cap-seconds> <poll-seconds> <file>...}"
25
+ CAP="${2:?}"
26
+ POLL="${3:?}"
27
+ shift 3
28
+ [ "$#" -gt 0 ] || { echo "watch-artifacts: no files given" >&2; exit 2; }
29
+
30
+ # Cap BELOW the host's background-job kill so every arm ends by PRINTING something. A job killed at the
31
+ # limit carries no wake reason and is indistinguishable from a real wake until you read the output —
32
+ # OBS-325, re-earned by a third watcher that had not been capped because the fix was applied only to the
33
+ # two that happened to be in view at the time. Class, not instance.
34
+ END=$((SECONDS + CAP))
35
+
36
+ # A file is DONE when the marker appears in its last few lines. Anchored to the tail on purpose: a report
37
+ # that merely *mentions* its own marker mid-body has not finished, and grepping the whole file would call
38
+ # that done. This brief tells seats to end the file with the marker, so the tail is where it must be.
39
+ done_file() {
40
+ [ -s "$1" ] || return 1
41
+ tail -5 "$1" 2>/dev/null | grep -qF -- "$MARKER"
42
+ }
43
+
44
+ while :; do
45
+ pending=()
46
+ ready=()
47
+ for f in "$@"; do
48
+ if done_file "$f"; then ready+=("$f"); else pending+=("$f"); fi
49
+ done
50
+
51
+ if [ "${#pending[@]}" -eq 0 ]; then
52
+ echo "WAKE: all ${#ready[@]} artifact(s) complete with marker '$MARKER'"
53
+ for f in ${ready[@]+"${ready[@]}"}; do echo " READY $(wc -c <"$f" | tr -d ' ') bytes $f"; done
54
+ exit 0
55
+ fi
56
+
57
+ if [ "$SECONDS" -ge "$END" ]; then
58
+ # Timing out is NOT failure and must not read as one: it is the heartbeat that proves the watcher was
59
+ # alive and still watching. Report both sides so the state is unambiguous on arrival.
60
+ echo "WAKE: cap ${CAP}s reached — ${#ready[@]} of $# complete, re-arm"
61
+ for f in ${ready[@]+"${ready[@]}"}; do echo " READY $(wc -c <"$f" | tr -d ' ') bytes $f"; done
62
+ for f in ${pending[@]+"${pending[@]}"}; do
63
+ if [ -s "$f" ]; then echo " PARTIAL $(wc -c <"$f" | tr -d ' ') bytes, no '$MARKER' yet $f"
64
+ else echo " ABSENT $f"; fi
65
+ done
66
+ exit 0
67
+ fi
68
+
69
+ sleep "$POLL"
70
+ done
@@ -28,7 +28,7 @@ WORKER="${1:?worker pane id required}"
28
28
  ORCH="${2:?orchestrator pane id required}"
29
29
  shift 2
30
30
  FAST_BLOCKED=false
31
- CAP=14400
31
+ CAP=240 # a backgrounded Bash job is killed at 300s; a longer cap can never be reached (OBS-325)
32
32
  SETTLE=60
33
33
  while [ $# -gt 0 ]; do
34
34
  case "$1" in
@@ -1,59 +0,0 @@
1
- import { type RunGraph } from "../../graph/schema.js";
2
- import { type JournalEvent } from "../../run/journal.js";
3
- import { FleetStaging } from "../staging.js";
4
- export type StudioInkView = {
5
- id: string;
6
- label: string;
7
- render(props: {
8
- cols: number;
9
- rows: number;
10
- }): string[];
11
- key?(name: string): void;
12
- };
13
- export type StudioInkIO = {
14
- input: NodeJS.ReadStream;
15
- output: NodeJS.WriteStream;
16
- views?: StudioInkView[];
17
- repoRoot?: string;
18
- globalDir?: string;
19
- staging?: FleetStaging;
20
- runsData?: RunsCockpitData;
21
- clock?: () => number;
22
- debug?: boolean;
23
- };
24
- export type RunsCockpitData = {
25
- runId?: string;
26
- events: JournalEvent[];
27
- graph: RunGraph;
28
- prompts?: Record<string, string[]>;
29
- };
30
- type DossierVerdict = {
31
- action: "retry" | "reroute" | "decompose" | "human";
32
- reason?: string;
33
- prompt?: string;
34
- };
35
- type ConsultDossierData = {
36
- taskId: string;
37
- verdicts: DossierVerdict[];
38
- };
39
- type RunsCockpitTask = {
40
- id: string;
41
- status: string;
42
- phase?: string;
43
- elapsed?: string;
44
- consultCount: number;
45
- };
46
- export declare function foldConsultDossier(data: RunsCockpitData, taskId: string): ConsultDossierData;
47
- export declare function buildRunsCockpitTasks(data: RunsCockpitData, nowMs: number): RunsCockpitTask[];
48
- export declare function StudioApp({ views: suppliedViews, repoRoot, globalDir, staging: suppliedStaging, runsData: suppliedRunsData, clock, columns, rows, }: {
49
- views?: StudioInkView[];
50
- repoRoot?: string;
51
- globalDir?: string;
52
- staging?: FleetStaging;
53
- runsData?: RunsCockpitData;
54
- clock?: () => number;
55
- columns?: number;
56
- rows?: number;
57
- }): import("react").JSX.Element;
58
- export declare function runStudioInk({ input, output, views, repoRoot, globalDir, staging, runsData, clock, debug, }: StudioInkIO): Promise<void>;
59
- export {};