tickmarkr 2.1.3 → 2.1.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/commands/status.d.ts +3 -0
- package/dist/cli/commands/status.js +23 -1
- package/dist/compile/collateral.d.ts +23 -2
- package/dist/compile/collateral.js +126 -11
- package/dist/compile/gsd.js +21 -1
- package/dist/compile/native.js +35 -3
- package/dist/gates/baseline.d.ts +23 -3
- package/dist/gates/baseline.js +75 -16
- package/dist/gates/llm.d.ts +1 -0
- package/dist/gates/llm.js +1 -0
- package/dist/gates/review.d.ts +9 -2
- package/dist/gates/review.js +51 -10
- package/dist/gates/run-gates.js +12 -1
- package/dist/run/daemon.d.ts +0 -1
- package/dist/run/daemon.js +158 -27
- package/dist/run/git.d.ts +65 -0
- package/dist/run/git.js +120 -8
- package/dist/run/journal.d.ts +47 -0
- package/dist/run/journal.js +156 -13
- package/dist/run/merge.js +13 -3
- package/dist/run/supervision.d.ts +1 -1
- package/dist/run/supervision.js +126 -35
- package/fixtures/gsd-sample/07-live-check/07-03-SUMMARY.md +4 -0
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +108 -2
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +64 -6
package/dist/run/supervision.js
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
|
-
import { mkdirSync, readFileSync, renameSync, rmSync, statSync, writeFileSync } from "node:fs";
|
|
1
|
+
import { mkdirSync, readdirSync, readFileSync, renameSync, rmSync, statSync, writeFileSync, } from "node:fs";
|
|
2
|
+
import { randomUUID } from "node:crypto";
|
|
2
3
|
import { dirname, join } from "node:path";
|
|
3
4
|
import { stateDirName, tickmarkrDir } from "../graph/graph.js";
|
|
4
5
|
// SUP-01: supervision liveness as FILE STATE, not as a report — lock.ts's proven shape, one file per
|
|
@@ -32,8 +33,9 @@ export const SUPERVISION_FUTURE_GRACE_MS = 1_000;
|
|
|
32
33
|
// SUP-05: CONTEXT is per SEAT, so its tiers are per seat too. One shared `context` tier would be read
|
|
33
34
|
// by both supervising seats and beaten by whichever of them still had a watcher, so a live overseer
|
|
34
35
|
// watcher would render the dead orchestrator one as armed — the mask this whole instrument exists to
|
|
35
|
-
// remove. The enumeration is CLOSED at one tier
|
|
36
|
-
// `watch`
|
|
36
|
+
// remove. The enumeration is CLOSED at one supervision tier and one context tier per supervising
|
|
37
|
+
// seat: orchestrator and overseer. `watch` is the sole process-owned tier and is armed seatlessly by
|
|
38
|
+
// the live unbounded board.
|
|
37
39
|
export const SUPERVISION_TIERS = [
|
|
38
40
|
"orchestrator", "orchestrator-context", "overseer", "overseer-context", "watch",
|
|
39
41
|
];
|
|
@@ -42,18 +44,59 @@ export const SUPERVISION_TIERS = [
|
|
|
42
44
|
// to a seat. Measured 2026-08-26: a consult seat of another tier ran the documented beat loop and the
|
|
43
45
|
// board read that tier armed with no seat of that tier having armed anything. ARMED-and-seatless reads
|
|
44
46
|
// as coverage, which is worse than ABSENT, so on these tiers a record that names no seat is not a beat.
|
|
45
|
-
export const SUPERVISION_SEAT_TIERS = [
|
|
47
|
+
export const SUPERVISION_SEAT_TIERS = [
|
|
48
|
+
"orchestrator", "orchestrator-context", "overseer", "overseer-context",
|
|
49
|
+
];
|
|
46
50
|
/** Does this tier's record have to name the seat it speaks for? */
|
|
47
51
|
export const isSeatTier = (tier) => SUPERVISION_SEAT_TIERS.includes(tier);
|
|
48
52
|
// PURE path math: stateDirName, never tickmarkrDir — the latter mkdirs the state dir and writes its
|
|
49
53
|
// .gitignore, so routing a READER through it would make status create the very tree it reports on.
|
|
50
|
-
|
|
54
|
+
const supervisionDir = (repoRoot) => join(repoRoot, stateDirName(repoRoot), "supervision");
|
|
55
|
+
export const supervisionBeatPath = (repoRoot, tier) => join(supervisionDir(repoRoot), `${tier}.beat`);
|
|
51
56
|
/** Where a watcher records that it STOOD DOWN. Its own file: the beat keeps meaning only "alive". */
|
|
52
|
-
export const supervisionStandDownPath = (repoRoot, tier) => join(
|
|
57
|
+
export const supervisionStandDownPath = (repoRoot, tier) => join(supervisionDir(repoRoot), `${tier}.standdown`);
|
|
58
|
+
// SUP-06: PRESENCE — one file per ARMED WATCHER, because a tier may legitimately have more than one.
|
|
59
|
+
// Two boards watch one repo the moment an operator opens a second pane, and the tier is armed while
|
|
60
|
+
// EITHER of them lives. The stand-down marker speaks for the whole tier, so the first board out
|
|
61
|
+
// writing it renders the second board's own tier DISARMED until that board's next beat — a live seat
|
|
62
|
+
// reported down, which is the under-claiming half of exactly the lie this instrument exists to remove.
|
|
63
|
+
// So the marker is written only by the LAST watcher out, and these files are how it knows it is last.
|
|
64
|
+
// Freshness decides presence, never a process table (SUP-02): a watcher that is killed cannot remove
|
|
65
|
+
// its own file, and an unremoved file ages past the same ceiling a beat does and stops counting.
|
|
66
|
+
const presencePrefix = (tier) => `${tier}.live.`;
|
|
67
|
+
const supervisionPresencePath = (repoRoot, tier, id) => join(supervisionDir(repoRoot), `${presencePrefix(tier)}${id}`);
|
|
68
|
+
/** Every presence file on this tier, by name. Missing directory ⇒ nobody is present. */
|
|
69
|
+
const presenceNames = (repoRoot, tier) => {
|
|
70
|
+
try {
|
|
71
|
+
return readdirSync(supervisionDir(repoRoot)).filter((n) => n.startsWith(presencePrefix(tier)));
|
|
72
|
+
}
|
|
73
|
+
catch {
|
|
74
|
+
return [];
|
|
75
|
+
}
|
|
76
|
+
};
|
|
77
|
+
/**
|
|
78
|
+
* Stale peers observed in ONE directory snapshot, or undefined when that same snapshot saw a live
|
|
79
|
+
* one. A later arm has a new id and is deliberately absent from the returned cleanup set.
|
|
80
|
+
*/
|
|
81
|
+
function stalePeersIfLast(repoRoot, tier, id, now = Date.now()) {
|
|
82
|
+
const own = `${presencePrefix(tier)}${id}`;
|
|
83
|
+
const stale = [];
|
|
84
|
+
for (const name of presenceNames(repoRoot, tier)) {
|
|
85
|
+
if (name === own)
|
|
86
|
+
continue;
|
|
87
|
+
try {
|
|
88
|
+
if (now - statSync(join(supervisionDir(repoRoot), name)).mtimeMs <= SUPERVISION_STALE_MS)
|
|
89
|
+
return undefined;
|
|
90
|
+
stale.push(name);
|
|
91
|
+
}
|
|
92
|
+
catch { /* a vanished peer needs no cleanup and is not evidence of a live watcher */ }
|
|
93
|
+
}
|
|
94
|
+
return stale;
|
|
95
|
+
}
|
|
53
96
|
// WRITER — a watcher's own call, on its own tier, every SUPERVISION_BEAT_MS. Never a reader's: the
|
|
54
97
|
// purity fence (status --watch leaves the state dir byte-identical) is the test that catches a reader
|
|
55
98
|
// that beats on the watcher's behalf, which would report every dead tier as healthy.
|
|
56
|
-
|
|
99
|
+
function writeSupervisionBeat(repoRoot, tier, seat, armId) {
|
|
57
100
|
// A seat tier may not be armed anonymously, and the refusal belongs HERE rather than only in the
|
|
58
101
|
// verb: any caller that could write a seatless record could arm a tier nobody occupies.
|
|
59
102
|
if (isSeatTier(tier) && !seat?.trim()) {
|
|
@@ -62,7 +105,13 @@ export function beatSupervision(repoRoot, tier, seat) {
|
|
|
62
105
|
tickmarkrDir(repoRoot); // the write path DOES create — beats land inside the gitignored state dir
|
|
63
106
|
const p = supervisionBeatPath(repoRoot, tier);
|
|
64
107
|
mkdirSync(dirname(p), { recursive: true });
|
|
65
|
-
writeFileSync(p, JSON.stringify({
|
|
108
|
+
writeFileSync(p, JSON.stringify({
|
|
109
|
+
tier, ...(seat ? { seat } : {}), ...(armId ? { armId } : {}),
|
|
110
|
+
pid: process.pid, beatAt: new Date().toISOString(),
|
|
111
|
+
}) + "\n");
|
|
112
|
+
}
|
|
113
|
+
export function beatSupervision(repoRoot, tier, seat) {
|
|
114
|
+
writeSupervisionBeat(repoRoot, tier, seat);
|
|
66
115
|
}
|
|
67
116
|
// THE WATCHER-FACING ENTRY POINT — the loop SUPERVISION_BEAT_MS actually drives. A supervising seat
|
|
68
117
|
// calls this once at the top of its watch and holds the handle for the duration; a seat that dies,
|
|
@@ -71,9 +120,11 @@ export function beatSupervision(repoRoot, tier, seat) {
|
|
|
71
120
|
// an unref'd interval that never holds the watcher's event loop open, and a beat failure that is
|
|
72
121
|
// swallowed rather than crashing the watcher — an unwritten beat ages out and reads STALE, which is
|
|
73
122
|
// the truth. The FIRST beat is swallowed on the same rule: a cosmetic instrument that could not write
|
|
74
|
-
// must not take the
|
|
75
|
-
//
|
|
76
|
-
//
|
|
123
|
+
// must not take the watcher down with it. The sole production in-repo callsite is `status --watch`
|
|
124
|
+
// when UNBOUNDED, which arms `watch` seatlessly for the life of the board — a bounded render is a
|
|
125
|
+
// reader and arms nothing, which is the purity fence D-02 tests. Supervising seats write their named
|
|
126
|
+
// tiers through the shipped beat verb instead. A tier nobody arms reads ABSENT — exactly what ABSENT
|
|
127
|
+
// means, not a false "healthy".
|
|
77
128
|
//
|
|
78
129
|
// Arming CLEARS any prior stand-down record: a tier that stood down and armed again is armed, and a
|
|
79
130
|
// marker left behind by the last run would otherwise report the live one as stood down forever.
|
|
@@ -86,40 +137,64 @@ export function armSupervision(repoRoot, tier, beatMs = SUPERVISION_BEAT_MS, sea
|
|
|
86
137
|
rmSync(supervisionStandDownPath(repoRoot, tier), { force: true, recursive: true });
|
|
87
138
|
}
|
|
88
139
|
catch { /* uncleared: the reader validates the marker and a newer beat outranks it — never masked */ }
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
const
|
|
140
|
+
// This watcher's own identity fences BOTH its presence and its stand-down against every later arm.
|
|
141
|
+
// pid alone collides between two boards in one host (and after pid reuse); a UUID never aliases the
|
|
142
|
+
// stale presence of a killed process that a later last-one-out cleanup may already have observed.
|
|
143
|
+
const id = `${process.pid}.${randomUUID()}`;
|
|
144
|
+
const presence = supervisionPresencePath(repoRoot, tier, id);
|
|
145
|
+
// Presence is refreshed with the beat, so it ages by the same clock and needs no separate loop.
|
|
146
|
+
const mark = () => {
|
|
94
147
|
try {
|
|
95
|
-
|
|
148
|
+
writeSupervisionBeat(repoRoot, tier, seat, id); // creates the directory presence is written into
|
|
149
|
+
writeFileSync(presence, JSON.stringify({ tier, pid: process.pid, id }) + "\n");
|
|
96
150
|
}
|
|
97
|
-
catch { /* repo gone / disk full —
|
|
98
|
-
}
|
|
151
|
+
catch { /* repo gone / disk full / no seat — the tier ages out rather than crashing its watcher */ }
|
|
152
|
+
};
|
|
153
|
+
mark();
|
|
154
|
+
const timer = setInterval(mark, beatMs);
|
|
99
155
|
timer.unref();
|
|
100
156
|
let stoodDown = false;
|
|
101
157
|
return {
|
|
102
|
-
// Stand down: stop beating AND say so.
|
|
103
|
-
// exit
|
|
104
|
-
// instant belongs to the first stand-down.
|
|
158
|
+
// Stand down: stop beating AND say so. Idempotence lets a watcher safely share cleanup across
|
|
159
|
+
// multiple exit paths; the recorded instant belongs to the first stand-down.
|
|
105
160
|
disarm: () => {
|
|
106
161
|
if (stoodDown)
|
|
107
162
|
return;
|
|
108
163
|
stoodDown = true;
|
|
109
164
|
clearInterval(timer);
|
|
165
|
+
try {
|
|
166
|
+
rmSync(presence, { force: true, recursive: true });
|
|
167
|
+
}
|
|
168
|
+
catch { /* ages out on its own */ }
|
|
169
|
+
// The marker speaks for the TIER, so only the last watcher out may write one: a peer still
|
|
170
|
+
// present means the tier is not down, and saying it is would render that live board's own tier
|
|
171
|
+
// DISARMED. The snapshot also fixes the cleanup set: a board arming after this decision receives
|
|
172
|
+
// a new id, so this older board can neither sweep its presence nor claim its beat stood down.
|
|
173
|
+
const stalePeers = stalePeersIfLast(repoRoot, tier, id);
|
|
174
|
+
if (stalePeers === undefined)
|
|
175
|
+
return;
|
|
110
176
|
// Published ATOMICALLY — written aside, renamed over — so no reader can ever meet a half-written
|
|
111
177
|
// marker. A torn marker is rejected anyway (see readStandDown), but a stand-down that reads as
|
|
112
178
|
// garbage is a stand-down that reports as a death, and the rename costs one line.
|
|
113
179
|
const p = supervisionStandDownPath(repoRoot, tier);
|
|
114
|
-
const tmp = `${p}.${
|
|
180
|
+
const tmp = `${p}.${id}.tmp`;
|
|
115
181
|
try {
|
|
116
182
|
mkdirSync(dirname(p), { recursive: true });
|
|
117
183
|
writeFileSync(tmp, JSON.stringify({
|
|
118
|
-
tier, ...(seat ? { seat } : {}),
|
|
184
|
+
tier, ...(seat ? { seat } : {}), armId: id,
|
|
185
|
+
pid: process.pid, disarmedAt: new Date().toISOString(),
|
|
119
186
|
}) + "\n");
|
|
120
187
|
renameSync(tmp, p);
|
|
121
188
|
}
|
|
122
189
|
catch { /* unrecordable stand-down ages out as STALE — pessimistic, which is the safe way to fail */ }
|
|
190
|
+
// Sweep only stale names in the pre-publication snapshot. Re-reading here used to catch and
|
|
191
|
+
// delete a newer board that armed between the peer check and this older board's rename.
|
|
192
|
+
for (const name of stalePeers) {
|
|
193
|
+
try {
|
|
194
|
+
rmSync(join(supervisionDir(repoRoot), name), { force: true, recursive: true });
|
|
195
|
+
}
|
|
196
|
+
catch { /* next sweep */ }
|
|
197
|
+
}
|
|
123
198
|
},
|
|
124
199
|
};
|
|
125
200
|
}
|
|
@@ -153,20 +228,26 @@ function readBeat(repoRoot, tier) {
|
|
|
153
228
|
// is not evidence that nobody armed the tier). On a SEAT tier it is the opposite: a record naming no
|
|
154
229
|
// seat leaves the tier armed and unattributable, which reads as coverage no seat is providing, so
|
|
155
230
|
// it is UNREADABLE — something is there and no beat any reader can attribute comes out of it.
|
|
156
|
-
const seat =
|
|
231
|
+
const { seat, armId } = beatMetadata(p);
|
|
157
232
|
if (isSeatTier(tier) && seat === undefined)
|
|
158
233
|
return "UNREADABLE";
|
|
159
|
-
return {
|
|
234
|
+
return {
|
|
235
|
+
mtimeMs: st.mtimeMs,
|
|
236
|
+
...(seat !== undefined ? { seat } : {}),
|
|
237
|
+
...(armId !== undefined ? { armId } : {}),
|
|
238
|
+
};
|
|
160
239
|
}
|
|
161
|
-
/**
|
|
162
|
-
function
|
|
240
|
+
/** Optional metadata declared by a beat; its mtime remains the only source of age. */
|
|
241
|
+
function beatMetadata(path) {
|
|
163
242
|
try {
|
|
164
243
|
const rec = JSON.parse(readFileSync(path, "utf8"));
|
|
165
|
-
|
|
244
|
+
const seat = typeof rec?.seat === "string" && rec.seat.trim() ? rec.seat : undefined;
|
|
245
|
+
const armId = typeof rec?.armId === "string" && rec.armId.trim() ? rec.armId : undefined;
|
|
246
|
+
return { ...(seat !== undefined ? { seat } : {}), ...(armId !== undefined ? { armId } : {}) };
|
|
166
247
|
}
|
|
167
248
|
catch {
|
|
168
|
-
return
|
|
169
|
-
} // unparseable bytes name no seat — the caller decides what that means
|
|
249
|
+
return {};
|
|
250
|
+
} // unparseable bytes name no seat or arm — the caller decides what that means
|
|
170
251
|
}
|
|
171
252
|
function beatLiveness(tier, beat, now) {
|
|
172
253
|
if (typeof beat !== "object")
|
|
@@ -203,6 +284,7 @@ function readStandDown(repoRoot, tier) {
|
|
|
203
284
|
if (!st.isFile())
|
|
204
285
|
return "UNREADABLE";
|
|
205
286
|
let seat;
|
|
287
|
+
let armId;
|
|
206
288
|
try {
|
|
207
289
|
const rec = JSON.parse(readFileSync(p, "utf8"));
|
|
208
290
|
if (rec?.tier !== tier)
|
|
@@ -210,6 +292,7 @@ function readStandDown(repoRoot, tier) {
|
|
|
210
292
|
if (typeof rec.disarmedAt !== "string" || Number.isNaN(Date.parse(rec.disarmedAt)))
|
|
211
293
|
return "UNREADABLE";
|
|
212
294
|
seat = typeof rec.seat === "string" && rec.seat.trim() ? rec.seat : undefined;
|
|
295
|
+
armId = typeof rec.armId === "string" && rec.armId.trim() ? rec.armId : undefined;
|
|
213
296
|
// A seat tier's hand-off names WHICH seat left, on the same rule as its beat: an anonymous
|
|
214
297
|
// stand-down on a per-seat tier says a watcher left without saying whose, so it is no record.
|
|
215
298
|
if (isSeatTier(tier) && seat === undefined)
|
|
@@ -218,18 +301,26 @@ function readStandDown(repoRoot, tier) {
|
|
|
218
301
|
catch {
|
|
219
302
|
return "UNREADABLE";
|
|
220
303
|
} // unparseable or unreadable bytes — not a stand-down anyone can read
|
|
221
|
-
return {
|
|
304
|
+
return {
|
|
305
|
+
mtimeMs: st.mtimeMs,
|
|
306
|
+
...(seat !== undefined ? { seat } : {}),
|
|
307
|
+
...(armId !== undefined ? { armId } : {}),
|
|
308
|
+
};
|
|
222
309
|
}
|
|
223
310
|
// THE TIER'S STATE — what every surface and every operator reads. A valid stand-down outranks the beat:
|
|
224
311
|
// the watcher that wrote it is gone ON PURPOSE, and its last beat ages out exactly like a dead one's
|
|
225
|
-
// would. It outranks the beat it FOLLOWED and no other — a
|
|
226
|
-
//
|
|
312
|
+
// would. It outranks the beat it FOLLOWED and no other — a later timestamp OR a FRESH different
|
|
313
|
+
// armed-watcher identity is another arm, so a marker whose rename lost that race cannot mask a live
|
|
314
|
+
// watcher. Once that foreign beat is stale, a newer clean hand-off must win: otherwise overlapping
|
|
315
|
+
// boards closed in last-beater-first order would leave the tier reporting a death forever.
|
|
227
316
|
export function supervisionStatus(repoRoot, tier, now = Date.now()) {
|
|
228
317
|
const beat = readBeat(repoRoot, tier);
|
|
229
318
|
const standDown = readStandDown(repoRoot, tier);
|
|
230
319
|
if (standDown === "UNREADABLE")
|
|
231
320
|
return { tier, state: "UNREADABLE" };
|
|
232
|
-
|
|
321
|
+
const beatOutranksStandDown = standDown !== "NONE" && typeof beat === "object" && (beat.mtimeMs > standDown.mtimeMs || (now - beat.mtimeMs <= SUPERVISION_STALE_MS &&
|
|
322
|
+
beat.armId !== undefined && standDown.armId !== undefined && beat.armId !== standDown.armId));
|
|
323
|
+
if (standDown !== "NONE" && !beatOutranksStandDown) {
|
|
233
324
|
// the seat that stood down is named by the marker, falling back to whatever its last beat named
|
|
234
325
|
const seat = standDown.seat ?? (typeof beat === "object" ? beat.seat : undefined);
|
|
235
326
|
return { tier, state: "DISARMED", ...(seat !== undefined ? { seat } : {}) };
|
package/package.json
CHANGED
|
@@ -249,6 +249,111 @@ detect: **the ruling would have made the worker commit the violation the task wa
|
|
|
249
249
|
|
|
250
250
|
### Context is a supervised resource, for BOTH tiers
|
|
251
251
|
|
|
252
|
+
**THE TIERS CLEAR EACH OTHER AT 50%. Neither tier clears itself on its own notice.** Operator directive,
|
|
253
|
+
2026-08-28, and it exists because **a seat cannot reliably observe its own exhaustion** — the seat that
|
|
254
|
+
most needs clearing is the one least able to notice, and this project has now measured that three ways:
|
|
255
|
+
an overseer ran nine hours at 86% unable to read its own number; a context watcher went **alive and blind**
|
|
256
|
+
when the run's own status text pushed the percentage off the statusline; and an orchestrator went
|
|
257
|
+
**366k → 970k of 1M between two checks** while its ACT wake sat unread in a detached log.
|
|
258
|
+
|
|
259
|
+
The protocol, in both directions:
|
|
260
|
+
|
|
261
|
+
1. **Overseer sees orch at ≥50%** → nudge it: write `HANDOFF-ORCH-<ver>.md`, then `/clear`, then re-read
|
|
262
|
+
its brief **and** its handoff, then **re-arm every watcher it listed** (a cleared session has none).
|
|
263
|
+
2. **The returning orch, now fresh, checks the OVERSEER.** If the overseer is at ≥50%, it directs the
|
|
264
|
+
overseer to write its handoff and clear, and **points it at `HANDOFF-OVERSEER-<ver>.md` by path**.
|
|
265
|
+
3. Whichever seat is fresh performs the check. **Never both at once** — the run keeps one supervising tier
|
|
266
|
+
at all times, and the seat holding the endgame goes second.
|
|
267
|
+
4. **The duty to clear the other tier must SURVIVE a clear**, so it belongs in BOTH handoff files as a
|
|
268
|
+
standing re-arm item — not only in the message that ordered it. Earned 2026-08-28: the orchestrator
|
|
269
|
+
performed the check on the overseer BEFORE its own clear, precisely because clearing first would have
|
|
270
|
+
wiped the instruction to do it, and its handoff did not record the duty.
|
|
271
|
+
5. **A seat cannot `/clear` ITSELF — so the OTHER TIER SENDS IT.** `/clear` is a CLI command typed into a
|
|
272
|
+
session and no tool invokes it in your own pane, but it is just text in someone else's: the partner
|
|
273
|
+
tier types it into your pane. **This is the whole reason the protocol is mutual**, and it means the
|
|
274
|
+
loop closes without the operator. The exchange, both directions, in this exact order:
|
|
275
|
+
|
|
276
|
+
```bash
|
|
277
|
+
# 1. the seat crossing 50% writes its handoff FIRST, ending with its terminal marker
|
|
278
|
+
# 2. it asks the partner, naming its own pane and handoff path:
|
|
279
|
+
herdr pane run <partner> "I am at <N>%. Clear me: send /clear to <my-pane>, then point me at <my-handoff>."
|
|
280
|
+
# 3. the PARTNER sends the clear, then VERIFIES before pointing:
|
|
281
|
+
herdr pane run <my-pane> "/clear"
|
|
282
|
+
# read the pane back — a cleared claude session shows an empty prompt and a reset context gauge
|
|
283
|
+
# 4. and only THEN, as a SEPARATE send, the re-orientation:
|
|
284
|
+
herdr pane run <my-pane> "You were cleared at <N>%. Read <handoff> and <brief>, re-arm EVERY watcher
|
|
285
|
+
they name — a cleared session has none — then confirm you are back."
|
|
286
|
+
```
|
|
287
|
+
|
|
288
|
+
⚠ **Steps 3 and 4 are two sends, never one.** A pointer batched with the clear lands *during* it and is
|
|
289
|
+
lost with the context it was meant to survive. Verify the clear landed by reading the prompt line
|
|
290
|
+
before sending the pointer — the same read-back every other send in this skill requires.
|
|
291
|
+
⚠ **The partner must not clear itself in the same window.** One supervising tier stays live at all
|
|
292
|
+
times; the seat holding the endgame goes second.
|
|
293
|
+
⚠ **Step 4 must state an EXPECTED-RETURN DEADLINE**, e.g. *"confirm you are back within 10 minutes."*
|
|
294
|
+
A clear order without one is an unbounded wait: see rule 6.
|
|
295
|
+
|
|
296
|
+
6. **THE RETURN LEG — the returning seat's FIRST act after re-arming is a verified notice to its partner.**
|
|
297
|
+
Not its second, not once the next milestone lands. **Measured 2026-08-28 (OBS-743):** an overseer cleared
|
|
298
|
+
at 00:05Z, was back at 00:07Z, *read the orchestrator's pane at 00:10Z to take its percentage* — and said
|
|
299
|
+
nothing. The orchestrator's last line had been *"T7 is mine until you're back."* It then held the task
|
|
300
|
+
alone for **53 minutes** across a reviewer flake, a retry, a merge and the whole tip verify, with no
|
|
301
|
+
signal that its supervising tier existed. The notice went out only because the operator noticed the gap.
|
|
302
|
+
|
|
303
|
+
**A one-way read is not a handshake.** The adopt step tells you to READ the partner's pane, which feels
|
|
304
|
+
like contact and transmits nothing — that is exactly how this gets skipped by a seat following the
|
|
305
|
+
protocol correctly.
|
|
306
|
+
|
|
307
|
+
The notice is ONE line (a newline submits early), sent with `herdr pane run` and **read back**, and it
|
|
308
|
+
carries four things:
|
|
309
|
+
```bash
|
|
310
|
+
herdr pane run <partner> "RETURN NOTICE <seat>: back on <my-pane> since <HH:MM>Z. WATCHERS I NOW HOLD:
|
|
311
|
+
<list>. SWEPT: <what was dead>. MISSION STATE AS I READ IT: <one clause>. Reply with every watcher YOU
|
|
312
|
+
still hold so we deconflict — do not arm anything I just named."
|
|
313
|
+
```
|
|
314
|
+
- **where and since when**, so the partner can stop holding your duties;
|
|
315
|
+
- **the watcher inventory you now hold** — coverage is the thing both tiers silently assume about each
|
|
316
|
+
other, and a returning seat that re-arms without saying so produces double-coverage that reads as
|
|
317
|
+
redundancy and is actually two tiers each trusting the other;
|
|
318
|
+
- **anything you swept**, because a dead watcher the partner armed is *its* belief about coverage, not
|
|
319
|
+
yours, and it will keep believing it;
|
|
320
|
+
- **an explicit deconfliction request.** Ask for the partner's inventory back; do not infer it.
|
|
321
|
+
|
|
322
|
+
⚠ **RE-ADOPT EVERY DETACHED WATCHER ON RETURN, and prove it from disk.** A detached watcher (`ppid 1`)
|
|
323
|
+
is the one kind that SURVIVES your clear, which is exactly why it rots unattended: `stat` its heartbeat
|
|
324
|
+
and treat **stale as dead**. Measured the same morning (OBS-742): a detached resolution watcher's
|
|
325
|
+
heartbeat was **157 minutes stale** and the task it existed to report had resolved **2h04m after its
|
|
326
|
+
last beat** — present, silent, and indistinguishable from healthy to anyone who did not look. Sweep it,
|
|
327
|
+
archive the heartbeat rather than deleting it so the gap stays measurable, and name it in the notice.
|
|
328
|
+
|
|
329
|
+
⚠ **A CLEAR AND A DEATH ARE THE SAME SILENCE.** A seat that clears and never returns — wrong pane,
|
|
330
|
+
crashed host, operator closed the tab — is indistinguishable from one mid-`/clear`. That is why step 4
|
|
331
|
+
states a deadline: **partner silent past it → escalate to the operator.** Without the deadline the
|
|
332
|
+
protocol's most dangerous state has no timeout, and the surviving tier waits forever on a peer that no
|
|
333
|
+
longer exists.
|
|
334
|
+
|
|
335
|
+
⚠ **`∑ NNNk tok` ON A CLAUDE STATUSLINE IS CUMULATIVE SESSION SPEND, NOT CONTEXT FILL.** The percentage
|
|
336
|
+
is the fill; the token total is what has been spent across every turn and keeps climbing after a compaction
|
|
337
|
+
or a clear. **Measured 2026-08-28, expensively:** this seat built a fallback watcher on `∑ Nk tok`, read
|
|
338
|
+
`970k` as 97% of a 1M window, and sent an urgent clear-order to an orchestrator that was actually at
|
|
339
|
+
**33%** — which then began a handoff and offered to discard a session two-thirds fresh, mid-endgame.
|
|
340
|
+
**Read the `%`. Never derive fill from the token total, and never build an instrument on a signal whose
|
|
341
|
+
SEMANTICS you have not verified against a second source.**
|
|
342
|
+
|
|
343
|
+
**Why 50% and not 85%:** a handoff written at 50% is written by a seat whose judgment is intact. One
|
|
344
|
+
written at 85% is written by a seat already degraded, about the decisions it is least able to summarise.
|
|
345
|
+
The threshold buys judgment, not headroom.
|
|
346
|
+
|
|
347
|
+
**Why the OTHER tier issues it:** a self-issued clear competes with whatever the seat is doing and loses.
|
|
348
|
+
An instruction from the other tier arrives as work, and the tier issuing it is not the tier that has to
|
|
349
|
+
overcome its own momentum to obey.
|
|
350
|
+
|
|
351
|
+
⚠ **A detached watcher gives COVERAGE and takes away NOTIFICATION.** A wake written to a log file that no
|
|
352
|
+
seat reads is not a wake. If the watcher must outlive a turn, it also needs a path that reaches a seat —
|
|
353
|
+
a beat the other tier reads, a notification, or an artifact the other tier watches. **Measured 2026-08-28:
|
|
354
|
+
`ACT: orch-215 at 970k/1000k — handoff + /clear NOW` fired correctly and sat unread in a scratchpad log
|
|
355
|
+
while the orchestrator kept working.**
|
|
356
|
+
|
|
252
357
|
Arm a context watcher on the orchestrator at spawn time and treat a threshold wake as a first-class event:
|
|
253
358
|
finish the step, write a handoff, `/clear` **plus a fresh brief — never `/compact`**, because a compaction
|
|
254
359
|
is a lossy summary nobody trusts while a clean session re-oriented from disk-verifiable state is reliable.
|
|
@@ -257,8 +362,9 @@ good, not after. If your own context cannot be read by the watcher, say so to th
|
|
|
257
362
|
number — an unmeasured budget is not a small budget.
|
|
258
363
|
|
|
259
364
|
```bash
|
|
260
|
-
|
|
261
|
-
.claude/skills/tickmarkr-overseer/scripts/watch-context.sh
|
|
365
|
+
# WARN 50 / ACT 50 — the mutual-clear threshold above, not a headroom alarm.
|
|
366
|
+
.claude/skills/tickmarkr-overseer/scripts/watch-context.sh orchestrator <orchestrator-agent-or-pane> 50 50 <handoff-file>
|
|
367
|
+
.claude/skills/tickmarkr-overseer/scripts/watch-context.sh overseer <overseer-agent-or-pane> 50 50 <handoff-file>
|
|
262
368
|
```
|
|
263
369
|
|
|
264
370
|
The first argument chooses the closed per-seat tier (`orchestrator-context` or `overseer-context`),
|
|
@@ -26,12 +26,18 @@
|
|
|
26
26
|
# last beat aged into a permanent stale — gradual growth never reached the auto-clear path at all.
|
|
27
27
|
# 4. EVERY TERMINAL EXIT STANDS THE TIER DOWN, so a watcher that finished reads DISARMED, not dead.
|
|
28
28
|
# Only a killed watcher reads STALE, which is exactly what STALE means.
|
|
29
|
+
# 5. A BLIND READ ALARMS (OBS-739). Ageing the tier is visible only to someone already reading the beat
|
|
30
|
+
# table; the seat that armed this watcher must be TOLD its instrument went blind. Silence and health
|
|
31
|
+
# must never look alike.
|
|
29
32
|
#
|
|
30
|
-
# usage: watch-context.sh <
|
|
33
|
+
# usage: watch-context.sh <role-slug> <agent|pane> <warn-pct> <act-pct> [handoff-file] [poll-s] [cap-s]
|
|
34
|
+
# <role-slug> is ANY seat role — orchestrator, overseer, surgeon, consult — and names the tier
|
|
35
|
+
# `<role>-context`. It is deliberately NOT a closed set: see OBS-730 at the guard below.
|
|
31
36
|
# TKR_AUTO_CLEAR=1 at act-pct WITH a fresh handoff, send /clear and re-brief instead of waking.
|
|
32
37
|
# TKR_REBRIEF=<path> the file the re-briefed seat is told to read (defaults to the handoff).
|
|
33
38
|
# TKR_HANDOFF_MAX_AGE_S how fresh "fresh" is (default 900).
|
|
34
39
|
# TKR_CLEAR_SETTLE_S seconds to let a cleared seat settle before the re-brief (default 6).
|
|
40
|
+
# TKR_BLIND_ALARM_S seconds unreadable before CONTEXT_BLIND alarms (default 120).
|
|
35
41
|
|
|
36
42
|
set -u
|
|
37
43
|
ROLE="${1:?supervising seat role required: orchestrator|overseer}"
|
|
@@ -45,10 +51,16 @@ MAXAGE="${TKR_HANDOFF_MAX_AGE_S:-900}"
|
|
|
45
51
|
REBRIEF="${TKR_REBRIEF:-$HANDOFF}"
|
|
46
52
|
SETTLE="${TKR_CLEAR_SETTLE_S:-6}"
|
|
47
53
|
|
|
54
|
+
# OBS-730: this rejected every role that was not orchestrator|overseer with exit 64 — while the skill
|
|
55
|
+
# mandates a context watcher on EVERY spawned seat. The rule and its own instrument disagreed, so the
|
|
56
|
+
# rule was unsatisfiable for surgeons, consults and every auxiliary seat, and the gap read as coverage.
|
|
57
|
+
# Any role slug names a tier; validate the SHAPE (a tier name reaches a shell) and never the membership.
|
|
48
58
|
case "$ROLE" in
|
|
49
|
-
|
|
50
|
-
|
|
59
|
+
*[!a-zA-Z0-9_-]*|'')
|
|
60
|
+
echo "watch-context.sh: role '$ROLE' must be a non-empty slug of [a-zA-Z0-9_-]" >&2; exit 64 ;;
|
|
51
61
|
esac
|
|
62
|
+
TIER="${ROLE}-context"
|
|
63
|
+
|
|
52
64
|
|
|
53
65
|
# The supervision beat interval (SUPERVISION_BEAT_MS = 10s). The loop ticks at the beat cadence or the
|
|
54
66
|
# caller's poll, whichever is SHORTER: a beat may only follow a successful read (rule 2), so the read
|
|
@@ -57,9 +69,33 @@ BEAT_EVERY=5
|
|
|
57
69
|
TICK=$(( POLL < BEAT_EVERY ? POLL : BEAT_EVERY ))
|
|
58
70
|
[ "$TICK" -ge 1 ] 2>/dev/null || TICK=1 # a zero or junk poll would spin, not watch
|
|
59
71
|
SEAT="$TARGET"
|
|
72
|
+
# ⚠ THE OTHER HALF OF OBS-730 LIVES IN THE PRODUCT, NOT HERE. `tickmarkr beat` enforces its own CLOSED
|
|
73
|
+
# tier set (`src/run/supervision.ts:45`), so widening only this script would let an auxiliary seat's
|
|
74
|
+
# watcher run while every beat failed SILENTLY — armed-looking, tier nonexistent. That is strictly worse
|
|
75
|
+
# than the exit 64 it replaced, and it is the precise failure this whole file exists to prevent.
|
|
76
|
+
# So PROBE ONCE and be loud about the answer. Watching still has real value without a tier — the warn,
|
|
77
|
+
# act and blind lines all still fire — but it must never be mistaken for registered supervision.
|
|
60
78
|
|
|
61
|
-
|
|
62
|
-
|
|
79
|
+
# ⚠ THE OTHER HALF OF OBS-730 LIVES IN THE PRODUCT, NOT HERE. `tickmarkr beat` enforces its own CLOSED
|
|
80
|
+
# tier set (`src/run/supervision.ts:45`), so widening only this script would let an auxiliary seat's
|
|
81
|
+
# watcher run while every beat failed SILENTLY — armed-looking, tier nonexistent, which is strictly
|
|
82
|
+
# WORSE than the exit 64 it replaced and is the precise failure this file exists to prevent.
|
|
83
|
+
# The refusal is reported by the FIRST REAL BEAT rather than by a startup probe: a probe that beats
|
|
84
|
+
# would emit one before any successful read and break rule 2 outright, and a probe that parses the
|
|
85
|
+
# usage banner binds this script to another command's help text. Beating is what we do anyway, so it
|
|
86
|
+
# perturbs nothing — and because a beat only ever follows a successful read, rule 2 still holds.
|
|
87
|
+
beat_refused=0
|
|
88
|
+
beat() {
|
|
89
|
+
tickmarkr beat "$TIER" --seat "$SEAT" >/dev/null 2>&1 && return 0
|
|
90
|
+
if [ "$beat_refused" -eq 0 ]; then
|
|
91
|
+
beat_refused=1
|
|
92
|
+
echo "TIER_UNREGISTERED ${TIER} — the product refused this tier (its set: src/run/supervision.ts:45)"
|
|
93
|
+
echo " watching CONTINUES and every warn/act/blind line below is real"
|
|
94
|
+
echo " but \`tickmarkr status\` will NOT show this seat as covered — never read that absence as safe"
|
|
95
|
+
fi
|
|
96
|
+
return 0
|
|
97
|
+
}
|
|
98
|
+
stand_down() { tickmarkr beat "$TIER" --stand-down --seat "$SEAT" >/dev/null 2>&1; return 0; }
|
|
63
99
|
# EVERY terminal exit — act, unsafe-act, cap — leaves through here, so none of them can forget to
|
|
64
100
|
# record the hand-off. A killed watcher never runs it, which is the one case that must read STALE.
|
|
65
101
|
trap stand_down EXIT
|
|
@@ -81,7 +117,7 @@ handoff_fresh() {
|
|
|
81
117
|
[ -f "$HANDOFF" ] || return 1
|
|
82
118
|
local age now mt
|
|
83
119
|
now=$(date +%s)
|
|
84
|
-
mt=$(stat -
|
|
120
|
+
mt=$(stat -c %Y "$HANDOFF" 2>/dev/null || stat -f %m "$HANDOFF" 2>/dev/null) || return 1
|
|
85
121
|
age=$((now - mt))
|
|
86
122
|
[ "$age" -le "$MAXAGE" ]
|
|
87
123
|
}
|
|
@@ -108,13 +144,35 @@ act_on() {
|
|
|
108
144
|
|
|
109
145
|
warned=0
|
|
110
146
|
elapsed=0
|
|
147
|
+
blind=0 # seconds in the current unreadable spell (OBS-739)
|
|
148
|
+
blind_alarmed=0
|
|
149
|
+
BLIND_ALARM_S="${TKR_BLIND_ALARM_S:-120}"
|
|
111
150
|
while [ "$elapsed" -lt "$CAP" ]; do
|
|
112
151
|
P=$(context_pct)
|
|
113
152
|
if [ -z "$P" ]; then
|
|
114
153
|
# Rule 2: no reading, no beat. The tier ages to STALE and a supervisor comes looking, which is the
|
|
115
154
|
# truth about a watcher that cannot see the seat it was armed on.
|
|
155
|
+
#
|
|
156
|
+
# Rule 5 (OBS-739): ALARM ON THE BLIND READ. Ageing a tier is not enough — it is only visible to
|
|
157
|
+
# someone already reading the beat table, and the measured failure was a watcher ALIVE AND BLIND for
|
|
158
|
+
# hours because the run's own status text pushed the percentage off the statusline. An instrument
|
|
159
|
+
# that cannot report its own absence is worse than none: the seat believes it has coverage and stops
|
|
160
|
+
# looking. So say so on stdout, where the supervising seat is actually woken. Once per blind spell,
|
|
161
|
+
# not every tick — a repeating alarm trains the reader to ignore it.
|
|
162
|
+
blind=$((blind + TICK))
|
|
163
|
+
if [ "$blind" -ge "$BLIND_ALARM_S" ] && [ "$blind_alarmed" -eq 0 ]; then
|
|
164
|
+
blind_alarmed=1
|
|
165
|
+
echo "CONTEXT_BLIND $TARGET — no percentage readable for ${blind}s; tier ${TIER} is ALIVE AND BLIND"
|
|
166
|
+
echo " this watcher is NOT providing coverage: read the seat by hand and re-arm on a clear statusline"
|
|
167
|
+
echo " do NOT substitute the token total — '∑ Nk tok' is cumulative SPEND, not context fill"
|
|
168
|
+
fi
|
|
116
169
|
sleep "$TICK"; elapsed=$((elapsed + TICK)); continue
|
|
117
170
|
fi
|
|
171
|
+
# a successful read closes the blind spell and re-arms the alarm for the next one
|
|
172
|
+
if [ "$blind_alarmed" -eq 1 ]; then
|
|
173
|
+
echo "CONTEXT_BLIND_CLEARED $TARGET — percentage readable again at ${P}% after ${blind}s blind"
|
|
174
|
+
fi
|
|
175
|
+
blind=0; blind_alarmed=0
|
|
118
176
|
beat
|
|
119
177
|
|
|
120
178
|
if [ "$P" -ge "$ACT" ] 2>/dev/null; then
|