tickmarkr 2.1.4 → 2.1.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/compile/collateral.d.ts +23 -2
- package/dist/compile/collateral.js +126 -11
- package/dist/compile/native.js +35 -3
- package/dist/gates/baseline.d.ts +23 -3
- package/dist/gates/baseline.js +75 -16
- package/dist/gates/llm.d.ts +1 -0
- package/dist/gates/llm.js +1 -0
- package/dist/gates/review.d.ts +9 -2
- package/dist/gates/review.js +51 -10
- package/dist/gates/run-gates.js +12 -1
- package/dist/run/daemon.js +110 -14
- package/dist/run/git.d.ts +50 -0
- package/dist/run/git.js +56 -2
- package/dist/run/journal.d.ts +23 -0
- package/dist/run/journal.js +96 -14
- package/dist/run/merge.js +13 -3
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +108 -2
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +63 -5
package/dist/run/journal.js
CHANGED
|
@@ -114,10 +114,11 @@ export function upheldFeedbackByTask(events) {
|
|
|
114
114
|
// for a resolved one is the silent-lie shape the gates exist to refuse, so the field says so outright.
|
|
115
115
|
export const UNIDENTIFIED = "<unidentified>";
|
|
116
116
|
const ANCHORED_RE = /^- (\S+?):(\d+) — (.*)$/; // "## Anchored review" rows (llm.ts)
|
|
117
|
-
const
|
|
117
|
+
const REVIEW_ROW_START_RE = /^- \[([^\]\r\n]+)\] /gm; // "- [material] …" (review.ts)
|
|
118
118
|
const JUDGE_ROW_RE = /^✗ ([\w.-]+): (.*)$/; // "✗ c1: …" (acceptance.ts) — id, then reason
|
|
119
119
|
const PATH_RE = /\b((?:[\w.@~+-]+\/)+[\w.@~+-]+\.\w{1,6})\b/;
|
|
120
120
|
const LINE_REF_RE = /(:\d+(?::\d+)?\b)|(\bline \d+\b)/gi;
|
|
121
|
+
const REVIEW_RATIONALE_SEPARATOR = " — rationale: ";
|
|
121
122
|
// ponytail: repo-relative tail from the first known top-level directory — enough to make an absolute
|
|
122
123
|
// worktree path and its repo-relative twin the same identity. Widen the marker list if a run ever
|
|
123
124
|
// names findings outside these roots.
|
|
@@ -143,10 +144,46 @@ function identifierIn(note) {
|
|
|
143
144
|
// code identity still has one stable identity of its own: its own words, with the volatile tokens
|
|
144
145
|
// swept out so line/path churn cannot mint a new symbol for the same finding. It is the reviewer's
|
|
145
146
|
// own bytes, never a guess, and it can never fuse two different findings into one.
|
|
146
|
-
function toFinding(cls, note, path, symbol) {
|
|
147
|
+
function toFinding(cls, note, path, symbol, rationale) {
|
|
147
148
|
const p = path || UNIDENTIFIED;
|
|
148
149
|
const s = symbol || normalizeGateFailure(note) || UNIDENTIFIED;
|
|
149
|
-
return {
|
|
150
|
+
return {
|
|
151
|
+
class: cls, path: p, symbol: s, note,
|
|
152
|
+
...(rationale !== undefined ? { rationale } : {}),
|
|
153
|
+
fingerprint: `${cls}|${p}|${s}`,
|
|
154
|
+
};
|
|
155
|
+
}
|
|
156
|
+
/**
|
|
157
|
+
* Decode review.ts's row rendering without treating physical lines as findings. A review finding's
|
|
158
|
+
* note and rationale are JSON strings before rendering and may therefore contain newlines; the next
|
|
159
|
+
* typed row (or the anchored-review block) is the record boundary. The fixed rationale separator is
|
|
160
|
+
* removed before identity is computed, so changing only why a concern was accepted re-seats it.
|
|
161
|
+
*/
|
|
162
|
+
function reviewDetailFindings(details) {
|
|
163
|
+
const anchoredAt = details.indexOf("\n\n## Anchored review");
|
|
164
|
+
let prose = anchoredAt === -1 ? details : details.slice(0, anchoredAt);
|
|
165
|
+
const inconsistencyAt = prose.search(/\nreview (?:finding|verdict) inconsistent:/);
|
|
166
|
+
if (inconsistencyAt !== -1)
|
|
167
|
+
prose = prose.slice(0, inconsistencyAt);
|
|
168
|
+
const starts = [...prose.matchAll(REVIEW_ROW_START_RE)];
|
|
169
|
+
return starts.map((start, i) => {
|
|
170
|
+
const contentStart = start.index + start[0].length;
|
|
171
|
+
const contentEnd = starts[i + 1]?.index ?? prose.length;
|
|
172
|
+
let content = prose.slice(contentStart, contentEnd);
|
|
173
|
+
// The newline before the next typed row is framing, while every earlier newline belongs to the
|
|
174
|
+
// reviewer's field. At EOF/anchored-review there is no framing newline to remove.
|
|
175
|
+
if (starts[i + 1] && content.endsWith("\n"))
|
|
176
|
+
content = content.slice(0, -1);
|
|
177
|
+
const deferred = String(start[1]).startsWith("deferred/");
|
|
178
|
+
const separatorAt = deferred ? content.indexOf(REVIEW_RATIONALE_SEPARATOR) : -1;
|
|
179
|
+
return separatorAt === -1
|
|
180
|
+
? { label: start[1], note: content }
|
|
181
|
+
: {
|
|
182
|
+
label: start[1],
|
|
183
|
+
note: content.slice(0, separatorAt),
|
|
184
|
+
rationale: content.slice(separatorAt + REVIEW_RATIONALE_SEPARATOR.length),
|
|
185
|
+
};
|
|
186
|
+
});
|
|
150
187
|
}
|
|
151
188
|
/**
|
|
152
189
|
* Structured findings for a BLOCKING review/judge gate result, parsed from the details the gate
|
|
@@ -163,24 +200,22 @@ function toFinding(cls, note, path, symbol) {
|
|
|
163
200
|
export function structuredFindings(gate, details, _scopeFiles = []) {
|
|
164
201
|
const lines = details.split("\n");
|
|
165
202
|
const rows = [];
|
|
166
|
-
const push = (cls, note, ownPath, fallbackSymbol = "") => {
|
|
203
|
+
const push = (cls, note, ownPath, fallbackSymbol = "", rationale) => {
|
|
167
204
|
const own = canonicalPath(ownPath || PATH_RE.exec(note)?.[1] || "");
|
|
168
205
|
const sym = identifierIn(note) || fallbackSymbol;
|
|
169
|
-
rows.push(toFinding(cls, note, own, sym));
|
|
206
|
+
rows.push(toFinding(cls, note, own, sym, rationale));
|
|
170
207
|
};
|
|
208
|
+
if (gate === "review") {
|
|
209
|
+
for (const finding of reviewDetailFindings(details)) {
|
|
210
|
+
push(`review:${finding.label}`, finding.note, "", "", finding.rationale);
|
|
211
|
+
}
|
|
212
|
+
}
|
|
171
213
|
for (const line of lines) {
|
|
172
214
|
const a = ANCHORED_RE.exec(line);
|
|
173
215
|
if (a) {
|
|
174
216
|
push(`${gate}:anchored`, a[3], a[1]);
|
|
175
217
|
continue;
|
|
176
218
|
}
|
|
177
|
-
if (gate === "review") {
|
|
178
|
-
const r = REVIEW_ROW_RE.exec(line);
|
|
179
|
-
if (r) {
|
|
180
|
-
push(`review:${r[1]}`, r[2], "");
|
|
181
|
-
continue;
|
|
182
|
-
}
|
|
183
|
-
}
|
|
184
219
|
if (gate === "acceptance") {
|
|
185
220
|
const j = JUDGE_ROW_RE.exec(line);
|
|
186
221
|
// the criterion id IS a stable symbol for an unmet acceptance criterion — the same criterion is
|
|
@@ -197,6 +232,29 @@ export function structuredFindings(gate, details, _scopeFiles = []) {
|
|
|
197
232
|
}
|
|
198
233
|
return rows;
|
|
199
234
|
}
|
|
235
|
+
// v2.1.5 T2: the reviewer's DEFERRAL channel, kept structured. `classifyReviewFindings`
|
|
236
|
+
// (gates/review.ts) renders a deferred finding as `- [deferred/<severity>] <note> — rationale: …`.
|
|
237
|
+
// The parser above preserves multiline fields and separates rationale from identity; an older journal
|
|
238
|
+
// whose row holds only prose still degrades through the same parse rather than to nothing. A deferred
|
|
239
|
+
// finding is a concern the reviewer SAW and chose not to block on; it is not a concern that was fixed.
|
|
240
|
+
const DEFERRED_CLASS_RE = /^review:deferred\b/;
|
|
241
|
+
export function isDeferredFinding(finding) {
|
|
242
|
+
return DEFERRED_CLASS_RE.test(finding.class);
|
|
243
|
+
}
|
|
244
|
+
/**
|
|
245
|
+
* The findings a PASSING review DEFERRED — the rows a blocking-only projection drops on the floor.
|
|
246
|
+
* A passing review's details are prose; without this the deferral has no identity a later round can
|
|
247
|
+
* match, and every structured reader of the journal is blind to a defect the reviewer itself named.
|
|
248
|
+
*/
|
|
249
|
+
export function deferredReviewFindings(details) {
|
|
250
|
+
return structuredFindings("review", details).filter(isDeferredFinding);
|
|
251
|
+
}
|
|
252
|
+
/** The exact review.ts details fragment represented by a structured review finding. */
|
|
253
|
+
export function renderStructuredReviewFinding(finding) {
|
|
254
|
+
const label = finding.class.startsWith("review:") ? finding.class.slice("review:".length) : finding.class;
|
|
255
|
+
const rationale = finding.rationale === undefined ? "" : `${REVIEW_RATIONALE_SEPARATOR}${finding.rationale}`;
|
|
256
|
+
return `- [${label}] ${finding.note}${rationale}`;
|
|
257
|
+
}
|
|
200
258
|
const findingRows = (event, gate) => {
|
|
201
259
|
if (Array.isArray(event.data.findings)) {
|
|
202
260
|
const rows = event.data.findings.filter((finding) => {
|
|
@@ -205,6 +263,7 @@ const findingRows = (event, gate) => {
|
|
|
205
263
|
const row = finding;
|
|
206
264
|
return typeof row.class === "string" && typeof row.path === "string"
|
|
207
265
|
&& typeof row.symbol === "string" && typeof row.note === "string"
|
|
266
|
+
&& (row.rationale === undefined || typeof row.rationale === "string")
|
|
208
267
|
&& typeof row.fingerprint === "string";
|
|
209
268
|
});
|
|
210
269
|
if (rows.length > 0)
|
|
@@ -509,6 +568,19 @@ export function journaledFailureBrief(events, taskId) {
|
|
|
509
568
|
* finding was dropped at the exact moment the operator paid for another attempt to fix it. A review
|
|
510
569
|
* that DECLINED (`skipped`) is not a verdict and neither adds nor retires — fail closed. Findings are
|
|
511
570
|
* keyed by fingerprint, so a reviewer restating one across rounds carries it once, not once per round.
|
|
571
|
+
*
|
|
572
|
+
* v2.1.5 T2: a passing review settles the findings it BLOCKED on. It does not settle the ones it
|
|
573
|
+
* DEFERRED — those it saw, declined to block on, and recorded a rationale for, and nothing has fixed
|
|
574
|
+
* them. So a pass retires the blocking set and re-seats its own deferrals, and the two retirements
|
|
575
|
+
* stay distinguishable: the blocking finding is gone, the deferral travels on as accepted work.
|
|
576
|
+
*
|
|
577
|
+
* A deferral's bound is the SAME single release as a blocking finding's — the operator accepting the
|
|
578
|
+
* review gate itself (`GATE_SATISFIED_RELEASE` stamped `gate: "review"`), the one approval in which a
|
|
579
|
+
* human actually looked at what the reviewer waved through. It is deliberately NOT bounded by a round
|
|
580
|
+
* count or by a time window: both retire a finding by arithmetic nobody read, which is the silent drop
|
|
581
|
+
* this fold exists to refuse. Nor can it accumulate — a reviewer restating the same path/note round
|
|
582
|
+
* after round re-seats ONE fingerprint, and a revised rationale replaces the prior rationale on that
|
|
583
|
+
* row. N rounds of the same concern therefore carry the newest accepted explanation once, not N rows.
|
|
512
584
|
*/
|
|
513
585
|
export function outstandingReviewFindings(events, taskId) {
|
|
514
586
|
const open = new Map();
|
|
@@ -522,8 +594,18 @@ export function outstandingReviewFindings(events, taskId) {
|
|
|
522
594
|
}
|
|
523
595
|
if (e.event !== "gate-result" || e.data.gate !== "review" || e.data.skipped === true)
|
|
524
596
|
continue;
|
|
525
|
-
if (e.data.pass !== false)
|
|
526
|
-
|
|
597
|
+
if (e.data.pass !== false) {
|
|
598
|
+
// a later review PASSED on this task: every finding it BLOCKED on is settled …
|
|
599
|
+
for (const [key, finding] of open)
|
|
600
|
+
if (!isDeferredFinding(finding))
|
|
601
|
+
open.delete(key);
|
|
602
|
+
// … and no deferral is, whether or not this pass restated it. A pass is silent about a
|
|
603
|
+
// deferral it does not mention: the concern is unfixed either way, and the reviewer that
|
|
604
|
+
// waved it through is not the release that accepts it. Retiring on omission would drop it on
|
|
605
|
+
// the very next round — the same silent drop by a different door.
|
|
606
|
+
for (const finding of findingRows(e, "review").filter(isDeferredFinding))
|
|
607
|
+
open.set(finding.fingerprint, finding);
|
|
608
|
+
}
|
|
527
609
|
else
|
|
528
610
|
for (const finding of findingRows(e, "review"))
|
|
529
611
|
open.set(finding.fingerprint, finding);
|
package/dist/run/merge.js
CHANGED
|
@@ -3,7 +3,7 @@ import { join } from "node:path";
|
|
|
3
3
|
import { shq } from "../adapters/types.js";
|
|
4
4
|
import { ceilingKillResult, classifyFailureOutput, effectiveCeilingMs, fingerprint, freshFailures, } from "../gates/baseline.js";
|
|
5
5
|
import { tickmarkrDir } from "../graph/graph.js";
|
|
6
|
-
import { gitHead, linkNodeModules, resolveIntegrationBranch, sh, shGit, shGitOk, WORKTREES_DIR } from "./git.js";
|
|
6
|
+
import { describeCapacity, gitHead, linkNodeModules, resolveIntegrationBranch, sameCapacity, sh, shGit, shGitOk, WORKTREES_DIR } from "./git.js";
|
|
7
7
|
export function integrationBranch(cfg, runId) {
|
|
8
8
|
return `${cfg.integrationBranchPrefix}${runId}`;
|
|
9
9
|
}
|
|
@@ -99,7 +99,13 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
99
99
|
// Battery parity on the infra rule too (T9): infrastructure-only output means the runner never
|
|
100
100
|
// completed a suite — nothing was verified, so nothing is forgivable, however familiar its
|
|
101
101
|
// fingerprints. Stricter-than-battery edge kept: unreadable output never forgives.
|
|
102
|
-
|
|
102
|
+
// T7: the SECOND reader of a baseline entry, and the same rule as the battery's (baseline.ts).
|
|
103
|
+
// Evidence crosses a session boundary here — the capture that would forgive this red is very
|
|
104
|
+
// often the previous session's — so a capture taken under a different resolved capacity forgives
|
|
105
|
+
// nothing at the tip either. An absent capacity is a pre-T7 baseline and keeps today's verdict.
|
|
106
|
+
const comparable = sameCapacity(entry?.capacity, r.capacity);
|
|
107
|
+
const forgiven = r.code !== 0 && baselineRed && failing.length === 0 && !unreadable && cause !== "infra"
|
|
108
|
+
&& comparable;
|
|
103
109
|
const pass = r.code === 0 || forgiven;
|
|
104
110
|
if (!pass)
|
|
105
111
|
writeFileSync(artifact, raw);
|
|
@@ -111,7 +117,11 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
111
117
|
fingerprints: r.code !== 0 ? fingerprint(stripped) : [],
|
|
112
118
|
details: r.code === 0 ? "exit 0"
|
|
113
119
|
: forgiven ? `exit ${r.code} but only baseline-recorded failures (forgiven vs baseline)`
|
|
114
|
-
:
|
|
120
|
+
: !comparable && baselineRed && failing.length === 0
|
|
121
|
+
? `exit ${r.code}; every failure is baseline-recorded, but that capture ran under `
|
|
122
|
+
+ `${describeCapacity(entry?.capacity)} and this verification ran under ${describeCapacity(r.capacity)} `
|
|
123
|
+
+ `— forgiveness across a changed capacity is not evidence`
|
|
124
|
+
: `exit ${r.code}`,
|
|
115
125
|
...(forgiven ? { forgiven: true } : {}),
|
|
116
126
|
...(cause ? { cause } : {}),
|
|
117
127
|
...(pass ? {} : { artifact }),
|
package/package.json
CHANGED
|
@@ -249,6 +249,111 @@ detect: **the ruling would have made the worker commit the violation the task wa
|
|
|
249
249
|
|
|
250
250
|
### Context is a supervised resource, for BOTH tiers
|
|
251
251
|
|
|
252
|
+
**THE TIERS CLEAR EACH OTHER AT 50%. Neither tier clears itself on its own notice.** Operator directive,
|
|
253
|
+
2026-08-28, and it exists because **a seat cannot reliably observe its own exhaustion** — the seat that
|
|
254
|
+
most needs clearing is the one least able to notice, and this project has now measured that three ways:
|
|
255
|
+
an overseer ran nine hours at 86% unable to read its own number; a context watcher went **alive and blind**
|
|
256
|
+
when the run's own status text pushed the percentage off the statusline; and an orchestrator went
|
|
257
|
+
**366k → 970k of 1M between two checks** while its ACT wake sat unread in a detached log.
|
|
258
|
+
|
|
259
|
+
The protocol, in both directions:
|
|
260
|
+
|
|
261
|
+
1. **Overseer sees orch at ≥50%** → nudge it: write `HANDOFF-ORCH-<ver>.md`, then `/clear`, then re-read
|
|
262
|
+
its brief **and** its handoff, then **re-arm every watcher it listed** (a cleared session has none).
|
|
263
|
+
2. **The returning orch, now fresh, checks the OVERSEER.** If the overseer is at ≥50%, it directs the
|
|
264
|
+
overseer to write its handoff and clear, and **points it at `HANDOFF-OVERSEER-<ver>.md` by path**.
|
|
265
|
+
3. Whichever seat is fresh performs the check. **Never both at once** — the run keeps one supervising tier
|
|
266
|
+
at all times, and the seat holding the endgame goes second.
|
|
267
|
+
4. **The duty to clear the other tier must SURVIVE a clear**, so it belongs in BOTH handoff files as a
|
|
268
|
+
standing re-arm item — not only in the message that ordered it. Earned 2026-08-28: the orchestrator
|
|
269
|
+
performed the check on the overseer BEFORE its own clear, precisely because clearing first would have
|
|
270
|
+
wiped the instruction to do it, and its handoff did not record the duty.
|
|
271
|
+
5. **A seat cannot `/clear` ITSELF — so the OTHER TIER SENDS IT.** `/clear` is a CLI command typed into a
|
|
272
|
+
session and no tool invokes it in your own pane, but it is just text in someone else's: the partner
|
|
273
|
+
tier types it into your pane. **This is the whole reason the protocol is mutual**, and it means the
|
|
274
|
+
loop closes without the operator. The exchange, both directions, in this exact order:
|
|
275
|
+
|
|
276
|
+
```bash
|
|
277
|
+
# 1. the seat crossing 50% writes its handoff FIRST, ending with its terminal marker
|
|
278
|
+
# 2. it asks the partner, naming its own pane and handoff path:
|
|
279
|
+
herdr pane run <partner> "I am at <N>%. Clear me: send /clear to <my-pane>, then point me at <my-handoff>."
|
|
280
|
+
# 3. the PARTNER sends the clear, then VERIFIES before pointing:
|
|
281
|
+
herdr pane run <my-pane> "/clear"
|
|
282
|
+
# read the pane back — a cleared claude session shows an empty prompt and a reset context gauge
|
|
283
|
+
# 4. and only THEN, as a SEPARATE send, the re-orientation:
|
|
284
|
+
herdr pane run <my-pane> "You were cleared at <N>%. Read <handoff> and <brief>, re-arm EVERY watcher
|
|
285
|
+
they name — a cleared session has none — then confirm you are back."
|
|
286
|
+
```
|
|
287
|
+
|
|
288
|
+
⚠ **Steps 3 and 4 are two sends, never one.** A pointer batched with the clear lands *during* it and is
|
|
289
|
+
lost with the context it was meant to survive. Verify the clear landed by reading the prompt line
|
|
290
|
+
before sending the pointer — the same read-back every other send in this skill requires.
|
|
291
|
+
⚠ **The partner must not clear itself in the same window.** One supervising tier stays live at all
|
|
292
|
+
times; the seat holding the endgame goes second.
|
|
293
|
+
⚠ **Step 4 must state an EXPECTED-RETURN DEADLINE**, e.g. *"confirm you are back within 10 minutes."*
|
|
294
|
+
A clear order without one is an unbounded wait: see rule 6.
|
|
295
|
+
|
|
296
|
+
6. **THE RETURN LEG — the returning seat's FIRST act after re-arming is a verified notice to its partner.**
|
|
297
|
+
Not its second, not once the next milestone lands. **Measured 2026-08-28 (OBS-743):** an overseer cleared
|
|
298
|
+
at 00:05Z, was back at 00:07Z, *read the orchestrator's pane at 00:10Z to take its percentage* — and said
|
|
299
|
+
nothing. The orchestrator's last line had been *"T7 is mine until you're back."* It then held the task
|
|
300
|
+
alone for **53 minutes** across a reviewer flake, a retry, a merge and the whole tip verify, with no
|
|
301
|
+
signal that its supervising tier existed. The notice went out only because the operator noticed the gap.
|
|
302
|
+
|
|
303
|
+
**A one-way read is not a handshake.** The adopt step tells you to READ the partner's pane, which feels
|
|
304
|
+
like contact and transmits nothing — that is exactly how this gets skipped by a seat following the
|
|
305
|
+
protocol correctly.
|
|
306
|
+
|
|
307
|
+
The notice is ONE line (a newline submits early), sent with `herdr pane run` and **read back**, and it
|
|
308
|
+
carries four things:
|
|
309
|
+
```bash
|
|
310
|
+
herdr pane run <partner> "RETURN NOTICE <seat>: back on <my-pane> since <HH:MM>Z. WATCHERS I NOW HOLD:
|
|
311
|
+
<list>. SWEPT: <what was dead>. MISSION STATE AS I READ IT: <one clause>. Reply with every watcher YOU
|
|
312
|
+
still hold so we deconflict — do not arm anything I just named."
|
|
313
|
+
```
|
|
314
|
+
- **where and since when**, so the partner can stop holding your duties;
|
|
315
|
+
- **the watcher inventory you now hold** — coverage is the thing both tiers silently assume about each
|
|
316
|
+
other, and a returning seat that re-arms without saying so produces double-coverage that reads as
|
|
317
|
+
redundancy and is actually two tiers each trusting the other;
|
|
318
|
+
- **anything you swept**, because a dead watcher the partner armed is *its* belief about coverage, not
|
|
319
|
+
yours, and it will keep believing it;
|
|
320
|
+
- **an explicit deconfliction request.** Ask for the partner's inventory back; do not infer it.
|
|
321
|
+
|
|
322
|
+
⚠ **RE-ADOPT EVERY DETACHED WATCHER ON RETURN, and prove it from disk.** A detached watcher (`ppid 1`)
|
|
323
|
+
is the one kind that SURVIVES your clear, which is exactly why it rots unattended: `stat` its heartbeat
|
|
324
|
+
and treat **stale as dead**. Measured the same morning (OBS-742): a detached resolution watcher's
|
|
325
|
+
heartbeat was **157 minutes stale** and the task it existed to report had resolved **2h04m after its
|
|
326
|
+
last beat** — present, silent, and indistinguishable from healthy to anyone who did not look. Sweep it,
|
|
327
|
+
archive the heartbeat rather than deleting it so the gap stays measurable, and name it in the notice.
|
|
328
|
+
|
|
329
|
+
⚠ **A CLEAR AND A DEATH ARE THE SAME SILENCE.** A seat that clears and never returns — wrong pane,
|
|
330
|
+
crashed host, operator closed the tab — is indistinguishable from one mid-`/clear`. That is why step 4
|
|
331
|
+
states a deadline: **partner silent past it → escalate to the operator.** Without the deadline the
|
|
332
|
+
protocol's most dangerous state has no timeout, and the surviving tier waits forever on a peer that no
|
|
333
|
+
longer exists.
|
|
334
|
+
|
|
335
|
+
⚠ **`∑ NNNk tok` ON A CLAUDE STATUSLINE IS CUMULATIVE SESSION SPEND, NOT CONTEXT FILL.** The percentage
|
|
336
|
+
is the fill; the token total is what has been spent across every turn and keeps climbing after a compaction
|
|
337
|
+
or a clear. **Measured 2026-08-28, expensively:** this seat built a fallback watcher on `∑ Nk tok`, read
|
|
338
|
+
`970k` as 97% of a 1M window, and sent an urgent clear-order to an orchestrator that was actually at
|
|
339
|
+
**33%** — which then began a handoff and offered to discard a session two-thirds fresh, mid-endgame.
|
|
340
|
+
**Read the `%`. Never derive fill from the token total, and never build an instrument on a signal whose
|
|
341
|
+
SEMANTICS you have not verified against a second source.**
|
|
342
|
+
|
|
343
|
+
**Why 50% and not 85%:** a handoff written at 50% is written by a seat whose judgment is intact. One
|
|
344
|
+
written at 85% is written by a seat already degraded, about the decisions it is least able to summarise.
|
|
345
|
+
The threshold buys judgment, not headroom.
|
|
346
|
+
|
|
347
|
+
**Why the OTHER tier issues it:** a self-issued clear competes with whatever the seat is doing and loses.
|
|
348
|
+
An instruction from the other tier arrives as work, and the tier issuing it is not the tier that has to
|
|
349
|
+
overcome its own momentum to obey.
|
|
350
|
+
|
|
351
|
+
⚠ **A detached watcher gives COVERAGE and takes away NOTIFICATION.** A wake written to a log file that no
|
|
352
|
+
seat reads is not a wake. If the watcher must outlive a turn, it also needs a path that reaches a seat —
|
|
353
|
+
a beat the other tier reads, a notification, or an artifact the other tier watches. **Measured 2026-08-28:
|
|
354
|
+
`ACT: orch-215 at 970k/1000k — handoff + /clear NOW` fired correctly and sat unread in a scratchpad log
|
|
355
|
+
while the orchestrator kept working.**
|
|
356
|
+
|
|
252
357
|
Arm a context watcher on the orchestrator at spawn time and treat a threshold wake as a first-class event:
|
|
253
358
|
finish the step, write a handoff, `/clear` **plus a fresh brief — never `/compact`**, because a compaction
|
|
254
359
|
is a lossy summary nobody trusts while a clean session re-oriented from disk-verifiable state is reliable.
|
|
@@ -257,8 +362,9 @@ good, not after. If your own context cannot be read by the watcher, say so to th
|
|
|
257
362
|
number — an unmeasured budget is not a small budget.
|
|
258
363
|
|
|
259
364
|
```bash
|
|
260
|
-
|
|
261
|
-
.claude/skills/tickmarkr-overseer/scripts/watch-context.sh
|
|
365
|
+
# WARN 50 / ACT 50 — the mutual-clear threshold above, not a headroom alarm.
|
|
366
|
+
.claude/skills/tickmarkr-overseer/scripts/watch-context.sh orchestrator <orchestrator-agent-or-pane> 50 50 <handoff-file>
|
|
367
|
+
.claude/skills/tickmarkr-overseer/scripts/watch-context.sh overseer <overseer-agent-or-pane> 50 50 <handoff-file>
|
|
262
368
|
```
|
|
263
369
|
|
|
264
370
|
The first argument chooses the closed per-seat tier (`orchestrator-context` or `overseer-context`),
|
|
@@ -26,12 +26,18 @@
|
|
|
26
26
|
# last beat aged into a permanent stale — gradual growth never reached the auto-clear path at all.
|
|
27
27
|
# 4. EVERY TERMINAL EXIT STANDS THE TIER DOWN, so a watcher that finished reads DISARMED, not dead.
|
|
28
28
|
# Only a killed watcher reads STALE, which is exactly what STALE means.
|
|
29
|
+
# 5. A BLIND READ ALARMS (OBS-739). Ageing the tier is visible only to someone already reading the beat
|
|
30
|
+
# table; the seat that armed this watcher must be TOLD its instrument went blind. Silence and health
|
|
31
|
+
# must never look alike.
|
|
29
32
|
#
|
|
30
|
-
# usage: watch-context.sh <
|
|
33
|
+
# usage: watch-context.sh <role-slug> <agent|pane> <warn-pct> <act-pct> [handoff-file] [poll-s] [cap-s]
|
|
34
|
+
# <role-slug> is ANY seat role — orchestrator, overseer, surgeon, consult — and names the tier
|
|
35
|
+
# `<role>-context`. It is deliberately NOT a closed set: see OBS-730 at the guard below.
|
|
31
36
|
# TKR_AUTO_CLEAR=1 at act-pct WITH a fresh handoff, send /clear and re-brief instead of waking.
|
|
32
37
|
# TKR_REBRIEF=<path> the file the re-briefed seat is told to read (defaults to the handoff).
|
|
33
38
|
# TKR_HANDOFF_MAX_AGE_S how fresh "fresh" is (default 900).
|
|
34
39
|
# TKR_CLEAR_SETTLE_S seconds to let a cleared seat settle before the re-brief (default 6).
|
|
40
|
+
# TKR_BLIND_ALARM_S seconds unreadable before CONTEXT_BLIND alarms (default 120).
|
|
35
41
|
|
|
36
42
|
set -u
|
|
37
43
|
ROLE="${1:?supervising seat role required: orchestrator|overseer}"
|
|
@@ -45,10 +51,16 @@ MAXAGE="${TKR_HANDOFF_MAX_AGE_S:-900}"
|
|
|
45
51
|
REBRIEF="${TKR_REBRIEF:-$HANDOFF}"
|
|
46
52
|
SETTLE="${TKR_CLEAR_SETTLE_S:-6}"
|
|
47
53
|
|
|
54
|
+
# OBS-730: this rejected every role that was not orchestrator|overseer with exit 64 — while the skill
|
|
55
|
+
# mandates a context watcher on EVERY spawned seat. The rule and its own instrument disagreed, so the
|
|
56
|
+
# rule was unsatisfiable for surgeons, consults and every auxiliary seat, and the gap read as coverage.
|
|
57
|
+
# Any role slug names a tier; validate the SHAPE (a tier name reaches a shell) and never the membership.
|
|
48
58
|
case "$ROLE" in
|
|
49
|
-
|
|
50
|
-
|
|
59
|
+
*[!a-zA-Z0-9_-]*|'')
|
|
60
|
+
echo "watch-context.sh: role '$ROLE' must be a non-empty slug of [a-zA-Z0-9_-]" >&2; exit 64 ;;
|
|
51
61
|
esac
|
|
62
|
+
TIER="${ROLE}-context"
|
|
63
|
+
|
|
52
64
|
|
|
53
65
|
# The supervision beat interval (SUPERVISION_BEAT_MS = 10s). The loop ticks at the beat cadence or the
|
|
54
66
|
# caller's poll, whichever is SHORTER: a beat may only follow a successful read (rule 2), so the read
|
|
@@ -57,9 +69,33 @@ BEAT_EVERY=5
|
|
|
57
69
|
TICK=$(( POLL < BEAT_EVERY ? POLL : BEAT_EVERY ))
|
|
58
70
|
[ "$TICK" -ge 1 ] 2>/dev/null || TICK=1 # a zero or junk poll would spin, not watch
|
|
59
71
|
SEAT="$TARGET"
|
|
72
|
+
# ⚠ THE OTHER HALF OF OBS-730 LIVES IN THE PRODUCT, NOT HERE. `tickmarkr beat` enforces its own CLOSED
|
|
73
|
+
# tier set (`src/run/supervision.ts:45`), so widening only this script would let an auxiliary seat's
|
|
74
|
+
# watcher run while every beat failed SILENTLY — armed-looking, tier nonexistent. That is strictly worse
|
|
75
|
+
# than the exit 64 it replaced, and it is the precise failure this whole file exists to prevent.
|
|
76
|
+
# So PROBE ONCE and be loud about the answer. Watching still has real value without a tier — the warn,
|
|
77
|
+
# act and blind lines all still fire — but it must never be mistaken for registered supervision.
|
|
60
78
|
|
|
61
|
-
|
|
62
|
-
|
|
79
|
+
# ⚠ THE OTHER HALF OF OBS-730 LIVES IN THE PRODUCT, NOT HERE. `tickmarkr beat` enforces its own CLOSED
|
|
80
|
+
# tier set (`src/run/supervision.ts:45`), so widening only this script would let an auxiliary seat's
|
|
81
|
+
# watcher run while every beat failed SILENTLY — armed-looking, tier nonexistent, which is strictly
|
|
82
|
+
# WORSE than the exit 64 it replaced and is the precise failure this file exists to prevent.
|
|
83
|
+
# The refusal is reported by the FIRST REAL BEAT rather than by a startup probe: a probe that beats
|
|
84
|
+
# would emit one before any successful read and break rule 2 outright, and a probe that parses the
|
|
85
|
+
# usage banner binds this script to another command's help text. Beating is what we do anyway, so it
|
|
86
|
+
# perturbs nothing — and because a beat only ever follows a successful read, rule 2 still holds.
|
|
87
|
+
beat_refused=0
|
|
88
|
+
beat() {
|
|
89
|
+
tickmarkr beat "$TIER" --seat "$SEAT" >/dev/null 2>&1 && return 0
|
|
90
|
+
if [ "$beat_refused" -eq 0 ]; then
|
|
91
|
+
beat_refused=1
|
|
92
|
+
echo "TIER_UNREGISTERED ${TIER} — the product refused this tier (its set: src/run/supervision.ts:45)"
|
|
93
|
+
echo " watching CONTINUES and every warn/act/blind line below is real"
|
|
94
|
+
echo " but \`tickmarkr status\` will NOT show this seat as covered — never read that absence as safe"
|
|
95
|
+
fi
|
|
96
|
+
return 0
|
|
97
|
+
}
|
|
98
|
+
stand_down() { tickmarkr beat "$TIER" --stand-down --seat "$SEAT" >/dev/null 2>&1; return 0; }
|
|
63
99
|
# EVERY terminal exit — act, unsafe-act, cap — leaves through here, so none of them can forget to
|
|
64
100
|
# record the hand-off. A killed watcher never runs it, which is the one case that must read STALE.
|
|
65
101
|
trap stand_down EXIT
|
|
@@ -108,13 +144,35 @@ act_on() {
|
|
|
108
144
|
|
|
109
145
|
warned=0
|
|
110
146
|
elapsed=0
|
|
147
|
+
blind=0 # seconds in the current unreadable spell (OBS-739)
|
|
148
|
+
blind_alarmed=0
|
|
149
|
+
BLIND_ALARM_S="${TKR_BLIND_ALARM_S:-120}"
|
|
111
150
|
while [ "$elapsed" -lt "$CAP" ]; do
|
|
112
151
|
P=$(context_pct)
|
|
113
152
|
if [ -z "$P" ]; then
|
|
114
153
|
# Rule 2: no reading, no beat. The tier ages to STALE and a supervisor comes looking, which is the
|
|
115
154
|
# truth about a watcher that cannot see the seat it was armed on.
|
|
155
|
+
#
|
|
156
|
+
# Rule 5 (OBS-739): ALARM ON THE BLIND READ. Ageing a tier is not enough — it is only visible to
|
|
157
|
+
# someone already reading the beat table, and the measured failure was a watcher ALIVE AND BLIND for
|
|
158
|
+
# hours because the run's own status text pushed the percentage off the statusline. An instrument
|
|
159
|
+
# that cannot report its own absence is worse than none: the seat believes it has coverage and stops
|
|
160
|
+
# looking. So say so on stdout, where the supervising seat is actually woken. Once per blind spell,
|
|
161
|
+
# not every tick — a repeating alarm trains the reader to ignore it.
|
|
162
|
+
blind=$((blind + TICK))
|
|
163
|
+
if [ "$blind" -ge "$BLIND_ALARM_S" ] && [ "$blind_alarmed" -eq 0 ]; then
|
|
164
|
+
blind_alarmed=1
|
|
165
|
+
echo "CONTEXT_BLIND $TARGET — no percentage readable for ${blind}s; tier ${TIER} is ALIVE AND BLIND"
|
|
166
|
+
echo " this watcher is NOT providing coverage: read the seat by hand and re-arm on a clear statusline"
|
|
167
|
+
echo " do NOT substitute the token total — '∑ Nk tok' is cumulative SPEND, not context fill"
|
|
168
|
+
fi
|
|
116
169
|
sleep "$TICK"; elapsed=$((elapsed + TICK)); continue
|
|
117
170
|
fi
|
|
171
|
+
# a successful read closes the blind spell and re-arms the alarm for the next one
|
|
172
|
+
if [ "$blind_alarmed" -eq 1 ]; then
|
|
173
|
+
echo "CONTEXT_BLIND_CLEARED $TARGET — percentage readable again at ${P}% after ${blind}s blind"
|
|
174
|
+
fi
|
|
175
|
+
blind=0; blind_alarmed=0
|
|
118
176
|
beat
|
|
119
177
|
|
|
120
178
|
if [ "$P" -ge "$ACT" ] 2>/dev/null; then
|