tickmarkr 2.1.4 → 2.1.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/commands/stats.d.ts +21 -0
- package/dist/cli/commands/stats.js +210 -0
- package/dist/cli/commands/status.js +28 -3
- package/dist/cli/index.d.ts +1 -1
- package/dist/cli/index.js +3 -1
- package/dist/compile/collateral.d.ts +23 -2
- package/dist/compile/collateral.js +126 -11
- package/dist/compile/native.js +35 -3
- package/dist/gates/baseline.d.ts +23 -3
- package/dist/gates/baseline.js +102 -33
- package/dist/gates/llm.d.ts +1 -0
- package/dist/gates/llm.js +1 -0
- package/dist/gates/review.d.ts +9 -2
- package/dist/gates/review.js +51 -10
- package/dist/gates/run-gates.js +12 -1
- package/dist/run/daemon.js +325 -19
- package/dist/run/git.d.ts +50 -0
- package/dist/run/git.js +56 -2
- package/dist/run/journal.d.ts +28 -0
- package/dist/run/journal.js +137 -20
- package/dist/run/merge.js +13 -3
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +120 -3
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +63 -5
package/dist/run/git.js
CHANGED
|
@@ -63,6 +63,53 @@ export const deriveForkCap = (concurrency, cores = availableParallelism()) => Ma
|
|
|
63
63
|
export const runWithForkBudget = (concurrency, fn) => forkBudget.run(String(deriveForkCap(concurrency)), fn);
|
|
64
64
|
/** The cap owned by the run on this async context; the standalone default outside one. */
|
|
65
65
|
export const resolvedForkCap = () => forkBudget.getStore() ?? DEFAULT_FORK_CAP;
|
|
66
|
+
const positiveInt = (v) => typeof v === "number" && Number.isInteger(v) && v > 0;
|
|
67
|
+
/**
|
|
68
|
+
* Three states, never two. A record carrying NO capacity is an older record from before this stamp
|
|
69
|
+
* existed: it keeps exactly the verdict it has today. A record carrying a capacity it cannot state —
|
|
70
|
+
* half the pair, an empty container, a zero, a negative, an unparseable value — is a NEWER record
|
|
71
|
+
* that is malformed, and reading it as an older one is how a fail-closed guard stops firing silently.
|
|
72
|
+
*/
|
|
73
|
+
export function readCapacity(value) {
|
|
74
|
+
if (value === undefined)
|
|
75
|
+
return { state: "absent" };
|
|
76
|
+
if (value === null || typeof value !== "object")
|
|
77
|
+
return { state: "malformed" };
|
|
78
|
+
const { forkCap, cores } = value;
|
|
79
|
+
return positiveInt(forkCap) && positiveInt(cores)
|
|
80
|
+
? { state: "present", capacity: { forkCap, cores } }
|
|
81
|
+
: { state: "malformed" };
|
|
82
|
+
}
|
|
83
|
+
/**
|
|
84
|
+
* May a verdict recorded under `recorded` be reused — forgiven, cached, replayed — by a session
|
|
85
|
+
* running under `current`? Absent → yes, unchanged. Present and identical → yes. Malformed, a
|
|
86
|
+
* different capacity, or a current capacity the caller could not state → no.
|
|
87
|
+
*/
|
|
88
|
+
export function sameCapacity(recorded, current) {
|
|
89
|
+
const read = readCapacity(recorded);
|
|
90
|
+
if (read.state === "absent")
|
|
91
|
+
return true;
|
|
92
|
+
if (read.state === "malformed" || current === undefined)
|
|
93
|
+
return false;
|
|
94
|
+
return read.capacity.forkCap === current.forkCap && read.capacity.cores === current.cores;
|
|
95
|
+
}
|
|
96
|
+
export const describeCapacity = (value) => {
|
|
97
|
+
const read = readCapacity(value);
|
|
98
|
+
return read.state === "present"
|
|
99
|
+
? `fork cap ${read.capacity.forkCap} of ${read.capacity.cores} cores`
|
|
100
|
+
: read.state === "absent" ? "an unrecorded capacity" : "a malformed capacity";
|
|
101
|
+
};
|
|
102
|
+
/**
|
|
103
|
+
* The capacity a child spawned on THIS async context would receive: the same precedence `shell`
|
|
104
|
+
* applies below — an operator export of the cap wins over the run's own derived value — beside the
|
|
105
|
+
* cores it was divided from. A caller holding a command's own result reads the capacity off THAT
|
|
106
|
+
* result (`ShResult.capacity`, stamped where the child's environment was built); this is for the
|
|
107
|
+
* decisions taken BEFORE any child exists — a cache hit, a reuse predicate.
|
|
108
|
+
*/
|
|
109
|
+
export const resolvedCapacity = () => ({
|
|
110
|
+
forkCap: Number(FORK_CAP_ENV in process.env ? process.env[FORK_CAP_ENV] : resolvedForkCap()),
|
|
111
|
+
cores: availableParallelism(),
|
|
112
|
+
});
|
|
66
113
|
/** The shipped shell ceiling: the fallback every caller gets when nothing measured a better one. */
|
|
67
114
|
export const DEFAULT_SHELL_TIMEOUT_MS = 600000;
|
|
68
115
|
/**
|
|
@@ -103,6 +150,13 @@ function shell(cmd, cwd, timeoutMs, login) {
|
|
|
103
150
|
// OBS-110: apply the run's own fork cap only when the operator has not already set one.
|
|
104
151
|
if (!(FORK_CAP_ENV in env))
|
|
105
152
|
env[FORK_CAP_ENV] = resolvedForkCap();
|
|
153
|
+
// T7: the capacity every result of this shell carries, read HERE — off the environment the child
|
|
154
|
+
// is about to receive, after the precedence above has settled. An operator export is already in
|
|
155
|
+
// `env`, so what gets recorded is the operator's number, which is the case a release was re-taken
|
|
156
|
+
// for; re-deriving the run's own budget after the command returned would stamp a cap no child ran
|
|
157
|
+
// under. `Number` of an unparseable export is NaN, which every reader treats as malformed and
|
|
158
|
+
// therefore fails closed — the honest direction when the cap in play cannot be stated.
|
|
159
|
+
const capacity = { forkCap: Number(env[FORK_CAP_ENV]), cores: availableParallelism() };
|
|
106
160
|
const attempt = () => new Promise((resolve) => {
|
|
107
161
|
const startedAt = Date.now();
|
|
108
162
|
// detached: bash gets its own process group so a timeout can kill the whole tree —
|
|
@@ -120,7 +174,7 @@ function shell(cmd, cwd, timeoutMs, login) {
|
|
|
120
174
|
clearTimeout(timer);
|
|
121
175
|
stdout += stdoutDecoder.end();
|
|
122
176
|
stderr += stderrDecoder.end();
|
|
123
|
-
resolve({ code, stdout, stderr: err ?? stderr, timedOut, durationMs: Date.now() - startedAt });
|
|
177
|
+
resolve({ code, stdout, stderr: err ?? stderr, timedOut, durationMs: Date.now() - startedAt, capacity });
|
|
124
178
|
};
|
|
125
179
|
const timer = setTimeout(() => {
|
|
126
180
|
timedOut = true;
|
|
@@ -170,7 +224,7 @@ function shell(cmd, cwd, timeoutMs, login) {
|
|
|
170
224
|
// Bounded, and the bound is what makes a persisting shortage a REPORTED failure rather than a
|
|
171
225
|
// wedged daemon: past it the caller gets the refusal's own text under exit 127, as before.
|
|
172
226
|
if (n >= SPAWN_ATTEMPT_LIMIT) {
|
|
173
|
-
return { code: 127, stdout: "", stderr: String(r.refused), durationMs: Date.now() - startedAt };
|
|
227
|
+
return { code: 127, stdout: "", stderr: String(r.refused), durationMs: Date.now() - startedAt, capacity };
|
|
174
228
|
}
|
|
175
229
|
await new Promise((wake) => setTimeout(wake, SPAWN_RETRY_BACKOFF_MS * n));
|
|
176
230
|
}
|
package/dist/run/journal.d.ts
CHANGED
|
@@ -29,6 +29,11 @@ export declare const ATTEMPT_CAP_RELEASE: "attempt-cap";
|
|
|
29
29
|
export declare const GATE_SATISFIED_RELEASE: "gate-satisfied";
|
|
30
30
|
export declare const REVIEW_UPHELD_RELEASE: "review-upheld";
|
|
31
31
|
export declare const RECHECK_RELEASE: "recheck";
|
|
32
|
+
export interface PreservedRef {
|
|
33
|
+
ref: string;
|
|
34
|
+
diffCommand: string;
|
|
35
|
+
}
|
|
36
|
+
export declare function preservedRefsByTask(events: JournalEvent[]): Map<string, PreservedRef[]>;
|
|
32
37
|
export declare function reviewRoundsSinceApproval(events: JournalEvent[], taskId: string): number;
|
|
33
38
|
export declare function upheldFeedbackByTask(events: JournalEvent[]): Map<string, string>;
|
|
34
39
|
export interface StructuredFinding {
|
|
@@ -36,6 +41,7 @@ export interface StructuredFinding {
|
|
|
36
41
|
path: string;
|
|
37
42
|
symbol: string;
|
|
38
43
|
note: string;
|
|
44
|
+
rationale?: string;
|
|
39
45
|
fingerprint: string;
|
|
40
46
|
}
|
|
41
47
|
export declare const UNIDENTIFIED = "<unidentified>";
|
|
@@ -52,6 +58,15 @@ export declare const UNIDENTIFIED = "<unidentified>";
|
|
|
52
58
|
* normalized words (see toFinding).
|
|
53
59
|
*/
|
|
54
60
|
export declare function structuredFindings(gate: string, details: string, _scopeFiles?: string[]): StructuredFinding[];
|
|
61
|
+
export declare function isDeferredFinding(finding: StructuredFinding): boolean;
|
|
62
|
+
/**
|
|
63
|
+
* The findings a PASSING review DEFERRED — the rows a blocking-only projection drops on the floor.
|
|
64
|
+
* A passing review's details are prose; without this the deferral has no identity a later round can
|
|
65
|
+
* match, and every structured reader of the journal is blind to a defect the reviewer itself named.
|
|
66
|
+
*/
|
|
67
|
+
export declare function deferredReviewFindings(details: string): StructuredFinding[];
|
|
68
|
+
/** The exact review.ts details fragment represented by a structured review finding. */
|
|
69
|
+
export declare function renderStructuredReviewFinding(finding: StructuredFinding): string;
|
|
55
70
|
export interface PriorRunJournal {
|
|
56
71
|
runId: string;
|
|
57
72
|
events: JournalEvent[];
|
|
@@ -126,6 +141,19 @@ export declare function journaledFailureBrief(events: JournalEvent[], taskId: st
|
|
|
126
141
|
* finding was dropped at the exact moment the operator paid for another attempt to fix it. A review
|
|
127
142
|
* that DECLINED (`skipped`) is not a verdict and neither adds nor retires — fail closed. Findings are
|
|
128
143
|
* keyed by fingerprint, so a reviewer restating one across rounds carries it once, not once per round.
|
|
144
|
+
*
|
|
145
|
+
* v2.1.5 T2: a passing review settles the findings it BLOCKED on. It does not settle the ones it
|
|
146
|
+
* DEFERRED — those it saw, declined to block on, and recorded a rationale for, and nothing has fixed
|
|
147
|
+
* them. So a pass retires the blocking set and re-seats its own deferrals, and the two retirements
|
|
148
|
+
* stay distinguishable: the blocking finding is gone, the deferral travels on as accepted work.
|
|
149
|
+
*
|
|
150
|
+
* A deferral's bound is the SAME single release as a blocking finding's — the operator accepting the
|
|
151
|
+
* review gate itself (`GATE_SATISFIED_RELEASE` stamped `gate: "review"`), the one approval in which a
|
|
152
|
+
* human actually looked at what the reviewer waved through. It is deliberately NOT bounded by a round
|
|
153
|
+
* count or by a time window: both retire a finding by arithmetic nobody read, which is the silent drop
|
|
154
|
+
* this fold exists to refuse. Nor can it accumulate — a reviewer restating the same path/note round
|
|
155
|
+
* after round re-seats ONE fingerprint, and a revised rationale replaces the prior rationale on that
|
|
156
|
+
* row. N rounds of the same concern therefore carry the newest accepted explanation once, not N rows.
|
|
129
157
|
*/
|
|
130
158
|
export declare function outstandingReviewFindings(events: JournalEvent[], taskId: string): StructuredFinding[];
|
|
131
159
|
/** The findings a funded repair must carry into the next dispatch, or undefined if none is pending. */
|
package/dist/run/journal.js
CHANGED
|
@@ -2,7 +2,7 @@ import { AsyncLocalStorage } from "node:async_hooks";
|
|
|
2
2
|
import { appendFileSync, existsSync, mkdirSync, readFileSync, readdirSync } from "node:fs";
|
|
3
3
|
import { join } from "node:path";
|
|
4
4
|
import { z } from "zod";
|
|
5
|
-
import { channelKey, TokenUsageSchema } from "../adapters/types.js";
|
|
5
|
+
import { channelKey, shq, TokenUsageSchema } from "../adapters/types.js";
|
|
6
6
|
import { stateDirName, taskContentDigest, tickmarkrDir } from "../graph/graph.js";
|
|
7
7
|
import { GATE_NAMES, TIERS } from "../graph/schema.js";
|
|
8
8
|
import { buildProfile, classify } from "../route/profile.js";
|
|
@@ -56,6 +56,24 @@ export const REVIEW_UPHELD_RELEASE = "review-upheld";
|
|
|
56
56
|
// so the corrected declaration is the thing that earns the green. Budget semantics match attempt-cap
|
|
57
57
|
// (fresh attempts, tried survives) because the park cost the task its remaining budget.
|
|
58
58
|
export const RECHECK_RELEASE = "recheck";
|
|
59
|
+
// OBS-738: one authority for every recovery surface. The ref is accepted only from the row that
|
|
60
|
+
// preservation itself writes; task-human prose, branch heads and commit history are deliberately
|
|
61
|
+
// absent from this fold. Keep every row in journal order — one task can be recreated more than once,
|
|
62
|
+
// and the terminal record owes the operator every resulting recovery handle, not merely the newest.
|
|
63
|
+
export function preservedRefsByTask(events) {
|
|
64
|
+
const byTask = new Map();
|
|
65
|
+
for (const event of events) {
|
|
66
|
+
if (event.event !== "worktree-preserved" || !event.taskId || typeof event.data.ref !== "string"
|
|
67
|
+
|| event.data.ref === "")
|
|
68
|
+
continue;
|
|
69
|
+
const ref = event.data.ref;
|
|
70
|
+
byTask.set(event.taskId, [
|
|
71
|
+
...(byTask.get(event.taskId) ?? []),
|
|
72
|
+
{ ref, diffCommand: `git diff ${shq(`${ref}^!`)}` },
|
|
73
|
+
]);
|
|
74
|
+
}
|
|
75
|
+
return byTask;
|
|
76
|
+
}
|
|
59
77
|
// OBS-189: review rounds are scoped to the current ENGAGEMENT — the stretch since the newest operator
|
|
60
78
|
// approval for the task. A whole-journal count re-parks an upheld task before its funded attempt can
|
|
61
79
|
// dispatch (measured live on run-20260726-213539), making a fresh journal the only escape. A T15
|
|
@@ -114,10 +132,11 @@ export function upheldFeedbackByTask(events) {
|
|
|
114
132
|
// for a resolved one is the silent-lie shape the gates exist to refuse, so the field says so outright.
|
|
115
133
|
export const UNIDENTIFIED = "<unidentified>";
|
|
116
134
|
const ANCHORED_RE = /^- (\S+?):(\d+) — (.*)$/; // "## Anchored review" rows (llm.ts)
|
|
117
|
-
const
|
|
135
|
+
const REVIEW_ROW_START_RE = /^- \[([^\]\r\n]+)\] /gm; // "- [material] …" (review.ts)
|
|
118
136
|
const JUDGE_ROW_RE = /^✗ ([\w.-]+): (.*)$/; // "✗ c1: …" (acceptance.ts) — id, then reason
|
|
119
137
|
const PATH_RE = /\b((?:[\w.@~+-]+\/)+[\w.@~+-]+\.\w{1,6})\b/;
|
|
120
138
|
const LINE_REF_RE = /(:\d+(?::\d+)?\b)|(\bline \d+\b)/gi;
|
|
139
|
+
const REVIEW_RATIONALE_SEPARATOR = " — rationale: ";
|
|
121
140
|
// ponytail: repo-relative tail from the first known top-level directory — enough to make an absolute
|
|
122
141
|
// worktree path and its repo-relative twin the same identity. Widen the marker list if a run ever
|
|
123
142
|
// names findings outside these roots.
|
|
@@ -143,10 +162,46 @@ function identifierIn(note) {
|
|
|
143
162
|
// code identity still has one stable identity of its own: its own words, with the volatile tokens
|
|
144
163
|
// swept out so line/path churn cannot mint a new symbol for the same finding. It is the reviewer's
|
|
145
164
|
// own bytes, never a guess, and it can never fuse two different findings into one.
|
|
146
|
-
function toFinding(cls, note, path, symbol) {
|
|
165
|
+
function toFinding(cls, note, path, symbol, rationale) {
|
|
147
166
|
const p = path || UNIDENTIFIED;
|
|
148
167
|
const s = symbol || normalizeGateFailure(note) || UNIDENTIFIED;
|
|
149
|
-
return {
|
|
168
|
+
return {
|
|
169
|
+
class: cls, path: p, symbol: s, note,
|
|
170
|
+
...(rationale !== undefined ? { rationale } : {}),
|
|
171
|
+
fingerprint: `${cls}|${p}|${s}`,
|
|
172
|
+
};
|
|
173
|
+
}
|
|
174
|
+
/**
|
|
175
|
+
* Decode review.ts's row rendering without treating physical lines as findings. A review finding's
|
|
176
|
+
* note and rationale are JSON strings before rendering and may therefore contain newlines; the next
|
|
177
|
+
* typed row (or the anchored-review block) is the record boundary. The fixed rationale separator is
|
|
178
|
+
* removed before identity is computed, so changing only why a concern was accepted re-seats it.
|
|
179
|
+
*/
|
|
180
|
+
function reviewDetailFindings(details) {
|
|
181
|
+
const anchoredAt = details.indexOf("\n\n## Anchored review");
|
|
182
|
+
let prose = anchoredAt === -1 ? details : details.slice(0, anchoredAt);
|
|
183
|
+
const inconsistencyAt = prose.search(/\nreview (?:finding|verdict) inconsistent:/);
|
|
184
|
+
if (inconsistencyAt !== -1)
|
|
185
|
+
prose = prose.slice(0, inconsistencyAt);
|
|
186
|
+
const starts = [...prose.matchAll(REVIEW_ROW_START_RE)];
|
|
187
|
+
return starts.map((start, i) => {
|
|
188
|
+
const contentStart = start.index + start[0].length;
|
|
189
|
+
const contentEnd = starts[i + 1]?.index ?? prose.length;
|
|
190
|
+
let content = prose.slice(contentStart, contentEnd);
|
|
191
|
+
// The newline before the next typed row is framing, while every earlier newline belongs to the
|
|
192
|
+
// reviewer's field. At EOF/anchored-review there is no framing newline to remove.
|
|
193
|
+
if (starts[i + 1] && content.endsWith("\n"))
|
|
194
|
+
content = content.slice(0, -1);
|
|
195
|
+
const deferred = String(start[1]).startsWith("deferred/");
|
|
196
|
+
const separatorAt = deferred ? content.indexOf(REVIEW_RATIONALE_SEPARATOR) : -1;
|
|
197
|
+
return separatorAt === -1
|
|
198
|
+
? { label: start[1], note: content }
|
|
199
|
+
: {
|
|
200
|
+
label: start[1],
|
|
201
|
+
note: content.slice(0, separatorAt),
|
|
202
|
+
rationale: content.slice(separatorAt + REVIEW_RATIONALE_SEPARATOR.length),
|
|
203
|
+
};
|
|
204
|
+
});
|
|
150
205
|
}
|
|
151
206
|
/**
|
|
152
207
|
* Structured findings for a BLOCKING review/judge gate result, parsed from the details the gate
|
|
@@ -163,24 +218,22 @@ function toFinding(cls, note, path, symbol) {
|
|
|
163
218
|
export function structuredFindings(gate, details, _scopeFiles = []) {
|
|
164
219
|
const lines = details.split("\n");
|
|
165
220
|
const rows = [];
|
|
166
|
-
const push = (cls, note, ownPath, fallbackSymbol = "") => {
|
|
221
|
+
const push = (cls, note, ownPath, fallbackSymbol = "", rationale) => {
|
|
167
222
|
const own = canonicalPath(ownPath || PATH_RE.exec(note)?.[1] || "");
|
|
168
223
|
const sym = identifierIn(note) || fallbackSymbol;
|
|
169
|
-
rows.push(toFinding(cls, note, own, sym));
|
|
224
|
+
rows.push(toFinding(cls, note, own, sym, rationale));
|
|
170
225
|
};
|
|
226
|
+
if (gate === "review") {
|
|
227
|
+
for (const finding of reviewDetailFindings(details)) {
|
|
228
|
+
push(`review:${finding.label}`, finding.note, "", "", finding.rationale);
|
|
229
|
+
}
|
|
230
|
+
}
|
|
171
231
|
for (const line of lines) {
|
|
172
232
|
const a = ANCHORED_RE.exec(line);
|
|
173
233
|
if (a) {
|
|
174
234
|
push(`${gate}:anchored`, a[3], a[1]);
|
|
175
235
|
continue;
|
|
176
236
|
}
|
|
177
|
-
if (gate === "review") {
|
|
178
|
-
const r = REVIEW_ROW_RE.exec(line);
|
|
179
|
-
if (r) {
|
|
180
|
-
push(`review:${r[1]}`, r[2], "");
|
|
181
|
-
continue;
|
|
182
|
-
}
|
|
183
|
-
}
|
|
184
237
|
if (gate === "acceptance") {
|
|
185
238
|
const j = JUDGE_ROW_RE.exec(line);
|
|
186
239
|
// the criterion id IS a stable symbol for an unmet acceptance criterion — the same criterion is
|
|
@@ -197,6 +250,29 @@ export function structuredFindings(gate, details, _scopeFiles = []) {
|
|
|
197
250
|
}
|
|
198
251
|
return rows;
|
|
199
252
|
}
|
|
253
|
+
// v2.1.5 T2: the reviewer's DEFERRAL channel, kept structured. `classifyReviewFindings`
|
|
254
|
+
// (gates/review.ts) renders a deferred finding as `- [deferred/<severity>] <note> — rationale: …`.
|
|
255
|
+
// The parser above preserves multiline fields and separates rationale from identity; an older journal
|
|
256
|
+
// whose row holds only prose still degrades through the same parse rather than to nothing. A deferred
|
|
257
|
+
// finding is a concern the reviewer SAW and chose not to block on; it is not a concern that was fixed.
|
|
258
|
+
const DEFERRED_CLASS_RE = /^review:deferred\b/;
|
|
259
|
+
export function isDeferredFinding(finding) {
|
|
260
|
+
return DEFERRED_CLASS_RE.test(finding.class);
|
|
261
|
+
}
|
|
262
|
+
/**
|
|
263
|
+
* The findings a PASSING review DEFERRED — the rows a blocking-only projection drops on the floor.
|
|
264
|
+
* A passing review's details are prose; without this the deferral has no identity a later round can
|
|
265
|
+
* match, and every structured reader of the journal is blind to a defect the reviewer itself named.
|
|
266
|
+
*/
|
|
267
|
+
export function deferredReviewFindings(details) {
|
|
268
|
+
return structuredFindings("review", details).filter(isDeferredFinding);
|
|
269
|
+
}
|
|
270
|
+
/** The exact review.ts details fragment represented by a structured review finding. */
|
|
271
|
+
export function renderStructuredReviewFinding(finding) {
|
|
272
|
+
const label = finding.class.startsWith("review:") ? finding.class.slice("review:".length) : finding.class;
|
|
273
|
+
const rationale = finding.rationale === undefined ? "" : `${REVIEW_RATIONALE_SEPARATOR}${finding.rationale}`;
|
|
274
|
+
return `- [${label}] ${finding.note}${rationale}`;
|
|
275
|
+
}
|
|
200
276
|
const findingRows = (event, gate) => {
|
|
201
277
|
if (Array.isArray(event.data.findings)) {
|
|
202
278
|
const rows = event.data.findings.filter((finding) => {
|
|
@@ -205,6 +281,7 @@ const findingRows = (event, gate) => {
|
|
|
205
281
|
const row = finding;
|
|
206
282
|
return typeof row.class === "string" && typeof row.path === "string"
|
|
207
283
|
&& typeof row.symbol === "string" && typeof row.note === "string"
|
|
284
|
+
&& (row.rationale === undefined || typeof row.rationale === "string")
|
|
208
285
|
&& typeof row.fingerprint === "string";
|
|
209
286
|
});
|
|
210
287
|
if (rows.length > 0)
|
|
@@ -509,6 +586,19 @@ export function journaledFailureBrief(events, taskId) {
|
|
|
509
586
|
* finding was dropped at the exact moment the operator paid for another attempt to fix it. A review
|
|
510
587
|
* that DECLINED (`skipped`) is not a verdict and neither adds nor retires — fail closed. Findings are
|
|
511
588
|
* keyed by fingerprint, so a reviewer restating one across rounds carries it once, not once per round.
|
|
589
|
+
*
|
|
590
|
+
* v2.1.5 T2: a passing review settles the findings it BLOCKED on. It does not settle the ones it
|
|
591
|
+
* DEFERRED — those it saw, declined to block on, and recorded a rationale for, and nothing has fixed
|
|
592
|
+
* them. So a pass retires the blocking set and re-seats its own deferrals, and the two retirements
|
|
593
|
+
* stay distinguishable: the blocking finding is gone, the deferral travels on as accepted work.
|
|
594
|
+
*
|
|
595
|
+
* A deferral's bound is the SAME single release as a blocking finding's — the operator accepting the
|
|
596
|
+
* review gate itself (`GATE_SATISFIED_RELEASE` stamped `gate: "review"`), the one approval in which a
|
|
597
|
+
* human actually looked at what the reviewer waved through. It is deliberately NOT bounded by a round
|
|
598
|
+
* count or by a time window: both retire a finding by arithmetic nobody read, which is the silent drop
|
|
599
|
+
* this fold exists to refuse. Nor can it accumulate — a reviewer restating the same path/note round
|
|
600
|
+
* after round re-seats ONE fingerprint, and a revised rationale replaces the prior rationale on that
|
|
601
|
+
* row. N rounds of the same concern therefore carry the newest accepted explanation once, not N rows.
|
|
512
602
|
*/
|
|
513
603
|
export function outstandingReviewFindings(events, taskId) {
|
|
514
604
|
const open = new Map();
|
|
@@ -522,8 +612,18 @@ export function outstandingReviewFindings(events, taskId) {
|
|
|
522
612
|
}
|
|
523
613
|
if (e.event !== "gate-result" || e.data.gate !== "review" || e.data.skipped === true)
|
|
524
614
|
continue;
|
|
525
|
-
if (e.data.pass !== false)
|
|
526
|
-
|
|
615
|
+
if (e.data.pass !== false) {
|
|
616
|
+
// a later review PASSED on this task: every finding it BLOCKED on is settled …
|
|
617
|
+
for (const [key, finding] of open)
|
|
618
|
+
if (!isDeferredFinding(finding))
|
|
619
|
+
open.delete(key);
|
|
620
|
+
// … and no deferral is, whether or not this pass restated it. A pass is silent about a
|
|
621
|
+
// deferral it does not mention: the concern is unfixed either way, and the reviewer that
|
|
622
|
+
// waved it through is not the release that accepts it. Retiring on omission would drop it on
|
|
623
|
+
// the very next round — the same silent drop by a different door.
|
|
624
|
+
for (const finding of findingRows(e, "review").filter(isDeferredFinding))
|
|
625
|
+
open.set(finding.fingerprint, finding);
|
|
626
|
+
}
|
|
527
627
|
else
|
|
528
628
|
for (const finding of findingRows(e, "review"))
|
|
529
629
|
open.set(finding.fingerprint, finding);
|
|
@@ -987,19 +1087,36 @@ export class Journal {
|
|
|
987
1087
|
? undefined
|
|
988
1088
|
: DecisionEventSchema.parse({ ...eventOrDecision, ts: new Date().toISOString() });
|
|
989
1089
|
const event = decisionRow?.event ?? eventOrDecision;
|
|
1090
|
+
const rowTaskId = decisionRow && "taskId" in decisionRow ? decisionRow.taskId : taskId;
|
|
990
1091
|
const inputData = decisionRow?.data ?? data;
|
|
1092
|
+
// OBS-738: terminal and resume records reduce the journal that precedes them. Neither re-derives
|
|
1093
|
+
// recovery facts from task-human prose: preserved refs come from preservedRefsByTask, and the
|
|
1094
|
+
// upheld brief comes from the established prompt/replay reducer.
|
|
1095
|
+
const priorEvents = event === "run-end" || event === "resume-restore" ? this.read() : [];
|
|
1096
|
+
const reducedData = event === "run-end"
|
|
1097
|
+
? (() => {
|
|
1098
|
+
const preservedRefs = [...preservedRefsByTask(priorEvents)].flatMap(([preservedTaskId, refs]) => refs.map(({ ref, diffCommand }) => ({ taskId: preservedTaskId, ref, diffCommand })));
|
|
1099
|
+
return preservedRefs.length > 0 ? { ...inputData, preservedRefs } : inputData;
|
|
1100
|
+
})()
|
|
1101
|
+
: event === "resume-restore" && rowTaskId && upheldFeedbackByTask(priorEvents).has(rowTaskId)
|
|
1102
|
+
? {
|
|
1103
|
+
...inputData,
|
|
1104
|
+
upheldFeedbackRestoredFor: rowTaskId,
|
|
1105
|
+
summary: `upheld feedback restored for ${rowTaskId}`,
|
|
1106
|
+
}
|
|
1107
|
+
: inputData;
|
|
991
1108
|
const evidence = judgePersistence.getStore();
|
|
992
1109
|
const failed = evidence?.invocations.filter((invocation) => invocation.transcript !== undefined) ?? [];
|
|
993
1110
|
const persistedData = event === "judge-retry" && failed.length > 0
|
|
994
1111
|
? {
|
|
995
|
-
...
|
|
1112
|
+
...reducedData,
|
|
996
1113
|
transcript: failed[0].transcript,
|
|
997
1114
|
...(failed[1] ? { retryTranscript: failed[1].transcript } : {}),
|
|
998
1115
|
}
|
|
999
|
-
:
|
|
1000
|
-
const row = decisionRow
|
|
1001
|
-
|
|
1002
|
-
|
|
1116
|
+
: reducedData;
|
|
1117
|
+
const row = decisionRow
|
|
1118
|
+
? { ...decisionRow, data: persistedData }
|
|
1119
|
+
: { ts: new Date().toISOString(), event, ...(taskId ? { taskId } : {}), data: persistedData };
|
|
1003
1120
|
// T3 secret redaction: only the persisted bytes are masked — the caller's data stays untouched in
|
|
1004
1121
|
// memory. The narrator receives the persisted (masked) row so a pane sink never shows a credential.
|
|
1005
1122
|
const line = redactSecrets(JSON.stringify(row));
|
package/dist/run/merge.js
CHANGED
|
@@ -3,7 +3,7 @@ import { join } from "node:path";
|
|
|
3
3
|
import { shq } from "../adapters/types.js";
|
|
4
4
|
import { ceilingKillResult, classifyFailureOutput, effectiveCeilingMs, fingerprint, freshFailures, } from "../gates/baseline.js";
|
|
5
5
|
import { tickmarkrDir } from "../graph/graph.js";
|
|
6
|
-
import { gitHead, linkNodeModules, resolveIntegrationBranch, sh, shGit, shGitOk, WORKTREES_DIR } from "./git.js";
|
|
6
|
+
import { describeCapacity, gitHead, linkNodeModules, resolveIntegrationBranch, sameCapacity, sh, shGit, shGitOk, WORKTREES_DIR } from "./git.js";
|
|
7
7
|
export function integrationBranch(cfg, runId) {
|
|
8
8
|
return `${cfg.integrationBranchPrefix}${runId}`;
|
|
9
9
|
}
|
|
@@ -99,7 +99,13 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
99
99
|
// Battery parity on the infra rule too (T9): infrastructure-only output means the runner never
|
|
100
100
|
// completed a suite — nothing was verified, so nothing is forgivable, however familiar its
|
|
101
101
|
// fingerprints. Stricter-than-battery edge kept: unreadable output never forgives.
|
|
102
|
-
|
|
102
|
+
// T7: the SECOND reader of a baseline entry, and the same rule as the battery's (baseline.ts).
|
|
103
|
+
// Evidence crosses a session boundary here — the capture that would forgive this red is very
|
|
104
|
+
// often the previous session's — so a capture taken under a different resolved capacity forgives
|
|
105
|
+
// nothing at the tip either. An absent capacity is a pre-T7 baseline and keeps today's verdict.
|
|
106
|
+
const comparable = sameCapacity(entry?.capacity, r.capacity);
|
|
107
|
+
const forgiven = r.code !== 0 && baselineRed && failing.length === 0 && !unreadable && cause !== "infra"
|
|
108
|
+
&& comparable;
|
|
103
109
|
const pass = r.code === 0 || forgiven;
|
|
104
110
|
if (!pass)
|
|
105
111
|
writeFileSync(artifact, raw);
|
|
@@ -111,7 +117,11 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
111
117
|
fingerprints: r.code !== 0 ? fingerprint(stripped) : [],
|
|
112
118
|
details: r.code === 0 ? "exit 0"
|
|
113
119
|
: forgiven ? `exit ${r.code} but only baseline-recorded failures (forgiven vs baseline)`
|
|
114
|
-
:
|
|
120
|
+
: !comparable && baselineRed && failing.length === 0
|
|
121
|
+
? `exit ${r.code}; every failure is baseline-recorded, but that capture ran under `
|
|
122
|
+
+ `${describeCapacity(entry?.capacity)} and this verification ran under ${describeCapacity(r.capacity)} `
|
|
123
|
+
+ `— forgiveness across a changed capacity is not evidence`
|
|
124
|
+
: `exit ${r.code}`,
|
|
115
125
|
...(forgiven ? { forgiven: true } : {}),
|
|
116
126
|
...(cause ? { cause } : {}),
|
|
117
127
|
...(pass ? {} : { artifact }),
|
package/package.json
CHANGED
|
@@ -103,7 +103,7 @@ through brief lineage. **An executor choice nobody made is still an executor cho
|
|
|
103
103
|
fraction (`ORCH · v1.19 4/5`, updated on every task-done); tickmarkr opens ONE TAB PER TASK, labelled
|
|
104
104
|
with the task id and holding that task's worker plus its judge/review/consult panes (tickmarkr
|
|
105
105
|
updates it). Never long context strings or ✓-chains.
|
|
106
|
-
2. **Orchestrator**: Launch the orchestrator with your agent host. Spawning on current herdr is two-step — the one-shot `agent start --cwd` form was removed in the herdr CLI redesign and now fails with `unknown option` (OBS-138): first create the pane with `herdr tab create --workspace <ws> --cwd <repo> --label "ORCH · <version>"` and parse `result.root_pane.pane_id` from its JSON, then start the agent in it. For Claude Code, use `herdr agent start orchestrator --kind claude --pane <root-pane-id> -- --permission-mode bypassPermissions` (append `--model <m>` after the `--` if the operator has a policy). For Codex, use `herdr agent start orchestrator --kind codex --pane <root-pane-id> -- --dangerously-bypass-approvals-and-sandbox` (add `--model <m>` to specify the model). The unsandboxed flag is REQUIRED: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation — do not downgrade it. Workers you never spawn — tickmarkr spawns its own visible worker panes. Auxiliary agents you do spawn (consultants, reviewers, scouts) follow the same forms: never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions --settings '{"promptSuggestionEnabled":false}'
|
|
106
|
+
2. **Orchestrator**: Launch the orchestrator with your agent host. Spawning on current herdr is two-step — the one-shot `agent start --cwd` form was removed in the herdr CLI redesign and now fails with `unknown option` (OBS-138): first create the pane with `herdr tab create --workspace <ws> --cwd <repo> --label "ORCH · <version>"` and parse `result.root_pane.pane_id` from its JSON, then start the agent in it. For Claude Code, use `herdr agent start orchestrator --kind claude --pane <root-pane-id> -- --permission-mode bypassPermissions` (append `--model <m>` after the `--` if the operator has a policy). For Codex, use `herdr agent start orchestrator --kind codex --pane <root-pane-id> -- --dangerously-bypass-approvals-and-sandbox` (add `--model <m>` to specify the model). The unsandboxed flag is REQUIRED: codex's `workspace-write` sandbox keeps `.git` refs read-only, so a sandboxed orchestrator's `tickmarkr run` dies at integration-branch creation — do not downgrade it. Workers you never spawn — tickmarkr spawns its own visible worker panes. Auxiliary agents you do spawn (consultants, reviewers, scouts) follow the same forms: never launch a claude session in plan mode or default permission mode for autonomous work — both stall on per-command approval prompts nobody is watching; claude is always `--permission-mode bypassPermissions --settings '{"promptSuggestionEnabled":false}'`. **For a codex consultant, use `-a never --sandbox workspace-write` — NOT `--sandbox read-only`.** ⚠ **`--sandbox read-only` CONTRADICTS this skill's own completion protocol and will hang the seat.** Every seat you spawn is told to deliver an ARTIFACT ending in a terminal MARKER, because that is the only completion signal the artifact watcher can key on (`done` is turn end). A read-only sandbox cannot write that artifact, so codex blocks on `Would you like to make the following edits?` for its OWN report — and the report exists ONLY in the pending edit, so abandoning the prompt destroys the work rather than merely delaying it. Measured 2026-08-28: a consultant spawned `--sandbox read-only` finished a 14,604-byte verdict, sat blocked on the write, and the operator saw the prompt before the supervising tier did. `read-only` is correct ONLY for a seat that writes nothing at all — which, under the artifact+marker rule, is no seat this skill tells you to spawn. When the prompt does appear, answer **"Yes, and don't ask again for these files"** rather than plain yes: plain yes re-blocks on the next write of the same file. **That `--settings` pair is not cosmetic and it is not optional:** claude-code's AUTOSUGGEST renders context-plausible ghost text into an idle seat's prompt line that is BYTE-IDENTICAL to a typed draft in text-format reads (OBS-482), so a supervising tier cannot tell a seat's own unsent work from a rendering artifact without `agent read --format ansi`. Turning the suggester off at spawn removes the ambiguity at its source instead of paying for the discrimination at every read. Verified against the shipped binary: `claude --settings '{"promptSuggestionEnabled":false}' -p …` exits 0 with a real response, and the key appears in the binary's own settings schema. **For kimi, pass `-y`** (`herdr agent start <name> --kind kimi --pane <id> -- -y`) — the adapter already launches its own workers that way (`src/adapters/kimi.ts:204`), and a kimi seat spawned without it sits on an approval prompt having done nothing. **Herdr cannot see that state**: it reports a kimi pane as `agent_status: working` with `screen_detection_skipped: true` while the prompt is up, so the BLOCKED-STATE watcher below is blind on this vendor and the spawn flag is the ONLY control. Every vendor you spawn needs its auto-approve form named here; a vendor absent from this list is a seat that will hang.
|
|
107
107
|
3. **Standing instructions travel as a brief FILE, never as pane text** — PTY input truncates at ~1024B and a
|
|
108
108
|
truncated brief silently drops policy. Write the full brief to `<repo>/.tickmarkr/overseer/ORCH-BRIEF.md`
|
|
109
109
|
(inside the tickmarkr state dir — already self-gitignored, no exclude step needed), then send one line:
|
|
@@ -249,6 +249,111 @@ detect: **the ruling would have made the worker commit the violation the task wa
|
|
|
249
249
|
|
|
250
250
|
### Context is a supervised resource, for BOTH tiers
|
|
251
251
|
|
|
252
|
+
**THE TIERS CLEAR EACH OTHER AT 50%. Neither tier clears itself on its own notice.** Operator directive,
|
|
253
|
+
2026-08-28, and it exists because **a seat cannot reliably observe its own exhaustion** — the seat that
|
|
254
|
+
most needs clearing is the one least able to notice, and this project has now measured that three ways:
|
|
255
|
+
an overseer ran nine hours at 86% unable to read its own number; a context watcher went **alive and blind**
|
|
256
|
+
when the run's own status text pushed the percentage off the statusline; and an orchestrator went
|
|
257
|
+
**366k → 970k of 1M between two checks** while its ACT wake sat unread in a detached log.
|
|
258
|
+
|
|
259
|
+
The protocol, in both directions:
|
|
260
|
+
|
|
261
|
+
1. **Overseer sees orch at ≥50%** → nudge it: write `HANDOFF-ORCH-<ver>.md`, then `/clear`, then re-read
|
|
262
|
+
its brief **and** its handoff, then **re-arm every watcher it listed** (a cleared session has none).
|
|
263
|
+
2. **The returning orch, now fresh, checks the OVERSEER.** If the overseer is at ≥50%, it directs the
|
|
264
|
+
overseer to write its handoff and clear, and **points it at `HANDOFF-OVERSEER-<ver>.md` by path**.
|
|
265
|
+
3. Whichever seat is fresh performs the check. **Never both at once** — the run keeps one supervising tier
|
|
266
|
+
at all times, and the seat holding the endgame goes second.
|
|
267
|
+
4. **The duty to clear the other tier must SURVIVE a clear**, so it belongs in BOTH handoff files as a
|
|
268
|
+
standing re-arm item — not only in the message that ordered it. Earned 2026-08-28: the orchestrator
|
|
269
|
+
performed the check on the overseer BEFORE its own clear, precisely because clearing first would have
|
|
270
|
+
wiped the instruction to do it, and its handoff did not record the duty.
|
|
271
|
+
5. **A seat cannot `/clear` ITSELF — so the OTHER TIER SENDS IT.** `/clear` is a CLI command typed into a
|
|
272
|
+
session and no tool invokes it in your own pane, but it is just text in someone else's: the partner
|
|
273
|
+
tier types it into your pane. **This is the whole reason the protocol is mutual**, and it means the
|
|
274
|
+
loop closes without the operator. The exchange, both directions, in this exact order:
|
|
275
|
+
|
|
276
|
+
```bash
|
|
277
|
+
# 1. the seat crossing 50% writes its handoff FIRST, ending with its terminal marker
|
|
278
|
+
# 2. it asks the partner, naming its own pane and handoff path:
|
|
279
|
+
herdr pane run <partner> "I am at <N>%. Clear me: send /clear to <my-pane>, then point me at <my-handoff>."
|
|
280
|
+
# 3. the PARTNER sends the clear, then VERIFIES before pointing:
|
|
281
|
+
herdr pane run <my-pane> "/clear"
|
|
282
|
+
# read the pane back — a cleared claude session shows an empty prompt and a reset context gauge
|
|
283
|
+
# 4. and only THEN, as a SEPARATE send, the re-orientation:
|
|
284
|
+
herdr pane run <my-pane> "You were cleared at <N>%. Read <handoff> and <brief>, re-arm EVERY watcher
|
|
285
|
+
they name — a cleared session has none — then confirm you are back."
|
|
286
|
+
```
|
|
287
|
+
|
|
288
|
+
⚠ **Steps 3 and 4 are two sends, never one.** A pointer batched with the clear lands *during* it and is
|
|
289
|
+
lost with the context it was meant to survive. Verify the clear landed by reading the prompt line
|
|
290
|
+
before sending the pointer — the same read-back every other send in this skill requires.
|
|
291
|
+
⚠ **The partner must not clear itself in the same window.** One supervising tier stays live at all
|
|
292
|
+
times; the seat holding the endgame goes second.
|
|
293
|
+
⚠ **Step 4 must state an EXPECTED-RETURN DEADLINE**, e.g. *"confirm you are back within 10 minutes."*
|
|
294
|
+
A clear order without one is an unbounded wait: see rule 6.
|
|
295
|
+
|
|
296
|
+
6. **THE RETURN LEG — the returning seat's FIRST act after re-arming is a verified notice to its partner.**
|
|
297
|
+
Not its second, not once the next milestone lands. **Measured 2026-08-28 (OBS-743):** an overseer cleared
|
|
298
|
+
at 00:05Z, was back at 00:07Z, *read the orchestrator's pane at 00:10Z to take its percentage* — and said
|
|
299
|
+
nothing. The orchestrator's last line had been *"T7 is mine until you're back."* It then held the task
|
|
300
|
+
alone for **53 minutes** across a reviewer flake, a retry, a merge and the whole tip verify, with no
|
|
301
|
+
signal that its supervising tier existed. The notice went out only because the operator noticed the gap.
|
|
302
|
+
|
|
303
|
+
**A one-way read is not a handshake.** The adopt step tells you to READ the partner's pane, which feels
|
|
304
|
+
like contact and transmits nothing — that is exactly how this gets skipped by a seat following the
|
|
305
|
+
protocol correctly.
|
|
306
|
+
|
|
307
|
+
The notice is ONE line (a newline submits early), sent with `herdr pane run` and **read back**, and it
|
|
308
|
+
carries four things:
|
|
309
|
+
```bash
|
|
310
|
+
herdr pane run <partner> "RETURN NOTICE <seat>: back on <my-pane> since <HH:MM>Z. WATCHERS I NOW HOLD:
|
|
311
|
+
<list>. SWEPT: <what was dead>. MISSION STATE AS I READ IT: <one clause>. Reply with every watcher YOU
|
|
312
|
+
still hold so we deconflict — do not arm anything I just named."
|
|
313
|
+
```
|
|
314
|
+
- **where and since when**, so the partner can stop holding your duties;
|
|
315
|
+
- **the watcher inventory you now hold** — coverage is the thing both tiers silently assume about each
|
|
316
|
+
other, and a returning seat that re-arms without saying so produces double-coverage that reads as
|
|
317
|
+
redundancy and is actually two tiers each trusting the other;
|
|
318
|
+
- **anything you swept**, because a dead watcher the partner armed is *its* belief about coverage, not
|
|
319
|
+
yours, and it will keep believing it;
|
|
320
|
+
- **an explicit deconfliction request.** Ask for the partner's inventory back; do not infer it.
|
|
321
|
+
|
|
322
|
+
⚠ **RE-ADOPT EVERY DETACHED WATCHER ON RETURN, and prove it from disk.** A detached watcher (`ppid 1`)
|
|
323
|
+
is the one kind that SURVIVES your clear, which is exactly why it rots unattended: `stat` its heartbeat
|
|
324
|
+
and treat **stale as dead**. Measured the same morning (OBS-742): a detached resolution watcher's
|
|
325
|
+
heartbeat was **157 minutes stale** and the task it existed to report had resolved **2h04m after its
|
|
326
|
+
last beat** — present, silent, and indistinguishable from healthy to anyone who did not look. Sweep it,
|
|
327
|
+
archive the heartbeat rather than deleting it so the gap stays measurable, and name it in the notice.
|
|
328
|
+
|
|
329
|
+
⚠ **A CLEAR AND A DEATH ARE THE SAME SILENCE.** A seat that clears and never returns — wrong pane,
|
|
330
|
+
crashed host, operator closed the tab — is indistinguishable from one mid-`/clear`. That is why step 4
|
|
331
|
+
states a deadline: **partner silent past it → escalate to the operator.** Without the deadline the
|
|
332
|
+
protocol's most dangerous state has no timeout, and the surviving tier waits forever on a peer that no
|
|
333
|
+
longer exists.
|
|
334
|
+
|
|
335
|
+
⚠ **`∑ NNNk tok` ON A CLAUDE STATUSLINE IS CUMULATIVE SESSION SPEND, NOT CONTEXT FILL.** The percentage
|
|
336
|
+
is the fill; the token total is what has been spent across every turn and keeps climbing after a compaction
|
|
337
|
+
or a clear. **Measured 2026-08-28, expensively:** this seat built a fallback watcher on `∑ Nk tok`, read
|
|
338
|
+
`970k` as 97% of a 1M window, and sent an urgent clear-order to an orchestrator that was actually at
|
|
339
|
+
**33%** — which then began a handoff and offered to discard a session two-thirds fresh, mid-endgame.
|
|
340
|
+
**Read the `%`. Never derive fill from the token total, and never build an instrument on a signal whose
|
|
341
|
+
SEMANTICS you have not verified against a second source.**
|
|
342
|
+
|
|
343
|
+
**Why 50% and not 85%:** a handoff written at 50% is written by a seat whose judgment is intact. One
|
|
344
|
+
written at 85% is written by a seat already degraded, about the decisions it is least able to summarise.
|
|
345
|
+
The threshold buys judgment, not headroom.
|
|
346
|
+
|
|
347
|
+
**Why the OTHER tier issues it:** a self-issued clear competes with whatever the seat is doing and loses.
|
|
348
|
+
An instruction from the other tier arrives as work, and the tier issuing it is not the tier that has to
|
|
349
|
+
overcome its own momentum to obey.
|
|
350
|
+
|
|
351
|
+
⚠ **A detached watcher gives COVERAGE and takes away NOTIFICATION.** A wake written to a log file that no
|
|
352
|
+
seat reads is not a wake. If the watcher must outlive a turn, it also needs a path that reaches a seat —
|
|
353
|
+
a beat the other tier reads, a notification, or an artifact the other tier watches. **Measured 2026-08-28:
|
|
354
|
+
`ACT: orch-215 at 970k/1000k — handoff + /clear NOW` fired correctly and sat unread in a scratchpad log
|
|
355
|
+
while the orchestrator kept working.**
|
|
356
|
+
|
|
252
357
|
Arm a context watcher on the orchestrator at spawn time and treat a threshold wake as a first-class event:
|
|
253
358
|
finish the step, write a handoff, `/clear` **plus a fresh brief — never `/compact`**, because a compaction
|
|
254
359
|
is a lossy summary nobody trusts while a clean session re-oriented from disk-verifiable state is reliable.
|
|
@@ -257,8 +362,9 @@ good, not after. If your own context cannot be read by the watcher, say so to th
|
|
|
257
362
|
number — an unmeasured budget is not a small budget.
|
|
258
363
|
|
|
259
364
|
```bash
|
|
260
|
-
|
|
261
|
-
.claude/skills/tickmarkr-overseer/scripts/watch-context.sh
|
|
365
|
+
# WARN 50 / ACT 50 — the mutual-clear threshold above, not a headroom alarm.
|
|
366
|
+
.claude/skills/tickmarkr-overseer/scripts/watch-context.sh orchestrator <orchestrator-agent-or-pane> 50 50 <handoff-file>
|
|
367
|
+
.claude/skills/tickmarkr-overseer/scripts/watch-context.sh overseer <overseer-agent-or-pane> 50 50 <handoff-file>
|
|
262
368
|
```
|
|
263
369
|
|
|
264
370
|
The first argument chooses the closed per-seat tier (`orchestrator-context` or `overseer-context`),
|
|
@@ -947,6 +1053,17 @@ twice.** They are mission-independent on purpose: nothing here names a task, a l
|
|
|
947
1053
|
concluded both forms were valid. The tell is unavailable unless the tool volunteers it. Corollary —
|
|
948
1054
|
an instrument that takes an input must be handed a DELIBERATELY BAD one before its clean runs are
|
|
949
1055
|
worth anything (rule 11 applied to tools, not just to gates).
|
|
1056
|
+
⚠ **AND THE COMMONEST WRONG INPUT IS A BASE REF: after the first merge, a task's diff against the
|
|
1057
|
+
run's `baseRef` is NEVER that task's diff.** Workers branch from the INTEGRATION TIP, so once any task
|
|
1058
|
+
has merged, `git diff baseRef..HEAD` in a later worktree reports that task PLUS every task merged
|
|
1059
|
+
before it, and the number looks entirely plausible. Diff from the task's OWN base — the integration
|
|
1060
|
+
commit it branched from — and say which base you used whenever you quote a size.
|
|
1061
|
+
**Measured 2026-08-28 in one run, twice, in both directions.** A supervising seat quoted "712
|
|
1062
|
+
insertions across 7 files" for a task whose real contribution was **300 across 2**; the surplus was two
|
|
1063
|
+
other tasks' merged work. On the next task the same trap was **larger** — 920 across 11 versus a true
|
|
1064
|
+
167 across 4 — and it was caught only because the other tier had just been burned by it. A scope
|
|
1065
|
+
judgement, a cost claim, or a review-size argument built on the baseRef diff is measuring three tasks
|
|
1066
|
+
and calling it one.
|
|
950
1067
|
14. **A unit is not a measurement.** A configured timeout is a KILL CEILING, not a duration — never compare
|
|
951
1068
|
it to a wall clock or quote it to an operator as an estimate.
|
|
952
1069
|
15. **Verify through the path that LOADS, not the path you edited.** Mirrored trees and symlinks mean your
|