tickmarkr 2.1.7 → 2.1.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/commands/doctor.d.ts +3 -0
- package/dist/cli/commands/doctor.js +23 -1
- package/dist/cli/commands/plan.d.ts +6 -2
- package/dist/cli/commands/plan.js +94 -1
- package/dist/cli/commands/status.js +50 -10
- package/dist/compile/index.js +15 -5
- package/dist/compile/native.js +30 -0
- package/dist/compile/ownership.d.ts +12 -2
- package/dist/compile/ownership.js +84 -13
- package/dist/drivers/herdr.d.ts +7 -5
- package/dist/drivers/herdr.js +142 -11
- package/dist/drivers/types.d.ts +5 -6
- package/dist/drivers/types.js +2 -48
- package/dist/gates/acceptance.d.ts +13 -0
- package/dist/gates/acceptance.js +43 -16
- package/dist/gates/run-gates.js +26 -23
- package/package.json +1 -1
- package/skills/tickmarkr-auto/SKILL.md +1 -1
- package/skills/tickmarkr-loop/SKILL.md +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +196 -24
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +109 -4
- package/skills/tickmarkr-overseer/scripts/watch-journal.sh +137 -0
package/dist/drivers/herdr.js
CHANGED
|
@@ -6,6 +6,7 @@ import { declaredInputBoxForWorkerName, matchesEmptyInputBox, matchesInputBox, m
|
|
|
6
6
|
import { consumePaneLaunchIntent, PANE_IDENTITY_ENV, paneIdentityLine } from "../brand.js";
|
|
7
7
|
import { createWorktree, sh } from "../run/git.js";
|
|
8
8
|
import { Journal } from "../run/journal.js";
|
|
9
|
+
import { readSupervision } from "../run/supervision.js";
|
|
9
10
|
import { herdrSealShellPrefix } from "./subprocess.js";
|
|
10
11
|
import { canonicalizeLegacyName, formatOwnedName, panesToClose, parseOwnedName } from "./types.js";
|
|
11
12
|
// VIS-09 P43-03: adopted safety floor from 43-MEASUREMENT.md (narrowest safe 53 → floor 108).
|
|
@@ -194,6 +195,64 @@ export class HerdrDriver {
|
|
|
194
195
|
// recovery to the file and the pipe while the operator's rail stays silent about it.
|
|
195
196
|
Journal.open(repoRoot, owned.runId, this.narrate).append("dispatch-retry", owned.taskId, data);
|
|
196
197
|
}
|
|
198
|
+
openRunJournal(runId) {
|
|
199
|
+
const roots = new Set(this.journalRoots.values());
|
|
200
|
+
roots.add(process.cwd());
|
|
201
|
+
for (const repoRoot of roots) {
|
|
202
|
+
try {
|
|
203
|
+
return { repoRoot, journal: Journal.open(repoRoot, runId, this.narrate) };
|
|
204
|
+
}
|
|
205
|
+
catch {
|
|
206
|
+
/* try the next known root */
|
|
207
|
+
}
|
|
208
|
+
}
|
|
209
|
+
return undefined;
|
|
210
|
+
}
|
|
211
|
+
liveSupervisionSeats(repoRoot) {
|
|
212
|
+
const seats = new Set();
|
|
213
|
+
for (const tier of readSupervision(repoRoot)) {
|
|
214
|
+
if (tier.state === "ARMED" && tier.seat)
|
|
215
|
+
seats.add(tier.seat);
|
|
216
|
+
}
|
|
217
|
+
return seats;
|
|
218
|
+
}
|
|
219
|
+
journalReconcile(handle, event, taskId, data) {
|
|
220
|
+
try {
|
|
221
|
+
handle?.journal.append(event, taskId, data);
|
|
222
|
+
}
|
|
223
|
+
catch {
|
|
224
|
+
/* reconcile remains cosmetic even when its audit row cannot be written */
|
|
225
|
+
}
|
|
226
|
+
}
|
|
227
|
+
paneReconcileData(pane, label, ownedName, sweeperRunId) {
|
|
228
|
+
return {
|
|
229
|
+
paneId: pane.paneId,
|
|
230
|
+
...(pane.tabId !== undefined ? { tabId: pane.tabId } : {}),
|
|
231
|
+
label,
|
|
232
|
+
ownedName,
|
|
233
|
+
ownedRunId: ownedName.runId,
|
|
234
|
+
runId: sweeperRunId,
|
|
235
|
+
sweeperRunId,
|
|
236
|
+
};
|
|
237
|
+
}
|
|
238
|
+
parsePaneList(handle, runId, stdout, stage) {
|
|
239
|
+
try {
|
|
240
|
+
const panes = JSON.parse(stdout).result?.panes;
|
|
241
|
+
if (!Array.isArray(panes))
|
|
242
|
+
throw new Error("pane list returned no panes array");
|
|
243
|
+
return panes;
|
|
244
|
+
}
|
|
245
|
+
catch (error) {
|
|
246
|
+
this.journalReconcile(handle, "pane-reconcile-list-failed", undefined, {
|
|
247
|
+
runId,
|
|
248
|
+
sweeperRunId: runId,
|
|
249
|
+
stage,
|
|
250
|
+
error: error instanceof Error ? error.message : String(error),
|
|
251
|
+
stdout,
|
|
252
|
+
});
|
|
253
|
+
return null;
|
|
254
|
+
}
|
|
255
|
+
}
|
|
197
256
|
/** v1.99 T2: bind this driver's own journal writes to the run's live narration sink. */
|
|
198
257
|
narrateWith(narrate) {
|
|
199
258
|
this.narrate = narrate;
|
|
@@ -1199,6 +1258,7 @@ export class HerdrDriver {
|
|
|
1199
1258
|
// resume/end) run with nothing in flight and take them too. Cosmetic by contract: every failure —
|
|
1200
1259
|
// herdr gone, pane vanished mid-sweep, unparseable listing — is swallowed; this method never throws.
|
|
1201
1260
|
async reconcile(desired, runId, opts) {
|
|
1261
|
+
const journalHandle = this.openRunJournal(runId);
|
|
1202
1262
|
try {
|
|
1203
1263
|
if (!this.ws)
|
|
1204
1264
|
return;
|
|
@@ -1206,25 +1266,96 @@ export class HerdrDriver {
|
|
|
1206
1266
|
// panesToClose skips any pane whose label doesn't parse as tickmarkr-owned (orchestrator/operator
|
|
1207
1267
|
// shells, undetected agents), so a fuller pane listing never widens the blast radius.
|
|
1208
1268
|
const list = await this.herdr("pane list");
|
|
1209
|
-
|
|
1210
|
-
|
|
1269
|
+
if (list.code !== 0) {
|
|
1270
|
+
this.journalReconcile(journalHandle, "pane-reconcile-list-failed", undefined, {
|
|
1271
|
+
runId,
|
|
1272
|
+
sweeperRunId: runId,
|
|
1273
|
+
stage: "pre-close",
|
|
1274
|
+
exitCode: list.code,
|
|
1275
|
+
error: list.stderr || list.stdout || `exit ${list.code}`,
|
|
1276
|
+
});
|
|
1277
|
+
return;
|
|
1278
|
+
}
|
|
1279
|
+
const panes = this.parsePaneList(journalHandle, runId, list.stdout, "pre-close");
|
|
1280
|
+
if (panes === null)
|
|
1281
|
+
return;
|
|
1282
|
+
const liveSeats = journalHandle ? this.liveSupervisionSeats(journalHandle.repoRoot) : new Set();
|
|
1283
|
+
for (const seat of opts?.liveSeats ?? [])
|
|
1284
|
+
liveSeats.add(seat);
|
|
1285
|
+
const toClose = panesToClose(panes.map((p) => ({ name: p.label, paneId: p.pane_id, tabId: p.tab_id, workspaceId: p.workspace_id })), desired, this.ws, runId, { ...opts, liveSeats });
|
|
1286
|
+
const paneById = new Map(panes
|
|
1287
|
+
.filter((p) => typeof p.pane_id === "string")
|
|
1288
|
+
.map((p) => [p.pane_id, p]));
|
|
1211
1289
|
const touched = new Set();
|
|
1212
1290
|
for (const c of toClose) {
|
|
1213
|
-
|
|
1214
|
-
|
|
1215
|
-
|
|
1291
|
+
const listed = paneById.get(c.paneId);
|
|
1292
|
+
const label = typeof listed?.label === "string" ? listed.label : "";
|
|
1293
|
+
const ownedName = parseOwnedName(label);
|
|
1294
|
+
if (!ownedName) {
|
|
1295
|
+
this.journalReconcile(journalHandle, "pane-reconcile-close-failed", undefined, {
|
|
1296
|
+
paneId: c.paneId,
|
|
1297
|
+
...(c.tabId !== undefined ? { tabId: c.tabId } : {}),
|
|
1298
|
+
label,
|
|
1299
|
+
runId,
|
|
1300
|
+
sweeperRunId: runId,
|
|
1301
|
+
error: "pane selected for reconcile no longer has a parseable owned label",
|
|
1302
|
+
});
|
|
1303
|
+
continue;
|
|
1304
|
+
}
|
|
1305
|
+
const data = this.paneReconcileData(c, label, ownedName, runId);
|
|
1306
|
+
const closed = await this.herdr(`pane close ${shq(c.paneId)}`);
|
|
1307
|
+
if (closed.code === 0) {
|
|
1308
|
+
if (typeof c.tabId === "string")
|
|
1309
|
+
touched.add(c.tabId);
|
|
1310
|
+
this.journalReconcile(journalHandle, "pane-reconcile-close", ownedName.taskId, data);
|
|
1311
|
+
}
|
|
1312
|
+
else {
|
|
1313
|
+
this.journalReconcile(journalHandle, "pane-reconcile-close-failed", ownedName.taskId, {
|
|
1314
|
+
...data,
|
|
1315
|
+
exitCode: closed.code,
|
|
1316
|
+
error: closed.stderr || closed.stdout || `exit ${closed.code}`,
|
|
1317
|
+
});
|
|
1318
|
+
}
|
|
1216
1319
|
}
|
|
1217
1320
|
if (touched.size === 0)
|
|
1218
1321
|
return;
|
|
1219
1322
|
// a tab our closes emptied was ours by construction (a tab with operator panes still has panes)
|
|
1220
1323
|
const pl = await this.herdr("pane list");
|
|
1221
|
-
|
|
1222
|
-
|
|
1223
|
-
|
|
1224
|
-
|
|
1324
|
+
if (pl.code !== 0) {
|
|
1325
|
+
this.journalReconcile(journalHandle, "pane-reconcile-list-failed", undefined, {
|
|
1326
|
+
runId,
|
|
1327
|
+
sweeperRunId: runId,
|
|
1328
|
+
stage: "post-close",
|
|
1329
|
+
exitCode: pl.code,
|
|
1330
|
+
error: pl.stderr || pl.stdout || `exit ${pl.code}`,
|
|
1331
|
+
});
|
|
1332
|
+
return;
|
|
1333
|
+
}
|
|
1334
|
+
const alivePanes = this.parsePaneList(journalHandle, runId, pl.stdout, "post-close");
|
|
1335
|
+
if (alivePanes === null)
|
|
1336
|
+
return;
|
|
1337
|
+
const alive = new Set(alivePanes.map((p) => p.tab_id));
|
|
1338
|
+
for (const tab of touched) {
|
|
1339
|
+
if (alive.has(tab))
|
|
1340
|
+
continue;
|
|
1341
|
+
const closed = await this.herdr(`tab close ${shq(tab)}`);
|
|
1342
|
+
if (closed.code !== 0) {
|
|
1343
|
+
this.journalReconcile(journalHandle, "tab-reconcile-close-failed", undefined, {
|
|
1344
|
+
tabId: tab,
|
|
1345
|
+
runId,
|
|
1346
|
+
sweeperRunId: runId,
|
|
1347
|
+
exitCode: closed.code,
|
|
1348
|
+
error: closed.stderr || closed.stdout || `exit ${closed.code}`,
|
|
1349
|
+
});
|
|
1350
|
+
}
|
|
1351
|
+
}
|
|
1225
1352
|
}
|
|
1226
|
-
catch {
|
|
1227
|
-
|
|
1353
|
+
catch (error) {
|
|
1354
|
+
this.journalReconcile(journalHandle, "pane-reconcile-failed", undefined, {
|
|
1355
|
+
runId,
|
|
1356
|
+
sweeperRunId: runId,
|
|
1357
|
+
error: error instanceof Error ? error.message : String(error),
|
|
1358
|
+
});
|
|
1228
1359
|
}
|
|
1229
1360
|
}
|
|
1230
1361
|
async worktree(repo, branch, baseRef) {
|
package/dist/drivers/types.d.ts
CHANGED
|
@@ -54,10 +54,12 @@ export interface FleetAgent {
|
|
|
54
54
|
tabId?: string;
|
|
55
55
|
workspaceId?: string;
|
|
56
56
|
}
|
|
57
|
-
export
|
|
57
|
+
export interface PanesToCloseOpts {
|
|
58
58
|
spareLiveLlm?: boolean;
|
|
59
59
|
endedRunIds?: Set<string>;
|
|
60
|
-
|
|
60
|
+
liveSeats?: Set<string>;
|
|
61
|
+
}
|
|
62
|
+
export declare function panesToClose(agents: FleetAgent[], desired: Set<string>, ws: string, runId: string, opts?: PanesToCloseOpts): {
|
|
61
63
|
paneId: string;
|
|
62
64
|
tabId?: string;
|
|
63
65
|
}[];
|
|
@@ -80,8 +82,5 @@ export interface ExecutorDriver {
|
|
|
80
82
|
narrateWith?(narrate: (event: JournalEvent) => void): void;
|
|
81
83
|
worktree(repo: string, branch: string, baseRef: string): Promise<string>;
|
|
82
84
|
narrator?: (cwd: string, command: string, runId?: string) => Promise<Slot>;
|
|
83
|
-
reconcile?: (desired: Set<string>, runId: string, opts?:
|
|
84
|
-
spareLiveLlm?: boolean;
|
|
85
|
-
endedRunIds?: Set<string>;
|
|
86
|
-
}) => Promise<void>;
|
|
85
|
+
reconcile?: (desired: Set<string>, runId: string, opts?: PanesToCloseOpts) => Promise<void>;
|
|
87
86
|
}
|
package/dist/drivers/types.js
CHANGED
|
@@ -20,54 +20,6 @@ export function parseOwnedName(name) {
|
|
|
20
20
|
export function isForeignName(name) {
|
|
21
21
|
return parseOwnedName(name) === null;
|
|
22
22
|
}
|
|
23
|
-
// v1.22b T1: workspace-aware fold over a fleet snapshot — decides which owned task panes are garbage
|
|
24
|
-
// right now: the desired-set/spareLiveLlm sweep (OBS-17 T2), scoped to THIS RUN'S OWN panes (by runId,
|
|
25
|
-
// OBS-772) in THIS RUN'S OWN WORKSPACE (OBS-769). Both conditions, and neither alone is the rule.
|
|
26
|
-
// Watch panes are operator-owned after run end and are reclaimed by the next run; foreign names
|
|
27
|
-
// (parseOwnedName fails) are never candidates.
|
|
28
|
-
//
|
|
29
|
-
// OBS-769 — WHY THE SWEEP STOPS AT THE WORKSPACE BOUNDARY. It used to close an owned pane carrying
|
|
30
|
-
// any OTHER runId in any other workspace, unconditionally, as a "misplaced leftover". Two tickmarkr
|
|
31
|
-
// runs in two repositories are lawful (the lock forbids two runs in ONE repository, not on one
|
|
32
|
-
// machine) and herdr gives each its own workspace — so that branch made every pair of concurrent
|
|
33
|
-
// runs kill each other's LIVE workers. Measured 2026-08-28: the run in w0 closed run ...2958's
|
|
34
|
-
// panes at 23:42:40.351/.392, and 53s later ...2958's own task-human sweep closed w0's live codex
|
|
35
|
-
// worker at 23:43:34.096. ...2958 ended 0/8. The death detector cannot see it: closing the pane
|
|
36
|
-
// makes paneAbsent, processTree, confirmedProcessTree and worktreeDelta true by ONE cause, and a
|
|
37
|
-
// closed pane can never accrue the CPU that the `cpu-accruing` hold reads.
|
|
38
|
-
// The comment this replaces claimed "only run age marks a misplaced pane garbage" — there was no age
|
|
39
|
-
// check in the code, and age is the wrong predicate anyway: w0's run STARTED EARLIER than ...2958,
|
|
40
|
-
// so an age rule would have licensed exactly the kill that landed. Run age says nothing about
|
|
41
|
-
// liveness, and a sweeping daemon cannot read another repository's run state. The workspace is the
|
|
42
|
-
// only ownership boundary available without cross-repo I/O, so it is the one enforced.
|
|
43
|
-
// Cost, named: an orphan pane from a dead run stranded in a workspace no later run opens is now left
|
|
44
|
-
// for the operator. That is cosmetic (`reconcile` is cosmetic by contract — "visibility is never a
|
|
45
|
-
// gate"), and a cosmetic cleanup must never be able to kill a live worker.
|
|
46
|
-
// OBS-772 — WHY THE runId LINE EXISTS, AND WHY THE WORKSPACE LINE ALONE WAS NOT THE FIX. The first
|
|
47
|
-
// repair was workspace-scoped only, and its own comment dismissed the residue — "two runs sharing one
|
|
48
|
-
// workspace would still sweep each other" — as unreachable, on the reasoning that one workspace per run
|
|
49
|
-
// is herdr's placement. That reasoned from ONE driver to the whole product. OrcaDriver has no workspace
|
|
50
|
-
// dimension at all: orca.ts passes a single ORCA_SPACE as the workspaceId for EVERY checkout and as
|
|
51
|
-
// `ws`, so `workspaceId !== ws` is never true there and every foreign pane fell straight through. Orca
|
|
52
|
-
// users had zero protection while the defect read as fixed. The runId line is the real rule and it is
|
|
53
|
-
// driver-agnostic: reconcile exists to clean up THIS RUN's panes, and a leftover from a dead run is
|
|
54
|
-
// exactly what cannot be told from a live run's pane without liveness data this process does not have.
|
|
55
|
-
// Both lines are kept — the workspace line preserves the pre-existing sparing of this run's own panes
|
|
56
|
-
// in another workspace, which the runId line alone would not.
|
|
57
|
-
// ⚠ WHAT THE runId LINE COST BEFORE OBS-777 — SUSPENDED, NOT NARROWED, and the price was larger than
|
|
58
|
-
// it read. Sparing every other runId suspended OBS-17's FOUNDING use case: "a killed daemon can't
|
|
59
|
-
// close its slots". This sweep was built to reclaim exactly those orphans, but could not reclaim ANY
|
|
60
|
-
// previous run's panes. Three separate pins asserted the old behaviour (reconcile.test.ts,
|
|
61
|
-
// orca-placement.test.ts,
|
|
62
|
-
// reconcile-live.test.ts); all three were changed deliberately, and the third is why this paragraph
|
|
63
|
-
// exists rather than a shorter one — two flipped pins is a trade, three is a pattern.
|
|
64
|
-
// OBS-777 RESTORES that reclamation: the CALLER passes `opts.endedRunIds`, a Set the daemon computes
|
|
65
|
-
// ONCE at run start from this repository's own `run-end` journals and dead lock holders. This fold
|
|
66
|
-
// stays pure — it gains one optional field, not a repo root — a foreign repository's runId is never
|
|
67
|
-
// resolvable and so stays spared by construction, and no driver learns about workspaces.
|
|
68
|
-
// ponytail: two conditions, no geometry reasoning, nothing driver-specific. `reconcile` is cosmetic by
|
|
69
|
-
// contract, and a cosmetic cleanup must never be able to kill a live worker — which is why the
|
|
70
|
-
// ended-run authority is the only safe way to restore the sweep without reviving the cross-run kill.
|
|
71
23
|
export function panesToClose(agents, desired, ws, runId, opts) {
|
|
72
24
|
const out = [];
|
|
73
25
|
for (const a of agents) {
|
|
@@ -76,6 +28,8 @@ export function panesToClose(agents, desired, ws, runId, opts) {
|
|
|
76
28
|
const owned = parseOwnedName(a.name);
|
|
77
29
|
if (!owned || owned.role === "watch")
|
|
78
30
|
continue;
|
|
31
|
+
if (opts?.liveSeats?.has(a.name))
|
|
32
|
+
continue;
|
|
79
33
|
if (owned.runId !== runId && !opts?.endedRunIds?.has(owned.runId))
|
|
80
34
|
continue;
|
|
81
35
|
if (a.workspaceId !== ws)
|
|
@@ -28,6 +28,19 @@ export interface VitestListedTest {
|
|
|
28
28
|
file: string;
|
|
29
29
|
projectName?: string;
|
|
30
30
|
}
|
|
31
|
+
export type VitestListResult = {
|
|
32
|
+
status: "listed";
|
|
33
|
+
tests: VitestListedTest[];
|
|
34
|
+
} | {
|
|
35
|
+
status: "failed";
|
|
36
|
+
error: string;
|
|
37
|
+
};
|
|
38
|
+
export declare function listVitestTests(cwd: string): Promise<VitestListResult>;
|
|
39
|
+
export interface NamedTestAudit {
|
|
40
|
+
criterion: string;
|
|
41
|
+
matches: VitestListedTest[];
|
|
42
|
+
}
|
|
43
|
+
export declare function auditNamedTestOracles(items: readonly AcceptanceItem[], listedTests: readonly VitestListedTest[]): NamedTestAudit[];
|
|
31
44
|
export type AcceptanceCorpusAuditResult = {
|
|
32
45
|
specPath: string;
|
|
33
46
|
status: "parse-failed";
|
package/dist/gates/acceptance.js
CHANGED
|
@@ -98,6 +98,47 @@ export function testFiltered(testCmd, name) {
|
|
|
98
98
|
const fwd = wrapped ? "-- " : "";
|
|
99
99
|
return `${testCmd} ${fwd}-t ${shq(pattern)}`;
|
|
100
100
|
}
|
|
101
|
+
const VitestListedTestsSchema = z.array(z.object({
|
|
102
|
+
name: z.string(),
|
|
103
|
+
file: z.string(),
|
|
104
|
+
projectName: z.string().optional(),
|
|
105
|
+
}));
|
|
106
|
+
export async function listVitestTests(cwd) {
|
|
107
|
+
const result = await sh(`${shq(join(cwd, "node_modules/.bin/vitest"))} list --json`, cwd);
|
|
108
|
+
if (result.code !== 0) {
|
|
109
|
+
return { status: "failed", error: (result.stderr || result.stdout || `exit ${result.code}`).trim() };
|
|
110
|
+
}
|
|
111
|
+
try {
|
|
112
|
+
const start = result.stdout.indexOf("[");
|
|
113
|
+
if (start < 0)
|
|
114
|
+
return { status: "failed", error: "runner emitted no JSON test listing" };
|
|
115
|
+
const parsed = VitestListedTestsSchema.safeParse(JSON.parse(result.stdout.slice(start)));
|
|
116
|
+
return parsed.success
|
|
117
|
+
? { status: "listed", tests: parsed.data }
|
|
118
|
+
: { status: "failed", error: z.prettifyError(parsed.error) };
|
|
119
|
+
}
|
|
120
|
+
catch (error) {
|
|
121
|
+
return { status: "failed", error: error instanceof Error ? error.message : String(error) };
|
|
122
|
+
}
|
|
123
|
+
}
|
|
124
|
+
export function auditNamedTestOracles(items, listedTests) {
|
|
125
|
+
const runnerNames = listedTests.map((listed) => ({
|
|
126
|
+
listed,
|
|
127
|
+
fullName: listed.name.split(" > ").join(" "),
|
|
128
|
+
}));
|
|
129
|
+
return items.flatMap((item) => {
|
|
130
|
+
if (typeof item !== "object" || item.oracle !== "test")
|
|
131
|
+
return [];
|
|
132
|
+
return [{
|
|
133
|
+
criterion: item.test,
|
|
134
|
+
// OBS-511: mirror the gate's leaf-anchored suffix rule — this denominator must count
|
|
135
|
+
// exactly the tests the shipped -t filter would select.
|
|
136
|
+
matches: runnerNames
|
|
137
|
+
.filter(({ fullName }) => fullName === item.test || fullName.endsWith(` ${item.test}`))
|
|
138
|
+
.map(({ listed }) => listed),
|
|
139
|
+
}];
|
|
140
|
+
});
|
|
141
|
+
}
|
|
101
142
|
function corpusSpecPaths(root) {
|
|
102
143
|
const paths = [];
|
|
103
144
|
const visit = (dir) => {
|
|
@@ -116,32 +157,18 @@ function corpusSpecPaths(root) {
|
|
|
116
157
|
// listing. Every discovered path contributes either all parser-produced acceptance items or one named
|
|
117
158
|
// parse failure; exceptions are evidence, never permission to shrink the corpus silently.
|
|
118
159
|
export function auditAcceptanceCorpus(corpusRoot, listedTests) {
|
|
119
|
-
const runnerNames = listedTests.map((listed) => ({
|
|
120
|
-
listed,
|
|
121
|
-
fullName: listed.name.split(" > ").join(" "),
|
|
122
|
-
}));
|
|
123
160
|
const results = [];
|
|
124
161
|
for (const specPath of corpusSpecPaths(corpusRoot)) {
|
|
125
162
|
try {
|
|
126
163
|
const graph = compileNative(specPath);
|
|
127
164
|
for (const task of graph.tasks) {
|
|
128
165
|
for (const item of task.acceptance) {
|
|
166
|
+
const namedTest = auditNamedTestOracles([item], listedTests)[0];
|
|
129
167
|
results.push({
|
|
130
168
|
specPath,
|
|
131
169
|
status: "parsed",
|
|
132
170
|
item,
|
|
133
|
-
...(
|
|
134
|
-
? {
|
|
135
|
-
namedTest: {
|
|
136
|
-
criterion: item.test,
|
|
137
|
-
// OBS-511: mirror the gate's leaf-anchored suffix rule — the audit's denominator
|
|
138
|
-
// must count exactly the tests the -t filter would select, or doctor and gate disagree.
|
|
139
|
-
matches: runnerNames
|
|
140
|
-
.filter(({ fullName }) => fullName === item.test || fullName.endsWith(` ${item.test}`))
|
|
141
|
-
.map(({ listed }) => listed),
|
|
142
|
-
},
|
|
143
|
-
}
|
|
144
|
-
: {}),
|
|
171
|
+
...(namedTest ? { namedTest } : {}),
|
|
145
172
|
});
|
|
146
173
|
}
|
|
147
174
|
}
|
package/dist/gates/run-gates.js
CHANGED
|
@@ -176,11 +176,11 @@ async function coveringTests(worktree, baseRef) {
|
|
|
176
176
|
}
|
|
177
177
|
const covering = tests.filter((t) => reachOf(t).has(file));
|
|
178
178
|
if (!covering.length)
|
|
179
|
-
|
|
179
|
+
continue; // nothing covers this file — keep every attributable selection already accumulated
|
|
180
180
|
for (const t of covering)
|
|
181
181
|
selected.add(t);
|
|
182
182
|
}
|
|
183
|
-
return [...selected].sort();
|
|
183
|
+
return selected.size ? [...selected].sort() : undefined;
|
|
184
184
|
}
|
|
185
185
|
/**
|
|
186
186
|
* The configured test command narrowed to these files. Mirrors testFiltered's `--` rule (acceptance.ts:104):
|
|
@@ -287,7 +287,7 @@ export async function runGates(task, ctx) {
|
|
|
287
287
|
}
|
|
288
288
|
return { results: sorted, commits };
|
|
289
289
|
};
|
|
290
|
-
const toolGates = ["build", "
|
|
290
|
+
const toolGates = ["build", "lint"].filter(enabled);
|
|
291
291
|
/**
|
|
292
292
|
* v1.87 T5: the shell gates run their commands against the WORKING TREE, while evidence, scope,
|
|
293
293
|
* the judged diff and the merge all read COMMITS. Uncommitted work is therefore visible to
|
|
@@ -337,28 +337,28 @@ export async function runGates(task, ctx) {
|
|
|
337
337
|
+ `Uncommitted at round end:\n${dirt}`,
|
|
338
338
|
meta: { dirtyWorktree: true, dirtyAtRoundEnd: true },
|
|
339
339
|
});
|
|
340
|
-
//
|
|
341
|
-
const runBattery = async (commands, selected) => {
|
|
342
|
-
if (!
|
|
340
|
+
// shell tools vs the shared baseline
|
|
341
|
+
const runBattery = async (commands, selected, gates = toolGates) => {
|
|
342
|
+
if (!gates.length)
|
|
343
343
|
return;
|
|
344
344
|
if (!v185) {
|
|
345
|
-
// ponytail: compareToBaseline batches
|
|
345
|
+
// ponytail: compareToBaseline batches adjacent tools — their starts are emitted at iteration,
|
|
346
346
|
// not at true execution start. They are collectively sub-second (measured), so the debounce
|
|
347
347
|
// suppresses them anyway; split compareToBaseline only if a tool gate ever gets slow.
|
|
348
|
-
// ponytail: legacy runs
|
|
348
|
+
// ponytail: legacy runs adjacent tools in ONE compareToBaseline call, so there is one interval
|
|
349
349
|
// to measure and each of its gates carries it. Split it only if this branch ever stops batching.
|
|
350
350
|
const batchAt = Date.now();
|
|
351
351
|
const batchLoadStart = loadProvider();
|
|
352
|
-
const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline,
|
|
352
|
+
const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [...gates]);
|
|
353
353
|
const batch = { durationMs: Date.now() - batchAt, load1Start: batchLoadStart, load1End: loadProvider() };
|
|
354
|
-
for (const g of
|
|
354
|
+
for (const g of gates)
|
|
355
355
|
spans.set(g, batch);
|
|
356
356
|
// The same refusal AFTER the commands, because a green command can dirty the tree the check
|
|
357
357
|
// above just proved clean. Batched, legacy cannot say WHICH command did it, so the refusal
|
|
358
358
|
// lands on the last gate that had one — the round dies there either way. A red battery is
|
|
359
359
|
// reported as the red it is: the round already ends, and the command output is the better lead.
|
|
360
360
|
const dirt = toolResults.every((r) => r.pass) ? await dirtyWorktree() : undefined;
|
|
361
|
-
const blame = dirt ? [...
|
|
361
|
+
const blame = dirt ? [...gates].reverse().find((g) => commands[g]) : undefined;
|
|
362
362
|
for (const r of toolResults) {
|
|
363
363
|
await emitStart(r.gate);
|
|
364
364
|
await record(r.gate === blame ? dirtyRefusal(blame, dirt, commands[blame]) : r);
|
|
@@ -366,8 +366,8 @@ export async function runGates(task, ctx) {
|
|
|
366
366
|
return;
|
|
367
367
|
}
|
|
368
368
|
// T4 (OBS-265): one command at a time, stopping at the first red — a failed build no longer buys
|
|
369
|
-
//
|
|
370
|
-
for (const g of
|
|
369
|
+
// any later tool before anyone reads its verdict.
|
|
370
|
+
for (const g of gates) {
|
|
371
371
|
await emitStart(g);
|
|
372
372
|
const [r] = await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g]));
|
|
373
373
|
// the screen's interval IS the test gate's first interval, so the split needs no second clock
|
|
@@ -437,9 +437,9 @@ export async function runGates(task, ctx) {
|
|
|
437
437
|
* screen IS the round's verdict: what it produced is journaled (in the order it ran) and the round
|
|
438
438
|
* ends there, so a drive-by out-of-scope edit costs <1s instead of the whole battery.
|
|
439
439
|
*
|
|
440
|
-
* A green screen changes nothing downstream. The
|
|
441
|
-
*
|
|
442
|
-
*
|
|
440
|
+
* A green screen changes nothing downstream. The returned record stays in GATE_NAMES order while
|
|
441
|
+
* the event stream reports the order gates actually ran; resume's GATE_NAMES walk over satisfied
|
|
442
|
+
* records therefore keeps declaration order without making the live stream lie about execution.
|
|
443
443
|
*
|
|
444
444
|
* ponytail: the price of that is re-reading two git checks (~40ms) in their canonical positions
|
|
445
445
|
* rather than teaching every consumer of the gate stream a second order. Both reads see the same
|
|
@@ -447,7 +447,7 @@ export async function runGates(task, ctx) {
|
|
|
447
447
|
* for. Charge it only when there IS a battery command to protect.
|
|
448
448
|
*/
|
|
449
449
|
const screenBlocks = async () => {
|
|
450
|
-
if (!toolGates.some((g) => ctx.commands[g]))
|
|
450
|
+
if (!toolGates.some((g) => ctx.commands[g]) && !(enabled("test") && ctx.commands.test))
|
|
451
451
|
return false;
|
|
452
452
|
const screened = [];
|
|
453
453
|
for (const [gate, compute] of [["evidence", evidenceResult], ["scope", scopeResult]]) {
|
|
@@ -626,12 +626,7 @@ export async function runGates(task, ctx) {
|
|
|
626
626
|
}
|
|
627
627
|
if (v185 && await screenBlocks())
|
|
628
628
|
return done();
|
|
629
|
-
|
|
630
|
-
// pays the full suite anyway, so a selection that misses costs a round and can never merge.
|
|
631
|
-
const selected = v185 && ctx.selectTests && enabled("test") && ctx.commands.test
|
|
632
|
-
? await coveringTests(ctx.worktree, ctx.baseRef)
|
|
633
|
-
: undefined;
|
|
634
|
-
await runBattery(selected ? { ...ctx.commands, test: testCommandForFiles(ctx.commands.test, selected) } : ctx.commands, selected);
|
|
629
|
+
await runBattery(ctx.commands);
|
|
635
630
|
if (failed())
|
|
636
631
|
return done();
|
|
637
632
|
if (enabled("evidence")) {
|
|
@@ -644,6 +639,14 @@ export async function runGates(task, ctx) {
|
|
|
644
639
|
if (failed())
|
|
645
640
|
return done();
|
|
646
641
|
}
|
|
642
|
+
// A non-final round may run only the tests covering its own diff; the merge-candidate round below
|
|
643
|
+
// pays the full suite anyway, so a selection that misses costs a round and can never merge.
|
|
644
|
+
const selected = v185 && ctx.selectTests && enabled("test") && ctx.commands.test
|
|
645
|
+
? await coveringTests(ctx.worktree, ctx.baseRef)
|
|
646
|
+
: undefined;
|
|
647
|
+
await runBattery(selected ? { ...ctx.commands, test: testCommandForFiles(ctx.commands.test, selected) } : ctx.commands, selected, enabled("test") ? ["test"] : []);
|
|
648
|
+
if (failed())
|
|
649
|
+
return done();
|
|
647
650
|
if (v185 && (enabled("acceptance") || enabled("review"))) {
|
|
648
651
|
// Judge and review are launched TOGETHER (96m of serialization over 5 runs). Enforcement is
|
|
649
652
|
// unchanged — it is still the AND of both, both still fail closed, and neither reads the other's
|
package/package.json
CHANGED
|
@@ -83,7 +83,7 @@ After sending, **confirm delivery** by reading the target pane and verifying the
|
|
|
83
83
|
1. **Prepare** — confirm the target list. Run the [binary preflight](#binary-preflight-before-compile-or-run). Check `git status`, confirm no tickmarkr run is active, and work from a non-main branch.
|
|
84
84
|
2. **Compile** — run `tickmarkr compile <spec-or-directory>`. Fix source-spec defects instead of editing the generated graph.
|
|
85
85
|
3. **Plan** — run `tickmarkr plan`. Review routes, capability-floor warnings, and human gates before execution.
|
|
86
|
-
4. **Run** — run `tickmarkr run`. Watch the run journal for its terminal
|
|
86
|
+
4. **Run** — run `tickmarkr run`. Watch the run journal for its terminal events rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every green clause for you. Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the relevant agent session.
|
|
87
87
|
5. **Verify and consolidate** — continue only after a green run. A run is green when the run-end event exists in the journal, the tip verify is not "failed", and the summary's `failed`, `human`, `blocked` and `pending` buckets are all empty — a run with a parked task is partial, not green. Tickmarkr consolidates accepted work on `tickmarkr/<runId>` and never signs off to the main branch. A human controls any later release merge.
|
|
88
88
|
6. **Record** — write `tickmarkr report <runId> --md` beside the source spec and commit the execution record when the repository tracks those records.
|
|
89
89
|
7. **Continue** — move to the next requested target. If a target fails or is parked, stop with the journal evidence rather than silently skipping it.
|
|
@@ -87,6 +87,6 @@ When spawning consultants (agents gathering synthesis input for decisions like S
|
|
|
87
87
|
1. **Prepare** — start from the requested spec. Run the [binary preflight](#binary-preflight-before-compile-or-run). Check `git status`, confirm no tickmarkr run is active, and work from a non-main branch.
|
|
88
88
|
2. **Compile** — run `tickmarkr compile <spec>`. Correct compilation errors in the spec, never in the generated graph.
|
|
89
89
|
3. **Plan** — run `tickmarkr plan`. Review the routing table, capability-floor warnings, and every human gate, including work that each gate blocks.
|
|
90
|
-
4. **Run** — run `tickmarkr run`. Watch the run journal for its terminal
|
|
90
|
+
4. **Run** — run `tickmarkr run`. Watch the run journal for its terminal events rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every green clause for you. Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the agent session; do not turn them into proxy questions.
|
|
91
91
|
5. **Verify and consolidate** — accept only a green run. A run is green when the run-end event exists in the journal, the tip verify is not "failed", and the summary's `failed`, `human`, `blocked` and `pending` buckets are all empty — a run with a parked task is partial, not green. Tickmarkr consolidates accepted task work on `tickmarkr/<runId>`; it never signs off to the main branch. A human may later merge that integration branch through the repository's normal release process.
|
|
92
92
|
6. **Record** — write `tickmarkr report <runId> --md` beside the source spec and commit the execution record when the repository tracks those records. Then [stand down](#stand-down-mission-end-and-retirement).
|