tickmarkr 1.96.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/brand.d.ts +31 -0
- package/dist/brand.js +44 -1
- package/dist/cli/commands/compile.js +32 -1
- package/dist/cli/commands/resume.js +9 -3
- package/dist/cli/commands/run.d.ts +61 -1
- package/dist/cli/commands/run.js +368 -17
- package/dist/cli/commands/status.js +303 -145
- package/dist/compile/collateral.d.ts +25 -0
- package/dist/compile/collateral.js +46 -11
- package/dist/compile/native.js +10 -0
- package/dist/drivers/herdr.d.ts +19 -13
- package/dist/drivers/herdr.js +88 -26
- package/dist/drivers/types.d.ts +2 -0
- package/dist/gates/acceptance.js +17 -7
- package/dist/gates/llm.d.ts +19 -0
- package/dist/gates/llm.js +104 -6
- package/dist/gates/run-gates.d.ts +18 -0
- package/dist/gates/run-gates.js +195 -29
- package/dist/gates/scope.d.ts +9 -1
- package/dist/gates/scope.js +22 -2
- package/dist/graph/graph.d.ts +1 -0
- package/dist/graph/graph.js +19 -2
- package/dist/report/compare.js +17 -2
- package/dist/run/daemon.d.ts +1 -8
- package/dist/run/daemon.js +231 -246
- package/dist/run/environment.d.ts +18 -1
- package/dist/run/environment.js +19 -2
- package/dist/run/journal.d.ts +50 -3
- package/dist/run/journal.js +181 -5
- package/dist/run/protocol.d.ts +4 -4
- package/dist/run/stall.d.ts +30 -0
- package/dist/run/stall.js +173 -0
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +20 -15
- package/skills/tickmarkr-overseer/scripts/seat-send.sh +98 -5
|
@@ -3,8 +3,12 @@ import { extname, join, relative } from "node:path";
|
|
|
3
3
|
import { filesGlob } from "../graph/files-glob.js";
|
|
4
4
|
import { criticalPathHits, DEFAULT_CONFIG, DEFAULT_REVIEW_CRITICAL_PATHS, effectiveReviewPolicy, loadConfig, } from "../config/config.js";
|
|
5
5
|
import { renderAcceptanceItem } from "../graph/schema.js";
|
|
6
|
-
// Advisory
|
|
7
|
-
//
|
|
6
|
+
// Advisory scan (OBS-12/13/14/21, OBS-76). NEVER expands files[] or fails compile — a warning the
|
|
7
|
+
// author acts on. Plain-text (no AST), capped + sorted for DISPLAY only.
|
|
8
|
+
// OBS-547: the collateral prediction is no longer plan-time-only. The daemon computes one full
|
|
9
|
+
// (uncapped) map per run and hands each task's slice to its scope gate, which classifies a red
|
|
10
|
+
// against it — the prediction never causes a refusal, it only says whether a red the gate already
|
|
11
|
+
// found was foreseeable (authoring defect) or not (chargeable quality failure).
|
|
8
12
|
/** Max files walked per root (sorted walk; rest ignored). */
|
|
9
13
|
const MAX_WALK_FILES = 400;
|
|
10
14
|
/** Max collateral test paths listed per task. */
|
|
@@ -95,15 +99,17 @@ function makeReader(repoRoot) {
|
|
|
95
99
|
};
|
|
96
100
|
}
|
|
97
101
|
/**
|
|
98
|
-
*
|
|
99
|
-
*
|
|
102
|
+
* OBS-547: the FULL per-task collateral prediction, uncapped. ONE map is computed per run (daemon.ts,
|
|
103
|
+
* at run start) and handed whole to the scope gate, so a hit the plan display hides is still there
|
|
104
|
+
* when the gate asks about it. The lint lines below are a capped VIEW of this same map — never a
|
|
105
|
+
* second scan. Tasks with no hits are absent.
|
|
100
106
|
*/
|
|
101
|
-
export function
|
|
107
|
+
export function collateralHits(tasks, repoRoot) {
|
|
108
|
+
const map = new Map();
|
|
102
109
|
const testFiles = walkCode(repoRoot, "tests");
|
|
103
110
|
if (!testFiles.length)
|
|
104
|
-
return
|
|
111
|
+
return map;
|
|
105
112
|
const read = makeReader(repoRoot);
|
|
106
|
-
const lines = [];
|
|
107
113
|
for (const t of tasks) {
|
|
108
114
|
// OBS-22: scopeGate accepts picomatch globs; advisory collateral warnings must agree.
|
|
109
115
|
const scoped = filesGlob(t.files.map((f) => f.replace(/^\.\//, "")));
|
|
@@ -122,14 +128,43 @@ export function collateralLints(tasks, repoRoot) {
|
|
|
122
128
|
if (mentions(text, needles))
|
|
123
129
|
hits.push(tf);
|
|
124
130
|
}
|
|
125
|
-
if (!hits.length)
|
|
126
|
-
continue;
|
|
127
131
|
// deterministic: walk already sorted; stable list
|
|
132
|
+
if (hits.length)
|
|
133
|
+
map.set(t.id, hits);
|
|
134
|
+
}
|
|
135
|
+
return map;
|
|
136
|
+
}
|
|
137
|
+
/** The verbatim repair an authoring defect owes: the files[] lines the spec is missing. */
|
|
138
|
+
export function filesRepair(taskId, paths) {
|
|
139
|
+
return paths.map((p) => `add ${p} to ${taskId}.files[]`).join("\n");
|
|
140
|
+
}
|
|
141
|
+
/**
|
|
142
|
+
* OBS-547: cross-reference at the RED. The gate supplies specificity (the paths actually touched),
|
|
143
|
+
* the prediction supplies classification — so there are no false positives to fear and no refusal to
|
|
144
|
+
* author. Feed this the FULL map for the task: a classifier bounded by the display cap calls a
|
|
145
|
+
* predicted 21st hit a quality failure.
|
|
146
|
+
*/
|
|
147
|
+
export function classifyScopeOffenders(taskId, hard, predicted) {
|
|
148
|
+
const named = new Set(predicted);
|
|
149
|
+
const hit = hard.filter((f) => named.has(f));
|
|
150
|
+
const missed = hard.filter((f) => !named.has(f));
|
|
151
|
+
return { authoring: hard.length > 0 && missed.length === 0, predicted: hit, missed, repair: filesRepair(taskId, hit) };
|
|
152
|
+
}
|
|
153
|
+
/**
|
|
154
|
+
* Return human-readable scope-lint lines for plan output (no `!` prefix — plan owns that).
|
|
155
|
+
* Each line names the task id and at least one missing collateral test path.
|
|
156
|
+
*/
|
|
157
|
+
export function collateralLints(tasks, repoRoot) {
|
|
158
|
+
const lines = [];
|
|
159
|
+
for (const [id, hits] of collateralHits(tasks, repoRoot)) {
|
|
128
160
|
const listed = hits.slice(0, MAX_HITS_PER_TASK).join(", ");
|
|
161
|
+
// OBS-547: the cap hides names, never predictions. Say so, and say where the hidden ones surface —
|
|
162
|
+
// a count with no route is exactly what left one run's victim unreadable.
|
|
129
163
|
const tail = hits.length > MAX_HITS_PER_TASK
|
|
130
|
-
? ` (${hits.length} total;
|
|
164
|
+
? ` (${hits.length} total; ${MAX_HITS_PER_TASK} shown, ${hits.length - MAX_HITS_PER_TASK} capped out of view`
|
|
165
|
+
+ ` but RETAINED for the scope gate — a matching scope red prints the hidden path with its files[] repair)`
|
|
131
166
|
: "";
|
|
132
|
-
lines.push(`${
|
|
167
|
+
lines.push(`${id}: likely collateral tests not in files[]: ${listed}${tail}`);
|
|
133
168
|
}
|
|
134
169
|
return lines;
|
|
135
170
|
}
|
package/dist/compile/native.js
CHANGED
|
@@ -728,6 +728,16 @@ acceptance is required on every task (a nested list of observable outcomes).
|
|
|
728
728
|
- Enumerating one axis exhaustively is what hides the others. A spec that guards PARTIAL coverage
|
|
729
729
|
site-by-site, member-by-member, can be defeated wholesale by CONDITIONAL coverage, which leaves
|
|
730
730
|
every enumeration satisfied. After you enumerate, ask what a single flag would do to the whole set.
|
|
731
|
+
- EVERY CRITERION NAMES THE PAIR IT DISCRIMINATES: the correct case that MUST PASS, and the
|
|
732
|
+
neighbouring plausible-wrong or false-clean case that MUST FAIL. Two easy examples that both pass are
|
|
733
|
+
not discrimination — they are two ways of being green, and the wrong half is the whole point: it is
|
|
734
|
+
what a reader would mistake for the mechanism, its VOCABULARY without its behaviour. Measured three
|
|
735
|
+
times in three days, each a criterion that pinned vocabulary instead of the discriminating case: a
|
|
736
|
+
suite that pinned an alarm's text while the alarm itself went unpinned; an export asserted SOMEWHERE,
|
|
737
|
+
which says nothing about the environment a process actually receives; and a gate green on "infra is
|
|
738
|
+
what execution says" while a signalled oracle still charged an attempt. Each was satisfied exactly as
|
|
739
|
+
written and each shipped the defect. If you cannot name the case that must FAIL, you have named a
|
|
740
|
+
topic, not a criterion — and the "so X fails" clause that ends a good criterion is where it goes.
|
|
731
741
|
- A criterion that names a behaviour must name the VALUE AT WHICH IT WOULD BREAK. Every criterion is a
|
|
732
742
|
claim about a variable — a status, a width, a count, an arrival time — and if it does not say which
|
|
733
743
|
value of that variable is the hard one, THE TEST WILL CHOOSE THE EASY ONE AND BE GREEN. Measured: a
|
package/dist/drivers/herdr.d.ts
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { type JournalEvent } from "../run/journal.js";
|
|
1
2
|
import { type ExecutorDriver, type NotifyOpts, type Slot, type SlotOpts } from "./types.js";
|
|
2
3
|
export declare const TRAILER_SAFE_FLOOR_COLS = 108;
|
|
3
4
|
export declare const TRAILER_WIDTH_MARGIN = 2;
|
|
@@ -22,20 +23,20 @@ export declare class DeliveryCorruptedError extends Error {
|
|
|
22
23
|
export type DriverJournal = (event: string, slotName: string, data: Record<string, unknown>) => void;
|
|
23
24
|
/** First-generation join direction from measured trailer-safe floor (43-MEASUREMENT.md). */
|
|
24
25
|
export declare function workerSplitDirection(paneCols: number | null, safeFloor?: number, margin?: number): "right" | "down";
|
|
25
|
-
export declare const
|
|
26
|
-
export
|
|
27
|
-
|
|
28
|
-
direction: "
|
|
29
|
-
/** herdr's split ratio is the FIRST child's share and a
|
|
30
|
-
*
|
|
31
|
-
ratio
|
|
32
|
-
/**
|
|
33
|
-
|
|
26
|
+
export declare const BOARD_HEIGHT_SHARE = 0.72;
|
|
27
|
+
export interface BoardPlacement {
|
|
28
|
+
/** Always down: the split is vertical, so the board can own the caller's FULL width. */
|
|
29
|
+
direction: "down";
|
|
30
|
+
/** herdr's split ratio is the FIRST child's share, and a down split's first child is the TOP
|
|
31
|
+
* region — the region the board occupies once it is swapped above the caller. */
|
|
32
|
+
ratio: number;
|
|
33
|
+
/** The new pane is swapped ABOVE the caller; the split alone would leave the board underneath. */
|
|
34
|
+
swap: "above";
|
|
34
35
|
}
|
|
35
|
-
/**
|
|
36
|
-
*
|
|
37
|
-
*
|
|
38
|
-
export declare function boardSplitPlan(
|
|
36
|
+
/** The single approved vertical-stack record. The caller's columns are accepted and deliberately
|
|
37
|
+
* ignored: this signature is where width used to decide the arrangement, and the parameter stays
|
|
38
|
+
* so that "the plan does not depend on it" is a property a caller (and a test) can exercise. */
|
|
39
|
+
export declare function boardSplitPlan(_callerCols?: number | null): BoardPlacement;
|
|
39
40
|
export declare function taskGroupOf(name: string): string | undefined;
|
|
40
41
|
/** The operator-facing TITLE for a tab this driver creates when no stage label is supplied: a task
|
|
41
42
|
* token for a worker (`T5`, `T5↻2` on a retry) and ROLE + task for a gate pane (`REVIEW T5`) — the
|
|
@@ -63,11 +64,14 @@ export declare class HerdrDriver implements ExecutorDriver {
|
|
|
63
64
|
private inputBoxes;
|
|
64
65
|
private bootstraps;
|
|
65
66
|
private journalRoots;
|
|
67
|
+
private narrate?;
|
|
66
68
|
private ws;
|
|
67
69
|
private callerPane;
|
|
68
70
|
private watches;
|
|
69
71
|
constructor(bin?: string, workersPerTab?: number, time?: HerdrTimeSource, journal?: DriverJournal | undefined);
|
|
70
72
|
private appendDispatchRetry;
|
|
73
|
+
/** v1.99 T2: bind this driver's own journal writes to the run's live narration sink. */
|
|
74
|
+
narrateWith(narrate: (event: JournalEvent) => void): void;
|
|
71
75
|
private serial;
|
|
72
76
|
private deliveryQueue;
|
|
73
77
|
private reserveDispatch;
|
|
@@ -77,6 +81,7 @@ export declare class HerdrDriver implements ExecutorDriver {
|
|
|
77
81
|
static available(): boolean;
|
|
78
82
|
private herdr;
|
|
79
83
|
private namedPaneId;
|
|
84
|
+
private paneStillOpen;
|
|
80
85
|
private paneId;
|
|
81
86
|
private paneWidth;
|
|
82
87
|
slot(cwd: string, name: string, opts?: SlotOpts): Promise<Slot>;
|
|
@@ -116,6 +121,7 @@ export declare class HerdrDriver implements ExecutorDriver {
|
|
|
116
121
|
private closeGrouped;
|
|
117
122
|
private ownedWatchPanes;
|
|
118
123
|
private watchSlot;
|
|
124
|
+
private discardSplit;
|
|
119
125
|
narrator(cwd: string, command: string, runId?: string): Promise<Slot>;
|
|
120
126
|
reconcile(desired: Set<string>, runId: string, opts?: {
|
|
121
127
|
spareLiveLlm?: boolean;
|
package/dist/drivers/herdr.js
CHANGED
|
@@ -69,19 +69,19 @@ export function workerSplitDirection(paneCols, safeFloor = TRAILER_SAFE_FLOOR_CO
|
|
|
69
69
|
}
|
|
70
70
|
// The watch board's geometry, deliberately NOT the trailer floor above. `workerSplitDirection`
|
|
71
71
|
// halves the caller and refuses a right split under 108+2 — that bound protects WORKER panes, which
|
|
72
|
-
// print a trailer; the supervising seat + board pair does not
|
|
73
|
-
//
|
|
74
|
-
//
|
|
75
|
-
//
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
return { direction: "
|
|
72
|
+
// print a trailer; the supervising seat + board pair does not, and neither does the board's own
|
|
73
|
+
// placement any more. Every width-derived variant of this placement has been wrong in the operator's
|
|
74
|
+
// tab: the halving floor sent a 189-column board below the seat (2026-08-18), and the width-first
|
|
75
|
+
// side split that replaced it puts the board and the narration shoulder to shoulder when the board is
|
|
76
|
+
// the surface the operator reads and the narration is the rail beneath it. The placement is now ONE
|
|
77
|
+
// record — the board stacked ABOVE the caller at full width, taking 72% of the height — and it is
|
|
78
|
+
// invariant: no terminal width, measured or unmeasurable, can select a different arrangement.
|
|
79
|
+
export const BOARD_HEIGHT_SHARE = 0.72; // board 72 / narration 28, the operator's stack
|
|
80
|
+
/** The single approved vertical-stack record. The caller's columns are accepted and deliberately
|
|
81
|
+
* ignored: this signature is where width used to decide the arrangement, and the parameter stays
|
|
82
|
+
* so that "the plan does not depend on it" is a property a caller (and a test) can exercise. */
|
|
83
|
+
export function boardSplitPlan(_callerCols) {
|
|
84
|
+
return { direction: "down", ratio: BOARD_HEIGHT_SHARE, swap: "above" };
|
|
85
85
|
}
|
|
86
86
|
/** The tab a slot belongs to: its TASK — worker, judge, review and consult panes for one task share it.
|
|
87
87
|
* Returns undefined for everything else, which keeps those on the dedicated-tab path.
|
|
@@ -147,6 +147,7 @@ export class HerdrDriver {
|
|
|
147
147
|
// the caller launched tickmarkr elsewhere (process.cwd is not run identity). The repo itself is
|
|
148
148
|
// also bound for judge/review/consult slots whose cwd is the root rather than a task worktree.
|
|
149
149
|
journalRoots = new Map();
|
|
150
|
+
narrate;
|
|
150
151
|
// VIS-10: the run's workspace id, captured once at construction (the daemon inherits it from the
|
|
151
152
|
// operator's env before the driver is built). Required at slot() time, never in the constructor —
|
|
152
153
|
// pickDriver and its unit test construct HerdrDriver without env, so slot() is the trust gate.
|
|
@@ -176,7 +177,14 @@ export class HerdrDriver {
|
|
|
176
177
|
if (!repoRoot) {
|
|
177
178
|
throw new Error(`cannot journal dispatch-retry: slot ${slot.name} has no daemon repo binding for ${slot.cwd}`);
|
|
178
179
|
}
|
|
179
|
-
|
|
180
|
+
// Bound to the live narration sink (`narrateWith`): this Journal is the driver's own — the
|
|
181
|
+
// daemon never appends this event and never sees it — so an unbound handle here persists the
|
|
182
|
+
// recovery to the file and the pipe while the operator's rail stays silent about it.
|
|
183
|
+
Journal.open(repoRoot, owned.runId, this.narrate).append("dispatch-retry", owned.taskId, data);
|
|
184
|
+
}
|
|
185
|
+
/** v1.99 T2: bind this driver's own journal writes to the run's live narration sink. */
|
|
186
|
+
narrateWith(narrate) {
|
|
187
|
+
this.narrate = narrate;
|
|
180
188
|
}
|
|
181
189
|
serial(fn) {
|
|
182
190
|
const p = this.groupSerial.then(fn, fn);
|
|
@@ -274,6 +282,20 @@ export class HerdrDriver {
|
|
|
274
282
|
return null;
|
|
275
283
|
}
|
|
276
284
|
}
|
|
285
|
+
// Is this pane id still in the listing? FAIL CLOSED: a listing we cannot read cannot prove a pane
|
|
286
|
+
// gone, and the caller uses this to decide whether a pane it tried to close is really off screen.
|
|
287
|
+
async paneStillOpen(paneId) {
|
|
288
|
+
const r = await this.herdr(`pane list`);
|
|
289
|
+
if (r.code !== 0)
|
|
290
|
+
return true;
|
|
291
|
+
try {
|
|
292
|
+
const panes = JSON.parse(r.stdout).result?.panes;
|
|
293
|
+
return !Array.isArray(panes) || panes.some((p) => p.pane_id === paneId);
|
|
294
|
+
}
|
|
295
|
+
catch {
|
|
296
|
+
return true;
|
|
297
|
+
}
|
|
298
|
+
}
|
|
277
299
|
// Before delivery, resolve fresh via the durable name because pane ids can compact. After delivery,
|
|
278
300
|
// pin the verified target so early liveness cannot drift to a label rebound onto another pane.
|
|
279
301
|
async paneId(slot) {
|
|
@@ -1059,19 +1081,18 @@ export class HerdrDriver {
|
|
|
1059
1081
|
return p.workspace_id === this.ws && typeof p.pane_id === "string" && owned?.role === "watch" && owned.taskId === "run";
|
|
1060
1082
|
}).map((p) => p.pane_id);
|
|
1061
1083
|
}
|
|
1062
|
-
// T2: the watch is a sibling of the daemon's own pane, never a separate tab —
|
|
1063
|
-
//
|
|
1064
|
-
//
|
|
1084
|
+
// T2: the watch is a sibling of the daemon's own pane, never a separate tab — stacked ABOVE it at
|
|
1085
|
+
// the caller's full width, always, whatever the terminal measures. Its durable owned name is how a
|
|
1086
|
+
// later daemon RECOGNIZES the board it must retire, so a run never stacks a second one.
|
|
1065
1087
|
async watchSlot(cwd, name) {
|
|
1066
1088
|
if (!this.ws)
|
|
1067
1089
|
throw new Error("herdr watch placement requires HERDR_WORKSPACE_ID — refusing unseeded pane");
|
|
1068
1090
|
if (!this.callerPane)
|
|
1069
1091
|
throw new Error("herdr watch placement requires HERDR_PANE_ID — refusing untargeted split");
|
|
1070
|
-
//
|
|
1071
|
-
// read —
|
|
1072
|
-
const plan = boardSplitPlan(
|
|
1073
|
-
const
|
|
1074
|
-
const sp = await this.herdr(`pane split ${shq(this.callerPane)} --direction ${plan.direction}${ratio} --no-focus`);
|
|
1092
|
+
// One invariant placement (boardSplitPlan): split the caller down, then swap the new pane above
|
|
1093
|
+
// it. No layout read decides this — width chose the arrangement twice and was wrong twice.
|
|
1094
|
+
const plan = boardSplitPlan();
|
|
1095
|
+
const sp = await this.herdr(`pane split ${shq(this.callerPane)} --direction ${plan.direction} --ratio ${plan.ratio} --no-focus`);
|
|
1075
1096
|
if (sp.code !== 0)
|
|
1076
1097
|
throw new Error(`herdr watch split failed: ${sp.stderr || sp.stdout}`);
|
|
1077
1098
|
let pane;
|
|
@@ -1083,18 +1104,59 @@ export class HerdrDriver {
|
|
|
1083
1104
|
}
|
|
1084
1105
|
if (typeof pane !== "string" || !pane)
|
|
1085
1106
|
throw new Error(`herdr watch split returned no pane id: ${sp.stdout}`);
|
|
1107
|
+
// The split leaves the board UNDER the caller; the swap is what makes the stack the requested
|
|
1108
|
+
// one. Verified, not assumed: a swap that failed would leave a board below the narration while
|
|
1109
|
+
// the daemon reported the geometry it asked for. Instead the split pane is closed and the failure
|
|
1110
|
+
// propagates — the daemon swallows it and runs boardless, which is honest about what is on screen.
|
|
1111
|
+
// `pane swap` answers a no-op with a ZERO exit and `changed:false` (herdr socket API: a swap it
|
|
1112
|
+
// declined is a non-error response), so an exit code alone proves nothing about the geometry —
|
|
1113
|
+
// that is exactly the path that would leave the board below the narration while the daemon
|
|
1114
|
+
// reported the stack. The documented `changed` flag is the verification; anything else — a
|
|
1115
|
+
// nonzero exit, `changed:false`, an unparseable result — fails closed.
|
|
1116
|
+
const swapped = await this.herdr(`pane swap --source-pane ${shq(pane)} --target-pane ${shq(this.callerPane)}`);
|
|
1117
|
+
let swapChanged;
|
|
1118
|
+
try {
|
|
1119
|
+
swapChanged = JSON.parse(swapped.stdout).result?.changed;
|
|
1120
|
+
}
|
|
1121
|
+
catch {
|
|
1122
|
+
/* fail closed below */
|
|
1123
|
+
}
|
|
1124
|
+
if (swapped.code !== 0 || swapChanged !== true) {
|
|
1125
|
+
await this.discardSplit(pane, `herdr watch swap ${plan.swap} failed: ${swapped.code !== 0
|
|
1126
|
+
? swapped.stderr || swapped.stdout
|
|
1127
|
+
: `herdr reported no swap took place: ${swapped.stdout || swapped.stderr}`}`);
|
|
1128
|
+
}
|
|
1086
1129
|
const renamed = await this.herdr(`pane rename ${shq(pane)} ${shq(name)}`);
|
|
1087
1130
|
if (renamed.code !== 0 || await this.namedPaneId(name) !== pane) {
|
|
1088
|
-
await this.
|
|
1089
|
-
throw new Error(`herdr watch rename failed: ${renamed.stderr || renamed.stdout}`);
|
|
1131
|
+
await this.discardSplit(pane, `herdr watch rename failed: ${renamed.stderr || renamed.stdout}`);
|
|
1090
1132
|
}
|
|
1091
1133
|
const seed = await this.herdr(`pane run ${shq(pane)} ${shq(`cd ${shq(cwd)}; export HERDR_WORKSPACE_ID=${shq(this.ws)}; ${herdrSealShellPrefix()}`)}`, cwd);
|
|
1092
1134
|
if (seed.code !== 0) {
|
|
1093
|
-
await this.
|
|
1094
|
-
throw new Error(`herdr watch seed failed: ${seed.stderr || seed.stdout}`);
|
|
1135
|
+
await this.discardSplit(pane, `herdr watch seed failed: ${seed.stderr || seed.stdout}`);
|
|
1095
1136
|
}
|
|
1096
1137
|
return { id: pane, name, cwd };
|
|
1097
1138
|
}
|
|
1139
|
+
// A board that could not be placed costs the OPERATOR a stray pane unless the split is really
|
|
1140
|
+
// taken back, so every close is followed by a pane-list verification rather than issued and
|
|
1141
|
+
// forgotten. A nonzero close and a success that frees nothing are both diagnosed from that same
|
|
1142
|
+
// observation. The daemon swallows it either way and runs boardless — but never silently keeps a
|
|
1143
|
+
// split the geometry it asked for does not include.
|
|
1144
|
+
async discardSplit(pane, why) {
|
|
1145
|
+
const closed = await this.herdr(`pane close ${shq(pane)}`);
|
|
1146
|
+
const stillOpen = await this.paneStillOpen(pane);
|
|
1147
|
+
const orphan = stillOpen
|
|
1148
|
+
? closed.code !== 0
|
|
1149
|
+
? closed.stderr || closed.stdout || `exit ${closed.code}`
|
|
1150
|
+
: "close reported success but the pane could not be proven absent from pane list"
|
|
1151
|
+
: null;
|
|
1152
|
+
if (orphan !== null) {
|
|
1153
|
+
// The daemon swallows narrator failures whole (visibility is never a gate), so the thrown
|
|
1154
|
+
// error dies in its catch. This line is the operator's only notice that a pane they did not
|
|
1155
|
+
// ask for is still on their screen and that no process will take it back.
|
|
1156
|
+
console.error(`tickmarkr: the watch split ${pane} survived its close (${orphan}) — close it by hand; the run continues boardless`);
|
|
1157
|
+
}
|
|
1158
|
+
throw new Error(orphan === null ? why : `${why} — and the split pane ${pane} survived its close (${orphan})`);
|
|
1159
|
+
}
|
|
1098
1160
|
// T6 narrator: the run's single live status surface, RUNNING THE COMMAND THIS CALL SUPPLIED. Only
|
|
1099
1161
|
// a board this driver instance itself opened is reused (this.watches); any other surviving board —
|
|
1100
1162
|
// a prior run's, or one already carrying this run's canonical name after a resume — is retired and
|
package/dist/drivers/types.d.ts
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import type { JournalEvent } from "../run/journal.js";
|
|
1
2
|
export interface Slot {
|
|
2
3
|
id: string;
|
|
3
4
|
name: string;
|
|
@@ -75,6 +76,7 @@ export interface ExecutorDriver {
|
|
|
75
76
|
nudge?(slot: Slot, message: string): Promise<boolean>;
|
|
76
77
|
notify(msg: string, opts?: NotifyOpts): Promise<void>;
|
|
77
78
|
close(slot: Slot): Promise<void>;
|
|
79
|
+
narrateWith?(narrate: (event: JournalEvent) => void): void;
|
|
78
80
|
worktree(repo: string, branch: string, baseRef: string): Promise<string>;
|
|
79
81
|
narrator?: (cwd: string, command: string, runId?: string) => Promise<Slot>;
|
|
80
82
|
reconcile?: (desired: Set<string>, runId: string, opts?: {
|
package/dist/gates/acceptance.js
CHANGED
|
@@ -179,20 +179,30 @@ function tail(out, n = 8) {
|
|
|
179
179
|
return "\n" + t.split("\n").slice(-n).join("\n");
|
|
180
180
|
}
|
|
181
181
|
/**
|
|
182
|
-
* OBS-540: classify only
|
|
183
|
-
*
|
|
184
|
-
*
|
|
185
|
-
*
|
|
182
|
+
* OBS-540/551: classify only execution evidence produced by the deterministic oracle process. That
|
|
183
|
+
* includes its output, exit status and runner-reported test count — never judge-authored prose. A
|
|
184
|
+
* nonzero command with an infra-only output shape, one of the enumerated signal exits, or a zero-run
|
|
185
|
+
* report without a failure identity returned no product verdict, so it remains fail-closed but cannot
|
|
186
|
+
* be billed as a diff failure. A parsed judge refusal is an evaluated quality verdict whatever its
|
|
187
|
+
* prose says. As in baseline classification, a named regression remains a verdict even if teardown
|
|
188
|
+
* later exits through a signal.
|
|
186
189
|
*/
|
|
187
190
|
function oracleExecutionFailure(label, code, stdout, stderr) {
|
|
188
191
|
const output = [stderr, stdout].filter((part) => part.length > 0).join("\n");
|
|
189
192
|
const classification = classifyFailureOutput(output);
|
|
190
|
-
|
|
193
|
+
const signalName = code === 143 ? "SIGTERM" : code === 137 ? "SIGKILL" : undefined;
|
|
194
|
+
const signalKilled = signalName !== undefined && classification !== "regression";
|
|
195
|
+
const signalEvidence = signalKilled
|
|
196
|
+
? `signal-shaped exit ${code} (${signalName}); `
|
|
197
|
+
: "";
|
|
198
|
+
const zeroRun = testsRan(output) === 0 && classification !== "regression";
|
|
199
|
+
const zeroRunEvidence = zeroRun ? "the runner reported zero tests run and no failure identity; " : "";
|
|
200
|
+
if (classification === "infra" || signalKilled || zeroRun) {
|
|
191
201
|
return {
|
|
192
202
|
gate: "acceptance",
|
|
193
203
|
pass: false,
|
|
194
|
-
details: `oracle failed: ${label} (exit ${code}) — infrastructure blocked execution before a verdict; this oracle verified nothing${tail(output)}`,
|
|
195
|
-
meta: { cause: "oracle-execution", classification, infra: true, retryable: false },
|
|
204
|
+
details: `oracle failed: ${label} (exit ${code}) — ${signalEvidence || zeroRunEvidence}infrastructure blocked execution before a verdict; this oracle verified nothing${tail(output)}`,
|
|
205
|
+
meta: { cause: "oracle-execution", classification: "infra", infra: true, retryable: false },
|
|
196
206
|
};
|
|
197
207
|
}
|
|
198
208
|
return {
|
package/dist/gates/llm.d.ts
CHANGED
|
@@ -33,6 +33,25 @@ export interface GateVia {
|
|
|
33
33
|
nameFor: (role: "judge" | "review", adapter: string) => string;
|
|
34
34
|
labelFor: (role: "judge" | "review") => string;
|
|
35
35
|
}
|
|
36
|
+
export declare const GATE_INACTIVITY_WINDOW_MS: number;
|
|
37
|
+
/** Test seam — shrink only the calibrated inactivity window; production always reads 12 minutes. */
|
|
38
|
+
export declare function setGateInactivityWindowMsForTests(ms: number): void;
|
|
39
|
+
export declare function resetGateInactivityWindowMsForTests(): void;
|
|
40
|
+
export interface GateCpuAccountant {
|
|
41
|
+
start(): Promise<void>;
|
|
42
|
+
read(): {
|
|
43
|
+
cpu: {
|
|
44
|
+
ms: number;
|
|
45
|
+
resolutionMs: number;
|
|
46
|
+
} | undefined;
|
|
47
|
+
gaps: number;
|
|
48
|
+
};
|
|
49
|
+
stop(): Promise<void>;
|
|
50
|
+
}
|
|
51
|
+
export type GateCpuAccountantFactory = (marker: string, cwd: string) => GateCpuAccountant;
|
|
52
|
+
/** Test seam for deterministic measurable/activity/gap samples without platform-specific ps output. */
|
|
53
|
+
export declare function setGateCpuAccountantFactoryForTests(factory: GateCpuAccountantFactory): void;
|
|
54
|
+
export declare function resetGateCpuAccountantFactoryForTests(): void;
|
|
36
55
|
export declare function captureLlmOutput<T>(run: () => Promise<T>): Promise<{
|
|
37
56
|
value: T;
|
|
38
57
|
outputs: string[];
|
package/dist/gates/llm.js
CHANGED
|
@@ -6,6 +6,7 @@ import { join } from "node:path";
|
|
|
6
6
|
import { formatOwnedName, parseOwnedName } from "../drivers/types.js";
|
|
7
7
|
import { bannerShell, paneDispatchCommand } from "../brand.js";
|
|
8
8
|
import { sh } from "../run/git.js";
|
|
9
|
+
import { harvestCpuFlatWindowMs, normalizeStallSnapshot, WorkerTreeCpuAccountant, } from "../run/stall.js";
|
|
9
10
|
export const GATE_PANE_SEP = " · ";
|
|
10
11
|
// v1.64 gate-integrity (repo-scan Tier A·1): the concrete completion-faking shortcuts every
|
|
11
12
|
// judge/review verdict must hunt for. Shared verbatim by the acceptance judge and review prompts.
|
|
@@ -94,6 +95,30 @@ export function rolePaneNameFromPrompt(prompt, fallback) {
|
|
|
94
95
|
return fallback;
|
|
95
96
|
}
|
|
96
97
|
const llmOutputCapture = new AsyncLocalStorage();
|
|
98
|
+
// v2.0 T1 (OBS-555): the empirical healthy-duration p95 is 10.6 minutes. The smallest whole
|
|
99
|
+
// minute above it plus the specified one-minute margin is twelve minutes. This default remains
|
|
100
|
+
// strictly below BOTH unchanged 900_000ms production dispatch timeouts: JUDGE_TIMEOUT_MS in
|
|
101
|
+
// acceptance.ts and reviewGate's literal timeout in review.ts. Scope's 300_000ms call is not a
|
|
102
|
+
// verdict gate and retains its existing one-wait behavior.
|
|
103
|
+
export const GATE_INACTIVITY_WINDOW_MS = 12 * 60_000;
|
|
104
|
+
const GATE_WAIT_SLICE_MS = 30_000;
|
|
105
|
+
let gateInactivityWindowMs = GATE_INACTIVITY_WINDOW_MS;
|
|
106
|
+
/** Test seam — shrink only the calibrated inactivity window; production always reads 12 minutes. */
|
|
107
|
+
export function setGateInactivityWindowMsForTests(ms) {
|
|
108
|
+
gateInactivityWindowMs = ms;
|
|
109
|
+
}
|
|
110
|
+
export function resetGateInactivityWindowMsForTests() {
|
|
111
|
+
gateInactivityWindowMs = GATE_INACTIVITY_WINDOW_MS;
|
|
112
|
+
}
|
|
113
|
+
const productionGateCpuAccountant = (marker, cwd) => new WorkerTreeCpuAccountant(marker, cwd);
|
|
114
|
+
let gateCpuAccountantFactory = productionGateCpuAccountant;
|
|
115
|
+
/** Test seam for deterministic measurable/activity/gap samples without platform-specific ps output. */
|
|
116
|
+
export function setGateCpuAccountantFactoryForTests(factory) {
|
|
117
|
+
gateCpuAccountantFactory = factory;
|
|
118
|
+
}
|
|
119
|
+
export function resetGateCpuAccountantFactoryForTests() {
|
|
120
|
+
gateCpuAccountantFactory = productionGateCpuAccountant;
|
|
121
|
+
}
|
|
97
122
|
// OBS-132: acceptance.ts owns verdict parsing and is deliberately byte-untouched. This async-scoped
|
|
98
123
|
// recorder lets run-gates observe the exact output that acceptance parsed without changing runLlm's
|
|
99
124
|
// return value or leaking concurrent tasks into one another. Callers retain output only when the
|
|
@@ -119,6 +144,8 @@ export async function runHeadless(adapter, model, prompt, cwd, timeoutMs = 30000
|
|
|
119
144
|
// as a visible named agent (herdr pane), with the quote-split completion wrapper.
|
|
120
145
|
export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
|
|
121
146
|
const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
|
|
147
|
+
let slot;
|
|
148
|
+
let accountant;
|
|
122
149
|
try {
|
|
123
150
|
const pf = join(dir, "prompt.md");
|
|
124
151
|
writeFileSync(pf, prompt);
|
|
@@ -131,19 +158,90 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
|
|
|
131
158
|
adapter.headlessCommand(pf, model),
|
|
132
159
|
gateExitTrailer(nonce),
|
|
133
160
|
].join("\n"));
|
|
134
|
-
|
|
161
|
+
slot = await via.driver.slot(cwd, rolePaneNameFromPrompt(prompt, via.name), via.label ? { label: via.label } : undefined);
|
|
135
162
|
via.onSlot?.(slot);
|
|
136
163
|
await via.driver.run(slot, paneDispatchCommand(scriptPath));
|
|
137
164
|
// nonce-suffixed exit only: a displayed bare "TICKMARKR_EXIT:" or another call's marker must not
|
|
138
165
|
// false-complete — same guard the worker path uses (daemon.ts:330-331).
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
166
|
+
const exitPattern = `TICKMARKR_EXIT_${nonce}:\\d`;
|
|
167
|
+
let out;
|
|
168
|
+
const gatePrompt = prompt.startsWith("TICKMARKR-JUDGE") || prompt.startsWith("TICKMARKR-REVIEW");
|
|
169
|
+
if (!gatePrompt) {
|
|
170
|
+
await via.driver.waitOutput(slot, exitPattern, timeoutMs, { regex: true });
|
|
171
|
+
out = await via.driver.read(slot, 400);
|
|
172
|
+
}
|
|
173
|
+
else {
|
|
174
|
+
// Review and judge waits are sliced so the two-leg inactivity policy can observe the pane and
|
|
175
|
+
// the exact dispatch script's process tree between waits. The accountant retains short-lived
|
|
176
|
+
// descendants with the same semantics as daemon workers; llm.ts depends only on stall.ts.
|
|
177
|
+
accountant = gateCpuAccountantFactory(scriptPath, cwd);
|
|
178
|
+
await accountant.start();
|
|
179
|
+
const startedAt = Date.now();
|
|
180
|
+
out = await via.driver.read(slot, 400);
|
|
181
|
+
let priorSnapshot = normalizeStallSnapshot(out);
|
|
182
|
+
const anchoredAt = Date.now();
|
|
183
|
+
let quietSince = anchoredAt;
|
|
184
|
+
const initialCpu = accountant.read();
|
|
185
|
+
let priorCpuMs = initialCpu.cpu?.ms;
|
|
186
|
+
let priorGaps = initialCpu.gaps;
|
|
187
|
+
let cpuFlatSince = initialCpu.cpu === undefined ? undefined : anchoredAt;
|
|
188
|
+
while (Date.now() - startedAt < timeoutMs) {
|
|
189
|
+
const remaining = timeoutMs - (Date.now() - startedAt);
|
|
190
|
+
// The adaptive test-window arm keeps a seam-adjusted case sliced too; production stays 30s.
|
|
191
|
+
const sliceMs = Math.max(1, Math.min(GATE_WAIT_SLICE_MS, Math.ceil(gateInactivityWindowMs / 4), remaining));
|
|
192
|
+
const matched = await via.driver.waitOutput(slot, exitPattern, sliceMs, { regex: true });
|
|
193
|
+
const raw = await via.driver.read(slot, 400);
|
|
194
|
+
out = raw;
|
|
195
|
+
// waitOutput is the driver's authoritative marker match. The raw check covers drivers whose
|
|
196
|
+
// wait timed out at the same boundary the marker landed; either way a trailer completes
|
|
197
|
+
// normally and is never mistaken for inactivity.
|
|
198
|
+
if (matched || new RegExp(exitPattern).test(raw))
|
|
199
|
+
break;
|
|
200
|
+
const now = Date.now();
|
|
201
|
+
const snapshot = normalizeStallSnapshot(raw);
|
|
202
|
+
if (snapshot !== priorSnapshot) {
|
|
203
|
+
priorSnapshot = snapshot;
|
|
204
|
+
quietSince = now;
|
|
205
|
+
}
|
|
206
|
+
const observation = accountant.read();
|
|
207
|
+
const cpu = observation.cpu;
|
|
208
|
+
if (observation.gaps !== priorGaps || cpu === undefined) {
|
|
209
|
+
// Missing evidence is a hold-open signal, never guessed inactivity. A later measurable
|
|
210
|
+
// sample starts a fresh complete window rather than inheriting quiet time across the gap.
|
|
211
|
+
priorGaps = observation.gaps;
|
|
212
|
+
priorCpuMs = undefined;
|
|
213
|
+
cpuFlatSince = undefined;
|
|
214
|
+
quietSince = now;
|
|
215
|
+
continue;
|
|
216
|
+
}
|
|
217
|
+
if (priorCpuMs === undefined || cpu.ms !== priorCpuMs) {
|
|
218
|
+
// Any process-tree CPU movement holds the call open and re-arms both clocks. Equality only
|
|
219
|
+
// becomes "flat" after the existing resolution-aware quantum window has elapsed.
|
|
220
|
+
priorCpuMs = cpu.ms;
|
|
221
|
+
cpuFlatSince = now;
|
|
222
|
+
quietSince = now;
|
|
223
|
+
continue;
|
|
224
|
+
}
|
|
225
|
+
const cpuFlatFor = now - (cpuFlatSince ?? now);
|
|
226
|
+
const snapshotQuietFor = now - quietSince;
|
|
227
|
+
if (cpuFlatFor >= harvestCpuFlatWindowMs(cpu.resolutionMs)
|
|
228
|
+
&& snapshotQuietFor >= gateInactivityWindowMs) {
|
|
229
|
+
break;
|
|
230
|
+
}
|
|
231
|
+
}
|
|
232
|
+
}
|
|
143
233
|
return dewrapPaneVerdict(out, nonce);
|
|
144
234
|
}
|
|
145
235
|
finally {
|
|
146
|
-
|
|
236
|
+
try {
|
|
237
|
+
await accountant?.stop();
|
|
238
|
+
if (slot && !via.keep)
|
|
239
|
+
await via.driver.close(slot);
|
|
240
|
+
}
|
|
241
|
+
finally {
|
|
242
|
+
// Unconditional and synchronous: a stop/close failure must not leak this call's prompt and script.
|
|
243
|
+
rmSync(dir, { recursive: true, force: true });
|
|
244
|
+
}
|
|
147
245
|
}
|
|
148
246
|
}
|
|
149
247
|
// OBS-155: a TUI renders the verdict as a bullet and HARD-wraps it at pane width with a 2-space
|
|
@@ -4,6 +4,23 @@ import { type GateName, type Task } from "../graph/schema.js";
|
|
|
4
4
|
import { type Baseline } from "./baseline.js";
|
|
5
5
|
import { type GateVia } from "./llm.js";
|
|
6
6
|
import type { GateResult } from "./types.js";
|
|
7
|
+
export type LoadProvider = () => number;
|
|
8
|
+
/** Test seam — inject deterministic load samples; production always reads os.loadavg. */
|
|
9
|
+
export declare function setLoadProviderForTests(provider: LoadProvider): void;
|
|
10
|
+
export declare function resetLoadProviderForTests(): void;
|
|
11
|
+
/**
|
|
12
|
+
* One gate's own measurement, taken WHERE THE GATE RUNS. `durationMs` sums that gate's execution
|
|
13
|
+
* intervals and nothing between them, so the composite `test` gate (a selected screen, then other
|
|
14
|
+
* gates, then the full suite) reports the two suites' cost rather than the span containing them —
|
|
15
|
+
* and no consumer has to re-derive a duration by subtracting journal timestamps, which measures the
|
|
16
|
+
* queue as well as the work. The load samples bracket the FIRST interval's start and the LAST
|
|
17
|
+
* interval's end: start is what a scheduler would have decided on, end is the state it left behind.
|
|
18
|
+
*/
|
|
19
|
+
export interface GateTelemetry {
|
|
20
|
+
durationMs: number;
|
|
21
|
+
load1Start: number;
|
|
22
|
+
load1End: number;
|
|
23
|
+
}
|
|
7
24
|
export type GateEvent = {
|
|
8
25
|
phase: "start";
|
|
9
26
|
gate: GateName;
|
|
@@ -31,6 +48,7 @@ export interface GateContext {
|
|
|
31
48
|
artifactDir?: string;
|
|
32
49
|
pipeline?: "v185" | "legacy";
|
|
33
50
|
selectTests?: boolean;
|
|
51
|
+
collateral?: ReadonlyArray<string>;
|
|
34
52
|
onGate?: (e: GateEvent) => void | Promise<void>;
|
|
35
53
|
}
|
|
36
54
|
/**
|