tickmarkr 2.5.2 → 2.5.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/qwen.d.ts +1 -0
- package/dist/adapters/qwen.js +8 -0
- package/dist/adapters/types.d.ts +1 -0
- package/dist/cli/commands/doctor.d.ts +6 -0
- package/dist/cli/commands/doctor.js +50 -1
- package/dist/cli/commands/run.js +2 -0
- package/dist/cli/commands/status.js +8 -14
- package/dist/compile/ownership.d.ts +9 -0
- package/dist/compile/ownership.js +65 -2
- package/dist/gates/baseline.d.ts +5 -1
- package/dist/gates/baseline.js +18 -5
- package/dist/gates/llm.d.ts +1 -1
- package/dist/gates/llm.js +34 -12
- package/dist/gates/review.d.ts +11 -0
- package/dist/gates/review.js +50 -8
- package/dist/gates/run-gates.js +2 -2
- package/dist/run/daemon.d.ts +3 -1
- package/dist/run/daemon.js +99 -32
- package/dist/run/journal.js +16 -7
- package/dist/run/operator-state.d.ts +1 -1
- package/dist/run/operator-state.js +5 -9
- package/package.json +1 -1
- package/skills/tickmarkr-auto/SKILL.md +1 -1
- package/skills/tickmarkr-loop/SKILL.md +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +61 -4
- package/skills/tickmarkr-overseer/scripts/watch-launch.sh +38 -0
package/dist/adapters/qwen.d.ts
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { type ClassifiedWorkerResult } from "./prompt.js";
|
|
2
2
|
import { type WorkerAdapter } from "./types.js";
|
|
3
3
|
export declare const QWEN_VERSION_IDENTITY: RegExp;
|
|
4
|
+
export declare const QWEN_HARNESS_BANNER_ROWS: readonly ["⚠ SAFE MODE — all customizations disabled (hooks, extensions, skills, MCP servers, QWEN.md). Restart without --safe-mode to resume normal operation.", "Warning: running headless with --yolo / approval-mode=yolo and no sandbox. All tool calls (shell, write, edit) auto-execute at this process's privilege level. Enable a sandbox via --sandbox / QWEN_SANDBOX, or set QWEN_CODE_SUPPRESS_YOLO_WARNING=1 to silence this notice."];
|
|
4
5
|
export declare function parseQwenResult(raw: string, nonce: string): ClassifiedWorkerResult;
|
|
5
6
|
export declare const qwen: WorkerAdapter;
|
package/dist/adapters/qwen.js
CHANGED
|
@@ -3,6 +3,13 @@ import { parseWorkerResult } from "./prompt.js";
|
|
|
3
3
|
import { channelsFromConfig, shq, } from "./types.js";
|
|
4
4
|
export const QWEN_VERSION_IDENTITY = /^\d+\.\d+\.\d+/;
|
|
5
5
|
const QWEN_SKIP_UPDATE = "QWEN_CODE_SKIP_UPDATE_CHECK_ONCE=true";
|
|
6
|
+
// Verbatim stderr from tests/fixtures/qwen/safe-mode.stderr (qwen 0.21.15). These bytes are launch
|
|
7
|
+
// harness, not evidence that a reviewer took a turn. Keep the declaration closed and row-shaped so
|
|
8
|
+
// llm.ts can sweep partial terminal paints without a broad SAFE MODE/warning regex.
|
|
9
|
+
export const QWEN_HARNESS_BANNER_ROWS = [
|
|
10
|
+
"⚠ SAFE MODE — all customizations disabled (hooks, extensions, skills, MCP servers, QWEN.md). Restart without --safe-mode to resume normal operation.",
|
|
11
|
+
"Warning: running headless with --yolo / approval-mode=yolo and no sandbox. All tool calls (shell, write, edit) auto-execute at this process's privilege level. Enable a sandbox via --sandbox / QWEN_SANDBOX, or set QWEN_CODE_SUPPRESS_YOLO_WARNING=1 to silence this notice.",
|
|
12
|
+
];
|
|
6
13
|
function decodeQwenEvents(events) {
|
|
7
14
|
const text = [];
|
|
8
15
|
let failed = false;
|
|
@@ -136,6 +143,7 @@ export const qwen = {
|
|
|
136
143
|
channels: (cfg) => channelsFromConfig("qwen", cfg),
|
|
137
144
|
hardcodedFlags: { binary: "qwen", flags: ["--safe-mode", "--approval-mode", "-m", "-o", "-p"] },
|
|
138
145
|
headlessCommand: (promptFile, model) => `${QWEN_SKIP_UPDATE} qwen --safe-mode --approval-mode yolo -m ${shq(model)} -o json -p '' < ${shq(promptFile)}`,
|
|
146
|
+
harnessBannerRows: QWEN_HARNESS_BANNER_ROWS,
|
|
139
147
|
// OBS-905: qwen has NO interactive form. The `-i "$(cat prompt)"` TUI launch put the whole prompt in
|
|
140
148
|
// argv (the OBS-889 leak-and-census shape) and produced a rendered transcript the JSON decoder above
|
|
141
149
|
// can never read — under the herdr driver every qwen task read "unparseable" and merged only by harvest.
|
package/dist/adapters/types.d.ts
CHANGED
|
@@ -171,6 +171,7 @@ export interface WorkerAdapter {
|
|
|
171
171
|
probe(): Promise<AuthHealth>;
|
|
172
172
|
channels(cfg: TickmarkrConfig): BillingChannel[];
|
|
173
173
|
headlessCommand(promptFile: string, model: string): string;
|
|
174
|
+
harnessBannerRows?: readonly string[];
|
|
174
175
|
interactiveCommand(promptFile: string, model: string): string | null;
|
|
175
176
|
interactiveSeed?: InteractiveSeed;
|
|
176
177
|
resumeCommand?(sessionId: string, promptFile: string, model: string): string;
|
|
@@ -8,6 +8,12 @@ import { type VitestListResult } from "../../gates/acceptance.js";
|
|
|
8
8
|
* concatenation and publishes no index, so the release listing is the only enumerable surface. */
|
|
9
9
|
export declare const LIVEBENCH_RELEASES_URL = "https://api.github.com/repos/LiveBench/livebench.github.io/contents/public";
|
|
10
10
|
export declare const LIVEBENCH_TABLE_MAX_AGE_DAYS = 90;
|
|
11
|
+
type ReviewDemotionSummary = {
|
|
12
|
+
reviewer: string;
|
|
13
|
+
count: number;
|
|
14
|
+
causes: Record<string, number>;
|
|
15
|
+
};
|
|
16
|
+
export declare function recentReviewDemotions(cwd: string, lastRuns?: number): ReviewDemotionSummary[];
|
|
11
17
|
export type DoctorOpts = {
|
|
12
18
|
banner?: boolean;
|
|
13
19
|
kimiTurnProbe?: (cwd: string) => Promise<KimiDoctorTurnResult>;
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { writeFileSync } from "node:fs";
|
|
2
2
|
import { spawnSync } from "node:child_process";
|
|
3
|
-
import { existsSync, readFileSync } from "node:fs";
|
|
3
|
+
import { existsSync, readFileSync, readdirSync } from "node:fs";
|
|
4
4
|
import { join } from "node:path";
|
|
5
5
|
import { detectPackageManager, turboContinueFindings } from "../../gates/baseline.js";
|
|
6
6
|
import { version } from "./version.js";
|
|
@@ -26,6 +26,46 @@ const initialFetch = globalThis.fetch;
|
|
|
26
26
|
const visual = () => process.stdout.isTTY === true && process.env.NO_COLOR === undefined;
|
|
27
27
|
const alignedStatusRow = (verdict, key, value) => ` ${statusRow(verdict, kvRow(key, value).slice(2))}`;
|
|
28
28
|
const attentionRow = (text) => ` ${statusRow("warn", text)}`;
|
|
29
|
+
// Use the routing profile's established 50-run recency horizon without importing journal machinery
|
|
30
|
+
// into doctor. Torn/malformed rows are ignored one at a time: diagnostics must survive a killed write.
|
|
31
|
+
export function recentReviewDemotions(cwd, lastRuns = 50) {
|
|
32
|
+
const runs = join(cwd, stateDirName(cwd), "runs");
|
|
33
|
+
if (!existsSync(runs))
|
|
34
|
+
return [];
|
|
35
|
+
const grouped = new Map();
|
|
36
|
+
const ids = readdirSync(runs, { withFileTypes: true })
|
|
37
|
+
.filter((entry) => entry.isDirectory() && entry.name.startsWith("run-")
|
|
38
|
+
&& existsSync(join(runs, entry.name, "journal.jsonl")))
|
|
39
|
+
.map((entry) => entry.name).sort().slice(-lastRuns);
|
|
40
|
+
for (const id of ids) {
|
|
41
|
+
const lines = readFileSync(join(runs, id, "journal.jsonl"), "utf8").split("\n");
|
|
42
|
+
for (const line of lines) {
|
|
43
|
+
if (!line.trim())
|
|
44
|
+
continue;
|
|
45
|
+
let row;
|
|
46
|
+
try {
|
|
47
|
+
row = JSON.parse(line);
|
|
48
|
+
}
|
|
49
|
+
catch {
|
|
50
|
+
continue;
|
|
51
|
+
}
|
|
52
|
+
if (!row || typeof row !== "object")
|
|
53
|
+
continue;
|
|
54
|
+
const event = row;
|
|
55
|
+
if (event.event !== "review-pool-demotion" || !event.data || typeof event.data !== "object")
|
|
56
|
+
continue;
|
|
57
|
+
const data = event.data;
|
|
58
|
+
if (typeof data.reviewer !== "string" || !data.reviewer.trim())
|
|
59
|
+
continue;
|
|
60
|
+
const cause = typeof data.cause === "string" && data.cause.trim() ? data.cause : "unknown";
|
|
61
|
+
const summary = grouped.get(data.reviewer) ?? { reviewer: data.reviewer, count: 0, causes: {} };
|
|
62
|
+
summary.count++;
|
|
63
|
+
summary.causes[cause] = (summary.causes[cause] ?? 0) + 1;
|
|
64
|
+
grouped.set(data.reviewer, summary);
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
return [...grouped.values()].sort((a, b) => a.reviewer.localeCompare(b.reviewer));
|
|
68
|
+
}
|
|
29
69
|
const ORCA_HOOK_ADAPTERS = {
|
|
30
70
|
claude: "claude-code",
|
|
31
71
|
"claude-code": "claude-code",
|
|
@@ -581,6 +621,15 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
|
|
|
581
621
|
const healthy = h.installed && (a.id !== kimi.id || h.authed);
|
|
582
622
|
return alignedStatusRow(healthy ? "pass" : "fail", a.id, state);
|
|
583
623
|
});
|
|
624
|
+
const reviewDemotions = recentReviewDemotions(cwd);
|
|
625
|
+
if (reviewDemotions.length) {
|
|
626
|
+
rows.push(legend("recent review-seat demotions:"));
|
|
627
|
+
for (const demotion of reviewDemotions) {
|
|
628
|
+
const causes = Object.entries(demotion.causes)
|
|
629
|
+
.map(([cause, count]) => `${cause}${count === demotion.count ? "" : ` ×${count}`}`).join(", ");
|
|
630
|
+
rows.push(alignedStatusRow("warn", demotion.reviewer, `${demotion.count} review seat${demotion.count === 1 ? "" : "s"} demoted · cause ${causes}`));
|
|
631
|
+
}
|
|
632
|
+
}
|
|
584
633
|
if (existsSync(graphPath(cwd))) {
|
|
585
634
|
try {
|
|
586
635
|
const graph = loadGraph(cwd);
|
package/dist/cli/commands/run.js
CHANGED
|
@@ -35,6 +35,7 @@ const RAIL_TONES = {
|
|
|
35
35
|
};
|
|
36
36
|
/** Approval-close lifecycle labels extend the established closed rail vocabulary. */
|
|
37
37
|
export const APPROVAL_RAIL_ROWS = {
|
|
38
|
+
"end-condition-held": { label: "close held", tone: "attention" },
|
|
38
39
|
"approval-window-start": { label: "approval window", tone: "attention" },
|
|
39
40
|
"approval-window-expired": { label: "approval window expired", tone: "attention" },
|
|
40
41
|
"tip-verify-cancelled": { label: "tip verify cancelled", tone: "attention" },
|
|
@@ -124,6 +125,7 @@ export const RAIL_ROWS = {
|
|
|
124
125
|
"gate-result": { label: "gate", tone: "pass" },
|
|
125
126
|
"baseline-wait": { label: "baseline wait", tone: "active" },
|
|
126
127
|
"suite-budget": { label: "suite budget", tone: "attention" },
|
|
128
|
+
"gate-replayed": { label: "gate replayed", tone: "attention" },
|
|
127
129
|
"gate-reused": { label: "gate reused", tone: "neutral" },
|
|
128
130
|
"judge-retry": { label: "judge retry", tone: "attention" },
|
|
129
131
|
"review-no-verdict": { label: "review unavailable", tone: "attention" },
|
|
@@ -825,23 +825,17 @@ const liveness = (events, daemon, now = Date.now()) => {
|
|
|
825
825
|
const state = daemon.state === "alive" ? "alive" : ended ? "finished" : "dead";
|
|
826
826
|
return `last event ${age} ago · daemon pid ${daemon.pid} ${state}${cause ? ` · ${cause}` : ""}`;
|
|
827
827
|
};
|
|
828
|
-
//
|
|
829
|
-
//
|
|
830
|
-
//
|
|
828
|
+
// OBS-978: the one comparator resume, plan and the operator fold read decides the join — it already
|
|
829
|
+
// audits the rehash chain. When it binds and the journal holds a graph-rehash, the newest row is the
|
|
830
|
+
// audit that bound the loaded graph; rehashAt marks the facts recorded against a prior graph.
|
|
831
831
|
const statusEngagement = (events, loadedHash) => {
|
|
832
|
-
|
|
832
|
+
if (!engagementComparable(events, loadedHash).comparable)
|
|
833
|
+
return { comparable: false };
|
|
833
834
|
for (let i = events.length - 1; i >= 0; i--) {
|
|
834
|
-
|
|
835
|
-
|
|
836
|
-
continue;
|
|
837
|
-
const auditsBaseline = baseline.comparable || baseline.reason === "mismatch"
|
|
838
|
-
? event.data.from === baseline.recorded
|
|
839
|
-
: event.data.from === null;
|
|
840
|
-
return event.data.to === loadedHash && auditsBaseline
|
|
841
|
-
? { comparable: true, rehashAt: i }
|
|
842
|
-
: { comparable: false };
|
|
835
|
+
if (events[i].event === "graph-rehash")
|
|
836
|
+
return { comparable: true, rehashAt: i };
|
|
843
837
|
}
|
|
844
|
-
return { comparable:
|
|
838
|
+
return { comparable: true };
|
|
845
839
|
};
|
|
846
840
|
// The journal's own reader rule (src/run/journal.ts readJsonl), applied to bytes already in hand:
|
|
847
841
|
// skip blanks, drop a line that will not parse (a torn trailing write after a crash), keep the rest.
|
|
@@ -13,6 +13,12 @@ export type OwnershipFinding = {
|
|
|
13
13
|
taskIds: string[];
|
|
14
14
|
corroboration?: OwnershipCorroboration;
|
|
15
15
|
detail: string;
|
|
16
|
+
} | {
|
|
17
|
+
code: "unowned-shape-oracle";
|
|
18
|
+
taskId: string;
|
|
19
|
+
source: string;
|
|
20
|
+
oracle: string;
|
|
21
|
+
detail: string;
|
|
16
22
|
} | {
|
|
17
23
|
code: "test-path-outside-allowlist";
|
|
18
24
|
taskId: string;
|
|
@@ -26,6 +32,9 @@ export type OwnershipFinding = {
|
|
|
26
32
|
path: string;
|
|
27
33
|
detail: string;
|
|
28
34
|
};
|
|
35
|
+
export declare const SHAPE_ORACLES: readonly ["tests/run/narration.test.ts", "tests/cli/brand-surfaces.test.ts", "tests/run/notify-identity.test.ts", "tests/run/outcome-projections.test.ts", "tests/cockpit/setup.test.ts"];
|
|
36
|
+
export declare const SHAPE_ORACLE_SOURCES: readonly ["src/run/daemon.ts", "src/cli/commands/run.ts"];
|
|
37
|
+
export declare const SHAPE_ORACLE_MAP: Record<string, readonly string[]>;
|
|
29
38
|
/**
|
|
30
39
|
* Cross-task ownership evidence. Findings are data: this checker never throws or changes the graph;
|
|
31
40
|
* the compile seam promotes only a corroborated unowned-test finding and reports every other shape.
|
|
@@ -2,6 +2,38 @@ import { readFileSync, readdirSync } from "node:fs";
|
|
|
2
2
|
import { basename, extname, join, posix } from "node:path";
|
|
3
3
|
import { filesGlob } from "../graph/files-glob.js";
|
|
4
4
|
import { collateralHits } from "./collateral.js";
|
|
5
|
+
export const SHAPE_ORACLES = [
|
|
6
|
+
"tests/run/narration.test.ts",
|
|
7
|
+
"tests/cli/brand-surfaces.test.ts",
|
|
8
|
+
"tests/run/notify-identity.test.ts",
|
|
9
|
+
"tests/run/outcome-projections.test.ts",
|
|
10
|
+
"tests/cockpit/setup.test.ts",
|
|
11
|
+
];
|
|
12
|
+
export const SHAPE_ORACLE_SOURCES = [
|
|
13
|
+
"src/run/daemon.ts",
|
|
14
|
+
"src/cli/commands/run.ts",
|
|
15
|
+
];
|
|
16
|
+
export const SHAPE_ORACLE_MAP = {
|
|
17
|
+
"src/run/daemon.ts": SHAPE_ORACLES,
|
|
18
|
+
"src/cli/commands/run.ts": SHAPE_ORACLES,
|
|
19
|
+
};
|
|
20
|
+
// Anchored-glob only: a files[] entry touches a mapped source when it names the source (or its
|
|
21
|
+
// extensionless stem) exactly, or when its glob's literal head — everything before the first
|
|
22
|
+
// wildcard — is the stem plus a literal dot, i.e. the wildcard only ever spans the extension
|
|
23
|
+
// ("src/run/daemon.*"). A broad multi-file glob like "src/**" that merely happens to cover the
|
|
24
|
+
// source is an unrelated task casting a wide net, not one touching daemon narration — that
|
|
25
|
+
// distinction is what broke every fixture using tests/fixtures/sample.prd.md's files: src/** task.
|
|
26
|
+
function touchesSource(files, source) {
|
|
27
|
+
const stem = source.replace(/\.(?:[cm]?[jt]sx?)$/, "");
|
|
28
|
+
return files.some((entry) => {
|
|
29
|
+
if (entry === source || entry === stem)
|
|
30
|
+
return true;
|
|
31
|
+
const special = entry.search(/[*?{[]/);
|
|
32
|
+
// Leg-2 v2.5.3: the anchored head is necessary, not sufficient — the pattern must also MATCH the
|
|
33
|
+
// source under the shared matcher, or "src/run/daemon.{js,jsx}" (never daemon.ts) would count as a touch.
|
|
34
|
+
return special !== -1 && entry.slice(0, special) === `${stem}.` && filesGlob([entry])(source);
|
|
35
|
+
});
|
|
36
|
+
}
|
|
5
37
|
const normalize = (path) => path.replace(/^\.\//, "").split("\\").join("/");
|
|
6
38
|
function testSources(repoRoot) {
|
|
7
39
|
const root = join(repoRoot, "tests");
|
|
@@ -155,6 +187,7 @@ export function ownershipFindings(tasks, repoRoot) {
|
|
|
155
187
|
const context = task.context.map(normalize);
|
|
156
188
|
return {
|
|
157
189
|
task,
|
|
190
|
+
files,
|
|
158
191
|
owns: files.length === 0 ? () => false : filesGlob(files),
|
|
159
192
|
allows: files.length === 0 ? () => true : filesGlob([...files, ...context]),
|
|
160
193
|
};
|
|
@@ -204,6 +237,23 @@ export function ownershipFindings(tasks, repoRoot) {
|
|
|
204
237
|
});
|
|
205
238
|
}
|
|
206
239
|
}
|
|
240
|
+
for (const entry of indexed) {
|
|
241
|
+
for (const [source, oracles] of Object.entries(SHAPE_ORACLE_MAP)) {
|
|
242
|
+
if (!touchesSource(entry.files, source))
|
|
243
|
+
continue;
|
|
244
|
+
for (const oracle of oracles) {
|
|
245
|
+
if (!entry.owns(oracle)) {
|
|
246
|
+
findings.push({
|
|
247
|
+
code: "unowned-shape-oracle",
|
|
248
|
+
taskId: entry.task.id,
|
|
249
|
+
source,
|
|
250
|
+
oracle,
|
|
251
|
+
detail: `${entry.task.id} touches ${source} without owning shape oracle ${oracle}`,
|
|
252
|
+
});
|
|
253
|
+
}
|
|
254
|
+
}
|
|
255
|
+
}
|
|
256
|
+
}
|
|
207
257
|
for (const source of sources) {
|
|
208
258
|
for (const owner of owners(source.path)) {
|
|
209
259
|
for (const path of repositoryPaths(source.text)) {
|
|
@@ -234,11 +284,24 @@ export function ownershipFindings(tasks, repoRoot) {
|
|
|
234
284
|
}
|
|
235
285
|
}
|
|
236
286
|
}
|
|
237
|
-
|
|
287
|
+
const target = (f) => {
|
|
288
|
+
switch (f.code) {
|
|
289
|
+
case "unowned-test": return f.test;
|
|
290
|
+
case "unowned-shape-oracle": return f.oracle;
|
|
291
|
+
case "test-path-outside-allowlist": return f.test;
|
|
292
|
+
case "unordered-context-write": return f.path;
|
|
293
|
+
}
|
|
294
|
+
};
|
|
295
|
+
return findings.sort((a, b) => {
|
|
296
|
+
const taskA = "taskId" in a ? a.taskId : "";
|
|
297
|
+
const taskB = "taskId" in b ? b.taskId : "";
|
|
298
|
+
return `${a.code}:${target(a)}:${taskA}`.localeCompare(`${b.code}:${target(b)}:${taskB}`);
|
|
299
|
+
});
|
|
238
300
|
}
|
|
239
301
|
export function renderOwnershipFinding(finding) {
|
|
240
302
|
return `tickmarkr: ownership-lint[${finding.code}]: ${finding.detail}`;
|
|
241
303
|
}
|
|
242
304
|
export function blocksCompile(finding) {
|
|
243
|
-
return finding.code === "unowned-test" && finding.corroboration !== undefined
|
|
305
|
+
return (finding.code === "unowned-test" && finding.corroboration !== undefined)
|
|
306
|
+
|| finding.code === "unowned-shape-oracle";
|
|
244
307
|
}
|
package/dist/gates/baseline.d.ts
CHANGED
|
@@ -161,6 +161,7 @@ export declare function detectVacuousOracles(cwd: string, tasks: ReadonlyArray<{
|
|
|
161
161
|
export declare function compareToBaseline(cwd: string, commands: Record<string, string>, baseline: Baseline, enabled: string[], opts?: {
|
|
162
162
|
rerunOf?: HostStarvedRerun;
|
|
163
163
|
infraRerun?: HostStarvedRerun;
|
|
164
|
+
selected?: readonly string[];
|
|
164
165
|
}): Promise<GateResult[]>;
|
|
165
166
|
export interface HostStarvedRerun {
|
|
166
167
|
durationMs: number;
|
|
@@ -184,7 +185,10 @@ export declare function resetCalmWindowForTests(): void;
|
|
|
184
185
|
export declare function waitForCalmWindow(): Promise<number>;
|
|
185
186
|
/** Summary totals survive digit-normalized fingerprints and exclude test-owned echoed output. */
|
|
186
187
|
export declare function runnerFileCount(raw: string): number | null;
|
|
187
|
-
export declare function fileCountDeficit(entry: BaselineCommand | undefined, raw: string
|
|
188
|
+
export declare function fileCountDeficit(entry: BaselineCommand | undefined, raw: string, opts?: {
|
|
189
|
+
name?: string;
|
|
190
|
+
selected?: readonly string[];
|
|
191
|
+
}): string | undefined;
|
|
188
192
|
/** Both readers classify fresh evidence first, retaining the whole-output guard against infra forgiveness. */
|
|
189
193
|
export declare function classifyFreshRunnerOutput(entry: BaselineCommand | undefined, raw: string, code: number): FailureClassification | undefined;
|
|
190
194
|
export {};
|
package/dist/gates/baseline.js
CHANGED
|
@@ -627,7 +627,16 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
|
|
|
627
627
|
details: `${g.meta?.classification === "infra" ? "infra; " : ""}runner-infra rerun after waiting ${opts.infraRerun.waitedMs}ms for a calm load window: ${withRerun.details.replace(/^infra; /, "")}`,
|
|
628
628
|
meta: { ...withRerun.meta, runnerInfraRerun: opts.infraRerun },
|
|
629
629
|
} : withRerun;
|
|
630
|
-
|
|
630
|
+
const withSelected = name === "test" && opts.selected
|
|
631
|
+
? {
|
|
632
|
+
...final,
|
|
633
|
+
meta: {
|
|
634
|
+
...final.meta,
|
|
635
|
+
...(Array.isArray(opts.selected) ? { selectedTests: [...opts.selected] } : {}),
|
|
636
|
+
},
|
|
637
|
+
}
|
|
638
|
+
: final;
|
|
639
|
+
results.push(r.capacity ? { ...withSelected, capacity: r.capacity } : withSelected);
|
|
631
640
|
};
|
|
632
641
|
// …and whether the entry that would forgive this command was measured in the same world. A
|
|
633
642
|
// baseline captured under a different fork cap forgives nothing: its fingerprints describe a
|
|
@@ -657,7 +666,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
|
|
|
657
666
|
continue;
|
|
658
667
|
}
|
|
659
668
|
const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
|
|
660
|
-
const deficit = fileCountDeficit(entry, raw);
|
|
669
|
+
const deficit = fileCountDeficit(entry, raw, { name, selected: opts.selected });
|
|
661
670
|
if (deficit) {
|
|
662
671
|
record({ gate: name, pass: false, details: deficit, meta: { classification: "infra", infra: true } });
|
|
663
672
|
continue;
|
|
@@ -689,7 +698,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
|
|
|
689
698
|
if (name === "test" && classification === "infra" && !rerunOf && !opts.infraRerun) {
|
|
690
699
|
const waitedMs = await waitForCalmWindow();
|
|
691
700
|
const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
|
|
692
|
-
results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { infraRerun: provenance }));
|
|
701
|
+
results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { ...opts, infraRerun: provenance }));
|
|
693
702
|
continue;
|
|
694
703
|
}
|
|
695
704
|
// OBS-896: every fresh failure must be timeout-class, and the suite must take at least twice its
|
|
@@ -698,7 +707,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
|
|
|
698
707
|
&& hostStarved(failing.join("\n"), r.durationMs ?? 0, entry?.durationMs)) {
|
|
699
708
|
const waitedMs = await waitForCalmWindow();
|
|
700
709
|
const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
|
|
701
|
-
results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { rerunOf: provenance }));
|
|
710
|
+
results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { ...opts, rerunOf: provenance }));
|
|
702
711
|
continue;
|
|
703
712
|
}
|
|
704
713
|
if (classification === "infra") {
|
|
@@ -833,7 +842,11 @@ export function runnerFileCount(raw) {
|
|
|
833
842
|
});
|
|
834
843
|
return counts.length ? counts.reduce((sum, count) => sum + count, 0) : null;
|
|
835
844
|
}
|
|
836
|
-
export function fileCountDeficit(entry, raw) {
|
|
845
|
+
export function fileCountDeficit(entry, raw, opts) {
|
|
846
|
+
// OBS-985: only a real, named selected-test run of the TEST gate is exempt — a truthy flag or an
|
|
847
|
+
// empty list named no selection and let a full-suite call opt itself out of the deficit guard.
|
|
848
|
+
if (opts?.name === "test" && opts.selected !== undefined && opts.selected.length > 0)
|
|
849
|
+
return undefined;
|
|
837
850
|
const actual = runnerFileCount(raw);
|
|
838
851
|
return entry?.fileCount != null && actual !== null && actual < entry.fileCount
|
|
839
852
|
? `infra; runner reported ${actual} test files, below baseline ${entry.fileCount} — suite incomplete`
|
package/dist/gates/llm.d.ts
CHANGED
|
@@ -65,7 +65,7 @@ export interface LlmRunResult {
|
|
|
65
65
|
seatAuthoredBytes?: number;
|
|
66
66
|
}
|
|
67
67
|
export declare const PROMPT_GLYPHS: readonly ["➜", "❯", "$", "%", ">>", ">"];
|
|
68
|
-
export declare function reviewSeatOutput(raw: string, nonce: string): string;
|
|
68
|
+
export declare function reviewSeatOutput(raw: string, nonce: string, adapterBannerRows?: readonly string[]): string;
|
|
69
69
|
export declare const REVIEW_FIRST_LIVENESS_MS = 30000;
|
|
70
70
|
export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number): Promise<string>;
|
|
71
71
|
export declare function runViaDriver(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via: LlmVia, timeoutMs?: number): Promise<string>;
|
package/dist/gates/llm.js
CHANGED
|
@@ -204,12 +204,14 @@ const IDENTITY_OPENERS = ["review ·", "tickmarkr"];
|
|
|
204
204
|
// Stages of the preamble walk. Each complete harness row is accepted only at or after its stage.
|
|
205
205
|
const ECHO = 0, START = 1, BANNER = 2, IDENTITY = 3, SEAT = 4;
|
|
206
206
|
// The char offset where the seat's own text begins, or -1 when the capture ends inside the preamble.
|
|
207
|
-
function seatStart(output) {
|
|
207
|
+
function seatStart(output, adapterBannerRows) {
|
|
208
208
|
const rows = output.split("\n");
|
|
209
209
|
if (rows.length > 1 && rows[rows.length - 1] === "")
|
|
210
210
|
rows.pop(); // the read's own line terminator
|
|
211
211
|
let stage = ECHO;
|
|
212
212
|
let bannerAt; // next banner row expected once the banner has begun
|
|
213
|
+
let adapterBannerAt;
|
|
214
|
+
let identitySeen = false;
|
|
213
215
|
let offset = 0;
|
|
214
216
|
for (let i = 0; i < rows.length; i++) {
|
|
215
217
|
const row = rows[i].replace(/[ \t]+$/, "");
|
|
@@ -217,7 +219,7 @@ function seatStart(output) {
|
|
|
217
219
|
const last = i === rows.length - 1;
|
|
218
220
|
// Complete harness rows: equality against the shape the preamble allows at this stage.
|
|
219
221
|
let accepted = false;
|
|
220
|
-
if (stage < SEAT && t.length === 0)
|
|
222
|
+
if (!identitySeen && stage < SEAT && t.length === 0)
|
|
221
223
|
accepted = true; // blank rows between preamble rows
|
|
222
224
|
else if (stage <= ECHO && completeEchoRow(row))
|
|
223
225
|
accepted = true;
|
|
@@ -225,17 +227,27 @@ function seatStart(output) {
|
|
|
225
227
|
stage = BANNER;
|
|
226
228
|
accepted = true;
|
|
227
229
|
}
|
|
228
|
-
else if (stage <= BANNER && bannerAt === undefined && BANNER_ROWS.includes(row)) {
|
|
230
|
+
else if (!identitySeen && stage <= BANNER && bannerAt === undefined && BANNER_ROWS.includes(row)) {
|
|
229
231
|
stage = BANNER;
|
|
230
232
|
bannerAt = BANNER_ROWS.indexOf(row) + 1;
|
|
231
233
|
accepted = true;
|
|
232
234
|
}
|
|
233
|
-
else if (stage <= BANNER && bannerAt !== undefined && row === BANNER_ROWS[bannerAt]) {
|
|
235
|
+
else if (!identitySeen && stage <= BANNER && bannerAt !== undefined && row === BANNER_ROWS[bannerAt]) {
|
|
234
236
|
bannerAt++;
|
|
235
237
|
accepted = true;
|
|
236
238
|
}
|
|
237
|
-
else if (stage <= IDENTITY &&
|
|
238
|
-
stage =
|
|
239
|
+
else if (stage <= IDENTITY && adapterBannerAt === undefined && adapterBannerRows.includes(row)) {
|
|
240
|
+
stage = BANNER;
|
|
241
|
+
adapterBannerAt = adapterBannerRows.indexOf(row) + 1;
|
|
242
|
+
accepted = true;
|
|
243
|
+
}
|
|
244
|
+
else if (stage <= IDENTITY && adapterBannerAt !== undefined && row === adapterBannerRows[adapterBannerAt]) {
|
|
245
|
+
adapterBannerAt++;
|
|
246
|
+
accepted = true;
|
|
247
|
+
}
|
|
248
|
+
else if (!identitySeen && stage <= IDENTITY && IDENTITY_LINE.test(t)) {
|
|
249
|
+
stage = IDENTITY;
|
|
250
|
+
identitySeen = true;
|
|
239
251
|
accepted = true;
|
|
240
252
|
}
|
|
241
253
|
if (accepted) {
|
|
@@ -253,9 +265,16 @@ function seatStart(output) {
|
|
|
253
265
|
? BANNER_ROWS.some((b) => b.startsWith(row))
|
|
254
266
|
: BANNER_ROWS[bannerAt]?.startsWith(row) === true))
|
|
255
267
|
return -1;
|
|
268
|
+
if (stage <= IDENTITY && (adapterBannerAt === undefined
|
|
269
|
+
? adapterBannerRows.some((b) => b.startsWith(row))
|
|
270
|
+
: adapterBannerRows[adapterBannerAt]?.startsWith(row) === true))
|
|
271
|
+
return -1;
|
|
256
272
|
// The identity row is painted right after the banner, so its prefix is a partial paint only there;
|
|
257
273
|
// with no banner in the capture, "review" or "tick" alone is the seat's own first row.
|
|
258
|
-
if (stage <= IDENTITY
|
|
274
|
+
if (!identitySeen && stage <= IDENTITY
|
|
275
|
+
&& (bannerAt === BANNER_ROWS.length
|
|
276
|
+
|| (adapterBannerRows.length > 0 && adapterBannerAt === adapterBannerRows.length))
|
|
277
|
+
&& IDENTITY_OPENERS.some((o) => o.startsWith(t)))
|
|
259
278
|
return -1;
|
|
260
279
|
return offset;
|
|
261
280
|
}
|
|
@@ -266,9 +285,9 @@ function seatStart(output) {
|
|
|
266
285
|
// Every capture taken before the preamble finishes therefore measures ZERO seat-authored bytes,
|
|
267
286
|
// which is what makes the caller's running Math.max safe: a partial banner counted once would be
|
|
268
287
|
// retained for the whole call and buy a silent seat its full ceiling.
|
|
269
|
-
export function reviewSeatOutput(raw, nonce) {
|
|
288
|
+
export function reviewSeatOutput(raw, nonce, adapterBannerRows = []) {
|
|
270
289
|
const output = stripVTControlCharacters(raw).replace(/\r\n?/g, "\n");
|
|
271
|
-
const start = seatStart(output);
|
|
290
|
+
const start = seatStart(output, adapterBannerRows);
|
|
272
291
|
if (start < 0)
|
|
273
292
|
return "";
|
|
274
293
|
const seat = output.slice(start);
|
|
@@ -287,7 +306,10 @@ async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 3000
|
|
|
287
306
|
const pf = join(dir, "prompt.md");
|
|
288
307
|
writeFileSync(pf, prompt);
|
|
289
308
|
const r = await sh(adapter.headlessCommand(pf, model), cwd, timeoutMs);
|
|
290
|
-
|
|
309
|
+
const output = r.stdout + "\n" + r.stderr;
|
|
310
|
+
const nonce = extractPromptNonce(prompt) ?? "";
|
|
311
|
+
return { output, exitCode: r.code, timedOut: r.timedOut === true,
|
|
312
|
+
seatAuthoredBytes: Buffer.byteLength(reviewSeatOutput(output, nonce, adapter.harnessBannerRows).trim()) };
|
|
291
313
|
}
|
|
292
314
|
finally {
|
|
293
315
|
rmSync(dir, { recursive: true, force: true });
|
|
@@ -349,7 +371,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
|
|
|
349
371
|
const startedAt = Date.now();
|
|
350
372
|
out = await via.driver.read(slot, 400);
|
|
351
373
|
const reviewing = prompt.startsWith("TICKMARKR-REVIEW");
|
|
352
|
-
seatAuthoredBytes = Buffer.byteLength(reviewSeatOutput(out, nonce));
|
|
374
|
+
seatAuthoredBytes = Buffer.byteLength(reviewSeatOutput(out, nonce, adapter.harnessBannerRows));
|
|
353
375
|
let firstLivenessObserved = false;
|
|
354
376
|
let priorSnapshot = normalizeStallSnapshot(out);
|
|
355
377
|
const anchoredAt = Date.now();
|
|
@@ -365,7 +387,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
|
|
|
365
387
|
const matched = await via.driver.waitOutput(slot, exitPattern, sliceMs, { regex: true });
|
|
366
388
|
const raw = await via.driver.read(slot, 400);
|
|
367
389
|
out = raw;
|
|
368
|
-
seatAuthoredBytes = Math.max(seatAuthoredBytes, Buffer.byteLength(reviewSeatOutput(raw, nonce)));
|
|
390
|
+
seatAuthoredBytes = Math.max(seatAuthoredBytes, Buffer.byteLength(reviewSeatOutput(raw, nonce, adapter.harnessBannerRows)));
|
|
369
391
|
// waitOutput is the driver's authoritative marker match. The raw check covers drivers whose
|
|
370
392
|
// wait timed out at the same boundary the marker landed; either way a trailer completes
|
|
371
393
|
// normally and is never mistaken for inactivity.
|
package/dist/gates/review.d.ts
CHANGED
|
@@ -53,6 +53,17 @@ export declare function isDiffCapPark(result: GateResult): boolean;
|
|
|
53
53
|
export declare function diffCapParkReason(results: GateResult[]): string | null;
|
|
54
54
|
export declare function modelId(model: string): string;
|
|
55
55
|
export { modelProvider };
|
|
56
|
+
/**
|
|
57
|
+
* One function decides whether a reviewer's `resolved` or `reraised` id names a carried fingerprint,
|
|
58
|
+
* comparing both sides with every whitespace run removed (`s.replace(/\s+/g, "")`).
|
|
59
|
+
*/
|
|
60
|
+
export declare function matchClosureId(candidate: unknown, fingerprint: string): boolean;
|
|
61
|
+
export declare function matchClosureId(candidate: unknown, fingerprints: Iterable<string>): string | undefined;
|
|
62
|
+
/**
|
|
63
|
+
* Validates closure ids in a review verdict: membership, duplication, and coverage of every prior id
|
|
64
|
+
* all route through matchClosureId.
|
|
65
|
+
*/
|
|
66
|
+
export declare function isReviewClosureInvalid(v: Pick<ReviewVerdict, "resolved" | "reraised"> | null | undefined, priorIds: ReadonlySet<string> | readonly string[]): boolean;
|
|
56
67
|
export declare function pickReviewer(author: Assignment, channels: BillingChannel[], exclude?: string[], // v1.1 failover: reviewer channels that already produced garbage for this task
|
|
57
68
|
prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
|
|
58
69
|
floor?: Tier, // task-declared only; config floors govern workers and must not silently move review seats
|
package/dist/gates/review.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { writeFileSync } from "node:fs";
|
|
1
|
+
import { existsSync, writeFileSync } from "node:fs";
|
|
2
2
|
import { join } from "node:path";
|
|
3
3
|
import { channelKey, shq } from "../adapters/types.js";
|
|
4
4
|
import { criticalPathHits, DEFAULT_DIFF_CAP, DEFAULT_REVIEW_CRITICAL_PATHS, declaredReviewPolicy, isReviewLeafPath, raiseReviewPolicy, REVIEW_VERSION_MIRRORS, TIER_RANK, } from "../config/config.js";
|
|
@@ -168,6 +168,31 @@ export function modelId(model) {
|
|
|
168
168
|
return model.slice(model.lastIndexOf("/") + 1);
|
|
169
169
|
}
|
|
170
170
|
export { modelProvider };
|
|
171
|
+
export function matchClosureId(candidate, target) {
|
|
172
|
+
if (typeof candidate !== "string")
|
|
173
|
+
return typeof target === "string" ? false : undefined;
|
|
174
|
+
const normCandidate = candidate.replace(/\s+/g, "");
|
|
175
|
+
if (typeof target === "string") {
|
|
176
|
+
return normCandidate === target.replace(/\s+/g, "");
|
|
177
|
+
}
|
|
178
|
+
for (const fp of target) {
|
|
179
|
+
if (typeof fp === "string" && normCandidate === fp.replace(/\s+/g, ""))
|
|
180
|
+
return fp;
|
|
181
|
+
}
|
|
182
|
+
return undefined;
|
|
183
|
+
}
|
|
184
|
+
/**
|
|
185
|
+
* Validates closure ids in a review verdict: membership, duplication, and coverage of every prior id
|
|
186
|
+
* all route through matchClosureId.
|
|
187
|
+
*/
|
|
188
|
+
export function isReviewClosureInvalid(v, priorIds) {
|
|
189
|
+
const priors = priorIds instanceof Set ? priorIds : new Set(priorIds);
|
|
190
|
+
const closureLists = [v?.resolved, v?.reraised];
|
|
191
|
+
const allCandidateIds = [...(v?.resolved ?? []), ...(v?.reraised ?? [])];
|
|
192
|
+
return !!v && (priors.size > 0 || closureLists.some((list) => list !== undefined)) && (closureLists.some((list) => !Array.isArray(list) || list.some((id) => !matchClosureId(id, priors)))
|
|
193
|
+
|| new Set(allCandidateIds.map((id) => matchClosureId(id, priors) ?? id)).size !== allCandidateIds.length
|
|
194
|
+
|| [...priors].some((id) => !allCandidateIds.some((candidate) => matchClosureId(candidate, id))));
|
|
195
|
+
}
|
|
171
196
|
// v1.53 T2: same entry grammar as routing.map.prefer (router.ts preferIndex — router is out of this
|
|
172
197
|
// module's dependency direction for a private fn, so the 3 lines live here too): `adapter` matches
|
|
173
198
|
// every channel of that adapter, `adapter:model` exactly one; unmatched channels sort after all entries.
|
|
@@ -357,7 +382,19 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
357
382
|
// make two otherwise-identical runs diverge in their journal bytes. The reviewer channel already
|
|
358
383
|
// disambiguates every call that matters: a retry always excludes the flaked channel (run-gates.ts),
|
|
359
384
|
// so it can never collide with the attempt it replaces.
|
|
360
|
-
const
|
|
385
|
+
const baseArtifactId = `${task.id}-${channelKey(reviewer).replace(/[^a-zA-Z0-9_.-]/g, "-")}`;
|
|
386
|
+
let artifactId = baseArtifactId;
|
|
387
|
+
if (artifactDir) {
|
|
388
|
+
if (existsSync(join(artifactDir, `review-brief-${baseArtifactId}.md`)) ||
|
|
389
|
+
existsSync(join(artifactDir, `review-raw-${baseArtifactId}.txt`))) {
|
|
390
|
+
let counter = 2;
|
|
391
|
+
while (existsSync(join(artifactDir, `review-brief-${baseArtifactId}-${counter}.md`)) ||
|
|
392
|
+
existsSync(join(artifactDir, `review-raw-${baseArtifactId}-${counter}.txt`))) {
|
|
393
|
+
counter++;
|
|
394
|
+
}
|
|
395
|
+
artifactId = `${baseArtifactId}-${counter}`;
|
|
396
|
+
}
|
|
397
|
+
}
|
|
361
398
|
const briefPath = artifactDir ? join(artifactDir, `review-brief-${artifactId}.md`) : undefined;
|
|
362
399
|
// Persistence is evidence, not a gate input: a full disk or a removed run dir never fails the gate.
|
|
363
400
|
let savedBrief;
|
|
@@ -397,10 +434,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
397
434
|
const v = extractVerdictJson(raw, nonce);
|
|
398
435
|
const findings = v && Array.isArray(v.findings) ? v.findings : null;
|
|
399
436
|
const priorIds = new Set(priorMaterials.map((finding) => finding.fingerprint));
|
|
400
|
-
const
|
|
401
|
-
const closureInvalid = !!v && (priorIds.size > 0 || closureLists.some((list) => list !== undefined)) && (closureLists.some((list) => !Array.isArray(list) || list.some((id) => typeof id !== "string" || !priorIds.has(id)))
|
|
402
|
-
|| new Set([...(v?.resolved ?? []), ...(v?.reraised ?? [])]).size !== (v?.resolved?.length ?? 0) + (v?.reraised?.length ?? 0)
|
|
403
|
-
|| [...priorIds].some((id) => !v?.resolved?.includes(id) && !v?.reraised?.includes(id)));
|
|
437
|
+
const closureInvalid = isReviewClosureInvalid(v, priorIds);
|
|
404
438
|
// findings decides the verdict on its own; the legacy path still needs approve + issues to parse.
|
|
405
439
|
if (!v || closureInvalid || (findings === null && (typeof v.approve !== "boolean" || !Array.isArray(v.issues)))) {
|
|
406
440
|
// OBS-196: name the cause and persist the raw bytes — a ruled-on "unparseable" without its
|
|
@@ -434,7 +468,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
434
468
|
const decided = findings !== null
|
|
435
469
|
? classifyReviewFindings(findings)
|
|
436
470
|
: classifyReviewIssues(v.approve, v.issues);
|
|
437
|
-
const reraised = priorMaterials.filter((finding) => v.reraised?.
|
|
471
|
+
const reraised = priorMaterials.filter((finding) => v.reraised?.some((id) => matchClosureId(id, finding.fingerprint)));
|
|
438
472
|
if (reraised.length) {
|
|
439
473
|
if (decided.pass)
|
|
440
474
|
decided.headline = "requested changes";
|
|
@@ -455,7 +489,15 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
455
489
|
details,
|
|
456
490
|
meta: {
|
|
457
491
|
...policyMeta, ...rotationMeta, reviewer: channelKey(reviewer), vendor: reviewer.vendor, provider,
|
|
458
|
-
...(priorMaterials.length ? {
|
|
492
|
+
...(priorMaterials.length ? {
|
|
493
|
+
resolved: v.resolved,
|
|
494
|
+
reraised: v.reraised,
|
|
495
|
+
normalisedMatches: (v.resolved ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
|
|
496
|
+
resolvedMatches: (v.resolved ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
|
|
497
|
+
reraisedMatches: (v.reraised ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
|
|
498
|
+
normalisedResolved: (v.resolved ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
|
|
499
|
+
normalisedReraised: (v.reraised ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
|
|
500
|
+
} : {}),
|
|
459
501
|
...(reraised.length ? { findings: [
|
|
460
502
|
...structuredFindings("review", details).filter((finding) => !reraised.some((prior) => prior.note === finding.note)),
|
|
461
503
|
...reraised,
|
package/dist/gates/run-gates.js
CHANGED
|
@@ -366,7 +366,7 @@ export async function runGates(task, ctx) {
|
|
|
366
366
|
// ponytail: legacy runs adjacent tools in ONE compareToBaseline call, so there is one interval
|
|
367
367
|
// to measure and each of its gates carries it. Split it only if this branch ever stops batching.
|
|
368
368
|
const finish = startMeasurement();
|
|
369
|
-
const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [...gates]);
|
|
369
|
+
const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [...gates], selected ? { selected } : {});
|
|
370
370
|
const batch = finish();
|
|
371
371
|
for (const g of gates)
|
|
372
372
|
addMeasurement(g, batch);
|
|
@@ -386,7 +386,7 @@ export async function runGates(task, ctx) {
|
|
|
386
386
|
// any later tool before anyone reads its verdict.
|
|
387
387
|
for (const g of gates) {
|
|
388
388
|
await emitStart(g);
|
|
389
|
-
const [r] = await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g]));
|
|
389
|
+
const [r] = await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], g === "test" && selected ? { selected } : {}));
|
|
390
390
|
// the screen's interval IS the test gate's first interval, so the split needs no second clock
|
|
391
391
|
if (g === "test" && selected)
|
|
392
392
|
selectedDurationMs = spans.get("test").durationMs;
|
package/dist/run/daemon.d.ts
CHANGED
|
@@ -109,7 +109,9 @@ export declare const SUITE_WAIT_CEILING_MS = 600000;
|
|
|
109
109
|
export declare const setSuiteWaitCeilingForTests: (ms: number) => void;
|
|
110
110
|
export declare const resetSuiteWaitCeilingForTests: () => void;
|
|
111
111
|
export declare const APPROVAL_POLL_MS = 250;
|
|
112
|
-
export declare const APPROVAL_WINDOW_MS =
|
|
112
|
+
export declare const APPROVAL_WINDOW_MS = 120000;
|
|
113
|
+
export declare const setApprovalWindowForTests: (ms: number) => void;
|
|
114
|
+
export declare const resetApprovalWindowForTests: () => void;
|
|
113
115
|
export declare const EARLY_LAUNCH_LIVENESS_MS = 60000;
|
|
114
116
|
/** Test seam — lowers the empty-pane liveness window without sleeping 60s per case. */
|
|
115
117
|
export declare function setEarlyLaunchLivenessMsForTests(ms: number): void;
|
package/dist/run/daemon.js
CHANGED
|
@@ -307,7 +307,13 @@ let suiteWaitCeilingMs = SUITE_WAIT_CEILING_MS;
|
|
|
307
307
|
export const setSuiteWaitCeilingForTests = (ms) => { suiteWaitCeilingMs = ms; };
|
|
308
308
|
export const resetSuiteWaitCeilingForTests = () => { suiteWaitCeilingMs = SUITE_WAIT_CEILING_MS; };
|
|
309
309
|
export const APPROVAL_POLL_MS = 250;
|
|
310
|
-
export const APPROVAL_WINDOW_MS =
|
|
310
|
+
export const APPROVAL_WINDOW_MS = 120_000;
|
|
311
|
+
// Keep ordinary park tests off the operator's production wait. Explicit timing tests
|
|
312
|
+
// use the same setter/reset pattern as the suite wait ceiling above.
|
|
313
|
+
const DEFAULT_APPROVAL_WINDOW_MS = process.env.VITEST ? 1 : APPROVAL_WINDOW_MS;
|
|
314
|
+
let approvalWindowMs = DEFAULT_APPROVAL_WINDOW_MS;
|
|
315
|
+
export const setApprovalWindowForTests = (ms) => { approvalWindowMs = ms; };
|
|
316
|
+
export const resetApprovalWindowForTests = () => { approvalWindowMs = DEFAULT_APPROVAL_WINDOW_MS; };
|
|
311
317
|
const PROVIDER_DEATH_REQUEUE_CAP = 2; // v1.46 T1: requeue same assignment twice, then fall through to the normal ladder
|
|
312
318
|
const PROVIDER_DEATH_BACKOFF_MS = 500; // short backoff before provider-death requeue
|
|
313
319
|
const NO_TRAILER_DEMOTION_STREAK = 2; // OBS-57: consecutive no-trailer windows demote a channel for the rest of the run
|
|
@@ -1768,6 +1774,17 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1768
1774
|
return `${identity} — release with ${commands.map((command) => `\`${command}\``).join(" or ")}`;
|
|
1769
1775
|
};
|
|
1770
1776
|
const park = async (t, reason, kind, assignment, attempts, startMs, gateFails = 0, consults = 0, tokens, metered = 0, retryMode = "fresh", details = {}) => {
|
|
1777
|
+
// OBS-979: a worker refusal can identify the missing authoring scope even when gates
|
|
1778
|
+
// subsequently supply the park's disposition. Keep that actionable path on the park itself.
|
|
1779
|
+
const worker = journal.read().reverse().find((e) => e.taskId === t.id && e.event === "worker-result");
|
|
1780
|
+
if (t.files.length > 0 && worker?.data.ok === false && typeof worker.data.summary === "string") {
|
|
1781
|
+
const allowed = filesGlob(t.files);
|
|
1782
|
+
const paths = [...worker.data.summary.matchAll(/(?:^|[\s`'"(])((?:[A-Za-z0-9_@.()[\]-]+\/)+[A-Za-z0-9_@.[\]-]+|[A-Za-z0-9_@-]+(?:\.[A-Za-z0-9_-]+)+)(?=$|[\s`'"),:;.!?])/g)]
|
|
1783
|
+
.map((match) => match[1].replace(/^\.\//, "").replace(/\.$/, ""))
|
|
1784
|
+
.filter((path) => !path.split("/").includes("..") && !allowed(path));
|
|
1785
|
+
if (paths.length)
|
|
1786
|
+
reason += ` — files[] repair hint: ${[...new Set(paths)].join(", ")}`;
|
|
1787
|
+
}
|
|
1771
1788
|
graph = setStatus(graph, t.id, "human");
|
|
1772
1789
|
saveGraph(repoRoot, graph);
|
|
1773
1790
|
journal.append("task-human", t.id, { ...details, reason, kind });
|
|
@@ -2026,6 +2043,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2026
2043
|
gate: g.gate, ...(unverdicted ? {} : { pass: g.pass }), details: g.details,
|
|
2027
2044
|
...(gateSubject ? { commit: gateSubject.commit, attempt: gateSubject.attempt } : {}),
|
|
2028
2045
|
...(gateSubject?.replayMeasurement ? { replayMeasurement: true } : {}),
|
|
2046
|
+
...(gateSubject?.replayedFromAttempt !== undefined ? { replayedFromAttempt: gateSubject.replayedFromAttempt } : {}),
|
|
2029
2047
|
...(g.meta?.skipped === true || noVerdictReview ? { skipped: true } : {}),
|
|
2030
2048
|
// T9: an infra-only exit is journaled AS one. The operator reading a red `test` row has to
|
|
2031
2049
|
// be able to tell "the suite found a defect" from "the runner never ran", and the merge
|
|
@@ -4136,33 +4154,64 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4136
4154
|
const gated = await gitHead(wt);
|
|
4137
4155
|
gateSubject = { commit: await gateCommitSubject(taskBase, gated, wt), attempt };
|
|
4138
4156
|
await trackedDriver.project?.(t.id, "in-review");
|
|
4157
|
+
// Only the immediately preceding attempt can lend a red. Re-journal its results
|
|
4158
|
+
// as this attempt's verdicts so all existing disposition and fingerprint accounting
|
|
4159
|
+
// sees the replay, without buying another command or reviewer invocation.
|
|
4160
|
+
const taskEvents = journal.read().filter((e) => e.taskId === t.id);
|
|
4161
|
+
const previousRound = taskEvents.map((e) => e.event === "phase-start" && e.data.phase === "gates").lastIndexOf(true);
|
|
4162
|
+
const previousRows = retryMode === "repair"
|
|
4163
|
+
? taskEvents.slice(previousRound + 1).filter((e) => e.event === "gate-result"
|
|
4164
|
+
&& e.data.attempt === attempt - 1)
|
|
4165
|
+
: [];
|
|
4139
4166
|
journal.phaseStart(t.id, "gates");
|
|
4140
|
-
|
|
4141
|
-
|
|
4142
|
-
|
|
4143
|
-
|
|
4144
|
-
|
|
4145
|
-
|
|
4146
|
-
|
|
4147
|
-
|
|
4148
|
-
|
|
4149
|
-
|
|
4150
|
-
|
|
4151
|
-
|
|
4152
|
-
|
|
4153
|
-
|
|
4154
|
-
|
|
4155
|
-
|
|
4156
|
-
|
|
4157
|
-
|
|
4158
|
-
|
|
4159
|
-
|
|
4160
|
-
|
|
4161
|
-
|
|
4162
|
-
|
|
4163
|
-
|
|
4164
|
-
|
|
4165
|
-
|
|
4167
|
+
const replay = previousRows.length > 0
|
|
4168
|
+
&& previousRows.every((e) => e.data.commit === gateSubject.commit)
|
|
4169
|
+
&& previousRows.some((e) => e.data.pass === false && e.data.skipped !== true);
|
|
4170
|
+
if (replay) {
|
|
4171
|
+
gateSubject.replayedFromAttempt = attempt - 1;
|
|
4172
|
+
results = previousRows.map(({ data }) => ({
|
|
4173
|
+
gate: String(data.gate), pass: data.pass === true, details: String(data.details),
|
|
4174
|
+
// The verdict is reused; its old timing is not a measurement of this attempt.
|
|
4175
|
+
meta: Object.fromEntries(Object.entries(data).filter(([key]) => !GATE_TELEMETRY_KEYS.includes(key) && key !== "capacity")),
|
|
4176
|
+
}));
|
|
4177
|
+
commits = await commitsAheadOf(taskBase, wt);
|
|
4178
|
+
for (const g of results) {
|
|
4179
|
+
journal.append("gate-replayed", t.id, {
|
|
4180
|
+
attempt, priorAttempt: attempt - 1, gate: g.gate, commit: gateSubject.commit,
|
|
4181
|
+
...(g.meta?.skipped === true ? { skipped: true } : { pass: g.pass }),
|
|
4182
|
+
details: g.details,
|
|
4183
|
+
});
|
|
4184
|
+
journalGateResult(g);
|
|
4185
|
+
}
|
|
4186
|
+
}
|
|
4187
|
+
else {
|
|
4188
|
+
({ results, commits } = await withSuiteWindow(t.id, t.gates.includes("test") && commands.test !== undefined, () => runReviewRecovery(t, {
|
|
4189
|
+
carriedFindings: outstandingFindings,
|
|
4190
|
+
worktree: wt, baseRef: taskBase, result, author: assignment,
|
|
4191
|
+
commands, baseline, channels: pools.review, judgeChannels: pools.judge, adapters, cfg, artifactDir: journal.dir,
|
|
4192
|
+
collateral: collateral.get(t.id) ?? [],
|
|
4193
|
+
pipeline: "v185", selectTests: !testGateFailed,
|
|
4194
|
+
via: cfg.visibility.llm === "pane"
|
|
4195
|
+
? {
|
|
4196
|
+
driver: trackedDriver,
|
|
4197
|
+
// D-07: judge/review panes self-clean when their verdict is read (keepLlm) — only "forever" keeps them.
|
|
4198
|
+
keep: keepLlm,
|
|
4199
|
+
onSlot: keepLlm ? (s) => keptSlots.push(s) : undefined,
|
|
4200
|
+
// T2 ownership contract: canonical names (tickmarkr:<role>:<task>:0:<runId>) so reconcile
|
|
4201
|
+
// owns judge/review panes; run-gates' -r1 retry suffix becomes attempt 1 in llm.ts.
|
|
4202
|
+
// Same-name reuse across worker attempts is safe: panes self-clean when read (keepLlm),
|
|
4203
|
+
// and herdr's DEFECT-01 reclaim covers a kept holdover under keepPanes:forever.
|
|
4204
|
+
nameFor: (role) => formatOwnedName({ role, taskId: t.id, attempt: 0, runId }),
|
|
4205
|
+
// role-tab label (SUP-01): role-first + task id, unique per concurrent instance within a run.
|
|
4206
|
+
// Duplicate labels from a resumed run or operator-made tabs are accepted (per-process state).
|
|
4207
|
+
labelFor: (role) => `${role.toUpperCase()} ${t.id}`,
|
|
4208
|
+
}
|
|
4209
|
+
: undefined,
|
|
4210
|
+
excludeReviewers: badReviewers,
|
|
4211
|
+
reviewHistory, demotedReviewers,
|
|
4212
|
+
onGate,
|
|
4213
|
+
})));
|
|
4214
|
+
}
|
|
4166
4215
|
results.forEach(classifySignalOnlyTest);
|
|
4167
4216
|
graph = addEvidence(graph, t.id, { commits, gateResults: results, artifacts: [promptFile] });
|
|
4168
4217
|
saveGraph(repoRoot, graph);
|
|
@@ -4377,6 +4426,19 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4377
4426
|
};
|
|
4378
4427
|
taskLoopStarted = true;
|
|
4379
4428
|
const inflight = new Map();
|
|
4429
|
+
// A settled worker can release a dependency or an approval in the same poll tick.
|
|
4430
|
+
// Audit the current graph at every empty-flight boundary, never a prior ready snapshot.
|
|
4431
|
+
const holdEndCondition = () => {
|
|
4432
|
+
sweepLiveApprovals();
|
|
4433
|
+
const freeSlots = Math.max(0, concurrency - inflight.size);
|
|
4434
|
+
const dispatchable = readyTasks(graph).filter((t) => !inflight.has(t.id));
|
|
4435
|
+
if (freeSlots === 0 || dispatchable.length === 0)
|
|
4436
|
+
return false;
|
|
4437
|
+
for (const task of dispatchable) {
|
|
4438
|
+
journal.append("end-condition-held", task.id, { deps: task.deps, freeSlots });
|
|
4439
|
+
}
|
|
4440
|
+
return true;
|
|
4441
|
+
};
|
|
4380
4442
|
closeLoop: while (true) {
|
|
4381
4443
|
let approvalDeadline;
|
|
4382
4444
|
while (true) {
|
|
@@ -4412,7 +4474,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4412
4474
|
.filter((t) => t.status === "pending").every(behindPark);
|
|
4413
4475
|
if (onlyParks) {
|
|
4414
4476
|
if (approvalDeadline === undefined) {
|
|
4415
|
-
const windowMs = opts.approvalWindowMs ??
|
|
4477
|
+
const windowMs = opts.approvalWindowMs ?? approvalWindowMs;
|
|
4416
4478
|
approvalDeadline = Date.now() + windowMs;
|
|
4417
4479
|
journal.append("approval-window-start", undefined, { windowMs, parked: [...parked] });
|
|
4418
4480
|
// The narrator may itself append a decision at this boundary.
|
|
@@ -4428,6 +4490,10 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4428
4490
|
}
|
|
4429
4491
|
journal.append("approval-window-expired", undefined, { parked: [...parked] });
|
|
4430
4492
|
}
|
|
4493
|
+
if (holdEndCondition()) {
|
|
4494
|
+
approvalDeadline = undefined;
|
|
4495
|
+
continue;
|
|
4496
|
+
}
|
|
4431
4497
|
break;
|
|
4432
4498
|
}
|
|
4433
4499
|
approvalDeadline = undefined;
|
|
@@ -4438,6 +4504,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4438
4504
|
waiters.push(new Promise((wake) => setTimeout(wake, APPROVAL_POLL_MS)));
|
|
4439
4505
|
}
|
|
4440
4506
|
await Promise.race(waiters); // aborted rejects on termination — unwinds the run
|
|
4507
|
+
if (inflight.size === 0)
|
|
4508
|
+
holdEndCondition();
|
|
4441
4509
|
}
|
|
4442
4510
|
// D-07: the sweep now closes only what's LEFT in keptSlots — done-closed worker slots were removed
|
|
4443
4511
|
// (no double-close) and self-cleaned LLM/consult panes were never added under keepLlm:false. This
|
|
@@ -4515,12 +4583,11 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
4515
4583
|
const approvalSerialization = await acquireApprovalSerialization(repoRoot, runId);
|
|
4516
4584
|
releaseApprovalSerialization = approvalSerialization.release;
|
|
4517
4585
|
// Close the last poll-to-run-end race while holding the same serializer as approve.
|
|
4518
|
-
|
|
4519
|
-
if (readyTasks(graph).length) {
|
|
4586
|
+
if (holdEndCondition()) {
|
|
4520
4587
|
releaseApprovalSerialization();
|
|
4521
4588
|
releaseApprovalSerialization = undefined;
|
|
4522
|
-
|
|
4523
|
-
|
|
4589
|
+
// Verification has settled: retain its verdict. The cache key will require a
|
|
4590
|
+
// new battery if dispatch actually moves the tip or changes its commands.
|
|
4524
4591
|
continue closeLoop;
|
|
4525
4592
|
}
|
|
4526
4593
|
const outstanding = outstandingApprovals(journal.read());
|
package/dist/run/journal.js
CHANGED
|
@@ -949,18 +949,27 @@ export function gateResultJournalData(gate, pass, details, meta = {}) {
|
|
|
949
949
|
const signalBasis = deriveSignalBasis(gate, pass, details, meta);
|
|
950
950
|
return { gate, pass, details, ...meta, signalBasis, signalQuality: signalQualityFromBasis(signalBasis) };
|
|
951
951
|
}
|
|
952
|
-
// T3 (Sol #2 / Fable F2): one canonical engagement identity, shared by status
|
|
953
|
-
// event records graphDefinitionHash (over compiled task definitions only — see
|
|
954
|
-
//
|
|
955
|
-
//
|
|
952
|
+
// T3 (Sol #2 / Fable F2) + OBS-978: one canonical engagement identity, shared by status, plan, the operator
|
|
953
|
+
// fold AND resume. The run-start event records graphDefinitionHash (over compiled task definitions only — see
|
|
954
|
+
// graph.graphDefinitionHash); each audited graph-rehash row (resume --graph-changed) then moves the identity to
|
|
955
|
+
// its `to`, so the recorded hash is the last audited rehash, else run-start. A row is audited when its `from`
|
|
956
|
+
// names the identity it replaced, or the run-start one (all pre-OBS-978 daemons wrote); a row auditing neither
|
|
957
|
+
// binds nothing — the journal is unbound until a release from null. unbound (also a pre-v1.44 journal) and
|
|
956
958
|
// mismatch are both not-comparable — status renders the notice either way; resume refuses either way and
|
|
957
959
|
// distinguishes the reason only for its message and the --graph-changed release event.
|
|
958
960
|
export function recordedGraphDefinitionHash(events) {
|
|
961
|
+
const start = events.find((e) => e.event === "run-start");
|
|
962
|
+
if (!start)
|
|
963
|
+
return undefined;
|
|
964
|
+
const origin = typeof start.data.graphDefinitionHash === "string" ? start.data.graphDefinitionHash : null;
|
|
965
|
+
let recorded = origin;
|
|
959
966
|
for (const e of events) {
|
|
960
|
-
if (e.event
|
|
961
|
-
|
|
967
|
+
if (e.event !== "graph-rehash")
|
|
968
|
+
continue;
|
|
969
|
+
const audited = e.data.from === recorded || e.data.from === origin;
|
|
970
|
+
recorded = audited && typeof e.data.to === "string" ? e.data.to : null;
|
|
962
971
|
}
|
|
963
|
-
return undefined;
|
|
972
|
+
return recorded ?? undefined;
|
|
964
973
|
}
|
|
965
974
|
// THE shared comparator (criterion: status and resume decide through one comparator). status reads
|
|
966
975
|
// .comparable; resume reads .comparable plus .reason/.recorded for its refusal message and the release.
|
|
@@ -10,7 +10,7 @@ export class OperatorStateFold {
|
|
|
10
10
|
tasks = new Map();
|
|
11
11
|
start;
|
|
12
12
|
startEvent;
|
|
13
|
-
|
|
13
|
+
rehashes = [];
|
|
14
14
|
end;
|
|
15
15
|
active = false;
|
|
16
16
|
approved = false;
|
|
@@ -34,7 +34,7 @@ export class OperatorStateFold {
|
|
|
34
34
|
this.tipFailed = false;
|
|
35
35
|
}
|
|
36
36
|
if (e.event === "graph-rehash")
|
|
37
|
-
this.
|
|
37
|
+
this.rehashes = [...this.rehashes, { ...e, data: { from: e.data.from, to: e.data.to } }];
|
|
38
38
|
if (e.event === "tip-verify-failed" || (e.event === "tip-verify" && e.data.pass === false))
|
|
39
39
|
this.tipFailed = true;
|
|
40
40
|
if (e.event === "run-end") {
|
|
@@ -146,13 +146,9 @@ export class OperatorStateFold {
|
|
|
146
146
|
comparableTo(hash) {
|
|
147
147
|
if (!hash)
|
|
148
148
|
return false;
|
|
149
|
-
|
|
150
|
-
const
|
|
151
|
-
|
|
152
|
-
const from = baseline.comparable ? baseline.recorded : baseline.reason === "mismatch" ? baseline.recorded : null;
|
|
153
|
-
return this.latestGraphRehash.data.to === hash && this.latestGraphRehash.data.from === from;
|
|
154
|
-
}
|
|
155
|
-
return baseline.comparable;
|
|
149
|
+
// The shared comparator audits the rehash chain; the fold keeps only the rows it reads.
|
|
150
|
+
const events = [this.startEvent, ...this.rehashes].filter((e) => e !== undefined);
|
|
151
|
+
return engagementComparable(events, hash).comparable;
|
|
156
152
|
}
|
|
157
153
|
}
|
|
158
154
|
/** C1/C6 share this pure reader; callers supply the same observation and journal snapshot. */
|
package/package.json
CHANGED
|
@@ -92,7 +92,7 @@ After sending, **confirm delivery** by reading the target pane and verifying the
|
|
|
92
92
|
1. **Prepare** — confirm the target list. Run the [binary preflight](#binary-preflight-before-compile-or-run). Check `git status`, confirm no tickmarkr run is active, and work from a non-main branch.
|
|
93
93
|
2. **Compile** — run `tickmarkr compile <spec-or-directory>`. Fix source-spec defects instead of editing the generated graph.
|
|
94
94
|
3. **Plan** — run `tickmarkr plan`. Review routes, capability-floor warnings, and human gates before execution.
|
|
95
|
-
4. **Run** — run `tickmarkr run`.
|
|
95
|
+
4. **Run** — run `tickmarkr run`. A watch ending the seat's turn is no watch: keep a **blocking journal consumer** alive for the run's terminal events — the shipped watcher below, or a foreground `until grep` on the run's terminal events — and ensure it is re-armed at most every twenty minutes. Never rely on a `Monitor`-only wake. Watch the run journal rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every green clause for you. Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the relevant agent session.
|
|
96
96
|
5. **Verify and consolidate** — continue only after a green run. A run is green when the run-end event exists in the journal, the tip verify is not "failed", and the summary's `failed`, `human`, `blocked` and `pending` buckets are all empty — a run with a parked task is partial, not green. Tickmarkr consolidates accepted work on `tickmarkr/<runId>` and never signs off to the main branch. A human controls any later release merge.
|
|
97
97
|
6. **Record** — `tickmarkr report <runId> --md` prints Markdown to stdout; redirect explicitly beside the spec (for example `tickmarkr report <runId> --md > feature.record.md`) and commit the execution record when the repository tracks those records.
|
|
98
98
|
7. **Continue** — move to the next requested target. If a target fails or is parked, stop with the journal evidence rather than silently skipping it.
|
|
@@ -88,7 +88,7 @@ When spawning consultants (agents gathering synthesis input for decisions like S
|
|
|
88
88
|
1. **Prepare** — start from the requested spec. Run the [binary preflight](#binary-preflight-before-compile-or-run). Check `git status`, confirm no tickmarkr run is active, and work from a non-main branch.
|
|
89
89
|
2. **Compile** — run `tickmarkr compile <spec>`. Correct compilation errors in the spec, never in the generated graph.
|
|
90
90
|
3. **Plan** — run `tickmarkr plan`. Review the routing table, capability-floor warnings, and every human gate, including work that each gate blocks.
|
|
91
|
-
4. **Run** — run `tickmarkr run`.
|
|
91
|
+
4. **Run** — run `tickmarkr run`. A watch ending the seat's turn is no watch: keep a **blocking journal consumer** alive for the run's terminal events — the shipped watcher below, or a foreground `until grep` on the run's terminal events — and ensure it is re-armed at most every twenty minutes. Never rely on a `Monitor`-only wake. Watch the run journal rather than polling agents, using the shipped watcher — `.claude/skills/tickmarkr-overseer/scripts/watch-journal.sh <state-dir>/runs 20 28800` — which takes a line baseline at arm time, then wakes ONCE on `run-end`, `task-human`, `task-failed` or `consult-verdict` and grades the run-end summary against every green clause for you. Re-arm after every wake. ⛔ Never `tail -F | grep -m1` (run-end is the journal's last line, so tail never notices the broken pipe and the watcher hangs forever) and never a pane-level done wait (it fires on every agent turn end, not mission end). ⚠ A bare whole-file `grep -q '"event":"run-end"'` is the trap the watcher exists to avoid: on a resume it matches the PREVIOUS run's run-end and returns instantly, so a re-armed watcher reads as coverage that does not exist. Resolve blocked interactions in the agent session; do not turn them into proxy questions.
|
|
92
92
|
5. **Verify and consolidate** — accept only a green run. A run is green when the run-end event exists in the journal, the tip verify is not "failed", and the summary's `failed`, `human`, `blocked` and `pending` buckets are all empty — a run with a parked task is partial, not green. Tickmarkr consolidates accepted task work on `tickmarkr/<runId>`; it never signs off to the main branch. A human may later merge that integration branch through the repository's normal release process.
|
|
93
93
|
6. **Record** — `tickmarkr report <runId> --md` prints Markdown to stdout. Redirect it explicitly beside the source spec (for example `tickmarkr report <runId> --md > feature.record.md`) and commit the execution record when the repository tracks those records. Then [stand down](#stand-down-mission-end-and-retirement).
|
|
94
94
|
|
|
@@ -149,6 +149,14 @@ through brief lineage. **An executor choice nobody made is still an executor cho
|
|
|
149
149
|
guidance belongs in the memory file or the shipped docs.
|
|
150
150
|
4. Arm the watcher and your own supervision beat (Supervision). Report the hierarchy map (pane ids + names) to the user.
|
|
151
151
|
|
|
152
|
+
### Seat-spawn and Leg-2 recipes
|
|
153
|
+
|
|
154
|
+
Every mission to a Claude or Grok seat is delivered only with `herdr pane run <pane> "<message>"` and
|
|
155
|
+
verified by reading the pane back; never use `agent prompt` for mission delivery. Launch a Grok seat with
|
|
156
|
+
`herdr agent start <seat> --kind grok --pane <pane> -- -m grok-4.6`. For Leg-2, a Codex reviewer under
|
|
157
|
+
`workspace-write` must be briefed with an in-worktree verdict path such as
|
|
158
|
+
`<repo>/.tickmarkr/overseer/verdicts/<task>.md`, and its verdict must be written there before it is read.
|
|
159
|
+
|
|
152
160
|
## Supervising tickmarkr as the executor — WHO DOES WHAT
|
|
153
161
|
|
|
154
162
|
When the mission runs `/tickmarkr-auto` (tickmarkr dispatches the workers), supervision changes shape —
|
|
@@ -204,6 +212,9 @@ journal tail to decide what happens next, or sweeping orphans — you have taken
|
|
|
204
212
|
- **The journal is the source of truth**, not panes. Watchers go on `run-end` / `task-human` /
|
|
205
213
|
`task-failed` / `consult-verdict`; never sleep-poll inside an agent turn. **Never key a watcher on an
|
|
206
214
|
agent's `done`** — that is turn end and fires the moment a seat finishes acknowledging you.
|
|
215
|
+
A watch ending the seat's turn is no watch: keep a **blocking journal consumer** alive for those terminal
|
|
216
|
+
events — the shipped watcher below, or a foreground `until grep` on the run's terminal events — and
|
|
217
|
+
ensure it is re-armed at most every twenty minutes. Never rely on a `Monitor`-only wake.
|
|
207
218
|
**All four are covered by one shipped instrument** — `scripts/watch-journal.sh <runs-dir> [poll] [cap]
|
|
208
219
|
[events-csv]` — which arms on a line baseline, wakes once, and grades a `run-end` against every green
|
|
209
220
|
clause. `scripts/watch-parks.sh` stays the park-specific wake for THIS seat (it counts parks and speaks
|
|
@@ -476,6 +487,52 @@ number — an unmeasured budget is not a small budget.
|
|
|
476
487
|
.claude/skills/tickmarkr-overseer/scripts/watch-context.sh overseer <overseer-agent-or-pane> 50 50 <handoff-file>
|
|
477
488
|
```
|
|
478
489
|
|
|
490
|
+
### A GO has a deadline — arm `watch-launch.sh` in the same act as the GO
|
|
491
|
+
|
|
492
|
+
A GO that produces no run is a silent failure until someone notices; on 2026-09-11 an orchestrator's codex
|
|
493
|
+
sandbox was rooted at the main repo, the spec worktree was outside its writable roots, it stopped at the
|
|
494
|
+
denial without reporting, the overseer's 10-minute wake expired un-re-armed, and three hours passed.
|
|
495
|
+
Two rules close that hole:
|
|
496
|
+
|
|
497
|
+
- **Every orchestrator seat is sandbox-rooted at the worktree it will run in** (`cd <worktree>` before
|
|
498
|
+
`herdr agent start … --sandbox workspace-write`), and its brief says: *a denied path or refused command
|
|
499
|
+
is reported to the overseer pane within 60 s — never a silent stop.*
|
|
500
|
+
- **The overseer arms the launch watcher in the SAME act as the GO**, with the lock path the run will
|
|
501
|
+
create, and treats `LAUNCH_OVERDUE` as a first-class event (read the orchestrator pane, fix the seat,
|
|
502
|
+
re-issue the GO):
|
|
503
|
+
|
|
504
|
+
```bash
|
|
505
|
+
.claude/skills/tickmarkr-overseer/scripts/watch-launch.sh <worktree>/.tickmarkr/graph.lock 900 <overseer-pane> &
|
|
506
|
+
```
|
|
507
|
+
|
|
508
|
+
It prints `LAUNCH_OK` with the lock's contents when the run starts (exit 0) and, past the deadline, delivers
|
|
509
|
+
`LAUNCH OVERDUE …` to the overseer pane AND as an OS notification (exit 3). Any wake you arm yourself with
|
|
510
|
+
a cap (a background `until` loop) must be RE-ARMED on every expiry; an expired wake is not a watch.
|
|
511
|
+
|
|
512
|
+
|
|
513
|
+
### A GO has a deadline — arm `watch-launch.sh` in the same act as the GO
|
|
514
|
+
|
|
515
|
+
A GO that produces no run is a silent failure until someone notices; on 2026-09-11 an orchestrator's codex
|
|
516
|
+
sandbox was rooted at the main repo, the spec worktree was outside its writable roots, it stopped at the
|
|
517
|
+
denial without reporting, the overseer's 10-minute wake expired un-re-armed, and three hours passed.
|
|
518
|
+
Two rules close that hole:
|
|
519
|
+
|
|
520
|
+
- **Every orchestrator seat is sandbox-rooted at the worktree it will run in** (`cd <worktree>` before
|
|
521
|
+
`herdr agent start … --sandbox workspace-write`), and its brief says: *a denied path or refused command
|
|
522
|
+
is reported to the overseer pane within 60 s — never a silent stop.*
|
|
523
|
+
- **The overseer arms the launch watcher in the SAME act as the GO**, with the lock path the run will
|
|
524
|
+
create, and treats `LAUNCH_OVERDUE` as a first-class event (read the orchestrator pane, fix the seat,
|
|
525
|
+
re-issue the GO):
|
|
526
|
+
|
|
527
|
+
```bash
|
|
528
|
+
.claude/skills/tickmarkr-overseer/scripts/watch-launch.sh <worktree>/.tickmarkr/graph.lock 900 <overseer-pane> &
|
|
529
|
+
```
|
|
530
|
+
|
|
531
|
+
It prints `LAUNCH_OK` with the lock's contents when the run starts (exit 0) and, past the deadline, delivers
|
|
532
|
+
`LAUNCH OVERDUE …` to the overseer pane AND as an OS notification (exit 3). Any wake you arm yourself with
|
|
533
|
+
a cap (a background `until` loop) must be RE-ARMED on every expiry; an expired wake is not a watch.
|
|
534
|
+
|
|
535
|
+
|
|
479
536
|
The first argument chooses the closed per-seat tier (`orchestrator-context` or `overseer-context`),
|
|
480
537
|
and every beat names the second argument as that tier's seat. The watcher beats only after reading a
|
|
481
538
|
rendered percentage, keeps beating on the supervision cadence even when its requested poll is slower,
|
|
@@ -580,9 +637,9 @@ they are left implicit:
|
|
|
580
637
|
Send only when the seat is idle and the ANSI prompt line is empty or dim-only (the Esc/SGR discriminator
|
|
581
638
|
separates an autosuggest ghost from typed input), then read back activity or an ACK; presence is not
|
|
582
639
|
delivery. If a stale draft must be replaced, supersede it explicitly with
|
|
583
|
-
`
|
|
640
|
+
`herdr pane run <pane> "<-- disregard … ACTUAL: …"` instead of stacking another instruction behind it.
|
|
584
641
|
- **A MESSAGE TO A WORKING SEAT IS A QUEUED MESSAGE, AND THE QUEUE DRAINS ONLY AT TURN BOUNDARIES.**
|
|
585
|
-
Delivery is not arrival:
|
|
642
|
+
Delivery is not arrival: a message sent to a `working` claude seat lands in its queue (`Press up to
|
|
586
643
|
edit queued messages` on the seat's prompt line is the tell) and is READ only when the current turn
|
|
587
644
|
ends — and with in-process teammates a turn runs 20–40 minutes, so steering latency equals subagent
|
|
588
645
|
runtime. Measured 2026-08-17/18 (P98 leg 1): a FREEZE HOLD and a checker-release directive stacked
|
|
@@ -622,7 +679,7 @@ they are left implicit:
|
|
|
622
679
|
- **AGENT NAMES ARE GLOBAL ACROSS WORKSPACES — verify a seat you spawned by PANE ID, never by name.**
|
|
623
680
|
Names must be unique among live agents *everywhere*, not within your workspace, so another workspace can
|
|
624
681
|
already hold `opus`, `sol`, `reviewer` or `orch`. When it does, your `agent start` **fails**, your pane
|
|
625
|
-
is left a bare shell, and `agent list` / `agent read`
|
|
682
|
+
is left a bare shell, and `agent list` / `agent read` for that name then resolve to the
|
|
626
683
|
**stranger's seat**. Measured 2026-08-06 (OBS-392): a spawn of `fable` collided with a live seat in
|
|
627
684
|
another workspace; `agent list` reported `fable -> blocked` and it was read as *this* seat coming up
|
|
628
685
|
blocked. It was an operator research session sitting on a *"Resume full session?"* prompt. One more
|
|
@@ -644,7 +701,7 @@ they are left implicit:
|
|
|
644
701
|
Re-arm name-keyed watchers in the same act as the rename; file-keyed artifact watchers are
|
|
645
702
|
unaffected (one more reason to prefer them).
|
|
646
703
|
- Stale typed input is unclearable via CLI — supersede it:
|
|
647
|
-
`pane run "<-- disregard everything before this arrow (stale draft). ACTUAL: <message>"`.
|
|
704
|
+
`herdr pane run <pane> "<-- disregard everything before this arrow (stale draft). ACTUAL: <message>"`.
|
|
648
705
|
**But DISCRIMINATE before you supersede or file it: text on an idle seat's prompt line has FOUR
|
|
649
706
|
authors** — the seat's own draft, an operator, another agent's `agent send` (writes WITHOUT Enter),
|
|
650
707
|
and claude-code's AUTOSUGGEST, which renders context-plausible ghost text BYTE-IDENTICAL to a typed
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# watch-launch.sh — a GO that produced no run is a silent failure until someone notices. This watcher
|
|
3
|
+
# notices. Arm it in the SAME act as the GO (orchestrator briefed to compile → plan → run) and it waits
|
|
4
|
+
# for the run's lock; when the lock has not appeared by the deadline it delivers LAUNCH OVERDUE to the
|
|
5
|
+
# overseer's pane AND as an OS notification, so the wake reaches a seat instead of a log nobody reads.
|
|
6
|
+
#
|
|
7
|
+
# Why it exists (2026-09-11): an orchestrator's codex sandbox was rooted at the main repo, the spec
|
|
8
|
+
# worktree was outside its writable roots, it stopped at the denial without reporting, and the overseer's
|
|
9
|
+
# own 10-minute wake expired un-re-armed. Three hours passed before anyone looked. A launch has a
|
|
10
|
+
# deadline; silence past it is the event.
|
|
11
|
+
#
|
|
12
|
+
# usage: watch-launch.sh <lock-path> <deadline-s> <overseer-pane> [poll-s]
|
|
13
|
+
# <lock-path> the run's .tickmarkr/graph.lock in the worktree the run will be launched in
|
|
14
|
+
# <deadline-s> seconds from now by which the lock must exist (a compile+plan+launch takes minutes,
|
|
15
|
+
# never hours; 900 is a generous default for a 7-task spec)
|
|
16
|
+
# <overseer-pane> the pane that must hear about it (herdr pane id), e.g. wZ:p18S
|
|
17
|
+
# [poll-s] poll interval, default 15
|
|
18
|
+
# exit 0 LAUNCH_OK (lock seen; prints its contents) · exit 3 LAUNCH_OVERDUE (delivered) · exit 64 usage
|
|
19
|
+
set -u
|
|
20
|
+
LOCK="${1:-}"; DEADLINE="${2:-}"; PANE="${3:-}"; POLL="${4:-15}"
|
|
21
|
+
[ -n "$LOCK" ] && [ -n "$DEADLINE" ] && [ -n "$PANE" ] || { echo "usage: watch-launch.sh <lock-path> <deadline-s> <overseer-pane> [poll-s]" >&2; exit 64; }
|
|
22
|
+
start=$(date +%s)
|
|
23
|
+
while :; do
|
|
24
|
+
if [ -f "$LOCK" ]; then
|
|
25
|
+
printf 'LAUNCH_OK %s %s\n' "$(date -u +%H:%M:%SZ)" "$(cat "$LOCK" 2>/dev/null | tr -d '\n')"
|
|
26
|
+
exit 0
|
|
27
|
+
fi
|
|
28
|
+
now=$(date +%s)
|
|
29
|
+
if [ $((now - start)) -ge "$DEADLINE" ]; then
|
|
30
|
+
msg="LAUNCH OVERDUE $(date -u +%H:%M:%SZ): no lock at $LOCK after ${DEADLINE}s — read the orchestrator pane NOW (sandbox denial? preflight refusal? unsubmitted GO?)"
|
|
31
|
+
echo "LAUNCH_OVERDUE $msg"
|
|
32
|
+
# Both deliveries, always: a pane the overseer reads AND a notification the operator sees.
|
|
33
|
+
herdr pane run "$PANE" "$msg" >/dev/null 2>&1 || echo " (pane delivery failed — the notification is the only path)"
|
|
34
|
+
herdr notification show "$msg" >/dev/null 2>&1 || true
|
|
35
|
+
exit 3
|
|
36
|
+
fi
|
|
37
|
+
sleep "$POLL"
|
|
38
|
+
done
|