tickmarkr 1.93.0 → 1.97.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/model-lints.d.ts +1 -1
- package/dist/adapters/model-lints.js +116 -52
- package/dist/brand.d.ts +3 -0
- package/dist/brand.js +3 -1
- package/dist/cli/commands/beat.d.ts +1 -0
- package/dist/cli/commands/beat.js +50 -0
- package/dist/cli/commands/status.js +325 -127
- package/dist/cli/commands/verify.js +52 -26
- package/dist/cli/index.d.ts +1 -1
- package/dist/cli/index.js +3 -1
- package/dist/compile/gsd.js +34 -1
- package/dist/compile/native.js +26 -1
- package/dist/drivers/herdr.d.ts +26 -1
- package/dist/drivers/herdr.js +77 -29
- package/dist/gates/acceptance.js +26 -4
- package/dist/gates/baseline.d.ts +10 -1
- package/dist/gates/baseline.js +35 -6
- package/dist/run/daemon.d.ts +5 -0
- package/dist/run/daemon.js +25 -4
- package/dist/run/merge.d.ts +6 -1
- package/dist/run/merge.js +46 -11
- package/fixtures/gsd-sample/07-live-check/07-01-PLAN.md +1 -1
- package/fixtures/gsd-sample/PROJECT.md +19 -0
- package/fixtures/payload-shape/p99-shaped.spec.md +68 -0
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +42 -18
- package/skills/tickmarkr-overseer/scripts/seat-send.sh +98 -5
|
@@ -42,6 +42,13 @@ export function parseCriteria(text) {
|
|
|
42
42
|
// verify models the author as a "human" vendor channel — resolvable, excludes nothing real.
|
|
43
43
|
export const HUMAN_CHANNEL = { adapter: "human", vendor: "human", model: "human", channel: "sub", tier: "frontier" };
|
|
44
44
|
export const HUMAN_AUTHOR = { adapter: "human", model: "human", channel: "sub", tier: "frontier" };
|
|
45
|
+
// The battery's own dirty-worktree refusal, mirrored from run-gates.ts:220-232 — that check lives in
|
|
46
|
+
// a closure this command cannot reach, and verify may not reshape it. run-gates stays the authority:
|
|
47
|
+
// it re-checks at round entry and after every gate command, so a copy that ever drifted could only
|
|
48
|
+
// refuse EARLY with a stale sentence — never let a dirty tree through.
|
|
49
|
+
const DIRTY_WHY = `refusing to gate a dirty worktree: the shell gates run against the working tree while `
|
|
50
|
+
+ `evidence, scope and the merge read commits, so these uncommitted changes would be gated `
|
|
51
|
+
+ `and never merged (and the committed diff would never be run)`;
|
|
45
52
|
// GATE-FIX-4 defect 1 (false-RED on macOS): os.tmpdir() returns /var/folders/…, a symlink into
|
|
46
53
|
// /private/var — so a baseline captured under the repo path and a head battery run under the tmp
|
|
47
54
|
// path disagree on every path-bearing fingerprint, and verify reds a green diff. graph.ts's
|
|
@@ -105,6 +112,51 @@ export async function verify(argv, cwd = process.cwd()) {
|
|
|
105
112
|
evidence: { commits: [], artifacts: [], gateResults: [] },
|
|
106
113
|
};
|
|
107
114
|
const commands = detectGateCommands(cwd, cfg);
|
|
115
|
+
// PRECONDITIONS (OBS-541) — every check that can refuse this candidate, evaluated together and
|
|
116
|
+
// BEFORE the baseline capture below. Both read cheap local state (one `git status`, the doctor
|
|
117
|
+
// cache), and both used to be read AFTER the capture: the dirty tree by runGates' own round-entry
|
|
118
|
+
// refusal, the review seat by the resolution that sat under it. That cost one full capture per
|
|
119
|
+
// refusal — measured at 602s and 590s on two refusals of the same candidate — to learn something
|
|
120
|
+
// knowable in 50ms. Messages, exit taxonomy and fail-closed semantics are unchanged; only the
|
|
121
|
+
// order is. Anything else that can refuse before a gate runs belongs in this phase, above capture.
|
|
122
|
+
// `--untracked-files=all` is load-bearing twice over: it overrides a repo/user
|
|
123
|
+
// `status.showUntrackedFiles=no` (under which untracked work is INVISIBLE and a dirty tree would
|
|
124
|
+
// capture and gate GREEN), and it enumerates nested files individually instead of collapsing them
|
|
125
|
+
// to a bare `?? dir/`, so the refusal names every offending path. The `.tickmarkr-*` exemption is
|
|
126
|
+
// unaffected — the harness's droppings are root-level files, never directories.
|
|
127
|
+
const status = await shGit("GIT_OPTIONAL_LOCKS=0 git status --porcelain --untracked-files=all", cwd);
|
|
128
|
+
const dirt = status.code !== 0
|
|
129
|
+
? `git status failed (exit ${status.code}) — the worktree cannot be proven clean`
|
|
130
|
+
: status.stdout.split("\n").map((l) => l.trimEnd())
|
|
131
|
+
.filter((l) => l.trim() && !/^.. \.tickmarkr-[^/]*$/.test(l)).join("\n");
|
|
132
|
+
if (dirt)
|
|
133
|
+
throw new Error(`${DIRTY_WHY}:\n${dirt}`);
|
|
134
|
+
// LLM seats only when a semantic gate will run.
|
|
135
|
+
let channels = [];
|
|
136
|
+
let judgeChannels;
|
|
137
|
+
let author = HUMAN_AUTHOR;
|
|
138
|
+
const adapters = allAdapters();
|
|
139
|
+
if (wantAcceptance || wantReview) {
|
|
140
|
+
const health = readDoctor(cwd) ?? (await probeAll(adapters));
|
|
141
|
+
const pools = rolePools(cfg, adapters, health);
|
|
142
|
+
judgeChannels = pools.judge;
|
|
143
|
+
channels = pools.review;
|
|
144
|
+
if (values.author && values.author !== "human") {
|
|
145
|
+
const [adapter, ...rest] = values.author.split(":");
|
|
146
|
+
const model = rest.join(":");
|
|
147
|
+
const c = channels.find((ch) => ch.adapter === adapter && ch.model === model);
|
|
148
|
+
if (!c) {
|
|
149
|
+
throw new Error(`--author ${values.author} does not name a discoverable review channel — one of: ${channels.map(channelKey).join(", ") || "(none)"}`);
|
|
150
|
+
}
|
|
151
|
+
author = { adapter: c.adapter, model: c.model, channel: c.channel, tier: c.tier };
|
|
152
|
+
}
|
|
153
|
+
else {
|
|
154
|
+
channels = [...channels, HUMAN_CHANNEL];
|
|
155
|
+
}
|
|
156
|
+
if (wantReview && !channels.some((c) => c.vendor !== "human")) {
|
|
157
|
+
throw new Error("review gate needs at least one authed LLM channel (run `tickmarkr doctor`) — or pass --no-review");
|
|
158
|
+
}
|
|
159
|
+
}
|
|
108
160
|
// Baseline: --baseline file > cached capture for this merge-base > fresh capture on a detached
|
|
109
161
|
// temp worktree of the merge-base (so pre-existing failures on base are forgiven, exactly as a run).
|
|
110
162
|
// ALL verify state (cache, base worktree, artifacts) lives OUTSIDE the repo: verify gates the repo
|
|
@@ -135,32 +187,6 @@ export async function verify(argv, cwd = process.cwd()) {
|
|
|
135
187
|
await removeWorktree(cwd, baseDir);
|
|
136
188
|
}
|
|
137
189
|
}
|
|
138
|
-
// LLM seats only when a semantic gate will run.
|
|
139
|
-
let channels = [];
|
|
140
|
-
let judgeChannels;
|
|
141
|
-
let author = HUMAN_AUTHOR;
|
|
142
|
-
const adapters = allAdapters();
|
|
143
|
-
if (wantAcceptance || wantReview) {
|
|
144
|
-
const health = readDoctor(cwd) ?? (await probeAll(adapters));
|
|
145
|
-
const pools = rolePools(cfg, adapters, health);
|
|
146
|
-
judgeChannels = pools.judge;
|
|
147
|
-
channels = pools.review;
|
|
148
|
-
if (values.author && values.author !== "human") {
|
|
149
|
-
const [adapter, ...rest] = values.author.split(":");
|
|
150
|
-
const model = rest.join(":");
|
|
151
|
-
const c = channels.find((ch) => ch.adapter === adapter && ch.model === model);
|
|
152
|
-
if (!c) {
|
|
153
|
-
throw new Error(`--author ${values.author} does not name a discoverable review channel — one of: ${channels.map(channelKey).join(", ") || "(none)"}`);
|
|
154
|
-
}
|
|
155
|
-
author = { adapter: c.adapter, model: c.model, channel: c.channel, tier: c.tier };
|
|
156
|
-
}
|
|
157
|
-
else {
|
|
158
|
-
channels = [...channels, HUMAN_CHANNEL];
|
|
159
|
-
}
|
|
160
|
-
if (wantReview && !channels.some((c) => c.vendor !== "human")) {
|
|
161
|
-
throw new Error("review gate needs at least one authed LLM channel (run `tickmarkr doctor`) — or pass --no-review");
|
|
162
|
-
}
|
|
163
|
-
}
|
|
164
190
|
const artifactDir = join(stateDir, new Date().toISOString().replace(/[:.]/g, "-"));
|
|
165
191
|
mkdirSync(artifactDir, { recursive: true });
|
|
166
192
|
const { results } = await runGates(task, {
|
package/dist/cli/index.d.ts
CHANGED
|
@@ -5,7 +5,7 @@ export type CommandResult = string | {
|
|
|
5
5
|
};
|
|
6
6
|
export type CommandMap = Record<string, (argv: string[]) => Promise<CommandResult>>;
|
|
7
7
|
export declare const COMMANDS: CommandMap;
|
|
8
|
-
export declare const USAGE = "tickmarkr \u2014 spec-driven orchestration harness for AI coding agents\nusage: tickmarkr <command>\n init guided setup + doctor; init --agent [--force] [--docs] adds agent skills/docs\n doctor re-probe adapters, herdr, auth; print capability matrix (--fix writes the test-runner ignore when a safe edit exists)\n fleet interactive fleet editor (fleet --print for CI drift checks)\n compile <src> spec \u2192 .tickmarkr/graph.json (fails without acceptance criteria)\n scope <intent> draft a compiled native spec beside an answered intent (--force to overwrite)\n plan dry-run routing table + cost estimate + floor lints\n eval run checked-in fixtures against every channel in isolated temp repos\n run execute the graph (--concurrency N --driver herdr|subprocess --route-strict)\n status live run state\n verify run the gate battery standalone against merge-base(--base, HEAD)..HEAD \u2014 no daemon, one verdict (--base main --criteria <file> | --task <id> [--files <glob>] [--author adapter:model] [--no-review] [--json])\n resume <id> continue a run from its journal\n report <id> cost/quality report (--md for committable execution record)\n profile show learned routing profile (profile reset = forget history via cursor, keeps telemetry)\n ui open the Fleet Studio TUI (full-screen tabbed cockpit)\n unlock remove a stale/garbage run lock (refuses if the holder is alive)\n approve <id> <task> release a park (--uphold sides with the reviewer and funds a fixed attempt; --by <name> --reason <text>); takes effect on resume";
|
|
8
|
+
export declare const USAGE = "tickmarkr \u2014 spec-driven orchestration harness for AI coding agents\nusage: tickmarkr <command>\n init guided setup + doctor; init --agent [--force] [--docs] adds agent skills/docs\n doctor re-probe adapters, herdr, auth; print capability matrix (--fix writes the test-runner ignore when a safe edit exists)\n fleet interactive fleet editor (fleet --print for CI drift checks)\n compile <src> spec \u2192 .tickmarkr/graph.json (fails without acceptance criteria)\n scope <intent> draft a compiled native spec beside an answered intent (--force to overwrite)\n plan dry-run routing table + cost estimate + floor lints\n eval run checked-in fixtures against every channel in isolated temp repos\n run execute the graph (--concurrency N --driver herdr|subprocess --route-strict)\n status live run state\n verify run the gate battery standalone against merge-base(--base, HEAD)..HEAD \u2014 no daemon, one verdict (--base main --criteria <file> | --task <id> [--files <glob>] [--author adapter:model] [--no-review] [--json])\n resume <id> continue a run from its journal\n report <id> cost/quality report (--md for committable execution record)\n profile show learned routing profile (profile reset = forget history via cursor, keeps telemetry)\n ui open the Fleet Studio TUI (full-screen tabbed cockpit)\n unlock remove a stale/garbage run lock (refuses if the holder is alive)\n beat <tier> record one supervision beat for orchestrator|overseer|watch (--stand-down to hand off); a supervising seat's own watcher loop calls it, and status reads the tier STALE once the beats stop\n approve <id> <task> release a park (--uphold sides with the reviewer and funds a fixed attempt; --by <name> --reason <text>); takes effect on resume";
|
|
9
9
|
export declare function dispatch(cmd: string | undefined, argv: string[], commands?: CommandMap): Promise<{
|
|
10
10
|
out: string;
|
|
11
11
|
code: number;
|
package/dist/cli/index.js
CHANGED
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
import { realpathSync } from "node:fs";
|
|
3
3
|
import { pathToFileURL } from "node:url";
|
|
4
4
|
import { approve } from "./commands/approve.js";
|
|
5
|
+
import { beat } from "./commands/beat.js";
|
|
5
6
|
import { compile } from "./commands/compile.js";
|
|
6
7
|
import { doctor } from "./commands/doctor.js";
|
|
7
8
|
import { evalCommand } from "./commands/eval.js";
|
|
@@ -20,7 +21,7 @@ import { verify } from "./commands/verify.js";
|
|
|
20
21
|
import { version } from "./commands/version.js";
|
|
21
22
|
const normalize = (r) => typeof r === "string" ? { out: r, code: 0 } : r;
|
|
22
23
|
export const COMMANDS = {
|
|
23
|
-
init, doctor, fleet, compile, scope, plan, run, status, resume, report, profile, ui, unlock, approve, version, verify, eval: evalCommand,
|
|
24
|
+
init, doctor, fleet, compile, scope, plan, run, status, resume, report, profile, ui, unlock, approve, beat, version, verify, eval: evalCommand,
|
|
24
25
|
};
|
|
25
26
|
const VERSION_FLAGS = new Set(["version", "--version", "-v"]);
|
|
26
27
|
const HELP_CMDS = new Set(["help", "-h", "--help"]);
|
|
@@ -42,6 +43,7 @@ usage: tickmarkr <command>
|
|
|
42
43
|
profile show learned routing profile (profile reset = forget history via cursor, keeps telemetry)
|
|
43
44
|
ui open the Fleet Studio TUI (full-screen tabbed cockpit)
|
|
44
45
|
unlock remove a stale/garbage run lock (refuses if the holder is alive)
|
|
46
|
+
beat <tier> record one supervision beat for orchestrator|overseer|watch (--stand-down to hand off); a supervising seat's own watcher loop calls it, and status reads the tier STALE once the beats stop
|
|
45
47
|
approve <id> <task> release a park (--uphold sides with the reviewer and funds a fixed attempt; --by <name> --reason <text>); takes effect on resume`;
|
|
46
48
|
// pure, testable dispatcher: resolves a command, forwards argv, shapes the result — no side effects.
|
|
47
49
|
// unknown/missing cmd → USAGE (exit 1 if a cmd was typed, 0 for bare `tickmarkr`); a handler throw becomes
|
package/dist/compile/gsd.js
CHANGED
|
@@ -1,8 +1,10 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { spawnSync } from "node:child_process";
|
|
2
|
+
import { existsSync, readdirSync, readFileSync, realpathSync, statSync } from "node:fs";
|
|
2
3
|
import { basename, dirname, isAbsolute, join, relative } from "node:path";
|
|
3
4
|
import { parse as parseYaml } from "yaml";
|
|
4
5
|
import { AcceptanceItemSchema, TIERS, validateGraph } from "../graph/schema.js";
|
|
5
6
|
import { CompileError, assertWriteScope, inferShape, sha256 } from "./common.js";
|
|
7
|
+
import { classifyContextPath } from "./native.js";
|
|
6
8
|
// GSD artifact front-end (spec v1.3): one GSD *plan* is one tickmarkr *task* — a plan is
|
|
7
9
|
// worktree-sized; its inner <task> steps stay in the worker prompt via context[0] = the plan file.
|
|
8
10
|
// Artifact-level only: parses .planning/ markdown, never GSD repo/command internals.
|
|
@@ -213,6 +215,7 @@ function compileOne(file, storedPath) {
|
|
|
213
215
|
...(routingHints ? { routingHints } : {}),
|
|
214
216
|
},
|
|
215
217
|
content,
|
|
218
|
+
refs,
|
|
216
219
|
};
|
|
217
220
|
}
|
|
218
221
|
export function compileGsd(src, root) {
|
|
@@ -280,6 +283,36 @@ export function compileGsd(src, root) {
|
|
|
280
283
|
throw new CompileError(`${files[i]} depends on "${d}" but no such plan exists in ${src}`);
|
|
281
284
|
});
|
|
282
285
|
}
|
|
286
|
+
// Match native context reachability against the committed tree. This block only obtains the HEAD
|
|
287
|
+
// snapshot; classifyContextPath remains the single authority for missing, untracked, and glob refs.
|
|
288
|
+
const gitDir = dirname(files[0]);
|
|
289
|
+
const top = spawnSync("git", ["-C", gitDir, "rev-parse", "--show-toplevel"], { encoding: "utf8", maxBuffer: 1 << 28 });
|
|
290
|
+
const tree = spawnSync("git", ["-C", gitDir, "ls-tree", "--full-tree", "-r", "--name-only", "HEAD"], { encoding: "utf8", maxBuffer: 1 << 28 });
|
|
291
|
+
const repoRoot = typeof top.stdout === "string" ? top.stdout.trim() : "";
|
|
292
|
+
// `root` is the repository whose worktrees will consume these refs. A source nested inside some
|
|
293
|
+
// unrelated repository (as with vendored fixtures) must not be judged against that outer tree.
|
|
294
|
+
const authoritative = root === undefined
|
|
295
|
+
|| (top.status === 0 && repoRoot !== "" && realpathSync(root) === realpathSync(repoRoot));
|
|
296
|
+
if (top.status === 0 && tree.status === 0 && repoRoot && typeof tree.stdout === "string" && authoritative) {
|
|
297
|
+
const tracked = new Set(tree.stdout.split("\n").filter(Boolean));
|
|
298
|
+
const unreachable = [];
|
|
299
|
+
for (const [i, t] of tasks.entries()) {
|
|
300
|
+
for (const entry of compiled[i].refs) {
|
|
301
|
+
const { kind, suggestion } = classifyContextPath(entry, tracked, repoRoot);
|
|
302
|
+
if (kind === "untracked") {
|
|
303
|
+
console.warn(`tickmarkr: OBS-170: task ${t.id} context ${JSON.stringify(entry)} exists in your checkout but is NOT in a worker's worktree. To make it worker context: git add -f ${entry} && git commit. Staging alone is not enough.`);
|
|
304
|
+
}
|
|
305
|
+
else if (kind === "missing") {
|
|
306
|
+
unreachable.push(` ${t.id}: ${JSON.stringify(entry)}${suggestion ? `\n → did you mean ${suggestion} ?` : "\n → not found in the repository, and not a path."}`);
|
|
307
|
+
}
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
if (unreachable.length) {
|
|
311
|
+
throw new CompileError(`context: paths that do not exist in ${src}:\n${unreachable.join("\n")}\n\n` +
|
|
312
|
+
`A context: entry is a promise the worker can read it. These resolve to nothing in the repository,\n` +
|
|
313
|
+
`so the worker would be told to read a file that is not there. Fix the paths and recompile.`);
|
|
314
|
+
}
|
|
315
|
+
}
|
|
283
316
|
return validateGraph({
|
|
284
317
|
version: 1,
|
|
285
318
|
spec: { source: "gsd", paths: files, hash: sha256(compiled.map((c) => c.content).join("\n")) },
|
package/dist/compile/native.js
CHANGED
|
@@ -132,7 +132,15 @@ function strictNegativeCount(text) {
|
|
|
132
132
|
}
|
|
133
133
|
function namedCriterionPaths(text, tests) {
|
|
134
134
|
const paths = new Set();
|
|
135
|
-
|
|
135
|
+
// Longest extension first, plus a not-a-name-character tail: regex alternation is first-match-wins,
|
|
136
|
+
// so `js|json` extracted `src/i18n/ar/common.json` as `src/i18n/ar/common.js` and `ts|tsx` turned
|
|
137
|
+
// `Card.tsx` into `Card.ts`. criterion-scope is a THROWING lint, so a criterion naming a .json,
|
|
138
|
+
// .tsx or .jsx file that sits squarely inside its own files[] refused to compile, and the refusal
|
|
139
|
+
// named a path its author never wrote. Order and boundary both: either alone would fix it, and the
|
|
140
|
+
// pair makes a future extension added in the wrong place harmless (OBS-545).
|
|
141
|
+
const ext = "tsx|ts|jsx|json|js|mjs|cjs|md|txt";
|
|
142
|
+
const re = new RegExp(`(?:^|[\\s("'\`])((?:src|tests|fixtures|scripts)/[A-Za-z0-9_@{}*?.,/+-]+\\.(?:${ext}))(?![A-Za-z0-9])`, "g");
|
|
143
|
+
for (const match of text.matchAll(re)) {
|
|
136
144
|
paths.add(match[1]);
|
|
137
145
|
}
|
|
138
146
|
for (const match of text.matchAll(/\b([A-Za-z0-9_.-]+\.test\.ts)\b/g)) {
|
|
@@ -739,6 +747,23 @@ acceptance is required on every task (a nested list of observable outcomes).
|
|
|
739
747
|
hard value anywhere in the domain, the criterion asserts a universal that may be FALSE ABOUT THE
|
|
740
748
|
WORLD — bound it or say where it stops holding, rather than demanding a value that does not exist.
|
|
741
749
|
|
|
750
|
+
WHICH SIDE OF A RUN INHERITS ENVIRONMENT — AND IT DEPENDS ON THE DRIVER (OBS-542):
|
|
751
|
+
- Gate commands and "command:"/"test:" oracles INHERIT THE DAEMON'S ENVIRONMENT. They are children of
|
|
752
|
+
the daemon, so launching it as \`bash -c 'set -a; . .env.test; set +a; exec tickmarkr run'\` reaches
|
|
753
|
+
every one of them.
|
|
754
|
+
- Under the HERDR driver, a WORKER PANE runs with a FRESH AMBIENT ENVIRONMENT: none of the daemon's
|
|
755
|
+
exports reach it. It carries only what tickmarkr seeds explicitly — the run's workspace id, the
|
|
756
|
+
pane's identity, and VITEST_MAX_FORKS (the one daemon value that crosses, so a worker's suites and
|
|
757
|
+
the gate shells divide the machine the same way). A herdr worker cannot see a secret, a token, or
|
|
758
|
+
an exported PATH edit, by deliberate design.
|
|
759
|
+
- Under the SUBPROCESS driver, a worker inherits the daemon's environment wholesale (minus tickmarkr's
|
|
760
|
+
own control vars).
|
|
761
|
+
- The driver is resolved at RUN START, never at compile time, so a spec cannot know which one it gets.
|
|
762
|
+
CONSEQUENCE FOR CRITERIA: anything needing credentials, a bound port, or the network belongs in a
|
|
763
|
+
"command:"/"test:" oracle — NEVER in a worker's prose report. As prose it is satisfiable under the
|
|
764
|
+
subprocess driver and impossible under herdr, while the byte-identical command inside an oracle passes
|
|
765
|
+
under both. Measured: two burned attempts and three gate reds on one task, to learn one sentence.
|
|
766
|
+
|
|
742
767
|
ORDERING AND OWNERSHIP:
|
|
743
768
|
- Every path has exactly ONE owning task. Two tasks writing one file must be ORDERED by deps, or the
|
|
744
769
|
loser's work is silently dropped when the integration tip advances.
|
package/dist/drivers/herdr.d.ts
CHANGED
|
@@ -22,7 +22,32 @@ export declare class DeliveryCorruptedError extends Error {
|
|
|
22
22
|
export type DriverJournal = (event: string, slotName: string, data: Record<string, unknown>) => void;
|
|
23
23
|
/** First-generation join direction from measured trailer-safe floor (43-MEASUREMENT.md). */
|
|
24
24
|
export declare function workerSplitDirection(paneCols: number | null, safeFloor?: number, margin?: number): "right" | "down";
|
|
25
|
+
export declare const BOARD_TARGET_COLS = 110;
|
|
26
|
+
export declare const BOARD_SEAT_FLOOR_COLS = 40;
|
|
27
|
+
export interface BoardSplitPlan {
|
|
28
|
+
direction: "right" | "down";
|
|
29
|
+
/** herdr's split ratio is the FIRST child's share and a right split's first child is the caller,
|
|
30
|
+
* so this is what the SEAT keeps. Absent on `down` — the board then takes the full width. */
|
|
31
|
+
ratio?: number;
|
|
32
|
+
/** Columns the board gets under this plan; null when it takes the caller's whole width. */
|
|
33
|
+
boardCols: number | null;
|
|
34
|
+
}
|
|
35
|
+
/** Board-first placement beside the supervising seat: right only while the caller can fund the board
|
|
36
|
+
* its target AND leave the seat its floor; otherwise down at full width, never a squeezed board.
|
|
37
|
+
* An unmeasurable caller falls back to down like every other placement here (fail closed). */
|
|
38
|
+
export declare function boardSplitPlan(callerCols: number | null, boardCols?: number, seatFloor?: number): BoardSplitPlan;
|
|
25
39
|
export declare function taskGroupOf(name: string): string | undefined;
|
|
40
|
+
/** The operator-facing TITLE for a tab this driver creates when no stage label is supplied: a task
|
|
41
|
+
* token for a worker (`T5`, `T5↻2` on a retry) and ROLE + task for a gate pane (`REVIEW T5`) — the
|
|
42
|
+
* same vocabulary `renameGroupTab` and the gate call sites already speak. Any name with no task
|
|
43
|
+
* identity keeps today's behaviour and titles the tab after the slot itself.
|
|
44
|
+
*
|
|
45
|
+
* Load-bearing because this fallback IS reached in production: once a group's join degrades (D-09,
|
|
46
|
+
* `groupSlot`), every later member takes the per-slot path, and the durable PANE name went onto the
|
|
47
|
+
* tab — `tickmarkr:worker:T5:1:run-20260819-022723-0000000000001591`, 58 chars, four of them across
|
|
48
|
+
* one tab bar (OBS-45's class, live again 2026-08-19 on run …1591). A durable identity and a human
|
|
49
|
+
* title are different strings; `pane rename` still gets the identity, unchanged. */
|
|
50
|
+
export declare function tabLabelFor(name: string): string;
|
|
26
51
|
export declare class HerdrDriver implements ExecutorDriver {
|
|
27
52
|
private bin;
|
|
28
53
|
private workersPerTab;
|
|
@@ -89,7 +114,7 @@ export declare class HerdrDriver implements ExecutorDriver {
|
|
|
89
114
|
notify(msg: string, opts?: NotifyOpts): Promise<void>;
|
|
90
115
|
close(slot: Slot): Promise<void>;
|
|
91
116
|
private closeGrouped;
|
|
92
|
-
private
|
|
117
|
+
private ownedWatchPanes;
|
|
93
118
|
private watchSlot;
|
|
94
119
|
narrator(cwd: string, command: string, runId?: string): Promise<Slot>;
|
|
95
120
|
reconcile(desired: Set<string>, runId: string, opts?: {
|
package/dist/drivers/herdr.js
CHANGED
|
@@ -67,6 +67,22 @@ export function workerSplitDirection(paneCols, safeFloor = TRAILER_SAFE_FLOOR_CO
|
|
|
67
67
|
return "down";
|
|
68
68
|
return paneCols / 2 >= safeFloor + margin ? "right" : "down";
|
|
69
69
|
}
|
|
70
|
+
// The watch board's geometry, deliberately NOT the trailer floor above. `workerSplitDirection`
|
|
71
|
+
// halves the caller and refuses a right split under 108+2 — that bound protects WORKER panes, which
|
|
72
|
+
// print a trailer; the supervising seat + board pair does not. Applied to that pair on 2026-08-18 it
|
|
73
|
+
// sent a 189-column tab's board BELOW the seat and the operator corrected it (QUEUE-v194 criterion 1;
|
|
74
|
+
// skills/tickmarkr-overseer/SKILL.md: "the side placement outranks the halving floor"). So the board
|
|
75
|
+
// is allocated its measured width FIRST and the seat keeps the remainder.
|
|
76
|
+
export const BOARD_TARGET_COLS = 110; // §14a measured clean-render bound for the board
|
|
77
|
+
export const BOARD_SEAT_FLOOR_COLS = 40; // the seat beside it still has to be usable
|
|
78
|
+
/** Board-first placement beside the supervising seat: right only while the caller can fund the board
|
|
79
|
+
* its target AND leave the seat its floor; otherwise down at full width, never a squeezed board.
|
|
80
|
+
* An unmeasurable caller falls back to down like every other placement here (fail closed). */
|
|
81
|
+
export function boardSplitPlan(callerCols, boardCols = BOARD_TARGET_COLS, seatFloor = BOARD_SEAT_FLOOR_COLS) {
|
|
82
|
+
if (callerCols == null || callerCols < boardCols + seatFloor)
|
|
83
|
+
return { direction: "down", boardCols: null };
|
|
84
|
+
return { direction: "right", ratio: Math.round(((callerCols - boardCols) / callerCols) * 1e4) / 1e4, boardCols };
|
|
85
|
+
}
|
|
70
86
|
/** The tab a slot belongs to: its TASK — worker, judge, review and consult panes for one task share it.
|
|
71
87
|
* Returns undefined for everything else, which keeps those on the dedicated-tab path.
|
|
72
88
|
*
|
|
@@ -81,6 +97,24 @@ export function taskGroupOf(name) {
|
|
|
81
97
|
return undefined;
|
|
82
98
|
return taskId && taskId.trim() ? taskId : undefined;
|
|
83
99
|
}
|
|
100
|
+
/** The operator-facing TITLE for a tab this driver creates when no stage label is supplied: a task
|
|
101
|
+
* token for a worker (`T5`, `T5↻2` on a retry) and ROLE + task for a gate pane (`REVIEW T5`) — the
|
|
102
|
+
* same vocabulary `renameGroupTab` and the gate call sites already speak. Any name with no task
|
|
103
|
+
* identity keeps today's behaviour and titles the tab after the slot itself.
|
|
104
|
+
*
|
|
105
|
+
* Load-bearing because this fallback IS reached in production: once a group's join degrades (D-09,
|
|
106
|
+
* `groupSlot`), every later member takes the per-slot path, and the durable PANE name went onto the
|
|
107
|
+
* tab — `tickmarkr:worker:T5:1:run-20260819-022723-0000000000001591`, 58 chars, four of them across
|
|
108
|
+
* one tab bar (OBS-45's class, live again 2026-08-19 on run …1591). A durable identity and a human
|
|
109
|
+
* title are different strings; `pane rename` still gets the identity, unchanged. */
|
|
110
|
+
export function tabLabelFor(name) {
|
|
111
|
+
const { role, taskId, attempt } = canonicalizeLegacyName(name, "");
|
|
112
|
+
if (!TASK_TAB_ROLES.has(role) || !taskId.trim())
|
|
113
|
+
return name;
|
|
114
|
+
if (role !== "worker")
|
|
115
|
+
return `${role.toUpperCase()} ${taskId}`;
|
|
116
|
+
return attempt > 0 ? `${taskId}↻${attempt}` : taskId;
|
|
117
|
+
}
|
|
84
118
|
/** Gate panes ride with the task they belong to and never consume the tab cap; everything else does.
|
|
85
119
|
* Scoped to the three GATE roles deliberately — an earlier cut of this said "not a worker", which let
|
|
86
120
|
* role:"other" members (any unrecognised name) bypass the cap and silently disabled overflow for
|
|
@@ -303,7 +337,7 @@ export class HerdrDriver {
|
|
|
303
337
|
// label defaults to the slot name; group tabs pass the STAGE name instead — a first-member label
|
|
304
338
|
// outlives its member once keepPanes reaps it (run-20260709-104447: the codex pane sat in a tab
|
|
305
339
|
// named after a dead cursor worker and the operator read it as a mislabeled agent)
|
|
306
|
-
async tabSlot(cwd, name, label = name) {
|
|
340
|
+
async tabSlot(cwd, name, label = tabLabelFor(name)) {
|
|
307
341
|
// tab-per-slot: concurrent agents in one tab split it into sliver columns — TUIs exit or
|
|
308
342
|
// hard-wrap at COLUMNS≈2, shredding even the TICKMARKR_RESULT marker (v1.4 phase-1 incident).
|
|
309
343
|
// A dedicated named tab gives every agent a full-width pane; tab close() reaps it.
|
|
@@ -400,7 +434,17 @@ export class HerdrDriver {
|
|
|
400
434
|
const joined = await this.joinGroup(cwd, name, group, latest);
|
|
401
435
|
if (joined)
|
|
402
436
|
return joined;
|
|
403
|
-
|
|
437
|
+
// D-09 fail-safe: this and future members degrade to per-slot tabs. Best-effort notify, on
|
|
438
|
+
// the relabel-failure precedent below: the latch is PERMANENT for the group and was the one
|
|
439
|
+
// layout decision this driver made with no record anywhere — an operator reading a scattered
|
|
440
|
+
// tab bar had no way to learn a split had failed once, hours earlier.
|
|
441
|
+
state.splitUnsupported = true;
|
|
442
|
+
try {
|
|
443
|
+
await this.notify(`tickmarkr tab grouping degraded: ${group} can no longer join by split — every later member opens its own tab`);
|
|
444
|
+
}
|
|
445
|
+
catch {
|
|
446
|
+
/* cosmetic only — never blocks membership or teardown (v1.18 invariant) */
|
|
447
|
+
}
|
|
404
448
|
return this.tabSlot(cwd, name);
|
|
405
449
|
}
|
|
406
450
|
return this.newGeneration(cwd, name, group, state); // cap full → overflow to a new generation tab
|
|
@@ -991,7 +1035,11 @@ export class HerdrDriver {
|
|
|
991
1035
|
this.groups.delete(slot.group); // group dies when all generations gone
|
|
992
1036
|
}
|
|
993
1037
|
}
|
|
994
|
-
|
|
1038
|
+
// Every surviving tickmarkr-owned board in this workspace — a PRIOR run's and one already wearing
|
|
1039
|
+
// this run's own name alike. Both are retired before a new board opens (narrator): what a pane this
|
|
1040
|
+
// process did not create is actually RUNNING cannot be read back, and the pre-v1.94 implementation
|
|
1041
|
+
// launched a bare `tickmarkr status --watch`, which follows the newest journal.
|
|
1042
|
+
async ownedWatchPanes() {
|
|
995
1043
|
if (!this.ws)
|
|
996
1044
|
throw new Error("herdr watch placement requires HERDR_WORKSPACE_ID — refusing unseeded pane");
|
|
997
1045
|
const list = await this.herdr("pane list");
|
|
@@ -1006,20 +1054,24 @@ export class HerdrDriver {
|
|
|
1006
1054
|
}
|
|
1007
1055
|
if (!Array.isArray(panes))
|
|
1008
1056
|
throw new Error(`herdr pane list returned no panes: ${list.stdout}`);
|
|
1009
|
-
|
|
1057
|
+
return panes.filter((p) => {
|
|
1010
1058
|
const owned = typeof p.label === "string" ? parseOwnedName(p.label) : null;
|
|
1011
|
-
return p.workspace_id === this.ws && typeof p.pane_id === "string" && owned?.role === "watch" && owned.taskId === "run"
|
|
1012
|
-
});
|
|
1013
|
-
return prior?.pane_id ?? null;
|
|
1059
|
+
return p.workspace_id === this.ws && typeof p.pane_id === "string" && owned?.role === "watch" && owned.taskId === "run";
|
|
1060
|
+
}).map((p) => p.pane_id);
|
|
1014
1061
|
}
|
|
1015
|
-
// T2: the watch is a
|
|
1016
|
-
//
|
|
1062
|
+
// T2: the watch is a sibling of the daemon's own pane, never a separate tab — beside it when the
|
|
1063
|
+
// tab can fund the board its width, below it at full width when it cannot. Its durable owned name
|
|
1064
|
+
// is how a later daemon RECOGNIZES the board it must retire, so a run never stacks a second one.
|
|
1017
1065
|
async watchSlot(cwd, name) {
|
|
1018
1066
|
if (!this.ws)
|
|
1019
1067
|
throw new Error("herdr watch placement requires HERDR_WORKSPACE_ID — refusing unseeded pane");
|
|
1020
1068
|
if (!this.callerPane)
|
|
1021
1069
|
throw new Error("herdr watch placement requires HERDR_PANE_ID — refusing untargeted split");
|
|
1022
|
-
|
|
1070
|
+
// Board width first (boardSplitPlan), measured off the caller through the driver's own layout
|
|
1071
|
+
// read — never an unconditional right split, and never the worker halving rule.
|
|
1072
|
+
const plan = boardSplitPlan(await this.paneWidth(this.callerPane));
|
|
1073
|
+
const ratio = plan.ratio == null ? "" : ` --ratio ${plan.ratio}`;
|
|
1074
|
+
const sp = await this.herdr(`pane split ${shq(this.callerPane)} --direction ${plan.direction}${ratio} --no-focus`);
|
|
1023
1075
|
if (sp.code !== 0)
|
|
1024
1076
|
throw new Error(`herdr watch split failed: ${sp.stderr || sp.stdout}`);
|
|
1025
1077
|
let pane;
|
|
@@ -1043,31 +1095,27 @@ export class HerdrDriver {
|
|
|
1043
1095
|
}
|
|
1044
1096
|
return { id: pane, name, cwd };
|
|
1045
1097
|
}
|
|
1046
|
-
// T6 narrator: the run's single live status surface
|
|
1047
|
-
// a
|
|
1048
|
-
//
|
|
1049
|
-
//
|
|
1098
|
+
// T6 narrator: the run's single live status surface, RUNNING THE COMMAND THIS CALL SUPPLIED. Only
|
|
1099
|
+
// a board this driver instance itself opened is reused (this.watches); any other surviving board —
|
|
1100
|
+
// a prior run's, or one already carrying this run's canonical name after a resume — is retired and
|
|
1101
|
+
// re-split, because adoption cannot restart or even read the process inside it and a pre-v1.94 pane
|
|
1102
|
+
// is running the bare `tickmarkr status --watch`, which narrates the newest journal instead of this
|
|
1103
|
+
// run. The retirement is VERIFIED gone before the replacement splits: reconcile is no backstop here
|
|
1104
|
+
// (panesToClose skips role "watch" by design, types.ts:92), so an unverified close would leave two
|
|
1105
|
+
// boards bound to different runs. Failures propagate — the daemon swallows.
|
|
1050
1106
|
async narrator(cwd, command, runId) {
|
|
1051
1107
|
const name = runId ? formatOwnedName({ role: "watch", taskId: "run", attempt: 0, runId }) : `narrator-watch-${process.pid}`;
|
|
1052
1108
|
return this.serial(async () => {
|
|
1053
1109
|
const cached = this.watches.get(name);
|
|
1054
1110
|
if (cached)
|
|
1055
1111
|
return cached;
|
|
1056
|
-
const
|
|
1057
|
-
|
|
1058
|
-
|
|
1059
|
-
|
|
1060
|
-
|
|
1061
|
-
|
|
1062
|
-
|
|
1063
|
-
if (prior) {
|
|
1064
|
-
const renamed = await this.herdr(`pane rename ${shq(prior)} ${shq(name)}`);
|
|
1065
|
-
if (renamed.code !== 0 || await this.namedPaneId(name) !== prior) {
|
|
1066
|
-
throw new Error(`herdr watch reclaim failed: ${renamed.stderr || renamed.stdout}`);
|
|
1067
|
-
}
|
|
1068
|
-
const s = { id: prior, name, cwd };
|
|
1069
|
-
this.watches.set(name, s);
|
|
1070
|
-
return s;
|
|
1112
|
+
const stale = await this.ownedWatchPanes();
|
|
1113
|
+
for (const pane of stale)
|
|
1114
|
+
await this.herdr(`pane close ${shq(pane)}`);
|
|
1115
|
+
if (stale.length) {
|
|
1116
|
+
const survived = (await this.ownedWatchPanes()).filter((p) => stale.includes(p));
|
|
1117
|
+
if (survived.length)
|
|
1118
|
+
throw new Error(`herdr watch retire failed: ${survived.join(", ")} survived close — refusing a second board`);
|
|
1071
1119
|
}
|
|
1072
1120
|
const s = await this.watchSlot(cwd, name);
|
|
1073
1121
|
this.watches.set(name, s);
|
package/dist/gates/acceptance.js
CHANGED
|
@@ -10,6 +10,7 @@ import { checkTaskDiffCaps, fetchTaskDiff, isProtectedEvidence, setAsideReceiptP
|
|
|
10
10
|
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
|
|
11
11
|
import { classifyVerdictCause } from "./verdict-cause.js";
|
|
12
12
|
import { reviewableLogicDiff } from "./artifact-manifest.js";
|
|
13
|
+
import { classifyFailureOutput } from "./baseline.js";
|
|
13
14
|
// Fable F4: acceptance judge shares review's 900s timeout — 300s default killed frontier judges on cap-sized diffs.
|
|
14
15
|
const JUDGE_TIMEOUT_MS = 900_000;
|
|
15
16
|
const CitationSchema = z.object({ path: z.string(), line: z.number().int() });
|
|
@@ -177,6 +178,29 @@ function tail(out, n = 8) {
|
|
|
177
178
|
return "";
|
|
178
179
|
return "\n" + t.split("\n").slice(-n).join("\n");
|
|
179
180
|
}
|
|
181
|
+
/**
|
|
182
|
+
* OBS-540: classify only bytes produced by the deterministic oracle process. A nonzero command with
|
|
183
|
+
* an infra-only output shape returned no product verdict, so it remains fail-closed but cannot be
|
|
184
|
+
* billed as a diff failure. Judge reasons never reach this helper: the judge path starts after the
|
|
185
|
+
* deterministic loop and a parsed refusal is an evaluated quality verdict whatever its prose says.
|
|
186
|
+
*/
|
|
187
|
+
function oracleExecutionFailure(label, code, stdout, stderr) {
|
|
188
|
+
const output = [stderr, stdout].filter((part) => part.length > 0).join("\n");
|
|
189
|
+
const classification = classifyFailureOutput(output);
|
|
190
|
+
if (classification === "infra") {
|
|
191
|
+
return {
|
|
192
|
+
gate: "acceptance",
|
|
193
|
+
pass: false,
|
|
194
|
+
details: `oracle failed: ${label} (exit ${code}) — infrastructure blocked execution before a verdict; this oracle verified nothing${tail(output)}`,
|
|
195
|
+
meta: { cause: "oracle-execution", classification, infra: true, retryable: false },
|
|
196
|
+
};
|
|
197
|
+
}
|
|
198
|
+
return {
|
|
199
|
+
gate: "acceptance",
|
|
200
|
+
pass: false,
|
|
201
|
+
details: `oracle failed: ${label} (exit ${code})${tail(stderr || stdout)}`,
|
|
202
|
+
};
|
|
203
|
+
}
|
|
180
204
|
// v1.70 / OBS-129: the new-file line numbers each changed HUNK spans, per path — parsed from the
|
|
181
205
|
// unified-diff so a citation is validated against real changed regions, not a substring of the whole
|
|
182
206
|
// diff text (which also matches the diff's own +++/@@ headers). A hunk's span is every new-file line it
|
|
@@ -289,8 +313,7 @@ export async function acceptanceGate(task, worktree, baseRef, judge, via, opts =
|
|
|
289
313
|
if (isCommand(a)) {
|
|
290
314
|
const r = await sh(a.command, worktree);
|
|
291
315
|
if (r.code !== 0) {
|
|
292
|
-
return {
|
|
293
|
-
details: `oracle failed: $ ${a.command} (exit ${r.code})${tail(r.stderr || r.stdout)}` };
|
|
316
|
+
return oracleExecutionFailure(`$ ${a.command}`, r.code, r.stdout, r.stderr);
|
|
294
317
|
}
|
|
295
318
|
passedDet.push(`✓ $ ${a.command} (exit 0)`);
|
|
296
319
|
}
|
|
@@ -302,8 +325,7 @@ export async function acceptanceGate(task, worktree, baseRef, judge, via, opts =
|
|
|
302
325
|
const r = await sh(testFiltered(opts.testCmd, a.test), worktree);
|
|
303
326
|
const out = (r.stderr || "") + "\n" + (r.stdout || "");
|
|
304
327
|
if (r.code !== 0) {
|
|
305
|
-
return
|
|
306
|
-
details: `oracle failed: test "${a.test}" (exit ${r.code})${tail(r.stderr || r.stdout)}` };
|
|
328
|
+
return oracleExecutionFailure(`test "${a.test}"`, r.code, r.stdout, r.stderr);
|
|
307
329
|
}
|
|
308
330
|
// OBS-55: exit 0 alone is vacuous when the name filter matched zero tests — fail closed.
|
|
309
331
|
const ran = testsRan(out);
|
package/dist/gates/baseline.d.ts
CHANGED
|
@@ -3,13 +3,22 @@ import type { AcceptanceItem } from "../graph/schema.js";
|
|
|
3
3
|
import { type ShResult } from "../run/git.js";
|
|
4
4
|
import type { GateResult } from "./types.js";
|
|
5
5
|
export interface BaselineCommand {
|
|
6
|
-
|
|
6
|
+
/**
|
|
7
|
+
* Absent when the capture returned no verdict — see `infra`. A pre-v1.90 baseline can also lack it
|
|
8
|
+
* (legacy entries read as red-at-baseline, the `?? 1` default both readers share).
|
|
9
|
+
*/
|
|
10
|
+
exitCode?: number;
|
|
7
11
|
fingerprints: string[];
|
|
8
12
|
missingCommand?: boolean;
|
|
9
13
|
/** What this command actually took at capture, on a pristine tree. Absent in pre-v1.90 baselines. */
|
|
10
14
|
durationMs?: number;
|
|
11
15
|
/** The ceiling that measurement implies, persisted so every later battery uses the same number. */
|
|
12
16
|
ceilingMs?: number;
|
|
17
|
+
/**
|
|
18
|
+
* OBS-534 (T2): the capture was SIGKILLed at its ceiling. It never finished asking the question, so
|
|
19
|
+
* the entry carries a CAUSE and no verdict: no exit code, no fingerprints, nothing forgivable.
|
|
20
|
+
*/
|
|
21
|
+
infra?: true;
|
|
13
22
|
}
|
|
14
23
|
export interface Baseline {
|
|
15
24
|
commands: Record<string, BaselineCommand>;
|
package/dist/gates/baseline.js
CHANGED
|
@@ -107,7 +107,11 @@ const VOCAB_RE = /\b(?:error|fail(?:ed|ure|ing)?)\b/i;
|
|
|
107
107
|
// test-level failure — "AssertionError after spawn EAGAIN" names one and is a regression; "spawn
|
|
108
108
|
// EAGAIN" and "Error: spawn EAGAIN" name none and are infra. One regression line anywhere in the
|
|
109
109
|
// output makes the whole output a regression, whatever else the runner printed.
|
|
110
|
-
|
|
110
|
+
// OBS-540: command-oracle startup failures are execution evidence too. These two Playwright/keyring
|
|
111
|
+
// shapes are emitted by the process that was asked to run the oracle; they are deliberately kept in
|
|
112
|
+
// this runner-output classifier rather than applied to any judge-authored reason text. A real test
|
|
113
|
+
// failure still dominates below because one regression-shaped line makes the whole output regression.
|
|
114
|
+
const INFRA_RE = /\bE(?:AGAIN|MFILE|NFILE|NOMEM|NOSPC)\b|JavaScript heap out of memory|Cannot allocate memory|Resource temporarily unavailable|Token not found in system keyring|Process from config\.webServer was not able to start/i;
|
|
111
115
|
// A named error CLASS ("AssertionError", "TypeError", "MyDomainError") — never bare "Error", which
|
|
112
116
|
// is what an errno report itself is headed with (`Error: spawn EAGAIN`). The prefix is required.
|
|
113
117
|
const ERROR_CLASS_RE = /\b[A-Za-z][A-Za-z0-9]*Error\b/;
|
|
@@ -170,7 +174,11 @@ const renormalize = (fp) => normalizeLine(fp.replace(ANSI_RE, ""));
|
|
|
170
174
|
* whether the output carried no recognizable failure shape at all.
|
|
171
175
|
*/
|
|
172
176
|
export function freshFailures(entry, raw) {
|
|
173
|
-
|
|
177
|
+
// OBS-534 (T2): an infra-recorded entry is a KILL, not a verdict — whatever fingerprints it carries
|
|
178
|
+
// came from output flushed before the kill, over a suite that never finished. Nothing there is
|
|
179
|
+
// "pre-existing", so it forgives nothing. The rule lives here rather than in either caller, so the
|
|
180
|
+
// battery and tip verify inherit it from the one helper they already share.
|
|
181
|
+
const known = new Set(entry?.infra === true ? [] : (entry?.fingerprints ?? []).map(renormalize));
|
|
174
182
|
// OBS-42: diagnostic headings enrich fingerprints but cannot invalidate legacy baselines.
|
|
175
183
|
const current = fingerprint(raw);
|
|
176
184
|
const fresh = current.filter((f) => !known.has(f) && (!FAIL_ANCHOR_RE.test(f) || f.startsWith("FAIL ")));
|
|
@@ -318,6 +326,19 @@ export async function captureBaseline(cwd, commands) {
|
|
|
318
326
|
// ponytail: a capture that was itself killed records the ceiling as its "measurement", which
|
|
319
327
|
// scales the next ceiling up — the right direction for a suite that never finished once.
|
|
320
328
|
const durationMs = r.durationMs ?? 0;
|
|
329
|
+
// OBS-534 (T2): a capture SIGKILLed at its ceiling never returned a verdict, so `r.code` is the
|
|
330
|
+
// kill and not evidence about the command. Run 1501 recorded `test: {durationMs: 600007,
|
|
331
|
+
// exitCode: 1}` for exactly this — a kill written down as a red baseline. There the accident
|
|
332
|
+
// helped (the inflated ceiling is why every task gate passed); the same accident can mark a
|
|
333
|
+
// GREEN command permanently red-at-baseline and hand every later gate free forgiveness for
|
|
334
|
+
// failures the diff really did cause. So record the cause and nothing forgivable: no exit code,
|
|
335
|
+
// and none of the partial output the runner had flushed before the kill. The measurement stays —
|
|
336
|
+
// it is the one thing the kill did establish — and still scales the next ceiling up.
|
|
337
|
+
// `timedOut` is set in exactly one place (git.ts's kill timer), so no ordinary exit reaches here.
|
|
338
|
+
if (r.timedOut === true) {
|
|
339
|
+
base.commands[name] = { infra: true, fingerprints: [], durationMs, ceilingMs: effectiveCeilingMs({ durationMs }) };
|
|
340
|
+
continue;
|
|
341
|
+
}
|
|
321
342
|
base.commands[name] = {
|
|
322
343
|
exitCode: r.code,
|
|
323
344
|
// a command that exits 0 has no failures to fingerprint — recording any would be a lie the
|
|
@@ -393,7 +414,8 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
|
393
414
|
results.push({ gate: name, pass: true, details: `no ${name} command detected — skipped`, meta: { skipped: true } });
|
|
394
415
|
continue;
|
|
395
416
|
}
|
|
396
|
-
const
|
|
417
|
+
const entry = baseline.commands[name];
|
|
418
|
+
const ceilingMs = effectiveCeilingMs(entry);
|
|
397
419
|
const r = await sh(cmd, cwd, ceilingMs);
|
|
398
420
|
// Q24: the kill is read BEFORE the exit code is interpreted at all. A SIGKILLed battery has
|
|
399
421
|
// whatever partial output it had flushed — typically no failure shape — so every path below
|
|
@@ -431,9 +453,16 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
|
431
453
|
// below rather than reading as a verified green. Raise the ceiling by teaching isFailureShaped
|
|
432
454
|
// that runner's position rule (leading verdict + identifier, or identifier + separator + trailing
|
|
433
455
|
// verdict); loosening back to vocabulary re-opens OBS-278.
|
|
434
|
-
const { failing, unreadable } = freshFailures(
|
|
435
|
-
|
|
436
|
-
|
|
456
|
+
const { failing, unreadable } = freshFailures(entry, raw);
|
|
457
|
+
// OBS-534 (T2): only a recorded VERDICT can be forgiven. A green baseline has no red to forgive,
|
|
458
|
+
// and neither has a capture that was killed at its ceiling — it recorded a cause instead, so it
|
|
459
|
+
// fails closed on the same branch rather than reading as "only pre-existing failures". Legacy
|
|
460
|
+
// entries with no exitCode keep the `?? 1` red default both readers share (merge.ts:131).
|
|
461
|
+
const baselineRed = entry?.infra !== true && (entry?.exitCode ?? 1) !== 0;
|
|
462
|
+
if (!failing.length && !baselineRed) {
|
|
463
|
+
const closed = entry?.infra === true
|
|
464
|
+
? `the baseline capture for this command was killed at its ceiling and recorded no verdict, so nothing here is forgivable — it now exits ${r.code} with no recognizable failure lines — failing closed`
|
|
465
|
+
: `command was green at baseline but now exits ${r.code} with no recognizable failure lines — failing closed`;
|
|
437
466
|
const evidence = unrecognizedEvidence(raw);
|
|
438
467
|
results.push({
|
|
439
468
|
gate: name,
|
package/dist/run/daemon.d.ts
CHANGED
|
@@ -58,6 +58,11 @@ export interface RunSummary {
|
|
|
58
58
|
*/
|
|
59
59
|
export declare function outstandingApprovals(events: JournalEvent[]): string[];
|
|
60
60
|
export declare function formatSummary(s: RunSummary): string;
|
|
61
|
+
/** The narrator's command, bound to THIS run. `status` takes exactly one positional and it is the
|
|
62
|
+
* run id (cli/commands/status.ts positionalRunId), so naming it here is what stops the board from
|
|
63
|
+
* following the newest journal in a repo that already carries a second, newer run — a board showing
|
|
64
|
+
* the wrong run is a recorded incident (skills/tickmarkr-overseer/SKILL.md). */
|
|
65
|
+
export declare const watchCommand: (runId: string) => string;
|
|
61
66
|
/**
|
|
62
67
|
* R3 (OBS-186): a gate that DECLINED to run is not a gate that failed. The review gate's skip branch
|
|
63
68
|
* no longer forges `pass: true` to buy passage, so the merge decision has to read the same predicate
|