tickmarkr 1.90.5 → 1.90.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -51,11 +51,53 @@ const TRAILING_FAIL_RE = /^\s*(?:test\s+)?\S+(?:\s+\([^)]*\))?\s+(?:\.{3}|-{3,})
51
51
  // runner sharing the glyph protocol is read without being enumerated; a status strip drawing ✖
52
52
  // mid-line inside chrome matches nothing, the same position rule as the other shapes.
53
53
  const GLYPH_FAIL_RE = /^\s*(?:(?:[\w@./-]+:\s*)*)✖\s+\S/;
54
+ // vitest's default reporter draws the same glyph protocol with × (U+00D7, verbatim capture,
55
+ // GATE-FIX-4 drill: `intake-backend:test: × GF4BE deliberate failure — probe only`). It is a
56
+ // fingerprintable SHAPE (isFailureShaped) but deliberately NOT in namesFailure: vitest prints a
57
+ // FAIL twin naming the same test, and the HYG-08 meta.failingTests pin (tests/gates/baseline.test.ts)
58
+ // keeps headline naming to one line per failure — duplicating each failure as glyph + FAIL would
59
+ // re-headline the noise HYG-08 removed, while dropping the × from the fingerprint set would shrink
60
+ // the per-test discriminators DEFECT 4 exists to harvest.
61
+ const X_GLYPH_FAIL_RE = /^\s*(?:(?:[\w@./-]+:\s*)*)×\s+\S/;
62
+ // Q137s (verbatim capture, dossier run-2 consult dossier, turbo 2.9.14 + pnpm 10): monorepo
63
+ // drivers prefix child output and summarize failures in their own grammar. Four shapes, all
64
+ // positional — `<pkg>:<task>:` prefixes reuse GLYPH_FAIL_RE's prefix idiom; `Failed:`/`ERROR`
65
+ // require the driver's own `pkg#task` identifier or its literal terminus, so prose or drawn
66
+ // chrome containing the words matches nothing (OBS-278 discipline):
67
+ // `intake-frontend:lint: ELIFECYCLE Command failed with exit code 1.`
68
+ // `Failed: intake-frontend#lint`
69
+ // ` ERROR intake-frontend#lint: command (…) … exited (1)`
70
+ // ` ERROR run failed: command exited (1)`
71
+ const TURBO_FAIL_RE = /^\s*(?:[\w@./-]+:\s*)*ELIFECYCLE\s+Command failed\b|^\s*Failed:\s+\S+#\S+|^\s*ERROR\s+(?:\S+#\S+:|run failed\b)/;
72
+ // GATE-FIX-4 DEFECT 4 (dossier drill, control 3): turbo prefixes every child line with
73
+ // `<pkg>:<task>: `, so the per-test shapes above (FAIL_ANCHOR_RE etc.) — all anchored at line
74
+ // START — never fingerprinted; forgiveness compared only TURBO_FAIL_RE's package-level lines, and a
75
+ // package with one tolerated pre-existing red was blind to every NEW failure inside it (a deliberate
76
+ // failing test came back green). The counts route is closed too: normalizeLine masks digits, so
77
+ // `Tests 3 failed` vs `Tests 2 failed` is not a discriminator — per-test lines, which carry file
78
+ // paths and test names, are. This prefix is deliberately narrower than TURBO_FAIL_RE's label idiom:
79
+ // at least TWO colon-joined segments (`<pkg>:<task>:`, tasks may nest — `pkg:test:unit:`), every
80
+ // segment after the first starting with a letter — so `Error: boom` (one segment), `src/x.ts:12:`
81
+ // (digit segment) and `12:34 error` (eslint stylish) are never stripped.
82
+ const TURBO_PREFIX_RE = /^\s*[\w@./-]+(?::[A-Za-z_][\w.-]*)+:\s+/;
83
+ const stripTurboPrefix = (l) => {
84
+ const m = TURBO_PREFIX_RE.exec(l);
85
+ return m ? l.slice(m[0].length) : undefined;
86
+ };
54
87
  // Lines that NAME a failing test — the ones worth headlining to the operator. One list, so recognition
55
88
  // and reporting cannot drift apart (a shape that blocks but never gets named cost 3 attempts once).
56
- const namesFailure = (l) => FAIL_ANCHOR_RE.test(l) || RUNNER_FAIL_RE.test(l) || TRAILING_FAIL_RE.test(l) || GLYPH_FAIL_RE.test(l);
89
+ const namesFailure = (l) => FAIL_ANCHOR_RE.test(l) || RUNNER_FAIL_RE.test(l) || TRAILING_FAIL_RE.test(l) || GLYPH_FAIL_RE.test(l) || TURBO_FAIL_RE.test(l);
90
+ // The stripped form is a second READ of the same line, for the recognition/headline paths that ask
91
+ // "does anything here name a failure" — verdict classification (isInfraLine/namesRegression) keeps
92
+ // reading the raw line only, so infra/regression verdicts are byte-unchanged by the prefix pass.
93
+ const namesFailureEitherForm = (l) => {
94
+ if (namesFailure(l))
95
+ return true;
96
+ const stripped = stripTurboPrefix(l);
97
+ return stripped !== undefined && namesFailure(stripped);
98
+ };
57
99
  const isFailureShaped = (l) => namesFailure(l) || SUMMARY_FAIL_RE.test(l) || ERROR_ANCHOR_RE.test(l)
58
- || TSC_ERROR_RE.test(l) || LINTER_ERROR_RE.test(l);
100
+ || TSC_ERROR_RE.test(l) || LINTER_ERROR_RE.test(l) || X_GLYPH_FAIL_RE.test(l);
59
101
  const VOCAB_RE = /\b(?:error|fail(?:ed|ure|ing)?)\b/i;
60
102
  // T9 — the infra/regression discriminator. A runner that died because the MACHINE ran out of
61
103
  // processes, file descriptors or memory never finished asking the question, so its nonzero exit is
@@ -95,7 +137,18 @@ export function fingerprint(output) {
95
137
  .split("\n")
96
138
  .map((l) => l.replace(ANSI_RE, ""))
97
139
  .filter((l) => !PASS_LINE_RE.test(l));
98
- const shaped = lines.filter(isFailureShaped);
140
+ // GATE-FIX-4 DEFECT 4: every line is read twice — as printed, and with a turbo `<pkg>:<task>:`
141
+ // prefix removed. A recognized stripped line fingerprints as its STRIPPED text, so the same
142
+ // failure fingerprints identically whether turbo prefixed it or a bare runner printed it; the
143
+ // prefixed form keeps fingerprinting too (baseline-recorded package-level reds stay forgivable).
144
+ const shaped = [];
145
+ for (const l of lines) {
146
+ if (isFailureShaped(l))
147
+ shaped.push(l);
148
+ const stripped = stripTurboPrefix(l);
149
+ if (stripped !== undefined && !PASS_LINE_RE.test(stripped) && isFailureShaped(stripped))
150
+ shaped.push(stripped);
151
+ }
99
152
  if (!shaped.length)
100
153
  return lines.some((l) => l.trim()) ? [UNRECOGNIZED_FAILURE] : [];
101
154
  return [...new Set(shaped.map(normalizeLine))];
@@ -170,6 +223,36 @@ export function detectGateCommands(repoRoot, cfg) {
170
223
  }
171
224
  return out;
172
225
  }
226
+ /**
227
+ * Dossier GATE-FIX-4 §3 (2026-08-13): turbo schedules per-package tasks and ABORTS the whole
228
+ * run at the first failing package, so every package after the abort goes unverified — and the
229
+ * baseline gate then forgives the truncated output as "matching". The class appeared in three
230
+ * gates and four syntaxes on one repo (turbo's scheduler, `&&`, a JS for-loop process.exit, and
231
+ * turbo nested inside two of them). This preflight names only what it can see textually: the
232
+ * resolved gate command, and — for the synthesized `<pm> run <script>` form — that script's own
233
+ * body. The remedy deliberately demands a forwarding check first: the same repo's lint wrapper
234
+ * branched on argv.length, so a blind `-- --continue` silently replaced the gate with a
235
+ * different command (the trap their overseer caught by reading, not by symmetry).
236
+ */
237
+ export function turboContinueFindings(repoRoot, cfg) {
238
+ const pkgPath = join(repoRoot, "package.json");
239
+ const scripts = existsSync(pkgPath)
240
+ ? (JSON.parse(readFileSync(pkgPath, "utf8")).scripts ?? {})
241
+ : {};
242
+ const findings = [];
243
+ for (const [gate, cmd] of Object.entries(detectGateCommands(repoRoot, cfg))) {
244
+ // The command text plus every package.json script it invokes by name — one level, textual.
245
+ const referenced = Object.keys(scripts).filter((name) => new RegExp(`\\brun\\s+(?:-s\\s+)?${reEscape(name)}(?:\\s|$)`).test(cmd));
246
+ const surface = [cmd, ...referenced.map((name) => scripts[name])].join("\n");
247
+ if (/\bturbo\s+run\b/.test(surface) && !surface.includes("--continue")) {
248
+ findings.push({
249
+ gate,
250
+ detail: `runs turbo without --continue — turbo aborts at the first failing package, so later packages go unverified yet baseline-forgiven. Append --continue where turbo is invoked, and verify forwarding first (\`run ${gate} -- --continue --dry=text\` must echo it): wrappers that branch on argv can silently swap the gated command`,
251
+ });
252
+ }
253
+ }
254
+ return findings;
255
+ }
173
256
  const shellToken = (cmd) => {
174
257
  for (const raw of cmd.trim().split(/\s+/)) {
175
258
  if (!raw || /^[A-Za-z_][A-Za-z0-9_]*=/.test(raw))
@@ -292,12 +375,12 @@ function headlineDetails(raw, fresh) {
292
375
  const headlines = raw
293
376
  .split("\n")
294
377
  .map((l) => l.replace(ANSI_RE, ""))
295
- .filter((l) => namesFailure(l) || SUMMARY_FAIL_RE.test(l));
378
+ .filter((l) => namesFailureEitherForm(l) || SUMMARY_FAIL_RE.test(l));
296
379
  if (!headlines.length)
297
380
  return { details: `new failures vs baseline:\n${fresh.join("\n")}` };
298
381
  return {
299
382
  details: `failing tests:\n${headlines.join("\n")}\n\nnew failure fingerprints vs baseline (secondary):\n${fresh.join("\n")}`,
300
- meta: { failingTests: headlines.filter(namesFailure) },
383
+ meta: { failingTests: headlines.filter(namesFailureEitherForm) },
301
384
  };
302
385
  }
303
386
  export async function compareToBaseline(cwd, commands, baseline, enabled) {
@@ -2,7 +2,7 @@ import { mkdirSync, writeFileSync } from "node:fs";
2
2
  import { join } from "node:path";
3
3
  import { getAdapter } from "../adapters/registry.js";
4
4
  import { bannerShell, paneDispatchCommand } from "../brand.js";
5
- import { extractVerdictJson, gateExitTrailer, gatePaneName, generateVerdictNonce, verdictNonceLine } from "../gates/llm.js";
5
+ import { dewrapPaneVerdict, extractVerdictJson, gateExitTrailer, gatePaneName, generateVerdictNonce, verdictNonceLine } from "../gates/llm.js";
6
6
  import { classifyVerdictCause } from "../gates/verdict-cause.js";
7
7
  import { disallowedBy } from "../route/preference.js";
8
8
  import { sh } from "./git.js";
@@ -173,7 +173,22 @@ opts = {}) {
173
173
  await driver.close(slot);
174
174
  }
175
175
  }
176
- return parseConsultVerdict(out, nonce);
176
+ // D-OBS-12 (dossier runs 1-2, 3/3 valid verdicts destroyed): the pane path is a terminal
177
+ // scrape — a single-line ~2000-char verdict soft-wraps in rendering and brace balance dies.
178
+ // Judge/review got dewrapPaneVerdict for exactly this (llm.ts, OBS-209 lineage); the consult
179
+ // seat never did. Headless output is machine-read stdout and needs no reconstruction.
180
+ const effective = cfg.visibility.llm === "headless" ? out : dewrapPaneVerdict(out, nonce);
181
+ const parsed = parseConsultVerdict(effective, nonce);
182
+ if (!parsed.verdict) {
183
+ // Q144s / OBS-196 parity: an unparseable verdict persists its raw bytes (pre-dewrap,
184
+ // rendering verbatim) beside the prompt — three human parks carried zero diagnostic
185
+ // payload because this artifact did not exist.
186
+ try {
187
+ writeFileSync(join(dir, `${d.taskId}-${n}${seatIdx > 0 ? `-s${seatIdx}` : ""}-response.txt`), out);
188
+ }
189
+ catch { /* evidence persistence must never fail the seat walk */ }
190
+ }
191
+ return parsed;
177
192
  };
178
193
  // v1.54 T1: ranked seat failover. Walk consult.prefer (adapter:model entries) to the first entry
179
194
  // whose adapter is in the live channel set; a failed seat or unparseable verdict falls to the next;
@@ -920,6 +920,31 @@ export async function runDaemon(repoRoot, opts = {}) {
920
920
  let releaseApprovalSerialization;
921
921
  try {
922
922
  let graph = loadGraph(repoRoot);
923
+ // GATE-FIX-4 defect 4 (no-op run refusal): a fresh run on a graph with nothing dispatchable used
924
+ // to journal {run-start, run-end} with zero dispatches — and downstream readers (greenness exit,
925
+ // status, notify) treat that run-end as completion, so an all-terminal graph "went green" having
926
+ // done nothing. Refuse HERE, the earliest seam where the graph is known and BEFORE Journal.create
927
+ // makes the run dir: no run-start row, no run dir, and the error exits nonzero through the CLI's
928
+ // ordinary catch. Dispatchability reuses graph.ts's status math rather than restating it: a task
929
+ // is dispatchable iff it is ready now (readyTasks) or can still become ready — pending with a
930
+ // non-parked closure (pendingTasks), or unblocked later by in-flight running/gated residue.
931
+ // FRESH runs only: a resume where everything is terminal already owns its run-end and keeps its
932
+ // current replay-then-quiesce behavior (the run-narrate test pins run-resume → run-end).
933
+ if (!opts.resume && readyTasks(graph).length === 0 && pendingTasks(graph).length === 0
934
+ && !graph.tasks.some((t) => t.status === "running" || t.status === "gated")) {
935
+ // Every pending task here is blocked-on-terminal by construction (pendingTasks(graph) is empty),
936
+ // so the count is labeled truthfully instead of as still-runnable "pending".
937
+ const counts = new Map();
938
+ for (const t of graph.tasks) {
939
+ const bucket = t.status === "pending" ? "blocked-on-terminal" : t.status;
940
+ counts.set(bucket, (counts.get(bucket) ?? 0) + 1);
941
+ }
942
+ const breakdown = counts.size === 0 ? "graph has zero tasks"
943
+ : [...counts.entries()].map(([s, n]) => `${s} ${n}`).join(", ");
944
+ throw new Error(`nothing to dispatch — no task is ready and none can become ready (${breakdown}). `
945
+ + "Refusing to start a run that would journal run-start/run-end with zero dispatches. "
946
+ + "Release parked tasks with `tickmarkr approve <runId> <taskId>`, or compile a fresh graph with `tickmarkr compile`.");
947
+ }
923
948
  // v1.51 T2: the routing mode resolves BEFORE any routing input is built — run flag > spec front-matter
924
949
  // > repo > global > default. The resolved cfg carries mode-compiled floors; route() never sees the mode.
925
950
  const rm = resolveRunMode(repoRoot, { flag: opts.mode, spec: graph.mode, globalDir: opts.globalDir });
@@ -9,12 +9,14 @@ export declare function TextLines({ lines }: {
9
9
  export declare function ToggleMark({ active }: {
10
10
  active: boolean;
11
11
  }): import("react").JSX.Element;
12
- export declare function FleetListScreen({ title, legend, rows, cursor, details, }: {
12
+ export declare function FleetListScreen({ title, legend, rows, cursor, details, filter, }: {
13
13
  title: string;
14
14
  legend: string;
15
15
  rows: FleetListRow[];
16
16
  cursor: number;
17
17
  details?: string[];
18
+ /** active type-to-search string; rendered on the legend line so the operator sees the narrowing */
19
+ filter?: string;
18
20
  }): import("react").JSX.Element;
19
21
  export declare function FleetReviewScreen({ title, legend, diff, }: {
20
22
  title: string;
@@ -17,8 +17,8 @@ export function ToggleMark({ active }) {
17
17
  ? _jsx(Text, { color: "ansi256(41)", children: GLYPHS.toggleActive })
18
18
  : _jsx(Text, { dimColor: true, children: GLYPHS.toggleInactive });
19
19
  }
20
- export function FleetListScreen({ title, legend, rows, cursor, details = [], }) {
21
- return (_jsxs(Box, { flexDirection: "column", children: [_jsx(Text, { bold: true, children: title }), _jsx(Text, { dimColor: true, children: legend }), rows.map((row, index) => (_jsxs(Text, { bold: index === cursor, children: [index === cursor ? `${GLYPHS.pointer} ` : " ", row.content] }, row.id))), _jsx(TextLines, { lines: details })] }));
20
+ export function FleetListScreen({ title, legend, rows, cursor, details = [], filter, }) {
21
+ return (_jsxs(Box, { flexDirection: "column", children: [_jsx(Text, { bold: true, children: title }), _jsx(Text, { dimColor: true, children: filter ? `${legend} · search: ${filter}` : legend }), rows.map((row, index) => (_jsxs(Text, { bold: index === cursor, children: [index === cursor ? `${GLYPHS.pointer} ` : " ", row.content] }, row.id))), _jsx(TextLines, { lines: details })] }));
22
22
  }
23
23
  export function FleetReviewScreen({ title, legend, diff, }) {
24
24
  return (_jsxs(Box, { flexDirection: "column", children: [_jsx(Text, { bold: true, children: title }), _jsx(Text, { dimColor: true, children: legend }), _jsx(Text, { children: diff })] }));
@@ -8,6 +8,10 @@ export type FleetEditorState = {
8
8
  classifications: FleetClassification[];
9
9
  selectedMode: RoutingMode;
10
10
  map: Record<string, MapEntry>;
11
+ judgeSeat?: {
12
+ adapter: string;
13
+ model: string;
14
+ };
11
15
  steering: Record<FleetSteeringKey, string[] | undefined>;
12
16
  };
13
17
  export type FleetOverlayReview = {
@@ -72,7 +76,7 @@ export type FleetCandidateOption = {
72
76
  };
73
77
  };
74
78
  export declare function formatDoctorAge(ageMs: number | null): string;
75
- export declare function FleetApp({ ageMs, agents, initialDenyAdapters, initialDenyModels, modelGroups, initialMode, modeOptions, initialMap, modePreview, shapeRows, candidatesForShape, preferOptionsForShape, initialSteering, steeringOptionsFor, reviewOverlay, reloadGuard, }: {
79
+ export declare function FleetApp({ ageMs, agents, initialDenyAdapters, initialDenyModels, modelGroups, initialMode, modeOptions, initialMap, modePreview, shapeRows, candidatesForShape, preferOptionsForShape, initialSteering, steeringOptionsFor, reviewOverlay, reloadGuard, entry, initialJudge, judgeSeats, }: {
76
80
  ageMs: number | null;
77
81
  agents: AgentCli[];
78
82
  initialDenyAdapters: string[];
@@ -89,8 +93,13 @@ export declare function FleetApp({ ageMs, agents, initialDenyAdapters, initialDe
89
93
  steeringOptionsFor: (which: FleetSteeringKey, current: string[]) => string[];
90
94
  reviewOverlay: (state: FleetEditorState) => FleetOverlayReview;
91
95
  reloadGuard: (bytes: string) => string | null;
96
+ entry?: "presets" | "probe";
97
+ /** resolved config judge as "adapter:model" — shown on the (keep default) picker row */
98
+ initialJudge?: string;
99
+ /** discovered adapter:model seats — the judge picker universe */
100
+ judgeSeats?: string[];
92
101
  }): import("react").JSX.Element;
93
- export declare function runFleetInkEditor({ ageMs, adapters, health, initialDenyAdapters, initialDenyModels, modelGroups, initialMode, modeOptions, initialMap, modePreview, shapeRows, candidatesForShape, preferOptionsForShape, initialSteering, steeringOptionsFor, reviewOverlay, reloadGuard, initialInput, input, output, debug, }: {
102
+ export declare function runFleetInkEditor({ ageMs, adapters, health, initialDenyAdapters, initialDenyModels, modelGroups, initialMode, modeOptions, initialMap, modePreview, shapeRows, candidatesForShape, preferOptionsForShape, initialSteering, steeringOptionsFor, reviewOverlay, reloadGuard, entry, initialJudge, judgeSeats, initialInput, input, output, debug, }: {
94
103
  ageMs: number | null;
95
104
  adapters: WorkerAdapter[];
96
105
  health: Record<string, AuthHealth>;
@@ -108,6 +117,10 @@ export declare function runFleetInkEditor({ ageMs, adapters, health, initialDeny
108
117
  steeringOptionsFor: (which: FleetSteeringKey, current: string[]) => string[];
109
118
  reviewOverlay: (state: FleetEditorState) => FleetOverlayReview;
110
119
  reloadGuard: (bytes: string) => string | null;
120
+ /** "presets" opens on the routing-mode screen with the extra custom row; default preserves the probe-first walk */
121
+ entry?: "presets" | "probe";
122
+ initialJudge?: string;
123
+ judgeSeats?: string[];
111
124
  initialInput?: string[];
112
125
  input: NodeJS.ReadStream;
113
126
  output: NodeJS.WriteStream;