tickmarkr 2.5.2 → 2.5.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/dist/adapters/qwen.d.ts +1 -0
  2. package/dist/adapters/qwen.js +8 -0
  3. package/dist/adapters/types.d.ts +1 -0
  4. package/dist/cli/commands/doctor.d.ts +10 -0
  5. package/dist/cli/commands/doctor.js +50 -1
  6. package/dist/cli/commands/plan.js +144 -3
  7. package/dist/cli/commands/resume.js +2 -0
  8. package/dist/cli/commands/run.js +12 -0
  9. package/dist/cli/commands/status.js +8 -14
  10. package/dist/compile/collateral.d.ts +2 -0
  11. package/dist/compile/collateral.js +50 -9
  12. package/dist/compile/ownership.d.ts +16 -0
  13. package/dist/compile/ownership.js +113 -13
  14. package/dist/config/config.d.ts +9 -0
  15. package/dist/config/config.js +13 -2
  16. package/dist/drivers/herdr.d.ts +5 -0
  17. package/dist/drivers/herdr.js +14 -3
  18. package/dist/drivers/types.d.ts +1 -0
  19. package/dist/gates/baseline.d.ts +5 -1
  20. package/dist/gates/baseline.js +18 -5
  21. package/dist/gates/llm.d.ts +1 -1
  22. package/dist/gates/llm.js +34 -12
  23. package/dist/gates/review.d.ts +46 -2
  24. package/dist/gates/review.js +111 -21
  25. package/dist/gates/run-gates.d.ts +2 -0
  26. package/dist/gates/run-gates.js +19 -10
  27. package/dist/route/router.d.ts +13 -0
  28. package/dist/route/router.js +64 -11
  29. package/dist/run/daemon.d.ts +3 -1
  30. package/dist/run/daemon.js +332 -53
  31. package/dist/run/git.d.ts +25 -0
  32. package/dist/run/git.js +68 -1
  33. package/dist/run/journal.js +16 -7
  34. package/dist/run/operator-state.d.ts +1 -1
  35. package/dist/run/operator-state.js +5 -9
  36. package/dist/tui/cockpit/live-runtime.js +12 -10
  37. package/package.json +1 -1
  38. package/skills/tickmarkr-auto/SKILL.md +1 -1
  39. package/skills/tickmarkr-loop/SKILL.md +1 -1
  40. package/skills/tickmarkr-overseer/SKILL.md +82 -4
  41. package/skills/tickmarkr-overseer/scripts/watch-launch.sh +38 -0
@@ -1,7 +1,42 @@
1
- import { readFileSync, readdirSync } from "node:fs";
1
+ import { existsSync, readFileSync, readdirSync } from "node:fs";
2
2
  import { basename, extname, join, posix } from "node:path";
3
3
  import { filesGlob } from "../graph/files-glob.js";
4
4
  import { collateralHits } from "./collateral.js";
5
+ export const SHAPE_ORACLES = [
6
+ "tests/run/narration.test.ts",
7
+ "tests/cli/brand-surfaces.test.ts",
8
+ "tests/run/notify-identity.test.ts",
9
+ "tests/run/outcome-projections.test.ts",
10
+ "tests/cockpit/setup.test.ts",
11
+ ];
12
+ export const SHAPE_ORACLE_SOURCES = [
13
+ "src/run/daemon.ts",
14
+ "src/cli/commands/run.ts",
15
+ ];
16
+ export const SHAPE_ORACLE_MAP = {
17
+ "src/run/daemon.ts": SHAPE_ORACLES,
18
+ "src/cli/commands/run.ts": SHAPE_ORACLES,
19
+ };
20
+ // A files[] entry touches a mapped shape-oracle source when it names it literally or by stem, or when
21
+ // the entry's glob — anchored or broad — matches the source AND the source exists in the repository
22
+ // being compiled: "src/run/**" owning this repository's daemon owns the five oracles, while the
23
+ // shipped sample PRD's "src/**" task in a repository holding no mapped source compiles clean.
24
+ // Anchored globs like "src/run/daemon.*" target the source specifically and touch it regardless.
25
+ function touchesSource(files, source, repoRoot) {
26
+ const stem = source.replace(/\.(?:[cm]?[jt]sx?)$/, "");
27
+ return files.some((entry) => {
28
+ if (entry === source || entry === stem)
29
+ return true;
30
+ const special = entry.search(/[*?{[]/);
31
+ if (special === -1)
32
+ return false;
33
+ if (!filesGlob([entry])(source))
34
+ return false;
35
+ if (entry.slice(0, special) === `${stem}.`)
36
+ return true;
37
+ return repoRoot !== undefined && existsSync(join(repoRoot, source));
38
+ });
39
+ }
5
40
  const normalize = (path) => path.replace(/^\.\//, "").split("\\").join("/");
6
41
  function testSources(repoRoot) {
7
42
  const root = join(repoRoot, "tests");
@@ -101,20 +136,24 @@ function mentionsCommandEntry(text) {
101
136
  return /(?:^|\/)src\/cli\/index\.(?:ts|js)\b/.test(text)
102
137
  || /["'`]src["'`]\s*,\s*["'`]cli["'`]\s*,\s*["'`]index\.(?:ts|js)["'`]/.test(text);
103
138
  }
104
- function corroboration(test, matches) {
105
- // A .test.ts-shaped collateral fixture is not by itself a dedicated test. Requiring a runner leaf
106
- // keeps import-only scan fixtures advisory while every executable subject in the measured union stays.
139
+ function corroborateSource(test, sourcePath) {
107
140
  const executable = test.text.replace(/\/\*[\s\S]*?\*\//g, "").replace(/^\s*\/\/.*$/gm, "");
108
141
  if (!/\b(?:test|it)(?:\.(?:concurrent|each|fails|only|skip|todo))*\s*\(/.test(executable))
109
142
  return undefined;
110
- for (const match of matches) {
111
- if (directlyImports(test, match.source))
112
- return { kind: "direct-import", source: match.source };
113
- }
143
+ if (directlyImports(test, sourcePath))
144
+ return { kind: "direct-import", source: sourcePath };
114
145
  if (invokesChildProcessSpawn(test.text) && mentionsCommandEntry(test.text)) {
115
- const command = matches.find(({ source }) => /^src\/cli\/commands\/[^/]+\.(?:[cm]?[jt]sx?)$/.test(source));
116
- if (command)
117
- return { kind: "command-entry-spawn", source: command.source, entry: "src/cli/index.ts" };
146
+ if (/^src\/cli\/commands\/[^/]+\.(?:[cm]?[jt]sx?)$/.test(sourcePath)) {
147
+ return { kind: "command-entry-spawn", source: sourcePath, entry: "src/cli/index.ts" };
148
+ }
149
+ }
150
+ return undefined;
151
+ }
152
+ function corroboration(test, matches) {
153
+ for (const match of matches) {
154
+ const evidence = corroborateSource(test, match.source);
155
+ if (evidence)
156
+ return evidence;
118
157
  }
119
158
  return undefined;
120
159
  }
@@ -155,6 +194,7 @@ export function ownershipFindings(tasks, repoRoot) {
155
194
  const context = task.context.map(normalize);
156
195
  return {
157
196
  task,
197
+ files,
158
198
  owns: files.length === 0 ? () => false : filesGlob(files),
159
199
  allows: files.length === 0 ? () => true : filesGlob([...files, ...context]),
160
200
  };
@@ -204,6 +244,52 @@ export function ownershipFindings(tasks, repoRoot) {
204
244
  });
205
245
  }
206
246
  }
247
+ for (const entry of indexed) {
248
+ for (const [source, oracles] of Object.entries(SHAPE_ORACLE_MAP)) {
249
+ if (!touchesSource(entry.files, source, repoRoot))
250
+ continue;
251
+ for (const oracle of oracles) {
252
+ if (!entry.owns(oracle)) {
253
+ findings.push({
254
+ code: "unowned-shape-oracle",
255
+ taskId: entry.task.id,
256
+ source,
257
+ oracle,
258
+ detail: `${entry.task.id} touches ${source} without owning shape oracle ${oracle}`,
259
+ });
260
+ }
261
+ }
262
+ }
263
+ }
264
+ for (const test of sources) {
265
+ const testOwners = owners(test.path);
266
+ if (testOwners.length === 0)
267
+ continue;
268
+ const testStem = basename(test.path).replace(/\.test\.ts$/, "");
269
+ for (const sourcePath of allSources) {
270
+ if (owners(sourcePath).length > 0)
271
+ continue;
272
+ const sourceStem = basename(sourcePath, extname(sourcePath));
273
+ if (testStem !== sourceStem && !testStem.startsWith(`${sourceStem}-`))
274
+ continue;
275
+ const evidence = corroborateSource(test, sourcePath);
276
+ if (!evidence)
277
+ continue;
278
+ for (const owner of testOwners) {
279
+ findings.push({
280
+ code: "unowned-source-of-owned-test",
281
+ taskId: owner.task.id,
282
+ test: test.path,
283
+ source: sourcePath,
284
+ corroboration: evidence,
285
+ detail: `${test.path} owned by ${owner.task.id} is a dedicated test of ${sourcePath} but no task owns ${sourcePath}`
286
+ + (evidence.kind === "direct-import" ? `; it imports ${evidence.source} directly`
287
+ : evidence.kind === "command-entry-spawn"
288
+ ? `; it spawns ${evidence.entry} to exercise ${evidence.source}` : ""),
289
+ });
290
+ }
291
+ }
292
+ }
207
293
  for (const source of sources) {
208
294
  for (const owner of owners(source.path)) {
209
295
  for (const path of repositoryPaths(source.text)) {
@@ -234,11 +320,25 @@ export function ownershipFindings(tasks, repoRoot) {
234
320
  }
235
321
  }
236
322
  }
237
- return findings.sort((a, b) => `${a.code}:${"test" in a ? a.test : a.path}:${"taskId" in a ? a.taskId : ""}`.localeCompare(`${b.code}:${"test" in b ? b.test : b.path}:${"taskId" in b ? b.taskId : ""}`));
323
+ const target = (f) => {
324
+ switch (f.code) {
325
+ case "unowned-test": return f.test;
326
+ case "unowned-shape-oracle": return f.oracle;
327
+ case "test-path-outside-allowlist": return f.test;
328
+ case "unordered-context-write": return f.path;
329
+ case "unowned-source-of-owned-test": return f.test;
330
+ }
331
+ };
332
+ return findings.sort((a, b) => {
333
+ const taskA = "taskId" in a ? a.taskId : "";
334
+ const taskB = "taskId" in b ? b.taskId : "";
335
+ return `${a.code}:${target(a)}:${taskA}`.localeCompare(`${b.code}:${target(b)}:${taskB}`);
336
+ });
238
337
  }
239
338
  export function renderOwnershipFinding(finding) {
240
339
  return `tickmarkr: ownership-lint[${finding.code}]: ${finding.detail}`;
241
340
  }
242
341
  export function blocksCompile(finding) {
243
- return finding.code === "unowned-test" && finding.corroboration !== undefined;
342
+ return (finding.code === "unowned-test" && finding.corroboration !== undefined)
343
+ || finding.code === "unowned-shape-oracle";
244
344
  }
@@ -175,6 +175,10 @@ export declare const TickmarkrConfigSchema: z.ZodObject<{
175
175
  on: "on";
176
176
  off: "off";
177
177
  }>;
178
+ escalateTier: z.ZodDefault<z.ZodEnum<{
179
+ on: "on";
180
+ off: "off";
181
+ }>>;
178
182
  learnedTuning: z.ZodOptional<z.ZodObject<{
179
183
  halfLifeRuns: z.ZodOptional<z.ZodNumber>;
180
184
  availWeight: z.ZodOptional<z.ZodNumber>;
@@ -275,6 +279,11 @@ export declare const TickmarkrConfigSchema: z.ZodObject<{
275
279
  "judge-only": "judge-only";
276
280
  }>>;
277
281
  criticalPaths: z.ZodOptional<z.ZodArray<z.ZodString>>;
282
+ floor: z.ZodDefault<z.ZodUnion<readonly [z.ZodLiteral<"worker">, z.ZodEnum<{
283
+ cheap: "cheap";
284
+ mid: "mid";
285
+ frontier: "frontier";
286
+ }>]>>;
278
287
  }, z.core.$strip>;
279
288
  consult: z.ZodObject<{
280
289
  adapter: z.ZodString;
@@ -302,6 +302,8 @@ export const TickmarkrConfigSchema = z.object({
302
302
  }),
303
303
  floors: z.record(z.string(), TierEnum),
304
304
  learned: z.enum(["on", "off"]), // v1.6 ROUTE-09 kill switch; a typo (offf) fails loud via safeParse
305
+ // OBS-986 (ES-1): climb one tier on request when untried channels exist in higher tiers.
306
+ escalateTier: z.enum(["on", "off"]).default("on"),
305
307
  // v1.9 ROUTE-15 — optional overrides for profile.ts HALF_LIFE_RUNS/AVAIL_WEIGHT; absent ⇒ byte-identical defaults.
306
308
  // SIBLING of learned (not nested): routing.learned is the on/off enum switch.
307
309
  learnedTuning: z.object({
@@ -366,6 +368,11 @@ export const TickmarkrConfigSchema = z.object({
366
368
  // R3: globs no task may skip review on (input/demux/lifecycle, gates/run/drivers/adapters, anything
367
369
  // reaching a shell). A judge-only task intersecting one of these fails COMPILE, never silently skips.
368
370
  criticalPaths: z.array(z.string()).optional(),
371
+ // RF-1 (OBS-922 add.2/3): review.floor — `worker` (default) names no tier and leaves the seat
372
+ // ranking to the author's routed seat; a tier is folded in as max(task floor, this) by reviewGate
373
+ // and its retry — it can raise a seat and never lower one. Not in the template: the cockpit
374
+ // fixture pins the fresh-install bytes (see d0e89a9a).
375
+ floor: z.union([z.literal("worker"), TierEnum]).default("worker"),
369
376
  }),
370
377
  // v1.54 T1: prefer — ranked consult seat failover. Entries MUST be adapter:model (unlike
371
378
  // review.prefer's adapter|adapter:model grammar): a consult seat has no channel to inherit a
@@ -445,6 +452,7 @@ export const DEFAULT_CONFIG = {
445
452
  // a workspace that has accumulated ≥MIN_SAMPLES warm telemetry per cell. Preview any workspace's effect
446
453
  // first with `tickmarkr plan` / `tickmarkr report`; flip to "off" to pin exact static routing (the kill switch stands).
447
454
  learned: "on",
455
+ escalateTier: "on",
448
456
  allowUnverifiedModels: false,
449
457
  },
450
458
  // Seed table (spec §13). New models = edit this (or your config.yaml), never code.
@@ -590,7 +598,7 @@ export const DEFAULT_CONFIG = {
590
598
  judge: { adapter: "claude-code", model: "fable" },
591
599
  // R3: no `policy` floor — the neutral floor leaves the compiler's per-task assignment standing, so
592
600
  // the path-keyed rule is reachable out of the box rather than raised to full by construction.
593
- review: { complexityThreshold: 7, timeoutMs: 900_000, required: true, criticalPaths: [...DEFAULT_REVIEW_CRITICAL_PATHS] },
601
+ review: { complexityThreshold: 7, timeoutMs: 900_000, required: true, floor: "worker", criticalPaths: [...DEFAULT_REVIEW_CRITICAL_PATHS] },
594
602
  consult: { adapter: "claude-code", model: "fable", stallMinutes: 15 },
595
603
  // v1.4: gate LLM calls (judge/review/consult) run headless by default; pane opts back into visible agents.
596
604
  // v1.2: workers are the real agent TUI in the pane; "print" restores the -p-rendered-in-pane path.
@@ -784,6 +792,7 @@ export function configTemplate(overlay) {
784
792
  # floors: # tier authority — advisory minimum bands; 'tickmarkr plan' lints violations
785
793
  # migration: frontier
786
794
  # learned: on # default ON (ROUTE-14); cold profile = exact v1.5 static routing, warms per workspace. Set 'off' to pin static routing; preview with 'tickmarkr plan'
795
+ # escalateTier: on # on | off (default on): climb one tier on request when untried channels exist in higher tiers
787
796
  # learnedTuning: { halfLifeRuns: 5, availWeight: 0.05 } # optional; defaults byte-identical
788
797
  # explore: { mode: on, excludeShapes: [], excludeComplexityAtOrAbove: null, cap: 5 } # optional; absent ⇒ byte-identical
789
798
  # sla: { implement: 15 } # optional per-shape minutes — advisory plan lint only; absent ⇒ no lint
@@ -817,7 +826,9 @@ export function configTemplate(overlay) {
817
826
  # test: npm test
818
827
  # byShape:
819
828
  # docs: { acceptance: false, review: false } # baseline, evidence, and scope are mandatory
820
- # review: { complexityThreshold: 7, timeoutMs: 900000, required: true, prefer: [codex:gpt-5.6-sol, kimi] }
829
+ # review: { complexityThreshold: 7, timeoutMs: 900000, required: true, floor: worker, prefer: [codex:gpt-5.6-sol, kimi] }
830
+ # # floor: worker (default) | cheap | mid | frontier — the reviewer seats at or above
831
+ # # max(author tier, task floor, this tier, prior reviewer); a tier here only raises it
821
832
  # # prefer: ordered reviewer seat preference (adapter | adapter:model); ranks
822
833
  # # diversity-eligible channels only — never admits a same-vendor/same-model reviewer
823
834
  # consult: { adapter: claude-code, model: fable, stallMinutes: 15, prefer: [codex:gpt-5.6-sol, kimi:kimi-code/k3] }
@@ -128,6 +128,11 @@ export declare class HerdrDriver implements ExecutorDriver {
128
128
  close(slot: Slot): Promise<void>;
129
129
  private closeGrouped;
130
130
  private watchPanes;
131
+ /** WB-1 (OBS-988): the daemon proved this run's own board lost. Forget the cached slot FIRST — even a
132
+ * retire that fails must never let the narrator answer with the ghost again — then take the pane
133
+ * back on ownership alone: the dead UI can never acknowledge, so the stop request is left for a
134
+ * merely stuck one to find. Serialized with the narrator so no split races the close. */
135
+ retireLostWatch(slot: Slot): Promise<void>;
131
136
  /** Name collisions never confer repository ownership. Unknown boards stay protected. */
132
137
  private retireWatch;
133
138
  focus(target: FocusTarget): Promise<FocusResult>;
@@ -6,7 +6,7 @@ import { declaredInputBoxForWorkerName, matchesEmptyInputBox, matchesInputBox, m
6
6
  import { consumePaneLaunchIntent, PANE_IDENTITY_ENV, paneIdentityLine } from "../brand.js";
7
7
  import { createWorktree, sh } from "../run/git.js";
8
8
  import { Journal } from "../run/journal.js";
9
- import { readSupervision, readWatchBoard, reserveWatchBoard, stopWatchBoard, WATCH_OWNER_ENV } from "../run/supervision.js";
9
+ import { readSupervision, readWatchBoard, requestWatchBoardStop, reserveWatchBoard, stopWatchBoard, WATCH_OWNER_ENV } from "../run/supervision.js";
10
10
  import { herdrSealShellPrefix } from "./subprocess.js";
11
11
  import { canonicalizeLegacyName, formatOwnedName, panesToClose, parseOwnedName } from "./types.js";
12
12
  // VIS-09 P43-03: adopted safety floor from 43-MEASUREMENT.md (narrowest safe 53 → floor 108).
@@ -1169,8 +1169,16 @@ export class HerdrDriver {
1169
1169
  throw new Error("herdr pane list returned no panes");
1170
1170
  return panes.filter(p => p.workspace_id === this.ws && p.label === name);
1171
1171
  }
1172
+ /** WB-1 (OBS-988): the daemon proved this run's own board lost. Forget the cached slot FIRST — even a
1173
+ * retire that fails must never let the narrator answer with the ghost again — then take the pane
1174
+ * back on ownership alone: the dead UI can never acknowledge, so the stop request is left for a
1175
+ * merely stuck one to find. Serialized with the narrator so no split races the close. */
1176
+ async retireLostWatch(slot) {
1177
+ this.watches.delete(slot.name);
1178
+ await this.serial(() => this.retireWatch(slot, true));
1179
+ }
1172
1180
  /** Name collisions never confer repository ownership. Unknown boards stay protected. */
1173
- async retireWatch(slot) {
1181
+ async retireWatch(slot, lost = false) {
1174
1182
  const runId = parseOwnedName(slot.name)?.runId;
1175
1183
  const owner = runId ? readWatchBoard(slot.cwd, runId) : undefined;
1176
1184
  const matches = await this.watchPanes(slot.name);
@@ -1181,7 +1189,10 @@ export class HerdrDriver {
1181
1189
  owner.pane !== slot.id || matches.length !== 1 || matches[0]?.pane_id !== owner.pane) {
1182
1190
  throw new Error(`watch ownership unknown or foreign for ${slot.name}; existing board protected`);
1183
1191
  }
1184
- await stopWatchBoard(owner, this.time);
1192
+ if (lost)
1193
+ requestWatchBoardStop(owner); // a lost owner never answers; only a live one still gets the ack window
1194
+ else
1195
+ await stopWatchBoard(owner, this.time);
1185
1196
  const verified = await this.watchPanes(slot.name);
1186
1197
  if (verified.length !== 1 || verified[0]?.pane_id !== owner.pane)
1187
1198
  throw new Error("watch target changed after acknowledgement; pane protected");
@@ -106,6 +106,7 @@ export interface ExecutorDriver {
106
106
  narrateWith?(narrate: (event: JournalEvent) => void): void;
107
107
  worktree(repo: string, branch: string, baseRef: string): Promise<string>;
108
108
  narrator?: (cwd: string, command: string, runId?: string) => Promise<Slot>;
109
+ retireLostWatch?: (slot: Slot) => Promise<void>;
109
110
  /** Best-effort projection of a task's lifecycle onto the execution host. */
110
111
  project?: (taskId: string, state: "in-progress" | "in-review" | "completed") => Promise<void>;
111
112
  reconcile?: (desired: Set<string>, runId: string, opts?: PanesToCloseOpts) => Promise<void>;
@@ -161,6 +161,7 @@ export declare function detectVacuousOracles(cwd: string, tasks: ReadonlyArray<{
161
161
  export declare function compareToBaseline(cwd: string, commands: Record<string, string>, baseline: Baseline, enabled: string[], opts?: {
162
162
  rerunOf?: HostStarvedRerun;
163
163
  infraRerun?: HostStarvedRerun;
164
+ selected?: readonly string[];
164
165
  }): Promise<GateResult[]>;
165
166
  export interface HostStarvedRerun {
166
167
  durationMs: number;
@@ -184,7 +185,10 @@ export declare function resetCalmWindowForTests(): void;
184
185
  export declare function waitForCalmWindow(): Promise<number>;
185
186
  /** Summary totals survive digit-normalized fingerprints and exclude test-owned echoed output. */
186
187
  export declare function runnerFileCount(raw: string): number | null;
187
- export declare function fileCountDeficit(entry: BaselineCommand | undefined, raw: string): string | undefined;
188
+ export declare function fileCountDeficit(entry: BaselineCommand | undefined, raw: string, opts?: {
189
+ name?: string;
190
+ selected?: readonly string[];
191
+ }): string | undefined;
188
192
  /** Both readers classify fresh evidence first, retaining the whole-output guard against infra forgiveness. */
189
193
  export declare function classifyFreshRunnerOutput(entry: BaselineCommand | undefined, raw: string, code: number): FailureClassification | undefined;
190
194
  export {};
@@ -627,7 +627,16 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
627
627
  details: `${g.meta?.classification === "infra" ? "infra; " : ""}runner-infra rerun after waiting ${opts.infraRerun.waitedMs}ms for a calm load window: ${withRerun.details.replace(/^infra; /, "")}`,
628
628
  meta: { ...withRerun.meta, runnerInfraRerun: opts.infraRerun },
629
629
  } : withRerun;
630
- results.push(r.capacity ? { ...final, capacity: r.capacity } : final);
630
+ const withSelected = name === "test" && opts.selected
631
+ ? {
632
+ ...final,
633
+ meta: {
634
+ ...final.meta,
635
+ ...(Array.isArray(opts.selected) ? { selectedTests: [...opts.selected] } : {}),
636
+ },
637
+ }
638
+ : final;
639
+ results.push(r.capacity ? { ...withSelected, capacity: r.capacity } : withSelected);
631
640
  };
632
641
  // …and whether the entry that would forgive this command was measured in the same world. A
633
642
  // baseline captured under a different fork cap forgives nothing: its fingerprints describe a
@@ -657,7 +666,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
657
666
  continue;
658
667
  }
659
668
  const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
660
- const deficit = fileCountDeficit(entry, raw);
669
+ const deficit = fileCountDeficit(entry, raw, { name, selected: opts.selected });
661
670
  if (deficit) {
662
671
  record({ gate: name, pass: false, details: deficit, meta: { classification: "infra", infra: true } });
663
672
  continue;
@@ -689,7 +698,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
689
698
  if (name === "test" && classification === "infra" && !rerunOf && !opts.infraRerun) {
690
699
  const waitedMs = await waitForCalmWindow();
691
700
  const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
692
- results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { infraRerun: provenance }));
701
+ results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { ...opts, infraRerun: provenance }));
693
702
  continue;
694
703
  }
695
704
  // OBS-896: every fresh failure must be timeout-class, and the suite must take at least twice its
@@ -698,7 +707,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
698
707
  && hostStarved(failing.join("\n"), r.durationMs ?? 0, entry?.durationMs)) {
699
708
  const waitedMs = await waitForCalmWindow();
700
709
  const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
701
- results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { rerunOf: provenance }));
710
+ results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { ...opts, rerunOf: provenance }));
702
711
  continue;
703
712
  }
704
713
  if (classification === "infra") {
@@ -833,7 +842,11 @@ export function runnerFileCount(raw) {
833
842
  });
834
843
  return counts.length ? counts.reduce((sum, count) => sum + count, 0) : null;
835
844
  }
836
- export function fileCountDeficit(entry, raw) {
845
+ export function fileCountDeficit(entry, raw, opts) {
846
+ // OBS-985: only a real, named selected-test run of the TEST gate is exempt — a truthy flag or an
847
+ // empty list named no selection and let a full-suite call opt itself out of the deficit guard.
848
+ if (opts?.name === "test" && opts.selected !== undefined && opts.selected.length > 0)
849
+ return undefined;
837
850
  const actual = runnerFileCount(raw);
838
851
  return entry?.fileCount != null && actual !== null && actual < entry.fileCount
839
852
  ? `infra; runner reported ${actual} test files, below baseline ${entry.fileCount} — suite incomplete`
@@ -65,7 +65,7 @@ export interface LlmRunResult {
65
65
  seatAuthoredBytes?: number;
66
66
  }
67
67
  export declare const PROMPT_GLYPHS: readonly ["➜", "❯", "$", "%", ">>", ">"];
68
- export declare function reviewSeatOutput(raw: string, nonce: string): string;
68
+ export declare function reviewSeatOutput(raw: string, nonce: string, adapterBannerRows?: readonly string[]): string;
69
69
  export declare const REVIEW_FIRST_LIVENESS_MS = 30000;
70
70
  export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number): Promise<string>;
71
71
  export declare function runViaDriver(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via: LlmVia, timeoutMs?: number): Promise<string>;
package/dist/gates/llm.js CHANGED
@@ -204,12 +204,14 @@ const IDENTITY_OPENERS = ["review ·", "tickmarkr"];
204
204
  // Stages of the preamble walk. Each complete harness row is accepted only at or after its stage.
205
205
  const ECHO = 0, START = 1, BANNER = 2, IDENTITY = 3, SEAT = 4;
206
206
  // The char offset where the seat's own text begins, or -1 when the capture ends inside the preamble.
207
- function seatStart(output) {
207
+ function seatStart(output, adapterBannerRows) {
208
208
  const rows = output.split("\n");
209
209
  if (rows.length > 1 && rows[rows.length - 1] === "")
210
210
  rows.pop(); // the read's own line terminator
211
211
  let stage = ECHO;
212
212
  let bannerAt; // next banner row expected once the banner has begun
213
+ let adapterBannerAt;
214
+ let identitySeen = false;
213
215
  let offset = 0;
214
216
  for (let i = 0; i < rows.length; i++) {
215
217
  const row = rows[i].replace(/[ \t]+$/, "");
@@ -217,7 +219,7 @@ function seatStart(output) {
217
219
  const last = i === rows.length - 1;
218
220
  // Complete harness rows: equality against the shape the preamble allows at this stage.
219
221
  let accepted = false;
220
- if (stage < SEAT && t.length === 0)
222
+ if (!identitySeen && stage < SEAT && t.length === 0)
221
223
  accepted = true; // blank rows between preamble rows
222
224
  else if (stage <= ECHO && completeEchoRow(row))
223
225
  accepted = true;
@@ -225,17 +227,27 @@ function seatStart(output) {
225
227
  stage = BANNER;
226
228
  accepted = true;
227
229
  }
228
- else if (stage <= BANNER && bannerAt === undefined && BANNER_ROWS.includes(row)) {
230
+ else if (!identitySeen && stage <= BANNER && bannerAt === undefined && BANNER_ROWS.includes(row)) {
229
231
  stage = BANNER;
230
232
  bannerAt = BANNER_ROWS.indexOf(row) + 1;
231
233
  accepted = true;
232
234
  }
233
- else if (stage <= BANNER && bannerAt !== undefined && row === BANNER_ROWS[bannerAt]) {
235
+ else if (!identitySeen && stage <= BANNER && bannerAt !== undefined && row === BANNER_ROWS[bannerAt]) {
234
236
  bannerAt++;
235
237
  accepted = true;
236
238
  }
237
- else if (stage <= IDENTITY && IDENTITY_LINE.test(t)) {
238
- stage = SEAT;
239
+ else if (stage <= IDENTITY && adapterBannerAt === undefined && adapterBannerRows.includes(row)) {
240
+ stage = BANNER;
241
+ adapterBannerAt = adapterBannerRows.indexOf(row) + 1;
242
+ accepted = true;
243
+ }
244
+ else if (stage <= IDENTITY && adapterBannerAt !== undefined && row === adapterBannerRows[adapterBannerAt]) {
245
+ adapterBannerAt++;
246
+ accepted = true;
247
+ }
248
+ else if (!identitySeen && stage <= IDENTITY && IDENTITY_LINE.test(t)) {
249
+ stage = IDENTITY;
250
+ identitySeen = true;
239
251
  accepted = true;
240
252
  }
241
253
  if (accepted) {
@@ -253,9 +265,16 @@ function seatStart(output) {
253
265
  ? BANNER_ROWS.some((b) => b.startsWith(row))
254
266
  : BANNER_ROWS[bannerAt]?.startsWith(row) === true))
255
267
  return -1;
268
+ if (stage <= IDENTITY && (adapterBannerAt === undefined
269
+ ? adapterBannerRows.some((b) => b.startsWith(row))
270
+ : adapterBannerRows[adapterBannerAt]?.startsWith(row) === true))
271
+ return -1;
256
272
  // The identity row is painted right after the banner, so its prefix is a partial paint only there;
257
273
  // with no banner in the capture, "review" or "tick" alone is the seat's own first row.
258
- if (stage <= IDENTITY && bannerAt === BANNER_ROWS.length && IDENTITY_OPENERS.some((o) => o.startsWith(t)))
274
+ if (!identitySeen && stage <= IDENTITY
275
+ && (bannerAt === BANNER_ROWS.length
276
+ || (adapterBannerRows.length > 0 && adapterBannerAt === adapterBannerRows.length))
277
+ && IDENTITY_OPENERS.some((o) => o.startsWith(t)))
259
278
  return -1;
260
279
  return offset;
261
280
  }
@@ -266,9 +285,9 @@ function seatStart(output) {
266
285
  // Every capture taken before the preamble finishes therefore measures ZERO seat-authored bytes,
267
286
  // which is what makes the caller's running Math.max safe: a partial banner counted once would be
268
287
  // retained for the whole call and buy a silent seat its full ceiling.
269
- export function reviewSeatOutput(raw, nonce) {
288
+ export function reviewSeatOutput(raw, nonce, adapterBannerRows = []) {
270
289
  const output = stripVTControlCharacters(raw).replace(/\r\n?/g, "\n");
271
- const start = seatStart(output);
290
+ const start = seatStart(output, adapterBannerRows);
272
291
  if (start < 0)
273
292
  return "";
274
293
  const seat = output.slice(start);
@@ -287,7 +306,10 @@ async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 3000
287
306
  const pf = join(dir, "prompt.md");
288
307
  writeFileSync(pf, prompt);
289
308
  const r = await sh(adapter.headlessCommand(pf, model), cwd, timeoutMs);
290
- return { output: r.stdout + "\n" + r.stderr, exitCode: r.code, timedOut: r.timedOut === true, seatAuthoredBytes: Buffer.byteLength((r.stdout + r.stderr).trim()) };
309
+ const output = r.stdout + "\n" + r.stderr;
310
+ const nonce = extractPromptNonce(prompt) ?? "";
311
+ return { output, exitCode: r.code, timedOut: r.timedOut === true,
312
+ seatAuthoredBytes: Buffer.byteLength(reviewSeatOutput(output, nonce, adapter.harnessBannerRows).trim()) };
291
313
  }
292
314
  finally {
293
315
  rmSync(dir, { recursive: true, force: true });
@@ -349,7 +371,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
349
371
  const startedAt = Date.now();
350
372
  out = await via.driver.read(slot, 400);
351
373
  const reviewing = prompt.startsWith("TICKMARKR-REVIEW");
352
- seatAuthoredBytes = Buffer.byteLength(reviewSeatOutput(out, nonce));
374
+ seatAuthoredBytes = Buffer.byteLength(reviewSeatOutput(out, nonce, adapter.harnessBannerRows));
353
375
  let firstLivenessObserved = false;
354
376
  let priorSnapshot = normalizeStallSnapshot(out);
355
377
  const anchoredAt = Date.now();
@@ -365,7 +387,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
365
387
  const matched = await via.driver.waitOutput(slot, exitPattern, sliceMs, { regex: true });
366
388
  const raw = await via.driver.read(slot, 400);
367
389
  out = raw;
368
- seatAuthoredBytes = Math.max(seatAuthoredBytes, Buffer.byteLength(reviewSeatOutput(raw, nonce)));
390
+ seatAuthoredBytes = Math.max(seatAuthoredBytes, Buffer.byteLength(reviewSeatOutput(raw, nonce, adapter.harnessBannerRows)));
369
391
  // waitOutput is the driver's authoritative marker match. The raw check covers drivers whose
370
392
  // wait timed out at the same boundary the marker landed; either way a trailer completes
371
393
  // normally and is never mistaken for inactivity.
@@ -53,9 +53,53 @@ export declare function isDiffCapPark(result: GateResult): boolean;
53
53
  export declare function diffCapParkReason(results: GateResult[]): string | null;
54
54
  export declare function modelId(model: string): string;
55
55
  export { modelProvider };
56
+ /**
57
+ * One function decides whether a reviewer's `resolved` or `reraised` id names a carried fingerprint,
58
+ * comparing both sides with every whitespace run removed (`s.replace(/\s+/g, "")`).
59
+ */
60
+ export declare function matchClosureId(candidate: unknown, fingerprint: string): boolean;
61
+ export declare function matchClosureId(candidate: unknown, fingerprints: Iterable<string>): string | undefined;
62
+ /**
63
+ * Validates closure ids in a review verdict: membership, duplication, and coverage of every prior id
64
+ * all route through matchClosureId.
65
+ */
66
+ export declare function isReviewClosureInvalid(v: Pick<ReviewVerdict, "resolved" | "reraised"> | null | undefined, priorIds: ReadonlySet<string> | readonly string[]): boolean;
67
+ export type ReviewerFloorCause = "author-tier" | "task-floor" | "config" | "prior-reviewer";
68
+ /**
69
+ * RF-1 (OBS-922 add.2/3): the tier a reviewer must meet is the maximum of the author's tier, the
70
+ * task-declared floor, a configured `review.floor` tier and, on a second round or a retry, the prior
71
+ * reviewer's tier. The cause names the input that reached the maximum (earlier inputs win a tie, so a
72
+ * floor the author's tier already satisfies is attributed to the author).
73
+ */
74
+ export declare function resolveReviewerFloor(authorTier: Tier, taskFloor?: Tier, configFloor?: Tier, priorReviewerTier?: Tier): {
75
+ floor: Tier;
76
+ cause: ReviewerFloorCause;
77
+ };
78
+ /** A prior reviewer of THIS task: a channel key, or a journaled row's key plus the tier it was DISPATCHED at. */
79
+ export type PriorReviewer = string | {
80
+ reviewer: string;
81
+ tier?: unknown;
82
+ };
83
+ /**
84
+ * The highest tier among the task's prior reviewers — the prior reviewer's tier for RF-1. A recorded
85
+ * dispatch tier is historical evidence and wins over the current pool; an unrecorded one falls back to
86
+ * the seat's channel; a seat neither establishes (it left the pool on resume, or the journal holds
87
+ * garbage) holds frontier — fail closed, never silently dropped.
88
+ */
89
+ export declare function priorReviewerTier(channels: BillingChannel[], priorReviewers?: readonly PriorReviewer[]): Tier | undefined;
90
+ /**
91
+ * The gate's floor: author tier, task floor, review.floor (a tier — `worker` names none) and the seats
92
+ * the caller names as THIS TASK's prior reviewers (earlier rounds' seats, a flaked seat). Eligibility
93
+ * exclusions are NOT evidence — a retry bans a flaked seat's whole adapter, and those sibling channels
94
+ * never reviewed — and neither is the run-scoped LRU rotation history, which names unrelated tasks' seats.
95
+ */
96
+ export declare function gateReviewerFloor(task: Pick<Task, "routingHints">, cfg: TickmarkrConfig, author: Assignment, channels: BillingChannel[], priorReviewers?: readonly PriorReviewer[]): {
97
+ floor: Tier;
98
+ cause: ReviewerFloorCause;
99
+ };
56
100
  export declare function pickReviewer(author: Assignment, channels: BillingChannel[], exclude?: string[], // v1.1 failover: reviewer channels that already produced garbage for this task
57
101
  prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
58
- floor?: Tier, // task-declared only; config floors govern workers and must not silently move review seats
102
+ floor?: Tier, // task/config/prior floor from the caller; the author's own tier is ALWAYS applied here (RF-1)
59
103
  history?: string[], // run-scoped picks, oldest to newest; empty preserves the established ranking
60
104
  onSeat?: (seat: number, count: number) => void, demoted?: ReadonlySet<string>): BillingChannel | null;
61
105
  export type ReviewUnparseableCause = VerdictUnparseableCause | "launch-never-started" | "truncated" | "silent";
@@ -65,4 +109,4 @@ export type ReviewUnparseableCause = VerdictUnparseableCause | "launch-never-sta
65
109
  * judgement rather than a guarantee made by this renderer.
66
110
  */
67
111
  export declare function renderDeclaredWriteScope(files: ReadonlyArray<string>): string;
68
- export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string, reviewHistory?: string[], demotedReviewers?: ReadonlySet<string>, carriedFindings?: readonly StructuredFinding[]): Promise<GateResult>;
112
+ export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string, reviewHistory?: string[], demotedReviewers?: ReadonlySet<string>, carriedFindings?: readonly StructuredFinding[], priorReviewers?: readonly PriorReviewer[]): Promise<GateResult>;