tickmarkr 2.3.0 → 2.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/dist/adapters/catalog-remote.d.ts +1 -4
  2. package/dist/adapters/catalog-remote.js +52 -42
  3. package/dist/adapters/catalog.js +5 -3
  4. package/dist/adapters/claude-code.d.ts +1 -1
  5. package/dist/adapters/claude-code.js +8 -5
  6. package/dist/adapters/model-lints.d.ts +9 -5
  7. package/dist/adapters/model-lints.js +56 -15
  8. package/dist/adapters/model-windows.js +11 -0
  9. package/dist/adapters/prompt.js +1 -0
  10. package/dist/adapters/qwen.d.ts +5 -0
  11. package/dist/adapters/qwen.js +153 -0
  12. package/dist/adapters/types.d.ts +1 -0
  13. package/dist/adapters/types.js +1 -0
  14. package/dist/cli/commands/compile.js +32 -6
  15. package/dist/cli/commands/doctor.d.ts +4 -3
  16. package/dist/cli/commands/doctor.js +20 -6
  17. package/dist/cli/commands/fleet.d.ts +4 -0
  18. package/dist/cli/commands/fleet.js +53 -14
  19. package/dist/cli/commands/plan.js +29 -5
  20. package/dist/cli/commands/status.d.ts +1 -0
  21. package/dist/cli/commands/status.js +45 -1
  22. package/dist/cli/commands/verify.d.ts +1 -0
  23. package/dist/cli/commands/verify.js +5 -0
  24. package/dist/cli/index.d.ts +1 -1
  25. package/dist/cli/index.js +2 -2
  26. package/dist/compile/collateral.d.ts +2 -9
  27. package/dist/compile/collateral.js +2 -9
  28. package/dist/compile/index.d.ts +4 -1
  29. package/dist/compile/index.js +41 -7
  30. package/dist/compile/native.d.ts +4 -2
  31. package/dist/compile/native.js +55 -6
  32. package/dist/compile/ownership.js +34 -9
  33. package/dist/config/config.d.ts +1 -0
  34. package/dist/config/config.js +51 -5
  35. package/dist/drivers/herdr.d.ts +1 -0
  36. package/dist/drivers/herdr.js +11 -1
  37. package/dist/drivers/orca.d.ts +14 -1
  38. package/dist/drivers/orca.js +114 -15
  39. package/dist/drivers/types.d.ts +10 -0
  40. package/dist/gates/baseline.js +26 -7
  41. package/dist/gates/review.d.ts +4 -2
  42. package/dist/gates/review.js +24 -13
  43. package/dist/gates/run-gates.d.ts +5 -2
  44. package/dist/gates/run-gates.js +34 -17
  45. package/dist/route/preference.d.ts +4 -0
  46. package/dist/route/preference.js +40 -0
  47. package/dist/route/router.js +15 -2
  48. package/dist/run/consult.d.ts +1 -0
  49. package/dist/run/consult.js +34 -7
  50. package/dist/run/daemon.d.ts +4 -0
  51. package/dist/run/daemon.js +128 -50
  52. package/dist/run/git.d.ts +2 -0
  53. package/dist/run/git.js +36 -5
  54. package/dist/run/journal.js +4 -1
  55. package/dist/tui/ink/fleet-app.d.ts +4 -0
  56. package/dist/tui/ink/fleet-app.js +45 -16
  57. package/package.json +59 -1
@@ -66,9 +66,17 @@ export declare function panesToClose(agents: FleetAgent[], desired: Set<string>,
66
66
  tabId?: string;
67
67
  }[];
68
68
  export declare function canonicalizeLegacyName(name: string, runId: string): OwnedName;
69
+ export interface SlotPlacement {
70
+ surface?: string;
71
+ hostPlatform?: string;
72
+ }
69
73
  export interface ExecutorDriver {
70
74
  id: string;
71
75
  interactive: boolean;
76
+ /** The exact terminal read surface used for liveness evidence. */
77
+ readSource?: string;
78
+ /** Placement facts returned by drivers whose terminal host exposes them. */
79
+ describe?(slot: Slot): SlotPlacement | Promise<SlotPlacement>;
72
80
  slot(cwd: string, name: string, opts?: SlotOpts): Promise<Slot>;
73
81
  run(slot: Slot, cmd: string): Promise<void>;
74
82
  waitOutput(slot: Slot, pattern: string, timeoutMs: number, opts?: {
@@ -84,5 +92,7 @@ export interface ExecutorDriver {
84
92
  narrateWith?(narrate: (event: JournalEvent) => void): void;
85
93
  worktree(repo: string, branch: string, baseRef: string): Promise<string>;
86
94
  narrator?: (cwd: string, command: string, runId?: string) => Promise<Slot>;
95
+ /** Best-effort projection of a task's lifecycle onto the execution host. */
96
+ project?: (taskId: string, state: "in-progress" | "in-review" | "completed") => Promise<void>;
87
97
  reconcile?: (desired: Set<string>, runId: string, opts?: PanesToCloseOpts) => Promise<void>;
88
98
  }
@@ -5,10 +5,19 @@ import { DEFAULT_SHELL_TIMEOUT_MS, describeCapacity, sameCapacity, sh } from "..
5
5
  // codes that varied between baseline and worktree runs, was reported as a "new failure". Strip ANSI first;
6
6
  // a pass-marker line is never a failure. [\d;#] covers raw ANSI and digit-normalized ANSI ("\x1b[#m") from
7
7
  // baselines stored by pre-hardening code.
8
- const ANSI_RE = /\x1b\[[\d;#]*[A-Za-z]/g;
8
+ // OBS-891 (run 3372): vitest toggles the cursor (`\x1b[?25l` / `\x1b[?25h`) around its progress
9
+ // output, and the private-mode parameter byte `?` never matched [\d;#], so an echo-block HEADER glued
10
+ // to a cursor-show sequence stayed invisible to withoutVitestEchoBlocks and its whole block leaked as
11
+ // runner evidence — seven prose-only "infra" parks in one night. Full CSI grammar: parameter bytes
12
+ // 0x30–0x3F (plus `#` for digit-normalized stored baselines), intermediates 0x20–0x2F, final 0x40–0x7E.
13
+ const ANSI_RE = /\x1b\[[0-?#]*[ -/]*[@-~]/g;
9
14
  // ponytail: only leading ✓/✔ after optional "label:" prefixes (turbo/vitest), or tickmarkr's own run
10
15
  // summary, counts as a pass line — other runners' pass markers (PASS, ok) stay fingerprintable
11
16
  const PASS_LINE_RE = /^\s*(?:(?:[\w@./-]+:\s*)*[✓✔]|(?:\[tickmarkr\]\s+)?(?:tickmarkr\s+[\w.-]+:\s+)?(?:\d+|#)\s+done,\s+(?:\d+|#)\s+failed(?:,\s+(?:\d+|#)\s+awaiting human)?\b)/;
17
+ // OBS-888: tickmarkr's own operator lines (`tickmarkr: baseline capture for "test" … spawn EAGAIN`)
18
+ // are printed by this product, never by a runner about the work. When this repository's tests exercise
19
+ // the capture path they print them too, carrying errno tokens INFRA_RE would read as host evidence.
20
+ const OPERATOR_LINE_RE = /^\s*tickmarkr: /;
12
21
  // HYG-08 (D-01, incident run-20260711-154920): a failing test went unnamed for 3 attempts because details
13
22
  // headlined benign fingerprint-diff noise. These anchors harvest the runner's OWN failure naming from fresh
14
23
  // output to headline it. \s is fine in a TS regex — the BSD [[:space:]] rule binds shell grep only.
@@ -149,7 +158,10 @@ const isFingerprintShaped = (l) => isFailureShaped(l) || INFRA_RE.test(l);
149
158
  * unreadable-runner case the existing fail-closed path already owns.
150
159
  */
151
160
  export function classifyFailureOutput(output) {
152
- const lines = output.split("\n").map((l) => l.replace(ANSI_RE, "")).filter((l) => !PASS_LINE_RE.test(l));
161
+ // OBS-891: the gate reads the WHOLE output here when the fresh-fingerprint diff is empty, so an errno
162
+ // token inside a test's echoed stdout/stderr block must be as invisible to this classifier as it is
163
+ // to fingerprint(): test-owned output is never runner evidence about the work.
164
+ const lines = withoutVitestEchoBlocks(output).map((l) => l.replace(ANSI_RE, "")).filter((l) => !PASS_LINE_RE.test(l) && !OPERATOR_LINE_RE.test(l));
153
165
  if (lines.some(namesRegression))
154
166
  return "regression";
155
167
  return lines.some(isInfraLine) ? "infra" : undefined;
@@ -161,7 +173,11 @@ export function classifyFailureOutput(output) {
161
173
  * no failure in that incomplete environment can safely become pre-existing forgiveness. Keep this
162
174
  * separate from `classifyFailureOutput` so changing capture policy cannot move gate verdicts.
163
175
  */
164
- const VITEST_ECHO_BLOCK_RE = /^\s*std(?:out|err)\s+\|\s+\S+\.(?:test|spec)\.[cm]?[jt]sx?\s+>\s+\S/;
176
+ // OBS-888 row 1: vitest 3.2.7 heads an echo block `std{out,err} | <file> > <test>` when the log is
177
+ // attributed to a test, `stderr | unknown test` when it is not, `stderr | <file>` for file-level output
178
+ // and `stderr | <task id>` (digits and underscores) when the reporter no longer knows the task
179
+ // (dist/chunks/index.*.js, `headerText`). The stripper knew only the first form.
180
+ const VITEST_ECHO_BLOCK_RE = /^\s*std(?:out|err)\s+\|\s+(?:unknown test\s*$|\d+_[\d_]+\s*$|\S+\.(?:test|spec)\.[cm]?[jt]sx?(?:\s*$|\s+>\s+\S))/;
165
181
  /** Keep runner output while omitting Vitest's echoed test-owned stdout/stderr diagnostic blocks. */
166
182
  const withoutVitestEchoBlocks = (output) => {
167
183
  const outside = [];
@@ -186,7 +202,7 @@ const captureInvalidatingLines = (output) => {
186
202
  const invalidating = [];
187
203
  for (const line of withoutVitestEchoBlocks(output)) {
188
204
  const clean = line.replace(ANSI_RE, "");
189
- if (!PASS_LINE_RE.test(clean) && CAPTURE_EXHAUSTION_RE.test(clean))
205
+ if (!PASS_LINE_RE.test(clean) && !OPERATOR_LINE_RE.test(clean) && CAPTURE_EXHAUSTION_RE.test(clean))
190
206
  invalidating.push(line);
191
207
  }
192
208
  return invalidating;
@@ -230,16 +246,18 @@ export const UNRECOGNIZED_FAILURE = "<unrecognized failure output>";
230
246
  export function fingerprint(output) {
231
247
  const lines = withoutVitestEchoBlocks(output)
232
248
  .map((l) => l.replace(ANSI_RE, ""))
233
- .filter((l) => !PASS_LINE_RE.test(l));
249
+ .filter((l) => !PASS_LINE_RE.test(l) && !OPERATOR_LINE_RE.test(l));
234
250
  // GATE-FIX-4 DEFECT 4: every line is read twice — as printed, and with a turbo `<pkg>:<task>:`
235
251
  // prefix removed. A recognized stripped line fingerprints as its STRIPPED text, so the same
236
252
  // failure fingerprints identically whether turbo prefixed it or a bare runner printed it; the
237
253
  // prefixed form keeps fingerprinting too (baseline-recorded package-level reds stay forgivable).
238
254
  const shaped = [];
239
255
  for (const l of lines) {
256
+ const stripped = stripTurboPrefix(l);
257
+ if (stripped !== undefined && OPERATOR_LINE_RE.test(stripped))
258
+ continue; // OBS-888: operator prose under a turbo prefix
240
259
  if (isFingerprintShaped(l))
241
260
  shaped.push(l);
242
- const stripped = stripTurboPrefix(l);
243
261
  if (stripped !== undefined && !PASS_LINE_RE.test(stripped) && isFingerprintShaped(stripped))
244
262
  shaped.push(stripped);
245
263
  }
@@ -576,7 +594,8 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
576
594
  // result rather than re-derived after the fact. The skip row above ran no command and therefore
577
595
  // states no capacity — a row that never divided the machine must not claim that it did.
578
596
  const record = (g) => {
579
- results.push(r.capacity ? { ...g, capacity: r.capacity } : g);
597
+ const withReap = r.reapedGroup ? { ...g, meta: { ...g.meta, reapedGroup: true } } : g;
598
+ results.push(r.capacity ? { ...withReap, capacity: r.capacity } : withReap);
580
599
  };
581
600
  // …and whether the entry that would forgive this command was measured in the same world. A
582
601
  // baseline captured under a different fork cap forgives nothing: its fingerprints describe a
@@ -52,7 +52,9 @@ export declare function modelId(model: string): string;
52
52
  export declare function modelProvider(model: string, fallback?: string): string;
53
53
  export declare function pickReviewer(author: Assignment, channels: BillingChannel[], exclude?: string[], // v1.1 failover: reviewer channels that already produced garbage for this task
54
54
  prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
55
- floor?: Tier): BillingChannel | null;
55
+ floor?: Tier, // task-declared only; config floors govern workers and must not silently move review seats
56
+ history?: string[], // run-scoped picks, oldest to newest; empty preserves the established ranking
57
+ onSeat?: (seat: number) => void): BillingChannel | null;
56
58
  export type ReviewUnparseableCause = VerdictUnparseableCause;
57
59
  /**
58
60
  * This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
@@ -60,4 +62,4 @@ export type ReviewUnparseableCause = VerdictUnparseableCause;
60
62
  * judgement rather than a guarantee made by this renderer.
61
63
  */
62
64
  export declare function renderDeclaredWriteScope(files: ReadonlyArray<string>): string;
63
- export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string): Promise<GateResult>;
65
+ export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string, reviewHistory?: string[]): Promise<GateResult>;
@@ -192,7 +192,9 @@ function reviewPreferIndex(c, prefer) {
192
192
  }
193
193
  export function pickReviewer(author, channels, exclude = [], // v1.1 failover: reviewer channels that already produced garbage for this task
194
194
  prefer = [], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
195
- floor) {
195
+ floor, // task-declared only; config floors govern workers and must not silently move review seats
196
+ history = [], // run-scoped picks, oldest to newest; empty preserves the established ranking
197
+ onSeat) {
196
198
  // FLEET-05 success criterion 2: an author not resolvable in the channel list yields NO reviewer.
197
199
  // The old `?? author.adapter` fallback compared an adapter id to vendor names, matched nothing, and
198
200
  // admitted every reviewer — including the author's own channel (fail-OPEN). null lands on reviewGate's
@@ -200,18 +202,23 @@ floor) {
200
202
  const authorChannel = channels.find((c) => c.adapter === author.adapter && c.model === author.model);
201
203
  if (!authorChannel)
202
204
  return null;
203
- // Failover also guards true provider; the initial pick keeps the established stamped-vendor contract.
204
205
  const authorProvider = modelProvider(author.model, authorChannel.vendor);
205
- return (channels
206
+ const ranked = channels
206
207
  // two independent axes: different vendor AND different base-model identity (ADDED TO the vendor
207
- // rule, never replacing it — a future edit can't silently drop either). The diversity filter runs
208
- // BEFORE preference ranking: prefer sorts survivors only, so no entry can resurrect an excluded channel.
208
+ // rule, never replacing it — a future edit can't silently drop either). Failover additionally guards
209
+ // true provider identity; the initial pick keeps the established stamped-vendor contract. The diversity
210
+ // filter runs BEFORE preference ranking, so prefer cannot resurrect an excluded channel.
209
211
  .filter((c) => c.vendor !== authorChannel.vendor
210
212
  && (exclude.length === 0 || modelProvider(c.model, c.vendor) !== authorProvider)
211
213
  && modelId(c.model) !== modelId(author.model)
212
214
  && !exclude.includes(channelKey(c))
213
215
  && (floor === undefined || TIER_RANK[c.tier] >= TIER_RANK[floor]))
214
- .sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b))[0] ?? null);
216
+ .sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b));
217
+ const reviewer = [...ranked].sort((a, b) => history.lastIndexOf(channelKey(a)) - history.lastIndexOf(channelKey(b))
218
+ || ranked.indexOf(a) - ranked.indexOf(b))[0] ?? null;
219
+ if (reviewer)
220
+ onSeat?.(ranked.indexOf(reviewer) + 1);
221
+ return reviewer;
215
222
  }
216
223
  /**
217
224
  * This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
@@ -229,7 +236,7 @@ ${files.map((path) => `- ${path}`).join("\n")}`;
229
236
  export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
230
237
  // OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
231
238
  // direct tests) skips persistence and changes nothing else.
232
- artifactDir) {
239
+ artifactDir, reviewHistory) {
233
240
  // R3 (OBS-186): participation is keyed on PATHS. The compiler's assignment comes from the DECLARED
234
241
  // files[]; the operator's floor may RAISE it to full and can never lower it. `complexityThreshold` is
235
242
  // retired — the branch that returned a green skip on a complexity comparison is gone, and with it the
@@ -301,7 +308,8 @@ artifactDir) {
301
308
  // A reviewer floor is opt-in at the task. Applying cfg.routing.floors here would change the
302
309
  // historical seat for every task that never asked for review-tier coupling.
303
310
  const reviewerFloor = task.routingHints?.floor;
304
- const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor);
311
+ let rotationSeat;
312
+ const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor, reviewHistory, reviewHistory ? (seat) => { rotationSeat = seat; } : undefined);
305
313
  if (!reviewer) {
306
314
  // meta.noEligibleReviewer lets run-gates' review-retry keep the ORIGINAL unparseable result when
307
315
  // the retry finds no second seat — a truthful cause beats a synthetic no-reviewer failure.
@@ -312,6 +320,8 @@ artifactDir) {
312
320
  ? { gate: "review", pass: false, details: `unreadable — ${reason}; set review.required:false to waive`, meta: { noEligibleReviewer: true, unreadable: true, ...(reviewerFloor ? { reviewerFloor } : {}) } }
313
321
  : { gate: "review", pass: true, details: `WARNING: ${reason} — review waived by config`, meta: { noEligibleReviewer: true, ...(reviewerFloor ? { reviewerFloor } : {}) } };
314
322
  }
323
+ reviewHistory?.push(channelKey(reviewer));
324
+ const rotationMeta = rotationSeat === undefined ? {} : { rotationSeat };
315
325
  const measuredDiff = await fetchTaskDiff(worktree, baseRef, task.files);
316
326
  // Keep the reader payload identical to the text charged to the strict cap:
317
327
  // whole-file source deletions are represented by their citable operation fact.
@@ -362,9 +372,8 @@ The top-level comments array is optional. Use it only for actionable line-anchor
362
372
  // frontier reviewers routinely need >5min on a configured-cap-sized diff, and `claude -p` buffers all
363
373
  // output until completion — runLlm's 300s default killed reviews mid-flight, returning empty
364
374
  // stdout that read as "unparseable" and escalated to re-implementation of green code
365
- // (run-20260709-104447 P87-09). ponytail: literal 15min; make it cfg.review.timeoutMs if a
366
- // second knob-turner appears.
367
- 900_000);
375
+ // (run-20260709-104447 P87-09). The configured ceiling defaults to that measured 15 minutes.
376
+ cfg.review.timeoutMs);
368
377
  const raw = llm.output;
369
378
  const provider = modelProvider(reviewer.model, reviewer.vendor);
370
379
  const v = extractVerdictJson(raw, nonce);
@@ -393,15 +402,17 @@ The top-level comments array is optional. Use it only for actionable line-anchor
393
402
  return {
394
403
  gate: "review",
395
404
  pass: false,
396
- details: `${failure} (reviewer ${reviewer.adapter}:${reviewer.model}; vendor: ${reviewer.vendor}; provider: ${provider}; cause: ${cause}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
405
+ details: `${failure} (reviewer ${reviewer.adapter}:${reviewer.model}; vendor: ${reviewer.vendor}; provider: ${provider}; cause: ${cause}${cause === "timeout" ? `; killed at configured review timeout ${cfg.review.timeoutMs}ms` : ""}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
397
406
  meta: {
398
407
  ...policyMeta,
408
+ ...rotationMeta,
399
409
  reviewer: channelKey(reviewer),
400
410
  vendor: reviewer.vendor,
401
411
  provider,
402
412
  unparseable: true,
403
413
  cause,
404
414
  ...(cause === "empty-output" ? { bytes } : {}),
415
+ ...(cause === "timeout" ? { timeoutMs: cfg.review.timeoutMs } : {}),
405
416
  ...(concludedOnInactivity ? { classification: "infra", infra: true } : {}),
406
417
  },
407
418
  };
@@ -414,6 +425,6 @@ The top-level comments array is optional. Use it only for actionable line-anchor
414
425
  gate: "review",
415
426
  pass: decided.pass,
416
427
  details: appendAnchoredReview(prose, v),
417
- meta: { ...policyMeta, reviewer: channelKey(reviewer), vendor: reviewer.vendor, provider },
428
+ meta: { ...policyMeta, ...rotationMeta, reviewer: channelKey(reviewer), vendor: reviewer.vendor, provider },
418
429
  };
419
430
  }
@@ -13,13 +13,15 @@ export declare function resetLoadProviderForTests(): void;
13
13
  * intervals and nothing between them, so the composite `test` gate (a selected screen, then other
14
14
  * gates, then the full suite) reports the two suites' cost rather than the span containing them —
15
15
  * and no consumer has to re-derive a duration by subtracting journal timestamps, which measures the
16
- * queue as well as the work. The load samples bracket the FIRST interval's start and the LAST
17
- * interval's end: start is what a scheduler would have decided on, end is the state it left behind.
16
+ * queue as well as the work. Load is sampled at each interval's endpoints and every second within it;
17
+ * start preserves the scheduling input while max and mean retain sustained interior saturation.
18
18
  */
19
19
  export interface GateTelemetry {
20
20
  durationMs: number;
21
21
  load1Start: number;
22
22
  load1End: number;
23
+ load1Max: number;
24
+ load1Mean: number;
23
25
  }
24
26
  export type GateEvent = {
25
27
  phase: "start";
@@ -51,6 +53,7 @@ export interface GateContext {
51
53
  cfg: TickmarkrConfig;
52
54
  via?: GateVia;
53
55
  excludeReviewers?: string[];
56
+ reviewHistory?: string[];
54
57
  artifactDir?: string;
55
58
  pipeline?: "v185" | "legacy";
56
59
  selectTests?: boolean;
@@ -208,24 +208,42 @@ export async function runGates(task, ctx) {
208
208
  // executing is added HERE, at the call site that runs it, so a gate that runs twice (the test
209
209
  // gate's screen and its full suite) sums to its own cost and never to the span between them.
210
210
  const spans = new Map();
211
+ const loadSamples = new Map();
211
212
  // The test gate's two halves, kept apart as well as summed: `durationMs` alone cannot say whether
212
213
  // a slow round was a slow subset or a slow full suite, and the parked scheduler's threshold is
213
214
  // defined over the full-suite cost.
214
215
  let selectedDurationMs;
215
216
  let fullDurationMs;
216
- const measure = async (gate, run) => {
217
+ const startMeasurement = () => {
217
218
  const at = Date.now();
218
- const load1Start = loadProvider();
219
+ const samples = [loadProvider()];
220
+ const timer = setInterval(() => samples.push(loadProvider()), 1_000);
221
+ timer.unref();
222
+ return () => {
223
+ clearInterval(timer);
224
+ samples.push(loadProvider());
225
+ return { durationMs: Date.now() - at, samples };
226
+ };
227
+ };
228
+ const addMeasurement = (gate, measured) => {
229
+ const prior = spans.get(gate);
230
+ const samples = [...(loadSamples.get(gate) ?? []), ...measured.samples];
231
+ loadSamples.set(gate, samples);
232
+ spans.set(gate, {
233
+ durationMs: (prior?.durationMs ?? 0) + measured.durationMs,
234
+ load1Start: samples[0],
235
+ load1End: samples[samples.length - 1],
236
+ load1Max: Math.max(...samples),
237
+ load1Mean: samples.reduce((sum, value) => sum + value, 0) / samples.length,
238
+ });
239
+ };
240
+ const measure = async (gate, run) => {
241
+ const finish = startMeasurement();
219
242
  try {
220
243
  return await run();
221
244
  }
222
245
  finally {
223
- const prior = spans.get(gate);
224
- spans.set(gate, {
225
- durationMs: (prior?.durationMs ?? 0) + (Date.now() - at),
226
- load1Start: prior?.load1Start ?? load1Start,
227
- load1End: loadProvider(),
228
- });
246
+ addMeasurement(gate, finish());
229
247
  }
230
248
  };
231
249
  // The measurement is attached at the ONE seam every result leaves this function through, so a
@@ -347,12 +365,11 @@ export async function runGates(task, ctx) {
347
365
  // suppresses them anyway; split compareToBaseline only if a tool gate ever gets slow.
348
366
  // ponytail: legacy runs adjacent tools in ONE compareToBaseline call, so there is one interval
349
367
  // to measure and each of its gates carries it. Split it only if this branch ever stops batching.
350
- const batchAt = Date.now();
351
- const batchLoadStart = loadProvider();
368
+ const finish = startMeasurement();
352
369
  const toolResults = await compareToBaseline(ctx.worktree, commands, ctx.baseline, [...gates]);
353
- const batch = { durationMs: Date.now() - batchAt, load1Start: batchLoadStart, load1End: loadProvider() };
370
+ const batch = finish();
354
371
  for (const g of gates)
355
- spans.set(g, batch);
372
+ addMeasurement(g, batch);
356
373
  // The same refusal AFTER the commands, because a green command can dirty the tree the check
357
374
  // above just proved clean. Batched, legacy cannot say WHICH command did it, so the refusal
358
375
  // lands on the last gate that had one — the round dies there either way. A red battery is
@@ -575,7 +592,7 @@ export async function runGates(task, ctx) {
575
592
  invocations.push(...captured.invocations);
576
593
  return captured.value;
577
594
  };
578
- let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir));
595
+ let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir, ctx.reviewHistory));
579
596
  // OBS-193/574: an unparseable review verdict retries the REVIEW exactly once, preferring a
580
597
  // different adapter. Only a single-adapter eligible pool may fall back to another channel on the
581
598
  // flaked adapter. The flaked verdict never enters results; an exhausted pool preserves its cause.
@@ -598,7 +615,7 @@ export async function runGates(task, ctx) {
598
615
  const crossAdapter = pickReviewer(ctx.author, ctx.channels, [...priorExclusions, ...adapterExclusions], ctx.cfg.review.prefer ?? [], task.routingHints?.floor);
599
616
  const exclusion = crossAdapter ? "adapter" : "channel";
600
617
  const retryExclusions = [...priorExclusions, ...(crossAdapter ? adapterExclusions : [flaked])];
601
- const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, retryExclusions, ctx.artifactDir));
618
+ const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, retryExclusions, ctx.artifactDir, ctx.reviewHistory));
602
619
  if (second.meta?.noEligibleReviewer !== true) {
603
620
  const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
604
621
  const route = exclusion === "adapter"
@@ -630,11 +647,11 @@ export async function runGates(task, ctx) {
630
647
  // The check runs BEFORE any gate, so on a clean tree it belongs to no gate: charging every round's
631
648
  // first gate for it would inflate the one measurement the parked recalibrations key on. It becomes
632
649
  // that gate's interval only on the path where it IS what the gate did — the refusal below.
633
- const entryAt = Date.now();
634
- const entryLoad = loadProvider();
650
+ const finishEntry = startMeasurement();
635
651
  const entryDirt = sequence.length ? await dirtyWorktree() : undefined;
652
+ const entryMeasurement = finishEntry();
636
653
  if (entryDirt) {
637
- spans.set(sequence[0], { durationMs: Date.now() - entryAt, load1Start: entryLoad, load1End: loadProvider() });
654
+ addMeasurement(sequence[0], entryMeasurement);
638
655
  await emitStart(sequence[0]);
639
656
  await record(dirtyRefusal(sequence[0], entryDirt));
640
657
  return done();
@@ -5,6 +5,10 @@ export interface Disallowed {
5
5
  entry: string;
6
6
  }
7
7
  export type PreferenceRole = "worker" | "judge" | "review" | "consult";
8
+ export declare function routingModelProvider(model: string, fallback?: string): string;
9
+ export declare const modelRouteIdentity: (model: string, fallback?: string) => string;
10
+ export declare const channelRouteIdentity: (key: string, fallback?: string) => string;
11
+ export declare function routingEntrySeatLines(cfg: TickmarkrConfig): string[];
8
12
  export declare function excludedChannels(cfg: TickmarkrConfig, adapters: {
9
13
  id: string;
10
14
  }[] | string[], health: Record<string, AuthHealth>): {
@@ -1,6 +1,46 @@
1
1
  import { channelKey, channelsFromConfig } from "../adapters/types.js";
2
2
  import { validateGraph } from "../graph/schema.js";
3
3
  import { route, RoutingError } from "./router.js";
4
+ const PREFERENCE_ROLES = ["worker", "judge", "review", "consult"];
5
+ // Keep routing retries on the same identity review diversity uses: the served provider plus the
6
+ // unprefixed model id. Gate review owns the original modelProvider policy; this dependency-leaf copy
7
+ // avoids importing gates back into route/run and must move with that helper when the scope permits.
8
+ export function routingModelProvider(model, fallback = "unknown") {
9
+ const id = model.toLowerCase();
10
+ const prefix = id.includes("/") ? id.slice(0, id.indexOf("/")) : "";
11
+ if (prefix === "openai" || prefix === "openai-codex" || /^(?:gpt|o\d)/.test(id))
12
+ return "openai";
13
+ if (prefix === "anthropic" || /^(?:claude|opus|sonnet|haiku|fable)(?:-|$)/.test(id))
14
+ return "anthropic";
15
+ if (prefix === "google" || /^gemini(?:-|$)/.test(id))
16
+ return "google";
17
+ if (prefix === "xai" || /^grok(?:-|$)/.test(id))
18
+ return "xai";
19
+ if (["zai", "zhipu", "zai-coding-plan"].includes(prefix) || /^glm(?:-|$)/.test(id))
20
+ return "zhipu";
21
+ if (["kimi-code", "moonshot"].includes(prefix) || /^kimi(?:-|$)/.test(id))
22
+ return "moonshot";
23
+ return fallback;
24
+ }
25
+ export const modelRouteIdentity = (model, fallback = "unknown") => `${routingModelProvider(model, fallback)}/${model.slice(model.lastIndexOf("/") + 1).toLowerCase()}`;
26
+ export const channelRouteIdentity = (key, fallback = "unknown") => {
27
+ const i = key.indexOf(":");
28
+ return i < 0 ? key : modelRouteIdentity(key.slice(i + 1), fallback);
29
+ };
30
+ export function routingEntrySeatLines(cfg) {
31
+ const lines = [];
32
+ const add = (path, entries, roles) => {
33
+ for (const entry of entries ?? [])
34
+ lines.push(`${path} '${entry}' reaches seats: ${roles.join(", ")}`);
35
+ };
36
+ add("routing.allow.adapters", cfg.routing.allow?.adapters, PREFERENCE_ROLES);
37
+ add("routing.allow.models", cfg.routing.allow?.models, PREFERENCE_ROLES);
38
+ add("routing.deny.adapters", cfg.routing.deny?.adapters, PREFERENCE_ROLES);
39
+ add("routing.deny.models", cfg.routing.deny?.models, PREFERENCE_ROLES);
40
+ add("routing.deny.workers.adapters", cfg.routing.deny?.workers?.adapters, ["worker"]);
41
+ add("routing.deny.workers.models", cfg.routing.deny?.workers?.models, ["worker"]);
42
+ return lines;
43
+ }
4
44
  const adapterIds = (adapters) => typeof adapters[0] === "string" ? adapters : adapters.map((a) => a.id);
5
45
  export function excludedChannels(cfg, adapters, health) {
6
46
  const { allow, deny } = cfg.routing;
@@ -1,6 +1,6 @@
1
1
  import { channelKey, channelsFromConfig } from "../adapters/types.js";
2
2
  import { TIER_RANK } from "../config/config.js";
3
- import { disallowedBy } from "./preference.js";
3
+ import { channelRouteIdentity, disallowedBy, modelRouteIdentity, routingModelProvider } from "./preference.js";
4
4
  import { cellOf, EXPLORE_CAP, explorationBonus, learnedScore, MIN_SAMPLES } from "./profile.js";
5
5
  export const NO_EXPLORE_ENV = "TICKMARKR_NO_EXPLORE";
6
6
  // OBS-89 (v1.60): the TICKMARKR_QUALITY variable is RETIRED — nothing in src reads it anymore and
@@ -337,7 +337,20 @@ export function nextChannel(current, task, cfg, channels, tried, profile, exclud
337
337
  // profile-dependent filter. NO exploration bonus here (route():110 has one; a probe on the
338
338
  // failure path would spend a real retry). Absent profile ⇒ every score is 0 ⇒ third key
339
339
  // all-ties ⇒ the stable sort preserves the exact v1.7 candidate ORDER.
340
- const pool = channels.filter((c) => !tried.includes(channelKey(c)) && TIER_RANK[c.tier] >= TIER_RANK[current.tier]);
340
+ const triedKeys = new Set(tried);
341
+ const triedIdentities = new Set(tried.map((key) => {
342
+ const channel = channels.find((c) => channelKey(c) === key);
343
+ return channel ? modelRouteIdentity(channel.model, channel.vendor) : channelRouteIdentity(key);
344
+ }));
345
+ // excludeAdapter is expanded by the daemon into every channel key of the failed adapter. Once that
346
+ // complete set is present, the outage follows the current served provider across gateway aliases.
347
+ const currentAdapterExcluded = channels.some((c) => c.adapter === current.adapter)
348
+ && channels.filter((c) => c.adapter === current.adapter).every((c) => triedKeys.has(channelKey(c)));
349
+ const currentChannel = channels.find((c) => c.adapter === current.adapter && c.model === current.model);
350
+ const excludedProvider = currentAdapterExcluded ? routingModelProvider(current.model, currentChannel?.vendor) : undefined;
351
+ const pool = channels.filter((c) => !triedIdentities.has(modelRouteIdentity(c.model, c.vendor))
352
+ && (!excludedProvider || routingModelProvider(c.model, c.vendor) !== excludedProvider)
353
+ && TIER_RANK[c.tier] >= TIER_RANK[current.tier]);
341
354
  const scores = new Map(pool.map((c) => [channelKey(c), profile ? learnedScore(profile, task.shape, channelKey(c), c.channel, { availWeight: cfg.routing.learnedTuning?.availWeight }) : 0]));
342
355
  const scoreOf = (c) => scores.get(channelKey(c));
343
356
  const candidates = pool.sort((a, b) => TIER_RANK[a.tier] - TIER_RANK[b.tier] || marginalCostRank(a) - marginalCostRank(b) || scoreOf(b) - scoreOf(a));
@@ -9,6 +9,7 @@ export interface ConsultVerdict {
9
9
  reason?: string;
10
10
  guidance?: string;
11
11
  excludeAdapter?: string;
12
+ excludeProvider?: string;
12
13
  adapter?: string;
13
14
  model?: string;
14
15
  vendor?: string;
@@ -4,7 +4,7 @@ import { getAdapter } from "../adapters/registry.js";
4
4
  import { bannerShell, paneDispatchCommand } from "../brand.js";
5
5
  import { dewrapPaneVerdict, extractVerdictJson, gateExitTrailer, gatePaneName, generateVerdictNonce, verdictNonceLine } from "../gates/llm.js";
6
6
  import { classifyVerdictCause } from "../gates/verdict-cause.js";
7
- import { disallowedBy } from "../route/preference.js";
7
+ import { disallowedBy, routingModelProvider } from "../route/preference.js";
8
8
  import { sh } from "./git.js";
9
9
  import { redactSecrets } from "./redact.js";
10
10
  import { filterLlmTranscript } from "./stall.js";
@@ -59,6 +59,22 @@ export function augmentRetryBrief(feedback, opts) {
59
59
  return parts.join("\n\n");
60
60
  }
61
61
  const ACTIONS = ["retry", "reroute", "decompose", "human"];
62
+ function excludedProviderFromDossier(d, adapter) {
63
+ try {
64
+ const events = JSON.parse(d.journalTail);
65
+ const assignment = [...events].reverse().find((event) => event.event === "task-dispatch"
66
+ && event.data?.assignment?.adapter === adapter
67
+ && typeof event.data.assignment.model === "string")?.data?.assignment;
68
+ if (typeof assignment?.model !== "string")
69
+ return undefined;
70
+ const provider = routingModelProvider(assignment.model);
71
+ return provider === "unknown" ? undefined : provider;
72
+ }
73
+ catch {
74
+ // Legacy/non-JSON dossier tails retain the adapter exclusion without inventing a provider.
75
+ return undefined;
76
+ }
77
+ }
62
78
  export function parseConsultVerdict(out, nonce) {
63
79
  const v = extractVerdictJson(out, nonce);
64
80
  if (!v)
@@ -110,10 +126,10 @@ ${d.journalTail}
110
126
  Verdict meanings: retry = same assignment with your notes as feedback; reroute = different CLI/model;
111
127
  decompose = task too big, needs human re-planning; human = a person must look at this.
112
128
 
113
- On reroute only, optional excludeAdapter is an adapter id (e.g. "cursor-agent") that bans EVERY
114
- channel of that adapter for this task. Use it for environmental failures ("the CLI is blocked",
115
- trust dialog, broken install) — not when a single model produced bad code. Omit for model-level
116
- reroutes so other models of the same adapter remain eligible.
129
+ On reroute only, optional excludeAdapter is the failed adapter id (e.g. "cursor-agent"). Tickmarkr
130
+ resolves the adapter's current model to its provider and bans that provider for this task. Use it for
131
+ environmental/provider failures ("the CLI is blocked", trust dialog, broken install) — not when a
132
+ single model produced bad code. Omit for model-level reroutes so other models remain eligible.
117
133
 
118
134
  ${verdictNonceLine(nonce)}
119
135
 
@@ -230,8 +246,19 @@ opts = {}) {
230
246
  for (const [i, seat] of allowedSeats.entries()) {
231
247
  try {
232
248
  const parsed = await invokeSeat(seat.adapter, seat.model, i);
233
- if (parsed.verdict)
234
- return { ...parsed.verdict, ...seatIdentity(seat) };
249
+ if (parsed.verdict) {
250
+ const excludeProvider = parsed.verdict.excludeAdapter
251
+ ? excludedProviderFromDossier(d, parsed.verdict.excludeAdapter)
252
+ : undefined;
253
+ return {
254
+ ...parsed.verdict,
255
+ ...(excludeProvider ? {
256
+ excludeProvider,
257
+ notes: `${parsed.verdict.notes} — excluded provider ${excludeProvider}`,
258
+ } : {}),
259
+ ...seatIdentity(seat),
260
+ };
261
+ }
235
262
  }
236
263
  catch {
237
264
  // failed seat (unknown adapter, dead driver/pane, shell error) — fall to the next entry
@@ -68,6 +68,7 @@ export declare function formatSummary(s: RunSummary): string;
68
68
  * run id (cli/commands/status.ts positionalRunId), so naming it here is what stops the board from
69
69
  * following the newest journal in a repo that already carries a second, newer run — a board showing
70
70
  * the wrong run is a recorded incident (skills/tickmarkr-overseer/SKILL.md). */
71
+ export declare const daemonEntrypoint: string;
71
72
  export declare const watchCommand: (runId: string) => string;
72
73
  /**
73
74
  * R3 (OBS-186): a gate that DECLINED to run is not a gate that failed. The review gate's skip branch
@@ -103,6 +104,9 @@ export declare const gateSatisfied: (g: GateResult) => boolean;
103
104
  */
104
105
  export declare function decisiveReviewRounds(events: JournalEvent[]): JournalEvent[];
105
106
  export declare const SUITE_POLL_MS = 250;
107
+ export declare const SUITE_WAIT_CEILING_MS = 600000;
108
+ export declare const setSuiteWaitCeilingForTests: (ms: number) => void;
109
+ export declare const resetSuiteWaitCeilingForTests: () => void;
106
110
  export declare const APPROVAL_POLL_MS = 250;
107
111
  export declare const EARLY_LAUNCH_LIVENESS_MS = 60000;
108
112
  /** Test seam — lowers the empty-pane liveness window without sleeping 60s per case. */