tickmarkr 2.6.1 → 2.6.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (85) hide show
  1. package/README.md +14 -3
  2. package/dist/adapters/catalog-remote.js +89 -47
  3. package/dist/adapters/claude-code.js +9 -6
  4. package/dist/adapters/codex.js +7 -4
  5. package/dist/adapters/prompt.d.ts +1 -0
  6. package/dist/adapters/prompt.js +14 -6
  7. package/dist/adapters/registry.js +3 -3
  8. package/dist/adapters/types.d.ts +12 -4
  9. package/dist/adapters/types.js +6 -0
  10. package/dist/cli/commands/approve.d.ts +5 -1
  11. package/dist/cli/commands/approve.js +66 -23
  12. package/dist/cli/commands/compile.js +13 -3
  13. package/dist/cli/commands/doctor.d.ts +2 -0
  14. package/dist/cli/commands/doctor.js +11 -3
  15. package/dist/cli/commands/fleet.js +45 -7
  16. package/dist/cli/commands/report.js +18 -2
  17. package/dist/cli/commands/resume.js +4 -2
  18. package/dist/cli/commands/status.js +24 -19
  19. package/dist/cli/help.d.ts +2 -0
  20. package/dist/cli/help.js +9 -2
  21. package/dist/config/config.d.ts +20 -0
  22. package/dist/config/config.js +47 -8
  23. package/dist/config/fleet-overlay.d.ts +1 -0
  24. package/dist/config/fleet-overlay.js +56 -0
  25. package/dist/drivers/orca.d.ts +26 -1
  26. package/dist/drivers/orca.js +193 -60
  27. package/dist/eval/canary.d.ts +2 -1
  28. package/dist/eval/canary.js +2 -2
  29. package/dist/eval/dispatch.js +1 -0
  30. package/dist/gates/acceptance.d.ts +2 -1
  31. package/dist/gates/acceptance.js +7 -2
  32. package/dist/gates/baseline.d.ts +12 -1
  33. package/dist/gates/baseline.js +11 -4
  34. package/dist/gates/llm.d.ts +5 -4
  35. package/dist/gates/llm.js +13 -13
  36. package/dist/gates/review.d.ts +8 -0
  37. package/dist/gates/review.js +40 -4
  38. package/dist/gates/run-gates.d.ts +2 -1
  39. package/dist/gates/run-gates.js +27 -13
  40. package/dist/gates/test-manifest.d.ts +3 -1
  41. package/dist/gates/test-manifest.js +9 -2
  42. package/dist/graph/schema.d.ts +2 -0
  43. package/dist/graph/schema.js +2 -0
  44. package/dist/plan/scope.js +2 -2
  45. package/dist/route/preference.d.ts +20 -2
  46. package/dist/route/preference.js +48 -13
  47. package/dist/route/router.js +30 -15
  48. package/dist/run/consult.d.ts +13 -1
  49. package/dist/run/consult.js +14 -5
  50. package/dist/run/daemon.d.ts +37 -2
  51. package/dist/run/daemon.js +579 -141
  52. package/dist/run/git.d.ts +8 -0
  53. package/dist/run/git.js +14 -0
  54. package/dist/run/journal.d.ts +126 -3
  55. package/dist/run/journal.js +410 -37
  56. package/dist/run/merge.d.ts +3 -1
  57. package/dist/run/merge.js +3 -2
  58. package/dist/run/operator-summary.d.ts +3 -0
  59. package/dist/run/operator-summary.js +3 -1
  60. package/dist/run/protocol.d.ts +31 -1
  61. package/dist/run/protocol.js +3 -1
  62. package/dist/run/supervision.d.ts +7 -1
  63. package/dist/run/supervision.js +5 -2
  64. package/dist/tui/cockpit/board.js +3 -3
  65. package/dist/tui/cockpit/decision-actions.d.ts +8 -5
  66. package/dist/tui/cockpit/decision-actions.js +55 -32
  67. package/dist/tui/cockpit/derive.js +13 -2
  68. package/dist/tui/cockpit/live-runtime.d.ts +10 -0
  69. package/dist/tui/cockpit/live-runtime.js +50 -3
  70. package/dist/tui/cockpit/run-cockpit.d.ts +3 -0
  71. package/dist/tui/cockpit/run-cockpit.js +26 -1
  72. package/dist/tui/cockpit/run-view.d.ts +9 -3
  73. package/dist/tui/cockpit/run-view.js +60 -7
  74. package/dist/tui/cockpit/setup-cockpit.d.ts +4 -0
  75. package/dist/tui/cockpit/setup-cockpit.js +6 -3
  76. package/dist/tui/ink/fleet-app.d.ts +15 -3
  77. package/dist/tui/ink/fleet-app.js +91 -22
  78. package/package.json +2 -1
  79. package/schema/config.schema.json +818 -0
  80. package/skills/tickmarkr-loop/SKILL.md +8 -2
  81. package/skills/tickmarkr-overseer/SKILL.md +42 -0
  82. package/skills/tickmarkr-overseer/scripts/classify-vitest-log.sh +88 -0
  83. package/skills/tickmarkr-overseer/scripts/context-statusline.sh +81 -0
  84. package/skills/tickmarkr-overseer/scripts/grade-ci.sh +36 -34
  85. package/skills/tickmarkr-overseer/scripts/watch-journal.sh +6 -4
package/dist/gates/llm.js CHANGED
@@ -304,12 +304,12 @@ export const REVIEW_FIRST_LIVENESS_MS = 30_000;
304
304
  // ceiling. Below this many seat-authored bytes at the first beat the seat is `silent` — demoted and
305
305
  // re-routed then, not at the ceiling. Pane path only; a headless runner buffers and keeps its ceiling.
306
306
  export const REVIEW_SILENT_BYTE_FLOOR = 64;
307
- async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 300000) {
307
+ async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 300000, effort) {
308
308
  const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
309
309
  try {
310
310
  const pf = join(dir, "prompt.md");
311
311
  writeFileSync(pf, prompt);
312
- const r = await sh(adapter.headlessCommand(pf, model), cwd, timeoutMs);
312
+ const r = await sh(adapter.headlessCommand(pf, model, effort), cwd, timeoutMs);
313
313
  const output = r.stdout + "\n" + r.stderr;
314
314
  const nonce = extractPromptNonce(prompt) ?? "";
315
315
  return { output, exitCode: r.code, timedOut: r.timedOut === true,
@@ -319,12 +319,12 @@ async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 3000
319
319
  rmSync(dir, { recursive: true, force: true });
320
320
  }
321
321
  }
322
- export async function runHeadless(adapter, model, prompt, cwd, timeoutMs = 300000) {
323
- return (await runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs)).output;
322
+ export async function runHeadless(adapter, model, prompt, cwd, timeoutMs = 300000, effort) {
323
+ return (await runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs, effort)).output;
324
324
  }
325
325
  // v1.1 default path: the same headless CLI call, but dispatched through the driver
326
326
  // as a visible named agent (herdr pane), with the quote-split completion wrapper.
327
- async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
327
+ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs = 300000, effort) {
328
328
  const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
329
329
  let slot;
330
330
  let accountant;
@@ -338,7 +338,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
338
338
  writeFileSync(scriptPath, [
339
339
  "export BASH_SILENCE_DEPRECATION_WARNING=1",
340
340
  bannerShell(),
341
- adapter.headlessCommand(pf, model),
341
+ adapter.headlessCommand(pf, model, effort),
342
342
  gateExitTrailer(nonce),
343
343
  ].join("\n"));
344
344
  slot = await via.driver.slot(cwd, rolePaneNameFromPrompt(prompt, via.name), via.label ? { label: via.label } : undefined);
@@ -487,8 +487,8 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
487
487
  }
488
488
  }
489
489
  }
490
- export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
491
- return (await runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs)).output;
490
+ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs = 300000, effort) {
491
+ return (await runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs, effort)).output;
492
492
  }
493
493
  // OBS-155: a TUI renders the verdict as a bullet and HARD-wraps it at pane width with a 2-space
494
494
  // continuation indent, splitting words mid-token — so literal newlines land inside JSON string
@@ -549,15 +549,15 @@ export function dewrapPaneVerdict(out, nonce) {
549
549
  }
550
550
  return out;
551
551
  }
552
- export async function runLlmDetailed(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
552
+ export async function runLlmDetailed(adapter, model, prompt, cwd, via, timeoutMs = 300000, effort) {
553
553
  const result = await (via
554
- ? runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs)
555
- : runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs));
554
+ ? runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs, effort)
555
+ : runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs, effort));
556
556
  llmOutputCapture.getStore()?.push(result.output);
557
557
  return result;
558
558
  }
559
- export async function runLlm(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
560
- return (await runLlmDetailed(adapter, model, prompt, cwd, via, timeoutMs)).output;
559
+ export async function runLlm(adapter, model, prompt, cwd, via, timeoutMs = 300000, effort) {
560
+ return (await runLlmDetailed(adapter, model, prompt, cwd, via, timeoutMs, effort)).output;
561
561
  }
562
562
  export function extractJson(raw) {
563
563
  const fenced = [...raw.matchAll(/```json\s*\n([\s\S]*?)```/g)].at(-1);
@@ -135,4 +135,12 @@ export declare function renderGoalSection(goal: string, repoRoot?: string): stri
135
135
  * closure on a typo and that read as malformed — the block is what a closure list is copied from.
136
136
  */
137
137
  export declare function renderPriorMaterials(priorMaterials: readonly StructuredFinding[]): string;
138
+ /**
139
+ * OBS-1052(2): a seat that lost the top of a long brief, or believed it had already filed its review,
140
+ * answered in prose — and prose is no verdict. So the requirement, naming THIS call's nonce with a
141
+ * valid example, is both the first and the last instruction of the brief. It is best-effort wording:
142
+ * the parser stays the authority, and nothing here reads approval out of prose.
143
+ */
144
+ export declare function reviewResponseExample(nonce: string): string;
145
+ export declare function reviewResponseRequirement(nonce: string): string;
138
146
  export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string, reviewHistory?: string[], demotedReviewers?: ReadonlySet<string>, carriedFindings?: readonly StructuredFinding[], priorReviewers?: readonly PriorReviewer[], carriedAuthors?: readonly string[], operatorContext?: string): Promise<GateResult>;
@@ -11,7 +11,7 @@ import { redactSecrets } from "../run/redact.js";
11
11
  import { rankPreferredChannels, reviewPreferenceTieBreak } from "../route/role-pick.js";
12
12
  import { modelProvider } from "../route/preference.js";
13
13
  import { resolveStateDir } from "./cache.js";
14
- import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, parseAnchoredComments, runLlmDetailed, verdictNonceLine } from "./llm.js";
14
+ import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, dewrapPaneVerdict, extractVerdictJson, generateVerdictNonce, parseAnchoredComments, runLlmDetailed, verdictNonceLine } from "./llm.js";
15
15
  import { classifyVerdictCause } from "./verdict-cause.js";
16
16
  import { captureDiffCapFor, measureArtifactDiff, reviewableLogicDiff, } from "./artifact-manifest.js";
17
17
  export { isProtectedEvidence, PROTECTED_EVIDENCE_PREFIXES, REGENERABLE_CAPTURE_PATHS, setAsideReceiptPath, setAsideRegenerableCaptures, } from "./artifact-manifest.js";
@@ -373,6 +373,33 @@ ${fingerprints}
373
373
  \`\`\`
374
374
  ${priorMaterials.map((finding, i) => `${i + 1}. ${finding.note}`).join("\n\n")}`;
375
375
  }
376
+ /**
377
+ * OBS-1052(2): a seat that lost the top of a long brief, or believed it had already filed its review,
378
+ * answered in prose — and prose is no verdict. So the requirement, naming THIS call's nonce with a
379
+ * valid example, is both the first and the last instruction of the brief. It is best-effort wording:
380
+ * the parser stays the authority, and nothing here reads approval out of prose.
381
+ */
382
+ export function reviewResponseExample(nonce) {
383
+ return JSON.stringify({
384
+ nonce, approve: false, resolved: [], reraised: [],
385
+ findings: [{ note: "path/to/file.ts:42 — the defect, in one line", severity: "material", defer: false, rationale: "" }],
386
+ comments: [],
387
+ });
388
+ }
389
+ export function reviewResponseRequirement(nonce) {
390
+ return `## Response requirement
391
+ Your reply must end with exactly ONE JSON object whose "nonce" is "${nonce}" — this brief's nonce, never one from an earlier brief. A valid example (a rejection; replace every value with your own verdict):
392
+ ${reviewResponseExample(nonce)}
393
+ This holds even if you already filed or posted a review of this task elsewhere (an earlier session or brief, a PR comment): that review is not on record here, so restate it now as this JSON with nonce "${nonce}". Prose saying a review was filed or approved is recorded as no verdict; approval is never inferred from it.`;
394
+ }
395
+ // The example parses by design, so an echo of the brief (a CLI printing its prompt, a pane showing it)
396
+ // would otherwise read as the seat's own verdict — or as its participation when it wrote only prose.
397
+ // Removed verbatim or hard-wrapped (renderer whitespace and chrome between any two characters) before
398
+ // the verdict is extracted or its absence classified; the saved raw bytes keep it as evidence.
399
+ function withoutExampleEcho(raw, nonce) {
400
+ const chars = [...reviewResponseExample(nonce).replace(/\s+/g, "")];
401
+ return raw.replace(new RegExp(chars.map((c) => c.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")).join("[\\s│|]*"), "g"), "");
402
+ }
376
403
  export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
377
404
  // OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
378
405
  // direct tests) skips persistence and changes nothing else.
@@ -482,7 +509,10 @@ carriedAuthors = [], operatorContext) {
482
509
  const suiteBudget = ownTestFiles.length
483
510
  ? `You may run at most the task's own test files explicitly named in files[]; these are the only suites you may run: ${ownTestFiles.map((file) => `\`${file}\``).join(", ")}.`
484
511
  : "No suite may be run: files[] names no explicit test file owned by this task.";
512
+ const responseRequirement = reviewResponseRequirement(nonce);
485
513
  const prompt = `TICKMARKR-REVIEW
514
+ ${responseRequirement}
515
+
486
516
  You are a skeptical cross-vendor code reviewer. Another agent (vendor: ${author.adapter}) authored this diff.
487
517
  Look for correctness bugs, security issues, and acceptance-criteria gaps. Approve only if you would merge it.
488
518
 
@@ -527,6 +557,8 @@ For every prior material, put its fingerprint in exactly one of resolved (verifi
527
557
  (still a blocking defect). Use only the listed fingerprints; never omit one or put it in both lists.
528
558
  Approve iff no material finding remains and every prior material is resolved.
529
559
  The top-level comments array is optional. Use it only for actionable line-anchored feedback.
560
+
561
+ ${responseRequirement}
530
562
  `;
531
563
  // Filenames are journaled (daemon.ts lifts meta.rawPath/briefPath onto the gate-result row), so they
532
564
  // must be reproducible from the same inputs — the verdict nonce is cryptographically random and would
@@ -569,7 +601,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
569
601
  // output until completion — runLlm's 300s default killed reviews mid-flight, returning empty
570
602
  // stdout that read as "unparseable" and escalated to re-implementation of green code
571
603
  // (run-20260709-104447 P87-09). The configured ceiling defaults to that measured 15 minutes.
572
- cfg.review.timeoutMs);
604
+ cfg.review.timeoutMs, reviewer.effort);
573
605
  const raw = llm.output;
574
606
  let saved;
575
607
  if (artifactDir) {
@@ -582,7 +614,11 @@ The top-level comments array is optional. Use it only for actionable line-anchor
582
614
  }
583
615
  }
584
616
  const provider = modelProvider(reviewer.model, reviewer.vendor);
585
- const v = extractVerdictJson(raw, nonce);
617
+ // A pane's own dewrap stops at the first parseable nonce-bound object; once the example's echo is gone
618
+ // a genuinely wrapped verdict behind it is reconstructed here, exactly as llm.ts would have.
619
+ const echoFree = withoutExampleEcho(raw, nonce);
620
+ const seat = via ? dewrapPaneVerdict(echoFree, nonce) : echoFree;
621
+ const v = extractVerdictJson(seat, nonce);
586
622
  const findings = v && Array.isArray(v.findings) ? v.findings : null;
587
623
  const priorIds = priorMaterials;
588
624
  const closureInvalid = isReviewClosureInvalid(v, priorIds);
@@ -596,7 +632,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
596
632
  : llm.launchNeverStarted ? "launch-never-started"
597
633
  : llm.silentAtBeat ? "silent"
598
634
  : llm.timedOut ? (bytes > 0 ? "truncated" : "silent")
599
- : classifyVerdictCause(raw, nonce, "approve", llm);
635
+ : classifyVerdictCause(seat, nonce, "approve", llm);
600
636
  const failure = cause === "malformed-verdict"
601
637
  ? "review output unparseable"
602
638
  : cause === "closure-mismatch"
@@ -17,7 +17,8 @@ export declare function resetLoadProviderForTests(): void;
17
17
  /**
18
18
  * One gate's own measurement, taken WHERE THE GATE RUNS. `durationMs` sums that gate's execution
19
19
  * intervals and nothing between them, so the composite `test` gate (a selected screen, then other
20
- * gates, then the full suite) reports the two suites' cost rather than the span containing them —
20
+ * gates, then the full suite) reports the two suites' cost rather than the span containing them
21
+ * (split across the two rows when the screen is published before semantic gates, OBS-1176) —
21
22
  * and no consumer has to re-derive a duration by subtracting journal timestamps, which measures the
22
23
  * queue as well as the work. Load is sampled at each interval's endpoints and every second within it;
23
24
  * start preserves the scheduling input while max and mean retain sustained interior saturation.
@@ -2,7 +2,7 @@ import { randomUUID } from "node:crypto";
2
2
  import { existsSync, mkdtempSync, readFileSync, rmSync, statSync } from "node:fs";
3
3
  import { loadavg, tmpdir } from "node:os";
4
4
  import { join, posix } from "node:path";
5
- import { channelKey, shq } from "../adapters/types.js";
5
+ import { channelKey, configuredEffort, shq } from "../adapters/types.js";
6
6
  import { TIER_RANK } from "../config/config.js";
7
7
  import { getAdapter } from "../adapters/registry.js";
8
8
  import { GATE_NAMES } from "../graph/schema.js";
@@ -43,13 +43,14 @@ function instrumentLlmAdapter(adapter, clocks) {
43
43
  return new Proxy(adapter, {
44
44
  get(target, property) {
45
45
  if (property === "headlessCommand") {
46
- return (promptFile, model) => {
47
- const command = target.headlessCommand(promptFile, model);
46
+ return (promptFile, model, effort) => {
47
+ const command = target.headlessCommand(promptFile, model, effort);
48
48
  const dir = mkdtempSync(join(tmpdir(), "tickmarkr-gate-invocation-"));
49
49
  const startedAtPath = join(dir, "started-at");
50
50
  const completedAtPath = join(dir, "completed-at");
51
51
  clocks.push({
52
52
  channel: channelKey({ adapter: target.id, model }),
53
+ ...(effort ? { effort } : {}),
53
54
  preparedAt: Date.now(),
54
55
  startedAtPath,
55
56
  completedAtPath,
@@ -85,7 +86,7 @@ function finishLlmDispatches(clocks) {
85
86
  finally {
86
87
  rmSync(clock.dir, { recursive: true, force: true });
87
88
  }
88
- return { channel: clock.channel, durationMs: completedAt - startedAt };
89
+ return { channel: clock.channel, ...(clock.effort ? { effort: clock.effort } : {}), durationMs: completedAt - startedAt };
89
90
  });
90
91
  }
91
92
  async function captureLlmDispatches(adapters, run) {
@@ -341,10 +342,14 @@ export async function runGates(task, ctx) {
341
342
  const enabled = (g) => task.gates.includes(g) && (g !== "acceptance" && g !== "review" || shapeGates?.[g] !== false);
342
343
  const failed = () => results.some((r) => !r.pass);
343
344
  // T4 (OBS-265): a GREEN selected-test run is a screen, not the round's verdict — the merge-candidate
344
- // round re-runs the full suite on the same commit and THAT is what the round reports. Held here so
345
- // exactly one `test` gate-result ever leaves a round, always carrying which suite spoke for it.
345
+ // round re-runs the full suite on the same commit and THAT is what the round reports. With no
346
+ // semantic gate to act on it, the screen is held so its full suite speaks for it in one row.
346
347
  // (A RED screen IS the verdict: the round ends there, so it is recorded immediately.)
347
348
  let heldTest;
349
+ // OBS-1176: when acceptance/review WILL act on a green screen, the screen is published before they
350
+ // start, as its own selected row. The full suite afterwards is a second invocation on its own row —
351
+ // it carries only its own receipts and interval, so it neither erases nor re-counts the screen.
352
+ const publishScreen = enabled("acceptance") || enabled("review");
348
353
  // v2.0 T2 (OBS-554): this round's per-gate measurement. Every interval a gate actually spends
349
354
  // executing is added HERE, at the call site that runs it, so a gate that runs twice (the test
350
355
  // gate's screen and its full suite) sums to its own cost and never to the span between them.
@@ -711,6 +716,13 @@ export async function runGates(task, ctx) {
711
716
  const screened = { ...r, meta: { ...r.meta, selectedTests: selected } };
712
717
  if (!screened.pass)
713
718
  await record(screened);
719
+ else if (publishScreen) {
720
+ await record(screened);
721
+ // The screen's interval now lives on its own row; the full suite measures from zero.
722
+ spans.delete("test");
723
+ loadSamples.delete("test");
724
+ selectedDurationMs = undefined;
725
+ }
714
726
  else {
715
727
  heldTest = withTelemetry(screened);
716
728
  results.push(heldTest);
@@ -815,8 +827,8 @@ export async function runGates(task, ctx) {
815
827
  // Separate from `invocations` above deliberately: that array is transcript evidence and records
816
828
  // one entry per CAPTURED OUTPUT, so a dispatch that produced none contributes nothing to it.
817
829
  const invocationSpans = [];
818
- const invokeJudge = async (adapter, model, via) => {
819
- const captured = await captureLlmDispatches([adapter], ([instrumented]) => acceptanceGate(task, ctx.worktree, ctx.baseRef, { adapter: instrumented, model }, via, { testCmd: ctx.commands.test, diffCap: ctx.cfg.gates.diffCap }));
830
+ const invokeJudge = async (adapter, model, via, effort) => {
831
+ const captured = await captureLlmDispatches([adapter], ([instrumented]) => acceptanceGate(task, ctx.worktree, ctx.baseRef, { adapter: instrumented, model, effort }, via, { testCmd: ctx.commands.test, diffCap: ctx.cfg.gates.diffCap }));
820
832
  // The instrumented adapter is reached only by runLlm. Deterministic oracles and diff-cap exits
821
833
  // never call headlessCommand, so they produce no clock and cannot manufacture an invocation.
822
834
  invocationSpans.push(...captured.invocations);
@@ -838,7 +850,8 @@ export async function runGates(task, ctx) {
838
850
  }
839
851
  return captured.value;
840
852
  };
841
- let a = await invokeJudge(judgeAdapter, ctx.cfg.judge.model, jvia);
853
+ // OBS-1182: every judge seat launches at its OWN configured effort, never the worker's.
854
+ let a = await invokeJudge(judgeAdapter, ctx.cfg.judge.model, jvia, configuredEffort(ctx.cfg, ctx.cfg.judge));
842
855
  // GATE-09: an unparseable judge verdict retries the JUDGE exactly once on a failover channel — never
843
856
  // the worker (run-20260711-185020 P43-03 L70-72 billed a judge flake as a worker attempt). The flaked
844
857
  // first verdict NEVER enters results (no false gate-result journal event, no operator notify, no stale
@@ -874,7 +887,7 @@ export async function runGates(task, ctx) {
874
887
  ? { driver: ctx.via.driver, keep: ctx.via.keep, onSlot: ctx.via.onSlot, name: ctx.via.nameFor("judge", retryAdapter.id) + "-r1", label: ctx.via.labelFor("judge") }
875
888
  : undefined;
876
889
  // the retry IS a second acceptanceGate call: one code path, one parser, zero new parse leniency.
877
- a = await invokeJudge(retryAdapter, retry.model, retryJvia);
890
+ a = await invokeJudge(retryAdapter, retry.model, retryJvia, configuredEffort(ctx.cfg, retry));
878
891
  a = { ...a, meta: { ...a.meta, judgeRetry: { flaked: flakedKey, retried: channelKey({ adapter: retry.adapter, model: retry.model }) } } };
879
892
  }
880
893
  // No dispatch, no key: a deterministic-oracle round writes no `invocations` field rather than an
@@ -1090,9 +1103,10 @@ export async function runGates(task, ctx) {
1090
1103
  }
1091
1104
  // The merge-candidate round: every other gate is green, so THIS round is the one that can merge —
1092
1105
  // the full suite runs on the exact gated commit before the pipeline reports green. Nothing merges
1093
- // on a subset (spec: "nothing merges without a complete green suite"). Its verdict SUPERSEDES the
1094
- // held screen rather than joining it: one `test` entry in the record, one `test` end event in the
1095
- // stream, and `fullSuite` says which suite spoke while `selectedTests` keeps what the screen ran.
1106
+ // on a subset (spec: "nothing merges without a complete green suite"). Its verdict replaces the
1107
+ // screen's entry in the returned record (one `test` entry), and `fullSuite` says which suite spoke
1108
+ // while `selectedTests` keeps what the screen ran. In the stream, a held screen is superseded (one
1109
+ // `test` end event); a published screen keeps its own earlier event and this is the second.
1096
1110
  if (selected) {
1097
1111
  await emitStart("test");
1098
1112
  // This is the last shell command a round can run — the judge's named-test oracle (acceptance.ts)
@@ -151,7 +151,9 @@ export declare function manifestFileCount(cmd: string, cwd: string): Promise<num
151
151
  * selection (OBS-1166: a selected screen's retry rediscovered the whole selection and refused). So the
152
152
  * retry is built from the UN-narrowed base command, its own `--` rule, the stranded files as the only
153
153
  * positional filters, and an `--exclude` of every completed file; the caller then requires discovery
154
- * to prove the exact retry set before launch. */
154
+ * to prove the exact retry set before launch. OBS-1180: Vitest matches an absolute filter against the
155
+ * module path it resolved through every symlink, so the filters are rooted at the canonical realpath of
156
+ * `cwd` — a symlinked worktree root otherwise filters to an empty discovery. */
155
157
  export declare function singleForkRetryCommand(base: string, cwd: string, stranded: readonly string[], completed: readonly string[]): string;
156
158
  /** One configured runner execution, and its own collection under the same arguments and environment.
157
159
  * The installed runner is trusted (R28 add.1 option B); the nonce catches stale artifacts, not forgery. */
@@ -492,10 +492,17 @@ function strandedSingleForkFiles(files, nonce, run) {
492
492
  * selection (OBS-1166: a selected screen's retry rediscovered the whole selection and refused). So the
493
493
  * retry is built from the UN-narrowed base command, its own `--` rule, the stranded files as the only
494
494
  * positional filters, and an `--exclude` of every completed file; the caller then requires discovery
495
- * to prove the exact retry set before launch. */
495
+ * to prove the exact retry set before launch. OBS-1180: Vitest matches an absolute filter against the
496
+ * module path it resolved through every symlink, so the filters are rooted at the canonical realpath of
497
+ * `cwd` — a symlinked worktree root otherwise filters to an empty discovery. */
496
498
  export function singleForkRetryCommand(base, cwd, stranded, completed) {
499
+ let root = cwd;
500
+ try {
501
+ root = realpathSync(cwd);
502
+ }
503
+ catch { /* unreadable cwd — the exact rediscovery below fails closed */ }
497
504
  const excluded = completed.map(f => `--exclude=${shq(f.replace(/[\\*?[\]{}()!+@]/g, "\\$&"))}`).join(" ");
498
- return `${base}${runnerInvocation(base, cwd).separator} ${stranded.map(f => shq(join(cwd, f))).join(" ")} ${excluded}`;
505
+ return `${base}${runnerInvocation(base, cwd).separator} ${stranded.map(f => shq(join(root, f))).join(" ")} ${excluded}`;
499
506
  }
500
507
  /** One configured runner execution, and its own collection under the same arguments and environment.
501
508
  * The installed runner is trusted (R28 add.1 option B); the nonce catches stale artifacts, not forgery. */
@@ -4,12 +4,14 @@ export declare const GRAPH_ROUTING_MODES: readonly ["partner-led", "risk-based",
4
4
  export declare const STATUSES: readonly ["pending", "running", "gated", "failed", "done", "human"];
5
5
  export declare const GATE_NAMES: readonly ["build", "test", "lint", "evidence", "scope", "acceptance", "review"];
6
6
  export declare const TIERS: readonly ["cheap", "mid", "frontier"];
7
+ export declare const EFFORTS: readonly ["low", "medium", "high"];
7
8
  export declare const SPEC_SOURCES: readonly ["speckit", "gsd", "prd", "native"];
8
9
  export declare const ORACLES: readonly ["command", "test", "judge"];
9
10
  export type Shape = (typeof SHAPES)[number];
10
11
  export type TaskStatus = (typeof STATUSES)[number];
11
12
  export type GateName = (typeof GATE_NAMES)[number];
12
13
  export type Oracle = (typeof ORACLES)[number];
14
+ export type Effort = (typeof EFFORTS)[number];
13
15
  export type SpecSource = (typeof SPEC_SOURCES)[number];
14
16
  export declare const AcceptanceItemSchema: z.ZodUnion<readonly [z.ZodString, z.ZodObject<{
15
17
  oracle: z.ZodLiteral<"command">;
@@ -7,6 +7,8 @@ export const STATUSES = ["pending", "running", "gated", "failed", "done", "human
7
7
  export const GATE_NAMES = ["build", "test", "lint", "evidence", "scope", "acceptance", "review"];
8
8
  const MANDATORY_GATES = ["build", "test", "lint", "evidence", "scope"];
9
9
  export const TIERS = ["cheap", "mid", "frontier"];
10
+ // OBS-1182: launch effort — channel metadata beside tier, never part of model identity or channelKey.
11
+ export const EFFORTS = ["low", "medium", "high"];
10
12
  export const SPEC_SOURCES = ["speckit", "gsd", "prd", "native"];
11
13
  // v1.19 acceptance oracles: command (exit code), test (named test), judge (LLM, free-text rubric).
12
14
  // A plain string is the read-old/write-new compat form — semantically a judge oracle (spec §2).
@@ -169,7 +169,7 @@ function bindCandidate(candidate, channels) {
169
169
  `(doctor found: ${channels.map((ch) => `${ch.adapter}:${ch.model}`).join(", ") || "(nothing)"}) — ` +
170
170
  "re-run scope --preview and confirm again");
171
171
  }
172
- return { adapter: c.adapter, model: c.model, channel: c.channel, tier: c.tier };
172
+ return { adapter: c.adapter, model: c.model, channel: c.channel, tier: c.tier, ...(c.effort ? { effort: c.effort } : {}) };
173
173
  }
174
174
  // Adapter-level probes establish the current installation/auth state, but shipped adapters do not
175
175
  // return per-model verdicts from probe(). Preserve cached verdicts only where that fresh snapshot is
@@ -202,7 +202,7 @@ export async function scopeIntent(intentFile, repoRoot, options) {
202
202
  driver, name: `scope-${name}-${attempts}-${adapter.id}`, label: "SCOPE",
203
203
  keep: options.cfg.visibility.keepPanes === "forever",
204
204
  } : undefined;
205
- const draft = extractDraft(await runLlm(adapter, assignment.model, prompt, repoRoot, via));
205
+ const draft = extractDraft(await runLlm(adapter, assignment.model, prompt, repoRoot, via, undefined, assignment.effort));
206
206
  let tasks;
207
207
  try {
208
208
  tasks = validateDraft(draft);
@@ -11,6 +11,16 @@ export declare const routingModelProvider: typeof modelProvider;
11
11
  export declare const modelRouteIdentity: (model: string, fallback?: string) => string;
12
12
  export declare const channelRouteIdentity: (key: string, fallback?: string) => string;
13
13
  export declare function routingEntrySeatLines(cfg: TickmarkrConfig): string[];
14
+ export declare const observedIdentity: (health: Record<string, AuthHealth> | null | undefined, adapter: string, model: string) => string | undefined;
15
+ export declare const observedSeat: (health: Record<string, AuthHealth> | null | undefined, adapter: string, model: string) => {
16
+ adapter: string;
17
+ model: string;
18
+ identity: string;
19
+ } | {
20
+ adapter: string;
21
+ model: string;
22
+ identity?: undefined;
23
+ };
14
24
  export declare function excludedChannels(cfg: TickmarkrConfig, adapters: {
15
25
  id: string;
16
26
  }[] | string[], health: Record<string, AuthHealth>): {
@@ -59,6 +69,14 @@ export interface DenyPreferCollision {
59
69
  detail: string;
60
70
  disallowed: Disallowed;
61
71
  }
62
- export declare function preferEntryDenied(p: string, cfg: TickmarkrConfig): Disallowed | null;
63
- export declare function denyPreferCollisions(cfg: TickmarkrConfig, shapes?: Iterable<string>): DenyPreferCollision[];
72
+ export declare function preferEntryDenied(p: string, cfg: TickmarkrConfig, health?: Record<string, AuthHealth> | null): Disallowed | null;
73
+ export declare function denyPreferCollisions(cfg: TickmarkrConfig, shapes?: Iterable<string>, health?: Record<string, AuthHealth> | null): DenyPreferCollision[];
74
+ export interface DeadPoolEntry {
75
+ shape: string;
76
+ entry: string;
77
+ disallowed: Disallowed;
78
+ admitted: string[];
79
+ }
80
+ export declare function deadPoolEntries(cfg: TickmarkrConfig, shapes?: Iterable<string>, health?: Record<string, AuthHealth> | null): DeadPoolEntry[];
81
+ export declare function deadPoolEntryLine({ shape, entry, disallowed, admitted }: DeadPoolEntry): string;
64
82
  export declare function denyPreferCollisionLine({ kind, shape, detail, disallowed }: DenyPreferCollision): string;
@@ -47,6 +47,13 @@ export function routingEntrySeatLines(cfg) {
47
47
  add("routing.deny.workers.models", cfg.routing.deny?.workers?.models, ["worker"]);
48
48
  return lines;
49
49
  }
50
+ // OBS-1143: the probed identity doctor cached for adapter:model — the one discoverChannels routes
51
+ // with. Absent ⇒ unknown, and the deny matcher stays conservative (alias-family match).
52
+ export const observedIdentity = (health, adapter, model) => health?.[adapter]?.modelAuth?.[model]?.identity ?? health?.[adapter]?.modelIdentities?.[model];
53
+ export const observedSeat = (health, adapter, model) => {
54
+ const identity = observedIdentity(health, adapter, model);
55
+ return identity ? { adapter, model, identity } : { adapter, model };
56
+ };
50
57
  const adapterIds = (adapters) => typeof adapters[0] === "string" ? adapters : adapters.map((a) => a.id);
51
58
  export function excludedChannels(cfg, adapters, health) {
52
59
  const { allow, deny } = cfg.routing;
@@ -57,9 +64,7 @@ export function excludedChannels(cfg, adapters, health) {
57
64
  if (!health[id]?.installed || !health[id]?.authed)
58
65
  continue;
59
66
  for (const c of channelsFromConfig(id, cfg)) {
60
- const identity = health[id]?.modelAuth?.[c.model]?.identity ?? health[id]?.modelIdentities?.[c.model];
61
- const chan = identity ? { ...c, identity } : c;
62
- const d = disallowedBy(chan, cfg.routing);
67
+ const d = disallowedBy(observedSeat(health, c.adapter, c.model), cfg.routing);
63
68
  if (d)
64
69
  out.push({ key: channelKey(c), d });
65
70
  }
@@ -175,13 +180,20 @@ const disallowedFromPreferError = (msg) => {
175
180
  const m = msg.match(/prefer entry .+ is disallowed by routing\.(deny|allow) \(([^)]+)\)/);
176
181
  return m ? { by: m[1], entry: m[2] } : null;
177
182
  };
178
- export function preferEntryDenied(p, cfg) {
183
+ // OBS-1143: `health` is doctor's CACHED verdict (never a probe); its identities ride the channels
184
+ // route() reads them from, so the preflight answers exactly as the router would.
185
+ export function preferEntryDenied(p, cfg, health) {
179
186
  if (!cfg.routing.allow && !cfg.routing.deny)
180
187
  return null;
181
188
  const probe = structuredClone(cfg);
182
189
  probe.routing.map = { ...probe.routing.map, [PREFLIGHT_SHAPE]: { prefer: [p] } };
190
+ const observed = Object.keys(cfg.tiers).flatMap((id) => channelsFromConfig(id, cfg))
191
+ .flatMap((c) => {
192
+ const identity = observedIdentity(health, c.adapter, c.model);
193
+ return identity ? [{ ...c, identity }] : [];
194
+ });
183
195
  try {
184
- route(preflightTask, probe, []);
196
+ route(preflightTask, probe, observed);
185
197
  return null;
186
198
  }
187
199
  catch (e) {
@@ -194,7 +206,9 @@ export function preferEntryDenied(p, cfg) {
194
206
  // route — resume hands it the loaded graph's shape set, so a collision on a shape no resumed task
195
207
  // uses can no longer refuse the only crash-recovery path. Omitted ⇒ the whole routing map: doctor
196
208
  // audits the config itself, which has no graph to be scoped by.
197
- export function denyPreferCollisions(cfg, shapes) {
209
+ // OBS-1143: `health` (doctor's cached verdict) supplies each alias's observed identity; omitted or
210
+ // unrecorded ⇒ unknown, which stays conservative.
211
+ export function denyPreferCollisions(cfg, shapes, health) {
198
212
  if (!cfg.routing.allow && !cfg.routing.deny)
199
213
  return [];
200
214
  const inGraph = shapes === undefined ? undefined : new Set(shapes);
@@ -203,7 +217,7 @@ export function denyPreferCollisions(cfg, shapes) {
203
217
  if (inGraph && !inGraph.has(shape))
204
218
  continue;
205
219
  if (entry.pin) {
206
- const d = disallowedBy({ adapter: entry.pin.via, model: entry.pin.model }, cfg.routing);
220
+ const d = disallowedBy(observedSeat(health, entry.pin.via, entry.pin.model), cfg.routing);
207
221
  if (d) {
208
222
  out.push({
209
223
  kind: "pin",
@@ -214,20 +228,17 @@ export function denyPreferCollisions(cfg, shapes) {
214
228
  }
215
229
  }
216
230
  const prefer = entry.prefer ?? [];
217
- if (prefer.length && prefer.every((p) => preferEntryDenied(p, cfg) !== null)) {
231
+ if (prefer.length && prefer.every((p) => preferEntryDenied(p, cfg, health) !== null)) {
218
232
  out.push({
219
233
  kind: "prefer",
220
234
  shape,
221
235
  detail: prefer.join(" > "),
222
- disallowed: preferEntryDenied(prefer[0], cfg),
236
+ disallowed: preferEntryDenied(prefer[0], cfg, health),
223
237
  });
224
238
  }
225
239
  const pool = entry.pool;
226
240
  if (pool) {
227
- const denied = pool.channels.map((p) => {
228
- const i = p.indexOf(":");
229
- return disallowedBy({ adapter: p.slice(0, i), model: p.slice(i + 1) }, cfg.routing);
230
- });
241
+ const denied = pool.channels.map((p) => poolEntryDisallowed(p, cfg, health));
231
242
  // only a FULLY dead pool collides — a partial deny still leaves live members to route
232
243
  if (denied.every((d) => d !== null)) {
233
244
  out.push({ kind: "pool", shape, detail: pool.channels.join(" > "), disallowed: denied[0] });
@@ -236,6 +247,30 @@ export function denyPreferCollisions(cfg, shapes) {
236
247
  }
237
248
  return out;
238
249
  }
250
+ const poolEntryDisallowed = (p, cfg, health) => {
251
+ const i = p.indexOf(":");
252
+ return disallowedBy(observedSeat(health, p.slice(0, i), p.slice(i + 1)), cfg.routing);
253
+ };
254
+ // OBS-1144: every disallowed entry a pool carries, beside the admitted remainder the router routes
255
+ // instead — the router skips the entry; an empty remainder is the exhausted pool it refuses.
256
+ export function deadPoolEntries(cfg, shapes, health) {
257
+ if (!cfg.routing.allow && !cfg.routing.deny)
258
+ return [];
259
+ const inGraph = shapes === undefined ? undefined : new Set(shapes);
260
+ return Object.entries(cfg.routing.map).flatMap(([shape, entry]) => {
261
+ if (!entry.pool || (inGraph && !inGraph.has(shape)))
262
+ return [];
263
+ const judged = entry.pool.channels.map((p) => ({ p, d: poolEntryDisallowed(p, cfg, health) }));
264
+ const admitted = judged.filter(({ d }) => d === null).map(({ p }) => p);
265
+ return judged.flatMap(({ p, d }) => (d ? [{ shape, entry: p, disallowed: d, admitted }] : []));
266
+ });
267
+ }
268
+ export function deadPoolEntryLine({ shape, entry, disallowed, admitted }) {
269
+ const verdict = admitted.length
270
+ ? `skipped — routes the admitted remainder ${admitted.join(" > ")}`
271
+ : "the pool is exhausted — no admitted entry remains, so the shape is unroutable";
272
+ return `routing.map.${shape}.pool entry ${entry} is disallowed by routing.${disallowed.by} (${disallowed.entry}) — ${verdict}`;
273
+ }
239
274
  export function denyPreferCollisionLine({ kind, shape, detail, disallowed }) {
240
275
  const verb = kind === "pin" ? "is disallowed" : "fully disallowed";
241
276
  return `deny∩prefer: routing.map.${shape}.${kind} ${detail} ${verb} by routing.${disallowed.by} (${disallowed.entry}) — remove the ${disallowed.by} entry or adjust ${kind}`;