tickmarkr 2.6.2 → 2.6.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/README.md +12 -2
  2. package/dist/adapters/codex.d.ts +2 -1
  3. package/dist/adapters/codex.js +26 -5
  4. package/dist/adapters/model-lints.d.ts +1 -0
  5. package/dist/adapters/model-lints.js +48 -1
  6. package/dist/adapters/model-windows.d.ts +2 -1
  7. package/dist/adapters/model-windows.js +18 -7
  8. package/dist/adapters/types.d.ts +6 -0
  9. package/dist/cli/commands/approve.d.ts +6 -3
  10. package/dist/cli/commands/approve.js +43 -6
  11. package/dist/cli/commands/doctor.d.ts +8 -2
  12. package/dist/cli/commands/doctor.js +6 -1
  13. package/dist/cli/commands/fleet.js +264 -59
  14. package/dist/cli/commands/init.js +5 -3
  15. package/dist/cli/commands/plan.js +13 -8
  16. package/dist/cli/commands/report.d.ts +40 -2
  17. package/dist/cli/commands/report.js +278 -11
  18. package/dist/cli/commands/resume.js +2 -4
  19. package/dist/cli/commands/run.d.ts +11 -0
  20. package/dist/cli/commands/run.js +33 -5
  21. package/dist/cli/commands/status.js +49 -4
  22. package/dist/cli/commands/verify.js +332 -121
  23. package/dist/compile/native.js +137 -0
  24. package/dist/config/config.d.ts +40 -9
  25. package/dist/config/config.js +133 -15
  26. package/dist/config/fleet-overlay.d.ts +13 -3
  27. package/dist/config/fleet-overlay.js +12 -8
  28. package/dist/drivers/herdr.d.ts +12 -0
  29. package/dist/drivers/herdr.js +51 -0
  30. package/dist/drivers/orca.d.ts +9 -1
  31. package/dist/drivers/orca.js +29 -7
  32. package/dist/drivers/types.d.ts +2 -0
  33. package/dist/drivers/types.js +2 -2
  34. package/dist/gates/acceptance.d.ts +7 -0
  35. package/dist/gates/acceptance.js +27 -5
  36. package/dist/gates/baseline.d.ts +20 -1
  37. package/dist/gates/baseline.js +100 -20
  38. package/dist/gates/cache.d.ts +8 -0
  39. package/dist/gates/cache.js +12 -2
  40. package/dist/gates/llm.d.ts +6 -0
  41. package/dist/gates/llm.js +27 -8
  42. package/dist/gates/review.d.ts +6 -1
  43. package/dist/gates/review.js +122 -32
  44. package/dist/gates/run-gates.d.ts +54 -3
  45. package/dist/gates/run-gates.js +331 -45
  46. package/dist/gates/test-manifest.d.ts +42 -0
  47. package/dist/gates/test-manifest.js +69 -10
  48. package/dist/route/router.d.ts +23 -1
  49. package/dist/route/router.js +54 -16
  50. package/dist/run/consult.d.ts +3 -1
  51. package/dist/run/consult.js +4 -2
  52. package/dist/run/daemon.d.ts +2 -1
  53. package/dist/run/daemon.js +349 -79
  54. package/dist/run/interactive-seed.d.ts +4 -0
  55. package/dist/run/interactive-seed.js +35 -9
  56. package/dist/run/journal.d.ts +123 -1
  57. package/dist/run/journal.js +480 -17
  58. package/dist/run/lease.d.ts +13 -0
  59. package/dist/run/lease.js +45 -0
  60. package/dist/run/protocol.d.ts +15 -0
  61. package/dist/run/protocol.js +11 -1
  62. package/dist/run/receipt-resolver.d.ts +22 -0
  63. package/dist/run/receipt-resolver.js +40 -1
  64. package/dist/run/repair-selection.d.ts +11 -1
  65. package/dist/run/repair-selection.js +17 -9
  66. package/dist/run/wall-budget.d.ts +48 -0
  67. package/dist/run/wall-budget.js +280 -0
  68. package/dist/tui/cockpit/live-store.d.ts +6 -0
  69. package/dist/tui/cockpit/live-store.js +36 -11
  70. package/dist/tui/cockpit/run-cockpit.js +2 -2
  71. package/dist/tui/cockpit/run-view.d.ts +2 -1
  72. package/dist/tui/cockpit/run-view.js +13 -9
  73. package/dist/tui/cockpit/setup-cockpit.d.ts +2 -0
  74. package/dist/tui/cockpit/setup-cockpit.js +4 -0
  75. package/package.json +2 -1
  76. package/schema/config.schema.json +8 -1
  77. package/skills/tickmarkr-auto/SKILL.md +2 -2
  78. package/skills/tickmarkr-loop/SKILL.md +17 -5
  79. package/skills/tickmarkr-overseer/scripts/classify-vitest-log.sh +4 -1
package/dist/gates/llm.js CHANGED
@@ -81,19 +81,22 @@ export function gatePaneName(role, taskId, suffix = "") {
81
81
  // T2 ownership contract: a canonical owned fallback (the daemon's nameFor now emits one) passes
82
82
  // through untouched; run-gates' "-r1" judge-retry suffix becomes attempt+1 so the retry pane's name
83
83
  // stays contract-parseable (tickmarkr:judge:<task>:1:<runId>) instead of a corrupted-runId shape.
84
+ // C-3 (D-669): the hop suffix is -r<N> — the judge-flake retry is -r1 and the adjudicator -r2, so the two
85
+ // never share an owned name (a retained retry pane must not shadow the adjudicator's slot).
84
86
  export function rolePaneNameFromPrompt(prompt, fallback) {
85
- const retry = fallback.endsWith("-r1");
86
- const base = retry ? fallback.slice(0, -3) : fallback;
87
+ const hop = /-r([1-9])$/.exec(fallback);
88
+ const suffix = hop ? hop[0] : "";
89
+ const base = hop ? fallback.slice(0, -suffix.length) : fallback;
87
90
  const owned = parseOwnedName(base);
88
91
  if (owned)
89
- return retry ? formatOwnedName({ ...owned, attempt: owned.attempt + 1 }) : base;
92
+ return hop ? formatOwnedName({ ...owned, attempt: owned.attempt + Number(hop[1]) }) : base;
90
93
  const id = prompt.match(/## Task ([^\n:]+):/)?.[1];
91
94
  if (!id)
92
95
  return fallback;
93
96
  if (prompt.startsWith("TICKMARKR-JUDGE"))
94
- return gatePaneName("judge", id, retry ? "-r1" : "");
97
+ return gatePaneName("judge", id, suffix);
95
98
  if (prompt.startsWith("TICKMARKR-REVIEW"))
96
- return gatePaneName("review", id, retry ? "-r1" : "");
99
+ return gatePaneName("review", id, suffix);
97
100
  return fallback;
98
101
  }
99
102
  const llmOutputCapture = new AsyncLocalStorage();
@@ -299,6 +302,16 @@ export function reviewSeatOutput(raw, nonce, adapterBannerRows = []) {
299
302
  // often the seat's first byte than the harness's exit marker, and the next read completes either.
300
303
  return trailer ? seat.slice(0, trailer.index) : seat;
301
304
  }
305
+ /** OBS-1168(b): a judge/review seat whose pane could not be created or launched. Typed so the gates can
306
+ * contain it as seat recovery (another seat) or an infra park — never as a worker failure. */
307
+ export class SeatLaunchError extends Error {
308
+ seat;
309
+ constructor(seat, cause) {
310
+ super(`seat ${seat} failed to launch: ${cause instanceof Error ? cause.message : String(cause)}`);
311
+ this.seat = seat;
312
+ this.name = "SeatLaunchError";
313
+ }
314
+ }
302
315
  export const REVIEW_FIRST_LIVENESS_MS = 30_000;
303
316
  // OBS-1039: a seat that wrote ten bytes and went quiet escaped the zero-byte beat and sat to the
304
317
  // ceiling. Below this many seat-authored bytes at the first beat the seat is `silent` — demoted and
@@ -341,9 +354,15 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
341
354
  adapter.headlessCommand(pf, model, effort),
342
355
  gateExitTrailer(nonce),
343
356
  ].join("\n"));
344
- slot = await via.driver.slot(cwd, rolePaneNameFromPrompt(prompt, via.name), via.label ? { label: via.label } : undefined);
345
- via.onSlot?.(slot);
346
- await via.driver.run(slot, paneDispatchCommand(scriptPath));
357
+ try {
358
+ slot = await via.driver.slot(cwd, rolePaneNameFromPrompt(prompt, via.name), via.label ? { label: via.label } : undefined);
359
+ via.onSlot?.(slot);
360
+ await via.driver.run(slot, paneDispatchCommand(scriptPath));
361
+ }
362
+ catch (error) {
363
+ forceClose = true; // a half-launched pane is closed, never kept
364
+ throw new SeatLaunchError(`${adapter.id}:${model}`, error);
365
+ }
347
366
  if (via.driver.sendKey) {
348
367
  try {
349
368
  if (matchesTrustDialog(await via.driver.read(slot, 400), adapter.trustDialog)) {
@@ -51,6 +51,11 @@ export declare function fetchTaskDiff(worktree: string, baseRef: string, files?:
51
51
  export declare function checkDiffCap(gate: string, measured: number, cap: number, prefix?: string): GateResult | null;
52
52
  /** Apply the strict reviewable-logic cap and the finite, larger capture cap independently. */
53
53
  export declare function checkTaskDiffCaps(gate: string, measured: Pick<TaskDiffMeasurement, "logicBytes" | "captureBytes">, logicCap: number, prefix?: string): GateResult | null;
54
+ export declare function isGarbageReview(result: GateResult): result is GateResult & {
55
+ meta: {
56
+ reviewer: string;
57
+ };
58
+ };
54
59
  export declare function isDiffCapPark(result: GateResult): boolean;
55
60
  export declare function diffCapParkReason(results: GateResult[]): string | null;
56
61
  export declare function modelId(model: string): string;
@@ -111,7 +116,7 @@ prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, ne
111
116
  floor?: Tier, // task/config/prior floor from the caller; the author's own tier is ALWAYS applied here (RF-1)
112
117
  history?: string[], // run-scoped picks, oldest to newest; empty preserves the established ranking
113
118
  onSeat?: (seat: number, count: number) => void, demoted?: ReadonlySet<string>, excludeVendors?: ReadonlySet<string>, authors?: readonly string[]): BillingChannel | null;
114
- export type ReviewUnparseableCause = VerdictUnparseableCause | "launch-never-started" | "truncated" | "silent" | "closure-mismatch";
119
+ export type ReviewUnparseableCause = VerdictUnparseableCause | "launch-never-started" | "truncated" | "silent" | "closure-mismatch" | "seat-launch-failed";
115
120
  /**
116
121
  * This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
117
122
  * remains a separate stated input, so whether its touched paths fit the declaration stays a reviewer
@@ -6,12 +6,12 @@ import { filesGlob } from "../graph/files-glob.js";
6
6
  import { renderAcceptanceItem, TIERS } from "../graph/schema.js";
7
7
  import { getAdapter } from "../adapters/registry.js";
8
8
  import { shOk } from "../run/git.js";
9
- import { carryReviewFindings, observedReviewFingerprints, reviewFingerprintMatches, structuredFindings, UNIDENTIFIED } from "../run/journal.js";
9
+ import { carryReviewFindings, isDeferredFinding, observedReviewFingerprints, reviewFingerprintMatches, structuredFindings, UNIDENTIFIED } from "../run/journal.js";
10
10
  import { redactSecrets } from "../run/redact.js";
11
11
  import { rankPreferredChannels, reviewPreferenceTieBreak } from "../route/role-pick.js";
12
12
  import { modelProvider } from "../route/preference.js";
13
13
  import { resolveStateDir } from "./cache.js";
14
- import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, dewrapPaneVerdict, extractVerdictJson, generateVerdictNonce, parseAnchoredComments, runLlmDetailed, verdictNonceLine } from "./llm.js";
14
+ import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, dewrapPaneVerdict, extractVerdictJson, generateVerdictNonce, parseAnchoredComments, runLlmDetailed, SeatLaunchError, verdictNonceLine } from "./llm.js";
15
15
  import { classifyVerdictCause } from "./verdict-cause.js";
16
16
  import { captureDiffCapFor, measureArtifactDiff, reviewableLogicDiff, } from "./artifact-manifest.js";
17
17
  export { isProtectedEvidence, PROTECTED_EVIDENCE_PREFIXES, REGENERABLE_CAPTURE_PATHS, setAsideReceiptPath, setAsideRegenerableCaptures, } from "./artifact-manifest.js";
@@ -164,6 +164,12 @@ export function checkTaskDiffCaps(gate, measured, logicCap, prefix = "") {
164
164
  },
165
165
  };
166
166
  }
167
+ // C-12: a reviewer is excluded for garbage only on the typed malformed-verdict flag. A material finding whose
168
+ // prose mentions the word "unparseable" is a delivered verdict; matching the prose dropped T7's only reviewer.
169
+ export function isGarbageReview(result) {
170
+ return result.gate === "review" && !result.pass && result.meta?.unparseable === true
171
+ && typeof result.meta?.reviewer === "string";
172
+ }
167
173
  export function isDiffCapPark(result) {
168
174
  return result.pass === false
169
175
  && result.meta?.parkKind === "diff-cap"
@@ -400,6 +406,23 @@ function withoutExampleEcho(raw, nonce) {
400
406
  const chars = [...reviewResponseExample(nonce).replace(/\s+/g, "")];
401
407
  return raw.replace(new RegExp(chars.map((c) => c.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")).join("[\\s│|]*"), "g"), "");
402
408
  }
409
+ /**
410
+ * OBS-1195: the parent an anchored comment names in its own optional `finding` field — a 1-based
411
+ * index into this verdict's findings, or a carried prior's id. A deferred entry or a resolved prior
412
+ * settles the anchor; any other valid parent binds it open; anything else leaves it unbound.
413
+ */
414
+ function anchorParent(binding, findings, priors, resolved) {
415
+ if (typeof binding === "number") {
416
+ const entry = Array.isArray(findings) && Number.isInteger(binding) && binding >= 1 ? findings[binding - 1] : undefined;
417
+ if (!entry || typeof entry !== "object" || typeof entry.note !== "string")
418
+ return {};
419
+ return entry.defer === true ? { settled: "deferred" } : { entry: entry.note };
420
+ }
421
+ const prior = matchClosureId(binding, priors);
422
+ if (prior === undefined)
423
+ return {};
424
+ return resolved.includes(prior) ? { settled: "resolved", parent: prior } : { parent: prior };
425
+ }
403
426
  export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
404
427
  // OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
405
428
  // direct tests) skips persistence and changes nothing else.
@@ -550,13 +573,21 @@ block the merge) or "minor" (style, naming, or preference that should not block)
550
573
  block approval. For a minor concern you have decided not to block on, set "defer": true and give a
551
574
  one-line "rationale" — it is recorded in the review, never dropped.
552
575
  A fix you prescribe that would break suites outside the task's declared write scope (files[]) is a scope finding, never a material one.
576
+ Each material finding names the class of defect it belongs to and binds that class to the goal clause or
577
+ acceptance criterion it violates (or to the regression this diff introduces), states its input → consequence,
578
+ and marks its evidence executed (you ran the reproducer), static (you traced it by reading) or blocked (it
579
+ could not run here). Blocked evidence never turns a finding into a pass, and a worker's own case table or
580
+ enumeration never resolves a finding: judge the diff itself.
553
581
 
554
582
  Respond with ONLY this JSON:
555
- {"nonce": "${nonce}", "approve": true|false, "resolved": [], "reraised": [], "findings": [{"note": "...", "severity": "material"|"minor", "defer": false, "rationale": ""}], "comments": [{"path": "path/to/file", "line": 42, "body": "actionable feedback"}]}
583
+ {"nonce": "${nonce}", "approve": true|false, "resolved": [], "reraised": [], "findings": [{"note": "...", "severity": "material"|"minor", "defer": false, "rationale": ""}], "comments": [{"path": "path/to/file", "line": 42, "body": "actionable feedback", "finding": 1}]}
556
584
  For every prior material, put its fingerprint in exactly one of resolved (verified fixed) or reraised
557
585
  (still a blocking defect). Use only the listed fingerprints; never omit one or put it in both lists.
558
586
  Approve iff no material finding remains and every prior material is resolved.
559
587
  The top-level comments array is optional. Use it only for actionable line-anchored feedback.
588
+ A comment's optional "finding" names its parent: the 1-based index of its entry in findings, or a prior
589
+ fingerprint copied from above. A comment anchored to a deferred entry or a resolved prior does not block;
590
+ a comment naming no parent stays open.
560
591
 
561
592
  ${responseRequirement}
562
593
  `;
@@ -590,18 +621,33 @@ ${responseRequirement}
590
621
  savedBrief = undefined;
591
622
  }
592
623
  }
593
- const llm = await runLlmDetailed(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? {
594
- driver: via.driver,
595
- keep: via.keep,
596
- onSlot: via.onSlot,
597
- name: via.nameFor("review", reviewer.adapter),
598
- label: via.labelFor("review"),
599
- } : undefined,
600
- // frontier reviewers routinely need >5min on a configured-cap-sized diff, and `claude -p` buffers all
601
- // output until completion — runLlm's 300s default killed reviews mid-flight, returning empty
602
- // stdout that read as "unparseable" and escalated to re-implementation of green code
603
- // (run-20260709-104447 P87-09). The configured ceiling defaults to that measured 15 minutes.
604
- cfg.review.timeoutMs, reviewer.effort);
624
+ const provider = modelProvider(reviewer.model, reviewer.vendor);
625
+ let llm;
626
+ try {
627
+ llm = await runLlmDetailed(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? {
628
+ driver: via.driver,
629
+ keep: via.keep,
630
+ onSlot: via.onSlot,
631
+ name: via.nameFor("review", reviewer.adapter),
632
+ label: via.labelFor("review"),
633
+ } : undefined,
634
+ // frontier reviewers routinely need >5min on a configured-cap-sized diff, and `claude -p` buffers all
635
+ // output until completion — runLlm's 300s default killed reviews mid-flight, returning empty
636
+ // stdout that read as "unparseable" and escalated to re-implementation of green code
637
+ // (run-20260709-104447 P87-09). The configured ceiling defaults to that measured 15 minutes.
638
+ cfg.review.timeoutMs, reviewer.effort);
639
+ }
640
+ catch (error) {
641
+ if (!(error instanceof SeatLaunchError))
642
+ throw error;
643
+ // OBS-1168(b): the seat never launched, so there is no verdict and nothing about the WORK. A typed
644
+ // no-verdict re-routes to another seat in run-gates; an exhausted pool is an infra park.
645
+ return { gate: "review", pass: false,
646
+ details: `review dispatch failed — ${error.message} (reviewer ${reviewer.adapter}:${reviewer.model}; vendor: ${reviewer.vendor}; provider: ${provider}; cause: seat-launch-failed) — failing closed`,
647
+ meta: { ...policyMeta, ...rotationMeta, ...floorMeta, reviewer: channelKey(reviewer), reviewerTier: reviewer.tier,
648
+ vendor: reviewer.vendor, provider, noVerdict: true, classification: "infra", infra: true,
649
+ cause: "seat-launch-failed", ...(savedBrief ? { briefPath: savedBrief } : {}) } };
650
+ }
605
651
  const raw = llm.output;
606
652
  let saved;
607
653
  if (artifactDir) {
@@ -613,7 +659,6 @@ ${responseRequirement}
613
659
  saved = undefined; // persistence is evidence, not a gate input — never fail the gate on it
614
660
  }
615
661
  }
616
- const provider = modelProvider(reviewer.model, reviewer.vendor);
617
662
  // A pane's own dewrap stops at the first parseable nonce-bound object; once the example's echo is gone
618
663
  // a genuinely wrapped verdict behind it is reconstructed here, exactly as llm.ts would have.
619
664
  const echoFree = withoutExampleEcho(raw, nonce);
@@ -653,6 +698,8 @@ ${responseRequirement}
653
698
  ...(cause === "malformed-verdict" ? { unparseable: true } : { noVerdict: true, classification: "infra", infra: true }),
654
699
  cause,
655
700
  ...(closureMismatch ? { resolved: v?.resolved, reraised: v?.reraised, carriedFingerprints: priorIds.flatMap(observedReviewFingerprints) } : {}),
701
+ // OBS-1196: a whole verdict that breaks the closure protocol was DELIVERED; only undelivered bytes are re-asked.
702
+ ...(closureInvalid ? { closureInvalid: true } : {}),
656
703
  bytes, seatAuthoredBytes: bytes,
657
704
  ...(saved ? { rawPath: saved } : {}),
658
705
  ...(savedBrief ? { briefPath: savedBrief } : {}),
@@ -669,25 +716,24 @@ ${responseRequirement}
669
716
  if (decided.pass)
670
717
  decided.headline = "requested changes";
671
718
  decided.pass = false;
672
- // A reviewer may also restate a re-raised material in findings. Preserve the original
673
- // prose once so an unchanged defect keeps the same failure brief across repair rounds.
674
- for (const finding of reraised) {
675
- const line = `- [material] ${finding.note}`;
676
- if (!decided.lines.includes(line))
677
- decided.lines.push(line);
678
- }
679
719
  }
680
- const prose = `reviewer ${reviewer.adapter}:${reviewer.model} (vendor: ${reviewer.vendor}; provider: ${provider}): ${decided.headline}${decided.lines.length ? "\n" + decided.lines.join("\n") : ""}`;
681
- const details = appendAnchoredReview(prose, v);
720
+ // OBS-1195: an anchor leaves the blocking set only through an explicit, unambiguous parent
721
+ // disposition: its own `finding` names a deferred entry of this verdict or a prior it resolved.
722
+ // Path coincidence binds nothing; an unbound or ambiguous anchor stays open exactly as before.
723
+ const comments = parseAnchoredComments(v);
724
+ const rawComments = (comments.length ? v.comments : []);
725
+ const resolvedIds = (v.resolved ?? []).map((id) => matchClosureId(id, priorIds));
726
+ const anchors = comments.map((comment, i) => ({ ...comment, ...anchorParent(rawComments[i]?.finding, v.findings, priorIds, resolvedIds) }));
727
+ const openAnchors = { comments: anchors.filter((a) => !a.settled) };
728
+ const settledAnchors = anchors.flatMap((a, i) => a.settled
729
+ ? [{ path: a.path, line: a.line, body: a.body, disposition: a.settled, finding: rawComments[i].finding }] : []);
682
730
  // Only the verdict's anchors may supply missing evidence, and only when unambiguous.
683
731
  // Reuse the journal's path normalization without changing legacy details-only parsing.
684
- const anchoredPaths = new Set(parseAnchoredComments(v).map((comment) => {
685
- const anchor = structuredFindings("review", `- ${comment.path}:${comment.line} — anchor`)
686
- .find((finding) => finding.class === "review:anchored");
687
- return anchor?.path ?? comment.path;
688
- }));
732
+ const anchorRow = (comment) => structuredFindings("review", `- ${comment.path}:${comment.line} — ${comment.body}`)
733
+ .find((finding) => finding.class === "review:anchored");
734
+ const anchoredPaths = new Set(comments.map((comment) => anchorRow({ ...comment, body: "anchor" })?.path ?? comment.path));
689
735
  const anchoredPath = anchoredPaths.size === 1 ? [...anchoredPaths][0] : undefined;
690
- const ownDetails = appendAnchoredReview(ownLines.join("\n"), v);
736
+ const ownDetails = appendAnchoredReview(ownLines.join("\n"), openAnchors);
691
737
  const currentRows = (ownDetails.trim() ? structuredFindings("review", ownDetails) : [])
692
738
  .map((finding) => finding.path === UNIDENTIFIED && anchoredPath
693
739
  ? { ...finding, path: anchoredPath, fingerprint: `${finding.class}|${anchoredPath}|${finding.symbol}` }
@@ -708,7 +754,50 @@ ${responseRequirement}
708
754
  const unambiguousRows = linkedRows.map((finding) => finding.reraisedFrom
709
755
  && linkedRows.filter((row) => row.reraisedFrom === finding.reraisedFrom).length > 1
710
756
  ? { ...finding, reraisedFrom: undefined } : finding);
711
- const carriedRows = carryReviewFindings(reraised, unambiguousRows);
757
+ // OBS-1195: a deferred entry echoing a prior's id names that prior as the concern it defers. The
758
+ // link rides the row for the journal's anchor fold only; a deferral never re-seats a material chain.
759
+ // A prior any material row restates — even ambiguously, its link cleared above — is claimed, not
760
+ // deferred: the deferral gets no link, so its bound anchors stay open (fail closed).
761
+ const claimed = new Set(linkedRows.map((row) => row.reraisedFrom).filter((id) => id !== undefined));
762
+ const rows = unambiguousRows.map((finding) => {
763
+ if (!isDeferredFinding(finding))
764
+ return finding;
765
+ const entry = v.findings?.find((entry) => entry && typeof entry === "object" && entry.note === finding.note);
766
+ const id = matchClosureId(entry?.reraised, reraised);
767
+ return id && !claimed.has(id) ? { ...finding, reraisedFrom: id } : finding;
768
+ });
769
+ // OBS-1195: current prose is rendered once. A prior's original wording is echoed only when no
770
+ // current finding positively binds it; a bound chain keeps that spelling as lineage instead.
771
+ for (const finding of reraised) {
772
+ const line = `- [material] ${finding.note}`;
773
+ const bound = unambiguousRows.some((row) => row.reraisedFrom === finding.fingerprint);
774
+ if (!bound && !decided.lines.includes(line))
775
+ decided.lines.push(line);
776
+ }
777
+ const prose = `reviewer ${reviewer.adapter}:${reviewer.model} (vendor: ${reviewer.vendor}; provider: ${provider}): ${decided.headline}${decided.lines.length ? "\n" + decided.lines.join("\n") : ""}`;
778
+ const details = appendAnchoredReview(prose, openAnchors);
779
+ // OBS-1195: a prior's `reraisedFrom` is the link an EARLIER verdict drew. Carried forward unrestated,
780
+ // it would read as this verdict's own material claim on that parent and block the anchor fold from
781
+ // honouring an explicit deferral of it. Lineage lives in observedFingerprints; drop the stale link.
782
+ const carried = carryReviewFindings(reraised.map(({ reraisedFrom: _stale, ...prior }) => prior), rows);
783
+ // An open anchor bound to a current entry names that entry's carried chain, only when unique.
784
+ const parentOf = (anchor) => {
785
+ if (anchor.entry === undefined)
786
+ return anchor.parent;
787
+ const owners = carried.filter((f) => f.class !== "review:anchored" && f.note === anchor.entry);
788
+ return owners.length === 1 ? owners[0].fingerprint : undefined;
789
+ };
790
+ const carriedRows = carried.map((finding) => {
791
+ if (finding.class !== "review:anchored")
792
+ return finding;
793
+ // Every comment spelling this row must name the same parent, else the anchor stays unbound.
794
+ const parents = new Set(anchors.filter((a) => {
795
+ const row = a.settled ? undefined : anchorRow(a);
796
+ return row?.path === finding.path && row.note === finding.note;
797
+ }).map(parentOf));
798
+ const [parent] = parents;
799
+ return parents.size === 1 && parent ? { ...finding, boundTo: parent } : finding;
800
+ });
712
801
  return {
713
802
  gate: "review",
714
803
  pass: decided.pass,
@@ -723,6 +812,7 @@ ${responseRequirement}
723
812
  reraisedMatches: (v.reraised ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
724
813
  } : {}),
725
814
  ...(!decided.pass ? { findings: carriedRows } : {}),
815
+ ...(settledAnchors.length ? { settledAnchors } : {}),
726
816
  ...(saved ? { rawPath: saved } : {}),
727
817
  ...(savedBrief ? { briefPath: savedBrief } : {}),
728
818
  },
@@ -1,5 +1,5 @@
1
1
  import type { CommandReceiptAttribution } from "../run/protocol.js";
2
- import { type Assignment, type BillingChannel, type WorkerAdapter, type WorkerResult } from "../adapters/types.js";
2
+ import { type Assignment, type AuthHealth, type BillingChannel, type WorkerAdapter, type WorkerResult } from "../adapters/types.js";
3
3
  import { type TickmarkrConfig } from "../config/config.js";
4
4
  import { type GateName, type Task } from "../graph/schema.js";
5
5
  import { type Baseline, type GateEvidenceOptions } from "./baseline.js";
@@ -60,18 +60,23 @@ export interface GateContext {
60
60
  baseline: Baseline;
61
61
  channels: BillingChannel[];
62
62
  judgeChannels?: BillingChannel[];
63
+ /** OBS-1186: doctor's cached verdict — the observed identity of the configured judge seat. Absent ⇒ unknown (conservative deny). */
64
+ health?: Record<string, AuthHealth> | null;
63
65
  adapters: WorkerAdapter[];
64
66
  cfg: TickmarkrConfig;
65
67
  via?: GateVia;
66
68
  carriedFindings?: readonly StructuredFinding[];
67
69
  /** Approval reason bound to this attempt; guidance, never criterion or closure authority. */
68
70
  operatorContext?: string;
71
+ /** OBS-1151: this task's earlier parsed judgments, newest first (the journal's, so they survive resume). */
72
+ priorJudgments?: readonly PriorJudgment[];
69
73
  excludeReviewers?: string[];
70
74
  demotedReviewers?: Set<string>;
71
75
  reviewNoVerdicts?: Map<string, string[]>;
72
76
  recheck?: boolean;
73
- /** Explicit worker funding requires fresh red measurements, never a gate waiver. */
74
- cachedRedBypass?: "operator-rerun";
77
+ /** Explicit worker funding requires fresh red measurements, never a gate waiver. OBS-1106: so does
78
+ * a retry that landed nothing on a timeout-class red — its one fresh re-observation. */
79
+ cachedRedBypass?: "operator-rerun" | "timeout-fresh";
75
80
  carriedAuthors?: readonly string[];
76
81
  /** The attempt whose worker last wrote the gated checkout; a dirty-tree refusal stamps it on the
77
82
  * preserve commit and its row. Absent (standalone verify, gate-only restores) preserves as "unknown". */
@@ -92,6 +97,52 @@ export interface GateContext {
92
97
  * a command that already has `--` takes them directly. Every path is quoted — config flows into a shell.
93
98
  */
94
99
  export declare function testCommandForFiles(testCmd: string, files: string[]): string;
100
+ /** OBS-635: a screen costing at least this share of the full suite runs the full suite instead. */
101
+ export declare const SCREEN_PROMOTION_RATIO = 0.75;
102
+ /**
103
+ * The screen's share of the full suite's cost, from the per-file durations the harness measured at
104
+ * baseline capture — never a worker's timing. Undefined (unknown) unless every selected file has a
105
+ * measured duration and the measured total is positive; unknown keeps the conservative screen path.
106
+ */
107
+ export declare function screenCostRatio(baseline: Baseline, selected: readonly string[]): number | undefined;
108
+ /** A retry base no runner invocation parses. evaluateManifestedTest builds its stranded single-fork
109
+ * retry from the base it is handed, and one it cannot parse throws before any spawn — so this base
110
+ * disables that inner recovery: a worker-RPC-stranded re-observation comes back infra (the caller
111
+ * parks it as ambiguous) instead of launching a second execution. */
112
+ export declare const REOBSERVATION_RETRY_BASE = "tickmarkr-reobservation-refuses-stranded-retry";
113
+ /** OBS-1106 residual: ONE isolated re-observation of a timeout-shaped red's attributed failing files on
114
+ * the same checkout, narrowed exactly as a screen is. Never cached and never a verdict: the caller
115
+ * keeps the original red and reads this only to decide whether that red is chargeable. Exactly one
116
+ * execution — the bounded infra/host-starved retries and the stranded single-fork recovery are all
117
+ * refused, so a diagnostic never buys more. */
118
+ export declare function reobserveTestFiles(worktree: string, testCmd: string, baseline: Baseline, files: string[], artifactDir?: string): Promise<GateResult>;
119
+ /** OBS-1151: one parsed judgment as the journal keeps it — the judged commit, the seat, and per criterion
120
+ * its subject key, ruling and cited paths. Never a verdict to reuse: only a subject to compare against. */
121
+ export interface PriorJudgment {
122
+ commit: string;
123
+ judge?: string;
124
+ criteria: ReadonlyArray<{
125
+ id: string;
126
+ key: string;
127
+ met: boolean;
128
+ paths: readonly string[];
129
+ }>;
130
+ }
131
+ /** OBS-1151: a criterion's comparable subject — its canonical text, the task's declared bounds and the
132
+ * operator context. The cited files' blobs are compared separately, over the union of both citations. */
133
+ export declare function judgmentSubjectKey(task: Pick<Task, "files" | "outOfScope">, criterion: string, operatorContext?: string): string;
134
+ export interface JudgeContradiction {
135
+ id: string;
136
+ met: boolean;
137
+ priorMet: boolean;
138
+ priorCommit: string;
139
+ paths: string[];
140
+ }
141
+ /** OBS-1151: the criteria whose fresh ruling reverses the newest prior ruling on the same subject key whose
142
+ * cited paths hold the identical blob at both commits (older comparable priors are still found behind a
143
+ * newer prior on different blobs). A citation-less side or an
144
+ * unreadable blob is unknown, never identical — that criterion's fresh ruling is simply fresh. */
145
+ export declare function judgeContradictions(worktree: string, head: string, fresh: PriorJudgment["criteria"], priors: readonly PriorJudgment[]): Promise<JudgeContradiction[]>;
95
146
  export declare function runGates(task: Task, ctx: GateContext): Promise<{
96
147
  results: GateResult[];
97
148
  commits: string[];