tickmarkr 2.6.1 → 2.6.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. package/README.md +16 -3
  2. package/dist/adapters/catalog-remote.js +89 -47
  3. package/dist/adapters/claude-code.js +9 -6
  4. package/dist/adapters/codex.js +7 -4
  5. package/dist/adapters/prompt.d.ts +1 -0
  6. package/dist/adapters/prompt.js +14 -6
  7. package/dist/adapters/registry.js +3 -3
  8. package/dist/adapters/types.d.ts +12 -4
  9. package/dist/adapters/types.js +6 -0
  10. package/dist/cli/commands/approve.d.ts +11 -4
  11. package/dist/cli/commands/approve.js +82 -27
  12. package/dist/cli/commands/compile.js +13 -3
  13. package/dist/cli/commands/doctor.d.ts +8 -2
  14. package/dist/cli/commands/doctor.js +11 -3
  15. package/dist/cli/commands/fleet.js +87 -11
  16. package/dist/cli/commands/plan.js +13 -8
  17. package/dist/cli/commands/report.d.ts +2 -1
  18. package/dist/cli/commands/report.js +74 -8
  19. package/dist/cli/commands/resume.js +4 -2
  20. package/dist/cli/commands/status.js +43 -20
  21. package/dist/cli/help.d.ts +2 -0
  22. package/dist/cli/help.js +9 -2
  23. package/dist/compile/native.js +7 -0
  24. package/dist/config/config.d.ts +35 -2
  25. package/dist/config/config.js +86 -10
  26. package/dist/config/fleet-overlay.d.ts +13 -2
  27. package/dist/config/fleet-overlay.js +60 -0
  28. package/dist/drivers/herdr.d.ts +12 -0
  29. package/dist/drivers/herdr.js +51 -0
  30. package/dist/drivers/orca.d.ts +35 -2
  31. package/dist/drivers/orca.js +222 -67
  32. package/dist/drivers/types.d.ts +2 -0
  33. package/dist/drivers/types.js +2 -2
  34. package/dist/eval/canary.d.ts +2 -1
  35. package/dist/eval/canary.js +2 -2
  36. package/dist/eval/dispatch.js +1 -0
  37. package/dist/gates/acceptance.d.ts +9 -1
  38. package/dist/gates/acceptance.js +31 -4
  39. package/dist/gates/baseline.d.ts +32 -2
  40. package/dist/gates/baseline.js +111 -24
  41. package/dist/gates/cache.d.ts +8 -0
  42. package/dist/gates/cache.js +12 -2
  43. package/dist/gates/llm.d.ts +11 -4
  44. package/dist/gates/llm.js +40 -21
  45. package/dist/gates/review.d.ts +14 -1
  46. package/dist/gates/review.js +160 -34
  47. package/dist/gates/run-gates.d.ts +56 -4
  48. package/dist/gates/run-gates.js +358 -58
  49. package/dist/gates/test-manifest.d.ts +45 -1
  50. package/dist/gates/test-manifest.js +78 -12
  51. package/dist/graph/schema.d.ts +2 -0
  52. package/dist/graph/schema.js +2 -0
  53. package/dist/plan/scope.js +2 -2
  54. package/dist/route/preference.d.ts +20 -2
  55. package/dist/route/preference.js +48 -13
  56. package/dist/route/router.d.ts +12 -1
  57. package/dist/route/router.js +56 -24
  58. package/dist/run/consult.d.ts +15 -1
  59. package/dist/run/consult.js +18 -7
  60. package/dist/run/daemon.d.ts +38 -2
  61. package/dist/run/daemon.js +895 -192
  62. package/dist/run/git.d.ts +8 -0
  63. package/dist/run/git.js +14 -0
  64. package/dist/run/interactive-seed.d.ts +4 -0
  65. package/dist/run/interactive-seed.js +35 -9
  66. package/dist/run/journal.d.ts +152 -3
  67. package/dist/run/journal.js +551 -50
  68. package/dist/run/lease.d.ts +13 -0
  69. package/dist/run/lease.js +45 -0
  70. package/dist/run/merge.d.ts +3 -1
  71. package/dist/run/merge.js +3 -2
  72. package/dist/run/operator-summary.d.ts +3 -0
  73. package/dist/run/operator-summary.js +3 -1
  74. package/dist/run/protocol.d.ts +46 -1
  75. package/dist/run/protocol.js +14 -2
  76. package/dist/run/receipt-resolver.d.ts +22 -0
  77. package/dist/run/receipt-resolver.js +40 -1
  78. package/dist/run/repair-selection.d.ts +11 -1
  79. package/dist/run/repair-selection.js +17 -9
  80. package/dist/run/supervision.d.ts +7 -1
  81. package/dist/run/supervision.js +5 -2
  82. package/dist/run/wall-budget.d.ts +48 -0
  83. package/dist/run/wall-budget.js +280 -0
  84. package/dist/tui/cockpit/board.js +3 -3
  85. package/dist/tui/cockpit/decision-actions.d.ts +8 -5
  86. package/dist/tui/cockpit/decision-actions.js +55 -32
  87. package/dist/tui/cockpit/derive.js +13 -2
  88. package/dist/tui/cockpit/live-runtime.d.ts +10 -0
  89. package/dist/tui/cockpit/live-runtime.js +50 -3
  90. package/dist/tui/cockpit/run-cockpit.d.ts +3 -0
  91. package/dist/tui/cockpit/run-cockpit.js +27 -2
  92. package/dist/tui/cockpit/run-view.d.ts +9 -2
  93. package/dist/tui/cockpit/run-view.js +66 -9
  94. package/dist/tui/cockpit/setup-cockpit.d.ts +6 -0
  95. package/dist/tui/cockpit/setup-cockpit.js +10 -3
  96. package/dist/tui/ink/fleet-app.d.ts +15 -3
  97. package/dist/tui/ink/fleet-app.js +91 -22
  98. package/package.json +3 -1
  99. package/schema/config.schema.json +825 -0
  100. package/skills/tickmarkr-loop/SKILL.md +15 -3
  101. package/skills/tickmarkr-overseer/SKILL.md +42 -0
  102. package/skills/tickmarkr-overseer/scripts/classify-vitest-log.sh +91 -0
  103. package/skills/tickmarkr-overseer/scripts/context-statusline.sh +81 -0
  104. package/skills/tickmarkr-overseer/scripts/grade-ci.sh +36 -34
  105. package/skills/tickmarkr-overseer/scripts/watch-journal.sh +6 -4
@@ -6,12 +6,12 @@ import { filesGlob } from "../graph/files-glob.js";
6
6
  import { renderAcceptanceItem, TIERS } from "../graph/schema.js";
7
7
  import { getAdapter } from "../adapters/registry.js";
8
8
  import { shOk } from "../run/git.js";
9
- import { carryReviewFindings, observedReviewFingerprints, reviewFingerprintMatches, structuredFindings, UNIDENTIFIED } from "../run/journal.js";
9
+ import { carryReviewFindings, isDeferredFinding, observedReviewFingerprints, reviewFingerprintMatches, structuredFindings, UNIDENTIFIED } from "../run/journal.js";
10
10
  import { redactSecrets } from "../run/redact.js";
11
11
  import { rankPreferredChannels, reviewPreferenceTieBreak } from "../route/role-pick.js";
12
12
  import { modelProvider } from "../route/preference.js";
13
13
  import { resolveStateDir } from "./cache.js";
14
- import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, parseAnchoredComments, runLlmDetailed, verdictNonceLine } from "./llm.js";
14
+ import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, dewrapPaneVerdict, extractVerdictJson, generateVerdictNonce, parseAnchoredComments, runLlmDetailed, SeatLaunchError, verdictNonceLine } from "./llm.js";
15
15
  import { classifyVerdictCause } from "./verdict-cause.js";
16
16
  import { captureDiffCapFor, measureArtifactDiff, reviewableLogicDiff, } from "./artifact-manifest.js";
17
17
  export { isProtectedEvidence, PROTECTED_EVIDENCE_PREFIXES, REGENERABLE_CAPTURE_PATHS, setAsideReceiptPath, setAsideRegenerableCaptures, } from "./artifact-manifest.js";
@@ -164,6 +164,12 @@ export function checkTaskDiffCaps(gate, measured, logicCap, prefix = "") {
164
164
  },
165
165
  };
166
166
  }
167
+ // C-12: a reviewer is excluded for garbage only on the typed malformed-verdict flag. A material finding whose
168
+ // prose mentions the word "unparseable" is a delivered verdict; matching the prose dropped T7's only reviewer.
169
+ export function isGarbageReview(result) {
170
+ return result.gate === "review" && !result.pass && result.meta?.unparseable === true
171
+ && typeof result.meta?.reviewer === "string";
172
+ }
167
173
  export function isDiffCapPark(result) {
168
174
  return result.pass === false
169
175
  && result.meta?.parkKind === "diff-cap"
@@ -373,6 +379,50 @@ ${fingerprints}
373
379
  \`\`\`
374
380
  ${priorMaterials.map((finding, i) => `${i + 1}. ${finding.note}`).join("\n\n")}`;
375
381
  }
382
+ /**
383
+ * OBS-1052(2): a seat that lost the top of a long brief, or believed it had already filed its review,
384
+ * answered in prose — and prose is no verdict. So the requirement, naming THIS call's nonce with a
385
+ * valid example, is both the first and the last instruction of the brief. It is best-effort wording:
386
+ * the parser stays the authority, and nothing here reads approval out of prose.
387
+ */
388
+ export function reviewResponseExample(nonce) {
389
+ return JSON.stringify({
390
+ nonce, approve: false, resolved: [], reraised: [],
391
+ findings: [{ note: "path/to/file.ts:42 — the defect, in one line", severity: "material", defer: false, rationale: "" }],
392
+ comments: [],
393
+ });
394
+ }
395
+ export function reviewResponseRequirement(nonce) {
396
+ return `## Response requirement
397
+ Your reply must end with exactly ONE JSON object whose "nonce" is "${nonce}" — this brief's nonce, never one from an earlier brief. A valid example (a rejection; replace every value with your own verdict):
398
+ ${reviewResponseExample(nonce)}
399
+ This holds even if you already filed or posted a review of this task elsewhere (an earlier session or brief, a PR comment): that review is not on record here, so restate it now as this JSON with nonce "${nonce}". Prose saying a review was filed or approved is recorded as no verdict; approval is never inferred from it.`;
400
+ }
401
+ // The example parses by design, so an echo of the brief (a CLI printing its prompt, a pane showing it)
402
+ // would otherwise read as the seat's own verdict — or as its participation when it wrote only prose.
403
+ // Removed verbatim or hard-wrapped (renderer whitespace and chrome between any two characters) before
404
+ // the verdict is extracted or its absence classified; the saved raw bytes keep it as evidence.
405
+ function withoutExampleEcho(raw, nonce) {
406
+ const chars = [...reviewResponseExample(nonce).replace(/\s+/g, "")];
407
+ return raw.replace(new RegExp(chars.map((c) => c.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")).join("[\\s│|]*"), "g"), "");
408
+ }
409
+ /**
410
+ * OBS-1195: the parent an anchored comment names in its own optional `finding` field — a 1-based
411
+ * index into this verdict's findings, or a carried prior's id. A deferred entry or a resolved prior
412
+ * settles the anchor; any other valid parent binds it open; anything else leaves it unbound.
413
+ */
414
+ function anchorParent(binding, findings, priors, resolved) {
415
+ if (typeof binding === "number") {
416
+ const entry = Array.isArray(findings) && Number.isInteger(binding) && binding >= 1 ? findings[binding - 1] : undefined;
417
+ if (!entry || typeof entry !== "object" || typeof entry.note !== "string")
418
+ return {};
419
+ return entry.defer === true ? { settled: "deferred" } : { entry: entry.note };
420
+ }
421
+ const prior = matchClosureId(binding, priors);
422
+ if (prior === undefined)
423
+ return {};
424
+ return resolved.includes(prior) ? { settled: "resolved", parent: prior } : { parent: prior };
425
+ }
376
426
  export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
377
427
  // OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
378
428
  // direct tests) skips persistence and changes nothing else.
@@ -482,7 +532,10 @@ carriedAuthors = [], operatorContext) {
482
532
  const suiteBudget = ownTestFiles.length
483
533
  ? `You may run at most the task's own test files explicitly named in files[]; these are the only suites you may run: ${ownTestFiles.map((file) => `\`${file}\``).join(", ")}.`
484
534
  : "No suite may be run: files[] names no explicit test file owned by this task.";
535
+ const responseRequirement = reviewResponseRequirement(nonce);
485
536
  const prompt = `TICKMARKR-REVIEW
537
+ ${responseRequirement}
538
+
486
539
  You are a skeptical cross-vendor code reviewer. Another agent (vendor: ${author.adapter}) authored this diff.
487
540
  Look for correctness bugs, security issues, and acceptance-criteria gaps. Approve only if you would merge it.
488
541
 
@@ -520,13 +573,23 @@ block the merge) or "minor" (style, naming, or preference that should not block)
520
573
  block approval. For a minor concern you have decided not to block on, set "defer": true and give a
521
574
  one-line "rationale" — it is recorded in the review, never dropped.
522
575
  A fix you prescribe that would break suites outside the task's declared write scope (files[]) is a scope finding, never a material one.
576
+ Each material finding names the class of defect it belongs to and binds that class to the goal clause or
577
+ acceptance criterion it violates (or to the regression this diff introduces), states its input → consequence,
578
+ and marks its evidence executed (you ran the reproducer), static (you traced it by reading) or blocked (it
579
+ could not run here). Blocked evidence never turns a finding into a pass, and a worker's own case table or
580
+ enumeration never resolves a finding: judge the diff itself.
523
581
 
524
582
  Respond with ONLY this JSON:
525
- {"nonce": "${nonce}", "approve": true|false, "resolved": [], "reraised": [], "findings": [{"note": "...", "severity": "material"|"minor", "defer": false, "rationale": ""}], "comments": [{"path": "path/to/file", "line": 42, "body": "actionable feedback"}]}
583
+ {"nonce": "${nonce}", "approve": true|false, "resolved": [], "reraised": [], "findings": [{"note": "...", "severity": "material"|"minor", "defer": false, "rationale": ""}], "comments": [{"path": "path/to/file", "line": 42, "body": "actionable feedback", "finding": 1}]}
526
584
  For every prior material, put its fingerprint in exactly one of resolved (verified fixed) or reraised
527
585
  (still a blocking defect). Use only the listed fingerprints; never omit one or put it in both lists.
528
586
  Approve iff no material finding remains and every prior material is resolved.
529
587
  The top-level comments array is optional. Use it only for actionable line-anchored feedback.
588
+ A comment's optional "finding" names its parent: the 1-based index of its entry in findings, or a prior
589
+ fingerprint copied from above. A comment anchored to a deferred entry or a resolved prior does not block;
590
+ a comment naming no parent stays open.
591
+
592
+ ${responseRequirement}
530
593
  `;
531
594
  // Filenames are journaled (daemon.ts lifts meta.rawPath/briefPath onto the gate-result row), so they
532
595
  // must be reproducible from the same inputs — the verdict nonce is cryptographically random and would
@@ -558,18 +621,33 @@ The top-level comments array is optional. Use it only for actionable line-anchor
558
621
  savedBrief = undefined;
559
622
  }
560
623
  }
561
- const llm = await runLlmDetailed(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? {
562
- driver: via.driver,
563
- keep: via.keep,
564
- onSlot: via.onSlot,
565
- name: via.nameFor("review", reviewer.adapter),
566
- label: via.labelFor("review"),
567
- } : undefined,
568
- // frontier reviewers routinely need >5min on a configured-cap-sized diff, and `claude -p` buffers all
569
- // output until completion — runLlm's 300s default killed reviews mid-flight, returning empty
570
- // stdout that read as "unparseable" and escalated to re-implementation of green code
571
- // (run-20260709-104447 P87-09). The configured ceiling defaults to that measured 15 minutes.
572
- cfg.review.timeoutMs);
624
+ const provider = modelProvider(reviewer.model, reviewer.vendor);
625
+ let llm;
626
+ try {
627
+ llm = await runLlmDetailed(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? {
628
+ driver: via.driver,
629
+ keep: via.keep,
630
+ onSlot: via.onSlot,
631
+ name: via.nameFor("review", reviewer.adapter),
632
+ label: via.labelFor("review"),
633
+ } : undefined,
634
+ // frontier reviewers routinely need >5min on a configured-cap-sized diff, and `claude -p` buffers all
635
+ // output until completion — runLlm's 300s default killed reviews mid-flight, returning empty
636
+ // stdout that read as "unparseable" and escalated to re-implementation of green code
637
+ // (run-20260709-104447 P87-09). The configured ceiling defaults to that measured 15 minutes.
638
+ cfg.review.timeoutMs, reviewer.effort);
639
+ }
640
+ catch (error) {
641
+ if (!(error instanceof SeatLaunchError))
642
+ throw error;
643
+ // OBS-1168(b): the seat never launched, so there is no verdict and nothing about the WORK. A typed
644
+ // no-verdict re-routes to another seat in run-gates; an exhausted pool is an infra park.
645
+ return { gate: "review", pass: false,
646
+ details: `review dispatch failed — ${error.message} (reviewer ${reviewer.adapter}:${reviewer.model}; vendor: ${reviewer.vendor}; provider: ${provider}; cause: seat-launch-failed) — failing closed`,
647
+ meta: { ...policyMeta, ...rotationMeta, ...floorMeta, reviewer: channelKey(reviewer), reviewerTier: reviewer.tier,
648
+ vendor: reviewer.vendor, provider, noVerdict: true, classification: "infra", infra: true,
649
+ cause: "seat-launch-failed", ...(savedBrief ? { briefPath: savedBrief } : {}) } };
650
+ }
573
651
  const raw = llm.output;
574
652
  let saved;
575
653
  if (artifactDir) {
@@ -581,8 +659,11 @@ The top-level comments array is optional. Use it only for actionable line-anchor
581
659
  saved = undefined; // persistence is evidence, not a gate input — never fail the gate on it
582
660
  }
583
661
  }
584
- const provider = modelProvider(reviewer.model, reviewer.vendor);
585
- const v = extractVerdictJson(raw, nonce);
662
+ // A pane's own dewrap stops at the first parseable nonce-bound object; once the example's echo is gone
663
+ // a genuinely wrapped verdict behind it is reconstructed here, exactly as llm.ts would have.
664
+ const echoFree = withoutExampleEcho(raw, nonce);
665
+ const seat = via ? dewrapPaneVerdict(echoFree, nonce) : echoFree;
666
+ const v = extractVerdictJson(seat, nonce);
586
667
  const findings = v && Array.isArray(v.findings) ? v.findings : null;
587
668
  const priorIds = priorMaterials;
588
669
  const closureInvalid = isReviewClosureInvalid(v, priorIds);
@@ -596,7 +677,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
596
677
  : llm.launchNeverStarted ? "launch-never-started"
597
678
  : llm.silentAtBeat ? "silent"
598
679
  : llm.timedOut ? (bytes > 0 ? "truncated" : "silent")
599
- : classifyVerdictCause(raw, nonce, "approve", llm);
680
+ : classifyVerdictCause(seat, nonce, "approve", llm);
600
681
  const failure = cause === "malformed-verdict"
601
682
  ? "review output unparseable"
602
683
  : cause === "closure-mismatch"
@@ -617,6 +698,8 @@ The top-level comments array is optional. Use it only for actionable line-anchor
617
698
  ...(cause === "malformed-verdict" ? { unparseable: true } : { noVerdict: true, classification: "infra", infra: true }),
618
699
  cause,
619
700
  ...(closureMismatch ? { resolved: v?.resolved, reraised: v?.reraised, carriedFingerprints: priorIds.flatMap(observedReviewFingerprints) } : {}),
701
+ // OBS-1196: a whole verdict that breaks the closure protocol was DELIVERED; only undelivered bytes are re-asked.
702
+ ...(closureInvalid ? { closureInvalid: true } : {}),
620
703
  bytes, seatAuthoredBytes: bytes,
621
704
  ...(saved ? { rawPath: saved } : {}),
622
705
  ...(savedBrief ? { briefPath: savedBrief } : {}),
@@ -633,25 +716,24 @@ The top-level comments array is optional. Use it only for actionable line-anchor
633
716
  if (decided.pass)
634
717
  decided.headline = "requested changes";
635
718
  decided.pass = false;
636
- // A reviewer may also restate a re-raised material in findings. Preserve the original
637
- // prose once so an unchanged defect keeps the same failure brief across repair rounds.
638
- for (const finding of reraised) {
639
- const line = `- [material] ${finding.note}`;
640
- if (!decided.lines.includes(line))
641
- decided.lines.push(line);
642
- }
643
719
  }
644
- const prose = `reviewer ${reviewer.adapter}:${reviewer.model} (vendor: ${reviewer.vendor}; provider: ${provider}): ${decided.headline}${decided.lines.length ? "\n" + decided.lines.join("\n") : ""}`;
645
- const details = appendAnchoredReview(prose, v);
720
+ // OBS-1195: an anchor leaves the blocking set only through an explicit, unambiguous parent
721
+ // disposition: its own `finding` names a deferred entry of this verdict or a prior it resolved.
722
+ // Path coincidence binds nothing; an unbound or ambiguous anchor stays open exactly as before.
723
+ const comments = parseAnchoredComments(v);
724
+ const rawComments = (comments.length ? v.comments : []);
725
+ const resolvedIds = (v.resolved ?? []).map((id) => matchClosureId(id, priorIds));
726
+ const anchors = comments.map((comment, i) => ({ ...comment, ...anchorParent(rawComments[i]?.finding, v.findings, priorIds, resolvedIds) }));
727
+ const openAnchors = { comments: anchors.filter((a) => !a.settled) };
728
+ const settledAnchors = anchors.flatMap((a, i) => a.settled
729
+ ? [{ path: a.path, line: a.line, body: a.body, disposition: a.settled, finding: rawComments[i].finding }] : []);
646
730
  // Only the verdict's anchors may supply missing evidence, and only when unambiguous.
647
731
  // Reuse the journal's path normalization without changing legacy details-only parsing.
648
- const anchoredPaths = new Set(parseAnchoredComments(v).map((comment) => {
649
- const anchor = structuredFindings("review", `- ${comment.path}:${comment.line} — anchor`)
650
- .find((finding) => finding.class === "review:anchored");
651
- return anchor?.path ?? comment.path;
652
- }));
732
+ const anchorRow = (comment) => structuredFindings("review", `- ${comment.path}:${comment.line} — ${comment.body}`)
733
+ .find((finding) => finding.class === "review:anchored");
734
+ const anchoredPaths = new Set(comments.map((comment) => anchorRow({ ...comment, body: "anchor" })?.path ?? comment.path));
653
735
  const anchoredPath = anchoredPaths.size === 1 ? [...anchoredPaths][0] : undefined;
654
- const ownDetails = appendAnchoredReview(ownLines.join("\n"), v);
736
+ const ownDetails = appendAnchoredReview(ownLines.join("\n"), openAnchors);
655
737
  const currentRows = (ownDetails.trim() ? structuredFindings("review", ownDetails) : [])
656
738
  .map((finding) => finding.path === UNIDENTIFIED && anchoredPath
657
739
  ? { ...finding, path: anchoredPath, fingerprint: `${finding.class}|${anchoredPath}|${finding.symbol}` }
@@ -672,7 +754,50 @@ The top-level comments array is optional. Use it only for actionable line-anchor
672
754
  const unambiguousRows = linkedRows.map((finding) => finding.reraisedFrom
673
755
  && linkedRows.filter((row) => row.reraisedFrom === finding.reraisedFrom).length > 1
674
756
  ? { ...finding, reraisedFrom: undefined } : finding);
675
- const carriedRows = carryReviewFindings(reraised, unambiguousRows);
757
+ // OBS-1195: a deferred entry echoing a prior's id names that prior as the concern it defers. The
758
+ // link rides the row for the journal's anchor fold only; a deferral never re-seats a material chain.
759
+ // A prior any material row restates — even ambiguously, its link cleared above — is claimed, not
760
+ // deferred: the deferral gets no link, so its bound anchors stay open (fail closed).
761
+ const claimed = new Set(linkedRows.map((row) => row.reraisedFrom).filter((id) => id !== undefined));
762
+ const rows = unambiguousRows.map((finding) => {
763
+ if (!isDeferredFinding(finding))
764
+ return finding;
765
+ const entry = v.findings?.find((entry) => entry && typeof entry === "object" && entry.note === finding.note);
766
+ const id = matchClosureId(entry?.reraised, reraised);
767
+ return id && !claimed.has(id) ? { ...finding, reraisedFrom: id } : finding;
768
+ });
769
+ // OBS-1195: current prose is rendered once. A prior's original wording is echoed only when no
770
+ // current finding positively binds it; a bound chain keeps that spelling as lineage instead.
771
+ for (const finding of reraised) {
772
+ const line = `- [material] ${finding.note}`;
773
+ const bound = unambiguousRows.some((row) => row.reraisedFrom === finding.fingerprint);
774
+ if (!bound && !decided.lines.includes(line))
775
+ decided.lines.push(line);
776
+ }
777
+ const prose = `reviewer ${reviewer.adapter}:${reviewer.model} (vendor: ${reviewer.vendor}; provider: ${provider}): ${decided.headline}${decided.lines.length ? "\n" + decided.lines.join("\n") : ""}`;
778
+ const details = appendAnchoredReview(prose, openAnchors);
779
+ // OBS-1195: a prior's `reraisedFrom` is the link an EARLIER verdict drew. Carried forward unrestated,
780
+ // it would read as this verdict's own material claim on that parent and block the anchor fold from
781
+ // honouring an explicit deferral of it. Lineage lives in observedFingerprints; drop the stale link.
782
+ const carried = carryReviewFindings(reraised.map(({ reraisedFrom: _stale, ...prior }) => prior), rows);
783
+ // An open anchor bound to a current entry names that entry's carried chain, only when unique.
784
+ const parentOf = (anchor) => {
785
+ if (anchor.entry === undefined)
786
+ return anchor.parent;
787
+ const owners = carried.filter((f) => f.class !== "review:anchored" && f.note === anchor.entry);
788
+ return owners.length === 1 ? owners[0].fingerprint : undefined;
789
+ };
790
+ const carriedRows = carried.map((finding) => {
791
+ if (finding.class !== "review:anchored")
792
+ return finding;
793
+ // Every comment spelling this row must name the same parent, else the anchor stays unbound.
794
+ const parents = new Set(anchors.filter((a) => {
795
+ const row = a.settled ? undefined : anchorRow(a);
796
+ return row?.path === finding.path && row.note === finding.note;
797
+ }).map(parentOf));
798
+ const [parent] = parents;
799
+ return parents.size === 1 && parent ? { ...finding, boundTo: parent } : finding;
800
+ });
676
801
  return {
677
802
  gate: "review",
678
803
  pass: decided.pass,
@@ -687,6 +812,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
687
812
  reraisedMatches: (v.reraised ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
688
813
  } : {}),
689
814
  ...(!decided.pass ? { findings: carriedRows } : {}),
815
+ ...(settledAnchors.length ? { settledAnchors } : {}),
690
816
  ...(saved ? { rawPath: saved } : {}),
691
817
  ...(savedBrief ? { briefPath: savedBrief } : {}),
692
818
  },
@@ -1,5 +1,5 @@
1
1
  import type { CommandReceiptAttribution } from "../run/protocol.js";
2
- import { type Assignment, type BillingChannel, type WorkerAdapter, type WorkerResult } from "../adapters/types.js";
2
+ import { type Assignment, type AuthHealth, type BillingChannel, type WorkerAdapter, type WorkerResult } from "../adapters/types.js";
3
3
  import { type TickmarkrConfig } from "../config/config.js";
4
4
  import { type GateName, type Task } from "../graph/schema.js";
5
5
  import { type Baseline, type GateEvidenceOptions } from "./baseline.js";
@@ -17,7 +17,8 @@ export declare function resetLoadProviderForTests(): void;
17
17
  /**
18
18
  * One gate's own measurement, taken WHERE THE GATE RUNS. `durationMs` sums that gate's execution
19
19
  * intervals and nothing between them, so the composite `test` gate (a selected screen, then other
20
- * gates, then the full suite) reports the two suites' cost rather than the span containing them —
20
+ * gates, then the full suite) reports the two suites' cost rather than the span containing them
21
+ * (split across the two rows when the screen is published before semantic gates, OBS-1176) —
21
22
  * and no consumer has to re-derive a duration by subtracting journal timestamps, which measures the
22
23
  * queue as well as the work. Load is sampled at each interval's endpoints and every second within it;
23
24
  * start preserves the scheduling input while max and mean retain sustained interior saturation.
@@ -59,18 +60,23 @@ export interface GateContext {
59
60
  baseline: Baseline;
60
61
  channels: BillingChannel[];
61
62
  judgeChannels?: BillingChannel[];
63
+ /** OBS-1186: doctor's cached verdict — the observed identity of the configured judge seat. Absent ⇒ unknown (conservative deny). */
64
+ health?: Record<string, AuthHealth> | null;
62
65
  adapters: WorkerAdapter[];
63
66
  cfg: TickmarkrConfig;
64
67
  via?: GateVia;
65
68
  carriedFindings?: readonly StructuredFinding[];
66
69
  /** Approval reason bound to this attempt; guidance, never criterion or closure authority. */
67
70
  operatorContext?: string;
71
+ /** OBS-1151: this task's earlier parsed judgments, newest first (the journal's, so they survive resume). */
72
+ priorJudgments?: readonly PriorJudgment[];
68
73
  excludeReviewers?: string[];
69
74
  demotedReviewers?: Set<string>;
70
75
  reviewNoVerdicts?: Map<string, string[]>;
71
76
  recheck?: boolean;
72
- /** Explicit worker funding requires fresh red measurements, never a gate waiver. */
73
- cachedRedBypass?: "operator-rerun";
77
+ /** Explicit worker funding requires fresh red measurements, never a gate waiver. OBS-1106: so does
78
+ * a retry that landed nothing on a timeout-class red — its one fresh re-observation. */
79
+ cachedRedBypass?: "operator-rerun" | "timeout-fresh";
74
80
  carriedAuthors?: readonly string[];
75
81
  /** The attempt whose worker last wrote the gated checkout; a dirty-tree refusal stamps it on the
76
82
  * preserve commit and its row. Absent (standalone verify, gate-only restores) preserves as "unknown". */
@@ -91,6 +97,52 @@ export interface GateContext {
91
97
  * a command that already has `--` takes them directly. Every path is quoted — config flows into a shell.
92
98
  */
93
99
  export declare function testCommandForFiles(testCmd: string, files: string[]): string;
100
+ /** OBS-635: a screen costing at least this share of the full suite runs the full suite instead. */
101
+ export declare const SCREEN_PROMOTION_RATIO = 0.75;
102
+ /**
103
+ * The screen's share of the full suite's cost, from the per-file durations the harness measured at
104
+ * baseline capture — never a worker's timing. Undefined (unknown) unless every selected file has a
105
+ * measured duration and the measured total is positive; unknown keeps the conservative screen path.
106
+ */
107
+ export declare function screenCostRatio(baseline: Baseline, selected: readonly string[]): number | undefined;
108
+ /** A retry base no runner invocation parses. evaluateManifestedTest builds its stranded single-fork
109
+ * retry from the base it is handed, and one it cannot parse throws before any spawn — so this base
110
+ * disables that inner recovery: a worker-RPC-stranded re-observation comes back infra (the caller
111
+ * parks it as ambiguous) instead of launching a second execution. */
112
+ export declare const REOBSERVATION_RETRY_BASE = "tickmarkr-reobservation-refuses-stranded-retry";
113
+ /** OBS-1106 residual: ONE isolated re-observation of a timeout-shaped red's attributed failing files on
114
+ * the same checkout, narrowed exactly as a screen is. Never cached and never a verdict: the caller
115
+ * keeps the original red and reads this only to decide whether that red is chargeable. Exactly one
116
+ * execution — the bounded infra/host-starved retries and the stranded single-fork recovery are all
117
+ * refused, so a diagnostic never buys more. */
118
+ export declare function reobserveTestFiles(worktree: string, testCmd: string, baseline: Baseline, files: string[], artifactDir?: string): Promise<GateResult>;
119
+ /** OBS-1151: one parsed judgment as the journal keeps it — the judged commit, the seat, and per criterion
120
+ * its subject key, ruling and cited paths. Never a verdict to reuse: only a subject to compare against. */
121
+ export interface PriorJudgment {
122
+ commit: string;
123
+ judge?: string;
124
+ criteria: ReadonlyArray<{
125
+ id: string;
126
+ key: string;
127
+ met: boolean;
128
+ paths: readonly string[];
129
+ }>;
130
+ }
131
+ /** OBS-1151: a criterion's comparable subject — its canonical text, the task's declared bounds and the
132
+ * operator context. The cited files' blobs are compared separately, over the union of both citations. */
133
+ export declare function judgmentSubjectKey(task: Pick<Task, "files" | "outOfScope">, criterion: string, operatorContext?: string): string;
134
+ export interface JudgeContradiction {
135
+ id: string;
136
+ met: boolean;
137
+ priorMet: boolean;
138
+ priorCommit: string;
139
+ paths: string[];
140
+ }
141
+ /** OBS-1151: the criteria whose fresh ruling reverses the newest prior ruling on the same subject key whose
142
+ * cited paths hold the identical blob at both commits (older comparable priors are still found behind a
143
+ * newer prior on different blobs). A citation-less side or an
144
+ * unreadable blob is unknown, never identical — that criterion's fresh ruling is simply fresh. */
145
+ export declare function judgeContradictions(worktree: string, head: string, fresh: PriorJudgment["criteria"], priors: readonly PriorJudgment[]): Promise<JudgeContradiction[]>;
94
146
  export declare function runGates(task: Task, ctx: GateContext): Promise<{
95
147
  results: GateResult[];
96
148
  commits: string[];