tickmarkr 2.6.1 → 2.6.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -3
- package/dist/adapters/catalog-remote.js +89 -47
- package/dist/adapters/claude-code.js +9 -6
- package/dist/adapters/codex.js +7 -4
- package/dist/adapters/prompt.d.ts +1 -0
- package/dist/adapters/prompt.js +14 -6
- package/dist/adapters/registry.js +3 -3
- package/dist/adapters/types.d.ts +12 -4
- package/dist/adapters/types.js +6 -0
- package/dist/cli/commands/approve.d.ts +11 -4
- package/dist/cli/commands/approve.js +82 -27
- package/dist/cli/commands/compile.js +13 -3
- package/dist/cli/commands/doctor.d.ts +8 -2
- package/dist/cli/commands/doctor.js +11 -3
- package/dist/cli/commands/fleet.js +87 -11
- package/dist/cli/commands/plan.js +13 -8
- package/dist/cli/commands/report.d.ts +2 -1
- package/dist/cli/commands/report.js +74 -8
- package/dist/cli/commands/resume.js +4 -2
- package/dist/cli/commands/status.js +43 -20
- package/dist/cli/help.d.ts +2 -0
- package/dist/cli/help.js +9 -2
- package/dist/compile/native.js +7 -0
- package/dist/config/config.d.ts +35 -2
- package/dist/config/config.js +86 -10
- package/dist/config/fleet-overlay.d.ts +13 -2
- package/dist/config/fleet-overlay.js +60 -0
- package/dist/drivers/herdr.d.ts +12 -0
- package/dist/drivers/herdr.js +51 -0
- package/dist/drivers/orca.d.ts +35 -2
- package/dist/drivers/orca.js +222 -67
- package/dist/drivers/types.d.ts +2 -0
- package/dist/drivers/types.js +2 -2
- package/dist/eval/canary.d.ts +2 -1
- package/dist/eval/canary.js +2 -2
- package/dist/eval/dispatch.js +1 -0
- package/dist/gates/acceptance.d.ts +9 -1
- package/dist/gates/acceptance.js +31 -4
- package/dist/gates/baseline.d.ts +32 -2
- package/dist/gates/baseline.js +111 -24
- package/dist/gates/cache.d.ts +8 -0
- package/dist/gates/cache.js +12 -2
- package/dist/gates/llm.d.ts +11 -4
- package/dist/gates/llm.js +40 -21
- package/dist/gates/review.d.ts +14 -1
- package/dist/gates/review.js +160 -34
- package/dist/gates/run-gates.d.ts +56 -4
- package/dist/gates/run-gates.js +358 -58
- package/dist/gates/test-manifest.d.ts +45 -1
- package/dist/gates/test-manifest.js +78 -12
- package/dist/graph/schema.d.ts +2 -0
- package/dist/graph/schema.js +2 -0
- package/dist/plan/scope.js +2 -2
- package/dist/route/preference.d.ts +20 -2
- package/dist/route/preference.js +48 -13
- package/dist/route/router.d.ts +12 -1
- package/dist/route/router.js +56 -24
- package/dist/run/consult.d.ts +15 -1
- package/dist/run/consult.js +18 -7
- package/dist/run/daemon.d.ts +38 -2
- package/dist/run/daemon.js +895 -192
- package/dist/run/git.d.ts +8 -0
- package/dist/run/git.js +14 -0
- package/dist/run/interactive-seed.d.ts +4 -0
- package/dist/run/interactive-seed.js +35 -9
- package/dist/run/journal.d.ts +152 -3
- package/dist/run/journal.js +551 -50
- package/dist/run/lease.d.ts +13 -0
- package/dist/run/lease.js +45 -0
- package/dist/run/merge.d.ts +3 -1
- package/dist/run/merge.js +3 -2
- package/dist/run/operator-summary.d.ts +3 -0
- package/dist/run/operator-summary.js +3 -1
- package/dist/run/protocol.d.ts +46 -1
- package/dist/run/protocol.js +14 -2
- package/dist/run/receipt-resolver.d.ts +22 -0
- package/dist/run/receipt-resolver.js +40 -1
- package/dist/run/repair-selection.d.ts +11 -1
- package/dist/run/repair-selection.js +17 -9
- package/dist/run/supervision.d.ts +7 -1
- package/dist/run/supervision.js +5 -2
- package/dist/run/wall-budget.d.ts +48 -0
- package/dist/run/wall-budget.js +280 -0
- package/dist/tui/cockpit/board.js +3 -3
- package/dist/tui/cockpit/decision-actions.d.ts +8 -5
- package/dist/tui/cockpit/decision-actions.js +55 -32
- package/dist/tui/cockpit/derive.js +13 -2
- package/dist/tui/cockpit/live-runtime.d.ts +10 -0
- package/dist/tui/cockpit/live-runtime.js +50 -3
- package/dist/tui/cockpit/run-cockpit.d.ts +3 -0
- package/dist/tui/cockpit/run-cockpit.js +27 -2
- package/dist/tui/cockpit/run-view.d.ts +9 -2
- package/dist/tui/cockpit/run-view.js +66 -9
- package/dist/tui/cockpit/setup-cockpit.d.ts +6 -0
- package/dist/tui/cockpit/setup-cockpit.js +10 -3
- package/dist/tui/ink/fleet-app.d.ts +15 -3
- package/dist/tui/ink/fleet-app.js +91 -22
- package/package.json +3 -1
- package/schema/config.schema.json +825 -0
- package/skills/tickmarkr-loop/SKILL.md +15 -3
- package/skills/tickmarkr-overseer/SKILL.md +42 -0
- package/skills/tickmarkr-overseer/scripts/classify-vitest-log.sh +91 -0
- package/skills/tickmarkr-overseer/scripts/context-statusline.sh +81 -0
- package/skills/tickmarkr-overseer/scripts/grade-ci.sh +36 -34
- package/skills/tickmarkr-overseer/scripts/watch-journal.sh +6 -4
package/dist/gates/review.js
CHANGED
|
@@ -6,12 +6,12 @@ import { filesGlob } from "../graph/files-glob.js";
|
|
|
6
6
|
import { renderAcceptanceItem, TIERS } from "../graph/schema.js";
|
|
7
7
|
import { getAdapter } from "../adapters/registry.js";
|
|
8
8
|
import { shOk } from "../run/git.js";
|
|
9
|
-
import { carryReviewFindings, observedReviewFingerprints, reviewFingerprintMatches, structuredFindings, UNIDENTIFIED } from "../run/journal.js";
|
|
9
|
+
import { carryReviewFindings, isDeferredFinding, observedReviewFingerprints, reviewFingerprintMatches, structuredFindings, UNIDENTIFIED } from "../run/journal.js";
|
|
10
10
|
import { redactSecrets } from "../run/redact.js";
|
|
11
11
|
import { rankPreferredChannels, reviewPreferenceTieBreak } from "../route/role-pick.js";
|
|
12
12
|
import { modelProvider } from "../route/preference.js";
|
|
13
13
|
import { resolveStateDir } from "./cache.js";
|
|
14
|
-
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, parseAnchoredComments, runLlmDetailed, verdictNonceLine } from "./llm.js";
|
|
14
|
+
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, dewrapPaneVerdict, extractVerdictJson, generateVerdictNonce, parseAnchoredComments, runLlmDetailed, SeatLaunchError, verdictNonceLine } from "./llm.js";
|
|
15
15
|
import { classifyVerdictCause } from "./verdict-cause.js";
|
|
16
16
|
import { captureDiffCapFor, measureArtifactDiff, reviewableLogicDiff, } from "./artifact-manifest.js";
|
|
17
17
|
export { isProtectedEvidence, PROTECTED_EVIDENCE_PREFIXES, REGENERABLE_CAPTURE_PATHS, setAsideReceiptPath, setAsideRegenerableCaptures, } from "./artifact-manifest.js";
|
|
@@ -164,6 +164,12 @@ export function checkTaskDiffCaps(gate, measured, logicCap, prefix = "") {
|
|
|
164
164
|
},
|
|
165
165
|
};
|
|
166
166
|
}
|
|
167
|
+
// C-12: a reviewer is excluded for garbage only on the typed malformed-verdict flag. A material finding whose
|
|
168
|
+
// prose mentions the word "unparseable" is a delivered verdict; matching the prose dropped T7's only reviewer.
|
|
169
|
+
export function isGarbageReview(result) {
|
|
170
|
+
return result.gate === "review" && !result.pass && result.meta?.unparseable === true
|
|
171
|
+
&& typeof result.meta?.reviewer === "string";
|
|
172
|
+
}
|
|
167
173
|
export function isDiffCapPark(result) {
|
|
168
174
|
return result.pass === false
|
|
169
175
|
&& result.meta?.parkKind === "diff-cap"
|
|
@@ -373,6 +379,50 @@ ${fingerprints}
|
|
|
373
379
|
\`\`\`
|
|
374
380
|
${priorMaterials.map((finding, i) => `${i + 1}. ${finding.note}`).join("\n\n")}`;
|
|
375
381
|
}
|
|
382
|
+
/**
|
|
383
|
+
* OBS-1052(2): a seat that lost the top of a long brief, or believed it had already filed its review,
|
|
384
|
+
* answered in prose — and prose is no verdict. So the requirement, naming THIS call's nonce with a
|
|
385
|
+
* valid example, is both the first and the last instruction of the brief. It is best-effort wording:
|
|
386
|
+
* the parser stays the authority, and nothing here reads approval out of prose.
|
|
387
|
+
*/
|
|
388
|
+
export function reviewResponseExample(nonce) {
|
|
389
|
+
return JSON.stringify({
|
|
390
|
+
nonce, approve: false, resolved: [], reraised: [],
|
|
391
|
+
findings: [{ note: "path/to/file.ts:42 — the defect, in one line", severity: "material", defer: false, rationale: "" }],
|
|
392
|
+
comments: [],
|
|
393
|
+
});
|
|
394
|
+
}
|
|
395
|
+
export function reviewResponseRequirement(nonce) {
|
|
396
|
+
return `## Response requirement
|
|
397
|
+
Your reply must end with exactly ONE JSON object whose "nonce" is "${nonce}" — this brief's nonce, never one from an earlier brief. A valid example (a rejection; replace every value with your own verdict):
|
|
398
|
+
${reviewResponseExample(nonce)}
|
|
399
|
+
This holds even if you already filed or posted a review of this task elsewhere (an earlier session or brief, a PR comment): that review is not on record here, so restate it now as this JSON with nonce "${nonce}". Prose saying a review was filed or approved is recorded as no verdict; approval is never inferred from it.`;
|
|
400
|
+
}
|
|
401
|
+
// The example parses by design, so an echo of the brief (a CLI printing its prompt, a pane showing it)
|
|
402
|
+
// would otherwise read as the seat's own verdict — or as its participation when it wrote only prose.
|
|
403
|
+
// Removed verbatim or hard-wrapped (renderer whitespace and chrome between any two characters) before
|
|
404
|
+
// the verdict is extracted or its absence classified; the saved raw bytes keep it as evidence.
|
|
405
|
+
function withoutExampleEcho(raw, nonce) {
|
|
406
|
+
const chars = [...reviewResponseExample(nonce).replace(/\s+/g, "")];
|
|
407
|
+
return raw.replace(new RegExp(chars.map((c) => c.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")).join("[\\s│|]*"), "g"), "");
|
|
408
|
+
}
|
|
409
|
+
/**
|
|
410
|
+
* OBS-1195: the parent an anchored comment names in its own optional `finding` field — a 1-based
|
|
411
|
+
* index into this verdict's findings, or a carried prior's id. A deferred entry or a resolved prior
|
|
412
|
+
* settles the anchor; any other valid parent binds it open; anything else leaves it unbound.
|
|
413
|
+
*/
|
|
414
|
+
function anchorParent(binding, findings, priors, resolved) {
|
|
415
|
+
if (typeof binding === "number") {
|
|
416
|
+
const entry = Array.isArray(findings) && Number.isInteger(binding) && binding >= 1 ? findings[binding - 1] : undefined;
|
|
417
|
+
if (!entry || typeof entry !== "object" || typeof entry.note !== "string")
|
|
418
|
+
return {};
|
|
419
|
+
return entry.defer === true ? { settled: "deferred" } : { entry: entry.note };
|
|
420
|
+
}
|
|
421
|
+
const prior = matchClosureId(binding, priors);
|
|
422
|
+
if (prior === undefined)
|
|
423
|
+
return {};
|
|
424
|
+
return resolved.includes(prior) ? { settled: "resolved", parent: prior } : { parent: prior };
|
|
425
|
+
}
|
|
376
426
|
export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
|
|
377
427
|
// OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
|
|
378
428
|
// direct tests) skips persistence and changes nothing else.
|
|
@@ -482,7 +532,10 @@ carriedAuthors = [], operatorContext) {
|
|
|
482
532
|
const suiteBudget = ownTestFiles.length
|
|
483
533
|
? `You may run at most the task's own test files explicitly named in files[]; these are the only suites you may run: ${ownTestFiles.map((file) => `\`${file}\``).join(", ")}.`
|
|
484
534
|
: "No suite may be run: files[] names no explicit test file owned by this task.";
|
|
535
|
+
const responseRequirement = reviewResponseRequirement(nonce);
|
|
485
536
|
const prompt = `TICKMARKR-REVIEW
|
|
537
|
+
${responseRequirement}
|
|
538
|
+
|
|
486
539
|
You are a skeptical cross-vendor code reviewer. Another agent (vendor: ${author.adapter}) authored this diff.
|
|
487
540
|
Look for correctness bugs, security issues, and acceptance-criteria gaps. Approve only if you would merge it.
|
|
488
541
|
|
|
@@ -520,13 +573,23 @@ block the merge) or "minor" (style, naming, or preference that should not block)
|
|
|
520
573
|
block approval. For a minor concern you have decided not to block on, set "defer": true and give a
|
|
521
574
|
one-line "rationale" — it is recorded in the review, never dropped.
|
|
522
575
|
A fix you prescribe that would break suites outside the task's declared write scope (files[]) is a scope finding, never a material one.
|
|
576
|
+
Each material finding names the class of defect it belongs to and binds that class to the goal clause or
|
|
577
|
+
acceptance criterion it violates (or to the regression this diff introduces), states its input → consequence,
|
|
578
|
+
and marks its evidence executed (you ran the reproducer), static (you traced it by reading) or blocked (it
|
|
579
|
+
could not run here). Blocked evidence never turns a finding into a pass, and a worker's own case table or
|
|
580
|
+
enumeration never resolves a finding: judge the diff itself.
|
|
523
581
|
|
|
524
582
|
Respond with ONLY this JSON:
|
|
525
|
-
{"nonce": "${nonce}", "approve": true|false, "resolved": [], "reraised": [], "findings": [{"note": "...", "severity": "material"|"minor", "defer": false, "rationale": ""}], "comments": [{"path": "path/to/file", "line": 42, "body": "actionable feedback"}]}
|
|
583
|
+
{"nonce": "${nonce}", "approve": true|false, "resolved": [], "reraised": [], "findings": [{"note": "...", "severity": "material"|"minor", "defer": false, "rationale": ""}], "comments": [{"path": "path/to/file", "line": 42, "body": "actionable feedback", "finding": 1}]}
|
|
526
584
|
For every prior material, put its fingerprint in exactly one of resolved (verified fixed) or reraised
|
|
527
585
|
(still a blocking defect). Use only the listed fingerprints; never omit one or put it in both lists.
|
|
528
586
|
Approve iff no material finding remains and every prior material is resolved.
|
|
529
587
|
The top-level comments array is optional. Use it only for actionable line-anchored feedback.
|
|
588
|
+
A comment's optional "finding" names its parent: the 1-based index of its entry in findings, or a prior
|
|
589
|
+
fingerprint copied from above. A comment anchored to a deferred entry or a resolved prior does not block;
|
|
590
|
+
a comment naming no parent stays open.
|
|
591
|
+
|
|
592
|
+
${responseRequirement}
|
|
530
593
|
`;
|
|
531
594
|
// Filenames are journaled (daemon.ts lifts meta.rawPath/briefPath onto the gate-result row), so they
|
|
532
595
|
// must be reproducible from the same inputs — the verdict nonce is cryptographically random and would
|
|
@@ -558,18 +621,33 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
558
621
|
savedBrief = undefined;
|
|
559
622
|
}
|
|
560
623
|
}
|
|
561
|
-
const
|
|
562
|
-
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
624
|
+
const provider = modelProvider(reviewer.model, reviewer.vendor);
|
|
625
|
+
let llm;
|
|
626
|
+
try {
|
|
627
|
+
llm = await runLlmDetailed(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? {
|
|
628
|
+
driver: via.driver,
|
|
629
|
+
keep: via.keep,
|
|
630
|
+
onSlot: via.onSlot,
|
|
631
|
+
name: via.nameFor("review", reviewer.adapter),
|
|
632
|
+
label: via.labelFor("review"),
|
|
633
|
+
} : undefined,
|
|
634
|
+
// frontier reviewers routinely need >5min on a configured-cap-sized diff, and `claude -p` buffers all
|
|
635
|
+
// output until completion — runLlm's 300s default killed reviews mid-flight, returning empty
|
|
636
|
+
// stdout that read as "unparseable" and escalated to re-implementation of green code
|
|
637
|
+
// (run-20260709-104447 P87-09). The configured ceiling defaults to that measured 15 minutes.
|
|
638
|
+
cfg.review.timeoutMs, reviewer.effort);
|
|
639
|
+
}
|
|
640
|
+
catch (error) {
|
|
641
|
+
if (!(error instanceof SeatLaunchError))
|
|
642
|
+
throw error;
|
|
643
|
+
// OBS-1168(b): the seat never launched, so there is no verdict and nothing about the WORK. A typed
|
|
644
|
+
// no-verdict re-routes to another seat in run-gates; an exhausted pool is an infra park.
|
|
645
|
+
return { gate: "review", pass: false,
|
|
646
|
+
details: `review dispatch failed — ${error.message} (reviewer ${reviewer.adapter}:${reviewer.model}; vendor: ${reviewer.vendor}; provider: ${provider}; cause: seat-launch-failed) — failing closed`,
|
|
647
|
+
meta: { ...policyMeta, ...rotationMeta, ...floorMeta, reviewer: channelKey(reviewer), reviewerTier: reviewer.tier,
|
|
648
|
+
vendor: reviewer.vendor, provider, noVerdict: true, classification: "infra", infra: true,
|
|
649
|
+
cause: "seat-launch-failed", ...(savedBrief ? { briefPath: savedBrief } : {}) } };
|
|
650
|
+
}
|
|
573
651
|
const raw = llm.output;
|
|
574
652
|
let saved;
|
|
575
653
|
if (artifactDir) {
|
|
@@ -581,8 +659,11 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
581
659
|
saved = undefined; // persistence is evidence, not a gate input — never fail the gate on it
|
|
582
660
|
}
|
|
583
661
|
}
|
|
584
|
-
|
|
585
|
-
|
|
662
|
+
// A pane's own dewrap stops at the first parseable nonce-bound object; once the example's echo is gone
|
|
663
|
+
// a genuinely wrapped verdict behind it is reconstructed here, exactly as llm.ts would have.
|
|
664
|
+
const echoFree = withoutExampleEcho(raw, nonce);
|
|
665
|
+
const seat = via ? dewrapPaneVerdict(echoFree, nonce) : echoFree;
|
|
666
|
+
const v = extractVerdictJson(seat, nonce);
|
|
586
667
|
const findings = v && Array.isArray(v.findings) ? v.findings : null;
|
|
587
668
|
const priorIds = priorMaterials;
|
|
588
669
|
const closureInvalid = isReviewClosureInvalid(v, priorIds);
|
|
@@ -596,7 +677,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
596
677
|
: llm.launchNeverStarted ? "launch-never-started"
|
|
597
678
|
: llm.silentAtBeat ? "silent"
|
|
598
679
|
: llm.timedOut ? (bytes > 0 ? "truncated" : "silent")
|
|
599
|
-
: classifyVerdictCause(
|
|
680
|
+
: classifyVerdictCause(seat, nonce, "approve", llm);
|
|
600
681
|
const failure = cause === "malformed-verdict"
|
|
601
682
|
? "review output unparseable"
|
|
602
683
|
: cause === "closure-mismatch"
|
|
@@ -617,6 +698,8 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
617
698
|
...(cause === "malformed-verdict" ? { unparseable: true } : { noVerdict: true, classification: "infra", infra: true }),
|
|
618
699
|
cause,
|
|
619
700
|
...(closureMismatch ? { resolved: v?.resolved, reraised: v?.reraised, carriedFingerprints: priorIds.flatMap(observedReviewFingerprints) } : {}),
|
|
701
|
+
// OBS-1196: a whole verdict that breaks the closure protocol was DELIVERED; only undelivered bytes are re-asked.
|
|
702
|
+
...(closureInvalid ? { closureInvalid: true } : {}),
|
|
620
703
|
bytes, seatAuthoredBytes: bytes,
|
|
621
704
|
...(saved ? { rawPath: saved } : {}),
|
|
622
705
|
...(savedBrief ? { briefPath: savedBrief } : {}),
|
|
@@ -633,25 +716,24 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
633
716
|
if (decided.pass)
|
|
634
717
|
decided.headline = "requested changes";
|
|
635
718
|
decided.pass = false;
|
|
636
|
-
// A reviewer may also restate a re-raised material in findings. Preserve the original
|
|
637
|
-
// prose once so an unchanged defect keeps the same failure brief across repair rounds.
|
|
638
|
-
for (const finding of reraised) {
|
|
639
|
-
const line = `- [material] ${finding.note}`;
|
|
640
|
-
if (!decided.lines.includes(line))
|
|
641
|
-
decided.lines.push(line);
|
|
642
|
-
}
|
|
643
719
|
}
|
|
644
|
-
|
|
645
|
-
|
|
720
|
+
// OBS-1195: an anchor leaves the blocking set only through an explicit, unambiguous parent
|
|
721
|
+
// disposition: its own `finding` names a deferred entry of this verdict or a prior it resolved.
|
|
722
|
+
// Path coincidence binds nothing; an unbound or ambiguous anchor stays open exactly as before.
|
|
723
|
+
const comments = parseAnchoredComments(v);
|
|
724
|
+
const rawComments = (comments.length ? v.comments : []);
|
|
725
|
+
const resolvedIds = (v.resolved ?? []).map((id) => matchClosureId(id, priorIds));
|
|
726
|
+
const anchors = comments.map((comment, i) => ({ ...comment, ...anchorParent(rawComments[i]?.finding, v.findings, priorIds, resolvedIds) }));
|
|
727
|
+
const openAnchors = { comments: anchors.filter((a) => !a.settled) };
|
|
728
|
+
const settledAnchors = anchors.flatMap((a, i) => a.settled
|
|
729
|
+
? [{ path: a.path, line: a.line, body: a.body, disposition: a.settled, finding: rawComments[i].finding }] : []);
|
|
646
730
|
// Only the verdict's anchors may supply missing evidence, and only when unambiguous.
|
|
647
731
|
// Reuse the journal's path normalization without changing legacy details-only parsing.
|
|
648
|
-
const
|
|
649
|
-
|
|
650
|
-
|
|
651
|
-
return anchor?.path ?? comment.path;
|
|
652
|
-
}));
|
|
732
|
+
const anchorRow = (comment) => structuredFindings("review", `- ${comment.path}:${comment.line} — ${comment.body}`)
|
|
733
|
+
.find((finding) => finding.class === "review:anchored");
|
|
734
|
+
const anchoredPaths = new Set(comments.map((comment) => anchorRow({ ...comment, body: "anchor" })?.path ?? comment.path));
|
|
653
735
|
const anchoredPath = anchoredPaths.size === 1 ? [...anchoredPaths][0] : undefined;
|
|
654
|
-
const ownDetails = appendAnchoredReview(ownLines.join("\n"),
|
|
736
|
+
const ownDetails = appendAnchoredReview(ownLines.join("\n"), openAnchors);
|
|
655
737
|
const currentRows = (ownDetails.trim() ? structuredFindings("review", ownDetails) : [])
|
|
656
738
|
.map((finding) => finding.path === UNIDENTIFIED && anchoredPath
|
|
657
739
|
? { ...finding, path: anchoredPath, fingerprint: `${finding.class}|${anchoredPath}|${finding.symbol}` }
|
|
@@ -672,7 +754,50 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
672
754
|
const unambiguousRows = linkedRows.map((finding) => finding.reraisedFrom
|
|
673
755
|
&& linkedRows.filter((row) => row.reraisedFrom === finding.reraisedFrom).length > 1
|
|
674
756
|
? { ...finding, reraisedFrom: undefined } : finding);
|
|
675
|
-
|
|
757
|
+
// OBS-1195: a deferred entry echoing a prior's id names that prior as the concern it defers. The
|
|
758
|
+
// link rides the row for the journal's anchor fold only; a deferral never re-seats a material chain.
|
|
759
|
+
// A prior any material row restates — even ambiguously, its link cleared above — is claimed, not
|
|
760
|
+
// deferred: the deferral gets no link, so its bound anchors stay open (fail closed).
|
|
761
|
+
const claimed = new Set(linkedRows.map((row) => row.reraisedFrom).filter((id) => id !== undefined));
|
|
762
|
+
const rows = unambiguousRows.map((finding) => {
|
|
763
|
+
if (!isDeferredFinding(finding))
|
|
764
|
+
return finding;
|
|
765
|
+
const entry = v.findings?.find((entry) => entry && typeof entry === "object" && entry.note === finding.note);
|
|
766
|
+
const id = matchClosureId(entry?.reraised, reraised);
|
|
767
|
+
return id && !claimed.has(id) ? { ...finding, reraisedFrom: id } : finding;
|
|
768
|
+
});
|
|
769
|
+
// OBS-1195: current prose is rendered once. A prior's original wording is echoed only when no
|
|
770
|
+
// current finding positively binds it; a bound chain keeps that spelling as lineage instead.
|
|
771
|
+
for (const finding of reraised) {
|
|
772
|
+
const line = `- [material] ${finding.note}`;
|
|
773
|
+
const bound = unambiguousRows.some((row) => row.reraisedFrom === finding.fingerprint);
|
|
774
|
+
if (!bound && !decided.lines.includes(line))
|
|
775
|
+
decided.lines.push(line);
|
|
776
|
+
}
|
|
777
|
+
const prose = `reviewer ${reviewer.adapter}:${reviewer.model} (vendor: ${reviewer.vendor}; provider: ${provider}): ${decided.headline}${decided.lines.length ? "\n" + decided.lines.join("\n") : ""}`;
|
|
778
|
+
const details = appendAnchoredReview(prose, openAnchors);
|
|
779
|
+
// OBS-1195: a prior's `reraisedFrom` is the link an EARLIER verdict drew. Carried forward unrestated,
|
|
780
|
+
// it would read as this verdict's own material claim on that parent and block the anchor fold from
|
|
781
|
+
// honouring an explicit deferral of it. Lineage lives in observedFingerprints; drop the stale link.
|
|
782
|
+
const carried = carryReviewFindings(reraised.map(({ reraisedFrom: _stale, ...prior }) => prior), rows);
|
|
783
|
+
// An open anchor bound to a current entry names that entry's carried chain, only when unique.
|
|
784
|
+
const parentOf = (anchor) => {
|
|
785
|
+
if (anchor.entry === undefined)
|
|
786
|
+
return anchor.parent;
|
|
787
|
+
const owners = carried.filter((f) => f.class !== "review:anchored" && f.note === anchor.entry);
|
|
788
|
+
return owners.length === 1 ? owners[0].fingerprint : undefined;
|
|
789
|
+
};
|
|
790
|
+
const carriedRows = carried.map((finding) => {
|
|
791
|
+
if (finding.class !== "review:anchored")
|
|
792
|
+
return finding;
|
|
793
|
+
// Every comment spelling this row must name the same parent, else the anchor stays unbound.
|
|
794
|
+
const parents = new Set(anchors.filter((a) => {
|
|
795
|
+
const row = a.settled ? undefined : anchorRow(a);
|
|
796
|
+
return row?.path === finding.path && row.note === finding.note;
|
|
797
|
+
}).map(parentOf));
|
|
798
|
+
const [parent] = parents;
|
|
799
|
+
return parents.size === 1 && parent ? { ...finding, boundTo: parent } : finding;
|
|
800
|
+
});
|
|
676
801
|
return {
|
|
677
802
|
gate: "review",
|
|
678
803
|
pass: decided.pass,
|
|
@@ -687,6 +812,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
687
812
|
reraisedMatches: (v.reraised ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
|
|
688
813
|
} : {}),
|
|
689
814
|
...(!decided.pass ? { findings: carriedRows } : {}),
|
|
815
|
+
...(settledAnchors.length ? { settledAnchors } : {}),
|
|
690
816
|
...(saved ? { rawPath: saved } : {}),
|
|
691
817
|
...(savedBrief ? { briefPath: savedBrief } : {}),
|
|
692
818
|
},
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { CommandReceiptAttribution } from "../run/protocol.js";
|
|
2
|
-
import { type Assignment, type BillingChannel, type WorkerAdapter, type WorkerResult } from "../adapters/types.js";
|
|
2
|
+
import { type Assignment, type AuthHealth, type BillingChannel, type WorkerAdapter, type WorkerResult } from "../adapters/types.js";
|
|
3
3
|
import { type TickmarkrConfig } from "../config/config.js";
|
|
4
4
|
import { type GateName, type Task } from "../graph/schema.js";
|
|
5
5
|
import { type Baseline, type GateEvidenceOptions } from "./baseline.js";
|
|
@@ -17,7 +17,8 @@ export declare function resetLoadProviderForTests(): void;
|
|
|
17
17
|
/**
|
|
18
18
|
* One gate's own measurement, taken WHERE THE GATE RUNS. `durationMs` sums that gate's execution
|
|
19
19
|
* intervals and nothing between them, so the composite `test` gate (a selected screen, then other
|
|
20
|
-
* gates, then the full suite) reports the two suites' cost rather than the span containing them
|
|
20
|
+
* gates, then the full suite) reports the two suites' cost rather than the span containing them
|
|
21
|
+
* (split across the two rows when the screen is published before semantic gates, OBS-1176) —
|
|
21
22
|
* and no consumer has to re-derive a duration by subtracting journal timestamps, which measures the
|
|
22
23
|
* queue as well as the work. Load is sampled at each interval's endpoints and every second within it;
|
|
23
24
|
* start preserves the scheduling input while max and mean retain sustained interior saturation.
|
|
@@ -59,18 +60,23 @@ export interface GateContext {
|
|
|
59
60
|
baseline: Baseline;
|
|
60
61
|
channels: BillingChannel[];
|
|
61
62
|
judgeChannels?: BillingChannel[];
|
|
63
|
+
/** OBS-1186: doctor's cached verdict — the observed identity of the configured judge seat. Absent ⇒ unknown (conservative deny). */
|
|
64
|
+
health?: Record<string, AuthHealth> | null;
|
|
62
65
|
adapters: WorkerAdapter[];
|
|
63
66
|
cfg: TickmarkrConfig;
|
|
64
67
|
via?: GateVia;
|
|
65
68
|
carriedFindings?: readonly StructuredFinding[];
|
|
66
69
|
/** Approval reason bound to this attempt; guidance, never criterion or closure authority. */
|
|
67
70
|
operatorContext?: string;
|
|
71
|
+
/** OBS-1151: this task's earlier parsed judgments, newest first (the journal's, so they survive resume). */
|
|
72
|
+
priorJudgments?: readonly PriorJudgment[];
|
|
68
73
|
excludeReviewers?: string[];
|
|
69
74
|
demotedReviewers?: Set<string>;
|
|
70
75
|
reviewNoVerdicts?: Map<string, string[]>;
|
|
71
76
|
recheck?: boolean;
|
|
72
|
-
/** Explicit worker funding requires fresh red measurements, never a gate waiver.
|
|
73
|
-
|
|
77
|
+
/** Explicit worker funding requires fresh red measurements, never a gate waiver. OBS-1106: so does
|
|
78
|
+
* a retry that landed nothing on a timeout-class red — its one fresh re-observation. */
|
|
79
|
+
cachedRedBypass?: "operator-rerun" | "timeout-fresh";
|
|
74
80
|
carriedAuthors?: readonly string[];
|
|
75
81
|
/** The attempt whose worker last wrote the gated checkout; a dirty-tree refusal stamps it on the
|
|
76
82
|
* preserve commit and its row. Absent (standalone verify, gate-only restores) preserves as "unknown". */
|
|
@@ -91,6 +97,52 @@ export interface GateContext {
|
|
|
91
97
|
* a command that already has `--` takes them directly. Every path is quoted — config flows into a shell.
|
|
92
98
|
*/
|
|
93
99
|
export declare function testCommandForFiles(testCmd: string, files: string[]): string;
|
|
100
|
+
/** OBS-635: a screen costing at least this share of the full suite runs the full suite instead. */
|
|
101
|
+
export declare const SCREEN_PROMOTION_RATIO = 0.75;
|
|
102
|
+
/**
|
|
103
|
+
* The screen's share of the full suite's cost, from the per-file durations the harness measured at
|
|
104
|
+
* baseline capture — never a worker's timing. Undefined (unknown) unless every selected file has a
|
|
105
|
+
* measured duration and the measured total is positive; unknown keeps the conservative screen path.
|
|
106
|
+
*/
|
|
107
|
+
export declare function screenCostRatio(baseline: Baseline, selected: readonly string[]): number | undefined;
|
|
108
|
+
/** A retry base no runner invocation parses. evaluateManifestedTest builds its stranded single-fork
|
|
109
|
+
* retry from the base it is handed, and one it cannot parse throws before any spawn — so this base
|
|
110
|
+
* disables that inner recovery: a worker-RPC-stranded re-observation comes back infra (the caller
|
|
111
|
+
* parks it as ambiguous) instead of launching a second execution. */
|
|
112
|
+
export declare const REOBSERVATION_RETRY_BASE = "tickmarkr-reobservation-refuses-stranded-retry";
|
|
113
|
+
/** OBS-1106 residual: ONE isolated re-observation of a timeout-shaped red's attributed failing files on
|
|
114
|
+
* the same checkout, narrowed exactly as a screen is. Never cached and never a verdict: the caller
|
|
115
|
+
* keeps the original red and reads this only to decide whether that red is chargeable. Exactly one
|
|
116
|
+
* execution — the bounded infra/host-starved retries and the stranded single-fork recovery are all
|
|
117
|
+
* refused, so a diagnostic never buys more. */
|
|
118
|
+
export declare function reobserveTestFiles(worktree: string, testCmd: string, baseline: Baseline, files: string[], artifactDir?: string): Promise<GateResult>;
|
|
119
|
+
/** OBS-1151: one parsed judgment as the journal keeps it — the judged commit, the seat, and per criterion
|
|
120
|
+
* its subject key, ruling and cited paths. Never a verdict to reuse: only a subject to compare against. */
|
|
121
|
+
export interface PriorJudgment {
|
|
122
|
+
commit: string;
|
|
123
|
+
judge?: string;
|
|
124
|
+
criteria: ReadonlyArray<{
|
|
125
|
+
id: string;
|
|
126
|
+
key: string;
|
|
127
|
+
met: boolean;
|
|
128
|
+
paths: readonly string[];
|
|
129
|
+
}>;
|
|
130
|
+
}
|
|
131
|
+
/** OBS-1151: a criterion's comparable subject — its canonical text, the task's declared bounds and the
|
|
132
|
+
* operator context. The cited files' blobs are compared separately, over the union of both citations. */
|
|
133
|
+
export declare function judgmentSubjectKey(task: Pick<Task, "files" | "outOfScope">, criterion: string, operatorContext?: string): string;
|
|
134
|
+
export interface JudgeContradiction {
|
|
135
|
+
id: string;
|
|
136
|
+
met: boolean;
|
|
137
|
+
priorMet: boolean;
|
|
138
|
+
priorCommit: string;
|
|
139
|
+
paths: string[];
|
|
140
|
+
}
|
|
141
|
+
/** OBS-1151: the criteria whose fresh ruling reverses the newest prior ruling on the same subject key whose
|
|
142
|
+
* cited paths hold the identical blob at both commits (older comparable priors are still found behind a
|
|
143
|
+
* newer prior on different blobs). A citation-less side or an
|
|
144
|
+
* unreadable blob is unknown, never identical — that criterion's fresh ruling is simply fresh. */
|
|
145
|
+
export declare function judgeContradictions(worktree: string, head: string, fresh: PriorJudgment["criteria"], priors: readonly PriorJudgment[]): Promise<JudgeContradiction[]>;
|
|
94
146
|
export declare function runGates(task: Task, ctx: GateContext): Promise<{
|
|
95
147
|
results: GateResult[];
|
|
96
148
|
commits: string[];
|