tickmarkr 2.5.8 → 2.5.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/qwen.js +30 -3
- package/dist/cli/commands/approve.js +15 -3
- package/dist/cli/commands/beat.d.ts +2 -0
- package/dist/cli/commands/beat.js +28 -29
- package/dist/cli/commands/compile.js +11 -0
- package/dist/cli/commands/fleet.js +61 -53
- package/dist/cli/commands/verify.d.ts +4 -1
- package/dist/cli/commands/verify.js +16 -5
- package/dist/cli/help.d.ts +2 -0
- package/dist/cli/help.js +6 -4
- package/dist/config/config.d.ts +41 -3
- package/dist/config/config.js +48 -17
- package/dist/config/fleet-overlay.js +47 -62
- package/dist/gates/baseline.d.ts +65 -0
- package/dist/gates/baseline.js +163 -9
- package/dist/gates/review.d.ts +6 -4
- package/dist/gates/review.js +62 -21
- package/dist/gates/run-gates.d.ts +4 -1
- package/dist/gates/run-gates.js +21 -11
- package/dist/gates/test-manifest.d.ts +11 -1
- package/dist/gates/test-manifest.js +41 -21
- package/dist/run/daemon.js +84 -41
- package/dist/run/journal.d.ts +17 -2
- package/dist/run/journal.js +99 -11
- package/dist/run/merge.d.ts +13 -2
- package/dist/run/merge.js +74 -12
- package/dist/run/protocol.d.ts +82 -0
- package/dist/run/protocol.js +35 -0
- package/dist/run/receipt-resolver.d.ts +18 -0
- package/dist/run/receipt-resolver.js +132 -0
- package/dist/run/supervision.d.ts +14 -1
- package/dist/run/supervision.js +122 -24
- package/dist/tui/cockpit/evidence-view.d.ts +10 -1
- package/dist/tui/cockpit/evidence-view.js +37 -5
- package/dist/tui/ink/fleet-app.d.ts +12 -22
- package/dist/tui/ink/fleet-app.js +520 -131
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +61 -36
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +14 -2
package/dist/run/daemon.js
CHANGED
|
@@ -35,7 +35,7 @@ import { classifyRepairDisposition, resolveScopeHints } from "./repair-dispositi
|
|
|
35
35
|
import { applyScopeAmendments, activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, deferredReviewFindings, engagementComparable, formatPriorFindingEvidence, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, isDeferredFinding, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, outstandingConsultGuidance, outstandingReviewFindings, pendingApprovalActions, pendingRechecks, pendingRepairFindings, phaseForGate, readPriorRunEvidence, recordedTaskFailureKind, RECHECK_RELEASE, renderStructuredReviewFinding, repairReachSinceApproval, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, structuredFindings, upheldFeedbackByTask } from "./journal.js";
|
|
36
36
|
import { isDiffCapPark, pickReviewer } from "../gates/review.js";
|
|
37
37
|
import { acquireApprovalSerialization, acquireRunLock, isPidLive, releaseRunLock } from "./lock.js";
|
|
38
|
-
import { ensureIntegration, integrationBranch, integrationHead, mergeTask, verifyIntegrationTip } from "./merge.js";
|
|
38
|
+
import { ensureIntegration, integrationBranch, integrationHead, mergeTask, reusedTipEvidence, verifyIntegrationTip } from "./merge.js";
|
|
39
39
|
import { climbChannel, marginalCostRank, nextChannel, route } from "../route/router.js";
|
|
40
40
|
import { desiredPanes } from "./reconcile.js";
|
|
41
41
|
import { readTierLiveness, readWatchBoard, supervisionBeatPath } from "./supervision.js";
|
|
@@ -164,6 +164,27 @@ export function resolveRunMode(repoRoot, opts = {}) {
|
|
|
164
164
|
: undefined;
|
|
165
165
|
return { cfg: resolved.cfg, mode: resolved.mode, source, ...(conflict ? { conflict } : {}) };
|
|
166
166
|
}
|
|
167
|
+
// Bind approval context before dispatch consumes the release. Restore reconstructs the same
|
|
168
|
+
// binding from journal order; a newer approval, including one without a reason, supersedes it.
|
|
169
|
+
function approvalReviewContext(events, taskId, restore = false) {
|
|
170
|
+
let pending;
|
|
171
|
+
let bound;
|
|
172
|
+
let approved = false;
|
|
173
|
+
for (const row of events) {
|
|
174
|
+
if (row.taskId !== taskId)
|
|
175
|
+
continue;
|
|
176
|
+
if (row.event === "task-approved") {
|
|
177
|
+
pending = typeof row.data.reason === "string" ? row.data.reason.trim() || undefined : undefined;
|
|
178
|
+
approved = true;
|
|
179
|
+
}
|
|
180
|
+
else if (row.event === "task-dispatch" || (row.event === "recheck-battery" && approved)) {
|
|
181
|
+
bound = pending;
|
|
182
|
+
pending = undefined;
|
|
183
|
+
approved = false;
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
return restore && !approved ? bound : pending;
|
|
187
|
+
}
|
|
167
188
|
// T14: the events that prove an approval was ENACTED — narrowly causal, never merely subsequent.
|
|
168
189
|
// Ordinary, attempt-cap and review-upheld approvals buy a WORKER, so their proof is a dispatch;
|
|
169
190
|
// recheck has its own battery enactment below. Generic terminal events are deliberately NOT proof: an approved human-gate
|
|
@@ -664,7 +685,7 @@ function lastVerifyCycle(events) {
|
|
|
664
685
|
if (e.event === "tip-verify-start") {
|
|
665
686
|
const { tip, cmdHash } = e.data;
|
|
666
687
|
cur = typeof tip === "string" && typeof cmdHash === "string"
|
|
667
|
-
? { tip, cmdHash, gates: new Set(), failed: false, forgiven: false, capacities: [e.data.capacity] }
|
|
688
|
+
? { tip, cmdHash, gates: new Set(), failed: false, forgiven: false, capacities: [e.data.capacity], evidenceByGate: new Map() }
|
|
668
689
|
: undefined;
|
|
669
690
|
afterRunEnd = false;
|
|
670
691
|
continue;
|
|
@@ -687,7 +708,7 @@ function lastVerifyCycle(events) {
|
|
|
687
708
|
continue;
|
|
688
709
|
}
|
|
689
710
|
if (!cur || afterRunEnd || cur.tip !== tip || cur.cmdHash !== cmdHash) {
|
|
690
|
-
cur = { tip, cmdHash, gates: new Set(), failed: false, forgiven: false, capacities: [] };
|
|
711
|
+
cur = { tip, cmdHash, gates: new Set(), failed: false, forgiven: false, capacities: [], evidenceByGate: new Map() };
|
|
691
712
|
}
|
|
692
713
|
afterRunEnd = false;
|
|
693
714
|
// T7: EVERY verdict row's own capacity, not just the start row's. The start row is a statement of
|
|
@@ -699,6 +720,7 @@ function lastVerifyCycle(events) {
|
|
|
699
720
|
cur.failed = true;
|
|
700
721
|
else {
|
|
701
722
|
cur.gates.add(gate);
|
|
723
|
+
cur.evidenceByGate.set(gate, e.data);
|
|
702
724
|
// Q121s review M1: a forgiven green depends on the baseline that forgave it. Baselines are not
|
|
703
725
|
// part of the cache key, so a forgiven cycle is NEVER cache-eligible — an unchanged tip whose
|
|
704
726
|
// baseline later vanishes or changes must re-run the full verify, not inherit the green.
|
|
@@ -909,7 +931,7 @@ export async function verifyIntegrationTipCached(intWt, commands, journal, opts
|
|
|
909
931
|
// pass — `cached: true` keeps it honest about not having re-run the command.
|
|
910
932
|
for (const gate of gates) {
|
|
911
933
|
const cmd = gate === "test" && commands.tipTest ? commands.tipTest : commands[gate];
|
|
912
|
-
journal.append("tip-verify", undefined, { gate, cmd, pass: true, exitCode: 0, cached: true, tip, cmdHash, capacity });
|
|
934
|
+
journal.append("tip-verify", undefined, { ...reusedTipEvidence(last.evidenceByGate.get(gate) ?? {}), gate, cmd, pass: true, exitCode: 0, cached: true, tip, cmdHash, capacity });
|
|
913
935
|
}
|
|
914
936
|
return false;
|
|
915
937
|
}
|
|
@@ -918,12 +940,22 @@ export async function verifyIntegrationTipCached(intWt, commands, journal, opts
|
|
|
918
940
|
? cancellableTipBattery(intWt, commands, journal.dir, opts.baseline, opts.signal)
|
|
919
941
|
: verifyIntegrationTip(intWt, commands, journal.dir, opts.baseline))) {
|
|
920
942
|
const measuredCapacity = r.capacity ?? capacity;
|
|
943
|
+
const evidence = r.reused ? reusedTipEvidence(r) : {
|
|
944
|
+
...(r.originRunRoot ? { originRunRoot: r.originRunRoot } : {}),
|
|
945
|
+
...(r.cause ? { cause: r.cause } : {}),
|
|
946
|
+
...(r.evidenceReceipt ? { evidenceReceipt: r.evidenceReceipt } : {}),
|
|
947
|
+
...(r.evidenceReceipts ? { evidenceReceipts: r.evidenceReceipts } : {}),
|
|
948
|
+
...(r.evidenceAbsence ? { evidenceAbsence: r.evidenceAbsence } : {}),
|
|
949
|
+
...Object.fromEntries(["nonce", "stdoutPath", "stderrPath"]
|
|
950
|
+
.filter(key => r[key] !== undefined).map(key => [key, r[key]])),
|
|
951
|
+
};
|
|
921
952
|
if (r.pass) {
|
|
922
953
|
// Q121s: a forgiven pass journals its fingerprints — honest about what was carried, never a silent green.
|
|
923
|
-
journal.append("tip-verify", undefined, { gate: r.gate, cmd: r.cmd, pass: true, exitCode: r.exitCode, details: r.details, ...(r.reused ? { cached: true } : {}), ...(r.forgiven ? { forgiven: true, fingerprints: r.fingerprints } : {}), tip, cmdHash, capacity: measuredCapacity });
|
|
954
|
+
journal.append("tip-verify", undefined, { ...evidence, gate: r.gate, cmd: r.cmd, pass: true, exitCode: r.exitCode, details: r.details, ...(r.reused ? { cached: true } : {}), ...(r.forgiven ? { forgiven: true, fingerprints: r.fingerprints } : {}), tip, cmdHash, capacity: measuredCapacity });
|
|
924
955
|
}
|
|
925
956
|
else {
|
|
926
957
|
journal.append("tip-verify-failed", undefined, {
|
|
958
|
+
...evidence,
|
|
927
959
|
gate: r.gate,
|
|
928
960
|
cmd: r.cmd,
|
|
929
961
|
exitCode: r.exitCode,
|
|
@@ -2550,6 +2582,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2550
2582
|
// run; a park-instead policy can come later if it ever bites.
|
|
2551
2583
|
}
|
|
2552
2584
|
const taskHistory = journal.read().filter((e) => e.taskId === t.id);
|
|
2585
|
+
// Lifetime identity counts every dispatch, including legacy and unchargeable rows. Releases
|
|
2586
|
+
// reset the budget below, never this counter; observational annotations do not consume it.
|
|
2587
|
+
let nextWorkerDispatchOrdinal = taskHistory.filter((e) => e.event === "task-dispatch").length;
|
|
2553
2588
|
const lastApproval = taskHistory.map((e) => e.event).lastIndexOf("task-approved");
|
|
2554
2589
|
const priorClimb = taskHistory.slice(lastApproval + 1).reverse().find((e) => e.event === "tier-escalated");
|
|
2555
2590
|
if (!hintsChanged && priorClimb && taskHistory.indexOf(priorClimb) > taskHistory.map((e) => e.event).lastIndexOf("task-dispatch")) {
|
|
@@ -2566,7 +2601,10 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2566
2601
|
// exclusion-list-equality oracle. Daemon-side append only — no journal.ts write-path change (Phase 48
|
|
2567
2602
|
// stays unblocked); inert to replayStatuses (unknown events ignored, pinned at journal.test.ts:70-80).
|
|
2568
2603
|
if (rs)
|
|
2569
|
-
journal.append("resume-restore", t.id, {
|
|
2604
|
+
journal.append("resume-restore", t.id, {
|
|
2605
|
+
attempts: rs.attempts, tried: [...tried], assignment,
|
|
2606
|
+
workerDispatchOrdinal: previousDispatch?.data.workerDispatchOrdinal ?? null,
|
|
2607
|
+
});
|
|
2570
2608
|
// Keep one live list: recovery retries must see exclusions added by onGate during the round.
|
|
2571
2609
|
const badReviewers = [...replayedReviewerExclusions];
|
|
2572
2610
|
const noteReviewEvent = (e) => {
|
|
@@ -2733,6 +2771,13 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
2733
2771
|
// every recalibration this telemetry funds, a gap is honest and a zero is a lie. The
|
|
2734
2772
|
// seven-gate closed set is asserted end-to-end in tests/run/gate-telemetry.test.ts.
|
|
2735
2773
|
...gateMeasurement(g.meta),
|
|
2774
|
+
// Fresh producer evidence is copied verbatim; absent/historical evidence is never minted here.
|
|
2775
|
+
...(g.meta?.reused ? {} : {
|
|
2776
|
+
...(g.evidenceReceipt ? { evidenceReceipt: g.evidenceReceipt } : {}),
|
|
2777
|
+
...(g.evidenceReceipts ? { evidenceReceipts: g.evidenceReceipts } : {}),
|
|
2778
|
+
...Object.fromEntries(["nonce", "stdoutPath", "stderrPath", "classification"]
|
|
2779
|
+
.filter(key => g.meta?.[key] !== undefined).map(key => [key, g.meta[key]])),
|
|
2780
|
+
}),
|
|
2736
2781
|
// T7: the capacity the gate's own command child ran under, lifted verbatim from the result
|
|
2737
2782
|
// the battery produced — read where the shell built that child's environment, never
|
|
2738
2783
|
// re-derived from the run's own budget, which would answer a different number than the
|
|
@@ -3066,6 +3111,8 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3066
3111
|
continue;
|
|
3067
3112
|
priorResults.set(e.data.gate, e);
|
|
3068
3113
|
}
|
|
3114
|
+
const declaredGates = GATE_NAMES.filter((gate) => t.gates.includes(gate));
|
|
3115
|
+
const reused = [];
|
|
3069
3116
|
let remainingGates;
|
|
3070
3117
|
if (recheck) {
|
|
3071
3118
|
remainingGates = GATE_NAMES.filter((gate) => t.gates.includes(gate));
|
|
@@ -3119,48 +3166,40 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3119
3166
|
});
|
|
3120
3167
|
}
|
|
3121
3168
|
let reusable = sameWorld && sameProtocol && (exactCurrentCommit || canonicalCurrentCommit || recreatedLegacyCommit);
|
|
3122
|
-
const reused = [];
|
|
3123
|
-
const declaredGates = GATE_NAMES.filter((gate) => t.gates.includes(gate));
|
|
3124
3169
|
for (const gate of declaredGates) {
|
|
3125
3170
|
if (!reusable || replayedGates.results.get(gate) !== true)
|
|
3126
3171
|
reusable = false;
|
|
3127
3172
|
else
|
|
3128
3173
|
reused.push(gate);
|
|
3129
3174
|
}
|
|
3130
|
-
// OBS-1049: build outputs are not commits. A replayed build verdict says the tree WAS green;
|
|
3131
|
-
// it says nothing about whether the RECREATED checkout holds what that build produced (dist),
|
|
3132
|
-
// and the host's npm lifecycle (ignore-scripts) may never rebuild it before the suite runs.
|
|
3133
|
-
// So a replayed build re-runs its own command here as provisioning, journaled by name; a red
|
|
3134
|
-
// provisioning is not reused — build (and everything behind it) re-enters the battery as a gate.
|
|
3135
|
-
// Every resume on this path recreates the checkout (recreateTaskWorktree above), so "reused
|
|
3136
|
-
// build" is exactly "build replayed onto a recreated tree".
|
|
3137
|
-
// Journal order is the criterion's order: a GREEN provisioning is recorded AFTER the reuse
|
|
3138
|
-
// rows it belongs to (gate-reused build, then gate-provisioned build); a RED provisioning is
|
|
3139
|
-
// recorded alone, and no reuse row follows it.
|
|
3140
|
-
let provisionedRow;
|
|
3141
|
-
if (reused.includes("build") && commands.build !== undefined) {
|
|
3142
|
-
// Provisioning is a verification command like any gate: it waits for the run's baseline and
|
|
3143
|
-
// takes the command lease (suite admission), never a bare shell beside sibling gates.
|
|
3144
|
-
await waitForBaseline(t.id);
|
|
3145
|
-
const startedAt = Date.now();
|
|
3146
|
-
const provisioned = await withCommandContext(t.id, () => sh(commands.build, wt));
|
|
3147
|
-
provisionedRow = {
|
|
3148
|
-
gate: "build", commit: replayedGates.commit, exitCode: provisioned.code, durationMs: Date.now() - startedAt,
|
|
3149
|
-
};
|
|
3150
|
-
if (provisioned.code !== 0) {
|
|
3151
|
-
reused.splice(reused.indexOf("build"));
|
|
3152
|
-
journal.append("gate-provisioned", t.id, provisionedRow);
|
|
3153
|
-
provisionedRow = undefined;
|
|
3154
|
-
}
|
|
3155
|
-
}
|
|
3156
|
-
for (const gate of reused) {
|
|
3157
|
-
journal.append("gate-reused", t.id, { gate, commit: replayedGates.commit });
|
|
3158
|
-
}
|
|
3159
|
-
if (provisionedRow !== undefined)
|
|
3160
|
-
journal.append("gate-provisioned", t.id, provisionedRow);
|
|
3161
3175
|
remainingGates = declaredGates.slice(reused.length);
|
|
3162
3176
|
}
|
|
3177
|
+
// OBS-1094/1049: every replay recreates the checkout, but build outputs are not commits.
|
|
3178
|
+
// Provision whenever build will not run as a gate, including an operator waiver restore.
|
|
3179
|
+
// Defer reuse rows until provisioning succeeds: green keeps reuse-then-provision order;
|
|
3180
|
+
// red records only provisioning and puts build and every declared successor back in gates,
|
|
3181
|
+
// unless build itself was waived: provisioning must not undo the operator's release.
|
|
3182
|
+
let provisionedRow;
|
|
3183
|
+
if (!remainingGates.includes("build") && commands.build !== undefined) {
|
|
3184
|
+
await waitForBaseline(t.id);
|
|
3185
|
+
const startedAt = Date.now();
|
|
3186
|
+
const provisioned = await withCommandContext(t.id, () => sh(commands.build, wt));
|
|
3187
|
+
provisionedRow = {
|
|
3188
|
+
gate: "build", commit: !satisfiedGate && !recheck ? replayedGates.commit : currentTaskSubject,
|
|
3189
|
+
exitCode: provisioned.code, durationMs: Date.now() - startedAt,
|
|
3190
|
+
};
|
|
3191
|
+
if (provisioned.code !== 0 && satisfiedGate !== "build") {
|
|
3192
|
+
reused.length = 0;
|
|
3193
|
+
remainingGates = ["build", ...declaredGates.filter((gate) => gate !== "build")];
|
|
3194
|
+
}
|
|
3195
|
+
}
|
|
3196
|
+
for (const gate of reused) {
|
|
3197
|
+
journal.append("gate-reused", t.id, { gate, commit: replayedGates.commit });
|
|
3198
|
+
}
|
|
3199
|
+
if (provisionedRow !== undefined)
|
|
3200
|
+
journal.append("gate-provisioned", t.id, provisionedRow);
|
|
3163
3201
|
const resumedTask = { ...t, gates: remainingGates };
|
|
3202
|
+
const operatorContext = approvalReviewContext(journal.read(), t.id, true);
|
|
3164
3203
|
gateLoop: while (true) {
|
|
3165
3204
|
fatalStop.signal.throwIfAborted();
|
|
3166
3205
|
executionSignal()?.throwIfAborted();
|
|
@@ -3178,6 +3217,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3178
3217
|
journal.phaseStart(t.id, "gates");
|
|
3179
3218
|
const { results } = await withCommandContext(t.id, () => runReviewRecovery(resumedTask, {
|
|
3180
3219
|
carriedFindings: outstandingReviewFindings(journal.read(), t.id),
|
|
3220
|
+
operatorContext,
|
|
3181
3221
|
worktree: wt, baseRef: taskBase, result: priorResult, author: gateAuthor,
|
|
3182
3222
|
commands, baseline, channels: pools.review, judgeChannels: pools.judge, adapters, cfg, artifactDir: journal.dir,
|
|
3183
3223
|
collateral: collateral.get(t.id) ?? [],
|
|
@@ -3425,6 +3465,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3425
3465
|
// fresh re-dispatches. Read from the JOURNAL, before this dispatch's own event lands: a run that
|
|
3426
3466
|
// stopped between funding the repair and sending it resumes still carrying the findings.
|
|
3427
3467
|
const journaledSoFar = journal.read(); // read BEFORE this dispatch's own event lands
|
|
3468
|
+
const operatorContext = approvalReviewContext(journaledSoFar, t.id);
|
|
3428
3469
|
let repairFindings = pendingRepairFindings(journaledSoFar, t.id);
|
|
3429
3470
|
// OBS-254, one layer below the upheld brief: the ordinary gate-fail brief was loop-local, so any
|
|
3430
3471
|
// path that rebuilt this task's state (a resume, `--retry-failed`, a fresh daemon) dispatched a
|
|
@@ -3515,9 +3556,10 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3515
3556
|
// ledger cannot tell a carried dispatch from an amnesiac one — the exact question a run that
|
|
3516
3557
|
// spends two frontier attempts re-deriving a known defect has to be able to answer afterwards.
|
|
3517
3558
|
fatalStop.signal.throwIfAborted();
|
|
3559
|
+
const workerDispatchOrdinal = nextWorkerDispatchOrdinal++;
|
|
3518
3560
|
journal.append("task-dispatch", t.id, {
|
|
3519
3561
|
...(scopeApproval?.data.release === "scope-request" ? { files: t.files, graphDefinitionHash: graphDefinitionHash(graph) } : {}),
|
|
3520
|
-
assignment, attempt, provenance: dispatchProvenance([
|
|
3562
|
+
assignment, attempt, workerDispatchOrdinal, provenance: dispatchProvenance([
|
|
3521
3563
|
channelKey(assignment) === channelKey(r.assignment) ? r.provenance
|
|
3522
3564
|
: t.routingHints?.pin ? `pin ${t.routingHints.pin.via}:${t.routingHints.pin.model} not re-tried` : "ladder assignment",
|
|
3523
3565
|
`dispatch ${channelKey(assignment)}`, climbProvenance,
|
|
@@ -3624,7 +3666,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
3624
3666
|
// anything the live brief holds beyond the journaled findings (a consult's guidance) is kept:
|
|
3625
3667
|
// a repair adds the diff and the fix-only contract, it never subtracts what was already known.
|
|
3626
3668
|
feedback = feedback && !brief.includes(feedback) ? `${brief}\n\n${feedback}` : brief;
|
|
3627
|
-
journal.append("repair-dispatch", t.id, { diffBytes: Buffer.byteLength(diff, "utf8"), capped, ...(capped ? { droppedFiles } : {}) });
|
|
3669
|
+
journal.append("repair-dispatch", t.id, { workerDispatchOrdinal, diffBytes: Buffer.byteLength(diff, "utf8"), capped, ...(capped ? { droppedFiles } : {}) });
|
|
3628
3670
|
}
|
|
3629
3671
|
if (feedback || priorNamed.length > 0) {
|
|
3630
3672
|
feedback = augmentRetryBrief(feedback, { attempted: commitsToCarry, carried: carriedCommits, present: presentCommits });
|
|
@@ -5093,6 +5135,7 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
5093
5135
|
else {
|
|
5094
5136
|
({ results, commits } = await withCommandContext(t.id, () => runReviewRecovery(t, {
|
|
5095
5137
|
carriedFindings: outstandingFindings,
|
|
5138
|
+
operatorContext,
|
|
5096
5139
|
worktree: wt, baseRef: taskBase, result, author: assignment,
|
|
5097
5140
|
commands, baseline, channels: pools.review, judgeChannels: pools.judge, adapters, cfg, artifactDir: journal.dir,
|
|
5098
5141
|
collateral: collateral.get(t.id) ?? [],
|
package/dist/run/journal.d.ts
CHANGED
|
@@ -43,6 +43,15 @@ export interface StructuredFinding {
|
|
|
43
43
|
note: string;
|
|
44
44
|
rationale?: string;
|
|
45
45
|
fingerprint: string;
|
|
46
|
+
/** Closed list of spellings actually observed on this chain, including the current one. */
|
|
47
|
+
observedFingerprints?: string[];
|
|
48
|
+
/** Prior id validated against the reviewer's reraised list by the review gate. */
|
|
49
|
+
reraisedFrom?: string;
|
|
50
|
+
/** Resolved definition AND defect identity; bare symbol spelling is never lineage. */
|
|
51
|
+
codeIdentity?: {
|
|
52
|
+
definition: string;
|
|
53
|
+
defect: string;
|
|
54
|
+
};
|
|
46
55
|
}
|
|
47
56
|
declare const CONSULT_ACTIONS: readonly ["retry", "reroute", "decompose", "human"];
|
|
48
57
|
export type ConsultGuidanceAction = (typeof CONSULT_ACTIONS)[number];
|
|
@@ -187,6 +196,11 @@ export declare function pendingApprovalActions(events: JournalEvent[]): Map<stri
|
|
|
187
196
|
* because every approval erased exactly the finding the fresh attempt was funded to fix.
|
|
188
197
|
*/
|
|
189
198
|
export declare function journaledFailureBrief(events: JournalEvent[], taskId: string): string[];
|
|
199
|
+
/** Never infer a cross-path alias from a symbol, sentence, or hash. */
|
|
200
|
+
export declare function reviewFingerprintMatches(candidate: unknown, fingerprint: string): boolean;
|
|
201
|
+
export declare function observedReviewFingerprints(finding: StructuredFinding): string[];
|
|
202
|
+
/** Re-seat only an unambiguous, positively linked chain; retain the newest evidence path. */
|
|
203
|
+
export declare function carryReviewFindings(priors: readonly StructuredFinding[], rows: readonly StructuredFinding[]): StructuredFinding[];
|
|
190
204
|
/**
|
|
191
205
|
* T6: the review findings still OUTSTANDING on a task. A review finding is a property of the TASK,
|
|
192
206
|
* not of the attempt that drew it: it stays outstanding until a later review PASSES on the task (or
|
|
@@ -208,7 +222,8 @@ export declare function journaledFailureBrief(events: JournalEvent[], taskId: st
|
|
|
208
222
|
* other gate settled that gate, not this one. Reading "approved" as "settled" is how a still-open
|
|
209
223
|
* finding was dropped at the exact moment the operator paid for another attempt to fix it. A review
|
|
210
224
|
* that DECLINED (`skipped`) is not a verdict and neither adds nor retires — fail closed. Findings are
|
|
211
|
-
*
|
|
225
|
+
* linked by validated lineage across paths; observed spellings remain a closed list. Repeating the
|
|
226
|
+
* same row carries it once, while distinct defects sharing a symbol retain separate fingerprints.
|
|
212
227
|
*
|
|
213
228
|
* v2.1.5 T2: a passing review settles the findings it BLOCKED on. It does not settle the ones it
|
|
214
229
|
* DEFERRED — those it saw, declined to block on, and recorded a rationale for, and nothing has fixed
|
|
@@ -312,8 +327,8 @@ export declare const TelemetryRowSchema: z.ZodObject<{
|
|
|
312
327
|
}>>;
|
|
313
328
|
kind: z.ZodOptional<z.ZodLiteral<"judge">>;
|
|
314
329
|
judgeOutcome: z.ZodOptional<z.ZodEnum<{
|
|
315
|
-
parseable: "parseable";
|
|
316
330
|
unparseable: "unparseable";
|
|
331
|
+
parseable: "parseable";
|
|
317
332
|
}>>;
|
|
318
333
|
}, z.core.$strip>;
|
|
319
334
|
export type TelemetryRow = z.infer<typeof TelemetryRowSchema>;
|
package/dist/run/journal.js
CHANGED
|
@@ -730,6 +730,83 @@ export function journaledFailureBrief(events, taskId) {
|
|
|
730
730
|
}
|
|
731
731
|
return rows;
|
|
732
732
|
}
|
|
733
|
+
/** Never infer a cross-path alias from a symbol, sentence, or hash. */
|
|
734
|
+
export function reviewFingerprintMatches(candidate, fingerprint) {
|
|
735
|
+
return typeof candidate === "string"
|
|
736
|
+
&& candidate.replace(/^\s*fingerprint\s*:\s*/i, "").replace(/\s+/g, "") === fingerprint.replace(/\s+/g, "");
|
|
737
|
+
}
|
|
738
|
+
export function observedReviewFingerprints(finding) {
|
|
739
|
+
const observed = Array.isArray(finding.observedFingerprints)
|
|
740
|
+
? finding.observedFingerprints.filter((id) => typeof id === "string") : [];
|
|
741
|
+
return [...new Set([finding.fingerprint, ...observed])];
|
|
742
|
+
}
|
|
743
|
+
const reviewNoteIdentity = (note) => note.replace(LINE_REF_RE, "").replace(/\s+/g, " ").trim();
|
|
744
|
+
function distinctReviewFinding(row) {
|
|
745
|
+
const identity = JSON.stringify([reviewNoteIdentity(row.note), row.codeIdentity ?? null]);
|
|
746
|
+
const symbol = `${row.symbol}#${createHash("sha256").update(identity).digest("hex").slice(0, 12)}`;
|
|
747
|
+
return { ...row, symbol, fingerprint: `${row.class}|${row.path}|${symbol}` };
|
|
748
|
+
}
|
|
749
|
+
/** Re-seat only an unambiguous, positively linked chain; retain the newest evidence path. */
|
|
750
|
+
export function carryReviewFindings(priors, rows) {
|
|
751
|
+
const open = [...priors];
|
|
752
|
+
for (const row of rows) {
|
|
753
|
+
const identity = row.codeIdentity;
|
|
754
|
+
const matches = open.filter((prior) => {
|
|
755
|
+
if (prior.class !== row.class)
|
|
756
|
+
return false;
|
|
757
|
+
if (row.reraisedFrom)
|
|
758
|
+
return rows.filter((other) => other.reraisedFrom === row.reraisedFrom).length === 1
|
|
759
|
+
&& observedReviewFingerprints(prior).includes(row.reraisedFrom);
|
|
760
|
+
if (identity?.definition && identity.defect && prior.codeIdentity) {
|
|
761
|
+
return prior.codeIdentity.definition === identity.definition && prior.codeIdentity.defect === identity.defect;
|
|
762
|
+
}
|
|
763
|
+
return prior.fingerprint === row.fingerprint && reviewNoteIdentity(prior.note) === reviewNoteIdentity(row.note);
|
|
764
|
+
});
|
|
765
|
+
if (matches.length === 1) {
|
|
766
|
+
const prior = matches[0];
|
|
767
|
+
// Positive lineage does not grant this chain another open defect's display spelling.
|
|
768
|
+
let newest = row;
|
|
769
|
+
while (open.some((other) => other !== prior && observedReviewFingerprints(other).includes(newest.fingerprint))) {
|
|
770
|
+
newest = distinctReviewFinding(newest);
|
|
771
|
+
}
|
|
772
|
+
const observed = [...new Set([...observedReviewFingerprints(prior), ...observedReviewFingerprints(newest)])];
|
|
773
|
+
open[open.indexOf(prior)] = {
|
|
774
|
+
...newest,
|
|
775
|
+
...(prior.codeIdentity && !newest.codeIdentity ? { codeIdentity: prior.codeIdentity } : {}),
|
|
776
|
+
...(observed.length > 1 ? { observedFingerprints: observed } : {}),
|
|
777
|
+
};
|
|
778
|
+
}
|
|
779
|
+
else {
|
|
780
|
+
// Two defects may name the same definition. Preserve both, with a copyable full triple.
|
|
781
|
+
// Deferrals intentionally replace their rationale without changing their note.
|
|
782
|
+
const collision = open.some((prior) => observedReviewFingerprints(prior).includes(row.fingerprint));
|
|
783
|
+
if (collision) {
|
|
784
|
+
let distinct = distinctReviewFinding(row);
|
|
785
|
+
// A generated spelling may itself belong to a chain that has since moved away.
|
|
786
|
+
// Reuse only a matching row; otherwise keep minting until the spelling is unowned.
|
|
787
|
+
while (true) {
|
|
788
|
+
const owners = open.filter((prior) => observedReviewFingerprints(prior).includes(distinct.fingerprint));
|
|
789
|
+
if (owners.length === 0) {
|
|
790
|
+
open.push(distinct);
|
|
791
|
+
break;
|
|
792
|
+
}
|
|
793
|
+
if (owners.length === 1 && reviewNoteIdentity(owners[0].note) === reviewNoteIdentity(row.note)) {
|
|
794
|
+
const prior = owners[0];
|
|
795
|
+
open[open.indexOf(prior)] = {
|
|
796
|
+
...prior, ...distinct,
|
|
797
|
+
observedFingerprints: [...new Set([...observedReviewFingerprints(prior), ...observedReviewFingerprints(distinct)])],
|
|
798
|
+
};
|
|
799
|
+
break;
|
|
800
|
+
}
|
|
801
|
+
distinct = distinctReviewFinding(distinct);
|
|
802
|
+
}
|
|
803
|
+
}
|
|
804
|
+
else
|
|
805
|
+
open.push(row);
|
|
806
|
+
}
|
|
807
|
+
}
|
|
808
|
+
return open;
|
|
809
|
+
}
|
|
733
810
|
/**
|
|
734
811
|
* T6: the review findings still OUTSTANDING on a task. A review finding is a property of the TASK,
|
|
735
812
|
* not of the attempt that drew it: it stays outstanding until a later review PASSES on the task (or
|
|
@@ -751,7 +828,8 @@ export function journaledFailureBrief(events, taskId) {
|
|
|
751
828
|
* other gate settled that gate, not this one. Reading "approved" as "settled" is how a still-open
|
|
752
829
|
* finding was dropped at the exact moment the operator paid for another attempt to fix it. A review
|
|
753
830
|
* that DECLINED (`skipped`) is not a verdict and neither adds nor retires — fail closed. Findings are
|
|
754
|
-
*
|
|
831
|
+
* linked by validated lineage across paths; observed spellings remain a closed list. Repeating the
|
|
832
|
+
* same row carries it once, while distinct defects sharing a symbol retain separate fingerprints.
|
|
755
833
|
*
|
|
756
834
|
* v2.1.5 T2: a passing review settles the findings it BLOCKED on. It does not settle the ones it
|
|
757
835
|
* DEFERRED — those it saw, declined to block on, and recorded a rationale for, and nothing has fixed
|
|
@@ -767,34 +845,44 @@ export function journaledFailureBrief(events, taskId) {
|
|
|
767
845
|
* row. N rounds of the same concern therefore carry the newest accepted explanation once, not N rows.
|
|
768
846
|
*/
|
|
769
847
|
export function outstandingReviewFindings(events, taskId) {
|
|
770
|
-
|
|
848
|
+
let open = [];
|
|
771
849
|
for (const e of events) {
|
|
772
850
|
if (e.taskId !== taskId)
|
|
773
851
|
continue;
|
|
774
852
|
if (e.event === "task-approved") {
|
|
775
853
|
if (e.data.release === GATE_SATISFIED_RELEASE && e.data.gate === "review")
|
|
776
|
-
open
|
|
854
|
+
open = [];
|
|
777
855
|
continue;
|
|
778
856
|
}
|
|
779
857
|
if (e.event !== "gate-result" || e.data.gate !== "review" || e.data.skipped === true)
|
|
780
858
|
continue;
|
|
859
|
+
// Rejected or missing verdicts carry diagnostics, not accepted findings or closures.
|
|
860
|
+
if (e.data.unparseable === true || e.data.noVerdict === true || e.data.cause !== undefined)
|
|
861
|
+
continue;
|
|
862
|
+
// A failed review can resolve one chain while re-raising another. Only observed, uniquely
|
|
863
|
+
// matched spellings retire a chain; skipped/no-verdict rows were excluded above.
|
|
864
|
+
if (Array.isArray(e.data.resolved)) {
|
|
865
|
+
const resolved = e.data.resolved;
|
|
866
|
+
const settled = new Set(resolved.flatMap((id) => {
|
|
867
|
+
const matches = open.filter((finding) => finding.class === "review:material"
|
|
868
|
+
&& observedReviewFingerprints(finding).some((fp) => reviewFingerprintMatches(id, fp)));
|
|
869
|
+
return matches.length === 1 ? matches : [];
|
|
870
|
+
}));
|
|
871
|
+
open = open.filter((finding) => !settled.has(finding));
|
|
872
|
+
}
|
|
781
873
|
if (e.data.pass !== false) {
|
|
782
874
|
// a later review PASSED on this task: every finding it BLOCKED on is settled …
|
|
783
|
-
|
|
784
|
-
if (!isDeferredFinding(finding))
|
|
785
|
-
open.delete(key);
|
|
875
|
+
open = open.filter(isDeferredFinding);
|
|
786
876
|
// … and no deferral is, whether or not this pass restated it. A pass is silent about a
|
|
787
877
|
// deferral it does not mention: the concern is unfixed either way, and the reviewer that
|
|
788
878
|
// waved it through is not the release that accepts it. Retiring on omission would drop it on
|
|
789
879
|
// the very next round — the same silent drop by a different door.
|
|
790
|
-
|
|
791
|
-
open.set(finding.fingerprint, finding);
|
|
880
|
+
open = carryReviewFindings(open, findingRows(e, "review").filter(isDeferredFinding));
|
|
792
881
|
}
|
|
793
882
|
else
|
|
794
|
-
|
|
795
|
-
open.set(finding.fingerprint, finding);
|
|
883
|
+
open = carryReviewFindings(open, findingRows(e, "review"));
|
|
796
884
|
}
|
|
797
|
-
return
|
|
885
|
+
return open;
|
|
798
886
|
}
|
|
799
887
|
/** The findings a funded repair must carry into the next dispatch, or undefined if none is pending. */
|
|
800
888
|
export function pendingRepairFindings(events, taskId) {
|
package/dist/run/merge.d.ts
CHANGED
|
@@ -1,6 +1,15 @@
|
|
|
1
1
|
import type { TickmarkrConfig } from "../config/config.js";
|
|
2
|
-
import { type Baseline, type FailureClassification } from "../gates/baseline.js";
|
|
2
|
+
import { type Baseline, type GateEvidenceOptions, type FailureClassification } from "../gates/baseline.js";
|
|
3
|
+
import type { GateEvidenceReceipt } from "./protocol.js";
|
|
3
4
|
export interface TipVerifyResult {
|
|
5
|
+
evidenceReceipt?: GateEvidenceReceipt;
|
|
6
|
+
evidenceReceipts?: GateEvidenceReceipt[];
|
|
7
|
+
evidenceAbsence?: "provenance-refused" | "not-started" | "historical-cache";
|
|
8
|
+
/** Absolute directory against which the original receipt artifact paths resolve. */
|
|
9
|
+
originRunRoot?: string;
|
|
10
|
+
nonce?: string;
|
|
11
|
+
stdoutPath?: string;
|
|
12
|
+
stderrPath?: string;
|
|
4
13
|
gate: string;
|
|
5
14
|
cmd: string;
|
|
6
15
|
pass: boolean;
|
|
@@ -20,6 +29,8 @@ export interface TipVerifyResult {
|
|
|
20
29
|
*/
|
|
21
30
|
cause?: FailureClassification;
|
|
22
31
|
}
|
|
32
|
+
/** Carry execution evidence without minting an invocation or rebasing its artifact paths. */
|
|
33
|
+
export declare function reusedTipEvidence(source: Partial<TipVerifyResult>): Partial<TipVerifyResult>;
|
|
23
34
|
export declare function integrationBranch(cfg: TickmarkrConfig, runId: string): string;
|
|
24
35
|
export declare function ensureIntegration(repo: string, branch: string, baseRef: string): Promise<string>;
|
|
25
36
|
export declare function integrationHead(intWt: string): Promise<string>;
|
|
@@ -31,4 +42,4 @@ export declare function mergeTask(intWt: string, taskBranch: string, message: st
|
|
|
31
42
|
branchTip: string;
|
|
32
43
|
};
|
|
33
44
|
}>;
|
|
34
|
-
export declare function verifyIntegrationTip(intWt: string, commands: Record<string, string>, runDir: string, baseline?: Baseline): Promise<TipVerifyResult[]>;
|
|
45
|
+
export declare function verifyIntegrationTip(intWt: string, commands: Record<string, string>, runDir: string, baseline?: Baseline, evidenceOptions?: GateEvidenceOptions): Promise<TipVerifyResult[]>;
|