tickmarkr 2.5.8 → 2.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/claude-code.js +19 -6
- package/dist/adapters/prompt.js +1 -1
- package/dist/adapters/qwen.js +30 -3
- package/dist/adapters/types.d.ts +3 -0
- package/dist/cli/commands/approve.js +15 -3
- package/dist/cli/commands/beat.d.ts +2 -0
- package/dist/cli/commands/beat.js +28 -29
- package/dist/cli/commands/compile.js +11 -0
- package/dist/cli/commands/fleet.js +61 -53
- package/dist/cli/commands/init.js +1 -1
- package/dist/cli/commands/report.js +11 -1
- package/dist/cli/commands/verify.d.ts +4 -1
- package/dist/cli/commands/verify.js +18 -5
- package/dist/cli/help.d.ts +2 -0
- package/dist/cli/help.js +6 -4
- package/dist/compile/native.js +39 -4
- package/dist/config/config.d.ts +41 -3
- package/dist/config/config.js +48 -17
- package/dist/config/fleet-overlay.js +47 -62
- package/dist/drivers/orca.d.ts +1 -1
- package/dist/drivers/orca.js +21 -4
- package/dist/gates/baseline.d.ts +67 -0
- package/dist/gates/baseline.js +172 -10
- package/dist/gates/cache.d.ts +8 -0
- package/dist/gates/cache.js +26 -13
- package/dist/gates/review.d.ts +8 -7
- package/dist/gates/review.js +83 -45
- package/dist/gates/run-gates.d.ts +4 -1
- package/dist/gates/run-gates.js +32 -12
- package/dist/gates/test-manifest.d.ts +16 -1
- package/dist/gates/test-manifest.js +127 -27
- package/dist/gates/test-reporter.js +4 -0
- package/dist/graph/graph.d.ts +1 -1
- package/dist/graph/graph.js +6 -1
- package/dist/graph/schema.d.ts +2 -0
- package/dist/graph/schema.js +2 -0
- package/dist/route/preference.d.ts +1 -1
- package/dist/route/preference.js +10 -39
- package/dist/run/daemon.d.ts +7 -1
- package/dist/run/daemon.js +237 -53
- package/dist/run/git.d.ts +10 -0
- package/dist/run/git.js +49 -1
- package/dist/run/journal.d.ts +18 -3
- package/dist/run/journal.js +141 -17
- package/dist/run/merge.d.ts +13 -2
- package/dist/run/merge.js +98 -12
- package/dist/run/protocol.d.ts +82 -0
- package/dist/run/protocol.js +35 -0
- package/dist/run/receipt-resolver.d.ts +18 -0
- package/dist/run/receipt-resolver.js +132 -0
- package/dist/run/supervision.d.ts +14 -1
- package/dist/run/supervision.js +122 -24
- package/dist/tui/cockpit/evidence-view.d.ts +10 -1
- package/dist/tui/cockpit/evidence-view.js +37 -5
- package/dist/tui/cockpit/live-runtime.d.ts +4 -0
- package/dist/tui/cockpit/live-runtime.js +37 -3
- package/dist/tui/cockpit/live-store.d.ts +2 -0
- package/dist/tui/ink/fleet-app.d.ts +12 -22
- package/dist/tui/ink/fleet-app.js +520 -131
- package/package.json +1 -1
- package/schema/rungraph.schema.json +7 -0
- package/skills/tickmarkr-overseer/SKILL.md +91 -41
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +14 -2
package/dist/run/journal.d.ts
CHANGED
|
@@ -43,6 +43,15 @@ export interface StructuredFinding {
|
|
|
43
43
|
note: string;
|
|
44
44
|
rationale?: string;
|
|
45
45
|
fingerprint: string;
|
|
46
|
+
/** Closed list of spellings actually observed on this chain, including the current one. */
|
|
47
|
+
observedFingerprints?: string[];
|
|
48
|
+
/** Prior id validated against the reviewer's reraised list by the review gate. */
|
|
49
|
+
reraisedFrom?: string;
|
|
50
|
+
/** Resolved definition AND defect identity; bare symbol spelling is never lineage. */
|
|
51
|
+
codeIdentity?: {
|
|
52
|
+
definition: string;
|
|
53
|
+
defect: string;
|
|
54
|
+
};
|
|
46
55
|
}
|
|
47
56
|
declare const CONSULT_ACTIONS: readonly ["retry", "reroute", "decompose", "human"];
|
|
48
57
|
export type ConsultGuidanceAction = (typeof CONSULT_ACTIONS)[number];
|
|
@@ -187,6 +196,11 @@ export declare function pendingApprovalActions(events: JournalEvent[]): Map<stri
|
|
|
187
196
|
* because every approval erased exactly the finding the fresh attempt was funded to fix.
|
|
188
197
|
*/
|
|
189
198
|
export declare function journaledFailureBrief(events: JournalEvent[], taskId: string): string[];
|
|
199
|
+
/** Never infer a cross-path alias from a symbol, sentence, or hash. */
|
|
200
|
+
export declare function reviewFingerprintMatches(candidate: unknown, fingerprint: string): boolean;
|
|
201
|
+
export declare function observedReviewFingerprints(finding: StructuredFinding): string[];
|
|
202
|
+
/** Re-seat only an unambiguous, positively linked chain; retain the newest evidence path. */
|
|
203
|
+
export declare function carryReviewFindings(priors: readonly StructuredFinding[], rows: readonly StructuredFinding[]): StructuredFinding[];
|
|
190
204
|
/**
|
|
191
205
|
* T6: the review findings still OUTSTANDING on a task. A review finding is a property of the TASK,
|
|
192
206
|
* not of the attempt that drew it: it stays outstanding until a later review PASSES on the task (or
|
|
@@ -208,7 +222,8 @@ export declare function journaledFailureBrief(events: JournalEvent[], taskId: st
|
|
|
208
222
|
* other gate settled that gate, not this one. Reading "approved" as "settled" is how a still-open
|
|
209
223
|
* finding was dropped at the exact moment the operator paid for another attempt to fix it. A review
|
|
210
224
|
* that DECLINED (`skipped`) is not a verdict and neither adds nor retires — fail closed. Findings are
|
|
211
|
-
*
|
|
225
|
+
* linked by validated lineage across paths; observed spellings remain a closed list. Repeating the
|
|
226
|
+
* same row carries it once, while distinct defects sharing a symbol retain separate fingerprints.
|
|
212
227
|
*
|
|
213
228
|
* v2.1.5 T2: a passing review settles the findings it BLOCKED on. It does not settle the ones it
|
|
214
229
|
* DEFERRED — those it saw, declined to block on, and recorded a rationale for, and nothing has fixed
|
|
@@ -312,8 +327,8 @@ export declare const TelemetryRowSchema: z.ZodObject<{
|
|
|
312
327
|
}>>;
|
|
313
328
|
kind: z.ZodOptional<z.ZodLiteral<"judge">>;
|
|
314
329
|
judgeOutcome: z.ZodOptional<z.ZodEnum<{
|
|
315
|
-
parseable: "parseable";
|
|
316
330
|
unparseable: "unparseable";
|
|
331
|
+
parseable: "parseable";
|
|
317
332
|
}>>;
|
|
318
333
|
}, z.core.$strip>;
|
|
319
334
|
export type TelemetryRow = z.infer<typeof TelemetryRowSchema>;
|
|
@@ -422,7 +437,7 @@ export declare class Journal {
|
|
|
422
437
|
assignment?: Assignment;
|
|
423
438
|
} | null;
|
|
424
439
|
replayResumeState(): Map<string, ResumeState>;
|
|
425
|
-
replaySatisfiedGates(): Map<string, GateName>;
|
|
440
|
+
replaySatisfiedGates(currentSubjects?: ReadonlyMap<string, string>): Map<string, GateName>;
|
|
426
441
|
replayCurrentAttemptGateResults(): Map<string, CurrentAttemptGateReplay>;
|
|
427
442
|
replayExcludedChannels(): Set<string>;
|
|
428
443
|
telemetry(row: TelemetryRow): void;
|
package/dist/run/journal.js
CHANGED
|
@@ -730,6 +730,83 @@ export function journaledFailureBrief(events, taskId) {
|
|
|
730
730
|
}
|
|
731
731
|
return rows;
|
|
732
732
|
}
|
|
733
|
+
/** Never infer a cross-path alias from a symbol, sentence, or hash. */
|
|
734
|
+
export function reviewFingerprintMatches(candidate, fingerprint) {
|
|
735
|
+
return typeof candidate === "string"
|
|
736
|
+
&& candidate.replace(/^\s*fingerprint\s*:\s*/i, "").replace(/\s+/g, "") === fingerprint.replace(/\s+/g, "");
|
|
737
|
+
}
|
|
738
|
+
export function observedReviewFingerprints(finding) {
|
|
739
|
+
const observed = Array.isArray(finding.observedFingerprints)
|
|
740
|
+
? finding.observedFingerprints.filter((id) => typeof id === "string") : [];
|
|
741
|
+
return [...new Set([finding.fingerprint, ...observed])];
|
|
742
|
+
}
|
|
743
|
+
const reviewNoteIdentity = (note) => note.replace(LINE_REF_RE, "").replace(/\s+/g, " ").trim();
|
|
744
|
+
function distinctReviewFinding(row) {
|
|
745
|
+
const identity = JSON.stringify([reviewNoteIdentity(row.note), row.codeIdentity ?? null]);
|
|
746
|
+
const symbol = `${row.symbol}#${createHash("sha256").update(identity).digest("hex").slice(0, 12)}`;
|
|
747
|
+
return { ...row, symbol, fingerprint: `${row.class}|${row.path}|${symbol}` };
|
|
748
|
+
}
|
|
749
|
+
/** Re-seat only an unambiguous, positively linked chain; retain the newest evidence path. */
|
|
750
|
+
export function carryReviewFindings(priors, rows) {
|
|
751
|
+
const open = [...priors];
|
|
752
|
+
for (const row of rows) {
|
|
753
|
+
const identity = row.codeIdentity;
|
|
754
|
+
const matches = open.filter((prior) => {
|
|
755
|
+
if (prior.class !== row.class)
|
|
756
|
+
return false;
|
|
757
|
+
if (row.reraisedFrom)
|
|
758
|
+
return rows.filter((other) => other.reraisedFrom === row.reraisedFrom).length === 1
|
|
759
|
+
&& observedReviewFingerprints(prior).includes(row.reraisedFrom);
|
|
760
|
+
if (identity?.definition && identity.defect && prior.codeIdentity) {
|
|
761
|
+
return prior.codeIdentity.definition === identity.definition && prior.codeIdentity.defect === identity.defect;
|
|
762
|
+
}
|
|
763
|
+
return prior.fingerprint === row.fingerprint && reviewNoteIdentity(prior.note) === reviewNoteIdentity(row.note);
|
|
764
|
+
});
|
|
765
|
+
if (matches.length === 1) {
|
|
766
|
+
const prior = matches[0];
|
|
767
|
+
// Positive lineage does not grant this chain another open defect's display spelling.
|
|
768
|
+
let newest = row;
|
|
769
|
+
while (open.some((other) => other !== prior && observedReviewFingerprints(other).includes(newest.fingerprint))) {
|
|
770
|
+
newest = distinctReviewFinding(newest);
|
|
771
|
+
}
|
|
772
|
+
const observed = [...new Set([...observedReviewFingerprints(prior), ...observedReviewFingerprints(newest)])];
|
|
773
|
+
open[open.indexOf(prior)] = {
|
|
774
|
+
...newest,
|
|
775
|
+
...(prior.codeIdentity && !newest.codeIdentity ? { codeIdentity: prior.codeIdentity } : {}),
|
|
776
|
+
...(observed.length > 1 ? { observedFingerprints: observed } : {}),
|
|
777
|
+
};
|
|
778
|
+
}
|
|
779
|
+
else {
|
|
780
|
+
// Two defects may name the same definition. Preserve both, with a copyable full triple.
|
|
781
|
+
// Deferrals intentionally replace their rationale without changing their note.
|
|
782
|
+
const collision = open.some((prior) => observedReviewFingerprints(prior).includes(row.fingerprint));
|
|
783
|
+
if (collision) {
|
|
784
|
+
let distinct = distinctReviewFinding(row);
|
|
785
|
+
// A generated spelling may itself belong to a chain that has since moved away.
|
|
786
|
+
// Reuse only a matching row; otherwise keep minting until the spelling is unowned.
|
|
787
|
+
while (true) {
|
|
788
|
+
const owners = open.filter((prior) => observedReviewFingerprints(prior).includes(distinct.fingerprint));
|
|
789
|
+
if (owners.length === 0) {
|
|
790
|
+
open.push(distinct);
|
|
791
|
+
break;
|
|
792
|
+
}
|
|
793
|
+
if (owners.length === 1 && reviewNoteIdentity(owners[0].note) === reviewNoteIdentity(row.note)) {
|
|
794
|
+
const prior = owners[0];
|
|
795
|
+
open[open.indexOf(prior)] = {
|
|
796
|
+
...prior, ...distinct,
|
|
797
|
+
observedFingerprints: [...new Set([...observedReviewFingerprints(prior), ...observedReviewFingerprints(distinct)])],
|
|
798
|
+
};
|
|
799
|
+
break;
|
|
800
|
+
}
|
|
801
|
+
distinct = distinctReviewFinding(distinct);
|
|
802
|
+
}
|
|
803
|
+
}
|
|
804
|
+
else
|
|
805
|
+
open.push(row);
|
|
806
|
+
}
|
|
807
|
+
}
|
|
808
|
+
return open;
|
|
809
|
+
}
|
|
733
810
|
/**
|
|
734
811
|
* T6: the review findings still OUTSTANDING on a task. A review finding is a property of the TASK,
|
|
735
812
|
* not of the attempt that drew it: it stays outstanding until a later review PASSES on the task (or
|
|
@@ -751,7 +828,8 @@ export function journaledFailureBrief(events, taskId) {
|
|
|
751
828
|
* other gate settled that gate, not this one. Reading "approved" as "settled" is how a still-open
|
|
752
829
|
* finding was dropped at the exact moment the operator paid for another attempt to fix it. A review
|
|
753
830
|
* that DECLINED (`skipped`) is not a verdict and neither adds nor retires — fail closed. Findings are
|
|
754
|
-
*
|
|
831
|
+
* linked by validated lineage across paths; observed spellings remain a closed list. Repeating the
|
|
832
|
+
* same row carries it once, while distinct defects sharing a symbol retain separate fingerprints.
|
|
755
833
|
*
|
|
756
834
|
* v2.1.5 T2: a passing review settles the findings it BLOCKED on. It does not settle the ones it
|
|
757
835
|
* DEFERRED — those it saw, declined to block on, and recorded a rationale for, and nothing has fixed
|
|
@@ -767,34 +845,44 @@ export function journaledFailureBrief(events, taskId) {
|
|
|
767
845
|
* row. N rounds of the same concern therefore carry the newest accepted explanation once, not N rows.
|
|
768
846
|
*/
|
|
769
847
|
export function outstandingReviewFindings(events, taskId) {
|
|
770
|
-
|
|
848
|
+
let open = [];
|
|
771
849
|
for (const e of events) {
|
|
772
850
|
if (e.taskId !== taskId)
|
|
773
851
|
continue;
|
|
774
852
|
if (e.event === "task-approved") {
|
|
775
853
|
if (e.data.release === GATE_SATISFIED_RELEASE && e.data.gate === "review")
|
|
776
|
-
open
|
|
854
|
+
open = [];
|
|
777
855
|
continue;
|
|
778
856
|
}
|
|
779
857
|
if (e.event !== "gate-result" || e.data.gate !== "review" || e.data.skipped === true)
|
|
780
858
|
continue;
|
|
859
|
+
// Rejected or missing verdicts carry diagnostics, not accepted findings or closures.
|
|
860
|
+
if (e.data.unparseable === true || e.data.noVerdict === true || e.data.cause !== undefined)
|
|
861
|
+
continue;
|
|
862
|
+
// A failed review can resolve one chain while re-raising another. Only observed, uniquely
|
|
863
|
+
// matched spellings retire a chain; skipped/no-verdict rows were excluded above.
|
|
864
|
+
if (Array.isArray(e.data.resolved)) {
|
|
865
|
+
const resolved = e.data.resolved;
|
|
866
|
+
const settled = new Set(resolved.flatMap((id) => {
|
|
867
|
+
const matches = open.filter((finding) => finding.class === "review:material"
|
|
868
|
+
&& observedReviewFingerprints(finding).some((fp) => reviewFingerprintMatches(id, fp)));
|
|
869
|
+
return matches.length === 1 ? matches : [];
|
|
870
|
+
}));
|
|
871
|
+
open = open.filter((finding) => !settled.has(finding));
|
|
872
|
+
}
|
|
781
873
|
if (e.data.pass !== false) {
|
|
782
874
|
// a later review PASSED on this task: every finding it BLOCKED on is settled …
|
|
783
|
-
|
|
784
|
-
if (!isDeferredFinding(finding))
|
|
785
|
-
open.delete(key);
|
|
875
|
+
open = open.filter(isDeferredFinding);
|
|
786
876
|
// … and no deferral is, whether or not this pass restated it. A pass is silent about a
|
|
787
877
|
// deferral it does not mention: the concern is unfixed either way, and the reviewer that
|
|
788
878
|
// waved it through is not the release that accepts it. Retiring on omission would drop it on
|
|
789
879
|
// the very next round — the same silent drop by a different door.
|
|
790
|
-
|
|
791
|
-
open.set(finding.fingerprint, finding);
|
|
880
|
+
open = carryReviewFindings(open, findingRows(e, "review").filter(isDeferredFinding));
|
|
792
881
|
}
|
|
793
882
|
else
|
|
794
|
-
|
|
795
|
-
open.set(finding.fingerprint, finding);
|
|
883
|
+
open = carryReviewFindings(open, findingRows(e, "review"));
|
|
796
884
|
}
|
|
797
|
-
return
|
|
885
|
+
return open;
|
|
798
886
|
}
|
|
799
887
|
/** The findings a funded repair must carry into the next dispatch, or undefined if none is pending. */
|
|
800
888
|
export function pendingRepairFindings(events, taskId) {
|
|
@@ -1646,26 +1734,62 @@ export class Journal {
|
|
|
1646
1734
|
}
|
|
1647
1735
|
return m;
|
|
1648
1736
|
}
|
|
1649
|
-
// OBS-130/OBS-571:
|
|
1650
|
-
//
|
|
1651
|
-
//
|
|
1652
|
-
// before
|
|
1653
|
-
|
|
1654
|
-
replaySatisfiedGates() {
|
|
1737
|
+
// OBS-130/OBS-571: only an explicit approval grants gate authority; recreation consumes
|
|
1738
|
+
// its immediate enactment. OBS-1133: retain a review waiver's subject separately so a later
|
|
1739
|
+
// recheck may carry it. Tool waivers never survive recheck. The daemon supplies the recreated
|
|
1740
|
+
// subject before using a carried waiver; absent that, replay can only compare journaled subjects.
|
|
1741
|
+
replaySatisfiedGates(currentSubjects) {
|
|
1655
1742
|
const satisfied = new Map();
|
|
1743
|
+
const reviewSubjects = new Map();
|
|
1744
|
+
const subjects = new Map();
|
|
1745
|
+
const reviewWaivers = new Map();
|
|
1656
1746
|
for (const e of this.read()) {
|
|
1657
1747
|
if (!e.taskId)
|
|
1658
1748
|
continue;
|
|
1749
|
+
if (e.event === "task-dispatch") {
|
|
1750
|
+
reviewSubjects.delete(e.taskId);
|
|
1751
|
+
subjects.delete(e.taskId);
|
|
1752
|
+
reviewWaivers.delete(e.taskId);
|
|
1753
|
+
satisfied.delete(e.taskId);
|
|
1754
|
+
}
|
|
1755
|
+
if (e.event === "gate-result") {
|
|
1756
|
+
const commit = typeof e.data.commit === "string" && e.data.commit ? e.data.commit : undefined;
|
|
1757
|
+
if (commit)
|
|
1758
|
+
subjects.set(e.taskId, commit);
|
|
1759
|
+
else
|
|
1760
|
+
subjects.delete(e.taskId);
|
|
1761
|
+
if (reviewWaivers.has(e.taskId) && reviewWaivers.get(e.taskId) !== commit) {
|
|
1762
|
+
reviewWaivers.delete(e.taskId);
|
|
1763
|
+
}
|
|
1764
|
+
if (e.data.gate === "review") {
|
|
1765
|
+
if (commit && e.data.pass === false)
|
|
1766
|
+
reviewSubjects.set(e.taskId, commit);
|
|
1767
|
+
else
|
|
1768
|
+
reviewSubjects.delete(e.taskId);
|
|
1769
|
+
}
|
|
1770
|
+
}
|
|
1659
1771
|
if (e.event === "worktree-recreation") {
|
|
1660
1772
|
satisfied.delete(e.taskId);
|
|
1661
1773
|
continue;
|
|
1662
1774
|
}
|
|
1663
1775
|
if (e.event === "task-approved") {
|
|
1664
1776
|
satisfied.delete(e.taskId);
|
|
1777
|
+
if (e.data.release === RECHECK_RELEASE) {
|
|
1778
|
+
const waivedSubject = reviewWaivers.get(e.taskId);
|
|
1779
|
+
const subject = currentSubjects?.get(e.taskId) ?? subjects.get(e.taskId);
|
|
1780
|
+
if (waivedSubject && waivedSubject === subject)
|
|
1781
|
+
satisfied.set(e.taskId, "review");
|
|
1782
|
+
continue;
|
|
1783
|
+
}
|
|
1784
|
+
reviewWaivers.delete(e.taskId);
|
|
1665
1785
|
if (e.data.release === GATE_SATISFIED_RELEASE
|
|
1666
1786
|
&& typeof e.data.gate === "string"
|
|
1667
1787
|
&& GATE_NAMES.includes(e.data.gate)) {
|
|
1668
1788
|
satisfied.set(e.taskId, e.data.gate);
|
|
1789
|
+
const subject = reviewSubjects.get(e.taskId);
|
|
1790
|
+
if (e.data.gate === "review" && subject && subject === subjects.get(e.taskId)) {
|
|
1791
|
+
reviewWaivers.set(e.taskId, subject);
|
|
1792
|
+
}
|
|
1669
1793
|
}
|
|
1670
1794
|
}
|
|
1671
1795
|
}
|
package/dist/run/merge.d.ts
CHANGED
|
@@ -1,6 +1,15 @@
|
|
|
1
1
|
import type { TickmarkrConfig } from "../config/config.js";
|
|
2
|
-
import { type Baseline, type FailureClassification } from "../gates/baseline.js";
|
|
2
|
+
import { type Baseline, type GateEvidenceOptions, type FailureClassification } from "../gates/baseline.js";
|
|
3
|
+
import type { GateEvidenceReceipt } from "./protocol.js";
|
|
3
4
|
export interface TipVerifyResult {
|
|
5
|
+
evidenceReceipt?: GateEvidenceReceipt;
|
|
6
|
+
evidenceReceipts?: GateEvidenceReceipt[];
|
|
7
|
+
evidenceAbsence?: "provenance-refused" | "not-started" | "historical-cache";
|
|
8
|
+
/** Directory the evidence setup returned; receipt artifact paths resolve against it. */
|
|
9
|
+
originRunRoot?: string;
|
|
10
|
+
nonce?: string;
|
|
11
|
+
stdoutPath?: string;
|
|
12
|
+
stderrPath?: string;
|
|
4
13
|
gate: string;
|
|
5
14
|
cmd: string;
|
|
6
15
|
pass: boolean;
|
|
@@ -20,6 +29,8 @@ export interface TipVerifyResult {
|
|
|
20
29
|
*/
|
|
21
30
|
cause?: FailureClassification;
|
|
22
31
|
}
|
|
32
|
+
/** Carry execution evidence without minting an invocation or rebasing its artifact paths. */
|
|
33
|
+
export declare function reusedTipEvidence(source: Partial<TipVerifyResult>): Partial<TipVerifyResult>;
|
|
23
34
|
export declare function integrationBranch(cfg: TickmarkrConfig, runId: string): string;
|
|
24
35
|
export declare function ensureIntegration(repo: string, branch: string, baseRef: string): Promise<string>;
|
|
25
36
|
export declare function integrationHead(intWt: string): Promise<string>;
|
|
@@ -31,4 +42,4 @@ export declare function mergeTask(intWt: string, taskBranch: string, message: st
|
|
|
31
42
|
branchTip: string;
|
|
32
43
|
};
|
|
33
44
|
}>;
|
|
34
|
-
export declare function verifyIntegrationTip(intWt: string, commands: Record<string, string>, runDir: string, baseline?: Baseline): Promise<TipVerifyResult[]>;
|
|
45
|
+
export declare function verifyIntegrationTip(intWt: string, commands: Record<string, string>, runDir: string, baseline?: Baseline, evidenceOptions?: GateEvidenceOptions): Promise<TipVerifyResult[]>;
|
package/dist/run/merge.js
CHANGED
|
@@ -1,11 +1,34 @@
|
|
|
1
1
|
import { existsSync, readFileSync, writeFileSync } from "node:fs";
|
|
2
2
|
import { join } from "node:path";
|
|
3
3
|
import { shq } from "../adapters/types.js";
|
|
4
|
-
import { ceilingKillResult, classifyFreshRunnerOutput, classifyRunnerOutput, fileCountDeficit, waitForCalmWindow, effectiveCeilingMs, fingerprint, freshFailures, } from "../gates/baseline.js";
|
|
4
|
+
import { beginGateEvidence, ceilingKillResult, classifyFreshRunnerOutput, classifyRunnerOutput, fileCountDeficit, waitForCalmWindow, effectiveCeilingMs, fingerprint, freshFailures, } from "../gates/baseline.js";
|
|
5
5
|
import { evaluateManifestedTest, isVitestTestCommand } from "../gates/test-manifest.js";
|
|
6
6
|
import { computeVerificationIdentity, formatReusedDetails, getVerdictStore, getWorktreeTree, resolveStateDir, isInfraResult, } from "../gates/cache.js";
|
|
7
7
|
import { tickmarkrDir } from "../graph/graph.js";
|
|
8
|
-
import { describeCapacity, gitHead, linkNodeModules, resolveIntegrationBranch, resolvedCapacity, sameCapacity, sh, shGit, shGitOk, WORKTREES_DIR } from "./git.js";
|
|
8
|
+
import { dependencyLinkRefusal, describeCapacity, gitHead, linkNodeModules, resolveIntegrationBranch, resolvedCapacity, sameCapacity, sh, shGit, shGitOk, WORKTREES_DIR } from "./git.js";
|
|
9
|
+
import { executionSignal } from "./execution-budget.js";
|
|
10
|
+
// Diagnostic convenience logs are observational, just like receipt persistence.
|
|
11
|
+
function writeTipLog(path, text) {
|
|
12
|
+
try {
|
|
13
|
+
writeFileSync(path, text);
|
|
14
|
+
}
|
|
15
|
+
catch { /* the command verdict remains authoritative */ }
|
|
16
|
+
}
|
|
17
|
+
/** Carry execution evidence without minting an invocation or rebasing its artifact paths. */
|
|
18
|
+
export function reusedTipEvidence(source) {
|
|
19
|
+
return {
|
|
20
|
+
reused: true,
|
|
21
|
+
...(source.evidenceReceipt ? { evidenceReceipt: source.evidenceReceipt } : {}),
|
|
22
|
+
...(source.evidenceReceipts ? { evidenceReceipts: source.evidenceReceipts } : {}),
|
|
23
|
+
...(source.originRunRoot ? { originRunRoot: source.originRunRoot } : {}),
|
|
24
|
+
...(!source.evidenceReceipt
|
|
25
|
+
? { evidenceAbsence: source.evidenceAbsence ?? "historical-cache" }
|
|
26
|
+
: source.evidenceAbsence ? { evidenceAbsence: source.evidenceAbsence } : {}),
|
|
27
|
+
...(source.nonce !== undefined ? { nonce: source.nonce } : {}),
|
|
28
|
+
...(source.stdoutPath !== undefined ? { stdoutPath: source.stdoutPath } : {}),
|
|
29
|
+
...(source.stderrPath !== undefined ? { stderrPath: source.stderrPath } : {}),
|
|
30
|
+
};
|
|
31
|
+
}
|
|
9
32
|
export function integrationBranch(cfg, runId) {
|
|
10
33
|
return `${cfg.integrationBranchPrefix}${runId}`;
|
|
11
34
|
}
|
|
@@ -52,7 +75,7 @@ export async function mergeTask(intWt, taskBranch, message, gatedCommit) {
|
|
|
52
75
|
// terminus unreachable by construction and misattributed the red to the last merged task.
|
|
53
76
|
// Fail-closed edges kept: no baseline → strict; baseline green for that gate → strict; any fresh
|
|
54
77
|
// fingerprint, or output with no recognizable failure shape, → failed.
|
|
55
|
-
export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
78
|
+
export async function verifyIntegrationTip(intWt, commands, runDir, baseline, evidenceOptions = {}) {
|
|
56
79
|
const results = [];
|
|
57
80
|
let runStartCommands;
|
|
58
81
|
let journalFound = false;
|
|
@@ -105,11 +128,21 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
105
128
|
if (!commands.test && commands.tipTest) {
|
|
106
129
|
gatesToRun.push(["test", commands.tipTest]);
|
|
107
130
|
}
|
|
131
|
+
const dependencyRefusal = dependencyLinkRefusal(intWt);
|
|
132
|
+
if (dependencyRefusal) {
|
|
133
|
+
for (const [gate, cmd] of gatesToRun.length ? gatesToRun : [["build", ""]]) {
|
|
134
|
+
const artifact = join(runDir, `tip-verify-${gate}.log`);
|
|
135
|
+
writeTipLog(artifact, dependencyRefusal + "\n");
|
|
136
|
+
results.push({ gate, cmd, pass: false, exitCode: 1, fingerprints: [],
|
|
137
|
+
cause: "infra", details: dependencyRefusal, evidenceAbsence: "not-started", artifact });
|
|
138
|
+
}
|
|
139
|
+
return results;
|
|
140
|
+
}
|
|
108
141
|
if (journalFound && hasRunEvidenceOrMalformed && runStartCommands === undefined) {
|
|
109
142
|
for (const [gate, cmd] of gatesToRun) {
|
|
110
143
|
const artifact = join(runDir, `tip-verify-${gate}.log`);
|
|
111
144
|
const details = `tip verify could not establish command provenance: run-start evidence missing or malformed in journal`;
|
|
112
|
-
|
|
145
|
+
writeTipLog(artifact, details + "\n");
|
|
113
146
|
results.push({
|
|
114
147
|
gate,
|
|
115
148
|
cmd,
|
|
@@ -117,6 +150,7 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
117
150
|
exitCode: 1,
|
|
118
151
|
fingerprints: [],
|
|
119
152
|
details,
|
|
153
|
+
evidenceAbsence: "provenance-refused",
|
|
120
154
|
artifact,
|
|
121
155
|
});
|
|
122
156
|
}
|
|
@@ -127,6 +161,8 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
127
161
|
// Publish only after the entire verification completes. The daemon kills a cancelled
|
|
128
162
|
// verifier process, so completed early gates must remain in memory until then.
|
|
129
163
|
const pending = new Map();
|
|
164
|
+
// D-267 (1): each fresh receipt's origin is the root its own beginGateEvidence returned.
|
|
165
|
+
const evidenceRoots = new Map();
|
|
130
166
|
for (const [gate, cmd] of gatesToRun) {
|
|
131
167
|
if (runStartCommands !== undefined) {
|
|
132
168
|
const expectedCmd = gate === "test"
|
|
@@ -135,7 +171,7 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
135
171
|
if (expectedCmd === undefined || cmd !== expectedCmd) {
|
|
136
172
|
const artifact = join(runDir, `tip-verify-${gate}.log`);
|
|
137
173
|
const details = `tip verify command "${cmd}" for gate "${gate}" was not named in run-start row`;
|
|
138
|
-
|
|
174
|
+
writeTipLog(artifact, details + "\n");
|
|
139
175
|
results.push({
|
|
140
176
|
gate,
|
|
141
177
|
cmd,
|
|
@@ -143,6 +179,7 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
143
179
|
exitCode: 1,
|
|
144
180
|
fingerprints: [],
|
|
145
181
|
details,
|
|
182
|
+
evidenceAbsence: "provenance-refused",
|
|
146
183
|
artifact,
|
|
147
184
|
});
|
|
148
185
|
continue;
|
|
@@ -184,7 +221,7 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
184
221
|
gate,
|
|
185
222
|
cmd,
|
|
186
223
|
pass: hit.pass,
|
|
187
|
-
|
|
224
|
+
...reusedTipEvidence(hit.meta ?? {}),
|
|
188
225
|
exitCode: hit.exitCode ?? 0,
|
|
189
226
|
fingerprints: hit.meta?.fingerprints ?? [],
|
|
190
227
|
...(hit.meta?.forgiven ? { forgiven: true } : {}),
|
|
@@ -201,20 +238,34 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
201
238
|
// a detected vitest test command never reaches the stdout-count/file-count path below (this
|
|
202
239
|
// gate's own manifest is always the FULL suite; verify has no selected screen to hold). A
|
|
203
240
|
// scripted test command that is not the detected runner falls through unchanged.
|
|
241
|
+
const evidenceSetup = { artifactDir: runDir, runId: runDir, ...evidenceOptions };
|
|
204
242
|
if (gate === "test" && isVitestTestCommand(cmd, intWt)) {
|
|
243
|
+
// D-267 (1): stamp the root the evidence setup returned, then give that exact root to every
|
|
244
|
+
// manifested invocation. Do not reconstruct it from the caller's options: the receipt's
|
|
245
|
+
// artifact paths are relative to the root returned by beginGateEvidence.
|
|
246
|
+
const evidence = beginGateEvidence(intWt, gate, cmd, evidenceSetup);
|
|
247
|
+
const manifestedEvidence = { ...evidenceSetup, artifactDir: evidence.root };
|
|
248
|
+
evidenceRoots.set(gate, evidence.root);
|
|
205
249
|
const outcome = await evaluateManifestedTest(cmd, intWt, {
|
|
206
250
|
baselineDurations: entry?.fileDurations,
|
|
207
251
|
longestFile: entry?.longestFile,
|
|
208
252
|
overallCeilingMs: effectiveCeilingMs(entry),
|
|
209
253
|
artifactDir: runDir,
|
|
254
|
+
evidence: manifestedEvidence,
|
|
210
255
|
});
|
|
211
256
|
const artifact = join(runDir, `tip-verify-${gate}.log`);
|
|
212
257
|
if (!outcome.pass)
|
|
213
|
-
|
|
258
|
+
writeTipLog(artifact, outcome.details);
|
|
214
259
|
results.push({
|
|
215
260
|
gate,
|
|
216
261
|
cmd,
|
|
217
262
|
pass: outcome.pass,
|
|
263
|
+
evidenceReceipt: outcome.evidenceReceipt,
|
|
264
|
+
evidenceReceipts: outcome.evidenceReceipts,
|
|
265
|
+
...(!outcome.evidenceReceipt ? { evidenceAbsence: "not-started" } : {}),
|
|
266
|
+
nonce: outcome.meta.nonce,
|
|
267
|
+
stdoutPath: outcome.meta.stdoutPath,
|
|
268
|
+
stderrPath: outcome.meta.stderrPath,
|
|
218
269
|
exitCode: outcome.exitCode,
|
|
219
270
|
reportPath: outcome.reportPath,
|
|
220
271
|
spawnedCommand: outcome.meta.spawnedCommand,
|
|
@@ -230,18 +281,42 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
230
281
|
// `sh` defaults to. A suite whose capture measured 600007ms carries a recorded 1800021ms ceiling;
|
|
231
282
|
// running it under 600000ms here SIGKILLed a green tip three times while every per-task gate passed.
|
|
232
283
|
const ceilingMs = effectiveCeilingMs(entry);
|
|
233
|
-
|
|
284
|
+
const evidenceReceipts = [];
|
|
285
|
+
const execute = async () => {
|
|
286
|
+
const evidence = beginGateEvidence(intWt, gate, cmd, evidenceSetup);
|
|
287
|
+
evidenceRoots.set(gate, evidence.root);
|
|
288
|
+
try {
|
|
289
|
+
const result = await sh(cmd, intWt, ceilingMs, { env: evidenceOptions.env, onReceipt: receipt => evidence.observe(receipt) });
|
|
290
|
+
evidenceReceipts.push(...evidence.history, evidence.finish(result.stdout, result.stderr));
|
|
291
|
+
return result;
|
|
292
|
+
}
|
|
293
|
+
catch (error) {
|
|
294
|
+
const receipt = evidence.finish();
|
|
295
|
+
if (receipt.termination.kind !== "not-started" || executionSignal()?.aborted)
|
|
296
|
+
throw error;
|
|
297
|
+
evidenceReceipts.push(...evidence.history, receipt);
|
|
298
|
+
return { code: -1, stdout: "", stderr: String(error) };
|
|
299
|
+
}
|
|
300
|
+
};
|
|
301
|
+
let r = await execute();
|
|
234
302
|
let raw = r.stdout + "\n" + r.stderr;
|
|
235
303
|
let stripped = raw.split(intWt).join("");
|
|
236
304
|
let rerun;
|
|
237
|
-
if (gate === "test" && !r.timedOut && !fileCountDeficit(entry, stripped)
|
|
305
|
+
if (evidenceReceipts.at(-1)?.termination.kind !== "not-started" && gate === "test" && !r.timedOut && !fileCountDeficit(entry, stripped)
|
|
238
306
|
&& classifyFreshRunnerOutput(entry, stripped, r.code) === "infra") {
|
|
239
307
|
const waitedMs = await waitForCalmWindow();
|
|
240
308
|
rerun = `runner-infra rerun after waiting ${waitedMs}ms for a calm load window`;
|
|
241
|
-
r = await
|
|
309
|
+
r = await execute();
|
|
242
310
|
raw = r.stdout + "\n" + r.stderr;
|
|
243
311
|
stripped = raw.split(intWt).join("");
|
|
244
312
|
}
|
|
313
|
+
const evidenceReceipt = evidenceReceipts.at(-1);
|
|
314
|
+
if (evidenceReceipt.termination.kind === "not-started") {
|
|
315
|
+
results.push({ gate, cmd, pass: false, exitCode: r.code, fingerprints: [],
|
|
316
|
+
details: `infra; command failed to launch: ${r.stderr}`, cause: "infra",
|
|
317
|
+
evidenceReceipt, evidenceReceipts, evidenceAbsence: "not-started" });
|
|
318
|
+
continue;
|
|
319
|
+
}
|
|
245
320
|
const artifact = join(runDir, `tip-verify-${gate}.log`);
|
|
246
321
|
// Battery parity on the ceiling too (baseline.ts Q24): the kill is read BEFORE the exit code is
|
|
247
322
|
// interpreted at all. A SIGKILLed battery never returned a verdict, so no line of its partial
|
|
@@ -249,11 +324,12 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
249
324
|
// operator cannot act on. The kill reader's own text (ceiling + elapsed) and cause replace it.
|
|
250
325
|
const killed = ceilingKillResult(gate, r, ceilingMs);
|
|
251
326
|
if (killed) {
|
|
252
|
-
|
|
327
|
+
writeTipLog(artifact, raw);
|
|
253
328
|
results.push({
|
|
254
329
|
gate,
|
|
255
330
|
cmd,
|
|
256
331
|
pass: false,
|
|
332
|
+
evidenceReceipt, evidenceReceipts,
|
|
257
333
|
exitCode: r.code,
|
|
258
334
|
fingerprints: [],
|
|
259
335
|
details: `${rerun ? `infra; ${rerun}: ` : ""}${killed.details}`,
|
|
@@ -290,11 +366,12 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
290
366
|
&& comparable;
|
|
291
367
|
const pass = !deficit && (r.code === 0 || greenTeardown || forgiven);
|
|
292
368
|
if (!pass)
|
|
293
|
-
|
|
369
|
+
writeTipLog(artifact, raw);
|
|
294
370
|
const tipResult = {
|
|
295
371
|
gate,
|
|
296
372
|
cmd,
|
|
297
373
|
pass,
|
|
374
|
+
evidenceReceipt, evidenceReceipts,
|
|
298
375
|
exitCode: r.code,
|
|
299
376
|
fingerprints: r.code !== 0 && !greenTeardown ? fingerprint(stripped) : [],
|
|
300
377
|
details: `${cause === "infra" ? "infra; " : ""}${rerun ? `${rerun}: ` : ""}` + (deficit?.replace(/^infra; /, "") ?? (r.code === 0 ? "exit 0"
|
|
@@ -315,12 +392,21 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
315
392
|
// spawns in the daemon child, and a command that changed the tree invalidates reuse.
|
|
316
393
|
const completedTree = pending.size ? await getWorktreeTree(intWt) : undefined;
|
|
317
394
|
for (const result of results) {
|
|
395
|
+
if (!result.reused && result.evidenceReceipt) {
|
|
396
|
+
const evidenceRoot = evidenceRoots.get(result.gate);
|
|
397
|
+
if (evidenceRoot === undefined)
|
|
398
|
+
throw new Error(`tip verify receipt for ${result.gate} has no evidence root from its evidence setup`);
|
|
399
|
+
result.originRunRoot = evidenceRoot;
|
|
400
|
+
}
|
|
318
401
|
const id = pending.get(result.gate);
|
|
319
402
|
if (id && id.tree === completedTree && !isInfraResult(result)) {
|
|
320
403
|
store.set(id, { ...result, meta: {
|
|
321
404
|
fingerprints: result.fingerprints, forgiven: result.forgiven,
|
|
322
405
|
reportPath: result.reportPath, spawnedCommand: result.spawnedCommand,
|
|
323
406
|
artifact: result.artifact,
|
|
407
|
+
evidenceReceipt: result.evidenceReceipt, evidenceReceipts: result.evidenceReceipts,
|
|
408
|
+
evidenceAbsence: result.evidenceAbsence, originRunRoot: result.originRunRoot,
|
|
409
|
+
nonce: result.nonce, stdoutPath: result.stdoutPath, stderrPath: result.stderrPath,
|
|
324
410
|
} });
|
|
325
411
|
}
|
|
326
412
|
}
|
package/dist/run/protocol.d.ts
CHANGED
|
@@ -37,6 +37,88 @@ export declare const ShellReceiptSchema: z.ZodObject<{
|
|
|
37
37
|
durationMs: z.ZodOptional<z.ZodNumber>;
|
|
38
38
|
}, z.core.$strict>;
|
|
39
39
|
export type ShellReceipt = z.infer<typeof ShellReceiptSchema>;
|
|
40
|
+
export declare const EVIDENCE_AVAILABILITIES: readonly ["available", "not-started", "capture-failed", "killed", "expired", "missing"];
|
|
41
|
+
export declare const EvidenceArtifactSchema: z.ZodObject<{
|
|
42
|
+
path: z.ZodString;
|
|
43
|
+
availability: z.ZodEnum<{
|
|
44
|
+
available: "available";
|
|
45
|
+
"not-started": "not-started";
|
|
46
|
+
"capture-failed": "capture-failed";
|
|
47
|
+
killed: "killed";
|
|
48
|
+
expired: "expired";
|
|
49
|
+
missing: "missing";
|
|
50
|
+
}>;
|
|
51
|
+
sha256: z.ZodNullable<z.ZodString>;
|
|
52
|
+
retainedBytes: z.ZodNumber;
|
|
53
|
+
droppedBytes: z.ZodNumber;
|
|
54
|
+
truncated: z.ZodBoolean;
|
|
55
|
+
}, z.core.$strict>;
|
|
56
|
+
export declare const GateEvidenceReceiptSchema: z.ZodObject<{
|
|
57
|
+
invocationId: z.ZodString;
|
|
58
|
+
nonce: z.ZodOptional<z.ZodString>;
|
|
59
|
+
subject: z.ZodObject<{
|
|
60
|
+
runId: z.ZodString;
|
|
61
|
+
taskId: z.ZodNullable<z.ZodString>;
|
|
62
|
+
attempt: z.ZodNullable<z.ZodNumber>;
|
|
63
|
+
gate: z.ZodString;
|
|
64
|
+
subjectCommit: z.ZodNullable<z.ZodString>;
|
|
65
|
+
}, z.core.$strict>;
|
|
66
|
+
termination: z.ZodObject<{
|
|
67
|
+
kind: z.ZodEnum<{
|
|
68
|
+
unknown: "unknown";
|
|
69
|
+
signal: "signal";
|
|
70
|
+
"not-started": "not-started";
|
|
71
|
+
exit: "exit";
|
|
72
|
+
timeout: "timeout";
|
|
73
|
+
}>;
|
|
74
|
+
exitCode: z.ZodNullable<z.ZodNumber>;
|
|
75
|
+
signal: z.ZodNullable<z.ZodString>;
|
|
76
|
+
timedOut: z.ZodNullable<z.ZodBoolean>;
|
|
77
|
+
}, z.core.$strict>;
|
|
78
|
+
availability: z.ZodEnum<{
|
|
79
|
+
available: "available";
|
|
80
|
+
"not-started": "not-started";
|
|
81
|
+
"capture-failed": "capture-failed";
|
|
82
|
+
killed: "killed";
|
|
83
|
+
expired: "expired";
|
|
84
|
+
missing: "missing";
|
|
85
|
+
}>;
|
|
86
|
+
redaction: z.ZodObject<{
|
|
87
|
+
material: z.ZodBoolean;
|
|
88
|
+
}, z.core.$strict>;
|
|
89
|
+
stdout: z.ZodObject<{
|
|
90
|
+
path: z.ZodString;
|
|
91
|
+
availability: z.ZodEnum<{
|
|
92
|
+
available: "available";
|
|
93
|
+
"not-started": "not-started";
|
|
94
|
+
"capture-failed": "capture-failed";
|
|
95
|
+
killed: "killed";
|
|
96
|
+
expired: "expired";
|
|
97
|
+
missing: "missing";
|
|
98
|
+
}>;
|
|
99
|
+
sha256: z.ZodNullable<z.ZodString>;
|
|
100
|
+
retainedBytes: z.ZodNumber;
|
|
101
|
+
droppedBytes: z.ZodNumber;
|
|
102
|
+
truncated: z.ZodBoolean;
|
|
103
|
+
}, z.core.$strict>;
|
|
104
|
+
stderr: z.ZodObject<{
|
|
105
|
+
path: z.ZodString;
|
|
106
|
+
availability: z.ZodEnum<{
|
|
107
|
+
available: "available";
|
|
108
|
+
"not-started": "not-started";
|
|
109
|
+
"capture-failed": "capture-failed";
|
|
110
|
+
killed: "killed";
|
|
111
|
+
expired: "expired";
|
|
112
|
+
missing: "missing";
|
|
113
|
+
}>;
|
|
114
|
+
sha256: z.ZodNullable<z.ZodString>;
|
|
115
|
+
retainedBytes: z.ZodNumber;
|
|
116
|
+
droppedBytes: z.ZodNumber;
|
|
117
|
+
truncated: z.ZodBoolean;
|
|
118
|
+
}, z.core.$strict>;
|
|
119
|
+
}, z.core.$strict>;
|
|
120
|
+
export type GateEvidenceReceipt = z.infer<typeof GateEvidenceReceiptSchema>;
|
|
121
|
+
export type EvidenceArtifact = z.infer<typeof EvidenceArtifactSchema>;
|
|
40
122
|
/** A task-build row must carry the entire caller-owned correlation, even before spawn. */
|
|
41
123
|
export declare const CommandReceiptSchema: z.ZodObject<{
|
|
42
124
|
outcome: z.ZodEnum<{
|