tickmarkr 2.5.7 → 2.5.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/qwen.js +30 -3
- package/dist/cli/commands/approve.js +23 -4
- package/dist/cli/commands/beat.d.ts +2 -0
- package/dist/cli/commands/beat.js +28 -29
- package/dist/cli/commands/compile.js +18 -0
- package/dist/cli/commands/fleet.js +61 -53
- package/dist/cli/commands/plan.d.ts +5 -0
- package/dist/cli/commands/plan.js +28 -23
- package/dist/cli/commands/resume.js +1 -1
- package/dist/cli/commands/run.js +1 -1
- package/dist/cli/commands/verify.d.ts +4 -1
- package/dist/cli/commands/verify.js +16 -5
- package/dist/cli/help.d.ts +2 -0
- package/dist/cli/help.js +6 -4
- package/dist/compile/native.js +68 -7
- package/dist/compile/retired-literals.d.ts +22 -0
- package/dist/compile/retired-literals.js +271 -0
- package/dist/config/config.d.ts +41 -3
- package/dist/config/config.js +48 -17
- package/dist/config/fleet-overlay.js +47 -62
- package/dist/drivers/index.d.ts +4 -2
- package/dist/drivers/index.js +54 -6
- package/dist/gates/baseline.d.ts +65 -0
- package/dist/gates/baseline.js +163 -9
- package/dist/gates/review.d.ts +6 -4
- package/dist/gates/review.js +62 -21
- package/dist/gates/run-gates.d.ts +6 -1
- package/dist/gates/run-gates.js +25 -13
- package/dist/gates/test-manifest.d.ts +20 -1
- package/dist/gates/test-manifest.js +50 -22
- package/dist/gates/test-reporter.js +22 -1
- package/dist/graph/schema.d.ts +28 -0
- package/dist/graph/schema.js +13 -1
- package/dist/run/daemon.d.ts +19 -0
- package/dist/run/daemon.js +363 -118
- package/dist/run/journal.d.ts +54 -3
- package/dist/run/journal.js +142 -11
- package/dist/run/merge.d.ts +15 -2
- package/dist/run/merge.js +74 -11
- package/dist/run/protocol.d.ts +82 -0
- package/dist/run/protocol.js +35 -0
- package/dist/run/receipt-resolver.d.ts +18 -0
- package/dist/run/receipt-resolver.js +132 -0
- package/dist/run/repair-disposition.d.ts +41 -0
- package/dist/run/repair-disposition.js +77 -0
- package/dist/run/supervision.d.ts +14 -1
- package/dist/run/supervision.js +122 -24
- package/dist/tui/cockpit/evidence-view.d.ts +10 -1
- package/dist/tui/cockpit/evidence-view.js +37 -5
- package/dist/tui/cockpit/home-view.js +45 -30
- package/dist/tui/cockpit/live-store.d.ts +18 -0
- package/dist/tui/ink/fleet-app.d.ts +12 -22
- package/dist/tui/ink/fleet-app.js +520 -131
- package/package.json +1 -1
- package/schema/rungraph.schema.json +54 -0
- package/skills/tickmarkr-overseer/SKILL.md +168 -36
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +14 -2
package/dist/run/journal.d.ts
CHANGED
|
@@ -43,6 +43,15 @@ export interface StructuredFinding {
|
|
|
43
43
|
note: string;
|
|
44
44
|
rationale?: string;
|
|
45
45
|
fingerprint: string;
|
|
46
|
+
/** Closed list of spellings actually observed on this chain, including the current one. */
|
|
47
|
+
observedFingerprints?: string[];
|
|
48
|
+
/** Prior id validated against the reviewer's reraised list by the review gate. */
|
|
49
|
+
reraisedFrom?: string;
|
|
50
|
+
/** Resolved definition AND defect identity; bare symbol spelling is never lineage. */
|
|
51
|
+
codeIdentity?: {
|
|
52
|
+
definition: string;
|
|
53
|
+
defect: string;
|
|
54
|
+
};
|
|
46
55
|
}
|
|
47
56
|
declare const CONSULT_ACTIONS: readonly ["retry", "reroute", "decompose", "human"];
|
|
48
57
|
export type ConsultGuidanceAction = (typeof CONSULT_ACTIONS)[number];
|
|
@@ -132,6 +141,42 @@ export declare function repairReachSinceApproval(events: JournalEvent[], taskId:
|
|
|
132
141
|
export declare function repairsSinceApproval(events: JournalEvent[], taskId: string): number;
|
|
133
142
|
/** Recheck releases not yet enacted by a battery (or by a legacy worker launch). */
|
|
134
143
|
export declare function pendingRechecks(events: JournalEvent[]): Set<string>;
|
|
144
|
+
/**
|
|
145
|
+
* OBS-1075: what an accepted approval still authorises, typed by its release. `worker` funds one
|
|
146
|
+
* dispatch (plain, attempt-cap, review-upheld, scope-request); `battery` funds the recheck battery
|
|
147
|
+
* and no worker; `waiver` satisfies exactly its named gate; `inert` authorises nothing.
|
|
148
|
+
*/
|
|
149
|
+
export type PendingApprovalAction = {
|
|
150
|
+
taskId: string;
|
|
151
|
+
ts: string;
|
|
152
|
+
authority: "worker";
|
|
153
|
+
release: "plain" | typeof ATTEMPT_CAP_RELEASE | typeof REVIEW_UPHELD_RELEASE | "scope-request";
|
|
154
|
+
} | {
|
|
155
|
+
taskId: string;
|
|
156
|
+
ts: string;
|
|
157
|
+
authority: "battery";
|
|
158
|
+
release: typeof RECHECK_RELEASE;
|
|
159
|
+
} | {
|
|
160
|
+
taskId: string;
|
|
161
|
+
ts: string;
|
|
162
|
+
authority: "waiver";
|
|
163
|
+
release: typeof GATE_SATISFIED_RELEASE;
|
|
164
|
+
gate: GateName;
|
|
165
|
+
} | {
|
|
166
|
+
taskId: string;
|
|
167
|
+
ts: string;
|
|
168
|
+
authority: "inert";
|
|
169
|
+
release: unknown;
|
|
170
|
+
};
|
|
171
|
+
/**
|
|
172
|
+
* Per task, the newest task-approved row no later enactment has consumed. Pure: events in, typed
|
|
173
|
+
* actions out. Acceptance is not enactment and a run boundary is not enactment — only the row the
|
|
174
|
+
* release causes consumes it (ENACTED_BY), so an approval accepted just before an abnormal exit
|
|
175
|
+
* survives every restart until it is enacted. A later approval supersedes an earlier one. An unknown
|
|
176
|
+
* release, or a waiver without a known gate, is inert: it supersedes, authorises nothing, and no row
|
|
177
|
+
* enacts it.
|
|
178
|
+
*/
|
|
179
|
+
export declare function pendingApprovalActions(events: JournalEvent[]): Map<string, PendingApprovalAction>;
|
|
135
180
|
/**
|
|
136
181
|
* Why the last attempt failed, one row per journaled cause, in the daemon's own `source: details`
|
|
137
182
|
* shape. The daemon builds that brief in a loop-local variable, which dies with the process: a resumed
|
|
@@ -151,6 +196,11 @@ export declare function pendingRechecks(events: JournalEvent[]): Set<string>;
|
|
|
151
196
|
* because every approval erased exactly the finding the fresh attempt was funded to fix.
|
|
152
197
|
*/
|
|
153
198
|
export declare function journaledFailureBrief(events: JournalEvent[], taskId: string): string[];
|
|
199
|
+
/** Never infer a cross-path alias from a symbol, sentence, or hash. */
|
|
200
|
+
export declare function reviewFingerprintMatches(candidate: unknown, fingerprint: string): boolean;
|
|
201
|
+
export declare function observedReviewFingerprints(finding: StructuredFinding): string[];
|
|
202
|
+
/** Re-seat only an unambiguous, positively linked chain; retain the newest evidence path. */
|
|
203
|
+
export declare function carryReviewFindings(priors: readonly StructuredFinding[], rows: readonly StructuredFinding[]): StructuredFinding[];
|
|
154
204
|
/**
|
|
155
205
|
* T6: the review findings still OUTSTANDING on a task. A review finding is a property of the TASK,
|
|
156
206
|
* not of the attempt that drew it: it stays outstanding until a later review PASSES on the task (or
|
|
@@ -172,7 +222,8 @@ export declare function journaledFailureBrief(events: JournalEvent[], taskId: st
|
|
|
172
222
|
* other gate settled that gate, not this one. Reading "approved" as "settled" is how a still-open
|
|
173
223
|
* finding was dropped at the exact moment the operator paid for another attempt to fix it. A review
|
|
174
224
|
* that DECLINED (`skipped`) is not a verdict and neither adds nor retires — fail closed. Findings are
|
|
175
|
-
*
|
|
225
|
+
* linked by validated lineage across paths; observed spellings remain a closed list. Repeating the
|
|
226
|
+
* same row carries it once, while distinct defects sharing a symbol retain separate fingerprints.
|
|
176
227
|
*
|
|
177
228
|
* v2.1.5 T2: a passing review settles the findings it BLOCKED on. It does not settle the ones it
|
|
178
229
|
* DEFERRED — those it saw, declined to block on, and recorded a rationale for, and nothing has fixed
|
|
@@ -240,6 +291,7 @@ export declare const TelemetryRowSchema: z.ZodObject<{
|
|
|
240
291
|
"gate-fail": "gate-fail";
|
|
241
292
|
quota: "quota";
|
|
242
293
|
infra: "infra";
|
|
294
|
+
"scope-request": "scope-request";
|
|
243
295
|
dispatch: "dispatch";
|
|
244
296
|
"human-gate": "human-gate";
|
|
245
297
|
"reroute-exhausted": "reroute-exhausted";
|
|
@@ -247,7 +299,6 @@ export declare const TelemetryRowSchema: z.ZodObject<{
|
|
|
247
299
|
"merge-conflict": "merge-conflict";
|
|
248
300
|
"tip-moved": "tip-moved";
|
|
249
301
|
authoring: "authoring";
|
|
250
|
-
"scope-request": "scope-request";
|
|
251
302
|
"diff-cap": "diff-cap";
|
|
252
303
|
}>>;
|
|
253
304
|
tokens: z.ZodCatch<z.ZodOptional<z.ZodObject<{
|
|
@@ -276,8 +327,8 @@ export declare const TelemetryRowSchema: z.ZodObject<{
|
|
|
276
327
|
}>>;
|
|
277
328
|
kind: z.ZodOptional<z.ZodLiteral<"judge">>;
|
|
278
329
|
judgeOutcome: z.ZodOptional<z.ZodEnum<{
|
|
279
|
-
parseable: "parseable";
|
|
280
330
|
unparseable: "unparseable";
|
|
331
|
+
parseable: "parseable";
|
|
281
332
|
}>>;
|
|
282
333
|
}, z.core.$strip>;
|
|
283
334
|
export type TelemetryRow = z.infer<typeof TelemetryRowSchema>;
|
package/dist/run/journal.js
CHANGED
|
@@ -620,6 +620,49 @@ export function pendingRechecks(events) {
|
|
|
620
620
|
}
|
|
621
621
|
return pending;
|
|
622
622
|
}
|
|
623
|
+
const ENACTED_BY = {
|
|
624
|
+
worker: ["task-dispatch", "worker-launch"],
|
|
625
|
+
battery: ["recheck-battery"],
|
|
626
|
+
waiver: ["worktree-recreation"],
|
|
627
|
+
};
|
|
628
|
+
/**
|
|
629
|
+
* Per task, the newest task-approved row no later enactment has consumed. Pure: events in, typed
|
|
630
|
+
* actions out. Acceptance is not enactment and a run boundary is not enactment — only the row the
|
|
631
|
+
* release causes consumes it (ENACTED_BY), so an approval accepted just before an abnormal exit
|
|
632
|
+
* survives every restart until it is enacted. A later approval supersedes an earlier one. An unknown
|
|
633
|
+
* release, or a waiver without a known gate, is inert: it supersedes, authorises nothing, and no row
|
|
634
|
+
* enacts it.
|
|
635
|
+
*/
|
|
636
|
+
export function pendingApprovalActions(events) {
|
|
637
|
+
const pending = new Map();
|
|
638
|
+
for (const e of events) {
|
|
639
|
+
if (!e.taskId)
|
|
640
|
+
continue;
|
|
641
|
+
if (e.event === "task-approved") {
|
|
642
|
+
pending.set(e.taskId, approvalAction(e.taskId, e));
|
|
643
|
+
continue;
|
|
644
|
+
}
|
|
645
|
+
const action = pending.get(e.taskId);
|
|
646
|
+
if (action && action.authority !== "inert" && ENACTED_BY[action.authority].includes(e.event))
|
|
647
|
+
pending.delete(e.taskId);
|
|
648
|
+
}
|
|
649
|
+
return pending;
|
|
650
|
+
}
|
|
651
|
+
function approvalAction(taskId, e) {
|
|
652
|
+
const { release, gate } = e.data;
|
|
653
|
+
const base = { taskId, ts: e.ts };
|
|
654
|
+
if (release === undefined)
|
|
655
|
+
return { ...base, authority: "worker", release: "plain" };
|
|
656
|
+
if (release === ATTEMPT_CAP_RELEASE || release === REVIEW_UPHELD_RELEASE || release === "scope-request") {
|
|
657
|
+
return { ...base, authority: "worker", release };
|
|
658
|
+
}
|
|
659
|
+
if (release === RECHECK_RELEASE)
|
|
660
|
+
return { ...base, authority: "battery", release };
|
|
661
|
+
if (release === GATE_SATISFIED_RELEASE && typeof gate === "string" && GATE_NAMES.includes(gate)) {
|
|
662
|
+
return { ...base, authority: "waiver", release, gate: gate };
|
|
663
|
+
}
|
|
664
|
+
return { ...base, authority: "inert", release };
|
|
665
|
+
}
|
|
623
666
|
// Both retry decisions below govern exactly ONE dispatch: the next one. So both are read back from the
|
|
624
667
|
// journal at the moment that dispatch is built, never carried in a process variable — a stop between
|
|
625
668
|
// the decision and the dispatch (OBS-254's shape, one layer up) would otherwise send a normal prompt
|
|
@@ -687,6 +730,83 @@ export function journaledFailureBrief(events, taskId) {
|
|
|
687
730
|
}
|
|
688
731
|
return rows;
|
|
689
732
|
}
|
|
733
|
+
/** Never infer a cross-path alias from a symbol, sentence, or hash. */
|
|
734
|
+
export function reviewFingerprintMatches(candidate, fingerprint) {
|
|
735
|
+
return typeof candidate === "string"
|
|
736
|
+
&& candidate.replace(/^\s*fingerprint\s*:\s*/i, "").replace(/\s+/g, "") === fingerprint.replace(/\s+/g, "");
|
|
737
|
+
}
|
|
738
|
+
export function observedReviewFingerprints(finding) {
|
|
739
|
+
const observed = Array.isArray(finding.observedFingerprints)
|
|
740
|
+
? finding.observedFingerprints.filter((id) => typeof id === "string") : [];
|
|
741
|
+
return [...new Set([finding.fingerprint, ...observed])];
|
|
742
|
+
}
|
|
743
|
+
const reviewNoteIdentity = (note) => note.replace(LINE_REF_RE, "").replace(/\s+/g, " ").trim();
|
|
744
|
+
function distinctReviewFinding(row) {
|
|
745
|
+
const identity = JSON.stringify([reviewNoteIdentity(row.note), row.codeIdentity ?? null]);
|
|
746
|
+
const symbol = `${row.symbol}#${createHash("sha256").update(identity).digest("hex").slice(0, 12)}`;
|
|
747
|
+
return { ...row, symbol, fingerprint: `${row.class}|${row.path}|${symbol}` };
|
|
748
|
+
}
|
|
749
|
+
/** Re-seat only an unambiguous, positively linked chain; retain the newest evidence path. */
|
|
750
|
+
export function carryReviewFindings(priors, rows) {
|
|
751
|
+
const open = [...priors];
|
|
752
|
+
for (const row of rows) {
|
|
753
|
+
const identity = row.codeIdentity;
|
|
754
|
+
const matches = open.filter((prior) => {
|
|
755
|
+
if (prior.class !== row.class)
|
|
756
|
+
return false;
|
|
757
|
+
if (row.reraisedFrom)
|
|
758
|
+
return rows.filter((other) => other.reraisedFrom === row.reraisedFrom).length === 1
|
|
759
|
+
&& observedReviewFingerprints(prior).includes(row.reraisedFrom);
|
|
760
|
+
if (identity?.definition && identity.defect && prior.codeIdentity) {
|
|
761
|
+
return prior.codeIdentity.definition === identity.definition && prior.codeIdentity.defect === identity.defect;
|
|
762
|
+
}
|
|
763
|
+
return prior.fingerprint === row.fingerprint && reviewNoteIdentity(prior.note) === reviewNoteIdentity(row.note);
|
|
764
|
+
});
|
|
765
|
+
if (matches.length === 1) {
|
|
766
|
+
const prior = matches[0];
|
|
767
|
+
// Positive lineage does not grant this chain another open defect's display spelling.
|
|
768
|
+
let newest = row;
|
|
769
|
+
while (open.some((other) => other !== prior && observedReviewFingerprints(other).includes(newest.fingerprint))) {
|
|
770
|
+
newest = distinctReviewFinding(newest);
|
|
771
|
+
}
|
|
772
|
+
const observed = [...new Set([...observedReviewFingerprints(prior), ...observedReviewFingerprints(newest)])];
|
|
773
|
+
open[open.indexOf(prior)] = {
|
|
774
|
+
...newest,
|
|
775
|
+
...(prior.codeIdentity && !newest.codeIdentity ? { codeIdentity: prior.codeIdentity } : {}),
|
|
776
|
+
...(observed.length > 1 ? { observedFingerprints: observed } : {}),
|
|
777
|
+
};
|
|
778
|
+
}
|
|
779
|
+
else {
|
|
780
|
+
// Two defects may name the same definition. Preserve both, with a copyable full triple.
|
|
781
|
+
// Deferrals intentionally replace their rationale without changing their note.
|
|
782
|
+
const collision = open.some((prior) => observedReviewFingerprints(prior).includes(row.fingerprint));
|
|
783
|
+
if (collision) {
|
|
784
|
+
let distinct = distinctReviewFinding(row);
|
|
785
|
+
// A generated spelling may itself belong to a chain that has since moved away.
|
|
786
|
+
// Reuse only a matching row; otherwise keep minting until the spelling is unowned.
|
|
787
|
+
while (true) {
|
|
788
|
+
const owners = open.filter((prior) => observedReviewFingerprints(prior).includes(distinct.fingerprint));
|
|
789
|
+
if (owners.length === 0) {
|
|
790
|
+
open.push(distinct);
|
|
791
|
+
break;
|
|
792
|
+
}
|
|
793
|
+
if (owners.length === 1 && reviewNoteIdentity(owners[0].note) === reviewNoteIdentity(row.note)) {
|
|
794
|
+
const prior = owners[0];
|
|
795
|
+
open[open.indexOf(prior)] = {
|
|
796
|
+
...prior, ...distinct,
|
|
797
|
+
observedFingerprints: [...new Set([...observedReviewFingerprints(prior), ...observedReviewFingerprints(distinct)])],
|
|
798
|
+
};
|
|
799
|
+
break;
|
|
800
|
+
}
|
|
801
|
+
distinct = distinctReviewFinding(distinct);
|
|
802
|
+
}
|
|
803
|
+
}
|
|
804
|
+
else
|
|
805
|
+
open.push(row);
|
|
806
|
+
}
|
|
807
|
+
}
|
|
808
|
+
return open;
|
|
809
|
+
}
|
|
690
810
|
/**
|
|
691
811
|
* T6: the review findings still OUTSTANDING on a task. A review finding is a property of the TASK,
|
|
692
812
|
* not of the attempt that drew it: it stays outstanding until a later review PASSES on the task (or
|
|
@@ -708,7 +828,8 @@ export function journaledFailureBrief(events, taskId) {
|
|
|
708
828
|
* other gate settled that gate, not this one. Reading "approved" as "settled" is how a still-open
|
|
709
829
|
* finding was dropped at the exact moment the operator paid for another attempt to fix it. A review
|
|
710
830
|
* that DECLINED (`skipped`) is not a verdict and neither adds nor retires — fail closed. Findings are
|
|
711
|
-
*
|
|
831
|
+
* linked by validated lineage across paths; observed spellings remain a closed list. Repeating the
|
|
832
|
+
* same row carries it once, while distinct defects sharing a symbol retain separate fingerprints.
|
|
712
833
|
*
|
|
713
834
|
* v2.1.5 T2: a passing review settles the findings it BLOCKED on. It does not settle the ones it
|
|
714
835
|
* DEFERRED — those it saw, declined to block on, and recorded a rationale for, and nothing has fixed
|
|
@@ -724,34 +845,44 @@ export function journaledFailureBrief(events, taskId) {
|
|
|
724
845
|
* row. N rounds of the same concern therefore carry the newest accepted explanation once, not N rows.
|
|
725
846
|
*/
|
|
726
847
|
export function outstandingReviewFindings(events, taskId) {
|
|
727
|
-
|
|
848
|
+
let open = [];
|
|
728
849
|
for (const e of events) {
|
|
729
850
|
if (e.taskId !== taskId)
|
|
730
851
|
continue;
|
|
731
852
|
if (e.event === "task-approved") {
|
|
732
853
|
if (e.data.release === GATE_SATISFIED_RELEASE && e.data.gate === "review")
|
|
733
|
-
open
|
|
854
|
+
open = [];
|
|
734
855
|
continue;
|
|
735
856
|
}
|
|
736
857
|
if (e.event !== "gate-result" || e.data.gate !== "review" || e.data.skipped === true)
|
|
737
858
|
continue;
|
|
859
|
+
// Rejected or missing verdicts carry diagnostics, not accepted findings or closures.
|
|
860
|
+
if (e.data.unparseable === true || e.data.noVerdict === true || e.data.cause !== undefined)
|
|
861
|
+
continue;
|
|
862
|
+
// A failed review can resolve one chain while re-raising another. Only observed, uniquely
|
|
863
|
+
// matched spellings retire a chain; skipped/no-verdict rows were excluded above.
|
|
864
|
+
if (Array.isArray(e.data.resolved)) {
|
|
865
|
+
const resolved = e.data.resolved;
|
|
866
|
+
const settled = new Set(resolved.flatMap((id) => {
|
|
867
|
+
const matches = open.filter((finding) => finding.class === "review:material"
|
|
868
|
+
&& observedReviewFingerprints(finding).some((fp) => reviewFingerprintMatches(id, fp)));
|
|
869
|
+
return matches.length === 1 ? matches : [];
|
|
870
|
+
}));
|
|
871
|
+
open = open.filter((finding) => !settled.has(finding));
|
|
872
|
+
}
|
|
738
873
|
if (e.data.pass !== false) {
|
|
739
874
|
// a later review PASSED on this task: every finding it BLOCKED on is settled …
|
|
740
|
-
|
|
741
|
-
if (!isDeferredFinding(finding))
|
|
742
|
-
open.delete(key);
|
|
875
|
+
open = open.filter(isDeferredFinding);
|
|
743
876
|
// … and no deferral is, whether or not this pass restated it. A pass is silent about a
|
|
744
877
|
// deferral it does not mention: the concern is unfixed either way, and the reviewer that
|
|
745
878
|
// waved it through is not the release that accepts it. Retiring on omission would drop it on
|
|
746
879
|
// the very next round — the same silent drop by a different door.
|
|
747
|
-
|
|
748
|
-
open.set(finding.fingerprint, finding);
|
|
880
|
+
open = carryReviewFindings(open, findingRows(e, "review").filter(isDeferredFinding));
|
|
749
881
|
}
|
|
750
882
|
else
|
|
751
|
-
|
|
752
|
-
open.set(finding.fingerprint, finding);
|
|
883
|
+
open = carryReviewFindings(open, findingRows(e, "review"));
|
|
753
884
|
}
|
|
754
|
-
return
|
|
885
|
+
return open;
|
|
755
886
|
}
|
|
756
887
|
/** The findings a funded repair must carry into the next dispatch, or undefined if none is pending. */
|
|
757
888
|
export function pendingRepairFindings(events, taskId) {
|
package/dist/run/merge.d.ts
CHANGED
|
@@ -1,6 +1,15 @@
|
|
|
1
1
|
import type { TickmarkrConfig } from "../config/config.js";
|
|
2
|
-
import { type Baseline, type FailureClassification } from "../gates/baseline.js";
|
|
2
|
+
import { type Baseline, type GateEvidenceOptions, type FailureClassification } from "../gates/baseline.js";
|
|
3
|
+
import type { GateEvidenceReceipt } from "./protocol.js";
|
|
3
4
|
export interface TipVerifyResult {
|
|
5
|
+
evidenceReceipt?: GateEvidenceReceipt;
|
|
6
|
+
evidenceReceipts?: GateEvidenceReceipt[];
|
|
7
|
+
evidenceAbsence?: "provenance-refused" | "not-started" | "historical-cache";
|
|
8
|
+
/** Absolute directory against which the original receipt artifact paths resolve. */
|
|
9
|
+
originRunRoot?: string;
|
|
10
|
+
nonce?: string;
|
|
11
|
+
stdoutPath?: string;
|
|
12
|
+
stderrPath?: string;
|
|
4
13
|
gate: string;
|
|
5
14
|
cmd: string;
|
|
6
15
|
pass: boolean;
|
|
@@ -12,12 +21,16 @@ export interface TipVerifyResult {
|
|
|
12
21
|
spawnedCommand?: string;
|
|
13
22
|
/** Q121s: nonzero exit whose failures are ALL baseline-recorded — forgiven exactly as the battery forgives. */
|
|
14
23
|
forgiven?: boolean;
|
|
24
|
+
/** D-131: carried from a persisted per-gate verdict — this cycle did NOT execute the command. */
|
|
25
|
+
reused?: true;
|
|
15
26
|
/**
|
|
16
27
|
* OBS-534: what a nonzero exit is evidence OF, taken from the battery's own readers — `ceilingKillResult`
|
|
17
28
|
* for a kill, the shared runner classifier for everything else. `infra` means nothing was verified.
|
|
18
29
|
*/
|
|
19
30
|
cause?: FailureClassification;
|
|
20
31
|
}
|
|
32
|
+
/** Carry execution evidence without minting an invocation or rebasing its artifact paths. */
|
|
33
|
+
export declare function reusedTipEvidence(source: Partial<TipVerifyResult>): Partial<TipVerifyResult>;
|
|
21
34
|
export declare function integrationBranch(cfg: TickmarkrConfig, runId: string): string;
|
|
22
35
|
export declare function ensureIntegration(repo: string, branch: string, baseRef: string): Promise<string>;
|
|
23
36
|
export declare function integrationHead(intWt: string): Promise<string>;
|
|
@@ -29,4 +42,4 @@ export declare function mergeTask(intWt: string, taskBranch: string, message: st
|
|
|
29
42
|
branchTip: string;
|
|
30
43
|
};
|
|
31
44
|
}>;
|
|
32
|
-
export declare function verifyIntegrationTip(intWt: string, commands: Record<string, string>, runDir: string, baseline?: Baseline): Promise<TipVerifyResult[]>;
|
|
45
|
+
export declare function verifyIntegrationTip(intWt: string, commands: Record<string, string>, runDir: string, baseline?: Baseline, evidenceOptions?: GateEvidenceOptions): Promise<TipVerifyResult[]>;
|
package/dist/run/merge.js
CHANGED
|
@@ -1,11 +1,34 @@
|
|
|
1
1
|
import { existsSync, readFileSync, writeFileSync } from "node:fs";
|
|
2
|
-
import { join } from "node:path";
|
|
2
|
+
import { join, resolve } from "node:path";
|
|
3
3
|
import { shq } from "../adapters/types.js";
|
|
4
|
-
import { ceilingKillResult, classifyFreshRunnerOutput, classifyRunnerOutput, fileCountDeficit, waitForCalmWindow, effectiveCeilingMs, fingerprint, freshFailures, } from "../gates/baseline.js";
|
|
4
|
+
import { beginGateEvidence, ceilingKillResult, classifyFreshRunnerOutput, classifyRunnerOutput, fileCountDeficit, waitForCalmWindow, effectiveCeilingMs, fingerprint, freshFailures, } from "../gates/baseline.js";
|
|
5
5
|
import { evaluateManifestedTest, isVitestTestCommand } from "../gates/test-manifest.js";
|
|
6
6
|
import { computeVerificationIdentity, formatReusedDetails, getVerdictStore, getWorktreeTree, resolveStateDir, isInfraResult, } from "../gates/cache.js";
|
|
7
7
|
import { tickmarkrDir } from "../graph/graph.js";
|
|
8
8
|
import { describeCapacity, gitHead, linkNodeModules, resolveIntegrationBranch, resolvedCapacity, sameCapacity, sh, shGit, shGitOk, WORKTREES_DIR } from "./git.js";
|
|
9
|
+
import { executionSignal } from "./execution-budget.js";
|
|
10
|
+
// Diagnostic convenience logs are observational, just like receipt persistence.
|
|
11
|
+
function writeTipLog(path, text) {
|
|
12
|
+
try {
|
|
13
|
+
writeFileSync(path, text);
|
|
14
|
+
}
|
|
15
|
+
catch { /* the command verdict remains authoritative */ }
|
|
16
|
+
}
|
|
17
|
+
/** Carry execution evidence without minting an invocation or rebasing its artifact paths. */
|
|
18
|
+
export function reusedTipEvidence(source) {
|
|
19
|
+
return {
|
|
20
|
+
reused: true,
|
|
21
|
+
...(source.evidenceReceipt ? { evidenceReceipt: source.evidenceReceipt } : {}),
|
|
22
|
+
...(source.evidenceReceipts ? { evidenceReceipts: source.evidenceReceipts } : {}),
|
|
23
|
+
...(source.originRunRoot ? { originRunRoot: source.originRunRoot } : {}),
|
|
24
|
+
...(!source.evidenceReceipt
|
|
25
|
+
? { evidenceAbsence: source.evidenceAbsence ?? "historical-cache" }
|
|
26
|
+
: source.evidenceAbsence ? { evidenceAbsence: source.evidenceAbsence } : {}),
|
|
27
|
+
...(source.nonce !== undefined ? { nonce: source.nonce } : {}),
|
|
28
|
+
...(source.stdoutPath !== undefined ? { stdoutPath: source.stdoutPath } : {}),
|
|
29
|
+
...(source.stderrPath !== undefined ? { stderrPath: source.stderrPath } : {}),
|
|
30
|
+
};
|
|
31
|
+
}
|
|
9
32
|
export function integrationBranch(cfg, runId) {
|
|
10
33
|
return `${cfg.integrationBranchPrefix}${runId}`;
|
|
11
34
|
}
|
|
@@ -52,7 +75,7 @@ export async function mergeTask(intWt, taskBranch, message, gatedCommit) {
|
|
|
52
75
|
// terminus unreachable by construction and misattributed the red to the last merged task.
|
|
53
76
|
// Fail-closed edges kept: no baseline → strict; baseline green for that gate → strict; any fresh
|
|
54
77
|
// fingerprint, or output with no recognizable failure shape, → failed.
|
|
55
|
-
export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
78
|
+
export async function verifyIntegrationTip(intWt, commands, runDir, baseline, evidenceOptions = {}) {
|
|
56
79
|
const results = [];
|
|
57
80
|
let runStartCommands;
|
|
58
81
|
let journalFound = false;
|
|
@@ -109,7 +132,7 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
109
132
|
for (const [gate, cmd] of gatesToRun) {
|
|
110
133
|
const artifact = join(runDir, `tip-verify-${gate}.log`);
|
|
111
134
|
const details = `tip verify could not establish command provenance: run-start evidence missing or malformed in journal`;
|
|
112
|
-
|
|
135
|
+
writeTipLog(artifact, details + "\n");
|
|
113
136
|
results.push({
|
|
114
137
|
gate,
|
|
115
138
|
cmd,
|
|
@@ -117,6 +140,7 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
117
140
|
exitCode: 1,
|
|
118
141
|
fingerprints: [],
|
|
119
142
|
details,
|
|
143
|
+
evidenceAbsence: "provenance-refused",
|
|
120
144
|
artifact,
|
|
121
145
|
});
|
|
122
146
|
}
|
|
@@ -135,7 +159,7 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
135
159
|
if (expectedCmd === undefined || cmd !== expectedCmd) {
|
|
136
160
|
const artifact = join(runDir, `tip-verify-${gate}.log`);
|
|
137
161
|
const details = `tip verify command "${cmd}" for gate "${gate}" was not named in run-start row`;
|
|
138
|
-
|
|
162
|
+
writeTipLog(artifact, details + "\n");
|
|
139
163
|
results.push({
|
|
140
164
|
gate,
|
|
141
165
|
cmd,
|
|
@@ -143,6 +167,7 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
143
167
|
exitCode: 1,
|
|
144
168
|
fingerprints: [],
|
|
145
169
|
details,
|
|
170
|
+
evidenceAbsence: "provenance-refused",
|
|
146
171
|
artifact,
|
|
147
172
|
});
|
|
148
173
|
continue;
|
|
@@ -184,6 +209,7 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
184
209
|
gate,
|
|
185
210
|
cmd,
|
|
186
211
|
pass: hit.pass,
|
|
212
|
+
...reusedTipEvidence(hit.meta ?? {}),
|
|
187
213
|
exitCode: hit.exitCode ?? 0,
|
|
188
214
|
fingerprints: hit.meta?.fingerprints ?? [],
|
|
189
215
|
...(hit.meta?.forgiven ? { forgiven: true } : {}),
|
|
@@ -206,14 +232,21 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
206
232
|
longestFile: entry?.longestFile,
|
|
207
233
|
overallCeilingMs: effectiveCeilingMs(entry),
|
|
208
234
|
artifactDir: runDir,
|
|
235
|
+
evidence: { artifactDir: runDir, runId: runDir, ...evidenceOptions },
|
|
209
236
|
});
|
|
210
237
|
const artifact = join(runDir, `tip-verify-${gate}.log`);
|
|
211
238
|
if (!outcome.pass)
|
|
212
|
-
|
|
239
|
+
writeTipLog(artifact, outcome.details);
|
|
213
240
|
results.push({
|
|
214
241
|
gate,
|
|
215
242
|
cmd,
|
|
216
243
|
pass: outcome.pass,
|
|
244
|
+
evidenceReceipt: outcome.evidenceReceipt,
|
|
245
|
+
evidenceReceipts: outcome.evidenceReceipts,
|
|
246
|
+
...(!outcome.evidenceReceipt ? { evidenceAbsence: "not-started" } : {}),
|
|
247
|
+
nonce: outcome.meta.nonce,
|
|
248
|
+
stdoutPath: outcome.meta.stdoutPath,
|
|
249
|
+
stderrPath: outcome.meta.stderrPath,
|
|
217
250
|
exitCode: outcome.exitCode,
|
|
218
251
|
reportPath: outcome.reportPath,
|
|
219
252
|
spawnedCommand: outcome.meta.spawnedCommand,
|
|
@@ -229,18 +262,41 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
229
262
|
// `sh` defaults to. A suite whose capture measured 600007ms carries a recorded 1800021ms ceiling;
|
|
230
263
|
// running it under 600000ms here SIGKILLed a green tip three times while every per-task gate passed.
|
|
231
264
|
const ceilingMs = effectiveCeilingMs(entry);
|
|
232
|
-
|
|
265
|
+
const evidenceReceipts = [];
|
|
266
|
+
const execute = async () => {
|
|
267
|
+
const evidence = beginGateEvidence(intWt, gate, cmd, { artifactDir: runDir, runId: runDir, ...evidenceOptions });
|
|
268
|
+
try {
|
|
269
|
+
const result = await sh(cmd, intWt, ceilingMs, { env: evidenceOptions.env, onReceipt: receipt => evidence.observe(receipt) });
|
|
270
|
+
evidenceReceipts.push(...evidence.history, evidence.finish(result.stdout, result.stderr));
|
|
271
|
+
return result;
|
|
272
|
+
}
|
|
273
|
+
catch (error) {
|
|
274
|
+
const receipt = evidence.finish();
|
|
275
|
+
if (receipt.termination.kind !== "not-started" || executionSignal()?.aborted)
|
|
276
|
+
throw error;
|
|
277
|
+
evidenceReceipts.push(...evidence.history, receipt);
|
|
278
|
+
return { code: -1, stdout: "", stderr: String(error) };
|
|
279
|
+
}
|
|
280
|
+
};
|
|
281
|
+
let r = await execute();
|
|
233
282
|
let raw = r.stdout + "\n" + r.stderr;
|
|
234
283
|
let stripped = raw.split(intWt).join("");
|
|
235
284
|
let rerun;
|
|
236
|
-
if (gate === "test" && !r.timedOut && !fileCountDeficit(entry, stripped)
|
|
285
|
+
if (evidenceReceipts.at(-1)?.termination.kind !== "not-started" && gate === "test" && !r.timedOut && !fileCountDeficit(entry, stripped)
|
|
237
286
|
&& classifyFreshRunnerOutput(entry, stripped, r.code) === "infra") {
|
|
238
287
|
const waitedMs = await waitForCalmWindow();
|
|
239
288
|
rerun = `runner-infra rerun after waiting ${waitedMs}ms for a calm load window`;
|
|
240
|
-
r = await
|
|
289
|
+
r = await execute();
|
|
241
290
|
raw = r.stdout + "\n" + r.stderr;
|
|
242
291
|
stripped = raw.split(intWt).join("");
|
|
243
292
|
}
|
|
293
|
+
const evidenceReceipt = evidenceReceipts.at(-1);
|
|
294
|
+
if (evidenceReceipt.termination.kind === "not-started") {
|
|
295
|
+
results.push({ gate, cmd, pass: false, exitCode: r.code, fingerprints: [],
|
|
296
|
+
details: `infra; command failed to launch: ${r.stderr}`, cause: "infra",
|
|
297
|
+
evidenceReceipt, evidenceReceipts, evidenceAbsence: "not-started" });
|
|
298
|
+
continue;
|
|
299
|
+
}
|
|
244
300
|
const artifact = join(runDir, `tip-verify-${gate}.log`);
|
|
245
301
|
// Battery parity on the ceiling too (baseline.ts Q24): the kill is read BEFORE the exit code is
|
|
246
302
|
// interpreted at all. A SIGKILLed battery never returned a verdict, so no line of its partial
|
|
@@ -248,11 +304,12 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
248
304
|
// operator cannot act on. The kill reader's own text (ceiling + elapsed) and cause replace it.
|
|
249
305
|
const killed = ceilingKillResult(gate, r, ceilingMs);
|
|
250
306
|
if (killed) {
|
|
251
|
-
|
|
307
|
+
writeTipLog(artifact, raw);
|
|
252
308
|
results.push({
|
|
253
309
|
gate,
|
|
254
310
|
cmd,
|
|
255
311
|
pass: false,
|
|
312
|
+
evidenceReceipt, evidenceReceipts,
|
|
256
313
|
exitCode: r.code,
|
|
257
314
|
fingerprints: [],
|
|
258
315
|
details: `${rerun ? `infra; ${rerun}: ` : ""}${killed.details}`,
|
|
@@ -289,11 +346,12 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
289
346
|
&& comparable;
|
|
290
347
|
const pass = !deficit && (r.code === 0 || greenTeardown || forgiven);
|
|
291
348
|
if (!pass)
|
|
292
|
-
|
|
349
|
+
writeTipLog(artifact, raw);
|
|
293
350
|
const tipResult = {
|
|
294
351
|
gate,
|
|
295
352
|
cmd,
|
|
296
353
|
pass,
|
|
354
|
+
evidenceReceipt, evidenceReceipts,
|
|
297
355
|
exitCode: r.code,
|
|
298
356
|
fingerprints: r.code !== 0 && !greenTeardown ? fingerprint(stripped) : [],
|
|
299
357
|
details: `${cause === "infra" ? "infra; " : ""}${rerun ? `${rerun}: ` : ""}` + (deficit?.replace(/^infra; /, "") ?? (r.code === 0 ? "exit 0"
|
|
@@ -314,12 +372,17 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
|
|
|
314
372
|
// spawns in the daemon child, and a command that changed the tree invalidates reuse.
|
|
315
373
|
const completedTree = pending.size ? await getWorktreeTree(intWt) : undefined;
|
|
316
374
|
for (const result of results) {
|
|
375
|
+
if (!result.reused && result.evidenceReceipt)
|
|
376
|
+
result.originRunRoot = resolve(evidenceOptions.artifactDir ?? runDir);
|
|
317
377
|
const id = pending.get(result.gate);
|
|
318
378
|
if (id && id.tree === completedTree && !isInfraResult(result)) {
|
|
319
379
|
store.set(id, { ...result, meta: {
|
|
320
380
|
fingerprints: result.fingerprints, forgiven: result.forgiven,
|
|
321
381
|
reportPath: result.reportPath, spawnedCommand: result.spawnedCommand,
|
|
322
382
|
artifact: result.artifact,
|
|
383
|
+
evidenceReceipt: result.evidenceReceipt, evidenceReceipts: result.evidenceReceipts,
|
|
384
|
+
evidenceAbsence: result.evidenceAbsence, originRunRoot: result.originRunRoot,
|
|
385
|
+
nonce: result.nonce, stdoutPath: result.stdoutPath, stderrPath: result.stderrPath,
|
|
323
386
|
} });
|
|
324
387
|
}
|
|
325
388
|
}
|
package/dist/run/protocol.d.ts
CHANGED
|
@@ -37,6 +37,88 @@ export declare const ShellReceiptSchema: z.ZodObject<{
|
|
|
37
37
|
durationMs: z.ZodOptional<z.ZodNumber>;
|
|
38
38
|
}, z.core.$strict>;
|
|
39
39
|
export type ShellReceipt = z.infer<typeof ShellReceiptSchema>;
|
|
40
|
+
export declare const EVIDENCE_AVAILABILITIES: readonly ["available", "not-started", "capture-failed", "killed", "expired", "missing"];
|
|
41
|
+
export declare const EvidenceArtifactSchema: z.ZodObject<{
|
|
42
|
+
path: z.ZodString;
|
|
43
|
+
availability: z.ZodEnum<{
|
|
44
|
+
available: "available";
|
|
45
|
+
"not-started": "not-started";
|
|
46
|
+
"capture-failed": "capture-failed";
|
|
47
|
+
killed: "killed";
|
|
48
|
+
expired: "expired";
|
|
49
|
+
missing: "missing";
|
|
50
|
+
}>;
|
|
51
|
+
sha256: z.ZodNullable<z.ZodString>;
|
|
52
|
+
retainedBytes: z.ZodNumber;
|
|
53
|
+
droppedBytes: z.ZodNumber;
|
|
54
|
+
truncated: z.ZodBoolean;
|
|
55
|
+
}, z.core.$strict>;
|
|
56
|
+
export declare const GateEvidenceReceiptSchema: z.ZodObject<{
|
|
57
|
+
invocationId: z.ZodString;
|
|
58
|
+
nonce: z.ZodOptional<z.ZodString>;
|
|
59
|
+
subject: z.ZodObject<{
|
|
60
|
+
runId: z.ZodString;
|
|
61
|
+
taskId: z.ZodNullable<z.ZodString>;
|
|
62
|
+
attempt: z.ZodNullable<z.ZodNumber>;
|
|
63
|
+
gate: z.ZodString;
|
|
64
|
+
subjectCommit: z.ZodNullable<z.ZodString>;
|
|
65
|
+
}, z.core.$strict>;
|
|
66
|
+
termination: z.ZodObject<{
|
|
67
|
+
kind: z.ZodEnum<{
|
|
68
|
+
unknown: "unknown";
|
|
69
|
+
signal: "signal";
|
|
70
|
+
"not-started": "not-started";
|
|
71
|
+
exit: "exit";
|
|
72
|
+
timeout: "timeout";
|
|
73
|
+
}>;
|
|
74
|
+
exitCode: z.ZodNullable<z.ZodNumber>;
|
|
75
|
+
signal: z.ZodNullable<z.ZodString>;
|
|
76
|
+
timedOut: z.ZodNullable<z.ZodBoolean>;
|
|
77
|
+
}, z.core.$strict>;
|
|
78
|
+
availability: z.ZodEnum<{
|
|
79
|
+
available: "available";
|
|
80
|
+
"not-started": "not-started";
|
|
81
|
+
"capture-failed": "capture-failed";
|
|
82
|
+
killed: "killed";
|
|
83
|
+
expired: "expired";
|
|
84
|
+
missing: "missing";
|
|
85
|
+
}>;
|
|
86
|
+
redaction: z.ZodObject<{
|
|
87
|
+
material: z.ZodBoolean;
|
|
88
|
+
}, z.core.$strict>;
|
|
89
|
+
stdout: z.ZodObject<{
|
|
90
|
+
path: z.ZodString;
|
|
91
|
+
availability: z.ZodEnum<{
|
|
92
|
+
available: "available";
|
|
93
|
+
"not-started": "not-started";
|
|
94
|
+
"capture-failed": "capture-failed";
|
|
95
|
+
killed: "killed";
|
|
96
|
+
expired: "expired";
|
|
97
|
+
missing: "missing";
|
|
98
|
+
}>;
|
|
99
|
+
sha256: z.ZodNullable<z.ZodString>;
|
|
100
|
+
retainedBytes: z.ZodNumber;
|
|
101
|
+
droppedBytes: z.ZodNumber;
|
|
102
|
+
truncated: z.ZodBoolean;
|
|
103
|
+
}, z.core.$strict>;
|
|
104
|
+
stderr: z.ZodObject<{
|
|
105
|
+
path: z.ZodString;
|
|
106
|
+
availability: z.ZodEnum<{
|
|
107
|
+
available: "available";
|
|
108
|
+
"not-started": "not-started";
|
|
109
|
+
"capture-failed": "capture-failed";
|
|
110
|
+
killed: "killed";
|
|
111
|
+
expired: "expired";
|
|
112
|
+
missing: "missing";
|
|
113
|
+
}>;
|
|
114
|
+
sha256: z.ZodNullable<z.ZodString>;
|
|
115
|
+
retainedBytes: z.ZodNumber;
|
|
116
|
+
droppedBytes: z.ZodNumber;
|
|
117
|
+
truncated: z.ZodBoolean;
|
|
118
|
+
}, z.core.$strict>;
|
|
119
|
+
}, z.core.$strict>;
|
|
120
|
+
export type GateEvidenceReceipt = z.infer<typeof GateEvidenceReceiptSchema>;
|
|
121
|
+
export type EvidenceArtifact = z.infer<typeof EvidenceArtifactSchema>;
|
|
40
122
|
/** A task-build row must carry the entire caller-owned correlation, even before spawn. */
|
|
41
123
|
export declare const CommandReceiptSchema: z.ZodObject<{
|
|
42
124
|
outcome: z.ZodEnum<{
|