tickmarkr 1.69.0 → 1.71.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/kimi.js +12 -35
- package/dist/adapters/registry.d.ts +25 -1
- package/dist/adapters/registry.js +95 -4
- package/dist/adapters/types.d.ts +8 -0
- package/dist/cli/commands/doctor.js +8 -3
- package/dist/cli/commands/report.js +40 -5
- package/dist/cli/commands/resume.js +7 -1
- package/dist/cli/commands/status.js +20 -8
- package/dist/drivers/herdr.d.ts +5 -0
- package/dist/drivers/herdr.js +22 -3
- package/dist/gates/acceptance.d.ts +5 -1
- package/dist/gates/acceptance.js +72 -10
- package/dist/gates/review.d.ts +10 -2
- package/dist/gates/review.js +72 -19
- package/dist/report/bundle.d.ts +45 -0
- package/dist/report/bundle.js +158 -0
- package/dist/report/compare.d.ts +64 -0
- package/dist/report/compare.js +231 -0
- package/dist/route/preference.d.ts +9 -0
- package/dist/route/preference.js +61 -0
- package/dist/run/daemon.d.ts +4 -0
- package/dist/run/daemon.js +94 -11
- package/dist/run/environment.d.ts +12 -0
- package/dist/run/environment.js +41 -0
- package/dist/run/interactive-seed.d.ts +1 -0
- package/dist/run/interactive-seed.js +12 -4
- package/dist/run/journal.d.ts +9 -1
- package/dist/run/journal.js +53 -2
- package/package.json +1 -1
package/dist/gates/acceptance.js
CHANGED
|
@@ -7,13 +7,14 @@ import { checkDiffCap, fetchTaskDiff } from "./review.js";
|
|
|
7
7
|
import { COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
|
|
8
8
|
// Fable F4: acceptance judge shares review's 900s timeout — 300s default killed frontier judges on cap-sized diffs.
|
|
9
9
|
const JUDGE_TIMEOUT_MS = 900_000;
|
|
10
|
+
const CitationSchema = z.object({ path: z.string(), line: z.number().int() });
|
|
10
11
|
const JudgeVerdictRowSchema = z.object({
|
|
11
12
|
criterion: z.string(),
|
|
12
13
|
met: z.boolean(),
|
|
13
14
|
reason: z.string(),
|
|
14
|
-
//
|
|
15
|
-
//
|
|
16
|
-
evidence: z.string(),
|
|
15
|
+
// Required, either form: a row omitting evidence is malformed and fails closed like any other shape
|
|
16
|
+
// violation. Object → structured citation; string → legacy quote.
|
|
17
|
+
evidence: z.union([CitationSchema, z.string()]),
|
|
17
18
|
});
|
|
18
19
|
const JudgeVerdictSchema = z.object({
|
|
19
20
|
pass: z.boolean(),
|
|
@@ -103,6 +104,64 @@ function tail(out, n = 8) {
|
|
|
103
104
|
return "";
|
|
104
105
|
return "\n" + t.split("\n").slice(-n).join("\n");
|
|
105
106
|
}
|
|
107
|
+
// v1.70: the new-file line numbers each hunk actually ADDS, per changed path — parsed from the
|
|
108
|
+
// unified-diff hunk headers so a citation is validated against real changed locations, not a substring
|
|
109
|
+
// of the whole diff text (which also matches unchanged context lines and the diff's own +++/@@ headers).
|
|
110
|
+
// Only `+` lines are changed locations; context lines advance the new-file counter but are not changes.
|
|
111
|
+
// ponytail: assumes git's default a/ b/ prefixes (fetchTaskDiff uses plain `git diff`); revisit if a
|
|
112
|
+
// caller passes a --no-prefix diff.
|
|
113
|
+
function changedLinesByFile(diff) {
|
|
114
|
+
const byFile = new Map();
|
|
115
|
+
let path = null;
|
|
116
|
+
let newLine = 0;
|
|
117
|
+
let inHunk = false;
|
|
118
|
+
for (const raw of diff.split("\n")) {
|
|
119
|
+
if (raw.startsWith("diff --git")) {
|
|
120
|
+
inHunk = false;
|
|
121
|
+
path = null;
|
|
122
|
+
continue;
|
|
123
|
+
}
|
|
124
|
+
if (!inHunk && raw.startsWith("--- "))
|
|
125
|
+
continue;
|
|
126
|
+
if (!inHunk && raw.startsWith("+++ ")) {
|
|
127
|
+
const p = raw.slice(4).trim();
|
|
128
|
+
path = p === "/dev/null" ? null : p.replace(/^[ab]\//, "");
|
|
129
|
+
continue;
|
|
130
|
+
}
|
|
131
|
+
const hunk = /^@@ -\d+(?:,\d+)? \+(\d+)(?:,\d+)? @@/.exec(raw);
|
|
132
|
+
if (hunk) {
|
|
133
|
+
newLine = Number(hunk[1]);
|
|
134
|
+
inHunk = true;
|
|
135
|
+
continue;
|
|
136
|
+
}
|
|
137
|
+
if (!inHunk || path === null)
|
|
138
|
+
continue;
|
|
139
|
+
const c = raw[0];
|
|
140
|
+
if (c === "+") {
|
|
141
|
+
let set = byFile.get(path);
|
|
142
|
+
if (!set) {
|
|
143
|
+
set = new Set();
|
|
144
|
+
byFile.set(path, set);
|
|
145
|
+
}
|
|
146
|
+
set.add(newLine);
|
|
147
|
+
newLine++;
|
|
148
|
+
}
|
|
149
|
+
else if (c === " ") {
|
|
150
|
+
newLine++;
|
|
151
|
+
}
|
|
152
|
+
else if (c !== "-") {
|
|
153
|
+
inHunk = false; // "\ No newline", a trailing blank, or the next section — hunk body ended
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
return byFile;
|
|
157
|
+
}
|
|
158
|
+
// A citation is valid evidence iff the diff adds the cited line of the cited file. A legacy free-text
|
|
159
|
+
// quote (string) keeps v1.64's substring check — non-empty and present somewhere in the diff.
|
|
160
|
+
function citesChangedLocation(evidence, changed, diff) {
|
|
161
|
+
if (typeof evidence === "string")
|
|
162
|
+
return evidence.trim().length > 0 && diff.includes(evidence);
|
|
163
|
+
return changed.get(evidence.path)?.has(evidence.line) ?? false;
|
|
164
|
+
}
|
|
106
165
|
export async function acceptanceGate(task, worktree, baseRef, judge, via, opts = {}) {
|
|
107
166
|
// 1. deterministic oracles — exit code decides, fail-closed, zero LLM calls (spec §2, T2).
|
|
108
167
|
// A failure returns here, before any runLlm() call: a judge can never override it.
|
|
@@ -176,9 +235,9 @@ ${diff}
|
|
|
176
235
|
${verdictNonceLine(nonce)}
|
|
177
236
|
|
|
178
237
|
Respond with ONLY this JSON (no prose before or after):
|
|
179
|
-
{"nonce": "${nonce}", "pass": true|false, "criteria": [{"criterion": "c1", "met": true|false, "reason": "...", "evidence": "
|
|
238
|
+
{"nonce": "${nonce}", "pass": true|false, "criteria": [{"criterion": "c1", "met": true|false, "reason": "...", "evidence": {"path": "path/to/file", "line": 42}}]}
|
|
180
239
|
Each criteria[].criterion MUST be the stable id from the rubric (c1, c2, ...) exactly once.
|
|
181
|
-
Each criteria[].evidence MUST be a
|
|
240
|
+
Each criteria[].evidence MUST be a structured citation {"path", "line"} pointing at a line the diff above actually adds or changes (the new-file line number); a citation to an unchanged or nonexistent location voids the whole verdict.
|
|
182
241
|
`;
|
|
183
242
|
const raw = await runLlm(judge.adapter, judge.model, prompt, worktree, via, JUDGE_TIMEOUT_MS);
|
|
184
243
|
const extracted = extractVerdictJson(raw, nonce);
|
|
@@ -191,13 +250,16 @@ Each criteria[].evidence MUST be a short verbatim quote copied from the diff abo
|
|
|
191
250
|
meta: { unparseable: true, judge: channelKey({ adapter: judge.adapter.id, model: judge.model }) } };
|
|
192
251
|
}
|
|
193
252
|
const { verdict: v, inconsistencies } = checkJudgeVerdict(extracted, expectedIds);
|
|
194
|
-
// v1.
|
|
195
|
-
// above, never the worktree or any other artifact.
|
|
196
|
-
//
|
|
197
|
-
|
|
253
|
+
// v1.70: each citation must point at a line the diff actually changed — validated against the hunks
|
|
254
|
+
// of `diff` (the exact string embedded in the prompt above), never the worktree or any other artifact.
|
|
255
|
+
// A citation to an untouched file or a line outside every changed hunk is a hallucinated verdict:
|
|
256
|
+
// treated as unparseable so GATE-09 retries the judge on a failover channel. (A legacy free-text quote
|
|
257
|
+
// still validates by substring, for the fake seam and pre-v1.70 fixtures.)
|
|
258
|
+
const changed = changedLinesByFile(diff);
|
|
259
|
+
const fabricated = v.criteria.filter((row) => !citesChangedLocation(row.evidence, changed, diff));
|
|
198
260
|
if (fabricated.length) {
|
|
199
261
|
return { gate: "acceptance", pass: false,
|
|
200
|
-
details: warn + detBlock + `judge verdict
|
|
262
|
+
details: warn + detBlock + `judge verdict cites evidence absent from the judged diff (${fabricated.map((row) => row.criterion).join(", ")}) — treating as unparseable, failing closed`,
|
|
201
263
|
meta: { unparseable: true, judge: channelKey({ adapter: judge.adapter.id, model: judge.model }) } };
|
|
202
264
|
}
|
|
203
265
|
const pass = v.pass === true && inconsistencies.length === 0 && v.criteria.every((row) => row.met);
|
package/dist/gates/review.d.ts
CHANGED
|
@@ -3,9 +3,17 @@ import { type TickmarkrConfig } from "../config/config.js";
|
|
|
3
3
|
import { type Task } from "../graph/schema.js";
|
|
4
4
|
import { type GateVia } from "./llm.js";
|
|
5
5
|
import type { GateResult } from "./types.js";
|
|
6
|
+
export type ReviewSeverity = "material" | "minor";
|
|
7
|
+
export interface ReviewFinding {
|
|
8
|
+
note: string;
|
|
9
|
+
severity: ReviewSeverity;
|
|
10
|
+
defer?: boolean;
|
|
11
|
+
rationale?: string;
|
|
12
|
+
}
|
|
6
13
|
export interface ReviewVerdict {
|
|
7
|
-
approve
|
|
8
|
-
issues
|
|
14
|
+
approve?: boolean;
|
|
15
|
+
issues?: string[];
|
|
16
|
+
findings?: ReviewFinding[];
|
|
9
17
|
}
|
|
10
18
|
export declare function fetchTaskDiff(worktree: string, baseRef: string): Promise<{
|
|
11
19
|
full: string;
|
package/dist/gates/review.js
CHANGED
|
@@ -5,6 +5,63 @@ import { getAdapter } from "../adapters/registry.js";
|
|
|
5
5
|
import { shOk } from "../run/git.js";
|
|
6
6
|
import { marginalCostRank } from "../route/router.js";
|
|
7
7
|
import { COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
|
|
8
|
+
// legacy flat `issues` shape — every issue blocks; the approve flag must agree with the list.
|
|
9
|
+
function classifyReviewIssues(approve, issues) {
|
|
10
|
+
const inconsistencies = [];
|
|
11
|
+
issues.forEach((issue, i) => {
|
|
12
|
+
if (typeof issue !== "string")
|
|
13
|
+
inconsistencies.push(`review verdict inconsistent: issues[${i}] must be a string`);
|
|
14
|
+
});
|
|
15
|
+
if (approve && issues.length) {
|
|
16
|
+
inconsistencies.push("review verdict inconsistent: approve=true requires issues to be empty");
|
|
17
|
+
}
|
|
18
|
+
else if (!approve && !issues.length) {
|
|
19
|
+
inconsistencies.push("review verdict inconsistent: approve=false requires at least one issue");
|
|
20
|
+
}
|
|
21
|
+
const pass = approve === true && inconsistencies.length === 0;
|
|
22
|
+
const lines = issues.map((issue) => `- ${typeof issue === "string" ? issue : JSON.stringify(issue)}`);
|
|
23
|
+
lines.push(...inconsistencies);
|
|
24
|
+
return { pass, headline: pass ? "approved" : approve ? "approval rejected" : "requested changes", lines };
|
|
25
|
+
}
|
|
26
|
+
// v1.70 T5: classified findings — only material (non-deferred) findings block approval. Deferred
|
|
27
|
+
// findings carry their rationale into the details (never dropped). Malformed rows fail closed like any
|
|
28
|
+
// other shape violation, so a garbage "findings" array can never fake an approval.
|
|
29
|
+
function classifyReviewFindings(findings) {
|
|
30
|
+
const inconsistencies = [];
|
|
31
|
+
const lines = [];
|
|
32
|
+
let material = 0;
|
|
33
|
+
let deferred = 0;
|
|
34
|
+
findings.forEach((f, i) => {
|
|
35
|
+
if (!f || typeof f !== "object") {
|
|
36
|
+
inconsistencies.push(`review finding inconsistent: findings[${i}] must be an object`);
|
|
37
|
+
return;
|
|
38
|
+
}
|
|
39
|
+
const { note, severity, defer, rationale } = f;
|
|
40
|
+
if (typeof note !== "string")
|
|
41
|
+
inconsistencies.push(`review finding inconsistent: findings[${i}].note must be a string`);
|
|
42
|
+
if (severity !== "material" && severity !== "minor")
|
|
43
|
+
inconsistencies.push(`review finding inconsistent: findings[${i}].severity must be "material" or "minor"`);
|
|
44
|
+
if (defer !== undefined && typeof defer !== "boolean")
|
|
45
|
+
inconsistencies.push(`review finding inconsistent: findings[${i}].defer must be a boolean`);
|
|
46
|
+
const isDeferred = defer === true;
|
|
47
|
+
if (isDeferred && (typeof rationale !== "string" || !rationale.trim())) {
|
|
48
|
+
inconsistencies.push(`review finding inconsistent: deferred findings[${i}] requires a rationale`);
|
|
49
|
+
}
|
|
50
|
+
if (severity === "material" && !isDeferred)
|
|
51
|
+
material++;
|
|
52
|
+
if (isDeferred)
|
|
53
|
+
deferred++;
|
|
54
|
+
const label = isDeferred ? `deferred/${severity ?? "?"}` : String(severity ?? "?");
|
|
55
|
+
const why = isDeferred && typeof rationale === "string" ? ` — rationale: ${rationale}` : "";
|
|
56
|
+
lines.push(`- [${label}] ${typeof note === "string" ? note : JSON.stringify(note)}${why}`);
|
|
57
|
+
});
|
|
58
|
+
const pass = material === 0 && inconsistencies.length === 0;
|
|
59
|
+
lines.push(...inconsistencies);
|
|
60
|
+
const headline = pass
|
|
61
|
+
? deferred ? `approved (${deferred} deferred)` : "approved"
|
|
62
|
+
: `requested changes (${material} material)`;
|
|
63
|
+
return { pass, headline, lines };
|
|
64
|
+
}
|
|
8
65
|
// OBS-48: cap on zero-context diff bytes (git diff -U0), not context-padded full diff — scattered
|
|
9
66
|
// one-line hunks no longer trip at ~370 diff-bytes per changed line. Full diff still goes to the judge.
|
|
10
67
|
const DIFF_CAP_REMEDY = "split the task, or raise gates.diffCap";
|
|
@@ -95,8 +152,14 @@ ${diff}
|
|
|
95
152
|
|
|
96
153
|
${verdictNonceLine(nonce)}
|
|
97
154
|
|
|
155
|
+
Classify every concern as "material" (a correctness, security, or acceptance-criteria defect that must
|
|
156
|
+
block the merge) or "minor" (style, naming, or preference that should not block). ONLY material findings
|
|
157
|
+
block approval. For a minor concern you have decided not to block on, set "defer": true and give a
|
|
158
|
+
one-line "rationale" — it is recorded in the review, never dropped.
|
|
159
|
+
|
|
98
160
|
Respond with ONLY this JSON:
|
|
99
|
-
{"nonce": "${nonce}", "approve": true|false, "
|
|
161
|
+
{"nonce": "${nonce}", "approve": true|false, "findings": [{"note": "...", "severity": "material"|"minor", "defer": false, "rationale": ""}]}
|
|
162
|
+
Approve iff no material finding remains; an empty findings list is a clean approval.
|
|
100
163
|
`;
|
|
101
164
|
const raw = await runLlm(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? { driver: via.driver, keep: via.keep, onSlot: via.onSlot, name: via.nameFor("review", reviewer.adapter), label: via.labelFor("review") } : undefined,
|
|
102
165
|
// frontier reviewers routinely need >5min on a configured-cap-sized diff, and `claude -p` buffers all
|
|
@@ -106,7 +169,9 @@ Respond with ONLY this JSON:
|
|
|
106
169
|
// second knob-turner appears.
|
|
107
170
|
900_000);
|
|
108
171
|
const v = extractVerdictJson(raw, nonce);
|
|
109
|
-
|
|
172
|
+
const findings = v && Array.isArray(v.findings) ? v.findings : null;
|
|
173
|
+
// findings decides the verdict on its own; the legacy path still needs approve + issues to parse.
|
|
174
|
+
if (!v || (findings === null && (typeof v.approve !== "boolean" || !Array.isArray(v.issues)))) {
|
|
110
175
|
return {
|
|
111
176
|
gate: "review",
|
|
112
177
|
pass: false,
|
|
@@ -114,25 +179,13 @@ Respond with ONLY this JSON:
|
|
|
114
179
|
meta: { reviewer: channelKey(reviewer) },
|
|
115
180
|
};
|
|
116
181
|
}
|
|
117
|
-
const
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
if (typeof issue !== "string")
|
|
121
|
-
inconsistencies.push(`review verdict inconsistent: issues[${i}] must be a string`);
|
|
122
|
-
});
|
|
123
|
-
if (v.approve && issues.length) {
|
|
124
|
-
inconsistencies.push("review verdict inconsistent: approve=true requires issues to be empty");
|
|
125
|
-
}
|
|
126
|
-
else if (!v.approve && !issues.length) {
|
|
127
|
-
inconsistencies.push("review verdict inconsistent: approve=false requires at least one issue");
|
|
128
|
-
}
|
|
129
|
-
const pass = v.approve === true && inconsistencies.length === 0;
|
|
130
|
-
const lines = issues.map((issue) => `- ${typeof issue === "string" ? issue : JSON.stringify(issue)}`);
|
|
131
|
-
lines.push(...inconsistencies);
|
|
182
|
+
const decided = findings !== null
|
|
183
|
+
? classifyReviewFindings(findings)
|
|
184
|
+
: classifyReviewIssues(v.approve, v.issues);
|
|
132
185
|
return {
|
|
133
186
|
gate: "review",
|
|
134
|
-
pass,
|
|
135
|
-
details: `reviewer ${reviewer.adapter}:${reviewer.model} (${reviewer.vendor}): ${
|
|
187
|
+
pass: decided.pass,
|
|
188
|
+
details: `reviewer ${reviewer.adapter}:${reviewer.model} (${reviewer.vendor}): ${decided.headline}${decided.lines.length ? "\n" + decided.lines.join("\n") : ""}`,
|
|
136
189
|
meta: { reviewer: channelKey(reviewer) },
|
|
137
190
|
};
|
|
138
191
|
}
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
import type { EvidenceCitation } from "../gates/acceptance.js";
|
|
2
|
+
import type { RunEnvironment } from "../run/environment.js";
|
|
3
|
+
import type { JournalEvent } from "../run/journal.js";
|
|
4
|
+
export declare const BUNDLE_SCHEMA_VERSION = 1;
|
|
5
|
+
export declare const KNOWN_LIMITS: readonly string[];
|
|
6
|
+
export type BundleEvidence = string | EvidenceCitation;
|
|
7
|
+
export interface BundleJudgeCriterion {
|
|
8
|
+
criterion: string;
|
|
9
|
+
met: boolean;
|
|
10
|
+
reason: string;
|
|
11
|
+
evidence: BundleEvidence;
|
|
12
|
+
}
|
|
13
|
+
export interface BundleGateResult {
|
|
14
|
+
gate: string;
|
|
15
|
+
pass: boolean;
|
|
16
|
+
details: string;
|
|
17
|
+
}
|
|
18
|
+
export interface BundleTask {
|
|
19
|
+
taskId: string;
|
|
20
|
+
outcome: "done" | "failed" | "human" | "not-recorded";
|
|
21
|
+
gates: BundleGateResult[];
|
|
22
|
+
/** Judge criteria with evidence citations, when the journal recorded them structured. */
|
|
23
|
+
judgeCriteria: BundleJudgeCriterion[];
|
|
24
|
+
}
|
|
25
|
+
export interface BundleContentHashes {
|
|
26
|
+
/** sha256 of the canonical JSON of the source journal events used to build this packet. */
|
|
27
|
+
journal: string;
|
|
28
|
+
/** graphDefinitionHash from run-start when recorded. */
|
|
29
|
+
graphDefinitionHash?: string;
|
|
30
|
+
}
|
|
31
|
+
export interface ProofBundle {
|
|
32
|
+
/** Schema version a future reader must check before parsing the rest. */
|
|
33
|
+
schemaVersion: number;
|
|
34
|
+
runId: string;
|
|
35
|
+
environment?: RunEnvironment;
|
|
36
|
+
contentHashes: BundleContentHashes;
|
|
37
|
+
tasks: BundleTask[];
|
|
38
|
+
/** Plain-language known limits — this is not an unconditional proof of correctness. */
|
|
39
|
+
knownLimits: string[];
|
|
40
|
+
}
|
|
41
|
+
/**
|
|
42
|
+
* Build a schema-versioned proof packet from a run's journal events.
|
|
43
|
+
* Pure: no filesystem, no network. Caller writes the result.
|
|
44
|
+
*/
|
|
45
|
+
export declare function buildProofBundle(runId: string, events: JournalEvent[]): ProofBundle;
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
// v1.70 T4: pure proof-packet builder for `tickmarkr report --bundle <path>`.
|
|
2
|
+
// One portable, schema-versioned snapshot of a run — task outcomes, judge evidence
|
|
3
|
+
// citations, environment identity, content hashes — secrets redacted via the shared
|
|
4
|
+
// redactSecrets seam. No I/O and no network: the caller (report CLI) owns the write.
|
|
5
|
+
import { createHash } from "node:crypto";
|
|
6
|
+
import { redactSecrets } from "../run/redact.js";
|
|
7
|
+
import { recordedEnvironment } from "./compare.js";
|
|
8
|
+
// Bump when the packet shape changes in a way a future reader must branch on before parsing.
|
|
9
|
+
export const BUNDLE_SCHEMA_VERSION = 1;
|
|
10
|
+
// Plain-language known limits — the packet is a journal snapshot, not an unconditional proof.
|
|
11
|
+
export const KNOWN_LIMITS = [
|
|
12
|
+
"This packet is a portable snapshot of journaled run facts, not an independent re-verification of the work or its gates.",
|
|
13
|
+
"Judge evidence citations are copied from the journal as the judge recorded them; this packet does not re-validate citations against the judged diff.",
|
|
14
|
+
"Content hashes bind this packet to the journal bytes used to build it; they do not prove the underlying gates, merges, or tip verification were correct.",
|
|
15
|
+
"Secret-shaped strings are redacted by local pattern matching only; redaction is not a formal security audit.",
|
|
16
|
+
"Producing this packet never contacts a network and never uploads anything.",
|
|
17
|
+
];
|
|
18
|
+
function contentHash(events) {
|
|
19
|
+
return createHash("sha256").update(JSON.stringify(events)).digest("hex");
|
|
20
|
+
}
|
|
21
|
+
function graphHash(events) {
|
|
22
|
+
for (const e of events) {
|
|
23
|
+
if (e.event !== "run-start")
|
|
24
|
+
continue;
|
|
25
|
+
const h = e.data.graphDefinitionHash;
|
|
26
|
+
return typeof h === "string" ? h : undefined;
|
|
27
|
+
}
|
|
28
|
+
return undefined;
|
|
29
|
+
}
|
|
30
|
+
function taskIds(events) {
|
|
31
|
+
const seen = new Set();
|
|
32
|
+
const out = [];
|
|
33
|
+
for (const e of events) {
|
|
34
|
+
if (!e.taskId || seen.has(e.taskId))
|
|
35
|
+
continue;
|
|
36
|
+
seen.add(e.taskId);
|
|
37
|
+
out.push(e.taskId);
|
|
38
|
+
}
|
|
39
|
+
return out;
|
|
40
|
+
}
|
|
41
|
+
function outcomeFor(events, taskId, runEnd) {
|
|
42
|
+
for (let i = events.length - 1; i >= 0; i--) {
|
|
43
|
+
const e = events[i];
|
|
44
|
+
if (e.taskId !== taskId)
|
|
45
|
+
continue;
|
|
46
|
+
if (e.event === "task-done")
|
|
47
|
+
return "done";
|
|
48
|
+
if (e.event === "task-failed")
|
|
49
|
+
return "failed";
|
|
50
|
+
if (e.event === "task-human")
|
|
51
|
+
return "human";
|
|
52
|
+
}
|
|
53
|
+
if (runEnd) {
|
|
54
|
+
const d = runEnd.data;
|
|
55
|
+
if (Array.isArray(d.done) && d.done.includes(taskId))
|
|
56
|
+
return "done";
|
|
57
|
+
if (Array.isArray(d.failed) && d.failed.includes(taskId))
|
|
58
|
+
return "failed";
|
|
59
|
+
if (Array.isArray(d.human) && d.human.includes(taskId))
|
|
60
|
+
return "human";
|
|
61
|
+
}
|
|
62
|
+
return "not-recorded";
|
|
63
|
+
}
|
|
64
|
+
// Parse a structured evidence citation without rewriting path/line — unaltered copy.
|
|
65
|
+
function parseEvidence(raw) {
|
|
66
|
+
if (typeof raw === "string")
|
|
67
|
+
return raw;
|
|
68
|
+
if (!raw || typeof raw !== "object" || Array.isArray(raw))
|
|
69
|
+
return undefined;
|
|
70
|
+
const o = raw;
|
|
71
|
+
if (typeof o.path === "string" && typeof o.line === "number" && Number.isInteger(o.line)) {
|
|
72
|
+
return { path: o.path, line: o.line };
|
|
73
|
+
}
|
|
74
|
+
return undefined;
|
|
75
|
+
}
|
|
76
|
+
// Structured judge criteria on an acceptance gate-result (when journaled). Evidence is copied
|
|
77
|
+
// unaltered so a future reader can match citations byte-for-byte to what the judge recorded.
|
|
78
|
+
function parseJudgeCriteria(data) {
|
|
79
|
+
const raw = data.criteria;
|
|
80
|
+
if (!Array.isArray(raw))
|
|
81
|
+
return [];
|
|
82
|
+
const out = [];
|
|
83
|
+
for (const row of raw) {
|
|
84
|
+
if (!row || typeof row !== "object" || Array.isArray(row))
|
|
85
|
+
continue;
|
|
86
|
+
const r = row;
|
|
87
|
+
if (typeof r.criterion !== "string" || typeof r.met !== "boolean" || typeof r.reason !== "string")
|
|
88
|
+
continue;
|
|
89
|
+
const evidence = parseEvidence(r.evidence);
|
|
90
|
+
if (evidence === undefined)
|
|
91
|
+
continue;
|
|
92
|
+
out.push({ criterion: r.criterion, met: r.met, reason: r.reason, evidence });
|
|
93
|
+
}
|
|
94
|
+
return out;
|
|
95
|
+
}
|
|
96
|
+
// Deep-walk string leaves through redactSecrets. Numbers/booleans/null stay as-is.
|
|
97
|
+
// Structured evidence citations keep path/line unaltered (redact only free-text string evidence).
|
|
98
|
+
function redactValue(v) {
|
|
99
|
+
if (typeof v === "string")
|
|
100
|
+
return redactSecrets(v);
|
|
101
|
+
if (v === null || typeof v !== "object")
|
|
102
|
+
return v;
|
|
103
|
+
if (Array.isArray(v))
|
|
104
|
+
return v.map(redactValue);
|
|
105
|
+
const o = v;
|
|
106
|
+
// Preserve structured {path, line} citations unaltered (criterion "unaltered").
|
|
107
|
+
if (typeof o.path === "string" && typeof o.line === "number" && Object.keys(o).length === 2) {
|
|
108
|
+
return { path: o.path, line: o.line };
|
|
109
|
+
}
|
|
110
|
+
const out = {};
|
|
111
|
+
for (const [k, val] of Object.entries(o))
|
|
112
|
+
out[k] = redactValue(val);
|
|
113
|
+
return out;
|
|
114
|
+
}
|
|
115
|
+
/**
|
|
116
|
+
* Build a schema-versioned proof packet from a run's journal events.
|
|
117
|
+
* Pure: no filesystem, no network. Caller writes the result.
|
|
118
|
+
*/
|
|
119
|
+
export function buildProofBundle(runId, events) {
|
|
120
|
+
const runEnd = [...events].reverse().find((e) => e.event === "run-end");
|
|
121
|
+
const environment = recordedEnvironment(events);
|
|
122
|
+
const journalHash = contentHash(events);
|
|
123
|
+
const gHash = graphHash(events);
|
|
124
|
+
const tasks = taskIds(events).map((taskId) => {
|
|
125
|
+
const gateEvents = events.filter((e) => e.taskId === taskId && e.event === "gate-result");
|
|
126
|
+
const gates = [];
|
|
127
|
+
const judgeCriteria = [];
|
|
128
|
+
for (const g of gateEvents) {
|
|
129
|
+
const gate = typeof g.data.gate === "string" ? g.data.gate : "unknown";
|
|
130
|
+
const pass = g.data.pass === true;
|
|
131
|
+
const details = typeof g.data.details === "string" ? g.data.details : "";
|
|
132
|
+
gates.push({ gate, pass, details });
|
|
133
|
+
if (gate === "acceptance") {
|
|
134
|
+
for (const c of parseJudgeCriteria(g.data))
|
|
135
|
+
judgeCriteria.push(c);
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
return {
|
|
139
|
+
taskId,
|
|
140
|
+
outcome: outcomeFor(events, taskId, runEnd),
|
|
141
|
+
gates,
|
|
142
|
+
judgeCriteria,
|
|
143
|
+
};
|
|
144
|
+
});
|
|
145
|
+
const packet = {
|
|
146
|
+
schemaVersion: BUNDLE_SCHEMA_VERSION,
|
|
147
|
+
runId,
|
|
148
|
+
...(environment ? { environment } : {}),
|
|
149
|
+
contentHashes: {
|
|
150
|
+
journal: journalHash,
|
|
151
|
+
...(gHash ? { graphDefinitionHash: gHash } : {}),
|
|
152
|
+
},
|
|
153
|
+
tasks,
|
|
154
|
+
knownLimits: [...KNOWN_LIMITS],
|
|
155
|
+
};
|
|
156
|
+
// Redact secret-shaped strings anywhere in the packet. Structured {path,line} citations stay intact.
|
|
157
|
+
return redactValue(packet);
|
|
158
|
+
}
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
import type { RunEnvironment } from "../run/environment.js";
|
|
2
|
+
import type { JournalEvent, TelemetryRow } from "../run/journal.js";
|
|
3
|
+
import { type CostConfig } from "./cost.js";
|
|
4
|
+
export type EnvironmentCompare = {
|
|
5
|
+
comparable: true;
|
|
6
|
+
recorded: string;
|
|
7
|
+
} | {
|
|
8
|
+
comparable: false;
|
|
9
|
+
reason: "mismatch";
|
|
10
|
+
recorded: string;
|
|
11
|
+
} | {
|
|
12
|
+
comparable: false;
|
|
13
|
+
reason: "unbound";
|
|
14
|
+
};
|
|
15
|
+
export declare function hasRunStart(events: JournalEvent[]): boolean;
|
|
16
|
+
export declare function recordedEnvironment(events: JournalEvent[]): RunEnvironment | undefined;
|
|
17
|
+
export declare function environmentComparable(baselineEnv: RunEnvironment | undefined, currentEnv: RunEnvironment | undefined): EnvironmentCompare;
|
|
18
|
+
export interface RunMetrics {
|
|
19
|
+
durationMs: number | undefined;
|
|
20
|
+
gatePass: number;
|
|
21
|
+
gateFail: number;
|
|
22
|
+
gateTotal: number;
|
|
23
|
+
costUsd: number | undefined;
|
|
24
|
+
tokensTotal: number | undefined;
|
|
25
|
+
}
|
|
26
|
+
export declare function runMetrics(events: JournalEvent[], rows?: TelemetryRow[], cost?: CostConfig): RunMetrics;
|
|
27
|
+
export interface RunDelta {
|
|
28
|
+
durationMs: number | undefined;
|
|
29
|
+
gateFail: number;
|
|
30
|
+
costUsd: number | undefined;
|
|
31
|
+
tokensTotal: number | undefined;
|
|
32
|
+
}
|
|
33
|
+
export type CompareOutcome = {
|
|
34
|
+
ok: false;
|
|
35
|
+
reason: string;
|
|
36
|
+
} | {
|
|
37
|
+
ok: true;
|
|
38
|
+
runId: string;
|
|
39
|
+
baselineRunId: string;
|
|
40
|
+
comparability: EnvironmentCompare;
|
|
41
|
+
current: RunMetrics;
|
|
42
|
+
baseline: RunMetrics;
|
|
43
|
+
delta: RunDelta;
|
|
44
|
+
text: string;
|
|
45
|
+
};
|
|
46
|
+
export declare function renderComparison(opts: {
|
|
47
|
+
runId: string;
|
|
48
|
+
baselineRunId: string;
|
|
49
|
+
comparability: EnvironmentCompare;
|
|
50
|
+
baselineEnv: RunEnvironment | undefined;
|
|
51
|
+
currentEnv: RunEnvironment | undefined;
|
|
52
|
+
current: RunMetrics;
|
|
53
|
+
baseline: RunMetrics;
|
|
54
|
+
delta: RunDelta;
|
|
55
|
+
}): string;
|
|
56
|
+
export declare function compareRuns(opts: {
|
|
57
|
+
runId: string;
|
|
58
|
+
baselineRunId: string;
|
|
59
|
+
events: JournalEvent[];
|
|
60
|
+
baselineEvents: JournalEvent[];
|
|
61
|
+
rows?: TelemetryRow[];
|
|
62
|
+
baselineRows?: TelemetryRow[];
|
|
63
|
+
cost?: CostConfig;
|
|
64
|
+
}): CompareOutcome;
|