tickmarkr 1.69.0 → 1.70.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/commands/report.js +40 -5
- package/dist/gates/acceptance.d.ts +5 -1
- package/dist/gates/acceptance.js +72 -10
- package/dist/gates/review.d.ts +10 -2
- package/dist/gates/review.js +72 -19
- package/dist/report/bundle.d.ts +45 -0
- package/dist/report/bundle.js +158 -0
- package/dist/report/compare.d.ts +64 -0
- package/dist/report/compare.js +231 -0
- package/dist/run/daemon.js +27 -1
- package/dist/run/environment.d.ts +12 -0
- package/dist/run/environment.js +41 -0
- package/package.json +1 -1
|
@@ -1,8 +1,11 @@
|
|
|
1
|
+
import { writeFileSync } from "node:fs";
|
|
1
2
|
import { parseArgs } from "node:util";
|
|
2
3
|
import { ttyVisual } from "../../adapters/model-lints.js";
|
|
3
4
|
import { addUsage } from "../../adapters/types.js";
|
|
4
5
|
import { dim, rule, title } from "../../brand.js";
|
|
5
6
|
import { loadConfig } from "../../config/config.js";
|
|
7
|
+
import { buildProofBundle } from "../../report/bundle.js";
|
|
8
|
+
import { compareRuns } from "../../report/compare.js";
|
|
6
9
|
import { estimateCosts } from "../../report/cost.js";
|
|
7
10
|
import { cellsOf, cellSummary } from "../../route/profile.js";
|
|
8
11
|
import { Journal, loadRoutingProfile } from "../../run/journal.js";
|
|
@@ -316,17 +319,49 @@ const stylizeReport = (out) => {
|
|
|
316
319
|
export async function report(argv, cwd = process.cwd()) {
|
|
317
320
|
const { values, positionals } = parseArgs({
|
|
318
321
|
args: argv,
|
|
319
|
-
options: {
|
|
322
|
+
options: {
|
|
323
|
+
md: { type: "boolean" },
|
|
324
|
+
// v1.70 T3: baseline run id for cost/gate/duration delta + environment comparability guard
|
|
325
|
+
compare: { type: "string" },
|
|
326
|
+
// v1.70 T4: write a portable, schema-versioned proof packet (local file only — no network)
|
|
327
|
+
bundle: { type: "string" },
|
|
328
|
+
},
|
|
320
329
|
allowPositionals: true,
|
|
321
330
|
});
|
|
322
331
|
const runId = positionals[0] ?? Journal.latestRunId(cwd, { withJournal: true });
|
|
323
332
|
if (!runId)
|
|
324
|
-
throw new Error("no runs found — usage: tickmarkr report <run-id> [--md]");
|
|
333
|
+
throw new Error("no runs found — usage: tickmarkr report <run-id> [--md] [--compare <baseline-run-id>] [--bundle <path>]");
|
|
325
334
|
const j = Journal.open(cwd, runId);
|
|
326
335
|
const events = j.read();
|
|
336
|
+
const rows = j.readTelemetry();
|
|
337
|
+
const cfg = loadConfig(cwd);
|
|
338
|
+
let bundleNote = "";
|
|
339
|
+
if (values.bundle) {
|
|
340
|
+
// Local-only write of the pure proof packet — no network path exists in buildProofBundle.
|
|
341
|
+
const packet = buildProofBundle(runId, events);
|
|
342
|
+
writeFileSync(values.bundle, JSON.stringify(packet, null, 2) + "\n");
|
|
343
|
+
bundleNote = `wrote proof bundle → ${values.bundle}\n`;
|
|
344
|
+
}
|
|
345
|
+
let comparison = "";
|
|
346
|
+
if (values.compare) {
|
|
347
|
+
const baselineRunId = values.compare;
|
|
348
|
+
const baseline = Journal.open(cwd, baselineRunId);
|
|
349
|
+
const outcome = compareRuns({
|
|
350
|
+
runId,
|
|
351
|
+
baselineRunId,
|
|
352
|
+
events,
|
|
353
|
+
baselineEvents: baseline.read(),
|
|
354
|
+
rows,
|
|
355
|
+
baselineRows: baseline.readTelemetry(),
|
|
356
|
+
cost: cfg.cost,
|
|
357
|
+
});
|
|
358
|
+
// Fail closed: missing run-start yields a clear reason, never a partial table.
|
|
359
|
+
if (!outcome.ok)
|
|
360
|
+
throw new Error(outcome.reason);
|
|
361
|
+
comparison = "\n" + outcome.text;
|
|
362
|
+
}
|
|
327
363
|
if (values.md) {
|
|
328
|
-
|
|
329
|
-
return renderMarkdownRecord(runId, events, estimateCosts(rows, loadConfig(cwd).cost), rows);
|
|
364
|
+
return bundleNote + renderMarkdownRecord(runId, events, estimateCosts(rows, cfg.cost), rows) + comparison;
|
|
330
365
|
}
|
|
331
|
-
return stylizeReport(textReport(runId, events,
|
|
366
|
+
return bundleNote + stylizeReport(textReport(runId, events, rows, cwd)) + comparison;
|
|
332
367
|
}
|
|
@@ -2,13 +2,17 @@ import { type WorkerAdapter } from "../adapters/types.js";
|
|
|
2
2
|
import { type Task } from "../graph/schema.js";
|
|
3
3
|
import { type LlmVia } from "./llm.js";
|
|
4
4
|
import type { GateResult } from "./types.js";
|
|
5
|
+
export interface EvidenceCitation {
|
|
6
|
+
path: string;
|
|
7
|
+
line: number;
|
|
8
|
+
}
|
|
5
9
|
export interface JudgeVerdict {
|
|
6
10
|
pass: boolean;
|
|
7
11
|
criteria: Array<{
|
|
8
12
|
criterion: string;
|
|
9
13
|
met: boolean;
|
|
10
14
|
reason: string;
|
|
11
|
-
evidence: string;
|
|
15
|
+
evidence: string | EvidenceCitation;
|
|
12
16
|
}>;
|
|
13
17
|
}
|
|
14
18
|
export declare function judgeCriterionId(index: number): string;
|
package/dist/gates/acceptance.js
CHANGED
|
@@ -7,13 +7,14 @@ import { checkDiffCap, fetchTaskDiff } from "./review.js";
|
|
|
7
7
|
import { COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
|
|
8
8
|
// Fable F4: acceptance judge shares review's 900s timeout — 300s default killed frontier judges on cap-sized diffs.
|
|
9
9
|
const JUDGE_TIMEOUT_MS = 900_000;
|
|
10
|
+
const CitationSchema = z.object({ path: z.string(), line: z.number().int() });
|
|
10
11
|
const JudgeVerdictRowSchema = z.object({
|
|
11
12
|
criterion: z.string(),
|
|
12
13
|
met: z.boolean(),
|
|
13
14
|
reason: z.string(),
|
|
14
|
-
//
|
|
15
|
-
//
|
|
16
|
-
evidence: z.string(),
|
|
15
|
+
// Required, either form: a row omitting evidence is malformed and fails closed like any other shape
|
|
16
|
+
// violation. Object → structured citation; string → legacy quote.
|
|
17
|
+
evidence: z.union([CitationSchema, z.string()]),
|
|
17
18
|
});
|
|
18
19
|
const JudgeVerdictSchema = z.object({
|
|
19
20
|
pass: z.boolean(),
|
|
@@ -103,6 +104,64 @@ function tail(out, n = 8) {
|
|
|
103
104
|
return "";
|
|
104
105
|
return "\n" + t.split("\n").slice(-n).join("\n");
|
|
105
106
|
}
|
|
107
|
+
// v1.70: the new-file line numbers each hunk actually ADDS, per changed path — parsed from the
|
|
108
|
+
// unified-diff hunk headers so a citation is validated against real changed locations, not a substring
|
|
109
|
+
// of the whole diff text (which also matches unchanged context lines and the diff's own +++/@@ headers).
|
|
110
|
+
// Only `+` lines are changed locations; context lines advance the new-file counter but are not changes.
|
|
111
|
+
// ponytail: assumes git's default a/ b/ prefixes (fetchTaskDiff uses plain `git diff`); revisit if a
|
|
112
|
+
// caller passes a --no-prefix diff.
|
|
113
|
+
function changedLinesByFile(diff) {
|
|
114
|
+
const byFile = new Map();
|
|
115
|
+
let path = null;
|
|
116
|
+
let newLine = 0;
|
|
117
|
+
let inHunk = false;
|
|
118
|
+
for (const raw of diff.split("\n")) {
|
|
119
|
+
if (raw.startsWith("diff --git")) {
|
|
120
|
+
inHunk = false;
|
|
121
|
+
path = null;
|
|
122
|
+
continue;
|
|
123
|
+
}
|
|
124
|
+
if (!inHunk && raw.startsWith("--- "))
|
|
125
|
+
continue;
|
|
126
|
+
if (!inHunk && raw.startsWith("+++ ")) {
|
|
127
|
+
const p = raw.slice(4).trim();
|
|
128
|
+
path = p === "/dev/null" ? null : p.replace(/^[ab]\//, "");
|
|
129
|
+
continue;
|
|
130
|
+
}
|
|
131
|
+
const hunk = /^@@ -\d+(?:,\d+)? \+(\d+)(?:,\d+)? @@/.exec(raw);
|
|
132
|
+
if (hunk) {
|
|
133
|
+
newLine = Number(hunk[1]);
|
|
134
|
+
inHunk = true;
|
|
135
|
+
continue;
|
|
136
|
+
}
|
|
137
|
+
if (!inHunk || path === null)
|
|
138
|
+
continue;
|
|
139
|
+
const c = raw[0];
|
|
140
|
+
if (c === "+") {
|
|
141
|
+
let set = byFile.get(path);
|
|
142
|
+
if (!set) {
|
|
143
|
+
set = new Set();
|
|
144
|
+
byFile.set(path, set);
|
|
145
|
+
}
|
|
146
|
+
set.add(newLine);
|
|
147
|
+
newLine++;
|
|
148
|
+
}
|
|
149
|
+
else if (c === " ") {
|
|
150
|
+
newLine++;
|
|
151
|
+
}
|
|
152
|
+
else if (c !== "-") {
|
|
153
|
+
inHunk = false; // "\ No newline", a trailing blank, or the next section — hunk body ended
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
return byFile;
|
|
157
|
+
}
|
|
158
|
+
// A citation is valid evidence iff the diff adds the cited line of the cited file. A legacy free-text
|
|
159
|
+
// quote (string) keeps v1.64's substring check — non-empty and present somewhere in the diff.
|
|
160
|
+
function citesChangedLocation(evidence, changed, diff) {
|
|
161
|
+
if (typeof evidence === "string")
|
|
162
|
+
return evidence.trim().length > 0 && diff.includes(evidence);
|
|
163
|
+
return changed.get(evidence.path)?.has(evidence.line) ?? false;
|
|
164
|
+
}
|
|
106
165
|
export async function acceptanceGate(task, worktree, baseRef, judge, via, opts = {}) {
|
|
107
166
|
// 1. deterministic oracles — exit code decides, fail-closed, zero LLM calls (spec §2, T2).
|
|
108
167
|
// A failure returns here, before any runLlm() call: a judge can never override it.
|
|
@@ -176,9 +235,9 @@ ${diff}
|
|
|
176
235
|
${verdictNonceLine(nonce)}
|
|
177
236
|
|
|
178
237
|
Respond with ONLY this JSON (no prose before or after):
|
|
179
|
-
{"nonce": "${nonce}", "pass": true|false, "criteria": [{"criterion": "c1", "met": true|false, "reason": "...", "evidence": "
|
|
238
|
+
{"nonce": "${nonce}", "pass": true|false, "criteria": [{"criterion": "c1", "met": true|false, "reason": "...", "evidence": {"path": "path/to/file", "line": 42}}]}
|
|
180
239
|
Each criteria[].criterion MUST be the stable id from the rubric (c1, c2, ...) exactly once.
|
|
181
|
-
Each criteria[].evidence MUST be a
|
|
240
|
+
Each criteria[].evidence MUST be a structured citation {"path", "line"} pointing at a line the diff above actually adds or changes (the new-file line number); a citation to an unchanged or nonexistent location voids the whole verdict.
|
|
182
241
|
`;
|
|
183
242
|
const raw = await runLlm(judge.adapter, judge.model, prompt, worktree, via, JUDGE_TIMEOUT_MS);
|
|
184
243
|
const extracted = extractVerdictJson(raw, nonce);
|
|
@@ -191,13 +250,16 @@ Each criteria[].evidence MUST be a short verbatim quote copied from the diff abo
|
|
|
191
250
|
meta: { unparseable: true, judge: channelKey({ adapter: judge.adapter.id, model: judge.model }) } };
|
|
192
251
|
}
|
|
193
252
|
const { verdict: v, inconsistencies } = checkJudgeVerdict(extracted, expectedIds);
|
|
194
|
-
// v1.
|
|
195
|
-
// above, never the worktree or any other artifact.
|
|
196
|
-
//
|
|
197
|
-
|
|
253
|
+
// v1.70: each citation must point at a line the diff actually changed — validated against the hunks
|
|
254
|
+
// of `diff` (the exact string embedded in the prompt above), never the worktree or any other artifact.
|
|
255
|
+
// A citation to an untouched file or a line outside every changed hunk is a hallucinated verdict:
|
|
256
|
+
// treated as unparseable so GATE-09 retries the judge on a failover channel. (A legacy free-text quote
|
|
257
|
+
// still validates by substring, for the fake seam and pre-v1.70 fixtures.)
|
|
258
|
+
const changed = changedLinesByFile(diff);
|
|
259
|
+
const fabricated = v.criteria.filter((row) => !citesChangedLocation(row.evidence, changed, diff));
|
|
198
260
|
if (fabricated.length) {
|
|
199
261
|
return { gate: "acceptance", pass: false,
|
|
200
|
-
details: warn + detBlock + `judge verdict
|
|
262
|
+
details: warn + detBlock + `judge verdict cites evidence absent from the judged diff (${fabricated.map((row) => row.criterion).join(", ")}) — treating as unparseable, failing closed`,
|
|
201
263
|
meta: { unparseable: true, judge: channelKey({ adapter: judge.adapter.id, model: judge.model }) } };
|
|
202
264
|
}
|
|
203
265
|
const pass = v.pass === true && inconsistencies.length === 0 && v.criteria.every((row) => row.met);
|
package/dist/gates/review.d.ts
CHANGED
|
@@ -3,9 +3,17 @@ import { type TickmarkrConfig } from "../config/config.js";
|
|
|
3
3
|
import { type Task } from "../graph/schema.js";
|
|
4
4
|
import { type GateVia } from "./llm.js";
|
|
5
5
|
import type { GateResult } from "./types.js";
|
|
6
|
+
export type ReviewSeverity = "material" | "minor";
|
|
7
|
+
export interface ReviewFinding {
|
|
8
|
+
note: string;
|
|
9
|
+
severity: ReviewSeverity;
|
|
10
|
+
defer?: boolean;
|
|
11
|
+
rationale?: string;
|
|
12
|
+
}
|
|
6
13
|
export interface ReviewVerdict {
|
|
7
|
-
approve
|
|
8
|
-
issues
|
|
14
|
+
approve?: boolean;
|
|
15
|
+
issues?: string[];
|
|
16
|
+
findings?: ReviewFinding[];
|
|
9
17
|
}
|
|
10
18
|
export declare function fetchTaskDiff(worktree: string, baseRef: string): Promise<{
|
|
11
19
|
full: string;
|
package/dist/gates/review.js
CHANGED
|
@@ -5,6 +5,63 @@ import { getAdapter } from "../adapters/registry.js";
|
|
|
5
5
|
import { shOk } from "../run/git.js";
|
|
6
6
|
import { marginalCostRank } from "../route/router.js";
|
|
7
7
|
import { COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
|
|
8
|
+
// legacy flat `issues` shape — every issue blocks; the approve flag must agree with the list.
|
|
9
|
+
function classifyReviewIssues(approve, issues) {
|
|
10
|
+
const inconsistencies = [];
|
|
11
|
+
issues.forEach((issue, i) => {
|
|
12
|
+
if (typeof issue !== "string")
|
|
13
|
+
inconsistencies.push(`review verdict inconsistent: issues[${i}] must be a string`);
|
|
14
|
+
});
|
|
15
|
+
if (approve && issues.length) {
|
|
16
|
+
inconsistencies.push("review verdict inconsistent: approve=true requires issues to be empty");
|
|
17
|
+
}
|
|
18
|
+
else if (!approve && !issues.length) {
|
|
19
|
+
inconsistencies.push("review verdict inconsistent: approve=false requires at least one issue");
|
|
20
|
+
}
|
|
21
|
+
const pass = approve === true && inconsistencies.length === 0;
|
|
22
|
+
const lines = issues.map((issue) => `- ${typeof issue === "string" ? issue : JSON.stringify(issue)}`);
|
|
23
|
+
lines.push(...inconsistencies);
|
|
24
|
+
return { pass, headline: pass ? "approved" : approve ? "approval rejected" : "requested changes", lines };
|
|
25
|
+
}
|
|
26
|
+
// v1.70 T5: classified findings — only material (non-deferred) findings block approval. Deferred
|
|
27
|
+
// findings carry their rationale into the details (never dropped). Malformed rows fail closed like any
|
|
28
|
+
// other shape violation, so a garbage "findings" array can never fake an approval.
|
|
29
|
+
function classifyReviewFindings(findings) {
|
|
30
|
+
const inconsistencies = [];
|
|
31
|
+
const lines = [];
|
|
32
|
+
let material = 0;
|
|
33
|
+
let deferred = 0;
|
|
34
|
+
findings.forEach((f, i) => {
|
|
35
|
+
if (!f || typeof f !== "object") {
|
|
36
|
+
inconsistencies.push(`review finding inconsistent: findings[${i}] must be an object`);
|
|
37
|
+
return;
|
|
38
|
+
}
|
|
39
|
+
const { note, severity, defer, rationale } = f;
|
|
40
|
+
if (typeof note !== "string")
|
|
41
|
+
inconsistencies.push(`review finding inconsistent: findings[${i}].note must be a string`);
|
|
42
|
+
if (severity !== "material" && severity !== "minor")
|
|
43
|
+
inconsistencies.push(`review finding inconsistent: findings[${i}].severity must be "material" or "minor"`);
|
|
44
|
+
if (defer !== undefined && typeof defer !== "boolean")
|
|
45
|
+
inconsistencies.push(`review finding inconsistent: findings[${i}].defer must be a boolean`);
|
|
46
|
+
const isDeferred = defer === true;
|
|
47
|
+
if (isDeferred && (typeof rationale !== "string" || !rationale.trim())) {
|
|
48
|
+
inconsistencies.push(`review finding inconsistent: deferred findings[${i}] requires a rationale`);
|
|
49
|
+
}
|
|
50
|
+
if (severity === "material" && !isDeferred)
|
|
51
|
+
material++;
|
|
52
|
+
if (isDeferred)
|
|
53
|
+
deferred++;
|
|
54
|
+
const label = isDeferred ? `deferred/${severity ?? "?"}` : String(severity ?? "?");
|
|
55
|
+
const why = isDeferred && typeof rationale === "string" ? ` — rationale: ${rationale}` : "";
|
|
56
|
+
lines.push(`- [${label}] ${typeof note === "string" ? note : JSON.stringify(note)}${why}`);
|
|
57
|
+
});
|
|
58
|
+
const pass = material === 0 && inconsistencies.length === 0;
|
|
59
|
+
lines.push(...inconsistencies);
|
|
60
|
+
const headline = pass
|
|
61
|
+
? deferred ? `approved (${deferred} deferred)` : "approved"
|
|
62
|
+
: `requested changes (${material} material)`;
|
|
63
|
+
return { pass, headline, lines };
|
|
64
|
+
}
|
|
8
65
|
// OBS-48: cap on zero-context diff bytes (git diff -U0), not context-padded full diff — scattered
|
|
9
66
|
// one-line hunks no longer trip at ~370 diff-bytes per changed line. Full diff still goes to the judge.
|
|
10
67
|
const DIFF_CAP_REMEDY = "split the task, or raise gates.diffCap";
|
|
@@ -95,8 +152,14 @@ ${diff}
|
|
|
95
152
|
|
|
96
153
|
${verdictNonceLine(nonce)}
|
|
97
154
|
|
|
155
|
+
Classify every concern as "material" (a correctness, security, or acceptance-criteria defect that must
|
|
156
|
+
block the merge) or "minor" (style, naming, or preference that should not block). ONLY material findings
|
|
157
|
+
block approval. For a minor concern you have decided not to block on, set "defer": true and give a
|
|
158
|
+
one-line "rationale" — it is recorded in the review, never dropped.
|
|
159
|
+
|
|
98
160
|
Respond with ONLY this JSON:
|
|
99
|
-
{"nonce": "${nonce}", "approve": true|false, "
|
|
161
|
+
{"nonce": "${nonce}", "approve": true|false, "findings": [{"note": "...", "severity": "material"|"minor", "defer": false, "rationale": ""}]}
|
|
162
|
+
Approve iff no material finding remains; an empty findings list is a clean approval.
|
|
100
163
|
`;
|
|
101
164
|
const raw = await runLlm(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? { driver: via.driver, keep: via.keep, onSlot: via.onSlot, name: via.nameFor("review", reviewer.adapter), label: via.labelFor("review") } : undefined,
|
|
102
165
|
// frontier reviewers routinely need >5min on a configured-cap-sized diff, and `claude -p` buffers all
|
|
@@ -106,7 +169,9 @@ Respond with ONLY this JSON:
|
|
|
106
169
|
// second knob-turner appears.
|
|
107
170
|
900_000);
|
|
108
171
|
const v = extractVerdictJson(raw, nonce);
|
|
109
|
-
|
|
172
|
+
const findings = v && Array.isArray(v.findings) ? v.findings : null;
|
|
173
|
+
// findings decides the verdict on its own; the legacy path still needs approve + issues to parse.
|
|
174
|
+
if (!v || (findings === null && (typeof v.approve !== "boolean" || !Array.isArray(v.issues)))) {
|
|
110
175
|
return {
|
|
111
176
|
gate: "review",
|
|
112
177
|
pass: false,
|
|
@@ -114,25 +179,13 @@ Respond with ONLY this JSON:
|
|
|
114
179
|
meta: { reviewer: channelKey(reviewer) },
|
|
115
180
|
};
|
|
116
181
|
}
|
|
117
|
-
const
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
if (typeof issue !== "string")
|
|
121
|
-
inconsistencies.push(`review verdict inconsistent: issues[${i}] must be a string`);
|
|
122
|
-
});
|
|
123
|
-
if (v.approve && issues.length) {
|
|
124
|
-
inconsistencies.push("review verdict inconsistent: approve=true requires issues to be empty");
|
|
125
|
-
}
|
|
126
|
-
else if (!v.approve && !issues.length) {
|
|
127
|
-
inconsistencies.push("review verdict inconsistent: approve=false requires at least one issue");
|
|
128
|
-
}
|
|
129
|
-
const pass = v.approve === true && inconsistencies.length === 0;
|
|
130
|
-
const lines = issues.map((issue) => `- ${typeof issue === "string" ? issue : JSON.stringify(issue)}`);
|
|
131
|
-
lines.push(...inconsistencies);
|
|
182
|
+
const decided = findings !== null
|
|
183
|
+
? classifyReviewFindings(findings)
|
|
184
|
+
: classifyReviewIssues(v.approve, v.issues);
|
|
132
185
|
return {
|
|
133
186
|
gate: "review",
|
|
134
|
-
pass,
|
|
135
|
-
details: `reviewer ${reviewer.adapter}:${reviewer.model} (${reviewer.vendor}): ${
|
|
187
|
+
pass: decided.pass,
|
|
188
|
+
details: `reviewer ${reviewer.adapter}:${reviewer.model} (${reviewer.vendor}): ${decided.headline}${decided.lines.length ? "\n" + decided.lines.join("\n") : ""}`,
|
|
136
189
|
meta: { reviewer: channelKey(reviewer) },
|
|
137
190
|
};
|
|
138
191
|
}
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
import type { EvidenceCitation } from "../gates/acceptance.js";
|
|
2
|
+
import type { RunEnvironment } from "../run/environment.js";
|
|
3
|
+
import type { JournalEvent } from "../run/journal.js";
|
|
4
|
+
export declare const BUNDLE_SCHEMA_VERSION = 1;
|
|
5
|
+
export declare const KNOWN_LIMITS: readonly string[];
|
|
6
|
+
export type BundleEvidence = string | EvidenceCitation;
|
|
7
|
+
export interface BundleJudgeCriterion {
|
|
8
|
+
criterion: string;
|
|
9
|
+
met: boolean;
|
|
10
|
+
reason: string;
|
|
11
|
+
evidence: BundleEvidence;
|
|
12
|
+
}
|
|
13
|
+
export interface BundleGateResult {
|
|
14
|
+
gate: string;
|
|
15
|
+
pass: boolean;
|
|
16
|
+
details: string;
|
|
17
|
+
}
|
|
18
|
+
export interface BundleTask {
|
|
19
|
+
taskId: string;
|
|
20
|
+
outcome: "done" | "failed" | "human" | "not-recorded";
|
|
21
|
+
gates: BundleGateResult[];
|
|
22
|
+
/** Judge criteria with evidence citations, when the journal recorded them structured. */
|
|
23
|
+
judgeCriteria: BundleJudgeCriterion[];
|
|
24
|
+
}
|
|
25
|
+
export interface BundleContentHashes {
|
|
26
|
+
/** sha256 of the canonical JSON of the source journal events used to build this packet. */
|
|
27
|
+
journal: string;
|
|
28
|
+
/** graphDefinitionHash from run-start when recorded. */
|
|
29
|
+
graphDefinitionHash?: string;
|
|
30
|
+
}
|
|
31
|
+
export interface ProofBundle {
|
|
32
|
+
/** Schema version a future reader must check before parsing the rest. */
|
|
33
|
+
schemaVersion: number;
|
|
34
|
+
runId: string;
|
|
35
|
+
environment?: RunEnvironment;
|
|
36
|
+
contentHashes: BundleContentHashes;
|
|
37
|
+
tasks: BundleTask[];
|
|
38
|
+
/** Plain-language known limits — this is not an unconditional proof of correctness. */
|
|
39
|
+
knownLimits: string[];
|
|
40
|
+
}
|
|
41
|
+
/**
|
|
42
|
+
* Build a schema-versioned proof packet from a run's journal events.
|
|
43
|
+
* Pure: no filesystem, no network. Caller writes the result.
|
|
44
|
+
*/
|
|
45
|
+
export declare function buildProofBundle(runId: string, events: JournalEvent[]): ProofBundle;
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
// v1.70 T4: pure proof-packet builder for `tickmarkr report --bundle <path>`.
|
|
2
|
+
// One portable, schema-versioned snapshot of a run — task outcomes, judge evidence
|
|
3
|
+
// citations, environment identity, content hashes — secrets redacted via the shared
|
|
4
|
+
// redactSecrets seam. No I/O and no network: the caller (report CLI) owns the write.
|
|
5
|
+
import { createHash } from "node:crypto";
|
|
6
|
+
import { redactSecrets } from "../run/redact.js";
|
|
7
|
+
import { recordedEnvironment } from "./compare.js";
|
|
8
|
+
// Bump when the packet shape changes in a way a future reader must branch on before parsing.
|
|
9
|
+
export const BUNDLE_SCHEMA_VERSION = 1;
|
|
10
|
+
// Plain-language known limits — the packet is a journal snapshot, not an unconditional proof.
|
|
11
|
+
export const KNOWN_LIMITS = [
|
|
12
|
+
"This packet is a portable snapshot of journaled run facts, not an independent re-verification of the work or its gates.",
|
|
13
|
+
"Judge evidence citations are copied from the journal as the judge recorded them; this packet does not re-validate citations against the judged diff.",
|
|
14
|
+
"Content hashes bind this packet to the journal bytes used to build it; they do not prove the underlying gates, merges, or tip verification were correct.",
|
|
15
|
+
"Secret-shaped strings are redacted by local pattern matching only; redaction is not a formal security audit.",
|
|
16
|
+
"Producing this packet never contacts a network and never uploads anything.",
|
|
17
|
+
];
|
|
18
|
+
function contentHash(events) {
|
|
19
|
+
return createHash("sha256").update(JSON.stringify(events)).digest("hex");
|
|
20
|
+
}
|
|
21
|
+
function graphHash(events) {
|
|
22
|
+
for (const e of events) {
|
|
23
|
+
if (e.event !== "run-start")
|
|
24
|
+
continue;
|
|
25
|
+
const h = e.data.graphDefinitionHash;
|
|
26
|
+
return typeof h === "string" ? h : undefined;
|
|
27
|
+
}
|
|
28
|
+
return undefined;
|
|
29
|
+
}
|
|
30
|
+
function taskIds(events) {
|
|
31
|
+
const seen = new Set();
|
|
32
|
+
const out = [];
|
|
33
|
+
for (const e of events) {
|
|
34
|
+
if (!e.taskId || seen.has(e.taskId))
|
|
35
|
+
continue;
|
|
36
|
+
seen.add(e.taskId);
|
|
37
|
+
out.push(e.taskId);
|
|
38
|
+
}
|
|
39
|
+
return out;
|
|
40
|
+
}
|
|
41
|
+
function outcomeFor(events, taskId, runEnd) {
|
|
42
|
+
for (let i = events.length - 1; i >= 0; i--) {
|
|
43
|
+
const e = events[i];
|
|
44
|
+
if (e.taskId !== taskId)
|
|
45
|
+
continue;
|
|
46
|
+
if (e.event === "task-done")
|
|
47
|
+
return "done";
|
|
48
|
+
if (e.event === "task-failed")
|
|
49
|
+
return "failed";
|
|
50
|
+
if (e.event === "task-human")
|
|
51
|
+
return "human";
|
|
52
|
+
}
|
|
53
|
+
if (runEnd) {
|
|
54
|
+
const d = runEnd.data;
|
|
55
|
+
if (Array.isArray(d.done) && d.done.includes(taskId))
|
|
56
|
+
return "done";
|
|
57
|
+
if (Array.isArray(d.failed) && d.failed.includes(taskId))
|
|
58
|
+
return "failed";
|
|
59
|
+
if (Array.isArray(d.human) && d.human.includes(taskId))
|
|
60
|
+
return "human";
|
|
61
|
+
}
|
|
62
|
+
return "not-recorded";
|
|
63
|
+
}
|
|
64
|
+
// Parse a structured evidence citation without rewriting path/line — unaltered copy.
|
|
65
|
+
function parseEvidence(raw) {
|
|
66
|
+
if (typeof raw === "string")
|
|
67
|
+
return raw;
|
|
68
|
+
if (!raw || typeof raw !== "object" || Array.isArray(raw))
|
|
69
|
+
return undefined;
|
|
70
|
+
const o = raw;
|
|
71
|
+
if (typeof o.path === "string" && typeof o.line === "number" && Number.isInteger(o.line)) {
|
|
72
|
+
return { path: o.path, line: o.line };
|
|
73
|
+
}
|
|
74
|
+
return undefined;
|
|
75
|
+
}
|
|
76
|
+
// Structured judge criteria on an acceptance gate-result (when journaled). Evidence is copied
|
|
77
|
+
// unaltered so a future reader can match citations byte-for-byte to what the judge recorded.
|
|
78
|
+
function parseJudgeCriteria(data) {
|
|
79
|
+
const raw = data.criteria;
|
|
80
|
+
if (!Array.isArray(raw))
|
|
81
|
+
return [];
|
|
82
|
+
const out = [];
|
|
83
|
+
for (const row of raw) {
|
|
84
|
+
if (!row || typeof row !== "object" || Array.isArray(row))
|
|
85
|
+
continue;
|
|
86
|
+
const r = row;
|
|
87
|
+
if (typeof r.criterion !== "string" || typeof r.met !== "boolean" || typeof r.reason !== "string")
|
|
88
|
+
continue;
|
|
89
|
+
const evidence = parseEvidence(r.evidence);
|
|
90
|
+
if (evidence === undefined)
|
|
91
|
+
continue;
|
|
92
|
+
out.push({ criterion: r.criterion, met: r.met, reason: r.reason, evidence });
|
|
93
|
+
}
|
|
94
|
+
return out;
|
|
95
|
+
}
|
|
96
|
+
// Deep-walk string leaves through redactSecrets. Numbers/booleans/null stay as-is.
|
|
97
|
+
// Structured evidence citations keep path/line unaltered (redact only free-text string evidence).
|
|
98
|
+
function redactValue(v) {
|
|
99
|
+
if (typeof v === "string")
|
|
100
|
+
return redactSecrets(v);
|
|
101
|
+
if (v === null || typeof v !== "object")
|
|
102
|
+
return v;
|
|
103
|
+
if (Array.isArray(v))
|
|
104
|
+
return v.map(redactValue);
|
|
105
|
+
const o = v;
|
|
106
|
+
// Preserve structured {path, line} citations unaltered (criterion "unaltered").
|
|
107
|
+
if (typeof o.path === "string" && typeof o.line === "number" && Object.keys(o).length === 2) {
|
|
108
|
+
return { path: o.path, line: o.line };
|
|
109
|
+
}
|
|
110
|
+
const out = {};
|
|
111
|
+
for (const [k, val] of Object.entries(o))
|
|
112
|
+
out[k] = redactValue(val);
|
|
113
|
+
return out;
|
|
114
|
+
}
|
|
115
|
+
/**
|
|
116
|
+
* Build a schema-versioned proof packet from a run's journal events.
|
|
117
|
+
* Pure: no filesystem, no network. Caller writes the result.
|
|
118
|
+
*/
|
|
119
|
+
export function buildProofBundle(runId, events) {
|
|
120
|
+
const runEnd = [...events].reverse().find((e) => e.event === "run-end");
|
|
121
|
+
const environment = recordedEnvironment(events);
|
|
122
|
+
const journalHash = contentHash(events);
|
|
123
|
+
const gHash = graphHash(events);
|
|
124
|
+
const tasks = taskIds(events).map((taskId) => {
|
|
125
|
+
const gateEvents = events.filter((e) => e.taskId === taskId && e.event === "gate-result");
|
|
126
|
+
const gates = [];
|
|
127
|
+
const judgeCriteria = [];
|
|
128
|
+
for (const g of gateEvents) {
|
|
129
|
+
const gate = typeof g.data.gate === "string" ? g.data.gate : "unknown";
|
|
130
|
+
const pass = g.data.pass === true;
|
|
131
|
+
const details = typeof g.data.details === "string" ? g.data.details : "";
|
|
132
|
+
gates.push({ gate, pass, details });
|
|
133
|
+
if (gate === "acceptance") {
|
|
134
|
+
for (const c of parseJudgeCriteria(g.data))
|
|
135
|
+
judgeCriteria.push(c);
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
return {
|
|
139
|
+
taskId,
|
|
140
|
+
outcome: outcomeFor(events, taskId, runEnd),
|
|
141
|
+
gates,
|
|
142
|
+
judgeCriteria,
|
|
143
|
+
};
|
|
144
|
+
});
|
|
145
|
+
const packet = {
|
|
146
|
+
schemaVersion: BUNDLE_SCHEMA_VERSION,
|
|
147
|
+
runId,
|
|
148
|
+
...(environment ? { environment } : {}),
|
|
149
|
+
contentHashes: {
|
|
150
|
+
journal: journalHash,
|
|
151
|
+
...(gHash ? { graphDefinitionHash: gHash } : {}),
|
|
152
|
+
},
|
|
153
|
+
tasks,
|
|
154
|
+
knownLimits: [...KNOWN_LIMITS],
|
|
155
|
+
};
|
|
156
|
+
// Redact secret-shaped strings anywhere in the packet. Structured {path,line} citations stay intact.
|
|
157
|
+
return redactValue(packet);
|
|
158
|
+
}
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
import type { RunEnvironment } from "../run/environment.js";
|
|
2
|
+
import type { JournalEvent, TelemetryRow } from "../run/journal.js";
|
|
3
|
+
import { type CostConfig } from "./cost.js";
|
|
4
|
+
export type EnvironmentCompare = {
|
|
5
|
+
comparable: true;
|
|
6
|
+
recorded: string;
|
|
7
|
+
} | {
|
|
8
|
+
comparable: false;
|
|
9
|
+
reason: "mismatch";
|
|
10
|
+
recorded: string;
|
|
11
|
+
} | {
|
|
12
|
+
comparable: false;
|
|
13
|
+
reason: "unbound";
|
|
14
|
+
};
|
|
15
|
+
export declare function hasRunStart(events: JournalEvent[]): boolean;
|
|
16
|
+
export declare function recordedEnvironment(events: JournalEvent[]): RunEnvironment | undefined;
|
|
17
|
+
export declare function environmentComparable(baselineEnv: RunEnvironment | undefined, currentEnv: RunEnvironment | undefined): EnvironmentCompare;
|
|
18
|
+
export interface RunMetrics {
|
|
19
|
+
durationMs: number | undefined;
|
|
20
|
+
gatePass: number;
|
|
21
|
+
gateFail: number;
|
|
22
|
+
gateTotal: number;
|
|
23
|
+
costUsd: number | undefined;
|
|
24
|
+
tokensTotal: number | undefined;
|
|
25
|
+
}
|
|
26
|
+
export declare function runMetrics(events: JournalEvent[], rows?: TelemetryRow[], cost?: CostConfig): RunMetrics;
|
|
27
|
+
export interface RunDelta {
|
|
28
|
+
durationMs: number | undefined;
|
|
29
|
+
gateFail: number;
|
|
30
|
+
costUsd: number | undefined;
|
|
31
|
+
tokensTotal: number | undefined;
|
|
32
|
+
}
|
|
33
|
+
export type CompareOutcome = {
|
|
34
|
+
ok: false;
|
|
35
|
+
reason: string;
|
|
36
|
+
} | {
|
|
37
|
+
ok: true;
|
|
38
|
+
runId: string;
|
|
39
|
+
baselineRunId: string;
|
|
40
|
+
comparability: EnvironmentCompare;
|
|
41
|
+
current: RunMetrics;
|
|
42
|
+
baseline: RunMetrics;
|
|
43
|
+
delta: RunDelta;
|
|
44
|
+
text: string;
|
|
45
|
+
};
|
|
46
|
+
export declare function renderComparison(opts: {
|
|
47
|
+
runId: string;
|
|
48
|
+
baselineRunId: string;
|
|
49
|
+
comparability: EnvironmentCompare;
|
|
50
|
+
baselineEnv: RunEnvironment | undefined;
|
|
51
|
+
currentEnv: RunEnvironment | undefined;
|
|
52
|
+
current: RunMetrics;
|
|
53
|
+
baseline: RunMetrics;
|
|
54
|
+
delta: RunDelta;
|
|
55
|
+
}): string;
|
|
56
|
+
export declare function compareRuns(opts: {
|
|
57
|
+
runId: string;
|
|
58
|
+
baselineRunId: string;
|
|
59
|
+
events: JournalEvent[];
|
|
60
|
+
baselineEvents: JournalEvent[];
|
|
61
|
+
rows?: TelemetryRow[];
|
|
62
|
+
baselineRows?: TelemetryRow[];
|
|
63
|
+
cost?: CostConfig;
|
|
64
|
+
}): CompareOutcome;
|
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
import { estimateCosts } from "./cost.js";
|
|
2
|
+
export function hasRunStart(events) {
|
|
3
|
+
return events.some((e) => e.event === "run-start");
|
|
4
|
+
}
|
|
5
|
+
// First run-start's environment, fail-closed on a missing or malformed stamp.
|
|
6
|
+
export function recordedEnvironment(events) {
|
|
7
|
+
for (const e of events) {
|
|
8
|
+
if (e.event !== "run-start")
|
|
9
|
+
continue;
|
|
10
|
+
const env = e.data.environment;
|
|
11
|
+
if (!env || typeof env !== "object" || Array.isArray(env))
|
|
12
|
+
return undefined;
|
|
13
|
+
const o = env;
|
|
14
|
+
if (typeof o.tickmarkrVersion !== "string" || typeof o.configHash !== "string")
|
|
15
|
+
return undefined;
|
|
16
|
+
if (!o.adapterVersions || typeof o.adapterVersions !== "object" || Array.isArray(o.adapterVersions))
|
|
17
|
+
return undefined;
|
|
18
|
+
const adapterVersions = {};
|
|
19
|
+
for (const [k, v] of Object.entries(o.adapterVersions)) {
|
|
20
|
+
if (typeof v !== "string")
|
|
21
|
+
return undefined;
|
|
22
|
+
adapterVersions[k] = v;
|
|
23
|
+
}
|
|
24
|
+
return { tickmarkrVersion: o.tickmarkrVersion, configHash: o.configHash, adapterVersions };
|
|
25
|
+
}
|
|
26
|
+
return undefined;
|
|
27
|
+
}
|
|
28
|
+
// Canonical identity string for the `recorded` field — configHash is the axis the acceptance
|
|
29
|
+
// criteria name; full environment equality still decides .comparable.
|
|
30
|
+
function envFingerprint(env) {
|
|
31
|
+
return env.configHash;
|
|
32
|
+
}
|
|
33
|
+
function envEqual(a, b) {
|
|
34
|
+
if (a.tickmarkrVersion !== b.tickmarkrVersion || a.configHash !== b.configHash)
|
|
35
|
+
return false;
|
|
36
|
+
const ak = Object.keys(a.adapterVersions).sort();
|
|
37
|
+
const bk = Object.keys(b.adapterVersions).sort();
|
|
38
|
+
if (ak.length !== bk.length)
|
|
39
|
+
return false;
|
|
40
|
+
for (let i = 0; i < ak.length; i++) {
|
|
41
|
+
if (ak[i] !== bk[i])
|
|
42
|
+
return false;
|
|
43
|
+
if (a.adapterVersions[ak[i]] !== b.adapterVersions[bk[i]])
|
|
44
|
+
return false;
|
|
45
|
+
}
|
|
46
|
+
return true;
|
|
47
|
+
}
|
|
48
|
+
// THE environment-identity comparator (criterion: reuses engagementComparable's shape).
|
|
49
|
+
// unbound = either side lacks a usable stamp; mismatch = stamps disagree; comparable = equal.
|
|
50
|
+
export function environmentComparable(baselineEnv, currentEnv) {
|
|
51
|
+
if (baselineEnv === undefined || currentEnv === undefined)
|
|
52
|
+
return { comparable: false, reason: "unbound" };
|
|
53
|
+
const recorded = envFingerprint(baselineEnv);
|
|
54
|
+
return envEqual(baselineEnv, currentEnv)
|
|
55
|
+
? { comparable: true, recorded }
|
|
56
|
+
: { comparable: false, reason: "mismatch", recorded };
|
|
57
|
+
}
|
|
58
|
+
export function runMetrics(events, rows = [], cost = {}) {
|
|
59
|
+
const start = events.find((e) => e.event === "run-start");
|
|
60
|
+
const end = [...events].reverse().find((e) => e.event === "run-end");
|
|
61
|
+
let durationMs;
|
|
62
|
+
if (start && end) {
|
|
63
|
+
const from = Date.parse(start.ts);
|
|
64
|
+
const to = Date.parse(end.ts);
|
|
65
|
+
if (Number.isFinite(from) && Number.isFinite(to) && to >= from)
|
|
66
|
+
durationMs = to - from;
|
|
67
|
+
}
|
|
68
|
+
const gates = events.filter((e) => e.event === "gate-result");
|
|
69
|
+
const gatePass = gates.filter((e) => e.data.pass === true).length;
|
|
70
|
+
const gateFail = gates.filter((e) => e.data.pass === false).length;
|
|
71
|
+
let costUsd;
|
|
72
|
+
let costSum = 0;
|
|
73
|
+
let hasCost = false;
|
|
74
|
+
for (const p of estimateCosts(rows, cost)) {
|
|
75
|
+
if (p.apiUsd !== undefined) {
|
|
76
|
+
costSum += p.apiUsd;
|
|
77
|
+
hasCost = true;
|
|
78
|
+
}
|
|
79
|
+
else if (p.amortizedUsd) {
|
|
80
|
+
costSum += (p.amortizedUsd[0] + p.amortizedUsd[1]) / 2;
|
|
81
|
+
hasCost = true;
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
if (hasCost)
|
|
85
|
+
costUsd = Math.round(costSum * 1e6) / 1e6;
|
|
86
|
+
let tokensTotal;
|
|
87
|
+
let tokenSum = 0;
|
|
88
|
+
let hasTokens = false;
|
|
89
|
+
for (const r of rows) {
|
|
90
|
+
if (!r.tokens)
|
|
91
|
+
continue;
|
|
92
|
+
hasTokens = true;
|
|
93
|
+
const t = r.tokens;
|
|
94
|
+
tokenSum += t.input + t.output + (t.cacheRead ?? 0) + (t.cacheWrite ?? 0) + (t.reasoning ?? 0);
|
|
95
|
+
}
|
|
96
|
+
if (hasTokens)
|
|
97
|
+
tokensTotal = tokenSum;
|
|
98
|
+
return { durationMs, gatePass, gateFail, gateTotal: gates.length, costUsd, tokensTotal };
|
|
99
|
+
}
|
|
100
|
+
const n = (x) => x.toLocaleString("en-US");
|
|
101
|
+
const EM = "—";
|
|
102
|
+
function fmtDuration(ms) {
|
|
103
|
+
if (ms === undefined)
|
|
104
|
+
return EM;
|
|
105
|
+
const seconds = Math.round(ms / 1_000);
|
|
106
|
+
const minutes = Math.floor(seconds / 60);
|
|
107
|
+
return minutes ? `${minutes}m ${seconds % 60}s` : `${seconds}s`;
|
|
108
|
+
}
|
|
109
|
+
function fmtSignedDuration(ms) {
|
|
110
|
+
if (ms === undefined)
|
|
111
|
+
return EM;
|
|
112
|
+
if (ms === 0)
|
|
113
|
+
return "0s";
|
|
114
|
+
const sign = ms > 0 ? "+" : "-";
|
|
115
|
+
return `${sign}${fmtDuration(Math.abs(ms))}`;
|
|
116
|
+
}
|
|
117
|
+
function fmtUsd(v) {
|
|
118
|
+
if (v === undefined)
|
|
119
|
+
return EM;
|
|
120
|
+
return `$${v.toFixed(6)}`;
|
|
121
|
+
}
|
|
122
|
+
function fmtSignedUsd(v) {
|
|
123
|
+
if (v === undefined)
|
|
124
|
+
return EM;
|
|
125
|
+
if (v === 0)
|
|
126
|
+
return "$0.000000";
|
|
127
|
+
const sign = v > 0 ? "+" : "-";
|
|
128
|
+
return `${sign}$${Math.abs(v).toFixed(6)}`;
|
|
129
|
+
}
|
|
130
|
+
function fmtInt(v) {
|
|
131
|
+
if (v === undefined)
|
|
132
|
+
return EM;
|
|
133
|
+
return n(v);
|
|
134
|
+
}
|
|
135
|
+
function fmtSignedInt(v) {
|
|
136
|
+
if (v === undefined)
|
|
137
|
+
return EM;
|
|
138
|
+
if (v === 0)
|
|
139
|
+
return "0";
|
|
140
|
+
return v > 0 ? `+${n(v)}` : `-${n(Math.abs(v))}`;
|
|
141
|
+
}
|
|
142
|
+
function comparabilityLine(cmp, baselineEnv, currentEnv) {
|
|
143
|
+
if (cmp.comparable) {
|
|
144
|
+
return `full comparability (environment identity matches; configHash=${cmp.recorded})`;
|
|
145
|
+
}
|
|
146
|
+
if (cmp.reason === "unbound") {
|
|
147
|
+
return "comparability caveat — one or both runs lack a recorded environment identity; not apples-to-apples";
|
|
148
|
+
}
|
|
149
|
+
const base = baselineEnv?.configHash ?? EM;
|
|
150
|
+
const cur = currentEnv?.configHash ?? EM;
|
|
151
|
+
return `comparability caveat — environment identity disagrees (baseline configHash=${base} ≠ current configHash=${cur}; recorded baseline ${cmp.recorded}); not apples-to-apples`;
|
|
152
|
+
}
|
|
153
|
+
export function renderComparison(opts) {
|
|
154
|
+
const { runId, baselineRunId, comparability, baselineEnv, currentEnv, current, baseline, delta } = opts;
|
|
155
|
+
// Both run ids are always named so the reader knows which is the baseline.
|
|
156
|
+
const lines = [
|
|
157
|
+
`## Comparison`,
|
|
158
|
+
"",
|
|
159
|
+
`- **run:** ${runId}`,
|
|
160
|
+
`- **baseline:** ${baselineRunId}`,
|
|
161
|
+
`- **comparability:** ${comparabilityLine(comparability, baselineEnv, currentEnv)}`,
|
|
162
|
+
"",
|
|
163
|
+
"### Delta (current − baseline)",
|
|
164
|
+
"",
|
|
165
|
+
`| metric | baseline (${baselineRunId}) | current (${runId}) | delta |`,
|
|
166
|
+
`| --- | --- | --- | --- |`,
|
|
167
|
+
`| duration | ${fmtDuration(baseline.durationMs)} | ${fmtDuration(current.durationMs)} | ${fmtSignedDuration(delta.durationMs)} |`,
|
|
168
|
+
`| gate failures | ${fmtInt(baseline.gateFail)} | ${fmtInt(current.gateFail)} | ${fmtSignedInt(delta.gateFail)} |`,
|
|
169
|
+
`| gate pass rate | ${baseline.gateTotal ? `${baseline.gatePass}/${baseline.gateTotal}` : EM} | ${current.gateTotal ? `${current.gatePass}/${current.gateTotal}` : EM} | ${EM} |`,
|
|
170
|
+
`| cost | ${fmtUsd(baseline.costUsd)} | ${fmtUsd(current.costUsd)} | ${fmtSignedUsd(delta.costUsd)} |`,
|
|
171
|
+
`| tokens | ${fmtInt(baseline.tokensTotal)} | ${fmtInt(current.tokensTotal)} | ${fmtSignedInt(delta.tokensTotal)} |`,
|
|
172
|
+
"",
|
|
173
|
+
];
|
|
174
|
+
if (!comparability.comparable) {
|
|
175
|
+
lines.push("_Caveat: deltas are shown for inspection only — environment identity disagrees, so this is not an apples-to-apples table._", "");
|
|
176
|
+
}
|
|
177
|
+
return lines.join("\n").trimEnd() + "\n";
|
|
178
|
+
}
|
|
179
|
+
// Fail closed when either journal has no run-start (no partial / fabricated comparison).
|
|
180
|
+
// Mismatch of environment identity still yields a rendered delta, with an explicit caveat.
|
|
181
|
+
export function compareRuns(opts) {
|
|
182
|
+
if (!hasRunStart(opts.baselineEvents)) {
|
|
183
|
+
return {
|
|
184
|
+
ok: false,
|
|
185
|
+
reason: `baseline run ${opts.baselineRunId} has no recorded run-start event — cannot compare`,
|
|
186
|
+
};
|
|
187
|
+
}
|
|
188
|
+
if (!hasRunStart(opts.events)) {
|
|
189
|
+
return {
|
|
190
|
+
ok: false,
|
|
191
|
+
reason: `run ${opts.runId} has no recorded run-start event — cannot compare`,
|
|
192
|
+
};
|
|
193
|
+
}
|
|
194
|
+
const baselineEnv = recordedEnvironment(opts.baselineEvents);
|
|
195
|
+
const currentEnv = recordedEnvironment(opts.events);
|
|
196
|
+
const comparability = environmentComparable(baselineEnv, currentEnv);
|
|
197
|
+
const baseline = runMetrics(opts.baselineEvents, opts.baselineRows ?? [], opts.cost ?? {});
|
|
198
|
+
const current = runMetrics(opts.events, opts.rows ?? [], opts.cost ?? {});
|
|
199
|
+
const delta = {
|
|
200
|
+
durationMs: current.durationMs !== undefined && baseline.durationMs !== undefined
|
|
201
|
+
? current.durationMs - baseline.durationMs
|
|
202
|
+
: undefined,
|
|
203
|
+
gateFail: current.gateFail - baseline.gateFail,
|
|
204
|
+
costUsd: current.costUsd !== undefined && baseline.costUsd !== undefined
|
|
205
|
+
? Math.round((current.costUsd - baseline.costUsd) * 1e6) / 1e6
|
|
206
|
+
: undefined,
|
|
207
|
+
tokensTotal: current.tokensTotal !== undefined && baseline.tokensTotal !== undefined
|
|
208
|
+
? current.tokensTotal - baseline.tokensTotal
|
|
209
|
+
: undefined,
|
|
210
|
+
};
|
|
211
|
+
const text = renderComparison({
|
|
212
|
+
runId: opts.runId,
|
|
213
|
+
baselineRunId: opts.baselineRunId,
|
|
214
|
+
comparability,
|
|
215
|
+
baselineEnv,
|
|
216
|
+
currentEnv,
|
|
217
|
+
current,
|
|
218
|
+
baseline,
|
|
219
|
+
delta,
|
|
220
|
+
});
|
|
221
|
+
return {
|
|
222
|
+
ok: true,
|
|
223
|
+
runId: opts.runId,
|
|
224
|
+
baselineRunId: opts.baselineRunId,
|
|
225
|
+
comparability,
|
|
226
|
+
current,
|
|
227
|
+
baseline,
|
|
228
|
+
delta,
|
|
229
|
+
text,
|
|
230
|
+
};
|
|
231
|
+
}
|
package/dist/run/daemon.js
CHANGED
|
@@ -15,6 +15,7 @@ import { captureBaseline, detectGateCommands, detectVacuousOracles } from "../ga
|
|
|
15
15
|
import { runGates } from "../gates/run-gates.js";
|
|
16
16
|
import { addEvidence, attributeBlocked, blockedTasks, getTask, graphDefinitionHash, loadGraph, pendingTasks, readyTasks, saveGraph, setStatus } from "../graph/graph.js";
|
|
17
17
|
import { augmentRetryBrief, consult, renderRetryGuidance } from "./consult.js";
|
|
18
|
+
import { runEnvironment } from "./environment.js";
|
|
18
19
|
import { cleanupRunWorktrees, gitHead, linkNodeModules, sh, shGit, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
|
|
19
20
|
import { runInteractiveSeed } from "./interactive-seed.js";
|
|
20
21
|
import { classifyWorkerResultCause, engagementComparable, Journal, loadRoutingProfile, newRunId } from "./journal.js";
|
|
@@ -65,6 +66,10 @@ export function formatSummary(s) {
|
|
|
65
66
|
return `done: ${s.done.length}, failed: ${s.failed.length}, human: ${s.human.length}, blocked: ${s.blocked.length}, pending: ${s.pending.length}\nintegration branch: ${s.branch}${tip}`;
|
|
66
67
|
}
|
|
67
68
|
const MAX_ATTEMPTS = 10; // ponytail: hard cap so a pathological ladder can never loop forever
|
|
69
|
+
// v1.70 T5: request-changes review rounds a single task may draw before it parks for a human decision
|
|
70
|
+
// instead of cycling. Well below MAX_ATTEMPTS so review non-convergence is caught long before the
|
|
71
|
+
// global cap. ponytail: literal constant; lift to cfg.review.roundCap only if a second knob-turner appears.
|
|
72
|
+
const REVIEW_ROUND_CAP = 3;
|
|
68
73
|
const BLOCKED_POLL_MS = 30_000; // between trailer-wait slices, check whether the pane is blocked on a prompt
|
|
69
74
|
const PROVIDER_DEATH_REQUEUE_CAP = 2; // v1.46 T1: requeue same assignment twice, then fall through to the normal ladder
|
|
70
75
|
const PROVIDER_DEATH_BACKOFF_MS = 500; // short backoff before provider-death requeue
|
|
@@ -253,7 +258,11 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
253
258
|
baseRef = await gitHead(repoRoot);
|
|
254
259
|
baseline = await captureBaseline(repoRoot, commands);
|
|
255
260
|
writeFileSync(join(journal.dir, "baseline.json"), JSON.stringify(baseline, null, 2));
|
|
256
|
-
|
|
261
|
+
// v1.70 T2: environment identity beside the graph/branch identity — running tickmarkr version,
|
|
262
|
+
// loaded-config hash, and the probed CLI version of each adapter holding a channel in the run,
|
|
263
|
+
// gathered through the existing probe/config-load paths (no second mechanism).
|
|
264
|
+
const environment = runEnvironment(cfg, channels, health);
|
|
265
|
+
journal.append("run-start", undefined, { pid: process.pid, baseRef, commands, channels: channels.map(channelKey), branch, graphDefinitionHash: graphDefinitionHash(graph), mode: rm.mode.mode, modeSource: rm.source, environment, ...(prior ? { supersedes: prior.runId } : {}) }); // graphDefinitionHash: T3 engagement identity (status+resume share it); pid: v1.13 (VIS-11) liveness; mode/modeSource: v1.51 T2; supersedes: v1.53 T5
|
|
257
266
|
runStarted = true;
|
|
258
267
|
// v1.53 T5: mark the prior run AFTER this run's run-start exists, so the prior journal never
|
|
259
268
|
// names a successor that has no journal. Append-only — the prior journal is never rewritten.
|
|
@@ -410,6 +419,14 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
410
419
|
if (rs)
|
|
411
420
|
journal.append("resume-restore", t.id, { attempts: rs.attempts, tried: [...tried], assignment });
|
|
412
421
|
const badReviewers = []; // v1.1: reviewer channels that produced unparseable output for this task
|
|
422
|
+
// v1.70 T5 (review-convergence): failed review rounds this task has drawn, counted from the per-task
|
|
423
|
+
// review history already in the journal — the SAME review gate-result stream onGate reads to grow the
|
|
424
|
+
// reviewer-exclusion list (badReviewers), never a second parallel counter. request-changes rounds do
|
|
425
|
+
// not exclude their reviewer (a fix is re-checked by the same seat), so their count lives here in the
|
|
426
|
+
// shared history rather than in badReviewers' garbage-only exclusion subset.
|
|
427
|
+
const reviewRoundsDrawn = () => journal.read().filter((e) => e.taskId === t.id && e.event === "gate-result" &&
|
|
428
|
+
e.data.gate === "review" &&
|
|
429
|
+
e.data.pass === false).length;
|
|
413
430
|
let feedback = "";
|
|
414
431
|
let ladderIdx = 0;
|
|
415
432
|
let modeFallbackNoted = false; // v1.2: journal the interactive→print fallback once per task, not per attempt
|
|
@@ -510,6 +527,15 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
510
527
|
await park(t, `attempt cap (${MAX_ATTEMPTS}) reached`, "attempt-cap", assignment, attempt, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
511
528
|
return;
|
|
512
529
|
}
|
|
530
|
+
// v1.70 T5: a task that has already drawn REVIEW_ROUND_CAP request-changes review rounds parks for
|
|
531
|
+
// a human decision instead of dispatching another round. Condition on the review history (data),
|
|
532
|
+
// never the code path — and let an operator's approval (the human decision the cap asked for)
|
|
533
|
+
// release it, mirroring the humanGate guard's condition-on-approval precedent. Rounds only accrue
|
|
534
|
+
// after attempt 0, so the guard skips the journal read on the happy path.
|
|
535
|
+
if (attempt > 0 && !approved.has(t.id) && reviewRoundsDrawn() >= REVIEW_ROUND_CAP) {
|
|
536
|
+
await park(t, `review round cap (${REVIEW_ROUND_CAP}) reached — request-changes reviews not converging; a human should decide`, "gate-fail", assignment, attempt, startMs, gateFails, consults, tokens, metered, retryMode);
|
|
537
|
+
return;
|
|
538
|
+
}
|
|
513
539
|
// OBS-57: a demoted channel must not be re-dispatched on consult retry or provider requeue.
|
|
514
540
|
if (demotedChannels.has(channelKey(assignment))) {
|
|
515
541
|
const next = nextChannel(assignment, t, cfg, channels, tried, profile, demotedChannels);
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
import type { AuthHealth, BillingChannel } from "../adapters/types.js";
|
|
2
|
+
import type { TickmarkrConfig } from "../config/config.js";
|
|
3
|
+
export interface RunEnvironment {
|
|
4
|
+
tickmarkrVersion: string;
|
|
5
|
+
configHash: string;
|
|
6
|
+
adapterVersions: Record<string, string>;
|
|
7
|
+
}
|
|
8
|
+
export declare const UNKNOWN_ADAPTER_VERSION = "unknown";
|
|
9
|
+
export declare function tickmarkrVersion(): string;
|
|
10
|
+
export declare function configHash(cfg: TickmarkrConfig): string;
|
|
11
|
+
export declare function adapterVersions(channels: BillingChannel[], health: Record<string, AuthHealth>): Record<string, string>;
|
|
12
|
+
export declare function runEnvironment(cfg: TickmarkrConfig, channels: BillingChannel[], health: Record<string, AuthHealth>): RunEnvironment;
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
import { createHash } from "node:crypto";
|
|
2
|
+
import { readFileSync } from "node:fs";
|
|
3
|
+
import { dirname, join } from "node:path";
|
|
4
|
+
import { fileURLToPath } from "node:url";
|
|
5
|
+
// An adapter whose version probe failed is recorded, not dropped — "unknown", never a fabricated string.
|
|
6
|
+
export const UNKNOWN_ADAPTER_VERSION = "unknown";
|
|
7
|
+
// Same package.json read as src/cli/commands/version.ts (one resolution pattern, two consumers).
|
|
8
|
+
const pkgPath = join(dirname(fileURLToPath(import.meta.url)), "../../package.json");
|
|
9
|
+
export function tickmarkrVersion() {
|
|
10
|
+
const { version: v } = JSON.parse(readFileSync(pkgPath, "utf8"));
|
|
11
|
+
return v;
|
|
12
|
+
}
|
|
13
|
+
// Key order in a parsed config is an accident of the schema/merge layers, so the hash canonicalizes
|
|
14
|
+
// first: object keys sorted recursively, undefined dropped (JSON semantics), array order preserved.
|
|
15
|
+
function stableStringify(v) {
|
|
16
|
+
if (v === null || typeof v !== "object")
|
|
17
|
+
return JSON.stringify(v);
|
|
18
|
+
if (Array.isArray(v))
|
|
19
|
+
return `[${v.map(stableStringify).join(",")}]`;
|
|
20
|
+
const entries = Object.entries(v)
|
|
21
|
+
.filter(([, val]) => val !== undefined)
|
|
22
|
+
.sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0));
|
|
23
|
+
return `{${entries.map(([k, val]) => `${JSON.stringify(k)}:${stableStringify(val)}`).join(",")}}`;
|
|
24
|
+
}
|
|
25
|
+
// sha256 truncated to 16 hex — the graphDefinitionHash convention (stable, grep-friendly).
|
|
26
|
+
export function configHash(cfg) {
|
|
27
|
+
return createHash("sha256").update(stableStringify(cfg)).digest("hex").slice(0, 16);
|
|
28
|
+
}
|
|
29
|
+
// One entry per adapter with a channel in the run (not per channel). The version is whatever the
|
|
30
|
+
// adapter's own probe recorded in health; a missing/undefined probe result becomes "unknown".
|
|
31
|
+
export function adapterVersions(channels, health) {
|
|
32
|
+
const out = {};
|
|
33
|
+
for (const c of channels) {
|
|
34
|
+
if (!(c.adapter in out))
|
|
35
|
+
out[c.adapter] = health[c.adapter]?.version ?? UNKNOWN_ADAPTER_VERSION;
|
|
36
|
+
}
|
|
37
|
+
return out;
|
|
38
|
+
}
|
|
39
|
+
export function runEnvironment(cfg, channels, health) {
|
|
40
|
+
return { tickmarkrVersion: tickmarkrVersion(), configHash: configHash(cfg), adapterVersions: adapterVersions(channels, health) };
|
|
41
|
+
}
|