tickmarkr 1.81.0 → 1.84.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -4
- package/dist/brand.d.ts +4 -1
- package/dist/brand.js +46 -7
- package/dist/cli/commands/approve.js +53 -8
- package/dist/cli/commands/plan.js +17 -2
- package/dist/cli/commands/report.js +15 -6
- package/dist/cli/commands/ui.js +13 -1
- package/dist/cli/index.d.ts +1 -1
- package/dist/cli/index.js +1 -1
- package/dist/compile/collateral.d.ts +19 -0
- package/dist/compile/collateral.js +90 -0
- package/dist/compile/index.js +18 -4
- package/dist/drivers/herdr.d.ts +1 -0
- package/dist/drivers/herdr.js +19 -0
- package/dist/drivers/types.d.ts +1 -0
- package/dist/gates/acceptance.js +50 -8
- package/dist/gates/llm.js +34 -19
- package/dist/gates/review.d.ts +9 -1
- package/dist/gates/review.js +173 -7
- package/dist/gates/run-gates.d.ts +1 -0
- package/dist/gates/run-gates.js +18 -1
- package/dist/report/bundle.d.ts +6 -1
- package/dist/report/bundle.js +13 -3
- package/dist/run/daemon.d.ts +5 -0
- package/dist/run/daemon.js +188 -27
- package/dist/run/journal.d.ts +5 -0
- package/dist/run/journal.js +71 -1
- package/dist/tui/cockpit/capture.d.ts +107 -1
- package/dist/tui/cockpit/capture.js +287 -10
- package/dist/tui/cockpit/components.d.ts +40 -47
- package/dist/tui/cockpit/components.js +66 -50
- package/dist/tui/cockpit/derive.d.ts +73 -1
- package/dist/tui/cockpit/derive.js +211 -15
- package/dist/tui/cockpit/keys.d.ts +114 -19
- package/dist/tui/cockpit/keys.js +413 -44
- package/dist/tui/cockpit/layout.d.ts +231 -0
- package/dist/tui/cockpit/layout.js +307 -0
- package/dist/tui/cockpit/live.d.ts +65 -3
- package/dist/tui/cockpit/live.js +418 -48
- package/dist/tui/cockpit/pointer.d.ts +261 -0
- package/dist/tui/cockpit/pointer.js +610 -0
- package/dist/tui/cockpit/run-cockpit.d.ts +93 -6
- package/dist/tui/cockpit/run-cockpit.js +805 -164
- package/dist/tui/cockpit/setup-cockpit.d.ts +146 -1
- package/dist/tui/cockpit/setup-cockpit.js +337 -5
- package/dist/tui/cockpit/views.d.ts +72 -0
- package/dist/tui/cockpit/views.js +56 -0
- package/dist/tui/cockpit/width.d.ts +105 -0
- package/dist/tui/cockpit/width.js +276 -0
- package/fixtures/speckit-sample/tasks.md +2 -2
- package/package.json +3 -1
package/dist/gates/acceptance.js
CHANGED
|
@@ -3,7 +3,7 @@ import { channelKey, shq } from "../adapters/types.js";
|
|
|
3
3
|
import { DEFAULT_DIFF_CAP } from "../config/config.js";
|
|
4
4
|
import { renderAcceptanceItem } from "../graph/schema.js";
|
|
5
5
|
import { sh } from "../run/git.js";
|
|
6
|
-
import { checkDiffCap, fetchTaskDiff } from "./review.js";
|
|
6
|
+
import { checkDiffCap, fetchTaskDiff, isProtectedEvidence, setAsideReceiptPath } from "./review.js";
|
|
7
7
|
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
|
|
8
8
|
// Fable F4: acceptance judge shares review's 900s timeout — 300s default killed frontier judges on cap-sized diffs.
|
|
9
9
|
const JUDGE_TIMEOUT_MS = 900_000;
|
|
@@ -172,22 +172,51 @@ function compressLines(lines) {
|
|
|
172
172
|
}
|
|
173
173
|
return parts.join(", ");
|
|
174
174
|
}
|
|
175
|
+
const DIFF_SECTIONS = /(?=^diff --git )/m;
|
|
176
|
+
// the a-side path of a whole-file deletion section, or null if this section is not one.
|
|
177
|
+
function deletedPath(section) {
|
|
178
|
+
if (!/^deleted file mode /m.test(section))
|
|
179
|
+
return null;
|
|
180
|
+
const oldPath = /^--- (.+)$/m.exec(section)?.[1]
|
|
181
|
+
?? /^Binary files (.+) and \/dev\/null differ$/m.exec(section)?.[1];
|
|
182
|
+
if (!oldPath || oldPath === "/dev/null")
|
|
183
|
+
return null;
|
|
184
|
+
const unquoted = oldPath.startsWith('"') && oldPath.endsWith('"') ? oldPath.slice(1, -1) : oldPath;
|
|
185
|
+
return unquoted.replace(/^a\//, "");
|
|
186
|
+
}
|
|
175
187
|
// OBS-134: whole-file deletions are already fully described by their path. Sending every removed line
|
|
176
188
|
// spends the cap and judge context on content that cannot exist after the change. Added and modified
|
|
177
189
|
// sections pass through byte-for-byte, so their anti-flooding budget is unchanged.
|
|
190
|
+
// v1.82 T1 clause 5: two exemptions, because this filter triggers on the very `deleted file mode` line
|
|
191
|
+
// the set-aside preserves. Protected evidence (the frozen anchors, the captured journals) bypasses the
|
|
192
|
+
// collapse so its deletion half reaches the measured text and the judge complete; and a section already
|
|
193
|
+
// set aside is never reduced a second time, or this filter would erase the receipt that replaced it.
|
|
178
194
|
function judgeRelevantDiff(diff) {
|
|
179
|
-
return diff.split(
|
|
180
|
-
if (
|
|
195
|
+
return diff.split(DIFF_SECTIONS).map((section) => {
|
|
196
|
+
if (setAsideReceiptPath(section))
|
|
181
197
|
return section;
|
|
182
|
-
const
|
|
183
|
-
|
|
184
|
-
if (!oldPath || oldPath === "/dev/null")
|
|
198
|
+
const path = deletedPath(section);
|
|
199
|
+
if (!path || isProtectedEvidence(path))
|
|
185
200
|
return section;
|
|
186
|
-
const unquoted = oldPath.startsWith('"') && oldPath.endsWith('"') ? oldPath.slice(1, -1) : oldPath;
|
|
187
|
-
const path = unquoted.replace(/^a\//, "");
|
|
188
201
|
return `deleted file: ${path}\n`;
|
|
189
202
|
}).join("");
|
|
190
203
|
}
|
|
204
|
+
// v1.82 T1 clause 7: the two shapes this task creates carry no changed hunk, so a judge asked to cite a
|
|
205
|
+
// changed line has nothing to cite — three review rounds died here. Each yields ONE citable operation
|
|
206
|
+
// fact, `{path, line: 0}`, read back from the judged text itself. Only these RECORDED facts make line 0
|
|
207
|
+
// citable; a zero line on any other path is still rejected as fabricated. Ordinary whole-file deletions
|
|
208
|
+
// stay uncitable deliberately: the relevance filter collapses them before this runs, and repairing that
|
|
209
|
+
// pre-existing limitation is outside this contract.
|
|
210
|
+
const OPERATION_FACT_LINE = 0;
|
|
211
|
+
function operationFactPaths(diff) {
|
|
212
|
+
return diff.split(DIFF_SECTIONS).flatMap((section) => {
|
|
213
|
+
const receipt = setAsideReceiptPath(section);
|
|
214
|
+
if (receipt)
|
|
215
|
+
return [receipt];
|
|
216
|
+
const path = deletedPath(section);
|
|
217
|
+
return path && isProtectedEvidence(path) ? [path] : [];
|
|
218
|
+
});
|
|
219
|
+
}
|
|
191
220
|
// A citation is valid evidence iff the cited line falls inside a changed hunk of the cited file (OBS-129:
|
|
192
221
|
// changed-hunk span, not exact-added-line). A legacy free-text quote (string) keeps v1.64's substring
|
|
193
222
|
// check — non-empty and present somewhere in the diff.
|
|
@@ -252,6 +281,16 @@ export async function acceptanceGate(task, worktree, baseRef, judge, via, opts =
|
|
|
252
281
|
// OBS-129: give the judge the exact citable new-file line numbers per changed file, so it grounds each
|
|
253
282
|
// citation in a real changed line instead of miscounting. Validated below against this same `changed`.
|
|
254
283
|
const changed = changedLinesByFile(diff);
|
|
284
|
+
// clause 7: fold the recorded operation facts into the same index the validator checks, so a change
|
|
285
|
+
// made only of set-aside captures or removed protected evidence still yields citable evidence.
|
|
286
|
+
for (const path of operationFactPaths(diff)) {
|
|
287
|
+
let set = changed.get(path);
|
|
288
|
+
if (!set) {
|
|
289
|
+
set = new Set();
|
|
290
|
+
changed.set(path, set);
|
|
291
|
+
}
|
|
292
|
+
set.add(OPERATION_FACT_LINE);
|
|
293
|
+
}
|
|
255
294
|
const citable = [...changed.entries()].map(([p, set]) => `- ${p}: ${compressLines([...set])}`).join("\n");
|
|
256
295
|
const nonce = generateVerdictNonce();
|
|
257
296
|
const prompt = `TICKMARKR-JUDGE
|
|
@@ -274,6 +313,9 @@ ${diff}
|
|
|
274
313
|
|
|
275
314
|
## Citable evidence lines (new-file line numbers inside the changed hunks above)
|
|
276
315
|
${citable || "(the diff changes no lines)"}
|
|
316
|
+
Line 0 is a recorded file-operation fact, not a hunk line: the diff states what happened to that file
|
|
317
|
+
(a set-aside regenerable capture, or a removed protected file) without carrying its content. Cite it as
|
|
318
|
+
{"path": "<that path>", "line": 0} when a criterion is about that file.
|
|
277
319
|
|
|
278
320
|
${verdictNonceLine(nonce)}
|
|
279
321
|
|
package/dist/gates/llm.js
CHANGED
|
@@ -180,26 +180,41 @@ export function dewrapPaneVerdict(out, nonce) {
|
|
|
180
180
|
if (!out.includes(nonce))
|
|
181
181
|
return out;
|
|
182
182
|
const lines = out.split("\n");
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
183
|
+
// OBS-209: EVERY brace-start is a candidate, scanned newest-first. findIndex took only the first,
|
|
184
|
+
// so any earlier line beginning with `{` — a quoted snippet, a lone brace in the reviewer's own
|
|
185
|
+
// reasoning — captured the scan, and the real verdict below it was unreachable no matter how far
|
|
186
|
+
// the join extended. Measured on run-20260728-110135 T1: kimi's nonce-bound APPROVAL sat at line
|
|
187
|
+
// 382 behind a bare `{` at line 189, so a passing review was recorded `malformed-verdict` and
|
|
188
|
+
// parked the task. Newest-first matches extractJson, which takes the LAST balanced object.
|
|
189
|
+
const starts = [];
|
|
190
|
+
for (let i = 0; i < lines.length; i++) {
|
|
191
|
+
if (/^\s*(?:[•*-]\s+)?\{/.test(lines[i]))
|
|
192
|
+
starts.push(i);
|
|
193
|
+
}
|
|
194
|
+
for (let si = starts.length - 1; si >= 0; si--) {
|
|
195
|
+
const start = starts[si];
|
|
196
|
+
for (let end = start; end < lines.length; end++) {
|
|
197
|
+
const joined = lines
|
|
198
|
+
.slice(start, end + 1)
|
|
199
|
+
.map((line, i) => (i === 0 ? line.replace(/^\s*(?:[•*-]\s+)?/, "") : line.replace(/^\s+/, "")))
|
|
200
|
+
.join("")
|
|
201
|
+
.trimEnd();
|
|
202
|
+
// ponytail: no verdict is a megabyte; abandon a runaway candidate rather than rejoin the
|
|
203
|
+
// whole transcript once per brace-start. Raise the ceiling if a real verdict ever exceeds it.
|
|
204
|
+
if (joined.length > 1_000_000)
|
|
205
|
+
break;
|
|
206
|
+
if (!joined.endsWith("}"))
|
|
207
|
+
continue;
|
|
208
|
+
try {
|
|
209
|
+
const parsed = JSON.parse(joined);
|
|
210
|
+
// the nonce IS the acceptance test — never reconstruct a verdict this call did not ask for
|
|
211
|
+
if (parsed && typeof parsed === "object" && parsed.nonce === nonce) {
|
|
212
|
+
return `${out}\n${joined}`;
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
catch {
|
|
216
|
+
/* not yet a complete object — keep extending within the bounded region */
|
|
199
217
|
}
|
|
200
|
-
}
|
|
201
|
-
catch {
|
|
202
|
-
/* not yet a complete object — keep extending within the bounded region */
|
|
203
218
|
}
|
|
204
219
|
}
|
|
205
220
|
return out;
|
package/dist/gates/review.d.ts
CHANGED
|
@@ -20,6 +20,13 @@ export interface ReviewVerdict {
|
|
|
20
20
|
body: string;
|
|
21
21
|
}>;
|
|
22
22
|
}
|
|
23
|
+
export declare const REGENERABLE_CAPTURE_PATHS: readonly string[];
|
|
24
|
+
export declare const PROTECTED_EVIDENCE_PREFIXES: readonly ["tests/fixtures/cockpit/anchors/", "tests/fixtures/cockpit/sources/", "tests/fixtures/cockpit/colour/sources/"];
|
|
25
|
+
export declare function isProtectedEvidence(path: string): boolean;
|
|
26
|
+
/** The `{path}` a set-aside receipt names, or null if this section carries no receipt. */
|
|
27
|
+
export declare function setAsideReceiptPath(section: string): string | null;
|
|
28
|
+
/** Replace the content of every section confined to the regenerable frame corpora with a receipt. */
|
|
29
|
+
export declare function setAsideRegenerableCaptures(diff: string): string;
|
|
23
30
|
export declare function fetchTaskDiff(worktree: string, baseRef: string): Promise<{
|
|
24
31
|
full: string;
|
|
25
32
|
forCap: string;
|
|
@@ -30,4 +37,5 @@ export declare function diffCapParkReason(results: GateResult[]): string | null;
|
|
|
30
37
|
export declare function modelId(model: string): string;
|
|
31
38
|
export declare function pickReviewer(author: Assignment, channels: BillingChannel[], exclude?: string[], // v1.1 failover: reviewer channels that already produced garbage for this task
|
|
32
39
|
prefer?: string[]): BillingChannel | null;
|
|
33
|
-
export
|
|
40
|
+
export type ReviewUnparseableCause = "empty-output" | "no-verdict" | "malformed-verdict";
|
|
41
|
+
export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string): Promise<GateResult>;
|
package/dist/gates/review.js
CHANGED
|
@@ -1,8 +1,11 @@
|
|
|
1
|
+
import { writeFileSync } from "node:fs";
|
|
2
|
+
import { join } from "node:path";
|
|
1
3
|
import { channelKey } from "../adapters/types.js";
|
|
2
4
|
import { DEFAULT_DIFF_CAP, TIER_RANK } from "../config/config.js";
|
|
3
5
|
import { renderAcceptanceItem } from "../graph/schema.js";
|
|
4
6
|
import { getAdapter } from "../adapters/registry.js";
|
|
5
7
|
import { shOk } from "../run/git.js";
|
|
8
|
+
import { redactSecrets } from "../run/redact.js";
|
|
6
9
|
import { marginalCostRank } from "../route/router.js";
|
|
7
10
|
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
|
|
8
11
|
// legacy flat `issues` shape — every issue blocks; the approve flag must agree with the list.
|
|
@@ -65,9 +68,152 @@ function classifyReviewFindings(findings) {
|
|
|
65
68
|
// OBS-48: cap on zero-context diff bytes (git diff -U0), not context-padded full diff — scattered
|
|
66
69
|
// one-line hunks no longer trip at ~370 diff-bytes per changed line. Full diff still goes to the judge.
|
|
67
70
|
const DIFF_CAP_REMEDY = "split the task, or raise gates.diffCap";
|
|
71
|
+
// v1.82 T1 — the cap bounds what a READER MUST READ, not what a run must write. A regeneration of the
|
|
72
|
+
// frame corpora is ~134KB of `-U0` measurement before a source line changes, and nobody reads it: those
|
|
73
|
+
// frames are asserted byte-for-byte by the corpus tests. Counting them is the category error this
|
|
74
|
+
// removes. The two artifacts stay two (OBS-48: the cap measures -U0, the reader receives the full diff);
|
|
75
|
+
// the exclusion is applied to both, identically, right here so BOTH measuring gates inherit it.
|
|
76
|
+
//
|
|
77
|
+
// Clause 1 — membership is an EXACT PATH match against the shipped capture manifest, the same lists the
|
|
78
|
+
// regeneration path itself uses. Location, directory depth and file extension confer nothing: an
|
|
79
|
+
// unmanifested file sitting beside real frames is measured and shown in full. (The anchors deliberately
|
|
80
|
+
// share basenames with the frames; only the full path separates the oracle from its regenerable twin.)
|
|
81
|
+
//
|
|
82
|
+
// The members are LISTED here rather than imported from the manifest module, for one measured reason:
|
|
83
|
+
// that module is the Ink/React renderer, and importing it puts the whole TUI in every gate's module
|
|
84
|
+
// graph — which also memoises chalk's colour level at import time and turns the fleet suite red. A
|
|
85
|
+
// drift test in tests/gates/diff-cap.test.ts asserts this list is exactly GOLDEN_FRAME_CASES +
|
|
86
|
+
// COLOUR_FRAME_CASES and that every entry exists on disk, so it is a copy that cannot drift rather
|
|
87
|
+
// than a second source of truth: add or rename a frame case and that test goes red until this matches.
|
|
88
|
+
export const REGENERABLE_CAPTURE_PATHS = [
|
|
89
|
+
"tests/fixtures/cockpit/frames/run.width-stacked.80x24.txt",
|
|
90
|
+
"tests/fixtures/cockpit/frames/run.width-folded-keys.100x24.txt",
|
|
91
|
+
"tests/fixtures/cockpit/frames/run.width-three-column.140x24.txt",
|
|
92
|
+
"tests/fixtures/cockpit/frames/run.height-14.140x14.txt",
|
|
93
|
+
"tests/fixtures/cockpit/frames/run.height-18.140x18.txt",
|
|
94
|
+
"tests/fixtures/cockpit/frames/run.height-24.140x24.txt",
|
|
95
|
+
"tests/fixtures/cockpit/frames/run.height-40.140x40.txt",
|
|
96
|
+
"tests/fixtures/cockpit/frames/run.no-colour.140x24.txt",
|
|
97
|
+
"tests/fixtures/cockpit/frames/run.non-tty.140x24.txt",
|
|
98
|
+
"tests/fixtures/cockpit/frames/run.ci.140x24.txt",
|
|
99
|
+
"tests/fixtures/cockpit/frames/setup.width-stacked.80x24.txt",
|
|
100
|
+
"tests/fixtures/cockpit/frames/setup.width-folded-keys.100x24.txt",
|
|
101
|
+
"tests/fixtures/cockpit/frames/setup.width-three-column.140x24.txt",
|
|
102
|
+
"tests/fixtures/cockpit/frames/setup.height-14.140x14.txt",
|
|
103
|
+
"tests/fixtures/cockpit/frames/setup.height-18.140x18.txt",
|
|
104
|
+
"tests/fixtures/cockpit/frames/setup.height-24.140x24.txt",
|
|
105
|
+
"tests/fixtures/cockpit/frames/setup.height-40.140x40.txt",
|
|
106
|
+
"tests/fixtures/cockpit/frames/setup.no-colour.140x24.txt",
|
|
107
|
+
"tests/fixtures/cockpit/frames/setup.non-tty.140x24.txt",
|
|
108
|
+
"tests/fixtures/cockpit/frames/setup.ci.140x24.txt",
|
|
109
|
+
"tests/fixtures/cockpit/colour/run-20260718-000943.colour.140x24.txt",
|
|
110
|
+
"tests/fixtures/cockpit/colour/run-20260718-000943.no-colour.140x24.txt",
|
|
111
|
+
"tests/fixtures/cockpit/colour/run-20260725-025004.interrupted.colour.140x24.txt",
|
|
112
|
+
];
|
|
113
|
+
const CAPTURE_MANIFEST = new Set(REGENERABLE_CAPTURE_PATHS);
|
|
114
|
+
// Clause 2 — the frozen appearance anchors and the captured engagement journals are NEVER set aside and
|
|
115
|
+
// are exempt from every other reduction too (clause 5): they are reviewed, immutable evidence. The
|
|
116
|
+
// anchors are the oracle this milestone declares, and the fixture law bans editing a captured journal to
|
|
117
|
+
// satisfy an assertion — so both keep counting toward the cap and keep reaching a reader verbatim.
|
|
118
|
+
export const PROTECTED_EVIDENCE_PREFIXES = [
|
|
119
|
+
"tests/fixtures/cockpit/anchors/",
|
|
120
|
+
"tests/fixtures/cockpit/sources/",
|
|
121
|
+
"tests/fixtures/cockpit/colour/sources/",
|
|
122
|
+
];
|
|
123
|
+
export function isProtectedEvidence(path) {
|
|
124
|
+
return PROTECTED_EVIDENCE_PREFIXES.some((prefix) => path.startsWith(prefix));
|
|
125
|
+
}
|
|
126
|
+
const SET_ASIDE_RECEIPT = /^set aside: regenerable capture (.+?) — \d+ bytes withheld\b/m;
|
|
127
|
+
/** The `{path}` a set-aside receipt names, or null if this section carries no receipt. */
|
|
128
|
+
export function setAsideReceiptPath(section) {
|
|
129
|
+
return SET_ASIDE_RECEIPT.exec(section)?.[1] ?? null;
|
|
130
|
+
}
|
|
131
|
+
// `--- a/x` / `+++ b/x` → "x"; "/dev/null" → null, which is the ABSENCE of a side, not a membership
|
|
132
|
+
// failure (clause 3). git quotes paths carrying specials, so unquote before stripping the a/ b/ prefix.
|
|
133
|
+
function diffSidePath(raw) {
|
|
134
|
+
const v = raw.trim();
|
|
135
|
+
if (v === "/dev/null")
|
|
136
|
+
return null;
|
|
137
|
+
const unquoted = v.startsWith('"') && v.endsWith('"') ? v.slice(1, -1) : v;
|
|
138
|
+
return unquoted.replace(/^[ab]\//, "");
|
|
139
|
+
}
|
|
140
|
+
// The `--- a/x` / `+++ b/x` header pair and the hunk lines below it. Clause 4 — null when the section
|
|
141
|
+
// carries NO content hunk at all (a mode-only change, a pure rename, a binary marker): such a section is
|
|
142
|
+
// left exactly as git wrote it rather than handed a manufactured receipt.
|
|
143
|
+
function parseSection(section) {
|
|
144
|
+
const lines = section.split("\n");
|
|
145
|
+
const minus = lines.findIndex((l) => l.startsWith("--- "));
|
|
146
|
+
if (minus === -1 || !lines[minus + 1]?.startsWith("+++ "))
|
|
147
|
+
return null;
|
|
148
|
+
const body = lines.slice(minus + 2);
|
|
149
|
+
if (!body.some((l) => l.startsWith("@@ ")))
|
|
150
|
+
return null;
|
|
151
|
+
return { lines, minus, sides: [diffSidePath(lines[minus].slice(4)), diffSidePath(lines[minus + 1].slice(4))], body };
|
|
152
|
+
}
|
|
153
|
+
// The content a one-sided section carries: its hunk lines with the sign stripped, keeping git's
|
|
154
|
+
// `` markers so a trailing-newline difference still reads as a content
|
|
155
|
+
// difference. Hunk headers are dropped — a delete and an add of the same bytes never share them.
|
|
156
|
+
function hunkPayload(body, sign) {
|
|
157
|
+
return body.filter((l) => l.startsWith(sign) || l.startsWith("\\")).map((l) => (l[0] === sign ? l.slice(1) : l)).join("\n");
|
|
158
|
+
}
|
|
159
|
+
// Clause 4 — a hunk is NOT proof of a content change. Git spells a KIND change (regular file ⇄ symlink)
|
|
160
|
+
// as a delete plus an add of the SAME path, each carrying a hunk, so when the regular file's bytes are
|
|
161
|
+
// exactly the link target both halves carry identical payloads: the kind changed and the content did
|
|
162
|
+
// not. Nothing was withheld, so neither half earns a receipt — and neither half can see the other, so
|
|
163
|
+
// the pairing is found across sections before any one of them is set aside.
|
|
164
|
+
function kindOnlyPaths(sections) {
|
|
165
|
+
const removed = new Map();
|
|
166
|
+
const added = new Map();
|
|
167
|
+
for (const section of sections) {
|
|
168
|
+
const parsed = parseSection(section);
|
|
169
|
+
if (!parsed)
|
|
170
|
+
continue;
|
|
171
|
+
const [a, b] = parsed.sides;
|
|
172
|
+
if (a && !b)
|
|
173
|
+
removed.set(a, hunkPayload(parsed.body, "-"));
|
|
174
|
+
else if (b && !a)
|
|
175
|
+
added.set(b, hunkPayload(parsed.body, "+"));
|
|
176
|
+
}
|
|
177
|
+
return new Set([...removed].filter(([path, payload]) => added.get(path) === payload).map(([path]) => path));
|
|
178
|
+
}
|
|
179
|
+
function setAsideSection(section, kindOnly) {
|
|
180
|
+
const parsed = parseSection(section);
|
|
181
|
+
if (!parsed)
|
|
182
|
+
return section;
|
|
183
|
+
const { lines, minus, sides } = parsed;
|
|
184
|
+
// Clause 3 — the test is over the sides that name a real file. Requiring BOTH sides to be members
|
|
185
|
+
// makes every corpus addition (a-side /dev/null) and deletion (b-side /dev/null) ineligible, which is
|
|
186
|
+
// exactly the frames this milestone adds. A rename crossing the boundary in either direction has two
|
|
187
|
+
// real sides and one of them is not a member, so it stays whole — `rename from` line included.
|
|
188
|
+
const named = sides.filter((p) => p !== null);
|
|
189
|
+
if (!named.length || !named.every((p) => CAPTURE_MANIFEST.has(p)))
|
|
190
|
+
return section;
|
|
191
|
+
// Clause 4 — both halves of a content-identical kind change are left exactly as git wrote them. A
|
|
192
|
+
// kind change that DID move bytes is an ordinary content change and is set aside like any other.
|
|
193
|
+
if (named.some((p) => kindOnly.has(p)))
|
|
194
|
+
return section;
|
|
195
|
+
const withheld = lines.slice(minus).join("\n");
|
|
196
|
+
// Clause 6 — the claimed size is the UTF-8 BYTE length of what was withheld (the file headers and
|
|
197
|
+
// hunks this receipt replaces), never a JavaScript string length: box-drawing frames make the two
|
|
198
|
+
// disagree. And the receipt itself is part of the measured artifact, so N set-aside sections can never
|
|
199
|
+
// measure as nothing while producing an arbitrarily large reader payload.
|
|
200
|
+
const receipt = `set aside: regenerable capture ${named.at(-1)} — ${Buffer.byteLength(withheld, "utf8")} bytes withheld (regenerable frame corpus: asserted byte-for-byte by the corpus tests, never read)`;
|
|
201
|
+
// Clause 4 — everything git said happened survives verbatim: old/new mode, new file mode, deleted file
|
|
202
|
+
// mode, similarity index, rename from/to, index. Only content is replaced, so a deletion is never
|
|
203
|
+
// presented as a file that still exists and an addition is never presented as a modification.
|
|
204
|
+
return `${lines.slice(0, minus).join("\n")}\n${receipt}\n`;
|
|
205
|
+
}
|
|
206
|
+
/** Replace the content of every section confined to the regenerable frame corpora with a receipt. */
|
|
207
|
+
export function setAsideRegenerableCaptures(diff) {
|
|
208
|
+
if (!diff.includes("diff --git "))
|
|
209
|
+
return diff;
|
|
210
|
+
const sections = diff.split(/(?=^diff --git )/m);
|
|
211
|
+
const kindOnly = kindOnlyPaths(sections);
|
|
212
|
+
return sections.map((s) => setAsideSection(s, kindOnly)).join("");
|
|
213
|
+
}
|
|
68
214
|
export async function fetchTaskDiff(worktree, baseRef) {
|
|
69
|
-
const full = await shOk(`git diff '${baseRef}..HEAD'`, worktree);
|
|
70
|
-
const forCap = await shOk(`git diff -U0 '${baseRef}..HEAD'`, worktree);
|
|
215
|
+
const full = setAsideRegenerableCaptures(await shOk(`git diff '${baseRef}..HEAD'`, worktree));
|
|
216
|
+
const forCap = setAsideRegenerableCaptures(await shOk(`git diff -U0 '${baseRef}..HEAD'`, worktree));
|
|
71
217
|
return { full, forCap };
|
|
72
218
|
}
|
|
73
219
|
export function checkDiffCap(gate, measured, cap, prefix = "") {
|
|
@@ -119,15 +265,20 @@ prefer = []) {
|
|
|
119
265
|
.filter((c) => c.vendor !== authorChannel.vendor && modelId(c.model) !== modelId(author.model) && !exclude.includes(channelKey(c)))
|
|
120
266
|
.sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b))[0] ?? null);
|
|
121
267
|
}
|
|
122
|
-
export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers
|
|
268
|
+
export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
|
|
269
|
+
// OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
|
|
270
|
+
// direct tests) skips persistence and changes nothing else.
|
|
271
|
+
artifactDir) {
|
|
123
272
|
if (task.complexity < cfg.review.complexityThreshold) {
|
|
124
273
|
return { gate: "review", pass: true, details: `skipped — complexity ${task.complexity} < threshold ${cfg.review.complexityThreshold}`, meta: { skipped: true } };
|
|
125
274
|
}
|
|
126
275
|
const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? []);
|
|
127
276
|
if (!reviewer) {
|
|
277
|
+
// meta.noEligibleReviewer lets run-gates' review-retry keep the ORIGINAL unparseable result when
|
|
278
|
+
// the retry finds no second seat — a truthful cause beats a synthetic no-reviewer failure.
|
|
128
279
|
return cfg.review.required
|
|
129
|
-
? { gate: "review", pass: false, details: "no cross-vendor reviewer available (diversity rule); set review.required:false to waive" }
|
|
130
|
-
: { gate: "review", pass: true, details: "WARNING: no cross-vendor reviewer available — review waived by config" };
|
|
280
|
+
? { gate: "review", pass: false, details: "no cross-vendor reviewer available (diversity rule); set review.required:false to waive", meta: { noEligibleReviewer: true } }
|
|
281
|
+
: { gate: "review", pass: true, details: "WARNING: no cross-vendor reviewer available — review waived by config", meta: { noEligibleReviewer: true } };
|
|
131
282
|
}
|
|
132
283
|
const { full: diff, forCap } = await fetchTaskDiff(worktree, baseRef);
|
|
133
284
|
const diffCap = cfg.gates.diffCap ?? DEFAULT_DIFF_CAP;
|
|
@@ -173,11 +324,26 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
173
324
|
const findings = v && Array.isArray(v.findings) ? v.findings : null;
|
|
174
325
|
// findings decides the verdict on its own; the legacy path still needs approve + issues to parse.
|
|
175
326
|
if (!v || (findings === null && (typeof v.approve !== "boolean" || !Array.isArray(v.issues)))) {
|
|
327
|
+
// OBS-196: name the cause and persist the raw bytes — a ruled-on "unparseable" without its
|
|
328
|
+
// evidence cannot be audited, and a cutoff must never be indistinguishable from a parse defect.
|
|
329
|
+
const cause = raw.trim().length === 0
|
|
330
|
+
? "empty-output"
|
|
331
|
+
: !raw.includes(nonce) ? "no-verdict" : "malformed-verdict";
|
|
332
|
+
let saved;
|
|
333
|
+
if (artifactDir) {
|
|
334
|
+
try {
|
|
335
|
+
saved = join(artifactDir, `review-raw-${task.id}-${Date.now()}.txt`);
|
|
336
|
+
writeFileSync(saved, redactSecrets(raw));
|
|
337
|
+
}
|
|
338
|
+
catch {
|
|
339
|
+
saved = undefined; // persistence is evidence, not a gate input — never fail the gate on it
|
|
340
|
+
}
|
|
341
|
+
}
|
|
176
342
|
return {
|
|
177
343
|
gate: "review",
|
|
178
344
|
pass: false,
|
|
179
|
-
details: `review output unparseable (reviewer ${reviewer.adapter}:${reviewer.model}) — failing closed`,
|
|
180
|
-
meta: { reviewer: channelKey(reviewer) },
|
|
345
|
+
details: `review output unparseable (reviewer ${reviewer.adapter}:${reviewer.model}; cause: ${cause}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
|
|
346
|
+
meta: { reviewer: channelKey(reviewer), unparseable: true, cause },
|
|
181
347
|
};
|
|
182
348
|
}
|
|
183
349
|
const decided = findings !== null
|
|
@@ -26,6 +26,7 @@ export interface GateContext {
|
|
|
26
26
|
cfg: TickmarkrConfig;
|
|
27
27
|
via?: GateVia;
|
|
28
28
|
excludeReviewers?: string[];
|
|
29
|
+
artifactDir?: string;
|
|
29
30
|
onGate?: (e: GateEvent) => void | Promise<void>;
|
|
30
31
|
}
|
|
31
32
|
export declare function runGates(task: Task, ctx: GateContext): Promise<{
|
package/dist/gates/run-gates.js
CHANGED
|
@@ -128,7 +128,24 @@ export async function runGates(task, ctx) {
|
|
|
128
128
|
// 5. cross-vendor review
|
|
129
129
|
if (enabled("review")) {
|
|
130
130
|
await emitStart("review");
|
|
131
|
-
|
|
131
|
+
let rv = await reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, ctx.adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir);
|
|
132
|
+
// OBS-193: an unparseable review verdict retries the REVIEW exactly once on a different reviewer —
|
|
133
|
+
// never the worker (GATE-09's judge-retry shape: straight-line single `if`, meta-only detection,
|
|
134
|
+
// the flaked verdict never enters results). The exclusion rides reviewGate's own excludeReviewers
|
|
135
|
+
// parameter, so pickReviewer's diversity rules still govern the retry seat; a fleet with no second
|
|
136
|
+
// eligible seat keeps the ORIGINAL result so the recorded cause stays truthful (OBS-196).
|
|
137
|
+
if (rv.meta?.unparseable === true && typeof rv.meta.reviewer === "string") {
|
|
138
|
+
const flaked = rv.meta.reviewer;
|
|
139
|
+
const retryVia = ctx.via
|
|
140
|
+
? { ...ctx.via, nameFor: (role, adapter) => ctx.via.nameFor(role, adapter) + "-r1" }
|
|
141
|
+
: undefined;
|
|
142
|
+
const second = await reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, ctx.adapters, ctx.cfg, retryVia, [...(ctx.excludeReviewers ?? []), flaked], ctx.artifactDir);
|
|
143
|
+
if (second.meta?.noEligibleReviewer !== true) {
|
|
144
|
+
const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
|
|
145
|
+
rv = { ...second, meta: { ...second.meta, reviewRetry: { flaked, retried } } };
|
|
146
|
+
}
|
|
147
|
+
}
|
|
148
|
+
await record(rv);
|
|
132
149
|
}
|
|
133
150
|
return { results, commits };
|
|
134
151
|
}
|
package/dist/report/bundle.d.ts
CHANGED
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
import type { EvidenceCitation } from "../gates/acceptance.js";
|
|
2
2
|
import type { RunEnvironment } from "../run/environment.js";
|
|
3
3
|
import type { JournalEvent } from "../run/journal.js";
|
|
4
|
-
export declare const BUNDLE_SCHEMA_VERSION =
|
|
4
|
+
export declare const BUNDLE_SCHEMA_VERSION = 2;
|
|
5
|
+
export declare const gateDeclined: (data: Record<string, unknown>) => boolean;
|
|
5
6
|
export declare const KNOWN_LIMITS: readonly string[];
|
|
6
7
|
export type BundleEvidence = string | EvidenceCitation;
|
|
7
8
|
export interface BundleJudgeCriterion {
|
|
@@ -12,7 +13,11 @@ export interface BundleJudgeCriterion {
|
|
|
12
13
|
}
|
|
13
14
|
export interface BundleGateResult {
|
|
14
15
|
gate: string;
|
|
16
|
+
/** true only when the gate actually ran and passed — a declined gate is never counted as passed. */
|
|
15
17
|
pass: boolean;
|
|
18
|
+
/** T11: true when the gate declined to run (journaled skipped:true, e.g. review below its
|
|
19
|
+
* complexity threshold) — neither a pass nor a fail; the details carry the reason. */
|
|
20
|
+
declined?: boolean;
|
|
16
21
|
details: string;
|
|
17
22
|
}
|
|
18
23
|
export interface BundleTask {
|
package/dist/report/bundle.js
CHANGED
|
@@ -6,7 +6,15 @@ import { createHash } from "node:crypto";
|
|
|
6
6
|
import { redactSecrets } from "../run/redact.js";
|
|
7
7
|
import { recordedEnvironment } from "./compare.js";
|
|
8
8
|
// Bump when the packet shape changes in a way a future reader must branch on before parsing.
|
|
9
|
-
|
|
9
|
+
// 2 (T11): gates gained `declined`, and `pass` narrowed from "the gate did not fail" to "the gate
|
|
10
|
+
// ran and passed" — a schema-1 reader would read a decline as a plain failure.
|
|
11
|
+
export const BUNDLE_SCHEMA_VERSION = 2;
|
|
12
|
+
// T11: the one predicate every decline surface shares — the record, the text tickmark rate, the
|
|
13
|
+
// comparison metrics and this packet must agree on what "never ran" means. Modern runs journal
|
|
14
|
+
// `skipped: true`; runs from before that field carry the same decline only in the details prefix
|
|
15
|
+
// ("skipped — complexity 4 < threshold 7"). The ^ anchor is load-bearing: a review that genuinely
|
|
16
|
+
// ran can mention a skipped something mid-body, and that gate is a pass.
|
|
17
|
+
export const gateDeclined = (data) => data.skipped === true || /^skipped\b/.test(String(data.details ?? ""));
|
|
10
18
|
// Plain-language known limits — the packet is a journal snapshot, not an unconditional proof.
|
|
11
19
|
export const KNOWN_LIMITS = [
|
|
12
20
|
"This packet is a portable snapshot of journaled run facts, not an independent re-verification of the work or its gates.",
|
|
@@ -127,9 +135,11 @@ export function buildProofBundle(runId, events) {
|
|
|
127
135
|
const judgeCriteria = [];
|
|
128
136
|
for (const g of gateEvents) {
|
|
129
137
|
const gate = typeof g.data.gate === "string" ? g.data.gate : "unknown";
|
|
130
|
-
|
|
138
|
+
// T11: a declined gate never ran — flagged declined, never a pass.
|
|
139
|
+
const declined = gateDeclined(g.data);
|
|
140
|
+
const pass = g.data.pass === true && !declined;
|
|
131
141
|
const details = typeof g.data.details === "string" ? g.data.details : "";
|
|
132
|
-
gates.push({ gate, pass, details });
|
|
142
|
+
gates.push({ gate, pass, ...(declined ? { declined: true } : {}), details });
|
|
133
143
|
if (gate === "acceptance") {
|
|
134
144
|
for (const c of parseJudgeCriteria(g.data))
|
|
135
145
|
judgeCriteria.push(c);
|
package/dist/run/daemon.d.ts
CHANGED
|
@@ -46,4 +46,9 @@ export declare const EARLY_LAUNCH_LIVENESS_MS = 60000;
|
|
|
46
46
|
/** Test seam — lowers the empty-pane liveness window without sleeping 60s per case. */
|
|
47
47
|
export declare function setEarlyLaunchLivenessMsForTests(ms: number): void;
|
|
48
48
|
export declare function resetEarlyLaunchLivenessMsForTests(): void;
|
|
49
|
+
export declare const NUDGEABLE_ADAPTERS: Set<string>;
|
|
50
|
+
export declare const WORKER_NUDGE_MESSAGE = "tickmarkr liveness check: if the task is complete, print your TICKMARKR_RESULT completion trailer exactly as specified in your prompt now. If not, state your next concrete action and continue working.";
|
|
51
|
+
/** Test seam — shrink the nudge gate and grace without minute-long sleeps. */
|
|
52
|
+
export declare function setNudgeTimingForTests(silentMs: number, graceMs: number): void;
|
|
53
|
+
export declare function resetNudgeTimingForTests(): void;
|
|
49
54
|
export declare function runDaemon(repoRoot: string, opts?: RunOptions): Promise<RunSummary>;
|