tickmarkr 1.84.0 → 1.86.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -2
- package/dist/adapters/catalog-remote.d.ts +64 -0
- package/dist/adapters/catalog-remote.js +287 -0
- package/dist/adapters/catalog.d.ts +96 -0
- package/dist/adapters/catalog.js +176 -0
- package/dist/adapters/claude-code.d.ts +1 -0
- package/dist/adapters/claude-code.js +59 -1
- package/dist/adapters/fake.js +42 -4
- package/dist/adapters/model-lints.d.ts +25 -5
- package/dist/adapters/model-lints.js +184 -50
- package/dist/adapters/model-windows.d.ts +31 -0
- package/dist/adapters/model-windows.js +69 -0
- package/dist/adapters/prompt.d.ts +5 -1
- package/dist/adapters/prompt.js +13 -4
- package/dist/adapters/registry.d.ts +25 -26
- package/dist/adapters/registry.js +173 -110
- package/dist/adapters/types.d.ts +3 -0
- package/dist/adapters/types.js +36 -3
- package/dist/brand.d.ts +5 -1
- package/dist/brand.js +18 -2
- package/dist/cli/commands/doctor.d.ts +3 -0
- package/dist/cli/commands/doctor.js +43 -21
- package/dist/cli/commands/fleet.d.ts +7 -0
- package/dist/cli/commands/fleet.js +94 -74
- package/dist/cli/commands/init.js +118 -5
- package/dist/cli/commands/status.js +202 -46
- package/dist/compile/collateral.d.ts +86 -2
- package/dist/compile/collateral.js +294 -3
- package/dist/compile/gsd.d.ts +2 -1
- package/dist/compile/gsd.js +68 -2
- package/dist/compile/native.d.ts +14 -0
- package/dist/compile/native.js +161 -12
- package/dist/config/config.d.ts +82 -5
- package/dist/config/config.js +253 -66
- package/dist/config/fleet-overlay.d.ts +25 -20
- package/dist/config/fleet-overlay.js +195 -77
- package/dist/config/fleet-why.d.ts +23 -0
- package/dist/config/fleet-why.js +42 -0
- package/dist/drivers/herdr.d.ts +21 -3
- package/dist/drivers/herdr.js +344 -110
- package/dist/gates/acceptance.js +7 -2
- package/dist/gates/baseline.d.ts +1 -0
- package/dist/gates/baseline.js +91 -13
- package/dist/gates/llm.d.ts +0 -1
- package/dist/gates/llm.js +5 -30
- package/dist/gates/review.d.ts +9 -1
- package/dist/gates/review.js +105 -10
- package/dist/gates/run-gates.d.ts +9 -0
- package/dist/gates/run-gates.js +285 -41
- package/dist/gates/verdict-cause.d.ts +4 -0
- package/dist/gates/verdict-cause.js +63 -0
- package/dist/graph/schema.d.ts +6 -0
- package/dist/graph/schema.js +8 -5
- package/dist/route/router.d.ts +0 -5
- package/dist/route/router.js +16 -20
- package/dist/run/consult.d.ts +6 -0
- package/dist/run/consult.js +35 -25
- package/dist/run/daemon.d.ts +48 -2
- package/dist/run/daemon.js +1488 -330
- package/dist/run/journal.d.ts +56 -3
- package/dist/run/journal.js +358 -4
- package/dist/run/stall.d.ts +35 -1
- package/dist/run/stall.js +118 -8
- package/dist/tui/cockpit/capture.d.ts +12 -0
- package/dist/tui/cockpit/capture.js +37 -1
- package/dist/tui/cockpit/components.d.ts +2 -0
- package/dist/tui/cockpit/components.js +8 -8
- package/dist/tui/cockpit/derive.d.ts +29 -2
- package/dist/tui/cockpit/derive.js +219 -23
- package/dist/tui/cockpit/run-cockpit.js +128 -27
- package/dist/tui/cockpit/theme.d.ts +32 -26
- package/dist/tui/cockpit/theme.js +11 -5
- package/dist/tui/ink/components.d.ts +0 -15
- package/dist/tui/ink/components.js +0 -17
- package/dist/tui/ink/fleet-app.d.ts +4 -1
- package/dist/tui/ink/fleet-app.js +134 -13
- package/fixtures/sample.native.md +1 -1
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +354 -34
- package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +70 -0
- package/skills/tickmarkr-overseer/scripts/watch-panes.sh +1 -1
- package/dist/tui/ink/studio-app.d.ts +0 -59
- package/dist/tui/ink/studio-app.js +0 -320
- package/dist/tui/save.d.ts +0 -38
- package/dist/tui/save.js +0 -96
- package/dist/tui/staging.d.ts +0 -29
- package/dist/tui/staging.js +0 -78
package/dist/gates/acceptance.js
CHANGED
|
@@ -5,6 +5,7 @@ import { renderAcceptanceItem } from "../graph/schema.js";
|
|
|
5
5
|
import { sh } from "../run/git.js";
|
|
6
6
|
import { checkDiffCap, fetchTaskDiff, isProtectedEvidence, setAsideReceiptPath } from "./review.js";
|
|
7
7
|
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
|
|
8
|
+
import { classifyVerdictCause } from "./verdict-cause.js";
|
|
8
9
|
// Fable F4: acceptance judge shares review's 900s timeout — 300s default killed frontier judges on cap-sized diffs.
|
|
9
10
|
const JUDGE_TIMEOUT_MS = 900_000;
|
|
10
11
|
const CitationSchema = z.object({ path: z.string(), line: z.number().int() });
|
|
@@ -328,12 +329,16 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
328
329
|
const raw = await runLlm(judge.adapter, judge.model, prompt, worktree, via, JUDGE_TIMEOUT_MS);
|
|
329
330
|
const extracted = extractVerdictJson(raw, nonce);
|
|
330
331
|
if (!extracted) {
|
|
332
|
+
const cause = classifyVerdictCause(raw, nonce, "pass");
|
|
331
333
|
// GATE-09: structured meta names the flaked judge channel (mirrors review.ts:99 meta precedent) so
|
|
332
334
|
// run-gates can retry the judge on a failover channel without string-matching details (D-03).
|
|
333
335
|
// The parsed-verdict paths below are untouched — the flake signal is exactly extractJson→null here.
|
|
336
|
+
const failure = cause === "malformed-verdict"
|
|
337
|
+
? "judge output unparseable — failing closed"
|
|
338
|
+
: "judge dispatch failed — no structurally valid nonce-bound response; output unparseable — failing closed";
|
|
334
339
|
return { gate: "acceptance", pass: false,
|
|
335
|
-
details: warn + detBlock +
|
|
336
|
-
meta: { unparseable: true, judge: channelKey({ adapter: judge.adapter.id, model: judge.model }) } };
|
|
340
|
+
details: warn + detBlock + failure,
|
|
341
|
+
meta: { unparseable: true, cause, judge: channelKey({ adapter: judge.adapter.id, model: judge.model }) } };
|
|
337
342
|
}
|
|
338
343
|
const { verdict: v, inconsistencies } = checkJudgeVerdict(extracted, expectedIds);
|
|
339
344
|
// v1.70 / OBS-129: each citation must fall inside a changed hunk of `diff` (the exact string embedded in
|
package/dist/gates/baseline.d.ts
CHANGED
|
@@ -14,6 +14,7 @@ export interface BaselineWarning {
|
|
|
14
14
|
commands: string[];
|
|
15
15
|
reason: string;
|
|
16
16
|
}
|
|
17
|
+
export declare const UNRECOGNIZED_FAILURE = "<unrecognized failure output>";
|
|
17
18
|
export declare function fingerprint(output: string): string[];
|
|
18
19
|
export declare function detectGateCommands(repoRoot: string, cfg: TickmarkrConfig): Record<string, string>;
|
|
19
20
|
export declare function captureBaseline(cwd: string, commands: Record<string, string>): Promise<Baseline>;
|
package/dist/gates/baseline.js
CHANGED
|
@@ -14,16 +14,76 @@ const PASS_LINE_RE = /^\s*(?:(?:[\w@./-]+:\s*)*[✓✔]|(?:\[tickmarkr\]\s+)?(?:
|
|
|
14
14
|
// output to headline it. \s is fine in a TS regex — the BSD [[:space:]] rule binds shell grep only.
|
|
15
15
|
// OBS-42: vitest's diagnostic failure headings are shared anchors for baseline and tip verification.
|
|
16
16
|
const FAIL_ANCHOR_RE = /^\s*(?:FAIL\s+|[^\w]*(?:Unhandled Errors|Uncaught Exception)\b)/;
|
|
17
|
-
|
|
17
|
+
// OBS-278: a failure fingerprint is a SHAPE — the way a runner reports a failure — never error/fail
|
|
18
|
+
// vocabulary. A status surface that draws those words (zone +N, run attempt N, "1 failed") used to
|
|
19
|
+
// fingerprint as a fresh failure on every attempt, rejecting a task for rendering what it was chartered
|
|
20
|
+
// to render. Digits are written (?:\d+|#) so a shape matches both raw and digit-normalized lines.
|
|
21
|
+
// Run summaries: " Tests N failed | M passed (T)" (vitest), "# fail N" (TAP / node:test),
|
|
22
|
+
// "ℹ fail N" (node:test's spec reporter) and "test result: FAILED. …" (cargo / libtest).
|
|
23
|
+
const SUMMARY_FAIL_RE = /^\s*(?:Tests?\s+(?:Files?\s+)?(?:\d+|#)\s+failed|#\s+fail\s+(?!0\b)(?:\d+|#)\b|ℹ\s+fail\s+(?!0\b)(?:\d+|#)\b|test result:\s+FAILED\b)/;
|
|
24
|
+
const ERROR_ANCHOR_RE = /^\s*(?:Error|[A-Za-z_$][\w$]*Error):\s+\S/;
|
|
25
|
+
const TSC_ERROR_RE = /^\s*\S.*\((?:\d+|#),(?:\d+|#)\):\s+error\s+[A-Z]+(?:\d+|#):/i; // tsc
|
|
26
|
+
const LINTER_ERROR_RE = /^\s*(?:\d+|#):(?:\d+|#)\s+error\s+\S/; // eslint stylish
|
|
27
|
+
// The failure shapes non-Vitest runners name their tests with: pytest's short summary
|
|
28
|
+
// (`FAILED tests/t.py::test_x - AssertionError`, `ERROR tests/t.py::fixture`), go test
|
|
29
|
+
// (`--- FAIL: TestFoo (0.00s)`), TAP / node:test (`not ok 1 - name`, whose summary line lands in
|
|
30
|
+
// SUMMARY_FAIL_RE) and python unittest's failure-block header (`FAIL: test_x (mod.Class.test_x)`,
|
|
31
|
+
// verbatim capture, python3 -m unittest -v). Each anchors the runner's token at line START; the
|
|
32
|
+
// pytest/go forms also require a file/test identifier after it (`::`, a path separator, or an
|
|
33
|
+
// extension) — so a surface drawing "FAILED" mid-line, or a plain sentence like "FAILED to reach
|
|
34
|
+
// the zone", still matches nothing. Without these a genuinely new pytest/go/TAP/unittest failure on
|
|
35
|
+
// an already-red baseline was forgiven as pre-existing, because only shaped lines can reject.
|
|
36
|
+
const RUNNER_FAIL_RE = /^\s*(?:(?:FAILED|ERROR)\s+\S*(?:::|[/\\]|\.[A-Za-z]\w*\b)|FAIL:\s+\S|---\s+FAIL:\s+\S|not ok\b)/;
|
|
37
|
+
// The other half of the position rule: runners that put the verdict LAST. cargo/libtest names each
|
|
38
|
+
// failing test as `test tests::name ... FAILED` (verbatim capture, cargo 1.95.0) and details it with
|
|
39
|
+
// `thread 'tests::name' (81651804) panicked at src/lib.rs:7:24:`; python unittest -v writes
|
|
40
|
+
// `test_x (mod.Class.test_x) ... FAIL` (and `... ERROR`) — the same rule with the runner's
|
|
41
|
+
// qualified-name token between identifier and separator, and the short verdict spelling (verbatim
|
|
42
|
+
// capture, python3 -m unittest -v). Recognition is positional, not a vendor name: an identifier, the
|
|
43
|
+
// runner's own separator, then the verdict ENDING the line. A drawn status strip ("│ ✗ tip-verify
|
|
44
|
+
// FAILED · zone +3 │") has the verdict mid-line inside chrome, so it matches nothing here — which is
|
|
45
|
+
// why this can generalize without re-opening OBS-278.
|
|
46
|
+
const TRAILING_FAIL_RE = /^\s*(?:test\s+)?\S+(?:\s+\([^)]*\))?\s+(?:\.{3}|-{3,})\s+(?:FAIL(?:ED)?|ERROR)\s*$|^\s*thread\s+'[^']*'\s+.*panicked at\s+\S/;
|
|
47
|
+
// node:test's spec reporter speaks glyphs, not words: a failure is named by a leading ✖ (verbatim
|
|
48
|
+
// capture, Node v22 `--test-reporter=spec`: `✖ old failure (0.585875ms)`, totalled as `ℹ fail 1` in
|
|
49
|
+
// SUMMARY_FAIL_RE) — the exact counterpart of the ✓/✔ pass markers PASS_LINE_RE drops, with the same
|
|
50
|
+
// optional label prefixes. Recognition is positional — the runner's own marker at line start — so any
|
|
51
|
+
// runner sharing the glyph protocol is read without being enumerated; a status strip drawing ✖
|
|
52
|
+
// mid-line inside chrome matches nothing, the same position rule as the other shapes.
|
|
53
|
+
const GLYPH_FAIL_RE = /^\s*(?:(?:[\w@./-]+:\s*)*)✖\s+\S/;
|
|
54
|
+
// Lines that NAME a failing test — the ones worth headlining to the operator. One list, so recognition
|
|
55
|
+
// and reporting cannot drift apart (a shape that blocks but never gets named cost 3 attempts once).
|
|
56
|
+
const namesFailure = (l) => FAIL_ANCHOR_RE.test(l) || RUNNER_FAIL_RE.test(l) || TRAILING_FAIL_RE.test(l) || GLYPH_FAIL_RE.test(l);
|
|
57
|
+
const isFailureShaped = (l) => namesFailure(l) || SUMMARY_FAIL_RE.test(l) || ERROR_ANCHOR_RE.test(l)
|
|
58
|
+
|| TSC_ERROR_RE.test(l) || LINTER_ERROR_RE.test(l);
|
|
59
|
+
const VOCAB_RE = /\b(?:error|fail(?:ed|ure|ing)?)\b/i;
|
|
18
60
|
const normalizeLine = (l) => l.replace(/\d+/g, "#").replace(/\s+/g, " ").trim();
|
|
61
|
+
// A failing command whose output holds no shape any runner here names. The marker is content-free and
|
|
62
|
+
// constant: downstream consumers (tip verify journals fingerprint counts) still see that the command
|
|
63
|
+
// failed, while no line of the output — narration, status strip, stack trace — can enter the set. Being
|
|
64
|
+
// constant it is identical on every attempt, so it can never surface as a fresh fingerprint either.
|
|
65
|
+
export const UNRECOGNIZED_FAILURE = "<unrecognized failure output>";
|
|
66
|
+
// OBS-278 dies here: only a failure SHAPE is fingerprintable. Vocabulary is not a shape — a surface that
|
|
67
|
+
// renders "zone +3 · run attempt 2 · 1 failed" (what T9 is chartered to draw) contributes nothing, with
|
|
68
|
+
// or without box glyphs, so it cannot manufacture a fresh-fingerprint rejection on any future attempt.
|
|
19
69
|
export function fingerprint(output) {
|
|
20
70
|
const lines = output
|
|
21
71
|
.split("\n")
|
|
22
72
|
.map((l) => l.replace(ANSI_RE, ""))
|
|
23
|
-
.filter((l) => !PASS_LINE_RE.test(l)
|
|
24
|
-
|
|
25
|
-
|
|
73
|
+
.filter((l) => !PASS_LINE_RE.test(l));
|
|
74
|
+
const shaped = lines.filter(isFailureShaped);
|
|
75
|
+
if (!shaped.length)
|
|
76
|
+
return lines.some((l) => l.trim()) ? [UNRECOGNIZED_FAILURE] : [];
|
|
77
|
+
return [...new Set(shaped.map(normalizeLine))];
|
|
26
78
|
}
|
|
79
|
+
// The diagnostic channel for a runner we cannot read: raw output lines, never fingerprints, never a
|
|
80
|
+
// verdict. They exist so the operator sees SOMETHING when a green command turns red unrecognizably.
|
|
81
|
+
const unrecognizedEvidence = (raw) => raw
|
|
82
|
+
.split("\n")
|
|
83
|
+
.map((l) => l.replace(ANSI_RE, "").trimEnd())
|
|
84
|
+
.filter((l) => VOCAB_RE.test(l))
|
|
85
|
+
.slice(0, 10)
|
|
86
|
+
.join("\n");
|
|
27
87
|
// stored baselines may predate ANSI/pass-marker hardening — renormalize at compare time so existing
|
|
28
88
|
// on-disk baseline.json files stay comparable without recapture (compat invariant, CLAUDE.md)
|
|
29
89
|
const renormalize = (fp) => normalizeLine(fp.replace(ANSI_RE, ""));
|
|
@@ -68,7 +128,9 @@ export async function captureBaseline(cwd, commands) {
|
|
|
68
128
|
// ponytail: strip the executing cwd so repo-root capture and worktree compare fingerprint identically; /private-vs-/tmp symlink variance is out of scope
|
|
69
129
|
base.commands[name] = {
|
|
70
130
|
exitCode: r.code,
|
|
71
|
-
|
|
131
|
+
// a command that exits 0 has no failures to fingerprint — recording any would be a lie the
|
|
132
|
+
// compare step then has to forgive
|
|
133
|
+
fingerprints: r.code === 0 ? [] : fingerprint((r.stdout + "\n" + r.stderr).split(cwd).join("")),
|
|
72
134
|
missingCommand: missingConfiguredCommand(cmd, r),
|
|
73
135
|
};
|
|
74
136
|
}
|
|
@@ -119,12 +181,12 @@ function headlineDetails(raw, fresh) {
|
|
|
119
181
|
const headlines = raw
|
|
120
182
|
.split("\n")
|
|
121
183
|
.map((l) => l.replace(ANSI_RE, ""))
|
|
122
|
-
.filter((l) =>
|
|
184
|
+
.filter((l) => namesFailure(l) || SUMMARY_FAIL_RE.test(l));
|
|
123
185
|
if (!headlines.length)
|
|
124
186
|
return { details: `new failures vs baseline:\n${fresh.join("\n")}` };
|
|
125
187
|
return {
|
|
126
188
|
details: `failing tests:\n${headlines.join("\n")}\n\nnew failure fingerprints vs baseline (secondary):\n${fresh.join("\n")}`,
|
|
127
|
-
meta: { failingTests: headlines.filter(
|
|
189
|
+
meta: { failingTests: headlines.filter(namesFailure) },
|
|
128
190
|
};
|
|
129
191
|
}
|
|
130
192
|
export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
@@ -145,18 +207,34 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
|
145
207
|
const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
|
|
146
208
|
const known = new Set((baseline.commands[name]?.fingerprints ?? []).map(renormalize));
|
|
147
209
|
// OBS-42: diagnostic headings enrich fingerprints but cannot invalidate legacy baselines.
|
|
148
|
-
const
|
|
149
|
-
|
|
210
|
+
const current = fingerprint(raw);
|
|
211
|
+
const fresh = current.filter((f) => !known.has(f) && (!FAIL_ANCHOR_RE.test(f) || f.startsWith("FAIL ")));
|
|
212
|
+
// OBS-278: only a failure SHAPE is a verdict — everything fingerprint() keeps is one, except the
|
|
213
|
+
// unrecognized-output marker, which is evidence for the operator and never grounds to reject.
|
|
214
|
+
// ponytail: ceiling — a runner whose failure output holds no shape above and whose baseline is
|
|
215
|
+
// already red has its new failures forgiven, so forgiveness that rests on the marker SAYS so
|
|
216
|
+
// below rather than reading as a verified green. Raise the ceiling by teaching isFailureShaped
|
|
217
|
+
// that runner's position rule (leading verdict + identifier, or identifier + separator + trailing
|
|
218
|
+
// verdict); loosening back to vocabulary re-opens OBS-278.
|
|
219
|
+
const unreadable = current.includes(UNRECOGNIZED_FAILURE);
|
|
220
|
+
const failing = fresh.filter((f) => f !== UNRECOGNIZED_FAILURE);
|
|
221
|
+
if (!failing.length && (baseline.commands[name]?.exitCode ?? 1) === 0) {
|
|
222
|
+
const closed = `command was green at baseline but now exits ${r.code} with no recognizable failure lines — failing closed`;
|
|
223
|
+
const evidence = unrecognizedEvidence(raw);
|
|
150
224
|
results.push({
|
|
151
225
|
gate: name,
|
|
152
226
|
pass: false,
|
|
153
|
-
details:
|
|
227
|
+
details: evidence ? `${closed}\nunrecognized output:\n${evidence}` : closed,
|
|
154
228
|
});
|
|
155
229
|
continue;
|
|
156
230
|
}
|
|
157
|
-
results.push(
|
|
158
|
-
? { gate: name, pass: false, ...headlineDetails(raw,
|
|
159
|
-
: {
|
|
231
|
+
results.push(failing.length
|
|
232
|
+
? { gate: name, pass: false, ...headlineDetails(raw, failing) }
|
|
233
|
+
: {
|
|
234
|
+
gate: name,
|
|
235
|
+
pass: true,
|
|
236
|
+
details: `exit ${r.code} but only pre-existing failures (forgiven)${unreadable ? " — no failure shape recognized in this output, so a new failure from this runner is invisible to the baseline gate" : ""}`,
|
|
237
|
+
});
|
|
160
238
|
}
|
|
161
239
|
return results;
|
|
162
240
|
}
|
package/dist/gates/llm.d.ts
CHANGED
|
@@ -15,7 +15,6 @@ export declare function appendAnchoredReview(prose: string, verdict: unknown): s
|
|
|
15
15
|
export declare function verdictNonceLine(nonce: string): string;
|
|
16
16
|
export declare function extractPromptNonce(prompt: string): string | null;
|
|
17
17
|
export declare function gateExitTrailer(nonce: string): string;
|
|
18
|
-
export declare function augmentFakeVerdictOutput(adapter: WorkerAdapter, out: string, nonce: string, prompt?: string): string;
|
|
19
18
|
export type GatePaneRole = "judge" | "review" | "consult";
|
|
20
19
|
/** T8: role-first pane name for fleet visibility — judge · T4, review · T3, consult · T2. */
|
|
21
20
|
export declare function gatePaneName(role: GatePaneRole, taskId: string, suffix?: string): string;
|
package/dist/gates/llm.js
CHANGED
|
@@ -69,29 +69,6 @@ export function extractPromptNonce(prompt) {
|
|
|
69
69
|
export function gateExitTrailer(nonce) {
|
|
70
70
|
return `printf '\\nTICKMARKR_''EXIT_${nonce}:%s\\n' $?`;
|
|
71
71
|
}
|
|
72
|
-
// v1.64: scripted fake judge verdicts predate the required per-criterion evidence field — quote the
|
|
73
|
-
// first line of the prompt's own diff block into rows lacking one so zero-token fixtures keep their
|
|
74
|
-
// outcomes. Rows scripting an explicit evidence value pass through verbatim (tests exercise both paths).
|
|
75
|
-
function injectFakeEvidence(obj, prompt) {
|
|
76
|
-
if (!prompt.startsWith("TICKMARKR-JUDGE") || !Array.isArray(obj.criteria))
|
|
77
|
-
return obj;
|
|
78
|
-
const line = /```diff\n([\s\S]*?)```/.exec(prompt)?.[1].split("\n").find((l) => l.trim());
|
|
79
|
-
if (!line)
|
|
80
|
-
return obj;
|
|
81
|
-
const criteria = obj.criteria.map((row) => row && typeof row === "object" && !("evidence" in row) ? { ...row, evidence: line } : row);
|
|
82
|
-
return { ...obj, criteria };
|
|
83
|
-
}
|
|
84
|
-
// ponytail: fake adapter serves static verdict JSON without nonce; append a bound copy for zero-token tests.
|
|
85
|
-
export function augmentFakeVerdictOutput(adapter, out, nonce, prompt = "") {
|
|
86
|
-
if (adapter.id !== "fake")
|
|
87
|
-
return out;
|
|
88
|
-
const obj = extractJson(out);
|
|
89
|
-
if (!obj || typeof obj !== "object" || obj.nonce === nonce)
|
|
90
|
-
return out;
|
|
91
|
-
if (typeof obj.nonce === "string")
|
|
92
|
-
return out;
|
|
93
|
-
return `${out}\n${JSON.stringify(injectFakeEvidence({ ...obj, nonce }, prompt))}`;
|
|
94
|
-
}
|
|
95
72
|
/** T8: role-first pane name for fleet visibility — judge · T4, review · T3, consult · T2. */
|
|
96
73
|
export function gatePaneName(role, taskId, suffix = "") {
|
|
97
74
|
return `${role}${GATE_PANE_SEP}${taskId}${suffix}`;
|
|
@@ -130,11 +107,7 @@ export async function runHeadless(adapter, model, prompt, cwd, timeoutMs = 30000
|
|
|
130
107
|
const pf = join(mkdtempSync(join(tmpdir(), "tickmarkr-llm-")), "prompt.md");
|
|
131
108
|
writeFileSync(pf, prompt);
|
|
132
109
|
const r = await sh(adapter.headlessCommand(pf, model), cwd, timeoutMs);
|
|
133
|
-
|
|
134
|
-
let out = r.stdout + "\n" + r.stderr;
|
|
135
|
-
if (nonce)
|
|
136
|
-
out = augmentFakeVerdictOutput(adapter, out, nonce, prompt);
|
|
137
|
-
return out;
|
|
110
|
+
return r.stdout + "\n" + r.stderr;
|
|
138
111
|
}
|
|
139
112
|
// v1.1 default path: the same headless CLI call, but dispatched through the driver
|
|
140
113
|
// as a visible named agent (herdr pane), with the quote-split completion wrapper.
|
|
@@ -157,10 +130,9 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
|
|
|
157
130
|
// nonce-suffixed exit only: a displayed bare "TICKMARKR_EXIT:" or another call's marker must not
|
|
158
131
|
// false-complete — same guard the worker path uses (daemon.ts:330-331).
|
|
159
132
|
await via.driver.waitOutput(slot, `TICKMARKR_EXIT_${nonce}:\\d`, timeoutMs, { regex: true });
|
|
160
|
-
|
|
133
|
+
const out = await via.driver.read(slot, 400);
|
|
161
134
|
if (!via.keep)
|
|
162
135
|
await via.driver.close(slot);
|
|
163
|
-
out = augmentFakeVerdictOutput(adapter, out, nonce, prompt);
|
|
164
136
|
return dewrapPaneVerdict(out, nonce);
|
|
165
137
|
}
|
|
166
138
|
// OBS-155: a TUI renders the verdict as a bullet and HARD-wraps it at pane width with a 2-space
|
|
@@ -179,6 +151,9 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
|
|
|
179
151
|
export function dewrapPaneVerdict(out, nonce) {
|
|
180
152
|
if (!out.includes(nonce))
|
|
181
153
|
return out;
|
|
154
|
+
// Preserve already-readable responder bytes; only a genuinely wrapped verdict needs reconstruction.
|
|
155
|
+
if (extractVerdictJson(out, nonce))
|
|
156
|
+
return out;
|
|
182
157
|
const lines = out.split("\n");
|
|
183
158
|
// OBS-209: EVERY brace-start is a candidate, scanned newest-first. findIndex took only the first,
|
|
184
159
|
// so any earlier line beginning with `{` — a quoted snippet, a lone brace in the reviewer's own
|
package/dist/gates/review.d.ts
CHANGED
|
@@ -3,6 +3,7 @@ import { type TickmarkrConfig } from "../config/config.js";
|
|
|
3
3
|
import { type Task } from "../graph/schema.js";
|
|
4
4
|
import { type GateVia } from "./llm.js";
|
|
5
5
|
import type { GateResult } from "./types.js";
|
|
6
|
+
import { type VerdictUnparseableCause } from "./verdict-cause.js";
|
|
6
7
|
export type ReviewSeverity = "material" | "minor";
|
|
7
8
|
export interface ReviewFinding {
|
|
8
9
|
note: string;
|
|
@@ -27,6 +28,13 @@ export declare function isProtectedEvidence(path: string): boolean;
|
|
|
27
28
|
export declare function setAsideReceiptPath(section: string): string | null;
|
|
28
29
|
/** Replace the content of every section confined to the regenerable frame corpora with a receipt. */
|
|
29
30
|
export declare function setAsideRegenerableCaptures(diff: string): string;
|
|
31
|
+
/**
|
|
32
|
+
* The paths this task's diff ACTUALLY touched. `-z` so a path carrying spaces or non-ASCII bytes is
|
|
33
|
+
* never mangled by git's quoting, `--no-renames` so a rename reports BOTH sides: a file renamed OUT of
|
|
34
|
+
* the leaf class must be visible to the promotion test, and rename detection would hide the old side.
|
|
35
|
+
*/
|
|
36
|
+
export declare function changedPaths(worktree: string, baseRef: string): Promise<string[]>;
|
|
37
|
+
export declare function mirrorsVersionOnly(worktree: string, baseRef: string, path: string): Promise<boolean>;
|
|
30
38
|
export declare function fetchTaskDiff(worktree: string, baseRef: string): Promise<{
|
|
31
39
|
full: string;
|
|
32
40
|
forCap: string;
|
|
@@ -37,5 +45,5 @@ export declare function diffCapParkReason(results: GateResult[]): string | null;
|
|
|
37
45
|
export declare function modelId(model: string): string;
|
|
38
46
|
export declare function pickReviewer(author: Assignment, channels: BillingChannel[], exclude?: string[], // v1.1 failover: reviewer channels that already produced garbage for this task
|
|
39
47
|
prefer?: string[]): BillingChannel | null;
|
|
40
|
-
export type ReviewUnparseableCause =
|
|
48
|
+
export type ReviewUnparseableCause = VerdictUnparseableCause;
|
|
41
49
|
export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string): Promise<GateResult>;
|
package/dist/gates/review.js
CHANGED
|
@@ -1,13 +1,14 @@
|
|
|
1
1
|
import { writeFileSync } from "node:fs";
|
|
2
2
|
import { join } from "node:path";
|
|
3
|
-
import { channelKey } from "../adapters/types.js";
|
|
4
|
-
import { DEFAULT_DIFF_CAP, TIER_RANK } from "../config/config.js";
|
|
3
|
+
import { channelKey, shq } from "../adapters/types.js";
|
|
4
|
+
import { criticalPathHits, DEFAULT_DIFF_CAP, DEFAULT_REVIEW_CRITICAL_PATHS, declaredReviewPolicy, isReviewLeafPath, raiseReviewPolicy, REVIEW_VERSION_MIRRORS, TIER_RANK, } from "../config/config.js";
|
|
5
5
|
import { renderAcceptanceItem } from "../graph/schema.js";
|
|
6
6
|
import { getAdapter } from "../adapters/registry.js";
|
|
7
7
|
import { shOk } from "../run/git.js";
|
|
8
8
|
import { redactSecrets } from "../run/redact.js";
|
|
9
9
|
import { marginalCostRank } from "../route/router.js";
|
|
10
10
|
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
|
|
11
|
+
import { classifyVerdictCause } from "./verdict-cause.js";
|
|
11
12
|
// legacy flat `issues` shape — every issue blocks; the approve flag must agree with the list.
|
|
12
13
|
function classifyReviewIssues(approve, issues) {
|
|
13
14
|
const inconsistencies = [];
|
|
@@ -211,6 +212,34 @@ export function setAsideRegenerableCaptures(diff) {
|
|
|
211
212
|
const kindOnly = kindOnlyPaths(sections);
|
|
212
213
|
return sections.map((s) => setAsideSection(s, kindOnly)).join("");
|
|
213
214
|
}
|
|
215
|
+
/**
|
|
216
|
+
* The paths this task's diff ACTUALLY touched. `-z` so a path carrying spaces or non-ASCII bytes is
|
|
217
|
+
* never mangled by git's quoting, `--no-renames` so a rename reports BOTH sides: a file renamed OUT of
|
|
218
|
+
* the leaf class must be visible to the promotion test, and rename detection would hide the old side.
|
|
219
|
+
*/
|
|
220
|
+
export async function changedPaths(worktree, baseRef) {
|
|
221
|
+
const out = await shOk(`git diff --name-only --no-renames -z '${baseRef}..HEAD'`, worktree);
|
|
222
|
+
return [...new Set(out.split("\0").map((p) => p.trim()).filter(Boolean))].sort();
|
|
223
|
+
}
|
|
224
|
+
/**
|
|
225
|
+
* A root manifest is a version MIRROR only when the bump is all it changed. `package.json` carries the
|
|
226
|
+
* gate commands and the dependency set, so a diff that moves `scripts` or `dependencies` is executable
|
|
227
|
+
* behaviour wearing a leaf-class path — the one thing a path predicate can never see for itself. Every
|
|
228
|
+
* added or removed line must be a `"version":` line (a lockfile bump rewrites several of them); a
|
|
229
|
+
* manifest whose diff cannot be read at all fails closed, out of the class.
|
|
230
|
+
*/
|
|
231
|
+
const VERSION_FIELD_LINE_RE = /^[+-]\s*"version":\s*"[^"]*",?\s*$/;
|
|
232
|
+
export async function mirrorsVersionOnly(worktree, baseRef, path) {
|
|
233
|
+
let diff;
|
|
234
|
+
try {
|
|
235
|
+
diff = await shOk(`git diff -U0 '${baseRef}..HEAD' -- ${shq(path)}`, worktree);
|
|
236
|
+
}
|
|
237
|
+
catch {
|
|
238
|
+
return false;
|
|
239
|
+
}
|
|
240
|
+
const changed = diff.split("\n").filter((l) => /^[+-]/.test(l) && !/^(?:\+\+\+|---)/.test(l));
|
|
241
|
+
return changed.length > 0 && changed.every((l) => VERSION_FIELD_LINE_RE.test(l));
|
|
242
|
+
}
|
|
214
243
|
export async function fetchTaskDiff(worktree, baseRef) {
|
|
215
244
|
const full = setAsideRegenerableCaptures(await shOk(`git diff '${baseRef}..HEAD'`, worktree));
|
|
216
245
|
const forCap = setAsideRegenerableCaptures(await shOk(`git diff -U0 '${baseRef}..HEAD'`, worktree));
|
|
@@ -269,9 +298,74 @@ export async function reviewGate(task, worktree, baseRef, author, channels, adap
|
|
|
269
298
|
// OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
|
|
270
299
|
// direct tests) skips persistence and changes nothing else.
|
|
271
300
|
artifactDir) {
|
|
272
|
-
|
|
273
|
-
|
|
301
|
+
// R3 (OBS-186): participation is keyed on PATHS. The compiler's assignment comes from the DECLARED
|
|
302
|
+
// files[]; the operator's floor may RAISE it to full and can never lower it. `complexityThreshold` is
|
|
303
|
+
// retired — the branch that returned a green skip on a complexity comparison is gone, and with it the
|
|
304
|
+
// "a law caps complexity at 3, the gate starts at 7" unreachability OBS-186 measured.
|
|
305
|
+
//
|
|
306
|
+
// COLLATERAL this rescoped task closed: the run-gates/daemon participation assertions are rewritten
|
|
307
|
+
// path-keyed, the NamedFake review fixtures author their own nonce-bound verdict (a renamed fake is
|
|
308
|
+
// a distinct responder and does not inherit the registered fake's producer contract — the check is
|
|
309
|
+
// not weakened, the fixture is fixed), the merge decision reads `gateSatisfied`, and the daemon writes a
|
|
310
|
+
// parallel round's gate-result rows in GATE_NAMES order (src/run/daemon.ts).
|
|
311
|
+
//
|
|
312
|
+
// That last one is why: retiring the switch makes fixtures that used to SKIP review journal TWO
|
|
313
|
+
// verdict rows per round instead of one, and judge ‖ review publish in COMPLETION order — so the
|
|
314
|
+
// three journal-determinism oracles (tests/run/narration.test.ts's byte comparison and its
|
|
315
|
+
// throwing-sink event-order check, tests/run/notify-identity.test.ts's two-run equality, and this
|
|
316
|
+
// repo's own phase-start/gate-result pairing in tests/run/daemon.test.ts) start seeing a race.
|
|
317
|
+
// LATENT, not introduced: the operator's config has run `complexityThreshold: 0` since 2026-07-31,
|
|
318
|
+
// so production rounds have journaled both siblings all along — only the fixtures were blind to it.
|
|
319
|
+
// Fixed in the ledger rather than in the oracles, because determinism run-to-run is a property of
|
|
320
|
+
// the journal, not of three test files that happen to assert it.
|
|
321
|
+
const declaredPolicy = declaredReviewPolicy(task.files);
|
|
322
|
+
const policy = raiseReviewPolicy(declaredPolicy, cfg.review.policy);
|
|
323
|
+
// PROMOTION: the declared assignment is a claim about paths, and the diff is the evidence. A
|
|
324
|
+
// judge-only task whose diff left the leaf class is reviewed in full — the claim never outranks
|
|
325
|
+
// what actually happened, and an empty diff promotes too (a skip earned by an absence is not earned).
|
|
326
|
+
let promotedBy = null;
|
|
327
|
+
if (policy === "judge-only") {
|
|
328
|
+
const touched = await changedPaths(worktree, baseRef);
|
|
329
|
+
// Two ways a path leaves the leaf class. It is not a member (`docs/tool.ts`, `docs/Makefile`) — or
|
|
330
|
+
// it is a root version mirror whose diff moved more than the version field, which no path predicate
|
|
331
|
+
// can see. `package.json` carries the gate commands, so a scripts edit hiding behind a leaf-class
|
|
332
|
+
// path is exactly the promotion this buys.
|
|
333
|
+
const nonMembers = touched.filter((p) => !isReviewLeafPath(p));
|
|
334
|
+
const impostorMirrors = (await Promise.all(touched.filter((p) => REVIEW_VERSION_MIRRORS.has(p))
|
|
335
|
+
.map(async (p) => await mirrorsVersionOnly(worktree, baseRef, p) ? null : p))).filter((p) => p !== null);
|
|
336
|
+
// The fail-closed backstop for the compile lint. reviewParticipationErrors runs at a repo root the
|
|
337
|
+
// compile seam cannot always name (collateral.ts); THIS gate is handed the run's real config, so a
|
|
338
|
+
// critical path that reached dispatch is reviewed here whatever the lint saw. The shipped defaults
|
|
339
|
+
// are unioned in for the same reason they are there: a config that names none still has a floor.
|
|
340
|
+
const criticalHits = criticalPathHits([...task.files, ...touched], [...new Set([...DEFAULT_REVIEW_CRITICAL_PATHS, ...(cfg.review.criticalPaths ?? [])])]);
|
|
341
|
+
const escaped = [...new Set([...nonMembers, ...impostorMirrors, ...criticalHits])].sort();
|
|
342
|
+
if (touched.length > 0 && escaped.length === 0) {
|
|
343
|
+
// A declined review makes NO green claim. types.ts's T11 note ("pass stays true so enforcement is
|
|
344
|
+
// unchanged") describes the baseline skips — a build/test/lint command the repo never configured.
|
|
345
|
+
// R3 overrules it here: a cross-vendor review that did not run cannot report a pass, so the record
|
|
346
|
+
// carries verdict "skipped" with the policy that declined it and the reason, and pass is never
|
|
347
|
+
// true. The merge-predicate seam this opens is named in the collateral note above.
|
|
348
|
+
return {
|
|
349
|
+
gate: "review",
|
|
350
|
+
pass: false,
|
|
351
|
+
details: `skipped — reviewPolicy judge-only: every declared path is docs/CHANGELOG/RELEASING/version-mirror leaf work and the diff stayed in that class (${touched.join(", ")})`,
|
|
352
|
+
meta: {
|
|
353
|
+
skipped: true,
|
|
354
|
+
verdict: "skipped",
|
|
355
|
+
policy: "judge-only",
|
|
356
|
+
reason: "every declared path and every path this diff touched is provably leaf-class work",
|
|
357
|
+
paths: touched,
|
|
358
|
+
},
|
|
359
|
+
};
|
|
360
|
+
}
|
|
361
|
+
promotedBy = escaped.length > 0 ? escaped : [];
|
|
274
362
|
}
|
|
363
|
+
// Every verdict this gate reports from here on was produced under `full` — either declared full, or
|
|
364
|
+
// promoted here. `promotedBy` names the paths that bought the promotion, so the record shows WHY.
|
|
365
|
+
const policyMeta = {
|
|
366
|
+
policy: "full",
|
|
367
|
+
...(promotedBy ? { promotedFrom: declaredPolicy, promotedBy } : {}),
|
|
368
|
+
};
|
|
275
369
|
const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? []);
|
|
276
370
|
if (!reviewer) {
|
|
277
371
|
// meta.noEligibleReviewer lets run-gates' review-retry keep the ORIGINAL unparseable result when
|
|
@@ -326,9 +420,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
326
420
|
if (!v || (findings === null && (typeof v.approve !== "boolean" || !Array.isArray(v.issues)))) {
|
|
327
421
|
// OBS-196: name the cause and persist the raw bytes — a ruled-on "unparseable" without its
|
|
328
422
|
// evidence cannot be audited, and a cutoff must never be indistinguishable from a parse defect.
|
|
329
|
-
const cause = raw
|
|
330
|
-
? "empty-output"
|
|
331
|
-
: !raw.includes(nonce) ? "no-verdict" : "malformed-verdict";
|
|
423
|
+
const cause = classifyVerdictCause(raw, nonce, "approve");
|
|
332
424
|
let saved;
|
|
333
425
|
if (artifactDir) {
|
|
334
426
|
try {
|
|
@@ -339,11 +431,14 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
339
431
|
saved = undefined; // persistence is evidence, not a gate input — never fail the gate on it
|
|
340
432
|
}
|
|
341
433
|
}
|
|
434
|
+
const failure = cause === "malformed-verdict"
|
|
435
|
+
? "review output unparseable"
|
|
436
|
+
: "review dispatch failed — no structurally valid nonce-bound response; output unparseable";
|
|
342
437
|
return {
|
|
343
438
|
gate: "review",
|
|
344
439
|
pass: false,
|
|
345
|
-
details:
|
|
346
|
-
meta: { reviewer: channelKey(reviewer), unparseable: true, cause },
|
|
440
|
+
details: `${failure} (reviewer ${reviewer.adapter}:${reviewer.model}; cause: ${cause}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
|
|
441
|
+
meta: { ...policyMeta, reviewer: channelKey(reviewer), unparseable: true, cause },
|
|
347
442
|
};
|
|
348
443
|
}
|
|
349
444
|
const decided = findings !== null
|
|
@@ -354,6 +449,6 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
354
449
|
gate: "review",
|
|
355
450
|
pass: decided.pass,
|
|
356
451
|
details: appendAnchoredReview(prose, v),
|
|
357
|
-
meta: { reviewer: channelKey(reviewer) },
|
|
452
|
+
meta: { ...policyMeta, reviewer: channelKey(reviewer) },
|
|
358
453
|
};
|
|
359
454
|
}
|
|
@@ -9,6 +9,7 @@ export type GateEvent = {
|
|
|
9
9
|
gate: GateName;
|
|
10
10
|
index: number;
|
|
11
11
|
total: number;
|
|
12
|
+
parentAt?: number;
|
|
12
13
|
} | {
|
|
13
14
|
phase: "end";
|
|
14
15
|
gate: GateName;
|
|
@@ -27,8 +28,16 @@ export interface GateContext {
|
|
|
27
28
|
via?: GateVia;
|
|
28
29
|
excludeReviewers?: string[];
|
|
29
30
|
artifactDir?: string;
|
|
31
|
+
pipeline?: "v185" | "legacy";
|
|
32
|
+
selectTests?: boolean;
|
|
30
33
|
onGate?: (e: GateEvent) => void | Promise<void>;
|
|
31
34
|
}
|
|
35
|
+
/**
|
|
36
|
+
* The configured test command narrowed to these files. Mirrors testFiltered's `--` rule (acceptance.ts:104):
|
|
37
|
+
* npm/yarn/pnpm/npx script wrappers need one `--` to forward positional filters to the underlying runner;
|
|
38
|
+
* a command that already has `--` takes them directly. Every path is quoted — config flows into a shell.
|
|
39
|
+
*/
|
|
40
|
+
export declare function testCommandForFiles(testCmd: string, files: string[]): string;
|
|
32
41
|
export declare function runGates(task: Task, ctx: GateContext): Promise<{
|
|
33
42
|
results: GateResult[];
|
|
34
43
|
commits: string[];
|