tickmarkr 1.86.0 → 1.89.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/catalog.d.ts +30 -1
- package/dist/adapters/catalog.js +58 -2
- package/dist/adapters/fake.d.ts +2 -1
- package/dist/adapters/fake.js +7 -0
- package/dist/adapters/grok.js +11 -0
- package/dist/adapters/kimi.d.ts +2 -1
- package/dist/adapters/kimi.js +36 -0
- package/dist/adapters/opencode.js +17 -0
- package/dist/adapters/pi.js +11 -0
- package/dist/adapters/prompt.js +8 -1
- package/dist/adapters/registry.d.ts +10 -2
- package/dist/adapters/registry.js +126 -67
- package/dist/adapters/types.d.ts +34 -3
- package/dist/adapters/types.js +99 -1
- package/dist/cli/commands/approve.d.ts +2 -0
- package/dist/cli/commands/approve.js +104 -84
- package/dist/cli/commands/compile.d.ts +1 -1
- package/dist/cli/commands/compile.js +29 -12
- package/dist/cli/commands/init.js +1 -1
- package/dist/cli/commands/plan.d.ts +1 -1
- package/dist/cli/commands/plan.js +21 -2
- package/dist/cli/commands/report.js +49 -0
- package/dist/cli/commands/resume.js +7 -1
- package/dist/cli/commands/status.js +298 -96
- package/dist/cli/harness.d.ts +13 -0
- package/dist/cli/harness.js +50 -0
- package/dist/compile/collateral.js +4 -4
- package/dist/compile/index.d.ts +14 -3
- package/dist/compile/index.js +36 -10
- package/dist/compile/native.js +108 -25
- package/dist/drivers/subprocess.d.ts +6 -1
- package/dist/drivers/subprocess.js +9 -4
- package/dist/gates/acceptance.d.ts +21 -1
- package/dist/gates/acceptance.js +67 -22
- package/dist/gates/artifact-manifest.d.ts +119 -0
- package/dist/gates/artifact-manifest.js +357 -0
- package/dist/gates/baseline.d.ts +6 -0
- package/dist/gates/baseline.js +52 -7
- package/dist/gates/llm.js +37 -26
- package/dist/gates/review.d.ts +16 -11
- package/dist/gates/review.js +44 -150
- package/dist/gates/run-gates.d.ts +1 -0
- package/dist/gates/run-gates.js +145 -9
- package/dist/graph/schema.d.ts +3 -1
- package/dist/graph/schema.js +4 -1
- package/dist/route/preference.d.ts +1 -1
- package/dist/route/preference.js +8 -1
- package/dist/run/consult.js +14 -1
- package/dist/run/daemon.d.ts +42 -0
- package/dist/run/daemon.js +2322 -1963
- package/dist/run/git.d.ts +50 -0
- package/dist/run/git.js +113 -2
- package/dist/run/interactive-seed.d.ts +6 -2
- package/dist/run/interactive-seed.js +72 -5
- package/dist/run/journal.d.ts +9 -1
- package/dist/run/journal.js +99 -9
- package/dist/run/lock.d.ts +11 -0
- package/dist/run/lock.js +97 -6
- package/dist/run/outcome.d.ts +50 -0
- package/dist/run/outcome.js +152 -0
- package/dist/run/protocol.d.ts +460 -0
- package/dist/run/protocol.js +433 -0
- package/dist/run/supervision.d.ts +29 -0
- package/dist/run/supervision.js +189 -0
- package/fixtures/gateway-models.json +1 -0
- package/fixtures/wrapped-acceptance.native.md +29 -0
- package/package.json +1 -1
- package/schema/rungraph.schema.json +21 -2
- package/skills/tickmarkr-overseer/SKILL.md +366 -4
- package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +95 -8
- package/skills/tickmarkr-overseer/scripts/watch-contamination.sh +77 -0
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +86 -0
- package/skills/tickmarkr-overseer/scripts/watch-parks.sh +96 -0
- package/skills/tickmarkr-overseer/scripts/watch-pending-input.sh +183 -0
|
@@ -121,14 +121,14 @@ export function collateralLints(tasks, repoRoot) {
|
|
|
121
121
|
continue;
|
|
122
122
|
if (mentions(text, needles))
|
|
123
123
|
hits.push(tf);
|
|
124
|
-
if (hits.length >= MAX_HITS_PER_TASK)
|
|
125
|
-
break;
|
|
126
124
|
}
|
|
127
125
|
if (!hits.length)
|
|
128
126
|
continue;
|
|
129
127
|
// deterministic: walk already sorted; stable list
|
|
130
|
-
const listed = hits.join(", ");
|
|
131
|
-
const tail = hits.length
|
|
128
|
+
const listed = hits.slice(0, MAX_HITS_PER_TASK).join(", ");
|
|
129
|
+
const tail = hits.length > MAX_HITS_PER_TASK
|
|
130
|
+
? ` (${hits.length} total; capped at ${MAX_HITS_PER_TASK} shown)`
|
|
131
|
+
: "";
|
|
132
132
|
lines.push(`${t.id}: likely collateral tests not in files[]: ${listed}${tail}`);
|
|
133
133
|
}
|
|
134
134
|
return lines;
|
package/dist/compile/index.d.ts
CHANGED
|
@@ -1,4 +1,15 @@
|
|
|
1
|
-
import type
|
|
2
|
-
type SourceType =
|
|
3
|
-
export
|
|
1
|
+
import { type RunGraph, type SpecSource } from "../graph/schema.js";
|
|
2
|
+
type SourceType = SpecSource;
|
|
3
|
+
export type PlanIR = {
|
|
4
|
+
version: RunGraph["version"];
|
|
5
|
+
source: RunGraph["spec"]["source"];
|
|
6
|
+
paths: RunGraph["spec"]["paths"];
|
|
7
|
+
hash: RunGraph["spec"]["hash"];
|
|
8
|
+
base?: RunGraph["spec"]["base"];
|
|
9
|
+
mode?: RunGraph["mode"];
|
|
10
|
+
tasks: RunGraph["tasks"];
|
|
11
|
+
};
|
|
12
|
+
export type PlanFinalizationHook = (plan: PlanIR) => PlanIR;
|
|
13
|
+
export declare function finalizePlan(plan: PlanIR, src: string): RunGraph;
|
|
14
|
+
export declare function compileSource(src: string, type?: SourceType, root?: string, beforeFinalize?: PlanFinalizationHook): RunGraph;
|
|
4
15
|
export {};
|
package/dist/compile/index.js
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { existsSync, readFileSync, statSync } from "node:fs";
|
|
2
2
|
import { join } from "node:path";
|
|
3
|
+
import { validateGraph } from "../graph/schema.js";
|
|
3
4
|
import { taskUnitContractErrors } from "./collateral.js";
|
|
4
5
|
import { CompileError } from "./common.js";
|
|
5
6
|
import { compileGsd, isGsdPhaseDir } from "./gsd.js";
|
|
@@ -39,15 +40,40 @@ function enforceTaskUnitContract(g, src) {
|
|
|
39
40
|
}
|
|
40
41
|
return g;
|
|
41
42
|
}
|
|
42
|
-
export function
|
|
43
|
+
export function finalizePlan(plan, src) {
|
|
44
|
+
const graph = validateGraph({
|
|
45
|
+
version: plan.version,
|
|
46
|
+
...(plan.mode !== undefined ? { mode: plan.mode } : {}),
|
|
47
|
+
spec: {
|
|
48
|
+
source: plan.source,
|
|
49
|
+
paths: plan.paths,
|
|
50
|
+
hash: plan.hash,
|
|
51
|
+
...(plan.base !== undefined ? { base: plan.base } : {}),
|
|
52
|
+
},
|
|
53
|
+
tasks: plan.tasks,
|
|
54
|
+
});
|
|
55
|
+
return enforceTaskUnitContract(graph, src);
|
|
56
|
+
}
|
|
57
|
+
function compilePlan(src, type, root) {
|
|
43
58
|
const kind = type ?? detect(src);
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
59
|
+
const graph = kind === "speckit" ? compileSpecKit(src)
|
|
60
|
+
: kind === "gsd" ? compileGsd(src, root)
|
|
61
|
+
: kind === "native" ? compileNative(src)
|
|
62
|
+
: kind === "prd" ? compilePrd(src)
|
|
63
|
+
: null;
|
|
64
|
+
if (!graph) {
|
|
65
|
+
throw new CompileError(`cannot detect spec type for ${src} — pass a Spec Kit feature dir (with tasks.md), a GSD phase dir (with *-PLAN.md), or a marked native/generic PRD .md file, or use --type speckit|prd|gsd|native`);
|
|
66
|
+
}
|
|
67
|
+
return {
|
|
68
|
+
version: graph.version,
|
|
69
|
+
source: graph.spec.source,
|
|
70
|
+
paths: graph.spec.paths,
|
|
71
|
+
hash: graph.spec.hash,
|
|
72
|
+
...(graph.spec.base !== undefined ? { base: graph.spec.base } : {}),
|
|
73
|
+
...(graph.mode !== undefined ? { mode: graph.mode } : {}),
|
|
74
|
+
tasks: graph.tasks,
|
|
75
|
+
};
|
|
76
|
+
}
|
|
77
|
+
export function compileSource(src, type, root, beforeFinalize = (plan) => plan) {
|
|
78
|
+
return finalizePlan(beforeFinalize(compilePlan(src, type, root)), src);
|
|
53
79
|
}
|
package/dist/compile/native.js
CHANGED
|
@@ -82,14 +82,27 @@ export function compileNative(file) {
|
|
|
82
82
|
const drafts = [];
|
|
83
83
|
let plainCount = 0; // v1.19: plain-string acceptance items compiled as judge oracles (compat) — warn once
|
|
84
84
|
let specMode;
|
|
85
|
-
|
|
85
|
+
let specBase;
|
|
86
|
+
for (const [index, line] of content.split("\n").entries()) {
|
|
86
87
|
const heading = line.match(HEAD_RE);
|
|
87
88
|
if (heading) {
|
|
88
|
-
drafts.push({ id: heading[1], title: heading[2].trim(), fields: {}, acceptance: [], gates: [], hasGates: false, list: null, continuationField: null });
|
|
89
|
+
drafts.push({ id: heading[1], title: heading[2].trim(), fields: {}, acceptanceRaw: [], acceptance: [], gates: [], hasGates: false, list: null, itemOpen: false, continuationField: null });
|
|
89
90
|
continue;
|
|
90
91
|
}
|
|
91
92
|
const draft = drafts.at(-1);
|
|
92
93
|
if (!draft) {
|
|
94
|
+
const base = line.match(/^base:\s*(.*)$/);
|
|
95
|
+
if (base) {
|
|
96
|
+
const value = base[1].trim();
|
|
97
|
+
if (!value) {
|
|
98
|
+
throw new CompileError("spec front-matter base must not be empty — repair: write lower-case base: <commit-or-ref> at column zero");
|
|
99
|
+
}
|
|
100
|
+
specBase = value;
|
|
101
|
+
continue;
|
|
102
|
+
}
|
|
103
|
+
if (/^Base:\s*/.test(line)) {
|
|
104
|
+
throw new CompileError("native task unit contract rejects column-zero Base: as untyped prose — repair: write lower-case base: <commit-or-ref>");
|
|
105
|
+
}
|
|
93
106
|
// v1.51 T2: spec front-matter — a top-level `mode: <name>` line before the first task heading
|
|
94
107
|
// declares the engagement's routing mode (loses only to an explicit run flag).
|
|
95
108
|
const fm = line.match(/^mode:\s*(\S+)\s*$/);
|
|
@@ -107,6 +120,7 @@ export function compileNative(file) {
|
|
|
107
120
|
if (!FIELDS.has(name))
|
|
108
121
|
invalid(draft.id, field[1], "is unknown");
|
|
109
122
|
const value = field[2].trim();
|
|
123
|
+
draft.itemOpen = false;
|
|
110
124
|
if (name === "acceptance" || name === "gates") {
|
|
111
125
|
if (value)
|
|
112
126
|
invalid(draft.id, field[1], "must be a nested list");
|
|
@@ -155,31 +169,51 @@ export function compileNative(file) {
|
|
|
155
169
|
const value = nested[1].trim();
|
|
156
170
|
if (!value)
|
|
157
171
|
invalid(draft.id, draft.list, "must not contain empty entries");
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
else {
|
|
170
|
-
plainCount++;
|
|
171
|
-
draft.acceptance.push(value);
|
|
172
|
-
}
|
|
173
|
-
}
|
|
174
|
-
else {
|
|
175
|
-
draft[draft.list].push(value);
|
|
176
|
-
}
|
|
172
|
+
(draft.list === "acceptance" ? draft.acceptanceRaw : draft.gates).push(value);
|
|
173
|
+
draft.itemOpen = true;
|
|
174
|
+
continue;
|
|
175
|
+
}
|
|
176
|
+
// OBS-488: a deeper-indented non-item line under a list item is that item's wrapped
|
|
177
|
+
// continuation — join with a single space, exactly as OBS-60/OBS-308 guarantee for fields.
|
|
178
|
+
// 1.87.0 dropped these lines silently: 53/78 of run-551's criteria compiled to first-line
|
|
179
|
+
// stubs and every falsifier tail was invisible to the judge.
|
|
180
|
+
if (draft.list && draft.itemOpen && (line.startsWith(" ") || line.startsWith("\t")) && line.trim()) {
|
|
181
|
+
const items = draft.list === "acceptance" ? draft.acceptanceRaw : draft.gates;
|
|
182
|
+
items[items.length - 1] += ` ${line.trim()}`;
|
|
177
183
|
continue;
|
|
178
184
|
}
|
|
179
185
|
if (line.startsWith("- "))
|
|
180
186
|
invalid(draft.id, "field", `bullet is malformed: ${JSON.stringify(line)}`);
|
|
181
|
-
if (line.trim()
|
|
182
|
-
draft.
|
|
187
|
+
if (!line.trim()) {
|
|
188
|
+
draft.itemOpen = false;
|
|
189
|
+
continue;
|
|
190
|
+
}
|
|
191
|
+
// OBS-488 fail-closed invariant: a non-blank line inside a task block that no rule consumes
|
|
192
|
+
// is an error, never a silent drop — dropping author bytes is the meta-defect that hid the
|
|
193
|
+
// criterion truncation for months.
|
|
194
|
+
throw new CompileError(`Task ${draft.id} line ${index + 1}: no parse rule consumes this line: ${JSON.stringify(line)}.\n` +
|
|
195
|
+
`A non-blank line in a task block must be a "- field:" bullet, a " - " list item, or an\n` +
|
|
196
|
+
`indented continuation of the preceding field or list item. tickmarkr refuses to silently\n` +
|
|
197
|
+
`drop spec bytes (OBS-488) — move prose above the first task heading or into a field.`);
|
|
198
|
+
}
|
|
199
|
+
// OBS-488: typed-oracle prefixes parse on the COMPLETE joined item text, never its first
|
|
200
|
+
// physical line — a wrapped `command:` body or falsifier tail is part of the criterion.
|
|
201
|
+
for (const draft of drafts) {
|
|
202
|
+
for (const raw of draft.acceptanceRaw) {
|
|
203
|
+
const typed = raw.match(ORACLE_RE);
|
|
204
|
+
if (typed) {
|
|
205
|
+
const [, kind, body] = typed;
|
|
206
|
+
if (!body.trim())
|
|
207
|
+
invalid(draft.id, "acceptance", `${kind} oracle must carry a value`);
|
|
208
|
+
draft.acceptance.push(kind === "command" ? { oracle: "command", command: body.trim() }
|
|
209
|
+
: kind === "test" ? { oracle: "test", test: body.trim() }
|
|
210
|
+
: { oracle: "judge", text: body.trim() });
|
|
211
|
+
}
|
|
212
|
+
else {
|
|
213
|
+
plainCount++;
|
|
214
|
+
draft.acceptance.push(raw);
|
|
215
|
+
}
|
|
216
|
+
}
|
|
183
217
|
}
|
|
184
218
|
if (!drafts.length) {
|
|
185
219
|
throw new CompileError(`${file} has no task sections. Expected "## T1: Title" headings with field bullets.`);
|
|
@@ -358,7 +392,12 @@ export function compileNative(file) {
|
|
|
358
392
|
const result = validateGraph({
|
|
359
393
|
version: 1,
|
|
360
394
|
...(specMode ? { mode: specMode } : {}),
|
|
361
|
-
spec: {
|
|
395
|
+
spec: {
|
|
396
|
+
source: "native",
|
|
397
|
+
paths: [file],
|
|
398
|
+
hash: sha256(content),
|
|
399
|
+
...(specBase !== undefined ? { base: specBase } : {}),
|
|
400
|
+
},
|
|
362
401
|
tasks,
|
|
363
402
|
});
|
|
364
403
|
// v1.19 read-old/write-new: a plain-string acceptance item compiles as a judge oracle. This is the
|
|
@@ -378,7 +417,7 @@ export function compileNative(file) {
|
|
|
378
417
|
// OBS-170/OBS-184: warn per unreachable context: entry, classified by the action its author must
|
|
379
418
|
// take. Warn-only by operator ruling (2026-08-03) — fail-closed would have refused to compile the
|
|
380
419
|
// spec that shipped green. Fails open when git cannot answer, so non-repo fixtures stay silent.
|
|
381
|
-
const repo = trackedAtHead(dirname(file));
|
|
420
|
+
const repo = tasks.some((task) => task.context.length > 0) ? trackedAtHead(dirname(file)) : undefined;
|
|
382
421
|
if (repo) {
|
|
383
422
|
const unreachable = [];
|
|
384
423
|
for (const t of tasks) {
|
|
@@ -419,6 +458,8 @@ acceptance is required on every task (a nested list of observable outcomes).
|
|
|
419
458
|
Spec front-matter (top-level, before the first task heading):
|
|
420
459
|
mode: partner-led | risk-based | staff-led — this engagement's routing mode
|
|
421
460
|
(loses only to an explicit \`run --mode\` flag)
|
|
461
|
+
base: commit-or-ref — optional declared source base; the declaration must begin at column zero
|
|
462
|
+
(runtime enforcement happens before a run)
|
|
422
463
|
|
|
423
464
|
Fields available per task:
|
|
424
465
|
goal: outcome the task must achieve (defaults to the title if omitted)
|
|
@@ -450,6 +491,10 @@ acceptance is required on every task (a nested list of observable outcomes).
|
|
|
450
491
|
WHAT MAKES A CRITERION REAL:
|
|
451
492
|
- "test:" must name a real test asserting on recorded state, journal lines, or drawn frames, and its
|
|
452
493
|
title must match the criterion string verbatim. It also needs a collectable test path in files[].
|
|
494
|
+
- A "test:" oracle must name a TOP-LEVEL test case, never one nested under describe(). Vitest prefixes
|
|
495
|
+
nested cases with every suite title, while the acceptance gate builds an anchored pattern that
|
|
496
|
+
vitest -t applies against the FULL runner-visible name; the verbatim leaf title then selects ZERO
|
|
497
|
+
tests and the gate parks even though the suite is green. Measured: two parks in one run.
|
|
453
498
|
- NO criterion may be satisfiable by an absence, a rename, a source-text grep, or an empty collection.
|
|
454
499
|
"no file references X" is not a criterion — it passes in a repo where the feature was never built.
|
|
455
500
|
- "goal:" is NEVER verification. Prose in the goal enforces nothing: every obligation needs an
|
|
@@ -457,6 +502,35 @@ acceptance is required on every task (a nested list of observable outcomes).
|
|
|
457
502
|
task that owns both the thing it checks and the fixture it checks against.
|
|
458
503
|
- A source-only obligation (a comment or doc that a change makes false) has no lawful "test:" — verify
|
|
459
504
|
it with "judge:", which reads the DIFF and must cite a changed line in a file the task owns.
|
|
505
|
+
- A criterion that pins the SHAPE of a fix must also pin the CONDITIONS under which it runs, or a
|
|
506
|
+
correctly-shaped fix that runs SOMETIMES satisfies it. "cleanup is try/finally at all four sites"
|
|
507
|
+
is satisfied by \`finally { if (cond) cleanup() }\` — the shape is present and the fix is inert in
|
|
508
|
+
production. Say UNCONDITIONAL, or name the branch that may not exist.
|
|
509
|
+
- Ask of every criterion: COULD THIS BE SATISFIED BY CODE THAT NOTHING OUTSIDE THE TEST SUITE CALLS?
|
|
510
|
+
A criterion naming a CAPABILITY is satisfiable by a stub whose only caller is its own test, and the
|
|
511
|
+
judge cannot catch it — it reads the DIFF, and a stub's hunks are real. Name the PRODUCTION CALLER
|
|
512
|
+
that must exercise the capability, and the path it runs on; "X is supported" ships as dead code.
|
|
513
|
+
- Enumerating one axis exhaustively is what hides the others. A spec that guards PARTIAL coverage
|
|
514
|
+
site-by-site, member-by-member, can be defeated wholesale by CONDITIONAL coverage, which leaves
|
|
515
|
+
every enumeration satisfied. After you enumerate, ask what a single flag would do to the whole set.
|
|
516
|
+
- A criterion that names a behaviour must name the VALUE AT WHICH IT WOULD BREAK. Every criterion is a
|
|
517
|
+
claim about a variable — a status, a width, a count, an arrival time — and if it does not say which
|
|
518
|
+
value of that variable is the hard one, THE TEST WILL CHOOSE THE EASY ONE AND BE GREEN. Measured: a
|
|
519
|
+
grace-window criterion whose tests pinned pane status to "working" when the defect needed "blocked";
|
|
520
|
+
a width criterion tested at 2-cell task ids when the schema permits 64; a fork-budget criterion
|
|
521
|
+
tested only where concurrency <= cores when config accepts every positive integer. Each was
|
|
522
|
+
satisfied exactly as written, and each shipped the defect. State six things per criterion:
|
|
523
|
+
1. the PRODUCTION ENTRY POINT and the FINAL OBSERVABLE — what a caller invokes, what a reader sees;
|
|
524
|
+
2. the AUTHORITATIVE SOURCE of each value, never a nearby re-derivation that can disagree with it;
|
|
525
|
+
3. the QUANTIFIER, or the closed subject set the claim ranges over;
|
|
526
|
+
4. the VARIABLE whose change could falsify the claim;
|
|
527
|
+
5. TWO DISCRIMINATING CASES that vary that variable — or the invariance relation, if the claim is
|
|
528
|
+
that varying it changes nothing — drawn from the FULL domain the system accepts, never a
|
|
529
|
+
convenient subrange, because the domain is chosen by whoever must satisfy the criterion;
|
|
530
|
+
6. the concrete FALSE-CLEAN case: what a passing test would look like if the mechanism were absent.
|
|
531
|
+
A criterion that cannot name (4) is a claim about nothing and its test cannot fail. If (5) has no
|
|
532
|
+
hard value anywhere in the domain, the criterion asserts a universal that may be FALSE ABOUT THE
|
|
533
|
+
WORLD — bound it or say where it stops holding, rather than demanding a value that does not exist.
|
|
460
534
|
|
|
461
535
|
ORDERING AND OWNERSHIP:
|
|
462
536
|
- Every path has exactly ONE owning task. Two tasks writing one file must be ORDERED by deps, or the
|
|
@@ -467,6 +541,15 @@ acceptance is required on every task (a nested list of observable outcomes).
|
|
|
467
541
|
mechanism DOES for them; the most dangerous consumer is a file no task owns, because nothing fixes it
|
|
468
542
|
and no edge can order it.
|
|
469
543
|
|
|
544
|
+
EVIDENCE DISCIPLINE THE REVIEWERS ENFORCE (write criteria assuming these; violations are material):
|
|
545
|
+
- A result row whose shape is unknown, mixed, or internally contradictory is MALFORMED and must fail
|
|
546
|
+
closed. Recognizing one discriminator never licenses ignoring the rest of the row; malformed input
|
|
547
|
+
may never normalize to passed or any other gate-satisfying outcome.
|
|
548
|
+
- Preserve machine evidence according to the producer's protocol: never trim, discard, or change
|
|
549
|
+
delimiters before validation. Git path evidence MUST use -z and NUL parsing with bytes preserved;
|
|
550
|
+
inspect timeout, signal, and exit status before interpreting output, because a failed or timed-out
|
|
551
|
+
producer may never normalize to an empty successful result.
|
|
552
|
+
|
|
470
553
|
"timeout:" IS A KILL CEILING, NOT AN ESTIMATE. Never read a sum of timeouts as a predicted duration.
|
|
471
554
|
-->
|
|
472
555
|
|
|
@@ -1,7 +1,12 @@
|
|
|
1
1
|
import type { ExecutorDriver, NotifyOpts, Slot } from "./types.js";
|
|
2
2
|
export declare const MAX_BUF: number;
|
|
3
3
|
export declare const HERDR_CONTROL_VARS: readonly ["HERDR_ENV", "HERDR_SOCKET_PATH"];
|
|
4
|
-
/**
|
|
4
|
+
/**
|
|
5
|
+
* Copy of worker env with the fork cap applied and herdr control-plane vars stripped.
|
|
6
|
+
* The cap is the one the enclosing run resolved (resolvedForkCap) — a worker's suites divide the
|
|
7
|
+
* same machine the gate shells do, so both seams have to read the same run-owned number rather
|
|
8
|
+
* than a flat constant. The operator's own export still wins.
|
|
9
|
+
*/
|
|
5
10
|
export declare function sealHerdrEnv(env?: NodeJS.ProcessEnv): NodeJS.ProcessEnv;
|
|
6
11
|
/** Pane/login-shell form of the same worker env seal (herdr seed + daemon setup). */
|
|
7
12
|
export declare function herdrSealShellPrefix(env?: NodeJS.ProcessEnv): string;
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { spawn } from "node:child_process";
|
|
2
2
|
import { shq } from "../adapters/types.js";
|
|
3
|
-
import { createWorktree,
|
|
3
|
+
import { createWorktree, FORK_CAP_ENV, resolvedForkCap } from "../run/git.js";
|
|
4
4
|
// HARD-03: cap retained worker output so a chatty worker can't grow tickmarkr's heap unbounded.
|
|
5
5
|
// 2MB is safe against BOTH consumers: the largest read() call site asks for 1000 lines
|
|
6
6
|
// (daemon.ts:217/247), and every marker (TICKMARKR_RESULT, TICKMARKR_EXIT_<nonce>) is emitted at the
|
|
@@ -13,18 +13,23 @@ export const MAX_BUF = 2 * 1024 * 1024;
|
|
|
13
13
|
// (and its own herdr driver CLI calls) keep the live session. Socket path is the wire; HERDR_ENV
|
|
14
14
|
// is the "I am inside herdr" gate every agent skill checks before mutating panes.
|
|
15
15
|
export const HERDR_CONTROL_VARS = ["HERDR_ENV", "HERDR_SOCKET_PATH"];
|
|
16
|
-
/**
|
|
16
|
+
/**
|
|
17
|
+
* Copy of worker env with the fork cap applied and herdr control-plane vars stripped.
|
|
18
|
+
* The cap is the one the enclosing run resolved (resolvedForkCap) — a worker's suites divide the
|
|
19
|
+
* same machine the gate shells do, so both seams have to read the same run-owned number rather
|
|
20
|
+
* than a flat constant. The operator's own export still wins.
|
|
21
|
+
*/
|
|
17
22
|
export function sealHerdrEnv(env = process.env) {
|
|
18
23
|
const out = { ...env };
|
|
19
24
|
if (!(FORK_CAP_ENV in out))
|
|
20
|
-
out[FORK_CAP_ENV] =
|
|
25
|
+
out[FORK_CAP_ENV] = resolvedForkCap();
|
|
21
26
|
for (const k of HERDR_CONTROL_VARS)
|
|
22
27
|
delete out[k];
|
|
23
28
|
return out;
|
|
24
29
|
}
|
|
25
30
|
/** Pane/login-shell form of the same worker env seal (herdr seed + daemon setup). */
|
|
26
31
|
export function herdrSealShellPrefix(env = process.env) {
|
|
27
|
-
const forkCap = sealHerdrEnv(env)[FORK_CAP_ENV] ??
|
|
32
|
+
const forkCap = sealHerdrEnv(env)[FORK_CAP_ENV] ?? resolvedForkCap();
|
|
28
33
|
return `export ${FORK_CAP_ENV}=${shq(forkCap)}; ` +
|
|
29
34
|
HERDR_CONTROL_VARS.map((k) => `unset ${k}`).join("; ") + "; ";
|
|
30
35
|
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { type WorkerAdapter } from "../adapters/types.js";
|
|
2
|
-
import { type Task } from "../graph/schema.js";
|
|
2
|
+
import { type AcceptanceItem, type Task } from "../graph/schema.js";
|
|
3
3
|
import { type LlmVia } from "./llm.js";
|
|
4
4
|
import type { GateResult } from "./types.js";
|
|
5
5
|
export interface EvidenceCitation {
|
|
@@ -21,7 +21,27 @@ export interface JudgeVerdict {
|
|
|
21
21
|
}>;
|
|
22
22
|
}
|
|
23
23
|
export declare function judgeCriterionId(index: number): string;
|
|
24
|
+
export declare function testFilterPattern(name: string): string;
|
|
24
25
|
export declare function testFiltered(testCmd: string, name: string): string;
|
|
26
|
+
export interface VitestListedTest {
|
|
27
|
+
name: string;
|
|
28
|
+
file: string;
|
|
29
|
+
projectName?: string;
|
|
30
|
+
}
|
|
31
|
+
export type AcceptanceCorpusAuditResult = {
|
|
32
|
+
specPath: string;
|
|
33
|
+
status: "parse-failed";
|
|
34
|
+
error: string;
|
|
35
|
+
} | {
|
|
36
|
+
specPath: string;
|
|
37
|
+
status: "parsed";
|
|
38
|
+
item: AcceptanceItem;
|
|
39
|
+
namedTest?: {
|
|
40
|
+
criterion: string;
|
|
41
|
+
matches: VitestListedTest[];
|
|
42
|
+
};
|
|
43
|
+
};
|
|
44
|
+
export declare function auditAcceptanceCorpus(corpusRoot: string, listedTests: readonly VitestListedTest[]): AcceptanceCorpusAuditResult[];
|
|
25
45
|
export interface AcceptanceGateOpts {
|
|
26
46
|
testCmd?: string;
|
|
27
47
|
diffCap?: number;
|
package/dist/gates/acceptance.js
CHANGED
|
@@ -1,11 +1,15 @@
|
|
|
1
|
+
import { readdirSync } from "node:fs";
|
|
2
|
+
import { join } from "node:path";
|
|
1
3
|
import { z } from "zod";
|
|
2
4
|
import { channelKey, shq } from "../adapters/types.js";
|
|
5
|
+
import { compileNative } from "../compile/native.js";
|
|
3
6
|
import { DEFAULT_DIFF_CAP } from "../config/config.js";
|
|
4
7
|
import { renderAcceptanceItem } from "../graph/schema.js";
|
|
5
8
|
import { sh } from "../run/git.js";
|
|
6
|
-
import {
|
|
9
|
+
import { checkTaskDiffCaps, fetchTaskDiff, isProtectedEvidence, setAsideReceiptPath } from "./review.js";
|
|
7
10
|
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlm, verdictNonceLine } from "./llm.js";
|
|
8
11
|
import { classifyVerdictCause } from "./verdict-cause.js";
|
|
12
|
+
import { reviewableLogicDiff } from "./artifact-manifest.js";
|
|
9
13
|
// Fable F4: acceptance judge shares review's 900s timeout — 300s default killed frontier judges on cap-sized diffs.
|
|
10
14
|
const JUDGE_TIMEOUT_MS = 900_000;
|
|
11
15
|
const CitationSchema = z.object({ path: z.string(), line: z.number().int() });
|
|
@@ -70,16 +74,75 @@ const isJudge = (a) => typeof a === "string" || (typeof a === "object" && a.orac
|
|
|
70
74
|
function escapeRegExp(s) {
|
|
71
75
|
return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
72
76
|
}
|
|
77
|
+
// v1.89 T6: Vitest applies -t to the runner-visible full name (enclosing describe titles and the test
|
|
78
|
+
// title, space-joined). Escaping preserves literal criteria; anchoring makes equality, rather than
|
|
79
|
+
// substring containment, the named-test contract.
|
|
80
|
+
export function testFilterPattern(name) {
|
|
81
|
+
return `^${escapeRegExp(name)}$`;
|
|
82
|
+
}
|
|
73
83
|
// OBS-55: when the base command already contains `--`, append -t after forwarded args — a second `--`
|
|
74
84
|
// makes vitest treat -t as a positional file filter and the name filter is dropped.
|
|
75
85
|
export function testFiltered(testCmd, name) {
|
|
76
|
-
const pattern =
|
|
86
|
+
const pattern = testFilterPattern(name);
|
|
77
87
|
const wrapped = /^\s*(npm|yarn|pnpm|npx)\b/.test(testCmd);
|
|
78
88
|
if (wrapped && /\s--\s/.test(testCmd))
|
|
79
89
|
return `${testCmd} -t ${shq(pattern)}`;
|
|
80
90
|
const fwd = wrapped ? "-- " : "";
|
|
81
91
|
return `${testCmd} ${fwd}-t ${shq(pattern)}`;
|
|
82
92
|
}
|
|
93
|
+
function corpusSpecPaths(root) {
|
|
94
|
+
const paths = [];
|
|
95
|
+
const visit = (dir) => {
|
|
96
|
+
for (const entry of readdirSync(dir, { withFileTypes: true }).sort((a, b) => a.name.localeCompare(b.name))) {
|
|
97
|
+
const path = join(dir, entry.name);
|
|
98
|
+
if (entry.isDirectory())
|
|
99
|
+
visit(path);
|
|
100
|
+
else if (entry.isFile() && entry.name.endsWith(".spec.md"))
|
|
101
|
+
paths.push(path);
|
|
102
|
+
}
|
|
103
|
+
};
|
|
104
|
+
visit(root);
|
|
105
|
+
return paths;
|
|
106
|
+
}
|
|
107
|
+
// v1.89 T6: mechanically bind the complete spec-corpus denominator to the runner's all-project JSON
|
|
108
|
+
// listing. Every discovered path contributes either all parser-produced acceptance items or one named
|
|
109
|
+
// parse failure; exceptions are evidence, never permission to shrink the corpus silently.
|
|
110
|
+
export function auditAcceptanceCorpus(corpusRoot, listedTests) {
|
|
111
|
+
const runnerNames = listedTests.map((listed) => ({
|
|
112
|
+
listed,
|
|
113
|
+
fullName: listed.name.split(" > ").join(" "),
|
|
114
|
+
}));
|
|
115
|
+
const results = [];
|
|
116
|
+
for (const specPath of corpusSpecPaths(corpusRoot)) {
|
|
117
|
+
try {
|
|
118
|
+
const graph = compileNative(specPath);
|
|
119
|
+
for (const task of graph.tasks) {
|
|
120
|
+
for (const item of task.acceptance) {
|
|
121
|
+
results.push({
|
|
122
|
+
specPath,
|
|
123
|
+
status: "parsed",
|
|
124
|
+
item,
|
|
125
|
+
...(typeof item === "object" && item.oracle === "test"
|
|
126
|
+
? {
|
|
127
|
+
namedTest: {
|
|
128
|
+
criterion: item.test,
|
|
129
|
+
matches: runnerNames
|
|
130
|
+
.filter(({ fullName }) => fullName === item.test)
|
|
131
|
+
.map(({ listed }) => listed),
|
|
132
|
+
},
|
|
133
|
+
}
|
|
134
|
+
: {}),
|
|
135
|
+
});
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
catch (error) {
|
|
140
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
141
|
+
results.push({ specPath, status: "parse-failed", error: `${specPath}: ${message}` });
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
return results;
|
|
145
|
+
}
|
|
83
146
|
// vitest/jest summary: "Tests N passed | M skipped (T)" — count passed+failed as actually ran.
|
|
84
147
|
function testsRan(output) {
|
|
85
148
|
const lines = output.replace(/\x1b\[[\d;#]*[A-Za-z]/g, "").split("\n");
|
|
@@ -185,23 +248,6 @@ function deletedPath(section) {
|
|
|
185
248
|
const unquoted = oldPath.startsWith('"') && oldPath.endsWith('"') ? oldPath.slice(1, -1) : oldPath;
|
|
186
249
|
return unquoted.replace(/^a\//, "");
|
|
187
250
|
}
|
|
188
|
-
// OBS-134: whole-file deletions are already fully described by their path. Sending every removed line
|
|
189
|
-
// spends the cap and judge context on content that cannot exist after the change. Added and modified
|
|
190
|
-
// sections pass through byte-for-byte, so their anti-flooding budget is unchanged.
|
|
191
|
-
// v1.82 T1 clause 5: two exemptions, because this filter triggers on the very `deleted file mode` line
|
|
192
|
-
// the set-aside preserves. Protected evidence (the frozen anchors, the captured journals) bypasses the
|
|
193
|
-
// collapse so its deletion half reaches the measured text and the judge complete; and a section already
|
|
194
|
-
// set aside is never reduced a second time, or this filter would erase the receipt that replaced it.
|
|
195
|
-
function judgeRelevantDiff(diff) {
|
|
196
|
-
return diff.split(DIFF_SECTIONS).map((section) => {
|
|
197
|
-
if (setAsideReceiptPath(section))
|
|
198
|
-
return section;
|
|
199
|
-
const path = deletedPath(section);
|
|
200
|
-
if (!path || isProtectedEvidence(path))
|
|
201
|
-
return section;
|
|
202
|
-
return `deleted file: ${path}\n`;
|
|
203
|
-
}).join("");
|
|
204
|
-
}
|
|
205
251
|
// v1.82 T1 clause 7: the two shapes this task creates carry no changed hunk, so a judge asked to cite a
|
|
206
252
|
// changed line has nothing to cite — three review rounds died here. Each yields ONE citable operation
|
|
207
253
|
// fact, `{path, line: 0}`, read back from the judged text itself. Only these RECORDED facts make line 0
|
|
@@ -272,10 +318,9 @@ export async function acceptanceGate(task, worktree, baseRef, judge, via, opts =
|
|
|
272
318
|
? "WARNING: only judge oracles — no deterministic command/test oracle guards this task (deterministic preferred, spec §2).\n"
|
|
273
319
|
: "";
|
|
274
320
|
const fetched = await fetchTaskDiff(worktree, baseRef);
|
|
275
|
-
const diff =
|
|
276
|
-
const forCap = judgeRelevantDiff(fetched.forCap);
|
|
321
|
+
const diff = reviewableLogicDiff(fetched.full);
|
|
277
322
|
const diffCap = opts.diffCap ?? DEFAULT_DIFF_CAP;
|
|
278
|
-
const capFail =
|
|
323
|
+
const capFail = checkTaskDiffCaps("acceptance", fetched, diffCap, warn + detBlock);
|
|
279
324
|
if (capFail)
|
|
280
325
|
return capFail;
|
|
281
326
|
const expectedIds = judgeItems.map((_, index) => judgeCriterionId(index));
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Generated artifacts are not inferred from their directory or extension. A
|
|
3
|
+
* path receives capture accounting only when its manifest row names a
|
|
4
|
+
* registered producer and repeats that producer's current provenance exactly.
|
|
5
|
+
*/
|
|
6
|
+
export type CaptureProducerProvenance = {
|
|
7
|
+
readonly source: string;
|
|
8
|
+
readonly entrypoint: string;
|
|
9
|
+
readonly revision: string;
|
|
10
|
+
};
|
|
11
|
+
export type CaptureProducerRegistration = {
|
|
12
|
+
readonly id: string;
|
|
13
|
+
readonly provenance: CaptureProducerProvenance;
|
|
14
|
+
};
|
|
15
|
+
export type CaptureArtifactManifestEntry = {
|
|
16
|
+
readonly path: string;
|
|
17
|
+
readonly producer: string;
|
|
18
|
+
readonly provenance: CaptureProducerProvenance;
|
|
19
|
+
};
|
|
20
|
+
export type CaptureArtifactManifest = {
|
|
21
|
+
readonly version: 1;
|
|
22
|
+
readonly producers: readonly CaptureProducerRegistration[];
|
|
23
|
+
readonly artifacts: readonly CaptureArtifactManifestEntry[];
|
|
24
|
+
};
|
|
25
|
+
export declare const CAPTURE_PRODUCERS: readonly [{
|
|
26
|
+
readonly id: "cockpit-golden-frames";
|
|
27
|
+
readonly provenance: {
|
|
28
|
+
readonly source: "src/tui/cockpit/capture.ts";
|
|
29
|
+
readonly entrypoint: "regenerateGoldenFrames";
|
|
30
|
+
readonly revision: "golden-frame-v1";
|
|
31
|
+
};
|
|
32
|
+
}, {
|
|
33
|
+
readonly id: "cockpit-colour-frames";
|
|
34
|
+
readonly provenance: {
|
|
35
|
+
readonly source: "src/tui/cockpit/capture.ts";
|
|
36
|
+
readonly entrypoint: "regenerateColourFrames";
|
|
37
|
+
readonly revision: "colour-frame-v1";
|
|
38
|
+
};
|
|
39
|
+
}];
|
|
40
|
+
export declare const CAPTURE_ARTIFACT_MANIFEST: {
|
|
41
|
+
readonly version: 1;
|
|
42
|
+
readonly producers: readonly [{
|
|
43
|
+
readonly id: "cockpit-golden-frames";
|
|
44
|
+
readonly provenance: {
|
|
45
|
+
readonly source: "src/tui/cockpit/capture.ts";
|
|
46
|
+
readonly entrypoint: "regenerateGoldenFrames";
|
|
47
|
+
readonly revision: "golden-frame-v1";
|
|
48
|
+
};
|
|
49
|
+
}, {
|
|
50
|
+
readonly id: "cockpit-colour-frames";
|
|
51
|
+
readonly provenance: {
|
|
52
|
+
readonly source: "src/tui/cockpit/capture.ts";
|
|
53
|
+
readonly entrypoint: "regenerateColourFrames";
|
|
54
|
+
readonly revision: "colour-frame-v1";
|
|
55
|
+
};
|
|
56
|
+
}];
|
|
57
|
+
readonly artifacts: readonly ({
|
|
58
|
+
path: string;
|
|
59
|
+
producer: string;
|
|
60
|
+
provenance: CaptureProducerProvenance;
|
|
61
|
+
} | {
|
|
62
|
+
path: string;
|
|
63
|
+
producer: string;
|
|
64
|
+
provenance: CaptureProducerProvenance;
|
|
65
|
+
})[];
|
|
66
|
+
};
|
|
67
|
+
/** Compatibility name for the pre-manifest gate API; now derived from one manifest. */
|
|
68
|
+
export declare const REGENERABLE_CAPTURE_PATHS: readonly string[];
|
|
69
|
+
export declare const PROTECTED_EVIDENCE_PREFIXES: readonly ["tests/fixtures/cockpit/anchors/", "tests/fixtures/cockpit/sources/", "tests/fixtures/cockpit/colour/sources/"];
|
|
70
|
+
export declare function isProtectedEvidence(path: string): boolean;
|
|
71
|
+
export type ArtifactPathClassification = {
|
|
72
|
+
readonly path: string;
|
|
73
|
+
readonly kind: "logic";
|
|
74
|
+
readonly reason: "protected-evidence" | "unmanifested" | "malformed-manifest" | "missing-producer" | "stale-provenance";
|
|
75
|
+
readonly error?: string;
|
|
76
|
+
} | {
|
|
77
|
+
readonly path: string;
|
|
78
|
+
readonly kind: "capture";
|
|
79
|
+
readonly reason: "manifest-provenance";
|
|
80
|
+
readonly producer: string;
|
|
81
|
+
readonly provenance: CaptureProducerProvenance;
|
|
82
|
+
};
|
|
83
|
+
export declare function classifyArtifactPath(path: string, manifest?: unknown): ArtifactPathClassification;
|
|
84
|
+
/** The path named by a citable capture receipt, or null when there is none. */
|
|
85
|
+
export declare function setAsideReceiptPath(section: string): string | null;
|
|
86
|
+
/**
|
|
87
|
+
* Whole-file source deletions retain their operation fact instead of spending
|
|
88
|
+
* the reader cap on bytes that no longer exist. Capture receipts and protected
|
|
89
|
+
* evidence are deliberately exempt from this older reduction.
|
|
90
|
+
*/
|
|
91
|
+
export declare function reviewableLogicDiff(diff: string): string;
|
|
92
|
+
export type ArtifactDiffSection = {
|
|
93
|
+
readonly paths: readonly string[];
|
|
94
|
+
readonly kind: "capture" | "logic";
|
|
95
|
+
readonly reason: string;
|
|
96
|
+
readonly producer?: string;
|
|
97
|
+
readonly provenance?: CaptureProducerProvenance;
|
|
98
|
+
/** UTF-8 bytes left for a reviewer to read, including any receipt. */
|
|
99
|
+
readonly logicBytes: number;
|
|
100
|
+
/** UTF-8 bytes withheld behind the finite capture bound. */
|
|
101
|
+
readonly captureBytes: number;
|
|
102
|
+
};
|
|
103
|
+
export type ArtifactDiffMeasurement = {
|
|
104
|
+
readonly rendered: string;
|
|
105
|
+
readonly logicBytes: number;
|
|
106
|
+
readonly captureBytes: number;
|
|
107
|
+
readonly sections: readonly ArtifactDiffSection[];
|
|
108
|
+
};
|
|
109
|
+
/**
|
|
110
|
+
* Classify and compact a Git diff once. Invalid manifest facts remain ordinary
|
|
111
|
+
* logic. Valid capture content is replaced by one citable receipt, while its
|
|
112
|
+
* exact UTF-8 payload remains charged to the separate capture bucket.
|
|
113
|
+
*/
|
|
114
|
+
export declare function measureArtifactDiff(diff: string, manifest?: unknown): ArtifactDiffMeasurement;
|
|
115
|
+
/** The original API now delegates to the provenance-backed measurement. */
|
|
116
|
+
export declare function setAsideRegenerableCaptures(diff: string): string;
|
|
117
|
+
export declare const MIN_CAPTURE_DIFF_CAP = 1000000;
|
|
118
|
+
export declare const CAPTURE_DIFF_CAP_MULTIPLIER = 4;
|
|
119
|
+
export declare function captureDiffCapFor(logicCap: number): number;
|