tickmarkr 2.1.3 → 2.1.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/commands/status.d.ts +3 -0
- package/dist/cli/commands/status.js +23 -1
- package/dist/compile/collateral.d.ts +23 -2
- package/dist/compile/collateral.js +126 -11
- package/dist/compile/gsd.js +21 -1
- package/dist/compile/native.js +35 -3
- package/dist/gates/baseline.d.ts +23 -3
- package/dist/gates/baseline.js +75 -16
- package/dist/gates/llm.d.ts +1 -0
- package/dist/gates/llm.js +1 -0
- package/dist/gates/review.d.ts +9 -2
- package/dist/gates/review.js +51 -10
- package/dist/gates/run-gates.js +12 -1
- package/dist/run/daemon.d.ts +0 -1
- package/dist/run/daemon.js +158 -27
- package/dist/run/git.d.ts +65 -0
- package/dist/run/git.js +120 -8
- package/dist/run/journal.d.ts +47 -0
- package/dist/run/journal.js +156 -13
- package/dist/run/merge.js +13 -3
- package/dist/run/supervision.d.ts +1 -1
- package/dist/run/supervision.js +126 -35
- package/fixtures/gsd-sample/07-live-check/07-03-SUMMARY.md +4 -0
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +108 -2
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +64 -6
|
@@ -8,6 +8,9 @@ export type StatusOpts = {
|
|
|
8
8
|
readWorkerOutput?: (taskId: string, attempt: number, runId: string) => Promise<string | undefined>;
|
|
9
9
|
webhookUrl?: string;
|
|
10
10
|
postWebhook?: DecisionWebhookPost;
|
|
11
|
+
/** How often the live board rewrites its own `watch` beat. Fixtures choose it so a death drill can
|
|
12
|
+
* watch the record STOP inside a test's patience; unset means SUPERVISION_BEAT_MS. */
|
|
13
|
+
supervisionBeatMs?: number;
|
|
11
14
|
};
|
|
12
15
|
export declare const GATE_KEYS: {
|
|
13
16
|
readonly build: "B";
|
|
@@ -12,7 +12,7 @@ import { isPidLive } from "../../run/lock.js";
|
|
|
12
12
|
import { normalizeGateOutcome } from "../../run/outcome.js";
|
|
13
13
|
import { desiredPanes } from "../../run/reconcile.js";
|
|
14
14
|
import { normalizeStallSnapshot } from "../../run/stall.js";
|
|
15
|
-
import { readSupervision, supervisionText } from "../../run/supervision.js";
|
|
15
|
+
import { armSupervision, readSupervision, supervisionText } from "../../run/supervision.js";
|
|
16
16
|
import { deriveRunCockpitData, } from "../../tui/cockpit/derive.js";
|
|
17
17
|
import { COCKPIT_COLUMN_FLOOR } from "../../tui/cockpit/layout.js";
|
|
18
18
|
import { cellWidth, fitCells, wrapCells } from "../../tui/cockpit/width.js";
|
|
@@ -1370,6 +1370,20 @@ export async function status(argv, cwd = process.cwd(), opts = {}) {
|
|
|
1370
1370
|
}
|
|
1371
1371
|
return fresh;
|
|
1372
1372
|
};
|
|
1373
|
+
// SUP-06: the live board IS the `watch` tier, so it is the only process that can beat it — which is
|
|
1374
|
+
// why that tier could only ever print ABSENT about itself while every other seat had a beater. A word
|
|
1375
|
+
// one tier can only ever say in one of its five states trains every reader to skip it.
|
|
1376
|
+
//
|
|
1377
|
+
// Armed ONLY when UNBOUNDED. A bounded render is a READER, and the purity fence (D-02: a bounded
|
|
1378
|
+
// --watch leaves .tickmarkr/ byte-identical) is exactly the test that catches a reader beating on a
|
|
1379
|
+
// watcher's behalf, which would report every dead tier healthy. Armed BEFORE the first frame, so the
|
|
1380
|
+
// board's own first print already names its tier armed rather than one interval later.
|
|
1381
|
+
//
|
|
1382
|
+
// The handle is HELD and stood down in the finally. Discarding it is not a smaller version of this —
|
|
1383
|
+
// it INVERTS it: armSupervision's interval outlives the board in any host that outlives one board, so
|
|
1384
|
+
// a dead board keeps writing and reads ARMED. That is the over-claiming direction, the one an operator
|
|
1385
|
+
// acts on, and the one this instrument exists to close.
|
|
1386
|
+
const armed = bounded ? undefined : armSupervision(cwd, "watch", opts.supervisionBeatMs);
|
|
1373
1387
|
let titleSaved = false;
|
|
1374
1388
|
const restoreTitle = () => {
|
|
1375
1389
|
if (!titleSaved)
|
|
@@ -1427,6 +1441,14 @@ export async function status(argv, cwd = process.cwd(), opts = {}) {
|
|
|
1427
1441
|
}
|
|
1428
1442
|
}
|
|
1429
1443
|
finally {
|
|
1444
|
+
// Stops the interval as well as recording the hand-off: a board that returns must not leave a
|
|
1445
|
+
// timer beating for it. A board that is KILLED never reaches here, its beat ages out, and the
|
|
1446
|
+
// tier reads STALE — armed-then-lost, which is the truth about a killed board.
|
|
1447
|
+
// Boards OVERLAP — a second pane is one keystroke away — and the stand-down marker speaks for the
|
|
1448
|
+
// whole tier, so this hand-off is recorded only when no other board is still present (see
|
|
1449
|
+
// armSupervision's presence files). Otherwise the first pane closed would render the second
|
|
1450
|
+
// pane's own tier down while it is drawing frames.
|
|
1451
|
+
armed?.disarm();
|
|
1430
1452
|
if (titleSaved) {
|
|
1431
1453
|
process.removeListener("exit", restoreTitle);
|
|
1432
1454
|
restoreTitle();
|
|
@@ -30,6 +30,22 @@ export declare function classifyScopeOffenders(taskId: string, hard: ReadonlyArr
|
|
|
30
30
|
* Each line names the task id and at least one missing collateral test path.
|
|
31
31
|
*/
|
|
32
32
|
export declare function collateralLints(tasks: ReadonlyArray<Pick<Task, "id" | "files">>, repoRoot: string): string[];
|
|
33
|
+
/**
|
|
34
|
+
* Does every mention of `target` have an explicit, target-local READ relation?
|
|
35
|
+
*
|
|
36
|
+
* A `context:` entry grants READ authority and nothing else, so both blocking paths have to ask this
|
|
37
|
+
* before honouring one. Fail-closed by construction: every occurrence must match one of the narrow
|
|
38
|
+
* read forms below. Absence from a write-verb list is never evidence. Thus `ownedTable returns two`
|
|
39
|
+
* and the unlisted `ownedTable sorts the rows it already publishes` are refused, while an unrelated
|
|
40
|
+
* write in `the consumer adds a cache while reading ownedTable` does not revoke ownedTable's read
|
|
41
|
+
* authority. A target mentioned twice, once for reading and once ambiguously, is refused.
|
|
42
|
+
*
|
|
43
|
+
* This is intentionally a lexical representation rather than prose understanding. Keyed on NAMES:
|
|
44
|
+
* only the literal target is classified, so a dependency described conceptually stays invisible and
|
|
45
|
+
* an unfamiliar read phrasing stays refused. That closes the false-positive direction only; widen
|
|
46
|
+
* the affirmative grammar only with a control that goes red first.
|
|
47
|
+
*/
|
|
48
|
+
export declare function criterionReadsOnly(text: string, target: string): boolean;
|
|
33
49
|
export declare function newDirectoryLints(tasks: ReadonlyArray<Pick<Task, "id" | "files">>, repoRoot: string): string[];
|
|
34
50
|
/**
|
|
35
51
|
* OBS-76 class: sweep src/ for out-of-scope source files that reference a symbol the acceptance
|
|
@@ -104,7 +120,7 @@ export declare function goalDensityErrors(tasks: ReadonlyArray<Pick<Task, "id" |
|
|
|
104
120
|
* must be COMPLETE: any code path that cannot be read is its own error — the rule fails closed
|
|
105
121
|
* rather than trust a partial scan.
|
|
106
122
|
*/
|
|
107
|
-
export declare function symbolOwnershipErrors(tasks: ReadonlyArray<Pick<Task, "id" | "files" | "acceptance"
|
|
123
|
+
export declare function symbolOwnershipErrors(tasks: ReadonlyArray<Pick<Task, "id" | "files" | "acceptance"> & Partial<Pick<Task, "context">>>, repoRoot: string): string[];
|
|
108
124
|
/**
|
|
109
125
|
* The participation half of the config — everything the review-policy rules read. `byShape` rides
|
|
110
126
|
* along because `gates.byShape.<shape>.review: false` is a second, NON-monotone participation switch:
|
|
@@ -128,6 +144,11 @@ export declare function reviewParticipationErrors(tasks: ReadonlyArray<Pick<Task
|
|
|
128
144
|
* symbol-ownership lint and the participation config, and defaults to the invocation directory —
|
|
129
145
|
* correct for the CLI/daemon, which compile from inside the target repo.
|
|
130
146
|
*
|
|
147
|
+
* The read declaration (`context:`) rides on this parameter, on the task objects themselves — the
|
|
148
|
+
* aggregator never infers it from the repo, the config, or a sibling task. It is OPTIONAL: a caller
|
|
149
|
+
* whose task objects carry no `context` gets byte-identical verdicts to the ones it got before the
|
|
150
|
+
* field existed, because an absent declaration grants no authority at all.
|
|
151
|
+
*
|
|
131
152
|
* KNOWN GAP (R2 review, T6/T7 scope): the compile seam in src/compile/index.ts
|
|
132
153
|
* (`enforceTaskUnitContract`) does not thread `compileSource`'s `root` argument into this call, so a
|
|
133
154
|
* PROGRAMMATIC compile whose target repo differs from process.cwd() resolves the wrong root and can
|
|
@@ -138,4 +159,4 @@ export declare function reviewParticipationErrors(tasks: ReadonlyArray<Pick<Task
|
|
|
138
159
|
* critical path itself (src/gates/review.ts), so a lint this seam misses costs a late verdict rather
|
|
139
160
|
* than an unreviewed one. The compile lint is the early warning; the gate is the fail-closed backstop.
|
|
140
161
|
*/
|
|
141
|
-
export declare function taskUnitContractErrors(tasks: ReadonlyArray<Pick<Task, "id" | "goal" | "files" | "deps" | "acceptance" | "shape"
|
|
162
|
+
export declare function taskUnitContractErrors(tasks: ReadonlyArray<Pick<Task, "id" | "goal" | "files" | "deps" | "acceptance" | "shape"> & Partial<Pick<Task, "context">>>, repoRoot?: string, review?: ReviewParticipation): string[];
|
|
@@ -171,15 +171,95 @@ export function collateralLints(tasks, repoRoot) {
|
|
|
171
171
|
// v1.53 T4 (OBS-76): needles are code-shaped tokens only (camelCase / snake_case) — plain prose
|
|
172
172
|
// words never match, so prose-only criteria yield zero needles instead of alarm-fatigue noise.
|
|
173
173
|
// ponytail: token heuristic, not AST symbol resolution — promote after a version of precision data.
|
|
174
|
-
function
|
|
175
|
-
const out = new
|
|
174
|
+
function criteriaSymbolTexts(acceptance) {
|
|
175
|
+
const out = new Map();
|
|
176
176
|
for (const item of acceptance) {
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
177
|
+
const text = renderAcceptanceItem(item);
|
|
178
|
+
for (const tok of text.match(/\b[A-Za-z_][A-Za-z0-9_]*\b/g) ?? []) {
|
|
179
|
+
if (!tok.includes("_") && !/[a-z][A-Z]/.test(tok))
|
|
180
|
+
continue;
|
|
181
|
+
const texts = out.get(tok) ?? [];
|
|
182
|
+
if (!texts.includes(text))
|
|
183
|
+
texts.push(text);
|
|
184
|
+
out.set(tok, texts);
|
|
180
185
|
}
|
|
181
186
|
}
|
|
182
|
-
return
|
|
187
|
+
return out;
|
|
188
|
+
}
|
|
189
|
+
function criteriaSymbols(acceptance) {
|
|
190
|
+
return [...criteriaSymbolTexts(acceptance).keys()].sort();
|
|
191
|
+
}
|
|
192
|
+
// These are affirmative READ relations, not a list of things that are not writes. The distinction
|
|
193
|
+
// is the fail-closed boundary: an unknown predicate never earns context[] authority. A target is
|
|
194
|
+
// readable when it is the object of an explicit observation (`reading ownedTable`) or the subject
|
|
195
|
+
// of an explicitly pre-existing state (`ownedTable already publishes`). An unqualified declarative
|
|
196
|
+
// predicate (`ownedTable returns two rows`) is deliberately absent because it can demand a change.
|
|
197
|
+
const DIRECT_READ_BEFORE = /\b(?:consult|consults|consulting|inspect|inspects|inspecting|read|reads|reading|reference|references|referencing)\s+(?:(?:the|an?)\s+)?(?:(?:current|existing|unchanged)\s+)?$/i;
|
|
198
|
+
const PREEXISTING_STATE_AFTER = /^\s*(?:(?:and|or)\s+\S+\s+)*(?:(?:,\s*)?which\s+)?(?:already|currently)\s+(?:carries|carry|contains?|declares?|defines?|exposes?|holds?|owns?|provides?|publish|publishes|returns?|supplies|supply|yields?)\b/i;
|
|
199
|
+
const PASSIVE_READ_AFTER = /^\s+(?:is|remains)\s+(?:read|referenced|unchanged|untouched|as[- ]is)\b/i;
|
|
200
|
+
// Once a target-local read has been established, a same-clause continuation can revoke it. These
|
|
201
|
+
// vetoes are structural rather than an attempted exhaustive list of write verbs: an action followed
|
|
202
|
+
// by `it` / `them` / `the same ...` is target-directed but ambiguous, and a coordinated passive
|
|
203
|
+
// participle inherits the target as its subject. Both therefore fail closed. The scan stops at an
|
|
204
|
+
// actual sentence/clause boundary; a dot inside a later path is not one.
|
|
205
|
+
const TARGET_ANAPHOR_AFTER_COORDINATOR = /\b(?:and|or|then|before|after|while)\b[^,;.!?\n]{0,80}\b(?:it|them|those|the\s+same(?:\s+[A-Za-z][A-Za-z0-9_-]*)?)\b/i;
|
|
206
|
+
const TARGET_PASSIVE_AFTER_SUBJECT_READ = /\b(?:and|or|then|before|after)\s+(?:then\s+)?(?:(?:is|gets?|becomes?|being)\s+)?[A-Za-z]+(?:ed|en)\b/i;
|
|
207
|
+
function sameClauseAfterTarget(after) {
|
|
208
|
+
const boundary = after.search(/[;!?\n]|\.(?=\s|$)/);
|
|
209
|
+
return boundary === -1 ? after : after.slice(0, boundary);
|
|
210
|
+
}
|
|
211
|
+
function targetOccurrences(text, target) {
|
|
212
|
+
if (!target)
|
|
213
|
+
return [];
|
|
214
|
+
const out = [];
|
|
215
|
+
const wordAtStart = /[A-Za-z0-9_]/.test(target[0]);
|
|
216
|
+
const wordAtEnd = /[A-Za-z0-9_]/.test(target[target.length - 1]);
|
|
217
|
+
for (let at = text.indexOf(target); at !== -1; at = text.indexOf(target, at + target.length)) {
|
|
218
|
+
const before = text[at - 1];
|
|
219
|
+
const after = text[at + target.length];
|
|
220
|
+
if (wordAtStart && before !== undefined && /[A-Za-z0-9_]/.test(before))
|
|
221
|
+
continue;
|
|
222
|
+
if (wordAtEnd && after !== undefined && /[A-Za-z0-9_]/.test(after))
|
|
223
|
+
continue;
|
|
224
|
+
out.push(at);
|
|
225
|
+
}
|
|
226
|
+
return out;
|
|
227
|
+
}
|
|
228
|
+
/**
|
|
229
|
+
* Does every mention of `target` have an explicit, target-local READ relation?
|
|
230
|
+
*
|
|
231
|
+
* A `context:` entry grants READ authority and nothing else, so both blocking paths have to ask this
|
|
232
|
+
* before honouring one. Fail-closed by construction: every occurrence must match one of the narrow
|
|
233
|
+
* read forms below. Absence from a write-verb list is never evidence. Thus `ownedTable returns two`
|
|
234
|
+
* and the unlisted `ownedTable sorts the rows it already publishes` are refused, while an unrelated
|
|
235
|
+
* write in `the consumer adds a cache while reading ownedTable` does not revoke ownedTable's read
|
|
236
|
+
* authority. A target mentioned twice, once for reading and once ambiguously, is refused.
|
|
237
|
+
*
|
|
238
|
+
* This is intentionally a lexical representation rather than prose understanding. Keyed on NAMES:
|
|
239
|
+
* only the literal target is classified, so a dependency described conceptually stays invisible and
|
|
240
|
+
* an unfamiliar read phrasing stays refused. That closes the false-positive direction only; widen
|
|
241
|
+
* the affirmative grammar only with a control that goes red first.
|
|
242
|
+
*/
|
|
243
|
+
export function criterionReadsOnly(text, target) {
|
|
244
|
+
const occurrences = targetOccurrences(text, target);
|
|
245
|
+
return occurrences.length > 0 && occurrences.every((at) => {
|
|
246
|
+
// Backticks quote the target, not the relation, so discard only the adjacent delimiters.
|
|
247
|
+
const before = text.slice(Math.max(0, at - 96), at).replace(/`$/, "");
|
|
248
|
+
const after = text.slice(at + target.length).replace(/^`/, "");
|
|
249
|
+
const preexisting = PREEXISTING_STATE_AFTER.exec(after);
|
|
250
|
+
const passive = PASSIVE_READ_AFTER.exec(after);
|
|
251
|
+
if (!DIRECT_READ_BEFORE.test(before) && !preexisting && !passive)
|
|
252
|
+
return false;
|
|
253
|
+
const clause = sameClauseAfterTarget(after);
|
|
254
|
+
if (TARGET_ANAPHOR_AFTER_COORDINATOR.test(clause))
|
|
255
|
+
return false;
|
|
256
|
+
// A subject-position read (`target is read` / `target already publishes`) leaves the target as
|
|
257
|
+
// the inherited subject of a coordinated passive: `... and then rewritten by the consumer` is
|
|
258
|
+
// a change demand, not a read. Object-position reads need the anaphor veto above instead.
|
|
259
|
+
const subjectRead = preexisting ?? passive;
|
|
260
|
+
return !subjectRead
|
|
261
|
+
|| !TARGET_PASSIVE_AFTER_SUBJECT_READ.test(clause.slice(subjectRead[0].length));
|
|
262
|
+
});
|
|
183
263
|
}
|
|
184
264
|
const ARCH_PAGES = ["docs/codebase/ARCHITECTURE.md", "docs/codebase/STRUCTURE.md"];
|
|
185
265
|
function topLevelSrcDir(file) {
|
|
@@ -519,8 +599,8 @@ function walkAllCode(repoRoot) {
|
|
|
519
599
|
*/
|
|
520
600
|
export function symbolOwnershipErrors(tasks, repoRoot) {
|
|
521
601
|
const perTask = tasks
|
|
522
|
-
.map((t) => ({ t,
|
|
523
|
-
.filter((x) => x.
|
|
602
|
+
.map((t) => ({ t, bySymbol: criteriaSymbolTexts(t.acceptance ?? []) }))
|
|
603
|
+
.filter((x) => x.bySymbol.size);
|
|
524
604
|
if (!perTask.length)
|
|
525
605
|
return [];
|
|
526
606
|
const { files: codeFiles, unreadable } = walkAllCode(repoRoot);
|
|
@@ -531,7 +611,7 @@ export function symbolOwnershipErrors(tasks, repoRoot) {
|
|
|
531
611
|
for (const u of unreadable)
|
|
532
612
|
errors.push(incomplete(u));
|
|
533
613
|
// resolve each symbol to its definition site(s) once, shared across tasks
|
|
534
|
-
const allSymbols = [...new Set(perTask.flatMap((x) => x.
|
|
614
|
+
const allSymbols = [...new Set(perTask.flatMap((x) => [...x.bySymbol.keys()]))];
|
|
535
615
|
const res = new Map(allSymbols.map((s) => [s, definitionRe(s)]));
|
|
536
616
|
const sites = new Map(allSymbols.map((s) => [s, []]));
|
|
537
617
|
for (const cf of codeFiles) {
|
|
@@ -548,16 +628,46 @@ export function symbolOwnershipErrors(tasks, repoRoot) {
|
|
|
548
628
|
sites.get(sym).push(cf);
|
|
549
629
|
}
|
|
550
630
|
}
|
|
551
|
-
for (const { t,
|
|
631
|
+
for (const { t, bySymbol } of perTask) {
|
|
552
632
|
// OBS-22: scopeGate accepts picomatch globs; ownership must agree.
|
|
553
633
|
const scoped = filesGlob(t.files.map((f) => f.replace(/^\.\//, "")));
|
|
554
|
-
|
|
634
|
+
// THE RULE AT THIS SITE: a `context:`
|
|
635
|
+
// entry declares a path the worker may READ, so it answers "can the worker satisfy a criterion
|
|
636
|
+
// that only reads this symbol" (yes) and never "may the worker change it" (no — that authority
|
|
637
|
+
// comes from files[] alone). Hence the exemption is conditional on the criteria that actually
|
|
638
|
+
// name the symbol, and it is asked of the SYMBOL, not of the sentence: `criterionReadsOnly` fails
|
|
639
|
+
// closed, so a criterion demanding the symbol CHANGE — and any phrasing that does not prove it
|
|
640
|
+
// reads the symbol — is still refused, and still told to widen files[] where the change is what
|
|
641
|
+
// it asks for. An unrelated in-scope write does not change that per-symbol answer. Keyed on NAMES:
|
|
642
|
+
// only a path literally written in context[] is seen here, so a
|
|
643
|
+
// read dependency an author described in prose remains invisible to this lint. That closes the
|
|
644
|
+
// false-positive direction only — nothing here narrows what the rule refuses.
|
|
645
|
+
const declaredContext = (t.context ?? []).map((f) => f.replace(/^\.\//, ""));
|
|
646
|
+
const readable = declaredContext.length ? filesGlob(declaredContext) : () => false;
|
|
647
|
+
for (const [sym, texts] of bySymbol) {
|
|
555
648
|
const defs = sites.get(sym) ?? [];
|
|
556
649
|
if (defs.length !== 1)
|
|
557
650
|
continue; // unknown or ambiguous — silent by ruling
|
|
558
651
|
const site = defs[0];
|
|
559
652
|
if (scoped(site))
|
|
560
653
|
continue; // defined inside the task's own write surface
|
|
654
|
+
const writes = !texts.every((text) => criterionReadsOnly(text, sym));
|
|
655
|
+
if (readable(site) && !writes)
|
|
656
|
+
continue; // read authority is declared and read authority is all it needs
|
|
657
|
+
if (readable(site)) {
|
|
658
|
+
errors.push(`${t.id}: criterion identifier "${sym}" is defined only in ${site}, which is declared in `
|
|
659
|
+
+ `context[] but not in files[] — a context: entry grants READ authority only, and this `
|
|
660
|
+
+ `criterion requires changing "${sym}"; add ${site} to files[] or reword the criterion to `
|
|
661
|
+
+ `require only reading it (OBS-248).`);
|
|
662
|
+
continue;
|
|
663
|
+
}
|
|
664
|
+
if (!writes) {
|
|
665
|
+
errors.push(`${t.id}: criterion identifier "${sym}" is defined only in ${site}, which is in neither `
|
|
666
|
+
+ `files[] nor context[] — the criterion only reads "${sym}", so declare ${site} in `
|
|
667
|
+
+ `context[]; widening files[] would grant write authority this criterion never asks for `
|
|
668
|
+
+ `(OBS-248).`);
|
|
669
|
+
continue;
|
|
670
|
+
}
|
|
561
671
|
errors.push(`${t.id}: criterion identifier "${sym}" is defined only in ${site}, which is not in files[] — `
|
|
562
672
|
+ `add ${site} to files[] or reword the criterion; a worker scoped to files[] cannot satisfy a `
|
|
563
673
|
+ `criterion whose enabling symbol lives outside it (OBS-248).`);
|
|
@@ -630,6 +740,11 @@ function activeReviewParticipation(repoRoot) {
|
|
|
630
740
|
* symbol-ownership lint and the participation config, and defaults to the invocation directory —
|
|
631
741
|
* correct for the CLI/daemon, which compile from inside the target repo.
|
|
632
742
|
*
|
|
743
|
+
* The read declaration (`context:`) rides on this parameter, on the task objects themselves — the
|
|
744
|
+
* aggregator never infers it from the repo, the config, or a sibling task. It is OPTIONAL: a caller
|
|
745
|
+
* whose task objects carry no `context` gets byte-identical verdicts to the ones it got before the
|
|
746
|
+
* field existed, because an absent declaration grants no authority at all.
|
|
747
|
+
*
|
|
633
748
|
* KNOWN GAP (R2 review, T6/T7 scope): the compile seam in src/compile/index.ts
|
|
634
749
|
* (`enforceTaskUnitContract`) does not thread `compileSource`'s `root` argument into this call, so a
|
|
635
750
|
* PROGRAMMATIC compile whose target repo differs from process.cwd() resolves the wrong root and can
|
package/dist/compile/gsd.js
CHANGED
|
@@ -106,6 +106,22 @@ function extractWriteDirectives(body, storedPath) {
|
|
|
106
106
|
return writes;
|
|
107
107
|
}
|
|
108
108
|
const isPlainObject = (v) => typeof v === "object" && v !== null && !Array.isArray(v);
|
|
109
|
+
function summaryCompletionStatus(planFile) {
|
|
110
|
+
const summaryFile = join(dirname(planFile), `${basename(planFile).replace(/-PLAN\.md$/, "")}-SUMMARY.md`);
|
|
111
|
+
if (!existsSync(summaryFile))
|
|
112
|
+
return undefined;
|
|
113
|
+
try {
|
|
114
|
+
const content = readFileSync(summaryFile, "utf8");
|
|
115
|
+
const fmMatch = /^---\r?\n([\s\S]*?)\r?\n---/.exec(content);
|
|
116
|
+
if (!fmMatch)
|
|
117
|
+
return undefined;
|
|
118
|
+
const fm = parseYaml(fmMatch[1]);
|
|
119
|
+
return isPlainObject(fm) ? fm.status : undefined;
|
|
120
|
+
}
|
|
121
|
+
catch {
|
|
122
|
+
return undefined;
|
|
123
|
+
}
|
|
124
|
+
}
|
|
109
125
|
// schema-legal, branch-safe id segment: dots etc. → dashes; no "--" runs (task-branch separator)
|
|
110
126
|
// or trailing dash — the id schema rejects both
|
|
111
127
|
const sanitize = (s) => s
|
|
@@ -170,7 +186,11 @@ function compileOne(file, storedPath) {
|
|
|
170
186
|
.filter((p) => !p.includes("$") && !p.startsWith("~") && !p.startsWith("/") && !p.split("/").includes(".."));
|
|
171
187
|
const taskCount = [...body.matchAll(/<task[\s>]/g)].length;
|
|
172
188
|
const humanGate = fm.autonomous === false || /<task type="checkpoint:/.test(body);
|
|
173
|
-
const
|
|
189
|
+
const summaryStatus = summaryCompletionStatus(file);
|
|
190
|
+
// GSD source rule (`/gsd:quick status`): a SUMMARY is complete only when its frontmatter says
|
|
191
|
+
// `status: complete`; when `status` is absent it reports `INCOMPLETE`. The compiler must preserve
|
|
192
|
+
// that distinction because an artifact can record a reviewed failure rather than completion.
|
|
193
|
+
const done = summaryStatus === "complete";
|
|
174
194
|
const files = strings(fm.files_modified).map((f) => f.replace(/^\.\//, ""));
|
|
175
195
|
assertWriteScope(file, `P${key}`, files, extractWriteDirectives(body, storedPath));
|
|
176
196
|
// fail closed (D-03): a silently dropped floor/pin routes the task cheap instead of erroring
|
package/dist/compile/native.js
CHANGED
|
@@ -4,6 +4,7 @@ import { basename, dirname, join, resolve } from "node:path";
|
|
|
4
4
|
import { filesGlob } from "../graph/files-glob.js";
|
|
5
5
|
import picomatch from "picomatch";
|
|
6
6
|
import { GATE_NAMES, GRAPH_ROUTING_MODES, ORACLES, renderAcceptanceItem, SHAPES, TIERS, validateGraph, } from "../graph/schema.js";
|
|
7
|
+
import { criterionReadsOnly } from "./collateral.js";
|
|
7
8
|
import { CompileError, inferShape, sha256 } from "./common.js";
|
|
8
9
|
// OBS-170/OBS-184: `context:` is a promise to the worker, and nothing ever checked it could be kept.
|
|
9
10
|
// Workers run in `git worktree add <baseRef>` (run/git.ts:169-176), which materialises the base
|
|
@@ -169,20 +170,51 @@ function renderedObservables(text) {
|
|
|
169
170
|
function criterionScopeFinding(task, text, criterion, id, tests) {
|
|
170
171
|
if (task.files.length === 0)
|
|
171
172
|
return undefined; // empty files[] is deliberately unrestricted
|
|
172
|
-
const
|
|
173
|
+
const strip = (entry) => entry.replace(/^\.\//, ""); // Q120s shared matcher below
|
|
174
|
+
const inFiles = filesGlob(task.files.map(strip));
|
|
175
|
+
const context = (task.context ?? []).map(strip);
|
|
176
|
+
const inContext = context.length ? filesGlob(context) : () => false;
|
|
177
|
+
// THE RULE AT THIS SITE: `files:` clears a named path outright, because it is write authority and
|
|
178
|
+
// write authority covers reading too. `context:` clears one ONLY for a criterion that reads it and
|
|
179
|
+
// does not change it — a read declaration is not a second write surface, and unioning it in unconditionally
|
|
180
|
+
// would promote read authority to write authority for exactly the criterion that asks for the
|
|
181
|
+
// change. The question is asked per PATH — each occurrence of each declared path must carry its own
|
|
182
|
+
// affirmative read relation. A write elsewhere in the criterion therefore cannot revoke a named
|
|
183
|
+
// dependency's read exemption, and a read elsewhere cannot confer one on a path being changed. It
|
|
184
|
+
// is the same target-specific question the symbol-ownership rule asks (collateral.ts), and it fails
|
|
185
|
+
// closed: any phrasing that does not prove the criterion reads the path is refused, as it was before
|
|
186
|
+
// this check consulted the declaration at all. Keyed on NAMES: only a path literally written in files[]
|
|
187
|
+
// or context[] is matched, so a dependency the author described conceptually and never wrote out
|
|
188
|
+
// stays invisible to this check. That closes the false-positive direction only — a path in neither
|
|
189
|
+
// declaration is refused exactly as before.
|
|
190
|
+
const declared = (path) => inFiles(path) || (inContext(path) && criterionReadsOnly(text, path));
|
|
173
191
|
const named = namedCriterionPaths(text, tests);
|
|
174
192
|
const { exact, denominators } = renderedObservables(text);
|
|
175
193
|
const asserting = tests.filter((test) => exact.some((token) => test.text.includes(token))
|
|
176
194
|
|| denominators.some((denominator) => new RegExp(`(?:^|\\D)\\d+/${denominator}(?:\\D|$)`).test(test.text))).map((test) => test.path);
|
|
177
|
-
const missing = [...new Set([...named, ...asserting])].filter((path) => !
|
|
195
|
+
const missing = [...new Set([...named, ...asserting])].filter((path) => !declared(path));
|
|
178
196
|
if (missing.length === 0)
|
|
179
197
|
return undefined;
|
|
198
|
+
// The remedy names the authority the criterion actually needs. A criterion that only READS the
|
|
199
|
+
// producer is repaired by context[]; instructing its author to widen files[] is the product
|
|
200
|
+
// emitting the unsafe workaround itself. Only a criterion that requires CHANGING the producer is
|
|
201
|
+
// told to widen the write surface.
|
|
202
|
+
const readMissing = missing.filter((path) => criterionReadsOnly(text, path));
|
|
203
|
+
const writeMissing = missing.filter((path) => !criterionReadsOnly(text, path));
|
|
204
|
+
const remedy = [
|
|
205
|
+
readMissing.length === 0 ? "" : `the criterion only reads ${readMissing.length === 1 ? "it" : "them"} `
|
|
206
|
+
+ `(${readMissing.join(", ")}), so add ${readMissing.length === 1 ? "it" : "them"} to context[] — `
|
|
207
|
+
+ `files[] would grant write authority this criterion never asks for`,
|
|
208
|
+
writeMissing.length === 0 ? "" : `the criterion requires changing ${writeMissing.length === 1 ? "it" : "them"} `
|
|
209
|
+
+ `(or does not establish a read-only relation) (${writeMissing.join(", ")}), so add `
|
|
210
|
+
+ `${writeMissing.length === 1 ? "it" : "them"} to files[]`,
|
|
211
|
+
].filter(Boolean).join("; ");
|
|
180
212
|
return {
|
|
181
213
|
code: "criterion-scope",
|
|
182
214
|
fixtureId: id,
|
|
183
215
|
taskId: task.id,
|
|
184
216
|
criterion,
|
|
185
|
-
detail: `criterion names asserting file${missing.length === 1 ? "" : "s"} outside expanded files[]: ${missing.join(", ")}`,
|
|
217
|
+
detail: `criterion names asserting file${missing.length === 1 ? "" : "s"} outside expanded files[] and context[]: ${missing.join(", ")} — ${remedy}`,
|
|
186
218
|
};
|
|
187
219
|
}
|
|
188
220
|
function exportedIdentifier(root, files, identifier) {
|
package/dist/gates/baseline.d.ts
CHANGED
|
@@ -1,7 +1,20 @@
|
|
|
1
1
|
import type { TickmarkrConfig } from "../config/config.js";
|
|
2
2
|
import type { AcceptanceItem } from "../graph/schema.js";
|
|
3
|
-
import { type ShResult } from "../run/git.js";
|
|
3
|
+
import { type RunCapacity, type ShResult } from "../run/git.js";
|
|
4
4
|
import type { GateResult } from "./types.js";
|
|
5
|
+
/**
|
|
6
|
+
* T7: the capacity a gate's own command ran under rides the RESULT, beside the verdict it explains,
|
|
7
|
+
* rather than inside `meta` — `meta` is a machine-readable extras bag several callers compare
|
|
8
|
+
* wholesale, and identity that a later session keys reuse on does not belong in a bag. Declared here,
|
|
9
|
+
* at the one producer, because only a battery gate has a command whose child received a fork cap:
|
|
10
|
+
* every other gate leaves the field absent, which is the honest reading of "this gate divided
|
|
11
|
+
* nothing". The daemon lifts it verbatim onto the journal's gate row (src/run/daemon.ts).
|
|
12
|
+
*/
|
|
13
|
+
declare module "./types.js" {
|
|
14
|
+
interface GateResult {
|
|
15
|
+
capacity?: RunCapacity;
|
|
16
|
+
}
|
|
17
|
+
}
|
|
5
18
|
export interface BaselineCommand {
|
|
6
19
|
/**
|
|
7
20
|
* Absent when the capture returned no verdict — see `infra`. A pre-v1.90 baseline can also lack it
|
|
@@ -21,8 +34,15 @@ export interface BaselineCommand {
|
|
|
21
34
|
/** The ceiling that measurement implies, persisted so every later battery uses the same number. */
|
|
22
35
|
ceilingMs?: number;
|
|
23
36
|
/**
|
|
24
|
-
*
|
|
25
|
-
*
|
|
37
|
+
* T7: the capacity this command's capture child ran under — the fork cap it received and the cores
|
|
38
|
+
* that cap was divided from. Absent in every pre-T7 baseline, which is exactly what makes those
|
|
39
|
+
* entries keep their current forgiveness; a MALFORMED one fails closed instead (git.ts readCapacity).
|
|
40
|
+
*/
|
|
41
|
+
capacity?: RunCapacity;
|
|
42
|
+
/**
|
|
43
|
+
* The capture did not return a trustworthy verdict: it was SIGKILLed at its ceiling, or its output
|
|
44
|
+
* proves the machine was exhausted while it ran. The entry therefore carries a CAUSE and no
|
|
45
|
+
* verdict: no exit code, no fingerprints, nothing forgivable.
|
|
26
46
|
*/
|
|
27
47
|
infra?: true;
|
|
28
48
|
}
|
package/dist/gates/baseline.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { existsSync, readFileSync } from "node:fs";
|
|
2
2
|
import { join } from "node:path";
|
|
3
|
-
import { DEFAULT_SHELL_TIMEOUT_MS, sh } from "../run/git.js";
|
|
3
|
+
import { DEFAULT_SHELL_TIMEOUT_MS, describeCapacity, sameCapacity, sh } from "../run/git.js";
|
|
4
4
|
// incident #2 (run-20260709-104447): a vitest ✓ PASS line with "error" in the test NAME, wrapped in ANSI
|
|
5
5
|
// codes that varied between baseline and worktree runs, was reported as a "new failure". Strip ANSI first;
|
|
6
6
|
// a pass-marker line is never a failure. [\d;#] covers raw ANSI and digit-normalized ANSI ("\x1b[#m") from
|
|
@@ -118,6 +118,10 @@ const VOCAB_RE = /\b(?:error|fail(?:ed|ure|ing)?)\b/i;
|
|
|
118
118
|
// this runner-output classifier rather than applied to any judge-authored reason text. A real test
|
|
119
119
|
// failure still dominates below because one regression-shaped line makes the whole output regression.
|
|
120
120
|
const INFRA_RE = /\bE(?:AGAIN|MFILE|NFILE|NOMEM|NOSPC)\b|JavaScript heap out of memory|Cannot allocate memory|Resource temporarily unavailable|Token not found in system keyring|Process from config\.webServer was not able to start/i;
|
|
121
|
+
// Capture invalidation is deliberately narrower than the gate's infrastructure vocabulary above:
|
|
122
|
+
// keyring/config-webServer startup failures remain gate concerns, while this policy is specifically
|
|
123
|
+
// for evidence that the capture ran while the machine was resource-starved.
|
|
124
|
+
const CAPTURE_EXHAUSTION_RE = /\bE(?:AGAIN|MFILE|NFILE|NOMEM|NOSPC)\b|JavaScript heap out of memory|Cannot allocate memory|Resource temporarily unavailable/i;
|
|
121
125
|
// A named error CLASS ("AssertionError", "TypeError", "MyDomainError") — never bare "Error", which
|
|
122
126
|
// is what an errno report itself is headed with (`Error: spawn EAGAIN`). The prefix is required.
|
|
123
127
|
const ERROR_CLASS_RE = /\b[A-Za-z][A-Za-z0-9]*Error\b/;
|
|
@@ -133,6 +137,18 @@ export function classifyFailureOutput(output) {
|
|
|
133
137
|
return "regression";
|
|
134
138
|
return lines.some(isInfraLine) ? "infra" : undefined;
|
|
135
139
|
}
|
|
140
|
+
/**
|
|
141
|
+
* Capture validity asks a different question from gate classification. At a gate, one genuine
|
|
142
|
+
* regression line must outrank adjacent errno evidence so a real defect is never laundered as infra.
|
|
143
|
+
* At capture, any such evidence invalidates the whole measurement: once the machine was exhausted,
|
|
144
|
+
* no failure in that incomplete environment can safely become pre-existing forgiveness. Keep this
|
|
145
|
+
* separate from `classifyFailureOutput` so changing capture policy cannot move gate verdicts.
|
|
146
|
+
*/
|
|
147
|
+
const captureHasInvalidatingInfra = (output) => output
|
|
148
|
+
.split("\n")
|
|
149
|
+
.map((l) => l.replace(ANSI_RE, ""))
|
|
150
|
+
.filter((l) => !PASS_LINE_RE.test(l))
|
|
151
|
+
.some((l) => CAPTURE_EXHAUSTION_RE.test(l));
|
|
136
152
|
const normalizeLine = (l) => l.replace(/\d+/g, "#").replace(/\s+/g, " ").trim();
|
|
137
153
|
// Vitest's default reporter names file durations as
|
|
138
154
|
// `✓ |project| tests/example.test.ts (12 tests) 1.23s` (❯ for a red file). The runner may be
|
|
@@ -370,6 +386,15 @@ export function ceilingKillResult(gate, r, ceilingMs) {
|
|
|
370
386
|
* shell asks it to finish faster than the thing it is measuring.
|
|
371
387
|
*/
|
|
372
388
|
export const CAPTURE_CEILING_MS = 1_800_000;
|
|
389
|
+
const invalidCaptureEntry = (durationMs) => ({
|
|
390
|
+
infra: true,
|
|
391
|
+
fingerprints: [],
|
|
392
|
+
durationMs,
|
|
393
|
+
fileDurationSumMs: null,
|
|
394
|
+
impliedParallelism: null,
|
|
395
|
+
longestFile: null,
|
|
396
|
+
ceilingMs: effectiveCeilingMs({ durationMs }),
|
|
397
|
+
});
|
|
373
398
|
export async function captureBaseline(cwd, commands) {
|
|
374
399
|
const base = { commands: {} };
|
|
375
400
|
for (const [name, cmd] of Object.entries(commands)) {
|
|
@@ -396,18 +421,22 @@ export async function captureBaseline(cwd, commands) {
|
|
|
396
421
|
console.error(`tickmarkr: baseline capture for "${name}" was killed at its ${CAPTURE_CEILING_MS}ms ceiling — `
|
|
397
422
|
+ `it recorded NO fingerprints, so nothing is forgiven and every gate will treat a pre-existing `
|
|
398
423
|
+ `failure as a fresh one. Raise the ceiling or shorten the command.`);
|
|
399
|
-
base.commands[name] =
|
|
400
|
-
infra: true,
|
|
401
|
-
fingerprints: [],
|
|
402
|
-
durationMs,
|
|
403
|
-
fileDurationSumMs: null,
|
|
404
|
-
impliedParallelism: null,
|
|
405
|
-
longestFile: null,
|
|
406
|
-
ceilingMs: effectiveCeilingMs({ durationMs }),
|
|
407
|
-
};
|
|
424
|
+
base.commands[name] = invalidCaptureEntry(durationMs);
|
|
408
425
|
continue;
|
|
409
426
|
}
|
|
410
427
|
const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
|
|
428
|
+
// Run 2137: the child exited and printed ordinary FAIL/AssertionError lines, so the gate-side
|
|
429
|
+
// discriminator correctly called the mixed output a regression. But the same output also said
|
|
430
|
+
// `spawn EAGAIN`: the machine had run out of processes while the pristine-tree measurement was
|
|
431
|
+
// being taken. A capture cannot know which red lines predated that shortage and which it caused,
|
|
432
|
+
// so none may become a fingerprint every later task gets to forgive.
|
|
433
|
+
if (captureHasInvalidatingInfra(raw)) {
|
|
434
|
+
console.error(`tickmarkr: baseline capture for "${name}" completed with process/resource-exhaustion evidence — `
|
|
435
|
+
+ `it recorded NO exit-code verdict and NO fingerprints, so nothing is forgiven for this command; `
|
|
436
|
+
+ `the measurement cannot distinguish a pre-existing failure from one caused by exhaustion.`);
|
|
437
|
+
base.commands[name] = invalidCaptureEntry(durationMs);
|
|
438
|
+
continue;
|
|
439
|
+
}
|
|
411
440
|
base.commands[name] = {
|
|
412
441
|
exitCode: r.code,
|
|
413
442
|
// a command that exits 0 has no failures to fingerprint — recording any would be a lie the
|
|
@@ -417,6 +446,9 @@ export async function captureBaseline(cwd, commands) {
|
|
|
417
446
|
durationMs,
|
|
418
447
|
...fileTiming(raw, durationMs),
|
|
419
448
|
ceilingMs: effectiveCeilingMs({ durationMs }),
|
|
449
|
+
// T7: the world this measurement was taken in, so a later reader can ask whether its own world
|
|
450
|
+
// is the same one. Recorded from THIS command's own shell result, never re-derived here.
|
|
451
|
+
...(r.capacity ? { capacity: r.capacity } : {}),
|
|
420
452
|
};
|
|
421
453
|
}
|
|
422
454
|
const names = Object.keys(commands);
|
|
@@ -487,17 +519,28 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
|
487
519
|
const entry = baseline.commands[name];
|
|
488
520
|
const ceilingMs = effectiveCeilingMs(entry);
|
|
489
521
|
const r = await sh(cmd, cwd, ceilingMs);
|
|
522
|
+
// T7: every verdict below carries the capacity ITS OWN command ran under, taken off the shell
|
|
523
|
+
// result rather than re-derived after the fact. The skip row above ran no command and therefore
|
|
524
|
+
// states no capacity — a row that never divided the machine must not claim that it did.
|
|
525
|
+
const record = (g) => {
|
|
526
|
+
results.push(r.capacity ? { ...g, capacity: r.capacity } : g);
|
|
527
|
+
};
|
|
528
|
+
// …and whether the entry that would forgive this command was measured in the same world. A
|
|
529
|
+
// baseline captured under a different fork cap forgives nothing: its fingerprints describe a
|
|
530
|
+
// machine divided by a different number. Absent capacity (every pre-T7 baseline) still forgives
|
|
531
|
+
// exactly as it does today; a malformed one fails closed (git.ts sameCapacity).
|
|
532
|
+
const comparable = sameCapacity(entry?.capacity, r.capacity);
|
|
490
533
|
// Q24: the kill is read BEFORE the exit code is interpreted at all. A SIGKILLed battery has
|
|
491
534
|
// whatever partial output it had flushed — typically no failure shape — so every path below
|
|
492
535
|
// would otherwise turn a timeout into a claim about the work: "no recognizable failure lines"
|
|
493
536
|
// when the baseline was green, or a forgiven pre-existing red when it was not. Neither is true.
|
|
494
537
|
const killed = ceilingKillResult(name, r, ceilingMs);
|
|
495
538
|
if (killed) {
|
|
496
|
-
|
|
539
|
+
record(killed);
|
|
497
540
|
continue;
|
|
498
541
|
}
|
|
499
542
|
if (r.code === 0) {
|
|
500
|
-
|
|
543
|
+
record({ gate: name, pass: true, details: "exit 0" });
|
|
501
544
|
continue;
|
|
502
545
|
}
|
|
503
546
|
const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
|
|
@@ -508,7 +551,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
|
508
551
|
// exactly where it belongs: on failures the runner actually reported and the baseline already had.
|
|
509
552
|
const classification = classifyFailureOutput(raw);
|
|
510
553
|
if (classification === "infra") {
|
|
511
|
-
|
|
554
|
+
record({
|
|
512
555
|
gate: name,
|
|
513
556
|
pass: false,
|
|
514
557
|
details: `exit ${r.code} on infrastructure alone — the runner never completed a suite, so this gate verified nothing:\n${unrecognizedEvidence(raw) || raw.trim().split("\n").slice(0, 10).join("\n")}`,
|
|
@@ -534,7 +577,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
|
534
577
|
? `the baseline capture for this command was killed at its ceiling and recorded no verdict, so nothing here is forgivable — it now exits ${r.code} with no recognizable failure lines — failing closed`
|
|
535
578
|
: `command was green at baseline but now exits ${r.code} with no recognizable failure lines — failing closed`;
|
|
536
579
|
const evidence = unrecognizedEvidence(raw);
|
|
537
|
-
|
|
580
|
+
record({
|
|
538
581
|
gate: name,
|
|
539
582
|
pass: false,
|
|
540
583
|
details: evidence ? `${closed}\nunrecognized output:\n${evidence}` : closed,
|
|
@@ -545,10 +588,26 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
|
545
588
|
if (failing.length) {
|
|
546
589
|
const headlined = headlineDetails(raw, failing);
|
|
547
590
|
const meta = { ...headlined.meta, ...(classification ? { classification } : {}) };
|
|
548
|
-
|
|
591
|
+
record({ gate: name, pass: false, details: headlined.details, ...(Object.keys(meta).length ? { meta } : {}) });
|
|
592
|
+
continue;
|
|
593
|
+
}
|
|
594
|
+
// T7: everything below this line is forgiveness, and forgiveness is the one verdict that reads a
|
|
595
|
+
// record from another session. The failures are all baseline-recorded — but recorded under a
|
|
596
|
+
// capacity this command did not run under, so they are not evidence that these failures
|
|
597
|
+
// pre-existed the diff. A red does not become a green on fingerprints from a world it was not
|
|
598
|
+
// measured in; the operator gets both worlds named.
|
|
599
|
+
if (!comparable) {
|
|
600
|
+
record({
|
|
601
|
+
gate: name,
|
|
602
|
+
pass: false,
|
|
603
|
+
details: `exit ${r.code}; every failure is recorded in the baseline, but that capture ran under `
|
|
604
|
+
+ `${describeCapacity(entry?.capacity)} and this command ran under ${describeCapacity(r.capacity)} — `
|
|
605
|
+
+ `forgiveness across a changed capacity is not evidence, so this fails closed`,
|
|
606
|
+
meta: { capacityMismatch: true, ...(classification ? { classification } : {}) },
|
|
607
|
+
});
|
|
549
608
|
continue;
|
|
550
609
|
}
|
|
551
|
-
|
|
610
|
+
record({
|
|
552
611
|
gate: name,
|
|
553
612
|
pass: true,
|
|
554
613
|
details: `exit ${r.code} but only pre-existing failures (forgiven)${unreadable ? " — no failure shape recognized in this output, so a new failure from this runner is invisible to the baseline gate" : ""}`,
|
package/dist/gates/llm.d.ts
CHANGED
package/dist/gates/llm.js
CHANGED
|
@@ -226,6 +226,7 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
|
|
|
226
226
|
const snapshotQuietFor = now - quietSince;
|
|
227
227
|
if (cpuFlatFor >= harvestCpuFlatWindowMs(cpu.resolutionMs)
|
|
228
228
|
&& snapshotQuietFor >= gateInactivityWindowMs) {
|
|
229
|
+
via.onInactivity?.();
|
|
229
230
|
break;
|
|
230
231
|
}
|
|
231
232
|
}
|