tickmarkr 2.1.3 → 2.1.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -8,6 +8,9 @@ export type StatusOpts = {
8
8
  readWorkerOutput?: (taskId: string, attempt: number, runId: string) => Promise<string | undefined>;
9
9
  webhookUrl?: string;
10
10
  postWebhook?: DecisionWebhookPost;
11
+ /** How often the live board rewrites its own `watch` beat. Fixtures choose it so a death drill can
12
+ * watch the record STOP inside a test's patience; unset means SUPERVISION_BEAT_MS. */
13
+ supervisionBeatMs?: number;
11
14
  };
12
15
  export declare const GATE_KEYS: {
13
16
  readonly build: "B";
@@ -12,7 +12,7 @@ import { isPidLive } from "../../run/lock.js";
12
12
  import { normalizeGateOutcome } from "../../run/outcome.js";
13
13
  import { desiredPanes } from "../../run/reconcile.js";
14
14
  import { normalizeStallSnapshot } from "../../run/stall.js";
15
- import { readSupervision, supervisionText } from "../../run/supervision.js";
15
+ import { armSupervision, readSupervision, supervisionText } from "../../run/supervision.js";
16
16
  import { deriveRunCockpitData, } from "../../tui/cockpit/derive.js";
17
17
  import { COCKPIT_COLUMN_FLOOR } from "../../tui/cockpit/layout.js";
18
18
  import { cellWidth, fitCells, wrapCells } from "../../tui/cockpit/width.js";
@@ -1370,6 +1370,20 @@ export async function status(argv, cwd = process.cwd(), opts = {}) {
1370
1370
  }
1371
1371
  return fresh;
1372
1372
  };
1373
+ // SUP-06: the live board IS the `watch` tier, so it is the only process that can beat it — which is
1374
+ // why that tier could only ever print ABSENT about itself while every other seat had a beater. A word
1375
+ // one tier can only ever say in one of its five states trains every reader to skip it.
1376
+ //
1377
+ // Armed ONLY when UNBOUNDED. A bounded render is a READER, and the purity fence (D-02: a bounded
1378
+ // --watch leaves .tickmarkr/ byte-identical) is exactly the test that catches a reader beating on a
1379
+ // watcher's behalf, which would report every dead tier healthy. Armed BEFORE the first frame, so the
1380
+ // board's own first print already names its tier armed rather than one interval later.
1381
+ //
1382
+ // The handle is HELD and stood down in the finally. Discarding it is not a smaller version of this —
1383
+ // it INVERTS it: armSupervision's interval outlives the board in any host that outlives one board, so
1384
+ // a dead board keeps writing and reads ARMED. That is the over-claiming direction, the one an operator
1385
+ // acts on, and the one this instrument exists to close.
1386
+ const armed = bounded ? undefined : armSupervision(cwd, "watch", opts.supervisionBeatMs);
1373
1387
  let titleSaved = false;
1374
1388
  const restoreTitle = () => {
1375
1389
  if (!titleSaved)
@@ -1427,6 +1441,14 @@ export async function status(argv, cwd = process.cwd(), opts = {}) {
1427
1441
  }
1428
1442
  }
1429
1443
  finally {
1444
+ // Stops the interval as well as recording the hand-off: a board that returns must not leave a
1445
+ // timer beating for it. A board that is KILLED never reaches here, its beat ages out, and the
1446
+ // tier reads STALE — armed-then-lost, which is the truth about a killed board.
1447
+ // Boards OVERLAP — a second pane is one keystroke away — and the stand-down marker speaks for the
1448
+ // whole tier, so this hand-off is recorded only when no other board is still present (see
1449
+ // armSupervision's presence files). Otherwise the first pane closed would render the second
1450
+ // pane's own tier down while it is drawing frames.
1451
+ armed?.disarm();
1430
1452
  if (titleSaved) {
1431
1453
  process.removeListener("exit", restoreTitle);
1432
1454
  restoreTitle();
@@ -30,6 +30,22 @@ export declare function classifyScopeOffenders(taskId: string, hard: ReadonlyArr
30
30
  * Each line names the task id and at least one missing collateral test path.
31
31
  */
32
32
  export declare function collateralLints(tasks: ReadonlyArray<Pick<Task, "id" | "files">>, repoRoot: string): string[];
33
+ /**
34
+ * Does every mention of `target` have an explicit, target-local READ relation?
35
+ *
36
+ * A `context:` entry grants READ authority and nothing else, so both blocking paths have to ask this
37
+ * before honouring one. Fail-closed by construction: every occurrence must match one of the narrow
38
+ * read forms below. Absence from a write-verb list is never evidence. Thus `ownedTable returns two`
39
+ * and the unlisted `ownedTable sorts the rows it already publishes` are refused, while an unrelated
40
+ * write in `the consumer adds a cache while reading ownedTable` does not revoke ownedTable's read
41
+ * authority. A target mentioned twice, once for reading and once ambiguously, is refused.
42
+ *
43
+ * This is intentionally a lexical representation rather than prose understanding. Keyed on NAMES:
44
+ * only the literal target is classified, so a dependency described conceptually stays invisible and
45
+ * an unfamiliar read phrasing stays refused. That closes the false-positive direction only; widen
46
+ * the affirmative grammar only with a control that goes red first.
47
+ */
48
+ export declare function criterionReadsOnly(text: string, target: string): boolean;
33
49
  export declare function newDirectoryLints(tasks: ReadonlyArray<Pick<Task, "id" | "files">>, repoRoot: string): string[];
34
50
  /**
35
51
  * OBS-76 class: sweep src/ for out-of-scope source files that reference a symbol the acceptance
@@ -104,7 +120,7 @@ export declare function goalDensityErrors(tasks: ReadonlyArray<Pick<Task, "id" |
104
120
  * must be COMPLETE: any code path that cannot be read is its own error — the rule fails closed
105
121
  * rather than trust a partial scan.
106
122
  */
107
- export declare function symbolOwnershipErrors(tasks: ReadonlyArray<Pick<Task, "id" | "files" | "acceptance">>, repoRoot: string): string[];
123
+ export declare function symbolOwnershipErrors(tasks: ReadonlyArray<Pick<Task, "id" | "files" | "acceptance"> & Partial<Pick<Task, "context">>>, repoRoot: string): string[];
108
124
  /**
109
125
  * The participation half of the config — everything the review-policy rules read. `byShape` rides
110
126
  * along because `gates.byShape.<shape>.review: false` is a second, NON-monotone participation switch:
@@ -128,6 +144,11 @@ export declare function reviewParticipationErrors(tasks: ReadonlyArray<Pick<Task
128
144
  * symbol-ownership lint and the participation config, and defaults to the invocation directory —
129
145
  * correct for the CLI/daemon, which compile from inside the target repo.
130
146
  *
147
+ * The read declaration (`context:`) rides on this parameter, on the task objects themselves — the
148
+ * aggregator never infers it from the repo, the config, or a sibling task. It is OPTIONAL: a caller
149
+ * whose task objects carry no `context` gets byte-identical verdicts to the ones it got before the
150
+ * field existed, because an absent declaration grants no authority at all.
151
+ *
131
152
  * KNOWN GAP (R2 review, T6/T7 scope): the compile seam in src/compile/index.ts
132
153
  * (`enforceTaskUnitContract`) does not thread `compileSource`'s `root` argument into this call, so a
133
154
  * PROGRAMMATIC compile whose target repo differs from process.cwd() resolves the wrong root and can
@@ -138,4 +159,4 @@ export declare function reviewParticipationErrors(tasks: ReadonlyArray<Pick<Task
138
159
  * critical path itself (src/gates/review.ts), so a lint this seam misses costs a late verdict rather
139
160
  * than an unreviewed one. The compile lint is the early warning; the gate is the fail-closed backstop.
140
161
  */
141
- export declare function taskUnitContractErrors(tasks: ReadonlyArray<Pick<Task, "id" | "goal" | "files" | "deps" | "acceptance" | "shape">>, repoRoot?: string, review?: ReviewParticipation): string[];
162
+ export declare function taskUnitContractErrors(tasks: ReadonlyArray<Pick<Task, "id" | "goal" | "files" | "deps" | "acceptance" | "shape"> & Partial<Pick<Task, "context">>>, repoRoot?: string, review?: ReviewParticipation): string[];
@@ -171,15 +171,95 @@ export function collateralLints(tasks, repoRoot) {
171
171
  // v1.53 T4 (OBS-76): needles are code-shaped tokens only (camelCase / snake_case) — plain prose
172
172
  // words never match, so prose-only criteria yield zero needles instead of alarm-fatigue noise.
173
173
  // ponytail: token heuristic, not AST symbol resolution — promote after a version of precision data.
174
- function criteriaSymbols(acceptance) {
175
- const out = new Set();
174
+ function criteriaSymbolTexts(acceptance) {
175
+ const out = new Map();
176
176
  for (const item of acceptance) {
177
- for (const tok of renderAcceptanceItem(item).match(/\b[A-Za-z_][A-Za-z0-9_]*\b/g) ?? []) {
178
- if (tok.includes("_") || /[a-z][A-Z]/.test(tok))
179
- out.add(tok);
177
+ const text = renderAcceptanceItem(item);
178
+ for (const tok of text.match(/\b[A-Za-z_][A-Za-z0-9_]*\b/g) ?? []) {
179
+ if (!tok.includes("_") && !/[a-z][A-Z]/.test(tok))
180
+ continue;
181
+ const texts = out.get(tok) ?? [];
182
+ if (!texts.includes(text))
183
+ texts.push(text);
184
+ out.set(tok, texts);
180
185
  }
181
186
  }
182
- return [...out].sort();
187
+ return out;
188
+ }
189
+ function criteriaSymbols(acceptance) {
190
+ return [...criteriaSymbolTexts(acceptance).keys()].sort();
191
+ }
192
+ // These are affirmative READ relations, not a list of things that are not writes. The distinction
193
+ // is the fail-closed boundary: an unknown predicate never earns context[] authority. A target is
194
+ // readable when it is the object of an explicit observation (`reading ownedTable`) or the subject
195
+ // of an explicitly pre-existing state (`ownedTable already publishes`). An unqualified declarative
196
+ // predicate (`ownedTable returns two rows`) is deliberately absent because it can demand a change.
197
+ const DIRECT_READ_BEFORE = /\b(?:consult|consults|consulting|inspect|inspects|inspecting|read|reads|reading|reference|references|referencing)\s+(?:(?:the|an?)\s+)?(?:(?:current|existing|unchanged)\s+)?$/i;
198
+ const PREEXISTING_STATE_AFTER = /^\s*(?:(?:and|or)\s+\S+\s+)*(?:(?:,\s*)?which\s+)?(?:already|currently)\s+(?:carries|carry|contains?|declares?|defines?|exposes?|holds?|owns?|provides?|publish|publishes|returns?|supplies|supply|yields?)\b/i;
199
+ const PASSIVE_READ_AFTER = /^\s+(?:is|remains)\s+(?:read|referenced|unchanged|untouched|as[- ]is)\b/i;
200
+ // Once a target-local read has been established, a same-clause continuation can revoke it. These
201
+ // vetoes are structural rather than an attempted exhaustive list of write verbs: an action followed
202
+ // by `it` / `them` / `the same ...` is target-directed but ambiguous, and a coordinated passive
203
+ // participle inherits the target as its subject. Both therefore fail closed. The scan stops at an
204
+ // actual sentence/clause boundary; a dot inside a later path is not one.
205
+ const TARGET_ANAPHOR_AFTER_COORDINATOR = /\b(?:and|or|then|before|after|while)\b[^,;.!?\n]{0,80}\b(?:it|them|those|the\s+same(?:\s+[A-Za-z][A-Za-z0-9_-]*)?)\b/i;
206
+ const TARGET_PASSIVE_AFTER_SUBJECT_READ = /\b(?:and|or|then|before|after)\s+(?:then\s+)?(?:(?:is|gets?|becomes?|being)\s+)?[A-Za-z]+(?:ed|en)\b/i;
207
+ function sameClauseAfterTarget(after) {
208
+ const boundary = after.search(/[;!?\n]|\.(?=\s|$)/);
209
+ return boundary === -1 ? after : after.slice(0, boundary);
210
+ }
211
+ function targetOccurrences(text, target) {
212
+ if (!target)
213
+ return [];
214
+ const out = [];
215
+ const wordAtStart = /[A-Za-z0-9_]/.test(target[0]);
216
+ const wordAtEnd = /[A-Za-z0-9_]/.test(target[target.length - 1]);
217
+ for (let at = text.indexOf(target); at !== -1; at = text.indexOf(target, at + target.length)) {
218
+ const before = text[at - 1];
219
+ const after = text[at + target.length];
220
+ if (wordAtStart && before !== undefined && /[A-Za-z0-9_]/.test(before))
221
+ continue;
222
+ if (wordAtEnd && after !== undefined && /[A-Za-z0-9_]/.test(after))
223
+ continue;
224
+ out.push(at);
225
+ }
226
+ return out;
227
+ }
228
+ /**
229
+ * Does every mention of `target` have an explicit, target-local READ relation?
230
+ *
231
+ * A `context:` entry grants READ authority and nothing else, so both blocking paths have to ask this
232
+ * before honouring one. Fail-closed by construction: every occurrence must match one of the narrow
233
+ * read forms below. Absence from a write-verb list is never evidence. Thus `ownedTable returns two`
234
+ * and the unlisted `ownedTable sorts the rows it already publishes` are refused, while an unrelated
235
+ * write in `the consumer adds a cache while reading ownedTable` does not revoke ownedTable's read
236
+ * authority. A target mentioned twice, once for reading and once ambiguously, is refused.
237
+ *
238
+ * This is intentionally a lexical representation rather than prose understanding. Keyed on NAMES:
239
+ * only the literal target is classified, so a dependency described conceptually stays invisible and
240
+ * an unfamiliar read phrasing stays refused. That closes the false-positive direction only; widen
241
+ * the affirmative grammar only with a control that goes red first.
242
+ */
243
+ export function criterionReadsOnly(text, target) {
244
+ const occurrences = targetOccurrences(text, target);
245
+ return occurrences.length > 0 && occurrences.every((at) => {
246
+ // Backticks quote the target, not the relation, so discard only the adjacent delimiters.
247
+ const before = text.slice(Math.max(0, at - 96), at).replace(/`$/, "");
248
+ const after = text.slice(at + target.length).replace(/^`/, "");
249
+ const preexisting = PREEXISTING_STATE_AFTER.exec(after);
250
+ const passive = PASSIVE_READ_AFTER.exec(after);
251
+ if (!DIRECT_READ_BEFORE.test(before) && !preexisting && !passive)
252
+ return false;
253
+ const clause = sameClauseAfterTarget(after);
254
+ if (TARGET_ANAPHOR_AFTER_COORDINATOR.test(clause))
255
+ return false;
256
+ // A subject-position read (`target is read` / `target already publishes`) leaves the target as
257
+ // the inherited subject of a coordinated passive: `... and then rewritten by the consumer` is
258
+ // a change demand, not a read. Object-position reads need the anaphor veto above instead.
259
+ const subjectRead = preexisting ?? passive;
260
+ return !subjectRead
261
+ || !TARGET_PASSIVE_AFTER_SUBJECT_READ.test(clause.slice(subjectRead[0].length));
262
+ });
183
263
  }
184
264
  const ARCH_PAGES = ["docs/codebase/ARCHITECTURE.md", "docs/codebase/STRUCTURE.md"];
185
265
  function topLevelSrcDir(file) {
@@ -519,8 +599,8 @@ function walkAllCode(repoRoot) {
519
599
  */
520
600
  export function symbolOwnershipErrors(tasks, repoRoot) {
521
601
  const perTask = tasks
522
- .map((t) => ({ t, symbols: criteriaSymbols(t.acceptance ?? []) }))
523
- .filter((x) => x.symbols.length);
602
+ .map((t) => ({ t, bySymbol: criteriaSymbolTexts(t.acceptance ?? []) }))
603
+ .filter((x) => x.bySymbol.size);
524
604
  if (!perTask.length)
525
605
  return [];
526
606
  const { files: codeFiles, unreadable } = walkAllCode(repoRoot);
@@ -531,7 +611,7 @@ export function symbolOwnershipErrors(tasks, repoRoot) {
531
611
  for (const u of unreadable)
532
612
  errors.push(incomplete(u));
533
613
  // resolve each symbol to its definition site(s) once, shared across tasks
534
- const allSymbols = [...new Set(perTask.flatMap((x) => x.symbols))];
614
+ const allSymbols = [...new Set(perTask.flatMap((x) => [...x.bySymbol.keys()]))];
535
615
  const res = new Map(allSymbols.map((s) => [s, definitionRe(s)]));
536
616
  const sites = new Map(allSymbols.map((s) => [s, []]));
537
617
  for (const cf of codeFiles) {
@@ -548,16 +628,46 @@ export function symbolOwnershipErrors(tasks, repoRoot) {
548
628
  sites.get(sym).push(cf);
549
629
  }
550
630
  }
551
- for (const { t, symbols } of perTask) {
631
+ for (const { t, bySymbol } of perTask) {
552
632
  // OBS-22: scopeGate accepts picomatch globs; ownership must agree.
553
633
  const scoped = filesGlob(t.files.map((f) => f.replace(/^\.\//, "")));
554
- for (const sym of symbols) {
634
+ // THE RULE AT THIS SITE: a `context:`
635
+ // entry declares a path the worker may READ, so it answers "can the worker satisfy a criterion
636
+ // that only reads this symbol" (yes) and never "may the worker change it" (no — that authority
637
+ // comes from files[] alone). Hence the exemption is conditional on the criteria that actually
638
+ // name the symbol, and it is asked of the SYMBOL, not of the sentence: `criterionReadsOnly` fails
639
+ // closed, so a criterion demanding the symbol CHANGE — and any phrasing that does not prove it
640
+ // reads the symbol — is still refused, and still told to widen files[] where the change is what
641
+ // it asks for. An unrelated in-scope write does not change that per-symbol answer. Keyed on NAMES:
642
+ // only a path literally written in context[] is seen here, so a
643
+ // read dependency an author described in prose remains invisible to this lint. That closes the
644
+ // false-positive direction only — nothing here narrows what the rule refuses.
645
+ const declaredContext = (t.context ?? []).map((f) => f.replace(/^\.\//, ""));
646
+ const readable = declaredContext.length ? filesGlob(declaredContext) : () => false;
647
+ for (const [sym, texts] of bySymbol) {
555
648
  const defs = sites.get(sym) ?? [];
556
649
  if (defs.length !== 1)
557
650
  continue; // unknown or ambiguous — silent by ruling
558
651
  const site = defs[0];
559
652
  if (scoped(site))
560
653
  continue; // defined inside the task's own write surface
654
+ const writes = !texts.every((text) => criterionReadsOnly(text, sym));
655
+ if (readable(site) && !writes)
656
+ continue; // read authority is declared and read authority is all it needs
657
+ if (readable(site)) {
658
+ errors.push(`${t.id}: criterion identifier "${sym}" is defined only in ${site}, which is declared in `
659
+ + `context[] but not in files[] — a context: entry grants READ authority only, and this `
660
+ + `criterion requires changing "${sym}"; add ${site} to files[] or reword the criterion to `
661
+ + `require only reading it (OBS-248).`);
662
+ continue;
663
+ }
664
+ if (!writes) {
665
+ errors.push(`${t.id}: criterion identifier "${sym}" is defined only in ${site}, which is in neither `
666
+ + `files[] nor context[] — the criterion only reads "${sym}", so declare ${site} in `
667
+ + `context[]; widening files[] would grant write authority this criterion never asks for `
668
+ + `(OBS-248).`);
669
+ continue;
670
+ }
561
671
  errors.push(`${t.id}: criterion identifier "${sym}" is defined only in ${site}, which is not in files[] — `
562
672
  + `add ${site} to files[] or reword the criterion; a worker scoped to files[] cannot satisfy a `
563
673
  + `criterion whose enabling symbol lives outside it (OBS-248).`);
@@ -630,6 +740,11 @@ function activeReviewParticipation(repoRoot) {
630
740
  * symbol-ownership lint and the participation config, and defaults to the invocation directory —
631
741
  * correct for the CLI/daemon, which compile from inside the target repo.
632
742
  *
743
+ * The read declaration (`context:`) rides on this parameter, on the task objects themselves — the
744
+ * aggregator never infers it from the repo, the config, or a sibling task. It is OPTIONAL: a caller
745
+ * whose task objects carry no `context` gets byte-identical verdicts to the ones it got before the
746
+ * field existed, because an absent declaration grants no authority at all.
747
+ *
633
748
  * KNOWN GAP (R2 review, T6/T7 scope): the compile seam in src/compile/index.ts
634
749
  * (`enforceTaskUnitContract`) does not thread `compileSource`'s `root` argument into this call, so a
635
750
  * PROGRAMMATIC compile whose target repo differs from process.cwd() resolves the wrong root and can
@@ -106,6 +106,22 @@ function extractWriteDirectives(body, storedPath) {
106
106
  return writes;
107
107
  }
108
108
  const isPlainObject = (v) => typeof v === "object" && v !== null && !Array.isArray(v);
109
+ function summaryCompletionStatus(planFile) {
110
+ const summaryFile = join(dirname(planFile), `${basename(planFile).replace(/-PLAN\.md$/, "")}-SUMMARY.md`);
111
+ if (!existsSync(summaryFile))
112
+ return undefined;
113
+ try {
114
+ const content = readFileSync(summaryFile, "utf8");
115
+ const fmMatch = /^---\r?\n([\s\S]*?)\r?\n---/.exec(content);
116
+ if (!fmMatch)
117
+ return undefined;
118
+ const fm = parseYaml(fmMatch[1]);
119
+ return isPlainObject(fm) ? fm.status : undefined;
120
+ }
121
+ catch {
122
+ return undefined;
123
+ }
124
+ }
109
125
  // schema-legal, branch-safe id segment: dots etc. → dashes; no "--" runs (task-branch separator)
110
126
  // or trailing dash — the id schema rejects both
111
127
  const sanitize = (s) => s
@@ -170,7 +186,11 @@ function compileOne(file, storedPath) {
170
186
  .filter((p) => !p.includes("$") && !p.startsWith("~") && !p.startsWith("/") && !p.split("/").includes(".."));
171
187
  const taskCount = [...body.matchAll(/<task[\s>]/g)].length;
172
188
  const humanGate = fm.autonomous === false || /<task type="checkpoint:/.test(body);
173
- const done = existsSync(join(dirname(file), `${basename(file).replace(/-PLAN\.md$/, "")}-SUMMARY.md`));
189
+ const summaryStatus = summaryCompletionStatus(file);
190
+ // GSD source rule (`/gsd:quick status`): a SUMMARY is complete only when its frontmatter says
191
+ // `status: complete`; when `status` is absent it reports `INCOMPLETE`. The compiler must preserve
192
+ // that distinction because an artifact can record a reviewed failure rather than completion.
193
+ const done = summaryStatus === "complete";
174
194
  const files = strings(fm.files_modified).map((f) => f.replace(/^\.\//, ""));
175
195
  assertWriteScope(file, `P${key}`, files, extractWriteDirectives(body, storedPath));
176
196
  // fail closed (D-03): a silently dropped floor/pin routes the task cheap instead of erroring
@@ -4,6 +4,7 @@ import { basename, dirname, join, resolve } from "node:path";
4
4
  import { filesGlob } from "../graph/files-glob.js";
5
5
  import picomatch from "picomatch";
6
6
  import { GATE_NAMES, GRAPH_ROUTING_MODES, ORACLES, renderAcceptanceItem, SHAPES, TIERS, validateGraph, } from "../graph/schema.js";
7
+ import { criterionReadsOnly } from "./collateral.js";
7
8
  import { CompileError, inferShape, sha256 } from "./common.js";
8
9
  // OBS-170/OBS-184: `context:` is a promise to the worker, and nothing ever checked it could be kept.
9
10
  // Workers run in `git worktree add <baseRef>` (run/git.ts:169-176), which materialises the base
@@ -169,20 +170,51 @@ function renderedObservables(text) {
169
170
  function criterionScopeFinding(task, text, criterion, id, tests) {
170
171
  if (task.files.length === 0)
171
172
  return undefined; // empty files[] is deliberately unrestricted
172
- const inScope = filesGlob(task.files.map((entry) => entry.replace(/^\.\//, ""))); // Q120s shared matcher
173
+ const strip = (entry) => entry.replace(/^\.\//, ""); // Q120s shared matcher below
174
+ const inFiles = filesGlob(task.files.map(strip));
175
+ const context = (task.context ?? []).map(strip);
176
+ const inContext = context.length ? filesGlob(context) : () => false;
177
+ // THE RULE AT THIS SITE: `files:` clears a named path outright, because it is write authority and
178
+ // write authority covers reading too. `context:` clears one ONLY for a criterion that reads it and
179
+ // does not change it — a read declaration is not a second write surface, and unioning it in unconditionally
180
+ // would promote read authority to write authority for exactly the criterion that asks for the
181
+ // change. The question is asked per PATH — each occurrence of each declared path must carry its own
182
+ // affirmative read relation. A write elsewhere in the criterion therefore cannot revoke a named
183
+ // dependency's read exemption, and a read elsewhere cannot confer one on a path being changed. It
184
+ // is the same target-specific question the symbol-ownership rule asks (collateral.ts), and it fails
185
+ // closed: any phrasing that does not prove the criterion reads the path is refused, as it was before
186
+ // this check consulted the declaration at all. Keyed on NAMES: only a path literally written in files[]
187
+ // or context[] is matched, so a dependency the author described conceptually and never wrote out
188
+ // stays invisible to this check. That closes the false-positive direction only — a path in neither
189
+ // declaration is refused exactly as before.
190
+ const declared = (path) => inFiles(path) || (inContext(path) && criterionReadsOnly(text, path));
173
191
  const named = namedCriterionPaths(text, tests);
174
192
  const { exact, denominators } = renderedObservables(text);
175
193
  const asserting = tests.filter((test) => exact.some((token) => test.text.includes(token))
176
194
  || denominators.some((denominator) => new RegExp(`(?:^|\\D)\\d+/${denominator}(?:\\D|$)`).test(test.text))).map((test) => test.path);
177
- const missing = [...new Set([...named, ...asserting])].filter((path) => !inScope(path));
195
+ const missing = [...new Set([...named, ...asserting])].filter((path) => !declared(path));
178
196
  if (missing.length === 0)
179
197
  return undefined;
198
+ // The remedy names the authority the criterion actually needs. A criterion that only READS the
199
+ // producer is repaired by context[]; instructing its author to widen files[] is the product
200
+ // emitting the unsafe workaround itself. Only a criterion that requires CHANGING the producer is
201
+ // told to widen the write surface.
202
+ const readMissing = missing.filter((path) => criterionReadsOnly(text, path));
203
+ const writeMissing = missing.filter((path) => !criterionReadsOnly(text, path));
204
+ const remedy = [
205
+ readMissing.length === 0 ? "" : `the criterion only reads ${readMissing.length === 1 ? "it" : "them"} `
206
+ + `(${readMissing.join(", ")}), so add ${readMissing.length === 1 ? "it" : "them"} to context[] — `
207
+ + `files[] would grant write authority this criterion never asks for`,
208
+ writeMissing.length === 0 ? "" : `the criterion requires changing ${writeMissing.length === 1 ? "it" : "them"} `
209
+ + `(or does not establish a read-only relation) (${writeMissing.join(", ")}), so add `
210
+ + `${writeMissing.length === 1 ? "it" : "them"} to files[]`,
211
+ ].filter(Boolean).join("; ");
180
212
  return {
181
213
  code: "criterion-scope",
182
214
  fixtureId: id,
183
215
  taskId: task.id,
184
216
  criterion,
185
- detail: `criterion names asserting file${missing.length === 1 ? "" : "s"} outside expanded files[]: ${missing.join(", ")}`,
217
+ detail: `criterion names asserting file${missing.length === 1 ? "" : "s"} outside expanded files[] and context[]: ${missing.join(", ")} — ${remedy}`,
186
218
  };
187
219
  }
188
220
  function exportedIdentifier(root, files, identifier) {
@@ -1,7 +1,20 @@
1
1
  import type { TickmarkrConfig } from "../config/config.js";
2
2
  import type { AcceptanceItem } from "../graph/schema.js";
3
- import { type ShResult } from "../run/git.js";
3
+ import { type RunCapacity, type ShResult } from "../run/git.js";
4
4
  import type { GateResult } from "./types.js";
5
+ /**
6
+ * T7: the capacity a gate's own command ran under rides the RESULT, beside the verdict it explains,
7
+ * rather than inside `meta` — `meta` is a machine-readable extras bag several callers compare
8
+ * wholesale, and identity that a later session keys reuse on does not belong in a bag. Declared here,
9
+ * at the one producer, because only a battery gate has a command whose child received a fork cap:
10
+ * every other gate leaves the field absent, which is the honest reading of "this gate divided
11
+ * nothing". The daemon lifts it verbatim onto the journal's gate row (src/run/daemon.ts).
12
+ */
13
+ declare module "./types.js" {
14
+ interface GateResult {
15
+ capacity?: RunCapacity;
16
+ }
17
+ }
5
18
  export interface BaselineCommand {
6
19
  /**
7
20
  * Absent when the capture returned no verdict — see `infra`. A pre-v1.90 baseline can also lack it
@@ -21,8 +34,15 @@ export interface BaselineCommand {
21
34
  /** The ceiling that measurement implies, persisted so every later battery uses the same number. */
22
35
  ceilingMs?: number;
23
36
  /**
24
- * OBS-534 (T2): the capture was SIGKILLed at its ceiling. It never finished asking the question, so
25
- * the entry carries a CAUSE and no verdict: no exit code, no fingerprints, nothing forgivable.
37
+ * T7: the capacity this command's capture child ran under — the fork cap it received and the cores
38
+ * that cap was divided from. Absent in every pre-T7 baseline, which is exactly what makes those
39
+ * entries keep their current forgiveness; a MALFORMED one fails closed instead (git.ts readCapacity).
40
+ */
41
+ capacity?: RunCapacity;
42
+ /**
43
+ * The capture did not return a trustworthy verdict: it was SIGKILLed at its ceiling, or its output
44
+ * proves the machine was exhausted while it ran. The entry therefore carries a CAUSE and no
45
+ * verdict: no exit code, no fingerprints, nothing forgivable.
26
46
  */
27
47
  infra?: true;
28
48
  }
@@ -1,6 +1,6 @@
1
1
  import { existsSync, readFileSync } from "node:fs";
2
2
  import { join } from "node:path";
3
- import { DEFAULT_SHELL_TIMEOUT_MS, sh } from "../run/git.js";
3
+ import { DEFAULT_SHELL_TIMEOUT_MS, describeCapacity, sameCapacity, sh } from "../run/git.js";
4
4
  // incident #2 (run-20260709-104447): a vitest ✓ PASS line with "error" in the test NAME, wrapped in ANSI
5
5
  // codes that varied between baseline and worktree runs, was reported as a "new failure". Strip ANSI first;
6
6
  // a pass-marker line is never a failure. [\d;#] covers raw ANSI and digit-normalized ANSI ("\x1b[#m") from
@@ -118,6 +118,10 @@ const VOCAB_RE = /\b(?:error|fail(?:ed|ure|ing)?)\b/i;
118
118
  // this runner-output classifier rather than applied to any judge-authored reason text. A real test
119
119
  // failure still dominates below because one regression-shaped line makes the whole output regression.
120
120
  const INFRA_RE = /\bE(?:AGAIN|MFILE|NFILE|NOMEM|NOSPC)\b|JavaScript heap out of memory|Cannot allocate memory|Resource temporarily unavailable|Token not found in system keyring|Process from config\.webServer was not able to start/i;
121
+ // Capture invalidation is deliberately narrower than the gate's infrastructure vocabulary above:
122
+ // keyring/config-webServer startup failures remain gate concerns, while this policy is specifically
123
+ // for evidence that the capture ran while the machine was resource-starved.
124
+ const CAPTURE_EXHAUSTION_RE = /\bE(?:AGAIN|MFILE|NFILE|NOMEM|NOSPC)\b|JavaScript heap out of memory|Cannot allocate memory|Resource temporarily unavailable/i;
121
125
  // A named error CLASS ("AssertionError", "TypeError", "MyDomainError") — never bare "Error", which
122
126
  // is what an errno report itself is headed with (`Error: spawn EAGAIN`). The prefix is required.
123
127
  const ERROR_CLASS_RE = /\b[A-Za-z][A-Za-z0-9]*Error\b/;
@@ -133,6 +137,18 @@ export function classifyFailureOutput(output) {
133
137
  return "regression";
134
138
  return lines.some(isInfraLine) ? "infra" : undefined;
135
139
  }
140
+ /**
141
+ * Capture validity asks a different question from gate classification. At a gate, one genuine
142
+ * regression line must outrank adjacent errno evidence so a real defect is never laundered as infra.
143
+ * At capture, any such evidence invalidates the whole measurement: once the machine was exhausted,
144
+ * no failure in that incomplete environment can safely become pre-existing forgiveness. Keep this
145
+ * separate from `classifyFailureOutput` so changing capture policy cannot move gate verdicts.
146
+ */
147
+ const captureHasInvalidatingInfra = (output) => output
148
+ .split("\n")
149
+ .map((l) => l.replace(ANSI_RE, ""))
150
+ .filter((l) => !PASS_LINE_RE.test(l))
151
+ .some((l) => CAPTURE_EXHAUSTION_RE.test(l));
136
152
  const normalizeLine = (l) => l.replace(/\d+/g, "#").replace(/\s+/g, " ").trim();
137
153
  // Vitest's default reporter names file durations as
138
154
  // `✓ |project| tests/example.test.ts (12 tests) 1.23s` (❯ for a red file). The runner may be
@@ -370,6 +386,15 @@ export function ceilingKillResult(gate, r, ceilingMs) {
370
386
  * shell asks it to finish faster than the thing it is measuring.
371
387
  */
372
388
  export const CAPTURE_CEILING_MS = 1_800_000;
389
+ const invalidCaptureEntry = (durationMs) => ({
390
+ infra: true,
391
+ fingerprints: [],
392
+ durationMs,
393
+ fileDurationSumMs: null,
394
+ impliedParallelism: null,
395
+ longestFile: null,
396
+ ceilingMs: effectiveCeilingMs({ durationMs }),
397
+ });
373
398
  export async function captureBaseline(cwd, commands) {
374
399
  const base = { commands: {} };
375
400
  for (const [name, cmd] of Object.entries(commands)) {
@@ -396,18 +421,22 @@ export async function captureBaseline(cwd, commands) {
396
421
  console.error(`tickmarkr: baseline capture for "${name}" was killed at its ${CAPTURE_CEILING_MS}ms ceiling — `
397
422
  + `it recorded NO fingerprints, so nothing is forgiven and every gate will treat a pre-existing `
398
423
  + `failure as a fresh one. Raise the ceiling or shorten the command.`);
399
- base.commands[name] = {
400
- infra: true,
401
- fingerprints: [],
402
- durationMs,
403
- fileDurationSumMs: null,
404
- impliedParallelism: null,
405
- longestFile: null,
406
- ceilingMs: effectiveCeilingMs({ durationMs }),
407
- };
424
+ base.commands[name] = invalidCaptureEntry(durationMs);
408
425
  continue;
409
426
  }
410
427
  const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
428
+ // Run 2137: the child exited and printed ordinary FAIL/AssertionError lines, so the gate-side
429
+ // discriminator correctly called the mixed output a regression. But the same output also said
430
+ // `spawn EAGAIN`: the machine had run out of processes while the pristine-tree measurement was
431
+ // being taken. A capture cannot know which red lines predated that shortage and which it caused,
432
+ // so none may become a fingerprint every later task gets to forgive.
433
+ if (captureHasInvalidatingInfra(raw)) {
434
+ console.error(`tickmarkr: baseline capture for "${name}" completed with process/resource-exhaustion evidence — `
435
+ + `it recorded NO exit-code verdict and NO fingerprints, so nothing is forgiven for this command; `
436
+ + `the measurement cannot distinguish a pre-existing failure from one caused by exhaustion.`);
437
+ base.commands[name] = invalidCaptureEntry(durationMs);
438
+ continue;
439
+ }
411
440
  base.commands[name] = {
412
441
  exitCode: r.code,
413
442
  // a command that exits 0 has no failures to fingerprint — recording any would be a lie the
@@ -417,6 +446,9 @@ export async function captureBaseline(cwd, commands) {
417
446
  durationMs,
418
447
  ...fileTiming(raw, durationMs),
419
448
  ceilingMs: effectiveCeilingMs({ durationMs }),
449
+ // T7: the world this measurement was taken in, so a later reader can ask whether its own world
450
+ // is the same one. Recorded from THIS command's own shell result, never re-derived here.
451
+ ...(r.capacity ? { capacity: r.capacity } : {}),
420
452
  };
421
453
  }
422
454
  const names = Object.keys(commands);
@@ -487,17 +519,28 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
487
519
  const entry = baseline.commands[name];
488
520
  const ceilingMs = effectiveCeilingMs(entry);
489
521
  const r = await sh(cmd, cwd, ceilingMs);
522
+ // T7: every verdict below carries the capacity ITS OWN command ran under, taken off the shell
523
+ // result rather than re-derived after the fact. The skip row above ran no command and therefore
524
+ // states no capacity — a row that never divided the machine must not claim that it did.
525
+ const record = (g) => {
526
+ results.push(r.capacity ? { ...g, capacity: r.capacity } : g);
527
+ };
528
+ // …and whether the entry that would forgive this command was measured in the same world. A
529
+ // baseline captured under a different fork cap forgives nothing: its fingerprints describe a
530
+ // machine divided by a different number. Absent capacity (every pre-T7 baseline) still forgives
531
+ // exactly as it does today; a malformed one fails closed (git.ts sameCapacity).
532
+ const comparable = sameCapacity(entry?.capacity, r.capacity);
490
533
  // Q24: the kill is read BEFORE the exit code is interpreted at all. A SIGKILLed battery has
491
534
  // whatever partial output it had flushed — typically no failure shape — so every path below
492
535
  // would otherwise turn a timeout into a claim about the work: "no recognizable failure lines"
493
536
  // when the baseline was green, or a forgiven pre-existing red when it was not. Neither is true.
494
537
  const killed = ceilingKillResult(name, r, ceilingMs);
495
538
  if (killed) {
496
- results.push(killed);
539
+ record(killed);
497
540
  continue;
498
541
  }
499
542
  if (r.code === 0) {
500
- results.push({ gate: name, pass: true, details: "exit 0" });
543
+ record({ gate: name, pass: true, details: "exit 0" });
501
544
  continue;
502
545
  }
503
546
  const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
@@ -508,7 +551,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
508
551
  // exactly where it belongs: on failures the runner actually reported and the baseline already had.
509
552
  const classification = classifyFailureOutput(raw);
510
553
  if (classification === "infra") {
511
- results.push({
554
+ record({
512
555
  gate: name,
513
556
  pass: false,
514
557
  details: `exit ${r.code} on infrastructure alone — the runner never completed a suite, so this gate verified nothing:\n${unrecognizedEvidence(raw) || raw.trim().split("\n").slice(0, 10).join("\n")}`,
@@ -534,7 +577,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
534
577
  ? `the baseline capture for this command was killed at its ceiling and recorded no verdict, so nothing here is forgivable — it now exits ${r.code} with no recognizable failure lines — failing closed`
535
578
  : `command was green at baseline but now exits ${r.code} with no recognizable failure lines — failing closed`;
536
579
  const evidence = unrecognizedEvidence(raw);
537
- results.push({
580
+ record({
538
581
  gate: name,
539
582
  pass: false,
540
583
  details: evidence ? `${closed}\nunrecognized output:\n${evidence}` : closed,
@@ -545,10 +588,26 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
545
588
  if (failing.length) {
546
589
  const headlined = headlineDetails(raw, failing);
547
590
  const meta = { ...headlined.meta, ...(classification ? { classification } : {}) };
548
- results.push({ gate: name, pass: false, details: headlined.details, ...(Object.keys(meta).length ? { meta } : {}) });
591
+ record({ gate: name, pass: false, details: headlined.details, ...(Object.keys(meta).length ? { meta } : {}) });
592
+ continue;
593
+ }
594
+ // T7: everything below this line is forgiveness, and forgiveness is the one verdict that reads a
595
+ // record from another session. The failures are all baseline-recorded — but recorded under a
596
+ // capacity this command did not run under, so they are not evidence that these failures
597
+ // pre-existed the diff. A red does not become a green on fingerprints from a world it was not
598
+ // measured in; the operator gets both worlds named.
599
+ if (!comparable) {
600
+ record({
601
+ gate: name,
602
+ pass: false,
603
+ details: `exit ${r.code}; every failure is recorded in the baseline, but that capture ran under `
604
+ + `${describeCapacity(entry?.capacity)} and this command ran under ${describeCapacity(r.capacity)} — `
605
+ + `forgiveness across a changed capacity is not evidence, so this fails closed`,
606
+ meta: { capacityMismatch: true, ...(classification ? { classification } : {}) },
607
+ });
549
608
  continue;
550
609
  }
551
- results.push({
610
+ record({
552
611
  gate: name,
553
612
  pass: true,
554
613
  details: `exit ${r.code} but only pre-existing failures (forgiven)${unreadable ? " — no failure shape recognized in this output, so a new failure from this runner is invisible to the baseline gate" : ""}`,
@@ -25,6 +25,7 @@ export interface LlmVia {
25
25
  label?: string;
26
26
  keep?: boolean;
27
27
  onSlot?: (slot: Slot) => void;
28
+ onInactivity?: () => void;
28
29
  }
29
30
  export interface GateVia {
30
31
  driver: ExecutorDriver;
package/dist/gates/llm.js CHANGED
@@ -226,6 +226,7 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
226
226
  const snapshotQuietFor = now - quietSince;
227
227
  if (cpuFlatFor >= harvestCpuFlatWindowMs(cpu.resolutionMs)
228
228
  && snapshotQuietFor >= gateInactivityWindowMs) {
229
+ via.onInactivity?.();
229
230
  break;
230
231
  }
231
232
  }