tickmarkr 2.5.1 → 2.5.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/run/git.d.ts CHANGED
@@ -4,6 +4,8 @@ export { ROUTING_ENV_SEAMS };
4
4
  export declare const FORK_CAP_ENV = "VITEST_MAX_FORKS";
5
5
  export declare const SUITE_PARENT_ENV = "TICKMARKR_SUITE_PARENT";
6
6
  export declare const DEFAULT_FORK_CAP = "6";
7
+ /** Verification owns the serialized suite window; workers keep their concurrency budget. */
8
+ export declare const runWithVerificationBudget: <T>(capacity: RunCapacity, fn: () => Promise<T>) => Promise<T>;
7
9
  /**
8
10
  * OBS-618: how many PROCESSES one vitest fork can hold at its peak — the fork, a daemon it spawns,
9
11
  * and that daemon's own `git` child. The previous formula counted FORKS and assumed one process
@@ -38,7 +40,7 @@ export declare const resolvedForkCap: () => string;
38
40
  /**
39
41
  * T7: the CAPACITY a suite verdict was measured under — the fork cap the command's child actually
40
42
  * received, and the core count that cap was divided from. Two verdicts are comparable only when both
41
- * numbers match: a run resumed at a different concurrency divides the same machine by a different
43
+ * numbers match: a run resumed at a different capacity divides the machine by a different
42
44
  * number, so a green measured in that other world is not evidence about this one.
43
45
  *
44
46
  * This pair is the WHOLE comparable identity, and the load averages a gate row already carries beside
@@ -77,7 +79,7 @@ export declare function sameCapacity(recorded: unknown, current: RunCapacity | u
77
79
  export declare const describeCapacity: (value: unknown) => string;
78
80
  /**
79
81
  * The capacity a child spawned on THIS async context would receive: the same precedence `shell`
80
- * applies below — an operator export of the cap wins over the run's own derived value — beside the
82
+ * applies below — the verification budget captures any operator export at run start — beside the
81
83
  * cores it was divided from. A caller holding a command's own result reads the capacity off THAT
82
84
  * result (`ShResult.capacity`, stamped where the child's environment was built); this is for the
83
85
  * decisions taken BEFORE any child exists — a cache hit, a reuse predicate.
package/dist/run/git.js CHANGED
@@ -15,24 +15,11 @@ export { ROUTING_ENV_SEAMS };
15
15
  export const FORK_CAP_ENV = "VITEST_MAX_FORKS";
16
16
  export const SUITE_PARENT_ENV = "TICKMARKR_SUITE_PARENT";
17
17
  export const DEFAULT_FORK_CAP = "6";
18
- /**
19
- * The fork budget belongs to ONE run: how many gate suites can be in flight at once is exactly that
20
- * run's resolved concurrency, so the per-suite cap must divide the machine by THAT number and no
21
- * other. Two things rule out a module-level variable. A single Node process holds more than one
22
- * runDaemon call — the suites do it, and so does a supervisor driving two repositories — so a
23
- * captured-once global hands whichever run started first its cap to every other run, and only a
24
- * reset seam production never calls could hide it. And re-deriving the number at spawn time reads
25
- * whatever `process.argv` or the config overlay says NOW, not what the run resolved: `parseArgs`
26
- * settles `--concurrency 2 --concurrency 8` on the LAST occurrence, and an overlay is a mutable
27
- * file, so a re-derived cap can disagree with the concurrency the run is actually enforcing.
28
- *
29
- * AsyncLocalStorage is the stdlib answer to both. The store is entered once, around the run body,
30
- * from the single value `runDaemon` resolved; every shell that run spawns — baseline capture, gate
31
- * batteries, tip verify, worker environments — inherits it through the async context, and a
32
- * concurrent run's shells inherit their own. There is nothing to reset: leaving the run leaves
33
- * the store, so sequential runs cannot inherit each other either.
34
- */
18
+ /** Worker concurrency is resolved once per run and isolated from other async runs. */
35
19
  const forkBudget = new AsyncLocalStorage();
20
+ const verificationBudget = new AsyncLocalStorage();
21
+ /** Verification owns the serialized suite window; workers keep their concurrency budget. */
22
+ export const runWithVerificationBudget = (capacity, fn) => verificationBudget.run(capacity, fn);
36
23
  /**
37
24
  * OBS-618: how many PROCESSES one vitest fork can hold at its peak — the fork, a daemon it spawns,
38
25
  * and that daemon's own `git` child. The previous formula counted FORKS and assumed one process
@@ -102,12 +89,12 @@ export const describeCapacity = (value) => {
102
89
  };
103
90
  /**
104
91
  * The capacity a child spawned on THIS async context would receive: the same precedence `shell`
105
- * applies below — an operator export of the cap wins over the run's own derived value — beside the
92
+ * applies below — the verification budget captures any operator export at run start — beside the
106
93
  * cores it was divided from. A caller holding a command's own result reads the capacity off THAT
107
94
  * result (`ShResult.capacity`, stamped where the child's environment was built); this is for the
108
95
  * decisions taken BEFORE any child exists — a cache hit, a reuse predicate.
109
96
  */
110
- export const resolvedCapacity = () => ({
97
+ export const resolvedCapacity = () => verificationBudget.getStore() ?? ({
111
98
  forkCap: Number(FORK_CAP_ENV in process.env ? process.env[FORK_CAP_ENV] : resolvedForkCap()),
112
99
  cores: availableParallelism(),
113
100
  });
@@ -151,9 +138,10 @@ function shell(cmd, cwd, timeoutMs, login) {
151
138
  const env = { ...process.env };
152
139
  for (const k of ROUTING_ENV_SEAMS)
153
140
  delete env[k];
154
- // OBS-110: apply the run's own fork cap only when the operator has not already set one.
155
- if (!(FORK_CAP_ENV in env))
156
- env[FORK_CAP_ENV] = resolvedForkCap();
141
+ // A run freezes the operator override and cores at startup; admission can lower the round cap.
142
+ env[FORK_CAP_ENV] = verificationBudget.getStore()
143
+ ? String(resolvedCapacity().forkCap)
144
+ : env[FORK_CAP_ENV] ?? resolvedForkCap();
157
145
  // OBS-854: descendants can leave the checkout (nested fixture suites do), so cwd alone cannot
158
146
  // attribute them. Every daemon shell exports the daemon pid as their durable parentage marker.
159
147
  env[SUITE_PARENT_ENV] = String(process.pid);
@@ -163,7 +151,7 @@ function shell(cmd, cwd, timeoutMs, login) {
163
151
  // for; re-deriving the run's own budget after the command returned would stamp a cap no child ran
164
152
  // under. `Number` of an unparseable export is NaN, which every reader treats as malformed and
165
153
  // therefore fails closed — the honest direction when the cap in play cannot be stated.
166
- const capacity = { forkCap: Number(env[FORK_CAP_ENV]), cores: availableParallelism() };
154
+ const capacity = { forkCap: Number(env[FORK_CAP_ENV]), cores: resolvedCapacity().cores };
167
155
  const attempt = () => new Promise((resolve) => {
168
156
  const startedAt = Date.now();
169
157
  // detached: bash gets its own process group so a timeout can kill the whole tree —
@@ -12,7 +12,7 @@ export interface TipVerifyResult {
12
12
  forgiven?: boolean;
13
13
  /**
14
14
  * OBS-534: what a nonzero exit is evidence OF, taken from the battery's own readers — `ceilingKillResult`
15
- * for a kill, `classifyFailureOutput` for everything else. `infra` means nothing was verified.
15
+ * for a kill, the shared runner classifier for everything else. `infra` means nothing was verified.
16
16
  */
17
17
  cause?: FailureClassification;
18
18
  }
package/dist/run/merge.js CHANGED
@@ -1,7 +1,7 @@
1
- import { existsSync, writeFileSync } from "node:fs";
1
+ import { existsSync, readFileSync, writeFileSync } from "node:fs";
2
2
  import { join } from "node:path";
3
3
  import { shq } from "../adapters/types.js";
4
- import { ceilingKillResult, classifyFailureOutput, effectiveCeilingMs, fingerprint, freshFailures, } from "../gates/baseline.js";
4
+ import { ceilingKillResult, classifyFreshRunnerOutput, classifyRunnerOutput, fileCountDeficit, waitForCalmWindow, effectiveCeilingMs, fingerprint, freshFailures, } from "../gates/baseline.js";
5
5
  import { tickmarkrDir } from "../graph/graph.js";
6
6
  import { describeCapacity, gitHead, linkNodeModules, resolveIntegrationBranch, sameCapacity, sh, shGit, shGitOk, WORKTREES_DIR } from "./git.js";
7
7
  export function integrationBranch(cfg, runId) {
@@ -52,16 +52,121 @@ export async function mergeTask(intWt, taskBranch, message, gatedCommit) {
52
52
  // fingerprint, or output with no recognizable failure shape, → failed.
53
53
  export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
54
54
  const results = [];
55
+ let runStartCommands;
56
+ let journalFound = false;
57
+ let hasRunEvidenceOrMalformed = false;
58
+ const journalPath = join(runDir, "journal.jsonl");
59
+ if (existsSync(journalPath)) {
60
+ journalFound = true;
61
+ try {
62
+ const raw = readFileSync(journalPath, "utf8").trim();
63
+ if (!raw) {
64
+ hasRunEvidenceOrMalformed = true;
65
+ }
66
+ else {
67
+ const lines = raw.split("\n");
68
+ for (const line of lines) {
69
+ if (!line.trim())
70
+ continue;
71
+ try {
72
+ const parsed = JSON.parse(line);
73
+ if (parsed.event === "run-start" && parsed.data?.commands) {
74
+ runStartCommands = parsed.data.commands;
75
+ break;
76
+ }
77
+ if (parsed.event && !parsed.event.startsWith("tip-verify")) {
78
+ hasRunEvidenceOrMalformed = true;
79
+ }
80
+ }
81
+ catch {
82
+ // Recover valid rows: skip malformed rows while searching for run-start
83
+ hasRunEvidenceOrMalformed = true;
84
+ }
85
+ }
86
+ }
87
+ }
88
+ catch {
89
+ hasRunEvidenceOrMalformed = true;
90
+ }
91
+ }
92
+ const gatesToRun = [];
55
93
  for (const [gate, cmd] of Object.entries(commands)) {
56
- const entry = baseline?.commands[gate];
94
+ if (gate === "tipTest")
95
+ continue;
96
+ if (gate === "test") {
97
+ gatesToRun.push(["test", commands.tipTest ?? cmd]);
98
+ }
99
+ else {
100
+ gatesToRun.push([gate, cmd]);
101
+ }
102
+ }
103
+ if (!commands.test && commands.tipTest) {
104
+ gatesToRun.push(["test", commands.tipTest]);
105
+ }
106
+ if (journalFound && hasRunEvidenceOrMalformed && runStartCommands === undefined) {
107
+ for (const [gate, cmd] of gatesToRun) {
108
+ const artifact = join(runDir, `tip-verify-${gate}.log`);
109
+ const details = `tip verify could not establish command provenance: run-start evidence missing or malformed in journal`;
110
+ writeFileSync(artifact, details + "\n");
111
+ results.push({
112
+ gate,
113
+ cmd,
114
+ pass: false,
115
+ exitCode: 1,
116
+ fingerprints: [],
117
+ details,
118
+ artifact,
119
+ });
120
+ }
121
+ return results;
122
+ }
123
+ for (const [gate, cmd] of gatesToRun) {
124
+ if (runStartCommands !== undefined) {
125
+ const expectedCmd = gate === "test"
126
+ ? (runStartCommands.tipTest ?? runStartCommands.test)
127
+ : runStartCommands[gate];
128
+ if (expectedCmd === undefined || cmd !== expectedCmd) {
129
+ const artifact = join(runDir, `tip-verify-${gate}.log`);
130
+ const details = `tip verify command "${cmd}" for gate "${gate}" was not named in run-start row`;
131
+ writeFileSync(artifact, details + "\n");
132
+ results.push({
133
+ gate,
134
+ cmd,
135
+ pass: false,
136
+ exitCode: 1,
137
+ fingerprints: [],
138
+ details,
139
+ artifact,
140
+ });
141
+ continue;
142
+ }
143
+ }
144
+ // Leg-2 T3 M1 (RULING-230-15): the entry is chosen by the RECORDED identity — the same provenance the
145
+ // command was validated against above — never the current config. After a resume whose config task
146
+ // command converged on the recorded tip command, the current-config test read "no distinct tip" and
147
+ // forgave a tip red against the TASK entry. Only standalone verify (no run-start row) has no record.
148
+ const identity = runStartCommands ?? commands;
149
+ const hasTipCommand = Boolean(identity.tipTest && identity.tipTest !== identity.test);
150
+ const entry = gate === "test" && (hasTipCommand || (!identity.test && identity.tipTest))
151
+ ? baseline?.commands.tipTest
152
+ : baseline?.commands[gate];
57
153
  // OBS-534: the ceiling is the BATTERY's, derived by effectiveCeilingMs from the same baseline entry
58
154
  // this loop already reads for forgiveness two lines down — never the flat DEFAULT_SHELL_TIMEOUT_MS
59
155
  // `sh` defaults to. A suite whose capture measured 600007ms carries a recorded 1800021ms ceiling;
60
156
  // running it under 600000ms here SIGKILLed a green tip three times while every per-task gate passed.
61
157
  const ceilingMs = effectiveCeilingMs(entry);
62
- const r = await sh(cmd, intWt, ceilingMs);
63
- const raw = r.stdout + "\n" + r.stderr;
64
- const stripped = raw.split(intWt).join("");
158
+ let r = await sh(cmd, intWt, ceilingMs);
159
+ let raw = r.stdout + "\n" + r.stderr;
160
+ let stripped = raw.split(intWt).join("");
161
+ let rerun;
162
+ if (gate === "test" && !r.timedOut && !fileCountDeficit(entry, stripped)
163
+ && classifyFreshRunnerOutput(entry, stripped, r.code) === "infra") {
164
+ const waitedMs = await waitForCalmWindow();
165
+ rerun = `runner-infra rerun after waiting ${waitedMs}ms for a calm load window`;
166
+ r = await sh(cmd, intWt, ceilingMs);
167
+ raw = r.stdout + "\n" + r.stderr;
168
+ stripped = raw.split(intWt).join("");
169
+ }
65
170
  const artifact = join(runDir, `tip-verify-${gate}.log`);
66
171
  // Battery parity on the ceiling too (baseline.ts Q24): the kill is read BEFORE the exit code is
67
172
  // interpreted at all. A SIGKILLed battery never returned a verdict, so no line of its partial
@@ -76,14 +181,16 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
76
181
  pass: false,
77
182
  exitCode: r.code,
78
183
  fingerprints: [],
79
- details: killed.details,
184
+ details: `${rerun ? `infra; ${rerun}: ` : ""}${killed.details}`,
80
185
  cause: killed.meta?.classification,
81
186
  artifact,
82
187
  });
83
188
  continue;
84
189
  }
85
190
  const { failing, unreadable } = freshFailures(entry, stripped);
86
- const cause = r.code === 0 ? undefined : classifyFailureOutput(stripped);
191
+ const deficit = fileCountDeficit(entry, stripped);
192
+ const cause = deficit ? "infra" : classifyFreshRunnerOutput(entry, stripped, r.code);
193
+ const greenTeardown = classifyRunnerOutput(stripped, r.code) === "green-teardown";
87
194
  // `?? 1` is the battery's own default (baseline.ts compareToBaseline): an exitCode-less legacy
88
195
  // entry reads as red-at-baseline there, so it must read the same here or old baselines silently
89
196
  // lose forgiveness. OBS-534 (T2): a capture killed at its ceiling now records a CAUSE and no
@@ -106,7 +213,7 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
106
213
  const comparable = sameCapacity(entry?.capacity, r.capacity);
107
214
  const forgiven = r.code !== 0 && baselineRed && failing.length === 0 && !unreadable && cause !== "infra"
108
215
  && comparable;
109
- const pass = r.code === 0 || forgiven;
216
+ const pass = !deficit && (r.code === 0 || greenTeardown || forgiven);
110
217
  if (!pass)
111
218
  writeFileSync(artifact, raw);
112
219
  results.push({
@@ -114,15 +221,16 @@ export async function verifyIntegrationTip(intWt, commands, runDir, baseline) {
114
221
  cmd,
115
222
  pass,
116
223
  exitCode: r.code,
117
- fingerprints: r.code !== 0 ? fingerprint(stripped) : [],
118
- details: r.code === 0 ? "exit 0"
119
- : forgiven ? `exit ${r.code} but only baseline-recorded failures (forgiven vs baseline)`
120
- : !comparable && baselineRed && failing.length === 0
121
- ? `exit ${r.code}; every failure is baseline-recorded, but that capture ran under `
122
- + `${describeCapacity(entry?.capacity)} and this verification ran under ${describeCapacity(r.capacity)} `
123
- + `— forgiveness across a changed capacity is not evidence`
124
- : `exit ${r.code}`,
125
- ...(forgiven ? { forgiven: true } : {}),
224
+ fingerprints: r.code !== 0 && !greenTeardown ? fingerprint(stripped) : [],
225
+ details: `${cause === "infra" ? "infra; " : ""}${rerun ? `${rerun}: ` : ""}` + (deficit?.replace(/^infra; /, "") ?? (r.code === 0 ? "exit 0"
226
+ : greenTeardown ? `exit ${r.code} after a green suite summary; only the runner's teardown fingerprint followed it`
227
+ : forgiven ? `exit ${r.code} but only baseline-recorded failures (forgiven vs baseline)`
228
+ : !comparable && baselineRed && failing.length === 0
229
+ ? `exit ${r.code}; every failure is baseline-recorded, but that capture ran under `
230
+ + `${describeCapacity(entry?.capacity)} and this verification ran under ${describeCapacity(r.capacity)} `
231
+ + `— forgiveness across a changed capacity is not evidence`
232
+ : `exit ${r.code}`)),
233
+ ...(forgiven && !deficit ? { forgiven: true } : {}),
126
234
  ...(cause ? { cause } : {}),
127
235
  ...(pass ? {} : { artifact }),
128
236
  });
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "tickmarkr",
3
- "version": "2.5.1",
3
+ "version": "2.5.2",
4
4
  "description": "Spec in, verified work out.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -220,17 +220,20 @@ banner_pct() {
220
220
  stripped = bare; gsub(rule, "", stripped)
221
221
  if (bare != "" && stripped == "") last = NR }
222
222
  END { labelled = 0
223
- for (i = 1; i <= NR; i++) if (line[i] ~ /Context [0-9]+% used/) { print line[i]; labelled = 1 }
223
+ for (i = 1; i <= NR; i++) if (line[i] ~ /Context [0-9]+% used/ || line[i] ~ /ctx [0-9]+%/) { print line[i]; labelled = 1 }
224
224
  if (!labelled && last) for (i = last + 1; i <= NR; i++) if (line[i] ~ /[0-9]+%/) print line[i] }
225
225
  ' | tail -1)
226
226
  # No rule line in the window, or no percentage below it: say so out loud rather than guess.
227
227
  [ -n "$banner" ] || { printf 'UNREADABLE\n'; return 0; }
228
+ # OBS-964 add.6: a claude-code statusline plugin (Graft) can replace the model row with a self-labelled
229
+ # `ctx N%` row; the label IS the selector, so no model name is required on that row.
230
+ printf '%s\n' "$banner" | grep -Eq 'ctx [0-9]+%' ||
228
231
  printf '%s\n' "$banner" |
229
232
  grep -Eqi '(^|[^[:alnum:]])(claude|opus|sonnet|haiku|fable|gpt|gemini|glm|kimi|grok|composer|openai|zai)[[:alnum:]_./-]*([[:space:]]|$)' ||
230
233
  { printf 'UNREADABLE\n'; return 0; }
231
234
  # OBS-964: a codex banner carries TWO percentages ("Context 16% used · weekly 96% left"); the LAST one is the
232
235
  # quota, not the fill. Prefer the labelled context figure; fall back to the last bare % (claude-code banner).
233
- pct=$(printf '%s\n' "$banner" | grep -oE 'Context [0-9]+% used' | grep -oE '[0-9]+' | head -1)
236
+ pct=$(printf '%s\n' "$banner" | grep -oE '(Context [0-9]+% used|ctx [0-9]+%)' | grep -oE '[0-9]+' | head -1)
234
237
  # OBS-964 add.3: a codex footer while WORKING drops the labelled figure but keeps `weekly N% left`; strip the
235
238
  # quota phrase before the bare-% fallback so a quota can never be read as fill.
236
239
  [ -n "$pct" ] || pct=$(printf '%s\n' "$banner" | sed -E 's/weekly [0-9]+% left//g' | grep -oE '[0-9]+%' | tail -1 | tr -d '%')