tickmarkr 2.5.1 → 2.5.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/commands/approve.js +1 -1
- package/dist/cli/commands/doctor.d.ts +2 -0
- package/dist/cli/commands/init.js +1 -1
- package/dist/cli/commands/run.d.ts +5 -0
- package/dist/cli/commands/run.js +14 -1
- package/dist/compile/native.js +3 -0
- package/dist/config/config.d.ts +6 -0
- package/dist/config/config.js +6 -0
- package/dist/drivers/subprocess.d.ts +2 -3
- package/dist/drivers/subprocess.js +2 -3
- package/dist/gates/baseline.d.ts +9 -0
- package/dist/gates/baseline.js +85 -29
- package/dist/gates/llm.js +2 -2
- package/dist/gates/review.js +18 -11
- package/dist/run/daemon.d.ts +6 -3
- package/dist/run/daemon.js +3138 -2941
- package/dist/run/git.d.ts +4 -2
- package/dist/run/git.js +11 -23
- package/dist/run/merge.d.ts +1 -1
- package/dist/run/merge.js +126 -18
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +5 -2
|
@@ -114,7 +114,7 @@ export function approvalRunOwner(cwd, runId) {
|
|
|
114
114
|
/** The one sentence that says who enacts this release and what it buys. */
|
|
115
115
|
export function approvalEnactment(token, run) {
|
|
116
116
|
if (run.live) {
|
|
117
|
-
return `the live daemon enacts this at its next task boundary — it will ${APPROVAL_ENACTS[token]}`;
|
|
117
|
+
return `the live daemon enacts this at its next task boundary — it will ${APPROVAL_ENACTS[token]}, including in the approval window or by cancelling an in-progress tip verify`;
|
|
118
118
|
}
|
|
119
119
|
if (run.blockingRunId) {
|
|
120
120
|
return `release recorded; live run \`${run.blockingRunId}\` holds the repository lock, so resume \`${run.runId}\` after it ends to ${APPROVAL_ENACTS[token]}`;
|
|
@@ -162,6 +162,7 @@ export declare function doctorProbePreflight(cwd?: string, cfg?: {
|
|
|
162
162
|
gates: {
|
|
163
163
|
build?: string | undefined;
|
|
164
164
|
test?: string | undefined;
|
|
165
|
+
tipTest?: string | undefined;
|
|
165
166
|
lint?: string | undefined;
|
|
166
167
|
diffCap?: number | undefined;
|
|
167
168
|
byShape?: Partial<Record<"plan" | "spec" | "implement" | "tests" | "docs" | "migration" | "ui" | "refactor" | "chore", {
|
|
@@ -278,6 +279,7 @@ export declare function cachedDoctorDiagnostics(cwd?: string, adapters?: WorkerA
|
|
|
278
279
|
gates: {
|
|
279
280
|
build?: string | undefined;
|
|
280
281
|
test?: string | undefined;
|
|
282
|
+
tipTest?: string | undefined;
|
|
281
283
|
lint?: string | undefined;
|
|
282
284
|
diffCap?: number | undefined;
|
|
283
285
|
byShape?: Partial<Record<"plan" | "spec" | "implement" | "tests" | "docs" | "migration" | "ui" | "refactor" | "chore", {
|
|
@@ -114,7 +114,7 @@ A run is green only when ALL of these hold: the run-end event exists in the jour
|
|
|
114
114
|
|
|
115
115
|
### Verified handoffs
|
|
116
116
|
|
|
117
|
-
When relaying missions between agents, never use bare send-text (\`herdr agent send\` / pane send-text) — it omits Enter. Use \`herdr pane run <pane> "<message>"\` or \`herdr notification show "<message>"\`. Confirm delivery by reading the
|
|
117
|
+
When relaying missions between agents, never use bare send-text (\`herdr agent send\` / pane send-text) — it omits Enter. Use \`herdr pane run <pane> "<message>"\` or \`herdr notification show "<message>"\`. Confirm delivery by reading the TARGET composer afterward; if the sent text still sits unsubmitted after ~15 s, send-keys enter to that pane once and read again (OBS-975); never report "relayed" without read-back.
|
|
118
118
|
|
|
119
119
|
### Orient before you act — this block may be the ONLY guidance your host loaded
|
|
120
120
|
|
|
@@ -4,6 +4,11 @@ import { type JournalEvent } from "../../run/journal.js";
|
|
|
4
4
|
* it is a phase counter, while a phase-start CARRYING a gate is the gate start the rail draws. */
|
|
5
5
|
export declare const TTY_NOISE_EVENTS: readonly ["worker-contact", "worker-status"];
|
|
6
6
|
type RailTone = "pass" | "fail" | "attention" | "active" | "neutral";
|
|
7
|
+
/** Approval-close lifecycle labels extend the established closed rail vocabulary. */
|
|
8
|
+
export declare const APPROVAL_RAIL_ROWS: Record<string, {
|
|
9
|
+
label: string;
|
|
10
|
+
tone: RailTone;
|
|
11
|
+
}>;
|
|
7
12
|
/** Closed retained set: the short operator label and the row's default tone. Labels are the rail's
|
|
8
13
|
* own vocabulary — a raw journal event name is what this surface exists to stop printing. A `pass`
|
|
9
14
|
* or `ok` datum on the event overrides the default tone, so one gate row can read either way.
|
package/dist/cli/commands/run.js
CHANGED
|
@@ -33,6 +33,12 @@ const RAIL_TONES = {
|
|
|
33
33
|
active: { glyph: GLYPHS.pointer, paint: LIVE.running },
|
|
34
34
|
neutral: { glyph: GLYPHS.neutral, paint: LIVE.chrome },
|
|
35
35
|
};
|
|
36
|
+
/** Approval-close lifecycle labels extend the established closed rail vocabulary. */
|
|
37
|
+
export const APPROVAL_RAIL_ROWS = {
|
|
38
|
+
"approval-window-start": { label: "approval window", tone: "attention" },
|
|
39
|
+
"approval-window-expired": { label: "approval window expired", tone: "attention" },
|
|
40
|
+
"tip-verify-cancelled": { label: "tip verify cancelled", tone: "attention" },
|
|
41
|
+
};
|
|
36
42
|
/** Closed retained set: the short operator label and the row's default tone. Labels are the rail's
|
|
37
43
|
* own vocabulary — a raw journal event name is what this surface exists to stop printing. A `pass`
|
|
38
44
|
* or `ok` datum on the event overrides the default tone, so one gate row can read either way.
|
|
@@ -116,6 +122,8 @@ export const RAIL_ROWS = {
|
|
|
116
122
|
// gate starts and verdicts
|
|
117
123
|
"phase-start": { label: "gate start", tone: "active" },
|
|
118
124
|
"gate-result": { label: "gate", tone: "pass" },
|
|
125
|
+
"baseline-wait": { label: "baseline wait", tone: "active" },
|
|
126
|
+
"suite-budget": { label: "suite budget", tone: "attention" },
|
|
119
127
|
"gate-reused": { label: "gate reused", tone: "neutral" },
|
|
120
128
|
"judge-retry": { label: "judge retry", tone: "attention" },
|
|
121
129
|
"review-no-verdict": { label: "review unavailable", tone: "attention" },
|
|
@@ -307,6 +315,11 @@ const RAIL_PROJECTION = {
|
|
|
307
315
|
? `gated ${d.gatedCommit.slice(0, 12)}, tip ${d.branchTip.slice(0, 12)}`
|
|
308
316
|
: undefined,
|
|
309
317
|
"trust-auto-answer": (d) => typeof d.adapter === "string" ? `${d.adapter}${typeof d.phase === "string" ? ` ${d.phase}` : ""}` : undefined,
|
|
318
|
+
// SB-1: the census the ceiling released beside, and the budget the round ran under versus the one it
|
|
319
|
+
// would have had on an empty census — none of `{count, occupancyCap, conservativeCap}` is on the ladder
|
|
320
|
+
"suite-budget": (d) => typeof d.count === "number" && typeof d.conservativeCap === "number" && typeof d.occupancyCap === "number"
|
|
321
|
+
? `beside ${d.count}, cap ${d.conservativeCap} not ${d.occupancyCap}`
|
|
322
|
+
: undefined,
|
|
310
323
|
};
|
|
311
324
|
// The row's text: the formatter's detail first, then the salient fields that detail could not carry.
|
|
312
325
|
// Appending (never prefixing) keeps the operator's prose at the front of the row, so the clip a
|
|
@@ -333,7 +346,7 @@ const clipCells = (text, cells) => cells <= 0 ? "" : cellWidth(text) <= cells ?
|
|
|
333
346
|
export function narrationRow(event, runId, columns = process.stdout.columns ?? 80) {
|
|
334
347
|
if (TTY_NOISE_EVENTS.includes(event.event))
|
|
335
348
|
return null;
|
|
336
|
-
const row = RAIL_ROWS[event.event];
|
|
349
|
+
const row = APPROVAL_RAIL_ROWS[event.event] ?? RAIL_ROWS[event.event];
|
|
337
350
|
if (!row)
|
|
338
351
|
return null;
|
|
339
352
|
if (event.event === "phase-start" && typeof event.data.gate !== "string")
|
package/dist/compile/native.js
CHANGED
|
@@ -847,6 +847,9 @@ acceptance is required on every task (a nested list of observable outcomes).
|
|
|
847
847
|
hard value anywhere in the domain, the criterion asserts a universal that may be FALSE ABOUT THE
|
|
848
848
|
WORLD — bound it or say where it stops holding, rather than demanding a value that does not exist.
|
|
849
849
|
|
|
850
|
+
- A millisecond ceiling in a test is a BUDGET for the SLOWEST RUNNER.
|
|
851
|
+
Any ceiling under one second must carry a SLOWEST-RUNNER note and a member that OVERRUNS it.
|
|
852
|
+
|
|
850
853
|
PICK THE CRITERION FORM FROM WHO COULD BE WRONG:
|
|
851
854
|
- When the WORKER could be wrong because it can choose the value, use "test:" and pin the exact
|
|
852
855
|
literal it could otherwise choose; an example selected by its implementer proves only itself.
|
package/dist/config/config.d.ts
CHANGED
|
@@ -240,6 +240,7 @@ export declare const TickmarkrConfigSchema: z.ZodObject<{
|
|
|
240
240
|
gates: z.ZodObject<{
|
|
241
241
|
build: z.ZodOptional<z.ZodString>;
|
|
242
242
|
test: z.ZodOptional<z.ZodString>;
|
|
243
|
+
tipTest: z.ZodOptional<z.ZodString>;
|
|
243
244
|
lint: z.ZodOptional<z.ZodString>;
|
|
244
245
|
diffCap: z.ZodOptional<z.ZodNumber>;
|
|
245
246
|
byShape: z.ZodOptional<z.ZodOptional<z.ZodRecord<z.ZodEnum<{
|
|
@@ -339,6 +340,11 @@ export type InitConfigOverlay = {
|
|
|
339
340
|
llm?: TickmarkrConfig["visibility"]["llm"];
|
|
340
341
|
};
|
|
341
342
|
};
|
|
343
|
+
/**
|
|
344
|
+
* Default config overlay template.
|
|
345
|
+
* Seam: gates.test defines the per-task gate and baseline test command;
|
|
346
|
+
* gates.tipTest optionally defines the integration tip verify test command (defaults to gates.test).
|
|
347
|
+
*/
|
|
342
348
|
export declare function configTemplate(overlay?: InitConfigOverlay): string;
|
|
343
349
|
export { FLEET_OVERLAY_KEYS, fleetEditableEquals, fleetRepoOverlayFromDelta, renderFleetOverlayWrite, repoOverlayYaml, serializeFleetOverlay, unifiedYamlDiff, } from "./fleet-overlay.js";
|
|
344
350
|
export type { FleetOverlayWrite } from "./fleet-overlay.js";
|
package/dist/config/config.js
CHANGED
|
@@ -337,6 +337,7 @@ export const TickmarkrConfigSchema = z.object({
|
|
|
337
337
|
gates: z.object({
|
|
338
338
|
build: z.string(),
|
|
339
339
|
test: z.string(),
|
|
340
|
+
tipTest: z.string(),
|
|
340
341
|
lint: z.string(),
|
|
341
342
|
diffCap: z.number().int().positive(),
|
|
342
343
|
byShape: z.partialRecord(z.enum(SHAPES), ShapeGateParticipationSchema).optional(),
|
|
@@ -756,6 +757,11 @@ export function overlayBytesLoadError(repoRoot, bytes, opts = {}) {
|
|
|
756
757
|
return e.message;
|
|
757
758
|
}
|
|
758
759
|
}
|
|
760
|
+
/**
|
|
761
|
+
* Default config overlay template.
|
|
762
|
+
* Seam: gates.test defines the per-task gate and baseline test command;
|
|
763
|
+
* gates.tipTest optionally defines the integration tip verify test command (defaults to gates.test).
|
|
764
|
+
*/
|
|
759
765
|
export function configTemplate(overlay) {
|
|
760
766
|
const base = `# tickmarkr config overlay — merges over built-in defaults (repo beats global beats defaults)
|
|
761
767
|
# concurrency: 3
|
|
@@ -3,9 +3,8 @@ export declare const MAX_BUF: number;
|
|
|
3
3
|
export declare const HERDR_CONTROL_VARS: readonly ["HERDR_ENV", "HERDR_SOCKET_PATH", "ORCA_TERMINAL_HANDLE", "ORCA_PANE_KEY", "ORCA_TAB_ID"];
|
|
4
4
|
/**
|
|
5
5
|
* Copy of worker env with the fork cap applied and host control-plane vars stripped.
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
* than a flat constant. The operator's own export still wins.
|
|
6
|
+
* Workers retain the run's concurrency-derived cap (resolvedForkCap), independently of
|
|
7
|
+
* verification's suite-window budget. The operator's own export still wins.
|
|
9
8
|
*/
|
|
10
9
|
export declare function sealHerdrEnv(env?: NodeJS.ProcessEnv): NodeJS.ProcessEnv;
|
|
11
10
|
/** Pane/login-shell form of the same host-neutral worker env seal (herdr seed + daemon setup). */
|
|
@@ -22,9 +22,8 @@ export const HERDR_CONTROL_VARS = [
|
|
|
22
22
|
];
|
|
23
23
|
/**
|
|
24
24
|
* Copy of worker env with the fork cap applied and host control-plane vars stripped.
|
|
25
|
-
*
|
|
26
|
-
*
|
|
27
|
-
* than a flat constant. The operator's own export still wins.
|
|
25
|
+
* Workers retain the run's concurrency-derived cap (resolvedForkCap), independently of
|
|
26
|
+
* verification's suite-window budget. The operator's own export still wins.
|
|
28
27
|
*/
|
|
29
28
|
export function sealHerdrEnv(env = process.env) {
|
|
30
29
|
const out = { ...env };
|
package/dist/gates/baseline.d.ts
CHANGED
|
@@ -31,6 +31,8 @@ export interface BaselineCommand {
|
|
|
31
31
|
durationMs?: number;
|
|
32
32
|
/** Sum of the per-file durations named by the runner; null when its output names none. */
|
|
33
33
|
fileDurationSumMs?: number | null;
|
|
34
|
+
/** Total files reported by the runner summary; null when unavailable. */
|
|
35
|
+
fileCount?: number | null;
|
|
34
36
|
/** fileDurationSumMs / durationMs — average implied file concurrency, not a configured fork count. */
|
|
35
37
|
impliedParallelism?: number | null;
|
|
36
38
|
/** The slowest per-file entry named by the runner; null when per-file timing is unavailable. */
|
|
@@ -158,6 +160,7 @@ export declare function detectVacuousOracles(cwd: string, tasks: ReadonlyArray<{
|
|
|
158
160
|
}>): Promise<VacuousOracleWarning[]>;
|
|
159
161
|
export declare function compareToBaseline(cwd: string, commands: Record<string, string>, baseline: Baseline, enabled: string[], opts?: {
|
|
160
162
|
rerunOf?: HostStarvedRerun;
|
|
163
|
+
infraRerun?: HostStarvedRerun;
|
|
161
164
|
}): Promise<GateResult[]>;
|
|
162
165
|
export interface HostStarvedRerun {
|
|
163
166
|
durationMs: number;
|
|
@@ -178,4 +181,10 @@ interface CalmWindow {
|
|
|
178
181
|
}
|
|
179
182
|
export declare function setCalmWindowForTests(over: Partial<CalmWindow>): void;
|
|
180
183
|
export declare function resetCalmWindowForTests(): void;
|
|
184
|
+
export declare function waitForCalmWindow(): Promise<number>;
|
|
185
|
+
/** Summary totals survive digit-normalized fingerprints and exclude test-owned echoed output. */
|
|
186
|
+
export declare function runnerFileCount(raw: string): number | null;
|
|
187
|
+
export declare function fileCountDeficit(entry: BaselineCommand | undefined, raw: string): string | undefined;
|
|
188
|
+
/** Both readers classify fresh evidence first, retaining the whole-output guard against infra forgiveness. */
|
|
189
|
+
export declare function classifyFreshRunnerOutput(entry: BaselineCommand | undefined, raw: string, code: number): FailureClassification | undefined;
|
|
181
190
|
export {};
|
package/dist/gates/baseline.js
CHANGED
|
@@ -30,7 +30,7 @@ const FAIL_ANCHOR_RE = /^\s*(?:FAIL\s+|[^\w]*(?:Unhandled Errors|Uncaught Except
|
|
|
30
30
|
// to render. Digits are written (?:\d+|#) so a shape matches both raw and digit-normalized lines.
|
|
31
31
|
// Run summaries: " Tests N failed | M passed (T)" (vitest), "# fail N" (TAP / node:test),
|
|
32
32
|
// "ℹ fail N" (node:test's spec reporter) and "test result: FAILED. …" (cargo / libtest).
|
|
33
|
-
const SUMMARY_FAIL_RE = /^\s*(?:Tests?\s+(?:Files?\s+)?(?:\d+|#)\s+failed|#\s+fail\s+(?!0\b)(?:\d+|#)\b|ℹ\s+fail\s+(?!0\b)(?:\d+|#)\b|test result:\s+FAILED\b)/;
|
|
33
|
+
const SUMMARY_FAIL_RE = /^\s*(?:Tests?\s+(?:Files?\s+)?(?!0\b)(?:\d+|#)\s+failed|#\s+fail\s+(?!0\b)(?:\d+|#)\b|ℹ\s+fail\s+(?!0\b)(?:\d+|#)\b|test result:\s+FAILED\b)/;
|
|
34
34
|
const ERROR_ANCHOR_RE = /^\s*(?:Error|[A-Za-z_$][\w$]*Error):\s+\S/;
|
|
35
35
|
const TSC_ERROR_RE = /^\s*\S.*\((?:\d+|#),(?:\d+|#)\):\s+error\s+[A-Z]+(?:\d+|#):/i; // tsc
|
|
36
36
|
const LINTER_ERROR_RE = /^\s*(?:\d+|#):(?:\d+|#)\s+error\s+\S/; // eslint stylish
|
|
@@ -103,9 +103,8 @@ const stripTurboPrefix = (l) => {
|
|
|
103
103
|
// Lines that NAME a failing test — the ones worth headlining to the operator. One list, so recognition
|
|
104
104
|
// and reporting cannot drift apart (a shape that blocks but never gets named cost 3 attempts once).
|
|
105
105
|
const namesFailure = (l) => FAIL_ANCHOR_RE.test(l) || RUNNER_FAIL_RE.test(l) || TRAILING_FAIL_RE.test(l) || GLYPH_FAIL_RE.test(l) || TURBO_FAIL_RE.test(l);
|
|
106
|
-
// The stripped form
|
|
107
|
-
//
|
|
108
|
-
// reading the raw line only, so infra/regression verdicts are byte-unchanged by the prefix pass.
|
|
106
|
+
// The stripped form lets recognition and headlines read the runner beneath a turbo prefix.
|
|
107
|
+
// Classification applies the same prefix stripping before its infra/regression vetoes.
|
|
109
108
|
const namesFailureEitherForm = (l) => {
|
|
110
109
|
if (namesFailure(l))
|
|
111
110
|
return true;
|
|
@@ -163,9 +162,13 @@ export function classifyFailureOutput(output) {
|
|
|
163
162
|
// token inside a test's echoed stdout/stderr block must be as invisible to this classifier as it is
|
|
164
163
|
// to fingerprint(): test-owned output is never runner evidence about the work.
|
|
165
164
|
const lines = withoutVitestEchoBlocks(output).map((l) => l.replace(ANSI_RE, "")).filter((l) => !PASS_LINE_RE.test(l) && !OPERATOR_LINE_RE.test(l));
|
|
166
|
-
|
|
165
|
+
const evidence = lines.map((line) => stripTurboPrefix(line) ?? line).filter((line) => !PASS_LINE_RE.test(line) && !OPERATOR_LINE_RE.test(line));
|
|
166
|
+
const infra = evidence.some(isInfraLine);
|
|
167
|
+
// A diagnostic section heading names no failing test. It cannot outvote the RPC death
|
|
168
|
+
// beneath it, but an actual FAIL/AssertionError anywhere still takes precedence.
|
|
169
|
+
if (evidence.some((line) => !(infra && UNHANDLED_HEADER_RE.test(line)) && namesRegression(line)))
|
|
167
170
|
return "regression";
|
|
168
|
-
return
|
|
171
|
+
return infra ? "infra" : undefined;
|
|
169
172
|
}
|
|
170
173
|
/**
|
|
171
174
|
* Capture validity asks a different question from gate classification. At a gate, one genuine
|
|
@@ -338,6 +341,8 @@ export function detectGateCommands(repoRoot, cfg) {
|
|
|
338
341
|
else if (scripts[name])
|
|
339
342
|
out[name] = `${runPrefix} ${name}`;
|
|
340
343
|
}
|
|
344
|
+
if (cfg.gates.tipTest)
|
|
345
|
+
out.tipTest = cfg.gates.tipTest;
|
|
341
346
|
return out;
|
|
342
347
|
}
|
|
343
348
|
/**
|
|
@@ -452,6 +457,7 @@ const invalidCaptureEntry = (durationMs, invalidCause, invalidatingLines = []) =
|
|
|
452
457
|
fingerprints: [],
|
|
453
458
|
durationMs,
|
|
454
459
|
fileDurationSumMs: null,
|
|
460
|
+
fileCount: null,
|
|
455
461
|
impliedParallelism: null,
|
|
456
462
|
longestFile: null,
|
|
457
463
|
ceilingMs: effectiveCeilingMs({ durationMs }),
|
|
@@ -460,11 +466,16 @@ const invalidCaptureEntry = (durationMs, invalidCause, invalidatingLines = []) =
|
|
|
460
466
|
export async function captureBaseline(cwd, commands) {
|
|
461
467
|
const base = { commands: {} };
|
|
462
468
|
for (const [name, cmd] of Object.entries(commands)) {
|
|
469
|
+
if (name === "tipTest" && commands.test !== undefined && cmd === commands.test) {
|
|
470
|
+
continue;
|
|
471
|
+
}
|
|
463
472
|
const r = await sh(cmd, cwd, CAPTURE_CEILING_MS);
|
|
464
473
|
// ponytail: strip the executing cwd so repo-root capture and worktree compare fingerprint identically; /private-vs-/tmp symlink variance is out of scope
|
|
465
474
|
// ponytail: a capture that was itself killed records the ceiling as its "measurement", which
|
|
466
475
|
// scales the next ceiling up — the right direction for a suite that never finished once.
|
|
467
476
|
const durationMs = r.durationMs ?? 0;
|
|
477
|
+
const combinedOutput = r.stdout + "\n" + r.stderr;
|
|
478
|
+
const raw = combinedOutput.split(cwd).join("");
|
|
468
479
|
// OBS-534 (T2): a capture SIGKILLed at its ceiling never returned a verdict, so `r.code` is the
|
|
469
480
|
// kill and not evidence about the command. Run 1501 recorded `test: {durationMs: 600007,
|
|
470
481
|
// exitCode: 1}` for exactly this — a kill written down as a red baseline. There the accident
|
|
@@ -483,11 +494,9 @@ export async function captureBaseline(cwd, commands) {
|
|
|
483
494
|
console.error(`tickmarkr: baseline capture for "${name}" was killed at its ${CAPTURE_CEILING_MS}ms ceiling — `
|
|
484
495
|
+ `it recorded NO fingerprints, so nothing is forgiven and every gate will treat a pre-existing `
|
|
485
496
|
+ `failure as a fresh one. Raise the ceiling or shorten the command.`);
|
|
486
|
-
base.commands[name] = invalidCaptureEntry(durationMs, "ceiling-kill");
|
|
497
|
+
base.commands[name] = { ...invalidCaptureEntry(durationMs, "ceiling-kill"), fileCount: runnerFileCount(raw) };
|
|
487
498
|
continue;
|
|
488
499
|
}
|
|
489
|
-
const combinedOutput = r.stdout + "\n" + r.stderr;
|
|
490
|
-
const raw = combinedOutput.split(cwd).join("");
|
|
491
500
|
// Run 2137: the child exited and printed ordinary FAIL/AssertionError lines, so the gate-side
|
|
492
501
|
// discriminator correctly called the mixed output a regression. But the same output also said
|
|
493
502
|
// `spawn EAGAIN`: the machine had run out of processes while the pristine-tree measurement was
|
|
@@ -499,15 +508,15 @@ export async function captureBaseline(cwd, commands) {
|
|
|
499
508
|
+ `it recorded NO exit-code verdict and NO fingerprints, so nothing is forgiven for this command; `
|
|
500
509
|
+ `the measurement cannot distinguish a pre-existing failure from one caused by exhaustion. `
|
|
501
510
|
+ `First invalidating line: ${invalidatingLines[0]}`);
|
|
502
|
-
base.commands[name] = invalidCaptureEntry(durationMs, "resource-exhaustion", invalidatingLines);
|
|
511
|
+
base.commands[name] = { ...invalidCaptureEntry(durationMs, "resource-exhaustion", invalidatingLines), fileCount: runnerFileCount(raw) };
|
|
503
512
|
continue;
|
|
504
513
|
}
|
|
505
|
-
// OBS-
|
|
506
|
-
//
|
|
514
|
+
// OBS-966: a worker RPC timeout is infra even beside an all-green summary.
|
|
515
|
+
// Capture and both gate readers share this discriminator; genuine test failures still dominate.
|
|
507
516
|
const runnerVerdict = classifyRunnerOutput(raw, r.code);
|
|
508
517
|
if (runnerVerdict === "infra") {
|
|
509
|
-
console.error(`tickmarkr: baseline capture for "${name}" carries runner-infrastructure evidence
|
|
510
|
-
base.commands[name] = invalidCaptureEntry(durationMs, "infra");
|
|
518
|
+
console.error(`tickmarkr: baseline capture for "${name}" carries runner-infrastructure evidence — it recorded NO verdict; nothing is forgiven for this command`);
|
|
519
|
+
base.commands[name] = { ...invalidCaptureEntry(durationMs, "infra", withoutVitestEchoBlocks(combinedOutput).filter((line) => isInfraLine(line.replace(ANSI_RE, "")))), fileCount: runnerFileCount(raw) };
|
|
511
520
|
continue;
|
|
512
521
|
}
|
|
513
522
|
base.commands[name] = {
|
|
@@ -519,13 +528,14 @@ export async function captureBaseline(cwd, commands) {
|
|
|
519
528
|
missingCommand: missingConfiguredCommand(cmd, r),
|
|
520
529
|
durationMs,
|
|
521
530
|
...fileTiming(raw, durationMs),
|
|
531
|
+
fileCount: runnerFileCount(raw),
|
|
522
532
|
ceilingMs: effectiveCeilingMs({ durationMs }),
|
|
523
533
|
// T7: the world this measurement was taken in, so a later reader can ask whether its own world
|
|
524
534
|
// is the same one. Recorded from THIS command's own shell result, never re-derived here.
|
|
525
535
|
...(r.capacity ? { capacity: r.capacity } : {}),
|
|
526
536
|
};
|
|
527
537
|
}
|
|
528
|
-
const names = Object.keys(commands);
|
|
538
|
+
const names = Object.keys(commands).filter((name) => !(name === "tipTest" && commands.test !== undefined && commands[name] === commands.test));
|
|
529
539
|
const missing = names.filter((name) => base.commands[name]?.missingCommand === true);
|
|
530
540
|
if (names.length > 0 && missing.length === names.length) {
|
|
531
541
|
base.warnings = [{
|
|
@@ -609,10 +619,15 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
|
|
|
609
619
|
const withReapError = r.reapError ? { ...withReap, meta: { ...withReap.meta, reapError: r.reapError } } : withReap;
|
|
610
620
|
const withRerun = rerunOf ? {
|
|
611
621
|
...withReapError,
|
|
612
|
-
details:
|
|
622
|
+
details: `${g.meta?.classification === "infra" ? "infra; " : ""}host-starved rerun after waiting ${rerunOf.waitedMs}ms for a calm load window: ${withReapError.details.replace(/^infra; /, "")}`,
|
|
613
623
|
meta: { ...withReapError.meta, hostStarvedRerun: rerunOf },
|
|
614
624
|
} : withReapError;
|
|
615
|
-
|
|
625
|
+
const final = opts.infraRerun ? {
|
|
626
|
+
...withRerun,
|
|
627
|
+
details: `${g.meta?.classification === "infra" ? "infra; " : ""}runner-infra rerun after waiting ${opts.infraRerun.waitedMs}ms for a calm load window: ${withRerun.details.replace(/^infra; /, "")}`,
|
|
628
|
+
meta: { ...withRerun.meta, runnerInfraRerun: opts.infraRerun },
|
|
629
|
+
} : withRerun;
|
|
630
|
+
results.push(r.capacity ? { ...final, capacity: r.capacity } : final);
|
|
616
631
|
};
|
|
617
632
|
// …and whether the entry that would forgive this command was measured in the same world. A
|
|
618
633
|
// baseline captured under a different fork cap forgives nothing: its fingerprints describe a
|
|
@@ -641,11 +656,16 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
|
|
|
641
656
|
});
|
|
642
657
|
continue;
|
|
643
658
|
}
|
|
659
|
+
const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
|
|
660
|
+
const deficit = fileCountDeficit(entry, raw);
|
|
661
|
+
if (deficit) {
|
|
662
|
+
record({ gate: name, pass: false, details: deficit, meta: { classification: "infra", infra: true } });
|
|
663
|
+
continue;
|
|
664
|
+
}
|
|
644
665
|
if (r.code === 0) {
|
|
645
666
|
record({ gate: name, pass: true, details: "exit 0" });
|
|
646
667
|
continue;
|
|
647
668
|
}
|
|
648
|
-
const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
|
|
649
669
|
// OBS-885/887: the same classifier the capture applied names a completed green suite on both sides.
|
|
650
670
|
const runnerVerdict = classifyRunnerOutput(raw, r.code);
|
|
651
671
|
if (runnerVerdict === "green-teardown") {
|
|
@@ -665,12 +685,16 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
|
|
|
665
685
|
// that known assertion outvote the fresh birpc line turns machine failure into a worker defect.
|
|
666
686
|
// `classifyRunnerOutput` remains the single discriminator. When there is no fresh fingerprint,
|
|
667
687
|
// retain the whole-output read so a repeated infra abort can never be baseline-forgiven as green.
|
|
668
|
-
const
|
|
669
|
-
|
|
670
|
-
|
|
688
|
+
const classification = classifyFreshRunnerOutput(entry, raw, r.code);
|
|
689
|
+
if (name === "test" && classification === "infra" && !rerunOf && !opts.infraRerun) {
|
|
690
|
+
const waitedMs = await waitForCalmWindow();
|
|
691
|
+
const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
|
|
692
|
+
results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { infraRerun: provenance }));
|
|
693
|
+
continue;
|
|
694
|
+
}
|
|
671
695
|
// OBS-896: every fresh failure must be timeout-class, and the suite must take at least twice its
|
|
672
696
|
// own baseline measurement. The first read buys one calm rerun here, never a worker repair.
|
|
673
|
-
if (name === "test" && classification !== "infra" && failing.length && !rerunOf
|
|
697
|
+
if (name === "test" && classification !== "infra" && failing.length && !rerunOf && !opts.infraRerun
|
|
674
698
|
&& hostStarved(failing.join("\n"), r.durationMs ?? 0, entry?.durationMs)) {
|
|
675
699
|
const waitedMs = await waitForCalmWindow();
|
|
676
700
|
const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
|
|
@@ -684,7 +708,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
|
|
|
684
708
|
record({
|
|
685
709
|
gate: name,
|
|
686
710
|
pass: false,
|
|
687
|
-
details: `exit ${r.code}; the fresh failures carry infrastructure evidence alone — the runner never completed a suite, so this gate verified nothing:\n${evidence}`,
|
|
711
|
+
details: `infra; exit ${r.code}; the fresh failures carry infrastructure evidence alone — the runner never completed a suite, so this gate verified nothing:\n${evidence}`,
|
|
688
712
|
meta: { classification, infra: true },
|
|
689
713
|
});
|
|
690
714
|
continue;
|
|
@@ -695,9 +719,11 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
|
|
|
695
719
|
// entries with no exitCode keep the `?? 1` red default both readers share (merge.ts:131).
|
|
696
720
|
const baselineRed = entry?.infra !== true && (entry?.exitCode ?? 1) !== 0;
|
|
697
721
|
if (!failing.length && !baselineRed) {
|
|
698
|
-
const recordedCause = entry?.invalidCause === "
|
|
699
|
-
?
|
|
700
|
-
:
|
|
722
|
+
const recordedCause = entry?.invalidCause === "infra"
|
|
723
|
+
? "was invalidated by its recorded runner-infrastructure cause"
|
|
724
|
+
: entry?.invalidCause === "resource-exhaustion" || entry?.invalidatingLines?.length
|
|
725
|
+
? `was invalidated by its recorded process/resource-exhaustion cause${entry.invalidatingLines?.[0] ? ` (${entry.invalidatingLines[0]})` : ""}`
|
|
726
|
+
: "was killed at its ceiling";
|
|
701
727
|
const closed = entry?.infra === true
|
|
702
728
|
? `the baseline capture for this command ${recordedCause} and recorded no verdict, so nothing here is forgivable — it now exits ${r.code} with no recognizable failure lines — failing closed`
|
|
703
729
|
: `command was green at baseline but now exits ${r.code} with no recognizable failure lines — failing closed`;
|
|
@@ -742,7 +768,7 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
|
|
|
742
768
|
return results;
|
|
743
769
|
}
|
|
744
770
|
const SUMMARY_LINE_RE = /^[^\S\n]*Test Files[^\S\n]+(.+)$/m;
|
|
745
|
-
const TEARDOWN_RE = /\[
|
|
771
|
+
const TEARDOWN_RE = /\[birpc\] rpc is closed, cannot call\b/;
|
|
746
772
|
const UNHANDLED_HEADER_RE = /^\s*[^\w]*(?:Unhandled Errors|Uncaught Exception)\b/;
|
|
747
773
|
/** One interpretation of runner bytes, shared by baseline capture and every gate consumer. */
|
|
748
774
|
export function classifyRunnerOutput(raw, code) {
|
|
@@ -750,6 +776,8 @@ export function classifyRunnerOutput(raw, code) {
|
|
|
750
776
|
return undefined;
|
|
751
777
|
const lines = withoutVitestEchoBlocks(raw).map((l) => l.replace(ANSI_RE, ""));
|
|
752
778
|
const text = lines.join("\n");
|
|
779
|
+
if (lines.some((line) => /\[vitest-worker\]: Timeout calling\b/.test(line) && !PASS_LINE_RE.test(line) && !OPERATOR_LINE_RE.test(line)))
|
|
780
|
+
return classifyFailureOutput(text);
|
|
753
781
|
const summary = SUMMARY_LINE_RE.exec(text);
|
|
754
782
|
if (summary) {
|
|
755
783
|
const failed = [...summary[1].matchAll(/\b(\d+)\s+failed\b/g)].map((match) => Number(match[1]));
|
|
@@ -780,14 +808,42 @@ export function hostStarved(fresh, durationMs, referenceMs) {
|
|
|
780
808
|
const heads = fresh.split("\n").map((line) => line.replace(ANSI_RE, "")).filter((line) => ERROR_HEAD_RE.test(line));
|
|
781
809
|
return heads.length > 0 && heads.every((line) => TIMEOUT_CLASS_RE.test(line));
|
|
782
810
|
}
|
|
783
|
-
const DEFAULT_CALM = {
|
|
811
|
+
const DEFAULT_CALM = {
|
|
812
|
+
pollMs: process.env.VITEST ? 10 : 5_000,
|
|
813
|
+
maxWaitMs: process.env.VITEST ? 50 : 600_000,
|
|
814
|
+
loadProvider: () => loadavg()[0] ?? 0,
|
|
815
|
+
calmLoad: () => availableParallelism() / 2,
|
|
816
|
+
};
|
|
784
817
|
let calm = DEFAULT_CALM;
|
|
785
818
|
export function setCalmWindowForTests(over) { calm = { ...calm, ...over }; }
|
|
786
819
|
export function resetCalmWindowForTests() { calm = DEFAULT_CALM; }
|
|
787
|
-
async function waitForCalmWindow() {
|
|
820
|
+
export async function waitForCalmWindow() {
|
|
788
821
|
const started = Date.now();
|
|
789
822
|
while (calm.loadProvider() > calm.calmLoad() && Date.now() - started < calm.maxWaitMs) {
|
|
790
823
|
await new Promise((resolve) => setTimeout(resolve, calm.pollMs));
|
|
791
824
|
}
|
|
792
825
|
return Date.now() - started;
|
|
793
826
|
}
|
|
827
|
+
/** Summary totals survive digit-normalized fingerprints and exclude test-owned echoed output. */
|
|
828
|
+
export function runnerFileCount(raw) {
|
|
829
|
+
const counts = withoutVitestEchoBlocks(raw).flatMap((line) => {
|
|
830
|
+
const clean = line.replace(ANSI_RE, "");
|
|
831
|
+
const match = /^\s*Test Files\s+.*\((\d+)\)\s*$/.exec(stripTurboPrefix(clean) ?? clean);
|
|
832
|
+
return match ? [Number(match[1])] : [];
|
|
833
|
+
});
|
|
834
|
+
return counts.length ? counts.reduce((sum, count) => sum + count, 0) : null;
|
|
835
|
+
}
|
|
836
|
+
export function fileCountDeficit(entry, raw) {
|
|
837
|
+
const actual = runnerFileCount(raw);
|
|
838
|
+
return entry?.fileCount != null && actual !== null && actual < entry.fileCount
|
|
839
|
+
? `infra; runner reported ${actual} test files, below baseline ${entry.fileCount} — suite incomplete`
|
|
840
|
+
: undefined;
|
|
841
|
+
}
|
|
842
|
+
/** Both readers classify fresh evidence first, retaining the whole-output guard against infra forgiveness. */
|
|
843
|
+
export function classifyFreshRunnerOutput(entry, raw, code) {
|
|
844
|
+
if (classifyRunnerOutput(raw, code) === "green-teardown")
|
|
845
|
+
return undefined;
|
|
846
|
+
const { failing } = freshFailures(entry, raw);
|
|
847
|
+
const verdict = classifyRunnerOutput(failing.length ? failing.join("\n") : raw, code);
|
|
848
|
+
return verdict === "infra" || verdict === "regression" ? verdict : undefined;
|
|
849
|
+
}
|
package/dist/gates/llm.js
CHANGED
|
@@ -287,7 +287,7 @@ async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 3000
|
|
|
287
287
|
const pf = join(dir, "prompt.md");
|
|
288
288
|
writeFileSync(pf, prompt);
|
|
289
289
|
const r = await sh(adapter.headlessCommand(pf, model), cwd, timeoutMs);
|
|
290
|
-
return { output: r.stdout + "\n" + r.stderr, exitCode: r.code, timedOut: r.timedOut === true, seatAuthoredBytes: Buffer.byteLength(r.stdout + r.stderr) };
|
|
290
|
+
return { output: r.stdout + "\n" + r.stderr, exitCode: r.code, timedOut: r.timedOut === true, seatAuthoredBytes: Buffer.byteLength((r.stdout + r.stderr).trim()) };
|
|
291
291
|
}
|
|
292
292
|
finally {
|
|
293
293
|
rmSync(dir, { recursive: true, force: true });
|
|
@@ -427,7 +427,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
|
|
|
427
427
|
}
|
|
428
428
|
}
|
|
429
429
|
timedOut = Date.now() - startedAt >= timeoutMs && !new RegExp(exitPattern).test(out);
|
|
430
|
-
if (timedOut)
|
|
430
|
+
if (timedOut && reviewing)
|
|
431
431
|
forceClose = true;
|
|
432
432
|
}
|
|
433
433
|
const exitCode = Number(new RegExp(`TICKMARKR_EXIT_${nonce}:(\\d+)`).exec(out)?.[1]);
|
package/dist/gates/review.js
CHANGED
|
@@ -352,7 +352,12 @@ For every prior material, put its fingerprint in exactly one of resolved (verifi
|
|
|
352
352
|
Approve iff no material finding remains and every prior material is resolved.
|
|
353
353
|
The top-level comments array is optional. Use it only for actionable line-anchored feedback.
|
|
354
354
|
`;
|
|
355
|
-
|
|
355
|
+
// Filenames are journaled (daemon.ts lifts meta.rawPath/briefPath onto the gate-result row), so they
|
|
356
|
+
// must be reproducible from the same inputs — the verdict nonce is cryptographically random and would
|
|
357
|
+
// make two otherwise-identical runs diverge in their journal bytes. The reviewer channel already
|
|
358
|
+
// disambiguates every call that matters: a retry always excludes the flaked channel (run-gates.ts),
|
|
359
|
+
// so it can never collide with the attempt it replaces.
|
|
360
|
+
const artifactId = `${task.id}-${channelKey(reviewer).replace(/[^a-zA-Z0-9_.-]/g, "-")}`;
|
|
356
361
|
const briefPath = artifactDir ? join(artifactDir, `review-brief-${artifactId}.md`) : undefined;
|
|
357
362
|
// Persistence is evidence, not a gate input: a full disk or a removed run dir never fails the gate.
|
|
358
363
|
let savedBrief;
|
|
@@ -378,6 +383,16 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
378
383
|
// (run-20260709-104447 P87-09). The configured ceiling defaults to that measured 15 minutes.
|
|
379
384
|
cfg.review.timeoutMs);
|
|
380
385
|
const raw = llm.output;
|
|
386
|
+
let saved;
|
|
387
|
+
if (artifactDir) {
|
|
388
|
+
try {
|
|
389
|
+
saved = join(artifactDir, `review-raw-${artifactId}.txt`);
|
|
390
|
+
writeFileSync(saved, redactSecrets(raw));
|
|
391
|
+
}
|
|
392
|
+
catch {
|
|
393
|
+
saved = undefined; // persistence is evidence, not a gate input — never fail the gate on it
|
|
394
|
+
}
|
|
395
|
+
}
|
|
381
396
|
const provider = modelProvider(reviewer.model, reviewer.vendor);
|
|
382
397
|
const v = extractVerdictJson(raw, nonce);
|
|
383
398
|
const findings = v && Array.isArray(v.findings) ? v.findings : null;
|
|
@@ -394,16 +409,6 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
394
409
|
const cause = closureInvalid ? "malformed-verdict" : llm.launchNeverStarted ? "launch-never-started"
|
|
395
410
|
: llm.timedOut ? (bytes > 0 ? "truncated" : "silent")
|
|
396
411
|
: classifyVerdictCause(raw, nonce, "approve", llm);
|
|
397
|
-
let saved;
|
|
398
|
-
if (artifactDir) {
|
|
399
|
-
try {
|
|
400
|
-
saved = join(artifactDir, `review-raw-${artifactId}.txt`);
|
|
401
|
-
writeFileSync(saved, redactSecrets(raw));
|
|
402
|
-
}
|
|
403
|
-
catch {
|
|
404
|
-
saved = undefined; // persistence is evidence, not a gate input — never fail the gate on it
|
|
405
|
-
}
|
|
406
|
-
}
|
|
407
412
|
const failure = cause === "malformed-verdict"
|
|
408
413
|
? "review output unparseable"
|
|
409
414
|
: "review dispatch failed — no structurally valid nonce-bound response";
|
|
@@ -455,6 +460,8 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
455
460
|
...structuredFindings("review", details).filter((finding) => !reraised.some((prior) => prior.note === finding.note)),
|
|
456
461
|
...reraised,
|
|
457
462
|
] } : {}),
|
|
463
|
+
...(saved ? { rawPath: saved } : {}),
|
|
464
|
+
...(savedBrief ? { briefPath: savedBrief } : {}),
|
|
458
465
|
},
|
|
459
466
|
};
|
|
460
467
|
}
|
package/dist/run/daemon.d.ts
CHANGED
|
@@ -14,6 +14,8 @@ export interface RunOptions {
|
|
|
14
14
|
graphChanged?: boolean;
|
|
15
15
|
retryFailed?: boolean;
|
|
16
16
|
concurrency?: number;
|
|
17
|
+
/** Bounded wait at a drain caused solely by parked tasks. */
|
|
18
|
+
approvalWindowMs?: number;
|
|
17
19
|
driver?: ExecutorDriver;
|
|
18
20
|
driverOverride?: DriverChoice;
|
|
19
21
|
adapters?: WorkerAdapter[];
|
|
@@ -55,9 +57,8 @@ export interface RunSummary {
|
|
|
55
57
|
* T14, amended by v2.2 T3: approvals the run accepted and never acted on. `approved` above is still
|
|
56
58
|
* built ONCE at startup — replay determinism depends on it — but a live approval is no longer inert:
|
|
57
59
|
* the boundary sweep in the task loop releases what lands while the daemon runs, so an approval
|
|
58
|
-
* written mid-run is
|
|
59
|
-
* fold still
|
|
60
|
-
* run-end sample below — meets no further boundary, so nothing can release it before this run ends.
|
|
60
|
+
* written mid-run is enacted at a boundary, during the approval window, or by cancelling tip verify.
|
|
61
|
+
* This fold still exposes decisions that could not enact, including a failure before dispatch.
|
|
61
62
|
* Without this the run-end record stated only buckets and tipVerify, both accurate, over a milestone
|
|
62
63
|
* that was silently incomplete: run …230 ended tipVerify "passed" with two upheld approvals and zero
|
|
63
64
|
* subsequent dispatches. Scored per task on its NEWEST approval: a later approval is the live
|
|
@@ -108,6 +109,7 @@ export declare const SUITE_WAIT_CEILING_MS = 600000;
|
|
|
108
109
|
export declare const setSuiteWaitCeilingForTests: (ms: number) => void;
|
|
109
110
|
export declare const resetSuiteWaitCeilingForTests: () => void;
|
|
110
111
|
export declare const APPROVAL_POLL_MS = 250;
|
|
112
|
+
export declare const APPROVAL_WINDOW_MS = 1000;
|
|
111
113
|
export declare const EARLY_LAUNCH_LIVENESS_MS = 60000;
|
|
112
114
|
/** Test seam — lowers the empty-pane liveness window without sleeping 60s per case. */
|
|
113
115
|
export declare function setEarlyLaunchLivenessMsForTests(ms: number): void;
|
|
@@ -151,6 +153,7 @@ export declare function commandsHash(commands: Record<string, string>): string;
|
|
|
151
153
|
export declare function verifyIntegrationTipCached(intWt: string, commands: Record<string, string>, journal: Journal, opts?: {
|
|
152
154
|
lastMergedTask?: string;
|
|
153
155
|
baseline?: Baseline;
|
|
156
|
+
signal?: AbortSignal;
|
|
154
157
|
}): Promise<boolean>;
|
|
155
158
|
type SuitePidProbe = (pid: number) => number | undefined;
|
|
156
159
|
/** Count full-suite roots in one process-table snapshot. The probes are arguments so the ownership
|