tickmarkr 2.5.4 → 2.5.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/registry.js +6 -1
- package/dist/adapters/types.d.ts +3 -0
- package/dist/cli/commands/approve.d.ts +1 -0
- package/dist/cli/commands/approve.js +54 -6
- package/dist/cli/commands/doctor.js +44 -29
- package/dist/cli/commands/fleet.js +195 -33
- package/dist/cli/commands/init.js +196 -6
- package/dist/cli/commands/resume.js +4 -2
- package/dist/cli/commands/run.js +11 -2
- package/dist/cli/commands/verify.js +1 -1
- package/dist/cli/help.d.ts +2 -0
- package/dist/cli/help.js +3 -1
- package/dist/config/config.d.ts +14 -8
- package/dist/config/config.js +29 -26
- package/dist/config/fleet-overlay.d.ts +3 -9
- package/dist/config/fleet-overlay.js +68 -11
- package/dist/config/fleet-why.d.ts +7 -0
- package/dist/config/fleet-why.js +5 -0
- package/dist/drivers/index.d.ts +15 -1
- package/dist/drivers/index.js +38 -10
- package/dist/drivers/orca.d.ts +99 -10
- package/dist/drivers/orca.js +586 -97
- package/dist/gates/baseline.d.ts +3 -0
- package/dist/gates/baseline.js +2 -1
- package/dist/gates/cache.d.ts +100 -0
- package/dist/gates/cache.js +389 -0
- package/dist/gates/run-gates.d.ts +3 -0
- package/dist/gates/run-gates.js +117 -10
- package/dist/gates/test-manifest.d.ts +99 -0
- package/dist/gates/test-manifest.js +389 -0
- package/dist/gates/test-reporter.d.ts +4 -0
- package/dist/gates/test-reporter.js +49 -0
- package/dist/route/preference.d.ts +22 -1
- package/dist/route/preference.js +123 -25
- package/dist/route/router.js +31 -6
- package/dist/run/daemon.d.ts +13 -0
- package/dist/run/daemon.js +160 -70
- package/dist/run/git.d.ts +10 -1
- package/dist/run/git.js +43 -9
- package/dist/run/journal.d.ts +14 -2
- package/dist/run/journal.js +78 -10
- package/dist/run/lease.d.ts +14 -0
- package/dist/run/lease.js +87 -0
- package/dist/run/merge.d.ts +2 -0
- package/dist/run/merge.js +91 -3
- package/dist/run/operator-state.d.ts +11 -0
- package/dist/run/operator-state.js +17 -3
- package/dist/tui/cockpit/board.d.ts +96 -0
- package/dist/tui/cockpit/board.js +346 -0
- package/dist/tui/cockpit/decision-actions.js +2 -0
- package/dist/tui/cockpit/layout.d.ts +5 -1
- package/dist/tui/cockpit/layout.js +8 -3
- package/dist/tui/cockpit/live-runtime.js +71 -21
- package/dist/tui/cockpit/run-view.d.ts +7 -5
- package/dist/tui/cockpit/run-view.js +12 -11
- package/dist/tui/ink/fleet-app.d.ts +41 -27
- package/dist/tui/ink/fleet-app.js +204 -31
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +173 -113
package/dist/gates/run-gates.js
CHANGED
|
@@ -6,15 +6,17 @@ import { TIER_RANK } from "../config/config.js";
|
|
|
6
6
|
import { getAdapter } from "../adapters/registry.js";
|
|
7
7
|
import { GATE_NAMES } from "../graph/schema.js";
|
|
8
8
|
import { acceptanceGate } from "./acceptance.js";
|
|
9
|
-
import { compareToBaseline } from "./baseline.js";
|
|
9
|
+
import { compareToBaseline, effectiveCeilingMs } from "./baseline.js";
|
|
10
10
|
import { evidenceGate } from "./evidence.js";
|
|
11
11
|
import { captureLlmOutput } from "./llm.js";
|
|
12
12
|
import { disallowedBy } from "../route/preference.js";
|
|
13
13
|
import { marginalCostRank } from "../route/router.js";
|
|
14
14
|
import { gateReviewerFloor, pickReviewer, reviewGate } from "./review.js";
|
|
15
15
|
import { scopeGate } from "./scope.js";
|
|
16
|
-
import {
|
|
16
|
+
import { evaluateManifestedTest, isVitestTestCommand } from "./test-manifest.js";
|
|
17
|
+
import { shGit, resolvedCapacity } from "../run/git.js";
|
|
17
18
|
import { withJudgeInvocationEvidence } from "../run/journal.js";
|
|
19
|
+
import { computeVerificationIdentity, formatReusedRow, getVerdictStore, isInfraResult, resolveStateDir, reusedIdentity, } from "./cache.js";
|
|
18
20
|
const productionLoadProvider = () => loadavg()[0] ?? 0;
|
|
19
21
|
let loadProvider = productionLoadProvider;
|
|
20
22
|
/** Test seam — inject deterministic load samples; production always reads os.loadavg. */
|
|
@@ -192,9 +194,45 @@ export function testCommandForFiles(testCmd, files) {
|
|
|
192
194
|
const fwd = wrapped && !/\s--\s/.test(testCmd) ? " --" : "";
|
|
193
195
|
return `${testCmd}${fwd} ${files.map(shq).join(" ")}`;
|
|
194
196
|
}
|
|
197
|
+
/** The manifest-report path for a detected vitest test command — never the stdout-count/file-count path. */
|
|
198
|
+
async function runVitestManifestGate(worktree, cmd, baseline, selected, artifactDir) {
|
|
199
|
+
const entry = baseline.commands.test;
|
|
200
|
+
const outcome = await evaluateManifestedTest(cmd, worktree, {
|
|
201
|
+
baselineDurations: entry?.fileDurations,
|
|
202
|
+
longestFile: entry?.longestFile,
|
|
203
|
+
overallCeilingMs: effectiveCeilingMs(entry),
|
|
204
|
+
artifactDir,
|
|
205
|
+
});
|
|
206
|
+
const reportPath = outcome.reportPath;
|
|
207
|
+
return {
|
|
208
|
+
gate: "test",
|
|
209
|
+
pass: outcome.pass,
|
|
210
|
+
details: outcome.details,
|
|
211
|
+
meta: { ...outcome.meta, reportPath, ...(selected ? { selectedTests: [...selected] } : {}) },
|
|
212
|
+
};
|
|
213
|
+
}
|
|
214
|
+
const SIGNAL_EXIT_RE = /\b(?:SIGTERM|SIGKILL|signal\s+(?:9|15)|exit(?:s|ed|\s+code)?\s+(?:137|143))\b/i;
|
|
215
|
+
const FAILURE_IDENTITY_RE = /\b(?:AssertionError|FAIL\s+\S|Tests?\s+\d+\s+failed|expected\s+.+\s+to\s+)\b/i;
|
|
216
|
+
/** D1: apply the daemon's signal-only rider before either battery cache read or write. Its onGate
|
|
217
|
+
* classification happens after persistence, too late to keep a scripted runner's non-verdict out.
|
|
218
|
+
* Keep named failures as work verdicts and preserve details for failure-policy fingerprinting. */
|
|
219
|
+
function classifySignalOnlyTest(g) {
|
|
220
|
+
if (g.gate !== "test" || g.pass || g.meta?.infra === true || !SIGNAL_EXIT_RE.test(g.details))
|
|
221
|
+
return;
|
|
222
|
+
const named = Array.isArray(g.meta?.failingTests) && g.meta.failingTests.length > 0;
|
|
223
|
+
if (named || FAILURE_IDENTITY_RE.test(g.details))
|
|
224
|
+
return;
|
|
225
|
+
g.meta = { ...g.meta, classification: "infra", infra: true, retryable: false, kind: "signal-exit" };
|
|
226
|
+
}
|
|
195
227
|
export async function runGates(task, ctx) {
|
|
196
228
|
const results = [];
|
|
197
229
|
let commits = [];
|
|
230
|
+
const stateDir = ctx.stateDir ?? resolveStateDir(ctx.worktree, ctx.artifactDir);
|
|
231
|
+
const verdictStore = getVerdictStore(stateDir);
|
|
232
|
+
// VC-1: a reused verdict is journaled as its own row (the daemon appends every note by name) so
|
|
233
|
+
// the ledger names the reuse and the identity even where the gate-result row's details must stay
|
|
234
|
+
// the fresh verdict's (see formatReusedRow).
|
|
235
|
+
const noteReuse = (gate, r, id) => ctx.onGate?.({ phase: "note", gate, name: "gate-reused-verdict", payload: { gate, pass: r.pass, details: r.meta?.reusedDetails, ...reusedIdentity(id) }, result: r });
|
|
198
236
|
const shapeGates = ctx.cfg.gates.byShape?.[task.shape];
|
|
199
237
|
const enabled = (g) => task.gates.includes(g) && (g !== "acceptance" && g !== "review" || shapeGates?.[g] !== false);
|
|
200
238
|
const failed = () => results.some((r) => !r.pass);
|
|
@@ -359,7 +397,7 @@ export async function runGates(task, ctx) {
|
|
|
359
397
|
const runBattery = async (commands, selected, gates = toolGates) => {
|
|
360
398
|
if (!gates.length)
|
|
361
399
|
return;
|
|
362
|
-
if (!v185) {
|
|
400
|
+
if (!v185 && !(commands.test && isVitestTestCommand(commands.test, ctx.worktree))) {
|
|
363
401
|
// ponytail: compareToBaseline batches adjacent tools — their starts are emitted at iteration,
|
|
364
402
|
// not at true execution start. They are collectively sub-second (measured), so the debounce
|
|
365
403
|
// suppresses them anyway; split compareToBaseline only if a tool gate ever gets slow.
|
|
@@ -386,25 +424,60 @@ export async function runGates(task, ctx) {
|
|
|
386
424
|
// any later tool before anyone reads its verdict.
|
|
387
425
|
for (const g of gates) {
|
|
388
426
|
await emitStart(g);
|
|
389
|
-
const
|
|
427
|
+
const cmd = commands[g];
|
|
428
|
+
let r;
|
|
429
|
+
let cached = false;
|
|
430
|
+
let identity;
|
|
431
|
+
if (cmd !== undefined) {
|
|
432
|
+
identity = await computeVerificationIdentity({
|
|
433
|
+
worktree: ctx.worktree,
|
|
434
|
+
gate: g,
|
|
435
|
+
scope: ctx.verificationScope,
|
|
436
|
+
command: cmd,
|
|
437
|
+
baseline: ctx.baseline,
|
|
438
|
+
selectedSet: g === "test" ? selected : undefined,
|
|
439
|
+
capacity: resolvedCapacity(),
|
|
440
|
+
});
|
|
441
|
+
const hit = verdictStore.get(identity);
|
|
442
|
+
if (hit)
|
|
443
|
+
classifySignalOnlyTest(hit); // Older entries predate classification at the write seam.
|
|
444
|
+
if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")) {
|
|
445
|
+
r = formatReusedRow(hit, identity);
|
|
446
|
+
cached = true;
|
|
447
|
+
await noteReuse(g, r, identity);
|
|
448
|
+
}
|
|
449
|
+
}
|
|
450
|
+
if (!r) {
|
|
451
|
+
// VL-1: a detected vitest test command is judged by its own invocation-bound report — the
|
|
452
|
+
// stdout-count/file-count path (compareToBaseline's fileCountDeficit) never runs for it. Any
|
|
453
|
+
// other scripted test command keeps today's exit-code contract byte-identically.
|
|
454
|
+
const useManifest = g === "test" && commands.test !== undefined && isVitestTestCommand(commands.test, ctx.worktree);
|
|
455
|
+
r = useManifest
|
|
456
|
+
? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir))
|
|
457
|
+
: (await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], g === "test" && selected ? { selected } : {})))[0];
|
|
458
|
+
}
|
|
390
459
|
// the screen's interval IS the test gate's first interval, so the split needs no second clock
|
|
391
460
|
if (g === "test" && selected)
|
|
392
|
-
selectedDurationMs = spans.get("test")
|
|
461
|
+
selectedDurationMs = spans.get("test")?.durationMs ?? 0;
|
|
393
462
|
// The pre-battery check proves the tree clean ONCE; a command that exits 0 having rewritten a
|
|
394
463
|
// tracked file makes it dirty again, and every gate after it — including the next shell gate,
|
|
395
464
|
// which would then run against bytes HEAD does not hold — inherits that. So re-check after each
|
|
396
465
|
// command, the last one included, and fail the gate whose command did it. (A red command needs
|
|
397
466
|
// no check: it already ends the round, and its own output is the truer verdict.)
|
|
398
|
-
if (r.pass && commands[g]) {
|
|
467
|
+
if (!cached && r.pass && commands[g]) {
|
|
399
468
|
const dirt = await dirtyWorktree();
|
|
400
469
|
if (dirt) {
|
|
401
470
|
await record(dirtyRefusal(g, dirt, commands[g]));
|
|
402
471
|
return;
|
|
403
472
|
}
|
|
404
473
|
}
|
|
474
|
+
if (r)
|
|
475
|
+
classifySignalOnlyTest(r);
|
|
476
|
+
if (!cached && identity && r && !isInfraResult(r)) {
|
|
477
|
+
verdictStore.set(identity, { ...r, meta: { ...r.meta, source: "gate", runDir: ctx.artifactDir } });
|
|
478
|
+
}
|
|
405
479
|
if (g === "test" && selected) {
|
|
406
480
|
const screened = { ...r, meta: { ...r.meta, selectedTests: selected } };
|
|
407
|
-
// green: held (see heldTest) so the full suite below can supersede it with ONE verdict.
|
|
408
481
|
if (!screened.pass)
|
|
409
482
|
await record(screened);
|
|
410
483
|
else {
|
|
@@ -748,9 +821,43 @@ export async function runGates(task, ctx) {
|
|
|
748
821
|
// This is the last shell command a round can run — the judge's named-test oracle (acceptance.ts)
|
|
749
822
|
// may have run one before it, and every gate between the battery and here reads commits only, so
|
|
750
823
|
// a clean tree HERE is what makes "the gated commit is the tested tree" true at merge time.
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
|
|
824
|
+
// VL-1: the merge-candidate's manifest is the FULL set — a full suite whose report lacks one
|
|
825
|
+
// manifest file never reaches the pass branch below, so a selected-only green can never merge.
|
|
826
|
+
let full;
|
|
827
|
+
let cached = false;
|
|
828
|
+
let identity;
|
|
829
|
+
if (ctx.commands.test !== undefined) {
|
|
830
|
+
identity = await computeVerificationIdentity({
|
|
831
|
+
worktree: ctx.worktree,
|
|
832
|
+
gate: "test",
|
|
833
|
+
scope: ctx.verificationScope,
|
|
834
|
+
command: ctx.commands.test,
|
|
835
|
+
baseline: ctx.baseline,
|
|
836
|
+
selectedSet: undefined,
|
|
837
|
+
capacity: resolvedCapacity(),
|
|
838
|
+
});
|
|
839
|
+
const hit = verdictStore.get(identity);
|
|
840
|
+
if (hit)
|
|
841
|
+
classifySignalOnlyTest(hit);
|
|
842
|
+
if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")) {
|
|
843
|
+
full = formatReusedRow(hit, identity);
|
|
844
|
+
cached = true;
|
|
845
|
+
await noteReuse("test", full, identity);
|
|
846
|
+
}
|
|
847
|
+
}
|
|
848
|
+
if (!full) {
|
|
849
|
+
const fullUsesManifest = ctx.commands.test !== undefined && isVitestTestCommand(ctx.commands.test, ctx.worktree);
|
|
850
|
+
full = fullUsesManifest
|
|
851
|
+
? await measure("test", () => runVitestManifestGate(ctx.worktree, ctx.commands.test, ctx.baseline, undefined, ctx.artifactDir))
|
|
852
|
+
: (await measure("test", () => compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"])))[0];
|
|
853
|
+
}
|
|
854
|
+
fullDurationMs = spans.get("test") ? spans.get("test").durationMs - (selectedDurationMs ?? 0) : 0;
|
|
855
|
+
const dirt = (!cached && full.pass) ? await dirtyWorktree() : undefined;
|
|
856
|
+
if (full)
|
|
857
|
+
classifySignalOnlyTest(full);
|
|
858
|
+
if (!cached && identity && full && !dirt && !isInfraResult(full)) {
|
|
859
|
+
verdictStore.set(identity, { ...full, meta: { ...full.meta, source: "gate", runDir: ctx.artifactDir } });
|
|
860
|
+
}
|
|
754
861
|
const merged = withTelemetry(dirt
|
|
755
862
|
? dirtyRefusal("test", dirt, ctx.commands.test)
|
|
756
863
|
: { ...full, meta: { ...full.meta, fullSuite: true, selectedTests: selected } });
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
import type { BaselineFileDuration } from "./baseline.js";
|
|
2
|
+
export declare function isVitestTestCommand(cmd: string, cwd: string): boolean;
|
|
3
|
+
/** One identity for every path this module compares: repo-relative, forward-slash. `vitest list
|
|
4
|
+
* --json` and `TestModule.moduleId` both hand back an absolute filesystem path already resolved
|
|
5
|
+
* through any symlink in it (e.g. macOS's `/var` -> `/private/var`, under which every OS temp dir —
|
|
6
|
+
* and so every test fixture worktree — lives); `cwd` as tickmarkr holds it may not be. Resolving cwd
|
|
7
|
+
* before computing the relative path is what makes the two actually comparable. Manifests built from
|
|
8
|
+
* a selected-test screen are already repo-relative and pass through unchanged. */
|
|
9
|
+
export declare function toManifestPath(file: string, cwd: string): string;
|
|
10
|
+
export interface TestReportCompletion {
|
|
11
|
+
at: number;
|
|
12
|
+
status: "passed" | "failed";
|
|
13
|
+
/** Failure fingerprints for this file; absent/empty on a passed file. */
|
|
14
|
+
failures?: string[];
|
|
15
|
+
}
|
|
16
|
+
/** The runner's own machine report — requested/started/completed are the runner's claims about ITSELF. */
|
|
17
|
+
export interface TestReport {
|
|
18
|
+
nonce: string;
|
|
19
|
+
requested: string[];
|
|
20
|
+
started: Record<string, number>;
|
|
21
|
+
completed: Record<string, TestReportCompletion>;
|
|
22
|
+
/** Files the reporter observed complete MORE than once — `completed`'s object keys cannot show
|
|
23
|
+
* this themselves (a second write silently overwrites the first), so the reporter records the
|
|
24
|
+
* evidence separately before it is lost. */
|
|
25
|
+
duplicateCompletions?: string[];
|
|
26
|
+
/** Written last, once, when the runner reaches its own terminal state. Its absence means the run
|
|
27
|
+
* never certified completion — killed, crashed, or still in flight — and is never a verdict. */
|
|
28
|
+
certificate?: {
|
|
29
|
+
at: number;
|
|
30
|
+
exitCode: number;
|
|
31
|
+
};
|
|
32
|
+
}
|
|
33
|
+
/** Reads and structurally validates the report; a missing or malformed file is `undefined` — never a partial parse. */
|
|
34
|
+
export declare function readTestReport(path: string): TestReport | undefined;
|
|
35
|
+
export type ManifestVerdictKind = "pass" | "infra" | "work" | "fail-closed";
|
|
36
|
+
export interface ManifestVerdict {
|
|
37
|
+
kind: ManifestVerdictKind;
|
|
38
|
+
pass: boolean;
|
|
39
|
+
details: string;
|
|
40
|
+
meta: Record<string, unknown>;
|
|
41
|
+
}
|
|
42
|
+
/**
|
|
43
|
+
* The independent validator: given the manifest THIS invocation was asked to prove, its bound nonce
|
|
44
|
+
* and the independently observed process exit code, decide the
|
|
45
|
+
* verdict from the report alone. `killedFile` short-circuits every report-shaped check — a job this
|
|
46
|
+
* module killed for a per-file hang never reaches its report.
|
|
47
|
+
*/
|
|
48
|
+
export declare function verifyManifestReport(opts: {
|
|
49
|
+
manifest: readonly string[];
|
|
50
|
+
nonce: string;
|
|
51
|
+
exitCode: number | undefined;
|
|
52
|
+
report: TestReport | undefined;
|
|
53
|
+
killedFile?: string;
|
|
54
|
+
hangBudgetMs?: number;
|
|
55
|
+
}): ManifestVerdict;
|
|
56
|
+
/** How much longer than its baseline measurement one file may legitimately run before it is a hang. */
|
|
57
|
+
export declare const FILE_HANG_SLACK = 3;
|
|
58
|
+
export declare const DEFAULT_FILE_HANG_BUDGET_MS = 60000;
|
|
59
|
+
export declare function fileHangBudgetMs(file: string, baselineDurations?: readonly BaselineFileDuration[] | null, ceilingMs?: number, longestFile?: BaselineFileDuration | null): number;
|
|
60
|
+
export interface ManifestRunResult {
|
|
61
|
+
exitCode: number | undefined;
|
|
62
|
+
stdout: string;
|
|
63
|
+
stderr: string;
|
|
64
|
+
report: TestReport | undefined;
|
|
65
|
+
killedFile?: string;
|
|
66
|
+
hangBudgetMs?: number;
|
|
67
|
+
/** The child's own pid (its process GROUP id too, since it is spawned detached) — for a caller
|
|
68
|
+
* that wants to prove the group is really gone after a hang kill (`process.kill(-pid, 0)` throws). */
|
|
69
|
+
pid?: number;
|
|
70
|
+
}
|
|
71
|
+
/** Supervise the configured command and poll the runner's atomic lifecycle snapshots. Every
|
|
72
|
+
* timeout kills the detached process group, including descendants holding the output pipes. */
|
|
73
|
+
export declare function runManifestedTest(cmd: string, cwd: string, opts: {
|
|
74
|
+
manifest: readonly string[];
|
|
75
|
+
nonce: string;
|
|
76
|
+
reportPath: string;
|
|
77
|
+
env?: NodeJS.ProcessEnv;
|
|
78
|
+
baselineDurations?: readonly BaselineFileDuration[] | null;
|
|
79
|
+
longestFile?: BaselineFileDuration | null;
|
|
80
|
+
pollMs?: number;
|
|
81
|
+
overallCeilingMs?: number;
|
|
82
|
+
}): Promise<ManifestRunResult>;
|
|
83
|
+
export interface ManifestGateOutcome {
|
|
84
|
+
pass: boolean;
|
|
85
|
+
kind: ManifestVerdictKind;
|
|
86
|
+
details: string;
|
|
87
|
+
classification?: "infra" | "regression";
|
|
88
|
+
meta: Record<string, unknown>;
|
|
89
|
+
exitCode: number;
|
|
90
|
+
reportPath: string;
|
|
91
|
+
}
|
|
92
|
+
/** One configured runner execution, and its own collection under the same arguments and environment.
|
|
93
|
+
* The installed runner is trusted (R28 add.1 option B); the nonce catches stale artifacts, not forgery. */
|
|
94
|
+
export declare function evaluateManifestedTest(cmd: string, cwd: string, opts: {
|
|
95
|
+
baselineDurations?: readonly BaselineFileDuration[] | null;
|
|
96
|
+
longestFile?: BaselineFileDuration | null;
|
|
97
|
+
overallCeilingMs?: number;
|
|
98
|
+
artifactDir?: string;
|
|
99
|
+
}): Promise<ManifestGateOutcome>;
|