tickmarkr 2.5.4 → 2.5.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/dist/adapters/registry.js +6 -1
  2. package/dist/adapters/types.d.ts +3 -0
  3. package/dist/cli/commands/approve.d.ts +1 -0
  4. package/dist/cli/commands/approve.js +54 -6
  5. package/dist/cli/commands/doctor.js +44 -29
  6. package/dist/cli/commands/fleet.js +195 -33
  7. package/dist/cli/commands/init.js +196 -6
  8. package/dist/cli/commands/resume.js +4 -2
  9. package/dist/cli/commands/run.js +11 -2
  10. package/dist/cli/commands/verify.js +1 -1
  11. package/dist/cli/help.d.ts +2 -0
  12. package/dist/cli/help.js +3 -1
  13. package/dist/config/config.d.ts +14 -8
  14. package/dist/config/config.js +29 -26
  15. package/dist/config/fleet-overlay.d.ts +3 -9
  16. package/dist/config/fleet-overlay.js +68 -11
  17. package/dist/config/fleet-why.d.ts +7 -0
  18. package/dist/config/fleet-why.js +5 -0
  19. package/dist/drivers/index.d.ts +15 -1
  20. package/dist/drivers/index.js +38 -10
  21. package/dist/drivers/orca.d.ts +99 -10
  22. package/dist/drivers/orca.js +586 -97
  23. package/dist/gates/baseline.d.ts +3 -0
  24. package/dist/gates/baseline.js +2 -1
  25. package/dist/gates/cache.d.ts +100 -0
  26. package/dist/gates/cache.js +389 -0
  27. package/dist/gates/run-gates.d.ts +3 -0
  28. package/dist/gates/run-gates.js +117 -10
  29. package/dist/gates/test-manifest.d.ts +99 -0
  30. package/dist/gates/test-manifest.js +389 -0
  31. package/dist/gates/test-reporter.d.ts +4 -0
  32. package/dist/gates/test-reporter.js +49 -0
  33. package/dist/route/preference.d.ts +22 -1
  34. package/dist/route/preference.js +123 -25
  35. package/dist/route/router.js +31 -6
  36. package/dist/run/daemon.d.ts +13 -0
  37. package/dist/run/daemon.js +160 -70
  38. package/dist/run/git.d.ts +10 -1
  39. package/dist/run/git.js +43 -9
  40. package/dist/run/journal.d.ts +14 -2
  41. package/dist/run/journal.js +78 -10
  42. package/dist/run/lease.d.ts +14 -0
  43. package/dist/run/lease.js +87 -0
  44. package/dist/run/merge.d.ts +2 -0
  45. package/dist/run/merge.js +91 -3
  46. package/dist/run/operator-state.d.ts +11 -0
  47. package/dist/run/operator-state.js +17 -3
  48. package/dist/tui/cockpit/board.d.ts +96 -0
  49. package/dist/tui/cockpit/board.js +346 -0
  50. package/dist/tui/cockpit/decision-actions.js +2 -0
  51. package/dist/tui/cockpit/layout.d.ts +5 -1
  52. package/dist/tui/cockpit/layout.js +8 -3
  53. package/dist/tui/cockpit/live-runtime.js +71 -21
  54. package/dist/tui/cockpit/run-view.d.ts +7 -5
  55. package/dist/tui/cockpit/run-view.js +12 -11
  56. package/dist/tui/ink/fleet-app.d.ts +41 -27
  57. package/dist/tui/ink/fleet-app.js +204 -31
  58. package/package.json +1 -1
  59. package/skills/tickmarkr-overseer/SKILL.md +173 -113
@@ -6,15 +6,17 @@ import { TIER_RANK } from "../config/config.js";
6
6
  import { getAdapter } from "../adapters/registry.js";
7
7
  import { GATE_NAMES } from "../graph/schema.js";
8
8
  import { acceptanceGate } from "./acceptance.js";
9
- import { compareToBaseline } from "./baseline.js";
9
+ import { compareToBaseline, effectiveCeilingMs } from "./baseline.js";
10
10
  import { evidenceGate } from "./evidence.js";
11
11
  import { captureLlmOutput } from "./llm.js";
12
12
  import { disallowedBy } from "../route/preference.js";
13
13
  import { marginalCostRank } from "../route/router.js";
14
14
  import { gateReviewerFloor, pickReviewer, reviewGate } from "./review.js";
15
15
  import { scopeGate } from "./scope.js";
16
- import { shGit } from "../run/git.js";
16
+ import { evaluateManifestedTest, isVitestTestCommand } from "./test-manifest.js";
17
+ import { shGit, resolvedCapacity } from "../run/git.js";
17
18
  import { withJudgeInvocationEvidence } from "../run/journal.js";
19
+ import { computeVerificationIdentity, formatReusedRow, getVerdictStore, isInfraResult, resolveStateDir, reusedIdentity, } from "./cache.js";
18
20
  const productionLoadProvider = () => loadavg()[0] ?? 0;
19
21
  let loadProvider = productionLoadProvider;
20
22
  /** Test seam — inject deterministic load samples; production always reads os.loadavg. */
@@ -192,9 +194,45 @@ export function testCommandForFiles(testCmd, files) {
192
194
  const fwd = wrapped && !/\s--\s/.test(testCmd) ? " --" : "";
193
195
  return `${testCmd}${fwd} ${files.map(shq).join(" ")}`;
194
196
  }
197
+ /** The manifest-report path for a detected vitest test command — never the stdout-count/file-count path. */
198
+ async function runVitestManifestGate(worktree, cmd, baseline, selected, artifactDir) {
199
+ const entry = baseline.commands.test;
200
+ const outcome = await evaluateManifestedTest(cmd, worktree, {
201
+ baselineDurations: entry?.fileDurations,
202
+ longestFile: entry?.longestFile,
203
+ overallCeilingMs: effectiveCeilingMs(entry),
204
+ artifactDir,
205
+ });
206
+ const reportPath = outcome.reportPath;
207
+ return {
208
+ gate: "test",
209
+ pass: outcome.pass,
210
+ details: outcome.details,
211
+ meta: { ...outcome.meta, reportPath, ...(selected ? { selectedTests: [...selected] } : {}) },
212
+ };
213
+ }
214
+ const SIGNAL_EXIT_RE = /\b(?:SIGTERM|SIGKILL|signal\s+(?:9|15)|exit(?:s|ed|\s+code)?\s+(?:137|143))\b/i;
215
+ const FAILURE_IDENTITY_RE = /\b(?:AssertionError|FAIL\s+\S|Tests?\s+\d+\s+failed|expected\s+.+\s+to\s+)\b/i;
216
+ /** D1: apply the daemon's signal-only rider before either battery cache read or write. Its onGate
217
+ * classification happens after persistence, too late to keep a scripted runner's non-verdict out.
218
+ * Keep named failures as work verdicts and preserve details for failure-policy fingerprinting. */
219
+ function classifySignalOnlyTest(g) {
220
+ if (g.gate !== "test" || g.pass || g.meta?.infra === true || !SIGNAL_EXIT_RE.test(g.details))
221
+ return;
222
+ const named = Array.isArray(g.meta?.failingTests) && g.meta.failingTests.length > 0;
223
+ if (named || FAILURE_IDENTITY_RE.test(g.details))
224
+ return;
225
+ g.meta = { ...g.meta, classification: "infra", infra: true, retryable: false, kind: "signal-exit" };
226
+ }
195
227
  export async function runGates(task, ctx) {
196
228
  const results = [];
197
229
  let commits = [];
230
+ const stateDir = ctx.stateDir ?? resolveStateDir(ctx.worktree, ctx.artifactDir);
231
+ const verdictStore = getVerdictStore(stateDir);
232
+ // VC-1: a reused verdict is journaled as its own row (the daemon appends every note by name) so
233
+ // the ledger names the reuse and the identity even where the gate-result row's details must stay
234
+ // the fresh verdict's (see formatReusedRow).
235
+ const noteReuse = (gate, r, id) => ctx.onGate?.({ phase: "note", gate, name: "gate-reused-verdict", payload: { gate, pass: r.pass, details: r.meta?.reusedDetails, ...reusedIdentity(id) }, result: r });
198
236
  const shapeGates = ctx.cfg.gates.byShape?.[task.shape];
199
237
  const enabled = (g) => task.gates.includes(g) && (g !== "acceptance" && g !== "review" || shapeGates?.[g] !== false);
200
238
  const failed = () => results.some((r) => !r.pass);
@@ -359,7 +397,7 @@ export async function runGates(task, ctx) {
359
397
  const runBattery = async (commands, selected, gates = toolGates) => {
360
398
  if (!gates.length)
361
399
  return;
362
- if (!v185) {
400
+ if (!v185 && !(commands.test && isVitestTestCommand(commands.test, ctx.worktree))) {
363
401
  // ponytail: compareToBaseline batches adjacent tools — their starts are emitted at iteration,
364
402
  // not at true execution start. They are collectively sub-second (measured), so the debounce
365
403
  // suppresses them anyway; split compareToBaseline only if a tool gate ever gets slow.
@@ -386,25 +424,60 @@ export async function runGates(task, ctx) {
386
424
  // any later tool before anyone reads its verdict.
387
425
  for (const g of gates) {
388
426
  await emitStart(g);
389
- const [r] = await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], g === "test" && selected ? { selected } : {}));
427
+ const cmd = commands[g];
428
+ let r;
429
+ let cached = false;
430
+ let identity;
431
+ if (cmd !== undefined) {
432
+ identity = await computeVerificationIdentity({
433
+ worktree: ctx.worktree,
434
+ gate: g,
435
+ scope: ctx.verificationScope,
436
+ command: cmd,
437
+ baseline: ctx.baseline,
438
+ selectedSet: g === "test" ? selected : undefined,
439
+ capacity: resolvedCapacity(),
440
+ });
441
+ const hit = verdictStore.get(identity);
442
+ if (hit)
443
+ classifySignalOnlyTest(hit); // Older entries predate classification at the write seam.
444
+ if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")) {
445
+ r = formatReusedRow(hit, identity);
446
+ cached = true;
447
+ await noteReuse(g, r, identity);
448
+ }
449
+ }
450
+ if (!r) {
451
+ // VL-1: a detected vitest test command is judged by its own invocation-bound report — the
452
+ // stdout-count/file-count path (compareToBaseline's fileCountDeficit) never runs for it. Any
453
+ // other scripted test command keeps today's exit-code contract byte-identically.
454
+ const useManifest = g === "test" && commands.test !== undefined && isVitestTestCommand(commands.test, ctx.worktree);
455
+ r = useManifest
456
+ ? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir))
457
+ : (await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], g === "test" && selected ? { selected } : {})))[0];
458
+ }
390
459
  // the screen's interval IS the test gate's first interval, so the split needs no second clock
391
460
  if (g === "test" && selected)
392
- selectedDurationMs = spans.get("test").durationMs;
461
+ selectedDurationMs = spans.get("test")?.durationMs ?? 0;
393
462
  // The pre-battery check proves the tree clean ONCE; a command that exits 0 having rewritten a
394
463
  // tracked file makes it dirty again, and every gate after it — including the next shell gate,
395
464
  // which would then run against bytes HEAD does not hold — inherits that. So re-check after each
396
465
  // command, the last one included, and fail the gate whose command did it. (A red command needs
397
466
  // no check: it already ends the round, and its own output is the truer verdict.)
398
- if (r.pass && commands[g]) {
467
+ if (!cached && r.pass && commands[g]) {
399
468
  const dirt = await dirtyWorktree();
400
469
  if (dirt) {
401
470
  await record(dirtyRefusal(g, dirt, commands[g]));
402
471
  return;
403
472
  }
404
473
  }
474
+ if (r)
475
+ classifySignalOnlyTest(r);
476
+ if (!cached && identity && r && !isInfraResult(r)) {
477
+ verdictStore.set(identity, { ...r, meta: { ...r.meta, source: "gate", runDir: ctx.artifactDir } });
478
+ }
405
479
  if (g === "test" && selected) {
406
480
  const screened = { ...r, meta: { ...r.meta, selectedTests: selected } };
407
- // green: held (see heldTest) so the full suite below can supersede it with ONE verdict.
408
481
  if (!screened.pass)
409
482
  await record(screened);
410
483
  else {
@@ -748,9 +821,43 @@ export async function runGates(task, ctx) {
748
821
  // This is the last shell command a round can run — the judge's named-test oracle (acceptance.ts)
749
822
  // may have run one before it, and every gate between the battery and here reads commits only, so
750
823
  // a clean tree HERE is what makes "the gated commit is the tested tree" true at merge time.
751
- const [full] = await measure("test", () => compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"]));
752
- fullDurationMs = spans.get("test").durationMs - (selectedDurationMs ?? 0);
753
- const dirt = full.pass ? await dirtyWorktree() : undefined;
824
+ // VL-1: the merge-candidate's manifest is the FULL set — a full suite whose report lacks one
825
+ // manifest file never reaches the pass branch below, so a selected-only green can never merge.
826
+ let full;
827
+ let cached = false;
828
+ let identity;
829
+ if (ctx.commands.test !== undefined) {
830
+ identity = await computeVerificationIdentity({
831
+ worktree: ctx.worktree,
832
+ gate: "test",
833
+ scope: ctx.verificationScope,
834
+ command: ctx.commands.test,
835
+ baseline: ctx.baseline,
836
+ selectedSet: undefined,
837
+ capacity: resolvedCapacity(),
838
+ });
839
+ const hit = verdictStore.get(identity);
840
+ if (hit)
841
+ classifySignalOnlyTest(hit);
842
+ if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")) {
843
+ full = formatReusedRow(hit, identity);
844
+ cached = true;
845
+ await noteReuse("test", full, identity);
846
+ }
847
+ }
848
+ if (!full) {
849
+ const fullUsesManifest = ctx.commands.test !== undefined && isVitestTestCommand(ctx.commands.test, ctx.worktree);
850
+ full = fullUsesManifest
851
+ ? await measure("test", () => runVitestManifestGate(ctx.worktree, ctx.commands.test, ctx.baseline, undefined, ctx.artifactDir))
852
+ : (await measure("test", () => compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"])))[0];
853
+ }
854
+ fullDurationMs = spans.get("test") ? spans.get("test").durationMs - (selectedDurationMs ?? 0) : 0;
855
+ const dirt = (!cached && full.pass) ? await dirtyWorktree() : undefined;
856
+ if (full)
857
+ classifySignalOnlyTest(full);
858
+ if (!cached && identity && full && !dirt && !isInfraResult(full)) {
859
+ verdictStore.set(identity, { ...full, meta: { ...full.meta, source: "gate", runDir: ctx.artifactDir } });
860
+ }
754
861
  const merged = withTelemetry(dirt
755
862
  ? dirtyRefusal("test", dirt, ctx.commands.test)
756
863
  : { ...full, meta: { ...full.meta, fullSuite: true, selectedTests: selected } });
@@ -0,0 +1,99 @@
1
+ import type { BaselineFileDuration } from "./baseline.js";
2
+ export declare function isVitestTestCommand(cmd: string, cwd: string): boolean;
3
+ /** One identity for every path this module compares: repo-relative, forward-slash. `vitest list
4
+ * --json` and `TestModule.moduleId` both hand back an absolute filesystem path already resolved
5
+ * through any symlink in it (e.g. macOS's `/var` -> `/private/var`, under which every OS temp dir —
6
+ * and so every test fixture worktree — lives); `cwd` as tickmarkr holds it may not be. Resolving cwd
7
+ * before computing the relative path is what makes the two actually comparable. Manifests built from
8
+ * a selected-test screen are already repo-relative and pass through unchanged. */
9
+ export declare function toManifestPath(file: string, cwd: string): string;
10
+ export interface TestReportCompletion {
11
+ at: number;
12
+ status: "passed" | "failed";
13
+ /** Failure fingerprints for this file; absent/empty on a passed file. */
14
+ failures?: string[];
15
+ }
16
+ /** The runner's own machine report — requested/started/completed are the runner's claims about ITSELF. */
17
+ export interface TestReport {
18
+ nonce: string;
19
+ requested: string[];
20
+ started: Record<string, number>;
21
+ completed: Record<string, TestReportCompletion>;
22
+ /** Files the reporter observed complete MORE than once — `completed`'s object keys cannot show
23
+ * this themselves (a second write silently overwrites the first), so the reporter records the
24
+ * evidence separately before it is lost. */
25
+ duplicateCompletions?: string[];
26
+ /** Written last, once, when the runner reaches its own terminal state. Its absence means the run
27
+ * never certified completion — killed, crashed, or still in flight — and is never a verdict. */
28
+ certificate?: {
29
+ at: number;
30
+ exitCode: number;
31
+ };
32
+ }
33
+ /** Reads and structurally validates the report; a missing or malformed file is `undefined` — never a partial parse. */
34
+ export declare function readTestReport(path: string): TestReport | undefined;
35
+ export type ManifestVerdictKind = "pass" | "infra" | "work" | "fail-closed";
36
+ export interface ManifestVerdict {
37
+ kind: ManifestVerdictKind;
38
+ pass: boolean;
39
+ details: string;
40
+ meta: Record<string, unknown>;
41
+ }
42
+ /**
43
+ * The independent validator: given the manifest THIS invocation was asked to prove, its bound nonce
44
+ * and the independently observed process exit code, decide the
45
+ * verdict from the report alone. `killedFile` short-circuits every report-shaped check — a job this
46
+ * module killed for a per-file hang never reaches its report.
47
+ */
48
+ export declare function verifyManifestReport(opts: {
49
+ manifest: readonly string[];
50
+ nonce: string;
51
+ exitCode: number | undefined;
52
+ report: TestReport | undefined;
53
+ killedFile?: string;
54
+ hangBudgetMs?: number;
55
+ }): ManifestVerdict;
56
+ /** How much longer than its baseline measurement one file may legitimately run before it is a hang. */
57
+ export declare const FILE_HANG_SLACK = 3;
58
+ export declare const DEFAULT_FILE_HANG_BUDGET_MS = 60000;
59
+ export declare function fileHangBudgetMs(file: string, baselineDurations?: readonly BaselineFileDuration[] | null, ceilingMs?: number, longestFile?: BaselineFileDuration | null): number;
60
+ export interface ManifestRunResult {
61
+ exitCode: number | undefined;
62
+ stdout: string;
63
+ stderr: string;
64
+ report: TestReport | undefined;
65
+ killedFile?: string;
66
+ hangBudgetMs?: number;
67
+ /** The child's own pid (its process GROUP id too, since it is spawned detached) — for a caller
68
+ * that wants to prove the group is really gone after a hang kill (`process.kill(-pid, 0)` throws). */
69
+ pid?: number;
70
+ }
71
+ /** Supervise the configured command and poll the runner's atomic lifecycle snapshots. Every
72
+ * timeout kills the detached process group, including descendants holding the output pipes. */
73
+ export declare function runManifestedTest(cmd: string, cwd: string, opts: {
74
+ manifest: readonly string[];
75
+ nonce: string;
76
+ reportPath: string;
77
+ env?: NodeJS.ProcessEnv;
78
+ baselineDurations?: readonly BaselineFileDuration[] | null;
79
+ longestFile?: BaselineFileDuration | null;
80
+ pollMs?: number;
81
+ overallCeilingMs?: number;
82
+ }): Promise<ManifestRunResult>;
83
+ export interface ManifestGateOutcome {
84
+ pass: boolean;
85
+ kind: ManifestVerdictKind;
86
+ details: string;
87
+ classification?: "infra" | "regression";
88
+ meta: Record<string, unknown>;
89
+ exitCode: number;
90
+ reportPath: string;
91
+ }
92
+ /** One configured runner execution, and its own collection under the same arguments and environment.
93
+ * The installed runner is trusted (R28 add.1 option B); the nonce catches stale artifacts, not forgery. */
94
+ export declare function evaluateManifestedTest(cmd: string, cwd: string, opts: {
95
+ baselineDurations?: readonly BaselineFileDuration[] | null;
96
+ longestFile?: BaselineFileDuration | null;
97
+ overallCeilingMs?: number;
98
+ artifactDir?: string;
99
+ }): Promise<ManifestGateOutcome>;