tickmarkr 2.5.5 → 2.5.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/README.md +3 -1
  2. package/dist/adapters/prompt.js +21 -1
  3. package/dist/cli/commands/approve.js +69 -8
  4. package/dist/cli/commands/doctor.d.ts +10 -0
  5. package/dist/cli/commands/fleet.js +4 -0
  6. package/dist/cli/commands/plan.js +20 -4
  7. package/dist/cli/commands/status.js +95 -34
  8. package/dist/cli/commands/verify.js +108 -85
  9. package/dist/config/config.d.ts +11 -0
  10. package/dist/config/config.js +21 -12
  11. package/dist/config/fleet-overlay.js +55 -17
  12. package/dist/drivers/index.js +2 -1
  13. package/dist/drivers/orca.d.ts +21 -1
  14. package/dist/drivers/orca.js +209 -27
  15. package/dist/gates/baseline.d.ts +21 -5
  16. package/dist/gates/baseline.js +67 -17
  17. package/dist/gates/cache.d.ts +14 -13
  18. package/dist/gates/cache.js +17 -5
  19. package/dist/gates/llm.d.ts +3 -0
  20. package/dist/gates/llm.js +11 -0
  21. package/dist/gates/review.d.ts +28 -3
  22. package/dist/gates/review.js +118 -15
  23. package/dist/gates/run-gates.d.ts +10 -2
  24. package/dist/gates/run-gates.js +362 -108
  25. package/dist/gates/test-manifest.d.ts +33 -1
  26. package/dist/gates/test-manifest.js +132 -40
  27. package/dist/gates/test-reporter.js +20 -7
  28. package/dist/graph/graph.d.ts +4 -0
  29. package/dist/graph/graph.js +50 -1
  30. package/dist/run/activity.d.ts +28 -0
  31. package/dist/run/activity.js +194 -0
  32. package/dist/run/consult.js +5 -4
  33. package/dist/run/daemon.d.ts +11 -0
  34. package/dist/run/daemon.js +663 -128
  35. package/dist/run/execution-budget.d.ts +25 -0
  36. package/dist/run/execution-budget.js +142 -0
  37. package/dist/run/git.d.ts +46 -1
  38. package/dist/run/git.js +149 -12
  39. package/dist/run/journal.d.ts +23 -5
  40. package/dist/run/journal.js +98 -25
  41. package/dist/run/lease.d.ts +44 -0
  42. package/dist/run/lease.js +226 -3
  43. package/dist/run/operator-page-summary.d.ts +56 -0
  44. package/dist/run/operator-page-summary.js +68 -0
  45. package/dist/run/operator-summary.d.ts +69 -0
  46. package/dist/run/operator-summary.js +77 -0
  47. package/dist/run/protocol.d.ts +71 -0
  48. package/dist/run/protocol.js +32 -0
  49. package/dist/run/recovery.d.ts +8 -0
  50. package/dist/run/recovery.js +25 -0
  51. package/dist/run/repair-selection.d.ts +12 -0
  52. package/dist/run/repair-selection.js +56 -0
  53. package/dist/run/stall.d.ts +6 -1
  54. package/dist/run/stall.js +60 -3
  55. package/dist/tui/cockpit/board.d.ts +9 -0
  56. package/dist/tui/cockpit/board.js +10 -0
  57. package/dist/tui/cockpit/derive.d.ts +35 -0
  58. package/dist/tui/cockpit/derive.js +152 -10
  59. package/dist/tui/cockpit/evidence-view.d.ts +2 -0
  60. package/dist/tui/cockpit/evidence-view.js +42 -12
  61. package/dist/tui/cockpit/run-cockpit.d.ts +8 -1
  62. package/dist/tui/cockpit/run-cockpit.js +95 -1
  63. package/dist/tui/cockpit/run-view.d.ts +37 -0
  64. package/dist/tui/cockpit/run-view.js +189 -2
  65. package/dist/tui/ink/fleet-app.d.ts +10 -2
  66. package/dist/tui/ink/fleet-app.js +33 -15
  67. package/package.json +2 -2
  68. package/skills/tickmarkr-overseer/SKILL.md +55 -3
  69. package/skills/tickmarkr-overseer/scripts/grade-ci.sh +29 -2
@@ -9,9 +9,17 @@ export declare function isVitestTestCommand(cmd: string, cwd: string): boolean;
9
9
  export declare function toManifestPath(file: string, cwd: string): string;
10
10
  export interface TestReportCompletion {
11
11
  at: number;
12
- status: "passed" | "failed";
12
+ /** R41: `skipped` = the module ran its lifecycle but executed NO test body (every test skipped);
13
+ * it is present in the manifest accounting and never counted as executed success. */
14
+ status: "passed" | "failed" | "skipped";
13
15
  /** Failure fingerprints for this file; absent/empty on a passed file. */
14
16
  failures?: string[];
17
+ /** Per-module test-body counts the reporter observed (absent on reports from older reporters). */
18
+ tests?: {
19
+ passed: number;
20
+ failed: number;
21
+ skipped: number;
22
+ };
15
23
  }
16
24
  /** The runner's own machine report — requested/started/completed are the runner's claims about ITSELF. */
17
25
  export interface TestReport {
@@ -28,6 +36,9 @@ export interface TestReport {
28
36
  certificate?: {
29
37
  at: number;
30
38
  exitCode: number;
39
+ /** Unhandled and module collection errors observed by the reporter; absent on older reports. */
40
+ errors?: number;
41
+ diagnostics?: string[];
31
42
  };
32
43
  }
33
44
  /** Reads and structurally validates the report; a missing or malformed file is `undefined` — never a partial parse. */
@@ -89,6 +100,27 @@ export interface ManifestGateOutcome {
89
100
  exitCode: number;
90
101
  reportPath: string;
91
102
  }
103
+ export interface DiscoveredManifest {
104
+ files: string[];
105
+ listing: string;
106
+ separator: string;
107
+ listingExit: number | undefined;
108
+ listingStdout: string;
109
+ }
110
+ /** OBS-1044: THE discovery seam — the files the runner would collect under this exact invocation,
111
+ * from its own listing and never from a stdout summary. The gate and the baseline capture share it,
112
+ * so the count a capture records is the count the gate later asks the report to certify. Throws
113
+ * when the runner cannot list: a count that is not the runner's own is not a count. */
114
+ export declare function discoverTestManifest(cmd: string, cwd: string, opts: {
115
+ dir: string;
116
+ nonce: string;
117
+ env: NodeJS.ProcessEnv;
118
+ overallCeilingMs?: number;
119
+ }): Promise<DiscoveredManifest>;
120
+ /** The baseline capture's reading of the same seam: the manifest's file count, or null when the
121
+ * runner cannot list (a null compares nothing — it never manufactures a deficit). The suite is not
122
+ * run here; the capture already ran it once. */
123
+ export declare function manifestFileCount(cmd: string, cwd: string): Promise<number | null>;
92
124
  /** One configured runner execution, and its own collection under the same arguments and environment.
93
125
  * The installed runner is trusted (R28 add.1 option B); the nonce catches stale artifacts, not forgery. */
94
126
  export declare function evaluateManifestedTest(cmd: string, cwd: string, opts: {
@@ -1,10 +1,10 @@
1
- import { randomBytes } from "node:crypto";
1
+ import { createHash, randomBytes } from "node:crypto";
2
2
  import { existsSync, mkdtempSync, readFileSync, realpathSync, writeFileSync } from "node:fs";
3
3
  import { tmpdir } from "node:os";
4
4
  import { isAbsolute, join, relative, sep } from "node:path";
5
5
  import { TEST_REPORTER_SOURCE } from "./test-reporter.js";
6
6
  import { shq } from "../adapters/types.js";
7
- import { FORK_CAP_ENV, ROUTING_ENV_SEAMS, SUITE_PARENT_ENV, shell, resolvedCapacity } from "../run/git.js";
7
+ import { FORK_CAP_ENV, ROUTING_ENV_SEAMS, SUITE_PARENT_ENV, shell, resolvedCapacity, verificationProtocol } from "../run/git.js";
8
8
  /**
9
9
  * VL-1 (OBS-985 lineage): a test gate's completion must be the runner's OWN report, never a stdout
10
10
  * count. `fileCountDeficit` (baseline.ts) reads a summary LINE — a selected screen's smaller count
@@ -30,7 +30,12 @@ function scriptInvocation(cmd, cwd) {
30
30
  manager++;
31
31
  if (!["npm", "pnpm", "yarn"].includes(argv[manager]))
32
32
  return undefined;
33
- const offset = manager + (["run", "run-script"].includes(argv[manager + 1]) ? 2 : 1);
33
+ let offset = manager + (["run", "run-script"].includes(argv[manager + 1]) ? 2 : 1);
34
+ // OBS-1044: `npm run -s test` — the very spelling detectGateCommands synthesizes — carries the
35
+ // manager's own flags before the script name; a seam blind to them routed every default-configured
36
+ // npm repository past the manifest. Skip flags, never a `--` (which ends the manager's arguments).
37
+ while (offset > manager + 1 && /^-(?!-$)/.test(argv[offset] ?? ""))
38
+ offset++;
34
39
  const name = ["t", "tst"].includes(argv[offset]) ? "test" : argv[offset];
35
40
  try {
36
41
  const body = JSON.parse(readFileSync(join(cwd, "package.json"), "utf8")).scripts?.[name];
@@ -72,8 +77,13 @@ function runnerInvocation(cmd, cwd) {
72
77
  if (forwarded[0] === "run")
73
78
  forwarded.shift();
74
79
  const collection = forwarded.filter((a) => a !== "--run" && a !== "--watch" && a !== "--");
80
+ // R41: `--filesOnly` lists FILE SPECIFICATIONS — the unit the reporter certifies (module start/end) —
81
+ // instead of collected test cases, from which the ordinary listing omits every skipped test and so
82
+ // drops a module whose whole body is skipped (describe.skipIf): the runner then reported two files
83
+ // the manifest never expected (T4 236 vs 238, T8 288 vs 290 on run 91). Filters and the environment
84
+ // are forwarded exactly as before; only the unit enumerated changes.
75
85
  return {
76
- listing: [...env, ...prefix.map(shq), shq(binary), "list", ...collection, "--json"].join(" "),
86
+ listing: [...env, ...prefix.map(shq), shq(binary), "list", ...collection, "--filesOnly", "--json"].join(" "),
77
87
  separator: script?.npm && !/(?:^|\s)--(?:\s|$)/.test(cmd) ? " --" : "",
78
88
  };
79
89
  }
@@ -107,8 +117,15 @@ function isTestReportShape(v) {
107
117
  if (typeof c !== "object" || c === null)
108
118
  return false;
109
119
  const cc = c;
110
- if (typeof cc.at !== "number" || (cc.status !== "passed" && cc.status !== "failed"))
120
+ if (typeof cc.at !== "number" || (cc.status !== "passed" && cc.status !== "failed" && cc.status !== "skipped"))
111
121
  return false;
122
+ if (cc.tests !== undefined) {
123
+ if (typeof cc.tests !== "object" || cc.tests === null)
124
+ return false;
125
+ const t = cc.tests;
126
+ if (!["passed", "failed", "skipped"].every((k) => typeof t[k] === "number"))
127
+ return false;
128
+ }
112
129
  }
113
130
  if (r.duplicateCompletions !== undefined) {
114
131
  if (!Array.isArray(r.duplicateCompletions) || !r.duplicateCompletions.every((f) => typeof f === "string"))
@@ -208,12 +225,13 @@ export function verifyManifestReport(opts) {
208
225
  meta: { classification: "infra", infra: true, missingCertificate: true, ...(pending !== undefined ? { file: pending } : {}) },
209
226
  };
210
227
  }
211
- const failures = Object.values(report.completed).filter((c) => c.status === "failed")
212
- .flatMap((c) => c.failures?.length ? c.failures : ["<unnamed failure>"]);
228
+ const failedCompletions = Object.entries(report.completed).filter(([, c]) => c.status === "failed");
229
+ const failingFiles = failedCompletions.map(([file]) => file).sort();
230
+ const failures = failedCompletions.flatMap(([, c]) => c.failures?.length ? c.failures : ["<unnamed failure>"]);
213
231
  if (failures.length)
214
232
  return { kind: "work", pass: false,
215
233
  details: `test report names failing fingerprint(s):\n${failures.join("\n")}`,
216
- meta: { classification: "regression", failingTests: failures, processExit: exitCode } };
234
+ meta: { classification: "regression", failingTests: failures, failingFiles, processExit: exitCode } };
217
235
  if (exitCode === undefined) {
218
236
  return {
219
237
  kind: "fail-closed",
@@ -264,11 +282,25 @@ export function verifyManifestReport(opts) {
264
282
  meta: { classification: "infra", infra: true, file: missing, missingFromReport: true },
265
283
  };
266
284
  }
285
+ // R41: a skipped whole module is PRESENT (the lifecycle certified it) but is not executed test-body
286
+ // success. The verdict says how many modules executed and names every one that did not; a report
287
+ // in which nothing executed proves nothing about the tree and is never a passing verdict.
288
+ const skippedModules = manifest.filter((f) => report.completed[f]?.status === "skipped").sort();
289
+ const executedModules = manifest.length - skippedModules.length;
290
+ if (executedModules === 0) {
291
+ return {
292
+ kind: "fail-closed",
293
+ pass: false,
294
+ details: `every manifest module was skipped whole (${skippedModules.join(", ")}) — no test body executed, so this report certifies nothing about the tree; failing closed`,
295
+ meta: { classification: "infra", infra: true, noExecutedModules: true, manifestFiles: manifest.length, executedModules, skippedModules },
296
+ };
297
+ }
267
298
  return {
268
299
  kind: "pass",
269
300
  pass: true,
270
- details: `invocation-bound test report succeeded — ${manifest.length} manifest file(s) present exactly once`,
271
- meta: { manifestFiles: manifest.length },
301
+ details: `invocation-bound test report succeeded — ${manifest.length} manifest file(s) present exactly once; ${executedModules} executed`
302
+ + (skippedModules.length ? `, ${skippedModules.length} skipped whole (${skippedModules.join(", ")})` : ""),
303
+ meta: { manifestFiles: manifest.length, executedModules, skippedModules },
272
304
  };
273
305
  }
274
306
  /** How much longer than its baseline measurement one file may legitimately run before it is a hang. */
@@ -328,42 +360,83 @@ export function runManifestedTest(cmd, cwd, opts) {
328
360
  report: readTestReport(opts.reportPath), killedFile, hangBudgetMs, pid,
329
361
  })).finally(() => { clearInterval(poll); });
330
362
  }
331
- /** One configured runner execution, and its own collection under the same arguments and environment.
332
- * The installed runner is trusted (R28 add.1 option B); the nonce catches stale artifacts, not forgery. */
333
- export async function evaluateManifestedTest(cmd, cwd, opts) {
334
- const dir = opts.artifactDir ?? mkdtempSync(join(tmpdir(), "tickmarkr-test-report-"));
335
- const nonce = randomBytes(16).toString("hex");
336
- const reportPath = join(dir, `test-manifest-report-${nonce}.json`);
337
- const reporterPath = join(dir, `test-reporter-${nonce}.mjs`);
338
- let spawnedCommand = cmd;
363
+ /** The child environment every manifest invocation (listing and run) receives, and the lifecycle
364
+ * protocol it records. R41: the policy is whatever THIS process was launched with (npm reads
365
+ * `npm_config_ignore_scripts` from the environment over every npmrc); the verdict records it so a
366
+ * verdict measured with `pretest` hooks is never compared to one without. */
367
+ function manifestEnvironment(cwd) {
339
368
  const env = { ...process.env, PATH: `${join(cwd, "node_modules/.bin")}:${process.env.PATH ?? ""}`,
340
369
  [FORK_CAP_ENV]: String(resolvedCapacity().forkCap), [SUITE_PARENT_ENV]: String(process.pid) };
370
+ const verification = verificationProtocol(env, cwd);
341
371
  for (const key of ROUTING_ENV_SEAMS)
342
372
  delete env[key];
343
373
  // A gate can itself be tested under Vitest. Do not give the child the outer worker identity.
344
374
  for (const key of Object.keys(env))
345
375
  if (["VITEST", "TEST", "VITEST_WORKER_ID", "VITEST_POOL_ID"].includes(key))
346
376
  delete env[key];
377
+ return { env, verification };
378
+ }
379
+ /** OBS-1044: THE discovery seam — the files the runner would collect under this exact invocation,
380
+ * from its own listing and never from a stdout summary. The gate and the baseline capture share it,
381
+ * so the count a capture records is the count the gate later asks the report to certify. Throws
382
+ * when the runner cannot list: a count that is not the runner's own is not a count. */
383
+ export async function discoverTestManifest(cmd, cwd, opts) {
384
+ const invocation = runnerInvocation(cmd, cwd);
385
+ const listed = await runManifestedTest(invocation.listing, cwd, {
386
+ manifest: [], nonce: opts.nonce, reportPath: join(opts.dir, `listing-${opts.nonce}.json`), env: opts.env,
387
+ overallCeilingMs: opts.overallCeilingMs ?? DEFAULT_FILE_HANG_BUDGET_MS,
388
+ });
389
+ if (listed.exitCode !== 0)
390
+ throw new Error(`vitest cannot list files (exit ${listed.exitCode ?? "signal"}): ${listed.stderr || listed.stdout}`);
391
+ let files;
347
392
  try {
348
- const invocation = runnerInvocation(cmd, cwd);
349
- const listed = await runManifestedTest(invocation.listing, cwd, {
350
- manifest: [], nonce, reportPath: join(dir, `listing-${nonce}.json`), env,
351
- overallCeilingMs: opts.overallCeilingMs ?? DEFAULT_FILE_HANG_BUDGET_MS,
352
- });
353
- if (listed.exitCode !== 0)
354
- throw new Error(`vitest cannot list files (exit ${listed.exitCode ?? "signal"}): ${listed.stderr || listed.stdout}`);
355
- let files;
356
- try {
357
- const rows = JSON.parse(listed.stdout.slice(listed.stdout.indexOf("[")));
358
- if (!Array.isArray(rows) || !rows.every((r) => typeof r?.file === "string"))
359
- throw new Error("invalid listing");
360
- files = [...new Set(rows.map((r) => toManifestPath(r.file, cwd)))].sort();
361
- }
362
- catch {
363
- throw new Error(`vitest cannot list files: invalid JSON listing: ${listed.stdout}`);
364
- }
365
- if (!files.length)
366
- throw new Error("vitest cannot list files: empty manifest");
393
+ const rows = JSON.parse(listed.stdout.slice(listed.stdout.indexOf("[")));
394
+ if (!Array.isArray(rows) || !rows.every((r) => typeof r?.file === "string"))
395
+ throw new Error("invalid listing");
396
+ files = [...new Set(rows.map((r) => toManifestPath(r.file, cwd)))].sort();
397
+ }
398
+ catch {
399
+ throw new Error(`vitest cannot list files: invalid JSON listing: ${listed.stdout}`);
400
+ }
401
+ if (!files.length)
402
+ throw new Error("vitest cannot list files: empty manifest");
403
+ return { files, listing: invocation.listing, separator: invocation.separator, listingExit: listed.exitCode, listingStdout: listed.stdout };
404
+ }
405
+ /** The baseline capture's reading of the same seam: the manifest's file count, or null when the
406
+ * runner cannot list (a null compares nothing — it never manufactures a deficit). The suite is not
407
+ * run here; the capture already ran it once. */
408
+ export async function manifestFileCount(cmd, cwd) {
409
+ const dir = mkdtempSync(join(tmpdir(), "tickmarkr-test-manifest-"));
410
+ try {
411
+ const { files } = await discoverTestManifest(cmd, cwd, { dir, nonce: randomBytes(16).toString("hex"), env: manifestEnvironment(cwd).env });
412
+ return files.length;
413
+ }
414
+ catch {
415
+ return null;
416
+ }
417
+ }
418
+ /** One configured runner execution, and its own collection under the same arguments and environment.
419
+ * The installed runner is trusted (R28 add.1 option B); the nonce catches stale artifacts, not forgery. */
420
+ export async function evaluateManifestedTest(cmd, cwd, opts) {
421
+ const dir = opts.artifactDir ?? mkdtempSync(join(tmpdir(), "tickmarkr-test-report-"));
422
+ const nonce = randomBytes(16).toString("hex");
423
+ const reportPath = join(dir, `test-manifest-report-${nonce}.json`);
424
+ const reporterPath = join(dir, `test-reporter-${nonce}.mjs`);
425
+ let spawnedCommand = cmd;
426
+ let manifestPath;
427
+ const { env, verification } = manifestEnvironment(cwd);
428
+ try {
429
+ const invocation = await discoverTestManifest(cmd, cwd, { dir, nonce, env, overallCeilingMs: opts.overallCeilingMs });
430
+ const files = invocation.files;
431
+ // R41: the EXPECTED manifest is evidence in its own right — persisted beside the report with the
432
+ // exact discovery invocation, so a later reader can tell what this invocation was asked to prove
433
+ // without reconstructing it from the runner's own claims.
434
+ manifestPath = join(dir, `test-manifest-expected-${nonce}.json`);
435
+ writeFileSync(manifestPath, JSON.stringify({
436
+ nonce, verification, listingCommand: invocation.listing, listingExit: invocation.listingExit,
437
+ listingStdoutSha256: createHash("sha256").update(invocation.listingStdout).digest("hex"),
438
+ discoveredAt: Date.now(), files,
439
+ }, null, 2) + "\n");
367
440
  writeFileSync(reporterPath, TEST_REPORTER_SOURCE);
368
441
  spawnedCommand = `${cmd}${invocation.separator} --reporter=${shq(reporterPath)} --outputFile=${shq(reportPath)}`;
369
442
  const invoked = await runManifestedTest(spawnedCommand, cwd, {
@@ -375,15 +448,34 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
375
448
  });
376
449
  const verdict = verifyManifestReport({ manifest: files, nonce, exitCode: invoked.exitCode,
377
450
  report: invoked.report, killedFile: invoked.killedFile, hangBudgetMs: invoked.hangBudgetMs });
451
+ // Preserve the validator's verdict and classification; runner evidence only explains it.
452
+ const stdoutPath = join(dir, `test-runner-stdout-${nonce}.log`);
453
+ const stderrPath = join(dir, `test-runner-stderr-${nonce}.log`);
454
+ const stdoutTail = Buffer.from(invoked.stdout).subarray(-16 * 1024);
455
+ const stderrTail = Buffer.from(invoked.stderr).subarray(-16 * 1024);
456
+ writeFileSync(stdoutPath, stdoutTail);
457
+ writeFileSync(stderrPath, stderrTail);
458
+ const report = invoked.report?.nonce === nonce ? invoked.report : undefined;
459
+ const neverStarted = report ? files.filter(file => !(file in report.started)).length : "unknown";
460
+ const errors = report?.certificate?.errors;
461
+ const reporterErrors = typeof errors === "number" && Number.isInteger(errors) && errors >= 0 ? errors : "unknown";
462
+ const reportedDiagnostics = report?.certificate?.diagnostics;
463
+ const runnerErrors = Array.isArray(reportedDiagnostics) ? reportedDiagnostics.filter(error => typeof error === "string") : [];
464
+ const diagnostics = !verdict.pass
465
+ ? `\nclassification: ${verdict.meta.classification ?? "unknown"}; runner-level diagnostic: never-started ${neverStarted}; reporter errors ${reporterErrors}`
466
+ + [...runnerErrors,
467
+ stdoutTail.toString(), stderrTail.toString()].filter(Boolean).map(text => `\n${text}`).join("")
468
+ : "";
378
469
  return { pass: verdict.pass, kind: verdict.kind,
379
- details: verdict.details + (!invoked.report && invoked.stderr ? `\nvitest reporter: ${invoked.stderr}` : ""),
470
+ details: verdict.details + diagnostics,
380
471
  classification: verdict.meta.classification,
381
- meta: { ...verdict.meta, nonce, manifest: files, spawnedCommand, processExit: invoked.exitCode, pid: invoked.pid },
472
+ meta: { ...verdict.meta, nonce, manifest: files, manifestPath, listingCommand: invocation.listing, verification,
473
+ spawnedCommand, processExit: invoked.exitCode, pid: invoked.pid, stdoutPath, stderrPath },
382
474
  exitCode: invoked.exitCode ?? -1, reportPath };
383
475
  }
384
476
  catch (error) {
385
477
  return { pass: false, kind: "infra", classification: "infra", exitCode: -1, reportPath,
386
478
  details: error instanceof Error ? error.message : String(error),
387
- meta: { classification: "infra", infra: true, manifestDiscoveryFailed: true, spawnedCommand } };
479
+ meta: { classification: "infra", infra: true, manifestDiscoveryFailed: true, spawnedCommand, verification, ...(manifestPath ? { manifestPath } : {}) } };
388
480
  }
389
481
  }
@@ -27,22 +27,35 @@ export default class TickmarkrReporter {
27
27
  const file = this.file(module);
28
28
  const failed = module.state() === 'failed';
29
29
  const failures = [];
30
- if (failed) {
31
- for (const test of module.children.allTests()) {
32
- if (test.result().state === 'failed') {
30
+ // R41: count test bodies by their own state so a module whose every test was skipped (a
31
+ // describe.skipIf gate) is recorded as SKIPPED — present in the lifecycle, but never as
32
+ // executed test-body success. 'passed'/'failed' executed; anything else did not run.
33
+ const tests = { passed: 0, failed: 0, skipped: 0 };
34
+ for (const test of module.children.allTests()) {
35
+ const state = test.result().state;
36
+ if (state === 'failed') {
37
+ tests.failed++;
38
+ if (failed) {
33
39
  const errors = test.result().errors || [];
34
40
  failures.push(...(errors.length ? errors.map(e => 'FAIL ' + file + ' > ' + test.fullName + ': ' + e.message) : ['FAIL ' + file + ' > ' + test.fullName]));
35
41
  }
36
- }
37
- if (!failures.length) failures.push('FAIL ' + file);
42
+ } else if (state === 'passed') tests.passed++;
43
+ else tests.skipped++;
38
44
  }
45
+ if (failed && !failures.length) failures.push('FAIL ' + file);
46
+ const status = failed ? 'failed' : tests.passed + tests.failed === 0 ? 'skipped' : 'passed';
39
47
  if (file in this.report.completed) this.report.duplicateCompletions.push(file);
40
- this.report.completed[file] = { at: Date.now(), status: failed ? 'failed' : 'passed', failures };
48
+ this.report.completed[file] = { at: Date.now(), status, failures, tests };
41
49
  this.save();
42
50
  }
43
51
  onTestRunEnd(modules, errors, reason) {
44
52
  const failed = reason === 'failed' || errors.length > 0 || Object.values(this.report.completed).some(c => c.status === 'failed');
45
- this.report.certificate = { at: Date.now(), exitCode: failed ? 1 : 0 };
53
+ // Collection failures can reach run end without a module start/end event. Keep their
54
+ // identity as diagnostics, without inventing lifecycle records or changing the verdict.
55
+ const loadErrors = modules.flatMap(module => module.errors().map(e => this.file(module) + ': ' + e.message));
56
+ this.report.certificate = { at: Date.now(), exitCode: failed ? 1 : 0,
57
+ errors: errors.length + loadErrors.length,
58
+ diagnostics: [...loadErrors, ...errors.map(e => [e.testPath, e.name, e.message].filter(Boolean).join(': '))] };
46
59
  this.save();
47
60
  }
48
61
  }
@@ -21,6 +21,8 @@ export interface OnDiskSpecHash {
21
21
  * evidence for failed recompiles, while moved/deleted source files need not strand a compiled graph.
22
22
  */
23
23
  export declare function onDiskSpecHash(_repoRoot: string, graph: RunGraph): OnDiskSpecHash | undefined;
24
+ /** A task's definition minus its files[] (and run state): what a scope amendment must NOT move (OBS-1073). */
25
+ export declare function taskDefinitionFingerprint(task: Task): string;
24
26
  export declare function graphDefinitionHash(g: RunGraph): string;
25
27
  export declare function taskContentDigest(task: Pick<Task, "goal" | "files" | "acceptance">): string;
26
28
  export declare function tickmarkrDir(repoRoot: string): string;
@@ -33,7 +35,9 @@ export declare function addEvidence(g: RunGraph, id: string, patch: {
33
35
  artifacts?: string[];
34
36
  gateResults?: unknown[];
35
37
  }): RunGraph;
38
+ export declare function chainDepth(g: RunGraph): Map<string, number>;
36
39
  export declare function readyTasks(g: RunGraph): Task[];
40
+ export declare function dispatchWaves(g: RunGraph, concurrency: number): Map<string, number>;
37
41
  export declare function isComplete(g: RunGraph): boolean;
38
42
  export declare function isStalled(g: RunGraph): boolean;
39
43
  export declare function closureReaches(g: RunGraph, taskId: string, pred: (t: Task) => boolean): boolean;
@@ -80,6 +80,11 @@ export function onDiskSpecHash(_repoRoot, graph) {
80
80
  // single comparator in journal.ts (engagementComparable) so the journal↔graph join is decided once.
81
81
  // ponytail: sha256 truncated to 16 hex — stable, grep-friendly; promote to full digest only if a
82
82
  // collision ever bites (engagement ids are not a trust boundary, collisions just force a re-run).
83
+ /** A task's definition minus its files[] (and run state): what a scope amendment must NOT move (OBS-1073). */
84
+ export function taskDefinitionFingerprint(task) {
85
+ const { status: _status, evidence: _evidence, files: _files, ...def } = task;
86
+ return createHash("sha256").update(JSON.stringify(def)).digest("hex").slice(0, 16);
87
+ }
83
88
  export function graphDefinitionHash(g) {
84
89
  const definitions = g.tasks.map(({ status: _status, evidence: _evidence, ...def }) => def);
85
90
  return createHash("sha256").update(JSON.stringify({ version: g.version, spec: g.spec, tasks: definitions })).digest("hex").slice(0, 16);
@@ -156,9 +161,53 @@ export function addEvidence(g, id, patch) {
156
161
  : t),
157
162
  };
158
163
  }
164
+ // OBS-1018: chain depth = number of downstream edges on the longest dependency path from a task to
165
+ // a leaf below it (a task nothing depends on is 0). Counted over the whole graph, status-blind: a
166
+ // root's critical path does not shrink because part of it already ran.
167
+ export function chainDepth(g) {
168
+ const dependents = new Map(g.tasks.map((t) => [t.id, []]));
169
+ for (const t of g.tasks)
170
+ for (const d of t.deps)
171
+ dependents.get(d)?.push(t.id);
172
+ const depth = new Map();
173
+ const visit = (id) => {
174
+ const known = depth.get(id);
175
+ if (known !== undefined)
176
+ return known;
177
+ const below = dependents.get(id) ?? [];
178
+ const value = below.length ? 1 + Math.max(...below.map(visit)) : 0; // acyclic: validateGraph rejects cycles
179
+ depth.set(id, value);
180
+ return value;
181
+ };
182
+ for (const t of g.tasks)
183
+ visit(t.id);
184
+ return depth;
185
+ }
186
+ // OBS-1018: admission is critical-path order — deepest chain root first, ties in declaration order.
187
+ // The daemon's dispatch loop slices this list unchanged. Stated limit: resume-restore of previously
188
+ // in-flight attempts keeps its own precedence (src/run/daemon.ts); only fresh admission is ordered here.
159
189
  export function readyTasks(g) {
160
190
  const done = new Set(g.tasks.filter((t) => t.status === "done").map((t) => t.id));
161
- return g.tasks.filter((t) => t.status === "pending" && t.deps.every((d) => done.has(d)));
191
+ const depth = chainDepth(g);
192
+ return g.tasks
193
+ .filter((t) => t.status === "pending" && t.deps.every((d) => done.has(d)))
194
+ .sort((a, b) => depth.get(b.id) - depth.get(a.id)); // Array#sort is stable: equal depths keep declaration order
195
+ }
196
+ // OBS-1018: the dispatch wave each pending task would enter at `concurrency` slots if every wave
197
+ // took one tick — computed by draining readyTasks, so plan and daemon can never disagree on order.
198
+ // Tasks already past pending carry no wave.
199
+ export function dispatchWaves(g, concurrency) {
200
+ const waves = new Map();
201
+ let sim = g;
202
+ for (let wave = 1;; wave++) {
203
+ const batch = readyTasks(sim).slice(0, Math.max(1, concurrency));
204
+ if (!batch.length)
205
+ return waves;
206
+ for (const t of batch) {
207
+ waves.set(t.id, wave);
208
+ sim = setStatus(sim, t.id, "done");
209
+ }
210
+ }
162
211
  }
163
212
  export function isComplete(g) {
164
213
  return g.tasks.every((t) => t.status === "done");
@@ -1,4 +1,5 @@
1
1
  import { type JournalEvent } from "./journal.js";
2
+ import { type CommandReceipt, type TrackedJournalRow } from "./protocol.js";
2
3
  export interface ActivityTask {
3
4
  id: string;
4
5
  gates: readonly string[];
@@ -13,3 +14,30 @@ export interface ActivitySnapshot {
13
14
  cells: Map<string, string>;
14
15
  }
15
16
  export declare function foldActivity(events: JournalEvent[], tasks: readonly ActivityTask[]): ActivitySnapshot;
17
+ /** Recorded evidence, not a probe of whether a subprocess is still alive. */
18
+ export type BuildActivity = {
19
+ state: "start-unrecorded" | "awaiting-command";
20
+ } | {
21
+ state: CommandReceipt["outcome"] | "unresolved";
22
+ receipt: CommandReceipt;
23
+ };
24
+ export interface TaskActivityProjection {
25
+ taskId: string;
26
+ /** Journal attempt label (zero based), absent until an attempt is recorded. */
27
+ attempt?: number;
28
+ state: "unconfirmed" | "preparing" | "implementing" | "returned-for-verification" | "validating" | "reviewing" | "merging" | "terminal";
29
+ /** Concurrent gates retain separate entries; state alone is only a summary. */
30
+ phases: {
31
+ gate: string;
32
+ state: "validating" | "reviewing";
33
+ }[];
34
+ build: BuildActivity;
35
+ }
36
+ /**
37
+ * Pure, evidence-only successor to foldActivity. Feed Journal.readTracked() and the owning run ID;
38
+ * sourceIndex supplies journal order, never wall time. Unattributed legacy rows belong to their
39
+ * tracked run and current attempt/round. They cannot prove identities absent from the journal.
40
+ * A new gates phase opens a round; completed gates cannot reopen within that round. No declared
41
+ * gate order, graph status, or successful verdict predicts a later phase.
42
+ */
43
+ export declare function projectActivity(runId: string, rows: readonly TrackedJournalRow[], tasks: readonly ActivityTask[]): Map<string, TaskActivityProjection>;