tickmarkr 2.5.5 → 2.5.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -1
- package/dist/adapters/prompt.js +21 -1
- package/dist/cli/commands/approve.js +69 -8
- package/dist/cli/commands/doctor.d.ts +10 -0
- package/dist/cli/commands/fleet.js +4 -0
- package/dist/cli/commands/plan.js +20 -4
- package/dist/cli/commands/status.js +95 -34
- package/dist/cli/commands/verify.js +108 -85
- package/dist/config/config.d.ts +11 -0
- package/dist/config/config.js +21 -12
- package/dist/config/fleet-overlay.js +55 -17
- package/dist/drivers/index.js +2 -1
- package/dist/drivers/orca.d.ts +21 -1
- package/dist/drivers/orca.js +209 -27
- package/dist/gates/baseline.d.ts +21 -5
- package/dist/gates/baseline.js +67 -17
- package/dist/gates/cache.d.ts +14 -13
- package/dist/gates/cache.js +17 -5
- package/dist/gates/llm.d.ts +3 -0
- package/dist/gates/llm.js +11 -0
- package/dist/gates/review.d.ts +28 -3
- package/dist/gates/review.js +118 -15
- package/dist/gates/run-gates.d.ts +10 -2
- package/dist/gates/run-gates.js +362 -108
- package/dist/gates/test-manifest.d.ts +33 -1
- package/dist/gates/test-manifest.js +132 -40
- package/dist/gates/test-reporter.js +20 -7
- package/dist/graph/graph.d.ts +4 -0
- package/dist/graph/graph.js +50 -1
- package/dist/run/activity.d.ts +28 -0
- package/dist/run/activity.js +194 -0
- package/dist/run/consult.js +5 -4
- package/dist/run/daemon.d.ts +11 -0
- package/dist/run/daemon.js +663 -128
- package/dist/run/execution-budget.d.ts +25 -0
- package/dist/run/execution-budget.js +142 -0
- package/dist/run/git.d.ts +46 -1
- package/dist/run/git.js +149 -12
- package/dist/run/journal.d.ts +23 -5
- package/dist/run/journal.js +98 -25
- package/dist/run/lease.d.ts +44 -0
- package/dist/run/lease.js +226 -3
- package/dist/run/operator-page-summary.d.ts +56 -0
- package/dist/run/operator-page-summary.js +68 -0
- package/dist/run/operator-summary.d.ts +69 -0
- package/dist/run/operator-summary.js +77 -0
- package/dist/run/protocol.d.ts +71 -0
- package/dist/run/protocol.js +32 -0
- package/dist/run/recovery.d.ts +8 -0
- package/dist/run/recovery.js +25 -0
- package/dist/run/repair-selection.d.ts +12 -0
- package/dist/run/repair-selection.js +56 -0
- package/dist/run/stall.d.ts +6 -1
- package/dist/run/stall.js +60 -3
- package/dist/tui/cockpit/board.d.ts +9 -0
- package/dist/tui/cockpit/board.js +10 -0
- package/dist/tui/cockpit/derive.d.ts +35 -0
- package/dist/tui/cockpit/derive.js +152 -10
- package/dist/tui/cockpit/evidence-view.d.ts +2 -0
- package/dist/tui/cockpit/evidence-view.js +42 -12
- package/dist/tui/cockpit/run-cockpit.d.ts +8 -1
- package/dist/tui/cockpit/run-cockpit.js +95 -1
- package/dist/tui/cockpit/run-view.d.ts +37 -0
- package/dist/tui/cockpit/run-view.js +189 -2
- package/dist/tui/ink/fleet-app.d.ts +10 -2
- package/dist/tui/ink/fleet-app.js +33 -15
- package/package.json +2 -2
- package/skills/tickmarkr-overseer/SKILL.md +55 -3
- package/skills/tickmarkr-overseer/scripts/grade-ci.sh +29 -2
|
@@ -9,9 +9,17 @@ export declare function isVitestTestCommand(cmd: string, cwd: string): boolean;
|
|
|
9
9
|
export declare function toManifestPath(file: string, cwd: string): string;
|
|
10
10
|
export interface TestReportCompletion {
|
|
11
11
|
at: number;
|
|
12
|
-
|
|
12
|
+
/** R41: `skipped` = the module ran its lifecycle but executed NO test body (every test skipped);
|
|
13
|
+
* it is present in the manifest accounting and never counted as executed success. */
|
|
14
|
+
status: "passed" | "failed" | "skipped";
|
|
13
15
|
/** Failure fingerprints for this file; absent/empty on a passed file. */
|
|
14
16
|
failures?: string[];
|
|
17
|
+
/** Per-module test-body counts the reporter observed (absent on reports from older reporters). */
|
|
18
|
+
tests?: {
|
|
19
|
+
passed: number;
|
|
20
|
+
failed: number;
|
|
21
|
+
skipped: number;
|
|
22
|
+
};
|
|
15
23
|
}
|
|
16
24
|
/** The runner's own machine report — requested/started/completed are the runner's claims about ITSELF. */
|
|
17
25
|
export interface TestReport {
|
|
@@ -28,6 +36,9 @@ export interface TestReport {
|
|
|
28
36
|
certificate?: {
|
|
29
37
|
at: number;
|
|
30
38
|
exitCode: number;
|
|
39
|
+
/** Unhandled and module collection errors observed by the reporter; absent on older reports. */
|
|
40
|
+
errors?: number;
|
|
41
|
+
diagnostics?: string[];
|
|
31
42
|
};
|
|
32
43
|
}
|
|
33
44
|
/** Reads and structurally validates the report; a missing or malformed file is `undefined` — never a partial parse. */
|
|
@@ -89,6 +100,27 @@ export interface ManifestGateOutcome {
|
|
|
89
100
|
exitCode: number;
|
|
90
101
|
reportPath: string;
|
|
91
102
|
}
|
|
103
|
+
export interface DiscoveredManifest {
|
|
104
|
+
files: string[];
|
|
105
|
+
listing: string;
|
|
106
|
+
separator: string;
|
|
107
|
+
listingExit: number | undefined;
|
|
108
|
+
listingStdout: string;
|
|
109
|
+
}
|
|
110
|
+
/** OBS-1044: THE discovery seam — the files the runner would collect under this exact invocation,
|
|
111
|
+
* from its own listing and never from a stdout summary. The gate and the baseline capture share it,
|
|
112
|
+
* so the count a capture records is the count the gate later asks the report to certify. Throws
|
|
113
|
+
* when the runner cannot list: a count that is not the runner's own is not a count. */
|
|
114
|
+
export declare function discoverTestManifest(cmd: string, cwd: string, opts: {
|
|
115
|
+
dir: string;
|
|
116
|
+
nonce: string;
|
|
117
|
+
env: NodeJS.ProcessEnv;
|
|
118
|
+
overallCeilingMs?: number;
|
|
119
|
+
}): Promise<DiscoveredManifest>;
|
|
120
|
+
/** The baseline capture's reading of the same seam: the manifest's file count, or null when the
|
|
121
|
+
* runner cannot list (a null compares nothing — it never manufactures a deficit). The suite is not
|
|
122
|
+
* run here; the capture already ran it once. */
|
|
123
|
+
export declare function manifestFileCount(cmd: string, cwd: string): Promise<number | null>;
|
|
92
124
|
/** One configured runner execution, and its own collection under the same arguments and environment.
|
|
93
125
|
* The installed runner is trusted (R28 add.1 option B); the nonce catches stale artifacts, not forgery. */
|
|
94
126
|
export declare function evaluateManifestedTest(cmd: string, cwd: string, opts: {
|
|
@@ -1,10 +1,10 @@
|
|
|
1
|
-
import { randomBytes } from "node:crypto";
|
|
1
|
+
import { createHash, randomBytes } from "node:crypto";
|
|
2
2
|
import { existsSync, mkdtempSync, readFileSync, realpathSync, writeFileSync } from "node:fs";
|
|
3
3
|
import { tmpdir } from "node:os";
|
|
4
4
|
import { isAbsolute, join, relative, sep } from "node:path";
|
|
5
5
|
import { TEST_REPORTER_SOURCE } from "./test-reporter.js";
|
|
6
6
|
import { shq } from "../adapters/types.js";
|
|
7
|
-
import { FORK_CAP_ENV, ROUTING_ENV_SEAMS, SUITE_PARENT_ENV, shell, resolvedCapacity } from "../run/git.js";
|
|
7
|
+
import { FORK_CAP_ENV, ROUTING_ENV_SEAMS, SUITE_PARENT_ENV, shell, resolvedCapacity, verificationProtocol } from "../run/git.js";
|
|
8
8
|
/**
|
|
9
9
|
* VL-1 (OBS-985 lineage): a test gate's completion must be the runner's OWN report, never a stdout
|
|
10
10
|
* count. `fileCountDeficit` (baseline.ts) reads a summary LINE — a selected screen's smaller count
|
|
@@ -30,7 +30,12 @@ function scriptInvocation(cmd, cwd) {
|
|
|
30
30
|
manager++;
|
|
31
31
|
if (!["npm", "pnpm", "yarn"].includes(argv[manager]))
|
|
32
32
|
return undefined;
|
|
33
|
-
|
|
33
|
+
let offset = manager + (["run", "run-script"].includes(argv[manager + 1]) ? 2 : 1);
|
|
34
|
+
// OBS-1044: `npm run -s test` — the very spelling detectGateCommands synthesizes — carries the
|
|
35
|
+
// manager's own flags before the script name; a seam blind to them routed every default-configured
|
|
36
|
+
// npm repository past the manifest. Skip flags, never a `--` (which ends the manager's arguments).
|
|
37
|
+
while (offset > manager + 1 && /^-(?!-$)/.test(argv[offset] ?? ""))
|
|
38
|
+
offset++;
|
|
34
39
|
const name = ["t", "tst"].includes(argv[offset]) ? "test" : argv[offset];
|
|
35
40
|
try {
|
|
36
41
|
const body = JSON.parse(readFileSync(join(cwd, "package.json"), "utf8")).scripts?.[name];
|
|
@@ -72,8 +77,13 @@ function runnerInvocation(cmd, cwd) {
|
|
|
72
77
|
if (forwarded[0] === "run")
|
|
73
78
|
forwarded.shift();
|
|
74
79
|
const collection = forwarded.filter((a) => a !== "--run" && a !== "--watch" && a !== "--");
|
|
80
|
+
// R41: `--filesOnly` lists FILE SPECIFICATIONS — the unit the reporter certifies (module start/end) —
|
|
81
|
+
// instead of collected test cases, from which the ordinary listing omits every skipped test and so
|
|
82
|
+
// drops a module whose whole body is skipped (describe.skipIf): the runner then reported two files
|
|
83
|
+
// the manifest never expected (T4 236 vs 238, T8 288 vs 290 on run 91). Filters and the environment
|
|
84
|
+
// are forwarded exactly as before; only the unit enumerated changes.
|
|
75
85
|
return {
|
|
76
|
-
listing: [...env, ...prefix.map(shq), shq(binary), "list", ...collection, "--json"].join(" "),
|
|
86
|
+
listing: [...env, ...prefix.map(shq), shq(binary), "list", ...collection, "--filesOnly", "--json"].join(" "),
|
|
77
87
|
separator: script?.npm && !/(?:^|\s)--(?:\s|$)/.test(cmd) ? " --" : "",
|
|
78
88
|
};
|
|
79
89
|
}
|
|
@@ -107,8 +117,15 @@ function isTestReportShape(v) {
|
|
|
107
117
|
if (typeof c !== "object" || c === null)
|
|
108
118
|
return false;
|
|
109
119
|
const cc = c;
|
|
110
|
-
if (typeof cc.at !== "number" || (cc.status !== "passed" && cc.status !== "failed"))
|
|
120
|
+
if (typeof cc.at !== "number" || (cc.status !== "passed" && cc.status !== "failed" && cc.status !== "skipped"))
|
|
111
121
|
return false;
|
|
122
|
+
if (cc.tests !== undefined) {
|
|
123
|
+
if (typeof cc.tests !== "object" || cc.tests === null)
|
|
124
|
+
return false;
|
|
125
|
+
const t = cc.tests;
|
|
126
|
+
if (!["passed", "failed", "skipped"].every((k) => typeof t[k] === "number"))
|
|
127
|
+
return false;
|
|
128
|
+
}
|
|
112
129
|
}
|
|
113
130
|
if (r.duplicateCompletions !== undefined) {
|
|
114
131
|
if (!Array.isArray(r.duplicateCompletions) || !r.duplicateCompletions.every((f) => typeof f === "string"))
|
|
@@ -208,12 +225,13 @@ export function verifyManifestReport(opts) {
|
|
|
208
225
|
meta: { classification: "infra", infra: true, missingCertificate: true, ...(pending !== undefined ? { file: pending } : {}) },
|
|
209
226
|
};
|
|
210
227
|
}
|
|
211
|
-
const
|
|
212
|
-
|
|
228
|
+
const failedCompletions = Object.entries(report.completed).filter(([, c]) => c.status === "failed");
|
|
229
|
+
const failingFiles = failedCompletions.map(([file]) => file).sort();
|
|
230
|
+
const failures = failedCompletions.flatMap(([, c]) => c.failures?.length ? c.failures : ["<unnamed failure>"]);
|
|
213
231
|
if (failures.length)
|
|
214
232
|
return { kind: "work", pass: false,
|
|
215
233
|
details: `test report names failing fingerprint(s):\n${failures.join("\n")}`,
|
|
216
|
-
meta: { classification: "regression", failingTests: failures, processExit: exitCode } };
|
|
234
|
+
meta: { classification: "regression", failingTests: failures, failingFiles, processExit: exitCode } };
|
|
217
235
|
if (exitCode === undefined) {
|
|
218
236
|
return {
|
|
219
237
|
kind: "fail-closed",
|
|
@@ -264,11 +282,25 @@ export function verifyManifestReport(opts) {
|
|
|
264
282
|
meta: { classification: "infra", infra: true, file: missing, missingFromReport: true },
|
|
265
283
|
};
|
|
266
284
|
}
|
|
285
|
+
// R41: a skipped whole module is PRESENT (the lifecycle certified it) but is not executed test-body
|
|
286
|
+
// success. The verdict says how many modules executed and names every one that did not; a report
|
|
287
|
+
// in which nothing executed proves nothing about the tree and is never a passing verdict.
|
|
288
|
+
const skippedModules = manifest.filter((f) => report.completed[f]?.status === "skipped").sort();
|
|
289
|
+
const executedModules = manifest.length - skippedModules.length;
|
|
290
|
+
if (executedModules === 0) {
|
|
291
|
+
return {
|
|
292
|
+
kind: "fail-closed",
|
|
293
|
+
pass: false,
|
|
294
|
+
details: `every manifest module was skipped whole (${skippedModules.join(", ")}) — no test body executed, so this report certifies nothing about the tree; failing closed`,
|
|
295
|
+
meta: { classification: "infra", infra: true, noExecutedModules: true, manifestFiles: manifest.length, executedModules, skippedModules },
|
|
296
|
+
};
|
|
297
|
+
}
|
|
267
298
|
return {
|
|
268
299
|
kind: "pass",
|
|
269
300
|
pass: true,
|
|
270
|
-
details: `invocation-bound test report succeeded — ${manifest.length} manifest file(s) present exactly once
|
|
271
|
-
|
|
301
|
+
details: `invocation-bound test report succeeded — ${manifest.length} manifest file(s) present exactly once; ${executedModules} executed`
|
|
302
|
+
+ (skippedModules.length ? `, ${skippedModules.length} skipped whole (${skippedModules.join(", ")})` : ""),
|
|
303
|
+
meta: { manifestFiles: manifest.length, executedModules, skippedModules },
|
|
272
304
|
};
|
|
273
305
|
}
|
|
274
306
|
/** How much longer than its baseline measurement one file may legitimately run before it is a hang. */
|
|
@@ -328,42 +360,83 @@ export function runManifestedTest(cmd, cwd, opts) {
|
|
|
328
360
|
report: readTestReport(opts.reportPath), killedFile, hangBudgetMs, pid,
|
|
329
361
|
})).finally(() => { clearInterval(poll); });
|
|
330
362
|
}
|
|
331
|
-
/**
|
|
332
|
-
*
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
const reportPath = join(dir, `test-manifest-report-${nonce}.json`);
|
|
337
|
-
const reporterPath = join(dir, `test-reporter-${nonce}.mjs`);
|
|
338
|
-
let spawnedCommand = cmd;
|
|
363
|
+
/** The child environment every manifest invocation (listing and run) receives, and the lifecycle
|
|
364
|
+
* protocol it records. R41: the policy is whatever THIS process was launched with (npm reads
|
|
365
|
+
* `npm_config_ignore_scripts` from the environment over every npmrc); the verdict records it so a
|
|
366
|
+
* verdict measured with `pretest` hooks is never compared to one without. */
|
|
367
|
+
function manifestEnvironment(cwd) {
|
|
339
368
|
const env = { ...process.env, PATH: `${join(cwd, "node_modules/.bin")}:${process.env.PATH ?? ""}`,
|
|
340
369
|
[FORK_CAP_ENV]: String(resolvedCapacity().forkCap), [SUITE_PARENT_ENV]: String(process.pid) };
|
|
370
|
+
const verification = verificationProtocol(env, cwd);
|
|
341
371
|
for (const key of ROUTING_ENV_SEAMS)
|
|
342
372
|
delete env[key];
|
|
343
373
|
// A gate can itself be tested under Vitest. Do not give the child the outer worker identity.
|
|
344
374
|
for (const key of Object.keys(env))
|
|
345
375
|
if (["VITEST", "TEST", "VITEST_WORKER_ID", "VITEST_POOL_ID"].includes(key))
|
|
346
376
|
delete env[key];
|
|
377
|
+
return { env, verification };
|
|
378
|
+
}
|
|
379
|
+
/** OBS-1044: THE discovery seam — the files the runner would collect under this exact invocation,
|
|
380
|
+
* from its own listing and never from a stdout summary. The gate and the baseline capture share it,
|
|
381
|
+
* so the count a capture records is the count the gate later asks the report to certify. Throws
|
|
382
|
+
* when the runner cannot list: a count that is not the runner's own is not a count. */
|
|
383
|
+
export async function discoverTestManifest(cmd, cwd, opts) {
|
|
384
|
+
const invocation = runnerInvocation(cmd, cwd);
|
|
385
|
+
const listed = await runManifestedTest(invocation.listing, cwd, {
|
|
386
|
+
manifest: [], nonce: opts.nonce, reportPath: join(opts.dir, `listing-${opts.nonce}.json`), env: opts.env,
|
|
387
|
+
overallCeilingMs: opts.overallCeilingMs ?? DEFAULT_FILE_HANG_BUDGET_MS,
|
|
388
|
+
});
|
|
389
|
+
if (listed.exitCode !== 0)
|
|
390
|
+
throw new Error(`vitest cannot list files (exit ${listed.exitCode ?? "signal"}): ${listed.stderr || listed.stdout}`);
|
|
391
|
+
let files;
|
|
347
392
|
try {
|
|
348
|
-
const
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
393
|
+
const rows = JSON.parse(listed.stdout.slice(listed.stdout.indexOf("[")));
|
|
394
|
+
if (!Array.isArray(rows) || !rows.every((r) => typeof r?.file === "string"))
|
|
395
|
+
throw new Error("invalid listing");
|
|
396
|
+
files = [...new Set(rows.map((r) => toManifestPath(r.file, cwd)))].sort();
|
|
397
|
+
}
|
|
398
|
+
catch {
|
|
399
|
+
throw new Error(`vitest cannot list files: invalid JSON listing: ${listed.stdout}`);
|
|
400
|
+
}
|
|
401
|
+
if (!files.length)
|
|
402
|
+
throw new Error("vitest cannot list files: empty manifest");
|
|
403
|
+
return { files, listing: invocation.listing, separator: invocation.separator, listingExit: listed.exitCode, listingStdout: listed.stdout };
|
|
404
|
+
}
|
|
405
|
+
/** The baseline capture's reading of the same seam: the manifest's file count, or null when the
|
|
406
|
+
* runner cannot list (a null compares nothing — it never manufactures a deficit). The suite is not
|
|
407
|
+
* run here; the capture already ran it once. */
|
|
408
|
+
export async function manifestFileCount(cmd, cwd) {
|
|
409
|
+
const dir = mkdtempSync(join(tmpdir(), "tickmarkr-test-manifest-"));
|
|
410
|
+
try {
|
|
411
|
+
const { files } = await discoverTestManifest(cmd, cwd, { dir, nonce: randomBytes(16).toString("hex"), env: manifestEnvironment(cwd).env });
|
|
412
|
+
return files.length;
|
|
413
|
+
}
|
|
414
|
+
catch {
|
|
415
|
+
return null;
|
|
416
|
+
}
|
|
417
|
+
}
|
|
418
|
+
/** One configured runner execution, and its own collection under the same arguments and environment.
|
|
419
|
+
* The installed runner is trusted (R28 add.1 option B); the nonce catches stale artifacts, not forgery. */
|
|
420
|
+
export async function evaluateManifestedTest(cmd, cwd, opts) {
|
|
421
|
+
const dir = opts.artifactDir ?? mkdtempSync(join(tmpdir(), "tickmarkr-test-report-"));
|
|
422
|
+
const nonce = randomBytes(16).toString("hex");
|
|
423
|
+
const reportPath = join(dir, `test-manifest-report-${nonce}.json`);
|
|
424
|
+
const reporterPath = join(dir, `test-reporter-${nonce}.mjs`);
|
|
425
|
+
let spawnedCommand = cmd;
|
|
426
|
+
let manifestPath;
|
|
427
|
+
const { env, verification } = manifestEnvironment(cwd);
|
|
428
|
+
try {
|
|
429
|
+
const invocation = await discoverTestManifest(cmd, cwd, { dir, nonce, env, overallCeilingMs: opts.overallCeilingMs });
|
|
430
|
+
const files = invocation.files;
|
|
431
|
+
// R41: the EXPECTED manifest is evidence in its own right — persisted beside the report with the
|
|
432
|
+
// exact discovery invocation, so a later reader can tell what this invocation was asked to prove
|
|
433
|
+
// without reconstructing it from the runner's own claims.
|
|
434
|
+
manifestPath = join(dir, `test-manifest-expected-${nonce}.json`);
|
|
435
|
+
writeFileSync(manifestPath, JSON.stringify({
|
|
436
|
+
nonce, verification, listingCommand: invocation.listing, listingExit: invocation.listingExit,
|
|
437
|
+
listingStdoutSha256: createHash("sha256").update(invocation.listingStdout).digest("hex"),
|
|
438
|
+
discoveredAt: Date.now(), files,
|
|
439
|
+
}, null, 2) + "\n");
|
|
367
440
|
writeFileSync(reporterPath, TEST_REPORTER_SOURCE);
|
|
368
441
|
spawnedCommand = `${cmd}${invocation.separator} --reporter=${shq(reporterPath)} --outputFile=${shq(reportPath)}`;
|
|
369
442
|
const invoked = await runManifestedTest(spawnedCommand, cwd, {
|
|
@@ -375,15 +448,34 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
|
|
|
375
448
|
});
|
|
376
449
|
const verdict = verifyManifestReport({ manifest: files, nonce, exitCode: invoked.exitCode,
|
|
377
450
|
report: invoked.report, killedFile: invoked.killedFile, hangBudgetMs: invoked.hangBudgetMs });
|
|
451
|
+
// Preserve the validator's verdict and classification; runner evidence only explains it.
|
|
452
|
+
const stdoutPath = join(dir, `test-runner-stdout-${nonce}.log`);
|
|
453
|
+
const stderrPath = join(dir, `test-runner-stderr-${nonce}.log`);
|
|
454
|
+
const stdoutTail = Buffer.from(invoked.stdout).subarray(-16 * 1024);
|
|
455
|
+
const stderrTail = Buffer.from(invoked.stderr).subarray(-16 * 1024);
|
|
456
|
+
writeFileSync(stdoutPath, stdoutTail);
|
|
457
|
+
writeFileSync(stderrPath, stderrTail);
|
|
458
|
+
const report = invoked.report?.nonce === nonce ? invoked.report : undefined;
|
|
459
|
+
const neverStarted = report ? files.filter(file => !(file in report.started)).length : "unknown";
|
|
460
|
+
const errors = report?.certificate?.errors;
|
|
461
|
+
const reporterErrors = typeof errors === "number" && Number.isInteger(errors) && errors >= 0 ? errors : "unknown";
|
|
462
|
+
const reportedDiagnostics = report?.certificate?.diagnostics;
|
|
463
|
+
const runnerErrors = Array.isArray(reportedDiagnostics) ? reportedDiagnostics.filter(error => typeof error === "string") : [];
|
|
464
|
+
const diagnostics = !verdict.pass
|
|
465
|
+
? `\nclassification: ${verdict.meta.classification ?? "unknown"}; runner-level diagnostic: never-started ${neverStarted}; reporter errors ${reporterErrors}`
|
|
466
|
+
+ [...runnerErrors,
|
|
467
|
+
stdoutTail.toString(), stderrTail.toString()].filter(Boolean).map(text => `\n${text}`).join("")
|
|
468
|
+
: "";
|
|
378
469
|
return { pass: verdict.pass, kind: verdict.kind,
|
|
379
|
-
details: verdict.details +
|
|
470
|
+
details: verdict.details + diagnostics,
|
|
380
471
|
classification: verdict.meta.classification,
|
|
381
|
-
meta: { ...verdict.meta, nonce, manifest: files,
|
|
472
|
+
meta: { ...verdict.meta, nonce, manifest: files, manifestPath, listingCommand: invocation.listing, verification,
|
|
473
|
+
spawnedCommand, processExit: invoked.exitCode, pid: invoked.pid, stdoutPath, stderrPath },
|
|
382
474
|
exitCode: invoked.exitCode ?? -1, reportPath };
|
|
383
475
|
}
|
|
384
476
|
catch (error) {
|
|
385
477
|
return { pass: false, kind: "infra", classification: "infra", exitCode: -1, reportPath,
|
|
386
478
|
details: error instanceof Error ? error.message : String(error),
|
|
387
|
-
meta: { classification: "infra", infra: true, manifestDiscoveryFailed: true, spawnedCommand } };
|
|
479
|
+
meta: { classification: "infra", infra: true, manifestDiscoveryFailed: true, spawnedCommand, verification, ...(manifestPath ? { manifestPath } : {}) } };
|
|
388
480
|
}
|
|
389
481
|
}
|
|
@@ -27,22 +27,35 @@ export default class TickmarkrReporter {
|
|
|
27
27
|
const file = this.file(module);
|
|
28
28
|
const failed = module.state() === 'failed';
|
|
29
29
|
const failures = [];
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
30
|
+
// R41: count test bodies by their own state so a module whose every test was skipped (a
|
|
31
|
+
// describe.skipIf gate) is recorded as SKIPPED — present in the lifecycle, but never as
|
|
32
|
+
// executed test-body success. 'passed'/'failed' executed; anything else did not run.
|
|
33
|
+
const tests = { passed: 0, failed: 0, skipped: 0 };
|
|
34
|
+
for (const test of module.children.allTests()) {
|
|
35
|
+
const state = test.result().state;
|
|
36
|
+
if (state === 'failed') {
|
|
37
|
+
tests.failed++;
|
|
38
|
+
if (failed) {
|
|
33
39
|
const errors = test.result().errors || [];
|
|
34
40
|
failures.push(...(errors.length ? errors.map(e => 'FAIL ' + file + ' > ' + test.fullName + ': ' + e.message) : ['FAIL ' + file + ' > ' + test.fullName]));
|
|
35
41
|
}
|
|
36
|
-
}
|
|
37
|
-
|
|
42
|
+
} else if (state === 'passed') tests.passed++;
|
|
43
|
+
else tests.skipped++;
|
|
38
44
|
}
|
|
45
|
+
if (failed && !failures.length) failures.push('FAIL ' + file);
|
|
46
|
+
const status = failed ? 'failed' : tests.passed + tests.failed === 0 ? 'skipped' : 'passed';
|
|
39
47
|
if (file in this.report.completed) this.report.duplicateCompletions.push(file);
|
|
40
|
-
this.report.completed[file] = { at: Date.now(), status
|
|
48
|
+
this.report.completed[file] = { at: Date.now(), status, failures, tests };
|
|
41
49
|
this.save();
|
|
42
50
|
}
|
|
43
51
|
onTestRunEnd(modules, errors, reason) {
|
|
44
52
|
const failed = reason === 'failed' || errors.length > 0 || Object.values(this.report.completed).some(c => c.status === 'failed');
|
|
45
|
-
|
|
53
|
+
// Collection failures can reach run end without a module start/end event. Keep their
|
|
54
|
+
// identity as diagnostics, without inventing lifecycle records or changing the verdict.
|
|
55
|
+
const loadErrors = modules.flatMap(module => module.errors().map(e => this.file(module) + ': ' + e.message));
|
|
56
|
+
this.report.certificate = { at: Date.now(), exitCode: failed ? 1 : 0,
|
|
57
|
+
errors: errors.length + loadErrors.length,
|
|
58
|
+
diagnostics: [...loadErrors, ...errors.map(e => [e.testPath, e.name, e.message].filter(Boolean).join(': '))] };
|
|
46
59
|
this.save();
|
|
47
60
|
}
|
|
48
61
|
}
|
package/dist/graph/graph.d.ts
CHANGED
|
@@ -21,6 +21,8 @@ export interface OnDiskSpecHash {
|
|
|
21
21
|
* evidence for failed recompiles, while moved/deleted source files need not strand a compiled graph.
|
|
22
22
|
*/
|
|
23
23
|
export declare function onDiskSpecHash(_repoRoot: string, graph: RunGraph): OnDiskSpecHash | undefined;
|
|
24
|
+
/** A task's definition minus its files[] (and run state): what a scope amendment must NOT move (OBS-1073). */
|
|
25
|
+
export declare function taskDefinitionFingerprint(task: Task): string;
|
|
24
26
|
export declare function graphDefinitionHash(g: RunGraph): string;
|
|
25
27
|
export declare function taskContentDigest(task: Pick<Task, "goal" | "files" | "acceptance">): string;
|
|
26
28
|
export declare function tickmarkrDir(repoRoot: string): string;
|
|
@@ -33,7 +35,9 @@ export declare function addEvidence(g: RunGraph, id: string, patch: {
|
|
|
33
35
|
artifacts?: string[];
|
|
34
36
|
gateResults?: unknown[];
|
|
35
37
|
}): RunGraph;
|
|
38
|
+
export declare function chainDepth(g: RunGraph): Map<string, number>;
|
|
36
39
|
export declare function readyTasks(g: RunGraph): Task[];
|
|
40
|
+
export declare function dispatchWaves(g: RunGraph, concurrency: number): Map<string, number>;
|
|
37
41
|
export declare function isComplete(g: RunGraph): boolean;
|
|
38
42
|
export declare function isStalled(g: RunGraph): boolean;
|
|
39
43
|
export declare function closureReaches(g: RunGraph, taskId: string, pred: (t: Task) => boolean): boolean;
|
package/dist/graph/graph.js
CHANGED
|
@@ -80,6 +80,11 @@ export function onDiskSpecHash(_repoRoot, graph) {
|
|
|
80
80
|
// single comparator in journal.ts (engagementComparable) so the journal↔graph join is decided once.
|
|
81
81
|
// ponytail: sha256 truncated to 16 hex — stable, grep-friendly; promote to full digest only if a
|
|
82
82
|
// collision ever bites (engagement ids are not a trust boundary, collisions just force a re-run).
|
|
83
|
+
/** A task's definition minus its files[] (and run state): what a scope amendment must NOT move (OBS-1073). */
|
|
84
|
+
export function taskDefinitionFingerprint(task) {
|
|
85
|
+
const { status: _status, evidence: _evidence, files: _files, ...def } = task;
|
|
86
|
+
return createHash("sha256").update(JSON.stringify(def)).digest("hex").slice(0, 16);
|
|
87
|
+
}
|
|
83
88
|
export function graphDefinitionHash(g) {
|
|
84
89
|
const definitions = g.tasks.map(({ status: _status, evidence: _evidence, ...def }) => def);
|
|
85
90
|
return createHash("sha256").update(JSON.stringify({ version: g.version, spec: g.spec, tasks: definitions })).digest("hex").slice(0, 16);
|
|
@@ -156,9 +161,53 @@ export function addEvidence(g, id, patch) {
|
|
|
156
161
|
: t),
|
|
157
162
|
};
|
|
158
163
|
}
|
|
164
|
+
// OBS-1018: chain depth = number of downstream edges on the longest dependency path from a task to
|
|
165
|
+
// a leaf below it (a task nothing depends on is 0). Counted over the whole graph, status-blind: a
|
|
166
|
+
// root's critical path does not shrink because part of it already ran.
|
|
167
|
+
export function chainDepth(g) {
|
|
168
|
+
const dependents = new Map(g.tasks.map((t) => [t.id, []]));
|
|
169
|
+
for (const t of g.tasks)
|
|
170
|
+
for (const d of t.deps)
|
|
171
|
+
dependents.get(d)?.push(t.id);
|
|
172
|
+
const depth = new Map();
|
|
173
|
+
const visit = (id) => {
|
|
174
|
+
const known = depth.get(id);
|
|
175
|
+
if (known !== undefined)
|
|
176
|
+
return known;
|
|
177
|
+
const below = dependents.get(id) ?? [];
|
|
178
|
+
const value = below.length ? 1 + Math.max(...below.map(visit)) : 0; // acyclic: validateGraph rejects cycles
|
|
179
|
+
depth.set(id, value);
|
|
180
|
+
return value;
|
|
181
|
+
};
|
|
182
|
+
for (const t of g.tasks)
|
|
183
|
+
visit(t.id);
|
|
184
|
+
return depth;
|
|
185
|
+
}
|
|
186
|
+
// OBS-1018: admission is critical-path order — deepest chain root first, ties in declaration order.
|
|
187
|
+
// The daemon's dispatch loop slices this list unchanged. Stated limit: resume-restore of previously
|
|
188
|
+
// in-flight attempts keeps its own precedence (src/run/daemon.ts); only fresh admission is ordered here.
|
|
159
189
|
export function readyTasks(g) {
|
|
160
190
|
const done = new Set(g.tasks.filter((t) => t.status === "done").map((t) => t.id));
|
|
161
|
-
|
|
191
|
+
const depth = chainDepth(g);
|
|
192
|
+
return g.tasks
|
|
193
|
+
.filter((t) => t.status === "pending" && t.deps.every((d) => done.has(d)))
|
|
194
|
+
.sort((a, b) => depth.get(b.id) - depth.get(a.id)); // Array#sort is stable: equal depths keep declaration order
|
|
195
|
+
}
|
|
196
|
+
// OBS-1018: the dispatch wave each pending task would enter at `concurrency` slots if every wave
|
|
197
|
+
// took one tick — computed by draining readyTasks, so plan and daemon can never disagree on order.
|
|
198
|
+
// Tasks already past pending carry no wave.
|
|
199
|
+
export function dispatchWaves(g, concurrency) {
|
|
200
|
+
const waves = new Map();
|
|
201
|
+
let sim = g;
|
|
202
|
+
for (let wave = 1;; wave++) {
|
|
203
|
+
const batch = readyTasks(sim).slice(0, Math.max(1, concurrency));
|
|
204
|
+
if (!batch.length)
|
|
205
|
+
return waves;
|
|
206
|
+
for (const t of batch) {
|
|
207
|
+
waves.set(t.id, wave);
|
|
208
|
+
sim = setStatus(sim, t.id, "done");
|
|
209
|
+
}
|
|
210
|
+
}
|
|
162
211
|
}
|
|
163
212
|
export function isComplete(g) {
|
|
164
213
|
return g.tasks.every((t) => t.status === "done");
|
package/dist/run/activity.d.ts
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { type JournalEvent } from "./journal.js";
|
|
2
|
+
import { type CommandReceipt, type TrackedJournalRow } from "./protocol.js";
|
|
2
3
|
export interface ActivityTask {
|
|
3
4
|
id: string;
|
|
4
5
|
gates: readonly string[];
|
|
@@ -13,3 +14,30 @@ export interface ActivitySnapshot {
|
|
|
13
14
|
cells: Map<string, string>;
|
|
14
15
|
}
|
|
15
16
|
export declare function foldActivity(events: JournalEvent[], tasks: readonly ActivityTask[]): ActivitySnapshot;
|
|
17
|
+
/** Recorded evidence, not a probe of whether a subprocess is still alive. */
|
|
18
|
+
export type BuildActivity = {
|
|
19
|
+
state: "start-unrecorded" | "awaiting-command";
|
|
20
|
+
} | {
|
|
21
|
+
state: CommandReceipt["outcome"] | "unresolved";
|
|
22
|
+
receipt: CommandReceipt;
|
|
23
|
+
};
|
|
24
|
+
export interface TaskActivityProjection {
|
|
25
|
+
taskId: string;
|
|
26
|
+
/** Journal attempt label (zero based), absent until an attempt is recorded. */
|
|
27
|
+
attempt?: number;
|
|
28
|
+
state: "unconfirmed" | "preparing" | "implementing" | "returned-for-verification" | "validating" | "reviewing" | "merging" | "terminal";
|
|
29
|
+
/** Concurrent gates retain separate entries; state alone is only a summary. */
|
|
30
|
+
phases: {
|
|
31
|
+
gate: string;
|
|
32
|
+
state: "validating" | "reviewing";
|
|
33
|
+
}[];
|
|
34
|
+
build: BuildActivity;
|
|
35
|
+
}
|
|
36
|
+
/**
|
|
37
|
+
* Pure, evidence-only successor to foldActivity. Feed Journal.readTracked() and the owning run ID;
|
|
38
|
+
* sourceIndex supplies journal order, never wall time. Unattributed legacy rows belong to their
|
|
39
|
+
* tracked run and current attempt/round. They cannot prove identities absent from the journal.
|
|
40
|
+
* A new gates phase opens a round; completed gates cannot reopen within that round. No declared
|
|
41
|
+
* gate order, graph status, or successful verdict predicts a later phase.
|
|
42
|
+
*/
|
|
43
|
+
export declare function projectActivity(runId: string, rows: readonly TrackedJournalRow[], tasks: readonly ActivityTask[]): Map<string, TaskActivityProjection>;
|