tickmarkr 2.5.6 → 2.5.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/prompt.js +21 -1
- package/dist/cli/commands/approve.js +13 -6
- package/dist/cli/commands/compile.js +7 -0
- package/dist/cli/commands/plan.d.ts +5 -0
- package/dist/cli/commands/plan.js +28 -23
- package/dist/cli/commands/resume.js +1 -1
- package/dist/cli/commands/run.js +1 -1
- package/dist/cli/commands/status.js +95 -34
- package/dist/cli/commands/verify.js +6 -4
- package/dist/compile/native.js +68 -7
- package/dist/compile/retired-literals.d.ts +22 -0
- package/dist/compile/retired-literals.js +271 -0
- package/dist/drivers/index.d.ts +4 -2
- package/dist/drivers/index.js +54 -6
- package/dist/gates/baseline.d.ts +8 -3
- package/dist/gates/baseline.js +6 -3
- package/dist/gates/review.js +12 -1
- package/dist/gates/run-gates.d.ts +6 -1
- package/dist/gates/run-gates.js +67 -9
- package/dist/gates/test-manifest.d.ts +12 -0
- package/dist/gates/test-manifest.js +29 -3
- package/dist/gates/test-reporter.js +28 -2
- package/dist/graph/graph.d.ts +2 -0
- package/dist/graph/graph.js +5 -0
- package/dist/graph/schema.d.ts +28 -0
- package/dist/graph/schema.js +13 -1
- package/dist/run/activity.d.ts +28 -0
- package/dist/run/activity.js +194 -0
- package/dist/run/daemon.d.ts +28 -0
- package/dist/run/daemon.js +466 -110
- package/dist/run/git.d.ts +6 -1
- package/dist/run/git.js +63 -11
- package/dist/run/journal.d.ts +57 -5
- package/dist/run/journal.js +103 -9
- package/dist/run/merge.d.ts +2 -0
- package/dist/run/merge.js +1 -0
- package/dist/run/operator-page-summary.d.ts +56 -0
- package/dist/run/operator-page-summary.js +68 -0
- package/dist/run/operator-summary.d.ts +69 -0
- package/dist/run/operator-summary.js +77 -0
- package/dist/run/protocol.d.ts +71 -0
- package/dist/run/protocol.js +32 -0
- package/dist/run/repair-disposition.d.ts +41 -0
- package/dist/run/repair-disposition.js +77 -0
- package/dist/tui/cockpit/board.d.ts +9 -0
- package/dist/tui/cockpit/board.js +10 -0
- package/dist/tui/cockpit/derive.d.ts +35 -0
- package/dist/tui/cockpit/derive.js +152 -10
- package/dist/tui/cockpit/evidence-view.d.ts +2 -0
- package/dist/tui/cockpit/evidence-view.js +42 -12
- package/dist/tui/cockpit/home-view.js +45 -30
- package/dist/tui/cockpit/live-store.d.ts +18 -0
- package/dist/tui/cockpit/run-cockpit.d.ts +8 -1
- package/dist/tui/cockpit/run-cockpit.js +95 -1
- package/dist/tui/cockpit/run-view.d.ts +37 -0
- package/dist/tui/cockpit/run-view.js +189 -2
- package/package.json +1 -1
- package/schema/rungraph.schema.json +54 -0
- package/skills/tickmarkr-overseer/SKILL.md +162 -3
|
@@ -20,6 +20,15 @@ export interface TestReportCompletion {
|
|
|
20
20
|
failed: number;
|
|
21
21
|
skipped: number;
|
|
22
22
|
};
|
|
23
|
+
/** Bounded (4096 bytes) runner evidence per failed test — diff, actual/expected or stack head.
|
|
24
|
+
* Never part of the failure's identity: `failures` alone feeds details and fingerprints. */
|
|
25
|
+
evidence?: FailureEvidence[];
|
|
26
|
+
}
|
|
27
|
+
export interface FailureEvidence {
|
|
28
|
+
test: string;
|
|
29
|
+
text: string;
|
|
30
|
+
truncated?: true;
|
|
31
|
+
unavailable?: true;
|
|
23
32
|
}
|
|
24
33
|
/** The runner's own machine report — requested/started/completed are the runner's claims about ITSELF. */
|
|
25
34
|
export interface TestReport {
|
|
@@ -36,6 +45,9 @@ export interface TestReport {
|
|
|
36
45
|
certificate?: {
|
|
37
46
|
at: number;
|
|
38
47
|
exitCode: number;
|
|
48
|
+
/** Unhandled and module collection errors observed by the reporter; absent on older reports. */
|
|
49
|
+
errors?: number;
|
|
50
|
+
diagnostics?: string[];
|
|
39
51
|
};
|
|
40
52
|
}
|
|
41
53
|
/** Reads and structurally validates the report; a missing or malformed file is `undefined` — never a partial parse. */
|
|
@@ -103,6 +103,12 @@ export function toManifestPath(file, cwd) {
|
|
|
103
103
|
catch { /* cwd unreadable — best effort with the given path */ }
|
|
104
104
|
return relative(resolvedCwd, file).split(sep).join("/");
|
|
105
105
|
}
|
|
106
|
+
function isFailureEvidence(v) {
|
|
107
|
+
if (typeof v !== "object" || v === null)
|
|
108
|
+
return false;
|
|
109
|
+
const e = v;
|
|
110
|
+
return typeof e.test === "string" && typeof e.text === "string";
|
|
111
|
+
}
|
|
106
112
|
function isTestReportShape(v) {
|
|
107
113
|
if (typeof v !== "object" || v === null)
|
|
108
114
|
return false;
|
|
@@ -228,10 +234,12 @@ export function verifyManifestReport(opts) {
|
|
|
228
234
|
const failedCompletions = Object.entries(report.completed).filter(([, c]) => c.status === "failed");
|
|
229
235
|
const failingFiles = failedCompletions.map(([file]) => file).sort();
|
|
230
236
|
const failures = failedCompletions.flatMap(([, c]) => c.failures?.length ? c.failures : ["<unnamed failure>"]);
|
|
237
|
+
// Parsed defensively: a malformed entry is dropped, never a verdict change — evidence is diagnostics only.
|
|
238
|
+
const failureEvidence = failedCompletions.flatMap(([, c]) => Array.isArray(c.evidence) ? c.evidence.filter(isFailureEvidence) : []);
|
|
231
239
|
if (failures.length)
|
|
232
240
|
return { kind: "work", pass: false,
|
|
233
241
|
details: `test report names failing fingerprint(s):\n${failures.join("\n")}`,
|
|
234
|
-
meta: { classification: "regression", failingTests: failures, failingFiles, processExit: exitCode } };
|
|
242
|
+
meta: { classification: "regression", failingTests: failures, failingFiles, processExit: exitCode, failureEvidence } };
|
|
235
243
|
if (exitCode === undefined) {
|
|
236
244
|
return {
|
|
237
245
|
kind: "fail-closed",
|
|
@@ -448,11 +456,29 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
|
|
|
448
456
|
});
|
|
449
457
|
const verdict = verifyManifestReport({ manifest: files, nonce, exitCode: invoked.exitCode,
|
|
450
458
|
report: invoked.report, killedFile: invoked.killedFile, hangBudgetMs: invoked.hangBudgetMs });
|
|
459
|
+
// Preserve the validator's verdict and classification; runner evidence only explains it.
|
|
460
|
+
const stdoutPath = join(dir, `test-runner-stdout-${nonce}.log`);
|
|
461
|
+
const stderrPath = join(dir, `test-runner-stderr-${nonce}.log`);
|
|
462
|
+
const stdoutTail = Buffer.from(invoked.stdout).subarray(-16 * 1024);
|
|
463
|
+
const stderrTail = Buffer.from(invoked.stderr).subarray(-16 * 1024);
|
|
464
|
+
writeFileSync(stdoutPath, stdoutTail);
|
|
465
|
+
writeFileSync(stderrPath, stderrTail);
|
|
466
|
+
const report = invoked.report?.nonce === nonce ? invoked.report : undefined;
|
|
467
|
+
const neverStarted = report ? files.filter(file => !(file in report.started)).length : "unknown";
|
|
468
|
+
const errors = report?.certificate?.errors;
|
|
469
|
+
const reporterErrors = typeof errors === "number" && Number.isInteger(errors) && errors >= 0 ? errors : "unknown";
|
|
470
|
+
const reportedDiagnostics = report?.certificate?.diagnostics;
|
|
471
|
+
const runnerErrors = Array.isArray(reportedDiagnostics) ? reportedDiagnostics.filter(error => typeof error === "string") : [];
|
|
472
|
+
const diagnostics = !verdict.pass
|
|
473
|
+
? `\nclassification: ${verdict.meta.classification ?? "unknown"}; runner-level diagnostic: never-started ${neverStarted}; reporter errors ${reporterErrors}`
|
|
474
|
+
+ [...runnerErrors,
|
|
475
|
+
stdoutTail.toString(), stderrTail.toString()].filter(Boolean).map(text => `\n${text}`).join("")
|
|
476
|
+
: "";
|
|
451
477
|
return { pass: verdict.pass, kind: verdict.kind,
|
|
452
|
-
details: verdict.details +
|
|
478
|
+
details: verdict.details + diagnostics,
|
|
453
479
|
classification: verdict.meta.classification,
|
|
454
480
|
meta: { ...verdict.meta, nonce, manifest: files, manifestPath, listingCommand: invocation.listing, verification,
|
|
455
|
-
spawnedCommand, processExit: invoked.exitCode, pid: invoked.pid },
|
|
481
|
+
spawnedCommand, processExit: invoked.exitCode, pid: invoked.pid, stdoutPath, stderrPath },
|
|
456
482
|
exitCode: invoked.exitCode ?? -1, reportPath };
|
|
457
483
|
}
|
|
458
484
|
catch (error) {
|
|
@@ -11,6 +11,24 @@ export default class TickmarkrReporter {
|
|
|
11
11
|
this.cwd = realpathSync(process.cwd());
|
|
12
12
|
}
|
|
13
13
|
file(module) { return relative(this.cwd, module.moduleId).split(sep).join('/'); }
|
|
14
|
+
// Bounded assertion evidence, kept BESIDE the failure fingerprint and never inside it: the
|
|
15
|
+
// fingerprint is the failure's identity (the repeated-failure cap compares it), the evidence is
|
|
16
|
+
// what the runner knew and the message elided. ONE 4096-byte budget per failed test, shared by
|
|
17
|
+
// all of its errors (expect.soft yields several); never invented.
|
|
18
|
+
evidence(test, errors) {
|
|
19
|
+
const parts = [];
|
|
20
|
+
for (const e of errors) {
|
|
21
|
+
if (e && typeof e.diff === 'string' && e.diff) parts.push(e.diff);
|
|
22
|
+
else for (const k of ['actual', 'expected']) if (e && e[k] !== undefined && e[k] !== null) parts.push(k + ': ' + (typeof e[k] === 'string' ? e[k] : JSON.stringify(e[k])));
|
|
23
|
+
if (e && typeof e.stack === 'string' && e.stack) parts.push(e.stack.split('\n').slice(0, 8).join('\n'));
|
|
24
|
+
}
|
|
25
|
+
if (!parts.length) return { test, text: '', unavailable: true };
|
|
26
|
+
const full = parts.join('\n').replace(/\x1b\[[0-9;]*m/g, '');
|
|
27
|
+
if (Buffer.byteLength(full) <= 4096) return { test, text: full };
|
|
28
|
+
let text = full.slice(0, 4096);
|
|
29
|
+
while (Buffer.byteLength(text) > 4096) text = text.slice(0, -1);
|
|
30
|
+
return { test, text, truncated: true };
|
|
31
|
+
}
|
|
14
32
|
save() {
|
|
15
33
|
writeFileSync(this.path + '.tmp', JSON.stringify(this.report));
|
|
16
34
|
renameSync(this.path + '.tmp', this.path);
|
|
@@ -27,6 +45,7 @@ export default class TickmarkrReporter {
|
|
|
27
45
|
const file = this.file(module);
|
|
28
46
|
const failed = module.state() === 'failed';
|
|
29
47
|
const failures = [];
|
|
48
|
+
const evidence = [];
|
|
30
49
|
// R41: count test bodies by their own state so a module whose every test was skipped (a
|
|
31
50
|
// describe.skipIf gate) is recorded as SKIPPED — present in the lifecycle, but never as
|
|
32
51
|
// executed test-body success. 'passed'/'failed' executed; anything else did not run.
|
|
@@ -37,6 +56,8 @@ export default class TickmarkrReporter {
|
|
|
37
56
|
tests.failed++;
|
|
38
57
|
if (failed) {
|
|
39
58
|
const errors = test.result().errors || [];
|
|
59
|
+
const name = file + ' > ' + test.fullName;
|
|
60
|
+
evidence.push(this.evidence(name, errors));
|
|
40
61
|
failures.push(...(errors.length ? errors.map(e => 'FAIL ' + file + ' > ' + test.fullName + ': ' + e.message) : ['FAIL ' + file + ' > ' + test.fullName]));
|
|
41
62
|
}
|
|
42
63
|
} else if (state === 'passed') tests.passed++;
|
|
@@ -45,12 +66,17 @@ export default class TickmarkrReporter {
|
|
|
45
66
|
if (failed && !failures.length) failures.push('FAIL ' + file);
|
|
46
67
|
const status = failed ? 'failed' : tests.passed + tests.failed === 0 ? 'skipped' : 'passed';
|
|
47
68
|
if (file in this.report.completed) this.report.duplicateCompletions.push(file);
|
|
48
|
-
this.report.completed[file] = { at: Date.now(), status, failures, tests };
|
|
69
|
+
this.report.completed[file] = { at: Date.now(), status, failures, tests, ...(evidence.length ? { evidence } : {}) };
|
|
49
70
|
this.save();
|
|
50
71
|
}
|
|
51
72
|
onTestRunEnd(modules, errors, reason) {
|
|
52
73
|
const failed = reason === 'failed' || errors.length > 0 || Object.values(this.report.completed).some(c => c.status === 'failed');
|
|
53
|
-
|
|
74
|
+
// Collection failures can reach run end without a module start/end event. Keep their
|
|
75
|
+
// identity as diagnostics, without inventing lifecycle records or changing the verdict.
|
|
76
|
+
const loadErrors = modules.flatMap(module => module.errors().map(e => this.file(module) + ': ' + e.message));
|
|
77
|
+
this.report.certificate = { at: Date.now(), exitCode: failed ? 1 : 0,
|
|
78
|
+
errors: errors.length + loadErrors.length,
|
|
79
|
+
diagnostics: [...loadErrors, ...errors.map(e => [e.testPath, e.name, e.message].filter(Boolean).join(': '))] };
|
|
54
80
|
this.save();
|
|
55
81
|
}
|
|
56
82
|
}
|
package/dist/graph/graph.d.ts
CHANGED
|
@@ -21,6 +21,8 @@ export interface OnDiskSpecHash {
|
|
|
21
21
|
* evidence for failed recompiles, while moved/deleted source files need not strand a compiled graph.
|
|
22
22
|
*/
|
|
23
23
|
export declare function onDiskSpecHash(_repoRoot: string, graph: RunGraph): OnDiskSpecHash | undefined;
|
|
24
|
+
/** A task's definition minus its files[] (and run state): what a scope amendment must NOT move (OBS-1073). */
|
|
25
|
+
export declare function taskDefinitionFingerprint(task: Task): string;
|
|
24
26
|
export declare function graphDefinitionHash(g: RunGraph): string;
|
|
25
27
|
export declare function taskContentDigest(task: Pick<Task, "goal" | "files" | "acceptance">): string;
|
|
26
28
|
export declare function tickmarkrDir(repoRoot: string): string;
|
package/dist/graph/graph.js
CHANGED
|
@@ -80,6 +80,11 @@ export function onDiskSpecHash(_repoRoot, graph) {
|
|
|
80
80
|
// single comparator in journal.ts (engagementComparable) so the journal↔graph join is decided once.
|
|
81
81
|
// ponytail: sha256 truncated to 16 hex — stable, grep-friendly; promote to full digest only if a
|
|
82
82
|
// collision ever bites (engagement ids are not a trust boundary, collisions just force a re-run).
|
|
83
|
+
/** A task's definition minus its files[] (and run state): what a scope amendment must NOT move (OBS-1073). */
|
|
84
|
+
export function taskDefinitionFingerprint(task) {
|
|
85
|
+
const { status: _status, evidence: _evidence, files: _files, ...def } = task;
|
|
86
|
+
return createHash("sha256").update(JSON.stringify(def)).digest("hex").slice(0, 16);
|
|
87
|
+
}
|
|
83
88
|
export function graphDefinitionHash(g) {
|
|
84
89
|
const definitions = g.tasks.map(({ status: _status, evidence: _evidence, ...def }) => def);
|
|
85
90
|
return createHash("sha256").update(JSON.stringify({ version: g.version, spec: g.spec, tasks: definitions })).digest("hex").slice(0, 16);
|
package/dist/graph/schema.d.ts
CHANGED
|
@@ -19,12 +19,22 @@ export declare const AcceptanceItemSchema: z.ZodUnion<readonly [z.ZodString, z.Z
|
|
|
19
19
|
oracle: z.ZodLiteral<"test">;
|
|
20
20
|
test: z.ZodString;
|
|
21
21
|
text: z.ZodOptional<z.ZodString>;
|
|
22
|
+
landing: z.ZodOptional<z.ZodString>;
|
|
22
23
|
}, z.core.$strip>, z.ZodObject<{
|
|
23
24
|
oracle: z.ZodLiteral<"judge">;
|
|
24
25
|
text: z.ZodString;
|
|
25
26
|
}, z.core.$strip>]>;
|
|
26
27
|
export type AcceptanceItem = z.infer<typeof AcceptanceItemSchema>;
|
|
27
28
|
export declare function renderAcceptanceItem(item: AcceptanceItem): string;
|
|
29
|
+
export declare const PinSchema: z.ZodDiscriminatedUnion<[z.ZodObject<{
|
|
30
|
+
kind: z.ZodLiteral<"literal">;
|
|
31
|
+
text: z.ZodString;
|
|
32
|
+
glob: z.ZodString;
|
|
33
|
+
}, z.core.$strip>, z.ZodObject<{
|
|
34
|
+
kind: z.ZodLiteral<"fixture">;
|
|
35
|
+
paths: z.ZodArray<z.ZodString>;
|
|
36
|
+
}, z.core.$strip>], "kind">;
|
|
37
|
+
export type Pin = z.infer<typeof PinSchema>;
|
|
28
38
|
export declare const TaskSchema: z.ZodObject<{
|
|
29
39
|
id: z.ZodString;
|
|
30
40
|
title: z.ZodString;
|
|
@@ -52,6 +62,7 @@ export declare const TaskSchema: z.ZodObject<{
|
|
|
52
62
|
oracle: z.ZodLiteral<"test">;
|
|
53
63
|
test: z.ZodString;
|
|
54
64
|
text: z.ZodOptional<z.ZodString>;
|
|
65
|
+
landing: z.ZodOptional<z.ZodString>;
|
|
55
66
|
}, z.core.$strip>, z.ZodObject<{
|
|
56
67
|
oracle: z.ZodLiteral<"judge">;
|
|
57
68
|
text: z.ZodString;
|
|
@@ -61,6 +72,14 @@ export declare const TaskSchema: z.ZodObject<{
|
|
|
61
72
|
confidence: z.ZodOptional<z.ZodNumber>;
|
|
62
73
|
reason: z.ZodOptional<z.ZodString>;
|
|
63
74
|
}, z.core.$strip>>>;
|
|
75
|
+
pins: z.ZodOptional<z.ZodArray<z.ZodDiscriminatedUnion<[z.ZodObject<{
|
|
76
|
+
kind: z.ZodLiteral<"literal">;
|
|
77
|
+
text: z.ZodString;
|
|
78
|
+
glob: z.ZodString;
|
|
79
|
+
}, z.core.$strip>, z.ZodObject<{
|
|
80
|
+
kind: z.ZodLiteral<"fixture">;
|
|
81
|
+
paths: z.ZodArray<z.ZodString>;
|
|
82
|
+
}, z.core.$strip>], "kind">>>;
|
|
64
83
|
gates: z.ZodDefault<z.ZodArray<z.ZodEnum<{
|
|
65
84
|
build: "build";
|
|
66
85
|
test: "test";
|
|
@@ -144,6 +163,7 @@ export declare const RunGraphSchema: z.ZodObject<{
|
|
|
144
163
|
oracle: z.ZodLiteral<"test">;
|
|
145
164
|
test: z.ZodString;
|
|
146
165
|
text: z.ZodOptional<z.ZodString>;
|
|
166
|
+
landing: z.ZodOptional<z.ZodString>;
|
|
147
167
|
}, z.core.$strip>, z.ZodObject<{
|
|
148
168
|
oracle: z.ZodLiteral<"judge">;
|
|
149
169
|
text: z.ZodString;
|
|
@@ -153,6 +173,14 @@ export declare const RunGraphSchema: z.ZodObject<{
|
|
|
153
173
|
confidence: z.ZodOptional<z.ZodNumber>;
|
|
154
174
|
reason: z.ZodOptional<z.ZodString>;
|
|
155
175
|
}, z.core.$strip>>>;
|
|
176
|
+
pins: z.ZodOptional<z.ZodArray<z.ZodDiscriminatedUnion<[z.ZodObject<{
|
|
177
|
+
kind: z.ZodLiteral<"literal">;
|
|
178
|
+
text: z.ZodString;
|
|
179
|
+
glob: z.ZodString;
|
|
180
|
+
}, z.core.$strip>, z.ZodObject<{
|
|
181
|
+
kind: z.ZodLiteral<"fixture">;
|
|
182
|
+
paths: z.ZodArray<z.ZodString>;
|
|
183
|
+
}, z.core.$strip>], "kind">>>;
|
|
156
184
|
gates: z.ZodDefault<z.ZodArray<z.ZodEnum<{
|
|
157
185
|
build: "build";
|
|
158
186
|
test: "test";
|
package/dist/graph/schema.js
CHANGED
|
@@ -19,7 +19,9 @@ export const ORACLES = ["command", "test", "judge"];
|
|
|
19
19
|
export const AcceptanceItemSchema = z.union([
|
|
20
20
|
z.string().min(1),
|
|
21
21
|
z.object({ oracle: z.literal("command"), command: z.string().min(1), text: z.string().min(1).optional() }),
|
|
22
|
-
|
|
22
|
+
// v2.5.8 T8 (OBS-1064): landing = the collectable suite path a test criterion lands in, declared
|
|
23
|
+
// beside the verbatim title (never inside it). Optional; compile enforces the collectable glob.
|
|
24
|
+
z.object({ oracle: z.literal("test"), test: z.string().min(1), text: z.string().min(1).optional(), landing: z.string().min(1).optional() }),
|
|
23
25
|
z.object({ oracle: z.literal("judge"), text: z.string().min(1) }),
|
|
24
26
|
]);
|
|
25
27
|
// Shared text rendering of one acceptance item — every consumer (worker prompt, acceptance gate,
|
|
@@ -33,6 +35,15 @@ export function renderAcceptanceItem(item) {
|
|
|
33
35
|
return item.text ?? `test: ${item.test}`;
|
|
34
36
|
return item.text; // judge — bare text, byte-identical to a plain-string judge criterion
|
|
35
37
|
}
|
|
38
|
+
// v2.5.8 T7 (agreement C2): declared pin obligations — a LIMITED AUTHORING CONTRACT, not an assertion
|
|
39
|
+
// analyzer. literal = an exact text plus the glob it is pinned in (every matching file holding the text
|
|
40
|
+
// is obligated); fixture = a path set whose every matching file is itself obligated (byte-pinned output
|
|
41
|
+
// the change will move) and which carries no literal text. Declared here because z.object strips
|
|
42
|
+
// undeclared keys on every load.
|
|
43
|
+
export const PinSchema = z.discriminatedUnion("kind", [
|
|
44
|
+
z.object({ kind: z.literal("literal"), text: z.string().min(1), glob: z.string().min(1) }),
|
|
45
|
+
z.object({ kind: z.literal("fixture"), paths: z.array(z.string().min(1)).min(1) }),
|
|
46
|
+
]);
|
|
36
47
|
export const TaskSchema = z.object({
|
|
37
48
|
// ids land in git branch names and herdr pane names — branch-safe characters only, bounded
|
|
38
49
|
// length (refs hit filesystem limits), and never "--" (the task-branch separator, locked
|
|
@@ -59,6 +70,7 @@ export const TaskSchema = z.object({
|
|
|
59
70
|
reason: z.string().optional(),
|
|
60
71
|
}))
|
|
61
72
|
.optional(),
|
|
73
|
+
pins: z.array(PinSchema).optional(),
|
|
62
74
|
gates: z
|
|
63
75
|
.array(z.enum(GATE_NAMES))
|
|
64
76
|
.default(["build", "test", "lint", "evidence", "scope", "acceptance", "review"])
|
package/dist/run/activity.d.ts
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { type JournalEvent } from "./journal.js";
|
|
2
|
+
import { type CommandReceipt, type TrackedJournalRow } from "./protocol.js";
|
|
2
3
|
export interface ActivityTask {
|
|
3
4
|
id: string;
|
|
4
5
|
gates: readonly string[];
|
|
@@ -13,3 +14,30 @@ export interface ActivitySnapshot {
|
|
|
13
14
|
cells: Map<string, string>;
|
|
14
15
|
}
|
|
15
16
|
export declare function foldActivity(events: JournalEvent[], tasks: readonly ActivityTask[]): ActivitySnapshot;
|
|
17
|
+
/** Recorded evidence, not a probe of whether a subprocess is still alive. */
|
|
18
|
+
export type BuildActivity = {
|
|
19
|
+
state: "start-unrecorded" | "awaiting-command";
|
|
20
|
+
} | {
|
|
21
|
+
state: CommandReceipt["outcome"] | "unresolved";
|
|
22
|
+
receipt: CommandReceipt;
|
|
23
|
+
};
|
|
24
|
+
export interface TaskActivityProjection {
|
|
25
|
+
taskId: string;
|
|
26
|
+
/** Journal attempt label (zero based), absent until an attempt is recorded. */
|
|
27
|
+
attempt?: number;
|
|
28
|
+
state: "unconfirmed" | "preparing" | "implementing" | "returned-for-verification" | "validating" | "reviewing" | "merging" | "terminal";
|
|
29
|
+
/** Concurrent gates retain separate entries; state alone is only a summary. */
|
|
30
|
+
phases: {
|
|
31
|
+
gate: string;
|
|
32
|
+
state: "validating" | "reviewing";
|
|
33
|
+
}[];
|
|
34
|
+
build: BuildActivity;
|
|
35
|
+
}
|
|
36
|
+
/**
|
|
37
|
+
* Pure, evidence-only successor to foldActivity. Feed Journal.readTracked() and the owning run ID;
|
|
38
|
+
* sourceIndex supplies journal order, never wall time. Unattributed legacy rows belong to their
|
|
39
|
+
* tracked run and current attempt/round. They cannot prove identities absent from the journal.
|
|
40
|
+
* A new gates phase opens a round; completed gates cannot reopen within that round. No declared
|
|
41
|
+
* gate order, graph status, or successful verdict predicts a later phase.
|
|
42
|
+
*/
|
|
43
|
+
export declare function projectActivity(runId: string, rows: readonly TrackedJournalRow[], tasks: readonly ActivityTask[]): Map<string, TaskActivityProjection>;
|
package/dist/run/activity.js
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { formatJournalNarration } from "./journal.js";
|
|
2
|
+
import { readCommandReceipt } from "./protocol.js";
|
|
2
3
|
const channelOf = (assignment) => {
|
|
3
4
|
const a = assignment;
|
|
4
5
|
return typeof a?.adapter === "string" && typeof a.model === "string" ? `${a.adapter}:${a.model}` : "unknown channel";
|
|
@@ -94,3 +95,196 @@ export function foldActivity(events, tasks) {
|
|
|
94
95
|
const last = events.at(-1);
|
|
95
96
|
return { ...(last ? { now: formatJournalNarration(last) } : {}), cells };
|
|
96
97
|
}
|
|
98
|
+
const activityRecord = (value) => typeof value === "object" && value !== null && !Array.isArray(value);
|
|
99
|
+
const activityOrdinal = (value) => typeof value === "number" && Number.isInteger(value) && value >= 0 ? value : undefined;
|
|
100
|
+
/**
|
|
101
|
+
* Pure, evidence-only successor to foldActivity. Feed Journal.readTracked() and the owning run ID;
|
|
102
|
+
* sourceIndex supplies journal order, never wall time. Unattributed legacy rows belong to their
|
|
103
|
+
* tracked run and current attempt/round. They cannot prove identities absent from the journal.
|
|
104
|
+
* A new gates phase opens a round; completed gates cannot reopen within that round. No declared
|
|
105
|
+
* gate order, graph status, or successful verdict predicts a later phase.
|
|
106
|
+
*/
|
|
107
|
+
export function projectActivity(runId, rows, tasks) {
|
|
108
|
+
const states = new Map(tasks.map((task) => [task.id, {
|
|
109
|
+
projection: { taskId: task.id, state: "unconfirmed", phases: [], build: { state: "start-unrecorded" } },
|
|
110
|
+
round: 0,
|
|
111
|
+
closed: new Set(),
|
|
112
|
+
invocations: new Set(),
|
|
113
|
+
workerReturned: false,
|
|
114
|
+
suspended: false,
|
|
115
|
+
redispatch: false,
|
|
116
|
+
dispatchedSinceRound: false,
|
|
117
|
+
adoptRoundAttempt: false,
|
|
118
|
+
}]));
|
|
119
|
+
for (const row of [...rows].sort((a, b) => a.sourceIndex - b.sourceIndex)) {
|
|
120
|
+
if (row.runId !== runId || row.kind === "protocol-issue")
|
|
121
|
+
continue;
|
|
122
|
+
// Compatibility readers normalize missing legacy attempts to zero. Retain the physical
|
|
123
|
+
// payload here so that absence still means the currently recorded attempt.
|
|
124
|
+
const raw = row.raw;
|
|
125
|
+
if (!activityRecord(raw) || typeof raw.event !== "string" || !activityRecord(raw.data))
|
|
126
|
+
continue;
|
|
127
|
+
const data = raw.data;
|
|
128
|
+
if (typeof data.runId === "string" && data.runId !== runId)
|
|
129
|
+
continue;
|
|
130
|
+
if (raw.event === "run-resume" || raw.event === "run-end") {
|
|
131
|
+
for (const st of states.values()) {
|
|
132
|
+
const p = st.projection;
|
|
133
|
+
if (p.build.state === "started" && "receipt" in p.build) {
|
|
134
|
+
p.build = { state: "unresolved", receipt: p.build.receipt };
|
|
135
|
+
}
|
|
136
|
+
for (const phase of p.phases)
|
|
137
|
+
st.closed.add(phase.gate);
|
|
138
|
+
p.phases = [];
|
|
139
|
+
if (p.state !== "terminal")
|
|
140
|
+
p.state = "unconfirmed";
|
|
141
|
+
st.suspended = true;
|
|
142
|
+
st.redispatch = raw.event === "run-resume";
|
|
143
|
+
st.dispatchedSinceRound = false;
|
|
144
|
+
st.adoptRoundAttempt = false;
|
|
145
|
+
}
|
|
146
|
+
continue;
|
|
147
|
+
}
|
|
148
|
+
if (typeof raw.taskId !== "string")
|
|
149
|
+
continue;
|
|
150
|
+
const st = states.get(raw.taskId);
|
|
151
|
+
if (!st)
|
|
152
|
+
continue;
|
|
153
|
+
const p = st.projection;
|
|
154
|
+
const attempt = activityOrdinal(data.attempt);
|
|
155
|
+
if (attempt !== undefined && p.attempt !== undefined && attempt < p.attempt)
|
|
156
|
+
continue;
|
|
157
|
+
if (raw.event === "task-dispatch") {
|
|
158
|
+
const nextAttempt = attempt ?? 0;
|
|
159
|
+
if (p.attempt === nextAttempt && !st.redispatch)
|
|
160
|
+
continue;
|
|
161
|
+
p.attempt = nextAttempt;
|
|
162
|
+
p.state = "preparing";
|
|
163
|
+
p.phases = [];
|
|
164
|
+
p.build = { state: "start-unrecorded" };
|
|
165
|
+
st.closed.clear();
|
|
166
|
+
st.workerReturned = false;
|
|
167
|
+
st.suspended = false;
|
|
168
|
+
st.redispatch = false;
|
|
169
|
+
st.dispatchedSinceRound = true;
|
|
170
|
+
st.adoptRoundAttempt = false;
|
|
171
|
+
continue;
|
|
172
|
+
}
|
|
173
|
+
const round = activityOrdinal(data.gateRound);
|
|
174
|
+
if (round !== undefined && round !== st.round)
|
|
175
|
+
continue;
|
|
176
|
+
const roundGate = (raw.event === "gate-result" || raw.event === "gate-phase-start"
|
|
177
|
+
|| raw.event === "phase-start") && typeof data.gate === "string"
|
|
178
|
+
&& !st.suspended && !st.closed.has(data.gate);
|
|
179
|
+
if (attempt !== undefined && p.attempt !== undefined && attempt !== p.attempt
|
|
180
|
+
&& !(st.adoptRoundAttempt && roundGate))
|
|
181
|
+
continue;
|
|
182
|
+
if (st.adoptRoundAttempt && roundGate && attempt !== undefined) {
|
|
183
|
+
p.attempt = attempt;
|
|
184
|
+
st.adoptRoundAttempt = false;
|
|
185
|
+
}
|
|
186
|
+
if (p.attempt === undefined && attempt !== undefined)
|
|
187
|
+
p.attempt = attempt;
|
|
188
|
+
if (["task-done", "task-failed", "task-human", "task-approved", "merge"].includes(raw.event)) {
|
|
189
|
+
p.state = "terminal";
|
|
190
|
+
p.phases = [];
|
|
191
|
+
st.suspended = true;
|
|
192
|
+
continue;
|
|
193
|
+
}
|
|
194
|
+
// Only an explicit new battery may re-enter verification after a resume or completed task.
|
|
195
|
+
if (raw.event === "phase-start" && data.phase === "gates") {
|
|
196
|
+
st.round++;
|
|
197
|
+
// Resume verification has no dispatch and labels the round with the dispatch count,
|
|
198
|
+
// rather than the last worker's attempt. Its first attributed gate/receipt owns the label.
|
|
199
|
+
st.adoptRoundAttempt = !st.dispatchedSinceRound;
|
|
200
|
+
st.dispatchedSinceRound = false;
|
|
201
|
+
st.closed.clear();
|
|
202
|
+
p.phases = [];
|
|
203
|
+
p.build = { state: "start-unrecorded" };
|
|
204
|
+
if (p.state === "terminal")
|
|
205
|
+
p.state = "unconfirmed";
|
|
206
|
+
st.suspended = false;
|
|
207
|
+
continue;
|
|
208
|
+
}
|
|
209
|
+
if (raw.event === "build-receipt" || raw.event === "build-result") {
|
|
210
|
+
// The emitter adds gate/reason metadata beside the strict command receipt.
|
|
211
|
+
const { gate: _gate, reason: _reason, freshBuildRan: _fresh, ...payload } = data;
|
|
212
|
+
const parsed = readCommandReceipt(payload);
|
|
213
|
+
if (parsed.kind !== "receipt")
|
|
214
|
+
continue;
|
|
215
|
+
const receipt = parsed.receipt;
|
|
216
|
+
if (receipt.outcome === "started" && !receipt.confirmedStart)
|
|
217
|
+
continue;
|
|
218
|
+
const identity = receipt.attribution;
|
|
219
|
+
if (identity.runId !== runId || identity.taskId !== p.taskId
|
|
220
|
+
|| identity.gateRound !== st.round)
|
|
221
|
+
continue;
|
|
222
|
+
if (identity.attempt < (p.attempt ?? 0)
|
|
223
|
+
|| (!st.adoptRoundAttempt && identity.attempt !== (p.attempt ?? 0)))
|
|
224
|
+
continue;
|
|
225
|
+
const previous = "receipt" in p.build ? p.build.receipt : undefined;
|
|
226
|
+
const same = previous?.attribution.invocation === identity.invocation;
|
|
227
|
+
if (same) {
|
|
228
|
+
if (p.build.state !== "started" && p.build.state !== "unresolved")
|
|
229
|
+
continue;
|
|
230
|
+
if (receipt.outcome === "started")
|
|
231
|
+
continue;
|
|
232
|
+
}
|
|
233
|
+
else {
|
|
234
|
+
if (st.suspended || st.closed.has("build") || st.invocations.has(identity.invocation))
|
|
235
|
+
continue;
|
|
236
|
+
// A late terminal cannot displace a newer invocation. No-start outcomes are themselves
|
|
237
|
+
// complete receipts and need no preceding start; a first terminal remains useful evidence.
|
|
238
|
+
if (previous && receipt.confirmedStart && receipt.outcome !== "started")
|
|
239
|
+
continue;
|
|
240
|
+
}
|
|
241
|
+
st.invocations.add(identity.invocation);
|
|
242
|
+
if (st.adoptRoundAttempt) {
|
|
243
|
+
p.attempt = identity.attempt;
|
|
244
|
+
st.adoptRoundAttempt = false;
|
|
245
|
+
}
|
|
246
|
+
p.build = { state: receipt.outcome, receipt };
|
|
247
|
+
continue;
|
|
248
|
+
}
|
|
249
|
+
if (st.suspended)
|
|
250
|
+
continue;
|
|
251
|
+
if (raw.event === "worker-launch" && !st.workerReturned && p.phases.length === 0
|
|
252
|
+
&& (p.state === "preparing" || p.state === "unconfirmed"))
|
|
253
|
+
p.state = "implementing";
|
|
254
|
+
if (raw.event === "worker-result" && !st.workerReturned && p.phases.length === 0) {
|
|
255
|
+
st.workerReturned = true;
|
|
256
|
+
p.state = "returned-for-verification";
|
|
257
|
+
}
|
|
258
|
+
if (raw.event === "phase-start" && data.phase === "merge") {
|
|
259
|
+
for (const phase of p.phases)
|
|
260
|
+
st.closed.add(phase.gate);
|
|
261
|
+
p.phases = [];
|
|
262
|
+
p.state = "merging";
|
|
263
|
+
st.workerReturned = true;
|
|
264
|
+
continue;
|
|
265
|
+
}
|
|
266
|
+
if (raw.event === "gate-result" && typeof data.gate === "string") {
|
|
267
|
+
st.workerReturned = true;
|
|
268
|
+
// Infrastructure skips have no verdict; a retry may start again in this round.
|
|
269
|
+
if (typeof data.pass === "boolean" || (data.skipped !== true && data.infra !== true))
|
|
270
|
+
st.closed.add(data.gate);
|
|
271
|
+
if (data.gate === "build" && p.build.state === "awaiting-command")
|
|
272
|
+
p.build = { state: "start-unrecorded" };
|
|
273
|
+
p.phases = p.phases.filter((phase) => phase.gate !== data.gate);
|
|
274
|
+
if (p.state !== "merging")
|
|
275
|
+
p.state = p.phases[0]?.state ?? "unconfirmed";
|
|
276
|
+
}
|
|
277
|
+
if ((raw.event === "phase-start" || raw.event === "gate-phase-start")
|
|
278
|
+
&& typeof data.gate === "string" && !st.closed.has(data.gate) && p.state !== "merging") {
|
|
279
|
+
const gate = data.gate;
|
|
280
|
+
if (!p.phases.some((phase) => phase.gate === gate)) {
|
|
281
|
+
p.phases.push({ gate, state: gate === "review" ? "reviewing" : "validating" });
|
|
282
|
+
}
|
|
283
|
+
st.workerReturned = true;
|
|
284
|
+
p.state = p.phases[0].state;
|
|
285
|
+
if (gate === "build" && p.build.state === "start-unrecorded")
|
|
286
|
+
p.build = { state: "awaiting-command" };
|
|
287
|
+
}
|
|
288
|
+
}
|
|
289
|
+
return new Map([...states].map(([id, st]) => [id, st.projection]));
|
|
290
|
+
}
|
package/dist/run/daemon.d.ts
CHANGED
|
@@ -4,8 +4,18 @@ import { type DriverChoice } from "../drivers/index.js";
|
|
|
4
4
|
import { type ExecutorDriver, type Slot } from "../drivers/types.js";
|
|
5
5
|
import { type Baseline } from "../gates/baseline.js";
|
|
6
6
|
import type { GateResult } from "../gates/types.js";
|
|
7
|
+
import { type RunGraph } from "../graph/schema.js";
|
|
7
8
|
import { Journal, type JournalEvent } from "./journal.js";
|
|
8
9
|
export declare function closeLiveSlot(liveSlots: Set<Slot>, driver: Pick<ExecutorDriver, "close">, slot: Slot): Promise<void>;
|
|
10
|
+
/** Optional authoritative transport receipt. Screen text can prove execution, never nonacceptance. */
|
|
11
|
+
export interface DispatchObservation {
|
|
12
|
+
slotId: string;
|
|
13
|
+
command: string;
|
|
14
|
+
dispatchId: string;
|
|
15
|
+
outcome: "accepted" | "not-accepted";
|
|
16
|
+
authoritative: true;
|
|
17
|
+
}
|
|
18
|
+
export declare function heldWorkerTransport(driver: ExecutorDriver, slot: Slot, dispatchId: string, held: (data: Record<string, unknown>) => void, sleep?: (ms: number) => Promise<void>): Pick<ExecutorDriver, "run" | "read" | "status" | "waitOutput" | "waitAgentStatus">;
|
|
9
19
|
export declare function setAttemptHardTimeoutMsForTests(ms: number): void;
|
|
10
20
|
export declare function resetAttemptHardTimeoutMsForTests(): void;
|
|
11
21
|
export { harvestCpuFlatWindowMs, resetHarvestCpuFlatMsForTests, setHarvestCpuFlatMsForTests, workerTreeCpuMs, } from "./stall.js";
|
|
@@ -49,6 +59,8 @@ export interface RunSummary {
|
|
|
49
59
|
pending: string[];
|
|
50
60
|
blocked: string[];
|
|
51
61
|
tipVerify?: "passed" | "failed";
|
|
62
|
+
/** additive beside the legacy tipVerify enum: HOW the close's latest verification cycle earned its verdict */
|
|
63
|
+
tipProof?: TipProof;
|
|
52
64
|
lastMergedTask?: string;
|
|
53
65
|
/** T14: did every approval this run accepted actually get enacted, or did the run end over one? */
|
|
54
66
|
approvalDisposition?: "complete" | "outstanding";
|
|
@@ -156,6 +168,21 @@ export declare function resetHarvestSilentMsForTests(): void;
|
|
|
156
168
|
export declare const HARVESTED_RESULT_SUMMARY = "harvested: the worktree carries committed work; the worker emitted no TICKMARKR_RESULT trailer";
|
|
157
169
|
/** T4 (OBS-266): identity of the command SET a tip verify ran — a changed command is a different verify. */
|
|
158
170
|
export declare function commandsHash(commands: Record<string, string>): string;
|
|
171
|
+
export interface TipProof {
|
|
172
|
+
kind: "fresh" | "reused" | "failed" | "incomplete";
|
|
173
|
+
/** the commit the cycle's start row spoke for */
|
|
174
|
+
tip?: string;
|
|
175
|
+
}
|
|
176
|
+
/**
|
|
177
|
+
* OBS-1077 close rider: what the engagement's LATEST verification cycle proved — its
|
|
178
|
+
* `tip-verify-start` row and what followed, never a commit comparison. Exactly one kind per close:
|
|
179
|
+
* fresh (every tip command ran AND passed), reused (an eligible cached cycle carried forward),
|
|
180
|
+
* failed, or incomplete (cancelled, cut short, mixed or undelimited). Nothing before the start row
|
|
181
|
+
* is read, so an unfinished cycle inherits nothing from an earlier green one.
|
|
182
|
+
*/
|
|
183
|
+
export declare function runEndTipProof(events: readonly JournalEvent[]): TipProof;
|
|
184
|
+
/** The close notification's statement of the proof — one clause per kind. */
|
|
185
|
+
export declare function formatTipProof(p: TipProof): string;
|
|
159
186
|
/**
|
|
160
187
|
* OBS-34's strict tip verify, but it stops re-paying for an unmoved tip (~334m corpus-wide; 69.5m in
|
|
161
188
|
* one park-heavy run whose 18 resume cycles merged nothing new). The verify journals the SHA it
|
|
@@ -185,4 +212,5 @@ export declare function liveSuiteCount(repoRoot: string): Promise<number>;
|
|
|
185
212
|
/** Test seam — exercise the production observer's total read bound with a small real tree. */
|
|
186
213
|
export declare function setObserveBudgetBytesForTests(bytes: number): void;
|
|
187
214
|
export declare function resetObserveBudgetBytesForTests(): void;
|
|
215
|
+
export declare function recordFatalRunEnd(journal: Journal, runId: string, branch: string, err: unknown, graph?: RunGraph, phase?: string): TipProof | undefined;
|
|
188
216
|
export declare function runDaemon(repoRoot: string, opts?: RunOptions): Promise<RunSummary>;
|