tickmarkr 2.5.6 → 2.5.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/prompt.js +21 -1
- package/dist/cli/commands/approve.js +5 -5
- package/dist/cli/commands/status.js +95 -34
- package/dist/cli/commands/verify.js +6 -4
- package/dist/gates/baseline.d.ts +8 -3
- package/dist/gates/baseline.js +6 -3
- package/dist/gates/review.js +12 -1
- package/dist/gates/run-gates.d.ts +4 -1
- package/dist/gates/run-gates.js +65 -9
- package/dist/gates/test-manifest.d.ts +3 -0
- package/dist/gates/test-manifest.js +20 -2
- package/dist/gates/test-reporter.js +6 -1
- package/dist/graph/graph.d.ts +2 -0
- package/dist/graph/graph.js +5 -0
- package/dist/run/activity.d.ts +28 -0
- package/dist/run/activity.js +194 -0
- package/dist/run/daemon.d.ts +9 -0
- package/dist/run/daemon.js +200 -46
- package/dist/run/git.d.ts +6 -1
- package/dist/run/git.js +63 -11
- package/dist/run/journal.d.ts +20 -4
- package/dist/run/journal.js +60 -9
- package/dist/run/operator-page-summary.d.ts +56 -0
- package/dist/run/operator-page-summary.js +68 -0
- package/dist/run/operator-summary.d.ts +69 -0
- package/dist/run/operator-summary.js +77 -0
- package/dist/run/protocol.d.ts +71 -0
- package/dist/run/protocol.js +32 -0
- package/dist/tui/cockpit/board.d.ts +9 -0
- package/dist/tui/cockpit/board.js +10 -0
- package/dist/tui/cockpit/derive.d.ts +35 -0
- package/dist/tui/cockpit/derive.js +152 -10
- package/dist/tui/cockpit/evidence-view.d.ts +2 -0
- package/dist/tui/cockpit/evidence-view.js +42 -12
- package/dist/tui/cockpit/run-cockpit.d.ts +8 -1
- package/dist/tui/cockpit/run-cockpit.js +95 -1
- package/dist/tui/cockpit/run-view.d.ts +37 -0
- package/dist/tui/cockpit/run-view.js +189 -2
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +55 -3
package/dist/adapters/prompt.js
CHANGED
|
@@ -71,9 +71,29 @@ function workerVerdictWitness(raw, nonce, positions) {
|
|
|
71
71
|
// these — a result carrying any other summary is a PARSED trailer, i.e. the worker speaking.
|
|
72
72
|
export const NO_TRAILER_SUMMARY = "worker produced no TICKMARKR_RESULT trailer";
|
|
73
73
|
export const UNPARSEABLE_TRAILER_SUMMARY = "unparseable TICKMARKR_RESULT trailer";
|
|
74
|
+
// OBS-1062: every interactive TUI echoes the brief into the pane, and the brief carries the trailer
|
|
75
|
+
// TEMPLATE above — so the nonce token is on screen before the worker has said anything. A token
|
|
76
|
+
// whose object opens with the template's literal `"ok":true|false` (not a bool) is the brief being
|
|
77
|
+
// displayed, never the worker speaking; it must not count as trailer participation, or the daemon
|
|
78
|
+
// reaps the attempt as malformed at its first poll (v2.5.7 run …115246, both workers at 34 s).
|
|
79
|
+
const TEMPLATE_BODY = '{"ok":true|false';
|
|
80
|
+
// Bounded to THIS occurrence (review finding, Leg 0a R2): the `{` must sit before the next nonce
|
|
81
|
+
// token, or a malformed worker token followed by a later template redraw would be erased as echo.
|
|
82
|
+
function isTemplateEcho(raw, at, end) {
|
|
83
|
+
const open = raw.indexOf("{", at);
|
|
84
|
+
if (open === -1 || open >= end)
|
|
85
|
+
return false;
|
|
86
|
+
const joined = raw
|
|
87
|
+
.slice(open, open + 64)
|
|
88
|
+
.split("\n")
|
|
89
|
+
.map((l) => l.replace(/^[\s│|]+/, "").replace(/[\s│|]+$/, ""))
|
|
90
|
+
.join("");
|
|
91
|
+
return joined.startsWith(TEMPLATE_BODY);
|
|
92
|
+
}
|
|
74
93
|
export function parseWorkerResult(raw, nonce) {
|
|
75
94
|
const fail = (summary, cause) => ({ ok: false, summary, deviations: [], raw, cause });
|
|
76
|
-
const
|
|
95
|
+
const all = trailerTokenPositions(raw, nonce);
|
|
96
|
+
const positions = all.filter((at, i) => !isTemplateEcho(raw, at, all[i + 1] ?? raw.length));
|
|
77
97
|
// TUIs echo the prompt template, redraw lines, and HARD-wrap the JSON with per-line margins
|
|
78
98
|
// (cursor does; recent-unwrapped can't rejoin hard newlines). Scan occurrences backward — last
|
|
79
99
|
// parseable wins — joining wrapped lines, stripping margin/box chrome, and growing the candidate
|
|
@@ -1,11 +1,11 @@
|
|
|
1
|
-
import { graphDefinitionHash, loadGraph, saveGraph } from "../../graph/graph.js";
|
|
1
|
+
import { graphDefinitionHash, loadGraph, saveGraph, taskDefinitionFingerprint } from "../../graph/graph.js";
|
|
2
2
|
import { execFileSync } from "node:child_process";
|
|
3
3
|
import { userInfo } from "node:os";
|
|
4
4
|
import { loadConfig } from "../../config/config.js";
|
|
5
5
|
import { integrationBranch } from "../../run/merge.js";
|
|
6
6
|
import { separabilityErrors } from "../../compile/collateral.js";
|
|
7
7
|
import { GATE_NAMES } from "../../graph/schema.js";
|
|
8
|
-
import { applyScopeAmendments, engagementComparable, ATTEMPT_CAP_RELEASE, GATE_SATISFIED_RELEASE, Journal, RECHECK_RELEASE, REVIEW_UPHELD_RELEASE } from "../../run/journal.js";
|
|
8
|
+
import { applyScopeAmendments, engagementComparable, engagementReleased, ATTEMPT_CAP_RELEASE, GATE_SATISFIED_RELEASE, Journal, RECHECK_RELEASE, REVIEW_UPHELD_RELEASE } from "../../run/journal.js";
|
|
9
9
|
export const APPROVAL_DISPOSITIONS = ["dispatch", "waive-gate", "re-dispatch", "fund-fixed-attempt", "fresh-budget"];
|
|
10
10
|
/**
|
|
11
11
|
* What each disposition's release actually buys — the clause every operator-facing sentence about
|
|
@@ -178,7 +178,7 @@ export async function approve(argv, cwd = process.cwd()) {
|
|
|
178
178
|
throw new Error(`scope-request for ${taskId} requires --files <glob,…>`);
|
|
179
179
|
if (decisions)
|
|
180
180
|
throw new Error("--files cannot be combined with --waive, --uphold or --recheck");
|
|
181
|
-
const graph = applyScopeAmendments(loadGraph(cwd), journal);
|
|
181
|
+
const graph = applyScopeAmendments(loadGraph(cwd), journal, false, engagementReleased(events));
|
|
182
182
|
const from = graphDefinitionHash(graph);
|
|
183
183
|
if (!engagementComparable(journal.read(), from).comparable
|
|
184
184
|
|| (lastHuman?.data.graphDefinitionHash !== undefined && lastHuman.data.graphDefinitionHash !== from)) {
|
|
@@ -196,11 +196,11 @@ export async function approve(argv, cwd = process.cwd()) {
|
|
|
196
196
|
throw new Error(`refusing scope-request approval for ${taskId}: ${conflicts.join("\n")}`);
|
|
197
197
|
journal.append("task-approved", taskId, {
|
|
198
198
|
by, ...(reason ? { reason } : {}), via: "cli", release: "scope-request",
|
|
199
|
-
amendment: { from, to: graphDefinitionHash(amended), beforeFiles: task.files, files: amendedFiles, parkLine: park.line },
|
|
199
|
+
amendment: { from, to: graphDefinitionHash(amended), beforeFiles: task.files, files: amendedFiles, parkLine: park.line, definition: taskDefinitionFingerprint(task) },
|
|
200
200
|
});
|
|
201
201
|
// Do not write graph.json from this process while the daemon owns it: its sweep materializes
|
|
202
202
|
// the amendment without replacing a sibling's running state with this command's snapshot.
|
|
203
|
-
const projected = applyScopeAmendments(graph, journal);
|
|
203
|
+
const projected = applyScopeAmendments(graph, journal, false, engagementReleased(events));
|
|
204
204
|
const owner = approvalRunOwner(cwd, runId);
|
|
205
205
|
if (!owner.live && !owner.blockingRunId)
|
|
206
206
|
saveGraph(cwd, projected);
|
|
@@ -7,8 +7,11 @@ import { HerdrDriver } from "../../drivers/herdr.js";
|
|
|
7
7
|
import { formatOwnedName, parseOwnedName, } from "../../drivers/types.js";
|
|
8
8
|
import { blockedTasks, graphDefinitionHash, loadGraph, stateDirName } from "../../graph/graph.js";
|
|
9
9
|
import { GATE_NAMES } from "../../graph/schema.js";
|
|
10
|
-
import {
|
|
11
|
-
import {
|
|
10
|
+
import { projectActivity } from "../../run/activity.js";
|
|
11
|
+
import { projectOperatorSummary } from "../../run/operator-summary.js";
|
|
12
|
+
import { trackJournalRows } from "../../run/protocol.js";
|
|
13
|
+
import { newestPark, permittedDecisionVerbs } from "./approve.js";
|
|
14
|
+
import { Journal, formatJournalNarration, engagementComparable, isQualityFailureParkKind, parseRunId, preservedRefsByTask, recordedTaskFailureKind, runHasEnded, upheldFeedbackByTask, } from "../../run/journal.js";
|
|
12
15
|
import { isPidLive, runLockRunId, runStatusLine } from "../../run/lock.js";
|
|
13
16
|
import { normalizeGateOutcome } from "../../run/outcome.js";
|
|
14
17
|
import { desiredPanes } from "../../run/reconcile.js";
|
|
@@ -96,8 +99,8 @@ const taskClockReference = (events, taskId, content, fallback) => {
|
|
|
96
99
|
}
|
|
97
100
|
return fallback;
|
|
98
101
|
};
|
|
99
|
-
// Every attempt number a cell phrase carries is the engagement's own ladder (
|
|
100
|
-
//
|
|
102
|
+
// Every attempt number a cell phrase carries is the engagement's own ladder (journal dispatch
|
|
103
|
+
// labels are displayed from 1). Name that ruler wherever a bare one appears, not only in the one phrase
|
|
101
104
|
// shipping today — a phrase added upstream must not reach the operator unruled.
|
|
102
105
|
const labelAttemptRuler = (text) => text.replace(/(?<!engagement |run )\battempt (?=\d)/gu, "engagement attempt ");
|
|
103
106
|
// The timer must keep the process ALIVE: an unref'd timer here let the event loop drain after the
|
|
@@ -907,9 +910,58 @@ const readRunRecord = (cwd, graph, namedRunId) => {
|
|
|
907
910
|
* catch an empty or torn first write. Those snapshots own no task fold yet; once run-start exists,
|
|
908
911
|
* however, a fold failure is unexpected and must surface instead of becoming a false empty run.
|
|
909
912
|
*/
|
|
910
|
-
const recordTaskRows = (record, graph) => record.comparable && record.events.some((event) => event.event === "run-start")
|
|
911
|
-
? new Map(deriveRunCockpitData({ fileName: `${record.runId}.journal.jsonl`, raw: record.raw }, "status", { graph }).taskRows.map((row) => [row.taskId, row]))
|
|
913
|
+
const recordTaskRows = (record, graph, isDaemonAlive) => record.comparable && record.events.some((event) => event.event === "run-start")
|
|
914
|
+
? new Map(deriveRunCockpitData({ fileName: `${record.runId}.journal.jsonl`, raw: record.raw }, "status", { graph, isDaemonAlive }).taskRows.map((row) => [row.taskId, row]))
|
|
912
915
|
: new Map();
|
|
916
|
+
/** Adapt the single journal snapshot to the shared, evidence-only projections. */
|
|
917
|
+
const recordProjection = (record, graph, isDaemonAlive) => {
|
|
918
|
+
const rows = record ? recordTaskRows(record, graph, isDaemonAlive) : new Map();
|
|
919
|
+
const events = record?.comparable ? record.events : [];
|
|
920
|
+
const activity = projectActivity(record?.runId ?? "", trackJournalRows(record?.runId ?? "", events.map((raw, sourceIndex) => ({ raw, sourceIndex }))), graph.tasks);
|
|
921
|
+
const tasks = graph.tasks.map(task => {
|
|
922
|
+
const row = rows.get(task.id);
|
|
923
|
+
const evidence = events.filter(event => event.taskId === task.id);
|
|
924
|
+
const last = evidence.at(-1);
|
|
925
|
+
const recorded = activity.get(task.id);
|
|
926
|
+
let phase = last ? recorded.state : undefined;
|
|
927
|
+
// A completed battery is evidence of readiness, never evidence that merge started.
|
|
928
|
+
const gates = gateSnapshot(task, events, record?.rehashAt);
|
|
929
|
+
const boundary = events.reduce((index, event, at) => ["run-start", "run-resume", "run-end"].includes(event.event)
|
|
930
|
+
|| (event.taskId === task.id && (event.event === "task-dispatch"
|
|
931
|
+
|| (event.event === "phase-start" && event.data.phase === "gates"))) ? at : index, -1);
|
|
932
|
+
// Read the latest outcome, including non-verdicts. The gate rail retains earlier
|
|
933
|
+
// verdicts while a screen or infrastructure retry is pending; that historical
|
|
934
|
+
// pass must not make the current battery look ready to merge.
|
|
935
|
+
const currentResults = new Map(events.slice(boundary + 1).filter(event => event.taskId === task.id && event.event === "gate-result")
|
|
936
|
+
.map(event => [event.data.gate, normalizeGateOutcome(event.data).kind]));
|
|
937
|
+
if (recorded.state === "unconfirmed" && task.gates.length > 0 && !gates.priorGraph
|
|
938
|
+
&& task.gates.every(gate => currentResults.get(gate) === "passed")) {
|
|
939
|
+
phase = "awaiting merge phase";
|
|
940
|
+
}
|
|
941
|
+
const responsibility = [...evidence].reverse().find(event => typeof event.data.role === "string" || typeof event.data.agent === "string");
|
|
942
|
+
return {
|
|
943
|
+
id: task.id, deps: task.deps, status: row?.state ?? task.status,
|
|
944
|
+
phase, lastEvidenceAt: last?.ts,
|
|
945
|
+
responsible: responsibility ? {
|
|
946
|
+
role: typeof responsibility.data.role === "string" ? responsibility.data.role : undefined,
|
|
947
|
+
agent: typeof responsibility.data.agent === "string" ? responsibility.data.agent : undefined,
|
|
948
|
+
} : undefined,
|
|
949
|
+
};
|
|
950
|
+
});
|
|
951
|
+
const decisions = tasks.flatMap(task => {
|
|
952
|
+
const park = task.status === "human" ? newestPark(events, task.id) : undefined;
|
|
953
|
+
return park ? [{ taskId: task.id, park, verbs: permittedDecisionVerbs(park) }] : [];
|
|
954
|
+
});
|
|
955
|
+
return { rows, activity, summaries: new Map(projectOperatorSummary(tasks, decisions).map(task => [task.taskId, task])) };
|
|
956
|
+
};
|
|
957
|
+
const summaryText = (summary) => sanitizeTaskText([
|
|
958
|
+
`phase ${summary.phase ?? "unrecorded"}`,
|
|
959
|
+
`last evidence ${summary.lastEvidenceAt ?? "unrecorded"}`,
|
|
960
|
+
`responsible ${[summary.responsible?.role, summary.responsible?.agent].filter(Boolean).join(" / ") || "unrecorded"}`,
|
|
961
|
+
`blocker ${summary.blocker?.kind ?? "none"}`,
|
|
962
|
+
`next action ${summary.blocker?.nextAction ?? "unrecorded"}`,
|
|
963
|
+
`human-decision ${summary.blocker?.decisionRequired ? "yes" : "no"}`,
|
|
964
|
+
].join(" · "));
|
|
913
965
|
/** The graph tasks this run's record actually speaks about — a row with no recorded fact is silence. */
|
|
914
966
|
const recordedTasks = (graph, rows) => graph.tasks.filter((task) => {
|
|
915
967
|
const row = rows.get(task.id);
|
|
@@ -936,14 +988,11 @@ binaryVersion, now = Date.now(), animationFrame = 0, workerLiveness = new Map(),
|
|
|
936
988
|
const comparable = record?.comparable ?? false;
|
|
937
989
|
const rehashAt = record?.rehashAt;
|
|
938
990
|
const assignments = new Map();
|
|
939
|
-
let replayed = null;
|
|
940
991
|
const contexts = new Map();
|
|
941
992
|
// v1.53 T5: this run is dead — a newer run replaced it
|
|
942
993
|
const supersededBy = [...events].reverse()
|
|
943
994
|
.find((e) => e.event === "superseded" && typeof e.data.by === "string")?.data.by;
|
|
944
995
|
if (record && comparable) {
|
|
945
|
-
if (!journalRowsOnly)
|
|
946
|
-
replayed = Journal.open(cwd, record.runId).replayStatuses();
|
|
947
996
|
for (const e of events) {
|
|
948
997
|
if (!journalRowsOnly && e.event === "task-dispatch" && e.taskId) {
|
|
949
998
|
const a = e.data.assignment;
|
|
@@ -956,7 +1005,12 @@ binaryVersion, now = Date.now(), animationFrame = 0, workerLiveness = new Map(),
|
|
|
956
1005
|
}
|
|
957
1006
|
}
|
|
958
1007
|
}
|
|
959
|
-
const
|
|
1008
|
+
const projection = recordProjection(record, g, () => daemon.state !== "dead");
|
|
1009
|
+
const taskRows = projection.rows;
|
|
1010
|
+
// Preserve the plain table's compact status vocabulary using the same snapshot as its evidence.
|
|
1011
|
+
const replayed = new Map([...taskRows].flatMap(([id, row]) => row.state === undefined ? [] : [[id,
|
|
1012
|
+
row.state === "running" || row.state === "interrupted" ? "pending" : graphTaskStatus(row.state, "pending"),
|
|
1013
|
+
]]));
|
|
960
1014
|
const clockFallback = new Date(now);
|
|
961
1015
|
const recordedClock = new Date(events.at(-1)?.ts ?? now);
|
|
962
1016
|
const zoneReference = Number.isFinite(recordedClock.getTime()) ? recordedClock : clockFallback;
|
|
@@ -973,14 +1027,7 @@ binaryVersion, now = Date.now(), animationFrame = 0, workerLiveness = new Map(),
|
|
|
973
1027
|
};
|
|
974
1028
|
const renderedTasks = journalRowsOnly ? recordedTasks(g, taskRows) : g.tasks;
|
|
975
1029
|
const starved = new Set(blockedTasks(effective).map((t) => t.id));
|
|
976
|
-
//
|
|
977
|
-
// (a recompiled graph's journal must not animate the wrong tasks); with no or stale journal the
|
|
978
|
-
// dep-waiting cells still derive from the effective graph statuses.
|
|
979
|
-
const activity = foldActivity(comparable ? events : [], effective.tasks);
|
|
980
|
-
// RULING-v189-scope §14c: a card answers BOTH supervisor questions. `foldActivity` names what a
|
|
981
|
-
// task waits FOR; this names what waits ON it. Read from `effective.tasks` — whose statuses are
|
|
982
|
-
// the journal's replay, never the compiled graph's — so a dependent the record says is done stops
|
|
983
|
-
// being named, and a task all of whose dependents are done acquires no entry at all.
|
|
1030
|
+
// Reverse dependencies share the effective task statuses with the projection.
|
|
984
1031
|
const dependents = new Map();
|
|
985
1032
|
for (const task of effective.tasks) {
|
|
986
1033
|
if (task.status === "done")
|
|
@@ -1039,7 +1086,18 @@ binaryVersion, now = Date.now(), animationFrame = 0, workerLiveness = new Map(),
|
|
|
1039
1086
|
|| (st === "failed" && failureKind === undefined));
|
|
1040
1087
|
const livePhase = phases.get(t.id);
|
|
1041
1088
|
const isStarved = !livePhase && starved.has(t.id);
|
|
1042
|
-
const
|
|
1089
|
+
const summary = projection.summaries.get(t.id);
|
|
1090
|
+
// Attempt labels can restart at zero in legacy engagements. Pair the preparing caption's
|
|
1091
|
+
// counter and timestamp from the same dispatch, rather than retaining a prior higher label.
|
|
1092
|
+
const dispatch = [...events].reverse().find(event => event.taskId === t.id && event.event === "task-dispatch");
|
|
1093
|
+
const projectedPhrase = summary.blocker?.kind === "dependency-wait"
|
|
1094
|
+
? `dep-waiting on ${summary.blocker.prerequisites.join(", ")}`
|
|
1095
|
+
: summary.phase === "terminal"
|
|
1096
|
+
? st === "human" ? `parked${failureKind ? ` (${failureKind})` : ""}` : undefined
|
|
1097
|
+
: summary.phase === "preparing" && dispatch
|
|
1098
|
+
? `preparing · attempt ${(Number.isInteger(dispatch.data.attempt) ? dispatch.data.attempt : 0) + 1} since ${dispatch.ts.slice(11, 19)}`
|
|
1099
|
+
: summary.phase ?? undefined;
|
|
1100
|
+
const rawPhrase = livePhase ? phaseDetail(livePhase, now, workerLiveness.get(t.id)) : projectedPhrase;
|
|
1043
1101
|
const phrase = rawPhrase === undefined
|
|
1044
1102
|
? undefined
|
|
1045
1103
|
: journalRowsOnly
|
|
@@ -1053,12 +1111,12 @@ binaryVersion, now = Date.now(), animationFrame = 0, workerLiveness = new Map(),
|
|
|
1053
1111
|
? gateSnapshot(t, events, rehashAt)
|
|
1054
1112
|
: { states: defaultGateStates(t), priorGraph: false };
|
|
1055
1113
|
const pane = panes.get(t.id);
|
|
1056
|
-
return { t, st, merged, failureKind, redTier, label, assignCol, isStarved, phrase, channel, ctx, livePhase, pane, ...gates };
|
|
1114
|
+
return { t, st, merged, failureKind, redTier, label, assignCol, isStarved, phrase, channel, ctx, livePhase, pane, summary, ...gates };
|
|
1057
1115
|
});
|
|
1058
1116
|
if (!unicode) {
|
|
1059
|
-
// machine
|
|
1060
|
-
// use
|
|
1061
|
-
const rows = cells.flatMap(({ t, st, merged, label, assignCol, livePhase, states, priorGraph, pane }) => {
|
|
1117
|
+
// Keep the machine task-title columns intact; projection details occupy their own line.
|
|
1118
|
+
// Phase-aware frames use ASCII spinners so pipes never receive terminal-only braille/ANSI.
|
|
1119
|
+
const rows = cells.flatMap(({ t, st, merged, label, assignCol, livePhase, states, priorGraph, pane, summary }) => {
|
|
1062
1120
|
const chain = gateChain(states, false);
|
|
1063
1121
|
const prefix = livePhase ? ` ${ASCII_SPINNER[animationFrame % ASCII_SPINNER.length]} ${t.id} ` : ` ${surfaceTaskBox(st, merged)} ${t.id} `;
|
|
1064
1122
|
const suffix = ` ${chain}${priorGraph ? ` ${PRIOR_GRAPH_MARKER}` : ""} ${livePhase ? "running" : surfaceStatusWord(st)}${label} ${assignCol}${pane ? ` pane ${pane}` : ""}`;
|
|
@@ -1069,6 +1127,7 @@ binaryVersion, now = Date.now(), animationFrame = 0, workerLiveness = new Map(),
|
|
|
1069
1127
|
// (a pane name is 60 columns on its own), so the floor costs wrapping, never the graph.
|
|
1070
1128
|
return [
|
|
1071
1129
|
`${prefix}${shortGoal(t.title, Math.max(MACHINE_TITLE_FLOOR, width - prefix.length - suffix.length))}${suffix}`,
|
|
1130
|
+
` ${summaryText(summary)}`,
|
|
1072
1131
|
...recoveryLinesForTask(t.id).map((line) => ` ${line}`),
|
|
1073
1132
|
];
|
|
1074
1133
|
});
|
|
@@ -1142,7 +1201,8 @@ binaryVersion, now = Date.now(), animationFrame = 0, workerLiveness = new Map(),
|
|
|
1142
1201
|
besideLockup(fillLockup(dim(versionText), versionCells), factLines[1]),
|
|
1143
1202
|
...factLines.slice(2).map((line) => `${" ".repeat(lockupCells + lockupGapCells)}${line}`),
|
|
1144
1203
|
];
|
|
1145
|
-
const
|
|
1204
|
+
const lastEvent = comparable ? events.at(-1) : undefined;
|
|
1205
|
+
const nowLine = lastEvent ? [legend(` now: ${formatJournalNarration(lastEvent)}`)] : [];
|
|
1146
1206
|
const supervisionLegend = legend(` ${supervisionText(supervision)}`);
|
|
1147
1207
|
const taskSectionSummary = `${done} of ${total} merged · rows in graph order · gates left→right in pipeline order`;
|
|
1148
1208
|
const taskSection = ` ${ok("▌")} ${title("TASKS")} ${dim(taskSectionSummary)}`;
|
|
@@ -1220,6 +1280,7 @@ binaryVersion, now = Date.now(), animationFrame = 0, workerLiveness = new Map(),
|
|
|
1220
1280
|
` ${tone(cell.t.id)} ${dim(taskArea(cell.t))} ${sanitizeTaskText(cell.t.title)}`,
|
|
1221
1281
|
...wrapCells(` ${dim(machinery)}`, boardColumns, { continuationPrefix: " " }),
|
|
1222
1282
|
...wrapCells(` ${dim(noteFor(cell))}`, boardColumns, { continuationPrefix: " " }).filter((line) => line.trim()),
|
|
1283
|
+
...wrapCells(` ${dim(summaryText(cell.summary))}`, boardColumns, { continuationPrefix: " " }),
|
|
1223
1284
|
"",
|
|
1224
1285
|
];
|
|
1225
1286
|
});
|
|
@@ -1249,15 +1310,15 @@ binaryVersion, now = Date.now(), animationFrame = 0, workerLiveness = new Map(),
|
|
|
1249
1310
|
? running
|
|
1250
1311
|
: dim;
|
|
1251
1312
|
const titleTone = cell.redTier || cell.livePhase || (effort?.parks ?? 0) >= 3 ? title : dim;
|
|
1252
|
-
return Array.from({ length: height }, (_, index) => ` ${idTone(fitColumn(columns[0][index] ?? "", idWidth, 1))}`
|
|
1253
|
-
|
|
1254
|
-
|
|
1255
|
-
|
|
1256
|
-
|
|
1257
|
-
|
|
1258
|
-
|
|
1259
|
-
|
|
1260
|
-
|
|
1313
|
+
return [...Array.from({ length: height }, (_, index) => ` ${idTone(fitColumn(columns[0][index] ?? "", idWidth, 1))}`
|
|
1314
|
+
+ dim(fitColumn(columns[1][index] ?? "", areaWidth))
|
|
1315
|
+
+ dim(fitColumn(columns[2][index] ?? "", depsWidth))
|
|
1316
|
+
+ titleTone(fitColumn(columns[3][index] ?? "", taskWidth))
|
|
1317
|
+
+ fitCells(columns[4][index] ?? "", gatesWidth)
|
|
1318
|
+
+ " "
|
|
1319
|
+
+ dim(fitColumn(columns[5][index] ?? "", channelWidth))
|
|
1320
|
+
+ dim(fitColumn(columns[6][index] ?? "", attemptWidth, 1))
|
|
1321
|
+
+ dim(fitCells(columns[7][index] ?? "", noteWidth))), ...wrapCells(` ${dim(summaryText(cell.summary))}`, boardColumns, { continuationPrefix: " " })];
|
|
1261
1322
|
});
|
|
1262
1323
|
return [legend(headerRow), dim(ruleRow), ...rows];
|
|
1263
1324
|
};
|
|
@@ -1305,7 +1366,7 @@ const oneLine = (cwd, namedRunId) => {
|
|
|
1305
1366
|
const record = readRunRecord(cwd, graph, namedRunId);
|
|
1306
1367
|
if (!record)
|
|
1307
1368
|
return sanitizeTaskText(`tickmarkr · no runs yet · 0/${graph.tasks.length} done`);
|
|
1308
|
-
const rows =
|
|
1369
|
+
const { rows } = recordProjection(record, graph);
|
|
1309
1370
|
const claims = record.comparable
|
|
1310
1371
|
? [
|
|
1311
1372
|
`${recordedDone(recordedTasks(graph, rows), rows)}/${graph.tasks.length} done`,
|
|
@@ -150,10 +150,12 @@ export async function verify(argv, cwd = process.cwd()) {
|
|
|
150
150
|
allowPositionals: false,
|
|
151
151
|
});
|
|
152
152
|
const stateRoot = await verifyStateRoot(cwd);
|
|
153
|
+
const fileRoot = (file) => existsSync(join(cwd, ".tickmarkr", file)) ? cwd : stateRoot;
|
|
153
154
|
if (resolve(stateRoot) !== resolve(cwd)) {
|
|
154
|
-
console.error(`verify: state files
|
|
155
|
+
console.error(`verify: state files resolved read-only from ${join(stateRoot, ".tickmarkr")} (linked worktree; per-file origins: `
|
|
156
|
+
+ STATE_FILES.map(file => `${file}: ${join(fileRoot(file), ".tickmarkr", file)} (${fileRoot(file) === cwd ? "local" : "common root"})`).join("; ") + ")");
|
|
155
157
|
}
|
|
156
|
-
const cfg = loadConfig(
|
|
158
|
+
const cfg = loadConfig(fileRoot("config.yaml"));
|
|
157
159
|
const head = (await shGitOk("git rev-parse HEAD", cwd)).trim();
|
|
158
160
|
const baseTip = (await shGitOk(`git rev-parse '${values.base}'`, cwd).catch(() => {
|
|
159
161
|
throw new Error(`--base ${values.base} is not a resolvable ref — pass --base <ref> naming the branch this diff targets`);
|
|
@@ -167,7 +169,7 @@ export async function verify(argv, cwd = process.cwd()) {
|
|
|
167
169
|
let files = values.files ?? [];
|
|
168
170
|
let goal = `independent verification of the ${values.base}..HEAD diff`;
|
|
169
171
|
if (values.task) {
|
|
170
|
-
const t = getTask(loadGraph(
|
|
172
|
+
const t = getTask(loadGraph(fileRoot("graph.json")), values.task);
|
|
171
173
|
acceptance = t.acceptance;
|
|
172
174
|
if (!files.length)
|
|
173
175
|
files = t.files;
|
|
@@ -223,7 +225,7 @@ export async function verify(argv, cwd = process.cwd()) {
|
|
|
223
225
|
let author = HUMAN_AUTHOR;
|
|
224
226
|
const adapters = allAdapters();
|
|
225
227
|
if (wantAcceptance || wantReview) {
|
|
226
|
-
const health = readDoctor(
|
|
228
|
+
const health = readDoctor(fileRoot("doctor.json")) ?? (await probeAll(adapters));
|
|
227
229
|
const pools = rolePools(cfg, adapters, health);
|
|
228
230
|
judgeChannels = pools.judge;
|
|
229
231
|
channels = pools.review;
|
package/dist/gates/baseline.d.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import type { TickmarkrConfig } from "../config/config.js";
|
|
2
2
|
import type { AcceptanceItem } from "../graph/schema.js";
|
|
3
|
-
import { type RunCapacity, type ShResult } from "../run/git.js";
|
|
3
|
+
import { type RunCapacity, type ShellOptions, type ShResult } from "../run/git.js";
|
|
4
4
|
import type { GateResult } from "./types.js";
|
|
5
5
|
/**
|
|
6
6
|
* T7: the capacity a gate's own command ran under rides the RESULT, beside the verdict it explains,
|
|
@@ -157,7 +157,12 @@ export declare const CAPTURE_CEILING_MS = 1800000;
|
|
|
157
157
|
/** OBS-1044: a cached vitest entry whose count is a stdout sum may be inflated; only a manifest-derived
|
|
158
158
|
* count is safe to apply as a deficit floor. A null count compares nothing and is safe as-is. */
|
|
159
159
|
export declare function staleFileCountCommands(baseline: Baseline, commands: Record<string, string>, cwd: string): string[];
|
|
160
|
-
export
|
|
160
|
+
export interface BaselineReceiptOptions {
|
|
161
|
+
onReceipt?: ShellOptions["onReceipt"];
|
|
162
|
+
/** Only comparison's build command may carry task-build identity; capture never does. */
|
|
163
|
+
taskBuildAttribution?: ShellOptions["receiptAttribution"];
|
|
164
|
+
}
|
|
165
|
+
export declare function captureBaseline(cwd: string, commands: Record<string, string>, opts?: BaselineReceiptOptions): Promise<Baseline>;
|
|
161
166
|
export interface VacuousOracleWarning {
|
|
162
167
|
kind: "vacuous-oracle";
|
|
163
168
|
taskId: string;
|
|
@@ -171,7 +176,7 @@ export declare function detectVacuousOracles(cwd: string, tasks: ReadonlyArray<{
|
|
|
171
176
|
export interface RetryOptions {
|
|
172
177
|
authorizeRetry?: (cause: "infra" | "host-starved") => boolean;
|
|
173
178
|
}
|
|
174
|
-
export declare function compareToBaseline(cwd: string, commands: Record<string, string>, baseline: Baseline, enabled: string[], opts?: RetryOptions & {
|
|
179
|
+
export declare function compareToBaseline(cwd: string, commands: Record<string, string>, baseline: Baseline, enabled: string[], opts?: RetryOptions & BaselineReceiptOptions & {
|
|
175
180
|
rerunOf?: HostStarvedRerun;
|
|
176
181
|
infraRerun?: HostStarvedRerun;
|
|
177
182
|
selected?: readonly string[];
|
package/dist/gates/baseline.js
CHANGED
|
@@ -487,13 +487,13 @@ export function staleFileCountCommands(baseline, commands, cwd) {
|
|
|
487
487
|
})
|
|
488
488
|
.map(([name]) => name);
|
|
489
489
|
}
|
|
490
|
-
export async function captureBaseline(cwd, commands) {
|
|
490
|
+
export async function captureBaseline(cwd, commands, opts = {}) {
|
|
491
491
|
const base = { commands: {} };
|
|
492
492
|
for (const [name, cmd] of Object.entries(commands)) {
|
|
493
493
|
if (name === "tipTest" && commands.test !== undefined && cmd === commands.test) {
|
|
494
494
|
continue;
|
|
495
495
|
}
|
|
496
|
-
const r = await sh(cmd, cwd, CAPTURE_CEILING_MS);
|
|
496
|
+
const r = await sh(cmd, cwd, CAPTURE_CEILING_MS, { onReceipt: opts.onReceipt });
|
|
497
497
|
// ponytail: strip the executing cwd so repo-root capture and worktree compare fingerprint identically; /private-vs-/tmp symlink variance is out of scope
|
|
498
498
|
// ponytail: a capture that was itself killed records the ceiling as its "measurement", which
|
|
499
499
|
// scales the next ceiling up — the right direction for a suite that never finished once.
|
|
@@ -634,7 +634,10 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
|
|
|
634
634
|
}
|
|
635
635
|
const entry = baseline.commands[name];
|
|
636
636
|
const ceilingMs = effectiveCeilingMs(entry);
|
|
637
|
-
const r = await sh(cmd, cwd, ceilingMs
|
|
637
|
+
const r = await sh(cmd, cwd, ceilingMs, {
|
|
638
|
+
onReceipt: opts.onReceipt,
|
|
639
|
+
...(name === "build" ? { receiptAttribution: opts.taskBuildAttribution } : {}),
|
|
640
|
+
});
|
|
638
641
|
// T7: every verdict below carries the capacity ITS OWN command ran under, taken off the shell
|
|
639
642
|
// result rather than re-derived after the fact. The skip row above ran no command and therefore
|
|
640
643
|
// states no capacity — a row that never divided the machine must not claim that it did.
|
package/dist/gates/review.js
CHANGED
|
@@ -185,7 +185,10 @@ export { modelProvider };
|
|
|
185
185
|
export function matchClosureId(candidate, target) {
|
|
186
186
|
if (typeof candidate !== "string")
|
|
187
187
|
return typeof target === "string" ? false : undefined;
|
|
188
|
-
|
|
188
|
+
// OBS-1068: the brief's copy block printed each id as `Fingerprint: <id>` and told the seat to copy
|
|
189
|
+
// it EXACTLY — three faithful approvals in one run were discarded as closure-mismatch. A leading
|
|
190
|
+
// `Fingerprint:` label is the brief's own wording, never the reviewer's id; strip it before comparing.
|
|
191
|
+
const normCandidate = candidate.replace(/^\s*fingerprint\s*:\s*/i, "").replace(/\s+/g, "");
|
|
189
192
|
if (typeof target === "string") {
|
|
190
193
|
return normCandidate === target.replace(/\s+/g, "");
|
|
191
194
|
}
|
|
@@ -476,6 +479,11 @@ carriedAuthors = []) {
|
|
|
476
479
|
return capFail;
|
|
477
480
|
const nonce = generateVerdictNonce();
|
|
478
481
|
const repoRoot = daemonRepoRoot(worktree, artifactDir);
|
|
482
|
+
// OBS-880 add.1: guidance only. Do not expand scope globs into permission to run suites.
|
|
483
|
+
const ownTestFiles = [...new Set(task.files.filter((file) => !/[*?[\]{}()!]/.test(file) && /(?:^|\/)[^/]*\.(?:test|spec)\.[cm]?[jt]sx?$/.test(file)))];
|
|
484
|
+
const suiteBudget = ownTestFiles.length
|
|
485
|
+
? `You may run at most the task's own test files explicitly named in files[]; these are the only suites you may run: ${ownTestFiles.map((file) => `\`${file}\``).join(", ")}.`
|
|
486
|
+
: "No suite may be run: files[] names no explicit test file owned by this task.";
|
|
479
487
|
const prompt = `TICKMARKR-REVIEW
|
|
480
488
|
You are a skeptical cross-vendor code reviewer. Another agent (vendor: ${author.adapter}) authored this diff.
|
|
481
489
|
Look for correctness bugs, security issues, and acceptance-criteria gaps. Approve only if you would merge it.
|
|
@@ -489,6 +497,9 @@ ${task.acceptance.map((a) => `- ${renderAcceptanceItem(a)}`).join("\n")}
|
|
|
489
497
|
|
|
490
498
|
${renderDeclaredWriteScope(task.files)}
|
|
491
499
|
|
|
500
|
+
## Reviewer suite budget
|
|
501
|
+
${suiteBudget} Never run the whole suite (including an unfiltered npm test or vitest run). The gate suite owns the runner lease; a parallel full suite starves the gate.
|
|
502
|
+
|
|
492
503
|
${priorMaterials.length ? `${renderPriorMaterials(priorMaterials)}
|
|
493
504
|
|
|
494
505
|
` : ""}## Diff
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import type { CommandReceiptAttribution } from "../run/protocol.js";
|
|
1
2
|
import { type Assignment, type BillingChannel, type WorkerAdapter, type WorkerResult } from "../adapters/types.js";
|
|
2
3
|
import { type TickmarkrConfig } from "../config/config.js";
|
|
3
4
|
import { type GateName, type Task } from "../graph/schema.js";
|
|
@@ -42,9 +43,10 @@ export type GateEvent = {
|
|
|
42
43
|
gate: GateName;
|
|
43
44
|
name: string;
|
|
44
45
|
payload: Record<string, unknown>;
|
|
45
|
-
result
|
|
46
|
+
result?: GateResult;
|
|
46
47
|
};
|
|
47
48
|
export interface GateContext {
|
|
49
|
+
buildReceiptIdentity?: Omit<CommandReceiptAttribution, "invocation">;
|
|
48
50
|
authorizeInfraRetry?: (subject: string, cause?: VerificationRetryCause) => boolean;
|
|
49
51
|
verificationScope?: VerificationScope;
|
|
50
52
|
worktree: string;
|
|
@@ -62,6 +64,7 @@ export interface GateContext {
|
|
|
62
64
|
excludeReviewers?: string[];
|
|
63
65
|
demotedReviewers?: Set<string>;
|
|
64
66
|
reviewNoVerdicts?: Map<string, string[]>;
|
|
67
|
+
recheck?: boolean;
|
|
65
68
|
carriedAuthors?: readonly string[];
|
|
66
69
|
reviewHistory?: string[];
|
|
67
70
|
priorReviewers?: PriorReviewer[];
|
package/dist/gates/run-gates.js
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { randomUUID } from "node:crypto";
|
|
1
2
|
import { existsSync, mkdtempSync, readFileSync, rmSync, statSync } from "node:fs";
|
|
2
3
|
import { loadavg, tmpdir } from "node:os";
|
|
3
4
|
import { join, posix } from "node:path";
|
|
@@ -258,6 +259,40 @@ function classifySignalOnlyTest(g) {
|
|
|
258
259
|
}
|
|
259
260
|
export async function runGates(task, ctx) {
|
|
260
261
|
const results = [];
|
|
262
|
+
// Receipt identity belongs to this round, never to a cached verdict. Each call from the shell
|
|
263
|
+
// allocates a new invocation, including retries whose local spawn counter starts at one again.
|
|
264
|
+
let currentBuild;
|
|
265
|
+
let buildStarted = false;
|
|
266
|
+
let receiptNotes = Promise.resolve();
|
|
267
|
+
const beginBuild = () => {
|
|
268
|
+
buildStarted = false;
|
|
269
|
+
currentBuild = { ...(ctx.buildReceiptIdentity ?? {
|
|
270
|
+
runId: ctx.artifactDir ?? "standalone", taskId: task.id, attempt: 0, gateRound: 0,
|
|
271
|
+
}), invocation: randomUUID() };
|
|
272
|
+
return { ...currentBuild };
|
|
273
|
+
};
|
|
274
|
+
const buildReceipt = (receipt, reason) => {
|
|
275
|
+
const matches = currentBuild !== undefined && receipt.attribution !== undefined
|
|
276
|
+
&& Object.keys(currentBuild)
|
|
277
|
+
.every((key) => receipt.attribution[key] === currentBuild[key]);
|
|
278
|
+
const attributed = matches && (receipt.outcome === "started" || !receipt.confirmedStart || buildStarted);
|
|
279
|
+
if (attributed)
|
|
280
|
+
buildStarted = receipt.outcome === "started";
|
|
281
|
+
const { attribution, ...observation } = receipt;
|
|
282
|
+
const payload = attributed ? { ...receipt, gate: "build", ...(reason ? { reason, freshBuildRan: false } : {}) } : {
|
|
283
|
+
...observation, gate: "build", attributionStatus: "unattributed",
|
|
284
|
+
reportedAttribution: attribution,
|
|
285
|
+
};
|
|
286
|
+
// Shell observers are synchronous. Serialize asynchronous note sinks and drain before the
|
|
287
|
+
// verdict (also on cancellation), without letting an observer change execution or retry policy.
|
|
288
|
+
receiptNotes = receiptNotes.then(async () => {
|
|
289
|
+
await ctx.onGate?.({ phase: "note", gate: "build", name: "build-receipt", payload });
|
|
290
|
+
}).catch(() => { });
|
|
291
|
+
};
|
|
292
|
+
const noBuild = async (outcome, reason) => {
|
|
293
|
+
buildReceipt({ outcome, confirmedStart: false, attribution: beginBuild() }, reason);
|
|
294
|
+
await receiptNotes;
|
|
295
|
+
};
|
|
261
296
|
let selectionDecision;
|
|
262
297
|
let commits = [];
|
|
263
298
|
const stateDir = ctx.stateDir ?? resolveStateDir(ctx.worktree, ctx.artifactDir);
|
|
@@ -272,6 +307,14 @@ export async function runGates(task, ctx) {
|
|
|
272
307
|
// the ledger names the reuse and the identity even where the gate-result row's details must stay
|
|
273
308
|
// the fresh verdict's (see formatReusedRow).
|
|
274
309
|
const noteReuse = (gate, r, id) => ctx.onGate?.({ phase: "note", gate, name: "gate-reused-verdict", payload: { gate, pass: r.pass, details: r.meta?.reusedDetails, ...reusedIdentity(id) }, result: r });
|
|
310
|
+
// OBS-1055: on a recheck a cached red is the answer the operator just said was wrongly given; it is
|
|
311
|
+
// journaled as discarded and the gate runs. Returns true when the hit must NOT be reused.
|
|
312
|
+
const discardCachedRed = async (gate, hit) => {
|
|
313
|
+
if (!ctx.recheck || hit.pass)
|
|
314
|
+
return false;
|
|
315
|
+
await ctx.onGate?.({ phase: "note", gate, name: "recheck-rerun", payload: { gate, reason: "cached-red-discarded" }, result: hit });
|
|
316
|
+
return true;
|
|
317
|
+
};
|
|
275
318
|
const shapeGates = ctx.cfg.gates.byShape?.[task.shape];
|
|
276
319
|
const enabled = (g) => task.gates.includes(g) && (g !== "acceptance" && g !== "review" || shapeGates?.[g] !== false);
|
|
277
320
|
const failed = () => results.some((r) => !r.pass);
|
|
@@ -595,20 +638,30 @@ export async function runGates(task, ctx) {
|
|
|
595
638
|
const hit = verdictStore.get(identity);
|
|
596
639
|
if (hit)
|
|
597
640
|
classifySignalOnlyTest(hit); // Older entries predate classification at the write seam.
|
|
598
|
-
if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")
|
|
641
|
+
if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")
|
|
642
|
+
&& !(await discardCachedRed(g, hit))) {
|
|
599
643
|
r = formatReusedRow(hit, identity);
|
|
600
644
|
cached = true;
|
|
645
|
+
if (g === "build")
|
|
646
|
+
await noBuild("reused-result", "verdict reused; no fresh build ran in this invocation");
|
|
601
647
|
await noteReuse(g, r, identity);
|
|
602
648
|
}
|
|
603
649
|
}
|
|
604
650
|
if (!r) {
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
|
|
608
|
-
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
|
|
651
|
+
if (g === "build" && !cmd)
|
|
652
|
+
await noBuild("skipped", "no build command detected");
|
|
653
|
+
try {
|
|
654
|
+
// VL-1: a detected vitest test command is judged by its own invocation-bound report — the
|
|
655
|
+
// stdout-count/file-count path (compareToBaseline's fileCountDeficit) never runs for it. Any
|
|
656
|
+
// other scripted test command keeps today's exit-code contract byte-identically.
|
|
657
|
+
const useManifest = g === "test" && commands.test !== undefined && isVitestTestCommand(commands.test, ctx.worktree);
|
|
658
|
+
r = useManifest
|
|
659
|
+
? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir, retryOptions(identity)))
|
|
660
|
+
: (await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], { ...retryOptions(identity), ...(g === "build" ? { onReceipt: buildReceipt, taskBuildAttribution: beginBuild } : {}), ...(g === "test" && selected ? { selected } : {}) })))[0];
|
|
661
|
+
}
|
|
662
|
+
finally {
|
|
663
|
+
await receiptNotes;
|
|
664
|
+
}
|
|
612
665
|
}
|
|
613
666
|
// the screen's interval IS the test gate's first interval, so the split needs no second clock
|
|
614
667
|
if (g === "test" && selected)
|
|
@@ -929,6 +982,8 @@ export async function runGates(task, ctx) {
|
|
|
929
982
|
if (entryDirt) {
|
|
930
983
|
addMeasurement(sequence[0], entryMeasurement);
|
|
931
984
|
await emitStart(sequence[0]);
|
|
985
|
+
if (enabled("build"))
|
|
986
|
+
await noBuild("refused", "dirty worktree; no build start confirmed");
|
|
932
987
|
await record(await dirtyRefusal(sequence[0], entryDirt));
|
|
933
988
|
return done();
|
|
934
989
|
}
|
|
@@ -1037,7 +1092,8 @@ export async function runGates(task, ctx) {
|
|
|
1037
1092
|
const hit = verdictStore.get(identity);
|
|
1038
1093
|
if (hit)
|
|
1039
1094
|
classifySignalOnlyTest(hit);
|
|
1040
|
-
if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")
|
|
1095
|
+
if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")
|
|
1096
|
+
&& !(await discardCachedRed("test", hit))) {
|
|
1041
1097
|
full = formatReusedRow(hit, identity);
|
|
1042
1098
|
cached = true;
|
|
1043
1099
|
await noteReuse("test", full, identity);
|
|
@@ -36,6 +36,9 @@ export interface TestReport {
|
|
|
36
36
|
certificate?: {
|
|
37
37
|
at: number;
|
|
38
38
|
exitCode: number;
|
|
39
|
+
/** Unhandled and module collection errors observed by the reporter; absent on older reports. */
|
|
40
|
+
errors?: number;
|
|
41
|
+
diagnostics?: string[];
|
|
39
42
|
};
|
|
40
43
|
}
|
|
41
44
|
/** Reads and structurally validates the report; a missing or malformed file is `undefined` — never a partial parse. */
|
|
@@ -448,11 +448,29 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
|
|
|
448
448
|
});
|
|
449
449
|
const verdict = verifyManifestReport({ manifest: files, nonce, exitCode: invoked.exitCode,
|
|
450
450
|
report: invoked.report, killedFile: invoked.killedFile, hangBudgetMs: invoked.hangBudgetMs });
|
|
451
|
+
// Preserve the validator's verdict and classification; runner evidence only explains it.
|
|
452
|
+
const stdoutPath = join(dir, `test-runner-stdout-${nonce}.log`);
|
|
453
|
+
const stderrPath = join(dir, `test-runner-stderr-${nonce}.log`);
|
|
454
|
+
const stdoutTail = Buffer.from(invoked.stdout).subarray(-16 * 1024);
|
|
455
|
+
const stderrTail = Buffer.from(invoked.stderr).subarray(-16 * 1024);
|
|
456
|
+
writeFileSync(stdoutPath, stdoutTail);
|
|
457
|
+
writeFileSync(stderrPath, stderrTail);
|
|
458
|
+
const report = invoked.report?.nonce === nonce ? invoked.report : undefined;
|
|
459
|
+
const neverStarted = report ? files.filter(file => !(file in report.started)).length : "unknown";
|
|
460
|
+
const errors = report?.certificate?.errors;
|
|
461
|
+
const reporterErrors = typeof errors === "number" && Number.isInteger(errors) && errors >= 0 ? errors : "unknown";
|
|
462
|
+
const reportedDiagnostics = report?.certificate?.diagnostics;
|
|
463
|
+
const runnerErrors = Array.isArray(reportedDiagnostics) ? reportedDiagnostics.filter(error => typeof error === "string") : [];
|
|
464
|
+
const diagnostics = !verdict.pass
|
|
465
|
+
? `\nclassification: ${verdict.meta.classification ?? "unknown"}; runner-level diagnostic: never-started ${neverStarted}; reporter errors ${reporterErrors}`
|
|
466
|
+
+ [...runnerErrors,
|
|
467
|
+
stdoutTail.toString(), stderrTail.toString()].filter(Boolean).map(text => `\n${text}`).join("")
|
|
468
|
+
: "";
|
|
451
469
|
return { pass: verdict.pass, kind: verdict.kind,
|
|
452
|
-
details: verdict.details +
|
|
470
|
+
details: verdict.details + diagnostics,
|
|
453
471
|
classification: verdict.meta.classification,
|
|
454
472
|
meta: { ...verdict.meta, nonce, manifest: files, manifestPath, listingCommand: invocation.listing, verification,
|
|
455
|
-
spawnedCommand, processExit: invoked.exitCode, pid: invoked.pid },
|
|
473
|
+
spawnedCommand, processExit: invoked.exitCode, pid: invoked.pid, stdoutPath, stderrPath },
|
|
456
474
|
exitCode: invoked.exitCode ?? -1, reportPath };
|
|
457
475
|
}
|
|
458
476
|
catch (error) {
|