tickmarkr 2.1.6 → 2.1.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/commands/beat.js +37 -10
- package/dist/cli/commands/run.js +21 -2
- package/dist/cli/commands/stats.js +4 -14
- package/dist/cli/commands/status.js +23 -16
- package/dist/compile/index.d.ts +1 -1
- package/dist/compile/index.js +14 -5
- package/dist/compile/ownership.d.ts +25 -0
- package/dist/compile/ownership.js +142 -0
- package/dist/drivers/herdr.d.ts +1 -0
- package/dist/drivers/herdr.js +7 -6
- package/dist/drivers/types.d.ts +2 -0
- package/dist/drivers/types.js +54 -14
- package/dist/gates/baseline.js +12 -1
- package/dist/run/daemon.js +42 -7
- package/dist/run/journal.js +21 -1
- package/dist/run/outcome.js +8 -1
- package/dist/run/supervision.d.ts +15 -1
- package/dist/run/supervision.js +112 -14
- package/package.json +1 -1
- package/skills/tickmarkr-auto/SKILL.md +2 -1
- package/skills/tickmarkr-loop/SKILL.md +2 -1
- package/skills/tickmarkr-overseer/SKILL.md +9 -0
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +46 -32
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { mkdirSync, renameSync, rmSync, writeFileSync } from "node:fs";
|
|
2
2
|
import { dirname } from "node:path";
|
|
3
3
|
import { tickmarkrDir } from "../../graph/graph.js";
|
|
4
|
-
import { SUPERVISION_BEAT_MS, SUPERVISION_STALE_MS, SUPERVISION_TIERS, beatSupervision, supervisionStandDownPath, } from "../../run/supervision.js";
|
|
4
|
+
import { SUPERVISION_BEAT_MS, SUPERVISION_DEFAULT_THRESHOLD_PCT, SUPERVISION_STALE_MS, SUPERVISION_TIERS, beatSupervision, supervisionStandDownPath, } from "../../run/supervision.js";
|
|
5
5
|
// SUP-04: the writer side of supervision, as a VERB. `beatSupervision` and `SUPERVISION_BEAT_MS` shipped
|
|
6
6
|
// with exactly one in-repo caller — the daemon, on one tier — so `status` printed
|
|
7
7
|
// `orchestrator ARMED / overseer ABSENT / watch ABSENT` while a real overseer worked the run: two thirds
|
|
@@ -18,16 +18,22 @@ import { SUPERVISION_BEAT_MS, SUPERVISION_STALE_MS, SUPERVISION_TIERS, beatSuper
|
|
|
18
18
|
// reads UNREADABLE. Here the writes are unguarded: a failure leaves the process with a non-zero exit
|
|
19
19
|
// and a message instead of a claim. The LOOP belongs to the caller, which is what makes the beat
|
|
20
20
|
// evidence: stop calling it and the tier ages into STALE on its own.
|
|
21
|
+
const VALUE_OPTIONS = ["--seat", "--arm-id", "--pct", "--threshold-pct"];
|
|
21
22
|
export async function beat(argv, cwd = process.cwd()) {
|
|
22
23
|
const standDown = argv.includes("--stand-down");
|
|
23
24
|
const seat = seatOf(argv);
|
|
24
|
-
const
|
|
25
|
+
const pct = percentageOf(argv, "--pct");
|
|
26
|
+
const thresholdPct = percentageOf(argv, "--threshold-pct") ?? SUPERVISION_DEFAULT_THRESHOLD_PCT;
|
|
27
|
+
const armId = optionOf(argv, "--arm-id");
|
|
28
|
+
const named = argv.find((a, i) => !a.startsWith("--") && !VALUE_OPTIONS.includes(argv[i - 1] ?? ""));
|
|
25
29
|
if (!isTier(named)) {
|
|
26
|
-
throw new Error(`usage: tickmarkr beat <${SUPERVISION_TIERS.join("|")}> --seat <identity>
|
|
30
|
+
throw new Error(`usage: tickmarkr beat <${SUPERVISION_TIERS.join("|")}> --seat <identity> ` +
|
|
31
|
+
`[--arm-id <identity> --pct <0..100> --threshold-pct <0..100>] [--stand-down] — ` +
|
|
32
|
+
`got ${named ? `\`${named}\`` : "no tier"}`);
|
|
27
33
|
}
|
|
28
34
|
// SUP-05: NO SEAT, NO BEAT — and the refusal comes before any write, so a refused invocation leaves
|
|
29
|
-
// the tier exactly as it found it. This verb is one-shot: the
|
|
30
|
-
// already exited by the time anyone reads the record, so tier +
|
|
35
|
+
// the tier exactly as it found it. This verb is one-shot: the exitedWriterPid it records has
|
|
36
|
+
// already exited by the time anyone reads the record, so tier + writer + instant is a beat nobody can
|
|
31
37
|
// attribute to a seat. Measured 2026-08-26: a consult seat ran the documented loop verbatim and the
|
|
32
38
|
// board read that tier ARMED with no seat of that tier having armed anything, and a seatless ARMED
|
|
33
39
|
// reads as coverage — worse than ABSENT, because ABSENT sends someone to look.
|
|
@@ -42,16 +48,35 @@ export async function beat(argv, cwd = process.cwd()) {
|
|
|
42
48
|
// uncleared marker would render DISARMED while this verb claimed ARMED, so the removal is unguarded
|
|
43
49
|
// too — `force` makes the ordinary "no marker" case a no-op, and anything else is a real failure.
|
|
44
50
|
rmSync(supervisionStandDownPath(cwd, named), { force: true, recursive: true });
|
|
45
|
-
beatSupervision(cwd, named, seat);
|
|
51
|
+
beatSupervision(cwd, named, seat, pct === undefined ? undefined : { armId: armId ?? seat, pct, thresholdPct });
|
|
46
52
|
return `${named} ARMED as ${seat} — beat again every ${SUPERVISION_BEAT_MS / 1_000}s; the tier reads STALE ${SUPERVISION_STALE_MS / 1_000}s after the last beat`;
|
|
47
53
|
}
|
|
48
54
|
/** `--seat <identity>` or `--seat=<identity>`; blank and missing are the same answer — none. */
|
|
49
55
|
function seatOf(argv) {
|
|
50
|
-
const
|
|
51
|
-
const spaced = argv[argv.indexOf("--seat") + 1];
|
|
52
|
-
const seat = (inline ?? (argv.includes("--seat") ? spaced : undefined))?.trim();
|
|
56
|
+
const seat = optionOf(argv, "--seat")?.trim();
|
|
53
57
|
return seat && !seat.startsWith("--") ? seat : undefined;
|
|
54
58
|
}
|
|
59
|
+
/** A `--name value` or `--name=value` option, excluding a missing value or the next flag. */
|
|
60
|
+
function optionOf(argv, name) {
|
|
61
|
+
const inline = argv.find((a) => a.startsWith(`${name}=`))?.slice(name.length + 1).trim();
|
|
62
|
+
const spaced = argv[argv.indexOf(name) + 1]?.trim();
|
|
63
|
+
const value = inline ?? (argv.includes(name) ? spaced : undefined);
|
|
64
|
+
return value && !value.startsWith("--") ? value : undefined;
|
|
65
|
+
}
|
|
66
|
+
function percentageOf(argv, name) {
|
|
67
|
+
const raw = optionOf(argv, name);
|
|
68
|
+
if (raw === undefined) {
|
|
69
|
+
if (argv.includes(name) || argv.some((a) => a.startsWith(`${name}=`))) {
|
|
70
|
+
throw new Error(`${name} needs a number from 0 through 100`);
|
|
71
|
+
}
|
|
72
|
+
return undefined;
|
|
73
|
+
}
|
|
74
|
+
const value = Number(raw);
|
|
75
|
+
if (!Number.isFinite(value) || value < 0 || value > 100) {
|
|
76
|
+
throw new Error(`${name} must be a number from 0 through 100`);
|
|
77
|
+
}
|
|
78
|
+
return value;
|
|
79
|
+
}
|
|
55
80
|
// Stand-down is a RECORDED act, not a silence: the marker tells a reader this watcher left on purpose,
|
|
56
81
|
// so the tier reads DISARMED rather than ageing out as a death. Published atomically — written aside,
|
|
57
82
|
// renamed over — because a torn marker is rejected by the reader, and a rejected stand-down reports a
|
|
@@ -63,7 +88,9 @@ function standDownTier(repoRoot, tier, seat) {
|
|
|
63
88
|
mkdirSync(dirname(p), { recursive: true });
|
|
64
89
|
// The marker names the seat for the same reason the beat does: "someone stood this tier down" is not
|
|
65
90
|
// a hand-off anyone can act on, and on a seat tier the reader rejects an anonymous one outright.
|
|
66
|
-
writeFileSync(tmp, JSON.stringify({
|
|
91
|
+
writeFileSync(tmp, JSON.stringify({
|
|
92
|
+
tier, seat, exitedWriterPid: process.pid, disarmedAt: new Date().toISOString(),
|
|
93
|
+
}) + "\n");
|
|
67
94
|
renameSync(tmp, p);
|
|
68
95
|
return `${tier} DISARMED — ${seat} handed off; status reads it stood down, not dead`;
|
|
69
96
|
}
|
package/dist/cli/commands/run.js
CHANGED
|
@@ -167,7 +167,7 @@ const railTone = (event, fallback) => {
|
|
|
167
167
|
// `gate-result` is the one event whose data IS a gate result; every other row carrying a `gate`
|
|
168
168
|
// names a gate without reporting one (a start, a reuse, a retry) and keeps its declared tone.
|
|
169
169
|
if (event.event === "gate-result")
|
|
170
|
-
return GATE_OUTCOME_TONES[normalizeGateOutcome(event.data).kind]
|
|
170
|
+
return GATE_OUTCOME_TONES[normalizeGateOutcome(event.data).kind];
|
|
171
171
|
const verdict = event.data.pass ?? event.data.ok;
|
|
172
172
|
return verdict === true ? "pass" : verdict === false ? "fail" : fallback;
|
|
173
173
|
};
|
|
@@ -226,7 +226,24 @@ const railSalient = (data) => RAIL_SALIENT.flatMap(({ key, render }) => {
|
|
|
226
226
|
* This is one detail vocabulary stated twice, so `brand-surfaces.test.ts` pins the two against each
|
|
227
227
|
* other over the whole event corpus: the pipe's bytes must equal event/taskId/THIS detail put through
|
|
228
228
|
* the formatter's legacy squeeze-and-slice. A ladder that drifts fails there rather than in a tab.
|
|
229
|
+
* Gate-result verdict words are the deliberate exception: the pipe is byte-frozen, while the TTY rail
|
|
230
|
+
* names the normalized outcome so a held selected-test screen is not narrated as a pass.
|
|
229
231
|
*/
|
|
232
|
+
const gateResultDetail = (data) => {
|
|
233
|
+
if (typeof data.gate !== "string")
|
|
234
|
+
return undefined;
|
|
235
|
+
switch (normalizeGateOutcome(data).kind) {
|
|
236
|
+
case "passed":
|
|
237
|
+
return `${data.gate} passed`;
|
|
238
|
+
case "failed":
|
|
239
|
+
return `${data.gate} failed`;
|
|
240
|
+
case "held":
|
|
241
|
+
// The kind is already normalized; the selected-test list only chooses this surface noun.
|
|
242
|
+
return Array.isArray(data.selectedTests) ? `${data.gate} selected-test screen` : `${data.gate} held`;
|
|
243
|
+
default:
|
|
244
|
+
return data.gate;
|
|
245
|
+
}
|
|
246
|
+
};
|
|
230
247
|
const formatterDetail = ({ event, data }) => {
|
|
231
248
|
const assignment = data.assignment;
|
|
232
249
|
const direct = [data.summary, data.reason, data.error, data.step, data.action, data.lint, data.branch, data.from]
|
|
@@ -240,7 +257,9 @@ const formatterDetail = ({ event, data }) => {
|
|
|
240
257
|
}
|
|
241
258
|
if (event === "tip-verify")
|
|
242
259
|
return `${data.gate} passed`;
|
|
243
|
-
|
|
260
|
+
if (event === "gate-result")
|
|
261
|
+
return gateResultDetail(data);
|
|
262
|
+
return `${data.gate}`;
|
|
244
263
|
}
|
|
245
264
|
if (typeof data.code === "number")
|
|
246
265
|
return `exit ${data.code}`;
|
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import { existsSync, readdirSync } from "node:fs";
|
|
2
2
|
import { join } from "node:path";
|
|
3
|
-
import { classifyFailureOutput } from "../../gates/baseline.js";
|
|
4
3
|
import { stateDirName } from "../../graph/graph.js";
|
|
5
4
|
import { Journal } from "../../run/journal.js";
|
|
5
|
+
import { normalizeGateOutcome } from "../../run/outcome.js";
|
|
6
6
|
// The operator-facing schema is deliberately closed. Adding a fact to the table therefore requires
|
|
7
7
|
// adding it here as a conscious compatibility change, rather than letting incidental journal fields
|
|
8
8
|
// leak into an analytics surface.
|
|
@@ -35,14 +35,6 @@ const reviewerFrom = (data) => {
|
|
|
35
35
|
return undefined;
|
|
36
36
|
return /\breviewer(?:\s+|:\s*)([\w@./+-]+:[\w@./+-]+)/iu.exec(data.details)?.[1];
|
|
37
37
|
};
|
|
38
|
-
const redEvidence = (data) => {
|
|
39
|
-
const fingerprints = Array.isArray(data.fingerprints)
|
|
40
|
-
? data.fingerprints.filter((value) => typeof value === "string")
|
|
41
|
-
: [];
|
|
42
|
-
const prose = [data.details, data.error, data.reason]
|
|
43
|
-
.filter((value) => typeof value === "string");
|
|
44
|
-
return [...fingerprints, ...prose].join("\n");
|
|
45
|
-
};
|
|
46
38
|
const historyFor = (tasks, taskId) => {
|
|
47
39
|
let history = tasks.get(taskId);
|
|
48
40
|
if (!history) {
|
|
@@ -143,12 +135,10 @@ export function collectChannelStats(cwd = process.cwd()) {
|
|
|
143
135
|
if (reviewer)
|
|
144
136
|
channel.reviewers.add(reviewer);
|
|
145
137
|
}
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
const infra = event.data.infra === true || classifyFailureOutput(redEvidence(event.data)) === "infra";
|
|
149
|
-
if (infra)
|
|
138
|
+
const outcome = normalizeGateOutcome(event.data);
|
|
139
|
+
if (outcome.kind === "infra")
|
|
150
140
|
channel.infraReds += 1;
|
|
151
|
-
else
|
|
141
|
+
else if (outcome.kind === "failed")
|
|
152
142
|
channel.realReds += 1;
|
|
153
143
|
}
|
|
154
144
|
// A rescue is task-matched, not attempt-matched: one edge says the failed author and delivering
|
|
@@ -12,7 +12,7 @@ import { isPidLive } from "../../run/lock.js";
|
|
|
12
12
|
import { normalizeGateOutcome } from "../../run/outcome.js";
|
|
13
13
|
import { desiredPanes } from "../../run/reconcile.js";
|
|
14
14
|
import { normalizeStallSnapshot } from "../../run/stall.js";
|
|
15
|
-
import {
|
|
15
|
+
import { armWatchSupervision, readSupervision, supervisionText } from "../../run/supervision.js";
|
|
16
16
|
import { deriveRunCockpitData, } from "../../tui/cockpit/derive.js";
|
|
17
17
|
import { COCKPIT_COLUMN_FLOOR } from "../../tui/cockpit/layout.js";
|
|
18
18
|
import { cellWidth, fitCells, wrapCells } from "../../tui/cockpit/width.js";
|
|
@@ -109,6 +109,15 @@ const ASCII_SPINNER = ["|", "/", "-", "\\"];
|
|
|
109
109
|
const SAVE_TERMINAL_TITLE = "\x1b[22;0t";
|
|
110
110
|
const RESTORE_TERMINAL_TITLE = "\x1b[23;0t";
|
|
111
111
|
const decisionEvidence = (stateDir, runId, sequence) => `${stateDir}/runs/${runId}/journal.jsonl#L${sequence}`;
|
|
112
|
+
const DECISION_GATE_VERDICTS = {
|
|
113
|
+
passed: "passed",
|
|
114
|
+
failed: "failed",
|
|
115
|
+
skipped: "skipped",
|
|
116
|
+
declined: "skipped",
|
|
117
|
+
held: "unknown",
|
|
118
|
+
unavailable: "unknown",
|
|
119
|
+
infra: "unknown",
|
|
120
|
+
};
|
|
112
121
|
// v1.79 T4: a deterministic JSONL projection over journal truth. Sequence/evidence come from the
|
|
113
122
|
// append-only line position, timestamps and claims come from the row, and no watcher-local clock or
|
|
114
123
|
// filesystem write participates. Re-reading the same bytes therefore returns the same event bytes.
|
|
@@ -126,13 +135,7 @@ export const decisionEventsFromJournal = (events, runId, stateDir = ".tickmarkr"
|
|
|
126
135
|
return [{ ...base, type: "phase-change", tier: "routine", phase: event.data.phase }];
|
|
127
136
|
}
|
|
128
137
|
if (event.event === "gate-result" && typeof event.data.gate === "string") {
|
|
129
|
-
const verdict = event.data.
|
|
130
|
-
? "skipped"
|
|
131
|
-
: event.data.pass === true
|
|
132
|
-
? "passed"
|
|
133
|
-
: event.data.pass === false
|
|
134
|
-
? "failed"
|
|
135
|
-
: "unknown";
|
|
138
|
+
const verdict = DECISION_GATE_VERDICTS[normalizeGateOutcome(event.data).kind];
|
|
136
139
|
return [{
|
|
137
140
|
...base,
|
|
138
141
|
type: "gate-verdict",
|
|
@@ -381,6 +384,13 @@ export const gateBox = (state, unicode) => {
|
|
|
381
384
|
return state === "pass" ? "[x]" : state === "fail" ? "[!]" : state === "skip" ? "." : "[ ]";
|
|
382
385
|
};
|
|
383
386
|
export const defaultGateStates = (task) => GATE_NAMES.map((gate) => task.gates.includes(gate) ? "open" : "skip");
|
|
387
|
+
const GATE_OUTCOME_STATES = {
|
|
388
|
+
// The board state is a rendering vocabulary over the normalized outcome, not another field reader.
|
|
389
|
+
passed: "pass",
|
|
390
|
+
failed: "fail",
|
|
391
|
+
skipped: "skip",
|
|
392
|
+
declined: "skip",
|
|
393
|
+
};
|
|
384
394
|
const gateSnapshot = (task, events, rehashAt) => {
|
|
385
395
|
const outcomes = new Map();
|
|
386
396
|
const start = attemptStartIdx(events, task.id);
|
|
@@ -389,12 +399,9 @@ const gateSnapshot = (task, events, rehashAt) => {
|
|
|
389
399
|
const e = events[eventIndex];
|
|
390
400
|
if (e.taskId !== task.id || e.event !== "gate-result" || typeof e.data.gate !== "string")
|
|
391
401
|
continue;
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
outcomes.set(e.data.gate, { state: "pass", eventIndex });
|
|
396
|
-
else if (e.data.pass === false)
|
|
397
|
-
outcomes.set(e.data.gate, { state: "fail", eventIndex });
|
|
402
|
+
const state = GATE_OUTCOME_STATES[normalizeGateOutcome(e.data).kind];
|
|
403
|
+
if (state)
|
|
404
|
+
outcomes.set(e.data.gate, { state, eventIndex });
|
|
398
405
|
}
|
|
399
406
|
}
|
|
400
407
|
return {
|
|
@@ -519,7 +526,7 @@ const foldTaskEffort = (tasks, events) => {
|
|
|
519
526
|
task.dispatches += 1;
|
|
520
527
|
else if (event.event === "gate-result" && event.data.gate === "review"
|
|
521
528
|
&& event.data.replayMeasurement !== true
|
|
522
|
-
&& REVIEW_VERDICTS.has(normalizeGateOutcome(event.data
|
|
529
|
+
&& REVIEW_VERDICTS.has(normalizeGateOutcome(event.data).kind))
|
|
523
530
|
task.reviews += 1;
|
|
524
531
|
else if (event.event === "task-human")
|
|
525
532
|
task.parks += 1;
|
|
@@ -1408,7 +1415,7 @@ export async function status(argv, cwd = process.cwd(), opts = {}) {
|
|
|
1408
1415
|
// it INVERTS it: armSupervision's interval outlives the board in any host that outlives one board, so
|
|
1409
1416
|
// a dead board keeps writing and reads ARMED. That is the over-claiming direction, the one an operator
|
|
1410
1417
|
// acts on, and the one this instrument exists to close.
|
|
1411
|
-
const armed = bounded ? undefined :
|
|
1418
|
+
const armed = bounded ? undefined : armWatchSupervision(cwd, opts.supervisionBeatMs);
|
|
1412
1419
|
let titleSaved = false;
|
|
1413
1420
|
const restoreTitle = () => {
|
|
1414
1421
|
if (!titleSaved)
|
package/dist/compile/index.d.ts
CHANGED
|
@@ -10,6 +10,6 @@ export type PlanIR = {
|
|
|
10
10
|
tasks: RunGraph["tasks"];
|
|
11
11
|
};
|
|
12
12
|
export type PlanFinalizationHook = (plan: PlanIR) => PlanIR;
|
|
13
|
-
export declare function finalizePlan(plan: PlanIR, src: string): RunGraph;
|
|
13
|
+
export declare function finalizePlan(plan: PlanIR, src: string, repoRoot?: string): RunGraph;
|
|
14
14
|
export declare function compileSource(src: string, type?: SourceType, root?: string, beforeFinalize?: PlanFinalizationHook): RunGraph;
|
|
15
15
|
export {};
|
package/dist/compile/index.js
CHANGED
|
@@ -2,6 +2,7 @@ import { existsSync, readFileSync, statSync } from "node:fs";
|
|
|
2
2
|
import { join } from "node:path";
|
|
3
3
|
import { validateGraph } from "../graph/schema.js";
|
|
4
4
|
import { taskUnitContractErrors } from "./collateral.js";
|
|
5
|
+
import { ownershipFindings, renderOwnershipFinding } from "./ownership.js";
|
|
5
6
|
import { CompileError } from "./common.js";
|
|
6
7
|
import { compileGsd, isGsdPhaseDir } from "./gsd.js";
|
|
7
8
|
import { compileNative, TICKMARKR_NATIVE_MARKER } from "./native.js";
|
|
@@ -40,8 +41,8 @@ function enforceTaskUnitContract(g, src) {
|
|
|
40
41
|
}
|
|
41
42
|
return g;
|
|
42
43
|
}
|
|
43
|
-
export function finalizePlan(plan, src) {
|
|
44
|
-
const graph = validateGraph({
|
|
44
|
+
export function finalizePlan(plan, src, repoRoot) {
|
|
45
|
+
const graph = enforceTaskUnitContract(validateGraph({
|
|
45
46
|
version: plan.version,
|
|
46
47
|
...(plan.mode !== undefined ? { mode: plan.mode } : {}),
|
|
47
48
|
spec: {
|
|
@@ -51,8 +52,16 @@ export function finalizePlan(plan, src) {
|
|
|
51
52
|
...(plan.base !== undefined ? { base: plan.base } : {}),
|
|
52
53
|
},
|
|
53
54
|
tasks: plan.tasks,
|
|
54
|
-
});
|
|
55
|
-
|
|
55
|
+
}), src);
|
|
56
|
+
// overseer-217: report-only in 2.1.7. The 2.1.8 removal condition is one real authored-graph run
|
|
57
|
+
// plus a measured false-positive rate for the conventional source-name → test-name mapping.
|
|
58
|
+
// Until then this warning MUST NOT become a compile abort: a heuristic cannot deadlock compile.
|
|
59
|
+
if (repoRoot) {
|
|
60
|
+
for (const finding of ownershipFindings(graph.tasks, repoRoot)) {
|
|
61
|
+
console.warn(renderOwnershipFinding(finding));
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
return graph;
|
|
56
65
|
}
|
|
57
66
|
function compilePlan(src, type, root) {
|
|
58
67
|
const kind = type ?? detect(src);
|
|
@@ -75,5 +84,5 @@ function compilePlan(src, type, root) {
|
|
|
75
84
|
};
|
|
76
85
|
}
|
|
77
86
|
export function compileSource(src, type, root, beforeFinalize = (plan) => plan) {
|
|
78
|
-
return finalizePlan(beforeFinalize(compilePlan(src, type, root)), src);
|
|
87
|
+
return finalizePlan(beforeFinalize(compilePlan(src, type, root)), src, root);
|
|
79
88
|
}
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
import type { Task } from "../graph/schema.js";
|
|
2
|
+
export type OwnershipFinding = {
|
|
3
|
+
code: "unowned-test";
|
|
4
|
+
test: string;
|
|
5
|
+
taskIds: string[];
|
|
6
|
+
detail: string;
|
|
7
|
+
} | {
|
|
8
|
+
code: "test-path-outside-allowlist";
|
|
9
|
+
taskId: string;
|
|
10
|
+
test: string;
|
|
11
|
+
path: string;
|
|
12
|
+
detail: string;
|
|
13
|
+
} | {
|
|
14
|
+
code: "unordered-context-write";
|
|
15
|
+
taskId: string;
|
|
16
|
+
ownerTaskId: string;
|
|
17
|
+
path: string;
|
|
18
|
+
detail: string;
|
|
19
|
+
};
|
|
20
|
+
/**
|
|
21
|
+
* Advisory cross-task ownership check. Findings are data: callers may report them, but this checker
|
|
22
|
+
* never throws and never changes the graph.
|
|
23
|
+
*/
|
|
24
|
+
export declare function ownershipFindings(tasks: readonly Task[], repoRoot: string): OwnershipFinding[];
|
|
25
|
+
export declare function renderOwnershipFinding(finding: OwnershipFinding): string;
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
import { readFileSync, readdirSync } from "node:fs";
|
|
2
|
+
import { basename, extname, join } from "node:path";
|
|
3
|
+
import { filesGlob } from "../graph/files-glob.js";
|
|
4
|
+
import { collateralHits } from "./collateral.js";
|
|
5
|
+
const normalize = (path) => path.replace(/^\.\//, "").split("\\").join("/");
|
|
6
|
+
function testSources(repoRoot) {
|
|
7
|
+
const root = join(repoRoot, "tests");
|
|
8
|
+
try {
|
|
9
|
+
return readdirSync(root, { recursive: true, encoding: "utf8" })
|
|
10
|
+
.filter((path) => path.endsWith(".test.ts"))
|
|
11
|
+
.sort()
|
|
12
|
+
.flatMap((path) => {
|
|
13
|
+
try {
|
|
14
|
+
return [{ path: `tests/${normalize(path)}`, text: readFileSync(join(root, path), "utf8") }];
|
|
15
|
+
}
|
|
16
|
+
catch {
|
|
17
|
+
return [];
|
|
18
|
+
}
|
|
19
|
+
});
|
|
20
|
+
}
|
|
21
|
+
catch {
|
|
22
|
+
return [];
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
// Conventional, not universal: status-watch-alive.test.ts is dedicated to status.ts even though it
|
|
26
|
+
// reaches that command through the CLI entry point. Keep this heuristic advisory until authored-graph
|
|
27
|
+
// measurements establish its false-positive rate.
|
|
28
|
+
function namedSourceTasks(test, tasks) {
|
|
29
|
+
const stem = basename(test).replace(/\.test\.ts$/, "");
|
|
30
|
+
const ids = new Set();
|
|
31
|
+
for (const task of tasks) {
|
|
32
|
+
for (const entry of task.files.map(normalize).filter((path) => path.startsWith("src/") && !/[*?{[]/.test(path))) {
|
|
33
|
+
const source = basename(entry, extname(entry));
|
|
34
|
+
if (stem === source || stem.startsWith(`${source}-`))
|
|
35
|
+
ids.add(task.id);
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
return [...ids];
|
|
39
|
+
}
|
|
40
|
+
function repositoryPaths(text) {
|
|
41
|
+
const paths = new Set();
|
|
42
|
+
for (const match of text.matchAll(/["'`]((?:src|tests|fixtures|scripts|skills|docs|\.claude)\/[^"'`\s]+)["'`]/g)) {
|
|
43
|
+
paths.add(normalize(match[1]));
|
|
44
|
+
}
|
|
45
|
+
return [...paths].sort();
|
|
46
|
+
}
|
|
47
|
+
function dependencyOrdered(a, b, byId) {
|
|
48
|
+
const reaches = (from, target) => {
|
|
49
|
+
const seen = new Set();
|
|
50
|
+
const pending = [...from.deps];
|
|
51
|
+
while (pending.length > 0) {
|
|
52
|
+
const id = pending.pop();
|
|
53
|
+
if (id === target)
|
|
54
|
+
return true;
|
|
55
|
+
if (seen.has(id))
|
|
56
|
+
continue;
|
|
57
|
+
seen.add(id);
|
|
58
|
+
pending.push(...(byId.get(id)?.deps ?? []));
|
|
59
|
+
}
|
|
60
|
+
return false;
|
|
61
|
+
};
|
|
62
|
+
return reaches(a, b.id) || reaches(b, a.id);
|
|
63
|
+
}
|
|
64
|
+
/**
|
|
65
|
+
* Advisory cross-task ownership check. Findings are data: callers may report them, but this checker
|
|
66
|
+
* never throws and never changes the graph.
|
|
67
|
+
*/
|
|
68
|
+
export function ownershipFindings(tasks, repoRoot) {
|
|
69
|
+
const sources = testSources(repoRoot);
|
|
70
|
+
const byId = new Map(tasks.map((task) => [task.id, task]));
|
|
71
|
+
const indexed = tasks.map((task) => {
|
|
72
|
+
const files = task.files.map(normalize);
|
|
73
|
+
const context = task.context.map(normalize);
|
|
74
|
+
return {
|
|
75
|
+
task,
|
|
76
|
+
owns: files.length === 0 ? () => false : filesGlob(files),
|
|
77
|
+
allows: files.length === 0 ? () => true : filesGlob([...files, ...context]),
|
|
78
|
+
};
|
|
79
|
+
});
|
|
80
|
+
const owners = (path) => indexed.filter(({ owns }) => owns(path));
|
|
81
|
+
const predictedBy = new Map();
|
|
82
|
+
for (const [taskId, hits] of collateralHits(tasks, repoRoot)) {
|
|
83
|
+
for (const hit of hits) {
|
|
84
|
+
const ids = predictedBy.get(hit) ?? new Set();
|
|
85
|
+
ids.add(taskId);
|
|
86
|
+
predictedBy.set(hit, ids);
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
for (const source of sources) {
|
|
90
|
+
const ids = predictedBy.get(source.path) ?? new Set();
|
|
91
|
+
for (const taskId of namedSourceTasks(source.path, tasks))
|
|
92
|
+
ids.add(taskId);
|
|
93
|
+
if (ids.size > 0)
|
|
94
|
+
predictedBy.set(source.path, ids);
|
|
95
|
+
}
|
|
96
|
+
const findings = [];
|
|
97
|
+
for (const [test, taskIds] of predictedBy) {
|
|
98
|
+
if (owners(test).length === 0) {
|
|
99
|
+
const ids = [...taskIds].sort();
|
|
100
|
+
findings.push({
|
|
101
|
+
code: "unowned-test",
|
|
102
|
+
test,
|
|
103
|
+
taskIds: ids,
|
|
104
|
+
detail: `${test} is a dedicated test of source owned by ${ids.join(", ")} but no task owns the test`,
|
|
105
|
+
});
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
for (const source of sources) {
|
|
109
|
+
for (const owner of owners(source.path)) {
|
|
110
|
+
for (const path of repositoryPaths(source.text)) {
|
|
111
|
+
if (!owner.allows(path)) {
|
|
112
|
+
findings.push({
|
|
113
|
+
code: "test-path-outside-allowlist",
|
|
114
|
+
taskId: owner.task.id,
|
|
115
|
+
test: source.path,
|
|
116
|
+
path,
|
|
117
|
+
detail: `${source.path} owned by ${owner.task.id} references ${path} outside that task's files[] and context[]`,
|
|
118
|
+
});
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
for (const reader of indexed) {
|
|
124
|
+
for (const entry of reader.task.context.map(normalize)) {
|
|
125
|
+
for (const owner of owners(entry)) {
|
|
126
|
+
if (owner.task.id === reader.task.id || dependencyOrdered(reader.task, owner.task, byId))
|
|
127
|
+
continue;
|
|
128
|
+
findings.push({
|
|
129
|
+
code: "unordered-context-write",
|
|
130
|
+
taskId: reader.task.id,
|
|
131
|
+
ownerTaskId: owner.task.id,
|
|
132
|
+
path: entry,
|
|
133
|
+
detail: `${reader.task.id} names ${entry} as context while ${owner.task.id} owns it for writing, with no dependency order between them`,
|
|
134
|
+
});
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
return findings.sort((a, b) => `${a.code}:${"test" in a ? a.test : a.path}:${"taskId" in a ? a.taskId : ""}`.localeCompare(`${b.code}:${"test" in b ? b.test : b.path}:${"taskId" in b ? b.taskId : ""}`));
|
|
139
|
+
}
|
|
140
|
+
export function renderOwnershipFinding(finding) {
|
|
141
|
+
return `tickmarkr: ownership-lint[${finding.code}]: ${finding.detail}`;
|
|
142
|
+
}
|
package/dist/drivers/herdr.d.ts
CHANGED
|
@@ -125,6 +125,7 @@ export declare class HerdrDriver implements ExecutorDriver {
|
|
|
125
125
|
narrator(cwd: string, command: string, runId?: string): Promise<Slot>;
|
|
126
126
|
reconcile(desired: Set<string>, runId: string, opts?: {
|
|
127
127
|
spareLiveLlm?: boolean;
|
|
128
|
+
endedRunIds?: Set<string>;
|
|
128
129
|
}): Promise<void>;
|
|
129
130
|
worktree(repo: string, branch: string, baseRef: string): Promise<string>;
|
|
130
131
|
}
|
package/dist/drivers/herdr.js
CHANGED
|
@@ -1188,12 +1188,13 @@ export class HerdrDriver {
|
|
|
1188
1188
|
return s;
|
|
1189
1189
|
});
|
|
1190
1190
|
}
|
|
1191
|
-
// OBS-17 T2 / v1.22b T1: close
|
|
1192
|
-
// killed-daemon orphans
|
|
1193
|
-
//
|
|
1194
|
-
//
|
|
1195
|
-
//
|
|
1196
|
-
//
|
|
1191
|
+
// OBS-17 T2 / v1.22b T1: close THIS RUN'S OWN tickmarkr-owned panes that should not exist
|
|
1192
|
+
// (superseded attempts, killed-daemon orphans of this run), in this run's workspace, then reap the
|
|
1193
|
+
// tabs those closes emptied. Ownership is decided ONLY by parseOwnedName (drivers/types.ts
|
|
1194
|
+
// panesToClose) — foreign names never become candidates. OBS-769/OBS-772/OBS-777: the sweep does
|
|
1195
|
+
// NOT reach other workspaces and reaches another runId only when the daemon's repository-scoped
|
|
1196
|
+
// snapshot proves that run ended. Run age is not consulted and never was. spareLiveLlm: same-run
|
|
1197
|
+
// judge/review/consult panes have no journal row while live (their
|
|
1197
1198
|
// events land after the verdict is read), so mid-run sweeps spare them; boundary sweeps (start/
|
|
1198
1199
|
// resume/end) run with nothing in flight and take them too. Cosmetic by contract: every failure —
|
|
1199
1200
|
// herdr gone, pane vanished mid-sweep, unparseable listing — is swallowed; this method never throws.
|
package/dist/drivers/types.d.ts
CHANGED
|
@@ -56,6 +56,7 @@ export interface FleetAgent {
|
|
|
56
56
|
}
|
|
57
57
|
export declare function panesToClose(agents: FleetAgent[], desired: Set<string>, ws: string, runId: string, opts?: {
|
|
58
58
|
spareLiveLlm?: boolean;
|
|
59
|
+
endedRunIds?: Set<string>;
|
|
59
60
|
}): {
|
|
60
61
|
paneId: string;
|
|
61
62
|
tabId?: string;
|
|
@@ -81,5 +82,6 @@ export interface ExecutorDriver {
|
|
|
81
82
|
narrator?: (cwd: string, command: string, runId?: string) => Promise<Slot>;
|
|
82
83
|
reconcile?: (desired: Set<string>, runId: string, opts?: {
|
|
83
84
|
spareLiveLlm?: boolean;
|
|
85
|
+
endedRunIds?: Set<string>;
|
|
84
86
|
}) => Promise<void>;
|
|
85
87
|
}
|
package/dist/drivers/types.js
CHANGED
|
@@ -21,12 +21,53 @@ export function isForeignName(name) {
|
|
|
21
21
|
return parseOwnedName(name) === null;
|
|
22
22
|
}
|
|
23
23
|
// v1.22b T1: workspace-aware fold over a fleet snapshot — decides which owned task panes are garbage
|
|
24
|
-
// right now
|
|
25
|
-
//
|
|
26
|
-
// and closes regardless of `desired`; an owned pane from THIS run elsewhere is left alone — a live
|
|
27
|
-
// run can legitimately hold panes across workspaces, so only run age marks a misplaced pane garbage.
|
|
24
|
+
// right now: the desired-set/spareLiveLlm sweep (OBS-17 T2), scoped to THIS RUN'S OWN panes (by runId,
|
|
25
|
+
// OBS-772) in THIS RUN'S OWN WORKSPACE (OBS-769). Both conditions, and neither alone is the rule.
|
|
28
26
|
// Watch panes are operator-owned after run end and are reclaimed by the next run; foreign names
|
|
29
|
-
// (parseOwnedName fails) are never candidates
|
|
27
|
+
// (parseOwnedName fails) are never candidates.
|
|
28
|
+
//
|
|
29
|
+
// OBS-769 — WHY THE SWEEP STOPS AT THE WORKSPACE BOUNDARY. It used to close an owned pane carrying
|
|
30
|
+
// any OTHER runId in any other workspace, unconditionally, as a "misplaced leftover". Two tickmarkr
|
|
31
|
+
// runs in two repositories are lawful (the lock forbids two runs in ONE repository, not on one
|
|
32
|
+
// machine) and herdr gives each its own workspace — so that branch made every pair of concurrent
|
|
33
|
+
// runs kill each other's LIVE workers. Measured 2026-08-28: the run in w0 closed run ...2958's
|
|
34
|
+
// panes at 23:42:40.351/.392, and 53s later ...2958's own task-human sweep closed w0's live codex
|
|
35
|
+
// worker at 23:43:34.096. ...2958 ended 0/8. The death detector cannot see it: closing the pane
|
|
36
|
+
// makes paneAbsent, processTree, confirmedProcessTree and worktreeDelta true by ONE cause, and a
|
|
37
|
+
// closed pane can never accrue the CPU that the `cpu-accruing` hold reads.
|
|
38
|
+
// The comment this replaces claimed "only run age marks a misplaced pane garbage" — there was no age
|
|
39
|
+
// check in the code, and age is the wrong predicate anyway: w0's run STARTED EARLIER than ...2958,
|
|
40
|
+
// so an age rule would have licensed exactly the kill that landed. Run age says nothing about
|
|
41
|
+
// liveness, and a sweeping daemon cannot read another repository's run state. The workspace is the
|
|
42
|
+
// only ownership boundary available without cross-repo I/O, so it is the one enforced.
|
|
43
|
+
// Cost, named: an orphan pane from a dead run stranded in a workspace no later run opens is now left
|
|
44
|
+
// for the operator. That is cosmetic (`reconcile` is cosmetic by contract — "visibility is never a
|
|
45
|
+
// gate"), and a cosmetic cleanup must never be able to kill a live worker.
|
|
46
|
+
// OBS-772 — WHY THE runId LINE EXISTS, AND WHY THE WORKSPACE LINE ALONE WAS NOT THE FIX. The first
|
|
47
|
+
// repair was workspace-scoped only, and its own comment dismissed the residue — "two runs sharing one
|
|
48
|
+
// workspace would still sweep each other" — as unreachable, on the reasoning that one workspace per run
|
|
49
|
+
// is herdr's placement. That reasoned from ONE driver to the whole product. OrcaDriver has no workspace
|
|
50
|
+
// dimension at all: orca.ts passes a single ORCA_SPACE as the workspaceId for EVERY checkout and as
|
|
51
|
+
// `ws`, so `workspaceId !== ws` is never true there and every foreign pane fell straight through. Orca
|
|
52
|
+
// users had zero protection while the defect read as fixed. The runId line is the real rule and it is
|
|
53
|
+
// driver-agnostic: reconcile exists to clean up THIS RUN's panes, and a leftover from a dead run is
|
|
54
|
+
// exactly what cannot be told from a live run's pane without liveness data this process does not have.
|
|
55
|
+
// Both lines are kept — the workspace line preserves the pre-existing sparing of this run's own panes
|
|
56
|
+
// in another workspace, which the runId line alone would not.
|
|
57
|
+
// ⚠ WHAT THE runId LINE COST BEFORE OBS-777 — SUSPENDED, NOT NARROWED, and the price was larger than
|
|
58
|
+
// it read. Sparing every other runId suspended OBS-17's FOUNDING use case: "a killed daemon can't
|
|
59
|
+
// close its slots". This sweep was built to reclaim exactly those orphans, but could not reclaim ANY
|
|
60
|
+
// previous run's panes. Three separate pins asserted the old behaviour (reconcile.test.ts,
|
|
61
|
+
// orca-placement.test.ts,
|
|
62
|
+
// reconcile-live.test.ts); all three were changed deliberately, and the third is why this paragraph
|
|
63
|
+
// exists rather than a shorter one — two flipped pins is a trade, three is a pattern.
|
|
64
|
+
// OBS-777 RESTORES that reclamation: the CALLER passes `opts.endedRunIds`, a Set the daemon computes
|
|
65
|
+
// ONCE at run start from this repository's own `run-end` journals and dead lock holders. This fold
|
|
66
|
+
// stays pure — it gains one optional field, not a repo root — a foreign repository's runId is never
|
|
67
|
+
// resolvable and so stays spared by construction, and no driver learns about workspaces.
|
|
68
|
+
// ponytail: two conditions, no geometry reasoning, nothing driver-specific. `reconcile` is cosmetic by
|
|
69
|
+
// contract, and a cosmetic cleanup must never be able to kill a live worker — which is why the
|
|
70
|
+
// ended-run authority is the only safe way to restore the sweep without reviving the cross-run kill.
|
|
30
71
|
export function panesToClose(agents, desired, ws, runId, opts) {
|
|
31
72
|
const out = [];
|
|
32
73
|
for (const a of agents) {
|
|
@@ -35,15 +76,14 @@ export function panesToClose(agents, desired, ws, runId, opts) {
|
|
|
35
76
|
const owned = parseOwnedName(a.name);
|
|
36
77
|
if (!owned || owned.role === "watch")
|
|
37
78
|
continue;
|
|
38
|
-
if (
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
continue;
|
|
46
|
-
}
|
|
79
|
+
if (owned.runId !== runId && !opts?.endedRunIds?.has(owned.runId))
|
|
80
|
+
continue;
|
|
81
|
+
if (a.workspaceId !== ws)
|
|
82
|
+
continue; // OBS-769: another workspace is another run's business
|
|
83
|
+
if (desired.has(a.name))
|
|
84
|
+
continue;
|
|
85
|
+
if (opts?.spareLiveLlm && owned.runId === runId && (owned.role === "judge" || owned.role === "review" || owned.role === "consult"))
|
|
86
|
+
continue;
|
|
47
87
|
out.push({ paneId: a.paneId, tabId: a.tabId });
|
|
48
88
|
}
|
|
49
89
|
return out;
|
package/dist/gates/baseline.js
CHANGED
|
@@ -117,7 +117,18 @@ const VOCAB_RE = /\b(?:error|fail(?:ed|ure|ing)?)\b/i;
|
|
|
117
117
|
// shapes are emitted by the process that was asked to run the oracle; they are deliberately kept in
|
|
118
118
|
// this runner-output classifier rather than applied to any judge-authored reason text. A real test
|
|
119
119
|
// failure still dominates below because one regression-shaped line makes the whole output regression.
|
|
120
|
-
|
|
120
|
+
// OBS-791: `[vitest-worker]: Timeout calling "<method>"` is the SAME birpc mechanism one layer up.
|
|
121
|
+
// Vitest's bundled birpc has a fixed 60s DEFAULT_TIMEOUT with no override (fixed upstream only in
|
|
122
|
+
// vitest 4.x), so a suite whose tests all pass can still die at teardown when the worker->host RPC
|
|
123
|
+
// window closes. `.github/workflows/release.yml` has forgiven this exact fingerprint since 2026-08-11
|
|
124
|
+
// while this classifier called it a regression — the product charged a worker for the failure the
|
|
125
|
+
// release gate was written to forgive. The METHOD NAME IS DELIBERATELY NOT PINNED: the timeout is a
|
|
126
|
+
// property of the RPC window, not of `onTaskUpdate`, and pinning one method would forgive a run and
|
|
127
|
+
// charge its sibling for the same infrastructure event. This stays a CLOSED signature — the
|
|
128
|
+
// `[vitest-worker]: ` prefix is vitest's own RPC layer and never user assertion text — and the
|
|
129
|
+
// per-line vetoes below are untouched, so one real failure anywhere still makes the whole output a
|
|
130
|
+
// regression.
|
|
131
|
+
const INFRA_RE = /\bE(?:AGAIN|MFILE|NFILE|NOMEM|NOSPC)\b|JavaScript heap out of memory|Cannot allocate memory|Resource temporarily unavailable|Token not found in system keyring|Process from config\.webServer was not able to start|\[birpc\] rpc is closed, cannot call\b|\[vitest-worker\]: Timeout calling\b/i;
|
|
121
132
|
// Capture invalidation is deliberately narrower than the gate's infrastructure vocabulary above:
|
|
122
133
|
// keyring/config-webServer startup failures remain gate concerns, while this policy is specifically
|
|
123
134
|
// for evidence that the capture ran while the machine was resource-starved.
|