tickmarkr 2.1.7 → 2.1.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/commands/doctor.d.ts +3 -0
- package/dist/cli/commands/doctor.js +23 -1
- package/dist/cli/commands/plan.d.ts +6 -2
- package/dist/cli/commands/plan.js +94 -1
- package/dist/cli/commands/status.js +50 -10
- package/dist/compile/index.js +15 -5
- package/dist/compile/native.js +30 -0
- package/dist/compile/ownership.d.ts +12 -2
- package/dist/compile/ownership.js +84 -13
- package/dist/drivers/herdr.d.ts +7 -5
- package/dist/drivers/herdr.js +142 -11
- package/dist/drivers/types.d.ts +5 -6
- package/dist/drivers/types.js +2 -48
- package/dist/gates/acceptance.d.ts +13 -0
- package/dist/gates/acceptance.js +43 -16
- package/dist/gates/run-gates.js +26 -23
- package/package.json +1 -1
- package/skills/tickmarkr-auto/SKILL.md +1 -1
- package/skills/tickmarkr-loop/SKILL.md +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +196 -24
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +109 -4
- package/skills/tickmarkr-overseer/scripts/watch-journal.sh +137 -0
|
@@ -3,6 +3,7 @@ import type { WorkerAdapter } from "../../adapters/types.js";
|
|
|
3
3
|
import { type KimiDoctorTurnResult } from "../../adapters/kimi.js";
|
|
4
4
|
import { type CatalogReadResult } from "../../adapters/catalog-remote.js";
|
|
5
5
|
import { type ShResult } from "../../run/git.js";
|
|
6
|
+
import { type VitestListResult } from "../../gates/acceptance.js";
|
|
6
7
|
/** Where a newer `table_<date>.csv` is discovered — the deployed site builds filenames by
|
|
7
8
|
* concatenation and publishes no index, so the release listing is the only enumerable surface. */
|
|
8
9
|
export declare const LIVEBENCH_RELEASES_URL = "https://api.github.com/repos/LiveBench/livebench.github.io/contents/public";
|
|
@@ -20,6 +21,8 @@ export type DoctorOpts = {
|
|
|
20
21
|
orcaStatusProbe?: (cwd: string, binary: string) => Promise<ShResult>;
|
|
21
22
|
/** Test seam for shell-path discovery; absence remains a normal doctor row, never an exception. */
|
|
22
23
|
resolveOrcaBinary?: (cwd: string) => string | undefined;
|
|
24
|
+
/** Test seam for the runner-owned JSON listing used by the acceptance-oracle report row. */
|
|
25
|
+
listTests?: (cwd: string) => Promise<VitestListResult>;
|
|
23
26
|
};
|
|
24
27
|
type OrcaCapability = {
|
|
25
28
|
verdict: "pass" | "fail";
|
|
@@ -8,7 +8,7 @@ import { allAdapters, binaryShadowWarnings, detectCandidateClis, flagDriftWarnin
|
|
|
8
8
|
import { CLAUDE_ALIAS_IDENTITY_STAMPS, claudeCode, resolveClaudeAliasIdentity } from "../../adapters/claude-code.js";
|
|
9
9
|
import { shq } from "../../adapters/types.js";
|
|
10
10
|
import { BANNER, compactTokens, dim, fail, kvRow, legend, ok, rule, statusRow, title } from "../../brand.js";
|
|
11
|
-
import { tickmarkrDir, stateDirName } from "../../graph/graph.js";
|
|
11
|
+
import { graphPath, loadGraph, tickmarkrDir, stateDirName } from "../../graph/graph.js";
|
|
12
12
|
import { catalogModelAdvisory, catalogTierRanking, declaredModelWindow, hasWindowsConfig, modelLints, suggestOverlay, ttyVisual } from "../../adapters/model-lints.js";
|
|
13
13
|
import { loadConfig, overlayPreferShapes } from "../../config/config.js";
|
|
14
14
|
import { HerdrDriver } from "../../drivers/herdr.js";
|
|
@@ -17,6 +17,7 @@ import { kimi, probeKimiDoctorTurn } from "../../adapters/kimi.js";
|
|
|
17
17
|
import { denyPreferCollisionLine, denyPreferCollisions, disallowedBy, excludedChannels, exclusionLine, preferRanks } from "../../route/preference.js";
|
|
18
18
|
import { LIVEBENCH_TABLE_DATE, readCachedCatalog, refreshCatalogCommand } from "../../adapters/catalog-remote.js";
|
|
19
19
|
import { sh } from "../../run/git.js";
|
|
20
|
+
import { auditNamedTestOracles, listVitestTests } from "../../gates/acceptance.js";
|
|
20
21
|
/** Where a newer `table_<date>.csv` is discovered — the deployed site builds filenames by
|
|
21
22
|
* concatenation and publishes no index, so the release listing is the only enumerable surface. */
|
|
22
23
|
export const LIVEBENCH_RELEASES_URL = "https://api.github.com/repos/LiveBench/livebench.github.io/contents/public";
|
|
@@ -362,6 +363,27 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
|
|
|
362
363
|
const healthy = h.installed && (a.id !== kimi.id || h.authed);
|
|
363
364
|
return alignedStatusRow(healthy ? "pass" : "fail", a.id, state);
|
|
364
365
|
});
|
|
366
|
+
if (existsSync(graphPath(cwd))) {
|
|
367
|
+
try {
|
|
368
|
+
const graph = loadGraph(cwd);
|
|
369
|
+
const items = graph.tasks.flatMap((task) => task.acceptance);
|
|
370
|
+
if (items.some((item) => typeof item === "object" && item.oracle === "test")) {
|
|
371
|
+
const listing = await (opts.listTests ?? listVitestTests)(cwd);
|
|
372
|
+
if (listing.status === "failed") {
|
|
373
|
+
rows.push(alignedStatusRow("fail", "acceptance-oracles", `runner listing failed — ${listing.error.split("\n")[0]}`));
|
|
374
|
+
}
|
|
375
|
+
else {
|
|
376
|
+
const audited = auditNamedTestOracles(items, listing.tests);
|
|
377
|
+
const resolved = audited.filter((row) => row.matches.length === 1).length;
|
|
378
|
+
const verdict = resolved === audited.length ? "pass" : "fail";
|
|
379
|
+
rows.push(alignedStatusRow(verdict, "acceptance-oracles", `${resolved}/${audited.length} resolved — each resolved oracle's shipped filter matches exactly one runner-listed test`));
|
|
380
|
+
}
|
|
381
|
+
}
|
|
382
|
+
}
|
|
383
|
+
catch (error) {
|
|
384
|
+
rows.push(alignedStatusRow("fail", "acceptance-oracles", `graph unreadable — ${error instanceof Error ? error.message : String(error)}`));
|
|
385
|
+
}
|
|
386
|
+
}
|
|
365
387
|
if (catalog.warning) {
|
|
366
388
|
rows.push(attentionRow(`model catalog cache unreadable — ${catalog.warning}; using vendored fallback (advisory — routing unchanged)`));
|
|
367
389
|
}
|
|
@@ -1,2 +1,6 @@
|
|
|
1
|
-
import type
|
|
2
|
-
|
|
1
|
+
import { type VitestListResult } from "../../gates/acceptance.js";
|
|
2
|
+
import { type WorkerAdapter } from "../../adapters/types.js";
|
|
3
|
+
export type PlanOpts = {
|
|
4
|
+
listTests?: (cwd: string) => Promise<VitestListResult>;
|
|
5
|
+
};
|
|
6
|
+
export declare function plan(argv: string[], cwd?: string, adapters?: WorkerAdapter[], harnessFrom?: string | undefined, opts?: PlanOpts): Promise<string>;
|
|
@@ -3,15 +3,20 @@ import { formatModelAuthLine, contextWindowLints, modelLints, preferEntryLints,
|
|
|
3
3
|
import { GLYPHS, dim, rule, title, warn } from "../../brand.js";
|
|
4
4
|
import { parseArgs } from "node:util";
|
|
5
5
|
import { collateralLints, sourceScopeLints } from "../../compile/collateral.js";
|
|
6
|
+
import { classifyContextPath } from "../../compile/native.js";
|
|
6
7
|
import { DEFAULT_CONFIG, overlayPreferShapes, ROUTING_MODES, TIER_RANK } from "../../config/config.js";
|
|
7
8
|
import { loadGraph } from "../../graph/graph.js";
|
|
9
|
+
import { renderAcceptanceItem } from "../../graph/schema.js";
|
|
8
10
|
import { resolveRunMode } from "../../run/daemon.js";
|
|
9
11
|
import { excludedChannels, exclusionLine } from "../../route/preference.js";
|
|
10
12
|
import { staffLedEvidence } from "../../route/profile.js";
|
|
11
13
|
import { route, RoutingError } from "../../route/router.js";
|
|
14
|
+
import { auditNamedTestOracles, listVitestTests } from "../../gates/acceptance.js";
|
|
12
15
|
import { modelId } from "../../gates/review.js";
|
|
13
16
|
import { loadRoutingProfile } from "../../run/journal.js";
|
|
14
17
|
import { harnessLine, resolveHarness } from "../harness.js";
|
|
18
|
+
import { shq } from "../../adapters/types.js";
|
|
19
|
+
import { shGit } from "../../run/git.js";
|
|
15
20
|
// T4 (v1.50): TTY-only brand pass — the title helper frames the routing table, lint/unroutable
|
|
16
21
|
// markers carry the attention glyph, section labels dim to chrome (the doctor/status system).
|
|
17
22
|
// Gated on ttyVisual(): the non-TTY surface returns untouched (byte-pinned, machine-consumable).
|
|
@@ -35,12 +40,60 @@ const fleetCanCrossVendorReview = (channels) => {
|
|
|
35
40
|
return true;
|
|
36
41
|
return false;
|
|
37
42
|
};
|
|
43
|
+
const NAMED_CRITERION_PATH = /(?:^|[\s("'`])((?:src|tests|fixtures|scripts)\/[A-Za-z0-9_@{}*?.,/+-]+\.(?:tsx|ts|jsx|json|js|mjs|cjs|md|txt))(?![A-Za-z0-9])/g;
|
|
44
|
+
async function taskInputFindings(tasks, cwd) {
|
|
45
|
+
const pathsByTask = tasks.map((task) => {
|
|
46
|
+
const paths = new Set(task.files.map((path) => path.replace(/^\.\//, "")));
|
|
47
|
+
for (const item of task.acceptance) {
|
|
48
|
+
const texts = typeof item === "string"
|
|
49
|
+
? [item]
|
|
50
|
+
: [renderAcceptanceItem(item), ...Object.values(item).filter((value) => typeof value === "string")];
|
|
51
|
+
for (const text of texts) {
|
|
52
|
+
for (const match of text.matchAll(NAMED_CRITERION_PATH))
|
|
53
|
+
paths.add(match[1]);
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
return { task, paths: [...paths] };
|
|
57
|
+
});
|
|
58
|
+
if (pathsByTask.every(({ paths }) => paths.length === 0))
|
|
59
|
+
return [];
|
|
60
|
+
const [tree, diff] = await Promise.all([
|
|
61
|
+
shGit("git ls-tree --full-tree -r --name-only -z HEAD", cwd),
|
|
62
|
+
shGit("git diff --name-only -z HEAD --", cwd),
|
|
63
|
+
]);
|
|
64
|
+
if (tree.code !== 0 || diff.code !== 0)
|
|
65
|
+
return [];
|
|
66
|
+
const tracked = new Set(tree.stdout.split("\0").filter(Boolean));
|
|
67
|
+
const changed = new Set(diff.stdout.split("\0").filter(Boolean));
|
|
68
|
+
const findings = [];
|
|
69
|
+
for (const { task, paths } of pathsByTask) {
|
|
70
|
+
for (const path of paths) {
|
|
71
|
+
const state = classifyContextPath(path, tracked, cwd);
|
|
72
|
+
if (state.kind === "untracked") {
|
|
73
|
+
findings.push({
|
|
74
|
+
taskId: task.id,
|
|
75
|
+
severity: "refuse",
|
|
76
|
+
detail: `task path ${JSON.stringify(path)} exists in the working tree but no commit holds it`,
|
|
77
|
+
});
|
|
78
|
+
}
|
|
79
|
+
else if (state.kind === "ok"
|
|
80
|
+
&& (changed.has(path) || [...changed].some((candidate) => candidate.startsWith(`${path}/`)))) {
|
|
81
|
+
findings.push({
|
|
82
|
+
taskId: task.id,
|
|
83
|
+
severity: "warn",
|
|
84
|
+
detail: `tracked path ${JSON.stringify(path)} differs from HEAD — run git restore -- ${shq(path)} to discard the divergence, or commit it before dispatch`,
|
|
85
|
+
});
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
return findings;
|
|
90
|
+
}
|
|
38
91
|
// v1.89 T4: harnessFrom is the resolver's INPUT — a caller (the byte-pinned goldens) fixes the location
|
|
39
92
|
// and keeps this machine's absolute paths out of a fixture. The default is the INVOKED entrypoint,
|
|
40
93
|
// `process.argv[1]`: the bin symlink a global install puts on PATH, which resolves to dist/cli/index.js.
|
|
41
94
|
// It is NOT `import.meta.url` — that names dist/cli/commands/plan.js, an internal module of the harness
|
|
42
95
|
// rather than the harness that was invoked, so the banner would identify the wrong file entirely.
|
|
43
|
-
export async function plan(argv, cwd = process.cwd(), adapters = allAdapters(), harnessFrom = process.argv[1]) {
|
|
96
|
+
export async function plan(argv, cwd = process.cwd(), adapters = allAdapters(), harnessFrom = process.argv[1], opts = {}) {
|
|
44
97
|
// ponytail: hardcoded 24h TTL — promote to config when an operator asks. mtime is the signal because
|
|
45
98
|
// doctor.json has no probe timestamp and a schema field would break the existing-files compat invariant.
|
|
46
99
|
const DOCTOR_STALE_MS = 24 * 60 * 60 * 1000;
|
|
@@ -51,6 +104,37 @@ export async function plan(argv, cwd = process.cwd(), adapters = allAdapters(),
|
|
|
51
104
|
throw new Error(`--mode must be one of ${ROUTING_MODES.join(" | ")} (got ${values.mode})`);
|
|
52
105
|
}
|
|
53
106
|
const g = loadGraph(cwd);
|
|
107
|
+
const oracleRefusals = new Map();
|
|
108
|
+
if (g.tasks.some((task) => task.acceptance.some((item) => typeof item === "object" && item.oracle === "test"))) {
|
|
109
|
+
const listing = await (opts.listTests ?? listVitestTests)(cwd);
|
|
110
|
+
for (const task of g.tasks) {
|
|
111
|
+
const refusals = [];
|
|
112
|
+
if (listing.status === "failed") {
|
|
113
|
+
if (task.acceptance.some((item) => typeof item === "object" && item.oracle === "test")) {
|
|
114
|
+
refusals.push(`acceptance oracle unresolved — runner listing failed: ${listing.error.split("\n")[0]}`);
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
else {
|
|
118
|
+
for (const audit of auditNamedTestOracles(task.acceptance, listing.tests)) {
|
|
119
|
+
if (audit.matches.length === 0) {
|
|
120
|
+
refusals.push(`acceptance oracle ${JSON.stringify(audit.criterion)} matches zero runner-listed test names`);
|
|
121
|
+
}
|
|
122
|
+
else if (audit.matches.length > 1) {
|
|
123
|
+
refusals.push(`acceptance oracle ${JSON.stringify(audit.criterion)} matches ${audit.matches.length} runner-listed test names`);
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
if (refusals.length)
|
|
128
|
+
oracleRefusals.set(task.id, refusals);
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
const inputFindings = await taskInputFindings(g.tasks, cwd);
|
|
132
|
+
const inputRefusals = new Map();
|
|
133
|
+
for (const finding of inputFindings) {
|
|
134
|
+
if (finding.severity !== "refuse")
|
|
135
|
+
continue;
|
|
136
|
+
inputRefusals.set(finding.taskId, [...(inputRefusals.get(finding.taskId) ?? []), finding.detail]);
|
|
137
|
+
}
|
|
54
138
|
const { cfg, mode, source } = resolveRunMode(cwd, { flag: values.mode, spec: g.mode });
|
|
55
139
|
// readDoctor cache path: staleness line only fires here (probeAll fallback is fresh by construction).
|
|
56
140
|
const cached = readDoctor(cwd);
|
|
@@ -152,6 +236,11 @@ export async function plan(argv, cwd = process.cwd(), adapters = allAdapters(),
|
|
|
152
236
|
let cost = 0;
|
|
153
237
|
const routed = [];
|
|
154
238
|
for (const t of g.tasks) {
|
|
239
|
+
const refusals = [...(oracleRefusals.get(t.id) ?? []), ...(inputRefusals.get(t.id) ?? [])];
|
|
240
|
+
if (refusals.length) {
|
|
241
|
+
lines.push(` ${t.id.padEnd(6)} ${t.shape.padEnd(10)} !! pre-dispatch refusal — ${refusals.join("; ")}`);
|
|
242
|
+
continue;
|
|
243
|
+
}
|
|
155
244
|
try {
|
|
156
245
|
const r = route(t, cfg, channels, dispatchProfile);
|
|
157
246
|
routed.push({ taskId: t.id, adapter: r.assignment.adapter, model: r.assignment.model });
|
|
@@ -192,6 +281,10 @@ export async function plan(argv, cwd = process.cwd(), adapters = allAdapters(),
|
|
|
192
281
|
lints.push(`${t.id}: unroutable — ${msg}`);
|
|
193
282
|
}
|
|
194
283
|
}
|
|
284
|
+
const inputWarnings = inputFindings.filter((finding) => finding.severity === "warn");
|
|
285
|
+
if (inputWarnings.length) {
|
|
286
|
+
lines.push("", "input warnings:", ...inputWarnings.map((finding) => ` ! ${finding.taskId}: ${finding.detail}`));
|
|
287
|
+
}
|
|
195
288
|
lines.push("", `est. cost (API channels only, rough): ~$${cost.toFixed(2)} + judge/review/consult calls`);
|
|
196
289
|
// VIS-04: summary only when a profile is active AND something deviates. Labeled by the switch — off = preview.
|
|
197
290
|
if (profile && deviations) {
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { readFileSync } from "node:fs";
|
|
1
|
+
import { existsSync, readFileSync } from "node:fs";
|
|
2
2
|
import { join } from "node:path";
|
|
3
3
|
import { GLYPHS, LIVE } from "../../brand.js";
|
|
4
4
|
import { DEFAULT_CONFIG, loadConfig } from "../../config/config.js";
|
|
@@ -7,8 +7,8 @@ import { formatOwnedName, parseOwnedName, } from "../../drivers/types.js";
|
|
|
7
7
|
import { blockedTasks, graphDefinitionHash, loadGraph, stateDirName } from "../../graph/graph.js";
|
|
8
8
|
import { GATE_NAMES } from "../../graph/schema.js";
|
|
9
9
|
import { foldActivity } from "../../run/activity.js";
|
|
10
|
-
import { Journal, engagementComparable, isQualityFailureParkKind, preservedRefsByTask, recordedTaskFailureKind, runHasEnded, upheldFeedbackByTask, } from "../../run/journal.js";
|
|
11
|
-
import { isPidLive } from "../../run/lock.js";
|
|
10
|
+
import { Journal, engagementComparable, isQualityFailureParkKind, parseRunId, preservedRefsByTask, recordedTaskFailureKind, runHasEnded, upheldFeedbackByTask, } from "../../run/journal.js";
|
|
11
|
+
import { isPidLive, runLockOwner } from "../../run/lock.js";
|
|
12
12
|
import { normalizeGateOutcome } from "../../run/outcome.js";
|
|
13
13
|
import { desiredPanes } from "../../run/reconcile.js";
|
|
14
14
|
import { normalizeStallSnapshot } from "../../run/stall.js";
|
|
@@ -850,20 +850,56 @@ const parseJournalSnapshot = (raw) => raw.split("\n").flatMap((line) => {
|
|
|
850
850
|
return [];
|
|
851
851
|
}
|
|
852
852
|
});
|
|
853
|
+
// runLockOwner is the single lock payload/liveness reader. Its current path helper also initializes
|
|
854
|
+
// .tickmarkr metadata; an engine-written lock necessarily passed through that initializer already, so
|
|
855
|
+
// status avoids invoking it in stripped reader-purity fixtures where no engine lock can exist.
|
|
856
|
+
const canReadRunLockOwner = (cwd) => existsSync(join(cwd, stateDirName(cwd), ".gitignore"));
|
|
857
|
+
const reportableLockRunId = (owner) => {
|
|
858
|
+
if (!owner?.live || owner.runId === undefined)
|
|
859
|
+
return undefined;
|
|
860
|
+
try {
|
|
861
|
+
return parseRunId(owner.runId);
|
|
862
|
+
}
|
|
863
|
+
catch {
|
|
864
|
+
// Repository-wide locks can be held for non-run work such as compile; those are not a status run.
|
|
865
|
+
return undefined;
|
|
866
|
+
}
|
|
867
|
+
};
|
|
868
|
+
const readRunJournalRaw = (cwd, runId, allowMissingJournal) => {
|
|
869
|
+
const dir = join(cwd, stateDirName(cwd), "runs", runId);
|
|
870
|
+
try {
|
|
871
|
+
return readFileSync(join(dir, "journal.jsonl"), "utf8");
|
|
872
|
+
}
|
|
873
|
+
catch (error) {
|
|
874
|
+
if (error.code === "ENOENT") {
|
|
875
|
+
if (allowMissingJournal)
|
|
876
|
+
return "";
|
|
877
|
+
throw new Error(`no journal for ${runId} at ${dir}`);
|
|
878
|
+
}
|
|
879
|
+
throw error;
|
|
880
|
+
}
|
|
881
|
+
};
|
|
853
882
|
/**
|
|
854
883
|
* ONE journal read per answer, and every claim about the run folded from THOSE bytes — the board's
|
|
855
884
|
* task rows, activity, phases, gates, liveness and tip verification, and the compact one-line form
|
|
856
885
|
* alike. A second read is what lets two snapshots be presented as one state: a line the daemon
|
|
857
886
|
* appends between them pairs a task count from one instant with a verdict from another.
|
|
858
887
|
*
|
|
859
|
-
* An explicit <runId> is a resolution, not a hint:
|
|
860
|
-
*
|
|
888
|
+
* An explicit <runId> is a resolution, not a hint: it refuses an id without a readable journal, so
|
|
889
|
+
* status fails loudly naming that id instead of rendering any other run. The implicit form first
|
|
890
|
+
* asks the repository lock accessor which run is live, then falls back to the newest journalled run;
|
|
891
|
+
* a lock-selected run may still be in the directory-before-first-journal window, which renders as an
|
|
892
|
+
* empty snapshot for that run rather than borrowing the previous run's journal.
|
|
861
893
|
*/
|
|
862
894
|
const readRunRecord = (cwd, graph, namedRunId) => {
|
|
863
|
-
const
|
|
895
|
+
const explicitRunId = namedRunId === undefined ? undefined : parseRunId(namedRunId);
|
|
896
|
+
const lockedRunId = explicitRunId === undefined && canReadRunLockOwner(cwd)
|
|
897
|
+
? reportableLockRunId(runLockOwner(cwd))
|
|
898
|
+
: undefined;
|
|
899
|
+
const runId = explicitRunId ?? lockedRunId ?? Journal.latestRunId(cwd, { withJournal: true });
|
|
864
900
|
if (!runId)
|
|
865
901
|
return undefined;
|
|
866
|
-
const raw =
|
|
902
|
+
const raw = readRunJournalRaw(cwd, runId, lockedRunId === runId);
|
|
867
903
|
const events = parseJournalSnapshot(raw);
|
|
868
904
|
// The resume comparator is the fail-closed baseline; a matching graph-rehash is the daemon's
|
|
869
905
|
// append-only audit that authorizes this status replay after stop-amend-resume.
|
|
@@ -1380,15 +1416,19 @@ export async function status(argv, cwd = process.cwd(), opts = {}) {
|
|
|
1380
1416
|
let journalCursor = 0;
|
|
1381
1417
|
const consumeDecisionEvents = () => {
|
|
1382
1418
|
// A named run is followed, never re-resolved: --watch <runId> keeps reporting that run even as
|
|
1383
|
-
// newer runs start.
|
|
1384
|
-
const
|
|
1419
|
+
// newer runs start. The no-argument form tracks the live lock first, then latest journal.
|
|
1420
|
+
const explicitRunId = namedRunId === undefined ? undefined : parseRunId(namedRunId);
|
|
1421
|
+
const lockedRunId = explicitRunId === undefined && canReadRunLockOwner(cwd)
|
|
1422
|
+
? reportableLockRunId(runLockOwner(cwd))
|
|
1423
|
+
: undefined;
|
|
1424
|
+
const runId = explicitRunId ?? lockedRunId ?? Journal.latestRunId(cwd, { withJournal: true });
|
|
1385
1425
|
if (!runId)
|
|
1386
1426
|
return [];
|
|
1387
1427
|
if (decisionRunId !== runId) {
|
|
1388
1428
|
decisionRunId = runId;
|
|
1389
1429
|
journalCursor = 0;
|
|
1390
1430
|
}
|
|
1391
|
-
const journalEvents =
|
|
1431
|
+
const journalEvents = parseJournalSnapshot(readRunJournalRaw(cwd, runId, lockedRunId === runId));
|
|
1392
1432
|
if (journalEvents.length < journalCursor)
|
|
1393
1433
|
journalCursor = 0;
|
|
1394
1434
|
const fresh = decisionEventsFromJournal(journalEvents, runId, stateDirName(cwd))
|
package/dist/compile/index.js
CHANGED
|
@@ -2,7 +2,7 @@ import { existsSync, readFileSync, statSync } from "node:fs";
|
|
|
2
2
|
import { join } from "node:path";
|
|
3
3
|
import { validateGraph } from "../graph/schema.js";
|
|
4
4
|
import { taskUnitContractErrors } from "./collateral.js";
|
|
5
|
-
import { ownershipFindings, renderOwnershipFinding } from "./ownership.js";
|
|
5
|
+
import { blocksCompile, ownershipFindings, renderOwnershipFinding } from "./ownership.js";
|
|
6
6
|
import { CompileError } from "./common.js";
|
|
7
7
|
import { compileGsd, isGsdPhaseDir } from "./gsd.js";
|
|
8
8
|
import { compileNative, TICKMARKR_NATIVE_MARKER } from "./native.js";
|
|
@@ -53,13 +53,23 @@ export function finalizePlan(plan, src, repoRoot) {
|
|
|
53
53
|
},
|
|
54
54
|
tasks: plan.tasks,
|
|
55
55
|
}), src);
|
|
56
|
-
// overseer-217
|
|
57
|
-
//
|
|
58
|
-
//
|
|
56
|
+
// overseer-217 removal condition, now paid: on this milestone's authored graph the conventional
|
|
57
|
+
// name map emitted 21 raw unowned-test findings; review found 1 real and 20 false, while intersecting
|
|
58
|
+
// with a direct import or command-entry spawn retained the real one and left 0 false positives. That
|
|
59
|
+
// is a precision measurement, NOT a recall claim. The prior halted semantic-contract class has no
|
|
60
|
+
// name/import/ownership relation (one of its three members has no matching literal anywhere), so this
|
|
61
|
+
// rule could not have caught it and does not claim to. The promoted rule already earned a true positive
|
|
62
|
+
// during authoring: assigning the plan command to oracle-preflight left two dedicated plan tests unowned.
|
|
59
63
|
if (repoRoot) {
|
|
60
|
-
|
|
64
|
+
const findings = ownershipFindings(graph.tasks, repoRoot);
|
|
65
|
+
const blocking = findings.filter(blocksCompile);
|
|
66
|
+
for (const finding of findings.filter((item) => !blocksCompile(item))) {
|
|
61
67
|
console.warn(renderOwnershipFinding(finding));
|
|
62
68
|
}
|
|
69
|
+
if (blocking.length > 0) {
|
|
70
|
+
throw new CompileError(`${src} violates cross-task test ownership (${blocking.length} error${blocking.length === 1 ? "" : "s"}):\n`
|
|
71
|
+
+ blocking.map((finding) => ` - ${renderOwnershipFinding(finding)}`).join("\n"));
|
|
72
|
+
}
|
|
63
73
|
}
|
|
64
74
|
return graph;
|
|
65
75
|
}
|
package/dist/compile/native.js
CHANGED
|
@@ -789,6 +789,36 @@ acceptance is required on every task (a nested list of observable outcomes).
|
|
|
789
789
|
hard value anywhere in the domain, the criterion asserts a universal that may be FALSE ABOUT THE
|
|
790
790
|
WORLD — bound it or say where it stops holding, rather than demanding a value that does not exist.
|
|
791
791
|
|
|
792
|
+
PICK THE CRITERION FORM FROM WHO COULD BE WRONG:
|
|
793
|
+
- When the WORKER could be wrong because it can choose the value, use "test:" and pin the exact
|
|
794
|
+
literal it could otherwise choose; an example selected by its implementer proves only itself.
|
|
795
|
+
- When the AUTHOR could be wrong by omitting a member from a list, quantify universally over the
|
|
796
|
+
authoritative closed set; a hand-written enumeration can repeat the same omission as the code.
|
|
797
|
+
- When the REVIEWER could be wrong about a prose artefact, use "judge:" to replay a recorded incident
|
|
798
|
+
against the changed prose; a keyword check proves vocabulary, not that the artefact prevents a repeat.
|
|
799
|
+
|
|
800
|
+
PRE-SCOPE BY TEXT, ENUMERATE BLOCKERS BY EXECUTION:
|
|
801
|
+
- Before assigning files[], sweep text across the repository tree for names, callers, tests and prose.
|
|
802
|
+
A text sweep produces a candidate list; only running the change enumerates the real blocker set.
|
|
803
|
+
Keep the candidates for scope, then execute the production path and full gates before declaring the
|
|
804
|
+
set closed — this milestone paid a halted run to learn that the two populations are not identical.
|
|
805
|
+
- SPIKE-THE-CONTRACT-THEN-SCOPE trigger question: COULD A TEST THIS TASK DOES NOT OWN BE ASSERTING THE
|
|
806
|
+
THING I AM CHANGING? "I'D HAVE TO GREP TO KNOW" IS YES. This applies to observable contracts:
|
|
807
|
+
execution order, event-stream order, diagnostics/output sets, CLI surface, serialised formats, or
|
|
808
|
+
timing measurements. If yes, implement the change as a throwaway spike, run the full suite, read the
|
|
809
|
+
reds, THEN scope files[].
|
|
810
|
+
- Caveat: a spike measures ONE implementation. It converts unknown collateral into
|
|
811
|
+
measured-for-one-specimen collateral; it does NOT make its reds the closed blocker set for every
|
|
812
|
+
route. A worker taking a different route can still red on unowned collateral; that remains a PLAN
|
|
813
|
+
DEFECT, NEVER A RETRY.
|
|
814
|
+
- Measured price: 518 s implement + 831 s suite = 1,348 s ≈ 22.5 min. "Far cheaper than the alternative"
|
|
815
|
+
is WITHDRAWN on the direct leg: a direct failed run died at about 20 min, so on the direct leg the
|
|
816
|
+
two are EQUAL. The spike only pays when it avoids downstream halt, sweep, re-scope, rulings, and
|
|
817
|
+
another compile+plan.
|
|
818
|
+
- Do NOT run the spike for a change that is purely additive and unwinds cheaply. The rule is bounded by
|
|
819
|
+
the expense of a late defect, not by novelty; where the defect would surface and fix cheaply, the spike
|
|
820
|
+
is pure overhead.
|
|
821
|
+
|
|
792
822
|
WHICH SIDE OF A RUN INHERITS ENVIRONMENT — AND IT DEPENDS ON THE DRIVER (OBS-542):
|
|
793
823
|
- Gate commands and "command:"/"test:" oracles INHERIT THE DAEMON'S ENVIRONMENT. They are children of
|
|
794
824
|
the daemon, so launching it as \`bash -c 'set -a; . .env.test; set +a; exec tickmarkr run'\` reaches
|
|
@@ -1,8 +1,17 @@
|
|
|
1
1
|
import type { Task } from "../graph/schema.js";
|
|
2
|
+
export type OwnershipCorroboration = {
|
|
3
|
+
kind: "direct-import";
|
|
4
|
+
source: string;
|
|
5
|
+
} | {
|
|
6
|
+
kind: "command-entry-spawn";
|
|
7
|
+
source: string;
|
|
8
|
+
entry: "src/cli/index.ts";
|
|
9
|
+
};
|
|
2
10
|
export type OwnershipFinding = {
|
|
3
11
|
code: "unowned-test";
|
|
4
12
|
test: string;
|
|
5
13
|
taskIds: string[];
|
|
14
|
+
corroboration?: OwnershipCorroboration;
|
|
6
15
|
detail: string;
|
|
7
16
|
} | {
|
|
8
17
|
code: "test-path-outside-allowlist";
|
|
@@ -18,8 +27,9 @@ export type OwnershipFinding = {
|
|
|
18
27
|
detail: string;
|
|
19
28
|
};
|
|
20
29
|
/**
|
|
21
|
-
*
|
|
22
|
-
*
|
|
30
|
+
* Cross-task ownership evidence. Findings are data: this checker never throws or changes the graph;
|
|
31
|
+
* the compile seam promotes only a corroborated unowned-test finding and reports every other shape.
|
|
23
32
|
*/
|
|
24
33
|
export declare function ownershipFindings(tasks: readonly Task[], repoRoot: string): OwnershipFinding[];
|
|
25
34
|
export declare function renderOwnershipFinding(finding: OwnershipFinding): string;
|
|
35
|
+
export declare function blocksCompile(finding: OwnershipFinding): boolean;
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { readFileSync, readdirSync } from "node:fs";
|
|
2
|
-
import { basename, extname, join } from "node:path";
|
|
2
|
+
import { basename, extname, join, posix } from "node:path";
|
|
3
3
|
import { filesGlob } from "../graph/files-glob.js";
|
|
4
4
|
import { collateralHits } from "./collateral.js";
|
|
5
5
|
const normalize = (path) => path.replace(/^\.\//, "").split("\\").join("/");
|
|
@@ -22,20 +22,79 @@ function testSources(repoRoot) {
|
|
|
22
22
|
return [];
|
|
23
23
|
}
|
|
24
24
|
}
|
|
25
|
-
|
|
26
|
-
// reaches that command through the CLI entry point. Keep this heuristic advisory until authored-graph
|
|
27
|
-
// measurements establish its false-positive rate.
|
|
28
|
-
function namedSourceTasks(test, tasks) {
|
|
25
|
+
function namedSources(test, tasks) {
|
|
29
26
|
const stem = basename(test).replace(/\.test\.ts$/, "");
|
|
30
|
-
const
|
|
27
|
+
const matches = new Map();
|
|
31
28
|
for (const task of tasks) {
|
|
32
29
|
for (const entry of task.files.map(normalize).filter((path) => path.startsWith("src/") && !/[*?{[]/.test(path))) {
|
|
33
30
|
const source = basename(entry, extname(entry));
|
|
34
|
-
if (stem === source || stem.startsWith(`${source}-`))
|
|
35
|
-
|
|
31
|
+
if (stem === source || stem.startsWith(`${source}-`)) {
|
|
32
|
+
matches.set(`${task.id}:${entry}`, { taskId: task.id, source: entry });
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
return [...matches.values()];
|
|
37
|
+
}
|
|
38
|
+
const moduleKey = (path) => normalize(path).replace(/\.(?:[cm]?[jt]sx?)$/, "");
|
|
39
|
+
function directImportSpecifiers(text) {
|
|
40
|
+
// Comments cannot create an edge. Keep strings intact because they are the import target.
|
|
41
|
+
const source = text.replace(/\/\*[\s\S]*?\*\//g, "").replace(/^\s*\/\/.*$/gm, "");
|
|
42
|
+
const specifiers = new Set();
|
|
43
|
+
for (const match of source.matchAll(/\bimport\s+(?:type\s+)?(?:[\w$*{},\s]+?\s+from\s+)?["']([^"']+)["']/g)) {
|
|
44
|
+
specifiers.add(match[1]);
|
|
45
|
+
}
|
|
46
|
+
for (const match of source.matchAll(/\bimport\s*\(\s*["']([^"']+)["']\s*\)/g)) {
|
|
47
|
+
specifiers.add(match[1]);
|
|
48
|
+
}
|
|
49
|
+
return [...specifiers];
|
|
50
|
+
}
|
|
51
|
+
function directlyImports(test, source) {
|
|
52
|
+
// DIRECT is load-bearing: do not walk through imported helpers. src/run/journal.ts alone has 84
|
|
53
|
+
// test importers in the measured tree, so transitive closure would recreate the raw alarm flood.
|
|
54
|
+
const target = moduleKey(source);
|
|
55
|
+
return directImportSpecifiers(test.text).some((specifier) => {
|
|
56
|
+
const imported = specifier.startsWith(".")
|
|
57
|
+
? posix.normalize(posix.join(posix.dirname(test.path), specifier))
|
|
58
|
+
: specifier.startsWith("src/") ? specifier : "";
|
|
59
|
+
return imported !== "" && moduleKey(imported) === target;
|
|
60
|
+
});
|
|
61
|
+
}
|
|
62
|
+
function invokesChildProcessSpawn(text) {
|
|
63
|
+
const source = text.replace(/\/\*[\s\S]*?\*\//g, "").replace(/^\s*\/\/.*$/gm, "");
|
|
64
|
+
const bindings = new Set();
|
|
65
|
+
for (const match of source.matchAll(/\bimport\s*{([^}]*)}\s*from\s*["'](?:node:)?child_process["']/g)) {
|
|
66
|
+
for (const member of match[1].split(",")) {
|
|
67
|
+
const binding = member.trim().match(/^spawn(?:Sync)?(?:\s+as\s+([A-Za-z_$][\w$]*))?$/);
|
|
68
|
+
if (binding)
|
|
69
|
+
bindings.add(binding[1] ?? member.trim());
|
|
36
70
|
}
|
|
37
71
|
}
|
|
38
|
-
|
|
72
|
+
for (const match of source.matchAll(/\bimport\s*\*\s*as\s*([A-Za-z_$][\w$]*)\s*from\s*["'](?:node:)?child_process["']/g)) {
|
|
73
|
+
if (new RegExp(`\\b${match[1]}\\.spawn(?:Sync)?\\s*\\(`).test(source))
|
|
74
|
+
return true;
|
|
75
|
+
}
|
|
76
|
+
return [...bindings].some((binding) => new RegExp(`\\b${binding}\\s*\\(`).test(source));
|
|
77
|
+
}
|
|
78
|
+
function mentionsCommandEntry(text) {
|
|
79
|
+
return /(?:^|\/)src\/cli\/index\.(?:ts|js)\b/.test(text)
|
|
80
|
+
|| /["'`]src["'`]\s*,\s*["'`]cli["'`]\s*,\s*["'`]index\.(?:ts|js)["'`]/.test(text);
|
|
81
|
+
}
|
|
82
|
+
function corroboration(test, matches) {
|
|
83
|
+
// A .test.ts-shaped collateral fixture is not by itself a dedicated test. Requiring a runner leaf
|
|
84
|
+
// keeps import-only scan fixtures advisory while every executable subject in the measured union stays.
|
|
85
|
+
const executable = test.text.replace(/\/\*[\s\S]*?\*\//g, "").replace(/^\s*\/\/.*$/gm, "");
|
|
86
|
+
if (!/\b(?:test|it)(?:\.(?:concurrent|each|fails|only|skip|todo))*\s*\(/.test(executable))
|
|
87
|
+
return undefined;
|
|
88
|
+
for (const match of matches) {
|
|
89
|
+
if (directlyImports(test, match.source))
|
|
90
|
+
return { kind: "direct-import", source: match.source };
|
|
91
|
+
}
|
|
92
|
+
if (invokesChildProcessSpawn(test.text) && mentionsCommandEntry(test.text)) {
|
|
93
|
+
const command = matches.find(({ source }) => /^src\/cli\/commands\/[^/]+\.(?:[cm]?[jt]sx?)$/.test(source));
|
|
94
|
+
if (command)
|
|
95
|
+
return { kind: "command-entry-spawn", source: command.source, entry: "src/cli/index.ts" };
|
|
96
|
+
}
|
|
97
|
+
return undefined;
|
|
39
98
|
}
|
|
40
99
|
function repositoryPaths(text) {
|
|
41
100
|
const paths = new Set();
|
|
@@ -62,11 +121,12 @@ function dependencyOrdered(a, b, byId) {
|
|
|
62
121
|
return reaches(a, b.id) || reaches(b, a.id);
|
|
63
122
|
}
|
|
64
123
|
/**
|
|
65
|
-
*
|
|
66
|
-
*
|
|
124
|
+
* Cross-task ownership evidence. Findings are data: this checker never throws or changes the graph;
|
|
125
|
+
* the compile seam promotes only a corroborated unowned-test finding and reports every other shape.
|
|
67
126
|
*/
|
|
68
127
|
export function ownershipFindings(tasks, repoRoot) {
|
|
69
128
|
const sources = testSources(repoRoot);
|
|
129
|
+
const sourceByPath = new Map(sources.map((source) => [source.path, source]));
|
|
70
130
|
const byId = new Map(tasks.map((task) => [task.id, task]));
|
|
71
131
|
const indexed = tasks.map((task) => {
|
|
72
132
|
const files = task.files.map(normalize);
|
|
@@ -81,6 +141,8 @@ export function ownershipFindings(tasks, repoRoot) {
|
|
|
81
141
|
const predictedBy = new Map();
|
|
82
142
|
for (const [taskId, hits] of collateralHits(tasks, repoRoot)) {
|
|
83
143
|
for (const hit of hits) {
|
|
144
|
+
if (!sourceByPath.has(hit))
|
|
145
|
+
continue;
|
|
84
146
|
const ids = predictedBy.get(hit) ?? new Set();
|
|
85
147
|
ids.add(taskId);
|
|
86
148
|
predictedBy.set(hit, ids);
|
|
@@ -88,7 +150,7 @@ export function ownershipFindings(tasks, repoRoot) {
|
|
|
88
150
|
}
|
|
89
151
|
for (const source of sources) {
|
|
90
152
|
const ids = predictedBy.get(source.path) ?? new Set();
|
|
91
|
-
for (const taskId of
|
|
153
|
+
for (const { taskId } of namedSources(source.path, tasks))
|
|
92
154
|
ids.add(taskId);
|
|
93
155
|
if (ids.size > 0)
|
|
94
156
|
predictedBy.set(source.path, ids);
|
|
@@ -97,11 +159,17 @@ export function ownershipFindings(tasks, repoRoot) {
|
|
|
97
159
|
for (const [test, taskIds] of predictedBy) {
|
|
98
160
|
if (owners(test).length === 0) {
|
|
99
161
|
const ids = [...taskIds].sort();
|
|
162
|
+
const source = sourceByPath.get(test);
|
|
163
|
+
const evidence = corroboration(source, namedSources(test, tasks));
|
|
100
164
|
findings.push({
|
|
101
165
|
code: "unowned-test",
|
|
102
166
|
test,
|
|
103
167
|
taskIds: ids,
|
|
104
|
-
|
|
168
|
+
...(evidence ? { corroboration: evidence } : {}),
|
|
169
|
+
detail: `${test} is a dedicated test of source owned by ${ids.join(", ")} but no task owns the test`
|
|
170
|
+
+ (evidence?.kind === "direct-import" ? `; it imports ${evidence.source} directly`
|
|
171
|
+
: evidence?.kind === "command-entry-spawn"
|
|
172
|
+
? `; it spawns ${evidence.entry} to exercise ${evidence.source}` : ""),
|
|
105
173
|
});
|
|
106
174
|
}
|
|
107
175
|
}
|
|
@@ -140,3 +208,6 @@ export function ownershipFindings(tasks, repoRoot) {
|
|
|
140
208
|
export function renderOwnershipFinding(finding) {
|
|
141
209
|
return `tickmarkr: ownership-lint[${finding.code}]: ${finding.detail}`;
|
|
142
210
|
}
|
|
211
|
+
export function blocksCompile(finding) {
|
|
212
|
+
return finding.code === "unowned-test" && finding.corroboration !== undefined;
|
|
213
|
+
}
|
package/dist/drivers/herdr.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { type JournalEvent } from "../run/journal.js";
|
|
2
|
-
import { type ExecutorDriver, type NotifyOpts, type Slot, type SlotOpts } from "./types.js";
|
|
2
|
+
import { type ExecutorDriver, type NotifyOpts, type PanesToCloseOpts, type Slot, type SlotOpts } from "./types.js";
|
|
3
3
|
export declare const TRAILER_SAFE_FLOOR_COLS = 108;
|
|
4
4
|
export declare const TRAILER_WIDTH_MARGIN = 2;
|
|
5
5
|
export declare const DELIVERY_ATTEMPTS = 3;
|
|
@@ -70,6 +70,11 @@ export declare class HerdrDriver implements ExecutorDriver {
|
|
|
70
70
|
private watches;
|
|
71
71
|
constructor(bin?: string, workersPerTab?: number, time?: HerdrTimeSource, journal?: DriverJournal | undefined);
|
|
72
72
|
private appendDispatchRetry;
|
|
73
|
+
private openRunJournal;
|
|
74
|
+
private liveSupervisionSeats;
|
|
75
|
+
private journalReconcile;
|
|
76
|
+
private paneReconcileData;
|
|
77
|
+
private parsePaneList;
|
|
73
78
|
/** v1.99 T2: bind this driver's own journal writes to the run's live narration sink. */
|
|
74
79
|
narrateWith(narrate: (event: JournalEvent) => void): void;
|
|
75
80
|
private serial;
|
|
@@ -123,9 +128,6 @@ export declare class HerdrDriver implements ExecutorDriver {
|
|
|
123
128
|
private watchSlot;
|
|
124
129
|
private discardSplit;
|
|
125
130
|
narrator(cwd: string, command: string, runId?: string): Promise<Slot>;
|
|
126
|
-
reconcile(desired: Set<string>, runId: string, opts?:
|
|
127
|
-
spareLiveLlm?: boolean;
|
|
128
|
-
endedRunIds?: Set<string>;
|
|
129
|
-
}): Promise<void>;
|
|
131
|
+
reconcile(desired: Set<string>, runId: string, opts?: PanesToCloseOpts): Promise<void>;
|
|
130
132
|
worktree(repo: string, branch: string, baseRef: string): Promise<string>;
|
|
131
133
|
}
|