tickmarkr 2.4.0 → 2.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/catalog-remote.d.ts +11 -0
- package/dist/adapters/catalog-remote.js +55 -13
- package/dist/adapters/codex.js +6 -7
- package/dist/adapters/qwen.js +3 -3
- package/dist/adapters/registry.js +13 -1
- package/dist/cli/commands/compile.d.ts +3 -0
- package/dist/cli/commands/compile.js +77 -46
- package/dist/cli/commands/doctor.js +14 -9
- package/dist/cli/commands/fleet.js +11 -5
- package/dist/cli/commands/init.js +12 -13
- package/dist/cli/commands/plan.js +47 -7
- package/dist/cli/commands/run.js +20 -1
- package/dist/cli/commands/status.js +20 -21
- package/dist/cli/commands/version.js +2 -2
- package/dist/compile/collateral.d.ts +14 -5
- package/dist/compile/collateral.js +32 -26
- package/dist/compile/index.js +17 -6
- package/dist/compile/native.js +10 -2
- package/dist/compile/ownership.js +15 -9
- package/dist/drivers/herdr.d.ts +1 -0
- package/dist/drivers/herdr.js +32 -3
- package/dist/drivers/orca.d.ts +5 -1
- package/dist/drivers/orca.js +51 -2
- package/dist/drivers/types.d.ts +1 -1
- package/dist/gates/baseline.d.ts +26 -2
- package/dist/gates/baseline.js +90 -7
- package/dist/gates/review.d.ts +2 -2
- package/dist/gates/review.js +2 -18
- package/dist/graph/graph.d.ts +20 -0
- package/dist/graph/graph.js +66 -1
- package/dist/route/preference.d.ts +3 -1
- package/dist/route/preference.js +4 -4
- package/dist/run/consult.js +1 -0
- package/dist/run/daemon.d.ts +5 -0
- package/dist/run/daemon.js +151 -22
- package/dist/run/git.d.ts +2 -0
- package/dist/run/git.js +18 -4
- package/dist/run/journal.d.ts +1 -1
- package/dist/run/journal.js +1 -1
- package/dist/run/lock.d.ts +6 -0
- package/dist/run/lock.js +41 -1
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +39 -4
- package/skills/tickmarkr-overseer/scripts/grade-ci.sh +79 -0
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +90 -4
package/dist/drivers/orca.d.ts
CHANGED
|
@@ -15,6 +15,8 @@ export declare const NOT_WRITABLE_CODE = "terminal_not_writable";
|
|
|
15
15
|
export declare const RUNNING_STATUS = "running";
|
|
16
16
|
export declare const STATUS_GOVERNED_METHODS: readonly ["read", "waitOutput", "status", "waitAgentStatus"];
|
|
17
17
|
export declare const WORKTREE_ADOPTION_TIMEOUT_MS = 60000;
|
|
18
|
+
/** A missing slot gets the same bounded chance to appear as a reaped shell gets to settle. */
|
|
19
|
+
export declare const PENDING_PROJECT_GRACE_MS = 2000;
|
|
18
20
|
export interface OrcaExec {
|
|
19
21
|
(args: string[], cwd: string, timeoutMs?: number): Promise<ShResult>;
|
|
20
22
|
}
|
|
@@ -136,7 +138,7 @@ export declare class OrcaDriver implements ExecutorDriver {
|
|
|
136
138
|
describe(slot: Slot): {
|
|
137
139
|
surface?: string;
|
|
138
140
|
hostPlatform?: string;
|
|
139
|
-
};
|
|
141
|
+
} | undefined;
|
|
140
142
|
run(slot: Slot, cmd: string): Promise<void>;
|
|
141
143
|
private sendText;
|
|
142
144
|
private sendReceipt;
|
|
@@ -221,6 +223,8 @@ export declare class OrcaDriver implements ExecutorDriver {
|
|
|
221
223
|
* that still reports truncated at totalCount rows is a listing this sweep declines to judge on.
|
|
222
224
|
*/
|
|
223
225
|
private listAll;
|
|
226
|
+
private openRunJournal;
|
|
227
|
+
private dropExpiredProjects;
|
|
224
228
|
/**
|
|
225
229
|
* Sweep tickmarkr-owned terminals down to `desired`. Ownership is decided ONLY by parseOwnedName
|
|
226
230
|
* over the owned TAB title, through the same panesToClose fold herdr uses (drivers/types.ts): an
|
package/dist/drivers/orca.js
CHANGED
|
@@ -58,6 +58,8 @@ export const WORKTREE_ADOPTION_TIMEOUT_MS = 60_000;
|
|
|
58
58
|
const WORKTREE_ADOPTION_POLL_MS = 1_000;
|
|
59
59
|
const WORKTREE_ADOPTION_JOURNAL_MS = 2_000;
|
|
60
60
|
const NUDGE_ECHO_TIMEOUT_MS = 2_000;
|
|
61
|
+
/** A missing slot gets the same bounded chance to appear as a reaped shell gets to settle. */
|
|
62
|
+
export const PENDING_PROJECT_GRACE_MS = 2_000;
|
|
61
63
|
const SYSTEM_TIME = {
|
|
62
64
|
now: () => Date.now(),
|
|
63
65
|
sleep: (ms) => new Promise((resolve) => setTimeout(resolve, ms)),
|
|
@@ -366,7 +368,7 @@ export class OrcaDriver {
|
|
|
366
368
|
this.taskWorktrees.set(owned.taskId, worktree);
|
|
367
369
|
const pending = this.pendingProjects.get(owned.taskId);
|
|
368
370
|
if (pending) {
|
|
369
|
-
await this.setWorkspaceStatus(worktree, pending);
|
|
371
|
+
await this.setWorkspaceStatus(worktree, pending.state);
|
|
370
372
|
this.pendingProjects.delete(owned.taskId);
|
|
371
373
|
}
|
|
372
374
|
}
|
|
@@ -1007,7 +1009,7 @@ export class OrcaDriver {
|
|
|
1007
1009
|
if (!worktree) {
|
|
1008
1010
|
// The daemon projects in-progress immediately before it creates the task checkout/slot.
|
|
1009
1011
|
// Hold only that latest state; slot() applies it once the task's own path is known.
|
|
1010
|
-
this.pendingProjects.set(taskId, state);
|
|
1012
|
+
this.pendingProjects.set(taskId, { state, since: this.time.now() });
|
|
1011
1013
|
return;
|
|
1012
1014
|
}
|
|
1013
1015
|
await this.setWorkspaceStatus(worktree, state);
|
|
@@ -1076,6 +1078,52 @@ export class OrcaDriver {
|
|
|
1076
1078
|
throw new OrcaError("list", "terminal list is still truncated at totalCount rows", whole.raw);
|
|
1077
1079
|
return whole;
|
|
1078
1080
|
}
|
|
1081
|
+
openRunJournal(runId) {
|
|
1082
|
+
const roots = new Set(this.journalRoots.values());
|
|
1083
|
+
roots.add(process.cwd());
|
|
1084
|
+
for (const repoRoot of roots) {
|
|
1085
|
+
try {
|
|
1086
|
+
return Journal.open(repoRoot, runId, this.narrate);
|
|
1087
|
+
}
|
|
1088
|
+
catch {
|
|
1089
|
+
/* try the next daemon-bound root */
|
|
1090
|
+
}
|
|
1091
|
+
}
|
|
1092
|
+
return undefined;
|
|
1093
|
+
}
|
|
1094
|
+
// A projection exists only to bridge project() to the worker slot that follows it. After the
|
|
1095
|
+
// grace, desired remains the dispatch oracle: a declared worker is still being placed and must
|
|
1096
|
+
// keep its projection. Make genuine absence durable by appending first and deleting second; with
|
|
1097
|
+
// no writable run journal the entry remains eligible for a later reconcile.
|
|
1098
|
+
dropExpiredProjects(desired, runId) {
|
|
1099
|
+
const now = this.time.now();
|
|
1100
|
+
const desiredTasks = new Set();
|
|
1101
|
+
for (const name of desired) {
|
|
1102
|
+
const owned = parseOwnedName(name);
|
|
1103
|
+
if (owned?.role === "worker")
|
|
1104
|
+
desiredTasks.add(owned.taskId);
|
|
1105
|
+
}
|
|
1106
|
+
const expired = [...this.pendingProjects].filter(([, pending]) => now - pending.since > PENDING_PROJECT_GRACE_MS).filter(([taskId]) => !desiredTasks.has(taskId));
|
|
1107
|
+
if (expired.length === 0)
|
|
1108
|
+
return;
|
|
1109
|
+
const journal = this.openRunJournal(runId);
|
|
1110
|
+
if (!journal)
|
|
1111
|
+
return;
|
|
1112
|
+
for (const [taskId, pending] of expired) {
|
|
1113
|
+
try {
|
|
1114
|
+
const pendingMs = Math.max(0, now - pending.since);
|
|
1115
|
+
journal.append("project-unplaced", taskId, {
|
|
1116
|
+
state: pending.state,
|
|
1117
|
+
pendingMs,
|
|
1118
|
+
graceMs: PENDING_PROJECT_GRACE_MS,
|
|
1119
|
+
});
|
|
1120
|
+
this.pendingProjects.delete(taskId);
|
|
1121
|
+
}
|
|
1122
|
+
catch {
|
|
1123
|
+
/* reconcile is cosmetic; preserve the projection until a later journalled drop */
|
|
1124
|
+
}
|
|
1125
|
+
}
|
|
1126
|
+
}
|
|
1079
1127
|
/**
|
|
1080
1128
|
* Sweep tickmarkr-owned terminals down to `desired`. Ownership is decided ONLY by parseOwnedName
|
|
1081
1129
|
* over the owned TAB title, through the same panesToClose fold herdr uses (drivers/types.ts): an
|
|
@@ -1091,6 +1139,7 @@ export class OrcaDriver {
|
|
|
1091
1139
|
* Cosmetic by contract: every failure is swallowed, per candidate and overall.
|
|
1092
1140
|
*/
|
|
1093
1141
|
async reconcile(desired, runId, opts) {
|
|
1142
|
+
this.dropExpiredProjects(desired, runId);
|
|
1094
1143
|
try {
|
|
1095
1144
|
// Every call below is handle-addressed or explicitly selectored, so the CLI's own cwd selects
|
|
1096
1145
|
// nothing — it only has to exist, which the checkouts being swept no longer need to.
|
package/dist/drivers/types.d.ts
CHANGED
|
@@ -76,7 +76,7 @@ export interface ExecutorDriver {
|
|
|
76
76
|
/** The exact terminal read surface used for liveness evidence. */
|
|
77
77
|
readSource?: string;
|
|
78
78
|
/** Placement facts returned by drivers whose terminal host exposes them. */
|
|
79
|
-
describe?(slot: Slot): SlotPlacement | Promise<SlotPlacement
|
|
79
|
+
describe?(slot: Slot): SlotPlacement | Promise<SlotPlacement> | undefined;
|
|
80
80
|
slot(cwd: string, name: string, opts?: SlotOpts): Promise<Slot>;
|
|
81
81
|
run(slot: Slot, cmd: string): Promise<void>;
|
|
82
82
|
waitOutput(slot: Slot, pattern: string, timeoutMs: number, opts?: {
|
package/dist/gates/baseline.d.ts
CHANGED
|
@@ -24,7 +24,9 @@ export interface BaselineCommand {
|
|
|
24
24
|
fingerprints: string[];
|
|
25
25
|
missingCommand?: boolean;
|
|
26
26
|
/** Why a capture returned no verdict. */
|
|
27
|
-
invalidCause?: "ceiling-kill" | "resource-exhaustion";
|
|
27
|
+
invalidCause?: "ceiling-kill" | "resource-exhaustion" | "infra";
|
|
28
|
+
/** The runner's summary was green and only its teardown fingerprint followed. */
|
|
29
|
+
teardownFingerprint?: true;
|
|
28
30
|
/** What this command actually took at capture, on a pristine tree. Absent in pre-v1.90 baselines. */
|
|
29
31
|
durationMs?: number;
|
|
30
32
|
/** Sum of the per-file durations named by the runner; null when its output names none. */
|
|
@@ -154,4 +156,26 @@ export declare function detectVacuousOracles(cwd: string, tasks: ReadonlyArray<{
|
|
|
154
156
|
id: string;
|
|
155
157
|
acceptance: AcceptanceItem[];
|
|
156
158
|
}>): Promise<VacuousOracleWarning[]>;
|
|
157
|
-
export declare function compareToBaseline(cwd: string, commands: Record<string, string>, baseline: Baseline, enabled: string[]
|
|
159
|
+
export declare function compareToBaseline(cwd: string, commands: Record<string, string>, baseline: Baseline, enabled: string[], opts?: {
|
|
160
|
+
rerunOf?: HostStarvedRerun;
|
|
161
|
+
}): Promise<GateResult[]>;
|
|
162
|
+
export interface HostStarvedRerun {
|
|
163
|
+
durationMs: number;
|
|
164
|
+
referenceMs: number;
|
|
165
|
+
waitedMs: number;
|
|
166
|
+
}
|
|
167
|
+
export type RunnerVerdict = FailureClassification | "green-teardown" | undefined;
|
|
168
|
+
/** One interpretation of runner bytes, shared by baseline capture and every gate consumer. */
|
|
169
|
+
export declare function classifyRunnerOutput(raw: string, code: number): RunnerVerdict;
|
|
170
|
+
export declare const HOST_STARVED_DURATION_FACTOR = 2;
|
|
171
|
+
/** Every fresh failure head is timeout-class and the suite took twice its own baseline measurement. */
|
|
172
|
+
export declare function hostStarved(fresh: string, durationMs: number, referenceMs: number | undefined): boolean;
|
|
173
|
+
interface CalmWindow {
|
|
174
|
+
pollMs: number;
|
|
175
|
+
maxWaitMs: number;
|
|
176
|
+
loadProvider: () => number;
|
|
177
|
+
calmLoad: () => number;
|
|
178
|
+
}
|
|
179
|
+
export declare function setCalmWindowForTests(over: Partial<CalmWindow>): void;
|
|
180
|
+
export declare function resetCalmWindowForTests(): void;
|
|
181
|
+
export {};
|
package/dist/gates/baseline.js
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { existsSync, readFileSync } from "node:fs";
|
|
2
|
+
import { availableParallelism, loadavg } from "node:os";
|
|
2
3
|
import { join } from "node:path";
|
|
3
4
|
import { DEFAULT_SHELL_TIMEOUT_MS, describeCapacity, sameCapacity, sh } from "../run/git.js";
|
|
4
5
|
// incident #2 (run-20260709-104447): a vitest ✓ PASS line with "error" in the test NAME, wrapped in ANSI
|
|
@@ -501,11 +502,20 @@ export async function captureBaseline(cwd, commands) {
|
|
|
501
502
|
base.commands[name] = invalidCaptureEntry(durationMs, "resource-exhaustion", invalidatingLines);
|
|
502
503
|
continue;
|
|
503
504
|
}
|
|
505
|
+
// OBS-885/887: capture and gate ask the same classifier. A green summary followed only by the
|
|
506
|
+
// teardown fingerprint is a pass; infrastructure without a summary is no verdict to forgive.
|
|
507
|
+
const runnerVerdict = classifyRunnerOutput(raw, r.code);
|
|
508
|
+
if (runnerVerdict === "infra") {
|
|
509
|
+
console.error(`tickmarkr: baseline capture for "${name}" carries runner-infrastructure evidence and no green summary — it recorded NO verdict; nothing is forgiven for this command`);
|
|
510
|
+
base.commands[name] = invalidCaptureEntry(durationMs, "infra");
|
|
511
|
+
continue;
|
|
512
|
+
}
|
|
504
513
|
base.commands[name] = {
|
|
505
|
-
exitCode: r.code,
|
|
514
|
+
exitCode: runnerVerdict === "green-teardown" ? 0 : r.code,
|
|
515
|
+
...(runnerVerdict === "green-teardown" ? { teardownFingerprint: true } : {}),
|
|
506
516
|
// a command that exits 0 has no failures to fingerprint — recording any would be a lie the
|
|
507
517
|
// compare step then has to forgive
|
|
508
|
-
fingerprints: r.code === 0 ? [] : fingerprint(raw),
|
|
518
|
+
fingerprints: r.code === 0 || runnerVerdict === "green-teardown" ? [] : fingerprint(raw),
|
|
509
519
|
missingCommand: missingConfiguredCommand(cmd, r),
|
|
510
520
|
durationMs,
|
|
511
521
|
...fileTiming(raw, durationMs),
|
|
@@ -570,8 +580,9 @@ function headlineDetails(raw, fresh) {
|
|
|
570
580
|
meta: { failingTests: headlines.filter(namesFailureEitherForm) },
|
|
571
581
|
};
|
|
572
582
|
}
|
|
573
|
-
export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
583
|
+
export async function compareToBaseline(cwd, commands, baseline, enabled, opts = {}) {
|
|
574
584
|
const results = [];
|
|
585
|
+
const rerunOf = opts.rerunOf;
|
|
575
586
|
for (const name of enabled) {
|
|
576
587
|
const cmd = commands[name];
|
|
577
588
|
if (!cmd) {
|
|
@@ -595,7 +606,13 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
|
595
606
|
// states no capacity — a row that never divided the machine must not claim that it did.
|
|
596
607
|
const record = (g) => {
|
|
597
608
|
const withReap = r.reapedGroup ? { ...g, meta: { ...g.meta, reapedGroup: true } } : g;
|
|
598
|
-
|
|
609
|
+
const withReapError = r.reapError ? { ...withReap, meta: { ...withReap.meta, reapError: r.reapError } } : withReap;
|
|
610
|
+
const withRerun = rerunOf ? {
|
|
611
|
+
...withReapError,
|
|
612
|
+
details: `host-starved rerun after waiting ${rerunOf.waitedMs}ms for a calm load window: ${withReapError.details}`,
|
|
613
|
+
meta: { ...withReapError.meta, hostStarvedRerun: rerunOf },
|
|
614
|
+
} : withReapError;
|
|
615
|
+
results.push(r.capacity ? { ...withRerun, capacity: r.capacity } : withRerun);
|
|
599
616
|
};
|
|
600
617
|
// …and whether the entry that would forgive this command was measured in the same world. A
|
|
601
618
|
// baseline captured under a different fork cap forgives nothing: its fingerprints describe a
|
|
@@ -629,6 +646,12 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
|
629
646
|
continue;
|
|
630
647
|
}
|
|
631
648
|
const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
|
|
649
|
+
// OBS-885/887: the same classifier the capture applied names a completed green suite on both sides.
|
|
650
|
+
const runnerVerdict = classifyRunnerOutput(raw, r.code);
|
|
651
|
+
if (runnerVerdict === "green-teardown") {
|
|
652
|
+
record({ gate: name, pass: true, details: `exit ${r.code} after a green suite summary; only the runner's teardown fingerprint followed it`, meta: { teardownFingerprint: true } });
|
|
653
|
+
continue;
|
|
654
|
+
}
|
|
632
655
|
// OBS-278: only a failure SHAPE is a verdict — everything fingerprint() keeps is one, except the
|
|
633
656
|
// unrecognized-output marker, which is evidence for the operator and never grounds to reject.
|
|
634
657
|
// ponytail: ceiling — a runner whose failure output holds no shape above and whose baseline is
|
|
@@ -640,10 +663,20 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
|
640
663
|
// T9: classify the FRESH diff before charging it. The complete runner output can legitimately
|
|
641
664
|
// contain a baseline-recorded assertion beside a newly introduced infrastructure death; letting
|
|
642
665
|
// that known assertion outvote the fresh birpc line turns machine failure into a worker defect.
|
|
643
|
-
// `
|
|
666
|
+
// `classifyRunnerOutput` remains the single discriminator. When there is no fresh fingerprint,
|
|
644
667
|
// retain the whole-output read so a repeated infra abort can never be baseline-forgiven as green.
|
|
645
|
-
const
|
|
646
|
-
const
|
|
668
|
+
const freshVerdict = failing.length ? classifyRunnerOutput(failing.join("\n"), r.code) : undefined;
|
|
669
|
+
const freshClassification = freshVerdict === "infra" || freshVerdict === "regression" ? freshVerdict : undefined;
|
|
670
|
+
const classification = freshClassification ?? (!failing.length && (runnerVerdict === "infra" || runnerVerdict === "regression") ? runnerVerdict : undefined);
|
|
671
|
+
// OBS-896: every fresh failure must be timeout-class, and the suite must take at least twice its
|
|
672
|
+
// own baseline measurement. The first read buys one calm rerun here, never a worker repair.
|
|
673
|
+
if (name === "test" && classification !== "infra" && failing.length && !rerunOf
|
|
674
|
+
&& hostStarved(failing.join("\n"), r.durationMs ?? 0, entry?.durationMs)) {
|
|
675
|
+
const waitedMs = await waitForCalmWindow();
|
|
676
|
+
const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
|
|
677
|
+
results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { rerunOf: provenance }));
|
|
678
|
+
continue;
|
|
679
|
+
}
|
|
647
680
|
if (classification === "infra") {
|
|
648
681
|
const evidence = failing.length
|
|
649
682
|
? failing.slice(0, 10).join("\n")
|
|
@@ -708,3 +741,53 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
|
|
|
708
741
|
}
|
|
709
742
|
return results;
|
|
710
743
|
}
|
|
744
|
+
const SUMMARY_LINE_RE = /^[^\S\n]*Test Files[^\S\n]+(.+)$/m;
|
|
745
|
+
const TEARDOWN_RE = /\[vitest-worker\]: Timeout calling\b|\[birpc\] rpc is closed, cannot call\b/;
|
|
746
|
+
const UNHANDLED_HEADER_RE = /^\s*[^\w]*(?:Unhandled Errors|Uncaught Exception)\b/;
|
|
747
|
+
/** One interpretation of runner bytes, shared by baseline capture and every gate consumer. */
|
|
748
|
+
export function classifyRunnerOutput(raw, code) {
|
|
749
|
+
if (code === 0)
|
|
750
|
+
return undefined;
|
|
751
|
+
const lines = withoutVitestEchoBlocks(raw).map((l) => l.replace(ANSI_RE, ""));
|
|
752
|
+
const text = lines.join("\n");
|
|
753
|
+
const summary = SUMMARY_LINE_RE.exec(text);
|
|
754
|
+
if (summary) {
|
|
755
|
+
const failed = [...summary[1].matchAll(/\b(\d+)\s+failed\b/g)].map((match) => Number(match[1]));
|
|
756
|
+
const summaryGreen = failed.every((count) => count === 0);
|
|
757
|
+
const summaryLine = text.slice(0, summary.index).split("\n").length - 1;
|
|
758
|
+
const teardownLine = lines.findIndex((line, index) => index > summaryLine && TEARDOWN_RE.test(line));
|
|
759
|
+
const otherFailure = lines.some((line, index) => {
|
|
760
|
+
if (index === summaryLine || TEARDOWN_RE.test(line))
|
|
761
|
+
return false;
|
|
762
|
+
if (PASS_LINE_RE.test(line) || OPERATOR_LINE_RE.test(line))
|
|
763
|
+
return false;
|
|
764
|
+
if (index > summaryLine && UNHANDLED_HEADER_RE.test(line))
|
|
765
|
+
return false;
|
|
766
|
+
return namesRegression(line);
|
|
767
|
+
});
|
|
768
|
+
if (summaryGreen && teardownLine > summaryLine && !otherFailure)
|
|
769
|
+
return "green-teardown";
|
|
770
|
+
}
|
|
771
|
+
return classifyFailureOutput(text);
|
|
772
|
+
}
|
|
773
|
+
const TIMEOUT_CLASS_RE = /\b(?:Test|Hook) timed out in (?:\d+|#) ?ms\b|\bexceeded (?:\d+|#) ?(?:ms|s|seconds)\b|\btimed out after (?:\d+|#)/i;
|
|
774
|
+
const ERROR_HEAD_RE = /^\s*(?:[A-Za-z][A-Za-z0-9]*Error|Error):\s/;
|
|
775
|
+
export const HOST_STARVED_DURATION_FACTOR = 2;
|
|
776
|
+
/** Every fresh failure head is timeout-class and the suite took twice its own baseline measurement. */
|
|
777
|
+
export function hostStarved(fresh, durationMs, referenceMs) {
|
|
778
|
+
if (!referenceMs || durationMs < HOST_STARVED_DURATION_FACTOR * referenceMs)
|
|
779
|
+
return false;
|
|
780
|
+
const heads = fresh.split("\n").map((line) => line.replace(ANSI_RE, "")).filter((line) => ERROR_HEAD_RE.test(line));
|
|
781
|
+
return heads.length > 0 && heads.every((line) => TIMEOUT_CLASS_RE.test(line));
|
|
782
|
+
}
|
|
783
|
+
const DEFAULT_CALM = { pollMs: 5_000, maxWaitMs: 600_000, loadProvider: () => loadavg()[0] ?? 0, calmLoad: () => availableParallelism() / 2 };
|
|
784
|
+
let calm = DEFAULT_CALM;
|
|
785
|
+
export function setCalmWindowForTests(over) { calm = { ...calm, ...over }; }
|
|
786
|
+
export function resetCalmWindowForTests() { calm = DEFAULT_CALM; }
|
|
787
|
+
async function waitForCalmWindow() {
|
|
788
|
+
const started = Date.now();
|
|
789
|
+
while (calm.loadProvider() > calm.calmLoad() && Date.now() - started < calm.maxWaitMs) {
|
|
790
|
+
await new Promise((resolve) => setTimeout(resolve, calm.pollMs));
|
|
791
|
+
}
|
|
792
|
+
return Date.now() - started;
|
|
793
|
+
}
|
package/dist/gates/review.d.ts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { type Assignment, type BillingChannel, type WorkerAdapter } from "../adapters/types.js";
|
|
2
2
|
import { type TickmarkrConfig, type Tier } from "../config/config.js";
|
|
3
3
|
import { type Task } from "../graph/schema.js";
|
|
4
|
+
import { modelProvider } from "../route/preference.js";
|
|
4
5
|
import { type GateVia } from "./llm.js";
|
|
5
6
|
import type { GateResult } from "./types.js";
|
|
6
7
|
import { type VerdictUnparseableCause } from "./verdict-cause.js";
|
|
@@ -48,8 +49,7 @@ export declare function checkTaskDiffCaps(gate: string, measured: Pick<TaskDiffM
|
|
|
48
49
|
export declare function isDiffCapPark(result: GateResult): boolean;
|
|
49
50
|
export declare function diffCapParkReason(results: GateResult[]): string | null;
|
|
50
51
|
export declare function modelId(model: string): string;
|
|
51
|
-
|
|
52
|
-
export declare function modelProvider(model: string, fallback?: string): string;
|
|
52
|
+
export { modelProvider };
|
|
53
53
|
export declare function pickReviewer(author: Assignment, channels: BillingChannel[], exclude?: string[], // v1.1 failover: reviewer channels that already produced garbage for this task
|
|
54
54
|
prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
|
|
55
55
|
floor?: Tier, // task-declared only; config floors govern workers and must not silently move review seats
|
package/dist/gates/review.js
CHANGED
|
@@ -8,6 +8,7 @@ import { getAdapter } from "../adapters/registry.js";
|
|
|
8
8
|
import { shOk } from "../run/git.js";
|
|
9
9
|
import { redactSecrets } from "../run/redact.js";
|
|
10
10
|
import { marginalCostRank } from "../route/router.js";
|
|
11
|
+
import { modelProvider } from "../route/preference.js";
|
|
11
12
|
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlmDetailed, verdictNonceLine } from "./llm.js";
|
|
12
13
|
import { classifyVerdictCause } from "./verdict-cause.js";
|
|
13
14
|
import { captureDiffCapFor, measureArtifactDiff, reviewableLogicDiff, } from "./artifact-manifest.js";
|
|
@@ -165,24 +166,7 @@ export function diffCapParkReason(results) {
|
|
|
165
166
|
export function modelId(model) {
|
|
166
167
|
return model.slice(model.lastIndexOf("/") + 1);
|
|
167
168
|
}
|
|
168
|
-
|
|
169
|
-
export function modelProvider(model, fallback = "unknown") {
|
|
170
|
-
const id = model.toLowerCase();
|
|
171
|
-
const prefix = id.includes("/") ? id.slice(0, id.indexOf("/")) : "";
|
|
172
|
-
if (prefix === "openai" || prefix === "openai-codex" || /^(?:gpt|o\d)/.test(id))
|
|
173
|
-
return "openai";
|
|
174
|
-
if (prefix === "anthropic" || /^(?:claude|opus|sonnet|haiku|fable)(?:-|$)/.test(id))
|
|
175
|
-
return "anthropic";
|
|
176
|
-
if (prefix === "google" || /^gemini(?:-|$)/.test(id))
|
|
177
|
-
return "google";
|
|
178
|
-
if (prefix === "xai" || /^grok(?:-|$)/.test(id))
|
|
179
|
-
return "xai";
|
|
180
|
-
if (["zai", "zhipu", "zai-coding-plan"].includes(prefix) || /^glm(?:-|$)/.test(id))
|
|
181
|
-
return "zhipu";
|
|
182
|
-
if (["kimi-code", "moonshot"].includes(prefix) || /^kimi(?:-|$)/.test(id))
|
|
183
|
-
return "moonshot";
|
|
184
|
-
return fallback;
|
|
185
|
-
}
|
|
169
|
+
export { modelProvider };
|
|
186
170
|
// v1.53 T2: same entry grammar as routing.map.prefer (router.ts preferIndex — router is out of this
|
|
187
171
|
// module's dependency direction for a private fn, so the 3 lines live here too): `adapter` matches
|
|
188
172
|
// every channel of that adapter, `adapter:model` exactly one; unmatched channels sort after all entries.
|
package/dist/graph/graph.d.ts
CHANGED
|
@@ -1,6 +1,26 @@
|
|
|
1
1
|
import { type RunGraph, type Task, type TaskStatus } from "./schema.js";
|
|
2
2
|
export declare function stateDirName(_repoRoot: string): string;
|
|
3
3
|
export declare function graphPath(repoRoot: string): string;
|
|
4
|
+
export interface CompileRefusalRecord {
|
|
5
|
+
refusedAt: string;
|
|
6
|
+
source: string;
|
|
7
|
+
error: string;
|
|
8
|
+
}
|
|
9
|
+
export declare function compileRefusalPath(repoRoot: string): string;
|
|
10
|
+
export declare function readCompileRefusal(repoRoot: string): CompileRefusalRecord | undefined;
|
|
11
|
+
export declare function saveCompileRefusal(repoRoot: string, record: CompileRefusalRecord): void;
|
|
12
|
+
export declare function clearCompileRefusal(repoRoot: string): void;
|
|
13
|
+
export interface OnDiskSpecHash {
|
|
14
|
+
path: string;
|
|
15
|
+
hash: string;
|
|
16
|
+
}
|
|
17
|
+
/**
|
|
18
|
+
* Re-hash the exact single source file recorded by CLI compiles. Native and PRD record the source
|
|
19
|
+
* markdown; Spec Kit records its tasks.md. GSD combines several plan bodies and is deliberately not
|
|
20
|
+
* reconstructed here. A missing file is the one fail-open case: the refusal record is the fail-closed
|
|
21
|
+
* evidence for failed recompiles, while moved/deleted source files need not strand a compiled graph.
|
|
22
|
+
*/
|
|
23
|
+
export declare function onDiskSpecHash(_repoRoot: string, graph: RunGraph): OnDiskSpecHash | undefined;
|
|
4
24
|
export declare function graphDefinitionHash(g: RunGraph): string;
|
|
5
25
|
export declare function taskContentDigest(task: Pick<Task, "goal" | "files" | "acceptance">): string;
|
|
6
26
|
export declare function tickmarkrDir(repoRoot: string): string;
|
package/dist/graph/graph.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { createHash } from "node:crypto";
|
|
2
2
|
import { existsSync, mkdirSync, readFileSync, renameSync, rmSync, writeFileSync } from "node:fs";
|
|
3
|
-
import { join } from "node:path";
|
|
3
|
+
import { isAbsolute, join } from "node:path";
|
|
4
4
|
import { validateGraph } from "./schema.js";
|
|
5
5
|
export function stateDirName(_repoRoot) {
|
|
6
6
|
return ".tickmarkr";
|
|
@@ -8,6 +8,71 @@ export function stateDirName(_repoRoot) {
|
|
|
8
8
|
export function graphPath(repoRoot) {
|
|
9
9
|
return join(repoRoot, stateDirName(repoRoot), "graph.json");
|
|
10
10
|
}
|
|
11
|
+
export function compileRefusalPath(repoRoot) {
|
|
12
|
+
return join(repoRoot, stateDirName(repoRoot), "compile-refusal.json");
|
|
13
|
+
}
|
|
14
|
+
export function readCompileRefusal(repoRoot) {
|
|
15
|
+
const path = compileRefusalPath(repoRoot);
|
|
16
|
+
if (!existsSync(path))
|
|
17
|
+
return undefined;
|
|
18
|
+
let value;
|
|
19
|
+
try {
|
|
20
|
+
value = JSON.parse(readFileSync(path, "utf8"));
|
|
21
|
+
}
|
|
22
|
+
catch (error) {
|
|
23
|
+
throw new Error(`compile refusal record at ${path} is unreadable: ${error instanceof Error ? error.message : String(error)}`);
|
|
24
|
+
}
|
|
25
|
+
if (typeof value !== "object" || value === null
|
|
26
|
+
|| typeof value.refusedAt !== "string"
|
|
27
|
+
|| typeof value.source !== "string"
|
|
28
|
+
|| typeof value.error !== "string") {
|
|
29
|
+
throw new Error(`compile refusal record at ${path} is malformed`);
|
|
30
|
+
}
|
|
31
|
+
return value;
|
|
32
|
+
}
|
|
33
|
+
export function saveCompileRefusal(repoRoot, record) {
|
|
34
|
+
tickmarkrDir(repoRoot);
|
|
35
|
+
const path = compileRefusalPath(repoRoot);
|
|
36
|
+
const tmp = `${path}.${process.pid}.tmp`;
|
|
37
|
+
try {
|
|
38
|
+
writeFileSync(tmp, JSON.stringify(record, null, 2) + "\n");
|
|
39
|
+
renameSync(tmp, path);
|
|
40
|
+
}
|
|
41
|
+
catch (error) {
|
|
42
|
+
rmSync(tmp, { force: true });
|
|
43
|
+
throw error;
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
export function clearCompileRefusal(repoRoot) {
|
|
47
|
+
rmSync(compileRefusalPath(repoRoot), { force: true });
|
|
48
|
+
}
|
|
49
|
+
/**
|
|
50
|
+
* Re-hash the exact single source file recorded by CLI compiles. Native and PRD record the source
|
|
51
|
+
* markdown; Spec Kit records its tasks.md. GSD combines several plan bodies and is deliberately not
|
|
52
|
+
* reconstructed here. A missing file is the one fail-open case: the refusal record is the fail-closed
|
|
53
|
+
* evidence for failed recompiles, while moved/deleted source files need not strand a compiled graph.
|
|
54
|
+
*/
|
|
55
|
+
export function onDiskSpecHash(_repoRoot, graph) {
|
|
56
|
+
if (graph.spec.source === "gsd")
|
|
57
|
+
return undefined;
|
|
58
|
+
if (graph.spec.paths.length !== 1)
|
|
59
|
+
return undefined;
|
|
60
|
+
const recorded = graph.spec.paths[0];
|
|
61
|
+
// CLI compilation records an absolute source. Relative paths belong to legacy/programmatic
|
|
62
|
+
// graphs whose resolution context is unknowable here, so preserve their prior fail-open behavior.
|
|
63
|
+
if (!isAbsolute(recorded))
|
|
64
|
+
return undefined;
|
|
65
|
+
const path = recorded;
|
|
66
|
+
try {
|
|
67
|
+
const content = readFileSync(path);
|
|
68
|
+
return { path, hash: createHash("sha256").update(content).digest("hex") };
|
|
69
|
+
}
|
|
70
|
+
catch (error) {
|
|
71
|
+
if (error.code === "ENOENT")
|
|
72
|
+
return undefined;
|
|
73
|
+
throw new Error(`cannot verify compiled spec hash from ${path}: ${error instanceof Error ? error.message : String(error)}`);
|
|
74
|
+
}
|
|
75
|
+
}
|
|
11
76
|
// T3 (Sol #2 / Fable F2): ONE canonical engagement identity over COMPILED TASK DEFINITIONS only.
|
|
12
77
|
// status/evidence are runtime-mutated (the daemon flips status, accumulates evidence every attempt) so
|
|
13
78
|
// they are excluded — the identity survives a status flip or evidence growth but changes the instant a
|
|
@@ -5,7 +5,9 @@ export interface Disallowed {
|
|
|
5
5
|
entry: string;
|
|
6
6
|
}
|
|
7
7
|
export type PreferenceRole = "worker" | "judge" | "review" | "consult";
|
|
8
|
-
|
|
8
|
+
/** Provider identity comes from the served model, not a gateway adapter's stamped vendor. */
|
|
9
|
+
export declare function modelProvider(model: string, fallback?: string): string;
|
|
10
|
+
export declare const routingModelProvider: typeof modelProvider;
|
|
9
11
|
export declare const modelRouteIdentity: (model: string, fallback?: string) => string;
|
|
10
12
|
export declare const channelRouteIdentity: (key: string, fallback?: string) => string;
|
|
11
13
|
export declare function routingEntrySeatLines(cfg: TickmarkrConfig): string[];
|
package/dist/route/preference.js
CHANGED
|
@@ -2,10 +2,8 @@ import { channelKey, channelsFromConfig } from "../adapters/types.js";
|
|
|
2
2
|
import { validateGraph } from "../graph/schema.js";
|
|
3
3
|
import { route, RoutingError } from "./router.js";
|
|
4
4
|
const PREFERENCE_ROLES = ["worker", "judge", "review", "consult"];
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
// avoids importing gates back into route/run and must move with that helper when the scope permits.
|
|
8
|
-
export function routingModelProvider(model, fallback = "unknown") {
|
|
5
|
+
/** Provider identity comes from the served model, not a gateway adapter's stamped vendor. */
|
|
6
|
+
export function modelProvider(model, fallback = "unknown") {
|
|
9
7
|
const id = model.toLowerCase();
|
|
10
8
|
const prefix = id.includes("/") ? id.slice(0, id.indexOf("/")) : "";
|
|
11
9
|
if (prefix === "openai" || prefix === "openai-codex" || /^(?:gpt|o\d)/.test(id))
|
|
@@ -22,6 +20,8 @@ export function routingModelProvider(model, fallback = "unknown") {
|
|
|
22
20
|
return "moonshot";
|
|
23
21
|
return fallback;
|
|
24
22
|
}
|
|
23
|
+
// Router naming remains explicit while sharing the review gate's exact function object.
|
|
24
|
+
export const routingModelProvider = modelProvider;
|
|
25
25
|
export const modelRouteIdentity = (model, fallback = "unknown") => `${routingModelProvider(model, fallback)}/${model.slice(model.lastIndexOf("/") + 1).toLowerCase()}`;
|
|
26
26
|
export const channelRouteIdentity = (key, fallback = "unknown") => {
|
|
27
27
|
const i = key.indexOf(":");
|
package/dist/run/consult.js
CHANGED
|
@@ -63,6 +63,7 @@ function excludedProviderFromDossier(d, adapter) {
|
|
|
63
63
|
try {
|
|
64
64
|
const events = JSON.parse(d.journalTail);
|
|
65
65
|
const assignment = [...events].reverse().find((event) => event.event === "task-dispatch"
|
|
66
|
+
&& event.taskId === d.taskId
|
|
66
67
|
&& event.data?.assignment?.adapter === adapter
|
|
67
68
|
&& typeof event.data.assignment.model === "string")?.data?.assignment;
|
|
68
69
|
if (typeof assignment?.model !== "string")
|
package/dist/run/daemon.d.ts
CHANGED
|
@@ -117,6 +117,11 @@ export declare const WORKER_NUDGE_MESSAGE = "tickmarkr liveness check: if the ta
|
|
|
117
117
|
/** Test seam — shrink the nudge gate and grace without minute-long sleeps. */
|
|
118
118
|
export declare function setNudgeTimingForTests(silentMs: number, graceMs: number): void;
|
|
119
119
|
export declare function resetNudgeTimingForTests(): void;
|
|
120
|
+
export declare const WORKER_STARTUP_WINDOW_MS = 60000;
|
|
121
|
+
export declare const WORKER_STARTUP_WINDOW_BYTES: number;
|
|
122
|
+
/** Test seam — exercises both sides of the startup boundary without a production-length fixture. */
|
|
123
|
+
export declare function setWorkerStartupWindowMsForTests(ms: number): void;
|
|
124
|
+
export declare function resetWorkerStartupWindowMsForTests(): void;
|
|
120
125
|
/** Test seam — shrink the quota-banner silence gate without minute-long sleeps. */
|
|
121
126
|
export declare function setQuotaBannerSilentMsForTests(ms: number): void;
|
|
122
127
|
export declare function resetQuotaBannerSilentMsForTests(): void;
|