tickmarkr 1.67.0 → 1.69.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/kimi.d.ts +25 -1
- package/dist/adapters/kimi.js +80 -0
- package/dist/adapters/types.d.ts +6 -0
- package/dist/cli/commands/eval.d.ts +4 -0
- package/dist/cli/commands/eval.js +26 -0
- package/dist/cli/commands/status.d.ts +20 -0
- package/dist/cli/commands/status.js +9 -9
- package/dist/cli/index.d.ts +1 -1
- package/dist/cli/index.js +3 -1
- package/dist/eval/canary.d.ts +46 -0
- package/dist/eval/canary.js +113 -0
- package/dist/eval/dispatch.d.ts +31 -0
- package/dist/eval/dispatch.js +207 -0
- package/dist/eval/fixtures.d.ts +22 -0
- package/dist/eval/fixtures.js +85 -0
- package/dist/eval/report.d.ts +35 -0
- package/dist/eval/report.js +82 -0
- package/dist/eval/selfcheck.d.ts +22 -0
- package/dist/eval/selfcheck.js +177 -0
- package/dist/run/daemon.js +130 -83
- package/dist/run/git.d.ts +2 -0
- package/dist/run/git.js +9 -0
- package/dist/run/interactive-seed.d.ts +15 -0
- package/dist/run/interactive-seed.js +25 -0
- package/dist/tui/app.d.ts +3 -0
- package/dist/tui/app.js +14 -2
- package/dist/tui/views/consult-dossier.d.ts +49 -0
- package/dist/tui/views/consult-dossier.js +169 -0
- package/dist/tui/views/runs-view.d.ts +70 -0
- package/dist/tui/views/runs-view.js +387 -0
- package/fixtures/eval/canary/solution/a.txt +1 -0
- package/fixtures/eval/canary/spec.md +8 -0
- package/fixtures/eval/canary/start/a.txt +1 -0
- package/fixtures/eval/sample/solution/a.txt +1 -0
- package/fixtures/eval/sample/spec.md +8 -0
- package/fixtures/eval/sample/start/a.txt +1 -0
- package/fixtures/gsd-sample/07-live-check/07-01-PLAN.md +42 -0
- package/fixtures/gsd-sample/07-live-check/07-02-PLAN.md +21 -0
- package/fixtures/gsd-sample/07-live-check/07-03-PLAN.md +18 -0
- package/fixtures/gsd-sample/07-live-check/07-03-SUMMARY.md +1 -0
- package/fixtures/missing-mandatory-gate.native.md +10 -0
- package/fixtures/sample-pin.prd.md +18 -0
- package/fixtures/sample.native.md +35 -0
- package/fixtures/sample.prd.md +22 -0
- package/fixtures/speckit-sample/tasks.md +20 -0
- package/package.json +3 -2
package/dist/adapters/kimi.d.ts
CHANGED
|
@@ -1,6 +1,30 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import type { ExecutorDriver, Slot } from "../drivers/types.js";
|
|
2
|
+
import { type Assignment, type WorkerAdapter, type WorkerResult } from "./types.js";
|
|
2
3
|
export declare function kimiAuthed(credentialsText: string, nowMs: number): boolean;
|
|
3
4
|
export declare function parseKimiModels(raw: string): string[];
|
|
4
5
|
export declare function parseKimiResult(raw: string, nonce: string): WorkerResult;
|
|
5
6
|
export declare function kimiSessionId(output: string): string | undefined;
|
|
7
|
+
export declare function kimiBannerModel(banner: string): string | undefined;
|
|
8
|
+
export declare function kimiBannerSessionId(banner: string): string | undefined;
|
|
9
|
+
export type KimiBannerConfirm = {
|
|
10
|
+
ok: true;
|
|
11
|
+
sessionId?: string;
|
|
12
|
+
} | {
|
|
13
|
+
ok: false;
|
|
14
|
+
error: string;
|
|
15
|
+
};
|
|
16
|
+
export declare function confirmKimiSeedBanner(banner: string, assignedModel: string): KimiBannerConfirm;
|
|
17
|
+
export interface KimiInteractiveSeedResult {
|
|
18
|
+
output: string;
|
|
19
|
+
seedFailed: boolean;
|
|
20
|
+
seedError?: string;
|
|
21
|
+
sessionId?: string;
|
|
22
|
+
}
|
|
23
|
+
export declare function runKimiInteractiveSeed(opts: {
|
|
24
|
+
driver: Pick<ExecutorDriver, "run" | "waitOutput" | "read">;
|
|
25
|
+
slot: Slot;
|
|
26
|
+
assignment: Assignment;
|
|
27
|
+
promptFile: string;
|
|
28
|
+
taskTimeoutMinutes: number;
|
|
29
|
+
}): Promise<KimiInteractiveSeedResult>;
|
|
6
30
|
export declare const kimi: WorkerAdapter;
|
package/dist/adapters/kimi.js
CHANGED
|
@@ -68,16 +68,90 @@ export function parseKimiResult(raw, nonce) {
|
|
|
68
68
|
// line only: prompt/model prose can contain lookalike text, and the anchored charset keeps a
|
|
69
69
|
// captured id shell-safe by construction (shq in resumeCommand is the second layer). Last valid
|
|
70
70
|
// line wins — a run may echo stale resume lines mid-transcript.
|
|
71
|
+
//
|
|
72
|
+
// v1.69 T7: the native TUI cold-start banner also prints `Session: session_<uuid>` at launch.
|
|
73
|
+
// Prefer a banner Session line when present so seed-mode captures the id from the launch banner
|
|
74
|
+
// itself rather than waiting for a completion-time resume trailer (which the TUI may never emit).
|
|
71
75
|
const RESUME_TRAILER_RE = /^\s*To resume this session: kimi -r (session_[0-9a-f-]+)\s*$/;
|
|
76
|
+
const BANNER_SESSION_LINE_RE = /^\s*Session:\s*(session_[0-9a-f-]+)\s*$/;
|
|
72
77
|
export function kimiSessionId(output) {
|
|
73
78
|
let id;
|
|
74
79
|
for (const line of output.split("\n")) {
|
|
80
|
+
const banner = BANNER_SESSION_LINE_RE.exec(line);
|
|
81
|
+
if (banner) {
|
|
82
|
+
id = banner[1];
|
|
83
|
+
continue;
|
|
84
|
+
}
|
|
75
85
|
const m = RESUME_TRAILER_RE.exec(line);
|
|
76
86
|
if (m)
|
|
77
87
|
id = m[1];
|
|
78
88
|
}
|
|
79
89
|
return id;
|
|
80
90
|
}
|
|
91
|
+
// v1.69 T6: the native TUI takes -m <alias>, where config.toml aliases are the bare model suffix of
|
|
92
|
+
// the tickmarkr channel id (live probe 2026-07-22). Keep the mapping explicit and localized.
|
|
93
|
+
function kimiAlias(model) {
|
|
94
|
+
return model.replace(/^kimi-code\//, "");
|
|
95
|
+
}
|
|
96
|
+
// v1.69 T7: the cold-start banner prints the model alias and session id. Parse them from the
|
|
97
|
+
// banner text already captured for the readiness match — no new probe, no extra dispatch.
|
|
98
|
+
const BANNER_MODEL_RE = /^Model:\s*(.+)$/m;
|
|
99
|
+
const BANNER_SESSION_RE = /^Session:\s*(session_[0-9a-f-]+)$/m;
|
|
100
|
+
export function kimiBannerModel(banner) {
|
|
101
|
+
const m = BANNER_MODEL_RE.exec(banner);
|
|
102
|
+
if (!m)
|
|
103
|
+
return undefined;
|
|
104
|
+
const alias = m[1].trim();
|
|
105
|
+
if (!alias)
|
|
106
|
+
return undefined;
|
|
107
|
+
return `kimi-code/${alias}`;
|
|
108
|
+
}
|
|
109
|
+
export function kimiBannerSessionId(banner) {
|
|
110
|
+
return BANNER_SESSION_RE.exec(banner)?.[1];
|
|
111
|
+
}
|
|
112
|
+
// Fail closed on a named-model mismatch; a missing Model line is not a mismatch (banner partial).
|
|
113
|
+
export function confirmKimiSeedBanner(banner, assignedModel) {
|
|
114
|
+
const saw = kimiBannerModel(banner);
|
|
115
|
+
if (saw !== undefined && saw !== assignedModel) {
|
|
116
|
+
return { ok: false, error: `model mismatch: expected ${assignedModel}, saw ${saw}` };
|
|
117
|
+
}
|
|
118
|
+
return { ok: true, sessionId: kimiBannerSessionId(banner) };
|
|
119
|
+
}
|
|
120
|
+
// Shared launch-then-seed surface (T6) + banner confirm helpers (T7). One definition so the
|
|
121
|
+
// adapter property and runKimiInteractiveSeed cannot drift.
|
|
122
|
+
const KIMI_SEED = {
|
|
123
|
+
launch: (model) => `kimi -y -m ${shq(kimiAlias(model))}`,
|
|
124
|
+
readinessMatch: "Send /help for help information.",
|
|
125
|
+
seedLine: (promptFile) => `Read ${promptFile} and do exactly what it says.`,
|
|
126
|
+
};
|
|
127
|
+
// v1.69 T7: kimi-local launch-then-seed that confirms the banner model and captures the session id
|
|
128
|
+
// from the SAME pane text already present for the readiness match. No extra probe, no extra
|
|
129
|
+
// dispatch — one read of the launch banner, then seed or fail closed.
|
|
130
|
+
export async function runKimiInteractiveSeed(opts) {
|
|
131
|
+
await opts.driver.run(opts.slot, KIMI_SEED.launch(opts.assignment.model));
|
|
132
|
+
const ready = await opts.driver.waitOutput(opts.slot, KIMI_SEED.readinessMatch, opts.taskTimeoutMinutes * 60_000);
|
|
133
|
+
// Banner text already on the pane for the readiness match — the only text model/session use.
|
|
134
|
+
const banner = await opts.driver.read(opts.slot, 1000);
|
|
135
|
+
if (!ready) {
|
|
136
|
+
return { output: banner, seedFailed: true, seedError: `readiness pattern not seen: ${KIMI_SEED.readinessMatch}` };
|
|
137
|
+
}
|
|
138
|
+
const confirm = confirmKimiSeedBanner(banner, opts.assignment.model);
|
|
139
|
+
if (!confirm.ok) {
|
|
140
|
+
return { output: banner, seedFailed: true, seedError: confirm.error };
|
|
141
|
+
}
|
|
142
|
+
const seedText = KIMI_SEED.seedLine(opts.promptFile);
|
|
143
|
+
await opts.driver.run(opts.slot, seedText);
|
|
144
|
+
let output = "";
|
|
145
|
+
for (let attempt = 0; attempt < 5; attempt++) {
|
|
146
|
+
await new Promise((r) => setTimeout(r, 200));
|
|
147
|
+
output = await opts.driver.read(opts.slot, 1000);
|
|
148
|
+
const bottom = output.trimEnd().split("\n").pop() ?? "";
|
|
149
|
+
if (!bottom.includes(seedText)) {
|
|
150
|
+
return { output, seedFailed: false, sessionId: confirm.sessionId };
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
return { output, seedFailed: true, seedError: "seed line never left the input box", sessionId: confirm.sessionId };
|
|
154
|
+
}
|
|
81
155
|
export const kimi = {
|
|
82
156
|
id: "kimi",
|
|
83
157
|
vendor: "moonshot",
|
|
@@ -108,6 +182,12 @@ export const kimi = {
|
|
|
108
182
|
// ("unknown command '…'", live-verified 2026-07-17, OBS-67) and -p is non-interactive-only.
|
|
109
183
|
// null → the daemon's print fallback (types.ts:101) keeps kimi workers visible without a TUI.
|
|
110
184
|
interactiveCommand: () => null,
|
|
185
|
+
// v1.69 T6/T7: launch the real TUI, wait for the cold-start readiness marker, then submit the
|
|
186
|
+
// prompt as one user turn. Banner model/session confirmation for seed mode lives in
|
|
187
|
+
// runKimiInteractiveSeed / confirmKimiSeedBanner (same pane text as the readiness match — no
|
|
188
|
+
// extra probe). Live-probed 2026-07-22 (kimi 0.28.1): `kimi -y` opens the TUI with yolo active
|
|
189
|
+
// and prints the model/session lines plus "Send /help for help information." before the input box.
|
|
190
|
+
interactiveSeed: KIMI_SEED,
|
|
111
191
|
// v1.53 T3 resume — live-probed 2026-07-18: `-p` + `-S <id>` compose cleanly (no OBS-67-class
|
|
112
192
|
// flag rejection) and the resumed session carries prior conversation state. `-S <id>` is the
|
|
113
193
|
// deterministic form; `-c` rejected as primary — cwd-keyed, nondeterministic under worktree
|
package/dist/adapters/types.d.ts
CHANGED
|
@@ -48,6 +48,11 @@ export interface WorkerResult {
|
|
|
48
48
|
deviations: string[];
|
|
49
49
|
raw: string;
|
|
50
50
|
}
|
|
51
|
+
export interface InteractiveSeed {
|
|
52
|
+
launch(model: string): string;
|
|
53
|
+
readinessMatch: string;
|
|
54
|
+
seedLine(promptFile: string): string;
|
|
55
|
+
}
|
|
51
56
|
export interface ContextUsage {
|
|
52
57
|
tokens: number;
|
|
53
58
|
limit?: number;
|
|
@@ -78,6 +83,7 @@ export interface WorkerAdapter {
|
|
|
78
83
|
channels(cfg: TickmarkrConfig): BillingChannel[];
|
|
79
84
|
headlessCommand(promptFile: string, model: string): string;
|
|
80
85
|
interactiveCommand(promptFile: string, model: string): string | null;
|
|
86
|
+
interactiveSeed?: InteractiveSeed;
|
|
81
87
|
resumeCommand?(sessionId: string, promptFile: string, model: string): string;
|
|
82
88
|
sessionIdFrom?(output: string): string | undefined;
|
|
83
89
|
resumeUnknownContext?: boolean;
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
import { parseArgs } from "node:util";
|
|
2
|
+
import { discoverFixtures, resolveFixturesRoot, seedFixture } from "../../eval/fixtures.js";
|
|
3
|
+
export async function evalCommand(argv, cwd = process.cwd()) {
|
|
4
|
+
const { positionals } = parseArgs({ args: argv, allowPositionals: true });
|
|
5
|
+
const root = resolveFixturesRoot(positionals[0], cwd);
|
|
6
|
+
const { valid, invalid } = discoverFixtures(root);
|
|
7
|
+
const lines = [`tickmarkr eval — discovered ${valid.length} fixture${valid.length === 1 ? "" : "s"}`];
|
|
8
|
+
for (const f of valid)
|
|
9
|
+
lines.push(` ${f.id}`);
|
|
10
|
+
if (invalid.length) {
|
|
11
|
+
lines.push("", "invalid fixtures:");
|
|
12
|
+
for (const i of invalid)
|
|
13
|
+
lines.push(` ${i.id} — ${i.reason}`);
|
|
14
|
+
}
|
|
15
|
+
for (const f of valid) {
|
|
16
|
+
const seeded = await seedFixture(f);
|
|
17
|
+
try {
|
|
18
|
+
lines.push(` seeded ${f.id}`);
|
|
19
|
+
}
|
|
20
|
+
finally {
|
|
21
|
+
await seeded.cleanup();
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
const out = lines.join("\n");
|
|
25
|
+
return invalid.length ? { out, code: 1 } : out;
|
|
26
|
+
}
|
|
@@ -1,5 +1,25 @@
|
|
|
1
|
+
import { type Task, type TaskStatus } from "../../graph/schema.js";
|
|
2
|
+
import { type JournalEvent } from "../../run/journal.js";
|
|
1
3
|
export type StatusOpts = {
|
|
2
4
|
iterations?: number;
|
|
3
5
|
sleep?: (ms: number) => Promise<void>;
|
|
4
6
|
};
|
|
7
|
+
export declare const GATE_KEYS: {
|
|
8
|
+
readonly build: "B";
|
|
9
|
+
readonly test: "T";
|
|
10
|
+
readonly lint: "L";
|
|
11
|
+
readonly evidence: "E";
|
|
12
|
+
readonly scope: "S";
|
|
13
|
+
readonly acceptance: "A";
|
|
14
|
+
readonly review: "R";
|
|
15
|
+
};
|
|
16
|
+
export declare const taskBox: (status: TaskStatus) => string;
|
|
17
|
+
export declare const gateBox: (state: "open" | "pass" | "fail" | "skip", unicode: boolean) => string;
|
|
18
|
+
export type GateState = "open" | "pass" | "fail" | "skip";
|
|
19
|
+
export declare const defaultGateStates: (task: Task) => GateState[];
|
|
20
|
+
export declare const gateStates: (task: Task, events: JournalEvent[]) => GateState[];
|
|
21
|
+
export declare const gateChain: (states: GateState[], unicode: boolean) => string;
|
|
22
|
+
export declare const failedGates: (states: GateState[]) => string[];
|
|
23
|
+
export declare const humanGateSuffix: (t: Task, st: TaskStatus, states: GateState[]) => string;
|
|
24
|
+
export declare const shortGoal: (goal: string, max: number) => string;
|
|
5
25
|
export declare function status(argv: string[], cwd?: string, opts?: StatusOpts): Promise<string>;
|
|
@@ -9,7 +9,7 @@ const NOT_COMPARABLE_NOTICE = "graph recompiled since this run — task states n
|
|
|
9
9
|
// The timer must keep the process ALIVE: an unref'd timer here let the event loop drain after the
|
|
10
10
|
// first frame, so a live `--watch` printed once and exited 0 (OBS-11). Never unref this.
|
|
11
11
|
const defaultSleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
|
12
|
-
const GATE_KEYS = { build: "B", test: "T", lint: "L", evidence: "E", scope: "S", acceptance: "A", review: "R" };
|
|
12
|
+
export const GATE_KEYS = { build: "B", test: "T", lint: "L", evidence: "E", scope: "S", acceptance: "A", review: "R" };
|
|
13
13
|
const attemptStartIdx = (events, taskId) => {
|
|
14
14
|
let idx = -1;
|
|
15
15
|
for (let i = 0; i < events.length; i++) {
|
|
@@ -21,22 +21,22 @@ const attemptStartIdx = (events, taskId) => {
|
|
|
21
21
|
};
|
|
22
22
|
const visual = () => process.stdout.isTTY === true && process.env.NO_COLOR === undefined;
|
|
23
23
|
// non-TTY machine surface only — the TTY frame draws task verdicts via statusRow
|
|
24
|
-
const taskBox = (status) => {
|
|
24
|
+
export const taskBox = (status) => {
|
|
25
25
|
if (status === "done")
|
|
26
26
|
return "[x]";
|
|
27
27
|
if (status === "failed" || status === "human")
|
|
28
28
|
return "[!]";
|
|
29
29
|
return "[ ]";
|
|
30
30
|
};
|
|
31
|
-
const gateBox = (state, unicode) => {
|
|
31
|
+
export const gateBox = (state, unicode) => {
|
|
32
32
|
if (unicode) {
|
|
33
33
|
// shared glyph vocabulary: pass/fail verdicts, dash for skip, dim circle for not-yet-run
|
|
34
34
|
return state === "pass" ? GLYPHS.pass : state === "fail" ? GLYPHS.fail : state === "skip" ? GLYPHS.neutral : GLYPHS.toggleInactive;
|
|
35
35
|
}
|
|
36
36
|
return state === "pass" ? "[x]" : state === "fail" ? "[!]" : state === "skip" ? "." : "[ ]";
|
|
37
37
|
};
|
|
38
|
-
const defaultGateStates = (task) => GATE_NAMES.map((gate) => task.gates.includes(gate) ? "open" : "skip");
|
|
39
|
-
const gateStates = (task, events) => {
|
|
38
|
+
export const defaultGateStates = (task) => GATE_NAMES.map((gate) => task.gates.includes(gate) ? "open" : "skip");
|
|
39
|
+
export const gateStates = (task, events) => {
|
|
40
40
|
const outcomes = new Map();
|
|
41
41
|
const start = attemptStartIdx(events, task.id);
|
|
42
42
|
if (start >= 0) {
|
|
@@ -57,10 +57,10 @@ const gateStates = (task, events) => {
|
|
|
57
57
|
const GATE_STATE_TOKEN = { pass: ok, fail, skip: dim, open: dim };
|
|
58
58
|
// TTY cells are bare glyphs in fixed GATE_NAMES order — gate identity lives once in the frame
|
|
59
59
|
// legend, and in words on a failing row; non-TTY keeps the letter+box chips (byte-pinned surface)
|
|
60
|
-
const gateChain = (states, unicode) => GATE_NAMES.map((gate, i) => unicode
|
|
60
|
+
export const gateChain = (states, unicode) => GATE_NAMES.map((gate, i) => unicode
|
|
61
61
|
? GATE_STATE_TOKEN[states[i]](gateBox(states[i], true))
|
|
62
62
|
: `${GATE_KEYS[gate]}${gateBox(states[i], false)}`).join(" ");
|
|
63
|
-
const failedGates = (states) => GATE_NAMES.filter((_, i) => states[i] === "fail");
|
|
63
|
+
export const failedGates = (states) => GATE_NAMES.filter((_, i) => states[i] === "fail");
|
|
64
64
|
// plain-text form of the failed-gate words for column math; rendered with a dim dot + red names
|
|
65
65
|
const failedSuffix = (states) => {
|
|
66
66
|
const f = failedGates(states);
|
|
@@ -70,8 +70,8 @@ const failedSuffix = (states) => {
|
|
|
70
70
|
// awaited approval is named from task state alone — never from gate results. Failed gates win the
|
|
71
71
|
// cell when present: a post-approval park is not awaiting the designed gate. Plain text for column
|
|
72
72
|
// math; rendered with a dim dot + warn words. TTY-only — the non-TTY surface stays byte-pinned.
|
|
73
|
-
const humanGateSuffix = (t, st, states) => st === "human" && t.humanGate && failedGates(states).length === 0 ? " · awaiting approval" : "";
|
|
74
|
-
const shortGoal = (goal, max) => {
|
|
73
|
+
export const humanGateSuffix = (t, st, states) => st === "human" && t.humanGate && failedGates(states).length === 0 ? " · awaiting approval" : "";
|
|
74
|
+
export const shortGoal = (goal, max) => {
|
|
75
75
|
const clause = goal.split(/[,;.?!]/, 1)[0].trim();
|
|
76
76
|
if (clause.length <= max)
|
|
77
77
|
return clause;
|
package/dist/cli/index.d.ts
CHANGED
|
@@ -5,7 +5,7 @@ export type CommandResult = string | {
|
|
|
5
5
|
};
|
|
6
6
|
export type CommandMap = Record<string, (argv: string[]) => Promise<CommandResult>>;
|
|
7
7
|
export declare const COMMANDS: CommandMap;
|
|
8
|
-
export declare const USAGE = "tickmarkr \u2014 spec-driven orchestration harness for AI coding agents\nusage: tickmarkr <command>\n init guided setup + doctor; init --agent [--force] [--docs] adds agent skills/docs\n doctor re-probe adapters, herdr, auth; print capability matrix\n fleet interactive fleet editor (fleet --print for CI drift checks)\n compile <src> spec \u2192 .tickmarkr/graph.json (fails without acceptance criteria)\n scope <intent> draft a compiled native spec beside an answered intent (--force to overwrite)\n plan dry-run routing table + cost estimate + floor lints\n run execute the graph (--concurrency N --driver herdr|subprocess --route-strict)\n status live run state\n resume <id> continue a run from its journal\n report <id> cost/quality report (--md for committable execution record)\n profile show learned routing profile (profile reset = forget history via cursor, keeps telemetry)\n ui open the Fleet Studio TUI (full-screen tabbed cockpit)\n unlock remove a stale/garbage run lock (refuses if the holder is alive)\n approve <id> <task> approve a parked human gate (--by <name> --reason <text>); takes effect on resume";
|
|
8
|
+
export declare const USAGE = "tickmarkr \u2014 spec-driven orchestration harness for AI coding agents\nusage: tickmarkr <command>\n init guided setup + doctor; init --agent [--force] [--docs] adds agent skills/docs\n doctor re-probe adapters, herdr, auth; print capability matrix\n fleet interactive fleet editor (fleet --print for CI drift checks)\n compile <src> spec \u2192 .tickmarkr/graph.json (fails without acceptance criteria)\n scope <intent> draft a compiled native spec beside an answered intent (--force to overwrite)\n plan dry-run routing table + cost estimate + floor lints\n eval run checked-in fixtures against every channel in isolated temp repos\n run execute the graph (--concurrency N --driver herdr|subprocess --route-strict)\n status live run state\n resume <id> continue a run from its journal\n report <id> cost/quality report (--md for committable execution record)\n profile show learned routing profile (profile reset = forget history via cursor, keeps telemetry)\n ui open the Fleet Studio TUI (full-screen tabbed cockpit)\n unlock remove a stale/garbage run lock (refuses if the holder is alive)\n approve <id> <task> approve a parked human gate (--by <name> --reason <text>); takes effect on resume";
|
|
9
9
|
export declare function dispatch(cmd: string | undefined, argv: string[], commands?: CommandMap): Promise<{
|
|
10
10
|
out: string;
|
|
11
11
|
code: number;
|
package/dist/cli/index.js
CHANGED
|
@@ -4,6 +4,7 @@ import { pathToFileURL } from "node:url";
|
|
|
4
4
|
import { approve } from "./commands/approve.js";
|
|
5
5
|
import { compile } from "./commands/compile.js";
|
|
6
6
|
import { doctor } from "./commands/doctor.js";
|
|
7
|
+
import { evalCommand } from "./commands/eval.js";
|
|
7
8
|
import { fleet } from "./commands/fleet.js";
|
|
8
9
|
import { init } from "./commands/init.js";
|
|
9
10
|
import { plan } from "./commands/plan.js";
|
|
@@ -18,7 +19,7 @@ import { unlock } from "./commands/unlock.js";
|
|
|
18
19
|
import { version } from "./commands/version.js";
|
|
19
20
|
const normalize = (r) => typeof r === "string" ? { out: r, code: 0 } : r;
|
|
20
21
|
export const COMMANDS = {
|
|
21
|
-
init, doctor, fleet, compile, scope, plan, run, status, resume, report, profile, ui, unlock, approve, version,
|
|
22
|
+
init, doctor, fleet, compile, scope, plan, run, status, resume, report, profile, ui, unlock, approve, version, eval: evalCommand,
|
|
22
23
|
};
|
|
23
24
|
const VERSION_FLAGS = new Set(["version", "--version", "-v"]);
|
|
24
25
|
const HELP_CMDS = new Set(["help", "-h", "--help"]);
|
|
@@ -31,6 +32,7 @@ usage: tickmarkr <command>
|
|
|
31
32
|
compile <src> spec → .tickmarkr/graph.json (fails without acceptance criteria)
|
|
32
33
|
scope <intent> draft a compiled native spec beside an answered intent (--force to overwrite)
|
|
33
34
|
plan dry-run routing table + cost estimate + floor lints
|
|
35
|
+
eval run checked-in fixtures against every channel in isolated temp repos
|
|
34
36
|
run execute the graph (--concurrency N --driver herdr|subprocess --route-strict)
|
|
35
37
|
status live run state
|
|
36
38
|
resume <id> continue a run from its journal
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import type { WorkerAdapter } from "../adapters/types.js";
|
|
2
|
+
import { type Fixture } from "./fixtures.js";
|
|
3
|
+
export declare const CANARY_FIXTURE_ID = "canary";
|
|
4
|
+
export interface CanaryJudgeResult {
|
|
5
|
+
fixtureId: string;
|
|
6
|
+
expectedPass: boolean;
|
|
7
|
+
judgePass: boolean;
|
|
8
|
+
breach: boolean;
|
|
9
|
+
details: string;
|
|
10
|
+
}
|
|
11
|
+
export interface FixtureChannelResult {
|
|
12
|
+
fixtureId: string;
|
|
13
|
+
channelKey: string;
|
|
14
|
+
skipped: boolean;
|
|
15
|
+
pass?: boolean;
|
|
16
|
+
}
|
|
17
|
+
export interface ChannelQualificationTotal {
|
|
18
|
+
channelKey: string;
|
|
19
|
+
passed: number;
|
|
20
|
+
failed: number;
|
|
21
|
+
skipped: number;
|
|
22
|
+
total: number;
|
|
23
|
+
}
|
|
24
|
+
export declare function isCanaryFixture(fixture: Fixture): boolean;
|
|
25
|
+
export declare function isCanaryResult(result: FixtureChannelResult): boolean;
|
|
26
|
+
export declare function resolveCanaryFixture(root: string): Fixture | undefined;
|
|
27
|
+
/**
|
|
28
|
+
* Return a fixture list that always includes the held-out canary fixture.
|
|
29
|
+
* If the canary is already present, the list is returned unchanged.
|
|
30
|
+
*/
|
|
31
|
+
export declare function ensureCanary(fixtures: Fixture[], root: string): Fixture[];
|
|
32
|
+
/**
|
|
33
|
+
* Run the held-out canary fixture against a judge adapter.
|
|
34
|
+
* The fixture is seeded, a deliberately failing change is committed, and the judge oracle
|
|
35
|
+
* is evaluated against the resulting diff. A correct judge returns pass=false; any pass=true
|
|
36
|
+
* verdict is flagged as a judge-integrity breach.
|
|
37
|
+
*/
|
|
38
|
+
export declare function runCanaryJudge(fixture: Fixture, judgeAdapter: WorkerAdapter, model: string): Promise<CanaryJudgeResult>;
|
|
39
|
+
/**
|
|
40
|
+
* Aggregate per-channel qualification totals from fixture-channel results.
|
|
41
|
+
* Results marked as canary (by the supplied predicate, defaulting to isCanaryResult) are excluded
|
|
42
|
+
* from every total so the canary never contributes to a channel's qualification score.
|
|
43
|
+
*/
|
|
44
|
+
export declare function aggregateChannelTotals(results: FixtureChannelResult[], options?: {
|
|
45
|
+
isCanary?: (r: FixtureChannelResult) => boolean;
|
|
46
|
+
}): ChannelQualificationTotal[];
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
import { existsSync, lstatSync, writeFileSync } from "node:fs";
|
|
2
|
+
import { join } from "node:path";
|
|
3
|
+
import { compileSource } from "../compile/index.js";
|
|
4
|
+
import { acceptanceGate } from "../gates/acceptance.js";
|
|
5
|
+
import { shGitOk } from "../run/git.js";
|
|
6
|
+
import { seedFixture } from "./fixtures.js";
|
|
7
|
+
export const CANARY_FIXTURE_ID = "canary";
|
|
8
|
+
export function isCanaryFixture(fixture) {
|
|
9
|
+
return fixture.id === CANARY_FIXTURE_ID;
|
|
10
|
+
}
|
|
11
|
+
export function isCanaryResult(result) {
|
|
12
|
+
return result.fixtureId === CANARY_FIXTURE_ID;
|
|
13
|
+
}
|
|
14
|
+
export function resolveCanaryFixture(root) {
|
|
15
|
+
const path = join(root, CANARY_FIXTURE_ID);
|
|
16
|
+
if (!existsSync(path) || !lstatSync(path).isDirectory())
|
|
17
|
+
return undefined;
|
|
18
|
+
const startDir = join(path, "start");
|
|
19
|
+
const solutionDir = join(path, "solution");
|
|
20
|
+
if (!existsSync(startDir) || !lstatSync(startDir).isDirectory())
|
|
21
|
+
return undefined;
|
|
22
|
+
if (!existsSync(solutionDir) || !lstatSync(solutionDir).isDirectory())
|
|
23
|
+
return undefined;
|
|
24
|
+
return { id: CANARY_FIXTURE_ID, path, startDir, solutionDir };
|
|
25
|
+
}
|
|
26
|
+
/**
|
|
27
|
+
* Return a fixture list that always includes the held-out canary fixture.
|
|
28
|
+
* If the canary is already present, the list is returned unchanged.
|
|
29
|
+
*/
|
|
30
|
+
export function ensureCanary(fixtures, root) {
|
|
31
|
+
if (fixtures.some(isCanaryFixture))
|
|
32
|
+
return fixtures;
|
|
33
|
+
const canary = resolveCanaryFixture(root);
|
|
34
|
+
return canary ? [canary, ...fixtures] : fixtures;
|
|
35
|
+
}
|
|
36
|
+
function compileCanaryTask(fixture) {
|
|
37
|
+
const specPath = join(fixture.path, "spec.md");
|
|
38
|
+
const graph = compileSource(specPath);
|
|
39
|
+
if (graph.tasks.length !== 1) {
|
|
40
|
+
throw new Error(`canary spec must contain exactly one task (found ${graph.tasks.length})`);
|
|
41
|
+
}
|
|
42
|
+
return graph.tasks[0];
|
|
43
|
+
}
|
|
44
|
+
// Apply a known-bad change so the judged diff contains quotable evidence of an unmet criterion.
|
|
45
|
+
async function applyKnownBadChange(repo) {
|
|
46
|
+
writeFileSync(join(repo, "a.txt"), "canary-wrong");
|
|
47
|
+
await shGitOk("git add -A", repo);
|
|
48
|
+
await shGitOk("git commit -m 'canary bad change' --no-gpg-sign", repo);
|
|
49
|
+
}
|
|
50
|
+
/**
|
|
51
|
+
* Run the held-out canary fixture against a judge adapter.
|
|
52
|
+
* The fixture is seeded, a deliberately failing change is committed, and the judge oracle
|
|
53
|
+
* is evaluated against the resulting diff. A correct judge returns pass=false; any pass=true
|
|
54
|
+
* verdict is flagged as a judge-integrity breach.
|
|
55
|
+
*/
|
|
56
|
+
export async function runCanaryJudge(fixture, judgeAdapter, model) {
|
|
57
|
+
if (!isCanaryFixture(fixture)) {
|
|
58
|
+
throw new Error(`fixture ${fixture.id} is not the canary fixture`);
|
|
59
|
+
}
|
|
60
|
+
const task = compileCanaryTask(fixture);
|
|
61
|
+
const seeded = await seedFixture(fixture);
|
|
62
|
+
const initialCommit = (await shGitOk("git rev-parse HEAD", seeded.repo)).trim();
|
|
63
|
+
try {
|
|
64
|
+
await applyKnownBadChange(seeded.repo);
|
|
65
|
+
const gateResult = await acceptanceGate(task, seeded.repo, initialCommit, { adapter: judgeAdapter, model });
|
|
66
|
+
const expectedPass = false;
|
|
67
|
+
const judgePass = gateResult.pass;
|
|
68
|
+
const breach = judgePass === true && expectedPass === false;
|
|
69
|
+
return {
|
|
70
|
+
fixtureId: fixture.id,
|
|
71
|
+
expectedPass,
|
|
72
|
+
judgePass,
|
|
73
|
+
breach,
|
|
74
|
+
details: gateResult.details,
|
|
75
|
+
};
|
|
76
|
+
}
|
|
77
|
+
finally {
|
|
78
|
+
await seeded.cleanup();
|
|
79
|
+
}
|
|
80
|
+
}
|
|
81
|
+
/**
|
|
82
|
+
* Aggregate per-channel qualification totals from fixture-channel results.
|
|
83
|
+
* Results marked as canary (by the supplied predicate, defaulting to isCanaryResult) are excluded
|
|
84
|
+
* from every total so the canary never contributes to a channel's qualification score.
|
|
85
|
+
*/
|
|
86
|
+
export function aggregateChannelTotals(results, options = {}) {
|
|
87
|
+
const isCanary = options.isCanary ?? isCanaryResult;
|
|
88
|
+
const byChannel = new Map();
|
|
89
|
+
for (const r of results) {
|
|
90
|
+
if (isCanary(r))
|
|
91
|
+
continue;
|
|
92
|
+
let total = byChannel.get(r.channelKey);
|
|
93
|
+
if (!total) {
|
|
94
|
+
total = { channelKey: r.channelKey, passed: 0, failed: 0, skipped: 0, total: 0 };
|
|
95
|
+
byChannel.set(r.channelKey, total);
|
|
96
|
+
}
|
|
97
|
+
total.total++;
|
|
98
|
+
if (r.skipped) {
|
|
99
|
+
total.skipped++;
|
|
100
|
+
}
|
|
101
|
+
else if (r.pass === true) {
|
|
102
|
+
total.passed++;
|
|
103
|
+
}
|
|
104
|
+
else if (r.pass === false) {
|
|
105
|
+
total.failed++;
|
|
106
|
+
}
|
|
107
|
+
else {
|
|
108
|
+
// A result with no pass verdict is treated as neither passed nor failed; it still counts
|
|
109
|
+
// toward the total so the aggregate is not silently distorted.
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
return [...byChannel.values()].sort((a, b) => a.channelKey.localeCompare(b.channelKey));
|
|
113
|
+
}
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
import type { AuthHealth, BillingChannel, WorkerAdapter, WorkerResult } from "../adapters/types.js";
|
|
2
|
+
import type { TickmarkrConfig } from "../config/config.js";
|
|
3
|
+
import { type Fixture } from "./fixtures.js";
|
|
4
|
+
export interface AcceptanceRunResult {
|
|
5
|
+
pass: boolean;
|
|
6
|
+
details: string;
|
|
7
|
+
}
|
|
8
|
+
export interface ChannelResult {
|
|
9
|
+
channel: BillingChannel;
|
|
10
|
+
channelKey: string;
|
|
11
|
+
skipped: boolean;
|
|
12
|
+
skipReason?: string;
|
|
13
|
+
repo?: string;
|
|
14
|
+
worker?: WorkerResult;
|
|
15
|
+
acceptance?: AcceptanceRunResult;
|
|
16
|
+
}
|
|
17
|
+
export interface DispatchOptions {
|
|
18
|
+
fixture: Fixture;
|
|
19
|
+
channels: BillingChannel[];
|
|
20
|
+
adapters: WorkerAdapter[];
|
|
21
|
+
health: Record<string, AuthHealth>;
|
|
22
|
+
cfg: TickmarkrConfig;
|
|
23
|
+
}
|
|
24
|
+
/**
|
|
25
|
+
* Render one prompt for a fixture and dispatch that identical, unmodified prompt to every channel
|
|
26
|
+
* under test. Each channel runs inside its own isolated seeded repository, and the fixture's own
|
|
27
|
+
* deterministic acceptance oracles are evaluated against each channel's resulting diff independently.
|
|
28
|
+
* Channels that fail install or model authentication are skipped with a recorded reason rather than
|
|
29
|
+
* dispatched. The dispatch path reuses the existing adapter `invoke` and `parse` contract.
|
|
30
|
+
*/
|
|
31
|
+
export declare function dispatchFixture(opts: DispatchOptions): Promise<ChannelResult[]>;
|