tickmarkr 1.68.0 → 1.70.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/dist/adapters/kimi.d.ts +25 -1
  2. package/dist/adapters/kimi.js +80 -0
  3. package/dist/adapters/types.d.ts +6 -0
  4. package/dist/cli/commands/eval.d.ts +4 -0
  5. package/dist/cli/commands/eval.js +26 -0
  6. package/dist/cli/commands/report.js +40 -5
  7. package/dist/cli/index.d.ts +1 -1
  8. package/dist/cli/index.js +3 -1
  9. package/dist/eval/canary.d.ts +46 -0
  10. package/dist/eval/canary.js +113 -0
  11. package/dist/eval/dispatch.d.ts +31 -0
  12. package/dist/eval/dispatch.js +207 -0
  13. package/dist/eval/fixtures.d.ts +22 -0
  14. package/dist/eval/fixtures.js +85 -0
  15. package/dist/eval/report.d.ts +35 -0
  16. package/dist/eval/report.js +82 -0
  17. package/dist/eval/selfcheck.d.ts +22 -0
  18. package/dist/eval/selfcheck.js +177 -0
  19. package/dist/gates/acceptance.d.ts +5 -1
  20. package/dist/gates/acceptance.js +72 -10
  21. package/dist/gates/review.d.ts +10 -2
  22. package/dist/gates/review.js +72 -19
  23. package/dist/report/bundle.d.ts +45 -0
  24. package/dist/report/bundle.js +158 -0
  25. package/dist/report/compare.d.ts +64 -0
  26. package/dist/report/compare.js +231 -0
  27. package/dist/run/daemon.js +152 -103
  28. package/dist/run/environment.d.ts +12 -0
  29. package/dist/run/environment.js +41 -0
  30. package/dist/run/interactive-seed.d.ts +15 -0
  31. package/dist/run/interactive-seed.js +25 -0
  32. package/fixtures/eval/canary/solution/a.txt +1 -0
  33. package/fixtures/eval/canary/spec.md +8 -0
  34. package/fixtures/eval/canary/start/a.txt +1 -0
  35. package/fixtures/eval/sample/solution/a.txt +1 -0
  36. package/fixtures/eval/sample/spec.md +8 -0
  37. package/fixtures/eval/sample/start/a.txt +1 -0
  38. package/fixtures/gsd-sample/07-live-check/07-01-PLAN.md +42 -0
  39. package/fixtures/gsd-sample/07-live-check/07-02-PLAN.md +21 -0
  40. package/fixtures/gsd-sample/07-live-check/07-03-PLAN.md +18 -0
  41. package/fixtures/gsd-sample/07-live-check/07-03-SUMMARY.md +1 -0
  42. package/fixtures/missing-mandatory-gate.native.md +10 -0
  43. package/fixtures/sample-pin.prd.md +18 -0
  44. package/fixtures/sample.native.md +35 -0
  45. package/fixtures/sample.prd.md +22 -0
  46. package/fixtures/speckit-sample/tasks.md +20 -0
  47. package/package.json +3 -2
@@ -1,6 +1,30 @@
1
- import { type WorkerAdapter, type WorkerResult } from "./types.js";
1
+ import type { ExecutorDriver, Slot } from "../drivers/types.js";
2
+ import { type Assignment, type WorkerAdapter, type WorkerResult } from "./types.js";
2
3
  export declare function kimiAuthed(credentialsText: string, nowMs: number): boolean;
3
4
  export declare function parseKimiModels(raw: string): string[];
4
5
  export declare function parseKimiResult(raw: string, nonce: string): WorkerResult;
5
6
  export declare function kimiSessionId(output: string): string | undefined;
7
+ export declare function kimiBannerModel(banner: string): string | undefined;
8
+ export declare function kimiBannerSessionId(banner: string): string | undefined;
9
+ export type KimiBannerConfirm = {
10
+ ok: true;
11
+ sessionId?: string;
12
+ } | {
13
+ ok: false;
14
+ error: string;
15
+ };
16
+ export declare function confirmKimiSeedBanner(banner: string, assignedModel: string): KimiBannerConfirm;
17
+ export interface KimiInteractiveSeedResult {
18
+ output: string;
19
+ seedFailed: boolean;
20
+ seedError?: string;
21
+ sessionId?: string;
22
+ }
23
+ export declare function runKimiInteractiveSeed(opts: {
24
+ driver: Pick<ExecutorDriver, "run" | "waitOutput" | "read">;
25
+ slot: Slot;
26
+ assignment: Assignment;
27
+ promptFile: string;
28
+ taskTimeoutMinutes: number;
29
+ }): Promise<KimiInteractiveSeedResult>;
6
30
  export declare const kimi: WorkerAdapter;
@@ -68,16 +68,90 @@ export function parseKimiResult(raw, nonce) {
68
68
  // line only: prompt/model prose can contain lookalike text, and the anchored charset keeps a
69
69
  // captured id shell-safe by construction (shq in resumeCommand is the second layer). Last valid
70
70
  // line wins — a run may echo stale resume lines mid-transcript.
71
+ //
72
+ // v1.69 T7: the native TUI cold-start banner also prints `Session: session_<uuid>` at launch.
73
+ // Prefer a banner Session line when present so seed-mode captures the id from the launch banner
74
+ // itself rather than waiting for a completion-time resume trailer (which the TUI may never emit).
71
75
  const RESUME_TRAILER_RE = /^\s*To resume this session: kimi -r (session_[0-9a-f-]+)\s*$/;
76
+ const BANNER_SESSION_LINE_RE = /^\s*Session:\s*(session_[0-9a-f-]+)\s*$/;
72
77
  export function kimiSessionId(output) {
73
78
  let id;
74
79
  for (const line of output.split("\n")) {
80
+ const banner = BANNER_SESSION_LINE_RE.exec(line);
81
+ if (banner) {
82
+ id = banner[1];
83
+ continue;
84
+ }
75
85
  const m = RESUME_TRAILER_RE.exec(line);
76
86
  if (m)
77
87
  id = m[1];
78
88
  }
79
89
  return id;
80
90
  }
91
+ // v1.69 T6: the native TUI takes -m <alias>, where config.toml aliases are the bare model suffix of
92
+ // the tickmarkr channel id (live probe 2026-07-22). Keep the mapping explicit and localized.
93
+ function kimiAlias(model) {
94
+ return model.replace(/^kimi-code\//, "");
95
+ }
96
+ // v1.69 T7: the cold-start banner prints the model alias and session id. Parse them from the
97
+ // banner text already captured for the readiness match — no new probe, no extra dispatch.
98
+ const BANNER_MODEL_RE = /^Model:\s*(.+)$/m;
99
+ const BANNER_SESSION_RE = /^Session:\s*(session_[0-9a-f-]+)$/m;
100
+ export function kimiBannerModel(banner) {
101
+ const m = BANNER_MODEL_RE.exec(banner);
102
+ if (!m)
103
+ return undefined;
104
+ const alias = m[1].trim();
105
+ if (!alias)
106
+ return undefined;
107
+ return `kimi-code/${alias}`;
108
+ }
109
+ export function kimiBannerSessionId(banner) {
110
+ return BANNER_SESSION_RE.exec(banner)?.[1];
111
+ }
112
+ // Fail closed on a named-model mismatch; a missing Model line is not a mismatch (banner partial).
113
+ export function confirmKimiSeedBanner(banner, assignedModel) {
114
+ const saw = kimiBannerModel(banner);
115
+ if (saw !== undefined && saw !== assignedModel) {
116
+ return { ok: false, error: `model mismatch: expected ${assignedModel}, saw ${saw}` };
117
+ }
118
+ return { ok: true, sessionId: kimiBannerSessionId(banner) };
119
+ }
120
+ // Shared launch-then-seed surface (T6) + banner confirm helpers (T7). One definition so the
121
+ // adapter property and runKimiInteractiveSeed cannot drift.
122
+ const KIMI_SEED = {
123
+ launch: (model) => `kimi -y -m ${shq(kimiAlias(model))}`,
124
+ readinessMatch: "Send /help for help information.",
125
+ seedLine: (promptFile) => `Read ${promptFile} and do exactly what it says.`,
126
+ };
127
+ // v1.69 T7: kimi-local launch-then-seed that confirms the banner model and captures the session id
128
+ // from the SAME pane text already present for the readiness match. No extra probe, no extra
129
+ // dispatch — one read of the launch banner, then seed or fail closed.
130
+ export async function runKimiInteractiveSeed(opts) {
131
+ await opts.driver.run(opts.slot, KIMI_SEED.launch(opts.assignment.model));
132
+ const ready = await opts.driver.waitOutput(opts.slot, KIMI_SEED.readinessMatch, opts.taskTimeoutMinutes * 60_000);
133
+ // Banner text already on the pane for the readiness match — the only text model/session use.
134
+ const banner = await opts.driver.read(opts.slot, 1000);
135
+ if (!ready) {
136
+ return { output: banner, seedFailed: true, seedError: `readiness pattern not seen: ${KIMI_SEED.readinessMatch}` };
137
+ }
138
+ const confirm = confirmKimiSeedBanner(banner, opts.assignment.model);
139
+ if (!confirm.ok) {
140
+ return { output: banner, seedFailed: true, seedError: confirm.error };
141
+ }
142
+ const seedText = KIMI_SEED.seedLine(opts.promptFile);
143
+ await opts.driver.run(opts.slot, seedText);
144
+ let output = "";
145
+ for (let attempt = 0; attempt < 5; attempt++) {
146
+ await new Promise((r) => setTimeout(r, 200));
147
+ output = await opts.driver.read(opts.slot, 1000);
148
+ const bottom = output.trimEnd().split("\n").pop() ?? "";
149
+ if (!bottom.includes(seedText)) {
150
+ return { output, seedFailed: false, sessionId: confirm.sessionId };
151
+ }
152
+ }
153
+ return { output, seedFailed: true, seedError: "seed line never left the input box", sessionId: confirm.sessionId };
154
+ }
81
155
  export const kimi = {
82
156
  id: "kimi",
83
157
  vendor: "moonshot",
@@ -108,6 +182,12 @@ export const kimi = {
108
182
  // ("unknown command '…'", live-verified 2026-07-17, OBS-67) and -p is non-interactive-only.
109
183
  // null → the daemon's print fallback (types.ts:101) keeps kimi workers visible without a TUI.
110
184
  interactiveCommand: () => null,
185
+ // v1.69 T6/T7: launch the real TUI, wait for the cold-start readiness marker, then submit the
186
+ // prompt as one user turn. Banner model/session confirmation for seed mode lives in
187
+ // runKimiInteractiveSeed / confirmKimiSeedBanner (same pane text as the readiness match — no
188
+ // extra probe). Live-probed 2026-07-22 (kimi 0.28.1): `kimi -y` opens the TUI with yolo active
189
+ // and prints the model/session lines plus "Send /help for help information." before the input box.
190
+ interactiveSeed: KIMI_SEED,
111
191
  // v1.53 T3 resume — live-probed 2026-07-18: `-p` + `-S <id>` compose cleanly (no OBS-67-class
112
192
  // flag rejection) and the resumed session carries prior conversation state. `-S <id>` is the
113
193
  // deterministic form; `-c` rejected as primary — cwd-keyed, nondeterministic under worktree
@@ -48,6 +48,11 @@ export interface WorkerResult {
48
48
  deviations: string[];
49
49
  raw: string;
50
50
  }
51
+ export interface InteractiveSeed {
52
+ launch(model: string): string;
53
+ readinessMatch: string;
54
+ seedLine(promptFile: string): string;
55
+ }
51
56
  export interface ContextUsage {
52
57
  tokens: number;
53
58
  limit?: number;
@@ -78,6 +83,7 @@ export interface WorkerAdapter {
78
83
  channels(cfg: TickmarkrConfig): BillingChannel[];
79
84
  headlessCommand(promptFile: string, model: string): string;
80
85
  interactiveCommand(promptFile: string, model: string): string | null;
86
+ interactiveSeed?: InteractiveSeed;
81
87
  resumeCommand?(sessionId: string, promptFile: string, model: string): string;
82
88
  sessionIdFrom?(output: string): string | undefined;
83
89
  resumeUnknownContext?: boolean;
@@ -0,0 +1,4 @@
1
+ export declare function evalCommand(argv: string[], cwd?: string): Promise<string | {
2
+ out: string;
3
+ code: number;
4
+ }>;
@@ -0,0 +1,26 @@
1
+ import { parseArgs } from "node:util";
2
+ import { discoverFixtures, resolveFixturesRoot, seedFixture } from "../../eval/fixtures.js";
3
+ export async function evalCommand(argv, cwd = process.cwd()) {
4
+ const { positionals } = parseArgs({ args: argv, allowPositionals: true });
5
+ const root = resolveFixturesRoot(positionals[0], cwd);
6
+ const { valid, invalid } = discoverFixtures(root);
7
+ const lines = [`tickmarkr eval — discovered ${valid.length} fixture${valid.length === 1 ? "" : "s"}`];
8
+ for (const f of valid)
9
+ lines.push(` ${f.id}`);
10
+ if (invalid.length) {
11
+ lines.push("", "invalid fixtures:");
12
+ for (const i of invalid)
13
+ lines.push(` ${i.id} — ${i.reason}`);
14
+ }
15
+ for (const f of valid) {
16
+ const seeded = await seedFixture(f);
17
+ try {
18
+ lines.push(` seeded ${f.id}`);
19
+ }
20
+ finally {
21
+ await seeded.cleanup();
22
+ }
23
+ }
24
+ const out = lines.join("\n");
25
+ return invalid.length ? { out, code: 1 } : out;
26
+ }
@@ -1,8 +1,11 @@
1
+ import { writeFileSync } from "node:fs";
1
2
  import { parseArgs } from "node:util";
2
3
  import { ttyVisual } from "../../adapters/model-lints.js";
3
4
  import { addUsage } from "../../adapters/types.js";
4
5
  import { dim, rule, title } from "../../brand.js";
5
6
  import { loadConfig } from "../../config/config.js";
7
+ import { buildProofBundle } from "../../report/bundle.js";
8
+ import { compareRuns } from "../../report/compare.js";
6
9
  import { estimateCosts } from "../../report/cost.js";
7
10
  import { cellsOf, cellSummary } from "../../route/profile.js";
8
11
  import { Journal, loadRoutingProfile } from "../../run/journal.js";
@@ -316,17 +319,49 @@ const stylizeReport = (out) => {
316
319
  export async function report(argv, cwd = process.cwd()) {
317
320
  const { values, positionals } = parseArgs({
318
321
  args: argv,
319
- options: { md: { type: "boolean" } },
322
+ options: {
323
+ md: { type: "boolean" },
324
+ // v1.70 T3: baseline run id for cost/gate/duration delta + environment comparability guard
325
+ compare: { type: "string" },
326
+ // v1.70 T4: write a portable, schema-versioned proof packet (local file only — no network)
327
+ bundle: { type: "string" },
328
+ },
320
329
  allowPositionals: true,
321
330
  });
322
331
  const runId = positionals[0] ?? Journal.latestRunId(cwd, { withJournal: true });
323
332
  if (!runId)
324
- throw new Error("no runs found — usage: tickmarkr report <run-id> [--md]");
333
+ throw new Error("no runs found — usage: tickmarkr report <run-id> [--md] [--compare <baseline-run-id>] [--bundle <path>]");
325
334
  const j = Journal.open(cwd, runId);
326
335
  const events = j.read();
336
+ const rows = j.readTelemetry();
337
+ const cfg = loadConfig(cwd);
338
+ let bundleNote = "";
339
+ if (values.bundle) {
340
+ // Local-only write of the pure proof packet — no network path exists in buildProofBundle.
341
+ const packet = buildProofBundle(runId, events);
342
+ writeFileSync(values.bundle, JSON.stringify(packet, null, 2) + "\n");
343
+ bundleNote = `wrote proof bundle → ${values.bundle}\n`;
344
+ }
345
+ let comparison = "";
346
+ if (values.compare) {
347
+ const baselineRunId = values.compare;
348
+ const baseline = Journal.open(cwd, baselineRunId);
349
+ const outcome = compareRuns({
350
+ runId,
351
+ baselineRunId,
352
+ events,
353
+ baselineEvents: baseline.read(),
354
+ rows,
355
+ baselineRows: baseline.readTelemetry(),
356
+ cost: cfg.cost,
357
+ });
358
+ // Fail closed: missing run-start yields a clear reason, never a partial table.
359
+ if (!outcome.ok)
360
+ throw new Error(outcome.reason);
361
+ comparison = "\n" + outcome.text;
362
+ }
327
363
  if (values.md) {
328
- const rows = j.readTelemetry();
329
- return renderMarkdownRecord(runId, events, estimateCosts(rows, loadConfig(cwd).cost), rows);
364
+ return bundleNote + renderMarkdownRecord(runId, events, estimateCosts(rows, cfg.cost), rows) + comparison;
330
365
  }
331
- return stylizeReport(textReport(runId, events, j.readTelemetry(), cwd));
366
+ return bundleNote + stylizeReport(textReport(runId, events, rows, cwd)) + comparison;
332
367
  }
@@ -5,7 +5,7 @@ export type CommandResult = string | {
5
5
  };
6
6
  export type CommandMap = Record<string, (argv: string[]) => Promise<CommandResult>>;
7
7
  export declare const COMMANDS: CommandMap;
8
- export declare const USAGE = "tickmarkr \u2014 spec-driven orchestration harness for AI coding agents\nusage: tickmarkr <command>\n init guided setup + doctor; init --agent [--force] [--docs] adds agent skills/docs\n doctor re-probe adapters, herdr, auth; print capability matrix\n fleet interactive fleet editor (fleet --print for CI drift checks)\n compile <src> spec \u2192 .tickmarkr/graph.json (fails without acceptance criteria)\n scope <intent> draft a compiled native spec beside an answered intent (--force to overwrite)\n plan dry-run routing table + cost estimate + floor lints\n run execute the graph (--concurrency N --driver herdr|subprocess --route-strict)\n status live run state\n resume <id> continue a run from its journal\n report <id> cost/quality report (--md for committable execution record)\n profile show learned routing profile (profile reset = forget history via cursor, keeps telemetry)\n ui open the Fleet Studio TUI (full-screen tabbed cockpit)\n unlock remove a stale/garbage run lock (refuses if the holder is alive)\n approve <id> <task> approve a parked human gate (--by <name> --reason <text>); takes effect on resume";
8
+ export declare const USAGE = "tickmarkr \u2014 spec-driven orchestration harness for AI coding agents\nusage: tickmarkr <command>\n init guided setup + doctor; init --agent [--force] [--docs] adds agent skills/docs\n doctor re-probe adapters, herdr, auth; print capability matrix\n fleet interactive fleet editor (fleet --print for CI drift checks)\n compile <src> spec \u2192 .tickmarkr/graph.json (fails without acceptance criteria)\n scope <intent> draft a compiled native spec beside an answered intent (--force to overwrite)\n plan dry-run routing table + cost estimate + floor lints\n eval run checked-in fixtures against every channel in isolated temp repos\n run execute the graph (--concurrency N --driver herdr|subprocess --route-strict)\n status live run state\n resume <id> continue a run from its journal\n report <id> cost/quality report (--md for committable execution record)\n profile show learned routing profile (profile reset = forget history via cursor, keeps telemetry)\n ui open the Fleet Studio TUI (full-screen tabbed cockpit)\n unlock remove a stale/garbage run lock (refuses if the holder is alive)\n approve <id> <task> approve a parked human gate (--by <name> --reason <text>); takes effect on resume";
9
9
  export declare function dispatch(cmd: string | undefined, argv: string[], commands?: CommandMap): Promise<{
10
10
  out: string;
11
11
  code: number;
package/dist/cli/index.js CHANGED
@@ -4,6 +4,7 @@ import { pathToFileURL } from "node:url";
4
4
  import { approve } from "./commands/approve.js";
5
5
  import { compile } from "./commands/compile.js";
6
6
  import { doctor } from "./commands/doctor.js";
7
+ import { evalCommand } from "./commands/eval.js";
7
8
  import { fleet } from "./commands/fleet.js";
8
9
  import { init } from "./commands/init.js";
9
10
  import { plan } from "./commands/plan.js";
@@ -18,7 +19,7 @@ import { unlock } from "./commands/unlock.js";
18
19
  import { version } from "./commands/version.js";
19
20
  const normalize = (r) => typeof r === "string" ? { out: r, code: 0 } : r;
20
21
  export const COMMANDS = {
21
- init, doctor, fleet, compile, scope, plan, run, status, resume, report, profile, ui, unlock, approve, version,
22
+ init, doctor, fleet, compile, scope, plan, run, status, resume, report, profile, ui, unlock, approve, version, eval: evalCommand,
22
23
  };
23
24
  const VERSION_FLAGS = new Set(["version", "--version", "-v"]);
24
25
  const HELP_CMDS = new Set(["help", "-h", "--help"]);
@@ -31,6 +32,7 @@ usage: tickmarkr <command>
31
32
  compile <src> spec → .tickmarkr/graph.json (fails without acceptance criteria)
32
33
  scope <intent> draft a compiled native spec beside an answered intent (--force to overwrite)
33
34
  plan dry-run routing table + cost estimate + floor lints
35
+ eval run checked-in fixtures against every channel in isolated temp repos
34
36
  run execute the graph (--concurrency N --driver herdr|subprocess --route-strict)
35
37
  status live run state
36
38
  resume <id> continue a run from its journal
@@ -0,0 +1,46 @@
1
+ import type { WorkerAdapter } from "../adapters/types.js";
2
+ import { type Fixture } from "./fixtures.js";
3
+ export declare const CANARY_FIXTURE_ID = "canary";
4
+ export interface CanaryJudgeResult {
5
+ fixtureId: string;
6
+ expectedPass: boolean;
7
+ judgePass: boolean;
8
+ breach: boolean;
9
+ details: string;
10
+ }
11
+ export interface FixtureChannelResult {
12
+ fixtureId: string;
13
+ channelKey: string;
14
+ skipped: boolean;
15
+ pass?: boolean;
16
+ }
17
+ export interface ChannelQualificationTotal {
18
+ channelKey: string;
19
+ passed: number;
20
+ failed: number;
21
+ skipped: number;
22
+ total: number;
23
+ }
24
+ export declare function isCanaryFixture(fixture: Fixture): boolean;
25
+ export declare function isCanaryResult(result: FixtureChannelResult): boolean;
26
+ export declare function resolveCanaryFixture(root: string): Fixture | undefined;
27
+ /**
28
+ * Return a fixture list that always includes the held-out canary fixture.
29
+ * If the canary is already present, the list is returned unchanged.
30
+ */
31
+ export declare function ensureCanary(fixtures: Fixture[], root: string): Fixture[];
32
+ /**
33
+ * Run the held-out canary fixture against a judge adapter.
34
+ * The fixture is seeded, a deliberately failing change is committed, and the judge oracle
35
+ * is evaluated against the resulting diff. A correct judge returns pass=false; any pass=true
36
+ * verdict is flagged as a judge-integrity breach.
37
+ */
38
+ export declare function runCanaryJudge(fixture: Fixture, judgeAdapter: WorkerAdapter, model: string): Promise<CanaryJudgeResult>;
39
+ /**
40
+ * Aggregate per-channel qualification totals from fixture-channel results.
41
+ * Results marked as canary (by the supplied predicate, defaulting to isCanaryResult) are excluded
42
+ * from every total so the canary never contributes to a channel's qualification score.
43
+ */
44
+ export declare function aggregateChannelTotals(results: FixtureChannelResult[], options?: {
45
+ isCanary?: (r: FixtureChannelResult) => boolean;
46
+ }): ChannelQualificationTotal[];
@@ -0,0 +1,113 @@
1
+ import { existsSync, lstatSync, writeFileSync } from "node:fs";
2
+ import { join } from "node:path";
3
+ import { compileSource } from "../compile/index.js";
4
+ import { acceptanceGate } from "../gates/acceptance.js";
5
+ import { shGitOk } from "../run/git.js";
6
+ import { seedFixture } from "./fixtures.js";
7
+ export const CANARY_FIXTURE_ID = "canary";
8
+ export function isCanaryFixture(fixture) {
9
+ return fixture.id === CANARY_FIXTURE_ID;
10
+ }
11
+ export function isCanaryResult(result) {
12
+ return result.fixtureId === CANARY_FIXTURE_ID;
13
+ }
14
+ export function resolveCanaryFixture(root) {
15
+ const path = join(root, CANARY_FIXTURE_ID);
16
+ if (!existsSync(path) || !lstatSync(path).isDirectory())
17
+ return undefined;
18
+ const startDir = join(path, "start");
19
+ const solutionDir = join(path, "solution");
20
+ if (!existsSync(startDir) || !lstatSync(startDir).isDirectory())
21
+ return undefined;
22
+ if (!existsSync(solutionDir) || !lstatSync(solutionDir).isDirectory())
23
+ return undefined;
24
+ return { id: CANARY_FIXTURE_ID, path, startDir, solutionDir };
25
+ }
26
+ /**
27
+ * Return a fixture list that always includes the held-out canary fixture.
28
+ * If the canary is already present, the list is returned unchanged.
29
+ */
30
+ export function ensureCanary(fixtures, root) {
31
+ if (fixtures.some(isCanaryFixture))
32
+ return fixtures;
33
+ const canary = resolveCanaryFixture(root);
34
+ return canary ? [canary, ...fixtures] : fixtures;
35
+ }
36
+ function compileCanaryTask(fixture) {
37
+ const specPath = join(fixture.path, "spec.md");
38
+ const graph = compileSource(specPath);
39
+ if (graph.tasks.length !== 1) {
40
+ throw new Error(`canary spec must contain exactly one task (found ${graph.tasks.length})`);
41
+ }
42
+ return graph.tasks[0];
43
+ }
44
+ // Apply a known-bad change so the judged diff contains quotable evidence of an unmet criterion.
45
+ async function applyKnownBadChange(repo) {
46
+ writeFileSync(join(repo, "a.txt"), "canary-wrong");
47
+ await shGitOk("git add -A", repo);
48
+ await shGitOk("git commit -m 'canary bad change' --no-gpg-sign", repo);
49
+ }
50
+ /**
51
+ * Run the held-out canary fixture against a judge adapter.
52
+ * The fixture is seeded, a deliberately failing change is committed, and the judge oracle
53
+ * is evaluated against the resulting diff. A correct judge returns pass=false; any pass=true
54
+ * verdict is flagged as a judge-integrity breach.
55
+ */
56
+ export async function runCanaryJudge(fixture, judgeAdapter, model) {
57
+ if (!isCanaryFixture(fixture)) {
58
+ throw new Error(`fixture ${fixture.id} is not the canary fixture`);
59
+ }
60
+ const task = compileCanaryTask(fixture);
61
+ const seeded = await seedFixture(fixture);
62
+ const initialCommit = (await shGitOk("git rev-parse HEAD", seeded.repo)).trim();
63
+ try {
64
+ await applyKnownBadChange(seeded.repo);
65
+ const gateResult = await acceptanceGate(task, seeded.repo, initialCommit, { adapter: judgeAdapter, model });
66
+ const expectedPass = false;
67
+ const judgePass = gateResult.pass;
68
+ const breach = judgePass === true && expectedPass === false;
69
+ return {
70
+ fixtureId: fixture.id,
71
+ expectedPass,
72
+ judgePass,
73
+ breach,
74
+ details: gateResult.details,
75
+ };
76
+ }
77
+ finally {
78
+ await seeded.cleanup();
79
+ }
80
+ }
81
+ /**
82
+ * Aggregate per-channel qualification totals from fixture-channel results.
83
+ * Results marked as canary (by the supplied predicate, defaulting to isCanaryResult) are excluded
84
+ * from every total so the canary never contributes to a channel's qualification score.
85
+ */
86
+ export function aggregateChannelTotals(results, options = {}) {
87
+ const isCanary = options.isCanary ?? isCanaryResult;
88
+ const byChannel = new Map();
89
+ for (const r of results) {
90
+ if (isCanary(r))
91
+ continue;
92
+ let total = byChannel.get(r.channelKey);
93
+ if (!total) {
94
+ total = { channelKey: r.channelKey, passed: 0, failed: 0, skipped: 0, total: 0 };
95
+ byChannel.set(r.channelKey, total);
96
+ }
97
+ total.total++;
98
+ if (r.skipped) {
99
+ total.skipped++;
100
+ }
101
+ else if (r.pass === true) {
102
+ total.passed++;
103
+ }
104
+ else if (r.pass === false) {
105
+ total.failed++;
106
+ }
107
+ else {
108
+ // A result with no pass verdict is treated as neither passed nor failed; it still counts
109
+ // toward the total so the aggregate is not silently distorted.
110
+ }
111
+ }
112
+ return [...byChannel.values()].sort((a, b) => a.channelKey.localeCompare(b.channelKey));
113
+ }
@@ -0,0 +1,31 @@
1
+ import type { AuthHealth, BillingChannel, WorkerAdapter, WorkerResult } from "../adapters/types.js";
2
+ import type { TickmarkrConfig } from "../config/config.js";
3
+ import { type Fixture } from "./fixtures.js";
4
+ export interface AcceptanceRunResult {
5
+ pass: boolean;
6
+ details: string;
7
+ }
8
+ export interface ChannelResult {
9
+ channel: BillingChannel;
10
+ channelKey: string;
11
+ skipped: boolean;
12
+ skipReason?: string;
13
+ repo?: string;
14
+ worker?: WorkerResult;
15
+ acceptance?: AcceptanceRunResult;
16
+ }
17
+ export interface DispatchOptions {
18
+ fixture: Fixture;
19
+ channels: BillingChannel[];
20
+ adapters: WorkerAdapter[];
21
+ health: Record<string, AuthHealth>;
22
+ cfg: TickmarkrConfig;
23
+ }
24
+ /**
25
+ * Render one prompt for a fixture and dispatch that identical, unmodified prompt to every channel
26
+ * under test. Each channel runs inside its own isolated seeded repository, and the fixture's own
27
+ * deterministic acceptance oracles are evaluated against each channel's resulting diff independently.
28
+ * Channels that fail install or model authentication are skipped with a recorded reason rather than
29
+ * dispatched. The dispatch path reuses the existing adapter `invoke` and `parse` contract.
30
+ */
31
+ export declare function dispatchFixture(opts: DispatchOptions): Promise<ChannelResult[]>;