@mrace07/kairo 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. package/README.md +141 -0
  2. package/dist/application/coding-agent.d.ts +85 -0
  3. package/dist/application/coding-agent.js +765 -0
  4. package/dist/application/context-manager.d.ts +22 -0
  5. package/dist/application/context-manager.js +174 -0
  6. package/dist/application/context-selector.d.ts +11 -0
  7. package/dist/application/context-selector.js +74 -0
  8. package/dist/application/evaluated-agent.d.ts +7 -0
  9. package/dist/application/evaluated-agent.js +16 -0
  10. package/dist/application/evaluation-comparison.d.ts +34 -0
  11. package/dist/application/evaluation-comparison.js +91 -0
  12. package/dist/application/evaluation-harness.d.ts +19 -0
  13. package/dist/application/evaluation-harness.js +217 -0
  14. package/dist/application/failure-analyzer.d.ts +5 -0
  15. package/dist/application/failure-analyzer.js +37 -0
  16. package/dist/application/interaction-routing.d.ts +9 -0
  17. package/dist/application/interaction-routing.js +20 -0
  18. package/dist/application/live-evaluation.d.ts +8 -0
  19. package/dist/application/live-evaluation.js +185 -0
  20. package/dist/application/model-routing.d.ts +12 -0
  21. package/dist/application/model-routing.js +40 -0
  22. package/dist/application/model-system-instruction.d.ts +4 -0
  23. package/dist/application/model-system-instruction.js +4 -0
  24. package/dist/application/self-evaluation.d.ts +26 -0
  25. package/dist/application/self-evaluation.js +394 -0
  26. package/dist/application/task-metrics.d.ts +31 -0
  27. package/dist/application/task-metrics.js +42 -0
  28. package/dist/application/verification-planner.d.ts +12 -0
  29. package/dist/application/verification-planner.js +97 -0
  30. package/dist/domain/models.d.ts +247 -0
  31. package/dist/domain/models.js +1 -0
  32. package/dist/domain/ports.d.ts +87 -0
  33. package/dist/domain/ports.js +1 -0
  34. package/dist/domain/provider-error.d.ts +18 -0
  35. package/dist/domain/provider-error.js +17 -0
  36. package/dist/infrastructure/configuration/config.d.ts +24 -0
  37. package/dist/infrastructure/configuration/config.js +79 -0
  38. package/dist/infrastructure/filesystem/platform-paths.d.ts +8 -0
  39. package/dist/infrastructure/filesystem/platform-paths.js +18 -0
  40. package/dist/infrastructure/persistence/sqlite-session-store.d.ts +82 -0
  41. package/dist/infrastructure/persistence/sqlite-session-store.js +447 -0
  42. package/dist/infrastructure/providers/gemini-provider.d.ts +14 -0
  43. package/dist/infrastructure/providers/gemini-provider.js +90 -0
  44. package/dist/infrastructure/providers/groq-provider.d.ts +16 -0
  45. package/dist/infrastructure/providers/groq-provider.js +101 -0
  46. package/dist/infrastructure/providers/jev-safety-advisor.d.ts +18 -0
  47. package/dist/infrastructure/providers/jev-safety-advisor.js +95 -0
  48. package/dist/infrastructure/providers/mistral-provider.d.ts +15 -0
  49. package/dist/infrastructure/providers/mistral-provider.js +137 -0
  50. package/dist/infrastructure/providers/openrouter-provider.d.ts +15 -0
  51. package/dist/infrastructure/providers/openrouter-provider.js +104 -0
  52. package/dist/infrastructure/providers/provider-recovery.d.ts +10 -0
  53. package/dist/infrastructure/providers/provider-recovery.js +108 -0
  54. package/dist/infrastructure/providers/provider-registry.d.ts +22 -0
  55. package/dist/infrastructure/providers/provider-registry.js +67 -0
  56. package/dist/infrastructure/repository/repository-awareness.d.ts +12 -0
  57. package/dist/infrastructure/repository/repository-awareness.js +25 -0
  58. package/dist/infrastructure/repository/repository-profiler.d.ts +35 -0
  59. package/dist/infrastructure/repository/repository-profiler.js +498 -0
  60. package/dist/infrastructure/security/macos-keychain-store.d.ts +17 -0
  61. package/dist/infrastructure/security/macos-keychain-store.js +73 -0
  62. package/dist/infrastructure/tools/workspace-tools.d.ts +30 -0
  63. package/dist/infrastructure/tools/workspace-tools.js +321 -0
  64. package/dist/interface/cli/evaluation-comparison-report.d.ts +6 -0
  65. package/dist/interface/cli/evaluation-comparison-report.js +46 -0
  66. package/dist/interface/cli/evaluation-report.d.ts +14 -0
  67. package/dist/interface/cli/evaluation-report.js +122 -0
  68. package/dist/interface/cli/index.d.ts +2 -0
  69. package/dist/interface/cli/index.js +238 -0
  70. package/dist/interface/cli/provider-setup.d.ts +16 -0
  71. package/dist/interface/cli/provider-setup.js +86 -0
  72. package/dist/interface/cli/repl.d.ts +7 -0
  73. package/dist/interface/cli/repl.js +19 -0
  74. package/dist/interface/cli/task-trace.d.ts +7 -0
  75. package/dist/interface/cli/task-trace.js +48 -0
  76. package/dist/interface/cli/tui.d.ts +147 -0
  77. package/dist/interface/cli/tui.js +910 -0
  78. package/package.json +61 -0
@@ -0,0 +1,394 @@
1
+ import { spawn } from "node:child_process";
2
+ import { cp, mkdtemp, readFile, rm, symlink, writeFile } from "node:fs/promises";
3
+ import { tmpdir } from "node:os";
4
+ import { basename, join } from "node:path";
5
+ import { pathToFileURL } from "node:url";
6
+ import { SqliteSessionStore } from "../infrastructure/persistence/sqlite-session-store.js";
7
+ import { createProvider } from "../infrastructure/providers/provider-registry.js";
8
+ import { RepositoryProfiler } from "../infrastructure/repository/repository-profiler.js";
9
+ import { WorkspaceTools, definitions } from "../infrastructure/tools/workspace-tools.js";
10
+ import { CodingAgent } from "./coding-agent.js";
11
+ import { taskMetrics } from "./task-metrics.js";
12
+ import { runEvaluatedAgent } from "./evaluated-agent.js";
13
+ const replacement = async (workspace, path, before, after) => {
14
+ const target = join(workspace, path);
15
+ const source = await readFile(target, "utf8");
16
+ if (!source.includes(before))
17
+ throw new Error(`Self-eval seed no longer matches ${path}.`);
18
+ await writeFile(target, source.replace(before, after), "utf8");
19
+ };
20
+ /** Seven real Kairo safeguards, deliberately removed from isolated source copies. */
21
+ export const selfEvaluationScenarios = [
22
+ {
23
+ id: "verification-check-script",
24
+ prompt: "A JavaScript project with only a `check` package script is not offered a typecheck verification command. Fix verification discovery so the `check` script is recognized as typecheck, preserve package-manager command formats, add or update a focused test, and run relevant verification.",
25
+ seed: (workspace) => replacement(workspace, "src/application/verification-planner.ts", '["typecheck", ["typecheck", "type-check", "check"]]', '["typecheck", ["typecheck", "type-check"]]'),
26
+ async verify(workspace) {
27
+ const module = await import(pathToFileURL(join(workspace, "dist/application/verification-planner.js")).href);
28
+ const candidates = new module.VerificationPlanner().candidates({
29
+ packageManager: "pnpm",
30
+ scripts: { check: "tsc --noEmit" },
31
+ });
32
+ if (candidates[0]?.label !== "typecheck" || candidates[0]?.command !== "pnpm check")
33
+ throw new Error("A pnpm check script was not exposed as typecheck.");
34
+ },
35
+ },
36
+ {
37
+ id: "focused-verification-selection",
38
+ prompt: "Kairo no longer recommends the focused typecheck for a changed TypeScript source file. Restore changed-file-aware verification selection, add or update a focused test, and run relevant verification.",
39
+ seed: (workspace) => replacement(workspace, "src/application/verification-planner.ts", 'if (isSource && byLabel("typecheck"))', "if (false)"),
40
+ async verify(workspace) {
41
+ const module = await import(pathToFileURL(join(workspace, "dist/application/verification-planner.js")).href);
42
+ const selection = new module.VerificationPlanner().select({
43
+ sourceRoots: ["src"],
44
+ testRoots: ["test"],
45
+ configFiles: [],
46
+ verificationCandidates: [
47
+ { label: "test", command: "pnpm test" },
48
+ { label: "typecheck", command: "pnpm check" },
49
+ ],
50
+ }, ["src/login.ts"]);
51
+ if (selection?.command !== "pnpm check" || selection.scope !== "broad")
52
+ throw new Error("A changed source file did not select focused typechecking.");
53
+ },
54
+ },
55
+ {
56
+ id: "bun-lockfile-discovery",
57
+ prompt: "Repository profiling fails to recognize projects using Bun's current bun.lock file. Restore support without regressing bun.lockb, add or update a focused test, and run relevant verification.",
58
+ seed: (workspace) => replacement(workspace, "src/infrastructure/repository/repository-profiler.ts", 'paths.has("bun.lockb") || paths.has("bun.lock")', 'paths.has("bun.lockb")'),
59
+ async verify(workspace) {
60
+ const fixture = await mkdtemp(join(tmpdir(), "kairo-bun-grader-"));
61
+ try {
62
+ await Promise.all([
63
+ writeFile(join(fixture, "package.json"), '{"name":"bun-fixture"}\n'),
64
+ writeFile(join(fixture, "bun.lock"), "# bun lockfile\n"),
65
+ ]);
66
+ const module = await import(pathToFileURL(join(workspace, "dist/infrastructure/repository/repository-profiler.js"))
67
+ .href);
68
+ if ((await new module.RepositoryProfiler().profile(fixture)).packageManager !== "bun")
69
+ throw new Error("A project with bun.lock was not detected as Bun.");
70
+ }
71
+ finally {
72
+ await rm(fixture, { recursive: true, force: true });
73
+ }
74
+ },
75
+ },
76
+ {
77
+ id: "context-relationship-ranking",
78
+ prompt: "Repository context ranking no longer includes a file connected to an independently relevant file. Restore relationship-aware ranking without replacing direct lexical ranking, add or update a focused test, and run relevant verification.",
79
+ seed: (workspace) => replacement(workspace, "src/application/context-selector.ts", "const relationshipScore = file.relatedFiles.some((path) => direct.has(path)) ? 3 : 0;", "const relationshipScore = 0;"),
80
+ async verify(workspace) {
81
+ const module = await import(pathToFileURL(join(workspace, "dist/application/context-selector.js")).href);
82
+ const selected = new module.ContextSelector().select("token", {
83
+ root: workspace,
84
+ packageManager: "pnpm",
85
+ scripts: {},
86
+ configFiles: [],
87
+ sourceRoots: [],
88
+ testRoots: [],
89
+ ignoredPaths: [],
90
+ indexedFiles: ["src/entry.ts", "src/token.ts", "src/other.ts"],
91
+ files: [
92
+ {
93
+ path: "src/entry.ts",
94
+ terms: [],
95
+ symbols: [],
96
+ imports: [],
97
+ relatedFiles: ["src/token.ts"],
98
+ },
99
+ {
100
+ path: "src/token.ts",
101
+ terms: ["token"],
102
+ symbols: ["token"],
103
+ imports: [],
104
+ relatedFiles: [],
105
+ },
106
+ { path: "src/other.ts", terms: [], symbols: [], imports: [], relatedFiles: [] },
107
+ ],
108
+ verificationCandidates: [],
109
+ createdAt: 0,
110
+ });
111
+ if (!selected.includes("src/entry.ts"))
112
+ throw new Error("A related file was not ranked.");
113
+ },
114
+ },
115
+ {
116
+ id: "verifying-task-recovery",
117
+ prompt: "After a restart, tasks left in the verifying state are not recovered as interrupted. Restore durable restart recovery for that state, add or update a focused test, and run relevant verification.",
118
+ async seed(workspace) {
119
+ await replacement(workspace, "src/infrastructure/persistence/sqlite-session-store.ts", "status IN ('planning', 'acting', 'verifying')", "status IN ('planning', 'acting')");
120
+ await replacement(workspace, "src/infrastructure/persistence/sqlite-session-store.ts", "status IN ('planning', 'acting', 'verifying')", "status IN ('planning', 'acting')");
121
+ },
122
+ async verify(workspace) {
123
+ const directory = await mkdtemp(join(tmpdir(), "kairo-recovery-grader-"));
124
+ const fixture = join(directory, "state.sqlite");
125
+ const module = await import(pathToFileURL(join(workspace, "dist/infrastructure/persistence/sqlite-session-store.js"))
126
+ .href);
127
+ const first = await module.SqliteSessionStore.open(fixture);
128
+ const session = first.create(workspace);
129
+ const task = first.startTask(session.id, "Recover this task");
130
+ first.updateTask(task.id, { status: "verifying" });
131
+ first.close();
132
+ const resumed = await module.SqliteSessionStore.open(fixture);
133
+ try {
134
+ if (resumed.task(task.id)?.status !== "interrupted")
135
+ throw new Error("A verifying task was not interrupted on restart.");
136
+ }
137
+ finally {
138
+ resumed.close();
139
+ await rm(directory, { recursive: true, force: true });
140
+ }
141
+ },
142
+ },
143
+ {
144
+ id: "symlink-escape-read",
145
+ prompt: "The workspace read tool accepts a symlink inside the workspace that points outside it. Restore symlink-escape protection for reads without weakening normal workspace access, add or update a focused test, and run relevant verification.",
146
+ seed: (workspace) => replacement(workspace, "src/infrastructure/tools/workspace-tools.ts", ' const actual = await realpath(candidate);\n // Resolving first catches a path that looks local but exits through a symlink.\n if (!this.inside(actual)) throw new Error("Symlink escapes the workspace.");\n return actual;', " return candidate;"),
147
+ async verify(workspace) {
148
+ const root = await mkdtemp(join(tmpdir(), "kairo-workspace-grader-"));
149
+ const outside = await mkdtemp(join(tmpdir(), "kairo-outside-grader-"));
150
+ try {
151
+ await writeFile(join(outside, "secret.txt"), "not for the workspace\n");
152
+ await symlink(outside, join(root, "escape"));
153
+ const module = await import(pathToFileURL(join(workspace, "dist/infrastructure/tools/workspace-tools.js")).href);
154
+ const result = await (await module.WorkspaceTools.create(root)).execute({ id: "read", name: "read_file", args: { path: "escape/secret.txt" } });
155
+ if (result.ok || !result.output.includes("Symlink escapes the workspace."))
156
+ throw new Error("A read through an escaping symlink was accepted.");
157
+ }
158
+ finally {
159
+ await Promise.all([
160
+ rm(root, { recursive: true, force: true }),
161
+ rm(outside, { recursive: true, force: true }),
162
+ ]);
163
+ }
164
+ },
165
+ },
166
+ {
167
+ id: "repair-brief-evidence",
168
+ prompt: "The focused retry context omits extracted failure excerpts from the repair brief. Restore that evidence while preserving the summary and locations, add or update a focused test, and run relevant verification.",
169
+ seed: (workspace) => replacement(workspace, "src/application/context-manager.ts", ' `Evidence: ${latest.evidence.excerpts.join(" | ") || "inspect the command output"}`,\n', ""),
170
+ async verify(workspace) {
171
+ const store = await SqliteSessionStore.open(":memory:");
172
+ try {
173
+ const session = store.create(workspace);
174
+ const task = store.startTask(session.id, "Fix failure");
175
+ store.recordRepairAttempt({
176
+ id: "repair-evidence",
177
+ taskId: task.id,
178
+ command: "pnpm test",
179
+ evidence: {
180
+ summary: "Expected true but received false",
181
+ fileLocations: [{ path: "src/example.ts", line: 4 }],
182
+ excerpts: ["Expected true", "received false"],
183
+ },
184
+ selectedFiles: ["src/example.ts"],
185
+ createdAt: Date.now(),
186
+ });
187
+ const module = await import(pathToFileURL(join(workspace, "dist/application/context-manager.js")).href);
188
+ const brief = (await new module.ContextManager(store).prepare(session.id, task)).find((message) => message.content.startsWith("Repair attempt"))?.content;
189
+ if (!brief?.includes("Evidence: Expected true | received false"))
190
+ throw new Error("The repair brief did not preserve evidence.");
191
+ }
192
+ finally {
193
+ store.close();
194
+ }
195
+ },
196
+ },
197
+ ];
198
+ /** Runs real provider-backed tasks against clean Git-free snapshots and persists metadata-only evidence. */
199
+ export async function runSelfEvaluationSuite(options) {
200
+ const sourceRoot = options.sourceRoot ?? process.cwd();
201
+ await assertKairoSource(sourceRoot);
202
+ const trials = options.trials ?? 1;
203
+ if (!Number.isInteger(trials) || trials < 1 || trials > 5)
204
+ throw new Error("Self-evaluation trials must be an integer from 1 to 5.");
205
+ let run = options.evaluationStore.createEvaluationRun({
206
+ suite: "self",
207
+ provider: options.provider,
208
+ model: options.model,
209
+ sourceRevision: await sourceRevision(sourceRoot),
210
+ trialCount: trials,
211
+ startedAt: Date.now(),
212
+ completedAt: undefined,
213
+ });
214
+ const results = [];
215
+ try {
216
+ for (let trial = 1; trial <= trials; trial += 1)
217
+ for (const scenario of selfEvaluationScenarios) {
218
+ const startedAt = Date.now();
219
+ const result = await runScenario(sourceRoot, scenario, options, trial);
220
+ results.push(result);
221
+ options.evaluationStore.saveEvaluationAttempt(toAttempt(run.id, result, startedAt));
222
+ }
223
+ }
224
+ finally {
225
+ run = options.evaluationStore.completeEvaluationRun(run.id);
226
+ }
227
+ return { run, results };
228
+ }
229
+ async function prepareWorkspace(sourceRoot, scenario) {
230
+ const workspace = await mkdtemp(join(tmpdir(), `kairo-self-eval-${scenario.id}-`));
231
+ await cp(sourceRoot, workspace, {
232
+ recursive: true,
233
+ filter: (source) => ![".git", "node_modules", "dist", ".kairo", ".pnpm-store"].includes(basename(source)),
234
+ });
235
+ await assertCommand(workspace, ["install", "--offline", "--frozen-lockfile"]);
236
+ await scenario.seed(workspace);
237
+ return workspace;
238
+ }
239
+ async function runScenario(sourceRoot, scenario, options, trial) {
240
+ let workspace;
241
+ try {
242
+ workspace = await prepareWorkspace(sourceRoot, scenario);
243
+ const store = await SqliteSessionStore.open(":memory:");
244
+ try {
245
+ const session = store.create(workspace);
246
+ const tools = await WorkspaceTools.create(workspace);
247
+ store.saveRepositorySnapshot(session.id, await new RepositoryProfiler().profile(workspace));
248
+ const agent = new CodingAgent(createProvider({ provider: options.provider, model: options.model }, options.apiKey, definitions), store, tools, new FixtureApproval(), definitions, { provider: options.provider, model: options.model });
249
+ const failure = await runEvaluatedAgent(agent, session.id, scenario.prompt, options.onProgress);
250
+ const agentError = failure.error;
251
+ const task = agent.status(session.id);
252
+ let expectationPassed = false;
253
+ let error = agentError ?? task.error;
254
+ try {
255
+ if (!agentError) {
256
+ await assertCommand(workspace, ["test"]);
257
+ await scenario.verify(workspace);
258
+ expectationPassed = true;
259
+ }
260
+ }
261
+ catch (gradingError) {
262
+ error = `Grading failed: ${gradingError.message}`;
263
+ }
264
+ const metrics = taskMetrics(store.taskEvents(task.id));
265
+ const verified = task.verificationPassed === true;
266
+ return {
267
+ id: scenario.id,
268
+ trial,
269
+ passed: task.status === "completed" && verified && expectationPassed,
270
+ taskStatus: task.status,
271
+ verified,
272
+ expectationPassed,
273
+ error,
274
+ failureCategory: failure.category,
275
+ metrics: { ...metrics },
276
+ };
277
+ }
278
+ finally {
279
+ store.close();
280
+ }
281
+ }
282
+ catch (error) {
283
+ return {
284
+ id: scenario.id,
285
+ trial,
286
+ passed: false,
287
+ taskStatus: "failed",
288
+ verified: false,
289
+ expectationPassed: false,
290
+ error: error.message,
291
+ failureCategory: "setup",
292
+ metrics: emptyMetrics(),
293
+ };
294
+ }
295
+ finally {
296
+ if (workspace)
297
+ await rm(workspace, { recursive: true, force: true });
298
+ }
299
+ }
300
+ function toAttempt(runId, result, startedAt) {
301
+ return {
302
+ runId,
303
+ scenarioId: result.id,
304
+ trial: result.trial,
305
+ passed: result.passed,
306
+ taskStatus: result.taskStatus,
307
+ verified: result.verified,
308
+ expectationPassed: result.expectationPassed,
309
+ failureCategory: result.passed ? undefined : classifyFailure(result),
310
+ metrics: result.metrics,
311
+ durationMs: Date.now() - startedAt,
312
+ createdAt: Date.now(),
313
+ };
314
+ }
315
+ function classifyFailure(result) {
316
+ if (result.failureCategory)
317
+ return result.failureCategory;
318
+ if (result.error?.startsWith("Grading failed:"))
319
+ return "grader";
320
+ if (result.error?.includes("Self-eval seed") || result.error?.includes("pnpm install"))
321
+ return "setup";
322
+ if (!result.verified)
323
+ return "verification";
324
+ if (result.taskStatus === "failed")
325
+ return "agent";
326
+ return "unknown";
327
+ }
328
+ class FixtureApproval {
329
+ async approve(_call, _description) {
330
+ return true;
331
+ }
332
+ }
333
+ function assertCommand(workspace, args) {
334
+ return new Promise((resolve, reject) => {
335
+ const child = spawn("pnpm", args, { cwd: workspace, stdio: ["ignore", "pipe", "pipe"] });
336
+ let output = "";
337
+ const collect = (chunk) => {
338
+ output = `${output}${chunk}`.slice(-12_000);
339
+ };
340
+ child.stdout.on("data", collect);
341
+ child.stderr.on("data", collect);
342
+ const timer = setTimeout(() => child.kill("SIGTERM"), 120_000);
343
+ child.once("error", (error) => {
344
+ clearTimeout(timer);
345
+ reject(error);
346
+ });
347
+ child.once("close", (code) => {
348
+ clearTimeout(timer);
349
+ code === 0
350
+ ? resolve()
351
+ : reject(new Error(`pnpm ${args.join(" ")} exited with ${code}: ${output}`));
352
+ });
353
+ });
354
+ }
355
+ function sourceRevision(root) {
356
+ return new Promise((resolve) => {
357
+ const child = spawn("git", ["rev-parse", "--short=12", "HEAD"], {
358
+ cwd: root,
359
+ stdio: ["ignore", "pipe", "ignore"],
360
+ });
361
+ let value = "";
362
+ child.stdout.on("data", (chunk) => (value += chunk));
363
+ child.once("error", () => resolve("unknown"));
364
+ child.once("close", (code) => resolve(code === 0 && value.trim() ? value.trim() : "unknown"));
365
+ });
366
+ }
367
+ async function assertKairoSource(root) {
368
+ try {
369
+ const packageJson = JSON.parse(await readFile(join(root, "package.json"), "utf8"));
370
+ await readFile(join(root, "src/application/coding-agent.ts"), "utf8");
371
+ if (packageJson.name !== "kairo")
372
+ throw new Error();
373
+ }
374
+ catch {
375
+ throw new Error("Run `kairo eval self` from the Kairo repository root.");
376
+ }
377
+ }
378
+ function emptyMetrics() {
379
+ return {
380
+ modelTurns: 0,
381
+ toolExecutions: 0,
382
+ toolFailures: 0,
383
+ approvals: 0,
384
+ repairs: 0,
385
+ verificationPasses: 0,
386
+ verificationFailures: 0,
387
+ verificationSelections: 0,
388
+ focusedVerifications: 0,
389
+ broadVerifications: 0,
390
+ repairConverged: false,
391
+ modelMs: 0,
392
+ toolMs: 0,
393
+ };
394
+ }
@@ -0,0 +1,31 @@
1
+ import type { TaskEvent } from "../domain/models.js";
2
+ /** Derives counters from durable events so resume never resets or doubles stored totals. */
3
+ export declare function taskMetrics(events: TaskEvent[]): {
4
+ providerRetries: number;
5
+ providerWaitMs: number;
6
+ modelTurns: number;
7
+ modelFailures: number;
8
+ modelMs: number;
9
+ toolRequests: number;
10
+ toolExecutions: number;
11
+ toolFailures: number;
12
+ toolMs: number;
13
+ approvals: number;
14
+ denials: number;
15
+ approvalMs: number;
16
+ autonomousActions: number;
17
+ repairs: number;
18
+ verificationPasses: number;
19
+ verificationFailures: number;
20
+ verificationSelections: number;
21
+ focusedVerifications: number;
22
+ broadVerifications: number;
23
+ jevDecisions: number;
24
+ jevFailures: number;
25
+ jevMs: number;
26
+ jevRoutes: number;
27
+ jevSafetyChecks: number;
28
+ jevRecoveryChecks: number;
29
+ repairConverged: boolean;
30
+ unfinishedOperations: number;
31
+ };
@@ -0,0 +1,42 @@
1
+ /** Derives counters from durable events so resume never resets or doubles stored totals. */
2
+ export function taskMetrics(events) {
3
+ const count = (kind, outcome) => events.filter((event) => event.kind === kind && (!outcome || event.outcome === outcome)).length;
4
+ const duration = (kind) => events
5
+ .filter((event) => event.kind === kind)
6
+ .reduce((sum, event) => sum + (event.durationMs ?? 0), 0);
7
+ const started = events.filter((event) => event.kind === "model_started" || event.kind === "tool_started");
8
+ const unfinished = started.filter((event) => !events.some((end) => end.operationId === event.operationId &&
9
+ end.kind === (event.kind === "model_started" ? "model_finished" : "tool_finished")));
10
+ return {
11
+ providerRetries: count("provider_retry"),
12
+ providerWaitMs: duration("provider_retry_wait"),
13
+ modelTurns: count("model_started"),
14
+ modelFailures: count("model_finished", "failed"),
15
+ modelMs: duration("model_finished"),
16
+ toolRequests: count("tool_requested"),
17
+ toolExecutions: count("tool_started"),
18
+ toolFailures: count("tool_finished", "failed"),
19
+ toolMs: duration("tool_finished"),
20
+ approvals: count("approval", "approved"),
21
+ denials: count("approval", "denied"),
22
+ approvalMs: duration("approval"),
23
+ autonomousActions: count("autonomous"),
24
+ repairs: count("repair"),
25
+ verificationPasses: count("verification", "passed"),
26
+ verificationFailures: count("verification", "failed"),
27
+ verificationSelections: count("verification_selected"),
28
+ focusedVerifications: count("verification_selected", "focused"),
29
+ broadVerifications: count("verification_selected", "broad"),
30
+ jevDecisions: count("jev_completed"),
31
+ jevFailures: count("jev_failed"),
32
+ jevMs: duration("jev_completed") + duration("jev_failed"),
33
+ jevRoutes: events.filter((event) => event.kind === "jev_completed" && event.name === "route")
34
+ .length,
35
+ jevSafetyChecks: events.filter((event) => event.kind === "jev_completed" && event.name === "safety").length,
36
+ jevRecoveryChecks: events.filter((event) => event.kind === "jev_completed" && event.name === "recovery").length,
37
+ repairConverged: events.some((event) => event.kind === "repair") &&
38
+ events.findLast((event) => event.kind === "verification")?.outcome === "passed" &&
39
+ events.findLast((event) => event.kind === "status")?.outcome === "completed",
40
+ unfinishedOperations: unfinished.length,
41
+ };
42
+ }
@@ -0,0 +1,12 @@
1
+ import type { FailureEvidence, RepositorySnapshot, VerificationCandidate, VerificationSelection } from "../domain/models.js";
2
+ export declare class VerificationPlanner {
3
+ /** Converts recognized package scripts into safe, ordered verification suggestions. */
4
+ candidates(profile: Pick<RepositorySnapshot, "packageManager" | "scripts"> & Partial<Pick<RepositorySnapshot, "manifestFiles">>): VerificationCandidate[];
5
+ private lockfile;
6
+ /** Selects the narrowest known check that plausibly covers changed or failing files. */
7
+ select(profile: Pick<RepositorySnapshot, "sourceRoots" | "testRoots" | "configFiles" | "verificationCandidates">, changedFiles: string[], failure?: FailureEvidence): VerificationSelection | undefined;
8
+ /** Labels a command chosen outside the recommender without rejecting manual or model fallback. */
9
+ selectionForCommand(profile: Pick<RepositorySnapshot, "verificationCandidates"> | undefined, command: string, source: Exclude<VerificationSelection["source"], "recommended">): VerificationSelection;
10
+ /** Returns a broader project check only after a focused recommendation has passed. */
11
+ broader(profile: Pick<RepositorySnapshot, "verificationCandidates">, completed: VerificationSelection): VerificationSelection | undefined;
12
+ }
@@ -0,0 +1,97 @@
1
+ const labels = [
2
+ ["test", ["test", "test:unit", "test:run"]],
3
+ ["typecheck", ["typecheck", "type-check", "check"]],
4
+ ["lint", ["lint"]],
5
+ ["build", ["build"]],
6
+ ];
7
+ export class VerificationPlanner {
8
+ /** Converts recognized package scripts into safe, ordered verification suggestions. */
9
+ candidates(profile) {
10
+ const runner = profile.packageManager === "unknown" ? "npm run" : profile.packageManager;
11
+ const candidate = [];
12
+ for (const [label, names] of labels) {
13
+ const script = names.find((name) => profile.scripts[name]);
14
+ if (script)
15
+ candidate.push({
16
+ label,
17
+ command: runner === "npm run" ? `npm run ${script}` : `${runner} ${script}`,
18
+ scope: "broad",
19
+ reason: `Discovered ${script} package script.`,
20
+ evidence: [
21
+ {
22
+ path: profile.manifestFiles?.find((path) => path.endsWith("package.json")) ??
23
+ "package.json",
24
+ kind: "manifest",
25
+ },
26
+ ...(profile.packageManager === "unknown"
27
+ ? []
28
+ : [{ path: this.lockfile(profile.packageManager), kind: "lockfile" }]),
29
+ ],
30
+ });
31
+ }
32
+ return candidate;
33
+ }
34
+ lockfile(packageManager) {
35
+ return {
36
+ npm: "package-lock.json",
37
+ pnpm: "pnpm-lock.yaml",
38
+ yarn: "yarn.lock",
39
+ bun: "bun.lock",
40
+ }[packageManager];
41
+ }
42
+ /** Selects the narrowest known check that plausibly covers changed or failing files. */
43
+ select(profile, changedFiles, failure) {
44
+ const candidates = profile.verificationCandidates;
45
+ if (!candidates.length)
46
+ return undefined;
47
+ const changed = [
48
+ ...new Set([
49
+ ...changedFiles,
50
+ ...(failure ? failure.fileLocations.map((item) => item.path) : []),
51
+ ]),
52
+ ];
53
+ const byLabel = (label) => candidates.find((candidate) => candidate.label === label);
54
+ const selection = (candidate, scope, reason) => candidate && {
55
+ command: candidate.command,
56
+ label: candidate.label,
57
+ scope,
58
+ reason,
59
+ source: "recommended",
60
+ };
61
+ const isTest = changed.some((path) => profile.testRoots.some((root) => path === root || path.startsWith(`${root}/`)) ||
62
+ /(?:^|[./_-])(test|spec)(?:[._-]|$)/i.test(path));
63
+ if (isTest)
64
+ return selection(byLabel("test"), "broad", "Changed or failing test file is covered by the test script.");
65
+ const isSource = changed.some((path) => profile.sourceRoots.some((root) => path === root || path.startsWith(`${root}/`)));
66
+ if (isSource && byLabel("typecheck"))
67
+ return selection(byLabel("typecheck"), "broad", "Changed source file is covered by typechecking.");
68
+ const isConfig = changed.some((path) => profile.configFiles.includes(path) || /(?:^|\/)(?:package|tsconfig)\.json$/.test(path));
69
+ if (isConfig)
70
+ return selection(byLabel("test") ?? byLabel("typecheck"), "broad", "Configuration change needs a project-level check.");
71
+ return selection(byLabel("test") ?? byLabel("typecheck") ?? candidates[0], "broad", "No narrower coverage could be established.");
72
+ }
73
+ /** Labels a command chosen outside the recommender without rejecting manual or model fallback. */
74
+ selectionForCommand(profile, command, source) {
75
+ const candidate = profile?.verificationCandidates.find((item) => item.command === command);
76
+ return {
77
+ command,
78
+ label: candidate?.label ?? "custom",
79
+ scope: candidate?.scope ?? "broad",
80
+ reason: candidate?.reason ?? "Command selected outside automatic recommendation.",
81
+ source,
82
+ };
83
+ }
84
+ /** Returns a broader project check only after a focused recommendation has passed. */
85
+ broader(profile, completed) {
86
+ const candidate = profile.verificationCandidates.find((item) => item.label === "test" && item.command !== completed.command);
87
+ if (!candidate)
88
+ return undefined;
89
+ return {
90
+ command: candidate.command,
91
+ label: candidate.label,
92
+ scope: "broad",
93
+ reason: "Typechecking passed; run the project tests for behavioral coverage.",
94
+ source: "recommended",
95
+ };
96
+ }
97
+ }