@mrace07/kairo 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +141 -0
- package/dist/application/coding-agent.d.ts +85 -0
- package/dist/application/coding-agent.js +765 -0
- package/dist/application/context-manager.d.ts +22 -0
- package/dist/application/context-manager.js +174 -0
- package/dist/application/context-selector.d.ts +11 -0
- package/dist/application/context-selector.js +74 -0
- package/dist/application/evaluated-agent.d.ts +7 -0
- package/dist/application/evaluated-agent.js +16 -0
- package/dist/application/evaluation-comparison.d.ts +34 -0
- package/dist/application/evaluation-comparison.js +91 -0
- package/dist/application/evaluation-harness.d.ts +19 -0
- package/dist/application/evaluation-harness.js +217 -0
- package/dist/application/failure-analyzer.d.ts +5 -0
- package/dist/application/failure-analyzer.js +37 -0
- package/dist/application/interaction-routing.d.ts +9 -0
- package/dist/application/interaction-routing.js +20 -0
- package/dist/application/live-evaluation.d.ts +8 -0
- package/dist/application/live-evaluation.js +185 -0
- package/dist/application/model-routing.d.ts +12 -0
- package/dist/application/model-routing.js +40 -0
- package/dist/application/model-system-instruction.d.ts +4 -0
- package/dist/application/model-system-instruction.js +4 -0
- package/dist/application/self-evaluation.d.ts +26 -0
- package/dist/application/self-evaluation.js +394 -0
- package/dist/application/task-metrics.d.ts +31 -0
- package/dist/application/task-metrics.js +42 -0
- package/dist/application/verification-planner.d.ts +12 -0
- package/dist/application/verification-planner.js +97 -0
- package/dist/domain/models.d.ts +247 -0
- package/dist/domain/models.js +1 -0
- package/dist/domain/ports.d.ts +87 -0
- package/dist/domain/ports.js +1 -0
- package/dist/domain/provider-error.d.ts +18 -0
- package/dist/domain/provider-error.js +17 -0
- package/dist/infrastructure/configuration/config.d.ts +24 -0
- package/dist/infrastructure/configuration/config.js +79 -0
- package/dist/infrastructure/filesystem/platform-paths.d.ts +8 -0
- package/dist/infrastructure/filesystem/platform-paths.js +18 -0
- package/dist/infrastructure/persistence/sqlite-session-store.d.ts +82 -0
- package/dist/infrastructure/persistence/sqlite-session-store.js +447 -0
- package/dist/infrastructure/providers/gemini-provider.d.ts +14 -0
- package/dist/infrastructure/providers/gemini-provider.js +90 -0
- package/dist/infrastructure/providers/groq-provider.d.ts +16 -0
- package/dist/infrastructure/providers/groq-provider.js +101 -0
- package/dist/infrastructure/providers/jev-safety-advisor.d.ts +18 -0
- package/dist/infrastructure/providers/jev-safety-advisor.js +95 -0
- package/dist/infrastructure/providers/mistral-provider.d.ts +15 -0
- package/dist/infrastructure/providers/mistral-provider.js +137 -0
- package/dist/infrastructure/providers/openrouter-provider.d.ts +15 -0
- package/dist/infrastructure/providers/openrouter-provider.js +104 -0
- package/dist/infrastructure/providers/provider-recovery.d.ts +10 -0
- package/dist/infrastructure/providers/provider-recovery.js +108 -0
- package/dist/infrastructure/providers/provider-registry.d.ts +22 -0
- package/dist/infrastructure/providers/provider-registry.js +67 -0
- package/dist/infrastructure/repository/repository-awareness.d.ts +12 -0
- package/dist/infrastructure/repository/repository-awareness.js +25 -0
- package/dist/infrastructure/repository/repository-profiler.d.ts +35 -0
- package/dist/infrastructure/repository/repository-profiler.js +498 -0
- package/dist/infrastructure/security/macos-keychain-store.d.ts +17 -0
- package/dist/infrastructure/security/macos-keychain-store.js +73 -0
- package/dist/infrastructure/tools/workspace-tools.d.ts +30 -0
- package/dist/infrastructure/tools/workspace-tools.js +321 -0
- package/dist/interface/cli/evaluation-comparison-report.d.ts +6 -0
- package/dist/interface/cli/evaluation-comparison-report.js +46 -0
- package/dist/interface/cli/evaluation-report.d.ts +14 -0
- package/dist/interface/cli/evaluation-report.js +122 -0
- package/dist/interface/cli/index.d.ts +2 -0
- package/dist/interface/cli/index.js +238 -0
- package/dist/interface/cli/provider-setup.d.ts +16 -0
- package/dist/interface/cli/provider-setup.js +86 -0
- package/dist/interface/cli/repl.d.ts +7 -0
- package/dist/interface/cli/repl.js +19 -0
- package/dist/interface/cli/task-trace.d.ts +7 -0
- package/dist/interface/cli/task-trace.js +48 -0
- package/dist/interface/cli/tui.d.ts +147 -0
- package/dist/interface/cli/tui.js +910 -0
- package/package.json +61 -0
|
@@ -0,0 +1,321 @@
|
|
|
1
|
+
import { readdir, readFile, realpath, writeFile } from "node:fs/promises";
|
|
2
|
+
import { dirname, relative, resolve } from "node:path";
|
|
3
|
+
import { spawn } from "node:child_process";
|
|
4
|
+
const MAX_OUTPUT = 48_000;
|
|
5
|
+
const ignored = new Set([".git", "node_modules", "dist", "build", "coverage", ".next", ".kairo"]);
|
|
6
|
+
const truncate = (value) => value.length > MAX_OUTPUT ? `${value.slice(0, MAX_OUTPUT)}\n[output truncated]` : value;
|
|
7
|
+
export const definitions = [
|
|
8
|
+
{
|
|
9
|
+
name: "submit_plan",
|
|
10
|
+
description: "Submit the final structured implementation plan after inspecting the repository. Only available in planning mode.",
|
|
11
|
+
mutating: false,
|
|
12
|
+
parameters: {
|
|
13
|
+
type: "object",
|
|
14
|
+
properties: {
|
|
15
|
+
goal: { type: "string" },
|
|
16
|
+
assumptions: { type: "array", items: { type: "string" } },
|
|
17
|
+
files: {
|
|
18
|
+
type: "array",
|
|
19
|
+
items: {
|
|
20
|
+
type: "object",
|
|
21
|
+
properties: { path: { type: "string" }, reason: { type: "string" } },
|
|
22
|
+
required: ["path", "reason"],
|
|
23
|
+
},
|
|
24
|
+
},
|
|
25
|
+
steps: { type: "array", items: { type: "string" } },
|
|
26
|
+
verification: {
|
|
27
|
+
type: "object",
|
|
28
|
+
properties: { command: { type: "string" }, reason: { type: "string" } },
|
|
29
|
+
required: ["reason"],
|
|
30
|
+
},
|
|
31
|
+
risks: { type: "array", items: { type: "string" } },
|
|
32
|
+
},
|
|
33
|
+
required: ["goal", "assumptions", "files", "steps", "verification", "risks"],
|
|
34
|
+
},
|
|
35
|
+
},
|
|
36
|
+
{
|
|
37
|
+
name: "list_files",
|
|
38
|
+
description: "List workspace files under an optional relative directory. Omit path or use an empty path to list the workspace root.",
|
|
39
|
+
mutating: false,
|
|
40
|
+
parameters: { type: "object", properties: { path: { type: "string" } } },
|
|
41
|
+
},
|
|
42
|
+
{
|
|
43
|
+
name: "read_file",
|
|
44
|
+
description: "Read a UTF-8 text file in the workspace.",
|
|
45
|
+
mutating: false,
|
|
46
|
+
parameters: {
|
|
47
|
+
type: "object",
|
|
48
|
+
properties: { path: { type: "string" } },
|
|
49
|
+
required: ["path"],
|
|
50
|
+
},
|
|
51
|
+
},
|
|
52
|
+
{
|
|
53
|
+
name: "read_file_range",
|
|
54
|
+
description: "Read a bounded inclusive line range from a UTF-8 workspace file.",
|
|
55
|
+
mutating: false,
|
|
56
|
+
parameters: {
|
|
57
|
+
type: "object",
|
|
58
|
+
properties: {
|
|
59
|
+
path: { type: "string" },
|
|
60
|
+
startLine: { type: "number" },
|
|
61
|
+
endLine: { type: "number" },
|
|
62
|
+
},
|
|
63
|
+
required: ["path", "startLine", "endLine"],
|
|
64
|
+
},
|
|
65
|
+
},
|
|
66
|
+
{
|
|
67
|
+
name: "search_files",
|
|
68
|
+
description: "Search UTF-8 workspace files for literal text under an optional relative directory. Omit path or use an empty path to search the workspace root.",
|
|
69
|
+
mutating: false,
|
|
70
|
+
parameters: {
|
|
71
|
+
type: "object",
|
|
72
|
+
properties: { query: { type: "string" }, path: { type: "string" } },
|
|
73
|
+
required: ["query"],
|
|
74
|
+
},
|
|
75
|
+
},
|
|
76
|
+
{
|
|
77
|
+
name: "write_file",
|
|
78
|
+
description: "Create or replace a UTF-8 workspace file.",
|
|
79
|
+
mutating: true,
|
|
80
|
+
parameters: {
|
|
81
|
+
type: "object",
|
|
82
|
+
properties: { path: { type: "string" }, content: { type: "string" } },
|
|
83
|
+
required: ["path", "content"],
|
|
84
|
+
},
|
|
85
|
+
},
|
|
86
|
+
{
|
|
87
|
+
name: "edit_file",
|
|
88
|
+
description: "Replace one exact text occurrence in a workspace file.",
|
|
89
|
+
mutating: true,
|
|
90
|
+
parameters: {
|
|
91
|
+
type: "object",
|
|
92
|
+
properties: {
|
|
93
|
+
path: { type: "string" },
|
|
94
|
+
oldText: { type: "string" },
|
|
95
|
+
newText: { type: "string" },
|
|
96
|
+
},
|
|
97
|
+
required: ["path", "oldText", "newText"],
|
|
98
|
+
},
|
|
99
|
+
},
|
|
100
|
+
{
|
|
101
|
+
name: "run_command",
|
|
102
|
+
description: "Run a shell command inside the workspace.",
|
|
103
|
+
mutating: true,
|
|
104
|
+
parameters: {
|
|
105
|
+
type: "object",
|
|
106
|
+
properties: {
|
|
107
|
+
command: { type: "string" },
|
|
108
|
+
verification: {
|
|
109
|
+
type: "boolean",
|
|
110
|
+
description: "Set true only for an actual test, typecheck, lint, build, or explicit task assertion. Ordinary inspection commands are not verification.",
|
|
111
|
+
},
|
|
112
|
+
},
|
|
113
|
+
required: ["command"],
|
|
114
|
+
},
|
|
115
|
+
},
|
|
116
|
+
];
|
|
117
|
+
export class WorkspaceTools {
|
|
118
|
+
root;
|
|
119
|
+
/** Stores the already-resolved workspace root used for every safety check. */
|
|
120
|
+
constructor(root) {
|
|
121
|
+
this.root = root;
|
|
122
|
+
}
|
|
123
|
+
/** Resolves the requested workspace once before exposing any file tools. */
|
|
124
|
+
static async create(workspace) {
|
|
125
|
+
return new WorkspaceTools(await realpath(workspace));
|
|
126
|
+
}
|
|
127
|
+
/** Returns whether an absolute path remains inside the resolved workspace root. */
|
|
128
|
+
inside(path) {
|
|
129
|
+
const rel = relative(this.root, path);
|
|
130
|
+
return rel === "" || (!rel.startsWith("..") && !rel.includes("../"));
|
|
131
|
+
}
|
|
132
|
+
/** Validates a user path and rejects traversal or symlink escapes before access. */
|
|
133
|
+
async filePath(input, forWrite = false) {
|
|
134
|
+
if (typeof input !== "string" || !input.trim())
|
|
135
|
+
throw new Error("A non-empty path is required.");
|
|
136
|
+
const candidate = resolve(this.root, input);
|
|
137
|
+
if (!this.inside(candidate))
|
|
138
|
+
throw new Error("Path is outside the workspace.");
|
|
139
|
+
try {
|
|
140
|
+
const actual = await realpath(candidate);
|
|
141
|
+
// Resolving first catches a path that looks local but exits through a symlink.
|
|
142
|
+
if (!this.inside(actual))
|
|
143
|
+
throw new Error("Symlink escapes the workspace.");
|
|
144
|
+
return actual;
|
|
145
|
+
}
|
|
146
|
+
catch (error) {
|
|
147
|
+
if (!forWrite || error.code !== "ENOENT")
|
|
148
|
+
throw error;
|
|
149
|
+
const parent = await realpath(dirname(candidate));
|
|
150
|
+
if (!this.inside(parent))
|
|
151
|
+
throw new Error("Parent directory escapes the workspace.");
|
|
152
|
+
return candidate;
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
/** Resolves an optional directory, treating a blank value as the workspace root. */
|
|
156
|
+
directoryPath(input) {
|
|
157
|
+
const path = input === undefined || input === null || (typeof input === "string" && !input.trim())
|
|
158
|
+
? "."
|
|
159
|
+
: input;
|
|
160
|
+
return this.filePath(path);
|
|
161
|
+
}
|
|
162
|
+
/** Produces the exact action text shown in the terminal approval prompt. */
|
|
163
|
+
description(call) {
|
|
164
|
+
return call.name === "run_command"
|
|
165
|
+
? `Run command in ${this.root}:\n${String(call.args.command ?? "")}`
|
|
166
|
+
: `${call.name}: ${String(call.args.path ?? "workspace")}`;
|
|
167
|
+
}
|
|
168
|
+
/** Dispatches a validated tool request and converts executor errors into tool results. */
|
|
169
|
+
async execute(call) {
|
|
170
|
+
try {
|
|
171
|
+
switch (call.name) {
|
|
172
|
+
case "list_files":
|
|
173
|
+
return {
|
|
174
|
+
ok: true,
|
|
175
|
+
output: await this.list(await this.directoryPath(call.args.path)),
|
|
176
|
+
};
|
|
177
|
+
case "read_file":
|
|
178
|
+
return {
|
|
179
|
+
ok: true,
|
|
180
|
+
output: truncate(await readFile(await this.filePath(call.args.path), "utf8")),
|
|
181
|
+
};
|
|
182
|
+
case "read_file_range":
|
|
183
|
+
return {
|
|
184
|
+
ok: true,
|
|
185
|
+
output: await this.readRange(call.args),
|
|
186
|
+
};
|
|
187
|
+
case "search_files":
|
|
188
|
+
return {
|
|
189
|
+
ok: true,
|
|
190
|
+
output: await this.search(String(call.args.query ?? ""), await this.directoryPath(call.args.path)),
|
|
191
|
+
};
|
|
192
|
+
case "write_file": {
|
|
193
|
+
const path = await this.filePath(call.args.path, true);
|
|
194
|
+
if (typeof call.args.content !== "string")
|
|
195
|
+
throw new Error("content must be a string");
|
|
196
|
+
await writeFile(path, call.args.content, "utf8");
|
|
197
|
+
return { ok: true, output: `Wrote ${relative(this.root, path)}` };
|
|
198
|
+
}
|
|
199
|
+
case "edit_file":
|
|
200
|
+
return await this.edit(call.args);
|
|
201
|
+
case "run_command":
|
|
202
|
+
return await this.command(String(call.args.command ?? ""));
|
|
203
|
+
default:
|
|
204
|
+
throw new Error(`Unknown tool: ${call.name}`);
|
|
205
|
+
}
|
|
206
|
+
}
|
|
207
|
+
catch (error) {
|
|
208
|
+
return { ok: false, output: `Tool error: ${error.message}` };
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
/** Recursively lists a bounded set of non-ignored workspace files. */
|
|
212
|
+
async list(dir) {
|
|
213
|
+
const out = [];
|
|
214
|
+
const visit = async (current) => {
|
|
215
|
+
for (const entry of await readdir(current, { withFileTypes: true })) {
|
|
216
|
+
if (ignored.has(entry.name))
|
|
217
|
+
continue;
|
|
218
|
+
const full = resolve(current, entry.name);
|
|
219
|
+
const display = relative(this.root, full);
|
|
220
|
+
if (entry.isDirectory())
|
|
221
|
+
await visit(full);
|
|
222
|
+
else if (entry.isFile())
|
|
223
|
+
out.push(display);
|
|
224
|
+
if (out.length >= 500)
|
|
225
|
+
return;
|
|
226
|
+
}
|
|
227
|
+
};
|
|
228
|
+
await visit(dir);
|
|
229
|
+
return out.length ? truncate(out.sort().join("\n")) : "No files found.";
|
|
230
|
+
}
|
|
231
|
+
/** Searches readable workspace files for a literal query with bounded matches. */
|
|
232
|
+
async search(query, dir) {
|
|
233
|
+
if (!query)
|
|
234
|
+
throw new Error("query cannot be empty");
|
|
235
|
+
const files = (await this.list(dir)).split("\n");
|
|
236
|
+
const matches = [];
|
|
237
|
+
for (const file of files) {
|
|
238
|
+
if (file.startsWith("[") || matches.length >= 200)
|
|
239
|
+
break;
|
|
240
|
+
try {
|
|
241
|
+
const text = await readFile(resolve(this.root, file), "utf8");
|
|
242
|
+
text.split("\n").forEach((line, index) => {
|
|
243
|
+
if (line.includes(query) && matches.length < 200)
|
|
244
|
+
matches.push(`${file}:${index + 1}: ${line}`);
|
|
245
|
+
});
|
|
246
|
+
}
|
|
247
|
+
catch {
|
|
248
|
+
/* binary/unreadable files are skipped */
|
|
249
|
+
}
|
|
250
|
+
}
|
|
251
|
+
return matches.length ? truncate(matches.join("\n")) : "No matches found.";
|
|
252
|
+
}
|
|
253
|
+
/** Returns an inclusive, line-numbered slice while enforcing a small read limit. */
|
|
254
|
+
async readRange(args) {
|
|
255
|
+
const { startLine, endLine } = args;
|
|
256
|
+
if (!Number.isInteger(startLine) ||
|
|
257
|
+
!Number.isInteger(endLine) ||
|
|
258
|
+
startLine < 1 ||
|
|
259
|
+
endLine < startLine ||
|
|
260
|
+
endLine - startLine >= 500)
|
|
261
|
+
throw new Error("Use an inclusive line range from 1 to at most 500 lines.");
|
|
262
|
+
const lines = (await readFile(await this.filePath(args.path), "utf8")).split("\n");
|
|
263
|
+
const start = startLine;
|
|
264
|
+
const selected = lines.slice(start - 1, endLine);
|
|
265
|
+
return selected.length
|
|
266
|
+
? truncate(selected.map((line, index) => `${start + index}: ${line}`).join("\n"))
|
|
267
|
+
: "No lines found in this range.";
|
|
268
|
+
}
|
|
269
|
+
/** Replaces one unambiguous text occurrence in a workspace file. */
|
|
270
|
+
async edit(args) {
|
|
271
|
+
const path = await this.filePath(args.path, true);
|
|
272
|
+
if (typeof args.oldText !== "string" || typeof args.newText !== "string")
|
|
273
|
+
throw new Error("oldText and newText must be strings");
|
|
274
|
+
const text = await readFile(path, "utf8");
|
|
275
|
+
const first = text.indexOf(args.oldText);
|
|
276
|
+
if (first < 0)
|
|
277
|
+
throw new Error("oldText was not found");
|
|
278
|
+
if (text.indexOf(args.oldText, first + args.oldText.length) >= 0)
|
|
279
|
+
throw new Error("oldText is ambiguous; provide more context");
|
|
280
|
+
await writeFile(path, text.replace(args.oldText, args.newText), "utf8");
|
|
281
|
+
return { ok: true, output: `Edited ${relative(this.root, path)}` };
|
|
282
|
+
}
|
|
283
|
+
/** Runs a shell command in the workspace and captures bounded output and timing. */
|
|
284
|
+
command(command) {
|
|
285
|
+
if (!command.trim())
|
|
286
|
+
return Promise.resolve({ ok: false, output: "command cannot be empty" });
|
|
287
|
+
return new Promise((done) => {
|
|
288
|
+
const startedAt = Date.now();
|
|
289
|
+
const child = spawn(command, {
|
|
290
|
+
cwd: this.root,
|
|
291
|
+
shell: true,
|
|
292
|
+
stdio: ["ignore", "pipe", "pipe"],
|
|
293
|
+
});
|
|
294
|
+
let output = "";
|
|
295
|
+
const collect = (chunk) => {
|
|
296
|
+
output = truncate(output + chunk.toString());
|
|
297
|
+
};
|
|
298
|
+
child.stdout.on("data", collect);
|
|
299
|
+
child.stderr.on("data", collect);
|
|
300
|
+
const timer = setTimeout(() => child.kill("SIGTERM"), 60_000);
|
|
301
|
+
child.on("close", (code) => {
|
|
302
|
+
clearTimeout(timer);
|
|
303
|
+
done({
|
|
304
|
+
ok: code === 0,
|
|
305
|
+
output: truncate(output || `Command exited with ${code}`),
|
|
306
|
+
exitCode: code,
|
|
307
|
+
durationMs: Date.now() - startedAt,
|
|
308
|
+
});
|
|
309
|
+
});
|
|
310
|
+
child.on("error", (error) => {
|
|
311
|
+
clearTimeout(timer);
|
|
312
|
+
done({
|
|
313
|
+
ok: false,
|
|
314
|
+
output: error.message,
|
|
315
|
+
exitCode: null,
|
|
316
|
+
durationMs: Date.now() - startedAt,
|
|
317
|
+
});
|
|
318
|
+
});
|
|
319
|
+
});
|
|
320
|
+
}
|
|
321
|
+
}
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
import { type EvaluationComparison } from "../../application/evaluation-comparison.js";
|
|
2
|
+
import type { EvaluationAttempt, EvaluationRun } from "../../domain/models.js";
|
|
3
|
+
/** Displays the selected baseline without including any agent content. */
|
|
4
|
+
export declare function formatBaseline(run: EvaluationRun, attempts: EvaluationAttempt[]): string;
|
|
5
|
+
/** Renders informational aggregate and per-scenario reliability changes. */
|
|
6
|
+
export declare function formatComparison(comparison: EvaluationComparison): string;
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import { comparisonMetrics, summarizeAttempts, } from "../../application/evaluation-comparison.js";
|
|
2
|
+
/** Formats missing observations explicitly. */
|
|
3
|
+
function rate(value) {
|
|
4
|
+
return value === null ? "N/A" : `${(value * 100).toFixed(1)}%`;
|
|
5
|
+
}
|
|
6
|
+
/** Labels directional changes without implying statistical significance. */
|
|
7
|
+
function delta(value, higherIsBetter, suffix = "") {
|
|
8
|
+
if (value === null)
|
|
9
|
+
return "N/A (missing attempts)";
|
|
10
|
+
const label = Math.abs(value) < 1e-9
|
|
11
|
+
? "NO CHANGE"
|
|
12
|
+
: value > 0 === higherIsBetter
|
|
13
|
+
? "IMPROVEMENT"
|
|
14
|
+
: "REGRESSION";
|
|
15
|
+
return `${label} ${value > 0 ? "+" : ""}${value.toFixed(2)}${suffix}`;
|
|
16
|
+
}
|
|
17
|
+
/** Shows both observed denominators and, when compatible, their deltas. */
|
|
18
|
+
function changeLines(label, change) {
|
|
19
|
+
const { baseline: b, current: c, delta: d } = change;
|
|
20
|
+
return [
|
|
21
|
+
`Infrastructure failures: ${b.infrastructureFailures} -> ${c.infrastructureFailures}; coding attempts: ${b.codingAttempts} -> ${c.codingAttempts}; coding-outcome pass rate: ${rate(b.codingPassRate)} -> ${rate(c.codingPassRate)}${d ? ` | ${delta(d.codingPassRate === null ? null : d.codingPassRate * 100, true, " pp")}` : ""}`,
|
|
22
|
+
`${label}: ${b.passed}/${b.attempts} (${rate(b.passRate)}) -> ${c.passed}/${c.attempts} (${rate(c.passRate)})${d ? ` | ${delta(d.passRate === null ? null : d.passRate * 100, true, " pp")}; passed-attempt change ${d.passed > 0 ? "+" : ""}${d.passed}` : ""}`,
|
|
23
|
+
...comparisonMetrics.map((metric) => ` avg ${metric}: ${b.averages[metric]?.toFixed(2) ?? "N/A"} -> ${c.averages[metric]?.toFixed(2) ?? "N/A"}${d ? ` | ${delta(d.averages[metric], false)}` : ""}`),
|
|
24
|
+
];
|
|
25
|
+
}
|
|
26
|
+
/** Displays the selected baseline without including any agent content. */
|
|
27
|
+
export function formatBaseline(run, attempts) {
|
|
28
|
+
const summary = summarizeAttempts(attempts);
|
|
29
|
+
return `Self-evaluation baseline: ${run.id}\nModel: ${run.provider}/${run.model} Revision: ${run.sourceRevision} Trials: ${run.trialCount}\nReliability: ${summary.passed}/${summary.attempts} passed (${rate(summary.passRate)})`;
|
|
30
|
+
}
|
|
31
|
+
/** Renders informational aggregate and per-scenario reliability changes. */
|
|
32
|
+
export function formatComparison(comparison) {
|
|
33
|
+
const { baselineRun: b, currentRun: c } = comparison;
|
|
34
|
+
return [
|
|
35
|
+
"Self-evaluation baseline comparison (informational):",
|
|
36
|
+
`Baseline: ${b.id} model=${b.provider}/${b.model} revision=${b.sourceRevision} trials=${b.trialCount}`,
|
|
37
|
+
`Current: ${c.id} model=${c.provider}/${c.model} revision=${c.sourceRevision} trials=${c.trialCount}`,
|
|
38
|
+
...(!comparison.comparable
|
|
39
|
+
? ["Provider/model mismatch: runs are not directly comparable; deltas are omitted."]
|
|
40
|
+
: []),
|
|
41
|
+
"Rates and averages use observed attempts; absent attempts are not counted as failures. Lower resource averages do not establish better correctness.",
|
|
42
|
+
"Coding-outcome rates exclude classified infrastructure failures. Historical misclassifications cannot be reconstructed from saved metadata.",
|
|
43
|
+
...changeLines("Overall", comparison.overall),
|
|
44
|
+
...comparison.scenarios.flatMap((scenario) => changeLines(scenario.scenarioId, scenario)),
|
|
45
|
+
].join("\n");
|
|
46
|
+
}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
import type { EvaluationResult, EvaluationAttempt, EvaluationRun, LiveEvaluationResult, SelfEvaluationResult } from "../../domain/models.js";
|
|
2
|
+
import type { JevEvaluationReport } from "../../application/evaluation-harness.js";
|
|
3
|
+
/** Reports matched deterministic Jev-off/Jev-on trials without changing evaluator semantics. */
|
|
4
|
+
export declare function formatJevEvaluationReport(report: JevEvaluationReport): string;
|
|
5
|
+
/** Renders benchmark evidence in a compact human-readable report. */
|
|
6
|
+
export declare function formatEvaluationReport(results: EvaluationResult[]): string;
|
|
7
|
+
/** Renders deterministic task evidence and DeepEval's semantic completion verdict side by side. */
|
|
8
|
+
export declare function formatLiveEvaluationReport(results: LiveEvaluationResult[]): string;
|
|
9
|
+
/** Renders real Kairo repository tasks, including trial identity for stochastic agent runs. */
|
|
10
|
+
export declare function formatSelfEvaluationReport(results: SelfEvaluationResult[], runId?: string): string;
|
|
11
|
+
/** Renders compact metadata-only reliability history for baseline comparisons. */
|
|
12
|
+
export declare function formatEvaluationHistory(runs: EvaluationRun[]): string;
|
|
13
|
+
/** Renders one saved run without replaying sensitive task or model content. */
|
|
14
|
+
export declare function formatEvaluationRun(run: EvaluationRun, attempts: EvaluationAttempt[]): string;
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
/** Reports matched deterministic Jev-off/Jev-on trials without changing evaluator semantics. */
|
|
2
|
+
export function formatJevEvaluationReport(report) {
|
|
3
|
+
const summarize = (results) => ({
|
|
4
|
+
completed: results.filter((result) => result.passed).length,
|
|
5
|
+
repairs: results.reduce((sum, result) => sum + result.metrics.repairs, 0),
|
|
6
|
+
verificationFailures: results.reduce((sum, result) => sum + result.metrics.verificationFailures, 0),
|
|
7
|
+
escalations: results.filter((result) => result.taskStatus === "verification_required").length,
|
|
8
|
+
jevMs: results.reduce((sum, result) => sum + (result.metrics.jevMs ?? 0), 0),
|
|
9
|
+
routes: results.reduce((sum, result) => sum + (result.metrics.jevRoutes ?? 0), 0),
|
|
10
|
+
safety: results.reduce((sum, result) => sum + (result.metrics.jevSafetyChecks ?? 0), 0),
|
|
11
|
+
recovery: results.reduce((sum, result) => sum + (result.metrics.jevRecoveryChecks ?? 0), 0),
|
|
12
|
+
});
|
|
13
|
+
const off = summarize(report.off);
|
|
14
|
+
const on = summarize(report.on);
|
|
15
|
+
return [
|
|
16
|
+
"Kairo deterministic Jev evaluation (matched fixtures)",
|
|
17
|
+
`Jev off: ${off.completed}/${report.off.length} verified completion; repairs=${off.repairs}; verification failures=${off.verificationFailures}; escalations=${off.escalations}`,
|
|
18
|
+
`Jev on: ${on.completed}/${report.on.length} verified completion; repairs=${on.repairs}; verification failures=${on.verificationFailures}; escalations=${on.escalations}; Jev latency=${Math.round(on.jevMs)} ms; routing=${on.routes}; safety=${on.safety}; recovery=${on.recovery}`,
|
|
19
|
+
"This deterministic mode measures integration behavior and metadata overhead, not live Jev decision quality.",
|
|
20
|
+
].join("\n");
|
|
21
|
+
}
|
|
22
|
+
/** Renders benchmark evidence in a compact human-readable report. */
|
|
23
|
+
export function formatEvaluationReport(results) {
|
|
24
|
+
const passed = results.filter((result) => result.passed).length;
|
|
25
|
+
return [
|
|
26
|
+
`Kairo scripted evaluation: ${passed}/${results.length} passed`,
|
|
27
|
+
...results.map((result) => [
|
|
28
|
+
`${result.passed ? "PASS" : "FAIL"} ${result.id}`,
|
|
29
|
+
`status=${result.taskStatus}`,
|
|
30
|
+
`verified=${result.verified}`,
|
|
31
|
+
`expectation=${result.expectationPassed}`,
|
|
32
|
+
`turns=${result.metrics.modelTurns}`,
|
|
33
|
+
`tools=${result.metrics.toolExecutions}`,
|
|
34
|
+
`repairs=${result.metrics.repairs}`,
|
|
35
|
+
`checks=${result.metrics.verificationPasses}/${result.metrics.verificationFailures}`,
|
|
36
|
+
`verification=${result.metrics.focusedVerifications} focused/${result.metrics.broadVerifications} broad`,
|
|
37
|
+
`repairConverged=${result.metrics.repairConverged}`,
|
|
38
|
+
result.error ? `error=${result.error}` : "",
|
|
39
|
+
]
|
|
40
|
+
.filter(Boolean)
|
|
41
|
+
.join(" ")),
|
|
42
|
+
].join("\n");
|
|
43
|
+
}
|
|
44
|
+
/** Renders deterministic task evidence and DeepEval's semantic completion verdict side by side. */
|
|
45
|
+
export function formatLiveEvaluationReport(results) {
|
|
46
|
+
const passed = results.filter((result) => result.passed).length;
|
|
47
|
+
return [
|
|
48
|
+
`Kairo live DeepEval evaluation: ${passed}/${results.length} passed`,
|
|
49
|
+
...results.map((result) => [
|
|
50
|
+
`${result.passed ? "PASS" : "FAIL"} ${result.id}`,
|
|
51
|
+
`status=${result.taskStatus}`,
|
|
52
|
+
`verified=${result.verified}`,
|
|
53
|
+
`expectation=${result.expectationPassed}`,
|
|
54
|
+
`judge=${result.judge.passed ? "passed" : "failed"}`,
|
|
55
|
+
result.judge.score === undefined ? "" : `score=${result.judge.score.toFixed(2)}`,
|
|
56
|
+
result.judge.reason ? `reason=${result.judge.reason}` : "",
|
|
57
|
+
result.judge.error ? `judgeError=${result.judge.error}` : "",
|
|
58
|
+
result.error ? `error=${result.error}` : "",
|
|
59
|
+
]
|
|
60
|
+
.filter(Boolean)
|
|
61
|
+
.join(" ")),
|
|
62
|
+
].join("\n");
|
|
63
|
+
}
|
|
64
|
+
/** Renders real Kairo repository tasks, including trial identity for stochastic agent runs. */
|
|
65
|
+
export function formatSelfEvaluationReport(results, runId) {
|
|
66
|
+
const passed = results.filter((result) => result.passed).length;
|
|
67
|
+
return [
|
|
68
|
+
`Kairo self evaluation${runId ? ` (${runId})` : ""}: ${passed}/${results.length} passed`,
|
|
69
|
+
...results.map((result) => [
|
|
70
|
+
`${result.passed ? "PASS" : "FAIL"} ${result.id}`,
|
|
71
|
+
`trial=${result.trial}`,
|
|
72
|
+
`retries=${result.metrics.providerRetries ?? "N/A"} waitMs=${result.metrics.providerWaitMs ?? "N/A"}`,
|
|
73
|
+
result.failureCategory ? `failure=${result.failureCategory}` : "",
|
|
74
|
+
`status=${result.taskStatus}`,
|
|
75
|
+
`verified=${result.verified}`,
|
|
76
|
+
`expectation=${result.expectationPassed}`,
|
|
77
|
+
`turns=${result.metrics.modelTurns}`,
|
|
78
|
+
`tools=${result.metrics.toolExecutions}`,
|
|
79
|
+
`repairs=${result.metrics.repairs}`,
|
|
80
|
+
`verification=${result.metrics.focusedVerifications} focused/${result.metrics.broadVerifications} broad`,
|
|
81
|
+
`repairConverged=${result.metrics.repairConverged}`,
|
|
82
|
+
result.error ? `error=${result.error}` : "",
|
|
83
|
+
]
|
|
84
|
+
.filter(Boolean)
|
|
85
|
+
.join(" ")),
|
|
86
|
+
].join("\n");
|
|
87
|
+
}
|
|
88
|
+
/** Renders compact metadata-only reliability history for baseline comparisons. */
|
|
89
|
+
export function formatEvaluationHistory(runs) {
|
|
90
|
+
if (!runs.length)
|
|
91
|
+
return "No saved self-evaluation runs.";
|
|
92
|
+
return [
|
|
93
|
+
"Kairo self-evaluation history:",
|
|
94
|
+
...runs.map((run) => `${run.id} ${run.passedCount}/${run.attemptCount} passed trials=${run.trialCount} model=${run.provider}/${run.model} revision=${run.sourceRevision} ${new Date(run.startedAt).toISOString()}`),
|
|
95
|
+
].join("\n");
|
|
96
|
+
}
|
|
97
|
+
/** Renders one saved run without replaying sensitive task or model content. */
|
|
98
|
+
export function formatEvaluationRun(run, attempts) {
|
|
99
|
+
return [
|
|
100
|
+
`Kairo self evaluation: ${run.id}`,
|
|
101
|
+
`Reliability: ${run.passedCount}/${run.attemptCount} passed`,
|
|
102
|
+
`Model: ${run.provider}/${run.model} Revision: ${run.sourceRevision} Trials: ${run.trialCount}`,
|
|
103
|
+
`Started: ${new Date(run.startedAt).toISOString()}${run.completedAt ? ` Completed: ${new Date(run.completedAt).toISOString()}` : ""}`,
|
|
104
|
+
...attempts.map((attempt) => [
|
|
105
|
+
`${attempt.passed ? "PASS" : "FAIL"} ${attempt.scenarioId}`,
|
|
106
|
+
`trial=${attempt.trial}`,
|
|
107
|
+
`retries=${attempt.metrics.providerRetries ?? "N/A"} waitMs=${attempt.metrics.providerWaitMs ?? "N/A"}`,
|
|
108
|
+
`status=${attempt.taskStatus}`,
|
|
109
|
+
`verified=${attempt.verified}`,
|
|
110
|
+
`expectation=${attempt.expectationPassed}`,
|
|
111
|
+
`turns=${attempt.metrics.modelTurns}`,
|
|
112
|
+
`tools=${attempt.metrics.toolExecutions}`,
|
|
113
|
+
`repairs=${attempt.metrics.repairs}`,
|
|
114
|
+
`verification=${attempt.metrics.focusedVerifications} focused/${attempt.metrics.broadVerifications} broad`,
|
|
115
|
+
`repairConverged=${attempt.metrics.repairConverged}`,
|
|
116
|
+
`durationMs=${attempt.durationMs}`,
|
|
117
|
+
attempt.failureCategory ? `failure=${attempt.failureCategory}` : "",
|
|
118
|
+
]
|
|
119
|
+
.filter(Boolean)
|
|
120
|
+
.join(" ")),
|
|
121
|
+
].join("\n");
|
|
122
|
+
}
|