@mrace07/kairo 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +141 -0
- package/dist/application/coding-agent.d.ts +85 -0
- package/dist/application/coding-agent.js +765 -0
- package/dist/application/context-manager.d.ts +22 -0
- package/dist/application/context-manager.js +174 -0
- package/dist/application/context-selector.d.ts +11 -0
- package/dist/application/context-selector.js +74 -0
- package/dist/application/evaluated-agent.d.ts +7 -0
- package/dist/application/evaluated-agent.js +16 -0
- package/dist/application/evaluation-comparison.d.ts +34 -0
- package/dist/application/evaluation-comparison.js +91 -0
- package/dist/application/evaluation-harness.d.ts +19 -0
- package/dist/application/evaluation-harness.js +217 -0
- package/dist/application/failure-analyzer.d.ts +5 -0
- package/dist/application/failure-analyzer.js +37 -0
- package/dist/application/interaction-routing.d.ts +9 -0
- package/dist/application/interaction-routing.js +20 -0
- package/dist/application/live-evaluation.d.ts +8 -0
- package/dist/application/live-evaluation.js +185 -0
- package/dist/application/model-routing.d.ts +12 -0
- package/dist/application/model-routing.js +40 -0
- package/dist/application/model-system-instruction.d.ts +4 -0
- package/dist/application/model-system-instruction.js +4 -0
- package/dist/application/self-evaluation.d.ts +26 -0
- package/dist/application/self-evaluation.js +394 -0
- package/dist/application/task-metrics.d.ts +31 -0
- package/dist/application/task-metrics.js +42 -0
- package/dist/application/verification-planner.d.ts +12 -0
- package/dist/application/verification-planner.js +97 -0
- package/dist/domain/models.d.ts +247 -0
- package/dist/domain/models.js +1 -0
- package/dist/domain/ports.d.ts +87 -0
- package/dist/domain/ports.js +1 -0
- package/dist/domain/provider-error.d.ts +18 -0
- package/dist/domain/provider-error.js +17 -0
- package/dist/infrastructure/configuration/config.d.ts +24 -0
- package/dist/infrastructure/configuration/config.js +79 -0
- package/dist/infrastructure/filesystem/platform-paths.d.ts +8 -0
- package/dist/infrastructure/filesystem/platform-paths.js +18 -0
- package/dist/infrastructure/persistence/sqlite-session-store.d.ts +82 -0
- package/dist/infrastructure/persistence/sqlite-session-store.js +447 -0
- package/dist/infrastructure/providers/gemini-provider.d.ts +14 -0
- package/dist/infrastructure/providers/gemini-provider.js +90 -0
- package/dist/infrastructure/providers/groq-provider.d.ts +16 -0
- package/dist/infrastructure/providers/groq-provider.js +101 -0
- package/dist/infrastructure/providers/jev-safety-advisor.d.ts +18 -0
- package/dist/infrastructure/providers/jev-safety-advisor.js +95 -0
- package/dist/infrastructure/providers/mistral-provider.d.ts +15 -0
- package/dist/infrastructure/providers/mistral-provider.js +137 -0
- package/dist/infrastructure/providers/openrouter-provider.d.ts +15 -0
- package/dist/infrastructure/providers/openrouter-provider.js +104 -0
- package/dist/infrastructure/providers/provider-recovery.d.ts +10 -0
- package/dist/infrastructure/providers/provider-recovery.js +108 -0
- package/dist/infrastructure/providers/provider-registry.d.ts +22 -0
- package/dist/infrastructure/providers/provider-registry.js +67 -0
- package/dist/infrastructure/repository/repository-awareness.d.ts +12 -0
- package/dist/infrastructure/repository/repository-awareness.js +25 -0
- package/dist/infrastructure/repository/repository-profiler.d.ts +35 -0
- package/dist/infrastructure/repository/repository-profiler.js +498 -0
- package/dist/infrastructure/security/macos-keychain-store.d.ts +17 -0
- package/dist/infrastructure/security/macos-keychain-store.js +73 -0
- package/dist/infrastructure/tools/workspace-tools.d.ts +30 -0
- package/dist/infrastructure/tools/workspace-tools.js +321 -0
- package/dist/interface/cli/evaluation-comparison-report.d.ts +6 -0
- package/dist/interface/cli/evaluation-comparison-report.js +46 -0
- package/dist/interface/cli/evaluation-report.d.ts +14 -0
- package/dist/interface/cli/evaluation-report.js +122 -0
- package/dist/interface/cli/index.d.ts +2 -0
- package/dist/interface/cli/index.js +238 -0
- package/dist/interface/cli/provider-setup.d.ts +16 -0
- package/dist/interface/cli/provider-setup.js +86 -0
- package/dist/interface/cli/repl.d.ts +7 -0
- package/dist/interface/cli/repl.js +19 -0
- package/dist/interface/cli/task-trace.d.ts +7 -0
- package/dist/interface/cli/task-trace.js +48 -0
- package/dist/interface/cli/tui.d.ts +147 -0
- package/dist/interface/cli/tui.js +910 -0
- package/package.json +61 -0
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
import type { Message, Task } from "../domain/models.js";
|
|
2
|
+
import type { TaskStore } from "../domain/ports.js";
|
|
3
|
+
export declare class ContextManager {
|
|
4
|
+
private readonly store;
|
|
5
|
+
private readonly selector;
|
|
6
|
+
constructor(store: TaskStore);
|
|
7
|
+
/** Builds the bounded message list that is sent to the model for the next turn. */
|
|
8
|
+
prepare(sessionId: string, task: Task): Promise<Message[]>;
|
|
9
|
+
/** Converts older conversation evidence into a durable summary before it is omitted. */
|
|
10
|
+
compact(sessionId: string, task: Task): string;
|
|
11
|
+
/** Drops leading tool-only messages that Gemini cannot interpret as a fresh conversation. */
|
|
12
|
+
private cleanStart;
|
|
13
|
+
/** Renders repository facts and ranked files as concise model guidance. */
|
|
14
|
+
private profileContext;
|
|
15
|
+
/** Loads applicable agent instructions for this request without retaining their text. */
|
|
16
|
+
private instructions;
|
|
17
|
+
private readBounded;
|
|
18
|
+
/** Combines the request and latest failure evidence for relevance ranking. */
|
|
19
|
+
private retrievalQuery;
|
|
20
|
+
/** Renders the latest persisted failure as an actionable repair instruction. */
|
|
21
|
+
private repairBrief;
|
|
22
|
+
}
|
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
import { open } from "node:fs/promises";
|
|
2
|
+
import { dirname, join } from "node:path";
|
|
3
|
+
import { ContextSelector } from "./context-selector.js";
|
|
4
|
+
const AUTO_COMPACT_AFTER = 48;
|
|
5
|
+
const MODEL_MESSAGE_LIMIT = 32;
|
|
6
|
+
const MAX_INSTRUCTION_BYTES = 16 * 1024;
|
|
7
|
+
const MAX_INSTRUCTION_TOTAL_BYTES = 32 * 1024;
|
|
8
|
+
const excerpt = (value, length = 700) => value.length > length ? `${value.slice(0, length)}…` : value;
|
|
9
|
+
export class ContextManager {
|
|
10
|
+
store;
|
|
11
|
+
selector = new ContextSelector();
|
|
12
|
+
constructor(store) {
|
|
13
|
+
this.store = store;
|
|
14
|
+
}
|
|
15
|
+
/** Builds the bounded message list that is sent to the model for the next turn. */
|
|
16
|
+
async prepare(sessionId, task) {
|
|
17
|
+
if (this.store.messageCount(sessionId) >= AUTO_COMPACT_AFTER)
|
|
18
|
+
this.compact(sessionId, task);
|
|
19
|
+
const checkpoint = this.store.latestCheckpoint(sessionId);
|
|
20
|
+
const recent = this.cleanStart(this.store.recentMessages(sessionId, MODEL_MESSAGE_LIMIT));
|
|
21
|
+
const profile = this.store.repositorySnapshot(sessionId);
|
|
22
|
+
const repositoryContext = profile && {
|
|
23
|
+
role: "user",
|
|
24
|
+
content: await this.profileContext(task, profile),
|
|
25
|
+
createdAt: profile.createdAt,
|
|
26
|
+
};
|
|
27
|
+
const repairBrief = this.repairBrief(task);
|
|
28
|
+
const repairContext = repairBrief && {
|
|
29
|
+
role: "user",
|
|
30
|
+
content: repairBrief,
|
|
31
|
+
createdAt: Date.now(),
|
|
32
|
+
};
|
|
33
|
+
const context = [
|
|
34
|
+
...(repositoryContext ? [repositoryContext] : []),
|
|
35
|
+
...(repairContext ? [repairContext] : []),
|
|
36
|
+
...(checkpoint
|
|
37
|
+
? [
|
|
38
|
+
{
|
|
39
|
+
role: "user",
|
|
40
|
+
content: `Context checkpoint from earlier work:\n${checkpoint.summary}`,
|
|
41
|
+
createdAt: checkpoint.createdAt,
|
|
42
|
+
},
|
|
43
|
+
]
|
|
44
|
+
: []),
|
|
45
|
+
...recent,
|
|
46
|
+
];
|
|
47
|
+
return context;
|
|
48
|
+
}
|
|
49
|
+
/** Converts older conversation evidence into a durable summary before it is omitted. */
|
|
50
|
+
compact(sessionId, task) {
|
|
51
|
+
const current = this.store.latestCheckpoint(sessionId);
|
|
52
|
+
const lastId = this.store.lastMessageId(sessionId);
|
|
53
|
+
if (current?.throughMessageId === lastId)
|
|
54
|
+
return current.summary;
|
|
55
|
+
const recent = this.store
|
|
56
|
+
.recentMessages(sessionId, 12)
|
|
57
|
+
.map((message) => `${message.role}${message.toolName ? `:${message.toolName}` : ""}: ${excerpt(message.content, 420)}`)
|
|
58
|
+
.join("\n");
|
|
59
|
+
const summary = [
|
|
60
|
+
`Task: ${task.prompt}`,
|
|
61
|
+
`State: ${task.status}`,
|
|
62
|
+
`Changed files: ${task.changedFiles.length ? task.changedFiles.join(", ") : "none"}`,
|
|
63
|
+
`Verification: ${task.verificationCommand ? `${task.verificationCommand} (${task.verificationPassed ? "passed" : "not passed"})` : "not run"}`,
|
|
64
|
+
task.error ? `Last error: ${excerpt(task.error)}` : "",
|
|
65
|
+
"Recent durable evidence:",
|
|
66
|
+
recent,
|
|
67
|
+
]
|
|
68
|
+
.filter(Boolean)
|
|
69
|
+
.join("\n");
|
|
70
|
+
this.store.saveCheckpoint(sessionId, task.id, summary, lastId);
|
|
71
|
+
this.store.updateTask(task.id, { summary });
|
|
72
|
+
return summary;
|
|
73
|
+
}
|
|
74
|
+
/** Drops leading tool-only messages that Gemini cannot interpret as a fresh conversation. */
|
|
75
|
+
cleanStart(messages) {
|
|
76
|
+
const first = messages.findIndex((message) => message.role === "user" || (message.role === "model" && !message.toolCallId));
|
|
77
|
+
return first < 0 ? messages.slice(-1) : messages.slice(first);
|
|
78
|
+
}
|
|
79
|
+
/** Renders repository facts and ranked files as concise model guidance. */
|
|
80
|
+
async profileContext(task, profile) {
|
|
81
|
+
const relevantFiles = this.selector.select(this.retrievalQuery(task), profile);
|
|
82
|
+
const instructions = await this.instructions(profile, relevantFiles);
|
|
83
|
+
const verification = profile.verificationCandidates.map((candidate) => {
|
|
84
|
+
const evidence = candidate.evidence?.map((item) => item.path).join(", ");
|
|
85
|
+
return `${candidate.label} = ${candidate.command}${evidence ? ` [${evidence}]` : ""}`;
|
|
86
|
+
});
|
|
87
|
+
return [
|
|
88
|
+
"Repository snapshot:",
|
|
89
|
+
`Root: ${profile.root}`,
|
|
90
|
+
`Ecosystems: ${profile.ecosystems.join(", ") || "none detected"}`,
|
|
91
|
+
`Repository state: ${profile.fingerprint.kind}${profile.fingerprint.branch ? ` branch=${profile.fingerprint.branch}` : ""}${profile.fingerprint.head ? ` head=${profile.fingerprint.head.slice(0, 12)}` : ""}${profile.truncated ? " (inventory truncated)" : ""}`,
|
|
92
|
+
`Changed paths: ${profile.changedPaths.join(", ") || "clean or unavailable"}`,
|
|
93
|
+
`Package: ${profile.packageName ?? "unknown"}`,
|
|
94
|
+
`Package manager: ${profile.packageManager}`,
|
|
95
|
+
`Scripts: ${Object.keys(profile.scripts).length ? Object.keys(profile.scripts).sort().join(", ") : "none detected"}`,
|
|
96
|
+
`Source roots: ${profile.sourceRoots.join(", ") || "none detected"}`,
|
|
97
|
+
`Test roots: ${profile.testRoots.join(", ") || "none detected"}`,
|
|
98
|
+
`Instruction files: ${profile.instructionFiles.join(", ") || "none detected"}`,
|
|
99
|
+
`Manifests: ${profile.manifestFiles.join(", ") || "none detected"}`,
|
|
100
|
+
`CI files: ${profile.ciFiles.join(", ") || "none detected"}`,
|
|
101
|
+
`Build files: ${profile.buildFiles.join(", ") || "none detected"}`,
|
|
102
|
+
`Documentation: ${profile.documentationFiles.join(", ") || "none detected"}`,
|
|
103
|
+
`Available verification: ${verification.join("; ") || "none detected"}`,
|
|
104
|
+
`Relevant files for this task: ${relevantFiles.join(", ") || "use search_files to locate files"}`,
|
|
105
|
+
instructions,
|
|
106
|
+
"Use the profile as a guide, inspect files before edits, and choose an appropriate verification command after changes. For custom checks use run_command with verification=true; ordinary inspection commands do not verify a task. Every edit invalidates earlier verification.",
|
|
107
|
+
]
|
|
108
|
+
.filter(Boolean)
|
|
109
|
+
.join("\n");
|
|
110
|
+
}
|
|
111
|
+
/** Loads applicable agent instructions for this request without retaining their text. */
|
|
112
|
+
async instructions(profile, relevantFiles) {
|
|
113
|
+
let remaining = MAX_INSTRUCTION_TOTAL_BYTES;
|
|
114
|
+
const sections = [];
|
|
115
|
+
for (const path of profile.instructionFiles) {
|
|
116
|
+
const directory = dirname(path).replaceAll("\\", "/");
|
|
117
|
+
const global = directory === "." ||
|
|
118
|
+
path === ".github/copilot-instructions.md" ||
|
|
119
|
+
path.startsWith(".cursor/rules/");
|
|
120
|
+
if (!global && !relevantFiles.some((file) => file.startsWith(`${directory}/`)))
|
|
121
|
+
continue;
|
|
122
|
+
const length = Math.min(MAX_INSTRUCTION_BYTES, remaining);
|
|
123
|
+
if (length <= 0)
|
|
124
|
+
break;
|
|
125
|
+
const text = await this.readBounded(join(profile.root, path), length);
|
|
126
|
+
if (!text)
|
|
127
|
+
continue;
|
|
128
|
+
sections.push(`Instructions from ${path}:\n${text}`);
|
|
129
|
+
remaining -= Buffer.byteLength(text);
|
|
130
|
+
}
|
|
131
|
+
return sections.join("\n");
|
|
132
|
+
}
|
|
133
|
+
async readBounded(path, length) {
|
|
134
|
+
let handle;
|
|
135
|
+
try {
|
|
136
|
+
handle = await open(path, "r");
|
|
137
|
+
const buffer = Buffer.alloc(length);
|
|
138
|
+
const { bytesRead } = await handle.read(buffer, 0, length, 0);
|
|
139
|
+
return buffer.subarray(0, bytesRead).toString("utf8");
|
|
140
|
+
}
|
|
141
|
+
catch {
|
|
142
|
+
return "";
|
|
143
|
+
}
|
|
144
|
+
finally {
|
|
145
|
+
await handle?.close();
|
|
146
|
+
}
|
|
147
|
+
}
|
|
148
|
+
/** Combines the request and latest failure evidence for relevance ranking. */
|
|
149
|
+
retrievalQuery(task) {
|
|
150
|
+
return [task.prompt, task.error, task.verificationOutput]
|
|
151
|
+
.filter((value) => Boolean(value))
|
|
152
|
+
.map((value) => value.slice(0, 8_000))
|
|
153
|
+
.join("\n");
|
|
154
|
+
}
|
|
155
|
+
/** Renders the latest persisted failure as an actionable repair instruction. */
|
|
156
|
+
repairBrief(task) {
|
|
157
|
+
const attempts = this.store.repairAttempts(task.id);
|
|
158
|
+
const latest = attempts.at(-1);
|
|
159
|
+
if (!latest)
|
|
160
|
+
return "";
|
|
161
|
+
const locations = latest.evidence.fileLocations
|
|
162
|
+
.map((location) => `${location.path}${location.line ? `:${location.line}` : ""}`)
|
|
163
|
+
.join(", ");
|
|
164
|
+
return [
|
|
165
|
+
`Repair attempt ${attempts.length}/2 after \`${latest.command}\` failed.`,
|
|
166
|
+
`Failure: ${latest.evidence.summary}`,
|
|
167
|
+
`Locations: ${locations || "none extracted"}`,
|
|
168
|
+
`Evidence: ${latest.evidence.excerpts.join(" | ") || "inspect the command output"}`,
|
|
169
|
+
`Changed files: ${task.changedFiles.join(", ") || "none recorded"}`,
|
|
170
|
+
`Repair budget remaining: ${Math.max(0, 2 - attempts.length)}.`,
|
|
171
|
+
"Prioritize the affected files, make a materially different repair, then rerun the same focused verification.",
|
|
172
|
+
].join("\n");
|
|
173
|
+
}
|
|
174
|
+
}
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import type { RepositorySnapshot } from "../domain/models.js";
|
|
2
|
+
export declare class ContextSelector {
|
|
3
|
+
/** Returns the highest-scoring repository files for a task or verification failure. */
|
|
4
|
+
select(task: string, profile: RepositorySnapshot, limit?: number): string[];
|
|
5
|
+
/** Produces normal and camel-case-split query terms for code identifiers. */
|
|
6
|
+
private terms;
|
|
7
|
+
/** Adds graph proximity to a file's direct lexical and structural relevance. */
|
|
8
|
+
private score;
|
|
9
|
+
/** Scores paths, file text, symbols, and source/test roles independently. */
|
|
10
|
+
private directScore;
|
|
11
|
+
}
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
const STOP_WORDS = new Set([
|
|
2
|
+
"the",
|
|
3
|
+
"and",
|
|
4
|
+
"with",
|
|
5
|
+
"this",
|
|
6
|
+
"that",
|
|
7
|
+
"from",
|
|
8
|
+
"into",
|
|
9
|
+
"add",
|
|
10
|
+
"fix",
|
|
11
|
+
"make",
|
|
12
|
+
"code",
|
|
13
|
+
"file",
|
|
14
|
+
]);
|
|
15
|
+
export class ContextSelector {
|
|
16
|
+
/** Returns the highest-scoring repository files for a task or verification failure. */
|
|
17
|
+
select(task, profile, limit = 12) {
|
|
18
|
+
const terms = this.terms(task);
|
|
19
|
+
const structural = new Map((profile.files ?? []).map((file) => [file.path, file]));
|
|
20
|
+
const inventory = profile.entries?.length
|
|
21
|
+
? profile.entries
|
|
22
|
+
: profile.indexedFiles.map((path) => ({ path, kind: "other" }));
|
|
23
|
+
const files = inventory.map((entry) => ({
|
|
24
|
+
...(structural.get(entry.path) ?? {
|
|
25
|
+
path: entry.path,
|
|
26
|
+
terms: [],
|
|
27
|
+
symbols: [],
|
|
28
|
+
imports: [],
|
|
29
|
+
relatedFiles: [],
|
|
30
|
+
}),
|
|
31
|
+
kind: entry.kind,
|
|
32
|
+
}));
|
|
33
|
+
const direct = new Set(files.filter((file) => this.directScore(file, terms, profile) > 0).map((file) => file.path));
|
|
34
|
+
// Relationship proximity only boosts files that are connected to independently relevant files.
|
|
35
|
+
return files
|
|
36
|
+
.map((file) => ({ path: file.path, score: this.score(file, terms, profile, direct) }))
|
|
37
|
+
.filter((item) => item.score > 0)
|
|
38
|
+
.sort((left, right) => right.score - left.score || left.path.localeCompare(right.path))
|
|
39
|
+
.slice(0, limit)
|
|
40
|
+
.map((item) => item.path);
|
|
41
|
+
}
|
|
42
|
+
/** Produces normal and camel-case-split query terms for code identifiers. */
|
|
43
|
+
terms(value) {
|
|
44
|
+
return [value, value.replace(/([a-z])([A-Z])/g, "$1 $2")]
|
|
45
|
+
.flatMap((part) => part.toLowerCase().split(/[^a-z0-9_$]+/))
|
|
46
|
+
.filter((term) => term.length > 2 && !STOP_WORDS.has(term));
|
|
47
|
+
}
|
|
48
|
+
/** Adds graph proximity to a file's direct lexical and structural relevance. */
|
|
49
|
+
score(file, terms, profile, direct) {
|
|
50
|
+
const relationshipScore = file.relatedFiles.some((path) => direct.has(path)) ? 3 : 0;
|
|
51
|
+
return this.directScore(file, terms, profile) + relationshipScore;
|
|
52
|
+
}
|
|
53
|
+
/** Scores paths, file text, symbols, and source/test roles independently. */
|
|
54
|
+
directScore(file, terms, profile) {
|
|
55
|
+
const lower = file.path.toLowerCase();
|
|
56
|
+
const pathScore = terms.reduce((score, term) => score + (lower.includes(term) ? 4 : 0), 0);
|
|
57
|
+
const contentScore = terms.reduce((score, term) => score + (file.terms.includes(term) ? 3 : 0), 0);
|
|
58
|
+
const symbolScore = terms.reduce((score, term) => score + (file.symbols.includes(term) ? 6 : 0), 0);
|
|
59
|
+
const sourceScore = profile.sourceRoots.some((root) => lower.startsWith(`${root}/`)) ? 1 : 0;
|
|
60
|
+
const testScore = profile.testRoots.some((root) => lower.startsWith(`${root}/`)) ? 1 : 0;
|
|
61
|
+
const changedScore = profile.changedPaths?.includes(file.path) ? 8 : 0;
|
|
62
|
+
const kindTerms = {
|
|
63
|
+
instruction: ["instruction", "agent", "rule", "convention"],
|
|
64
|
+
manifest: ["dependency", "package", "manifest", "setup", "install"],
|
|
65
|
+
ci: ["ci", "pipeline", "workflow", "action"],
|
|
66
|
+
build: ["build", "docker", "make", "task"],
|
|
67
|
+
documentation: ["documentation", "docs", "readme", "architecture"],
|
|
68
|
+
test: ["test", "spec", "verify"],
|
|
69
|
+
config: ["config", "configuration", "setting"],
|
|
70
|
+
};
|
|
71
|
+
const kindScore = terms.some((term) => kindTerms[file.kind]?.includes(term)) ? 3 : 0;
|
|
72
|
+
return (pathScore + contentScore + symbolScore + sourceScore + testScore + changedScore + kindScore);
|
|
73
|
+
}
|
|
74
|
+
}
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
import { ProviderError } from "../domain/provider-error.js";
|
|
2
|
+
import type { CodingAgent } from "./coding-agent.js";
|
|
3
|
+
/** Captures failure before the caller closes the store, allowing metrics to be read. */
|
|
4
|
+
export declare function runEvaluatedAgent(agent: CodingAgent, sessionId: string, prompt: string, onProgress?: (text: string) => void): Promise<{
|
|
5
|
+
error?: string;
|
|
6
|
+
category?: "agent" | ProviderError["category"];
|
|
7
|
+
}>;
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { ProviderError } from "../domain/provider-error.js";
|
|
2
|
+
/** Captures failure before the caller closes the store, allowing metrics to be read. */
|
|
3
|
+
export async function runEvaluatedAgent(agent, sessionId, prompt, onProgress) {
|
|
4
|
+
try {
|
|
5
|
+
await agent.run(sessionId, prompt, (text) => {
|
|
6
|
+
if (text.startsWith("\n[") && /: (?:retry|stopped retrying)/.test(text))
|
|
7
|
+
onProgress?.(text);
|
|
8
|
+
});
|
|
9
|
+
return {};
|
|
10
|
+
}
|
|
11
|
+
catch (error) {
|
|
12
|
+
return error instanceof ProviderError
|
|
13
|
+
? { error: error.message, category: error.category }
|
|
14
|
+
: { error: "Agent execution failed.", category: "agent" };
|
|
15
|
+
}
|
|
16
|
+
}
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
import type { EvaluationAttempt, EvaluationRun } from "../domain/models.js";
|
|
2
|
+
import type { EvaluationStore } from "../domain/ports.js";
|
|
3
|
+
export declare const comparisonMetrics: readonly ["modelTurns", "toolExecutions", "repairs", "verificationFailures", "durationMs"];
|
|
4
|
+
type Metric = (typeof comparisonMetrics)[number];
|
|
5
|
+
export interface ReliabilitySummary {
|
|
6
|
+
infrastructureFailures: number;
|
|
7
|
+
codingAttempts: number;
|
|
8
|
+
codingPassRate: number | null;
|
|
9
|
+
attempts: number;
|
|
10
|
+
passed: number;
|
|
11
|
+
passRate: number | null;
|
|
12
|
+
averages: Record<Metric, number | null>;
|
|
13
|
+
}
|
|
14
|
+
export interface ReliabilityChange {
|
|
15
|
+
baseline: ReliabilitySummary;
|
|
16
|
+
current: ReliabilitySummary;
|
|
17
|
+
delta: ReliabilitySummary | null;
|
|
18
|
+
}
|
|
19
|
+
export interface EvaluationComparison {
|
|
20
|
+
baselineRun: EvaluationRun;
|
|
21
|
+
currentRun: EvaluationRun;
|
|
22
|
+
comparable: boolean;
|
|
23
|
+
overall: ReliabilityChange;
|
|
24
|
+
scenarios: Array<ReliabilityChange & {
|
|
25
|
+
scenarioId: string;
|
|
26
|
+
}>;
|
|
27
|
+
}
|
|
28
|
+
/** Averages observed attempts; missing data stays unavailable rather than becoming zero. */
|
|
29
|
+
export declare function summarizeAttempts(attempts: EvaluationAttempt[]): ReliabilitySummary;
|
|
30
|
+
/** Compares observed trial distributions, retaining missing scenarios on either side. */
|
|
31
|
+
export declare function compareEvaluations(baselineRun: EvaluationRun, baselineAttempts: EvaluationAttempt[], currentRun: EvaluationRun, currentAttempts: EvaluationAttempt[]): EvaluationComparison;
|
|
32
|
+
/** Loads a saved comparison; no baseline is a normal state for automatic reporting. */
|
|
33
|
+
export declare function compareWithBaseline(store: EvaluationStore, runId: string): EvaluationComparison | undefined;
|
|
34
|
+
export {};
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
export const comparisonMetrics = [
|
|
2
|
+
"modelTurns",
|
|
3
|
+
"toolExecutions",
|
|
4
|
+
"repairs",
|
|
5
|
+
"verificationFailures",
|
|
6
|
+
"durationMs",
|
|
7
|
+
];
|
|
8
|
+
/** Averages observed attempts; missing data stays unavailable rather than becoming zero. */
|
|
9
|
+
export function summarizeAttempts(attempts) {
|
|
10
|
+
const infrastructure = new Set([
|
|
11
|
+
"quota",
|
|
12
|
+
"authentication",
|
|
13
|
+
"network",
|
|
14
|
+
"service",
|
|
15
|
+
"request",
|
|
16
|
+
"setup",
|
|
17
|
+
]);
|
|
18
|
+
const coding = attempts.filter((attempt) => !infrastructure.has(attempt.failureCategory ?? ""));
|
|
19
|
+
const passed = attempts.filter((attempt) => attempt.passed).length;
|
|
20
|
+
const averages = Object.fromEntries(comparisonMetrics.map((metric) => {
|
|
21
|
+
const values = attempts
|
|
22
|
+
.map((attempt) => (metric === "durationMs" ? attempt.durationMs : attempt.metrics[metric]))
|
|
23
|
+
.filter((value) => Number.isFinite(value));
|
|
24
|
+
return [
|
|
25
|
+
metric,
|
|
26
|
+
values.length ? values.reduce((sum, value) => sum + value, 0) / values.length : null,
|
|
27
|
+
];
|
|
28
|
+
}));
|
|
29
|
+
return {
|
|
30
|
+
infrastructureFailures: attempts.length - coding.length,
|
|
31
|
+
codingAttempts: coding.length,
|
|
32
|
+
codingPassRate: coding.length
|
|
33
|
+
? coding.filter((attempt) => attempt.passed).length / coding.length
|
|
34
|
+
: null,
|
|
35
|
+
attempts: attempts.length,
|
|
36
|
+
passed,
|
|
37
|
+
passRate: attempts.length ? passed / attempts.length : null,
|
|
38
|
+
averages,
|
|
39
|
+
};
|
|
40
|
+
}
|
|
41
|
+
/** Compares observed trial distributions, retaining missing scenarios on either side. */
|
|
42
|
+
export function compareEvaluations(baselineRun, baselineAttempts, currentRun, currentAttempts) {
|
|
43
|
+
const comparable = baselineRun.provider === currentRun.provider && baselineRun.model === currentRun.model;
|
|
44
|
+
const difference = (before, after) => before === null || after === null ? null : after - before;
|
|
45
|
+
const change = (before, after) => {
|
|
46
|
+
const baseline = summarizeAttempts(before);
|
|
47
|
+
const current = summarizeAttempts(after);
|
|
48
|
+
return {
|
|
49
|
+
baseline,
|
|
50
|
+
current,
|
|
51
|
+
delta: comparable
|
|
52
|
+
? {
|
|
53
|
+
infrastructureFailures: current.infrastructureFailures - baseline.infrastructureFailures,
|
|
54
|
+
codingAttempts: current.codingAttempts - baseline.codingAttempts,
|
|
55
|
+
codingPassRate: difference(baseline.codingPassRate, current.codingPassRate),
|
|
56
|
+
attempts: current.attempts - baseline.attempts,
|
|
57
|
+
passed: current.passed - baseline.passed,
|
|
58
|
+
passRate: difference(baseline.passRate, current.passRate),
|
|
59
|
+
averages: Object.fromEntries(comparisonMetrics.map((metric) => [
|
|
60
|
+
metric,
|
|
61
|
+
difference(baseline.averages[metric], current.averages[metric]),
|
|
62
|
+
])),
|
|
63
|
+
}
|
|
64
|
+
: null,
|
|
65
|
+
};
|
|
66
|
+
};
|
|
67
|
+
const scenarios = [
|
|
68
|
+
...new Set([...baselineAttempts, ...currentAttempts].map((attempt) => attempt.scenarioId)),
|
|
69
|
+
].sort();
|
|
70
|
+
return {
|
|
71
|
+
baselineRun,
|
|
72
|
+
currentRun,
|
|
73
|
+
comparable,
|
|
74
|
+
overall: change(baselineAttempts, currentAttempts),
|
|
75
|
+
scenarios: scenarios.map((scenarioId) => ({
|
|
76
|
+
scenarioId,
|
|
77
|
+
...change(baselineAttempts.filter((attempt) => attempt.scenarioId === scenarioId), currentAttempts.filter((attempt) => attempt.scenarioId === scenarioId)),
|
|
78
|
+
})),
|
|
79
|
+
};
|
|
80
|
+
}
|
|
81
|
+
/** Loads a saved comparison; no baseline is a normal state for automatic reporting. */
|
|
82
|
+
export function compareWithBaseline(store, runId) {
|
|
83
|
+
const current = store.evaluationRun(runId);
|
|
84
|
+
if (!current)
|
|
85
|
+
throw new Error(`Evaluation run not found: ${runId}`);
|
|
86
|
+
if (current.suite !== "self" || current.completedAt === undefined)
|
|
87
|
+
throw new Error("Comparison requires a completed self-evaluation run.");
|
|
88
|
+
const baseline = store.evaluationBaseline();
|
|
89
|
+
return (baseline &&
|
|
90
|
+
compareEvaluations(baseline, store.evaluationAttempts(baseline.id), current, store.evaluationAttempts(current.id)));
|
|
91
|
+
}
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
import type { EvaluationResult, ModelTurn } from "../domain/models.js";
|
|
2
|
+
import type { JevSafetyAdvisor } from "../domain/ports.js";
|
|
3
|
+
export type EvaluationScenario = {
|
|
4
|
+
id: string;
|
|
5
|
+
prompt: string;
|
|
6
|
+
steps: ModelTurn[];
|
|
7
|
+
expect(workspace: string): Promise<boolean>;
|
|
8
|
+
};
|
|
9
|
+
export declare const evaluationScenarios: EvaluationScenario[];
|
|
10
|
+
/** Runs all fixture cases in disposable copies and returns only the final aggregate evidence. */
|
|
11
|
+
export declare function runEvaluationSuite(): Promise<EvaluationResult[]>;
|
|
12
|
+
export type JevEvaluationReport = {
|
|
13
|
+
off: EvaluationResult[];
|
|
14
|
+
on: EvaluationResult[];
|
|
15
|
+
};
|
|
16
|
+
/** Runs matched deterministic fixtures to isolate Jev decision overhead from model quality. */
|
|
17
|
+
export declare function runJevEvaluationSuite(): Promise<JevEvaluationReport>;
|
|
18
|
+
/** Runs one scenario with isolated storage so benchmark state cannot affect the user workspace. */
|
|
19
|
+
export declare function runScenario(scenario: EvaluationScenario, jev?: JevSafetyAdvisor): Promise<EvaluationResult>;
|
|
@@ -0,0 +1,217 @@
|
|
|
1
|
+
import { cp, mkdtemp, readFile, rm } from "node:fs/promises";
|
|
2
|
+
import { tmpdir } from "node:os";
|
|
3
|
+
import { join } from "node:path";
|
|
4
|
+
import { SqliteSessionStore } from "../infrastructure/persistence/sqlite-session-store.js";
|
|
5
|
+
import { RepositoryProfiler } from "../infrastructure/repository/repository-profiler.js";
|
|
6
|
+
import { WorkspaceTools, definitions } from "../infrastructure/tools/workspace-tools.js";
|
|
7
|
+
import { CodingAgent } from "./coding-agent.js";
|
|
8
|
+
import { taskMetrics } from "./task-metrics.js";
|
|
9
|
+
const fixtures = join(process.cwd(), "evals", "fixtures");
|
|
10
|
+
/** Supplies fixed model decisions so the first benchmark measures agent mechanics reproducibly. */
|
|
11
|
+
class ScenarioProvider {
|
|
12
|
+
steps;
|
|
13
|
+
index = 0;
|
|
14
|
+
constructor(steps) {
|
|
15
|
+
this.steps = steps;
|
|
16
|
+
}
|
|
17
|
+
/** Returns the next deterministic model decision for this benchmark case. */
|
|
18
|
+
async stream(_messages, onText) {
|
|
19
|
+
const step = this.steps[this.index++];
|
|
20
|
+
if (!step)
|
|
21
|
+
return { text: "Scenario finished.", toolCalls: [] };
|
|
22
|
+
if (step.text)
|
|
23
|
+
onText(step.text);
|
|
24
|
+
return step;
|
|
25
|
+
}
|
|
26
|
+
}
|
|
27
|
+
/** Eval runs operate only in a disposable fixture copy, so approvals are deterministic and isolated. */
|
|
28
|
+
class FixtureApproval {
|
|
29
|
+
/** Approves only actions inside the disposable fixture copy created by this harness. */
|
|
30
|
+
async approve(_call, _description) {
|
|
31
|
+
return true;
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
export const evaluationScenarios = [
|
|
35
|
+
{
|
|
36
|
+
id: "create-file",
|
|
37
|
+
prompt: "Create result.txt containing done and verify the project tests.",
|
|
38
|
+
steps: [
|
|
39
|
+
{
|
|
40
|
+
text: "",
|
|
41
|
+
toolCalls: [
|
|
42
|
+
{ id: "write", name: "write_file", args: { path: "result.txt", content: "done\n" } },
|
|
43
|
+
],
|
|
44
|
+
},
|
|
45
|
+
{
|
|
46
|
+
text: "",
|
|
47
|
+
toolCalls: [
|
|
48
|
+
{
|
|
49
|
+
id: "test",
|
|
50
|
+
name: "run_command",
|
|
51
|
+
args: { command: "node --test test.mjs", verification: true },
|
|
52
|
+
},
|
|
53
|
+
],
|
|
54
|
+
},
|
|
55
|
+
{ text: "Created and verified.", toolCalls: [] },
|
|
56
|
+
],
|
|
57
|
+
async expect(workspace) {
|
|
58
|
+
return (await readFile(join(workspace, "result.txt"), "utf8")) === "done\n";
|
|
59
|
+
},
|
|
60
|
+
},
|
|
61
|
+
{
|
|
62
|
+
id: "repair-failing-test",
|
|
63
|
+
prompt: "Fix the failing message test and verify it.",
|
|
64
|
+
steps: [
|
|
65
|
+
{
|
|
66
|
+
text: "",
|
|
67
|
+
toolCalls: [
|
|
68
|
+
{
|
|
69
|
+
id: "first-edit",
|
|
70
|
+
name: "edit_file",
|
|
71
|
+
args: { path: "message.txt", oldText: "bad\n", newText: "still-bad\n" },
|
|
72
|
+
},
|
|
73
|
+
],
|
|
74
|
+
},
|
|
75
|
+
{
|
|
76
|
+
text: "",
|
|
77
|
+
toolCalls: [
|
|
78
|
+
{
|
|
79
|
+
id: "first-test",
|
|
80
|
+
name: "run_command",
|
|
81
|
+
args: { command: "node --test test.mjs", verification: true },
|
|
82
|
+
},
|
|
83
|
+
],
|
|
84
|
+
},
|
|
85
|
+
{
|
|
86
|
+
text: "",
|
|
87
|
+
toolCalls: [
|
|
88
|
+
{
|
|
89
|
+
id: "fix",
|
|
90
|
+
name: "edit_file",
|
|
91
|
+
args: { path: "message.txt", oldText: "still-bad\n", newText: "good\n" },
|
|
92
|
+
},
|
|
93
|
+
],
|
|
94
|
+
},
|
|
95
|
+
{
|
|
96
|
+
text: "",
|
|
97
|
+
toolCalls: [
|
|
98
|
+
{
|
|
99
|
+
id: "second-test",
|
|
100
|
+
name: "run_command",
|
|
101
|
+
args: { command: "node --test test.mjs", verification: true },
|
|
102
|
+
},
|
|
103
|
+
],
|
|
104
|
+
},
|
|
105
|
+
{ text: "Fixed and verified.", toolCalls: [] },
|
|
106
|
+
],
|
|
107
|
+
async expect(workspace) {
|
|
108
|
+
return (await readFile(join(workspace, "message.txt"), "utf8")) === "good\n";
|
|
109
|
+
},
|
|
110
|
+
},
|
|
111
|
+
];
|
|
112
|
+
/** Runs all fixture cases in disposable copies and returns only the final aggregate evidence. */
|
|
113
|
+
export async function runEvaluationSuite() {
|
|
114
|
+
const results = [];
|
|
115
|
+
for (const scenario of evaluationScenarios)
|
|
116
|
+
results.push(await runScenario(scenario));
|
|
117
|
+
return results;
|
|
118
|
+
}
|
|
119
|
+
/** Runs matched deterministic fixtures to isolate Jev decision overhead from model quality. */
|
|
120
|
+
export async function runJevEvaluationSuite() {
|
|
121
|
+
const advisor = {
|
|
122
|
+
async assess() {
|
|
123
|
+
return { risk: "low", confidence: 1 };
|
|
124
|
+
},
|
|
125
|
+
async route() {
|
|
126
|
+
return { value: "build", confidence: 1 };
|
|
127
|
+
},
|
|
128
|
+
async recover() {
|
|
129
|
+
return { value: "repair", confidence: 1 };
|
|
130
|
+
},
|
|
131
|
+
async modelTier() {
|
|
132
|
+
return { value: "balanced", confidence: 1 };
|
|
133
|
+
},
|
|
134
|
+
};
|
|
135
|
+
return {
|
|
136
|
+
off: await runEvaluationSuite(),
|
|
137
|
+
on: await Promise.all(evaluationScenarios.map((scenario) => runScenario(scenario, advisor))),
|
|
138
|
+
};
|
|
139
|
+
}
|
|
140
|
+
/** Runs one scenario with isolated storage so benchmark state cannot affect the user workspace. */
|
|
141
|
+
export async function runScenario(scenario, jev) {
|
|
142
|
+
const root = await mkdtemp(join(tmpdir(), `kairo-eval-${scenario.id}-`));
|
|
143
|
+
try {
|
|
144
|
+
await cp(join(fixtures, scenario.id), root, { recursive: true });
|
|
145
|
+
const store = await SqliteSessionStore.open(":memory:");
|
|
146
|
+
try {
|
|
147
|
+
const session = store.create(root);
|
|
148
|
+
const tools = await WorkspaceTools.create(root);
|
|
149
|
+
store.saveRepositorySnapshot(session.id, await new RepositoryProfiler().profile(root));
|
|
150
|
+
const agent = new CodingAgent(new ScenarioProvider(scenario.steps), store, tools, new FixtureApproval(), definitions, undefined, jev);
|
|
151
|
+
await agent.run(session.id, scenario.prompt, () => { });
|
|
152
|
+
const task = agent.status(session.id);
|
|
153
|
+
const expectationPassed = await scenario.expect(root);
|
|
154
|
+
const metrics = taskMetrics(store.taskEvents(task.id));
|
|
155
|
+
const verified = task.verificationPassed === true;
|
|
156
|
+
return {
|
|
157
|
+
id: scenario.id,
|
|
158
|
+
passed: task.status === "completed" && verified && expectationPassed,
|
|
159
|
+
taskStatus: task.status,
|
|
160
|
+
verified,
|
|
161
|
+
expectationPassed,
|
|
162
|
+
metrics: {
|
|
163
|
+
modelTurns: metrics.modelTurns,
|
|
164
|
+
toolExecutions: metrics.toolExecutions,
|
|
165
|
+
toolFailures: metrics.toolFailures,
|
|
166
|
+
approvals: metrics.approvals,
|
|
167
|
+
repairs: metrics.repairs,
|
|
168
|
+
verificationPasses: metrics.verificationPasses,
|
|
169
|
+
verificationFailures: metrics.verificationFailures,
|
|
170
|
+
verificationSelections: metrics.verificationSelections,
|
|
171
|
+
focusedVerifications: metrics.focusedVerifications,
|
|
172
|
+
broadVerifications: metrics.broadVerifications,
|
|
173
|
+
repairConverged: metrics.repairConverged,
|
|
174
|
+
modelMs: metrics.modelMs,
|
|
175
|
+
toolMs: metrics.toolMs,
|
|
176
|
+
jevDecisions: metrics.jevDecisions,
|
|
177
|
+
jevFailures: metrics.jevFailures,
|
|
178
|
+
jevMs: metrics.jevMs,
|
|
179
|
+
jevRoutes: metrics.jevRoutes,
|
|
180
|
+
jevSafetyChecks: metrics.jevSafetyChecks,
|
|
181
|
+
jevRecoveryChecks: metrics.jevRecoveryChecks,
|
|
182
|
+
},
|
|
183
|
+
};
|
|
184
|
+
}
|
|
185
|
+
finally {
|
|
186
|
+
store.close();
|
|
187
|
+
}
|
|
188
|
+
}
|
|
189
|
+
catch (error) {
|
|
190
|
+
return {
|
|
191
|
+
id: scenario.id,
|
|
192
|
+
passed: false,
|
|
193
|
+
taskStatus: "failed",
|
|
194
|
+
verified: false,
|
|
195
|
+
expectationPassed: false,
|
|
196
|
+
error: error.message,
|
|
197
|
+
metrics: {
|
|
198
|
+
modelTurns: 0,
|
|
199
|
+
toolExecutions: 0,
|
|
200
|
+
toolFailures: 0,
|
|
201
|
+
approvals: 0,
|
|
202
|
+
repairs: 0,
|
|
203
|
+
verificationPasses: 0,
|
|
204
|
+
verificationFailures: 0,
|
|
205
|
+
verificationSelections: 0,
|
|
206
|
+
focusedVerifications: 0,
|
|
207
|
+
broadVerifications: 0,
|
|
208
|
+
repairConverged: false,
|
|
209
|
+
modelMs: 0,
|
|
210
|
+
toolMs: 0,
|
|
211
|
+
},
|
|
212
|
+
};
|
|
213
|
+
}
|
|
214
|
+
finally {
|
|
215
|
+
await rm(root, { recursive: true, force: true });
|
|
216
|
+
}
|
|
217
|
+
}
|