editcodewithai 0.0.1 → 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +57 -9
- package/dist/benchmark.d.ts +37 -0
- package/dist/benchmark.js +123 -0
- package/dist/benchmarks/benchmarkRunner.d.ts +5 -0
- package/dist/benchmarks/benchmarkRunner.js +85 -0
- package/dist/benchmarks/challenges.d.ts +2 -0
- package/dist/benchmarks/challenges.js +158 -0
- package/dist/benchmarks/fileUtils.d.ts +15 -0
- package/dist/benchmarks/fileUtils.js +55 -0
- package/dist/benchmarks/index.d.ts +5 -0
- package/dist/benchmarks/index.js +44 -0
- package/dist/benchmarks/llm.d.ts +10 -0
- package/dist/benchmarks/llm.js +29 -0
- package/dist/benchmarks/models.d.ts +1 -0
- package/dist/benchmarks/models.js +21 -0
- package/dist/benchmarks/runner.d.ts +4 -0
- package/dist/benchmarks/runner.js +35 -0
- package/dist/benchmarks/types.d.ts +16 -0
- package/dist/benchmarks/types.js +2 -0
- package/dist/cli.d.ts +2 -0
- package/dist/cli.js +157 -0
- package/dist/computeInitialDocument.d.ts +13 -0
- package/dist/computeInitialDocument.js +135 -0
- package/dist/fileUtils.d.ts +20 -0
- package/dist/fileUtils.js +62 -0
- package/dist/index.d.ts +3 -19
- package/dist/index.js +19 -139
- package/dist/metadata.d.ts +12 -0
- package/dist/metadata.js +51 -0
- package/dist/prompt.d.ts +8 -0
- package/dist/prompt.js +31 -0
- package/dist/types.d.ts +21 -0
- package/dist/types.js +2 -0
- package/package.json +12 -8
package/README.md
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
# editcodewithai
|
|
2
|
-
|
|
2
|
+
|
|
3
|
+
Edit Code With AI
|
|
3
4
|
|
|
4
5
|
## A Simple Open Source AI Code Editor
|
|
5
6
|
|
|
@@ -21,30 +22,77 @@ npm install
|
|
|
21
22
|
## Usage
|
|
22
23
|
|
|
23
24
|
```typescript
|
|
24
|
-
import { performAiEdit } from
|
|
25
|
+
import { performAiEdit } from "editcodewithai";
|
|
26
|
+
import { VizFiles } from "@vizhub/viz-types";
|
|
27
|
+
|
|
28
|
+
// Define your LLM function that will process the prompt
|
|
29
|
+
const myLlmFunction = async (prompt: string) => {
|
|
30
|
+
// Call your preferred LLM API here
|
|
31
|
+
// This example assumes using OpenRouter
|
|
32
|
+
const response = await fetch(
|
|
33
|
+
"https://openrouter.ai/api/v1/chat/completions",
|
|
34
|
+
{
|
|
35
|
+
method: "POST",
|
|
36
|
+
headers: {
|
|
37
|
+
"Content-Type": "application/json",
|
|
38
|
+
Authorization: `Bearer ${apiKey}`,
|
|
39
|
+
},
|
|
40
|
+
body: JSON.stringify({
|
|
41
|
+
model: "openai/gpt-4",
|
|
42
|
+
messages: [{ role: "user", content: prompt }],
|
|
43
|
+
}),
|
|
44
|
+
}
|
|
45
|
+
);
|
|
46
|
+
|
|
47
|
+
const data = await response.json();
|
|
48
|
+
return {
|
|
49
|
+
content: data.choices[0].message.content,
|
|
50
|
+
generationId: data.id,
|
|
51
|
+
};
|
|
52
|
+
};
|
|
53
|
+
|
|
54
|
+
// Your files
|
|
55
|
+
const files: VizFiles = {
|
|
56
|
+
file1: {
|
|
57
|
+
name: "index.js",
|
|
58
|
+
text: "console.log('Hello world');",
|
|
59
|
+
},
|
|
60
|
+
};
|
|
25
61
|
|
|
62
|
+
// Perform the AI edit
|
|
26
63
|
const result = await performAiEdit({
|
|
27
64
|
prompt: "Update the code to use async/await",
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
// Your files object
|
|
31
|
-
},
|
|
65
|
+
files: files,
|
|
66
|
+
llmFunction: myLlmFunction,
|
|
32
67
|
apiKey: "your-openrouter-api-key",
|
|
33
|
-
baseURL: "https://openrouter.ai/api/v1"
|
|
34
68
|
});
|
|
35
69
|
```
|
|
36
70
|
|
|
37
|
-
|
|
71
|
+
### Parameters
|
|
72
|
+
|
|
73
|
+
The `performAiEdit` function accepts the following parameters:
|
|
74
|
+
|
|
75
|
+
- `prompt`: A string containing instructions for the AI
|
|
76
|
+
- `files`: A `VizFiles` object (map of file IDs to file objects)
|
|
77
|
+
- `llmFunction`: A function that takes a prompt string and returns a Promise with the LLM response
|
|
78
|
+
- `apiKey`: Your OpenRouter API key for retrieving cost metadata
|
|
79
|
+
|
|
80
|
+
Each file in the `files` object should contain:
|
|
81
|
+
|
|
38
82
|
- `name`: The filename (e.g. "index.js")
|
|
39
83
|
- `text`: The file contents as a string
|
|
40
84
|
|
|
41
|
-
|
|
85
|
+
### Return Value
|
|
86
|
+
|
|
87
|
+
The function returns an object with:
|
|
88
|
+
|
|
42
89
|
- `changedFiles`: Updated files with modifications
|
|
43
90
|
- `openRouterGenerationId`: ID of the generation
|
|
44
91
|
- `upstreamCostCents`: Cost in cents
|
|
45
92
|
- `provider`: The AI provider used
|
|
46
93
|
- `inputTokens`: Number of input tokens
|
|
47
94
|
- `outputTokens`: Number of output tokens
|
|
95
|
+
- `promptTemplateVersion`: Version of the prompt template used
|
|
48
96
|
|
|
49
97
|
## Similar open source projects
|
|
50
98
|
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
import { LlmFunction } from "./types";
|
|
2
|
+
/**
|
|
3
|
+
* Creates an OpenRouter LLM function with the provided API key and model using LangChain
|
|
4
|
+
* @param apiKey OpenRouter API key
|
|
5
|
+
* @param model Model to use (defaults to "anthropic/claude-3.5-sonnet")
|
|
6
|
+
* @returns LLM function that can be used with performAiEdit
|
|
7
|
+
*/
|
|
8
|
+
export declare function createOpenRouterLlmFunction(apiKey: string, model?: string): LlmFunction;
|
|
9
|
+
/**
|
|
10
|
+
* Runs a benchmark test using OpenRouter to implement an "add" function
|
|
11
|
+
* @param apiKey OpenRouter API key
|
|
12
|
+
* @param model Optional model to use (defaults to "openai/gpt-4")
|
|
13
|
+
* @returns Benchmark results including the implementation and performance metrics
|
|
14
|
+
*/
|
|
15
|
+
export declare function runAddFunctionBenchmark(apiKey: string, model?: string): Promise<{
|
|
16
|
+
success: boolean;
|
|
17
|
+
implementation: any;
|
|
18
|
+
isCorrect: boolean;
|
|
19
|
+
testOutput: any;
|
|
20
|
+
elapsedTime: number;
|
|
21
|
+
costCents: any;
|
|
22
|
+
provider: any;
|
|
23
|
+
inputTokens: any;
|
|
24
|
+
outputTokens: any;
|
|
25
|
+
error?: undefined;
|
|
26
|
+
} | {
|
|
27
|
+
success: boolean;
|
|
28
|
+
error: string;
|
|
29
|
+
implementation?: undefined;
|
|
30
|
+
isCorrect?: undefined;
|
|
31
|
+
testOutput?: undefined;
|
|
32
|
+
elapsedTime?: undefined;
|
|
33
|
+
costCents?: undefined;
|
|
34
|
+
provider?: undefined;
|
|
35
|
+
inputTokens?: undefined;
|
|
36
|
+
outputTokens?: undefined;
|
|
37
|
+
}>;
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
import { performAiEdit } from "./index";
|
|
2
|
+
import { ChatOpenAI } from "@langchain/openai";
|
|
3
|
+
import { StringOutputParser } from "@langchain/core/output_parsers";
|
|
4
|
+
/**
|
|
5
|
+
* Creates an OpenRouter LLM function with the provided API key and model using LangChain
|
|
6
|
+
* @param apiKey OpenRouter API key
|
|
7
|
+
* @param model Model to use (defaults to "anthropic/claude-3.5-sonnet")
|
|
8
|
+
* @returns LLM function that can be used with performAiEdit
|
|
9
|
+
*/
|
|
10
|
+
export function createOpenRouterLlmFunction(apiKey, model = "anthropic/claude-3.5-sonnet") {
|
|
11
|
+
return async (prompt) => {
|
|
12
|
+
try {
|
|
13
|
+
const options = {
|
|
14
|
+
modelName: model,
|
|
15
|
+
configuration: {
|
|
16
|
+
apiKey,
|
|
17
|
+
baseURL: "https://openrouter.ai/api/v1",
|
|
18
|
+
},
|
|
19
|
+
streaming: false,
|
|
20
|
+
};
|
|
21
|
+
const chatModel = new ChatOpenAI(options);
|
|
22
|
+
const result = await chatModel.invoke(prompt);
|
|
23
|
+
const parser = new StringOutputParser();
|
|
24
|
+
const resultString = await parser.invoke(result);
|
|
25
|
+
return {
|
|
26
|
+
content: resultString,
|
|
27
|
+
generationId: Date.now().toString(), // OpenRouter doesn't return an ID through LangChain
|
|
28
|
+
};
|
|
29
|
+
}
|
|
30
|
+
catch (error) {
|
|
31
|
+
if (error instanceof Error) {
|
|
32
|
+
throw new Error(`OpenRouter API error: ${error.message}`);
|
|
33
|
+
}
|
|
34
|
+
else {
|
|
35
|
+
throw new Error(`OpenRouter API error: ${String(error)}`);
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
};
|
|
39
|
+
}
|
|
40
|
+
/**
|
|
41
|
+
* Runs a benchmark test using OpenRouter to implement an "add" function
|
|
42
|
+
* @param apiKey OpenRouter API key
|
|
43
|
+
* @param model Optional model to use (defaults to "openai/gpt-4")
|
|
44
|
+
* @returns Benchmark results including the implementation and performance metrics
|
|
45
|
+
*/
|
|
46
|
+
export async function runAddFunctionBenchmark(apiKey, model = "openai/gpt-4") {
|
|
47
|
+
console.log(`Running benchmark with model: ${model}`);
|
|
48
|
+
// Create test file with empty add function
|
|
49
|
+
const files = {
|
|
50
|
+
file1: {
|
|
51
|
+
name: "index.js",
|
|
52
|
+
text: "function add(a, b) {\n // TODO: Implement this function\n}\n\nmodule.exports = { add };",
|
|
53
|
+
},
|
|
54
|
+
};
|
|
55
|
+
const prompt = "Implement the 'add' function to add two numbers together and return the result.";
|
|
56
|
+
// Create LLM function
|
|
57
|
+
const llmFunction = createOpenRouterLlmFunction(apiKey, model);
|
|
58
|
+
console.log("Sending request to OpenRouter...");
|
|
59
|
+
const startTime = Date.now();
|
|
60
|
+
try {
|
|
61
|
+
// Perform the AI edit
|
|
62
|
+
const result = await performAiEdit({
|
|
63
|
+
prompt,
|
|
64
|
+
files,
|
|
65
|
+
llmFunction,
|
|
66
|
+
apiKey,
|
|
67
|
+
});
|
|
68
|
+
const endTime = Date.now();
|
|
69
|
+
const elapsedTime = (endTime - startTime) / 1000;
|
|
70
|
+
// Extract the implementation
|
|
71
|
+
const implementation = result.changedFiles.file1?.text || "";
|
|
72
|
+
// Validate the implementation
|
|
73
|
+
let isCorrect = false;
|
|
74
|
+
let testOutput = null;
|
|
75
|
+
try {
|
|
76
|
+
// Create a function from the implementation to test it
|
|
77
|
+
const funcStr = implementation + "\nreturn add(3, 4);";
|
|
78
|
+
const testFunc = new Function(funcStr);
|
|
79
|
+
testOutput = testFunc();
|
|
80
|
+
isCorrect = testOutput === 7;
|
|
81
|
+
}
|
|
82
|
+
catch (error) {
|
|
83
|
+
console.error("Error testing implementation:", error);
|
|
84
|
+
}
|
|
85
|
+
return {
|
|
86
|
+
success: true,
|
|
87
|
+
implementation,
|
|
88
|
+
isCorrect,
|
|
89
|
+
testOutput,
|
|
90
|
+
elapsedTime,
|
|
91
|
+
costCents: result.upstreamCostCents,
|
|
92
|
+
provider: result.provider,
|
|
93
|
+
inputTokens: result.inputTokens,
|
|
94
|
+
outputTokens: result.outputTokens,
|
|
95
|
+
};
|
|
96
|
+
}
|
|
97
|
+
catch (error) {
|
|
98
|
+
console.error("Benchmark failed:", error);
|
|
99
|
+
return {
|
|
100
|
+
success: false,
|
|
101
|
+
error: error instanceof Error ? error.message : String(error),
|
|
102
|
+
};
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
/**
|
|
106
|
+
* Example usage of the benchmark function
|
|
107
|
+
*/
|
|
108
|
+
// Check if this file is being executed directly (CommonJS approach)
|
|
109
|
+
if (require.main === module) {
|
|
110
|
+
// This code runs when the file is executed directly
|
|
111
|
+
const apiKey = process.env.OPENROUTER_API_KEY;
|
|
112
|
+
if (!apiKey) {
|
|
113
|
+
console.error("Please set the OPENROUTER_API_KEY environment variable");
|
|
114
|
+
process.exit(1);
|
|
115
|
+
}
|
|
116
|
+
runAddFunctionBenchmark(apiKey)
|
|
117
|
+
.then((result) => {
|
|
118
|
+
console.log("Benchmark result:", JSON.stringify(result, null, 2));
|
|
119
|
+
})
|
|
120
|
+
.catch((error) => {
|
|
121
|
+
console.error("Error running benchmark:", error);
|
|
122
|
+
});
|
|
123
|
+
}
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.runAllChallenges = runAllChallenges;
|
|
4
|
+
const file_system_1 = require("langchain/cache/file_system");
|
|
5
|
+
const index_1 = require("../index");
|
|
6
|
+
const llm_1 = require("./llm");
|
|
7
|
+
const fileUtils_1 = require("./fileUtils");
|
|
8
|
+
const runner_1 = require("./runner");
|
|
9
|
+
const challenges_1 = require("./challenges");
|
|
10
|
+
const models_1 = require("./models");
|
|
11
|
+
/**
|
|
12
|
+
* Main function: runs the challenges on different models, calls the AI to fill them in,
|
|
13
|
+
* executes each test, and writes pass/fail results to CSV.
|
|
14
|
+
*/
|
|
15
|
+
async function runAllChallenges() {
|
|
16
|
+
const apiKey = process.env.OPENROUTER_API_KEY || "";
|
|
17
|
+
if (!apiKey) {
|
|
18
|
+
throw new Error("Please set OPENROUTER_API_KEY in your .env");
|
|
19
|
+
}
|
|
20
|
+
// Setup cache
|
|
21
|
+
const cacheDirBase = "benchmarks/llmResponseCache";
|
|
22
|
+
(0, fileUtils_1.ensureCacheDir)(cacheDirBase);
|
|
23
|
+
const cache = await file_system_1.LocalFileCache.create(cacheDirBase);
|
|
24
|
+
// Store results
|
|
25
|
+
const results = [];
|
|
26
|
+
// Run each model against each challenge
|
|
27
|
+
for (const model of models_1.models) {
|
|
28
|
+
try {
|
|
29
|
+
await runModelChallenges(model, apiKey, cache, results);
|
|
30
|
+
}
|
|
31
|
+
catch (error) {
|
|
32
|
+
console.error(`Error running model ${model}:`, error);
|
|
33
|
+
// Add failure results for remaining challenges
|
|
34
|
+
for (const challenge of challenges_1.challenges) {
|
|
35
|
+
results.push({
|
|
36
|
+
challenge: challenge.name,
|
|
37
|
+
model,
|
|
38
|
+
passFail: "error",
|
|
39
|
+
});
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
// Write results to CSV
|
|
44
|
+
const csv = (0, fileUtils_1.writeResultsToCsv)(results, "benchmarks/results.csv");
|
|
45
|
+
console.log("\nAll challenges completed. See 'benchmarks/results.csv' for summary.\n");
|
|
46
|
+
console.log(csv);
|
|
47
|
+
}
|
|
48
|
+
/**
|
|
49
|
+
* Runs all challenges for a specific model
|
|
50
|
+
*/
|
|
51
|
+
async function runModelChallenges(model, apiKey, cache, results) {
|
|
52
|
+
// Create the LLM function for this model
|
|
53
|
+
const llmFunction = (0, llm_1.createOpenRouterLlmFunction)(model, apiKey, cache);
|
|
54
|
+
// Run each challenge for this model
|
|
55
|
+
for (const challenge of challenges_1.challenges) {
|
|
56
|
+
console.log(`\n=== Challenge: ${challenge.name} | Model: ${model} ===`);
|
|
57
|
+
try {
|
|
58
|
+
// Ask AI to fill out the placeholder TODOs
|
|
59
|
+
const aiResult = await (0, index_1.performAiEdit)({
|
|
60
|
+
prompt: challenge.prompt,
|
|
61
|
+
files: challenge.files,
|
|
62
|
+
llmFunction,
|
|
63
|
+
});
|
|
64
|
+
// Write returned files to disk, including LLM response
|
|
65
|
+
const challengeDir = (0, fileUtils_1.writeChallengeFiles)(challenge.name, model, aiResult.changedFiles, aiResult.rawResponse);
|
|
66
|
+
// Run index.js in a child process
|
|
67
|
+
const exitCode = (0, runner_1.runNodeTest)(challengeDir);
|
|
68
|
+
const passFail = exitCode === 0 ? "pass" : "fail";
|
|
69
|
+
// Record the result
|
|
70
|
+
results.push({
|
|
71
|
+
challenge: challenge.name,
|
|
72
|
+
model,
|
|
73
|
+
passFail,
|
|
74
|
+
});
|
|
75
|
+
}
|
|
76
|
+
catch (error) {
|
|
77
|
+
console.error(`Error running challenge ${challenge.name}:`, error);
|
|
78
|
+
results.push({
|
|
79
|
+
challenge: challenge.name,
|
|
80
|
+
model,
|
|
81
|
+
passFail: "error",
|
|
82
|
+
});
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
}
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.challenges = void 0;
|
|
4
|
+
// Challenges definition. Each has multiple files (some with TODO).
|
|
5
|
+
// "index.mjs" includes the test code that calls process.exit(1) if it fails.
|
|
6
|
+
exports.challenges = [
|
|
7
|
+
{
|
|
8
|
+
name: "add",
|
|
9
|
+
prompt: "Implement the 'add' function to correctly add two numbers (a+b) and pass the test in index.mjs.",
|
|
10
|
+
files: {
|
|
11
|
+
file1: {
|
|
12
|
+
name: "index.mjs",
|
|
13
|
+
text: `
|
|
14
|
+
import { add } from "./functions.mjs";
|
|
15
|
+
|
|
16
|
+
// A simple test:
|
|
17
|
+
const result = add(3, 4);
|
|
18
|
+
if (result !== 7) {
|
|
19
|
+
console.error("Test failed: expected 7, got", result);
|
|
20
|
+
process.exit(1);
|
|
21
|
+
}
|
|
22
|
+
console.log("Add test passed");
|
|
23
|
+
process.exit(0);
|
|
24
|
+
`,
|
|
25
|
+
},
|
|
26
|
+
file2: {
|
|
27
|
+
name: "functions.mjs",
|
|
28
|
+
text: `
|
|
29
|
+
// TODO: Implement the add function
|
|
30
|
+
export function add(a, b) {
|
|
31
|
+
// TODO
|
|
32
|
+
}
|
|
33
|
+
`,
|
|
34
|
+
},
|
|
35
|
+
},
|
|
36
|
+
},
|
|
37
|
+
{
|
|
38
|
+
name: "multiply",
|
|
39
|
+
prompt: "Implement the 'multiply' function to correctly multiply two numbers and pass the unit test in index.mjs.",
|
|
40
|
+
files: {
|
|
41
|
+
file1: {
|
|
42
|
+
name: "index.mjs",
|
|
43
|
+
text: `
|
|
44
|
+
import { multiply } from "./functions.mjs";
|
|
45
|
+
|
|
46
|
+
const result = multiply(6, 7);
|
|
47
|
+
if (result !== 42) {
|
|
48
|
+
console.error("Test failed: expected 42, got", result);
|
|
49
|
+
process.exit(1);
|
|
50
|
+
}
|
|
51
|
+
console.log("Multiply test passed");
|
|
52
|
+
process.exit(0);
|
|
53
|
+
`,
|
|
54
|
+
},
|
|
55
|
+
file2: {
|
|
56
|
+
name: "functions.mjs",
|
|
57
|
+
text: `
|
|
58
|
+
// TODO: Implement the multiply function
|
|
59
|
+
export function multiply(a, b) {
|
|
60
|
+
// TODO
|
|
61
|
+
}
|
|
62
|
+
`,
|
|
63
|
+
},
|
|
64
|
+
},
|
|
65
|
+
},
|
|
66
|
+
{
|
|
67
|
+
name: "square",
|
|
68
|
+
prompt: "Implement the 'square' function that returns x*x.",
|
|
69
|
+
files: {
|
|
70
|
+
file1: {
|
|
71
|
+
name: "index.mjs",
|
|
72
|
+
text: `
|
|
73
|
+
import { square } from "./functions.mjs";
|
|
74
|
+
|
|
75
|
+
const input = 5;
|
|
76
|
+
const result = square(input);
|
|
77
|
+
if (result !== 25) {
|
|
78
|
+
console.error("Test failed: expected 25, got", result);
|
|
79
|
+
process.exit(1);
|
|
80
|
+
}
|
|
81
|
+
console.log("Square test passed");
|
|
82
|
+
process.exit(0);
|
|
83
|
+
`,
|
|
84
|
+
},
|
|
85
|
+
file2: {
|
|
86
|
+
name: "functions.mjs",
|
|
87
|
+
text: `
|
|
88
|
+
// TODO: Implement the square function
|
|
89
|
+
export function square(x) {
|
|
90
|
+
// TODO
|
|
91
|
+
}
|
|
92
|
+
`,
|
|
93
|
+
},
|
|
94
|
+
},
|
|
95
|
+
},
|
|
96
|
+
{
|
|
97
|
+
name: "toUpperCase",
|
|
98
|
+
prompt: "Implement the toUpperCase function that returns the given string in uppercase.",
|
|
99
|
+
files: {
|
|
100
|
+
file1: {
|
|
101
|
+
name: "index.mjs",
|
|
102
|
+
text: `
|
|
103
|
+
import { toUpperCase } from "./functions.mjs";
|
|
104
|
+
|
|
105
|
+
const input = "hello";
|
|
106
|
+
const result = toUpperCase(input);
|
|
107
|
+
if (result !== "HELLO") {
|
|
108
|
+
console.error("Test failed: expected 'HELLO', got", result);
|
|
109
|
+
process.exit(1);
|
|
110
|
+
}
|
|
111
|
+
console.log("toUpperCase test passed");
|
|
112
|
+
process.exit(0);
|
|
113
|
+
`,
|
|
114
|
+
},
|
|
115
|
+
file2: {
|
|
116
|
+
name: "functions.mjs",
|
|
117
|
+
text: `
|
|
118
|
+
// TODO: Implement the toUpperCase function
|
|
119
|
+
export function toUpperCase(str) {
|
|
120
|
+
// TODO
|
|
121
|
+
}
|
|
122
|
+
`,
|
|
123
|
+
},
|
|
124
|
+
},
|
|
125
|
+
},
|
|
126
|
+
{
|
|
127
|
+
name: "reverseString",
|
|
128
|
+
prompt: "Implement the reverseString function that reverses the given string.",
|
|
129
|
+
files: {
|
|
130
|
+
file1: {
|
|
131
|
+
name: "index.mjs",
|
|
132
|
+
text: `
|
|
133
|
+
import { reverseString } from "./functions.mjs";
|
|
134
|
+
|
|
135
|
+
const input = "OpenAI";
|
|
136
|
+
const expected = "IAnepO";
|
|
137
|
+
|
|
138
|
+
const result = reverseString(input);
|
|
139
|
+
if (result !== expected) {
|
|
140
|
+
console.error("Test failed: expected", expected, "but got", result);
|
|
141
|
+
process.exit(1);
|
|
142
|
+
}
|
|
143
|
+
console.log("reverseString test passed");
|
|
144
|
+
process.exit(0);
|
|
145
|
+
`,
|
|
146
|
+
},
|
|
147
|
+
file2: {
|
|
148
|
+
name: "functions.mjs",
|
|
149
|
+
text: `
|
|
150
|
+
// TODO: Implement the reverseString function
|
|
151
|
+
export function reverseString(str) {
|
|
152
|
+
// TODO
|
|
153
|
+
}
|
|
154
|
+
`,
|
|
155
|
+
},
|
|
156
|
+
},
|
|
157
|
+
},
|
|
158
|
+
];
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
import { VizFiles } from "@vizhub/viz-types";
|
|
2
|
+
import { ChallengeResult } from "./types";
|
|
3
|
+
/**
|
|
4
|
+
* Writes the files for a given challenge into a folder named after challenge.name + model
|
|
5
|
+
* e.g. "benchmarks/challenges/add/gpt4" to keep them separate per model.
|
|
6
|
+
*/
|
|
7
|
+
export declare function writeChallengeFiles(challengeName: string, model: string, changedFiles: VizFiles, llmResponse?: string): string;
|
|
8
|
+
/**
|
|
9
|
+
* Writes results to a CSV file
|
|
10
|
+
*/
|
|
11
|
+
export declare function writeResultsToCsv(results: ChallengeResult[], filePath?: string): string;
|
|
12
|
+
/**
|
|
13
|
+
* Ensures a cache directory exists
|
|
14
|
+
*/
|
|
15
|
+
export declare function ensureCacheDir(cacheDirBase: string): void;
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
var __importDefault = (this && this.__importDefault) || function (mod) {
|
|
3
|
+
return (mod && mod.__esModule) ? mod : { "default": mod };
|
|
4
|
+
};
|
|
5
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
6
|
+
exports.writeChallengeFiles = writeChallengeFiles;
|
|
7
|
+
exports.writeResultsToCsv = writeResultsToCsv;
|
|
8
|
+
exports.ensureCacheDir = ensureCacheDir;
|
|
9
|
+
const fs_1 = __importDefault(require("fs"));
|
|
10
|
+
const path_1 = __importDefault(require("path"));
|
|
11
|
+
/**
|
|
12
|
+
* Writes the files for a given challenge into a folder named after challenge.name + model
|
|
13
|
+
* e.g. "benchmarks/challenges/add/gpt4" to keep them separate per model.
|
|
14
|
+
*/
|
|
15
|
+
function writeChallengeFiles(challengeName, model, changedFiles, llmResponse) {
|
|
16
|
+
const challengeDir = path_1.default.join("benchmarks", "challenges", challengeName, model);
|
|
17
|
+
if (!fs_1.default.existsSync(challengeDir)) {
|
|
18
|
+
fs_1.default.mkdirSync(challengeDir, { recursive: true });
|
|
19
|
+
}
|
|
20
|
+
Object.keys(changedFiles).forEach((fileKey) => {
|
|
21
|
+
const { name, text } = changedFiles[fileKey];
|
|
22
|
+
const filePath = path_1.default.join(challengeDir, name);
|
|
23
|
+
fs_1.default.writeFileSync(filePath, text, "utf-8");
|
|
24
|
+
});
|
|
25
|
+
// Write the LLM response to a file if provided
|
|
26
|
+
if (llmResponse) {
|
|
27
|
+
const responsePath = path_1.default.join(challengeDir, "llmResponse.md");
|
|
28
|
+
fs_1.default.writeFileSync(responsePath, llmResponse, "utf-8");
|
|
29
|
+
}
|
|
30
|
+
return challengeDir;
|
|
31
|
+
}
|
|
32
|
+
/**
|
|
33
|
+
* Writes results to a CSV file
|
|
34
|
+
*/
|
|
35
|
+
function writeResultsToCsv(results, filePath = "benchmarks/results.csv") {
|
|
36
|
+
let csv = "challenge,model,passFail\n";
|
|
37
|
+
for (const r of results) {
|
|
38
|
+
csv += `${r.challenge},${r.model},${r.passFail}\n`;
|
|
39
|
+
}
|
|
40
|
+
// Ensure benchmarks directory exists
|
|
41
|
+
const dir = path_1.default.dirname(filePath);
|
|
42
|
+
if (!fs_1.default.existsSync(dir)) {
|
|
43
|
+
fs_1.default.mkdirSync(dir, { recursive: true });
|
|
44
|
+
}
|
|
45
|
+
fs_1.default.writeFileSync(filePath, csv, "utf-8");
|
|
46
|
+
return csv;
|
|
47
|
+
}
|
|
48
|
+
/**
|
|
49
|
+
* Ensures a cache directory exists
|
|
50
|
+
*/
|
|
51
|
+
function ensureCacheDir(cacheDirBase) {
|
|
52
|
+
if (!fs_1.default.existsSync(cacheDirBase)) {
|
|
53
|
+
fs_1.default.mkdirSync(cacheDirBase, { recursive: true });
|
|
54
|
+
}
|
|
55
|
+
}
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
var __importDefault = (this && this.__importDefault) || function (mod) {
|
|
3
|
+
return (mod && mod.__esModule) ? mod : { "default": mod };
|
|
4
|
+
};
|
|
5
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
6
|
+
exports.writeResultsToCsv = exports.writeChallengeFiles = exports.runNodeTest = exports.createOpenRouterLlmFunction = exports.challenges = exports.runAllChallenges = void 0;
|
|
7
|
+
/**********************************************************************
|
|
8
|
+
* benchmark.ts
|
|
9
|
+
* --------------------------------------------------------------------
|
|
10
|
+
* Requires environment variables to be set in a .env file
|
|
11
|
+
* --------------------------------------------------------------------
|
|
12
|
+
* Defines 5 challenges, each with index.js (ES module) that performs
|
|
13
|
+
* a simple unit test. The code calls process.exit(0) on success or
|
|
14
|
+
* process.exit(1) on failure.
|
|
15
|
+
*
|
|
16
|
+
* Runs all challenges on 5 different models:
|
|
17
|
+
* ["gpt3", "gpt4", "gpt5", "deepseek", "qwen32"]
|
|
18
|
+
*
|
|
19
|
+
* Writes final results to "results.csv" with columns:
|
|
20
|
+
* challenge, model, passFail, exitCode
|
|
21
|
+
**********************************************************************/
|
|
22
|
+
const dotenv_1 = __importDefault(require("dotenv"));
|
|
23
|
+
const benchmarkRunner_1 = require("./benchmarkRunner");
|
|
24
|
+
// Load environment variables from .env file
|
|
25
|
+
dotenv_1.default.config();
|
|
26
|
+
// Run if invoked directly
|
|
27
|
+
if (require.main === module) {
|
|
28
|
+
(0, benchmarkRunner_1.runAllChallenges)().catch((error) => {
|
|
29
|
+
console.error("Error running challenges:", error);
|
|
30
|
+
process.exit(1);
|
|
31
|
+
});
|
|
32
|
+
}
|
|
33
|
+
// Re-export for programmatic usage
|
|
34
|
+
var benchmarkRunner_2 = require("./benchmarkRunner");
|
|
35
|
+
Object.defineProperty(exports, "runAllChallenges", { enumerable: true, get: function () { return benchmarkRunner_2.runAllChallenges; } });
|
|
36
|
+
var challenges_1 = require("./challenges");
|
|
37
|
+
Object.defineProperty(exports, "challenges", { enumerable: true, get: function () { return challenges_1.challenges; } });
|
|
38
|
+
var llm_1 = require("./llm");
|
|
39
|
+
Object.defineProperty(exports, "createOpenRouterLlmFunction", { enumerable: true, get: function () { return llm_1.createOpenRouterLlmFunction; } });
|
|
40
|
+
var runner_1 = require("./runner");
|
|
41
|
+
Object.defineProperty(exports, "runNodeTest", { enumerable: true, get: function () { return runner_1.runNodeTest; } });
|
|
42
|
+
var fileUtils_1 = require("./fileUtils");
|
|
43
|
+
Object.defineProperty(exports, "writeChallengeFiles", { enumerable: true, get: function () { return fileUtils_1.writeChallengeFiles; } });
|
|
44
|
+
Object.defineProperty(exports, "writeResultsToCsv", { enumerable: true, get: function () { return fileUtils_1.writeResultsToCsv; } });
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import { LlmFunction } from "../types";
|
|
2
|
+
import { LocalFileCache } from "langchain/cache/file_system";
|
|
3
|
+
/**
|
|
4
|
+
* Creates an LLM function that connects to OpenRouter
|
|
5
|
+
* @param model The model identifier to use
|
|
6
|
+
* @param apiKey OpenRouter API key
|
|
7
|
+
* @param cache Optional local file cache to avoid duplicate requests
|
|
8
|
+
* @returns A function that can be used with performAiEdit
|
|
9
|
+
*/
|
|
10
|
+
export declare function createOpenRouterLlmFunction(model: string, apiKey: string, cache?: LocalFileCache): LlmFunction;
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.createOpenRouterLlmFunction = createOpenRouterLlmFunction;
|
|
4
|
+
const openai_1 = require("@langchain/openai");
|
|
5
|
+
const output_parsers_1 = require("@langchain/core/output_parsers");
|
|
6
|
+
/**
|
|
7
|
+
* Creates an LLM function that connects to OpenRouter
|
|
8
|
+
* @param model The model identifier to use
|
|
9
|
+
* @param apiKey OpenRouter API key
|
|
10
|
+
* @param cache Optional local file cache to avoid duplicate requests
|
|
11
|
+
* @returns A function that can be used with performAiEdit
|
|
12
|
+
*/
|
|
13
|
+
function createOpenRouterLlmFunction(model, apiKey, cache) {
|
|
14
|
+
return async (prompt) => {
|
|
15
|
+
// Create OpenAI chat model with OpenRouter configuration
|
|
16
|
+
const chatModel = new openai_1.ChatOpenAI({
|
|
17
|
+
modelName: model,
|
|
18
|
+
configuration: { apiKey, baseURL: "https://openrouter.ai/api/v1" },
|
|
19
|
+
streaming: false,
|
|
20
|
+
cache,
|
|
21
|
+
});
|
|
22
|
+
// Invoke the model
|
|
23
|
+
const result = await chatModel.invoke(prompt);
|
|
24
|
+
// Parse to string
|
|
25
|
+
const parser = new output_parsers_1.StringOutputParser();
|
|
26
|
+
const resultString = await parser.invoke(result);
|
|
27
|
+
return { content: resultString, generationId: result.lc_kwargs.id };
|
|
28
|
+
};
|
|
29
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export declare const models: string[];
|