editcodewithai 0.1.0 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +86 -30
- package/dist/fileUtils.d.ts +3 -9
- package/dist/fileUtils.js +15 -11
- package/dist/index.d.ts +2 -1
- package/dist/index.js +2 -2
- package/dist/types.d.ts +8 -7
- package/package.json +19 -9
- package/dist/benchmark.d.ts +0 -37
- package/dist/benchmark.js +0 -123
- package/dist/benchmarks/benchmarkRunner.d.ts +0 -5
- package/dist/benchmarks/benchmarkRunner.js +0 -85
- package/dist/benchmarks/challenges.d.ts +0 -2
- package/dist/benchmarks/challenges.js +0 -158
- package/dist/benchmarks/fileUtils.d.ts +0 -15
- package/dist/benchmarks/fileUtils.js +0 -55
- package/dist/benchmarks/index.d.ts +0 -5
- package/dist/benchmarks/index.js +0 -44
- package/dist/benchmarks/llm.d.ts +0 -10
- package/dist/benchmarks/llm.js +0 -29
- package/dist/benchmarks/models.d.ts +0 -1
- package/dist/benchmarks/models.js +0 -21
- package/dist/benchmarks/runner.d.ts +0 -4
- package/dist/benchmarks/runner.js +0 -35
- package/dist/benchmarks/types.d.ts +0 -16
- package/dist/benchmarks/types.js +0 -2
- package/dist/cli.d.ts +0 -2
- package/dist/cli.js +0 -157
- package/dist/computeInitialDocument.d.ts +0 -13
- package/dist/computeInitialDocument.js +0 -135
package/README.md
CHANGED
|
@@ -1,22 +1,19 @@
|
|
|
1
1
|
# editcodewithai
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
A lightweight, flexible library for AI-powered code editing.
|
|
4
4
|
|
|
5
|
-
|
|
5
|
+
See also [vizhub-benchmarks](https://github.com/vizhub-core/vizhub-benchmarks).
|
|
6
6
|
|
|
7
|
-
|
|
7
|
+
## Overview
|
|
8
8
|
|
|
9
|
-
|
|
9
|
+
`editcodewithai` is a JavaScript/TypeScript library that enables AI-powered code editing in your applications. It provides a simple interface to send code files and instructions to an LLM (Large Language Model) and receive edited code in return.
|
|
10
10
|
|
|
11
|
-
|
|
12
|
-
# Clone the repository
|
|
13
|
-
git clone https://github.com/yourusername/editcodewithai.git
|
|
11
|
+
The library is designed to be model-agnostic, allowing you to use any LLM provider while handling the prompt engineering, file parsing, and response processing for you.
|
|
14
12
|
|
|
15
|
-
|
|
16
|
-
cd editcodewithai
|
|
13
|
+
## Installation
|
|
17
14
|
|
|
18
|
-
|
|
19
|
-
npm install
|
|
15
|
+
```bash
|
|
16
|
+
npm install editcodewithai
|
|
20
17
|
```
|
|
21
18
|
|
|
22
19
|
## Usage
|
|
@@ -38,10 +35,10 @@ const myLlmFunction = async (prompt: string) => {
|
|
|
38
35
|
Authorization: `Bearer ${apiKey}`,
|
|
39
36
|
},
|
|
40
37
|
body: JSON.stringify({
|
|
41
|
-
model: "
|
|
38
|
+
model: "anthropic/claude-3.5-sonnet",
|
|
42
39
|
messages: [{ role: "user", content: prompt }],
|
|
43
40
|
}),
|
|
44
|
-
}
|
|
41
|
+
},
|
|
45
42
|
);
|
|
46
43
|
|
|
47
44
|
const data = await response.json();
|
|
@@ -66,35 +63,58 @@ const result = await performAiEdit({
|
|
|
66
63
|
llmFunction: myLlmFunction,
|
|
67
64
|
apiKey: "your-openrouter-api-key",
|
|
68
65
|
});
|
|
66
|
+
|
|
67
|
+
console.log(result.changedFiles);
|
|
69
68
|
```
|
|
70
69
|
|
|
71
|
-
|
|
70
|
+
## API Reference
|
|
72
71
|
|
|
73
|
-
|
|
72
|
+
### performAiEdit(params)
|
|
74
73
|
|
|
75
|
-
|
|
76
|
-
- `files`: A `VizFiles` object (map of file IDs to file objects)
|
|
77
|
-
- `llmFunction`: A function that takes a prompt string and returns a Promise with the LLM response
|
|
78
|
-
- `apiKey`: Your OpenRouter API key for retrieving cost metadata
|
|
74
|
+
The main function that processes files with an AI model and returns edited code.
|
|
79
75
|
|
|
80
|
-
|
|
76
|
+
#### Parameters
|
|
81
77
|
|
|
82
|
-
|
|
78
|
+
| Parameter | Type | Description |
|
|
79
|
+
| ------------- | ------------- | ----------------------------------------------------------------- |
|
|
80
|
+
| `prompt` | `string` | Instructions for the AI on how to modify the code |
|
|
81
|
+
| `files` | `VizFiles` | Object containing file information (see below) |
|
|
82
|
+
| `llmFunction` | `LlmFunction` | Function that sends the prompt to an LLM and returns the response |
|
|
83
|
+
| `apiKey` | `string` | OpenRouter API key for retrieving cost metadata |
|
|
84
|
+
|
|
85
|
+
The `VizFiles` type is a map of file IDs to file objects, where each file object has:
|
|
86
|
+
|
|
87
|
+
- `name`: The filename (e.g., "index.js")
|
|
83
88
|
- `text`: The file contents as a string
|
|
84
89
|
|
|
85
|
-
|
|
90
|
+
The `LlmFunction` type is a function that takes a prompt string and returns a Promise with:
|
|
91
|
+
|
|
92
|
+
- `content`: The LLM's response text
|
|
93
|
+
- `generationId`: A unique ID for the generation (used for cost tracking)
|
|
94
|
+
|
|
95
|
+
#### Return Value
|
|
86
96
|
|
|
87
97
|
The function returns an object with:
|
|
88
98
|
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
99
|
+
| Property | Type | Description |
|
|
100
|
+
| ------------------------ | ---------- | ------------------------------------------ |
|
|
101
|
+
| `changedFiles` | `VizFiles` | Updated files with AI modifications |
|
|
102
|
+
| `openRouterGenerationId` | `string` | ID of the generation from the LLM provider |
|
|
103
|
+
| `upstreamCostCents` | `number` | Cost of the API call in cents |
|
|
104
|
+
| `provider` | `string` | The AI provider used (e.g., "openai") |
|
|
105
|
+
| `inputTokens` | `number` | Number of input tokens used |
|
|
106
|
+
| `outputTokens` | `number` | Number of output tokens generated |
|
|
107
|
+
| `promptTemplateVersion` | `number` | Version of the prompt template used |
|
|
108
|
+
|
|
109
|
+
### File Operations
|
|
96
110
|
|
|
97
|
-
|
|
111
|
+
The library handles several file operations automatically:
|
|
112
|
+
|
|
113
|
+
- **Updating existing files**: When the AI modifies a file's content
|
|
114
|
+
- **Creating new files**: When the AI suggests new files to add
|
|
115
|
+
- **Deleting files**: When the AI returns empty content for a file
|
|
116
|
+
|
|
117
|
+
## Similar Projects
|
|
98
118
|
|
|
99
119
|
- **Aider**: An AI pair programming tool that integrates with your terminal to assist in code editing within your local git repository. [https://aider.chat/](https://aider.chat/)
|
|
100
120
|
|
|
@@ -109,3 +129,39 @@ The function returns an object with:
|
|
|
109
129
|
- **Void**: An open-source alternative to proprietary AI code editors, offering AI-assisted coding features while prioritizing user privacy and control. [https://void.dev/](https://void.dev/)
|
|
110
130
|
|
|
111
131
|
- **Cody**: An advanced AI coding assistant developed by Sourcegraph, integrating seamlessly with popular IDEs to provide features like AI-driven chat, code autocompletion, and inline editing. [https://github.com/sourcegraph/cody](https://github.com/sourcegraph/cody)
|
|
132
|
+
|
|
133
|
+
## Contributing
|
|
134
|
+
|
|
135
|
+
To contribute to this project:
|
|
136
|
+
|
|
137
|
+
```bash
|
|
138
|
+
# Clone the repository
|
|
139
|
+
git clone https://github.com/yourusername/editcodewithai.git
|
|
140
|
+
|
|
141
|
+
# Navigate to project directory
|
|
142
|
+
cd editcodewithai
|
|
143
|
+
|
|
144
|
+
# Install dependencies
|
|
145
|
+
npm install
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
Run tests to ensure everything is working correctly:
|
|
149
|
+
|
|
150
|
+
```bash
|
|
151
|
+
npm test
|
|
152
|
+
```
|
|
153
|
+
|
|
154
|
+
Please submit pull requests with clear descriptions of changes and ensure all tests pass. Protocol for wrapping up a PR:
|
|
155
|
+
|
|
156
|
+
```
|
|
157
|
+
npm test
|
|
158
|
+
npm run typecheck
|
|
159
|
+
npm run prettier
|
|
160
|
+
# Verify the README is up to date
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
Please create an issue first before creating a PR to discuss the changes you want to make. This helps ensure that your contributions align with the project's goals and vision.
|
|
164
|
+
|
|
165
|
+
## License
|
|
166
|
+
|
|
167
|
+
This project is licensed under the MIT License. See the [LICENSE](LICENSE) file for details.
|
package/dist/fileUtils.d.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { VizFiles, VizFile } from "@vizhub/viz-types";
|
|
1
|
+
import { VizFiles, VizFile, FileCollection } from "@vizhub/viz-types";
|
|
2
2
|
/**
|
|
3
3
|
* If the LLM outputs empty text for a file, we interpret this
|
|
4
4
|
* as a request to delete the file.
|
|
@@ -7,14 +7,8 @@ export declare function shouldDeleteFile(file?: VizFile): boolean;
|
|
|
7
7
|
/**
|
|
8
8
|
* Processes files for the prompt by truncating large files
|
|
9
9
|
*/
|
|
10
|
-
export declare function prepareFilesForPrompt(files: VizFiles):
|
|
11
|
-
name: string;
|
|
12
|
-
text: string;
|
|
13
|
-
}[];
|
|
10
|
+
export declare function prepareFilesForPrompt(files: VizFiles): FileCollection;
|
|
14
11
|
/**
|
|
15
12
|
* Merges original files with changes from the LLM
|
|
16
13
|
*/
|
|
17
|
-
export declare function mergeFileChanges(originalFiles: VizFiles, parsedFiles:
|
|
18
|
-
name: string;
|
|
19
|
-
text: string;
|
|
20
|
-
}[]): VizFiles;
|
|
14
|
+
export declare function mergeFileChanges(originalFiles: VizFiles, parsedFiles: FileCollection): VizFiles;
|
package/dist/fileUtils.js
CHANGED
|
@@ -17,15 +17,16 @@ function shouldDeleteFile(file) {
|
|
|
17
17
|
* Processes files for the prompt by truncating large files
|
|
18
18
|
*/
|
|
19
19
|
function prepareFilesForPrompt(files) {
|
|
20
|
-
|
|
21
|
-
|
|
20
|
+
const result = {};
|
|
21
|
+
Object.values(files).forEach((file) => {
|
|
22
22
|
// Example: truncate large files, etc.
|
|
23
|
-
|
|
23
|
+
result[file.name] = file.text
|
|
24
24
|
.split("\n")
|
|
25
25
|
.slice(0, file.name.endsWith(".csv") || file.name.endsWith(".json") ? 50 : 500)
|
|
26
26
|
.map((line) => line.slice(0, 200))
|
|
27
|
-
.join("\n")
|
|
28
|
-
})
|
|
27
|
+
.join("\n");
|
|
28
|
+
});
|
|
29
|
+
return result;
|
|
29
30
|
}
|
|
30
31
|
/**
|
|
31
32
|
* Merges original files with changes from the LLM
|
|
@@ -34,7 +35,10 @@ function mergeFileChanges(originalFiles, parsedFiles) {
|
|
|
34
35
|
// Start with existing files
|
|
35
36
|
let changedFiles = Object.keys(originalFiles).reduce((acc, fileId) => {
|
|
36
37
|
const original = originalFiles[fileId];
|
|
37
|
-
const
|
|
38
|
+
const changedText = parsedFiles[original.name];
|
|
39
|
+
const changedFile = changedText !== undefined
|
|
40
|
+
? { name: original.name, text: changedText }
|
|
41
|
+
: undefined;
|
|
38
42
|
if (shouldDeleteFile(changedFile)) {
|
|
39
43
|
// Exclude from new set
|
|
40
44
|
return acc;
|
|
@@ -47,14 +51,14 @@ function mergeFileChanges(originalFiles, parsedFiles) {
|
|
|
47
51
|
return acc;
|
|
48
52
|
}, {});
|
|
49
53
|
// Handle newly-created files
|
|
50
|
-
parsedFiles.forEach((
|
|
51
|
-
const existingFile = Object.values(changedFiles).find((file) => file.name ===
|
|
54
|
+
Object.entries(parsedFiles).forEach(([fileName, fileText]) => {
|
|
55
|
+
const existingFile = Object.values(changedFiles).find((file) => file.name === fileName);
|
|
52
56
|
// If no existing file and not empty => it's a new file
|
|
53
|
-
if (!existingFile &&
|
|
57
|
+
if (!existingFile && fileText.trim() !== "") {
|
|
54
58
|
const newFileId = (0, viz_utils_1.generateVizFileId)();
|
|
55
59
|
changedFiles[newFileId] = {
|
|
56
|
-
name:
|
|
57
|
-
text:
|
|
60
|
+
name: fileName,
|
|
61
|
+
text: fileText,
|
|
58
62
|
};
|
|
59
63
|
}
|
|
60
64
|
});
|
package/dist/index.d.ts
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
|
-
import { PerformAiEditParams, PerformAiEditResult } from "./types";
|
|
1
|
+
import type { PerformAiEditParams, PerformAiEditResult } from "./types";
|
|
2
|
+
export type { LlmFunction, PerformAiEditParams, PerformAiEditResult, } from "./types";
|
|
2
3
|
/**
|
|
3
4
|
* Core AI logic for:
|
|
4
5
|
* - Building the prompt (including context)
|
package/dist/index.js
CHANGED
|
@@ -14,9 +14,9 @@ const debug = false;
|
|
|
14
14
|
* - Retrieving cost metadata
|
|
15
15
|
*/
|
|
16
16
|
async function performAiEdit({ prompt, files, llmFunction, apiKey, }) {
|
|
17
|
-
// 1.
|
|
17
|
+
// 1. Format the existing files into the "markdown code block" format
|
|
18
18
|
const preparedFiles = (0, fileUtils_1.prepareFilesForPrompt)(files);
|
|
19
|
-
const filesContext = (0, llm_code_format_1.
|
|
19
|
+
const filesContext = (0, llm_code_format_1.formatMarkdownFiles)(preparedFiles);
|
|
20
20
|
// 2. Assemble the final prompt
|
|
21
21
|
const fullPrompt = (0, prompt_1.assembleFullPrompt)({ filesContext, prompt });
|
|
22
22
|
debug && console.log("[performAiEdit] fullPrompt:", fullPrompt);
|
package/dist/types.d.ts
CHANGED
|
@@ -8,14 +8,15 @@ export interface PerformAiEditParams {
|
|
|
8
8
|
files: VizFiles;
|
|
9
9
|
llmFunction: LlmFunction;
|
|
10
10
|
apiKey?: string;
|
|
11
|
+
baseURL?: string;
|
|
11
12
|
}
|
|
12
13
|
export interface PerformAiEditResult {
|
|
13
14
|
changedFiles: VizFiles;
|
|
14
|
-
openRouterGenerationId
|
|
15
|
-
upstreamCostCents
|
|
16
|
-
provider
|
|
17
|
-
inputTokens
|
|
18
|
-
outputTokens
|
|
19
|
-
promptTemplateVersion
|
|
20
|
-
rawResponse
|
|
15
|
+
openRouterGenerationId?: string;
|
|
16
|
+
upstreamCostCents?: number;
|
|
17
|
+
provider?: string;
|
|
18
|
+
inputTokens?: number;
|
|
19
|
+
outputTokens?: number;
|
|
20
|
+
promptTemplateVersion?: number;
|
|
21
|
+
rawResponse?: string;
|
|
21
22
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "editcodewithai",
|
|
3
|
-
"version": "
|
|
3
|
+
"version": "1.1.0",
|
|
4
4
|
"description": "Edit Code With AI",
|
|
5
5
|
"main": "dist/index.js",
|
|
6
6
|
"types": "dist/index.d.ts",
|
|
@@ -13,7 +13,11 @@
|
|
|
13
13
|
"build": "tsc",
|
|
14
14
|
"prepublishOnly": "npm run build",
|
|
15
15
|
"typecheck": "tsc --noEmit",
|
|
16
|
-
"benchmark": "
|
|
16
|
+
"benchmark": "vite-node src/benchmarks/cli.ts run; cp -r ./benchmarks grader-app/public",
|
|
17
|
+
"grade": "ts-node src/benchmarks/cli.ts grade",
|
|
18
|
+
"benchmark:help": "ts-node src/benchmarks/cli.ts help",
|
|
19
|
+
"upgrade": "ncu -u",
|
|
20
|
+
"prettier": "prettier {*.*,**/*.*} --write"
|
|
17
21
|
},
|
|
18
22
|
"repository": {
|
|
19
23
|
"type": "git",
|
|
@@ -36,16 +40,22 @@
|
|
|
36
40
|
},
|
|
37
41
|
"homepage": "https://github.com/vizhub-core/editcodewithai#readme",
|
|
38
42
|
"dependencies": {
|
|
39
|
-
"@
|
|
40
|
-
"@
|
|
41
|
-
"
|
|
42
|
-
"
|
|
43
|
-
"dotenv": "^16.4.7",
|
|
44
|
-
"llm-code-format": "^0.2.2"
|
|
43
|
+
"@vizhub/viz-types": "^0.1.0",
|
|
44
|
+
"@vizhub/viz-utils": "^1.0.1",
|
|
45
|
+
"dotenv": "^16.5.0",
|
|
46
|
+
"llm-code-format": "^2.0.1"
|
|
45
47
|
},
|
|
46
48
|
"devDependencies": {
|
|
49
|
+
"@langchain/core": "^0.3.44",
|
|
50
|
+
"@langchain/openai": "^0.5.5",
|
|
51
|
+
"langchain": "^0.3.21",
|
|
52
|
+
"cors": "^2.8.5",
|
|
53
|
+
"express": "^5.1.0",
|
|
54
|
+
"npm-check-updates": "^17.1.18",
|
|
55
|
+
"prettier": "^3.5.3",
|
|
56
|
+
"puppeteer": "^24.6.1",
|
|
47
57
|
"ts-node": "^10.9.2",
|
|
48
|
-
"typescript": "^5.8.
|
|
58
|
+
"typescript": "^5.8.3",
|
|
49
59
|
"vitest": "^3.1.1"
|
|
50
60
|
}
|
|
51
61
|
}
|
package/dist/benchmark.d.ts
DELETED
|
@@ -1,37 +0,0 @@
|
|
|
1
|
-
import { LlmFunction } from "./types";
|
|
2
|
-
/**
|
|
3
|
-
* Creates an OpenRouter LLM function with the provided API key and model using LangChain
|
|
4
|
-
* @param apiKey OpenRouter API key
|
|
5
|
-
* @param model Model to use (defaults to "anthropic/claude-3.5-sonnet")
|
|
6
|
-
* @returns LLM function that can be used with performAiEdit
|
|
7
|
-
*/
|
|
8
|
-
export declare function createOpenRouterLlmFunction(apiKey: string, model?: string): LlmFunction;
|
|
9
|
-
/**
|
|
10
|
-
* Runs a benchmark test using OpenRouter to implement an "add" function
|
|
11
|
-
* @param apiKey OpenRouter API key
|
|
12
|
-
* @param model Optional model to use (defaults to "openai/gpt-4")
|
|
13
|
-
* @returns Benchmark results including the implementation and performance metrics
|
|
14
|
-
*/
|
|
15
|
-
export declare function runAddFunctionBenchmark(apiKey: string, model?: string): Promise<{
|
|
16
|
-
success: boolean;
|
|
17
|
-
implementation: any;
|
|
18
|
-
isCorrect: boolean;
|
|
19
|
-
testOutput: any;
|
|
20
|
-
elapsedTime: number;
|
|
21
|
-
costCents: any;
|
|
22
|
-
provider: any;
|
|
23
|
-
inputTokens: any;
|
|
24
|
-
outputTokens: any;
|
|
25
|
-
error?: undefined;
|
|
26
|
-
} | {
|
|
27
|
-
success: boolean;
|
|
28
|
-
error: string;
|
|
29
|
-
implementation?: undefined;
|
|
30
|
-
isCorrect?: undefined;
|
|
31
|
-
testOutput?: undefined;
|
|
32
|
-
elapsedTime?: undefined;
|
|
33
|
-
costCents?: undefined;
|
|
34
|
-
provider?: undefined;
|
|
35
|
-
inputTokens?: undefined;
|
|
36
|
-
outputTokens?: undefined;
|
|
37
|
-
}>;
|
package/dist/benchmark.js
DELETED
|
@@ -1,123 +0,0 @@
|
|
|
1
|
-
import { performAiEdit } from "./index";
|
|
2
|
-
import { ChatOpenAI } from "@langchain/openai";
|
|
3
|
-
import { StringOutputParser } from "@langchain/core/output_parsers";
|
|
4
|
-
/**
|
|
5
|
-
* Creates an OpenRouter LLM function with the provided API key and model using LangChain
|
|
6
|
-
* @param apiKey OpenRouter API key
|
|
7
|
-
* @param model Model to use (defaults to "anthropic/claude-3.5-sonnet")
|
|
8
|
-
* @returns LLM function that can be used with performAiEdit
|
|
9
|
-
*/
|
|
10
|
-
export function createOpenRouterLlmFunction(apiKey, model = "anthropic/claude-3.5-sonnet") {
|
|
11
|
-
return async (prompt) => {
|
|
12
|
-
try {
|
|
13
|
-
const options = {
|
|
14
|
-
modelName: model,
|
|
15
|
-
configuration: {
|
|
16
|
-
apiKey,
|
|
17
|
-
baseURL: "https://openrouter.ai/api/v1",
|
|
18
|
-
},
|
|
19
|
-
streaming: false,
|
|
20
|
-
};
|
|
21
|
-
const chatModel = new ChatOpenAI(options);
|
|
22
|
-
const result = await chatModel.invoke(prompt);
|
|
23
|
-
const parser = new StringOutputParser();
|
|
24
|
-
const resultString = await parser.invoke(result);
|
|
25
|
-
return {
|
|
26
|
-
content: resultString,
|
|
27
|
-
generationId: Date.now().toString(), // OpenRouter doesn't return an ID through LangChain
|
|
28
|
-
};
|
|
29
|
-
}
|
|
30
|
-
catch (error) {
|
|
31
|
-
if (error instanceof Error) {
|
|
32
|
-
throw new Error(`OpenRouter API error: ${error.message}`);
|
|
33
|
-
}
|
|
34
|
-
else {
|
|
35
|
-
throw new Error(`OpenRouter API error: ${String(error)}`);
|
|
36
|
-
}
|
|
37
|
-
}
|
|
38
|
-
};
|
|
39
|
-
}
|
|
40
|
-
/**
|
|
41
|
-
* Runs a benchmark test using OpenRouter to implement an "add" function
|
|
42
|
-
* @param apiKey OpenRouter API key
|
|
43
|
-
* @param model Optional model to use (defaults to "openai/gpt-4")
|
|
44
|
-
* @returns Benchmark results including the implementation and performance metrics
|
|
45
|
-
*/
|
|
46
|
-
export async function runAddFunctionBenchmark(apiKey, model = "openai/gpt-4") {
|
|
47
|
-
console.log(`Running benchmark with model: ${model}`);
|
|
48
|
-
// Create test file with empty add function
|
|
49
|
-
const files = {
|
|
50
|
-
file1: {
|
|
51
|
-
name: "index.js",
|
|
52
|
-
text: "function add(a, b) {\n // TODO: Implement this function\n}\n\nmodule.exports = { add };",
|
|
53
|
-
},
|
|
54
|
-
};
|
|
55
|
-
const prompt = "Implement the 'add' function to add two numbers together and return the result.";
|
|
56
|
-
// Create LLM function
|
|
57
|
-
const llmFunction = createOpenRouterLlmFunction(apiKey, model);
|
|
58
|
-
console.log("Sending request to OpenRouter...");
|
|
59
|
-
const startTime = Date.now();
|
|
60
|
-
try {
|
|
61
|
-
// Perform the AI edit
|
|
62
|
-
const result = await performAiEdit({
|
|
63
|
-
prompt,
|
|
64
|
-
files,
|
|
65
|
-
llmFunction,
|
|
66
|
-
apiKey,
|
|
67
|
-
});
|
|
68
|
-
const endTime = Date.now();
|
|
69
|
-
const elapsedTime = (endTime - startTime) / 1000;
|
|
70
|
-
// Extract the implementation
|
|
71
|
-
const implementation = result.changedFiles.file1?.text || "";
|
|
72
|
-
// Validate the implementation
|
|
73
|
-
let isCorrect = false;
|
|
74
|
-
let testOutput = null;
|
|
75
|
-
try {
|
|
76
|
-
// Create a function from the implementation to test it
|
|
77
|
-
const funcStr = implementation + "\nreturn add(3, 4);";
|
|
78
|
-
const testFunc = new Function(funcStr);
|
|
79
|
-
testOutput = testFunc();
|
|
80
|
-
isCorrect = testOutput === 7;
|
|
81
|
-
}
|
|
82
|
-
catch (error) {
|
|
83
|
-
console.error("Error testing implementation:", error);
|
|
84
|
-
}
|
|
85
|
-
return {
|
|
86
|
-
success: true,
|
|
87
|
-
implementation,
|
|
88
|
-
isCorrect,
|
|
89
|
-
testOutput,
|
|
90
|
-
elapsedTime,
|
|
91
|
-
costCents: result.upstreamCostCents,
|
|
92
|
-
provider: result.provider,
|
|
93
|
-
inputTokens: result.inputTokens,
|
|
94
|
-
outputTokens: result.outputTokens,
|
|
95
|
-
};
|
|
96
|
-
}
|
|
97
|
-
catch (error) {
|
|
98
|
-
console.error("Benchmark failed:", error);
|
|
99
|
-
return {
|
|
100
|
-
success: false,
|
|
101
|
-
error: error instanceof Error ? error.message : String(error),
|
|
102
|
-
};
|
|
103
|
-
}
|
|
104
|
-
}
|
|
105
|
-
/**
|
|
106
|
-
* Example usage of the benchmark function
|
|
107
|
-
*/
|
|
108
|
-
// Check if this file is being executed directly (CommonJS approach)
|
|
109
|
-
if (require.main === module) {
|
|
110
|
-
// This code runs when the file is executed directly
|
|
111
|
-
const apiKey = process.env.OPENROUTER_API_KEY;
|
|
112
|
-
if (!apiKey) {
|
|
113
|
-
console.error("Please set the OPENROUTER_API_KEY environment variable");
|
|
114
|
-
process.exit(1);
|
|
115
|
-
}
|
|
116
|
-
runAddFunctionBenchmark(apiKey)
|
|
117
|
-
.then((result) => {
|
|
118
|
-
console.log("Benchmark result:", JSON.stringify(result, null, 2));
|
|
119
|
-
})
|
|
120
|
-
.catch((error) => {
|
|
121
|
-
console.error("Error running benchmark:", error);
|
|
122
|
-
});
|
|
123
|
-
}
|
|
@@ -1,85 +0,0 @@
|
|
|
1
|
-
"use strict";
|
|
2
|
-
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
-
exports.runAllChallenges = runAllChallenges;
|
|
4
|
-
const file_system_1 = require("langchain/cache/file_system");
|
|
5
|
-
const index_1 = require("../index");
|
|
6
|
-
const llm_1 = require("./llm");
|
|
7
|
-
const fileUtils_1 = require("./fileUtils");
|
|
8
|
-
const runner_1 = require("./runner");
|
|
9
|
-
const challenges_1 = require("./challenges");
|
|
10
|
-
const models_1 = require("./models");
|
|
11
|
-
/**
|
|
12
|
-
* Main function: runs the challenges on different models, calls the AI to fill them in,
|
|
13
|
-
* executes each test, and writes pass/fail results to CSV.
|
|
14
|
-
*/
|
|
15
|
-
async function runAllChallenges() {
|
|
16
|
-
const apiKey = process.env.OPENROUTER_API_KEY || "";
|
|
17
|
-
if (!apiKey) {
|
|
18
|
-
throw new Error("Please set OPENROUTER_API_KEY in your .env");
|
|
19
|
-
}
|
|
20
|
-
// Setup cache
|
|
21
|
-
const cacheDirBase = "benchmarks/llmResponseCache";
|
|
22
|
-
(0, fileUtils_1.ensureCacheDir)(cacheDirBase);
|
|
23
|
-
const cache = await file_system_1.LocalFileCache.create(cacheDirBase);
|
|
24
|
-
// Store results
|
|
25
|
-
const results = [];
|
|
26
|
-
// Run each model against each challenge
|
|
27
|
-
for (const model of models_1.models) {
|
|
28
|
-
try {
|
|
29
|
-
await runModelChallenges(model, apiKey, cache, results);
|
|
30
|
-
}
|
|
31
|
-
catch (error) {
|
|
32
|
-
console.error(`Error running model ${model}:`, error);
|
|
33
|
-
// Add failure results for remaining challenges
|
|
34
|
-
for (const challenge of challenges_1.challenges) {
|
|
35
|
-
results.push({
|
|
36
|
-
challenge: challenge.name,
|
|
37
|
-
model,
|
|
38
|
-
passFail: "error",
|
|
39
|
-
});
|
|
40
|
-
}
|
|
41
|
-
}
|
|
42
|
-
}
|
|
43
|
-
// Write results to CSV
|
|
44
|
-
const csv = (0, fileUtils_1.writeResultsToCsv)(results, "benchmarks/results.csv");
|
|
45
|
-
console.log("\nAll challenges completed. See 'benchmarks/results.csv' for summary.\n");
|
|
46
|
-
console.log(csv);
|
|
47
|
-
}
|
|
48
|
-
/**
|
|
49
|
-
* Runs all challenges for a specific model
|
|
50
|
-
*/
|
|
51
|
-
async function runModelChallenges(model, apiKey, cache, results) {
|
|
52
|
-
// Create the LLM function for this model
|
|
53
|
-
const llmFunction = (0, llm_1.createOpenRouterLlmFunction)(model, apiKey, cache);
|
|
54
|
-
// Run each challenge for this model
|
|
55
|
-
for (const challenge of challenges_1.challenges) {
|
|
56
|
-
console.log(`\n=== Challenge: ${challenge.name} | Model: ${model} ===`);
|
|
57
|
-
try {
|
|
58
|
-
// Ask AI to fill out the placeholder TODOs
|
|
59
|
-
const aiResult = await (0, index_1.performAiEdit)({
|
|
60
|
-
prompt: challenge.prompt,
|
|
61
|
-
files: challenge.files,
|
|
62
|
-
llmFunction,
|
|
63
|
-
});
|
|
64
|
-
// Write returned files to disk, including LLM response
|
|
65
|
-
const challengeDir = (0, fileUtils_1.writeChallengeFiles)(challenge.name, model, aiResult.changedFiles, aiResult.rawResponse);
|
|
66
|
-
// Run index.js in a child process
|
|
67
|
-
const exitCode = (0, runner_1.runNodeTest)(challengeDir);
|
|
68
|
-
const passFail = exitCode === 0 ? "pass" : "fail";
|
|
69
|
-
// Record the result
|
|
70
|
-
results.push({
|
|
71
|
-
challenge: challenge.name,
|
|
72
|
-
model,
|
|
73
|
-
passFail,
|
|
74
|
-
});
|
|
75
|
-
}
|
|
76
|
-
catch (error) {
|
|
77
|
-
console.error(`Error running challenge ${challenge.name}:`, error);
|
|
78
|
-
results.push({
|
|
79
|
-
challenge: challenge.name,
|
|
80
|
-
model,
|
|
81
|
-
passFail: "error",
|
|
82
|
-
});
|
|
83
|
-
}
|
|
84
|
-
}
|
|
85
|
-
}
|