editcodewithai 0.1.0 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +83 -29
- package/dist/fileUtils.d.ts +3 -9
- package/dist/fileUtils.js +15 -11
- package/dist/index.js +2 -2
- package/dist/types.d.ts +8 -7
- package/package.json +19 -7
- package/dist/benchmark.d.ts +0 -37
- package/dist/benchmark.js +0 -123
- package/dist/benchmarks/benchmarkRunner.d.ts +0 -5
- package/dist/benchmarks/benchmarkRunner.js +0 -85
- package/dist/benchmarks/challenges.d.ts +0 -2
- package/dist/benchmarks/challenges.js +0 -158
- package/dist/benchmarks/fileUtils.d.ts +0 -15
- package/dist/benchmarks/fileUtils.js +0 -55
- package/dist/benchmarks/index.d.ts +0 -5
- package/dist/benchmarks/index.js +0 -44
- package/dist/benchmarks/llm.d.ts +0 -10
- package/dist/benchmarks/llm.js +0 -29
- package/dist/benchmarks/models.d.ts +0 -1
- package/dist/benchmarks/models.js +0 -21
- package/dist/benchmarks/runner.d.ts +0 -4
- package/dist/benchmarks/runner.js +0 -35
- package/dist/benchmarks/types.d.ts +0 -16
- package/dist/benchmarks/types.js +0 -2
- package/dist/cli.d.ts +0 -2
- package/dist/cli.js +0 -157
- package/dist/computeInitialDocument.d.ts +0 -13
- package/dist/computeInitialDocument.js +0 -135
package/README.md
CHANGED
|
@@ -1,22 +1,17 @@
|
|
|
1
1
|
# editcodewithai
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
A lightweight, flexible library for AI-powered code editing.
|
|
4
4
|
|
|
5
|
-
##
|
|
5
|
+
## Overview
|
|
6
6
|
|
|
7
|
-
|
|
7
|
+
`editcodewithai` is a JavaScript/TypeScript library that enables AI-powered code editing in your applications. It provides a simple interface to send code files and instructions to an LLM (Large Language Model) and receive edited code in return.
|
|
8
|
+
|
|
9
|
+
The library is designed to be model-agnostic, allowing you to use any LLM provider while handling the prompt engineering, file parsing, and response processing for you.
|
|
8
10
|
|
|
9
11
|
## Installation
|
|
10
12
|
|
|
11
13
|
```bash
|
|
12
|
-
|
|
13
|
-
git clone https://github.com/yourusername/editcodewithai.git
|
|
14
|
-
|
|
15
|
-
# Navigate to project directory
|
|
16
|
-
cd editcodewithai
|
|
17
|
-
|
|
18
|
-
# Install dependencies
|
|
19
|
-
npm install
|
|
14
|
+
npm install editcodewithai
|
|
20
15
|
```
|
|
21
16
|
|
|
22
17
|
## Usage
|
|
@@ -41,7 +36,7 @@ const myLlmFunction = async (prompt: string) => {
|
|
|
41
36
|
model: "openai/gpt-4",
|
|
42
37
|
messages: [{ role: "user", content: prompt }],
|
|
43
38
|
}),
|
|
44
|
-
}
|
|
39
|
+
},
|
|
45
40
|
);
|
|
46
41
|
|
|
47
42
|
const data = await response.json();
|
|
@@ -66,35 +61,58 @@ const result = await performAiEdit({
|
|
|
66
61
|
llmFunction: myLlmFunction,
|
|
67
62
|
apiKey: "your-openrouter-api-key",
|
|
68
63
|
});
|
|
64
|
+
|
|
65
|
+
console.log(result.changedFiles);
|
|
69
66
|
```
|
|
70
67
|
|
|
71
|
-
|
|
68
|
+
## API Reference
|
|
69
|
+
|
|
70
|
+
### performAiEdit(params)
|
|
72
71
|
|
|
73
|
-
The
|
|
72
|
+
The main function that processes files with an AI model and returns edited code.
|
|
74
73
|
|
|
75
|
-
|
|
76
|
-
- `files`: A `VizFiles` object (map of file IDs to file objects)
|
|
77
|
-
- `llmFunction`: A function that takes a prompt string and returns a Promise with the LLM response
|
|
78
|
-
- `apiKey`: Your OpenRouter API key for retrieving cost metadata
|
|
74
|
+
#### Parameters
|
|
79
75
|
|
|
80
|
-
|
|
76
|
+
| Parameter | Type | Description |
|
|
77
|
+
| ------------- | ------------- | ----------------------------------------------------------------- |
|
|
78
|
+
| `prompt` | `string` | Instructions for the AI on how to modify the code |
|
|
79
|
+
| `files` | `VizFiles` | Object containing file information (see below) |
|
|
80
|
+
| `llmFunction` | `LlmFunction` | Function that sends the prompt to an LLM and returns the response |
|
|
81
|
+
| `apiKey` | `string` | OpenRouter API key for retrieving cost metadata |
|
|
81
82
|
|
|
82
|
-
|
|
83
|
+
The `VizFiles` type is a map of file IDs to file objects, where each file object has:
|
|
84
|
+
|
|
85
|
+
- `name`: The filename (e.g., "index.js")
|
|
83
86
|
- `text`: The file contents as a string
|
|
84
87
|
|
|
85
|
-
|
|
88
|
+
The `LlmFunction` type is a function that takes a prompt string and returns a Promise with:
|
|
89
|
+
|
|
90
|
+
- `content`: The LLM's response text
|
|
91
|
+
- `generationId`: A unique ID for the generation (used for cost tracking)
|
|
92
|
+
|
|
93
|
+
#### Return Value
|
|
86
94
|
|
|
87
95
|
The function returns an object with:
|
|
88
96
|
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
97
|
+
| Property | Type | Description |
|
|
98
|
+
| ------------------------ | ---------- | ------------------------------------------ |
|
|
99
|
+
| `changedFiles` | `VizFiles` | Updated files with AI modifications |
|
|
100
|
+
| `openRouterGenerationId` | `string` | ID of the generation from the LLM provider |
|
|
101
|
+
| `upstreamCostCents` | `number` | Cost of the API call in cents |
|
|
102
|
+
| `provider` | `string` | The AI provider used (e.g., "openai") |
|
|
103
|
+
| `inputTokens` | `number` | Number of input tokens used |
|
|
104
|
+
| `outputTokens` | `number` | Number of output tokens generated |
|
|
105
|
+
| `promptTemplateVersion` | `number` | Version of the prompt template used |
|
|
106
|
+
|
|
107
|
+
### File Operations
|
|
108
|
+
|
|
109
|
+
The library handles several file operations automatically:
|
|
96
110
|
|
|
97
|
-
|
|
111
|
+
- **Updating existing files**: When the AI modifies a file's content
|
|
112
|
+
- **Creating new files**: When the AI suggests new files to add
|
|
113
|
+
- **Deleting files**: When the AI returns empty content for a file
|
|
114
|
+
|
|
115
|
+
## Similar Projects
|
|
98
116
|
|
|
99
117
|
- **Aider**: An AI pair programming tool that integrates with your terminal to assist in code editing within your local git repository. [https://aider.chat/](https://aider.chat/)
|
|
100
118
|
|
|
@@ -109,3 +127,39 @@ The function returns an object with:
|
|
|
109
127
|
- **Void**: An open-source alternative to proprietary AI code editors, offering AI-assisted coding features while prioritizing user privacy and control. [https://void.dev/](https://void.dev/)
|
|
110
128
|
|
|
111
129
|
- **Cody**: An advanced AI coding assistant developed by Sourcegraph, integrating seamlessly with popular IDEs to provide features like AI-driven chat, code autocompletion, and inline editing. [https://github.com/sourcegraph/cody](https://github.com/sourcegraph/cody)
|
|
130
|
+
|
|
131
|
+
## Contributing
|
|
132
|
+
|
|
133
|
+
To contribute to this project:
|
|
134
|
+
|
|
135
|
+
```bash
|
|
136
|
+
# Clone the repository
|
|
137
|
+
git clone https://github.com/yourusername/editcodewithai.git
|
|
138
|
+
|
|
139
|
+
# Navigate to project directory
|
|
140
|
+
cd editcodewithai
|
|
141
|
+
|
|
142
|
+
# Install dependencies
|
|
143
|
+
npm install
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
Run tests to ensure everything is working correctly:
|
|
147
|
+
|
|
148
|
+
```bash
|
|
149
|
+
npm test
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
Please submit pull requests with clear descriptions of changes and ensure all tests pass. Protocol for wrapping up a PR:
|
|
153
|
+
|
|
154
|
+
```
|
|
155
|
+
npm test
|
|
156
|
+
npm run typecheck
|
|
157
|
+
npm run prettier
|
|
158
|
+
# Verify the README is up to date
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
Please create an issue first before creating a PR to discuss the changes you want to make. This helps ensure that your contributions align with the project's goals and vision.
|
|
162
|
+
|
|
163
|
+
## License
|
|
164
|
+
|
|
165
|
+
This project is licensed under the MIT License. See the [LICENSE](LICENSE) file for details.
|
package/dist/fileUtils.d.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { VizFiles, VizFile } from "@vizhub/viz-types";
|
|
1
|
+
import { VizFiles, VizFile, FileCollection } from "@vizhub/viz-types";
|
|
2
2
|
/**
|
|
3
3
|
* If the LLM outputs empty text for a file, we interpret this
|
|
4
4
|
* as a request to delete the file.
|
|
@@ -7,14 +7,8 @@ export declare function shouldDeleteFile(file?: VizFile): boolean;
|
|
|
7
7
|
/**
|
|
8
8
|
* Processes files for the prompt by truncating large files
|
|
9
9
|
*/
|
|
10
|
-
export declare function prepareFilesForPrompt(files: VizFiles):
|
|
11
|
-
name: string;
|
|
12
|
-
text: string;
|
|
13
|
-
}[];
|
|
10
|
+
export declare function prepareFilesForPrompt(files: VizFiles): FileCollection;
|
|
14
11
|
/**
|
|
15
12
|
* Merges original files with changes from the LLM
|
|
16
13
|
*/
|
|
17
|
-
export declare function mergeFileChanges(originalFiles: VizFiles, parsedFiles:
|
|
18
|
-
name: string;
|
|
19
|
-
text: string;
|
|
20
|
-
}[]): VizFiles;
|
|
14
|
+
export declare function mergeFileChanges(originalFiles: VizFiles, parsedFiles: FileCollection): VizFiles;
|
package/dist/fileUtils.js
CHANGED
|
@@ -17,15 +17,16 @@ function shouldDeleteFile(file) {
|
|
|
17
17
|
* Processes files for the prompt by truncating large files
|
|
18
18
|
*/
|
|
19
19
|
function prepareFilesForPrompt(files) {
|
|
20
|
-
|
|
21
|
-
|
|
20
|
+
const result = {};
|
|
21
|
+
Object.values(files).forEach((file) => {
|
|
22
22
|
// Example: truncate large files, etc.
|
|
23
|
-
|
|
23
|
+
result[file.name] = file.text
|
|
24
24
|
.split("\n")
|
|
25
25
|
.slice(0, file.name.endsWith(".csv") || file.name.endsWith(".json") ? 50 : 500)
|
|
26
26
|
.map((line) => line.slice(0, 200))
|
|
27
|
-
.join("\n")
|
|
28
|
-
})
|
|
27
|
+
.join("\n");
|
|
28
|
+
});
|
|
29
|
+
return result;
|
|
29
30
|
}
|
|
30
31
|
/**
|
|
31
32
|
* Merges original files with changes from the LLM
|
|
@@ -34,7 +35,10 @@ function mergeFileChanges(originalFiles, parsedFiles) {
|
|
|
34
35
|
// Start with existing files
|
|
35
36
|
let changedFiles = Object.keys(originalFiles).reduce((acc, fileId) => {
|
|
36
37
|
const original = originalFiles[fileId];
|
|
37
|
-
const
|
|
38
|
+
const changedText = parsedFiles[original.name];
|
|
39
|
+
const changedFile = changedText !== undefined
|
|
40
|
+
? { name: original.name, text: changedText }
|
|
41
|
+
: undefined;
|
|
38
42
|
if (shouldDeleteFile(changedFile)) {
|
|
39
43
|
// Exclude from new set
|
|
40
44
|
return acc;
|
|
@@ -47,14 +51,14 @@ function mergeFileChanges(originalFiles, parsedFiles) {
|
|
|
47
51
|
return acc;
|
|
48
52
|
}, {});
|
|
49
53
|
// Handle newly-created files
|
|
50
|
-
parsedFiles.forEach((
|
|
51
|
-
const existingFile = Object.values(changedFiles).find((file) => file.name ===
|
|
54
|
+
Object.entries(parsedFiles).forEach(([fileName, fileText]) => {
|
|
55
|
+
const existingFile = Object.values(changedFiles).find((file) => file.name === fileName);
|
|
52
56
|
// If no existing file and not empty => it's a new file
|
|
53
|
-
if (!existingFile &&
|
|
57
|
+
if (!existingFile && fileText.trim() !== "") {
|
|
54
58
|
const newFileId = (0, viz_utils_1.generateVizFileId)();
|
|
55
59
|
changedFiles[newFileId] = {
|
|
56
|
-
name:
|
|
57
|
-
text:
|
|
60
|
+
name: fileName,
|
|
61
|
+
text: fileText,
|
|
58
62
|
};
|
|
59
63
|
}
|
|
60
64
|
});
|
package/dist/index.js
CHANGED
|
@@ -14,9 +14,9 @@ const debug = false;
|
|
|
14
14
|
* - Retrieving cost metadata
|
|
15
15
|
*/
|
|
16
16
|
async function performAiEdit({ prompt, files, llmFunction, apiKey, }) {
|
|
17
|
-
// 1.
|
|
17
|
+
// 1. Format the existing files into the "markdown code block" format
|
|
18
18
|
const preparedFiles = (0, fileUtils_1.prepareFilesForPrompt)(files);
|
|
19
|
-
const filesContext = (0, llm_code_format_1.
|
|
19
|
+
const filesContext = (0, llm_code_format_1.formatMarkdownFiles)(preparedFiles);
|
|
20
20
|
// 2. Assemble the final prompt
|
|
21
21
|
const fullPrompt = (0, prompt_1.assembleFullPrompt)({ filesContext, prompt });
|
|
22
22
|
debug && console.log("[performAiEdit] fullPrompt:", fullPrompt);
|
package/dist/types.d.ts
CHANGED
|
@@ -8,14 +8,15 @@ export interface PerformAiEditParams {
|
|
|
8
8
|
files: VizFiles;
|
|
9
9
|
llmFunction: LlmFunction;
|
|
10
10
|
apiKey?: string;
|
|
11
|
+
baseURL?: string;
|
|
11
12
|
}
|
|
12
13
|
export interface PerformAiEditResult {
|
|
13
14
|
changedFiles: VizFiles;
|
|
14
|
-
openRouterGenerationId
|
|
15
|
-
upstreamCostCents
|
|
16
|
-
provider
|
|
17
|
-
inputTokens
|
|
18
|
-
outputTokens
|
|
19
|
-
promptTemplateVersion
|
|
20
|
-
rawResponse
|
|
15
|
+
openRouterGenerationId?: string;
|
|
16
|
+
upstreamCostCents?: number;
|
|
17
|
+
provider?: string;
|
|
18
|
+
inputTokens?: number;
|
|
19
|
+
outputTokens?: number;
|
|
20
|
+
promptTemplateVersion?: number;
|
|
21
|
+
rawResponse?: string;
|
|
21
22
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "editcodewithai",
|
|
3
|
-
"version": "
|
|
3
|
+
"version": "1.0.0",
|
|
4
4
|
"description": "Edit Code With AI",
|
|
5
5
|
"main": "dist/index.js",
|
|
6
6
|
"types": "dist/index.d.ts",
|
|
@@ -13,7 +13,11 @@
|
|
|
13
13
|
"build": "tsc",
|
|
14
14
|
"prepublishOnly": "npm run build",
|
|
15
15
|
"typecheck": "tsc --noEmit",
|
|
16
|
-
"benchmark": "ts-node src/benchmarks/
|
|
16
|
+
"benchmark": "ts-node src/benchmarks/cli.ts run; cp -r ./benchmarks grader-app/public",
|
|
17
|
+
"grade": "ts-node src/benchmarks/cli.ts grade",
|
|
18
|
+
"benchmark:help": "ts-node src/benchmarks/cli.ts help",
|
|
19
|
+
"upgrade": "ncu -u",
|
|
20
|
+
"prettier": "prettier {*.*,**/*.*} --write"
|
|
17
21
|
},
|
|
18
22
|
"repository": {
|
|
19
23
|
"type": "git",
|
|
@@ -37,15 +41,23 @@
|
|
|
37
41
|
"homepage": "https://github.com/vizhub-core/editcodewithai#readme",
|
|
38
42
|
"dependencies": {
|
|
39
43
|
"@langchain/core": "^0.3.43",
|
|
40
|
-
"@langchain/openai": "^0.5.
|
|
41
|
-
"@
|
|
42
|
-
"@vizhub/viz-
|
|
44
|
+
"@langchain/openai": "^0.5.4",
|
|
45
|
+
"@types/d3": "^7.4.3",
|
|
46
|
+
"@vizhub/viz-types": "^0.1.0",
|
|
47
|
+
"@vizhub/viz-utils": "^0.1.0",
|
|
48
|
+
"d3": "^7.9.0",
|
|
43
49
|
"dotenv": "^16.4.7",
|
|
44
|
-
"
|
|
50
|
+
"langchain": "^0.3.20",
|
|
51
|
+
"llm-code-format": "^2.0.1"
|
|
45
52
|
},
|
|
46
53
|
"devDependencies": {
|
|
54
|
+
"cors": "^2.8.5",
|
|
55
|
+
"express": "^5.1.0",
|
|
56
|
+
"npm-check-updates": "^17.1.16",
|
|
57
|
+
"prettier": "^3.5.3",
|
|
58
|
+
"puppeteer": "^24.6.0",
|
|
47
59
|
"ts-node": "^10.9.2",
|
|
48
|
-
"typescript": "^5.8.
|
|
60
|
+
"typescript": "^5.8.3",
|
|
49
61
|
"vitest": "^3.1.1"
|
|
50
62
|
}
|
|
51
63
|
}
|
package/dist/benchmark.d.ts
DELETED
|
@@ -1,37 +0,0 @@
|
|
|
1
|
-
import { LlmFunction } from "./types";
|
|
2
|
-
/**
|
|
3
|
-
* Creates an OpenRouter LLM function with the provided API key and model using LangChain
|
|
4
|
-
* @param apiKey OpenRouter API key
|
|
5
|
-
* @param model Model to use (defaults to "anthropic/claude-3.5-sonnet")
|
|
6
|
-
* @returns LLM function that can be used with performAiEdit
|
|
7
|
-
*/
|
|
8
|
-
export declare function createOpenRouterLlmFunction(apiKey: string, model?: string): LlmFunction;
|
|
9
|
-
/**
|
|
10
|
-
* Runs a benchmark test using OpenRouter to implement an "add" function
|
|
11
|
-
* @param apiKey OpenRouter API key
|
|
12
|
-
* @param model Optional model to use (defaults to "openai/gpt-4")
|
|
13
|
-
* @returns Benchmark results including the implementation and performance metrics
|
|
14
|
-
*/
|
|
15
|
-
export declare function runAddFunctionBenchmark(apiKey: string, model?: string): Promise<{
|
|
16
|
-
success: boolean;
|
|
17
|
-
implementation: any;
|
|
18
|
-
isCorrect: boolean;
|
|
19
|
-
testOutput: any;
|
|
20
|
-
elapsedTime: number;
|
|
21
|
-
costCents: any;
|
|
22
|
-
provider: any;
|
|
23
|
-
inputTokens: any;
|
|
24
|
-
outputTokens: any;
|
|
25
|
-
error?: undefined;
|
|
26
|
-
} | {
|
|
27
|
-
success: boolean;
|
|
28
|
-
error: string;
|
|
29
|
-
implementation?: undefined;
|
|
30
|
-
isCorrect?: undefined;
|
|
31
|
-
testOutput?: undefined;
|
|
32
|
-
elapsedTime?: undefined;
|
|
33
|
-
costCents?: undefined;
|
|
34
|
-
provider?: undefined;
|
|
35
|
-
inputTokens?: undefined;
|
|
36
|
-
outputTokens?: undefined;
|
|
37
|
-
}>;
|
package/dist/benchmark.js
DELETED
|
@@ -1,123 +0,0 @@
|
|
|
1
|
-
import { performAiEdit } from "./index";
|
|
2
|
-
import { ChatOpenAI } from "@langchain/openai";
|
|
3
|
-
import { StringOutputParser } from "@langchain/core/output_parsers";
|
|
4
|
-
/**
|
|
5
|
-
* Creates an OpenRouter LLM function with the provided API key and model using LangChain
|
|
6
|
-
* @param apiKey OpenRouter API key
|
|
7
|
-
* @param model Model to use (defaults to "anthropic/claude-3.5-sonnet")
|
|
8
|
-
* @returns LLM function that can be used with performAiEdit
|
|
9
|
-
*/
|
|
10
|
-
export function createOpenRouterLlmFunction(apiKey, model = "anthropic/claude-3.5-sonnet") {
|
|
11
|
-
return async (prompt) => {
|
|
12
|
-
try {
|
|
13
|
-
const options = {
|
|
14
|
-
modelName: model,
|
|
15
|
-
configuration: {
|
|
16
|
-
apiKey,
|
|
17
|
-
baseURL: "https://openrouter.ai/api/v1",
|
|
18
|
-
},
|
|
19
|
-
streaming: false,
|
|
20
|
-
};
|
|
21
|
-
const chatModel = new ChatOpenAI(options);
|
|
22
|
-
const result = await chatModel.invoke(prompt);
|
|
23
|
-
const parser = new StringOutputParser();
|
|
24
|
-
const resultString = await parser.invoke(result);
|
|
25
|
-
return {
|
|
26
|
-
content: resultString,
|
|
27
|
-
generationId: Date.now().toString(), // OpenRouter doesn't return an ID through LangChain
|
|
28
|
-
};
|
|
29
|
-
}
|
|
30
|
-
catch (error) {
|
|
31
|
-
if (error instanceof Error) {
|
|
32
|
-
throw new Error(`OpenRouter API error: ${error.message}`);
|
|
33
|
-
}
|
|
34
|
-
else {
|
|
35
|
-
throw new Error(`OpenRouter API error: ${String(error)}`);
|
|
36
|
-
}
|
|
37
|
-
}
|
|
38
|
-
};
|
|
39
|
-
}
|
|
40
|
-
/**
|
|
41
|
-
* Runs a benchmark test using OpenRouter to implement an "add" function
|
|
42
|
-
* @param apiKey OpenRouter API key
|
|
43
|
-
* @param model Optional model to use (defaults to "openai/gpt-4")
|
|
44
|
-
* @returns Benchmark results including the implementation and performance metrics
|
|
45
|
-
*/
|
|
46
|
-
export async function runAddFunctionBenchmark(apiKey, model = "openai/gpt-4") {
|
|
47
|
-
console.log(`Running benchmark with model: ${model}`);
|
|
48
|
-
// Create test file with empty add function
|
|
49
|
-
const files = {
|
|
50
|
-
file1: {
|
|
51
|
-
name: "index.js",
|
|
52
|
-
text: "function add(a, b) {\n // TODO: Implement this function\n}\n\nmodule.exports = { add };",
|
|
53
|
-
},
|
|
54
|
-
};
|
|
55
|
-
const prompt = "Implement the 'add' function to add two numbers together and return the result.";
|
|
56
|
-
// Create LLM function
|
|
57
|
-
const llmFunction = createOpenRouterLlmFunction(apiKey, model);
|
|
58
|
-
console.log("Sending request to OpenRouter...");
|
|
59
|
-
const startTime = Date.now();
|
|
60
|
-
try {
|
|
61
|
-
// Perform the AI edit
|
|
62
|
-
const result = await performAiEdit({
|
|
63
|
-
prompt,
|
|
64
|
-
files,
|
|
65
|
-
llmFunction,
|
|
66
|
-
apiKey,
|
|
67
|
-
});
|
|
68
|
-
const endTime = Date.now();
|
|
69
|
-
const elapsedTime = (endTime - startTime) / 1000;
|
|
70
|
-
// Extract the implementation
|
|
71
|
-
const implementation = result.changedFiles.file1?.text || "";
|
|
72
|
-
// Validate the implementation
|
|
73
|
-
let isCorrect = false;
|
|
74
|
-
let testOutput = null;
|
|
75
|
-
try {
|
|
76
|
-
// Create a function from the implementation to test it
|
|
77
|
-
const funcStr = implementation + "\nreturn add(3, 4);";
|
|
78
|
-
const testFunc = new Function(funcStr);
|
|
79
|
-
testOutput = testFunc();
|
|
80
|
-
isCorrect = testOutput === 7;
|
|
81
|
-
}
|
|
82
|
-
catch (error) {
|
|
83
|
-
console.error("Error testing implementation:", error);
|
|
84
|
-
}
|
|
85
|
-
return {
|
|
86
|
-
success: true,
|
|
87
|
-
implementation,
|
|
88
|
-
isCorrect,
|
|
89
|
-
testOutput,
|
|
90
|
-
elapsedTime,
|
|
91
|
-
costCents: result.upstreamCostCents,
|
|
92
|
-
provider: result.provider,
|
|
93
|
-
inputTokens: result.inputTokens,
|
|
94
|
-
outputTokens: result.outputTokens,
|
|
95
|
-
};
|
|
96
|
-
}
|
|
97
|
-
catch (error) {
|
|
98
|
-
console.error("Benchmark failed:", error);
|
|
99
|
-
return {
|
|
100
|
-
success: false,
|
|
101
|
-
error: error instanceof Error ? error.message : String(error),
|
|
102
|
-
};
|
|
103
|
-
}
|
|
104
|
-
}
|
|
105
|
-
/**
|
|
106
|
-
* Example usage of the benchmark function
|
|
107
|
-
*/
|
|
108
|
-
// Check if this file is being executed directly (CommonJS approach)
|
|
109
|
-
if (require.main === module) {
|
|
110
|
-
// This code runs when the file is executed directly
|
|
111
|
-
const apiKey = process.env.OPENROUTER_API_KEY;
|
|
112
|
-
if (!apiKey) {
|
|
113
|
-
console.error("Please set the OPENROUTER_API_KEY environment variable");
|
|
114
|
-
process.exit(1);
|
|
115
|
-
}
|
|
116
|
-
runAddFunctionBenchmark(apiKey)
|
|
117
|
-
.then((result) => {
|
|
118
|
-
console.log("Benchmark result:", JSON.stringify(result, null, 2));
|
|
119
|
-
})
|
|
120
|
-
.catch((error) => {
|
|
121
|
-
console.error("Error running benchmark:", error);
|
|
122
|
-
});
|
|
123
|
-
}
|
|
@@ -1,85 +0,0 @@
|
|
|
1
|
-
"use strict";
|
|
2
|
-
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
-
exports.runAllChallenges = runAllChallenges;
|
|
4
|
-
const file_system_1 = require("langchain/cache/file_system");
|
|
5
|
-
const index_1 = require("../index");
|
|
6
|
-
const llm_1 = require("./llm");
|
|
7
|
-
const fileUtils_1 = require("./fileUtils");
|
|
8
|
-
const runner_1 = require("./runner");
|
|
9
|
-
const challenges_1 = require("./challenges");
|
|
10
|
-
const models_1 = require("./models");
|
|
11
|
-
/**
|
|
12
|
-
* Main function: runs the challenges on different models, calls the AI to fill them in,
|
|
13
|
-
* executes each test, and writes pass/fail results to CSV.
|
|
14
|
-
*/
|
|
15
|
-
async function runAllChallenges() {
|
|
16
|
-
const apiKey = process.env.OPENROUTER_API_KEY || "";
|
|
17
|
-
if (!apiKey) {
|
|
18
|
-
throw new Error("Please set OPENROUTER_API_KEY in your .env");
|
|
19
|
-
}
|
|
20
|
-
// Setup cache
|
|
21
|
-
const cacheDirBase = "benchmarks/llmResponseCache";
|
|
22
|
-
(0, fileUtils_1.ensureCacheDir)(cacheDirBase);
|
|
23
|
-
const cache = await file_system_1.LocalFileCache.create(cacheDirBase);
|
|
24
|
-
// Store results
|
|
25
|
-
const results = [];
|
|
26
|
-
// Run each model against each challenge
|
|
27
|
-
for (const model of models_1.models) {
|
|
28
|
-
try {
|
|
29
|
-
await runModelChallenges(model, apiKey, cache, results);
|
|
30
|
-
}
|
|
31
|
-
catch (error) {
|
|
32
|
-
console.error(`Error running model ${model}:`, error);
|
|
33
|
-
// Add failure results for remaining challenges
|
|
34
|
-
for (const challenge of challenges_1.challenges) {
|
|
35
|
-
results.push({
|
|
36
|
-
challenge: challenge.name,
|
|
37
|
-
model,
|
|
38
|
-
passFail: "error",
|
|
39
|
-
});
|
|
40
|
-
}
|
|
41
|
-
}
|
|
42
|
-
}
|
|
43
|
-
// Write results to CSV
|
|
44
|
-
const csv = (0, fileUtils_1.writeResultsToCsv)(results, "benchmarks/results.csv");
|
|
45
|
-
console.log("\nAll challenges completed. See 'benchmarks/results.csv' for summary.\n");
|
|
46
|
-
console.log(csv);
|
|
47
|
-
}
|
|
48
|
-
/**
|
|
49
|
-
* Runs all challenges for a specific model
|
|
50
|
-
*/
|
|
51
|
-
async function runModelChallenges(model, apiKey, cache, results) {
|
|
52
|
-
// Create the LLM function for this model
|
|
53
|
-
const llmFunction = (0, llm_1.createOpenRouterLlmFunction)(model, apiKey, cache);
|
|
54
|
-
// Run each challenge for this model
|
|
55
|
-
for (const challenge of challenges_1.challenges) {
|
|
56
|
-
console.log(`\n=== Challenge: ${challenge.name} | Model: ${model} ===`);
|
|
57
|
-
try {
|
|
58
|
-
// Ask AI to fill out the placeholder TODOs
|
|
59
|
-
const aiResult = await (0, index_1.performAiEdit)({
|
|
60
|
-
prompt: challenge.prompt,
|
|
61
|
-
files: challenge.files,
|
|
62
|
-
llmFunction,
|
|
63
|
-
});
|
|
64
|
-
// Write returned files to disk, including LLM response
|
|
65
|
-
const challengeDir = (0, fileUtils_1.writeChallengeFiles)(challenge.name, model, aiResult.changedFiles, aiResult.rawResponse);
|
|
66
|
-
// Run index.js in a child process
|
|
67
|
-
const exitCode = (0, runner_1.runNodeTest)(challengeDir);
|
|
68
|
-
const passFail = exitCode === 0 ? "pass" : "fail";
|
|
69
|
-
// Record the result
|
|
70
|
-
results.push({
|
|
71
|
-
challenge: challenge.name,
|
|
72
|
-
model,
|
|
73
|
-
passFail,
|
|
74
|
-
});
|
|
75
|
-
}
|
|
76
|
-
catch (error) {
|
|
77
|
-
console.error(`Error running challenge ${challenge.name}:`, error);
|
|
78
|
-
results.push({
|
|
79
|
-
challenge: challenge.name,
|
|
80
|
-
model,
|
|
81
|
-
passFail: "error",
|
|
82
|
-
});
|
|
83
|
-
}
|
|
84
|
-
}
|
|
85
|
-
}
|
|
@@ -1,158 +0,0 @@
|
|
|
1
|
-
"use strict";
|
|
2
|
-
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
-
exports.challenges = void 0;
|
|
4
|
-
// Challenges definition. Each has multiple files (some with TODO).
|
|
5
|
-
// "index.mjs" includes the test code that calls process.exit(1) if it fails.
|
|
6
|
-
exports.challenges = [
|
|
7
|
-
{
|
|
8
|
-
name: "add",
|
|
9
|
-
prompt: "Implement the 'add' function to correctly add two numbers (a+b) and pass the test in index.mjs.",
|
|
10
|
-
files: {
|
|
11
|
-
file1: {
|
|
12
|
-
name: "index.mjs",
|
|
13
|
-
text: `
|
|
14
|
-
import { add } from "./functions.mjs";
|
|
15
|
-
|
|
16
|
-
// A simple test:
|
|
17
|
-
const result = add(3, 4);
|
|
18
|
-
if (result !== 7) {
|
|
19
|
-
console.error("Test failed: expected 7, got", result);
|
|
20
|
-
process.exit(1);
|
|
21
|
-
}
|
|
22
|
-
console.log("Add test passed");
|
|
23
|
-
process.exit(0);
|
|
24
|
-
`,
|
|
25
|
-
},
|
|
26
|
-
file2: {
|
|
27
|
-
name: "functions.mjs",
|
|
28
|
-
text: `
|
|
29
|
-
// TODO: Implement the add function
|
|
30
|
-
export function add(a, b) {
|
|
31
|
-
// TODO
|
|
32
|
-
}
|
|
33
|
-
`,
|
|
34
|
-
},
|
|
35
|
-
},
|
|
36
|
-
},
|
|
37
|
-
{
|
|
38
|
-
name: "multiply",
|
|
39
|
-
prompt: "Implement the 'multiply' function to correctly multiply two numbers and pass the unit test in index.mjs.",
|
|
40
|
-
files: {
|
|
41
|
-
file1: {
|
|
42
|
-
name: "index.mjs",
|
|
43
|
-
text: `
|
|
44
|
-
import { multiply } from "./functions.mjs";
|
|
45
|
-
|
|
46
|
-
const result = multiply(6, 7);
|
|
47
|
-
if (result !== 42) {
|
|
48
|
-
console.error("Test failed: expected 42, got", result);
|
|
49
|
-
process.exit(1);
|
|
50
|
-
}
|
|
51
|
-
console.log("Multiply test passed");
|
|
52
|
-
process.exit(0);
|
|
53
|
-
`,
|
|
54
|
-
},
|
|
55
|
-
file2: {
|
|
56
|
-
name: "functions.mjs",
|
|
57
|
-
text: `
|
|
58
|
-
// TODO: Implement the multiply function
|
|
59
|
-
export function multiply(a, b) {
|
|
60
|
-
// TODO
|
|
61
|
-
}
|
|
62
|
-
`,
|
|
63
|
-
},
|
|
64
|
-
},
|
|
65
|
-
},
|
|
66
|
-
{
|
|
67
|
-
name: "square",
|
|
68
|
-
prompt: "Implement the 'square' function that returns x*x.",
|
|
69
|
-
files: {
|
|
70
|
-
file1: {
|
|
71
|
-
name: "index.mjs",
|
|
72
|
-
text: `
|
|
73
|
-
import { square } from "./functions.mjs";
|
|
74
|
-
|
|
75
|
-
const input = 5;
|
|
76
|
-
const result = square(input);
|
|
77
|
-
if (result !== 25) {
|
|
78
|
-
console.error("Test failed: expected 25, got", result);
|
|
79
|
-
process.exit(1);
|
|
80
|
-
}
|
|
81
|
-
console.log("Square test passed");
|
|
82
|
-
process.exit(0);
|
|
83
|
-
`,
|
|
84
|
-
},
|
|
85
|
-
file2: {
|
|
86
|
-
name: "functions.mjs",
|
|
87
|
-
text: `
|
|
88
|
-
// TODO: Implement the square function
|
|
89
|
-
export function square(x) {
|
|
90
|
-
// TODO
|
|
91
|
-
}
|
|
92
|
-
`,
|
|
93
|
-
},
|
|
94
|
-
},
|
|
95
|
-
},
|
|
96
|
-
{
|
|
97
|
-
name: "toUpperCase",
|
|
98
|
-
prompt: "Implement the toUpperCase function that returns the given string in uppercase.",
|
|
99
|
-
files: {
|
|
100
|
-
file1: {
|
|
101
|
-
name: "index.mjs",
|
|
102
|
-
text: `
|
|
103
|
-
import { toUpperCase } from "./functions.mjs";
|
|
104
|
-
|
|
105
|
-
const input = "hello";
|
|
106
|
-
const result = toUpperCase(input);
|
|
107
|
-
if (result !== "HELLO") {
|
|
108
|
-
console.error("Test failed: expected 'HELLO', got", result);
|
|
109
|
-
process.exit(1);
|
|
110
|
-
}
|
|
111
|
-
console.log("toUpperCase test passed");
|
|
112
|
-
process.exit(0);
|
|
113
|
-
`,
|
|
114
|
-
},
|
|
115
|
-
file2: {
|
|
116
|
-
name: "functions.mjs",
|
|
117
|
-
text: `
|
|
118
|
-
// TODO: Implement the toUpperCase function
|
|
119
|
-
export function toUpperCase(str) {
|
|
120
|
-
// TODO
|
|
121
|
-
}
|
|
122
|
-
`,
|
|
123
|
-
},
|
|
124
|
-
},
|
|
125
|
-
},
|
|
126
|
-
{
|
|
127
|
-
name: "reverseString",
|
|
128
|
-
prompt: "Implement the reverseString function that reverses the given string.",
|
|
129
|
-
files: {
|
|
130
|
-
file1: {
|
|
131
|
-
name: "index.mjs",
|
|
132
|
-
text: `
|
|
133
|
-
import { reverseString } from "./functions.mjs";
|
|
134
|
-
|
|
135
|
-
const input = "OpenAI";
|
|
136
|
-
const expected = "IAnepO";
|
|
137
|
-
|
|
138
|
-
const result = reverseString(input);
|
|
139
|
-
if (result !== expected) {
|
|
140
|
-
console.error("Test failed: expected", expected, "but got", result);
|
|
141
|
-
process.exit(1);
|
|
142
|
-
}
|
|
143
|
-
console.log("reverseString test passed");
|
|
144
|
-
process.exit(0);
|
|
145
|
-
`,
|
|
146
|
-
},
|
|
147
|
-
file2: {
|
|
148
|
-
name: "functions.mjs",
|
|
149
|
-
text: `
|
|
150
|
-
// TODO: Implement the reverseString function
|
|
151
|
-
export function reverseString(str) {
|
|
152
|
-
// TODO
|
|
153
|
-
}
|
|
154
|
-
`,
|
|
155
|
-
},
|
|
156
|
-
},
|
|
157
|
-
},
|
|
158
|
-
];
|
|
@@ -1,15 +0,0 @@
|
|
|
1
|
-
import { VizFiles } from "@vizhub/viz-types";
|
|
2
|
-
import { ChallengeResult } from "./types";
|
|
3
|
-
/**
|
|
4
|
-
* Writes the files for a given challenge into a folder named after challenge.name + model
|
|
5
|
-
* e.g. "benchmarks/challenges/add/gpt4" to keep them separate per model.
|
|
6
|
-
*/
|
|
7
|
-
export declare function writeChallengeFiles(challengeName: string, model: string, changedFiles: VizFiles, llmResponse?: string): string;
|
|
8
|
-
/**
|
|
9
|
-
* Writes results to a CSV file
|
|
10
|
-
*/
|
|
11
|
-
export declare function writeResultsToCsv(results: ChallengeResult[], filePath?: string): string;
|
|
12
|
-
/**
|
|
13
|
-
* Ensures a cache directory exists
|
|
14
|
-
*/
|
|
15
|
-
export declare function ensureCacheDir(cacheDirBase: string): void;
|
|
@@ -1,55 +0,0 @@
|
|
|
1
|
-
"use strict";
|
|
2
|
-
var __importDefault = (this && this.__importDefault) || function (mod) {
|
|
3
|
-
return (mod && mod.__esModule) ? mod : { "default": mod };
|
|
4
|
-
};
|
|
5
|
-
Object.defineProperty(exports, "__esModule", { value: true });
|
|
6
|
-
exports.writeChallengeFiles = writeChallengeFiles;
|
|
7
|
-
exports.writeResultsToCsv = writeResultsToCsv;
|
|
8
|
-
exports.ensureCacheDir = ensureCacheDir;
|
|
9
|
-
const fs_1 = __importDefault(require("fs"));
|
|
10
|
-
const path_1 = __importDefault(require("path"));
|
|
11
|
-
/**
|
|
12
|
-
* Writes the files for a given challenge into a folder named after challenge.name + model
|
|
13
|
-
* e.g. "benchmarks/challenges/add/gpt4" to keep them separate per model.
|
|
14
|
-
*/
|
|
15
|
-
function writeChallengeFiles(challengeName, model, changedFiles, llmResponse) {
|
|
16
|
-
const challengeDir = path_1.default.join("benchmarks", "challenges", challengeName, model);
|
|
17
|
-
if (!fs_1.default.existsSync(challengeDir)) {
|
|
18
|
-
fs_1.default.mkdirSync(challengeDir, { recursive: true });
|
|
19
|
-
}
|
|
20
|
-
Object.keys(changedFiles).forEach((fileKey) => {
|
|
21
|
-
const { name, text } = changedFiles[fileKey];
|
|
22
|
-
const filePath = path_1.default.join(challengeDir, name);
|
|
23
|
-
fs_1.default.writeFileSync(filePath, text, "utf-8");
|
|
24
|
-
});
|
|
25
|
-
// Write the LLM response to a file if provided
|
|
26
|
-
if (llmResponse) {
|
|
27
|
-
const responsePath = path_1.default.join(challengeDir, "llmResponse.md");
|
|
28
|
-
fs_1.default.writeFileSync(responsePath, llmResponse, "utf-8");
|
|
29
|
-
}
|
|
30
|
-
return challengeDir;
|
|
31
|
-
}
|
|
32
|
-
/**
|
|
33
|
-
* Writes results to a CSV file
|
|
34
|
-
*/
|
|
35
|
-
function writeResultsToCsv(results, filePath = "benchmarks/results.csv") {
|
|
36
|
-
let csv = "challenge,model,passFail\n";
|
|
37
|
-
for (const r of results) {
|
|
38
|
-
csv += `${r.challenge},${r.model},${r.passFail}\n`;
|
|
39
|
-
}
|
|
40
|
-
// Ensure benchmarks directory exists
|
|
41
|
-
const dir = path_1.default.dirname(filePath);
|
|
42
|
-
if (!fs_1.default.existsSync(dir)) {
|
|
43
|
-
fs_1.default.mkdirSync(dir, { recursive: true });
|
|
44
|
-
}
|
|
45
|
-
fs_1.default.writeFileSync(filePath, csv, "utf-8");
|
|
46
|
-
return csv;
|
|
47
|
-
}
|
|
48
|
-
/**
|
|
49
|
-
* Ensures a cache directory exists
|
|
50
|
-
*/
|
|
51
|
-
function ensureCacheDir(cacheDirBase) {
|
|
52
|
-
if (!fs_1.default.existsSync(cacheDirBase)) {
|
|
53
|
-
fs_1.default.mkdirSync(cacheDirBase, { recursive: true });
|
|
54
|
-
}
|
|
55
|
-
}
|
package/dist/benchmarks/index.js
DELETED
|
@@ -1,44 +0,0 @@
|
|
|
1
|
-
"use strict";
|
|
2
|
-
var __importDefault = (this && this.__importDefault) || function (mod) {
|
|
3
|
-
return (mod && mod.__esModule) ? mod : { "default": mod };
|
|
4
|
-
};
|
|
5
|
-
Object.defineProperty(exports, "__esModule", { value: true });
|
|
6
|
-
exports.writeResultsToCsv = exports.writeChallengeFiles = exports.runNodeTest = exports.createOpenRouterLlmFunction = exports.challenges = exports.runAllChallenges = void 0;
|
|
7
|
-
/**********************************************************************
|
|
8
|
-
* benchmark.ts
|
|
9
|
-
* --------------------------------------------------------------------
|
|
10
|
-
* Requires environment variables to be set in a .env file
|
|
11
|
-
* --------------------------------------------------------------------
|
|
12
|
-
* Defines 5 challenges, each with index.js (ES module) that performs
|
|
13
|
-
* a simple unit test. The code calls process.exit(0) on success or
|
|
14
|
-
* process.exit(1) on failure.
|
|
15
|
-
*
|
|
16
|
-
* Runs all challenges on 5 different models:
|
|
17
|
-
* ["gpt3", "gpt4", "gpt5", "deepseek", "qwen32"]
|
|
18
|
-
*
|
|
19
|
-
* Writes final results to "results.csv" with columns:
|
|
20
|
-
* challenge, model, passFail, exitCode
|
|
21
|
-
**********************************************************************/
|
|
22
|
-
const dotenv_1 = __importDefault(require("dotenv"));
|
|
23
|
-
const benchmarkRunner_1 = require("./benchmarkRunner");
|
|
24
|
-
// Load environment variables from .env file
|
|
25
|
-
dotenv_1.default.config();
|
|
26
|
-
// Run if invoked directly
|
|
27
|
-
if (require.main === module) {
|
|
28
|
-
(0, benchmarkRunner_1.runAllChallenges)().catch((error) => {
|
|
29
|
-
console.error("Error running challenges:", error);
|
|
30
|
-
process.exit(1);
|
|
31
|
-
});
|
|
32
|
-
}
|
|
33
|
-
// Re-export for programmatic usage
|
|
34
|
-
var benchmarkRunner_2 = require("./benchmarkRunner");
|
|
35
|
-
Object.defineProperty(exports, "runAllChallenges", { enumerable: true, get: function () { return benchmarkRunner_2.runAllChallenges; } });
|
|
36
|
-
var challenges_1 = require("./challenges");
|
|
37
|
-
Object.defineProperty(exports, "challenges", { enumerable: true, get: function () { return challenges_1.challenges; } });
|
|
38
|
-
var llm_1 = require("./llm");
|
|
39
|
-
Object.defineProperty(exports, "createOpenRouterLlmFunction", { enumerable: true, get: function () { return llm_1.createOpenRouterLlmFunction; } });
|
|
40
|
-
var runner_1 = require("./runner");
|
|
41
|
-
Object.defineProperty(exports, "runNodeTest", { enumerable: true, get: function () { return runner_1.runNodeTest; } });
|
|
42
|
-
var fileUtils_1 = require("./fileUtils");
|
|
43
|
-
Object.defineProperty(exports, "writeChallengeFiles", { enumerable: true, get: function () { return fileUtils_1.writeChallengeFiles; } });
|
|
44
|
-
Object.defineProperty(exports, "writeResultsToCsv", { enumerable: true, get: function () { return fileUtils_1.writeResultsToCsv; } });
|
package/dist/benchmarks/llm.d.ts
DELETED
|
@@ -1,10 +0,0 @@
|
|
|
1
|
-
import { LlmFunction } from "../types";
|
|
2
|
-
import { LocalFileCache } from "langchain/cache/file_system";
|
|
3
|
-
/**
|
|
4
|
-
* Creates an LLM function that connects to OpenRouter
|
|
5
|
-
* @param model The model identifier to use
|
|
6
|
-
* @param apiKey OpenRouter API key
|
|
7
|
-
* @param cache Optional local file cache to avoid duplicate requests
|
|
8
|
-
* @returns A function that can be used with performAiEdit
|
|
9
|
-
*/
|
|
10
|
-
export declare function createOpenRouterLlmFunction(model: string, apiKey: string, cache?: LocalFileCache): LlmFunction;
|
package/dist/benchmarks/llm.js
DELETED
|
@@ -1,29 +0,0 @@
|
|
|
1
|
-
"use strict";
|
|
2
|
-
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
-
exports.createOpenRouterLlmFunction = createOpenRouterLlmFunction;
|
|
4
|
-
const openai_1 = require("@langchain/openai");
|
|
5
|
-
const output_parsers_1 = require("@langchain/core/output_parsers");
|
|
6
|
-
/**
|
|
7
|
-
* Creates an LLM function that connects to OpenRouter
|
|
8
|
-
* @param model The model identifier to use
|
|
9
|
-
* @param apiKey OpenRouter API key
|
|
10
|
-
* @param cache Optional local file cache to avoid duplicate requests
|
|
11
|
-
* @returns A function that can be used with performAiEdit
|
|
12
|
-
*/
|
|
13
|
-
function createOpenRouterLlmFunction(model, apiKey, cache) {
|
|
14
|
-
return async (prompt) => {
|
|
15
|
-
// Create OpenAI chat model with OpenRouter configuration
|
|
16
|
-
const chatModel = new openai_1.ChatOpenAI({
|
|
17
|
-
modelName: model,
|
|
18
|
-
configuration: { apiKey, baseURL: "https://openrouter.ai/api/v1" },
|
|
19
|
-
streaming: false,
|
|
20
|
-
cache,
|
|
21
|
-
});
|
|
22
|
-
// Invoke the model
|
|
23
|
-
const result = await chatModel.invoke(prompt);
|
|
24
|
-
// Parse to string
|
|
25
|
-
const parser = new output_parsers_1.StringOutputParser();
|
|
26
|
-
const resultString = await parser.invoke(result);
|
|
27
|
-
return { content: resultString, generationId: result.lc_kwargs.id };
|
|
28
|
-
};
|
|
29
|
-
}
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
export declare const models: string[];
|
|
@@ -1,21 +0,0 @@
|
|
|
1
|
-
"use strict";
|
|
2
|
-
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
-
exports.models = void 0;
|
|
4
|
-
exports.models = [
|
|
5
|
-
"anthropic/claude-3.7-sonnet",
|
|
6
|
-
"anthropic/claude-3.7-sonnet:thinking",
|
|
7
|
-
"google/gemini-2.0-flash-001",
|
|
8
|
-
"anthropic/claude-3.5-sonnet",
|
|
9
|
-
"deepseek/deepseek-r1",
|
|
10
|
-
"openai/o3-mini-high",
|
|
11
|
-
"openai/o3-mini",
|
|
12
|
-
"deepseek/deepseek-chat",
|
|
13
|
-
"qwen/qwen-2.5-72b-instruct",
|
|
14
|
-
"qwen/qwq-32b",
|
|
15
|
-
"anthropic/claude-3.5-haiku",
|
|
16
|
-
"qwen/qwen-2.5-7b-instruct",
|
|
17
|
-
"amazon/nova-micro-v1",
|
|
18
|
-
"meta-llama/llama-3.1-8b-instruct",
|
|
19
|
-
"meta-llama/llama-3.2-3b-instruct",
|
|
20
|
-
"meta-llama/llama-3.2-1b-instruct",
|
|
21
|
-
];
|
|
@@ -1,35 +0,0 @@
|
|
|
1
|
-
"use strict";
|
|
2
|
-
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
-
exports.runNodeTest = runNodeTest;
|
|
4
|
-
const child_process_1 = require("child_process");
|
|
5
|
-
/**
|
|
6
|
-
* Runs "index.mjs" for a given challenge directory with Node and returns exit code.
|
|
7
|
-
*/
|
|
8
|
-
function runNodeTest(challengeDir) {
|
|
9
|
-
const result = runNodeProcess(challengeDir);
|
|
10
|
-
logTestResult(challengeDir, result);
|
|
11
|
-
return result.status ?? 1; // If status is null, consider that a failure
|
|
12
|
-
}
|
|
13
|
-
/**
|
|
14
|
-
* Executes the Node.js process for a test
|
|
15
|
-
*/
|
|
16
|
-
function runNodeProcess(challengeDir) {
|
|
17
|
-
const child = (0, child_process_1.spawnSync)("node", ["index.mjs"], {
|
|
18
|
-
cwd: challengeDir,
|
|
19
|
-
encoding: "utf-8",
|
|
20
|
-
});
|
|
21
|
-
return {
|
|
22
|
-
status: child.status,
|
|
23
|
-
stdout: child.stdout,
|
|
24
|
-
stderr: child.stderr
|
|
25
|
-
};
|
|
26
|
-
}
|
|
27
|
-
/**
|
|
28
|
-
* Logs the test result to the console
|
|
29
|
-
*/
|
|
30
|
-
function logTestResult(challengeDir, result) {
|
|
31
|
-
console.log(`--- Running test in ${challengeDir} ---`);
|
|
32
|
-
console.log("stdout:\n", result.stdout.trim());
|
|
33
|
-
console.log("stderr:\n", result.stderr.trim());
|
|
34
|
-
console.log("exit code:", result.status, "\n");
|
|
35
|
-
}
|
|
@@ -1,16 +0,0 @@
|
|
|
1
|
-
import { VizFiles } from "@vizhub/viz-types";
|
|
2
|
-
export interface Challenge {
|
|
3
|
-
name: string;
|
|
4
|
-
prompt: string;
|
|
5
|
-
files: VizFiles;
|
|
6
|
-
}
|
|
7
|
-
export interface ChallengeResult {
|
|
8
|
-
challenge: string;
|
|
9
|
-
model: string;
|
|
10
|
-
passFail: "pass" | "fail" | "error";
|
|
11
|
-
}
|
|
12
|
-
export interface TestRunResult {
|
|
13
|
-
status: number | null;
|
|
14
|
-
stdout: string;
|
|
15
|
-
stderr: string;
|
|
16
|
-
}
|
package/dist/benchmarks/types.js
DELETED
package/dist/cli.d.ts
DELETED
package/dist/cli.js
DELETED
|
@@ -1,157 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
|
-
"use strict";
|
|
3
|
-
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
4
|
-
if (k2 === undefined) k2 = k;
|
|
5
|
-
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
6
|
-
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
7
|
-
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
8
|
-
}
|
|
9
|
-
Object.defineProperty(o, k2, desc);
|
|
10
|
-
}) : (function(o, m, k, k2) {
|
|
11
|
-
if (k2 === undefined) k2 = k;
|
|
12
|
-
o[k2] = m[k];
|
|
13
|
-
}));
|
|
14
|
-
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
15
|
-
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
16
|
-
}) : function(o, v) {
|
|
17
|
-
o["default"] = v;
|
|
18
|
-
});
|
|
19
|
-
var __importStar = (this && this.__importStar) || (function () {
|
|
20
|
-
var ownKeys = function(o) {
|
|
21
|
-
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
22
|
-
var ar = [];
|
|
23
|
-
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
24
|
-
return ar;
|
|
25
|
-
};
|
|
26
|
-
return ownKeys(o);
|
|
27
|
-
};
|
|
28
|
-
return function (mod) {
|
|
29
|
-
if (mod && mod.__esModule) return mod;
|
|
30
|
-
var result = {};
|
|
31
|
-
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
32
|
-
__setModuleDefault(result, mod);
|
|
33
|
-
return result;
|
|
34
|
-
};
|
|
35
|
-
})();
|
|
36
|
-
var __importDefault = (this && this.__importDefault) || function (mod) {
|
|
37
|
-
return (mod && mod.__esModule) ? mod : { "default": mod };
|
|
38
|
-
};
|
|
39
|
-
Object.defineProperty(exports, "__esModule", { value: true });
|
|
40
|
-
const commander_1 = require("commander");
|
|
41
|
-
const path = __importStar(require("path"));
|
|
42
|
-
const readline = __importStar(require("readline"));
|
|
43
|
-
const index_1 = require("./index");
|
|
44
|
-
const llm_1 = require("./benchmarks/llm");
|
|
45
|
-
const computeInitialDocument_1 = require("./computeInitialDocument");
|
|
46
|
-
const dotenv_1 = __importDefault(require("dotenv"));
|
|
47
|
-
const fs = __importStar(require("fs/promises"));
|
|
48
|
-
// Load environment variables
|
|
49
|
-
dotenv_1.default.config();
|
|
50
|
-
const program = new commander_1.Command();
|
|
51
|
-
program
|
|
52
|
-
.name("editcode")
|
|
53
|
-
.description("Edit code files using AI")
|
|
54
|
-
.option("-p, --prompt <prompt>", "The instruction for editing the code")
|
|
55
|
-
.option("-d, --dir <directory>", "Directory to process", process.cwd())
|
|
56
|
-
.option("--dry-run", "Show changes without writing them", false)
|
|
57
|
-
.option("--model <model>", "OpenRouter model to use", "anthropic/claude-3.7-sonnet")
|
|
58
|
-
.action(async (options) => {
|
|
59
|
-
try {
|
|
60
|
-
// Get prompt from user if not provided as argument
|
|
61
|
-
if (!options.prompt) {
|
|
62
|
-
const rl = readline.createInterface({
|
|
63
|
-
input: process.stdin,
|
|
64
|
-
output: process.stdout
|
|
65
|
-
});
|
|
66
|
-
options.prompt = await new Promise((resolve) => {
|
|
67
|
-
rl.question("> ", (answer) => {
|
|
68
|
-
rl.close();
|
|
69
|
-
resolve(answer);
|
|
70
|
-
});
|
|
71
|
-
});
|
|
72
|
-
}
|
|
73
|
-
if (!options.prompt.trim()) {
|
|
74
|
-
console.error("Error: Prompt cannot be empty");
|
|
75
|
-
process.exit(1);
|
|
76
|
-
}
|
|
77
|
-
// Debug: Log environment
|
|
78
|
-
console.log('Environment variables:', {
|
|
79
|
-
OPENROUTER_API_KEY: process.env.OPENROUTER_API_KEY ? '[PRESENT]' : '[MISSING]',
|
|
80
|
-
NODE_ENV: process.env.NODE_ENV,
|
|
81
|
-
PATH: process.env.PATH?.substring(0, 50) + '...'
|
|
82
|
-
});
|
|
83
|
-
// Validate API key
|
|
84
|
-
const apiKey = process.env.OPENROUTER_API_KEY;
|
|
85
|
-
if (!apiKey) {
|
|
86
|
-
console.error("Error: OPENROUTER_API_KEY environment variable is required");
|
|
87
|
-
process.exit(1);
|
|
88
|
-
}
|
|
89
|
-
// Get absolute path
|
|
90
|
-
const fullPath = path.resolve(options.dir);
|
|
91
|
-
console.log(`Processing directory: ${fullPath}`);
|
|
92
|
-
// Use computeInitialDocument to get files (respects .ignore files)
|
|
93
|
-
const initialDocument = (0, computeInitialDocument_1.computeInitialDocument)({ fullPath });
|
|
94
|
-
// Convert to VizFiles format
|
|
95
|
-
const vizFiles = {};
|
|
96
|
-
Object.entries(initialDocument.files).forEach(([id, file]) => {
|
|
97
|
-
if (file.text !== null) {
|
|
98
|
-
// Skip directories
|
|
99
|
-
vizFiles[id] = {
|
|
100
|
-
name: path.relative(process.cwd(), path.join(fullPath, file.name)),
|
|
101
|
-
text: file.text,
|
|
102
|
-
};
|
|
103
|
-
}
|
|
104
|
-
});
|
|
105
|
-
const fileCount = Object.keys(vizFiles).length;
|
|
106
|
-
if (fileCount === 0) {
|
|
107
|
-
console.error("No files found to process");
|
|
108
|
-
process.exit(1);
|
|
109
|
-
}
|
|
110
|
-
console.log(`Found ${fileCount} files to process`);
|
|
111
|
-
// Create LLM function
|
|
112
|
-
const llmFunction = (0, llm_1.createOpenRouterLlmFunction)(options.model, apiKey);
|
|
113
|
-
// Perform the edit
|
|
114
|
-
console.log("\nRequesting changes from AI...");
|
|
115
|
-
const result = await (0, index_1.performAiEdit)({
|
|
116
|
-
prompt: options.prompt,
|
|
117
|
-
files: vizFiles,
|
|
118
|
-
llmFunction,
|
|
119
|
-
apiKey,
|
|
120
|
-
});
|
|
121
|
-
// Show changes
|
|
122
|
-
console.log("\nProposed changes:");
|
|
123
|
-
for (const [id, file] of Object.entries(result.changedFiles)) {
|
|
124
|
-
const original = vizFiles[id];
|
|
125
|
-
if (!original) {
|
|
126
|
-
console.log(`\nNew file: ${file.name}`);
|
|
127
|
-
console.log(file.text);
|
|
128
|
-
}
|
|
129
|
-
else if (original.text !== file.text) {
|
|
130
|
-
console.log(`\nModified: ${file.name}`);
|
|
131
|
-
console.log(file.text);
|
|
132
|
-
}
|
|
133
|
-
}
|
|
134
|
-
// Write changes if not dry run
|
|
135
|
-
if (!options.dryRun) {
|
|
136
|
-
console.log("\nWriting changes...");
|
|
137
|
-
for (const [_, file] of Object.entries(result.changedFiles)) {
|
|
138
|
-
const fullPath = path.resolve(file.name);
|
|
139
|
-
await fs.mkdir(path.dirname(fullPath), { recursive: true });
|
|
140
|
-
await fs.writeFile(fullPath, file.text);
|
|
141
|
-
console.log(`Updated ${fullPath}`);
|
|
142
|
-
}
|
|
143
|
-
}
|
|
144
|
-
// Show metadata
|
|
145
|
-
console.log("\nMetadata:");
|
|
146
|
-
console.log(`Model: ${options.model}`);
|
|
147
|
-
console.log(`Provider: ${result.provider}`);
|
|
148
|
-
console.log(`Cost: ${result.upstreamCostCents / 100} USD`);
|
|
149
|
-
console.log(`Input tokens: ${result.inputTokens}`);
|
|
150
|
-
console.log(`Output tokens: ${result.outputTokens}`);
|
|
151
|
-
}
|
|
152
|
-
catch (error) {
|
|
153
|
-
console.error("Error:", error);
|
|
154
|
-
process.exit(1);
|
|
155
|
-
}
|
|
156
|
-
});
|
|
157
|
-
program.parse();
|
|
@@ -1,13 +0,0 @@
|
|
|
1
|
-
interface InitialDocument {
|
|
2
|
-
files: {
|
|
3
|
-
[key: string]: {
|
|
4
|
-
text: string | null;
|
|
5
|
-
name: string;
|
|
6
|
-
};
|
|
7
|
-
};
|
|
8
|
-
}
|
|
9
|
-
export declare const isDirectory: (path: string) => boolean;
|
|
10
|
-
export declare const computeInitialDocument: ({ fullPath }: {
|
|
11
|
-
fullPath: string;
|
|
12
|
-
}) => InitialDocument;
|
|
13
|
-
export {};
|
|
@@ -1,135 +0,0 @@
|
|
|
1
|
-
"use strict";
|
|
2
|
-
var __importDefault = (this && this.__importDefault) || function (mod) {
|
|
3
|
-
return (mod && mod.__esModule) ? mod : { "default": mod };
|
|
4
|
-
};
|
|
5
|
-
Object.defineProperty(exports, "__esModule", { value: true });
|
|
6
|
-
exports.computeInitialDocument = exports.isDirectory = void 0;
|
|
7
|
-
const fs_1 = __importDefault(require("fs"));
|
|
8
|
-
const path_1 = __importDefault(require("path"));
|
|
9
|
-
const ignore_1 = __importDefault(require("ignore"));
|
|
10
|
-
// Configuration constants
|
|
11
|
-
const ignoreFilePattern = '**/{.ignore,.gitignore}';
|
|
12
|
-
const baseIgnore = [
|
|
13
|
-
'.git',
|
|
14
|
-
'node_modules',
|
|
15
|
-
'dist',
|
|
16
|
-
'*.log',
|
|
17
|
-
'.DS_Store',
|
|
18
|
-
'coverage',
|
|
19
|
-
'.env',
|
|
20
|
-
'.env.*',
|
|
21
|
-
];
|
|
22
|
-
const debugIgnore = false;
|
|
23
|
-
const debugDirectories = false;
|
|
24
|
-
const enableDirectories = true;
|
|
25
|
-
/**
|
|
26
|
-
* @param {string} fullPath - absolute path of the workspace root
|
|
27
|
-
* @param {string} currentDirectoryPath - path where the ignore file is found, relative to fullPath
|
|
28
|
-
* @param {string} fileName - name of the ignore file
|
|
29
|
-
* @returns {string[]} parsed lines
|
|
30
|
-
*/
|
|
31
|
-
const parseIgnoreFile = (fullPath, currentDirectory, fileName) => {
|
|
32
|
-
const filePath = path_1.default.join(fullPath, currentDirectory, fileName);
|
|
33
|
-
const content = fs_1.default.readFileSync(filePath, 'utf8');
|
|
34
|
-
const globs = content
|
|
35
|
-
.split(/[\n\r]+/)
|
|
36
|
-
.filter(
|
|
37
|
-
// remove blank line and comments
|
|
38
|
-
(line) => line.length > 0 && !line.startsWith('#'))
|
|
39
|
-
.map((line) => {
|
|
40
|
-
const { bang, slash, glob } = line.match(/^(?<bang>!?)(?<slash>\/?)(?<glob>.*)$/).groups;
|
|
41
|
-
const hasSlash = Boolean(slash) || /\/.*\S/.test(glob);
|
|
42
|
-
const relativeGlob = path_1.default.posix.join(currentDirectory.replace(
|
|
43
|
-
// escape characters with special meaning in glob expressions
|
|
44
|
-
/[*?!# \[\]\\]/g, (char) => '\\' + char),
|
|
45
|
-
// a pattern that doesn't include a slash (not counting a trailing one) matches files in any descendant directory of the current one
|
|
46
|
-
hasSlash ? '' : '**', glob);
|
|
47
|
-
// preserve leading `!` and `/` characters
|
|
48
|
-
return bang + slash + relativeGlob;
|
|
49
|
-
});
|
|
50
|
-
if (debugIgnore) {
|
|
51
|
-
console.debug('at', currentDirectory, 'parsing', fileName, 'obtained globs', globs);
|
|
52
|
-
}
|
|
53
|
-
return globs;
|
|
54
|
-
};
|
|
55
|
-
const isDirectory = (path) => path.endsWith('/');
|
|
56
|
-
exports.isDirectory = isDirectory;
|
|
57
|
-
// Lists files from the file system,
|
|
58
|
-
// converts them into the internal data structure.
|
|
59
|
-
const computeInitialDocument = ({ fullPath }) => {
|
|
60
|
-
// Initialize the document using our data structure for representing files.
|
|
61
|
-
const initialDocument = {
|
|
62
|
-
files: {},
|
|
63
|
-
};
|
|
64
|
-
/**
|
|
65
|
-
* Stack for recursively traversing directories.
|
|
66
|
-
* @type {string[]}
|
|
67
|
-
*/
|
|
68
|
-
let files = [];
|
|
69
|
-
const ignoreFileMatcher = (0, ignore_1.default)().add(ignoreFilePattern);
|
|
70
|
-
const isIgnoreFile = (fileName) => ignoreFileMatcher.ignores(fileName);
|
|
71
|
-
const unsearchedDirectories = [
|
|
72
|
-
{
|
|
73
|
-
currentDirectory: '.',
|
|
74
|
-
ignore: (0, ignore_1.default)().add(baseIgnore),
|
|
75
|
-
},
|
|
76
|
-
];
|
|
77
|
-
while (unsearchedDirectories.length !== 0) {
|
|
78
|
-
const { currentDirectory, ignore: parentIgnore } = unsearchedDirectories.pop();
|
|
79
|
-
const currentDirectoryPath = path_1.default.join(fullPath, currentDirectory);
|
|
80
|
-
const dirEntries = fs_1.default
|
|
81
|
-
.readdirSync(currentDirectoryPath, {
|
|
82
|
-
withFileTypes: true,
|
|
83
|
-
})
|
|
84
|
-
.filter((dirent) => enableDirectories ? true : dirent.isFile());
|
|
85
|
-
// find .ignore or .gitignore files in the current directory
|
|
86
|
-
const ignoreFiles = dirEntries
|
|
87
|
-
.filter((dirent) => dirent.isFile() && isIgnoreFile(dirent.name))
|
|
88
|
-
.map((file) => file.name);
|
|
89
|
-
let ignore = parentIgnore;
|
|
90
|
-
if (ignoreFiles.length > 0) {
|
|
91
|
-
const globs = ignoreFiles.flatMap((fileName) => parseIgnoreFile(fullPath, currentDirectory, fileName));
|
|
92
|
-
ignore = (0, ignore_1.default)().add(parentIgnore).add(globs);
|
|
93
|
-
}
|
|
94
|
-
const newFiles = dirEntries
|
|
95
|
-
.filter((dirent) => {
|
|
96
|
-
const relativePath = path_1.default.posix.join(currentDirectory, dirent.name) +
|
|
97
|
-
(dirent.isDirectory() ? '/' : '');
|
|
98
|
-
const keep = !ignore.ignores(relativePath);
|
|
99
|
-
if (debugIgnore && !keep) {
|
|
100
|
-
console.debug('at', currentDirectory, 'ignoring', relativePath);
|
|
101
|
-
}
|
|
102
|
-
return keep;
|
|
103
|
-
})
|
|
104
|
-
// Add a trailing slash for directories
|
|
105
|
-
.map((dirent) => {
|
|
106
|
-
const relativePath = path_1.default.posix.join(currentDirectory, dirent.name);
|
|
107
|
-
if (!dirent.isDirectory()) {
|
|
108
|
-
return relativePath;
|
|
109
|
-
}
|
|
110
|
-
unsearchedDirectories.push({
|
|
111
|
-
currentDirectory: relativePath,
|
|
112
|
-
ignore,
|
|
113
|
-
});
|
|
114
|
-
return relativePath + '/';
|
|
115
|
-
});
|
|
116
|
-
files.push(...newFiles);
|
|
117
|
-
}
|
|
118
|
-
files.forEach((file) => {
|
|
119
|
-
const id = Math.floor(Math.random() * 10000000).toString();
|
|
120
|
-
initialDocument.files[id] = {
|
|
121
|
-
text: (0, exports.isDirectory)(file)
|
|
122
|
-
? null
|
|
123
|
-
: fs_1.default.readFileSync(path_1.default.join(fullPath, file), 'utf-8'),
|
|
124
|
-
name: file,
|
|
125
|
-
};
|
|
126
|
-
});
|
|
127
|
-
if (debugDirectories) {
|
|
128
|
-
console.log('files:');
|
|
129
|
-
console.log(files);
|
|
130
|
-
console.log('initialDocument:');
|
|
131
|
-
console.log(JSON.stringify(initialDocument, null, 2));
|
|
132
|
-
}
|
|
133
|
-
return initialDocument;
|
|
134
|
-
};
|
|
135
|
-
exports.computeInitialDocument = computeInitialDocument;
|