forge-workflow 0.0.3 → 0.0.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/commands/dev.md +340 -314
- package/.claude/commands/plan.md +521 -478
- package/.claude/commands/premerge.md +176 -179
- package/.claude/commands/research.md +42 -42
- package/.claude/commands/review.md +442 -442
- package/.claude/commands/rollback.md +721 -721
- package/.claude/commands/ship.md +164 -134
- package/.claude/commands/sonarcloud.md +152 -152
- package/.claude/commands/status.md +48 -77
- package/.claude/commands/validate.md +282 -237
- package/.claude/commands/verify.md +221 -221
- package/.claude/rules/greptile-review-process.md +285 -285
- package/.claude/rules/workflow.md +105 -105
- package/.claude/scripts/greptile-resolve.sh +526 -526
- package/.claude/scripts/load-env.sh +32 -32
- package/.cline/workflows/dev.md +337 -311
- package/.cline/workflows/plan.md +518 -475
- package/.cline/workflows/premerge.md +173 -176
- package/.cline/workflows/research.md +39 -39
- package/.cline/workflows/review.md +439 -439
- package/.cline/workflows/rollback.md +718 -718
- package/.cline/workflows/ship.md +161 -131
- package/.cline/workflows/sonarcloud.md +146 -146
- package/.cline/workflows/status.md +45 -74
- package/.cline/workflows/validate.md +279 -234
- package/.cline/workflows/verify.md +218 -218
- package/.codex/config.toml +11 -11
- package/.codex/skills/dev/SKILL.md +340 -314
- package/.codex/skills/plan/SKILL.md +521 -478
- package/.codex/skills/premerge/SKILL.md +176 -179
- package/.codex/skills/research/SKILL.md +42 -42
- package/.codex/skills/review/SKILL.md +442 -442
- package/.codex/skills/rollback/SKILL.md +721 -721
- package/.codex/skills/ship/SKILL.md +164 -134
- package/.codex/skills/sonarcloud/SKILL.md +149 -149
- package/.codex/skills/status/SKILL.md +48 -77
- package/.codex/skills/validate/SKILL.md +282 -237
- package/.codex/skills/verify/SKILL.md +221 -221
- package/.cursor/commands/dev.md +337 -311
- package/.cursor/commands/plan.md +518 -475
- package/.cursor/commands/premerge.md +173 -176
- package/.cursor/commands/research.md +39 -39
- package/.cursor/commands/review.md +439 -439
- package/.cursor/commands/rollback.md +718 -718
- package/.cursor/commands/ship.md +161 -131
- package/.cursor/commands/sonarcloud.md +146 -146
- package/.cursor/commands/status.md +45 -74
- package/.cursor/commands/validate.md +279 -234
- package/.cursor/commands/verify.md +218 -218
- package/.cursor/rules/permissions-guidance.mdc +37 -37
- package/.forge/hooks/check-tdd.js +240 -240
- package/.github/PLUGIN_TEMPLATE.json +32 -32
- package/.github/prompts/dev.prompt.md +342 -316
- package/.github/prompts/plan.prompt.md +523 -480
- package/.github/prompts/premerge.prompt.md +178 -181
- package/.github/prompts/research.prompt.md +44 -44
- package/.github/prompts/review.prompt.md +444 -444
- package/.github/prompts/rollback.prompt.md +723 -723
- package/.github/prompts/ship.prompt.md +166 -136
- package/.github/prompts/sonarcloud.prompt.md +151 -151
- package/.github/prompts/status.prompt.md +50 -79
- package/.github/prompts/validate.prompt.md +284 -239
- package/.github/prompts/verify.prompt.md +223 -223
- package/.github/workflows/beads-to-github.yml +56 -0
- package/.github/workflows/github-to-beads.yml +97 -0
- package/.kilocode/workflows/dev.md +341 -315
- package/.kilocode/workflows/plan.md +522 -479
- package/.kilocode/workflows/premerge.md +177 -180
- package/.kilocode/workflows/research.md +43 -43
- package/.kilocode/workflows/review.md +443 -443
- package/.kilocode/workflows/rollback.md +722 -722
- package/.kilocode/workflows/ship.md +165 -135
- package/.kilocode/workflows/sonarcloud.md +150 -150
- package/.kilocode/workflows/status.md +49 -78
- package/.kilocode/workflows/validate.md +283 -238
- package/.kilocode/workflows/verify.md +222 -222
- package/.mcp.json.example +12 -12
- package/.opencode/commands/dev.md +340 -314
- package/.opencode/commands/plan.md +521 -478
- package/.opencode/commands/premerge.md +176 -179
- package/.opencode/commands/research.md +42 -42
- package/.opencode/commands/review.md +442 -442
- package/.opencode/commands/rollback.md +721 -721
- package/.opencode/commands/ship.md +164 -134
- package/.opencode/commands/sonarcloud.md +149 -149
- package/.opencode/commands/status.md +48 -77
- package/.opencode/commands/validate.md +282 -237
- package/.opencode/commands/verify.md +221 -221
- package/.roo/commands/dev.md +341 -315
- package/.roo/commands/plan.md +522 -479
- package/.roo/commands/premerge.md +177 -180
- package/.roo/commands/research.md +43 -43
- package/.roo/commands/review.md +443 -443
- package/.roo/commands/rollback.md +722 -722
- package/.roo/commands/ship.md +165 -135
- package/.roo/commands/sonarcloud.md +150 -150
- package/.roo/commands/status.md +49 -78
- package/.roo/commands/validate.md +283 -238
- package/.roo/commands/verify.md +222 -222
- package/AGENTS.md +175 -169
- package/CLAUDE.md +100 -99
- package/LICENSE +21 -21
- package/README.md +429 -414
- package/bin/forge-cmd.js +313 -313
- package/bin/{forge-validate.js → forge-preflight.js} +309 -303
- package/bin/forge.js +4596 -4232
- package/docs/AGENT_INSTALL_PROMPT.md +342 -342
- package/docs/BEADS_GITHUB_SYNC.md +251 -0
- package/docs/ENHANCED_ONBOARDING.md +602 -602
- package/docs/EXAMPLES.md +482 -482
- package/docs/GREPTILE_SETUP.md +400 -400
- package/docs/MANUAL_REVIEW_GUIDE.md +106 -106
- package/docs/ROADMAP.md +359 -359
- package/docs/SETUP.md +663 -632
- package/docs/TOOLCHAIN.md +630 -630
- package/docs/VALIDATION.md +363 -363
- package/install.sh +40 -1058
- package/lefthook.yml +39 -39
- package/lib/agents/README.md +198 -198
- package/lib/agents/claude.plugin.json +28 -28
- package/lib/agents/cline.plugin.json +22 -22
- package/lib/agents/codex.plugin.json +19 -19
- package/lib/agents/copilot.plugin.json +24 -24
- package/lib/agents/cursor.plugin.json +25 -25
- package/lib/agents/kilocode.plugin.json +22 -22
- package/lib/agents/opencode.plugin.json +20 -20
- package/lib/agents/roo.plugin.json +23 -23
- package/lib/agents-config.js +2112 -2112
- package/lib/beads-health-check.js +143 -0
- package/lib/beads-setup.js +341 -0
- package/lib/beads-sync-scaffold.js +260 -0
- package/lib/commands/dev.js +513 -513
- package/lib/commands/plan.js +692 -692
- package/lib/commands/recommend.js +119 -119
- package/lib/commands/ship.js +377 -377
- package/lib/commands/status.js +378 -378
- package/lib/commands/validate.js +602 -602
- package/lib/context-merge.js +359 -359
- package/lib/dep-guard/analyzer.js +294 -294
- package/lib/dep-guard/behavior-detector.js +98 -98
- package/lib/dep-guard/contract-detector.js +162 -162
- package/lib/dep-guard/import-detector.js +498 -498
- package/lib/dep-guard/path-utils.js +13 -13
- package/lib/dep-guard/rubric.js +120 -120
- package/lib/dep-guard/task-parser.js +318 -318
- package/lib/detect-agent.js +191 -0
- package/lib/detect-worktree.js +47 -0
- package/lib/file-hash.js +26 -0
- package/lib/husky-migration.js +450 -0
- package/lib/lefthook-check.js +65 -0
- package/lib/pat-setup.js +207 -0
- package/lib/plugin-catalog.js +350 -350
- package/lib/plugin-manager.js +166 -166
- package/lib/plugin-recommender.js +141 -141
- package/lib/project-discovery.js +491 -491
- package/lib/setup-action-log.js +139 -0
- package/lib/setup-summary-renderer.js +106 -0
- package/lib/setup-utils.js +96 -0
- package/lib/setup.js +192 -118
- package/lib/smart-merge.js +64 -0
- package/lib/symlink-utils.js +81 -0
- package/lib/workflow-profiles.js +197 -197
- package/package.json +131 -129
- package/scripts/beads-context.sh +291 -0
- package/scripts/beads-context.test.js +563 -0
- package/scripts/behavioral-judge.sh +378 -0
- package/scripts/benchmark.js +85 -0
- package/scripts/branch-protection.js +183 -0
- package/scripts/check-agents.js +172 -0
- package/scripts/commitlint.js +42 -0
- package/scripts/conflict-detect.sh +323 -0
- package/scripts/dep-guard-analyze.js +71 -0
- package/scripts/dep-guard.sh +811 -0
- package/scripts/eval_win.py +249 -0
- package/scripts/file-index.sh +399 -0
- package/scripts/github-beads-sync/comment.mjs +64 -0
- package/scripts/github-beads-sync/config.mjs +148 -0
- package/scripts/github-beads-sync/github-api.mjs +131 -0
- package/scripts/github-beads-sync/index.mjs +332 -0
- package/scripts/github-beads-sync/label-mapper.mjs +54 -0
- package/scripts/github-beads-sync/mapping.mjs +78 -0
- package/scripts/github-beads-sync/reverse-sync-cli.mjs +31 -0
- package/scripts/github-beads-sync/reverse-sync.mjs +138 -0
- package/scripts/github-beads-sync/run-bd.mjs +159 -0
- package/scripts/github-beads-sync/sanitize.mjs +121 -0
- package/scripts/github-beads-sync.config.json +26 -0
- package/scripts/improve-command.js +375 -0
- package/scripts/lib/eval-runner.js +229 -0
- package/scripts/lib/eval-schema.js +135 -0
- package/scripts/lib/eval-storage.js +78 -0
- package/scripts/lib/grading.js +203 -0
- package/scripts/lib/transcript-parser.js +63 -0
- package/scripts/lint.js +47 -0
- package/scripts/migrate-to-bun-test.js +412 -0
- package/scripts/run-command-eval.js +236 -0
- package/scripts/smart-status.sh +782 -0
- package/scripts/sync-commands.js +571 -0
- package/scripts/sync-utils.sh +460 -0
- package/scripts/test-dashboard.js +123 -0
- package/scripts/test.js +44 -0
- package/scripts/validate.sh +94 -0
- package/skills/parallel-deep-research/SKILL.md +108 -108
- package/skills/parallel-deep-research/evals/README.md +27 -27
- package/skills/parallel-deep-research/evals/evals.json +62 -62
- package/skills/sonarcloud-analysis/SKILL.md +171 -171
- package/skills/sonarcloud-analysis/evals/README.md +27 -27
- package/skills/sonarcloud-analysis/evals/evals.json +50 -50
- package/skills/sonarcloud-analysis/references/api-reference.md +466 -466
- package/docs/WORKFLOW.md +0 -400
|
@@ -0,0 +1,412 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* Migration script: node:test + node:assert/strict → bun:test (CJS style)
|
|
4
|
+
*
|
|
5
|
+
* Strategy:
|
|
6
|
+
* - Use require('bun:test') to stay CJS-compatible (all other requires stay as-is)
|
|
7
|
+
* - Convert assert.* calls to expect() style
|
|
8
|
+
* - Handle multiline assert calls by processing the full file as text with careful regex
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
const fs = require('node:fs');
|
|
12
|
+
const path = require('node:path');
|
|
13
|
+
|
|
14
|
+
const TEST_DIRS = [
|
|
15
|
+
'test',
|
|
16
|
+
'test/cli',
|
|
17
|
+
'test/commands',
|
|
18
|
+
'test/e2e',
|
|
19
|
+
'test/integration',
|
|
20
|
+
'test/workflows',
|
|
21
|
+
];
|
|
22
|
+
|
|
23
|
+
const ROOT = path.join(__dirname, '..');
|
|
24
|
+
|
|
25
|
+
function getTestFiles() {
|
|
26
|
+
const files = [];
|
|
27
|
+
for (const dir of TEST_DIRS) {
|
|
28
|
+
const fullDir = path.join(ROOT, dir);
|
|
29
|
+
if (!fs.existsSync(fullDir)) continue;
|
|
30
|
+
const entries = fs.readdirSync(fullDir);
|
|
31
|
+
for (const entry of entries) {
|
|
32
|
+
if (entry.endsWith('.test.js')) {
|
|
33
|
+
files.push(path.join(fullDir, entry));
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
return files;
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
/**
|
|
41
|
+
* Find the closing paren index, accounting for nested parens/brackets/braces/strings
|
|
42
|
+
*/
|
|
43
|
+
function findMatchingParen(str, startIdx) {
|
|
44
|
+
let depth = 0;
|
|
45
|
+
let inString = false;
|
|
46
|
+
let stringChar = '';
|
|
47
|
+
let i = startIdx;
|
|
48
|
+
|
|
49
|
+
while (i < str.length) {
|
|
50
|
+
const ch = str[i];
|
|
51
|
+
const prev = i > 0 ? str[i - 1] : '';
|
|
52
|
+
|
|
53
|
+
if (inString) {
|
|
54
|
+
if (ch === stringChar && prev !== '\\') {
|
|
55
|
+
inString = false;
|
|
56
|
+
} else if (ch === '\\') {
|
|
57
|
+
i++; // skip next char
|
|
58
|
+
}
|
|
59
|
+
} else {
|
|
60
|
+
if (ch === '"' || ch === "'" || ch === '`') {
|
|
61
|
+
inString = true;
|
|
62
|
+
stringChar = ch;
|
|
63
|
+
} else if (ch === '(' || ch === '[' || ch === '{') {
|
|
64
|
+
depth++;
|
|
65
|
+
} else if (ch === ')' || ch === ']' || ch === '}') {
|
|
66
|
+
depth--;
|
|
67
|
+
if (depth === 0) return i;
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
i++;
|
|
71
|
+
}
|
|
72
|
+
return -1;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/**
|
|
76
|
+
* Find top-level comma in args string (not inside parens/brackets/strings)
|
|
77
|
+
*/
|
|
78
|
+
function findTopLevelComma(str) {
|
|
79
|
+
let depth = 0;
|
|
80
|
+
let inString = false;
|
|
81
|
+
let stringChar = '';
|
|
82
|
+
for (let i = 0; i < str.length; i++) {
|
|
83
|
+
const ch = str[i];
|
|
84
|
+
const prev = i > 0 ? str[i - 1] : '';
|
|
85
|
+
if (inString) {
|
|
86
|
+
if (ch === stringChar && prev !== '\\') inString = false;
|
|
87
|
+
else if (ch === '\\') i++;
|
|
88
|
+
} else {
|
|
89
|
+
if (ch === '"' || ch === "'" || ch === '`') {
|
|
90
|
+
inString = true;
|
|
91
|
+
stringChar = ch;
|
|
92
|
+
} else if (ch === '(' || ch === '[' || ch === '{') depth++;
|
|
93
|
+
else if (ch === ')' || ch === ']' || ch === '}') depth--;
|
|
94
|
+
else if (ch === ',' && depth === 0) return i;
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
return -1;
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/**
|
|
101
|
+
* Process assert.* calls in content using bracket-aware parsing
|
|
102
|
+
* Returns new content with all assert calls converted
|
|
103
|
+
*/
|
|
104
|
+
function convertAssertCalls(content) {
|
|
105
|
+
// We'll process the content character by character, replacing assert.X(...) calls
|
|
106
|
+
let result = '';
|
|
107
|
+
let i = 0;
|
|
108
|
+
|
|
109
|
+
while (i < content.length) {
|
|
110
|
+
// Look for "assert."
|
|
111
|
+
const assertIdx = content.indexOf('assert.', i);
|
|
112
|
+
if (assertIdx === -1) {
|
|
113
|
+
result += content.slice(i);
|
|
114
|
+
break;
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
// Copy everything up to "assert."
|
|
118
|
+
result += content.slice(i, assertIdx);
|
|
119
|
+
i = assertIdx;
|
|
120
|
+
|
|
121
|
+
// Determine which assert method
|
|
122
|
+
const rest = content.slice(i + 'assert.'.length);
|
|
123
|
+
|
|
124
|
+
let method = '';
|
|
125
|
+
let j = 0;
|
|
126
|
+
while (j < rest.length && /[a-zA-Z]/.test(rest[j])) {
|
|
127
|
+
method += rest[j];
|
|
128
|
+
j++;
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
// Check the character after the method name is '('
|
|
132
|
+
if (rest[j] !== '(') {
|
|
133
|
+
// Not a function call, copy and move on
|
|
134
|
+
result += content.slice(i, i + 'assert.'.length + method.length);
|
|
135
|
+
i += 'assert.'.length + method.length;
|
|
136
|
+
continue;
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
// Find the matching closing paren for the call
|
|
140
|
+
const openParenAbsolute = assertIdx + 'assert.'.length + method.length;
|
|
141
|
+
const closeParenAbsolute = findMatchingParen(content, openParenAbsolute);
|
|
142
|
+
|
|
143
|
+
if (closeParenAbsolute === -1) {
|
|
144
|
+
// Can't find closing paren, copy as-is
|
|
145
|
+
result += content.slice(i, openParenAbsolute + 1);
|
|
146
|
+
i = openParenAbsolute + 1;
|
|
147
|
+
continue;
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
// Extract the args string (everything between the outer parens)
|
|
151
|
+
const argsStr = content.slice(openParenAbsolute + 1, closeParenAbsolute);
|
|
152
|
+
|
|
153
|
+
// Convert based on method
|
|
154
|
+
let converted;
|
|
155
|
+
|
|
156
|
+
switch (method) {
|
|
157
|
+
case 'ok': {
|
|
158
|
+
// assert.ok(expr) or assert.ok(expr, 'msg')
|
|
159
|
+
const commaIdx = findTopLevelComma(argsStr);
|
|
160
|
+
const expr = commaIdx !== -1 ? argsStr.slice(0, commaIdx).trim() : argsStr.trim();
|
|
161
|
+
converted = `expect(${expr}).toBeTruthy()`;
|
|
162
|
+
break;
|
|
163
|
+
}
|
|
164
|
+
case 'strictEqual': {
|
|
165
|
+
// assert.strictEqual(a, b) or assert.strictEqual(a, b, 'msg')
|
|
166
|
+
const c1 = findTopLevelComma(argsStr);
|
|
167
|
+
if (c1 === -1) { converted = null; break; }
|
|
168
|
+
const a = argsStr.slice(0, c1).trim();
|
|
169
|
+
const rest2 = argsStr.slice(c1 + 1);
|
|
170
|
+
const c2 = findTopLevelComma(rest2);
|
|
171
|
+
const b = c2 !== -1 ? rest2.slice(0, c2).trim() : rest2.trim();
|
|
172
|
+
converted = `expect(${a}).toBe(${b})`;
|
|
173
|
+
break;
|
|
174
|
+
}
|
|
175
|
+
case 'notStrictEqual': {
|
|
176
|
+
const c1 = findTopLevelComma(argsStr);
|
|
177
|
+
if (c1 === -1) { converted = null; break; }
|
|
178
|
+
const a = argsStr.slice(0, c1).trim();
|
|
179
|
+
const rest2 = argsStr.slice(c1 + 1);
|
|
180
|
+
const c2 = findTopLevelComma(rest2);
|
|
181
|
+
const b = c2 !== -1 ? rest2.slice(0, c2).trim() : rest2.trim();
|
|
182
|
+
converted = `expect(${a}).not.toBe(${b})`;
|
|
183
|
+
break;
|
|
184
|
+
}
|
|
185
|
+
case 'deepStrictEqual': {
|
|
186
|
+
const c1 = findTopLevelComma(argsStr);
|
|
187
|
+
if (c1 === -1) { converted = null; break; }
|
|
188
|
+
const a = argsStr.slice(0, c1).trim();
|
|
189
|
+
const rest2 = argsStr.slice(c1 + 1);
|
|
190
|
+
const c2 = findTopLevelComma(rest2);
|
|
191
|
+
const b = c2 !== -1 ? rest2.slice(0, c2).trim() : rest2.trim();
|
|
192
|
+
converted = `expect(${a}).toEqual(${b})`;
|
|
193
|
+
break;
|
|
194
|
+
}
|
|
195
|
+
case 'equal': {
|
|
196
|
+
const c1 = findTopLevelComma(argsStr);
|
|
197
|
+
if (c1 === -1) { converted = null; break; }
|
|
198
|
+
const a = argsStr.slice(0, c1).trim();
|
|
199
|
+
const rest2 = argsStr.slice(c1 + 1);
|
|
200
|
+
const c2 = findTopLevelComma(rest2);
|
|
201
|
+
const b = c2 !== -1 ? rest2.slice(0, c2).trim() : rest2.trim();
|
|
202
|
+
converted = `expect(${a}).toBe(${b})`;
|
|
203
|
+
break;
|
|
204
|
+
}
|
|
205
|
+
case 'notEqual': {
|
|
206
|
+
const c1 = findTopLevelComma(argsStr);
|
|
207
|
+
if (c1 === -1) { converted = null; break; }
|
|
208
|
+
const a = argsStr.slice(0, c1).trim();
|
|
209
|
+
const rest2 = argsStr.slice(c1 + 1);
|
|
210
|
+
const c2 = findTopLevelComma(rest2);
|
|
211
|
+
const b = c2 !== -1 ? rest2.slice(0, c2).trim() : rest2.trim();
|
|
212
|
+
converted = `expect(${a}).not.toBe(${b})`;
|
|
213
|
+
break;
|
|
214
|
+
}
|
|
215
|
+
case 'fail': {
|
|
216
|
+
// assert.fail(msg) → throw new Error(msg)
|
|
217
|
+
converted = `throw new Error(${argsStr.trim()})`;
|
|
218
|
+
break;
|
|
219
|
+
}
|
|
220
|
+
case 'throws': {
|
|
221
|
+
// assert.throws(fn) or assert.throws(fn, options)
|
|
222
|
+
const c1 = findTopLevelComma(argsStr);
|
|
223
|
+
const fn = c1 !== -1 ? argsStr.slice(0, c1).trim() : argsStr.trim();
|
|
224
|
+
converted = `expect(${fn}).toThrow()`;
|
|
225
|
+
break;
|
|
226
|
+
}
|
|
227
|
+
case 'doesNotThrow': {
|
|
228
|
+
// assert.doesNotThrow(fn) or assert.doesNotThrow(fn, 'msg')
|
|
229
|
+
const c1 = findTopLevelComma(argsStr);
|
|
230
|
+
const fn = c1 !== -1 ? argsStr.slice(0, c1).trim() : argsStr.trim();
|
|
231
|
+
converted = `expect(${fn}).not.toThrow()`;
|
|
232
|
+
break;
|
|
233
|
+
}
|
|
234
|
+
case 'doesNotReject': {
|
|
235
|
+
// await assert.doesNotReject(asyncFn) → await expect(asyncFn()).resolves.not.toThrow()
|
|
236
|
+
// We handle this by making expect(asyncFn).resolves.not.toThrow()
|
|
237
|
+
// Check if there's an 'await' before this call
|
|
238
|
+
const c1 = findTopLevelComma(argsStr);
|
|
239
|
+
const fn = c1 !== -1 ? argsStr.slice(0, c1).trim() : argsStr.trim();
|
|
240
|
+
// The fn is an async function expression, call it: fn()
|
|
241
|
+
// But we'll wrap it: expect(fn).resolves.not.toThrow()
|
|
242
|
+
converted = `expect(${fn}).resolves.not.toThrow()`;
|
|
243
|
+
break;
|
|
244
|
+
}
|
|
245
|
+
case 'match': {
|
|
246
|
+
// assert.match(str, regex) → expect(str).toMatch(regex)
|
|
247
|
+
const c1 = findTopLevelComma(argsStr);
|
|
248
|
+
if (c1 === -1) { converted = null; break; }
|
|
249
|
+
const str = argsStr.slice(0, c1).trim();
|
|
250
|
+
const rest2 = argsStr.slice(c1 + 1).trim();
|
|
251
|
+
// rest2 may have trailing message, but regex is the main arg
|
|
252
|
+
const c2 = findTopLevelComma(rest2);
|
|
253
|
+
const regex = c2 !== -1 ? rest2.slice(0, c2).trim() : rest2.trim();
|
|
254
|
+
converted = `expect(${str}).toMatch(${regex})`;
|
|
255
|
+
break;
|
|
256
|
+
}
|
|
257
|
+
default:
|
|
258
|
+
converted = null;
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
if (converted !== null) {
|
|
262
|
+
result += converted;
|
|
263
|
+
} else {
|
|
264
|
+
// Couldn't convert, copy as-is
|
|
265
|
+
result += content.slice(i, closeParenAbsolute + 1);
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
i = closeParenAbsolute + 1;
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
return result;
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
function migrateFile(filePath) {
|
|
275
|
+
let content = fs.readFileSync(filePath, 'utf-8');
|
|
276
|
+
|
|
277
|
+
// Skip if already using bun:test
|
|
278
|
+
if (content.includes("'bun:test'") || content.includes('"bun:test"')) {
|
|
279
|
+
return { skipped: true, reason: 'already using bun:test' };
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
// Skip if not using node:test
|
|
283
|
+
if (!content.includes("'node:test'") && !content.includes('"node:test"')) {
|
|
284
|
+
return { skipped: true, reason: 'does not use node:test' };
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
const original = content;
|
|
288
|
+
|
|
289
|
+
// ---- Step 1: Extract which symbols are imported from node:test ----
|
|
290
|
+
const nodeTestImportMatch = content.match(
|
|
291
|
+
/const\s*\{([^}]+)\}\s*=\s*require\(['"]node:test['"]\)/
|
|
292
|
+
);
|
|
293
|
+
|
|
294
|
+
let importedSymbols = [];
|
|
295
|
+
if (nodeTestImportMatch) {
|
|
296
|
+
importedSymbols = nodeTestImportMatch[1]
|
|
297
|
+
.split(',')
|
|
298
|
+
.map(s => s.trim())
|
|
299
|
+
.filter(Boolean);
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
// ---- Step 2: Map node:test symbols to bun:test equivalents ----
|
|
303
|
+
const symbolMap = {
|
|
304
|
+
'describe': 'describe',
|
|
305
|
+
'test': 'test',
|
|
306
|
+
'it': 'it',
|
|
307
|
+
'before': 'beforeAll',
|
|
308
|
+
'after': 'afterAll',
|
|
309
|
+
'beforeEach': 'beforeEach',
|
|
310
|
+
'afterEach': 'afterEach',
|
|
311
|
+
};
|
|
312
|
+
|
|
313
|
+
// Build the bun:test symbols list
|
|
314
|
+
// Handle aliased imports like "after: _after"
|
|
315
|
+
const bunSymbols = [];
|
|
316
|
+
for (const sym of importedSymbols) {
|
|
317
|
+
if (sym.includes(':')) {
|
|
318
|
+
// e.g. "after: _after"
|
|
319
|
+
const [origName, alias] = sym.split(':').map(s => s.trim());
|
|
320
|
+
const bunEquiv = symbolMap[origName] || origName;
|
|
321
|
+
bunSymbols.push(`${bunEquiv}: ${alias}`);
|
|
322
|
+
} else {
|
|
323
|
+
const bunEquiv = symbolMap[sym] || sym;
|
|
324
|
+
if (bunEquiv !== sym) {
|
|
325
|
+
// e.g. before → beforeAll: we need the body to still call `before()`,
|
|
326
|
+
// so we alias: beforeAll: before -- wait, no.
|
|
327
|
+
// The body uses "before()" which came from the import.
|
|
328
|
+
// So if we imported "before" as "before" but now want "beforeAll",
|
|
329
|
+
// we need to rename the usage in the body too, OR import as:
|
|
330
|
+
// const { beforeAll: before } = require('bun:test')
|
|
331
|
+
// That way the body code is unchanged.
|
|
332
|
+
bunSymbols.push(`${bunEquiv}: ${sym}`);
|
|
333
|
+
} else {
|
|
334
|
+
bunSymbols.push(sym);
|
|
335
|
+
}
|
|
336
|
+
}
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
// Add expect if assert is used
|
|
340
|
+
const hasAssert = content.includes('assert.');
|
|
341
|
+
if (hasAssert && !bunSymbols.includes('expect')) {
|
|
342
|
+
bunSymbols.push('expect');
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
// ---- Step 3: Replace the node:test require line ----
|
|
346
|
+
const bunRequire = `const { ${bunSymbols.join(', ')} } = require('bun:test');`;
|
|
347
|
+
content = content.replace(
|
|
348
|
+
/const\s*\{[^}]+\}\s*=\s*require\(['"]node:test['"]\)\s*;?/,
|
|
349
|
+
bunRequire
|
|
350
|
+
);
|
|
351
|
+
|
|
352
|
+
// ---- Step 4: Remove node:assert/strict require line ----
|
|
353
|
+
content = content.replace(
|
|
354
|
+
/\nconst\s+assert\s*=\s*require\(['"]node:assert\/strict['"]\)\s*;?/g,
|
|
355
|
+
''
|
|
356
|
+
);
|
|
357
|
+
content = content.replace(
|
|
358
|
+
/\nconst\s+assert\s*=\s*require\(['"]node:assert['"]\)\s*;?/g,
|
|
359
|
+
''
|
|
360
|
+
);
|
|
361
|
+
|
|
362
|
+
// ---- Step 5: Convert assert.* calls ----
|
|
363
|
+
if (hasAssert) {
|
|
364
|
+
content = convertAssertCalls(content);
|
|
365
|
+
}
|
|
366
|
+
|
|
367
|
+
// ---- Step 6: Clean up extra blank lines ----
|
|
368
|
+
content = content.replace(/\n{3,}/g, '\n\n');
|
|
369
|
+
|
|
370
|
+
if (content === original) {
|
|
371
|
+
return { skipped: true, reason: 'no changes made' };
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
fs.writeFileSync(filePath, content, 'utf-8');
|
|
375
|
+
return { migrated: true };
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
// ---- Main ----
|
|
379
|
+
const files = getTestFiles();
|
|
380
|
+
console.log(`Found ${files.length} test files to check\n`);
|
|
381
|
+
|
|
382
|
+
let migrated = 0;
|
|
383
|
+
let skipped = 0;
|
|
384
|
+
const errors = [];
|
|
385
|
+
|
|
386
|
+
for (const file of files) {
|
|
387
|
+
const rel = path.relative(ROOT, file);
|
|
388
|
+
try {
|
|
389
|
+
const result = migrateFile(file);
|
|
390
|
+
if (result.migrated) {
|
|
391
|
+
console.log(`✓ Migrated: ${rel}`);
|
|
392
|
+
migrated++;
|
|
393
|
+
} else {
|
|
394
|
+
console.log(` Skipped: ${rel} (${result.reason})`);
|
|
395
|
+
skipped++;
|
|
396
|
+
}
|
|
397
|
+
} catch (err) {
|
|
398
|
+
console.error(`✗ Error: ${rel}: ${err.message}`);
|
|
399
|
+
errors.push({ file: rel, error: err.message });
|
|
400
|
+
}
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
console.log(`\n--- Summary ---`);
|
|
404
|
+
console.log(`Migrated: ${migrated}`);
|
|
405
|
+
console.log(`Skipped: ${skipped}`);
|
|
406
|
+
console.log(`Errors: ${errors.length}`);
|
|
407
|
+
if (errors.length > 0) {
|
|
408
|
+
for (const e of errors) {
|
|
409
|
+
console.error(` ✗ ${e.file}: ${e.error}`);
|
|
410
|
+
}
|
|
411
|
+
process.exit(1);
|
|
412
|
+
}
|
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* End-to-end eval pipeline — wire it all together.
|
|
3
|
+
*
|
|
4
|
+
* Load eval set → create worktree → for each query: run setup → execute
|
|
5
|
+
* command → parse transcript → grade → run teardown → reset worktree →
|
|
6
|
+
* save results → destroy worktree.
|
|
7
|
+
*
|
|
8
|
+
* Usage:
|
|
9
|
+
* bun scripts/run-command-eval.js <eval-set-path> [--timeout <ms>] [--threshold <score>]
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
const { execFileSync } = require('child_process');
|
|
13
|
+
const { loadEvalSet } = require('./lib/eval-schema');
|
|
14
|
+
const { parseTranscript } = require('./lib/transcript-parser');
|
|
15
|
+
const {
|
|
16
|
+
createEvalWorktree,
|
|
17
|
+
destroyEvalWorktree,
|
|
18
|
+
resetWorktree,
|
|
19
|
+
executeCommand,
|
|
20
|
+
} = require('./lib/eval-runner');
|
|
21
|
+
const { gradeTranscript } = require('./lib/grading');
|
|
22
|
+
const { saveEvalResult } = require('./lib/eval-storage');
|
|
23
|
+
|
|
24
|
+
// ---------------------------------------------------------------------------
|
|
25
|
+
// parseArgs
|
|
26
|
+
// ---------------------------------------------------------------------------
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* Parse CLI arguments.
|
|
30
|
+
*
|
|
31
|
+
* @param {string[]} argv — process.argv.slice(2)
|
|
32
|
+
* @returns {{ evalSetPath: string, timeout: number, threshold: number }}
|
|
33
|
+
*/
|
|
34
|
+
function parseArgs(argv) {
|
|
35
|
+
let evalSetPath = null;
|
|
36
|
+
let timeout = 120000;
|
|
37
|
+
let threshold = 0.7;
|
|
38
|
+
|
|
39
|
+
for (let i = 0; i < argv.length; i++) {
|
|
40
|
+
if (argv[i] === '--timeout' && i + 1 < argv.length) {
|
|
41
|
+
timeout = Number(argv[i + 1]);
|
|
42
|
+
i++; // skip next
|
|
43
|
+
} else if (argv[i] === '--threshold' && i + 1 < argv.length) {
|
|
44
|
+
threshold = Number(argv[i + 1]);
|
|
45
|
+
i++; // skip next
|
|
46
|
+
} else if (!argv[i].startsWith('--')) {
|
|
47
|
+
evalSetPath = argv[i];
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
if (!evalSetPath) {
|
|
52
|
+
throw new Error('Usage: run-command-eval <eval-set-path> [--timeout <ms>] [--threshold <score>]');
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
return { evalSetPath, timeout, threshold };
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
// ---------------------------------------------------------------------------
|
|
59
|
+
// runShellCommand — run a setup/teardown shell command in a worktree
|
|
60
|
+
// ---------------------------------------------------------------------------
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* Run a shell command (setup or teardown) in the worktree.
|
|
64
|
+
* Uses execFileSync with bash -c to avoid direct shell injection.
|
|
65
|
+
*
|
|
66
|
+
* @param {string} command — the shell command string
|
|
67
|
+
* @param {string} worktreePath — cwd for the command
|
|
68
|
+
*/
|
|
69
|
+
function runShellCommand(command, worktreePath) {
|
|
70
|
+
execFileSync('bash', ['-c', command], {
|
|
71
|
+
cwd: worktreePath,
|
|
72
|
+
encoding: 'utf-8',
|
|
73
|
+
stdio: ['pipe', 'pipe', 'pipe'],
|
|
74
|
+
});
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
// ---------------------------------------------------------------------------
|
|
78
|
+
// runEvalPipeline
|
|
79
|
+
// ---------------------------------------------------------------------------
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* Main orchestrator — run the full eval pipeline.
|
|
83
|
+
*
|
|
84
|
+
* @param {string} evalSetPath — path to the .eval.json file
|
|
85
|
+
* @param {object} [options]
|
|
86
|
+
* @param {number} [options.timeout=120000] — per-query timeout in ms
|
|
87
|
+
* @param {number} [options.threshold=0.7] — pass/fail score cutoff
|
|
88
|
+
* @param {Function} [options._invokeGrader] — injectable grader for testing
|
|
89
|
+
* @param {string} [options._basePath] — eval-logs base path for testing
|
|
90
|
+
* @param {boolean} [options._skipWorktree=false] — skip worktree creation for unit tests
|
|
91
|
+
* @param {Function} [options._executeOverride] — injectable command executor for testing
|
|
92
|
+
* @returns {Promise<{ command: string, results: Array, overall_score: number, passed: boolean, duration_ms: number }>}
|
|
93
|
+
*/
|
|
94
|
+
async function runEvalPipeline(evalSetPath, options = {}) {
|
|
95
|
+
const timeout = options.timeout || 120000;
|
|
96
|
+
const threshold = options.threshold != null ? options.threshold : 0.7;
|
|
97
|
+
const skipWorktree = options._skipWorktree || false;
|
|
98
|
+
const execOverride = options._executeOverride || null;
|
|
99
|
+
const invokeGrader = options._invokeGrader || null;
|
|
100
|
+
const basePath = options._basePath || undefined;
|
|
101
|
+
|
|
102
|
+
const startTime = Date.now();
|
|
103
|
+
|
|
104
|
+
// 1. Load eval set
|
|
105
|
+
const evalSet = loadEvalSet(evalSetPath);
|
|
106
|
+
const { command, queries } = evalSet;
|
|
107
|
+
|
|
108
|
+
// 2. Create eval worktree (unless skipped for testing)
|
|
109
|
+
let worktreePath = null;
|
|
110
|
+
if (!skipWorktree) {
|
|
111
|
+
const wt = await createEvalWorktree();
|
|
112
|
+
worktreePath = wt.path;
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
const queryResults = [];
|
|
116
|
+
|
|
117
|
+
try {
|
|
118
|
+
// 3. For each query in eval set
|
|
119
|
+
for (const query of queries) {
|
|
120
|
+
// a. Run setup command if present
|
|
121
|
+
if (query.setup && worktreePath) {
|
|
122
|
+
try {
|
|
123
|
+
runShellCommand(query.setup, worktreePath);
|
|
124
|
+
} catch (_err) {
|
|
125
|
+
// Setup failure is non-fatal — continue with the query
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
// b. Execute the command
|
|
130
|
+
let execResult;
|
|
131
|
+
if (execOverride) {
|
|
132
|
+
execResult = await execOverride(command, query.prompt, worktreePath, timeout);
|
|
133
|
+
} else {
|
|
134
|
+
execResult = await executeCommand(command, query.prompt, worktreePath, timeout);
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
// c. Parse transcript
|
|
138
|
+
const transcript = parseTranscript(execResult.stdout);
|
|
139
|
+
|
|
140
|
+
// d. Grade with gradeTranscript
|
|
141
|
+
let gradeResult;
|
|
142
|
+
const gradeOpts = { timeout };
|
|
143
|
+
if (invokeGrader) {
|
|
144
|
+
gradeResult = await invokeGrader(transcript, query.assertions, gradeOpts);
|
|
145
|
+
} else {
|
|
146
|
+
gradeResult = await gradeTranscript(transcript, query.assertions, gradeOpts);
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
// e. Run teardown command if present
|
|
150
|
+
if (query.teardown && worktreePath) {
|
|
151
|
+
try {
|
|
152
|
+
runShellCommand(query.teardown, worktreePath);
|
|
153
|
+
} catch (_err) {
|
|
154
|
+
// Teardown failure is non-fatal
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
// f. Reset worktree
|
|
159
|
+
if (worktreePath) {
|
|
160
|
+
await resetWorktree(worktreePath);
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
// g. Collect results
|
|
164
|
+
queryResults.push({
|
|
165
|
+
name: query.name,
|
|
166
|
+
prompt: query.prompt,
|
|
167
|
+
score: gradeResult.score,
|
|
168
|
+
assertions: gradeResult.assertions,
|
|
169
|
+
exitCode: execResult.exitCode,
|
|
170
|
+
timedOut: execResult.timedOut,
|
|
171
|
+
});
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
// 4. Compute overall score (average of query scores)
|
|
175
|
+
const totalScore = queryResults.reduce((sum, qr) => sum + qr.score, 0);
|
|
176
|
+
const overallScore = queryResults.length > 0 ? totalScore / queryResults.length : 0;
|
|
177
|
+
|
|
178
|
+
// 5. Build the final result
|
|
179
|
+
const endTime = Date.now();
|
|
180
|
+
const result = {
|
|
181
|
+
command,
|
|
182
|
+
results: queryResults,
|
|
183
|
+
overall_score: overallScore,
|
|
184
|
+
passed: overallScore >= threshold,
|
|
185
|
+
duration_ms: endTime - startTime,
|
|
186
|
+
timestamp: new Date().toISOString(),
|
|
187
|
+
};
|
|
188
|
+
|
|
189
|
+
// 6. Save results
|
|
190
|
+
const savedPath = saveEvalResult(result, basePath);
|
|
191
|
+
result.savedTo = savedPath;
|
|
192
|
+
|
|
193
|
+
return result;
|
|
194
|
+
} finally {
|
|
195
|
+
// 7. Destroy worktree (in finally block)
|
|
196
|
+
if (worktreePath) {
|
|
197
|
+
try {
|
|
198
|
+
await destroyEvalWorktree(worktreePath);
|
|
199
|
+
} catch (_err) {
|
|
200
|
+
// Best-effort cleanup
|
|
201
|
+
}
|
|
202
|
+
}
|
|
203
|
+
}
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
// ---------------------------------------------------------------------------
|
|
207
|
+
// CLI entry point
|
|
208
|
+
// ---------------------------------------------------------------------------
|
|
209
|
+
|
|
210
|
+
if (require.main === module) {
|
|
211
|
+
const args = parseArgs(process.argv.slice(2));
|
|
212
|
+
|
|
213
|
+
runEvalPipeline(args.evalSetPath, args)
|
|
214
|
+
.then((result) => {
|
|
215
|
+
const passedCount = result.results.filter((r) => r.score >= 0.5).length;
|
|
216
|
+
const totalCount = result.results.length;
|
|
217
|
+
|
|
218
|
+
console.log(`Command: ${result.command}`);
|
|
219
|
+
console.log(`Queries: ${passedCount}/${totalCount} passed`);
|
|
220
|
+
console.log(`Overall score: ${result.overall_score.toFixed(2)}`);
|
|
221
|
+
console.log(
|
|
222
|
+
`Result: ${result.passed ? 'PASS' : 'FAIL'} (threshold: ${args.threshold})`
|
|
223
|
+
);
|
|
224
|
+
if (result.savedTo) {
|
|
225
|
+
console.log(`Saved to: ${result.savedTo}`);
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
process.exit(result.passed ? 0 : 1);
|
|
229
|
+
})
|
|
230
|
+
.catch((err) => {
|
|
231
|
+
console.error(`Error: ${err.message}`);
|
|
232
|
+
process.exit(1);
|
|
233
|
+
});
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
module.exports = { runEvalPipeline, parseArgs };
|