forge-workflow 0.0.3 → 0.0.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (209) hide show
  1. package/.claude/commands/dev.md +340 -314
  2. package/.claude/commands/plan.md +521 -478
  3. package/.claude/commands/premerge.md +176 -179
  4. package/.claude/commands/research.md +42 -42
  5. package/.claude/commands/review.md +442 -442
  6. package/.claude/commands/rollback.md +721 -721
  7. package/.claude/commands/ship.md +164 -134
  8. package/.claude/commands/sonarcloud.md +152 -152
  9. package/.claude/commands/status.md +48 -77
  10. package/.claude/commands/validate.md +282 -237
  11. package/.claude/commands/verify.md +221 -221
  12. package/.claude/rules/greptile-review-process.md +285 -285
  13. package/.claude/rules/workflow.md +105 -105
  14. package/.claude/scripts/greptile-resolve.sh +526 -526
  15. package/.claude/scripts/load-env.sh +32 -32
  16. package/.cline/workflows/dev.md +337 -311
  17. package/.cline/workflows/plan.md +518 -475
  18. package/.cline/workflows/premerge.md +173 -176
  19. package/.cline/workflows/research.md +39 -39
  20. package/.cline/workflows/review.md +439 -439
  21. package/.cline/workflows/rollback.md +718 -718
  22. package/.cline/workflows/ship.md +161 -131
  23. package/.cline/workflows/sonarcloud.md +146 -146
  24. package/.cline/workflows/status.md +45 -74
  25. package/.cline/workflows/validate.md +279 -234
  26. package/.cline/workflows/verify.md +218 -218
  27. package/.codex/config.toml +11 -11
  28. package/.codex/skills/dev/SKILL.md +340 -314
  29. package/.codex/skills/plan/SKILL.md +521 -478
  30. package/.codex/skills/premerge/SKILL.md +176 -179
  31. package/.codex/skills/research/SKILL.md +42 -42
  32. package/.codex/skills/review/SKILL.md +442 -442
  33. package/.codex/skills/rollback/SKILL.md +721 -721
  34. package/.codex/skills/ship/SKILL.md +164 -134
  35. package/.codex/skills/sonarcloud/SKILL.md +149 -149
  36. package/.codex/skills/status/SKILL.md +48 -77
  37. package/.codex/skills/validate/SKILL.md +282 -237
  38. package/.codex/skills/verify/SKILL.md +221 -221
  39. package/.cursor/commands/dev.md +337 -311
  40. package/.cursor/commands/plan.md +518 -475
  41. package/.cursor/commands/premerge.md +173 -176
  42. package/.cursor/commands/research.md +39 -39
  43. package/.cursor/commands/review.md +439 -439
  44. package/.cursor/commands/rollback.md +718 -718
  45. package/.cursor/commands/ship.md +161 -131
  46. package/.cursor/commands/sonarcloud.md +146 -146
  47. package/.cursor/commands/status.md +45 -74
  48. package/.cursor/commands/validate.md +279 -234
  49. package/.cursor/commands/verify.md +218 -218
  50. package/.cursor/rules/permissions-guidance.mdc +37 -37
  51. package/.forge/hooks/check-tdd.js +240 -240
  52. package/.github/PLUGIN_TEMPLATE.json +32 -32
  53. package/.github/prompts/dev.prompt.md +342 -316
  54. package/.github/prompts/plan.prompt.md +523 -480
  55. package/.github/prompts/premerge.prompt.md +178 -181
  56. package/.github/prompts/research.prompt.md +44 -44
  57. package/.github/prompts/review.prompt.md +444 -444
  58. package/.github/prompts/rollback.prompt.md +723 -723
  59. package/.github/prompts/ship.prompt.md +166 -136
  60. package/.github/prompts/sonarcloud.prompt.md +151 -151
  61. package/.github/prompts/status.prompt.md +50 -79
  62. package/.github/prompts/validate.prompt.md +284 -239
  63. package/.github/prompts/verify.prompt.md +223 -223
  64. package/.github/workflows/beads-to-github.yml +56 -0
  65. package/.github/workflows/github-to-beads.yml +97 -0
  66. package/.kilocode/workflows/dev.md +341 -315
  67. package/.kilocode/workflows/plan.md +522 -479
  68. package/.kilocode/workflows/premerge.md +177 -180
  69. package/.kilocode/workflows/research.md +43 -43
  70. package/.kilocode/workflows/review.md +443 -443
  71. package/.kilocode/workflows/rollback.md +722 -722
  72. package/.kilocode/workflows/ship.md +165 -135
  73. package/.kilocode/workflows/sonarcloud.md +150 -150
  74. package/.kilocode/workflows/status.md +49 -78
  75. package/.kilocode/workflows/validate.md +283 -238
  76. package/.kilocode/workflows/verify.md +222 -222
  77. package/.mcp.json.example +12 -12
  78. package/.opencode/commands/dev.md +340 -314
  79. package/.opencode/commands/plan.md +521 -478
  80. package/.opencode/commands/premerge.md +176 -179
  81. package/.opencode/commands/research.md +42 -42
  82. package/.opencode/commands/review.md +442 -442
  83. package/.opencode/commands/rollback.md +721 -721
  84. package/.opencode/commands/ship.md +164 -134
  85. package/.opencode/commands/sonarcloud.md +149 -149
  86. package/.opencode/commands/status.md +48 -77
  87. package/.opencode/commands/validate.md +282 -237
  88. package/.opencode/commands/verify.md +221 -221
  89. package/.roo/commands/dev.md +341 -315
  90. package/.roo/commands/plan.md +522 -479
  91. package/.roo/commands/premerge.md +177 -180
  92. package/.roo/commands/research.md +43 -43
  93. package/.roo/commands/review.md +443 -443
  94. package/.roo/commands/rollback.md +722 -722
  95. package/.roo/commands/ship.md +165 -135
  96. package/.roo/commands/sonarcloud.md +150 -150
  97. package/.roo/commands/status.md +49 -78
  98. package/.roo/commands/validate.md +283 -238
  99. package/.roo/commands/verify.md +222 -222
  100. package/AGENTS.md +175 -169
  101. package/CLAUDE.md +100 -99
  102. package/LICENSE +21 -21
  103. package/README.md +429 -414
  104. package/bin/forge-cmd.js +313 -313
  105. package/bin/{forge-validate.js → forge-preflight.js} +309 -303
  106. package/bin/forge.js +4596 -4232
  107. package/docs/AGENT_INSTALL_PROMPT.md +342 -342
  108. package/docs/BEADS_GITHUB_SYNC.md +251 -0
  109. package/docs/ENHANCED_ONBOARDING.md +602 -602
  110. package/docs/EXAMPLES.md +482 -482
  111. package/docs/GREPTILE_SETUP.md +400 -400
  112. package/docs/MANUAL_REVIEW_GUIDE.md +106 -106
  113. package/docs/ROADMAP.md +359 -359
  114. package/docs/SETUP.md +663 -632
  115. package/docs/TOOLCHAIN.md +630 -630
  116. package/docs/VALIDATION.md +363 -363
  117. package/install.sh +40 -1058
  118. package/lefthook.yml +39 -39
  119. package/lib/agents/README.md +198 -198
  120. package/lib/agents/claude.plugin.json +28 -28
  121. package/lib/agents/cline.plugin.json +22 -22
  122. package/lib/agents/codex.plugin.json +19 -19
  123. package/lib/agents/copilot.plugin.json +24 -24
  124. package/lib/agents/cursor.plugin.json +25 -25
  125. package/lib/agents/kilocode.plugin.json +22 -22
  126. package/lib/agents/opencode.plugin.json +20 -20
  127. package/lib/agents/roo.plugin.json +23 -23
  128. package/lib/agents-config.js +2112 -2112
  129. package/lib/beads-health-check.js +143 -0
  130. package/lib/beads-setup.js +341 -0
  131. package/lib/beads-sync-scaffold.js +260 -0
  132. package/lib/commands/dev.js +513 -513
  133. package/lib/commands/plan.js +692 -692
  134. package/lib/commands/recommend.js +119 -119
  135. package/lib/commands/ship.js +377 -377
  136. package/lib/commands/status.js +378 -378
  137. package/lib/commands/validate.js +602 -602
  138. package/lib/context-merge.js +359 -359
  139. package/lib/dep-guard/analyzer.js +294 -294
  140. package/lib/dep-guard/behavior-detector.js +98 -98
  141. package/lib/dep-guard/contract-detector.js +162 -162
  142. package/lib/dep-guard/import-detector.js +498 -498
  143. package/lib/dep-guard/path-utils.js +13 -13
  144. package/lib/dep-guard/rubric.js +120 -120
  145. package/lib/dep-guard/task-parser.js +318 -318
  146. package/lib/detect-agent.js +191 -0
  147. package/lib/detect-worktree.js +47 -0
  148. package/lib/file-hash.js +26 -0
  149. package/lib/husky-migration.js +450 -0
  150. package/lib/lefthook-check.js +65 -0
  151. package/lib/pat-setup.js +207 -0
  152. package/lib/plugin-catalog.js +350 -350
  153. package/lib/plugin-manager.js +166 -166
  154. package/lib/plugin-recommender.js +141 -141
  155. package/lib/project-discovery.js +491 -491
  156. package/lib/setup-action-log.js +139 -0
  157. package/lib/setup-summary-renderer.js +106 -0
  158. package/lib/setup-utils.js +96 -0
  159. package/lib/setup.js +192 -118
  160. package/lib/smart-merge.js +64 -0
  161. package/lib/symlink-utils.js +81 -0
  162. package/lib/workflow-profiles.js +197 -197
  163. package/package.json +131 -129
  164. package/scripts/beads-context.sh +291 -0
  165. package/scripts/beads-context.test.js +563 -0
  166. package/scripts/behavioral-judge.sh +378 -0
  167. package/scripts/benchmark.js +85 -0
  168. package/scripts/branch-protection.js +183 -0
  169. package/scripts/check-agents.js +172 -0
  170. package/scripts/commitlint.js +42 -0
  171. package/scripts/conflict-detect.sh +323 -0
  172. package/scripts/dep-guard-analyze.js +71 -0
  173. package/scripts/dep-guard.sh +811 -0
  174. package/scripts/eval_win.py +249 -0
  175. package/scripts/file-index.sh +399 -0
  176. package/scripts/github-beads-sync/comment.mjs +64 -0
  177. package/scripts/github-beads-sync/config.mjs +148 -0
  178. package/scripts/github-beads-sync/github-api.mjs +131 -0
  179. package/scripts/github-beads-sync/index.mjs +332 -0
  180. package/scripts/github-beads-sync/label-mapper.mjs +54 -0
  181. package/scripts/github-beads-sync/mapping.mjs +78 -0
  182. package/scripts/github-beads-sync/reverse-sync-cli.mjs +31 -0
  183. package/scripts/github-beads-sync/reverse-sync.mjs +138 -0
  184. package/scripts/github-beads-sync/run-bd.mjs +159 -0
  185. package/scripts/github-beads-sync/sanitize.mjs +121 -0
  186. package/scripts/github-beads-sync.config.json +26 -0
  187. package/scripts/improve-command.js +375 -0
  188. package/scripts/lib/eval-runner.js +229 -0
  189. package/scripts/lib/eval-schema.js +135 -0
  190. package/scripts/lib/eval-storage.js +78 -0
  191. package/scripts/lib/grading.js +203 -0
  192. package/scripts/lib/transcript-parser.js +63 -0
  193. package/scripts/lint.js +47 -0
  194. package/scripts/migrate-to-bun-test.js +412 -0
  195. package/scripts/run-command-eval.js +236 -0
  196. package/scripts/smart-status.sh +782 -0
  197. package/scripts/sync-commands.js +571 -0
  198. package/scripts/sync-utils.sh +460 -0
  199. package/scripts/test-dashboard.js +123 -0
  200. package/scripts/test.js +44 -0
  201. package/scripts/validate.sh +94 -0
  202. package/skills/parallel-deep-research/SKILL.md +108 -108
  203. package/skills/parallel-deep-research/evals/README.md +27 -27
  204. package/skills/parallel-deep-research/evals/evals.json +62 -62
  205. package/skills/sonarcloud-analysis/SKILL.md +171 -171
  206. package/skills/sonarcloud-analysis/evals/README.md +27 -27
  207. package/skills/sonarcloud-analysis/evals/evals.json +50 -50
  208. package/skills/sonarcloud-analysis/references/api-reference.md +466 -466
  209. package/docs/WORKFLOW.md +0 -400
@@ -0,0 +1,375 @@
1
+ /**
2
+ * Semi-autonomous improvement loop: analyze eval failures, rewrite command,
3
+ * re-evaluate, and stop on regression or plateau.
4
+ *
5
+ * Usage:
6
+ * bun scripts/improve-command.js <command-path> --eval-set <path> [--max-iterations <N>]
7
+ */
8
+
9
+ const { execFileSync } = require('child_process');
10
+ const fs = require('fs');
11
+ const { DEFAULT_BASE_PATH, loadEvalHistory } = require('./lib/eval-storage');
12
+
13
+ // ---------------------------------------------------------------------------
14
+ // analyzeFailures
15
+ // ---------------------------------------------------------------------------
16
+
17
+ /**
18
+ * Extract failing assertions with their query context from an eval result.
19
+ *
20
+ * @param {object} evalResult - result from runEvalPipeline
21
+ * @returns {Array<{ query: string, assertion: string, reasoning: string }>}
22
+ */
23
+ function analyzeFailures(evalResult) {
24
+ const failures = [];
25
+
26
+ for (const queryResult of evalResult.results) {
27
+ for (const assertion of queryResult.assertions) {
28
+ if (!assertion.pass) {
29
+ failures.push({
30
+ query: queryResult.prompt,
31
+ assertion: assertion.check,
32
+ reasoning: assertion.reasoning,
33
+ });
34
+ }
35
+ }
36
+ }
37
+
38
+ return failures;
39
+ }
40
+
41
+ // ---------------------------------------------------------------------------
42
+ // buildRewritePrompt
43
+ // ---------------------------------------------------------------------------
44
+
45
+ /**
46
+ * Build a prompt asking for a command rewrite that fixes the identified failures.
47
+ *
48
+ * @param {string} commandContent - current command markdown content
49
+ * @param {Array<{ query: string, assertion: string, reasoning: string }>} failures
50
+ * @param {object[]} history - array of prior eval results (from loadEvalHistory)
51
+ * @returns {string}
52
+ */
53
+ function buildRewritePrompt(commandContent, failures, history) {
54
+ let prompt = '';
55
+
56
+ prompt += 'You are a command prompt engineer. Rewrite the following command to fix the failing assertions.\n\n';
57
+
58
+ prompt += '## Current Command\n\n';
59
+ prompt += commandContent + '\n\n';
60
+
61
+ prompt += '## Failing Assertions\n\n';
62
+ for (const failure of failures) {
63
+ prompt += `- Query: "${failure.query}"\n`;
64
+ prompt += ` Assertion: "${failure.assertion}"\n`;
65
+ prompt += ` Reasoning: "${failure.reasoning}"\n\n`;
66
+ }
67
+
68
+ if (history && history.length > 0) {
69
+ prompt += '## Prior Eval Attempts\n\n';
70
+ prompt += 'Avoid repeating approaches that have already been tried. Here is a summary of prior attempts:\n\n';
71
+ for (let i = 0; i < history.length; i++) {
72
+ const attempt = history[i];
73
+ const failCount = attempt.results
74
+ ? attempt.results.filter((result) => result.assertions && result.assertions.some((assertion) => !assertion.pass)).length
75
+ : 0;
76
+ prompt += `- Attempt ${i + 1}: score ${attempt.overall_score}, ${failCount} failing queries\n`;
77
+ }
78
+ prompt += '\n';
79
+
80
+ // Detect flaky assertions: the same check both passed and failed across sessions.
81
+ const assertionOutcomes = new Map();
82
+ for (const attempt of history) {
83
+ if (!attempt.results) continue;
84
+ for (const result of attempt.results) {
85
+ if (!result.assertions) continue;
86
+ for (const assertion of result.assertions) {
87
+ if (!assertionOutcomes.has(assertion.check)) {
88
+ assertionOutcomes.set(assertion.check, { pass: 0, fail: 0 });
89
+ }
90
+ const entry = assertionOutcomes.get(assertion.check);
91
+ if (assertion.pass) {
92
+ entry.pass++;
93
+ } else {
94
+ entry.fail++;
95
+ }
96
+ }
97
+ }
98
+ }
99
+
100
+ const flakyAssertions = [];
101
+ for (const [check, outcomes] of assertionOutcomes) {
102
+ if (outcomes.pass > 0 && outcomes.fail > 0) {
103
+ flakyAssertions.push({ check, pass: outcomes.pass, fail: outcomes.fail });
104
+ }
105
+ }
106
+
107
+ if (flakyAssertions.length > 0) {
108
+ prompt += '## Flaky/Inconsistent Assertions\n\n';
109
+ prompt += 'These assertions are flaky - they pass in some sessions and fail in others. ';
110
+ prompt += 'Do not waste iterations on these; they may depend on environment rather than command quality.\n\n';
111
+ for (const flakyAssertion of flakyAssertions) {
112
+ prompt += `- "${flakyAssertion.check}" - passed ${flakyAssertion.pass}x, failed ${flakyAssertion.fail}x across sessions\n`;
113
+ }
114
+ prompt += '\n';
115
+ }
116
+ }
117
+
118
+ prompt += '## Instructions\n\n';
119
+ prompt += 'Return ONLY the rewritten command markdown content. No explanation, no code fences.\n';
120
+
121
+ return prompt;
122
+ }
123
+
124
+ // ---------------------------------------------------------------------------
125
+ // generateDiff
126
+ // ---------------------------------------------------------------------------
127
+
128
+ /**
129
+ * Simple line-by-line diff showing added (+) and removed (-) lines.
130
+ *
131
+ * @param {string} original
132
+ * @param {string} modified
133
+ * @returns {string}
134
+ */
135
+ function generateDiff(original, modified) {
136
+ const originalLines = original.split('\n');
137
+ const modifiedLines = modified.split('\n');
138
+
139
+ if (original === modified) {
140
+ return '';
141
+ }
142
+
143
+ const lines = [];
144
+ let i = 0;
145
+ let j = 0;
146
+
147
+ while (i < originalLines.length || j < modifiedLines.length) {
148
+ if (i < originalLines.length && j < modifiedLines.length && originalLines[i] === modifiedLines[j]) {
149
+ lines.push(' ' + originalLines[i]);
150
+ i++;
151
+ j++;
152
+ } else {
153
+ const originalLineInModified = j < modifiedLines.length ? modifiedLines.indexOf(originalLines[i], j) : -1;
154
+ const modifiedLineInOriginal = i < originalLines.length ? originalLines.indexOf(modifiedLines[j], i) : -1;
155
+
156
+ if (i >= originalLines.length) {
157
+ lines.push('+' + modifiedLines[j]);
158
+ j++;
159
+ } else if (j >= modifiedLines.length) {
160
+ lines.push('-' + originalLines[i]);
161
+ i++;
162
+ } else if (
163
+ originalLineInModified !== -1 &&
164
+ (modifiedLineInOriginal === -1 || originalLineInModified - j <= modifiedLineInOriginal - i)
165
+ ) {
166
+ lines.push('+' + modifiedLines[j]);
167
+ j++;
168
+ } else {
169
+ lines.push('-' + originalLines[i]);
170
+ i++;
171
+ }
172
+ }
173
+ }
174
+
175
+ return lines.join('\n');
176
+ }
177
+
178
+ /**
179
+ * Default rewrite invocation via `claude -p`.
180
+ *
181
+ * @param {string} prompt
182
+ * @param {{ timeout?: number, _execFileSync?: Function }} [options]
183
+ * @returns {Promise<string>}
184
+ */
185
+ async function defaultRewriteCommand(prompt, options = {}) {
186
+ const timeout = options.timeout || 120_000;
187
+ const execFile = options._execFileSync || execFileSync;
188
+
189
+ return execFile(
190
+ 'claude',
191
+ ['-p', prompt, '--output-format', 'text', '--no-session-persistence'],
192
+ {
193
+ encoding: 'utf-8',
194
+ timeout,
195
+ maxBuffer: 10 * 1024 * 1024,
196
+ }
197
+ );
198
+ }
199
+
200
+ // ---------------------------------------------------------------------------
201
+ // runImprovementLoop
202
+ // ---------------------------------------------------------------------------
203
+
204
+ /**
205
+ * Main improvement loop orchestrator.
206
+ *
207
+ * @param {string} commandPath - path to the command markdown file
208
+ * @param {string} evalSetPath - path to the .eval.json file
209
+ * @param {object} [options]
210
+ * @param {number} [options.maxIterations=3]
211
+ * @param {Function} [options._runEval] - injectable eval function for testing
212
+ * @param {Function} [options._rewriteCommand] - injectable rewriter for testing
213
+ * @param {string} [options._basePath] - eval-logs base path for testing
214
+ * @returns {Promise<{ original: string, best: string, originalScore: number, bestScore: number, iterations: number, reason: string, diff: string }>}
215
+ */
216
+ async function runImprovementLoop(commandPath, evalSetPath, options = {}) {
217
+ const maxIterations = options.maxIterations || 3;
218
+ const runEval = options._runEval;
219
+ const rewriteCommand = options._rewriteCommand || ((prompt) => defaultRewriteCommand(prompt, options));
220
+ const basePath = options._basePath || DEFAULT_BASE_PATH;
221
+
222
+ const originalContent = fs.readFileSync(commandPath, 'utf8');
223
+ let bestContent = originalContent;
224
+
225
+ try {
226
+ const baselineResult = await runEval(evalSetPath);
227
+ const originalScore = baselineResult.overall_score;
228
+ const history = loadEvalHistory(baselineResult.command, basePath);
229
+
230
+ let bestScore = originalScore;
231
+ let latestResult = baselineResult;
232
+ let previousScore = originalScore;
233
+ let iterations = 0;
234
+ let reason = 'max_iterations';
235
+
236
+ for (let iter = 1; iter <= maxIterations; iter++) {
237
+ iterations = iter;
238
+
239
+ const failures = analyzeFailures(latestResult);
240
+ const prompt = buildRewritePrompt(
241
+ fs.readFileSync(commandPath, 'utf8'),
242
+ failures,
243
+ history
244
+ );
245
+
246
+ const newContent = await rewriteCommand(prompt);
247
+ fs.writeFileSync(commandPath, newContent, 'utf8');
248
+
249
+ const newResult = await runEval(evalSetPath);
250
+ const newScore = newResult.overall_score;
251
+
252
+ if (newScore < bestScore) {
253
+ fs.writeFileSync(commandPath, bestContent, 'utf8');
254
+ reason = 'regression';
255
+ break;
256
+ }
257
+
258
+ if (newScore === previousScore) {
259
+ if (newScore >= bestScore) {
260
+ bestContent = newContent;
261
+ bestScore = newScore;
262
+ }
263
+ fs.writeFileSync(commandPath, bestContent, 'utf8');
264
+ reason = 'plateau';
265
+ break;
266
+ }
267
+
268
+ if (newScore > bestScore) {
269
+ bestContent = newContent;
270
+ bestScore = newScore;
271
+ }
272
+
273
+ previousScore = newScore;
274
+ latestResult = newResult;
275
+
276
+ if (iter === maxIterations) {
277
+ reason = 'max_iterations';
278
+ }
279
+ }
280
+
281
+ fs.writeFileSync(commandPath, bestContent, 'utf8');
282
+
283
+ return {
284
+ original: originalContent,
285
+ best: bestContent,
286
+ originalScore,
287
+ bestScore,
288
+ iterations,
289
+ reason,
290
+ diff: generateDiff(originalContent, bestContent),
291
+ };
292
+ } catch (err) {
293
+ try {
294
+ fs.writeFileSync(commandPath, bestContent, 'utf8');
295
+ } catch (_restoreErr) {
296
+ // Preserve the original failure if restoring also fails.
297
+ }
298
+ throw err;
299
+ }
300
+ }
301
+
302
+ // ---------------------------------------------------------------------------
303
+ // CLI entry point
304
+ // ---------------------------------------------------------------------------
305
+
306
+ /**
307
+ * Parse CLI arguments.
308
+ *
309
+ * @param {string[]} argv - process.argv.slice(2)
310
+ * @returns {{ commandPath: string, evalSetPath: string, maxIterations: number }}
311
+ */
312
+ function parseCliArgs(argv) {
313
+ let commandPath = null;
314
+ let evalSetPath = null;
315
+ let maxIterations = 3;
316
+
317
+ for (let i = 0; i < argv.length; i++) {
318
+ if (argv[i] === '--eval-set' && i + 1 < argv.length) {
319
+ evalSetPath = argv[i + 1];
320
+ i++;
321
+ } else if (argv[i] === '--max-iterations' && i + 1 < argv.length) {
322
+ maxIterations = Number(argv[i + 1]);
323
+ i++;
324
+ } else if (!argv[i].startsWith('--')) {
325
+ commandPath = argv[i];
326
+ }
327
+ }
328
+
329
+ if (!commandPath) {
330
+ throw new Error('Usage: improve-command <command-path> --eval-set <path> [--max-iterations <N>]');
331
+ }
332
+ if (!evalSetPath) {
333
+ throw new Error('Missing required --eval-set <path>');
334
+ }
335
+
336
+ return { commandPath, evalSetPath, maxIterations };
337
+ }
338
+
339
+ if (require.main === module) {
340
+ const args = parseCliArgs(process.argv.slice(2));
341
+ const { runEvalPipeline } = require('./run-command-eval');
342
+
343
+ runImprovementLoop(args.commandPath, args.evalSetPath, {
344
+ maxIterations: args.maxIterations,
345
+ _runEval: (evalPath) => runEvalPipeline(evalPath),
346
+ })
347
+ .then((result) => {
348
+ console.log('\n=== Improvement Summary ===');
349
+ console.log(`Original score: ${result.originalScore.toFixed(2)}`);
350
+ console.log(`Best score: ${result.bestScore.toFixed(2)}`);
351
+ console.log(`Iterations: ${result.iterations}`);
352
+ console.log(`Reason: ${result.reason}`);
353
+
354
+ if (result.diff) {
355
+ console.log('\n=== Diff (original -> best) ===');
356
+ console.log(result.diff);
357
+ } else {
358
+ console.log('\nNo changes made.');
359
+ }
360
+
361
+ console.log('\nDiff shown above. Review and decide whether to apply.');
362
+ })
363
+ .catch((err) => {
364
+ console.error(`Error: ${err.message}`);
365
+ process.exit(1);
366
+ });
367
+ }
368
+
369
+ module.exports = {
370
+ analyzeFailures,
371
+ buildRewritePrompt,
372
+ defaultRewriteCommand,
373
+ generateDiff,
374
+ runImprovementLoop,
375
+ };
@@ -0,0 +1,229 @@
1
+ /**
2
+ * Eval runner core — worktree isolation + command execution.
3
+ *
4
+ * Provides building blocks for the eval pipeline:
5
+ * - createEvalWorktree() — spin up an isolated worktree
6
+ * - destroyEvalWorktree() — tear it down (force, even if dirty)
7
+ * - resetWorktree() — reset between eval queries
8
+ * - executeCommand() — run a claude CLI command in a worktree
9
+ */
10
+
11
+ const path = require('path');
12
+ const { execSync } = require('node:child_process');
13
+
14
+ // ── helpers ──────────────────────────────────────────────────────────
15
+
16
+ /**
17
+ * Detect the repo root by walking up from cwd.
18
+ * Works from both the main repo and from within worktrees.
19
+ */
20
+ function getRepoRoot() {
21
+ const root = execSync('git rev-parse --show-toplevel', {
22
+ encoding: 'utf-8',
23
+ stdio: ['pipe', 'pipe', 'pipe'],
24
+ }).trim();
25
+ return root;
26
+ }
27
+
28
+ /**
29
+ * Get the .worktrees directory path for eval worktrees.
30
+ * Eval worktrees live under <repo-root>/.worktrees/
31
+ */
32
+ function getWorktreesDir() {
33
+ const root = getRepoRoot();
34
+ return path.join(root, '.worktrees');
35
+ }
36
+
37
+ // ── createEvalWorktree ───────────────────────────────────────────────
38
+
39
+ /**
40
+ * Create a git worktree with a unique name for eval isolation.
41
+ *
42
+ * @returns {Promise<{ path: string, branch: string }>}
43
+ */
44
+ async function createEvalWorktree() {
45
+ const timestamp = Date.now();
46
+ const pid = process.pid;
47
+ const name = `eval-${timestamp}-${pid}`;
48
+ const branch = `eval-${timestamp}-${pid}`;
49
+ const worktreesDir = getWorktreesDir();
50
+ const wtPath = path.join(worktreesDir, name);
51
+
52
+ // Create the worktree with a detached HEAD first, then create branch
53
+ execSync(`git worktree add -b "${branch}" "${wtPath}" HEAD`, {
54
+ cwd: getRepoRoot(),
55
+ encoding: 'utf-8',
56
+ stdio: ['pipe', 'pipe', 'pipe'],
57
+ });
58
+
59
+ return { path: wtPath, branch };
60
+ }
61
+
62
+ // ── destroyEvalWorktree ──────────────────────────────────────────────
63
+
64
+ /**
65
+ * Remove a worktree and its temporary branch.
66
+ * Succeeds even if the worktree is dirty.
67
+ *
68
+ * @param {string} worktreePath — absolute path to the worktree
69
+ * @returns {Promise<void>}
70
+ */
71
+ async function destroyEvalWorktree(worktreePath) {
72
+ const repoRoot = getRepoRoot();
73
+
74
+ // Query actual branch for this worktree (more reliable than inferring from dir name)
75
+ let branch;
76
+ try {
77
+ branch = execSync('git branch --show-current', {
78
+ cwd: worktreePath,
79
+ encoding: 'utf-8',
80
+ stdio: ['pipe', 'pipe', 'pipe'],
81
+ }).trim();
82
+ } catch (_err) {
83
+ // Worktree may be corrupted — fall back to directory name
84
+ branch = path.basename(worktreePath);
85
+ }
86
+
87
+ // Remove the worktree (--force handles dirty state)
88
+ execSync(`git worktree remove --force "${worktreePath}"`, {
89
+ cwd: repoRoot,
90
+ encoding: 'utf-8',
91
+ stdio: ['pipe', 'pipe', 'pipe'],
92
+ });
93
+
94
+ // Prune to clean up references
95
+ execSync('git worktree prune', {
96
+ cwd: repoRoot,
97
+ encoding: 'utf-8',
98
+ stdio: ['pipe', 'pipe', 'pipe'],
99
+ });
100
+
101
+ // Delete the temporary branch (force in case it's not fully merged)
102
+ if (branch && branch.startsWith('eval-')) {
103
+ try {
104
+ execSync(`git branch -D "${branch}"`, {
105
+ cwd: repoRoot,
106
+ encoding: 'utf-8',
107
+ stdio: ['pipe', 'pipe', 'pipe'],
108
+ });
109
+ } catch (_err) {
110
+ // Branch may already be gone — ignore
111
+ }
112
+ }
113
+ }
114
+
115
+ // ── resetWorktree ────────────────────────────────────────────────────
116
+
117
+ /**
118
+ * Reset a worktree to a clean state (tracked files restored, untracked removed).
119
+ *
120
+ * @param {string} worktreePath — absolute path to the worktree
121
+ * @returns {Promise<void>}
122
+ */
123
+ async function resetWorktree(worktreePath) {
124
+ // Restore tracked files
125
+ execSync('git checkout -- .', {
126
+ cwd: worktreePath,
127
+ encoding: 'utf-8',
128
+ stdio: ['pipe', 'pipe', 'pipe'],
129
+ });
130
+
131
+ // Remove untracked files, directories, and ignored files (full reset between runs)
132
+ execSync('git clean -fdx', {
133
+ cwd: worktreePath,
134
+ encoding: 'utf-8',
135
+ stdio: ['pipe', 'pipe', 'pipe'],
136
+ });
137
+ }
138
+
139
+ // ── executeCommand ───────────────────────────────────────────────────
140
+
141
+ /**
142
+ * Execute a command in an eval worktree.
143
+ *
144
+ * In production, runs `claude -p "<prompt>" --output-format stream-json --verbose --no-session-persistence`.
145
+ * Accepts an optional `cmdOverride` array for testing (avoids invoking real LLM).
146
+ *
147
+ * @param {string} _command — label only (e.g., "/status")
148
+ * @param {string} prompt — the prompt to send
149
+ * @param {string} worktreePath — absolute path to the worktree (used as cwd)
150
+ * @param {number} [timeout=120000] — timeout in milliseconds
151
+ * @param {string[]} [cmdOverride] — optional command array for testing
152
+ * @returns {Promise<{ stdout: string, stderr: string, exitCode: number, timedOut: boolean }>}
153
+ */
154
+ async function executeCommand(_command, prompt, worktreePath, timeout = 120000, cmdOverride) {
155
+ // Build the command to run
156
+ const cmd = cmdOverride || [
157
+ 'claude',
158
+ '-p',
159
+ prompt,
160
+ '--output-format',
161
+ 'stream-json',
162
+ '--verbose',
163
+ '--no-session-persistence',
164
+ ];
165
+
166
+ // Build environment: inherit current env, strip CLAUDECODE, set FORGE_EVAL
167
+ const env = { ...process.env };
168
+ delete env.CLAUDECODE;
169
+ env.FORGE_EVAL = '1';
170
+
171
+ // Spawn with Bun.spawn (array form, no shell interpolation)
172
+ const proc = Bun.spawn(cmd, { // eslint-disable-line no-undef -- Bun global provided by runtime
173
+ cwd: worktreePath,
174
+ env,
175
+ stdout: 'pipe',
176
+ stderr: 'pipe',
177
+ });
178
+
179
+ let timedOut = false;
180
+ let timeoutId;
181
+
182
+ // Set up timeout
183
+ const timeoutPromise = new Promise((resolve) => {
184
+ timeoutId = setTimeout(() => {
185
+ timedOut = true;
186
+ proc.kill();
187
+ resolve();
188
+ }, timeout);
189
+ });
190
+
191
+ // Wait for process to exit (or timeout)
192
+ const exitPromise = proc.exited.then(() => {
193
+ clearTimeout(timeoutId);
194
+ });
195
+
196
+ await Promise.race([exitPromise, timeoutPromise]);
197
+
198
+ // Read stdout and stderr
199
+ let stdout = '';
200
+ let stderr = '';
201
+
202
+ try {
203
+ stdout = await new Response(proc.stdout).text();
204
+ } catch (_err) {
205
+ // stream may be closed on kill — ignore
206
+ }
207
+
208
+ try {
209
+ stderr = await new Response(proc.stderr).text();
210
+ } catch (_err) {
211
+ // stream may be closed on kill — ignore
212
+ }
213
+
214
+ const exitCode = timedOut ? (proc.exitCode ?? 1) : proc.exitCode;
215
+
216
+ return {
217
+ stdout,
218
+ stderr,
219
+ exitCode,
220
+ timedOut,
221
+ };
222
+ }
223
+
224
+ module.exports = {
225
+ createEvalWorktree,
226
+ destroyEvalWorktree,
227
+ resetWorktree,
228
+ executeCommand,
229
+ };