forge-workflow 0.0.4 → 0.0.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/commands/dev.md +345 -340
- package/.claude/commands/plan.md +566 -521
- package/.claude/commands/premerge.md +186 -176
- package/.claude/commands/research.md +42 -42
- package/.claude/commands/review.md +448 -442
- package/.claude/commands/rollback.md +721 -721
- package/.claude/commands/ship.md +212 -164
- package/.claude/commands/sonarcloud.md +152 -152
- package/.claude/commands/status.md +90 -48
- package/.claude/commands/validate.md +288 -282
- package/.claude/commands/verify.md +269 -221
- package/.claude/rules/greptile-review-process.md +285 -285
- package/.claude/rules/workflow.md +121 -105
- package/.claude/scripts/greptile-resolve.sh +558 -526
- package/.claude/scripts/load-env.sh +32 -32
- package/.cline/workflows/dev.md +342 -337
- package/.cline/workflows/plan.md +563 -518
- package/.cline/workflows/premerge.md +183 -173
- package/.cline/workflows/research.md +39 -39
- package/.cline/workflows/review.md +445 -439
- package/.cline/workflows/rollback.md +718 -718
- package/.cline/workflows/ship.md +209 -161
- package/.cline/workflows/sonarcloud.md +146 -146
- package/.cline/workflows/status.md +87 -45
- package/.cline/workflows/validate.md +285 -279
- package/.cline/workflows/verify.md +266 -218
- package/.codex/config.toml +11 -11
- package/.codex/skills/dev/SKILL.md +345 -340
- package/.codex/skills/plan/SKILL.md +566 -521
- package/.codex/skills/premerge/SKILL.md +186 -176
- package/.codex/skills/research/SKILL.md +42 -42
- package/.codex/skills/review/SKILL.md +448 -442
- package/.codex/skills/rollback/SKILL.md +721 -721
- package/.codex/skills/ship/SKILL.md +212 -164
- package/.codex/skills/sonarcloud/SKILL.md +149 -149
- package/.codex/skills/status/SKILL.md +90 -48
- package/.codex/skills/validate/SKILL.md +288 -282
- package/.codex/skills/verify/SKILL.md +269 -221
- package/.cursor/commands/dev.md +342 -337
- package/.cursor/commands/plan.md +563 -518
- package/.cursor/commands/premerge.md +183 -173
- package/.cursor/commands/research.md +39 -39
- package/.cursor/commands/review.md +445 -439
- package/.cursor/commands/rollback.md +718 -718
- package/.cursor/commands/ship.md +209 -161
- package/.cursor/commands/sonarcloud.md +146 -146
- package/.cursor/commands/status.md +87 -45
- package/.cursor/commands/validate.md +285 -279
- package/.cursor/commands/verify.md +266 -218
- package/.cursor/rules/permissions-guidance.mdc +37 -37
- package/.forge/hooks/check-tdd.js +240 -240
- package/.github/PLUGIN_TEMPLATE.json +32 -32
- package/.github/prompts/dev.prompt.md +347 -342
- package/.github/prompts/plan.prompt.md +568 -523
- package/.github/prompts/premerge.prompt.md +188 -178
- package/.github/prompts/research.prompt.md +44 -44
- package/.github/prompts/review.prompt.md +450 -444
- package/.github/prompts/rollback.prompt.md +723 -723
- package/.github/prompts/ship.prompt.md +214 -166
- package/.github/prompts/sonarcloud.prompt.md +151 -151
- package/.github/prompts/status.prompt.md +92 -50
- package/.github/prompts/validate.prompt.md +290 -284
- package/.github/prompts/verify.prompt.md +271 -223
- package/.github/workflows/beads-to-github.yml +56 -0
- package/.github/workflows/github-to-beads.yml +97 -0
- package/.kilocode/workflows/dev.md +346 -341
- package/.kilocode/workflows/plan.md +567 -522
- package/.kilocode/workflows/premerge.md +187 -177
- package/.kilocode/workflows/research.md +43 -43
- package/.kilocode/workflows/review.md +449 -443
- package/.kilocode/workflows/rollback.md +722 -722
- package/.kilocode/workflows/ship.md +213 -165
- package/.kilocode/workflows/sonarcloud.md +150 -150
- package/.kilocode/workflows/status.md +91 -49
- package/.kilocode/workflows/validate.md +289 -283
- package/.kilocode/workflows/verify.md +270 -222
- package/.mcp.json.example +12 -12
- package/.opencode/commands/dev.md +345 -340
- package/.opencode/commands/plan.md +566 -521
- package/.opencode/commands/premerge.md +186 -176
- package/.opencode/commands/research.md +42 -42
- package/.opencode/commands/review.md +448 -442
- package/.opencode/commands/rollback.md +721 -721
- package/.opencode/commands/ship.md +212 -164
- package/.opencode/commands/sonarcloud.md +149 -149
- package/.opencode/commands/status.md +90 -48
- package/.opencode/commands/validate.md +288 -282
- package/.opencode/commands/verify.md +269 -221
- package/.roo/commands/dev.md +346 -341
- package/.roo/commands/plan.md +567 -522
- package/.roo/commands/premerge.md +187 -177
- package/.roo/commands/research.md +43 -43
- package/.roo/commands/review.md +449 -443
- package/.roo/commands/rollback.md +722 -722
- package/.roo/commands/ship.md +213 -165
- package/.roo/commands/sonarcloud.md +150 -150
- package/.roo/commands/status.md +91 -49
- package/.roo/commands/validate.md +289 -283
- package/.roo/commands/verify.md +270 -222
- package/AGENTS.md +272 -175
- package/CLAUDE.md +110 -100
- package/README.md +429 -416
- package/bin/forge-cmd.js +317 -313
- package/bin/forge-preflight.js +322 -309
- package/bin/forge.js +4765 -4303
- package/docs/AGENT_INSTALL_PROMPT.md +342 -342
- package/docs/BEADS_GITHUB_SYNC.md +251 -251
- package/docs/ENHANCED_ONBOARDING.md +612 -602
- package/docs/EXAMPLES.md +482 -482
- package/docs/GREPTILE_SETUP.md +400 -400
- package/docs/MANUAL_REVIEW_GUIDE.md +106 -106
- package/docs/ROADMAP.md +359 -359
- package/docs/SETUP.md +663 -631
- package/docs/TOOLCHAIN.md +653 -630
- package/docs/VALIDATION.md +363 -363
- package/install.sh +40 -1056
- package/lefthook.yml +50 -39
- package/lib/agents/README.md +198 -198
- package/lib/agents/claude.plugin.json +28 -28
- package/lib/agents/cline.plugin.json +22 -22
- package/lib/agents/codex.plugin.json +19 -19
- package/lib/agents/copilot.plugin.json +24 -24
- package/lib/agents/cursor.plugin.json +25 -25
- package/lib/agents/kilocode.plugin.json +22 -22
- package/lib/agents/opencode.plugin.json +20 -20
- package/lib/agents/roo.plugin.json +23 -23
- package/lib/agents-config.js +2112 -2112
- package/lib/beads-health-check.js +143 -0
- package/lib/beads-setup.js +341 -0
- package/lib/beads-sync-scaffold.js +260 -0
- package/lib/commands/_registry.js +134 -0
- package/lib/commands/clean.js +181 -0
- package/lib/commands/dev.js +571 -513
- package/lib/commands/plan.js +692 -692
- package/lib/commands/push.js +196 -0
- package/lib/commands/recommend.js +119 -119
- package/lib/commands/ship.js +377 -377
- package/lib/commands/status.js +378 -378
- package/lib/commands/sync.js +55 -0
- package/lib/commands/team.js +37 -0
- package/lib/commands/test.js +207 -0
- package/lib/commands/validate.js +602 -602
- package/lib/commands/worktree.js +310 -0
- package/lib/context-merge.js +359 -359
- package/lib/dep-guard/analyzer.js +294 -294
- package/lib/dep-guard/behavior-detector.js +98 -98
- package/lib/dep-guard/contract-detector.js +162 -162
- package/lib/dep-guard/import-detector.js +498 -498
- package/lib/dep-guard/path-utils.js +13 -13
- package/lib/dep-guard/rubric.js +120 -120
- package/lib/dep-guard/task-parser.js +318 -318
- package/lib/detect-agent.js +191 -191
- package/lib/detect-worktree.js +47 -47
- package/lib/docs-command.js +51 -0
- package/lib/docs-copy.js +50 -0
- package/lib/file-hash.js +26 -26
- package/lib/freshness-token.js +148 -0
- package/lib/greptile-match.js +80 -0
- package/lib/husky-migration.js +450 -0
- package/lib/lefthook-check.js +65 -0
- package/lib/pat-setup.js +207 -0
- package/lib/plugin-catalog.js +350 -350
- package/lib/plugin-manager.js +166 -166
- package/lib/plugin-recommender.js +141 -141
- package/lib/project-discovery.js +491 -491
- package/lib/reset.js +309 -0
- package/lib/setup-action-log.js +139 -139
- package/lib/setup-summary-renderer.js +106 -106
- package/lib/setup-utils.js +96 -0
- package/lib/setup.js +192 -192
- package/lib/smart-merge.js +64 -0
- package/lib/symlink-utils.js +81 -0
- package/lib/task-ownership.js +117 -0
- package/lib/workflow-profiles.js +197 -197
- package/package.json +131 -128
- package/scripts/beads-context.sh +426 -0
- package/scripts/beads-context.test.js +567 -0
- package/scripts/behavioral-judge.sh +378 -0
- package/scripts/benchmark.js +85 -0
- package/scripts/branch-protection.js +183 -0
- package/scripts/check-agents.js +172 -0
- package/scripts/check-forge-token.js +98 -0
- package/scripts/commitlint.js +42 -0
- package/scripts/conflict-detect.sh +323 -0
- package/scripts/dep-guard-analyze.js +71 -0
- package/scripts/dep-guard.sh +789 -0
- package/scripts/eval_win.py +249 -0
- package/scripts/file-index.sh +493 -0
- package/scripts/forge-team/index.sh +86 -0
- package/scripts/forge-team/lib/agent-prompt.sh +52 -0
- package/scripts/forge-team/lib/claim.sh +256 -0
- package/scripts/forge-team/lib/dashboard.sh +341 -0
- package/scripts/forge-team/lib/epic.sh +332 -0
- package/scripts/forge-team/lib/hooks.sh +253 -0
- package/scripts/forge-team/lib/identity.sh +235 -0
- package/scripts/forge-team/lib/sync-github.sh +317 -0
- package/scripts/forge-team/lib/verify.sh +284 -0
- package/scripts/forge-team/lib/workload.sh +296 -0
- package/scripts/forge-team/tests/agent-prompt.test.sh +72 -0
- package/scripts/forge-team/tests/claim.test.sh +179 -0
- package/scripts/forge-team/tests/dashboard.test.sh +170 -0
- package/scripts/forge-team/tests/dispatcher.test.sh +79 -0
- package/scripts/forge-team/tests/epic.test.sh +176 -0
- package/scripts/forge-team/tests/hooks.test.sh +239 -0
- package/scripts/forge-team/tests/identity.test.sh +176 -0
- package/scripts/forge-team/tests/integration.test.sh +371 -0
- package/scripts/forge-team/tests/sync-github.test.sh +209 -0
- package/scripts/forge-team/tests/verify.test.sh +314 -0
- package/scripts/forge-team/tests/workflow-integration.test.sh +43 -0
- package/scripts/forge-team/tests/workload.test.sh +209 -0
- package/scripts/github-beads-sync/comment.mjs +64 -0
- package/scripts/github-beads-sync/config.mjs +148 -0
- package/scripts/github-beads-sync/github-api.mjs +131 -0
- package/scripts/github-beads-sync/index.mjs +332 -0
- package/scripts/github-beads-sync/label-mapper.mjs +54 -0
- package/scripts/github-beads-sync/mapping.mjs +78 -0
- package/scripts/github-beads-sync/reverse-sync-cli.mjs +31 -0
- package/scripts/github-beads-sync/reverse-sync.mjs +138 -0
- package/scripts/github-beads-sync/run-bd.mjs +159 -0
- package/scripts/github-beads-sync/sanitize.mjs +121 -0
- package/scripts/github-beads-sync.config.json +26 -0
- package/scripts/improve-command.js +375 -0
- package/scripts/lib/eval-runner.js +268 -0
- package/scripts/lib/eval-schema.js +135 -0
- package/scripts/lib/eval-storage.js +78 -0
- package/scripts/lib/grading.js +203 -0
- package/scripts/lib/jsonl-lock.sh +48 -0
- package/scripts/lib/sanitize.sh +116 -0
- package/scripts/lib/transcript-parser.js +63 -0
- package/scripts/lint.js +47 -0
- package/scripts/migrate-to-bun-test.js +412 -0
- package/scripts/pr-coordinator.sh +706 -0
- package/scripts/run-command-eval.js +236 -0
- package/scripts/smart-status.sh +809 -0
- package/scripts/sync-commands.js +571 -0
- package/scripts/sync-utils.sh +455 -0
- package/scripts/test-dashboard.js +123 -0
- package/scripts/test.js +46 -0
- package/scripts/validate.sh +94 -0
- package/skills/parallel-deep-research/SKILL.md +108 -108
- package/skills/parallel-deep-research/evals/README.md +27 -27
- package/skills/parallel-deep-research/evals/evals.json +62 -62
- package/skills/sonarcloud-analysis/SKILL.md +171 -171
- package/skills/sonarcloud-analysis/evals/README.md +27 -27
- package/skills/sonarcloud-analysis/evals/evals.json +50 -50
- package/skills/sonarcloud-analysis/references/api-reference.md +466 -466
- package/.cursor/hooks/state/continual-learning-index.json +0 -19
- package/.cursor/hooks/state/continual-learning.json +0 -8
|
@@ -0,0 +1,375 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Semi-autonomous improvement loop: analyze eval failures, rewrite command,
|
|
3
|
+
* re-evaluate, and stop on regression or plateau.
|
|
4
|
+
*
|
|
5
|
+
* Usage:
|
|
6
|
+
* bun scripts/improve-command.js <command-path> --eval-set <path> [--max-iterations <N>]
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
const { execFileSync } = require('child_process');
|
|
10
|
+
const fs = require('fs');
|
|
11
|
+
const { DEFAULT_BASE_PATH, loadEvalHistory } = require('./lib/eval-storage');
|
|
12
|
+
|
|
13
|
+
// ---------------------------------------------------------------------------
|
|
14
|
+
// analyzeFailures
|
|
15
|
+
// ---------------------------------------------------------------------------
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Extract failing assertions with their query context from an eval result.
|
|
19
|
+
*
|
|
20
|
+
* @param {object} evalResult - result from runEvalPipeline
|
|
21
|
+
* @returns {Array<{ query: string, assertion: string, reasoning: string }>}
|
|
22
|
+
*/
|
|
23
|
+
function analyzeFailures(evalResult) {
|
|
24
|
+
const failures = [];
|
|
25
|
+
|
|
26
|
+
for (const queryResult of evalResult.results) {
|
|
27
|
+
for (const assertion of queryResult.assertions) {
|
|
28
|
+
if (!assertion.pass) {
|
|
29
|
+
failures.push({
|
|
30
|
+
query: queryResult.prompt,
|
|
31
|
+
assertion: assertion.check,
|
|
32
|
+
reasoning: assertion.reasoning,
|
|
33
|
+
});
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
return failures;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
// ---------------------------------------------------------------------------
|
|
42
|
+
// buildRewritePrompt
|
|
43
|
+
// ---------------------------------------------------------------------------
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* Build a prompt asking for a command rewrite that fixes the identified failures.
|
|
47
|
+
*
|
|
48
|
+
* @param {string} commandContent - current command markdown content
|
|
49
|
+
* @param {Array<{ query: string, assertion: string, reasoning: string }>} failures
|
|
50
|
+
* @param {object[]} history - array of prior eval results (from loadEvalHistory)
|
|
51
|
+
* @returns {string}
|
|
52
|
+
*/
|
|
53
|
+
function buildRewritePrompt(commandContent, failures, history) {
|
|
54
|
+
let prompt = '';
|
|
55
|
+
|
|
56
|
+
prompt += 'You are a command prompt engineer. Rewrite the following command to fix the failing assertions.\n\n';
|
|
57
|
+
|
|
58
|
+
prompt += '## Current Command\n\n';
|
|
59
|
+
prompt += commandContent + '\n\n';
|
|
60
|
+
|
|
61
|
+
prompt += '## Failing Assertions\n\n';
|
|
62
|
+
for (const failure of failures) {
|
|
63
|
+
prompt += `- Query: "${failure.query}"\n`;
|
|
64
|
+
prompt += ` Assertion: "${failure.assertion}"\n`;
|
|
65
|
+
prompt += ` Reasoning: "${failure.reasoning}"\n\n`;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
if (history && history.length > 0) {
|
|
69
|
+
prompt += '## Prior Eval Attempts\n\n';
|
|
70
|
+
prompt += 'Avoid repeating approaches that have already been tried. Here is a summary of prior attempts:\n\n';
|
|
71
|
+
for (let i = 0; i < history.length; i++) {
|
|
72
|
+
const attempt = history[i];
|
|
73
|
+
const failCount = attempt.results
|
|
74
|
+
? attempt.results.filter((result) => result.assertions && result.assertions.some((assertion) => !assertion.pass)).length
|
|
75
|
+
: 0;
|
|
76
|
+
prompt += `- Attempt ${i + 1}: score ${attempt.overall_score}, ${failCount} failing queries\n`;
|
|
77
|
+
}
|
|
78
|
+
prompt += '\n';
|
|
79
|
+
|
|
80
|
+
// Detect flaky assertions: the same check both passed and failed across sessions.
|
|
81
|
+
const assertionOutcomes = new Map();
|
|
82
|
+
for (const attempt of history) {
|
|
83
|
+
if (!attempt.results) continue;
|
|
84
|
+
for (const result of attempt.results) {
|
|
85
|
+
if (!result.assertions) continue;
|
|
86
|
+
for (const assertion of result.assertions) {
|
|
87
|
+
if (!assertionOutcomes.has(assertion.check)) {
|
|
88
|
+
assertionOutcomes.set(assertion.check, { pass: 0, fail: 0 });
|
|
89
|
+
}
|
|
90
|
+
const entry = assertionOutcomes.get(assertion.check);
|
|
91
|
+
if (assertion.pass) {
|
|
92
|
+
entry.pass++;
|
|
93
|
+
} else {
|
|
94
|
+
entry.fail++;
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
const flakyAssertions = [];
|
|
101
|
+
for (const [check, outcomes] of assertionOutcomes) {
|
|
102
|
+
if (outcomes.pass > 0 && outcomes.fail > 0) {
|
|
103
|
+
flakyAssertions.push({ check, pass: outcomes.pass, fail: outcomes.fail });
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
if (flakyAssertions.length > 0) {
|
|
108
|
+
prompt += '## Flaky/Inconsistent Assertions\n\n';
|
|
109
|
+
prompt += 'These assertions are flaky - they pass in some sessions and fail in others. ';
|
|
110
|
+
prompt += 'Do not waste iterations on these; they may depend on environment rather than command quality.\n\n';
|
|
111
|
+
for (const flakyAssertion of flakyAssertions) {
|
|
112
|
+
prompt += `- "${flakyAssertion.check}" - passed ${flakyAssertion.pass}x, failed ${flakyAssertion.fail}x across sessions\n`;
|
|
113
|
+
}
|
|
114
|
+
prompt += '\n';
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
prompt += '## Instructions\n\n';
|
|
119
|
+
prompt += 'Return ONLY the rewritten command markdown content. No explanation, no code fences.\n';
|
|
120
|
+
|
|
121
|
+
return prompt;
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
// ---------------------------------------------------------------------------
|
|
125
|
+
// generateDiff
|
|
126
|
+
// ---------------------------------------------------------------------------
|
|
127
|
+
|
|
128
|
+
/**
|
|
129
|
+
* Simple line-by-line diff showing added (+) and removed (-) lines.
|
|
130
|
+
*
|
|
131
|
+
* @param {string} original
|
|
132
|
+
* @param {string} modified
|
|
133
|
+
* @returns {string}
|
|
134
|
+
*/
|
|
135
|
+
function generateDiff(original, modified) {
|
|
136
|
+
const originalLines = original.split('\n');
|
|
137
|
+
const modifiedLines = modified.split('\n');
|
|
138
|
+
|
|
139
|
+
if (original === modified) {
|
|
140
|
+
return '';
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
const lines = [];
|
|
144
|
+
let i = 0;
|
|
145
|
+
let j = 0;
|
|
146
|
+
|
|
147
|
+
while (i < originalLines.length || j < modifiedLines.length) {
|
|
148
|
+
if (i < originalLines.length && j < modifiedLines.length && originalLines[i] === modifiedLines[j]) {
|
|
149
|
+
lines.push(' ' + originalLines[i]);
|
|
150
|
+
i++;
|
|
151
|
+
j++;
|
|
152
|
+
} else {
|
|
153
|
+
const originalLineInModified = j < modifiedLines.length ? modifiedLines.indexOf(originalLines[i], j) : -1;
|
|
154
|
+
const modifiedLineInOriginal = i < originalLines.length ? originalLines.indexOf(modifiedLines[j], i) : -1;
|
|
155
|
+
|
|
156
|
+
if (i >= originalLines.length) {
|
|
157
|
+
lines.push('+' + modifiedLines[j]);
|
|
158
|
+
j++;
|
|
159
|
+
} else if (j >= modifiedLines.length) {
|
|
160
|
+
lines.push('-' + originalLines[i]);
|
|
161
|
+
i++;
|
|
162
|
+
} else if (
|
|
163
|
+
originalLineInModified !== -1 &&
|
|
164
|
+
(modifiedLineInOriginal === -1 || originalLineInModified - j <= modifiedLineInOriginal - i)
|
|
165
|
+
) {
|
|
166
|
+
lines.push('+' + modifiedLines[j]);
|
|
167
|
+
j++;
|
|
168
|
+
} else {
|
|
169
|
+
lines.push('-' + originalLines[i]);
|
|
170
|
+
i++;
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
return lines.join('\n');
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
/**
|
|
179
|
+
* Default rewrite invocation via `claude -p`.
|
|
180
|
+
*
|
|
181
|
+
* @param {string} prompt
|
|
182
|
+
* @param {{ timeout?: number, _execFileSync?: Function }} [options]
|
|
183
|
+
* @returns {Promise<string>}
|
|
184
|
+
*/
|
|
185
|
+
async function defaultRewriteCommand(prompt, options = {}) {
|
|
186
|
+
const timeout = options.timeout || 120_000;
|
|
187
|
+
const execFile = options._execFileSync || execFileSync;
|
|
188
|
+
|
|
189
|
+
return execFile(
|
|
190
|
+
'claude',
|
|
191
|
+
['-p', prompt, '--output-format', 'text', '--no-session-persistence'],
|
|
192
|
+
{
|
|
193
|
+
encoding: 'utf-8',
|
|
194
|
+
timeout,
|
|
195
|
+
maxBuffer: 10 * 1024 * 1024,
|
|
196
|
+
}
|
|
197
|
+
);
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
// ---------------------------------------------------------------------------
|
|
201
|
+
// runImprovementLoop
|
|
202
|
+
// ---------------------------------------------------------------------------
|
|
203
|
+
|
|
204
|
+
/**
|
|
205
|
+
* Main improvement loop orchestrator.
|
|
206
|
+
*
|
|
207
|
+
* @param {string} commandPath - path to the command markdown file
|
|
208
|
+
* @param {string} evalSetPath - path to the .eval.json file
|
|
209
|
+
* @param {object} [options]
|
|
210
|
+
* @param {number} [options.maxIterations=3]
|
|
211
|
+
* @param {Function} [options._runEval] - injectable eval function for testing
|
|
212
|
+
* @param {Function} [options._rewriteCommand] - injectable rewriter for testing
|
|
213
|
+
* @param {string} [options._basePath] - eval-logs base path for testing
|
|
214
|
+
* @returns {Promise<{ original: string, best: string, originalScore: number, bestScore: number, iterations: number, reason: string, diff: string }>}
|
|
215
|
+
*/
|
|
216
|
+
async function runImprovementLoop(commandPath, evalSetPath, options = {}) {
|
|
217
|
+
const maxIterations = options.maxIterations || 3;
|
|
218
|
+
const runEval = options._runEval;
|
|
219
|
+
const rewriteCommand = options._rewriteCommand || ((prompt) => defaultRewriteCommand(prompt, options));
|
|
220
|
+
const basePath = options._basePath || DEFAULT_BASE_PATH;
|
|
221
|
+
|
|
222
|
+
const originalContent = fs.readFileSync(commandPath, 'utf8');
|
|
223
|
+
let bestContent = originalContent;
|
|
224
|
+
|
|
225
|
+
try {
|
|
226
|
+
const baselineResult = await runEval(evalSetPath);
|
|
227
|
+
const originalScore = baselineResult.overall_score;
|
|
228
|
+
const history = loadEvalHistory(baselineResult.command, basePath);
|
|
229
|
+
|
|
230
|
+
let bestScore = originalScore;
|
|
231
|
+
let latestResult = baselineResult;
|
|
232
|
+
let previousScore = originalScore;
|
|
233
|
+
let iterations = 0;
|
|
234
|
+
let reason = 'max_iterations';
|
|
235
|
+
|
|
236
|
+
for (let iter = 1; iter <= maxIterations; iter++) {
|
|
237
|
+
iterations = iter;
|
|
238
|
+
|
|
239
|
+
const failures = analyzeFailures(latestResult);
|
|
240
|
+
const prompt = buildRewritePrompt(
|
|
241
|
+
fs.readFileSync(commandPath, 'utf8'),
|
|
242
|
+
failures,
|
|
243
|
+
history
|
|
244
|
+
);
|
|
245
|
+
|
|
246
|
+
const newContent = await rewriteCommand(prompt);
|
|
247
|
+
fs.writeFileSync(commandPath, newContent, 'utf8');
|
|
248
|
+
|
|
249
|
+
const newResult = await runEval(evalSetPath);
|
|
250
|
+
const newScore = newResult.overall_score;
|
|
251
|
+
|
|
252
|
+
if (newScore < bestScore) {
|
|
253
|
+
fs.writeFileSync(commandPath, bestContent, 'utf8');
|
|
254
|
+
reason = 'regression';
|
|
255
|
+
break;
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
if (newScore === previousScore) {
|
|
259
|
+
if (newScore >= bestScore) {
|
|
260
|
+
bestContent = newContent;
|
|
261
|
+
bestScore = newScore;
|
|
262
|
+
}
|
|
263
|
+
fs.writeFileSync(commandPath, bestContent, 'utf8');
|
|
264
|
+
reason = 'plateau';
|
|
265
|
+
break;
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
if (newScore > bestScore) {
|
|
269
|
+
bestContent = newContent;
|
|
270
|
+
bestScore = newScore;
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
previousScore = newScore;
|
|
274
|
+
latestResult = newResult;
|
|
275
|
+
|
|
276
|
+
if (iter === maxIterations) {
|
|
277
|
+
reason = 'max_iterations';
|
|
278
|
+
}
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
fs.writeFileSync(commandPath, bestContent, 'utf8');
|
|
282
|
+
|
|
283
|
+
return {
|
|
284
|
+
original: originalContent,
|
|
285
|
+
best: bestContent,
|
|
286
|
+
originalScore,
|
|
287
|
+
bestScore,
|
|
288
|
+
iterations,
|
|
289
|
+
reason,
|
|
290
|
+
diff: generateDiff(originalContent, bestContent),
|
|
291
|
+
};
|
|
292
|
+
} catch (err) {
|
|
293
|
+
try {
|
|
294
|
+
fs.writeFileSync(commandPath, bestContent, 'utf8');
|
|
295
|
+
} catch (_restoreErr) {
|
|
296
|
+
// Preserve the original failure if restoring also fails.
|
|
297
|
+
}
|
|
298
|
+
throw err;
|
|
299
|
+
}
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
// ---------------------------------------------------------------------------
|
|
303
|
+
// CLI entry point
|
|
304
|
+
// ---------------------------------------------------------------------------
|
|
305
|
+
|
|
306
|
+
/**
|
|
307
|
+
* Parse CLI arguments.
|
|
308
|
+
*
|
|
309
|
+
* @param {string[]} argv - process.argv.slice(2)
|
|
310
|
+
* @returns {{ commandPath: string, evalSetPath: string, maxIterations: number }}
|
|
311
|
+
*/
|
|
312
|
+
function parseCliArgs(argv) {
|
|
313
|
+
let commandPath = null;
|
|
314
|
+
let evalSetPath = null;
|
|
315
|
+
let maxIterations = 3;
|
|
316
|
+
|
|
317
|
+
for (let i = 0; i < argv.length; i++) {
|
|
318
|
+
if (argv[i] === '--eval-set' && i + 1 < argv.length) {
|
|
319
|
+
evalSetPath = argv[i + 1];
|
|
320
|
+
i++;
|
|
321
|
+
} else if (argv[i] === '--max-iterations' && i + 1 < argv.length) {
|
|
322
|
+
maxIterations = Number(argv[i + 1]);
|
|
323
|
+
i++;
|
|
324
|
+
} else if (!argv[i].startsWith('--')) {
|
|
325
|
+
commandPath = argv[i];
|
|
326
|
+
}
|
|
327
|
+
}
|
|
328
|
+
|
|
329
|
+
if (!commandPath) {
|
|
330
|
+
throw new Error('Usage: improve-command <command-path> --eval-set <path> [--max-iterations <N>]');
|
|
331
|
+
}
|
|
332
|
+
if (!evalSetPath) {
|
|
333
|
+
throw new Error('Missing required --eval-set <path>');
|
|
334
|
+
}
|
|
335
|
+
|
|
336
|
+
return { commandPath, evalSetPath, maxIterations };
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
if (require.main === module) {
|
|
340
|
+
const args = parseCliArgs(process.argv.slice(2));
|
|
341
|
+
const { runEvalPipeline } = require('./run-command-eval');
|
|
342
|
+
|
|
343
|
+
runImprovementLoop(args.commandPath, args.evalSetPath, {
|
|
344
|
+
maxIterations: args.maxIterations,
|
|
345
|
+
_runEval: (evalPath) => runEvalPipeline(evalPath),
|
|
346
|
+
})
|
|
347
|
+
.then((result) => {
|
|
348
|
+
console.log('\n=== Improvement Summary ===');
|
|
349
|
+
console.log(`Original score: ${result.originalScore.toFixed(2)}`);
|
|
350
|
+
console.log(`Best score: ${result.bestScore.toFixed(2)}`);
|
|
351
|
+
console.log(`Iterations: ${result.iterations}`);
|
|
352
|
+
console.log(`Reason: ${result.reason}`);
|
|
353
|
+
|
|
354
|
+
if (result.diff) {
|
|
355
|
+
console.log('\n=== Diff (original -> best) ===');
|
|
356
|
+
console.log(result.diff);
|
|
357
|
+
} else {
|
|
358
|
+
console.log('\nNo changes made.');
|
|
359
|
+
}
|
|
360
|
+
|
|
361
|
+
console.log('\nDiff shown above. Review and decide whether to apply.');
|
|
362
|
+
})
|
|
363
|
+
.catch((err) => {
|
|
364
|
+
console.error(`Error: ${err.message}`);
|
|
365
|
+
process.exit(1);
|
|
366
|
+
});
|
|
367
|
+
}
|
|
368
|
+
|
|
369
|
+
module.exports = {
|
|
370
|
+
analyzeFailures,
|
|
371
|
+
buildRewritePrompt,
|
|
372
|
+
defaultRewriteCommand,
|
|
373
|
+
generateDiff,
|
|
374
|
+
runImprovementLoop,
|
|
375
|
+
};
|
|
@@ -0,0 +1,268 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Eval runner core — worktree isolation + command execution.
|
|
3
|
+
*
|
|
4
|
+
* Provides building blocks for the eval pipeline:
|
|
5
|
+
* - createEvalWorktree() — spin up an isolated worktree
|
|
6
|
+
* - destroyEvalWorktree() — tear it down (force, even if dirty)
|
|
7
|
+
* - resetWorktree() — reset between eval queries
|
|
8
|
+
* - executeCommand() — run a claude CLI command in a worktree
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
const path = require('path');
|
|
12
|
+
const { execSync } = require('node:child_process');
|
|
13
|
+
|
|
14
|
+
// ── active worktree tracking (cleanup on crash) ─────────────────────
|
|
15
|
+
// Tracks active eval worktrees so we can clean up on process exit/crash.
|
|
16
|
+
// Prevents orphaned eval-* branches when interrupted.
|
|
17
|
+
// Note: execSync is safe here — all paths are internally generated, never user input.
|
|
18
|
+
const activeEvalWorktrees = new Map(); // path -> branch
|
|
19
|
+
|
|
20
|
+
function cleanupActiveWorktrees() {
|
|
21
|
+
if (activeEvalWorktrees.size === 0) return;
|
|
22
|
+
let repoRoot;
|
|
23
|
+
try { repoRoot = getRepoRoot(); } catch (_err) { return; }
|
|
24
|
+
for (const [wtPath, branch] of activeEvalWorktrees) {
|
|
25
|
+
try {
|
|
26
|
+
execSync(`git worktree remove --force "${wtPath}"`, { cwd: repoRoot, stdio: 'pipe' });
|
|
27
|
+
} catch (_err) { /* already removed */ }
|
|
28
|
+
if (branch && branch.startsWith('eval-')) {
|
|
29
|
+
try {
|
|
30
|
+
execSync(`git branch -D "${branch}"`, { cwd: repoRoot, stdio: 'pipe' });
|
|
31
|
+
} catch (_err) { /* already deleted */ }
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
try { execSync('git worktree prune', { cwd: repoRoot, stdio: 'pipe' }); } catch (_err) { /* ignore */ }
|
|
35
|
+
activeEvalWorktrees.clear();
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
process.on('exit', cleanupActiveWorktrees);
|
|
39
|
+
process.on('SIGINT', () => {
|
|
40
|
+
const hadWork = activeEvalWorktrees.size > 0;
|
|
41
|
+
cleanupActiveWorktrees();
|
|
42
|
+
if (hadWork) process.exit(130);
|
|
43
|
+
});
|
|
44
|
+
process.on('SIGTERM', () => {
|
|
45
|
+
const hadWork = activeEvalWorktrees.size > 0;
|
|
46
|
+
cleanupActiveWorktrees();
|
|
47
|
+
if (hadWork) process.exit(143);
|
|
48
|
+
});
|
|
49
|
+
|
|
50
|
+
// ── helpers ──────────────────────────────────────────────────────────
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Detect the repo root by walking up from cwd.
|
|
54
|
+
* Works from both the main repo and from within worktrees.
|
|
55
|
+
*/
|
|
56
|
+
function getRepoRoot() {
|
|
57
|
+
const root = execSync('git rev-parse --show-toplevel', {
|
|
58
|
+
encoding: 'utf-8',
|
|
59
|
+
stdio: ['pipe', 'pipe', 'pipe'],
|
|
60
|
+
}).trim();
|
|
61
|
+
return root;
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* Get the .worktrees directory path for eval worktrees.
|
|
66
|
+
* Eval worktrees live under <repo-root>/.worktrees/
|
|
67
|
+
*/
|
|
68
|
+
function getWorktreesDir() {
|
|
69
|
+
const root = getRepoRoot();
|
|
70
|
+
return path.join(root, '.worktrees');
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
// ── createEvalWorktree ───────────────────────────────────────────────
|
|
74
|
+
|
|
75
|
+
/**
|
|
76
|
+
* Create a git worktree with a unique name for eval isolation.
|
|
77
|
+
*
|
|
78
|
+
* @returns {Promise<{ path: string, branch: string }>}
|
|
79
|
+
*/
|
|
80
|
+
async function createEvalWorktree() {
|
|
81
|
+
const timestamp = Date.now();
|
|
82
|
+
const pid = process.pid;
|
|
83
|
+
const name = `eval-${timestamp}-${pid}`;
|
|
84
|
+
const branch = `eval-${timestamp}-${pid}`;
|
|
85
|
+
const worktreesDir = getWorktreesDir();
|
|
86
|
+
const wtPath = path.join(worktreesDir, name);
|
|
87
|
+
|
|
88
|
+
// Create the worktree with a detached HEAD first, then create branch
|
|
89
|
+
execSync(`git worktree add -b "${branch}" "${wtPath}" HEAD`, {
|
|
90
|
+
cwd: getRepoRoot(),
|
|
91
|
+
encoding: 'utf-8',
|
|
92
|
+
stdio: ['pipe', 'pipe', 'pipe'],
|
|
93
|
+
});
|
|
94
|
+
|
|
95
|
+
activeEvalWorktrees.set(wtPath, branch);
|
|
96
|
+
return { path: wtPath, branch };
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
// ── destroyEvalWorktree ──────────────────────────────────────────────
|
|
100
|
+
|
|
101
|
+
/**
|
|
102
|
+
* Remove a worktree and its temporary branch.
|
|
103
|
+
* Succeeds even if the worktree is dirty.
|
|
104
|
+
*
|
|
105
|
+
* @param {string} worktreePath — absolute path to the worktree
|
|
106
|
+
* @returns {Promise<void>}
|
|
107
|
+
*/
|
|
108
|
+
async function destroyEvalWorktree(worktreePath) {
|
|
109
|
+
const repoRoot = getRepoRoot();
|
|
110
|
+
|
|
111
|
+
// Query actual branch for this worktree (more reliable than inferring from dir name)
|
|
112
|
+
let branch;
|
|
113
|
+
try {
|
|
114
|
+
branch = execSync('git branch --show-current', {
|
|
115
|
+
cwd: worktreePath,
|
|
116
|
+
encoding: 'utf-8',
|
|
117
|
+
stdio: ['pipe', 'pipe', 'pipe'],
|
|
118
|
+
}).trim();
|
|
119
|
+
} catch (_err) {
|
|
120
|
+
// Worktree may be corrupted — fall back to directory name
|
|
121
|
+
branch = path.basename(worktreePath);
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
// Remove the worktree (--force handles dirty state)
|
|
125
|
+
execSync(`git worktree remove --force "${worktreePath}"`, {
|
|
126
|
+
cwd: repoRoot,
|
|
127
|
+
encoding: 'utf-8',
|
|
128
|
+
stdio: ['pipe', 'pipe', 'pipe'],
|
|
129
|
+
});
|
|
130
|
+
|
|
131
|
+
// Prune to clean up references
|
|
132
|
+
execSync('git worktree prune', {
|
|
133
|
+
cwd: repoRoot,
|
|
134
|
+
encoding: 'utf-8',
|
|
135
|
+
stdio: ['pipe', 'pipe', 'pipe'],
|
|
136
|
+
});
|
|
137
|
+
|
|
138
|
+
activeEvalWorktrees.delete(worktreePath);
|
|
139
|
+
|
|
140
|
+
// Delete the temporary branch (force in case it's not fully merged)
|
|
141
|
+
if (branch && branch.startsWith('eval-')) {
|
|
142
|
+
try {
|
|
143
|
+
execSync(`git branch -D "${branch}"`, {
|
|
144
|
+
cwd: repoRoot,
|
|
145
|
+
encoding: 'utf-8',
|
|
146
|
+
stdio: ['pipe', 'pipe', 'pipe'],
|
|
147
|
+
});
|
|
148
|
+
} catch (_err) {
|
|
149
|
+
// Branch may already be gone — ignore
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
// ── resetWorktree ────────────────────────────────────────────────────
|
|
155
|
+
|
|
156
|
+
/**
|
|
157
|
+
* Reset a worktree to a clean state (tracked files restored, untracked removed).
|
|
158
|
+
*
|
|
159
|
+
* @param {string} worktreePath — absolute path to the worktree
|
|
160
|
+
* @returns {Promise<void>}
|
|
161
|
+
*/
|
|
162
|
+
async function resetWorktree(worktreePath) {
|
|
163
|
+
// Restore tracked files
|
|
164
|
+
execSync('git checkout -- .', {
|
|
165
|
+
cwd: worktreePath,
|
|
166
|
+
encoding: 'utf-8',
|
|
167
|
+
stdio: ['pipe', 'pipe', 'pipe'],
|
|
168
|
+
});
|
|
169
|
+
|
|
170
|
+
// Remove untracked files, directories, and ignored files (full reset between runs)
|
|
171
|
+
execSync('git clean -fdx', {
|
|
172
|
+
cwd: worktreePath,
|
|
173
|
+
encoding: 'utf-8',
|
|
174
|
+
stdio: ['pipe', 'pipe', 'pipe'],
|
|
175
|
+
});
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
// ── executeCommand ───────────────────────────────────────────────────
|
|
179
|
+
|
|
180
|
+
/**
|
|
181
|
+
* Execute a command in an eval worktree.
|
|
182
|
+
*
|
|
183
|
+
* In production, runs `claude -p "<prompt>" --output-format stream-json --verbose --no-session-persistence`.
|
|
184
|
+
* Accepts an optional `cmdOverride` array for testing (avoids invoking real LLM).
|
|
185
|
+
*
|
|
186
|
+
* @param {string} _command — label only (e.g., "/status")
|
|
187
|
+
* @param {string} prompt — the prompt to send
|
|
188
|
+
* @param {string} worktreePath — absolute path to the worktree (used as cwd)
|
|
189
|
+
* @param {number} [timeout=120000] — timeout in milliseconds
|
|
190
|
+
* @param {string[]} [cmdOverride] — optional command array for testing
|
|
191
|
+
* @returns {Promise<{ stdout: string, stderr: string, exitCode: number, timedOut: boolean }>}
|
|
192
|
+
*/
|
|
193
|
+
async function executeCommand(_command, prompt, worktreePath, timeout = 120000, cmdOverride) {
|
|
194
|
+
// Build the command to run
|
|
195
|
+
const cmd = cmdOverride || [
|
|
196
|
+
'claude',
|
|
197
|
+
'-p',
|
|
198
|
+
prompt,
|
|
199
|
+
'--output-format',
|
|
200
|
+
'stream-json',
|
|
201
|
+
'--verbose',
|
|
202
|
+
'--no-session-persistence',
|
|
203
|
+
];
|
|
204
|
+
|
|
205
|
+
// Build environment: inherit current env, strip CLAUDECODE, set FORGE_EVAL
|
|
206
|
+
const env = { ...process.env };
|
|
207
|
+
delete env.CLAUDECODE;
|
|
208
|
+
env.FORGE_EVAL = '1';
|
|
209
|
+
|
|
210
|
+
// Spawn with Bun.spawn (array form, no shell interpolation)
|
|
211
|
+
const proc = Bun.spawn(cmd, { // eslint-disable-line no-undef -- Bun global provided by runtime
|
|
212
|
+
cwd: worktreePath,
|
|
213
|
+
env,
|
|
214
|
+
stdout: 'pipe',
|
|
215
|
+
stderr: 'pipe',
|
|
216
|
+
});
|
|
217
|
+
|
|
218
|
+
let timedOut = false;
|
|
219
|
+
let timeoutId;
|
|
220
|
+
|
|
221
|
+
// Set up timeout
|
|
222
|
+
const timeoutPromise = new Promise((resolve) => {
|
|
223
|
+
timeoutId = setTimeout(() => {
|
|
224
|
+
timedOut = true;
|
|
225
|
+
proc.kill();
|
|
226
|
+
resolve();
|
|
227
|
+
}, timeout);
|
|
228
|
+
});
|
|
229
|
+
|
|
230
|
+
// Wait for process to exit (or timeout)
|
|
231
|
+
const exitPromise = proc.exited.then(() => {
|
|
232
|
+
clearTimeout(timeoutId);
|
|
233
|
+
});
|
|
234
|
+
|
|
235
|
+
await Promise.race([exitPromise, timeoutPromise]);
|
|
236
|
+
|
|
237
|
+
// Read stdout and stderr
|
|
238
|
+
let stdout = '';
|
|
239
|
+
let stderr = '';
|
|
240
|
+
|
|
241
|
+
try {
|
|
242
|
+
stdout = await new Response(proc.stdout).text();
|
|
243
|
+
} catch (_err) {
|
|
244
|
+
// stream may be closed on kill — ignore
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
try {
|
|
248
|
+
stderr = await new Response(proc.stderr).text();
|
|
249
|
+
} catch (_err) {
|
|
250
|
+
// stream may be closed on kill — ignore
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
const exitCode = timedOut ? (proc.exitCode ?? 1) : proc.exitCode;
|
|
254
|
+
|
|
255
|
+
return {
|
|
256
|
+
stdout,
|
|
257
|
+
stderr,
|
|
258
|
+
exitCode,
|
|
259
|
+
timedOut,
|
|
260
|
+
};
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
module.exports = {
|
|
264
|
+
createEvalWorktree,
|
|
265
|
+
destroyEvalWorktree,
|
|
266
|
+
resetWorktree,
|
|
267
|
+
executeCommand,
|
|
268
|
+
};
|