claude-dev-env 8.26.3 → 8.26.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/SKILL.md +3 -3
- package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/evals/review/README.md +2 -2
- package/.agents/skills/pull-request/SKILL.md +16 -0
- package/bin/ever-shipped-skills.mjs +1 -0
- package/bin/ever-shipped-skills.test.mjs +16 -2
- package/bin/install.prune.test.mjs +23 -0
- package/package.json +1 -1
- /package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/evals/review/baseline-responses.json +0 -0
- /package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/evals/review/baseline.md +0 -0
- /package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/evals/review/cases.json +0 -0
- /package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/evals/review/review_eval_support/config/constants.py +0 -0
- /package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/evals/review/review_eval_support/execution.py +0 -0
- /package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/evals/review/review_eval_support/grading.py +0 -0
- /package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/evals/review/run.py +0 -0
- /package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/evals/review/test_run.py +0 -0
- /package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/reference/build-evaluation.md +0 -0
- /package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/reference/graders-and-commands.md +0 -0
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
---
|
|
2
|
-
name:
|
|
2
|
+
name: build-eval
|
|
3
3
|
description: >-
|
|
4
4
|
Evaluate skills with a bounded direct Codex review suite or `claude plugin eval`.
|
|
5
5
|
Use labeled inputs and deterministic grading for output correctness; use a plugin
|
|
@@ -7,7 +7,7 @@ description: >-
|
|
|
7
7
|
Use when the user asks to eval or test a skill.
|
|
8
8
|
---
|
|
9
9
|
|
|
10
|
-
#
|
|
10
|
+
# Build eval
|
|
11
11
|
|
|
12
12
|
For building an evaluation of an AI workflow, follow [the design process](reference/build-evaluation.md).
|
|
13
13
|
For a direct Codex output-quality evaluation, use [the review suite](evals/review/README.md).
|
|
@@ -100,7 +100,7 @@ Otherwise work the steps in order.
|
|
|
100
100
|
and the `claude -p` recipe for flag-gated features.
|
|
101
101
|
|
|
102
102
|
```text
|
|
103
|
-
|
|
103
|
+
build-eval/
|
|
104
104
|
├── SKILL.md
|
|
105
105
|
└── reference/
|
|
106
106
|
└── graders-and-commands.md
|
|
@@ -21,9 +21,9 @@ Report finding precision and recall, clean-case specificity and false-positive r
|
|
|
21
21
|
From the repository root:
|
|
22
22
|
|
|
23
23
|
```powershell
|
|
24
|
-
$suite = 'packages/claude-dev-env/.agents/skills/
|
|
24
|
+
$suite = 'packages/claude-dev-env/.agents/skills/build-eval/evals/review/run.py'
|
|
25
25
|
python $suite validate
|
|
26
|
-
python -m pytest packages/claude-dev-env/.agents/skills/
|
|
26
|
+
python -m pytest packages/claude-dev-env/.agents/skills/build-eval/evals/review/test_run.py -q
|
|
27
27
|
python $suite live --limit 2 --timeout 90 --model gpt-6.1-sol --effort low --output review-smoke
|
|
28
28
|
python $suite live --split heldout --limit 6 --timeout 90 --model gpt-6.1-sol --effort low --output review-heldout
|
|
29
29
|
```
|
|
@@ -105,6 +105,20 @@ Check each behavior claim against the final diff and verification evidence.
|
|
|
105
105
|
Preserve whether a rule is added, removed, or narrowed, and distinguish tests
|
|
106
106
|
added from tests run. Refresh these claims after a rebase or correction.
|
|
107
107
|
|
|
108
|
+
Open with the concrete problem and resulting behavior in one or two short
|
|
109
|
+
paragraphs of plain prose. Write for a reviewer who has not read the worker's
|
|
110
|
+
conversation. Include only claims supported by the final diff. State commands,
|
|
111
|
+
results, and material limits under `## Verification`.
|
|
112
|
+
|
|
113
|
+
Keep substantive review and test evidence. Put extensive logs, raw diffs, token
|
|
114
|
+
counts, and agent transcripts in an existing linked artifact or a closed
|
|
115
|
+
`<details>` section after the primary explanation. Let justified evidence grow
|
|
116
|
+
without a total-length cap. A collapsed transcript still needs a useful opening.
|
|
117
|
+
|
|
118
|
+
Load a repository's PR-description skill and run the checker command it supplies.
|
|
119
|
+
Apply these steps to automated publishers too. A green code check proves no
|
|
120
|
+
description claim by itself.
|
|
121
|
+
|
|
108
122
|
### 3. Run the local linter
|
|
109
123
|
|
|
110
124
|
Resolve the active managed root (`CLAUDE_CONFIG_DIR` when set, `~/.claude`
|
|
@@ -158,6 +172,8 @@ without exposing author values.
|
|
|
158
172
|
|
|
159
173
|
Confirm the published claims still describe the verified head and preserve the
|
|
160
174
|
scope of its evidence, including whether checks used saved artifacts or a new run.
|
|
175
|
+
Read the remote opening with evidence sections collapsed. Confirm that it explains
|
|
176
|
+
the change on its own and that the evidence links lead to the cited results.
|
|
161
177
|
|
|
162
178
|
## Exit handling
|
|
163
179
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { strict as assert } from 'node:assert';
|
|
2
2
|
import { test } from 'node:test';
|
|
3
|
-
import { existsSync, readdirSync } from 'node:fs';
|
|
3
|
+
import { existsSync, readFileSync, readdirSync } from 'node:fs';
|
|
4
4
|
import { dirname, join } from 'node:path';
|
|
5
5
|
import { fileURLToPath } from 'node:url';
|
|
6
6
|
import { EVER_SHIPPED_SKILL_NAMES } from './ever-shipped-skills.mjs';
|
|
@@ -41,7 +41,21 @@ test('EVER_SHIPPED_SKILL_NAMES includes the windows scheduled task skill', () =>
|
|
|
41
41
|
assert.equal(EVER_SHIPPED_SKILL_NAMES.has('windows-scheduled-task'), true);
|
|
42
42
|
});
|
|
43
43
|
|
|
44
|
-
test('
|
|
44
|
+
test('build-eval ships under its matching name without the retired active directory', () => {
|
|
45
|
+
const sourceSkillsDirectory = join(
|
|
46
|
+
PACKAGE_DIRECTORY,
|
|
47
|
+
PACKAGE_AGENTS_HOME_DIRECTORY_NAME,
|
|
48
|
+
MANAGED_SKILLS_DIRECTORY_NAME,
|
|
49
|
+
);
|
|
50
|
+
assert.ok(EVER_SHIPPED_SKILL_NAMES.has('build-eval'));
|
|
51
|
+
assert.match(
|
|
52
|
+
readFileSync(join(sourceSkillsDirectory, 'build-eval', 'SKILL.md'), 'utf8'),
|
|
53
|
+
/^---\r?\nname: build-eval\r?\n/,
|
|
54
|
+
);
|
|
55
|
+
assert.equal(existsSync(join(sourceSkillsDirectory, 'plugin-eval-standalone-skill')), false);
|
|
56
|
+
});
|
|
57
|
+
|
|
58
|
+
test('EVER_SHIPPED_SKILL_NAMES retains the retired plugin eval standalone skill', () => {
|
|
45
59
|
assert.equal(EVER_SHIPPED_SKILL_NAMES.has('plugin-eval-standalone-skill'), true);
|
|
46
60
|
});
|
|
47
61
|
|
|
@@ -24,6 +24,7 @@ const RETIRED_SKILL_DIRECTORIES = [
|
|
|
24
24
|
'fixbugs',
|
|
25
25
|
'pr-scope-resolve',
|
|
26
26
|
'post-audit-findings',
|
|
27
|
+
'plugin-eval-standalone-skill',
|
|
27
28
|
'pr-consistency-audit',
|
|
28
29
|
'bdd-protocol',
|
|
29
30
|
'hitl',
|
|
@@ -607,6 +608,28 @@ test('a full reinstall over a pre-manifest dirty tree prunes retired skills and
|
|
|
607
608
|
}
|
|
608
609
|
});
|
|
609
610
|
|
|
611
|
+
test('a full install replaces the retired plugin eval name with build-eval', () => {
|
|
612
|
+
const sandbox = createSandbox();
|
|
613
|
+
const retiredSkillName = 'plugin-eval-standalone-skill';
|
|
614
|
+
try {
|
|
615
|
+
plantSkillDirectory(sandbox.skillsDirectory, retiredSkillName, true);
|
|
616
|
+
|
|
617
|
+
runInstaller(sandbox.homeDirectory, []);
|
|
618
|
+
|
|
619
|
+
assert.equal(existsSync(join(sandbox.skillsDirectory, retiredSkillName)), false);
|
|
620
|
+
assert.ok(prunedSkillBackupContains(sandbox.claudeDirectory, retiredSkillName));
|
|
621
|
+
assert.match(
|
|
622
|
+
readFileSync(join(sandbox.skillsDirectory, 'build-eval', 'SKILL.md'), 'utf8'),
|
|
623
|
+
/^---\r?\nname: build-eval\r?\n/,
|
|
624
|
+
);
|
|
625
|
+
const manifest = readManifest(sandbox.manifestPath);
|
|
626
|
+
assert.ok(manifest.skills.includes('build-eval'));
|
|
627
|
+
assert.equal(manifest.skills.includes(retiredSkillName), false);
|
|
628
|
+
} finally {
|
|
629
|
+
rmSync(sandbox.homeDirectory, { recursive: true, force: true });
|
|
630
|
+
}
|
|
631
|
+
});
|
|
632
|
+
|
|
610
633
|
test('a full reinstall over an old-format manifest without a skills key still prunes via the ever-shipped fallback', () => {
|
|
611
634
|
const sandbox = createSandbox();
|
|
612
635
|
try {
|
package/package.json
CHANGED
|
File without changes
|
/package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/evals/review/baseline.md
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
/package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/evals/review/test_run.py
RENAMED
|
File without changes
|
/package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/reference/build-evaluation.md
RENAMED
|
File without changes
|
|
File without changes
|