claude-dev-env 8.26.4 → 8.26.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/SKILL.md +3 -3
- package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/evals/review/README.md +2 -2
- package/bin/ever-shipped-skills.mjs +1 -0
- package/bin/ever-shipped-skills.test.mjs +16 -2
- package/bin/install.prune.test.mjs +23 -0
- package/package.json +1 -1
- /package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/evals/review/baseline-responses.json +0 -0
- /package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/evals/review/baseline.md +0 -0
- /package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/evals/review/cases.json +0 -0
- /package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/evals/review/review_eval_support/config/constants.py +0 -0
- /package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/evals/review/review_eval_support/execution.py +0 -0
- /package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/evals/review/review_eval_support/grading.py +0 -0
- /package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/evals/review/run.py +0 -0
- /package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/evals/review/test_run.py +0 -0
- /package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/reference/build-evaluation.md +0 -0
- /package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/reference/graders-and-commands.md +0 -0
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
---
|
|
2
|
-
name:
|
|
2
|
+
name: build-eval
|
|
3
3
|
description: >-
|
|
4
4
|
Evaluate skills with a bounded direct Codex review suite or `claude plugin eval`.
|
|
5
5
|
Use labeled inputs and deterministic grading for output correctness; use a plugin
|
|
@@ -7,7 +7,7 @@ description: >-
|
|
|
7
7
|
Use when the user asks to eval or test a skill.
|
|
8
8
|
---
|
|
9
9
|
|
|
10
|
-
#
|
|
10
|
+
# Build eval
|
|
11
11
|
|
|
12
12
|
For building an evaluation of an AI workflow, follow [the design process](reference/build-evaluation.md).
|
|
13
13
|
For a direct Codex output-quality evaluation, use [the review suite](evals/review/README.md).
|
|
@@ -100,7 +100,7 @@ Otherwise work the steps in order.
|
|
|
100
100
|
and the `claude -p` recipe for flag-gated features.
|
|
101
101
|
|
|
102
102
|
```text
|
|
103
|
-
|
|
103
|
+
build-eval/
|
|
104
104
|
├── SKILL.md
|
|
105
105
|
└── reference/
|
|
106
106
|
└── graders-and-commands.md
|
|
@@ -21,9 +21,9 @@ Report finding precision and recall, clean-case specificity and false-positive r
|
|
|
21
21
|
From the repository root:
|
|
22
22
|
|
|
23
23
|
```powershell
|
|
24
|
-
$suite = 'packages/claude-dev-env/.agents/skills/
|
|
24
|
+
$suite = 'packages/claude-dev-env/.agents/skills/build-eval/evals/review/run.py'
|
|
25
25
|
python $suite validate
|
|
26
|
-
python -m pytest packages/claude-dev-env/.agents/skills/
|
|
26
|
+
python -m pytest packages/claude-dev-env/.agents/skills/build-eval/evals/review/test_run.py -q
|
|
27
27
|
python $suite live --limit 2 --timeout 90 --model gpt-6.1-sol --effort low --output review-smoke
|
|
28
28
|
python $suite live --split heldout --limit 6 --timeout 90 --model gpt-6.1-sol --effort low --output review-heldout
|
|
29
29
|
```
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { strict as assert } from 'node:assert';
|
|
2
2
|
import { test } from 'node:test';
|
|
3
|
-
import { existsSync, readdirSync } from 'node:fs';
|
|
3
|
+
import { existsSync, readFileSync, readdirSync } from 'node:fs';
|
|
4
4
|
import { dirname, join } from 'node:path';
|
|
5
5
|
import { fileURLToPath } from 'node:url';
|
|
6
6
|
import { EVER_SHIPPED_SKILL_NAMES } from './ever-shipped-skills.mjs';
|
|
@@ -41,7 +41,21 @@ test('EVER_SHIPPED_SKILL_NAMES includes the windows scheduled task skill', () =>
|
|
|
41
41
|
assert.equal(EVER_SHIPPED_SKILL_NAMES.has('windows-scheduled-task'), true);
|
|
42
42
|
});
|
|
43
43
|
|
|
44
|
-
test('
|
|
44
|
+
test('build-eval ships under its matching name without the retired active directory', () => {
|
|
45
|
+
const sourceSkillsDirectory = join(
|
|
46
|
+
PACKAGE_DIRECTORY,
|
|
47
|
+
PACKAGE_AGENTS_HOME_DIRECTORY_NAME,
|
|
48
|
+
MANAGED_SKILLS_DIRECTORY_NAME,
|
|
49
|
+
);
|
|
50
|
+
assert.ok(EVER_SHIPPED_SKILL_NAMES.has('build-eval'));
|
|
51
|
+
assert.match(
|
|
52
|
+
readFileSync(join(sourceSkillsDirectory, 'build-eval', 'SKILL.md'), 'utf8'),
|
|
53
|
+
/^---\r?\nname: build-eval\r?\n/,
|
|
54
|
+
);
|
|
55
|
+
assert.equal(existsSync(join(sourceSkillsDirectory, 'plugin-eval-standalone-skill')), false);
|
|
56
|
+
});
|
|
57
|
+
|
|
58
|
+
test('EVER_SHIPPED_SKILL_NAMES retains the retired plugin eval standalone skill', () => {
|
|
45
59
|
assert.equal(EVER_SHIPPED_SKILL_NAMES.has('plugin-eval-standalone-skill'), true);
|
|
46
60
|
});
|
|
47
61
|
|
|
@@ -24,6 +24,7 @@ const RETIRED_SKILL_DIRECTORIES = [
|
|
|
24
24
|
'fixbugs',
|
|
25
25
|
'pr-scope-resolve',
|
|
26
26
|
'post-audit-findings',
|
|
27
|
+
'plugin-eval-standalone-skill',
|
|
27
28
|
'pr-consistency-audit',
|
|
28
29
|
'bdd-protocol',
|
|
29
30
|
'hitl',
|
|
@@ -607,6 +608,28 @@ test('a full reinstall over a pre-manifest dirty tree prunes retired skills and
|
|
|
607
608
|
}
|
|
608
609
|
});
|
|
609
610
|
|
|
611
|
+
test('a full install replaces the retired plugin eval name with build-eval', () => {
|
|
612
|
+
const sandbox = createSandbox();
|
|
613
|
+
const retiredSkillName = 'plugin-eval-standalone-skill';
|
|
614
|
+
try {
|
|
615
|
+
plantSkillDirectory(sandbox.skillsDirectory, retiredSkillName, true);
|
|
616
|
+
|
|
617
|
+
runInstaller(sandbox.homeDirectory, []);
|
|
618
|
+
|
|
619
|
+
assert.equal(existsSync(join(sandbox.skillsDirectory, retiredSkillName)), false);
|
|
620
|
+
assert.ok(prunedSkillBackupContains(sandbox.claudeDirectory, retiredSkillName));
|
|
621
|
+
assert.match(
|
|
622
|
+
readFileSync(join(sandbox.skillsDirectory, 'build-eval', 'SKILL.md'), 'utf8'),
|
|
623
|
+
/^---\r?\nname: build-eval\r?\n/,
|
|
624
|
+
);
|
|
625
|
+
const manifest = readManifest(sandbox.manifestPath);
|
|
626
|
+
assert.ok(manifest.skills.includes('build-eval'));
|
|
627
|
+
assert.equal(manifest.skills.includes(retiredSkillName), false);
|
|
628
|
+
} finally {
|
|
629
|
+
rmSync(sandbox.homeDirectory, { recursive: true, force: true });
|
|
630
|
+
}
|
|
631
|
+
});
|
|
632
|
+
|
|
610
633
|
test('a full reinstall over an old-format manifest without a skills key still prunes via the ever-shipped fallback', () => {
|
|
611
634
|
const sandbox = createSandbox();
|
|
612
635
|
try {
|
package/package.json
CHANGED
|
File without changes
|
/package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/evals/review/baseline.md
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
/package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/evals/review/test_run.py
RENAMED
|
File without changes
|
/package/.agents/skills/{plugin-eval-standalone-skill → build-eval}/reference/build-evaluation.md
RENAMED
|
File without changes
|
|
File without changes
|