praxis-agent 0.46.4 → 0.47.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -0
- package/dist/cli-runtime.d.ts +2 -0
- package/dist/cli-runtime.js +29 -2
- package/dist/evals/eval-contract.d.ts +102 -0
- package/dist/evals/eval-contract.js +57 -0
- package/dist/evals/eval-graders.d.ts +6 -0
- package/dist/evals/eval-graders.js +94 -0
- package/dist/evals/project-eval-runner.d.ts +59 -0
- package/dist/evals/project-eval-runner.js +359 -0
- package/dist/evals/project-eval-schema.d.ts +35 -0
- package/dist/evals/project-eval-schema.js +395 -0
- package/dist/evals/project-eval-workspace.d.ts +30 -0
- package/dist/evals/project-eval-workspace.js +108 -0
- package/dist/evals/project-eval.d.ts +77 -0
- package/dist/evals/project-eval.js +214 -0
- package/dist/platform/bounded-process-runner.d.ts +2 -0
- package/dist/platform/bounded-process-runner.js +4 -2
- package/dist/plugins/claude-plugin-eval-graders.d.ts +2 -33
- package/dist/plugins/claude-plugin-eval-graders.js +7 -66
- package/dist/plugins/claude-plugin-eval-runner.d.ts +4 -29
- package/dist/plugins/claude-plugin-eval-runner.js +3 -58
- package/dist/plugins/claude-plugin-eval-schema.d.ts +9 -34
- package/dist/tools/local-tools.js +16 -1
- package/package.json +1 -1
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
import { join, relative, resolve } from 'node:path';
|
|
2
|
+
import { writeFileAtomically } from '../platform/atomic-write.js';
|
|
3
|
+
import { runProjectEvalCase, } from './project-eval-runner.js';
|
|
4
|
+
import { discoverProjectEvalCases } from './project-eval-schema.js';
|
|
5
|
+
export const PROJECT_EVAL_HELP = `Usage: praxis eval [options] <target>
|
|
6
|
+
|
|
7
|
+
Run deterministic project outcome evaluations in isolated workspaces.
|
|
8
|
+
|
|
9
|
+
Options:
|
|
10
|
+
--case <glob> Filter case names
|
|
11
|
+
--tag <tag[,tag]> Filter tags; repeatable
|
|
12
|
+
--runs <1..50> Override run count
|
|
13
|
+
--model <model> Override model
|
|
14
|
+
--allow-tools <rules> Grant gated tools; comma-separated and repeatable
|
|
15
|
+
--run-verification Enable verifier subprocesses
|
|
16
|
+
--output-dir <dir> Write artifacts to this directory
|
|
17
|
+
--keep-temp Preserve temporary workspaces
|
|
18
|
+
--json Print exactly one aggregate JSON value
|
|
19
|
+
--verbose Print run progress to stderr
|
|
20
|
+
-h, --help Display help`;
|
|
21
|
+
function takeValue(argv, index, option) {
|
|
22
|
+
const value = argv[index + 1];
|
|
23
|
+
if (!value || value.startsWith('-'))
|
|
24
|
+
throw new Error(`${option} requires a value`);
|
|
25
|
+
return value;
|
|
26
|
+
}
|
|
27
|
+
function listValues(value, option) {
|
|
28
|
+
const values = value.split(',').map((item) => item.trim());
|
|
29
|
+
if (values.some((item) => item.length === 0))
|
|
30
|
+
throw new Error(`${option} contains an empty value`);
|
|
31
|
+
return values;
|
|
32
|
+
}
|
|
33
|
+
export function parseProjectEvalOptions(argv) {
|
|
34
|
+
const options = {
|
|
35
|
+
tags: [],
|
|
36
|
+
allowTools: [],
|
|
37
|
+
runVerification: false,
|
|
38
|
+
keepTemp: false,
|
|
39
|
+
json: false,
|
|
40
|
+
verbose: false,
|
|
41
|
+
};
|
|
42
|
+
const operands = [];
|
|
43
|
+
for (let index = 0; index < argv.length; index += 1) {
|
|
44
|
+
const value = argv[index];
|
|
45
|
+
if (!value)
|
|
46
|
+
continue;
|
|
47
|
+
if (value === '-h' || value === '--help')
|
|
48
|
+
return { ...options, help: true };
|
|
49
|
+
if (value === '--run-verification')
|
|
50
|
+
options.runVerification = true;
|
|
51
|
+
else if (value === '--keep-temp')
|
|
52
|
+
options.keepTemp = true;
|
|
53
|
+
else if (value === '--json')
|
|
54
|
+
options.json = true;
|
|
55
|
+
else if (value === '--verbose')
|
|
56
|
+
options.verbose = true;
|
|
57
|
+
else if (value === '--case' ||
|
|
58
|
+
value === '--model' ||
|
|
59
|
+
value === '--output-dir' ||
|
|
60
|
+
value === '--runs' ||
|
|
61
|
+
value === '--tag' ||
|
|
62
|
+
value === '--allow-tools') {
|
|
63
|
+
const selected = takeValue(argv, index, value);
|
|
64
|
+
index += 1;
|
|
65
|
+
if (value === '--case')
|
|
66
|
+
options.caseGlob = selected;
|
|
67
|
+
else if (value === '--model')
|
|
68
|
+
options.model = selected;
|
|
69
|
+
else if (value === '--output-dir')
|
|
70
|
+
options.outputDir = selected;
|
|
71
|
+
else if (value === '--runs') {
|
|
72
|
+
const runs = Number(selected);
|
|
73
|
+
if (!Number.isInteger(runs) || runs < 1 || runs > 50)
|
|
74
|
+
throw new Error('--runs must be an integer from 1 to 50');
|
|
75
|
+
options.runs = runs;
|
|
76
|
+
}
|
|
77
|
+
else if (value === '--tag')
|
|
78
|
+
options.tags.push(...listValues(selected, value));
|
|
79
|
+
else
|
|
80
|
+
options.allowTools.push(...listValues(selected, value));
|
|
81
|
+
}
|
|
82
|
+
else if (value.startsWith('-'))
|
|
83
|
+
throw new Error(`Unknown eval option: ${value}`);
|
|
84
|
+
else
|
|
85
|
+
operands.push(value);
|
|
86
|
+
}
|
|
87
|
+
if (operands.length !== 1)
|
|
88
|
+
throw new Error('eval requires one target');
|
|
89
|
+
const target = operands[0];
|
|
90
|
+
if (!target)
|
|
91
|
+
throw new Error('eval requires one target');
|
|
92
|
+
options.target = target;
|
|
93
|
+
return options;
|
|
94
|
+
}
|
|
95
|
+
function aggregateUsage(results) {
|
|
96
|
+
const totals = {
|
|
97
|
+
input_tokens: 0,
|
|
98
|
+
output_tokens: 0,
|
|
99
|
+
cache_read_input_tokens: 0,
|
|
100
|
+
cache_creation_input_tokens: 0,
|
|
101
|
+
web_search_requests: 0,
|
|
102
|
+
};
|
|
103
|
+
for (const result of results) {
|
|
104
|
+
if (!result.usage)
|
|
105
|
+
continue;
|
|
106
|
+
totals.input_tokens += result.usage.inputTokens;
|
|
107
|
+
totals.output_tokens += result.usage.outputTokens;
|
|
108
|
+
totals.cache_read_input_tokens += result.usage.cacheReadInputTokens ?? 0;
|
|
109
|
+
totals.cache_creation_input_tokens +=
|
|
110
|
+
result.usage.cacheCreationInputTokens ?? 0;
|
|
111
|
+
totals.web_search_requests += result.usage.webSearchRequests ?? 0;
|
|
112
|
+
}
|
|
113
|
+
return totals;
|
|
114
|
+
}
|
|
115
|
+
function runSummary(result, outputDirectory) {
|
|
116
|
+
return {
|
|
117
|
+
case: result.case,
|
|
118
|
+
run: result.run,
|
|
119
|
+
model: result.model,
|
|
120
|
+
passed: result.passed,
|
|
121
|
+
score: result.score,
|
|
122
|
+
turns: result.turns,
|
|
123
|
+
usage: result.usage,
|
|
124
|
+
cost_usd: result.cost_usd,
|
|
125
|
+
cost_known: result.cost_known,
|
|
126
|
+
duration_ms: result.duration_ms,
|
|
127
|
+
termination: result.termination,
|
|
128
|
+
error: result.error,
|
|
129
|
+
artifact_dir: relative(outputDirectory, join(outputDirectory, result.case, `run-${result.run}`)).replaceAll('\\', '/'),
|
|
130
|
+
};
|
|
131
|
+
}
|
|
132
|
+
export async function executeProjectEvalCommand(argv, io, dependencies, callerCwd = process.cwd(), signal) {
|
|
133
|
+
const options = parseProjectEvalOptions(argv);
|
|
134
|
+
if (options.help) {
|
|
135
|
+
io.stdout(PROJECT_EVAL_HELP);
|
|
136
|
+
return 0;
|
|
137
|
+
}
|
|
138
|
+
if (!options.target)
|
|
139
|
+
throw new Error('eval requires one target');
|
|
140
|
+
const target = resolve(callerCwd, options.target);
|
|
141
|
+
const cases = await discoverProjectEvalCases(target, options.caseGlob, options.tags);
|
|
142
|
+
const outputDirectory = options.outputDir
|
|
143
|
+
? resolve(callerCwd, options.outputDir)
|
|
144
|
+
: join(dependencies.configRoot, 'evals', 'results', new Date().toISOString().replaceAll(':', '-'));
|
|
145
|
+
const plannedRunCount = cases.reduce((total, definition) => total + (options.runs ?? definition.runs), 0);
|
|
146
|
+
const results = [];
|
|
147
|
+
const started = Date.now();
|
|
148
|
+
let interrupted = signal?.aborted ?? false;
|
|
149
|
+
runs: for (const definition of cases) {
|
|
150
|
+
for (let runIndex = 1; runIndex <= (options.runs ?? definition.runs); runIndex += 1) {
|
|
151
|
+
if (signal?.aborted) {
|
|
152
|
+
interrupted = true;
|
|
153
|
+
break runs;
|
|
154
|
+
}
|
|
155
|
+
const result = await runProjectEvalCase({
|
|
156
|
+
case: definition,
|
|
157
|
+
factory: dependencies.runtimeFactory,
|
|
158
|
+
run: runIndex,
|
|
159
|
+
allowTools: options.allowTools,
|
|
160
|
+
...(options.model === undefined ? {} : { model: options.model }),
|
|
161
|
+
keepTemp: options.keepTemp,
|
|
162
|
+
runVerification: options.runVerification,
|
|
163
|
+
outputDir: outputDirectory,
|
|
164
|
+
version: dependencies.version ?? 'unknown',
|
|
165
|
+
...(signal === undefined ? {} : { signal }),
|
|
166
|
+
});
|
|
167
|
+
results.push(result);
|
|
168
|
+
if (options.verbose)
|
|
169
|
+
io.stderr(`${definition.name} run ${runIndex}: ${result.passed ? 'passed' : 'failed'}\n`);
|
|
170
|
+
if (signal?.aborted || result.termination === 'interrupted') {
|
|
171
|
+
interrupted = true;
|
|
172
|
+
break runs;
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
}
|
|
176
|
+
const passed = results.filter((result) => result.passed).length;
|
|
177
|
+
const knownCostResults = results.filter((result) => result.cost_known);
|
|
178
|
+
const aggregate = {
|
|
179
|
+
schema_version: '1.0',
|
|
180
|
+
version: dependencies.version ?? 'unknown',
|
|
181
|
+
start: new Date(started).toISOString(),
|
|
182
|
+
duration_ms: Date.now() - started,
|
|
183
|
+
target,
|
|
184
|
+
output_dir: outputDirectory,
|
|
185
|
+
model: options.model ?? null,
|
|
186
|
+
case_count: cases.length,
|
|
187
|
+
planned_run_count: plannedRunCount,
|
|
188
|
+
completed_run_count: results.length,
|
|
189
|
+
run_count: results.length,
|
|
190
|
+
passed,
|
|
191
|
+
failed: results.length - passed,
|
|
192
|
+
pass_rate: results.length === 0 ? 0 : passed / results.length,
|
|
193
|
+
total_turns: results.reduce((total, result) => total + result.turns, 0),
|
|
194
|
+
usage_totals: aggregateUsage(results),
|
|
195
|
+
usage_known_runs: results.filter((result) => result.usage !== null).length,
|
|
196
|
+
usage_unknown_runs: results.filter((result) => result.usage === null)
|
|
197
|
+
.length,
|
|
198
|
+
known_cost_total_usd: knownCostResults.length === 0
|
|
199
|
+
? null
|
|
200
|
+
: knownCostResults.reduce((total, result) => total + (result.cost_usd ?? 0), 0),
|
|
201
|
+
known_cost_runs: knownCostResults.length,
|
|
202
|
+
unknown_cost_runs: results.length - knownCostResults.length,
|
|
203
|
+
partial: interrupted || results.length < plannedRunCount,
|
|
204
|
+
interrupted,
|
|
205
|
+
runs: results.map((result) => runSummary(result, outputDirectory)),
|
|
206
|
+
};
|
|
207
|
+
await writeFileAtomically(join(outputDirectory, 'aggregate-result.json'), JSON.stringify(aggregate, null, 2));
|
|
208
|
+
if (options.json)
|
|
209
|
+
io.stdout(`${JSON.stringify(aggregate)}\n`);
|
|
210
|
+
else
|
|
211
|
+
io.stdout(`${passed}/${results.length} passed\n`);
|
|
212
|
+
return interrupted ? 130 : passed === results.length ? 0 : 1;
|
|
213
|
+
}
|
|
214
|
+
//# sourceMappingURL=project-eval.js.map
|
|
@@ -19,6 +19,8 @@ export interface RunProcessOptions {
|
|
|
19
19
|
signal?: AbortSignal;
|
|
20
20
|
onOutput?: (output: string) => void | Promise<void>;
|
|
21
21
|
env?: Readonly<Record<string, string>>;
|
|
22
|
+
inheritEnvironment?: boolean;
|
|
23
|
+
redactExplicitEnvironment?: boolean;
|
|
22
24
|
controlOutputBytes?: number;
|
|
23
25
|
controlOutputFd?: number;
|
|
24
26
|
scriptInput?: string;
|
|
@@ -84,7 +84,7 @@ export class BoundedProcessRunner {
|
|
|
84
84
|
if (options.signal?.aborted)
|
|
85
85
|
return Promise.reject(abortError());
|
|
86
86
|
return new Promise((resolve, reject) => {
|
|
87
|
-
const sensitiveValues = sensitiveEnvironmentValues(process.env);
|
|
87
|
+
const sensitiveValues = sensitiveEnvironmentValues(process.env, ...(options.redactExplicitEnvironment ? [options.env ?? {}] : []));
|
|
88
88
|
const longestSensitiveValueBytes = sensitiveValues.reduce((longest, value) => Math.max(longest, Buffer.byteLength(value)), 0);
|
|
89
89
|
const rawOutputLimit = this.options.maxOutputBytes + Math.max(3, longestSensitiveValueBytes);
|
|
90
90
|
const controlOutputFd = options.controlOutputBytes === undefined
|
|
@@ -108,7 +108,9 @@ export class BoundedProcessRunner {
|
|
|
108
108
|
const child = spawn(options.command, options.args, {
|
|
109
109
|
cwd: options.cwd ?? this.options.cwd,
|
|
110
110
|
detached: process.platform !== 'win32',
|
|
111
|
-
env:
|
|
111
|
+
env: options.inheritEnvironment === false
|
|
112
|
+
? sanitizeChildEnvironment(options.env ?? {}, {})
|
|
113
|
+
: sanitizeChildEnvironment(options.env ?? {}),
|
|
112
114
|
stdio,
|
|
113
115
|
});
|
|
114
116
|
const chunks = { stdout: [], stderr: [] };
|
|
@@ -1,37 +1,6 @@
|
|
|
1
1
|
import type { ClaudePluginEvalCase } from './claude-plugin-eval-schema.js';
|
|
2
|
-
export
|
|
3
|
-
|
|
4
|
-
tool?: string;
|
|
5
|
-
input?: Record<string, unknown>;
|
|
6
|
-
[key: string]: unknown;
|
|
7
|
-
}
|
|
8
|
-
export interface EvalRunArtifacts {
|
|
9
|
-
lastMessage: string;
|
|
10
|
-
trace: readonly EvalTraceEvent[];
|
|
11
|
-
cwd: string;
|
|
12
|
-
}
|
|
13
|
-
export interface EvalGraderResult {
|
|
14
|
-
name: string;
|
|
15
|
-
passed: boolean;
|
|
16
|
-
weight: number;
|
|
17
|
-
explanation: string;
|
|
18
|
-
judge_votes?: readonly boolean[];
|
|
19
|
-
evidence?: string;
|
|
20
|
-
with_only?: boolean;
|
|
21
|
-
}
|
|
22
|
-
export interface EvalJudge {
|
|
23
|
-
vote(request: {
|
|
24
|
-
criteria: string;
|
|
25
|
-
focus: string;
|
|
26
|
-
baseline?: string;
|
|
27
|
-
model: string;
|
|
28
|
-
signal?: AbortSignal;
|
|
29
|
-
}): Promise<{
|
|
30
|
-
passed: boolean;
|
|
31
|
-
explanation?: string;
|
|
32
|
-
costUsd: number;
|
|
33
|
-
}>;
|
|
34
|
-
}
|
|
2
|
+
export type { EvalTraceEvent, EvalRunArtifacts, EvalGraderResult, EvalJudge, } from '../evals/eval-contract.js';
|
|
3
|
+
import type { EvalRunArtifacts, EvalGraderResult, EvalJudge } from '../evals/eval-contract.js';
|
|
35
4
|
export declare function gradeClaudePluginEvalRun(options: {
|
|
36
5
|
case: ClaudePluginEvalCase;
|
|
37
6
|
artifacts: EvalRunArtifacts;
|
|
@@ -1,20 +1,6 @@
|
|
|
1
1
|
import { readFile } from 'node:fs/promises';
|
|
2
|
-
import { minimatch } from 'minimatch';
|
|
3
2
|
import { resolveContainedPath } from './claude-plugin-eval-schema.js';
|
|
4
|
-
|
|
5
|
-
if (expected && typeof expected === 'object' && !Array.isArray(expected)) {
|
|
6
|
-
if (!actual || typeof actual !== 'object' || Array.isArray(actual))
|
|
7
|
-
return false;
|
|
8
|
-
return Object.entries(expected).every(([key, value]) => subset(actual[key], value));
|
|
9
|
-
}
|
|
10
|
-
return Object.is(actual, expected);
|
|
11
|
-
}
|
|
12
|
-
function matches(event, match) {
|
|
13
|
-
const selected = typeof match === 'string' ? { tool: match } : match;
|
|
14
|
-
return (event.type === 'tool-call' &&
|
|
15
|
-
event.tool === selected.tool &&
|
|
16
|
-
(!selected.input_match || subset(event.input, selected.input_match)));
|
|
17
|
-
}
|
|
3
|
+
import { gradeDeterministicEvalRun } from '../evals/eval-graders.js';
|
|
18
4
|
async function focusText(focus, artifacts) {
|
|
19
5
|
if (focus === 'last_message')
|
|
20
6
|
return artifacts.lastMessage;
|
|
@@ -32,56 +18,6 @@ async function focusText(focus, artifacts) {
|
|
|
32
18
|
}
|
|
33
19
|
return readFile(await resolveContainedPath(artifacts.cwd, focus.path, 'Grader focus'), 'utf8');
|
|
34
20
|
}
|
|
35
|
-
async function freeGrade(grader, artifacts) {
|
|
36
|
-
let passed = false;
|
|
37
|
-
let evidence = '';
|
|
38
|
-
if (grader.type === 'regex') {
|
|
39
|
-
const target = await focusText(grader.target, artifacts);
|
|
40
|
-
const expression = new RegExp(grader.pattern, grader.flags.includes('g') ? grader.flags : `${grader.flags}g`);
|
|
41
|
-
const count = [...target.matchAll(expression)].length;
|
|
42
|
-
passed =
|
|
43
|
-
grader.match === 'contains'
|
|
44
|
-
? count > 0
|
|
45
|
-
: grader.match === 'not_contains'
|
|
46
|
-
? count === 0
|
|
47
|
-
: count === Number(grader.match.slice(6));
|
|
48
|
-
evidence = `matches=${count}`;
|
|
49
|
-
}
|
|
50
|
-
else if (grader.type === 'tool_used') {
|
|
51
|
-
const count = artifacts.trace.filter((event) => matches(event, {
|
|
52
|
-
tool: grader.tool,
|
|
53
|
-
...(grader.input_match ? { input_match: grader.input_match } : {}),
|
|
54
|
-
})).length;
|
|
55
|
-
passed = count >= grader.min && count <= grader.max;
|
|
56
|
-
evidence = `uses=${count}`;
|
|
57
|
-
}
|
|
58
|
-
else if (grader.type === 'tool_order') {
|
|
59
|
-
const before = artifacts.trace.findIndex((event) => matches(event, grader.before));
|
|
60
|
-
const after = artifacts.trace.findIndex((event, index) => index > before && matches(event, grader.after));
|
|
61
|
-
passed = before >= 0 && after > before;
|
|
62
|
-
evidence = `before=${before},after=${after}`;
|
|
63
|
-
}
|
|
64
|
-
else {
|
|
65
|
-
const { glob } = await import('node:fs/promises');
|
|
66
|
-
let count = 0;
|
|
67
|
-
for await (const path of glob('**/*', { cwd: artifacts.cwd })) {
|
|
68
|
-
if (!minimatch(path, grader.path))
|
|
69
|
-
continue;
|
|
70
|
-
await resolveContainedPath(artifacts.cwd, path, 'file_exists match');
|
|
71
|
-
count += 1;
|
|
72
|
-
}
|
|
73
|
-
passed = grader.exists ? count > 0 : count === 0;
|
|
74
|
-
evidence = `files=${count}`;
|
|
75
|
-
}
|
|
76
|
-
return {
|
|
77
|
-
name: grader.name,
|
|
78
|
-
passed,
|
|
79
|
-
weight: grader.weight,
|
|
80
|
-
explanation: passed ? 'passed' : 'failed',
|
|
81
|
-
evidence,
|
|
82
|
-
...(grader.arm === 'with-only' ? { with_only: true } : {}),
|
|
83
|
-
};
|
|
84
|
-
}
|
|
85
21
|
export async function gradeClaudePluginEvalRun(options) {
|
|
86
22
|
const results = [];
|
|
87
23
|
let skippedPaidGraders = false;
|
|
@@ -90,7 +26,12 @@ export async function gradeClaudePluginEvalRun(options) {
|
|
|
90
26
|
if (options.arm === 'without' && item.arm === 'with-only')
|
|
91
27
|
continue;
|
|
92
28
|
if (item.type !== 'llm' && item.type !== 'baseline') {
|
|
93
|
-
|
|
29
|
+
const [result] = await gradeDeterministicEvalRun({
|
|
30
|
+
graders: [item],
|
|
31
|
+
artifacts: options.artifacts,
|
|
32
|
+
});
|
|
33
|
+
if (result)
|
|
34
|
+
results.push(result);
|
|
94
35
|
continue;
|
|
95
36
|
}
|
|
96
37
|
if (options.skipPaid) {
|
|
@@ -1,33 +1,9 @@
|
|
|
1
|
-
import type { RuntimeEvent } from '../core/runtime.js';
|
|
2
1
|
import type { DataPlane } from '../persistence/data-plane.js';
|
|
3
2
|
import type { ClaudePluginEvalCase } from './claude-plugin-eval-schema.js';
|
|
4
|
-
import type { EvalRunArtifacts } from '
|
|
5
|
-
export
|
|
6
|
-
export
|
|
7
|
-
|
|
8
|
-
text: string;
|
|
9
|
-
turns: number;
|
|
10
|
-
costUsd?: number;
|
|
11
|
-
}>;
|
|
12
|
-
close?(): Promise<void>;
|
|
13
|
-
}
|
|
14
|
-
export interface PluginEvalRuntimeFactory {
|
|
15
|
-
create(options: {
|
|
16
|
-
dataPlane: DataPlane;
|
|
17
|
-
cwd: string;
|
|
18
|
-
configRoot: string;
|
|
19
|
-
home: string;
|
|
20
|
-
model?: string;
|
|
21
|
-
maxTurns: number;
|
|
22
|
-
pluginDirectories: readonly string[];
|
|
23
|
-
allowedTools: readonly string[];
|
|
24
|
-
appendSystemPrompt?: string;
|
|
25
|
-
historyFile?: string;
|
|
26
|
-
addDirs: readonly string[];
|
|
27
|
-
env: Readonly<Record<string, string>>;
|
|
28
|
-
eventSink(event: RuntimeEvent): void;
|
|
29
|
-
}): Promise<PluginEvalRuntime>;
|
|
30
|
-
}
|
|
3
|
+
import type { EvalRunArtifacts, EvalRuntime, EvalRuntimeFactory } from '../evals/eval-contract.js';
|
|
4
|
+
export { DEFAULT_EVAL_ALLOWED_TOOLS, resolveEvalAllowedTools, } from '../evals/eval-contract.js';
|
|
5
|
+
export type PluginEvalRuntime = EvalRuntime;
|
|
6
|
+
export type PluginEvalRuntimeFactory = EvalRuntimeFactory;
|
|
31
7
|
export interface EvalSingleRunResult {
|
|
32
8
|
text: string;
|
|
33
9
|
turns: number;
|
|
@@ -39,7 +15,6 @@ export interface EvalSingleRunResult {
|
|
|
39
15
|
termination: 'timeout' | 'interrupted' | null;
|
|
40
16
|
tempRoot?: string;
|
|
41
17
|
}
|
|
42
|
-
export declare function resolveEvalAllowedTools(requested: readonly string[], grants: readonly string[]): string[];
|
|
43
18
|
export declare function runClaudePluginEvalOnce(options: {
|
|
44
19
|
case: ClaudePluginEvalCase;
|
|
45
20
|
factory: PluginEvalRuntimeFactory;
|
|
@@ -3,50 +3,11 @@ import { tmpdir } from 'node:os';
|
|
|
3
3
|
import { join } from 'node:path';
|
|
4
4
|
import { BoundedProcessRunner, joinedProcessOutput, } from '../platform/bounded-process-runner.js';
|
|
5
5
|
import { resolveContainedPath } from './claude-plugin-eval-schema.js';
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
'Glob',
|
|
9
|
-
'Grep',
|
|
10
|
-
'Skill',
|
|
11
|
-
];
|
|
6
|
+
import { normalizeEvalTraceEvent, resolveEvalAllowedTools, } from '../evals/eval-contract.js';
|
|
7
|
+
export { DEFAULT_EVAL_ALLOWED_TOOLS, resolveEvalAllowedTools, } from '../evals/eval-contract.js';
|
|
12
8
|
function abortError() {
|
|
13
9
|
return new DOMException('Eval run timed out', 'AbortError');
|
|
14
10
|
}
|
|
15
|
-
function toolName(rule) {
|
|
16
|
-
const separator = rule.indexOf('(');
|
|
17
|
-
return separator < 0 ? rule : rule.slice(0, separator);
|
|
18
|
-
}
|
|
19
|
-
function gatedTool(rule) {
|
|
20
|
-
const name = toolName(rule);
|
|
21
|
-
return (['Bash', 'Write', 'Edit', 'NotebookEdit', 'WebFetch', 'WebSearch'].includes(name) || name.startsWith('mcp__'));
|
|
22
|
-
}
|
|
23
|
-
function operatorGrants(requested, grants) {
|
|
24
|
-
const name = toolName(requested);
|
|
25
|
-
const wildcardMatch = (pattern) => {
|
|
26
|
-
const parts = pattern.split('*');
|
|
27
|
-
if (parts.length === 1)
|
|
28
|
-
return false;
|
|
29
|
-
if (!requested.startsWith(parts[0] ?? ''))
|
|
30
|
-
return false;
|
|
31
|
-
let cursor = parts[0]?.length ?? 0;
|
|
32
|
-
for (const part of parts.slice(1, -1)) {
|
|
33
|
-
const index = requested.indexOf(part, cursor);
|
|
34
|
-
if (index < 0)
|
|
35
|
-
return false;
|
|
36
|
-
cursor = index + part.length;
|
|
37
|
-
}
|
|
38
|
-
const last = parts.at(-1) ?? '';
|
|
39
|
-
return last.length === 0 || requested.slice(cursor).endsWith(last);
|
|
40
|
-
};
|
|
41
|
-
return grants.some((grant) => grant === requested || grant === name || wildcardMatch(grant));
|
|
42
|
-
}
|
|
43
|
-
export function resolveEvalAllowedTools(requested, grants) {
|
|
44
|
-
const selected = requested.length ? requested : DEFAULT_EVAL_ALLOWED_TOOLS;
|
|
45
|
-
for (const rule of selected)
|
|
46
|
-
if (gatedTool(rule) && !operatorGrants(rule, grants))
|
|
47
|
-
throw new Error(`Eval case requests gated tool ${rule}; grant it with --allow-tools`);
|
|
48
|
-
return [...new Set(selected)];
|
|
49
|
-
}
|
|
50
11
|
async function scaffold(caseDef, cwd, home, signal) {
|
|
51
12
|
if (!caseDef.context.scaffoldScript)
|
|
52
13
|
return;
|
|
@@ -70,23 +31,7 @@ async function scaffold(caseDef, cwd, home, signal) {
|
|
|
70
31
|
if (result.code !== 0)
|
|
71
32
|
throw new Error(`Scaffold failed (${result.code}): ${joinedProcessOutput(result)}`);
|
|
72
33
|
}
|
|
73
|
-
|
|
74
|
-
if (event.type === 'tool-call')
|
|
75
|
-
return {
|
|
76
|
-
type: 'tool-call',
|
|
77
|
-
tool: event.call.name,
|
|
78
|
-
input: event.call.input,
|
|
79
|
-
id: event.call.id,
|
|
80
|
-
};
|
|
81
|
-
if (event.type === 'tool-result')
|
|
82
|
-
return {
|
|
83
|
-
type: 'tool-result',
|
|
84
|
-
callId: event.callId,
|
|
85
|
-
content: event.content,
|
|
86
|
-
isError: event.isError,
|
|
87
|
-
};
|
|
88
|
-
return event;
|
|
89
|
-
}
|
|
34
|
+
const traceEvent = normalizeEvalTraceEvent;
|
|
90
35
|
export async function runClaudePluginEvalOnce(options) {
|
|
91
36
|
const tempRoot = await mkdtemp(join(tmpdir(), 'praxis-eval-'));
|
|
92
37
|
const cwd = join(tempRoot, 'cwd');
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import type { EvalArm as SharedEvalArm, EvalDeterministicGrader, EvalFocus as SharedEvalFocus, EvalToolMatch as SharedEvalToolMatch } from '../evals/eval-contract.js';
|
|
1
2
|
export declare const EVAL_SCHEMA_VERSION = "1.0";
|
|
2
3
|
export declare const MAX_EVAL_FILE_BYTES: number;
|
|
3
4
|
export declare const MAX_EVAL_GRADERS = 256;
|
|
@@ -5,48 +6,23 @@ export declare const MAX_EVAL_ARRAY_ITEMS = 256;
|
|
|
5
6
|
export declare const MAX_EVAL_STRING_LENGTH: number;
|
|
6
7
|
export declare const MAX_EVAL_OBJECT_DEPTH = 16;
|
|
7
8
|
export declare const MAX_EVAL_OBJECT_NODES = 4096;
|
|
8
|
-
export type EvalFocus =
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
export type EvalArm = 'with-only' | 'both';
|
|
13
|
-
interface EvalGraderBase {
|
|
9
|
+
export type EvalFocus = SharedEvalFocus;
|
|
10
|
+
export type EvalArm = SharedEvalArm;
|
|
11
|
+
export type EvalToolMatch = SharedEvalToolMatch;
|
|
12
|
+
export type ClaudePluginEvalGrader = EvalDeterministicGrader | {
|
|
14
13
|
name: string;
|
|
15
14
|
weight: number;
|
|
16
15
|
arm?: EvalArm;
|
|
17
|
-
}
|
|
18
|
-
export type ClaudePluginEvalGrader = (EvalGraderBase & {
|
|
19
|
-
type: 'regex';
|
|
20
|
-
target: EvalFocus;
|
|
21
|
-
pattern: string;
|
|
22
|
-
flags: string;
|
|
23
|
-
match: 'contains' | 'not_contains' | `count:${number}`;
|
|
24
|
-
}) | (EvalGraderBase & {
|
|
25
|
-
type: 'tool_order';
|
|
26
|
-
before: EvalToolMatch;
|
|
27
|
-
after: EvalToolMatch;
|
|
28
|
-
}) | (EvalGraderBase & {
|
|
29
|
-
type: 'tool_used';
|
|
30
|
-
tool: string;
|
|
31
|
-
input_match?: Record<string, unknown>;
|
|
32
|
-
min: number;
|
|
33
|
-
max: number;
|
|
34
|
-
}) | (EvalGraderBase & {
|
|
35
|
-
type: 'file_exists';
|
|
36
|
-
path: string;
|
|
37
|
-
exists: boolean;
|
|
38
|
-
}) | (EvalGraderBase & {
|
|
39
16
|
type: 'llm';
|
|
40
17
|
criteria: string;
|
|
41
18
|
focus: EvalFocus;
|
|
42
|
-
}
|
|
19
|
+
} | {
|
|
20
|
+
name: string;
|
|
21
|
+
weight: number;
|
|
22
|
+
arm?: EvalArm;
|
|
43
23
|
type: 'baseline';
|
|
44
24
|
baseline_file: string;
|
|
45
25
|
criteria: string;
|
|
46
|
-
});
|
|
47
|
-
export type EvalToolMatch = string | {
|
|
48
|
-
tool: string;
|
|
49
|
-
input_match?: Record<string, unknown>;
|
|
50
26
|
};
|
|
51
27
|
export interface ClaudePluginEvalCase {
|
|
52
28
|
schemaVersion: string;
|
|
@@ -76,5 +52,4 @@ export interface ClaudePluginEvalCase {
|
|
|
76
52
|
}
|
|
77
53
|
export declare function loadClaudePluginEvalCase(caseDir: string): Promise<ClaudePluginEvalCase>;
|
|
78
54
|
export declare function resolveContainedPath(root: string, candidate: string, label: string): Promise<string>;
|
|
79
|
-
export {};
|
|
80
55
|
//# sourceMappingURL=claude-plugin-eval-schema.d.ts.map
|
|
@@ -4,6 +4,7 @@ import { mkdir, mkdtemp, open, realpath, stat, writeFile, } from 'node:fs/promis
|
|
|
4
4
|
import { basename, dirname, extname, isAbsolute, join, relative, resolve, } from 'node:path';
|
|
5
5
|
import { homedir, tmpdir } from 'node:os';
|
|
6
6
|
import sharp from 'sharp';
|
|
7
|
+
import { countTokens } from '@anthropic-ai/tokenizer';
|
|
7
8
|
import { commandShell, commandShellArguments, } from '../platform/command-shell.js';
|
|
8
9
|
import { BoundedProcessRunner, joinedProcessOutput, } from '../platform/bounded-process-runner.js';
|
|
9
10
|
import { globFiles } from './glob.js';
|
|
@@ -462,6 +463,8 @@ function truncateOutput(content, maxBytes) {
|
|
|
462
463
|
? `${retained.content}\n[output truncated]`
|
|
463
464
|
: retained.content;
|
|
464
465
|
}
|
|
466
|
+
const TEXT_READ_MAX_BYTES = 256 * 1024;
|
|
467
|
+
const TEXT_READ_MAX_TOKENS = 25_000;
|
|
465
468
|
function abortError() {
|
|
466
469
|
return new DOMException('Tool execution aborted', 'AbortError');
|
|
467
470
|
}
|
|
@@ -1037,8 +1040,20 @@ export class LocalToolRegistry {
|
|
|
1037
1040
|
: selected
|
|
1038
1041
|
.map((line, index) => `${(offset === 0 ? 0 : offset) + index}\t${line}`)
|
|
1039
1042
|
.join('\n');
|
|
1043
|
+
if (!notebook) {
|
|
1044
|
+
const contentBytes = Buffer.byteLength(content);
|
|
1045
|
+
if (contentBytes > TEXT_READ_MAX_BYTES) {
|
|
1046
|
+
throw new Error(`Read result is ${contentBytes} bytes, which exceeds the ${formatKilobytes(TEXT_READ_MAX_BYTES)} limit. Use offset and limit to read specific portions.`);
|
|
1047
|
+
}
|
|
1048
|
+
const contentTokens = countTokens(content);
|
|
1049
|
+
if (contentTokens > TEXT_READ_MAX_TOKENS) {
|
|
1050
|
+
throw new Error(`Read result is ${contentTokens} tokens, which exceeds the ${TEXT_READ_MAX_TOKENS} token limit. Use offset and limit to read specific portions.`);
|
|
1051
|
+
}
|
|
1052
|
+
}
|
|
1040
1053
|
return {
|
|
1041
|
-
content:
|
|
1054
|
+
content: notebook
|
|
1055
|
+
? truncateOutput(content, this.maxOutputBytes)
|
|
1056
|
+
: content,
|
|
1042
1057
|
isError: false,
|
|
1043
1058
|
accessedPaths: [filePath],
|
|
1044
1059
|
...(notebook
|