praxis-agent 0.46.4 → 0.47.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,214 @@
1
+ import { join, relative, resolve } from 'node:path';
2
+ import { writeFileAtomically } from '../platform/atomic-write.js';
3
+ import { runProjectEvalCase, } from './project-eval-runner.js';
4
+ import { discoverProjectEvalCases } from './project-eval-schema.js';
5
+ export const PROJECT_EVAL_HELP = `Usage: praxis eval [options] <target>
6
+
7
+ Run deterministic project outcome evaluations in isolated workspaces.
8
+
9
+ Options:
10
+ --case <glob> Filter case names
11
+ --tag <tag[,tag]> Filter tags; repeatable
12
+ --runs <1..50> Override run count
13
+ --model <model> Override model
14
+ --allow-tools <rules> Grant gated tools; comma-separated and repeatable
15
+ --run-verification Enable verifier subprocesses
16
+ --output-dir <dir> Write artifacts to this directory
17
+ --keep-temp Preserve temporary workspaces
18
+ --json Print exactly one aggregate JSON value
19
+ --verbose Print run progress to stderr
20
+ -h, --help Display help`;
21
+ function takeValue(argv, index, option) {
22
+ const value = argv[index + 1];
23
+ if (!value || value.startsWith('-'))
24
+ throw new Error(`${option} requires a value`);
25
+ return value;
26
+ }
27
+ function listValues(value, option) {
28
+ const values = value.split(',').map((item) => item.trim());
29
+ if (values.some((item) => item.length === 0))
30
+ throw new Error(`${option} contains an empty value`);
31
+ return values;
32
+ }
33
+ export function parseProjectEvalOptions(argv) {
34
+ const options = {
35
+ tags: [],
36
+ allowTools: [],
37
+ runVerification: false,
38
+ keepTemp: false,
39
+ json: false,
40
+ verbose: false,
41
+ };
42
+ const operands = [];
43
+ for (let index = 0; index < argv.length; index += 1) {
44
+ const value = argv[index];
45
+ if (!value)
46
+ continue;
47
+ if (value === '-h' || value === '--help')
48
+ return { ...options, help: true };
49
+ if (value === '--run-verification')
50
+ options.runVerification = true;
51
+ else if (value === '--keep-temp')
52
+ options.keepTemp = true;
53
+ else if (value === '--json')
54
+ options.json = true;
55
+ else if (value === '--verbose')
56
+ options.verbose = true;
57
+ else if (value === '--case' ||
58
+ value === '--model' ||
59
+ value === '--output-dir' ||
60
+ value === '--runs' ||
61
+ value === '--tag' ||
62
+ value === '--allow-tools') {
63
+ const selected = takeValue(argv, index, value);
64
+ index += 1;
65
+ if (value === '--case')
66
+ options.caseGlob = selected;
67
+ else if (value === '--model')
68
+ options.model = selected;
69
+ else if (value === '--output-dir')
70
+ options.outputDir = selected;
71
+ else if (value === '--runs') {
72
+ const runs = Number(selected);
73
+ if (!Number.isInteger(runs) || runs < 1 || runs > 50)
74
+ throw new Error('--runs must be an integer from 1 to 50');
75
+ options.runs = runs;
76
+ }
77
+ else if (value === '--tag')
78
+ options.tags.push(...listValues(selected, value));
79
+ else
80
+ options.allowTools.push(...listValues(selected, value));
81
+ }
82
+ else if (value.startsWith('-'))
83
+ throw new Error(`Unknown eval option: ${value}`);
84
+ else
85
+ operands.push(value);
86
+ }
87
+ if (operands.length !== 1)
88
+ throw new Error('eval requires one target');
89
+ const target = operands[0];
90
+ if (!target)
91
+ throw new Error('eval requires one target');
92
+ options.target = target;
93
+ return options;
94
+ }
95
+ function aggregateUsage(results) {
96
+ const totals = {
97
+ input_tokens: 0,
98
+ output_tokens: 0,
99
+ cache_read_input_tokens: 0,
100
+ cache_creation_input_tokens: 0,
101
+ web_search_requests: 0,
102
+ };
103
+ for (const result of results) {
104
+ if (!result.usage)
105
+ continue;
106
+ totals.input_tokens += result.usage.inputTokens;
107
+ totals.output_tokens += result.usage.outputTokens;
108
+ totals.cache_read_input_tokens += result.usage.cacheReadInputTokens ?? 0;
109
+ totals.cache_creation_input_tokens +=
110
+ result.usage.cacheCreationInputTokens ?? 0;
111
+ totals.web_search_requests += result.usage.webSearchRequests ?? 0;
112
+ }
113
+ return totals;
114
+ }
115
+ function runSummary(result, outputDirectory) {
116
+ return {
117
+ case: result.case,
118
+ run: result.run,
119
+ model: result.model,
120
+ passed: result.passed,
121
+ score: result.score,
122
+ turns: result.turns,
123
+ usage: result.usage,
124
+ cost_usd: result.cost_usd,
125
+ cost_known: result.cost_known,
126
+ duration_ms: result.duration_ms,
127
+ termination: result.termination,
128
+ error: result.error,
129
+ artifact_dir: relative(outputDirectory, join(outputDirectory, result.case, `run-${result.run}`)).replaceAll('\\', '/'),
130
+ };
131
+ }
132
+ export async function executeProjectEvalCommand(argv, io, dependencies, callerCwd = process.cwd(), signal) {
133
+ const options = parseProjectEvalOptions(argv);
134
+ if (options.help) {
135
+ io.stdout(PROJECT_EVAL_HELP);
136
+ return 0;
137
+ }
138
+ if (!options.target)
139
+ throw new Error('eval requires one target');
140
+ const target = resolve(callerCwd, options.target);
141
+ const cases = await discoverProjectEvalCases(target, options.caseGlob, options.tags);
142
+ const outputDirectory = options.outputDir
143
+ ? resolve(callerCwd, options.outputDir)
144
+ : join(dependencies.configRoot, 'evals', 'results', new Date().toISOString().replaceAll(':', '-'));
145
+ const plannedRunCount = cases.reduce((total, definition) => total + (options.runs ?? definition.runs), 0);
146
+ const results = [];
147
+ const started = Date.now();
148
+ let interrupted = signal?.aborted ?? false;
149
+ runs: for (const definition of cases) {
150
+ for (let runIndex = 1; runIndex <= (options.runs ?? definition.runs); runIndex += 1) {
151
+ if (signal?.aborted) {
152
+ interrupted = true;
153
+ break runs;
154
+ }
155
+ const result = await runProjectEvalCase({
156
+ case: definition,
157
+ factory: dependencies.runtimeFactory,
158
+ run: runIndex,
159
+ allowTools: options.allowTools,
160
+ ...(options.model === undefined ? {} : { model: options.model }),
161
+ keepTemp: options.keepTemp,
162
+ runVerification: options.runVerification,
163
+ outputDir: outputDirectory,
164
+ version: dependencies.version ?? 'unknown',
165
+ ...(signal === undefined ? {} : { signal }),
166
+ });
167
+ results.push(result);
168
+ if (options.verbose)
169
+ io.stderr(`${definition.name} run ${runIndex}: ${result.passed ? 'passed' : 'failed'}\n`);
170
+ if (signal?.aborted || result.termination === 'interrupted') {
171
+ interrupted = true;
172
+ break runs;
173
+ }
174
+ }
175
+ }
176
+ const passed = results.filter((result) => result.passed).length;
177
+ const knownCostResults = results.filter((result) => result.cost_known);
178
+ const aggregate = {
179
+ schema_version: '1.0',
180
+ version: dependencies.version ?? 'unknown',
181
+ start: new Date(started).toISOString(),
182
+ duration_ms: Date.now() - started,
183
+ target,
184
+ output_dir: outputDirectory,
185
+ model: options.model ?? null,
186
+ case_count: cases.length,
187
+ planned_run_count: plannedRunCount,
188
+ completed_run_count: results.length,
189
+ run_count: results.length,
190
+ passed,
191
+ failed: results.length - passed,
192
+ pass_rate: results.length === 0 ? 0 : passed / results.length,
193
+ total_turns: results.reduce((total, result) => total + result.turns, 0),
194
+ usage_totals: aggregateUsage(results),
195
+ usage_known_runs: results.filter((result) => result.usage !== null).length,
196
+ usage_unknown_runs: results.filter((result) => result.usage === null)
197
+ .length,
198
+ known_cost_total_usd: knownCostResults.length === 0
199
+ ? null
200
+ : knownCostResults.reduce((total, result) => total + (result.cost_usd ?? 0), 0),
201
+ known_cost_runs: knownCostResults.length,
202
+ unknown_cost_runs: results.length - knownCostResults.length,
203
+ partial: interrupted || results.length < plannedRunCount,
204
+ interrupted,
205
+ runs: results.map((result) => runSummary(result, outputDirectory)),
206
+ };
207
+ await writeFileAtomically(join(outputDirectory, 'aggregate-result.json'), JSON.stringify(aggregate, null, 2));
208
+ if (options.json)
209
+ io.stdout(`${JSON.stringify(aggregate)}\n`);
210
+ else
211
+ io.stdout(`${passed}/${results.length} passed\n`);
212
+ return interrupted ? 130 : passed === results.length ? 0 : 1;
213
+ }
214
+ //# sourceMappingURL=project-eval.js.map
@@ -19,6 +19,8 @@ export interface RunProcessOptions {
19
19
  signal?: AbortSignal;
20
20
  onOutput?: (output: string) => void | Promise<void>;
21
21
  env?: Readonly<Record<string, string>>;
22
+ inheritEnvironment?: boolean;
23
+ redactExplicitEnvironment?: boolean;
22
24
  controlOutputBytes?: number;
23
25
  controlOutputFd?: number;
24
26
  scriptInput?: string;
@@ -84,7 +84,7 @@ export class BoundedProcessRunner {
84
84
  if (options.signal?.aborted)
85
85
  return Promise.reject(abortError());
86
86
  return new Promise((resolve, reject) => {
87
- const sensitiveValues = sensitiveEnvironmentValues(process.env);
87
+ const sensitiveValues = sensitiveEnvironmentValues(process.env, ...(options.redactExplicitEnvironment ? [options.env ?? {}] : []));
88
88
  const longestSensitiveValueBytes = sensitiveValues.reduce((longest, value) => Math.max(longest, Buffer.byteLength(value)), 0);
89
89
  const rawOutputLimit = this.options.maxOutputBytes + Math.max(3, longestSensitiveValueBytes);
90
90
  const controlOutputFd = options.controlOutputBytes === undefined
@@ -108,7 +108,9 @@ export class BoundedProcessRunner {
108
108
  const child = spawn(options.command, options.args, {
109
109
  cwd: options.cwd ?? this.options.cwd,
110
110
  detached: process.platform !== 'win32',
111
- env: { ...sanitizeChildEnvironment(), ...options.env },
111
+ env: options.inheritEnvironment === false
112
+ ? sanitizeChildEnvironment(options.env ?? {}, {})
113
+ : sanitizeChildEnvironment(options.env ?? {}),
112
114
  stdio,
113
115
  });
114
116
  const chunks = { stdout: [], stderr: [] };
@@ -1,37 +1,6 @@
1
1
  import type { ClaudePluginEvalCase } from './claude-plugin-eval-schema.js';
2
- export interface EvalTraceEvent {
3
- type: string;
4
- tool?: string;
5
- input?: Record<string, unknown>;
6
- [key: string]: unknown;
7
- }
8
- export interface EvalRunArtifacts {
9
- lastMessage: string;
10
- trace: readonly EvalTraceEvent[];
11
- cwd: string;
12
- }
13
- export interface EvalGraderResult {
14
- name: string;
15
- passed: boolean;
16
- weight: number;
17
- explanation: string;
18
- judge_votes?: readonly boolean[];
19
- evidence?: string;
20
- with_only?: boolean;
21
- }
22
- export interface EvalJudge {
23
- vote(request: {
24
- criteria: string;
25
- focus: string;
26
- baseline?: string;
27
- model: string;
28
- signal?: AbortSignal;
29
- }): Promise<{
30
- passed: boolean;
31
- explanation?: string;
32
- costUsd: number;
33
- }>;
34
- }
2
+ export type { EvalTraceEvent, EvalRunArtifacts, EvalGraderResult, EvalJudge, } from '../evals/eval-contract.js';
3
+ import type { EvalRunArtifacts, EvalGraderResult, EvalJudge } from '../evals/eval-contract.js';
35
4
  export declare function gradeClaudePluginEvalRun(options: {
36
5
  case: ClaudePluginEvalCase;
37
6
  artifacts: EvalRunArtifacts;
@@ -1,20 +1,6 @@
1
1
  import { readFile } from 'node:fs/promises';
2
- import { minimatch } from 'minimatch';
3
2
  import { resolveContainedPath } from './claude-plugin-eval-schema.js';
4
- function subset(actual, expected) {
5
- if (expected && typeof expected === 'object' && !Array.isArray(expected)) {
6
- if (!actual || typeof actual !== 'object' || Array.isArray(actual))
7
- return false;
8
- return Object.entries(expected).every(([key, value]) => subset(actual[key], value));
9
- }
10
- return Object.is(actual, expected);
11
- }
12
- function matches(event, match) {
13
- const selected = typeof match === 'string' ? { tool: match } : match;
14
- return (event.type === 'tool-call' &&
15
- event.tool === selected.tool &&
16
- (!selected.input_match || subset(event.input, selected.input_match)));
17
- }
3
+ import { gradeDeterministicEvalRun } from '../evals/eval-graders.js';
18
4
  async function focusText(focus, artifacts) {
19
5
  if (focus === 'last_message')
20
6
  return artifacts.lastMessage;
@@ -32,56 +18,6 @@ async function focusText(focus, artifacts) {
32
18
  }
33
19
  return readFile(await resolveContainedPath(artifacts.cwd, focus.path, 'Grader focus'), 'utf8');
34
20
  }
35
- async function freeGrade(grader, artifacts) {
36
- let passed = false;
37
- let evidence = '';
38
- if (grader.type === 'regex') {
39
- const target = await focusText(grader.target, artifacts);
40
- const expression = new RegExp(grader.pattern, grader.flags.includes('g') ? grader.flags : `${grader.flags}g`);
41
- const count = [...target.matchAll(expression)].length;
42
- passed =
43
- grader.match === 'contains'
44
- ? count > 0
45
- : grader.match === 'not_contains'
46
- ? count === 0
47
- : count === Number(grader.match.slice(6));
48
- evidence = `matches=${count}`;
49
- }
50
- else if (grader.type === 'tool_used') {
51
- const count = artifacts.trace.filter((event) => matches(event, {
52
- tool: grader.tool,
53
- ...(grader.input_match ? { input_match: grader.input_match } : {}),
54
- })).length;
55
- passed = count >= grader.min && count <= grader.max;
56
- evidence = `uses=${count}`;
57
- }
58
- else if (grader.type === 'tool_order') {
59
- const before = artifacts.trace.findIndex((event) => matches(event, grader.before));
60
- const after = artifacts.trace.findIndex((event, index) => index > before && matches(event, grader.after));
61
- passed = before >= 0 && after > before;
62
- evidence = `before=${before},after=${after}`;
63
- }
64
- else {
65
- const { glob } = await import('node:fs/promises');
66
- let count = 0;
67
- for await (const path of glob('**/*', { cwd: artifacts.cwd })) {
68
- if (!minimatch(path, grader.path))
69
- continue;
70
- await resolveContainedPath(artifacts.cwd, path, 'file_exists match');
71
- count += 1;
72
- }
73
- passed = grader.exists ? count > 0 : count === 0;
74
- evidence = `files=${count}`;
75
- }
76
- return {
77
- name: grader.name,
78
- passed,
79
- weight: grader.weight,
80
- explanation: passed ? 'passed' : 'failed',
81
- evidence,
82
- ...(grader.arm === 'with-only' ? { with_only: true } : {}),
83
- };
84
- }
85
21
  export async function gradeClaudePluginEvalRun(options) {
86
22
  const results = [];
87
23
  let skippedPaidGraders = false;
@@ -90,7 +26,12 @@ export async function gradeClaudePluginEvalRun(options) {
90
26
  if (options.arm === 'without' && item.arm === 'with-only')
91
27
  continue;
92
28
  if (item.type !== 'llm' && item.type !== 'baseline') {
93
- results.push(await freeGrade(item, options.artifacts));
29
+ const [result] = await gradeDeterministicEvalRun({
30
+ graders: [item],
31
+ artifacts: options.artifacts,
32
+ });
33
+ if (result)
34
+ results.push(result);
94
35
  continue;
95
36
  }
96
37
  if (options.skipPaid) {
@@ -1,33 +1,9 @@
1
- import type { RuntimeEvent } from '../core/runtime.js';
2
1
  import type { DataPlane } from '../persistence/data-plane.js';
3
2
  import type { ClaudePluginEvalCase } from './claude-plugin-eval-schema.js';
4
- import type { EvalRunArtifacts } from './claude-plugin-eval-graders.js';
5
- export declare const DEFAULT_EVAL_ALLOWED_TOOLS: readonly ["Read", "Glob", "Grep", "Skill"];
6
- export interface PluginEvalRuntime {
7
- run(prompt: string, signal: AbortSignal): Promise<{
8
- text: string;
9
- turns: number;
10
- costUsd?: number;
11
- }>;
12
- close?(): Promise<void>;
13
- }
14
- export interface PluginEvalRuntimeFactory {
15
- create(options: {
16
- dataPlane: DataPlane;
17
- cwd: string;
18
- configRoot: string;
19
- home: string;
20
- model?: string;
21
- maxTurns: number;
22
- pluginDirectories: readonly string[];
23
- allowedTools: readonly string[];
24
- appendSystemPrompt?: string;
25
- historyFile?: string;
26
- addDirs: readonly string[];
27
- env: Readonly<Record<string, string>>;
28
- eventSink(event: RuntimeEvent): void;
29
- }): Promise<PluginEvalRuntime>;
30
- }
3
+ import type { EvalRunArtifacts, EvalRuntime, EvalRuntimeFactory } from '../evals/eval-contract.js';
4
+ export { DEFAULT_EVAL_ALLOWED_TOOLS, resolveEvalAllowedTools, } from '../evals/eval-contract.js';
5
+ export type PluginEvalRuntime = EvalRuntime;
6
+ export type PluginEvalRuntimeFactory = EvalRuntimeFactory;
31
7
  export interface EvalSingleRunResult {
32
8
  text: string;
33
9
  turns: number;
@@ -39,7 +15,6 @@ export interface EvalSingleRunResult {
39
15
  termination: 'timeout' | 'interrupted' | null;
40
16
  tempRoot?: string;
41
17
  }
42
- export declare function resolveEvalAllowedTools(requested: readonly string[], grants: readonly string[]): string[];
43
18
  export declare function runClaudePluginEvalOnce(options: {
44
19
  case: ClaudePluginEvalCase;
45
20
  factory: PluginEvalRuntimeFactory;
@@ -3,50 +3,11 @@ import { tmpdir } from 'node:os';
3
3
  import { join } from 'node:path';
4
4
  import { BoundedProcessRunner, joinedProcessOutput, } from '../platform/bounded-process-runner.js';
5
5
  import { resolveContainedPath } from './claude-plugin-eval-schema.js';
6
- export const DEFAULT_EVAL_ALLOWED_TOOLS = [
7
- 'Read',
8
- 'Glob',
9
- 'Grep',
10
- 'Skill',
11
- ];
6
+ import { normalizeEvalTraceEvent, resolveEvalAllowedTools, } from '../evals/eval-contract.js';
7
+ export { DEFAULT_EVAL_ALLOWED_TOOLS, resolveEvalAllowedTools, } from '../evals/eval-contract.js';
12
8
  function abortError() {
13
9
  return new DOMException('Eval run timed out', 'AbortError');
14
10
  }
15
- function toolName(rule) {
16
- const separator = rule.indexOf('(');
17
- return separator < 0 ? rule : rule.slice(0, separator);
18
- }
19
- function gatedTool(rule) {
20
- const name = toolName(rule);
21
- return (['Bash', 'Write', 'Edit', 'NotebookEdit', 'WebFetch', 'WebSearch'].includes(name) || name.startsWith('mcp__'));
22
- }
23
- function operatorGrants(requested, grants) {
24
- const name = toolName(requested);
25
- const wildcardMatch = (pattern) => {
26
- const parts = pattern.split('*');
27
- if (parts.length === 1)
28
- return false;
29
- if (!requested.startsWith(parts[0] ?? ''))
30
- return false;
31
- let cursor = parts[0]?.length ?? 0;
32
- for (const part of parts.slice(1, -1)) {
33
- const index = requested.indexOf(part, cursor);
34
- if (index < 0)
35
- return false;
36
- cursor = index + part.length;
37
- }
38
- const last = parts.at(-1) ?? '';
39
- return last.length === 0 || requested.slice(cursor).endsWith(last);
40
- };
41
- return grants.some((grant) => grant === requested || grant === name || wildcardMatch(grant));
42
- }
43
- export function resolveEvalAllowedTools(requested, grants) {
44
- const selected = requested.length ? requested : DEFAULT_EVAL_ALLOWED_TOOLS;
45
- for (const rule of selected)
46
- if (gatedTool(rule) && !operatorGrants(rule, grants))
47
- throw new Error(`Eval case requests gated tool ${rule}; grant it with --allow-tools`);
48
- return [...new Set(selected)];
49
- }
50
11
  async function scaffold(caseDef, cwd, home, signal) {
51
12
  if (!caseDef.context.scaffoldScript)
52
13
  return;
@@ -70,23 +31,7 @@ async function scaffold(caseDef, cwd, home, signal) {
70
31
  if (result.code !== 0)
71
32
  throw new Error(`Scaffold failed (${result.code}): ${joinedProcessOutput(result)}`);
72
33
  }
73
- function traceEvent(event) {
74
- if (event.type === 'tool-call')
75
- return {
76
- type: 'tool-call',
77
- tool: event.call.name,
78
- input: event.call.input,
79
- id: event.call.id,
80
- };
81
- if (event.type === 'tool-result')
82
- return {
83
- type: 'tool-result',
84
- callId: event.callId,
85
- content: event.content,
86
- isError: event.isError,
87
- };
88
- return event;
89
- }
34
+ const traceEvent = normalizeEvalTraceEvent;
90
35
  export async function runClaudePluginEvalOnce(options) {
91
36
  const tempRoot = await mkdtemp(join(tmpdir(), 'praxis-eval-'));
92
37
  const cwd = join(tempRoot, 'cwd');
@@ -1,3 +1,4 @@
1
+ import type { EvalArm as SharedEvalArm, EvalDeterministicGrader, EvalFocus as SharedEvalFocus, EvalToolMatch as SharedEvalToolMatch } from '../evals/eval-contract.js';
1
2
  export declare const EVAL_SCHEMA_VERSION = "1.0";
2
3
  export declare const MAX_EVAL_FILE_BYTES: number;
3
4
  export declare const MAX_EVAL_GRADERS = 256;
@@ -5,48 +6,23 @@ export declare const MAX_EVAL_ARRAY_ITEMS = 256;
5
6
  export declare const MAX_EVAL_STRING_LENGTH: number;
6
7
  export declare const MAX_EVAL_OBJECT_DEPTH = 16;
7
8
  export declare const MAX_EVAL_OBJECT_NODES = 4096;
8
- export type EvalFocus = 'last_message' | 'trace' | 'files' | {
9
- source: 'file';
10
- path: string;
11
- };
12
- export type EvalArm = 'with-only' | 'both';
13
- interface EvalGraderBase {
9
+ export type EvalFocus = SharedEvalFocus;
10
+ export type EvalArm = SharedEvalArm;
11
+ export type EvalToolMatch = SharedEvalToolMatch;
12
+ export type ClaudePluginEvalGrader = EvalDeterministicGrader | {
14
13
  name: string;
15
14
  weight: number;
16
15
  arm?: EvalArm;
17
- }
18
- export type ClaudePluginEvalGrader = (EvalGraderBase & {
19
- type: 'regex';
20
- target: EvalFocus;
21
- pattern: string;
22
- flags: string;
23
- match: 'contains' | 'not_contains' | `count:${number}`;
24
- }) | (EvalGraderBase & {
25
- type: 'tool_order';
26
- before: EvalToolMatch;
27
- after: EvalToolMatch;
28
- }) | (EvalGraderBase & {
29
- type: 'tool_used';
30
- tool: string;
31
- input_match?: Record<string, unknown>;
32
- min: number;
33
- max: number;
34
- }) | (EvalGraderBase & {
35
- type: 'file_exists';
36
- path: string;
37
- exists: boolean;
38
- }) | (EvalGraderBase & {
39
16
  type: 'llm';
40
17
  criteria: string;
41
18
  focus: EvalFocus;
42
- }) | (EvalGraderBase & {
19
+ } | {
20
+ name: string;
21
+ weight: number;
22
+ arm?: EvalArm;
43
23
  type: 'baseline';
44
24
  baseline_file: string;
45
25
  criteria: string;
46
- });
47
- export type EvalToolMatch = string | {
48
- tool: string;
49
- input_match?: Record<string, unknown>;
50
26
  };
51
27
  export interface ClaudePluginEvalCase {
52
28
  schemaVersion: string;
@@ -76,5 +52,4 @@ export interface ClaudePluginEvalCase {
76
52
  }
77
53
  export declare function loadClaudePluginEvalCase(caseDir: string): Promise<ClaudePluginEvalCase>;
78
54
  export declare function resolveContainedPath(root: string, candidate: string, label: string): Promise<string>;
79
- export {};
80
55
  //# sourceMappingURL=claude-plugin-eval-schema.d.ts.map
@@ -4,6 +4,7 @@ import { mkdir, mkdtemp, open, realpath, stat, writeFile, } from 'node:fs/promis
4
4
  import { basename, dirname, extname, isAbsolute, join, relative, resolve, } from 'node:path';
5
5
  import { homedir, tmpdir } from 'node:os';
6
6
  import sharp from 'sharp';
7
+ import { countTokens } from '@anthropic-ai/tokenizer';
7
8
  import { commandShell, commandShellArguments, } from '../platform/command-shell.js';
8
9
  import { BoundedProcessRunner, joinedProcessOutput, } from '../platform/bounded-process-runner.js';
9
10
  import { globFiles } from './glob.js';
@@ -462,6 +463,8 @@ function truncateOutput(content, maxBytes) {
462
463
  ? `${retained.content}\n[output truncated]`
463
464
  : retained.content;
464
465
  }
466
+ const TEXT_READ_MAX_BYTES = 256 * 1024;
467
+ const TEXT_READ_MAX_TOKENS = 25_000;
465
468
  function abortError() {
466
469
  return new DOMException('Tool execution aborted', 'AbortError');
467
470
  }
@@ -1037,8 +1040,20 @@ export class LocalToolRegistry {
1037
1040
  : selected
1038
1041
  .map((line, index) => `${(offset === 0 ? 0 : offset) + index}\t${line}`)
1039
1042
  .join('\n');
1043
+ if (!notebook) {
1044
+ const contentBytes = Buffer.byteLength(content);
1045
+ if (contentBytes > TEXT_READ_MAX_BYTES) {
1046
+ throw new Error(`Read result is ${contentBytes} bytes, which exceeds the ${formatKilobytes(TEXT_READ_MAX_BYTES)} limit. Use offset and limit to read specific portions.`);
1047
+ }
1048
+ const contentTokens = countTokens(content);
1049
+ if (contentTokens > TEXT_READ_MAX_TOKENS) {
1050
+ throw new Error(`Read result is ${contentTokens} tokens, which exceeds the ${TEXT_READ_MAX_TOKENS} token limit. Use offset and limit to read specific portions.`);
1051
+ }
1052
+ }
1040
1053
  return {
1041
- content: truncateOutput(content, this.maxOutputBytes),
1054
+ content: notebook
1055
+ ? truncateOutput(content, this.maxOutputBytes)
1056
+ : content,
1042
1057
  isError: false,
1043
1058
  accessedPaths: [filePath],
1044
1059
  ...(notebook
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "praxis-agent",
3
- "version": "0.46.4",
3
+ "version": "0.47.0",
4
4
  "description": "Local-first, single-user general agent for the command line.",
5
5
  "license": "MIT",
6
6
  "author": "wuqisen",