@nexrall/code-core 1.4.74 → 1.4.76
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -3
- package/dist/agent/askOnce.d.ts +10 -0
- package/dist/agent/askOnce.d.ts.map +1 -1
- package/dist/agent/askOnce.js +10 -2
- package/dist/agent/planMode.d.ts.map +1 -1
- package/dist/agent/planMode.js +16 -3
- package/dist/index.d.ts +2 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +2 -0
- package/dist/permissions/autoMode.d.ts +89 -0
- package/dist/permissions/autoMode.d.ts.map +1 -0
- package/dist/permissions/autoMode.js +346 -0
- package/dist/permissions/rules.d.ts +6 -0
- package/dist/permissions/rules.d.ts.map +1 -1
- package/dist/permissions/rules.js +33 -0
- package/dist/plugins/eval.d.ts +199 -0
- package/dist/plugins/eval.d.ts.map +1 -0
- package/dist/plugins/eval.js +470 -0
- package/package.json +1 -1
|
@@ -0,0 +1,470 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
// S21 — `nex plugin eval`: run a plugin against a suite of test cases and score
|
|
3
|
+
// the results (Claude Code's `claude plugin eval`, shipped in their W37).
|
|
4
|
+
//
|
|
5
|
+
// This file is the PURE half: case discovery/parsing, the grader vocabulary,
|
|
6
|
+
// the trace-based graders, and the scoring/aggregation. The caller owns ALL
|
|
7
|
+
// I/O — spawning each run, reading the child's event log, and judging (the
|
|
8
|
+
// `llm`/`baseline` graders call a model through an injected `judge` function so
|
|
9
|
+
// this module stays unit-testable without a backend).
|
|
10
|
+
//
|
|
11
|
+
// Suite layout (CC's, minus what we do not support yet — see below):
|
|
12
|
+
//
|
|
13
|
+
// my-plugin/
|
|
14
|
+
// evals/
|
|
15
|
+
// first-case/
|
|
16
|
+
// prompt.md # frontmatter = case fields; body = the prompt
|
|
17
|
+
// graders/
|
|
18
|
+
// criteria.md # frontmatter = grader; body = llm criteria
|
|
19
|
+
// skill-fired.md
|
|
20
|
+
// results/ # written by each run; never discovered as a case
|
|
21
|
+
//
|
|
22
|
+
// Deliberately NOT supported yet (refused loudly, never silently skipped):
|
|
23
|
+
// · case.yaml — needs a real YAML parser, which this repo does not carry
|
|
24
|
+
// (the frontmatter reader is "not a full YAML parser" by design). A case
|
|
25
|
+
// that ships one is an ERROR, because silently ignoring its fields would
|
|
26
|
+
// run a DIFFERENT suite than the author wrote.
|
|
27
|
+
// · mock MCP servers (`evals/mocks/`) — a follow-up; a case using them is
|
|
28
|
+
// likewise refused rather than run without the mocks.
|
|
29
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
30
|
+
if (k2 === undefined) k2 = k;
|
|
31
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
32
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
33
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
34
|
+
}
|
|
35
|
+
Object.defineProperty(o, k2, desc);
|
|
36
|
+
}) : (function(o, m, k, k2) {
|
|
37
|
+
if (k2 === undefined) k2 = k;
|
|
38
|
+
o[k2] = m[k];
|
|
39
|
+
}));
|
|
40
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
41
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
42
|
+
}) : function(o, v) {
|
|
43
|
+
o["default"] = v;
|
|
44
|
+
});
|
|
45
|
+
var __importStar = (this && this.__importStar) || (function () {
|
|
46
|
+
var ownKeys = function(o) {
|
|
47
|
+
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
48
|
+
var ar = [];
|
|
49
|
+
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
50
|
+
return ar;
|
|
51
|
+
};
|
|
52
|
+
return ownKeys(o);
|
|
53
|
+
};
|
|
54
|
+
return function (mod) {
|
|
55
|
+
if (mod && mod.__esModule) return mod;
|
|
56
|
+
var result = {};
|
|
57
|
+
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
58
|
+
__setModuleDefault(result, mod);
|
|
59
|
+
return result;
|
|
60
|
+
};
|
|
61
|
+
})();
|
|
62
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
63
|
+
exports.EVAL_MAX_TIMEOUT_S = exports.EVAL_MAX_TURNS_CEILING = exports.EVAL_DEFAULT_TIMEOUT_S = exports.EVAL_DEFAULT_MAX_TURNS = exports.EVAL_DEFAULT_THRESHOLD = exports.EVAL_MAX_RUNS = exports.EVAL_DEFAULT_RUNS = exports.EVAL_SCHEMA_VERSION = void 0;
|
|
64
|
+
exports.parseList = parseList;
|
|
65
|
+
exports.parseGraderFile = parseGraderFile;
|
|
66
|
+
exports.parseEvalCase = parseEvalCase;
|
|
67
|
+
exports.discoverEvalCases = discoverEvalCases;
|
|
68
|
+
exports.targetText = targetText;
|
|
69
|
+
exports.evaluateGrader = evaluateGrader;
|
|
70
|
+
exports.gradersForArm = gradersForArm;
|
|
71
|
+
exports.scoreRun = scoreRun;
|
|
72
|
+
exports.aggregateReport = aggregateReport;
|
|
73
|
+
const fs = __importStar(require("fs"));
|
|
74
|
+
const path = __importStar(require("path"));
|
|
75
|
+
const frontmatter_1 = require("../util/frontmatter");
|
|
76
|
+
const nestedInstructions_1 = require("../agent/nestedInstructions");
|
|
77
|
+
exports.EVAL_SCHEMA_VERSION = 1;
|
|
78
|
+
/** CC's default: each case runs three times. */
|
|
79
|
+
exports.EVAL_DEFAULT_RUNS = 3;
|
|
80
|
+
exports.EVAL_MAX_RUNS = 50;
|
|
81
|
+
exports.EVAL_DEFAULT_THRESHOLD = 1.0;
|
|
82
|
+
exports.EVAL_DEFAULT_MAX_TURNS = 10;
|
|
83
|
+
exports.EVAL_DEFAULT_TIMEOUT_S = 300;
|
|
84
|
+
/** CC caps both; larger values are refused, not clamped silently. */
|
|
85
|
+
exports.EVAL_MAX_TURNS_CEILING = 200;
|
|
86
|
+
exports.EVAL_MAX_TIMEOUT_S = 3600;
|
|
87
|
+
/** `match: count:N` in a regex grader. */
|
|
88
|
+
const COUNT_RE = /^count:(\d+)$/;
|
|
89
|
+
// ─── Parsing helpers ─────────────────────────────────────────────────────────
|
|
90
|
+
/** `[a, b]` / `a, b` / `a` → ['a','b'] — the two shapes real files use. */
|
|
91
|
+
function parseList(value) {
|
|
92
|
+
if (value === undefined)
|
|
93
|
+
return [];
|
|
94
|
+
const inner = value.trim().replace(/^\[|\]$/g, '');
|
|
95
|
+
return inner
|
|
96
|
+
.split(',')
|
|
97
|
+
.map((s) => s.trim().replace(/^["']|["']$/g, ''))
|
|
98
|
+
.filter(Boolean);
|
|
99
|
+
}
|
|
100
|
+
function parsePositiveInt(value, dflt, ceiling, what) {
|
|
101
|
+
if (value === undefined || value.trim() === '')
|
|
102
|
+
return dflt;
|
|
103
|
+
const n = Number(value.trim());
|
|
104
|
+
if (!Number.isInteger(n) || n < 1)
|
|
105
|
+
throw new Error(`${what} must be a positive integer (got "${value}")`);
|
|
106
|
+
if (n > ceiling)
|
|
107
|
+
throw new Error(`${what} ${n} exceeds the ceiling of ${ceiling}`);
|
|
108
|
+
return n;
|
|
109
|
+
}
|
|
110
|
+
/**
|
|
111
|
+
* One grader file → EvalGrader. Throws on anything that would make the grader
|
|
112
|
+
* silently unscorable (unknown type, missing pattern/tool/path, bad count).
|
|
113
|
+
*/
|
|
114
|
+
function parseGraderFile(file, raw) {
|
|
115
|
+
const { meta, body } = (0, frontmatter_1.parseFrontmatter)(raw);
|
|
116
|
+
const type = (meta.type ?? '').trim();
|
|
117
|
+
if (!['regex', 'tool_used', 'tool_order', 'file_exists', 'llm', 'baseline'].includes(type)) {
|
|
118
|
+
throw new Error(`grader "${file}": type must be regex|tool_used|tool_order|file_exists|llm|baseline (got "${meta.type ?? ''}")`);
|
|
119
|
+
}
|
|
120
|
+
const weightRaw = (meta.weight ?? '1').trim();
|
|
121
|
+
const weight = Number(weightRaw);
|
|
122
|
+
if (!Number.isFinite(weight) || weight <= 0)
|
|
123
|
+
throw new Error(`grader "${file}": weight must be a positive number`);
|
|
124
|
+
const arm = (meta.arm ?? 'both').trim() === 'with-only' ? 'with-only' : 'both';
|
|
125
|
+
if (meta.arm !== undefined && !['both', 'with-only'].includes(meta.arm.trim())) {
|
|
126
|
+
throw new Error(`grader "${file}": arm must be both or with-only (got "${meta.arm}")`);
|
|
127
|
+
}
|
|
128
|
+
const g = { file, type, weight, arm };
|
|
129
|
+
if (type === 'regex') {
|
|
130
|
+
if (!meta.pattern)
|
|
131
|
+
throw new Error(`grader "${file}": regex grader needs a \`pattern\``);
|
|
132
|
+
g.pattern = meta.pattern;
|
|
133
|
+
g.flags = meta.flags?.trim() || '';
|
|
134
|
+
const match = (meta.match ?? 'contains').trim();
|
|
135
|
+
if (!(match === 'contains' || match === 'not_contains' || COUNT_RE.test(match))) {
|
|
136
|
+
throw new Error(`grader "${file}": match must be contains | not_contains | count:<n> (got "${match}")`);
|
|
137
|
+
}
|
|
138
|
+
g.match = match;
|
|
139
|
+
const target = (meta.target ?? 'last_message').trim();
|
|
140
|
+
if (!['last_message', 'trace', 'file'].includes(target)) {
|
|
141
|
+
throw new Error(`grader "${file}": target must be last_message | trace | file (got "${target}")`);
|
|
142
|
+
}
|
|
143
|
+
g.target = target;
|
|
144
|
+
if (target === 'file') {
|
|
145
|
+
if (!meta.path)
|
|
146
|
+
throw new Error(`grader "${file}": target "file" needs a \`path\``);
|
|
147
|
+
g.pathPattern = meta.path;
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
else if (type === 'tool_used') {
|
|
151
|
+
if (!meta.tool)
|
|
152
|
+
throw new Error(`grader "${file}": tool_used grader needs a \`tool\``);
|
|
153
|
+
g.tool = meta.tool.trim();
|
|
154
|
+
g.inputMatch = meta.input_match?.trim() || undefined;
|
|
155
|
+
g.min = parsePositiveInt(meta.min, 1, 10000, 'min');
|
|
156
|
+
g.max = meta.max !== undefined && meta.max.trim() !== '' ? parsePositiveInt(meta.max, 1, 10000, 'max') : undefined;
|
|
157
|
+
if (g.max !== undefined && g.max < (g.min ?? 1))
|
|
158
|
+
throw new Error(`grader "${file}": max < min`);
|
|
159
|
+
}
|
|
160
|
+
else if (type === 'tool_order') {
|
|
161
|
+
if (!meta.before || !meta.after)
|
|
162
|
+
throw new Error(`grader "${file}": tool_order needs \`before\` and \`after\``);
|
|
163
|
+
g.before = meta.before.trim();
|
|
164
|
+
g.after = meta.after.trim();
|
|
165
|
+
}
|
|
166
|
+
else if (type === 'file_exists') {
|
|
167
|
+
if (!meta.path)
|
|
168
|
+
throw new Error(`grader "${file}": file_exists needs a \`path\``);
|
|
169
|
+
g.path = meta.path.trim();
|
|
170
|
+
const ex = (meta.exists ?? 'true').trim().toLowerCase();
|
|
171
|
+
g.exists = !(ex === 'false' || ex === '0' || ex === 'no');
|
|
172
|
+
}
|
|
173
|
+
else {
|
|
174
|
+
// llm / baseline: criteria from the body (the file IS the criteria text).
|
|
175
|
+
const criteria = body.trim();
|
|
176
|
+
if (!criteria)
|
|
177
|
+
throw new Error(`grader "${file}": ${type} grader needs criteria in the file body`);
|
|
178
|
+
g.criteria = criteria;
|
|
179
|
+
if (type === 'llm') {
|
|
180
|
+
const focus = (meta.focus ?? 'last_message').trim();
|
|
181
|
+
if (!['last_message', 'trace'].includes(focus)) {
|
|
182
|
+
throw new Error(`grader "${file}": focus must be last_message | trace (got "${focus}")`);
|
|
183
|
+
}
|
|
184
|
+
g.focus = focus;
|
|
185
|
+
}
|
|
186
|
+
else {
|
|
187
|
+
if (!meta.baseline_file)
|
|
188
|
+
throw new Error(`grader "${file}": baseline grader needs \`baseline_file\``);
|
|
189
|
+
g.baselineFile = meta.baseline_file.trim();
|
|
190
|
+
}
|
|
191
|
+
}
|
|
192
|
+
return g;
|
|
193
|
+
}
|
|
194
|
+
/** One case directory → EvalCase. See the header for the layout. */
|
|
195
|
+
function parseEvalCase(dir) {
|
|
196
|
+
const promptFile = path.join(dir, 'prompt.md');
|
|
197
|
+
const caseYaml = path.join(dir, 'case.yaml');
|
|
198
|
+
if (fs.existsSync(caseYaml)) {
|
|
199
|
+
throw new Error('case.yaml is not supported yet (this repo carries no YAML parser) — move its fields into prompt.md frontmatter');
|
|
200
|
+
}
|
|
201
|
+
if (fs.existsSync(path.join(dir, 'mocks'))) {
|
|
202
|
+
throw new Error('mock MCP servers (mocks/) are not supported yet');
|
|
203
|
+
}
|
|
204
|
+
let raw;
|
|
205
|
+
try {
|
|
206
|
+
raw = fs.readFileSync(promptFile, 'utf-8');
|
|
207
|
+
}
|
|
208
|
+
catch {
|
|
209
|
+
throw new Error('no prompt.md (a case needs one; case.yaml alone is not supported yet)');
|
|
210
|
+
}
|
|
211
|
+
const { meta, body } = (0, frontmatter_1.parseFrontmatter)(raw);
|
|
212
|
+
const prompt = body.trim();
|
|
213
|
+
if (!prompt)
|
|
214
|
+
throw new Error('prompt.md has an empty prompt body');
|
|
215
|
+
const gradersDir = path.join(dir, 'graders');
|
|
216
|
+
let graderFiles;
|
|
217
|
+
try {
|
|
218
|
+
graderFiles = fs
|
|
219
|
+
.readdirSync(gradersDir)
|
|
220
|
+
.filter((f) => f.endsWith('.md'))
|
|
221
|
+
.sort();
|
|
222
|
+
}
|
|
223
|
+
catch {
|
|
224
|
+
throw new Error('no graders/ directory (at least one grader is required)');
|
|
225
|
+
}
|
|
226
|
+
if (!graderFiles.length)
|
|
227
|
+
throw new Error('graders/ holds no .md files (at least one grader is required)');
|
|
228
|
+
const graders = graderFiles.map((f) => parseGraderFile(f, fs.readFileSync(path.join(gradersDir, f), 'utf-8')));
|
|
229
|
+
return {
|
|
230
|
+
name: meta.name?.trim() || path.basename(dir),
|
|
231
|
+
dir,
|
|
232
|
+
prompt,
|
|
233
|
+
runs: parsePositiveInt(meta.runs, exports.EVAL_DEFAULT_RUNS, exports.EVAL_MAX_RUNS, 'runs'),
|
|
234
|
+
maxTurns: parsePositiveInt(meta.max_turns, exports.EVAL_DEFAULT_MAX_TURNS, exports.EVAL_MAX_TURNS_CEILING, 'max_turns'),
|
|
235
|
+
timeoutSeconds: parsePositiveInt(meta.timeout_seconds, exports.EVAL_DEFAULT_TIMEOUT_S, exports.EVAL_MAX_TIMEOUT_S, 'timeout_seconds'),
|
|
236
|
+
allowedTools: parseList(meta.allowed_tools),
|
|
237
|
+
tags: parseList(meta.tags),
|
|
238
|
+
expectedOutcome: meta.expected_outcome?.trim() || undefined,
|
|
239
|
+
graders,
|
|
240
|
+
};
|
|
241
|
+
}
|
|
242
|
+
/** Every case under `<pluginDir>/evals` (the results dir is never a case). */
|
|
243
|
+
function discoverEvalCases(pluginDir) {
|
|
244
|
+
const evalsDir = path.join(pluginDir, 'evals');
|
|
245
|
+
const errors = [];
|
|
246
|
+
let names;
|
|
247
|
+
try {
|
|
248
|
+
names = fs
|
|
249
|
+
.readdirSync(evalsDir)
|
|
250
|
+
.filter((n) => n !== 'results' && !n.startsWith('.'))
|
|
251
|
+
.sort();
|
|
252
|
+
}
|
|
253
|
+
catch {
|
|
254
|
+
return { cases: [], errors: [] };
|
|
255
|
+
}
|
|
256
|
+
const cases = [];
|
|
257
|
+
for (const n of names) {
|
|
258
|
+
const dir = path.join(evalsDir, n);
|
|
259
|
+
let isDir;
|
|
260
|
+
try {
|
|
261
|
+
isDir = fs.statSync(dir).isDirectory();
|
|
262
|
+
}
|
|
263
|
+
catch {
|
|
264
|
+
isDir = false;
|
|
265
|
+
}
|
|
266
|
+
if (!isDir)
|
|
267
|
+
continue;
|
|
268
|
+
try {
|
|
269
|
+
cases.push(parseEvalCase(dir));
|
|
270
|
+
}
|
|
271
|
+
catch (e) {
|
|
272
|
+
errors.push({ dir, error: e.message });
|
|
273
|
+
}
|
|
274
|
+
}
|
|
275
|
+
return { cases, errors };
|
|
276
|
+
}
|
|
277
|
+
// ─── The graders ─────────────────────────────────────────────────────────────
|
|
278
|
+
/** The text a grader's target names. Exposed for tests and the report's `why`. */
|
|
279
|
+
function targetText(g, trace) {
|
|
280
|
+
if (g.type === 'regex' && g.target === 'trace') {
|
|
281
|
+
return trace.toolCalls.map((c) => `${c.tool} ${JSON.stringify(c.input)}`).join('\n');
|
|
282
|
+
}
|
|
283
|
+
if (g.type === 'regex' && g.target === 'file') {
|
|
284
|
+
const content = trace.readFile(g.pathPattern ?? '');
|
|
285
|
+
return content; // null = no matching file; the grader fails with that reason
|
|
286
|
+
}
|
|
287
|
+
if (g.type === 'llm' && g.focus === 'trace') {
|
|
288
|
+
return trace.toolCalls.map((c) => `${c.tool} ${JSON.stringify(c.input)}`).join('\n');
|
|
289
|
+
}
|
|
290
|
+
return trace.lastMessage;
|
|
291
|
+
}
|
|
292
|
+
/** null = the pattern does not compile; the grader FAILS with that reason. */
|
|
293
|
+
function compileRegex(g) {
|
|
294
|
+
try {
|
|
295
|
+
return new RegExp(g.pattern ?? '', g.flags ?? '');
|
|
296
|
+
}
|
|
297
|
+
catch {
|
|
298
|
+
return null;
|
|
299
|
+
}
|
|
300
|
+
}
|
|
301
|
+
function countMatches(re, text) {
|
|
302
|
+
// A global clone: a non-global regex .match() returns the first hit only, and
|
|
303
|
+
// count:N must count every occurrence.
|
|
304
|
+
const flags = re.flags.includes('g') ? re.flags : re.flags + 'g';
|
|
305
|
+
return (text.match(new RegExp(re.source, flags)) ?? []).length;
|
|
306
|
+
}
|
|
307
|
+
/**
|
|
308
|
+
* Score ONE grader over ONE run. Async because llm/baseline call the judge.
|
|
309
|
+
* A grader that cannot be evaluated FAILS (with the reason in `why`) — it must
|
|
310
|
+
* never pass by accident, or a broken suite would look green.
|
|
311
|
+
*/
|
|
312
|
+
async function evaluateGrader(g, trace, judge) {
|
|
313
|
+
try {
|
|
314
|
+
switch (g.type) {
|
|
315
|
+
case 'regex': {
|
|
316
|
+
const text = targetText(g, trace);
|
|
317
|
+
if (text === null)
|
|
318
|
+
return { passed: false, why: `no file matches ${g.pathPattern} (regex ${g.pattern})` };
|
|
319
|
+
const re = compileRegex(g);
|
|
320
|
+
if (!re)
|
|
321
|
+
return { passed: false, why: `invalid pattern: ${g.pattern}` };
|
|
322
|
+
const match = g.match ?? 'contains';
|
|
323
|
+
if (match === 'contains') {
|
|
324
|
+
return { passed: re.test(text), why: re.test(text) ? 'found' : `not found in ${g.target ?? 'last_message'}` };
|
|
325
|
+
}
|
|
326
|
+
if (match === 'not_contains') {
|
|
327
|
+
return {
|
|
328
|
+
passed: !re.test(text),
|
|
329
|
+
why: re.test(text) ? `unexpectedly found in ${g.target ?? 'last_message'}` : 'absent, as required',
|
|
330
|
+
};
|
|
331
|
+
}
|
|
332
|
+
const cm = COUNT_RE.exec(match);
|
|
333
|
+
const want = Number(cm[1]);
|
|
334
|
+
const got = countMatches(re, text);
|
|
335
|
+
return { passed: got === want, why: `count ${got}, wanted ${want}` };
|
|
336
|
+
}
|
|
337
|
+
case 'tool_used': {
|
|
338
|
+
const n = trace.toolCalls.filter((c) => {
|
|
339
|
+
if (c.tool !== g.tool)
|
|
340
|
+
return false;
|
|
341
|
+
if (!g.inputMatch)
|
|
342
|
+
return true;
|
|
343
|
+
try {
|
|
344
|
+
return new RegExp(g.inputMatch).test(JSON.stringify(c.input));
|
|
345
|
+
}
|
|
346
|
+
catch {
|
|
347
|
+
return false;
|
|
348
|
+
}
|
|
349
|
+
}).length;
|
|
350
|
+
const min = g.min ?? 1;
|
|
351
|
+
const ok = n >= min && (g.max === undefined || n <= g.max);
|
|
352
|
+
return {
|
|
353
|
+
passed: ok,
|
|
354
|
+
why: `${g.tool} used ${n} time(s)` +
|
|
355
|
+
(ok ? '' : ` (wanted ${g.max === undefined ? `>=${min}` : `${min}..${g.max}`})`),
|
|
356
|
+
};
|
|
357
|
+
}
|
|
358
|
+
case 'tool_order': {
|
|
359
|
+
const iBefore = trace.toolCalls.findIndex((c) => c.tool === g.before);
|
|
360
|
+
const iAfter = trace.toolCalls.findIndex((c) => c.tool === g.after);
|
|
361
|
+
if (iBefore === -1)
|
|
362
|
+
return { passed: false, why: `${g.before} never called` };
|
|
363
|
+
if (iAfter === -1)
|
|
364
|
+
return { passed: false, why: `${g.after} never called` };
|
|
365
|
+
return iBefore < iAfter
|
|
366
|
+
? { passed: true, why: `${g.before} before ${g.after}` }
|
|
367
|
+
: { passed: false, why: `${g.after} came before ${g.before}` };
|
|
368
|
+
}
|
|
369
|
+
case 'file_exists': {
|
|
370
|
+
const re = (0, nestedInstructions_1.pathGlobToRegExp)(g.path ?? '');
|
|
371
|
+
const hit = re ? trace.files.some((f) => re.test(f)) : false;
|
|
372
|
+
const want = g.exists !== false;
|
|
373
|
+
return {
|
|
374
|
+
passed: hit === want,
|
|
375
|
+
why: want ? (hit ? 'present' : 'missing') : hit ? 'present, wanted absent' : 'absent, as required',
|
|
376
|
+
};
|
|
377
|
+
}
|
|
378
|
+
case 'llm': {
|
|
379
|
+
if (!judge)
|
|
380
|
+
return { passed: false, why: 'no judge available (pass --judge-model or provide NEXRALL_EVAL_JUDGE_MODEL)' };
|
|
381
|
+
const focusText = targetText(g, trace) ?? '';
|
|
382
|
+
const verdict = await judge({ criteria: g.criteria ?? '', focusText });
|
|
383
|
+
if (verdict === null)
|
|
384
|
+
return { passed: false, why: 'judge could not be reached' };
|
|
385
|
+
return { passed: verdict, why: verdict ? 'judge: pass' : 'judge: fail' };
|
|
386
|
+
}
|
|
387
|
+
case 'baseline': {
|
|
388
|
+
if (!judge)
|
|
389
|
+
return { passed: false, why: 'no judge available (pass --judge-model)' };
|
|
390
|
+
const baselineText = trace.readFile(g.baselineFile ?? '');
|
|
391
|
+
if (baselineText === null)
|
|
392
|
+
return { passed: false, why: `baseline file ${g.baselineFile} not found` };
|
|
393
|
+
const verdict = await judge({ criteria: g.criteria ?? '', focusText: trace.lastMessage, baselineText });
|
|
394
|
+
if (verdict === null)
|
|
395
|
+
return { passed: false, why: 'judge could not be reached' };
|
|
396
|
+
return { passed: verdict, why: verdict ? 'at least as good as the baseline' : 'worse than the baseline' };
|
|
397
|
+
}
|
|
398
|
+
}
|
|
399
|
+
}
|
|
400
|
+
catch (e) {
|
|
401
|
+
return { passed: false, why: e.message };
|
|
402
|
+
}
|
|
403
|
+
}
|
|
404
|
+
/** Which graders count in an arm: with-only graders never score the baseline. */
|
|
405
|
+
function gradersForArm(graders, arm) {
|
|
406
|
+
return arm === 'with' ? graders : graders.filter((g) => g.arm !== 'with-only');
|
|
407
|
+
}
|
|
408
|
+
/** Score one run: weighted fraction of its graders that passed. */
|
|
409
|
+
async function scoreRun(graders, trace, judge, arm) {
|
|
410
|
+
const active = gradersForArm(graders, arm);
|
|
411
|
+
if (!active.length)
|
|
412
|
+
return { score: 0, results: [] };
|
|
413
|
+
const results = [];
|
|
414
|
+
let passedWeight = 0;
|
|
415
|
+
let totalWeight = 0;
|
|
416
|
+
for (const g of active) {
|
|
417
|
+
const v = await evaluateGrader(g, trace, judge);
|
|
418
|
+
totalWeight += g.weight;
|
|
419
|
+
if (v.passed)
|
|
420
|
+
passedWeight += g.weight;
|
|
421
|
+
results.push({ file: g.file, type: g.type, weight: g.weight, passed: v.passed, why: v.why });
|
|
422
|
+
}
|
|
423
|
+
return { score: totalWeight > 0 ? passedWeight / totalWeight : 0, results };
|
|
424
|
+
}
|
|
425
|
+
const mean = (xs) => (xs.length ? xs.reduce((a, b) => a + b, 0) / xs.length : null);
|
|
426
|
+
/**
|
|
427
|
+
* Fold run outcomes into the report. Case score = mean of the plugin-arm run
|
|
428
|
+
* scores; `delta` = with − without (null when the baseline did not run).
|
|
429
|
+
* A case PASSES at ≥ threshold (CC's default: every grader, every run).
|
|
430
|
+
*/
|
|
431
|
+
function aggregateReport(i) {
|
|
432
|
+
const cases = i.cases.map((c) => {
|
|
433
|
+
const withScore = mean(c.with.map((r) => r.score));
|
|
434
|
+
const withoutScore = mean(c.without.map((r) => r.score));
|
|
435
|
+
return {
|
|
436
|
+
name: c.name,
|
|
437
|
+
with: c.with,
|
|
438
|
+
without: c.without,
|
|
439
|
+
withScore,
|
|
440
|
+
withoutScore,
|
|
441
|
+
delta: withScore !== null && withoutScore !== null ? withScore - withoutScore : null,
|
|
442
|
+
passed: withScore !== null && withScore >= i.threshold,
|
|
443
|
+
};
|
|
444
|
+
});
|
|
445
|
+
const scores = cases.map((c) => c.withScore).filter((s) => s !== null);
|
|
446
|
+
const deltas = cases.map((c) => c.delta).filter((d) => d !== null);
|
|
447
|
+
const allRuns = cases.flatMap((c) => [...c.with, ...c.without]);
|
|
448
|
+
const reported = allRuns.filter((r) => r.costReported && r.costUsd !== null);
|
|
449
|
+
return {
|
|
450
|
+
schemaVersion: exports.EVAL_SCHEMA_VERSION,
|
|
451
|
+
plugin: i.plugin,
|
|
452
|
+
startedAt: i.startedAt,
|
|
453
|
+
durationMs: i.durationMs,
|
|
454
|
+
threshold: i.threshold,
|
|
455
|
+
runs: i.runs,
|
|
456
|
+
...(i.judgeModel ? { judgeModel: i.judgeModel } : {}),
|
|
457
|
+
...(i.partial ? { partial: true, partialReason: i.partial.reason } : {}),
|
|
458
|
+
cases,
|
|
459
|
+
errors: i.errors,
|
|
460
|
+
aggregates: {
|
|
461
|
+
casesPassed: cases.filter((c) => c.passed).length,
|
|
462
|
+
casesTotal: cases.length,
|
|
463
|
+
overallScore: scores.length ? scores.reduce((a, b) => a + b, 0) / scores.length : 0,
|
|
464
|
+
meanDelta: deltas.length ? deltas.reduce((a, b) => a + b, 0) / deltas.length : null,
|
|
465
|
+
},
|
|
466
|
+
costUsd: reported.reduce((a, r) => a + (r.costUsd ?? 0), 0),
|
|
467
|
+
costMissing: allRuns.length - reported.length,
|
|
468
|
+
};
|
|
469
|
+
}
|
|
470
|
+
//# sourceMappingURL=eval.js.map
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@nexrall/code-core",
|
|
3
|
-
"version": "1.4.
|
|
3
|
+
"version": "1.4.76",
|
|
4
4
|
"description": "Core agent loop, tools, and extension primitives for Nexrall Code — embed an AI coding agent in any Node.js application.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "Nexrall <support@nexrall.com> (https://nexrall.com)",
|