@nexrall/code-core 1.4.73 → 1.4.75
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -3
- package/dist/agent/compaction.d.ts.map +1 -1
- package/dist/agent/compaction.js +17 -13
- package/dist/agent/hooks.d.ts +13 -0
- package/dist/agent/hooks.d.ts.map +1 -1
- package/dist/agent/hooks.js +1 -0
- package/dist/agent/loop.d.ts.map +1 -1
- package/dist/agent/loop.js +67 -2
- package/dist/agent/sharedTasks.d.ts +21 -6
- package/dist/agent/sharedTasks.d.ts.map +1 -1
- package/dist/agent/sharedTasks.js +69 -17
- package/dist/agent/taskEvents.d.ts +73 -0
- package/dist/agent/taskEvents.d.ts.map +1 -0
- package/dist/agent/taskEvents.js +297 -0
- package/dist/index.d.ts +2 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +2 -0
- package/dist/plugins/eval.d.ts +199 -0
- package/dist/plugins/eval.d.ts.map +1 -0
- package/dist/plugins/eval.js +470 -0
- package/dist/tools/executor.d.ts +6 -0
- package/dist/tools/executor.d.ts.map +1 -1
- package/dist/tools/executor.js +116 -6
- package/dist/types.d.ts +10 -0
- package/dist/types.d.ts.map +1 -1
- package/package.json +1 -1
|
@@ -0,0 +1,470 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
// S21 — `nex plugin eval`: run a plugin against a suite of test cases and score
|
|
3
|
+
// the results (Claude Code's `claude plugin eval`, shipped in their W37).
|
|
4
|
+
//
|
|
5
|
+
// This file is the PURE half: case discovery/parsing, the grader vocabulary,
|
|
6
|
+
// the trace-based graders, and the scoring/aggregation. The caller owns ALL
|
|
7
|
+
// I/O — spawning each run, reading the child's event log, and judging (the
|
|
8
|
+
// `llm`/`baseline` graders call a model through an injected `judge` function so
|
|
9
|
+
// this module stays unit-testable without a backend).
|
|
10
|
+
//
|
|
11
|
+
// Suite layout (CC's, minus what we do not support yet — see below):
|
|
12
|
+
//
|
|
13
|
+
// my-plugin/
|
|
14
|
+
// evals/
|
|
15
|
+
// first-case/
|
|
16
|
+
// prompt.md # frontmatter = case fields; body = the prompt
|
|
17
|
+
// graders/
|
|
18
|
+
// criteria.md # frontmatter = grader; body = llm criteria
|
|
19
|
+
// skill-fired.md
|
|
20
|
+
// results/ # written by each run; never discovered as a case
|
|
21
|
+
//
|
|
22
|
+
// Deliberately NOT supported yet (refused loudly, never silently skipped):
|
|
23
|
+
// · case.yaml — needs a real YAML parser, which this repo does not carry
|
|
24
|
+
// (the frontmatter reader is "not a full YAML parser" by design). A case
|
|
25
|
+
// that ships one is an ERROR, because silently ignoring its fields would
|
|
26
|
+
// run a DIFFERENT suite than the author wrote.
|
|
27
|
+
// · mock MCP servers (`evals/mocks/`) — a follow-up; a case using them is
|
|
28
|
+
// likewise refused rather than run without the mocks.
|
|
29
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
30
|
+
if (k2 === undefined) k2 = k;
|
|
31
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
32
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
33
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
34
|
+
}
|
|
35
|
+
Object.defineProperty(o, k2, desc);
|
|
36
|
+
}) : (function(o, m, k, k2) {
|
|
37
|
+
if (k2 === undefined) k2 = k;
|
|
38
|
+
o[k2] = m[k];
|
|
39
|
+
}));
|
|
40
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
41
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
42
|
+
}) : function(o, v) {
|
|
43
|
+
o["default"] = v;
|
|
44
|
+
});
|
|
45
|
+
var __importStar = (this && this.__importStar) || (function () {
|
|
46
|
+
var ownKeys = function(o) {
|
|
47
|
+
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
48
|
+
var ar = [];
|
|
49
|
+
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
50
|
+
return ar;
|
|
51
|
+
};
|
|
52
|
+
return ownKeys(o);
|
|
53
|
+
};
|
|
54
|
+
return function (mod) {
|
|
55
|
+
if (mod && mod.__esModule) return mod;
|
|
56
|
+
var result = {};
|
|
57
|
+
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
58
|
+
__setModuleDefault(result, mod);
|
|
59
|
+
return result;
|
|
60
|
+
};
|
|
61
|
+
})();
|
|
62
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
63
|
+
exports.EVAL_MAX_TIMEOUT_S = exports.EVAL_MAX_TURNS_CEILING = exports.EVAL_DEFAULT_TIMEOUT_S = exports.EVAL_DEFAULT_MAX_TURNS = exports.EVAL_DEFAULT_THRESHOLD = exports.EVAL_MAX_RUNS = exports.EVAL_DEFAULT_RUNS = exports.EVAL_SCHEMA_VERSION = void 0;
|
|
64
|
+
exports.parseList = parseList;
|
|
65
|
+
exports.parseGraderFile = parseGraderFile;
|
|
66
|
+
exports.parseEvalCase = parseEvalCase;
|
|
67
|
+
exports.discoverEvalCases = discoverEvalCases;
|
|
68
|
+
exports.targetText = targetText;
|
|
69
|
+
exports.evaluateGrader = evaluateGrader;
|
|
70
|
+
exports.gradersForArm = gradersForArm;
|
|
71
|
+
exports.scoreRun = scoreRun;
|
|
72
|
+
exports.aggregateReport = aggregateReport;
|
|
73
|
+
const fs = __importStar(require("fs"));
|
|
74
|
+
const path = __importStar(require("path"));
|
|
75
|
+
const frontmatter_1 = require("../util/frontmatter");
|
|
76
|
+
const nestedInstructions_1 = require("../agent/nestedInstructions");
|
|
77
|
+
exports.EVAL_SCHEMA_VERSION = 1;
|
|
78
|
+
/** CC's default: each case runs three times. */
|
|
79
|
+
exports.EVAL_DEFAULT_RUNS = 3;
|
|
80
|
+
exports.EVAL_MAX_RUNS = 50;
|
|
81
|
+
exports.EVAL_DEFAULT_THRESHOLD = 1.0;
|
|
82
|
+
exports.EVAL_DEFAULT_MAX_TURNS = 10;
|
|
83
|
+
exports.EVAL_DEFAULT_TIMEOUT_S = 300;
|
|
84
|
+
/** CC caps both; larger values are refused, not clamped silently. */
|
|
85
|
+
exports.EVAL_MAX_TURNS_CEILING = 200;
|
|
86
|
+
exports.EVAL_MAX_TIMEOUT_S = 3600;
|
|
87
|
+
/** `match: count:N` in a regex grader. */
|
|
88
|
+
const COUNT_RE = /^count:(\d+)$/;
|
|
89
|
+
// ─── Parsing helpers ─────────────────────────────────────────────────────────
|
|
90
|
+
/** `[a, b]` / `a, b` / `a` → ['a','b'] — the two shapes real files use. */
|
|
91
|
+
function parseList(value) {
|
|
92
|
+
if (value === undefined)
|
|
93
|
+
return [];
|
|
94
|
+
const inner = value.trim().replace(/^\[|\]$/g, '');
|
|
95
|
+
return inner
|
|
96
|
+
.split(',')
|
|
97
|
+
.map((s) => s.trim().replace(/^["']|["']$/g, ''))
|
|
98
|
+
.filter(Boolean);
|
|
99
|
+
}
|
|
100
|
+
function parsePositiveInt(value, dflt, ceiling, what) {
|
|
101
|
+
if (value === undefined || value.trim() === '')
|
|
102
|
+
return dflt;
|
|
103
|
+
const n = Number(value.trim());
|
|
104
|
+
if (!Number.isInteger(n) || n < 1)
|
|
105
|
+
throw new Error(`${what} must be a positive integer (got "${value}")`);
|
|
106
|
+
if (n > ceiling)
|
|
107
|
+
throw new Error(`${what} ${n} exceeds the ceiling of ${ceiling}`);
|
|
108
|
+
return n;
|
|
109
|
+
}
|
|
110
|
+
/**
|
|
111
|
+
* One grader file → EvalGrader. Throws on anything that would make the grader
|
|
112
|
+
* silently unscorable (unknown type, missing pattern/tool/path, bad count).
|
|
113
|
+
*/
|
|
114
|
+
function parseGraderFile(file, raw) {
|
|
115
|
+
const { meta, body } = (0, frontmatter_1.parseFrontmatter)(raw);
|
|
116
|
+
const type = (meta.type ?? '').trim();
|
|
117
|
+
if (!['regex', 'tool_used', 'tool_order', 'file_exists', 'llm', 'baseline'].includes(type)) {
|
|
118
|
+
throw new Error(`grader "${file}": type must be regex|tool_used|tool_order|file_exists|llm|baseline (got "${meta.type ?? ''}")`);
|
|
119
|
+
}
|
|
120
|
+
const weightRaw = (meta.weight ?? '1').trim();
|
|
121
|
+
const weight = Number(weightRaw);
|
|
122
|
+
if (!Number.isFinite(weight) || weight <= 0)
|
|
123
|
+
throw new Error(`grader "${file}": weight must be a positive number`);
|
|
124
|
+
const arm = (meta.arm ?? 'both').trim() === 'with-only' ? 'with-only' : 'both';
|
|
125
|
+
if (meta.arm !== undefined && !['both', 'with-only'].includes(meta.arm.trim())) {
|
|
126
|
+
throw new Error(`grader "${file}": arm must be both or with-only (got "${meta.arm}")`);
|
|
127
|
+
}
|
|
128
|
+
const g = { file, type, weight, arm };
|
|
129
|
+
if (type === 'regex') {
|
|
130
|
+
if (!meta.pattern)
|
|
131
|
+
throw new Error(`grader "${file}": regex grader needs a \`pattern\``);
|
|
132
|
+
g.pattern = meta.pattern;
|
|
133
|
+
g.flags = meta.flags?.trim() || '';
|
|
134
|
+
const match = (meta.match ?? 'contains').trim();
|
|
135
|
+
if (!(match === 'contains' || match === 'not_contains' || COUNT_RE.test(match))) {
|
|
136
|
+
throw new Error(`grader "${file}": match must be contains | not_contains | count:<n> (got "${match}")`);
|
|
137
|
+
}
|
|
138
|
+
g.match = match;
|
|
139
|
+
const target = (meta.target ?? 'last_message').trim();
|
|
140
|
+
if (!['last_message', 'trace', 'file'].includes(target)) {
|
|
141
|
+
throw new Error(`grader "${file}": target must be last_message | trace | file (got "${target}")`);
|
|
142
|
+
}
|
|
143
|
+
g.target = target;
|
|
144
|
+
if (target === 'file') {
|
|
145
|
+
if (!meta.path)
|
|
146
|
+
throw new Error(`grader "${file}": target "file" needs a \`path\``);
|
|
147
|
+
g.pathPattern = meta.path;
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
else if (type === 'tool_used') {
|
|
151
|
+
if (!meta.tool)
|
|
152
|
+
throw new Error(`grader "${file}": tool_used grader needs a \`tool\``);
|
|
153
|
+
g.tool = meta.tool.trim();
|
|
154
|
+
g.inputMatch = meta.input_match?.trim() || undefined;
|
|
155
|
+
g.min = parsePositiveInt(meta.min, 1, 10000, 'min');
|
|
156
|
+
g.max = meta.max !== undefined && meta.max.trim() !== '' ? parsePositiveInt(meta.max, 1, 10000, 'max') : undefined;
|
|
157
|
+
if (g.max !== undefined && g.max < (g.min ?? 1))
|
|
158
|
+
throw new Error(`grader "${file}": max < min`);
|
|
159
|
+
}
|
|
160
|
+
else if (type === 'tool_order') {
|
|
161
|
+
if (!meta.before || !meta.after)
|
|
162
|
+
throw new Error(`grader "${file}": tool_order needs \`before\` and \`after\``);
|
|
163
|
+
g.before = meta.before.trim();
|
|
164
|
+
g.after = meta.after.trim();
|
|
165
|
+
}
|
|
166
|
+
else if (type === 'file_exists') {
|
|
167
|
+
if (!meta.path)
|
|
168
|
+
throw new Error(`grader "${file}": file_exists needs a \`path\``);
|
|
169
|
+
g.path = meta.path.trim();
|
|
170
|
+
const ex = (meta.exists ?? 'true').trim().toLowerCase();
|
|
171
|
+
g.exists = !(ex === 'false' || ex === '0' || ex === 'no');
|
|
172
|
+
}
|
|
173
|
+
else {
|
|
174
|
+
// llm / baseline: criteria from the body (the file IS the criteria text).
|
|
175
|
+
const criteria = body.trim();
|
|
176
|
+
if (!criteria)
|
|
177
|
+
throw new Error(`grader "${file}": ${type} grader needs criteria in the file body`);
|
|
178
|
+
g.criteria = criteria;
|
|
179
|
+
if (type === 'llm') {
|
|
180
|
+
const focus = (meta.focus ?? 'last_message').trim();
|
|
181
|
+
if (!['last_message', 'trace'].includes(focus)) {
|
|
182
|
+
throw new Error(`grader "${file}": focus must be last_message | trace (got "${focus}")`);
|
|
183
|
+
}
|
|
184
|
+
g.focus = focus;
|
|
185
|
+
}
|
|
186
|
+
else {
|
|
187
|
+
if (!meta.baseline_file)
|
|
188
|
+
throw new Error(`grader "${file}": baseline grader needs \`baseline_file\``);
|
|
189
|
+
g.baselineFile = meta.baseline_file.trim();
|
|
190
|
+
}
|
|
191
|
+
}
|
|
192
|
+
return g;
|
|
193
|
+
}
|
|
194
|
+
/** One case directory → EvalCase. See the header for the layout. */
|
|
195
|
+
function parseEvalCase(dir) {
|
|
196
|
+
const promptFile = path.join(dir, 'prompt.md');
|
|
197
|
+
const caseYaml = path.join(dir, 'case.yaml');
|
|
198
|
+
if (fs.existsSync(caseYaml)) {
|
|
199
|
+
throw new Error('case.yaml is not supported yet (this repo carries no YAML parser) — move its fields into prompt.md frontmatter');
|
|
200
|
+
}
|
|
201
|
+
if (fs.existsSync(path.join(dir, 'mocks'))) {
|
|
202
|
+
throw new Error('mock MCP servers (mocks/) are not supported yet');
|
|
203
|
+
}
|
|
204
|
+
let raw;
|
|
205
|
+
try {
|
|
206
|
+
raw = fs.readFileSync(promptFile, 'utf-8');
|
|
207
|
+
}
|
|
208
|
+
catch {
|
|
209
|
+
throw new Error('no prompt.md (a case needs one; case.yaml alone is not supported yet)');
|
|
210
|
+
}
|
|
211
|
+
const { meta, body } = (0, frontmatter_1.parseFrontmatter)(raw);
|
|
212
|
+
const prompt = body.trim();
|
|
213
|
+
if (!prompt)
|
|
214
|
+
throw new Error('prompt.md has an empty prompt body');
|
|
215
|
+
const gradersDir = path.join(dir, 'graders');
|
|
216
|
+
let graderFiles;
|
|
217
|
+
try {
|
|
218
|
+
graderFiles = fs
|
|
219
|
+
.readdirSync(gradersDir)
|
|
220
|
+
.filter((f) => f.endsWith('.md'))
|
|
221
|
+
.sort();
|
|
222
|
+
}
|
|
223
|
+
catch {
|
|
224
|
+
throw new Error('no graders/ directory (at least one grader is required)');
|
|
225
|
+
}
|
|
226
|
+
if (!graderFiles.length)
|
|
227
|
+
throw new Error('graders/ holds no .md files (at least one grader is required)');
|
|
228
|
+
const graders = graderFiles.map((f) => parseGraderFile(f, fs.readFileSync(path.join(gradersDir, f), 'utf-8')));
|
|
229
|
+
return {
|
|
230
|
+
name: meta.name?.trim() || path.basename(dir),
|
|
231
|
+
dir,
|
|
232
|
+
prompt,
|
|
233
|
+
runs: parsePositiveInt(meta.runs, exports.EVAL_DEFAULT_RUNS, exports.EVAL_MAX_RUNS, 'runs'),
|
|
234
|
+
maxTurns: parsePositiveInt(meta.max_turns, exports.EVAL_DEFAULT_MAX_TURNS, exports.EVAL_MAX_TURNS_CEILING, 'max_turns'),
|
|
235
|
+
timeoutSeconds: parsePositiveInt(meta.timeout_seconds, exports.EVAL_DEFAULT_TIMEOUT_S, exports.EVAL_MAX_TIMEOUT_S, 'timeout_seconds'),
|
|
236
|
+
allowedTools: parseList(meta.allowed_tools),
|
|
237
|
+
tags: parseList(meta.tags),
|
|
238
|
+
expectedOutcome: meta.expected_outcome?.trim() || undefined,
|
|
239
|
+
graders,
|
|
240
|
+
};
|
|
241
|
+
}
|
|
242
|
+
/** Every case under `<pluginDir>/evals` (the results dir is never a case). */
|
|
243
|
+
function discoverEvalCases(pluginDir) {
|
|
244
|
+
const evalsDir = path.join(pluginDir, 'evals');
|
|
245
|
+
const errors = [];
|
|
246
|
+
let names;
|
|
247
|
+
try {
|
|
248
|
+
names = fs
|
|
249
|
+
.readdirSync(evalsDir)
|
|
250
|
+
.filter((n) => n !== 'results' && !n.startsWith('.'))
|
|
251
|
+
.sort();
|
|
252
|
+
}
|
|
253
|
+
catch {
|
|
254
|
+
return { cases: [], errors: [] };
|
|
255
|
+
}
|
|
256
|
+
const cases = [];
|
|
257
|
+
for (const n of names) {
|
|
258
|
+
const dir = path.join(evalsDir, n);
|
|
259
|
+
let isDir;
|
|
260
|
+
try {
|
|
261
|
+
isDir = fs.statSync(dir).isDirectory();
|
|
262
|
+
}
|
|
263
|
+
catch {
|
|
264
|
+
isDir = false;
|
|
265
|
+
}
|
|
266
|
+
if (!isDir)
|
|
267
|
+
continue;
|
|
268
|
+
try {
|
|
269
|
+
cases.push(parseEvalCase(dir));
|
|
270
|
+
}
|
|
271
|
+
catch (e) {
|
|
272
|
+
errors.push({ dir, error: e.message });
|
|
273
|
+
}
|
|
274
|
+
}
|
|
275
|
+
return { cases, errors };
|
|
276
|
+
}
|
|
277
|
+
// ─── The graders ─────────────────────────────────────────────────────────────
|
|
278
|
+
/** The text a grader's target names. Exposed for tests and the report's `why`. */
|
|
279
|
+
function targetText(g, trace) {
|
|
280
|
+
if (g.type === 'regex' && g.target === 'trace') {
|
|
281
|
+
return trace.toolCalls.map((c) => `${c.tool} ${JSON.stringify(c.input)}`).join('\n');
|
|
282
|
+
}
|
|
283
|
+
if (g.type === 'regex' && g.target === 'file') {
|
|
284
|
+
const content = trace.readFile(g.pathPattern ?? '');
|
|
285
|
+
return content; // null = no matching file; the grader fails with that reason
|
|
286
|
+
}
|
|
287
|
+
if (g.type === 'llm' && g.focus === 'trace') {
|
|
288
|
+
return trace.toolCalls.map((c) => `${c.tool} ${JSON.stringify(c.input)}`).join('\n');
|
|
289
|
+
}
|
|
290
|
+
return trace.lastMessage;
|
|
291
|
+
}
|
|
292
|
+
/** null = the pattern does not compile; the grader FAILS with that reason. */
|
|
293
|
+
function compileRegex(g) {
|
|
294
|
+
try {
|
|
295
|
+
return new RegExp(g.pattern ?? '', g.flags ?? '');
|
|
296
|
+
}
|
|
297
|
+
catch {
|
|
298
|
+
return null;
|
|
299
|
+
}
|
|
300
|
+
}
|
|
301
|
+
function countMatches(re, text) {
|
|
302
|
+
// A global clone: a non-global regex .match() returns the first hit only, and
|
|
303
|
+
// count:N must count every occurrence.
|
|
304
|
+
const flags = re.flags.includes('g') ? re.flags : re.flags + 'g';
|
|
305
|
+
return (text.match(new RegExp(re.source, flags)) ?? []).length;
|
|
306
|
+
}
|
|
307
|
+
/**
|
|
308
|
+
* Score ONE grader over ONE run. Async because llm/baseline call the judge.
|
|
309
|
+
* A grader that cannot be evaluated FAILS (with the reason in `why`) — it must
|
|
310
|
+
* never pass by accident, or a broken suite would look green.
|
|
311
|
+
*/
|
|
312
|
+
async function evaluateGrader(g, trace, judge) {
|
|
313
|
+
try {
|
|
314
|
+
switch (g.type) {
|
|
315
|
+
case 'regex': {
|
|
316
|
+
const text = targetText(g, trace);
|
|
317
|
+
if (text === null)
|
|
318
|
+
return { passed: false, why: `no file matches ${g.pathPattern} (regex ${g.pattern})` };
|
|
319
|
+
const re = compileRegex(g);
|
|
320
|
+
if (!re)
|
|
321
|
+
return { passed: false, why: `invalid pattern: ${g.pattern}` };
|
|
322
|
+
const match = g.match ?? 'contains';
|
|
323
|
+
if (match === 'contains') {
|
|
324
|
+
return { passed: re.test(text), why: re.test(text) ? 'found' : `not found in ${g.target ?? 'last_message'}` };
|
|
325
|
+
}
|
|
326
|
+
if (match === 'not_contains') {
|
|
327
|
+
return {
|
|
328
|
+
passed: !re.test(text),
|
|
329
|
+
why: re.test(text) ? `unexpectedly found in ${g.target ?? 'last_message'}` : 'absent, as required',
|
|
330
|
+
};
|
|
331
|
+
}
|
|
332
|
+
const cm = COUNT_RE.exec(match);
|
|
333
|
+
const want = Number(cm[1]);
|
|
334
|
+
const got = countMatches(re, text);
|
|
335
|
+
return { passed: got === want, why: `count ${got}, wanted ${want}` };
|
|
336
|
+
}
|
|
337
|
+
case 'tool_used': {
|
|
338
|
+
const n = trace.toolCalls.filter((c) => {
|
|
339
|
+
if (c.tool !== g.tool)
|
|
340
|
+
return false;
|
|
341
|
+
if (!g.inputMatch)
|
|
342
|
+
return true;
|
|
343
|
+
try {
|
|
344
|
+
return new RegExp(g.inputMatch).test(JSON.stringify(c.input));
|
|
345
|
+
}
|
|
346
|
+
catch {
|
|
347
|
+
return false;
|
|
348
|
+
}
|
|
349
|
+
}).length;
|
|
350
|
+
const min = g.min ?? 1;
|
|
351
|
+
const ok = n >= min && (g.max === undefined || n <= g.max);
|
|
352
|
+
return {
|
|
353
|
+
passed: ok,
|
|
354
|
+
why: `${g.tool} used ${n} time(s)` +
|
|
355
|
+
(ok ? '' : ` (wanted ${g.max === undefined ? `>=${min}` : `${min}..${g.max}`})`),
|
|
356
|
+
};
|
|
357
|
+
}
|
|
358
|
+
case 'tool_order': {
|
|
359
|
+
const iBefore = trace.toolCalls.findIndex((c) => c.tool === g.before);
|
|
360
|
+
const iAfter = trace.toolCalls.findIndex((c) => c.tool === g.after);
|
|
361
|
+
if (iBefore === -1)
|
|
362
|
+
return { passed: false, why: `${g.before} never called` };
|
|
363
|
+
if (iAfter === -1)
|
|
364
|
+
return { passed: false, why: `${g.after} never called` };
|
|
365
|
+
return iBefore < iAfter
|
|
366
|
+
? { passed: true, why: `${g.before} before ${g.after}` }
|
|
367
|
+
: { passed: false, why: `${g.after} came before ${g.before}` };
|
|
368
|
+
}
|
|
369
|
+
case 'file_exists': {
|
|
370
|
+
const re = (0, nestedInstructions_1.pathGlobToRegExp)(g.path ?? '');
|
|
371
|
+
const hit = re ? trace.files.some((f) => re.test(f)) : false;
|
|
372
|
+
const want = g.exists !== false;
|
|
373
|
+
return {
|
|
374
|
+
passed: hit === want,
|
|
375
|
+
why: want ? (hit ? 'present' : 'missing') : hit ? 'present, wanted absent' : 'absent, as required',
|
|
376
|
+
};
|
|
377
|
+
}
|
|
378
|
+
case 'llm': {
|
|
379
|
+
if (!judge)
|
|
380
|
+
return { passed: false, why: 'no judge available (pass --judge-model or provide NEXRALL_EVAL_JUDGE_MODEL)' };
|
|
381
|
+
const focusText = targetText(g, trace) ?? '';
|
|
382
|
+
const verdict = await judge({ criteria: g.criteria ?? '', focusText });
|
|
383
|
+
if (verdict === null)
|
|
384
|
+
return { passed: false, why: 'judge could not be reached' };
|
|
385
|
+
return { passed: verdict, why: verdict ? 'judge: pass' : 'judge: fail' };
|
|
386
|
+
}
|
|
387
|
+
case 'baseline': {
|
|
388
|
+
if (!judge)
|
|
389
|
+
return { passed: false, why: 'no judge available (pass --judge-model)' };
|
|
390
|
+
const baselineText = trace.readFile(g.baselineFile ?? '');
|
|
391
|
+
if (baselineText === null)
|
|
392
|
+
return { passed: false, why: `baseline file ${g.baselineFile} not found` };
|
|
393
|
+
const verdict = await judge({ criteria: g.criteria ?? '', focusText: trace.lastMessage, baselineText });
|
|
394
|
+
if (verdict === null)
|
|
395
|
+
return { passed: false, why: 'judge could not be reached' };
|
|
396
|
+
return { passed: verdict, why: verdict ? 'at least as good as the baseline' : 'worse than the baseline' };
|
|
397
|
+
}
|
|
398
|
+
}
|
|
399
|
+
}
|
|
400
|
+
catch (e) {
|
|
401
|
+
return { passed: false, why: e.message };
|
|
402
|
+
}
|
|
403
|
+
}
|
|
404
|
+
/** Which graders count in an arm: with-only graders never score the baseline. */
|
|
405
|
+
function gradersForArm(graders, arm) {
|
|
406
|
+
return arm === 'with' ? graders : graders.filter((g) => g.arm !== 'with-only');
|
|
407
|
+
}
|
|
408
|
+
/** Score one run: weighted fraction of its graders that passed. */
|
|
409
|
+
async function scoreRun(graders, trace, judge, arm) {
|
|
410
|
+
const active = gradersForArm(graders, arm);
|
|
411
|
+
if (!active.length)
|
|
412
|
+
return { score: 0, results: [] };
|
|
413
|
+
const results = [];
|
|
414
|
+
let passedWeight = 0;
|
|
415
|
+
let totalWeight = 0;
|
|
416
|
+
for (const g of active) {
|
|
417
|
+
const v = await evaluateGrader(g, trace, judge);
|
|
418
|
+
totalWeight += g.weight;
|
|
419
|
+
if (v.passed)
|
|
420
|
+
passedWeight += g.weight;
|
|
421
|
+
results.push({ file: g.file, type: g.type, weight: g.weight, passed: v.passed, why: v.why });
|
|
422
|
+
}
|
|
423
|
+
return { score: totalWeight > 0 ? passedWeight / totalWeight : 0, results };
|
|
424
|
+
}
|
|
425
|
+
const mean = (xs) => (xs.length ? xs.reduce((a, b) => a + b, 0) / xs.length : null);
|
|
426
|
+
/**
|
|
427
|
+
* Fold run outcomes into the report. Case score = mean of the plugin-arm run
|
|
428
|
+
* scores; `delta` = with − without (null when the baseline did not run).
|
|
429
|
+
* A case PASSES at ≥ threshold (CC's default: every grader, every run).
|
|
430
|
+
*/
|
|
431
|
+
function aggregateReport(i) {
|
|
432
|
+
const cases = i.cases.map((c) => {
|
|
433
|
+
const withScore = mean(c.with.map((r) => r.score));
|
|
434
|
+
const withoutScore = mean(c.without.map((r) => r.score));
|
|
435
|
+
return {
|
|
436
|
+
name: c.name,
|
|
437
|
+
with: c.with,
|
|
438
|
+
without: c.without,
|
|
439
|
+
withScore,
|
|
440
|
+
withoutScore,
|
|
441
|
+
delta: withScore !== null && withoutScore !== null ? withScore - withoutScore : null,
|
|
442
|
+
passed: withScore !== null && withScore >= i.threshold,
|
|
443
|
+
};
|
|
444
|
+
});
|
|
445
|
+
const scores = cases.map((c) => c.withScore).filter((s) => s !== null);
|
|
446
|
+
const deltas = cases.map((c) => c.delta).filter((d) => d !== null);
|
|
447
|
+
const allRuns = cases.flatMap((c) => [...c.with, ...c.without]);
|
|
448
|
+
const reported = allRuns.filter((r) => r.costReported && r.costUsd !== null);
|
|
449
|
+
return {
|
|
450
|
+
schemaVersion: exports.EVAL_SCHEMA_VERSION,
|
|
451
|
+
plugin: i.plugin,
|
|
452
|
+
startedAt: i.startedAt,
|
|
453
|
+
durationMs: i.durationMs,
|
|
454
|
+
threshold: i.threshold,
|
|
455
|
+
runs: i.runs,
|
|
456
|
+
...(i.judgeModel ? { judgeModel: i.judgeModel } : {}),
|
|
457
|
+
...(i.partial ? { partial: true, partialReason: i.partial.reason } : {}),
|
|
458
|
+
cases,
|
|
459
|
+
errors: i.errors,
|
|
460
|
+
aggregates: {
|
|
461
|
+
casesPassed: cases.filter((c) => c.passed).length,
|
|
462
|
+
casesTotal: cases.length,
|
|
463
|
+
overallScore: scores.length ? scores.reduce((a, b) => a + b, 0) / scores.length : 0,
|
|
464
|
+
meanDelta: deltas.length ? deltas.reduce((a, b) => a + b, 0) / deltas.length : null,
|
|
465
|
+
},
|
|
466
|
+
costUsd: reported.reduce((a, r) => a + (r.costUsd ?? 0), 0),
|
|
467
|
+
costMissing: allRuns.length - reported.length,
|
|
468
|
+
};
|
|
469
|
+
}
|
|
470
|
+
//# sourceMappingURL=eval.js.map
|
package/dist/tools/executor.d.ts
CHANGED
|
@@ -29,6 +29,12 @@ export declare function runCapture(cmd: string, args: string[], opts: {
|
|
|
29
29
|
timeout: number;
|
|
30
30
|
maxBuffer: number;
|
|
31
31
|
}): Promise<CaptureResult>;
|
|
32
|
+
/**
|
|
33
|
+
* Path of the ripgrep binary this process would use — the NEXRALL_RG_PATH override, `rg`
|
|
34
|
+
* from PATH, or an editor-bundled copy — or undefined when the grep fallback is in charge.
|
|
35
|
+
* `nex doctor` reports through this so its verdict matches what the search tools do.
|
|
36
|
+
*/
|
|
37
|
+
export declare function resolveRipgrepBinary(): string | undefined;
|
|
32
38
|
/**
|
|
33
39
|
* Glob → anchored RegExp. Scans the pattern once, so the `*` / `?` rules can never
|
|
34
40
|
* rewrite the regex emitted for `**` (the old chained .replace() calls turned
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"executor.d.ts","sourceRoot":"","sources":["../../src/tools/executor.ts"],"names":[],"mappings":"AAOA,OAAO,KAAK,EAAE,UAAU,EAAE,MAAM,UAAU,CAAC;AAC3C,OAAO,EAAyB,KAAK,aAAa,EAAE,MAAM,WAAW,CAAC;AAyuCtE;;;;;;;GAOG;AACH,wBAAgB,iBAAiB,CAAC,MAAM,EAAE,MAAM,EAAE,KAAK,EAAE,MAAM,EAAE,QAAQ,SAAmB,GAAG,MAAM,CAgBpG;AAED;;;;;;;GAOG;AACH,MAAM,WAAW,aAAa;IAC5B,MAAM,EAAE,MAAM,CAAC;IACf,MAAM,EAAE,MAAM,CAAC;IACf,MAAM,EAAE,MAAM,GAAG,IAAI,CAAC;IACtB,MAAM,EAAE,MAAM,CAAC,OAAO,GAAG,IAAI,CAAC;IAC9B,KAAK,CAAC,EAAE,MAAM,CAAC,cAAc,CAAC;CAC/B;AACD,wBAAgB,UAAU,CACxB,GAAG,EAAE,MAAM,EACX,IAAI,EAAE,MAAM,EAAE,EACd,IAAI,EAAE;IAAE,GAAG,CAAC,EAAE,MAAM,CAAC;IAAC,OAAO,EAAE,MAAM,CAAC;IAAC,SAAS,EAAE,MAAM,CAAA;CAAE,GACzD,OAAO,CAAC,aAAa,CAAC,CAmDxB;
|
|
1
|
+
{"version":3,"file":"executor.d.ts","sourceRoot":"","sources":["../../src/tools/executor.ts"],"names":[],"mappings":"AAOA,OAAO,KAAK,EAAE,UAAU,EAAE,MAAM,UAAU,CAAC;AAC3C,OAAO,EAAyB,KAAK,aAAa,EAAE,MAAM,WAAW,CAAC;AAyuCtE;;;;;;;GAOG;AACH,wBAAgB,iBAAiB,CAAC,MAAM,EAAE,MAAM,EAAE,KAAK,EAAE,MAAM,EAAE,QAAQ,SAAmB,GAAG,MAAM,CAgBpG;AAED;;;;;;;GAOG;AACH,MAAM,WAAW,aAAa;IAC5B,MAAM,EAAE,MAAM,CAAC;IACf,MAAM,EAAE,MAAM,CAAC;IACf,MAAM,EAAE,MAAM,GAAG,IAAI,CAAC;IACtB,MAAM,EAAE,MAAM,CAAC,OAAO,GAAG,IAAI,CAAC;IAC9B,KAAK,CAAC,EAAE,MAAM,CAAC,cAAc,CAAC;CAC/B;AACD,wBAAgB,UAAU,CACxB,GAAG,EAAE,MAAM,EACX,IAAI,EAAE,MAAM,EAAE,EACd,IAAI,EAAE;IAAE,GAAG,CAAC,EAAE,MAAM,CAAC;IAAC,OAAO,EAAE,MAAM,CAAC;IAAC,SAAS,EAAE,MAAM,CAAA;CAAE,GACzD,OAAO,CAAC,aAAa,CAAC,CAmDxB;AAqID;;;;GAIG;AACH,wBAAgB,oBAAoB,IAAI,MAAM,GAAG,SAAS,CAEzD;AA2zCD;;;;;GAKG;AACH,wBAAgB,WAAW,CAAC,OAAO,EAAE,MAAM,GAAG,MAAM,CAuCnD;AAq8BD,wBAAsB,WAAW,CAC/B,IAAI,EAAE,MAAM,EACZ,KAAK,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,EAC9B,WAAW,CAAC,EAAE;IAAE,OAAO,EAAE,OAAO,CAAA;CAAE,EAClC,OAAO,CAAC,EAAE,aAAa,EACvB,OAAO,CAAC,EAAE,MAAM,EAChB,UAAU,CAAC,EAAE,MAAM,EACnB,QAAQ,CAAC,EAAE,CAAC,KAAK,EAAE,MAAM,KAAK,IAAI,EAMlC,QAAQ,CAAC,EAAE;IAAE,EAAE,EAAE,MAAM,CAAC;IAAC,IAAI,EAAE,MAAM,CAAA;CAAE,GACtC,OAAO,CAAC,UAAU,CAAC,CAmCrB"}
|
package/dist/tools/executor.js
CHANGED
|
@@ -35,6 +35,7 @@ var __importStar = (this && this.__importStar) || (function () {
|
|
|
35
35
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
36
36
|
exports.capExternalOutput = capExternalOutput;
|
|
37
37
|
exports.runCapture = runCapture;
|
|
38
|
+
exports.resolveRipgrepBinary = resolveRipgrepBinary;
|
|
38
39
|
exports.globToRegex = globToRegex;
|
|
39
40
|
exports.executeTool = executeTool;
|
|
40
41
|
const fs = __importStar(require("fs"));
|
|
@@ -1314,18 +1315,119 @@ function runCapture(cmd, args, opts) {
|
|
|
1314
1315
|
}
|
|
1315
1316
|
// Detect ripgrep once per process — preferred over grep (faster, respects .gitignore).
|
|
1316
1317
|
//
|
|
1317
|
-
// Binary resolution
|
|
1318
|
-
//
|
|
1319
|
-
//
|
|
1320
|
-
//
|
|
1318
|
+
// Binary resolution, in order:
|
|
1319
|
+
// 1. NEXRALL_RG_PATH (the VS Code extension points it at its own bundled rg),
|
|
1320
|
+
// 2. `rg` on PATH,
|
|
1321
|
+
// 3. the copy bundled inside every VS Code-family editor — the one ripgrep most dev
|
|
1322
|
+
// machines actually have, even when nothing put it on PATH. Layouts differ by
|
|
1323
|
+
// editor and version (asar.unpacked vs plain node_modules, per-arch subdir), so
|
|
1324
|
+
// probe a short list of known shapes instead of trusting any single path,
|
|
1325
|
+
// 4. common user-install dirs (Homebrew, cargo, /usr/local, snap).
|
|
1326
|
+
// Before the editor probe, a machine with no rg on PATH silently fell back to grep and
|
|
1327
|
+
// a full, ignore-unaware directory walk even though an editor next door shipped rg 15.
|
|
1321
1328
|
let _rgChecked = false;
|
|
1322
1329
|
let _rgAvailable = false;
|
|
1323
1330
|
let _rgBin = 'rg';
|
|
1331
|
+
/** One editor's bundled-ripgrep candidates, given its `Resources/app`-equivalent dir. */
|
|
1332
|
+
function bundledRipgrepPaths(appDir, platform, arch) {
|
|
1333
|
+
const exe = platform === 'win32' ? 'rg.exe' : 'rg';
|
|
1334
|
+
const other = arch === 'arm64' ? 'x64' : 'arm64';
|
|
1335
|
+
const out = [];
|
|
1336
|
+
for (const nm of ['node_modules.asar.unpacked', 'node_modules']) {
|
|
1337
|
+
out.push(path.join(appDir, nm, '@vscode', 'ripgrep', 'bin', exe));
|
|
1338
|
+
for (const tag of [`${platform}-${arch}`, `${platform}-${other}`]) {
|
|
1339
|
+
out.push(path.join(appDir, nm, '@vscode', 'ripgrep-universal', 'bin', tag, exe));
|
|
1340
|
+
}
|
|
1341
|
+
}
|
|
1342
|
+
return out;
|
|
1343
|
+
}
|
|
1344
|
+
function listDirsSafe(dir) {
|
|
1345
|
+
try {
|
|
1346
|
+
return fs
|
|
1347
|
+
.readdirSync(dir, { withFileTypes: true })
|
|
1348
|
+
.filter((d) => d.isDirectory())
|
|
1349
|
+
.map((d) => d.name);
|
|
1350
|
+
}
|
|
1351
|
+
catch {
|
|
1352
|
+
return [];
|
|
1353
|
+
}
|
|
1354
|
+
}
|
|
1355
|
+
function editorRipgrepCandidates(platform, arch) {
|
|
1356
|
+
const out = [];
|
|
1357
|
+
if (platform === 'darwin') {
|
|
1358
|
+
const apps = [
|
|
1359
|
+
'Visual Studio Code.app',
|
|
1360
|
+
'Visual Studio Code - Insiders.app',
|
|
1361
|
+
'VSCodium.app',
|
|
1362
|
+
'Cursor.app',
|
|
1363
|
+
'Windsurf.app',
|
|
1364
|
+
'Antigravity IDE.app',
|
|
1365
|
+
];
|
|
1366
|
+
for (const root of ['/Applications', path.join(os.homedir(), 'Applications')]) {
|
|
1367
|
+
for (const app of apps) {
|
|
1368
|
+
out.push(...bundledRipgrepPaths(path.join(root, app, 'Contents', 'Resources', 'app'), platform, arch));
|
|
1369
|
+
}
|
|
1370
|
+
}
|
|
1371
|
+
}
|
|
1372
|
+
else if (platform === 'win32') {
|
|
1373
|
+
const appDirs = [];
|
|
1374
|
+
if (process.env.LOCALAPPDATA) {
|
|
1375
|
+
appDirs.push(path.join(process.env.LOCALAPPDATA, 'Programs', 'Microsoft VS Code'));
|
|
1376
|
+
}
|
|
1377
|
+
if (process.env.ProgramFiles) {
|
|
1378
|
+
appDirs.push(path.join(process.env.ProgramFiles, 'Microsoft VS Code'));
|
|
1379
|
+
}
|
|
1380
|
+
appDirs.push(path.join(os.homedir(), 'AppData', 'Local', 'Programs', 'Microsoft VS Code'));
|
|
1381
|
+
for (const dir of appDirs)
|
|
1382
|
+
out.push(...bundledRipgrepPaths(path.join(dir, 'resources', 'app'), platform, arch));
|
|
1383
|
+
}
|
|
1384
|
+
else {
|
|
1385
|
+
for (const root of ['/usr/share/code', '/usr/lib/code', '/opt/visual-studio-code', '/snap/code/current']) {
|
|
1386
|
+
out.push(...bundledRipgrepPaths(path.join(root, 'resources', 'app'), platform, arch));
|
|
1387
|
+
}
|
|
1388
|
+
// Remote/server installs keep one directory per commit hash or server name.
|
|
1389
|
+
for (const base of ['.vscode-server', '.cursor-server', '.windsurf-server', '.vscodium-server']) {
|
|
1390
|
+
const home = path.join(os.homedir(), base);
|
|
1391
|
+
for (const dir of listDirsSafe(path.join(home, 'bin'))) {
|
|
1392
|
+
out.push(...bundledRipgrepPaths(path.join(home, 'bin', dir), platform, arch));
|
|
1393
|
+
}
|
|
1394
|
+
for (const dir of listDirsSafe(path.join(home, 'cli', 'servers'))) {
|
|
1395
|
+
out.push(...bundledRipgrepPaths(path.join(home, 'cli', 'servers', dir, 'server'), platform, arch));
|
|
1396
|
+
out.push(...bundledRipgrepPaths(path.join(home, 'cli', 'servers', dir), platform, arch));
|
|
1397
|
+
}
|
|
1398
|
+
}
|
|
1399
|
+
}
|
|
1400
|
+
return out;
|
|
1401
|
+
}
|
|
1402
|
+
function ripgrepCandidates() {
|
|
1403
|
+
const out = [];
|
|
1404
|
+
const env = process.env.NEXRALL_RG_PATH?.trim();
|
|
1405
|
+
if (env)
|
|
1406
|
+
out.push(env);
|
|
1407
|
+
out.push('rg');
|
|
1408
|
+
if (process.platform === 'win32')
|
|
1409
|
+
out.push('rg.exe');
|
|
1410
|
+
out.push(...editorRipgrepCandidates(process.platform, process.arch));
|
|
1411
|
+
const home = os.homedir();
|
|
1412
|
+
if (process.platform === 'darwin') {
|
|
1413
|
+
out.push('/opt/homebrew/bin/rg', '/usr/local/bin/rg', path.join(home, '.cargo', 'bin', 'rg'));
|
|
1414
|
+
}
|
|
1415
|
+
else if (process.platform === 'win32') {
|
|
1416
|
+
out.push(path.join(home, '.cargo', 'bin', 'rg.exe'));
|
|
1417
|
+
}
|
|
1418
|
+
else {
|
|
1419
|
+
out.push('/usr/bin/rg', '/usr/local/bin/rg', '/snap/bin/rg', path.join(home, '.local', 'bin', 'rg'), path.join(home, '.cargo', 'bin', 'rg'));
|
|
1420
|
+
}
|
|
1421
|
+
return [...new Set(out)];
|
|
1422
|
+
}
|
|
1324
1423
|
function ripgrepAvailable() {
|
|
1325
1424
|
if (!_rgChecked) {
|
|
1326
1425
|
_rgChecked = true;
|
|
1327
|
-
const
|
|
1328
|
-
|
|
1426
|
+
for (const bin of ripgrepCandidates()) {
|
|
1427
|
+
// Skip absolute paths that plainly don't exist (other machines' editor bundles):
|
|
1428
|
+
// a miss still costs a spawnSync, and the editor list alone can hold dozens.
|
|
1429
|
+
if (path.isAbsolute(bin) && !fs.existsSync(bin))
|
|
1430
|
+
continue;
|
|
1329
1431
|
const probe = (0, child_process_1.spawnSync)(bin, ['--version'], { encoding: 'utf-8', timeout: 3000 });
|
|
1330
1432
|
if (probe.status === 0) {
|
|
1331
1433
|
_rgBin = bin;
|
|
@@ -1340,6 +1442,14 @@ function ripgrepAvailable() {
|
|
|
1340
1442
|
function rgBin() {
|
|
1341
1443
|
return _rgBin;
|
|
1342
1444
|
}
|
|
1445
|
+
/**
|
|
1446
|
+
* Path of the ripgrep binary this process would use — the NEXRALL_RG_PATH override, `rg`
|
|
1447
|
+
* from PATH, or an editor-bundled copy — or undefined when the grep fallback is in charge.
|
|
1448
|
+
* `nex doctor` reports through this so its verdict matches what the search tools do.
|
|
1449
|
+
*/
|
|
1450
|
+
function resolveRipgrepBinary() {
|
|
1451
|
+
return ripgrepAvailable() ? _rgBin : undefined;
|
|
1452
|
+
}
|
|
1343
1453
|
// ── Cross-file breakage warning ──────────────────────────────────────────────
|
|
1344
1454
|
// After an edit removes/renames an exported symbol, scan the rest of the repo for
|
|
1345
1455
|
// surviving references. Returns a short warning string (or '' when clean). Best-
|
package/dist/types.d.ts
CHANGED
|
@@ -495,6 +495,16 @@ export interface AgentLoopOptions {
|
|
|
495
495
|
* @peer-name: ...") the way Claude Code's own inline preview does.
|
|
496
496
|
*/
|
|
497
497
|
onPeerMessage?: (messages: import('./agent/peerTransport').PeerMessage[]) => void;
|
|
498
|
+
/**
|
|
499
|
+
* Fired once per turn boundary where cross-session task-board notes (see
|
|
500
|
+
* agent/taskEvents.ts) are being folded into the conversation — lets a UI
|
|
501
|
+
* show the same event the model is about to see ("the task you were
|
|
502
|
+
* waiting on just completed") instead of the user having to run `nex tasks`
|
|
503
|
+
* to find out. Only CONSEQUENTIAL events produce notes (`completed` /
|
|
504
|
+
* `reopened` / `removed` with live waiters), so a UI on this callback is
|
|
505
|
+
* exactly as quiet as the model is.
|
|
506
|
+
*/
|
|
507
|
+
onTaskNotes?: (notes: import('./agent/taskEvents').TaskEventNote[]) => void;
|
|
498
508
|
/**
|
|
499
509
|
* This session's own peer identity (see agent/peerRegistry.ts's
|
|
500
510
|
* registerPeer), if it registered as a discoverable peer at all —
|