@nexrall/code-core 1.4.74 → 1.4.75
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -3
- package/dist/index.d.ts +1 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -0
- package/dist/plugins/eval.d.ts +199 -0
- package/dist/plugins/eval.d.ts.map +1 -0
- package/dist/plugins/eval.js +470 -0
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -12,6 +12,7 @@ npm install @nexrall/code-core
|
|
|
12
12
|
|--------|-------------|
|
|
13
13
|
| `@nexrall/code-core` | Everything via the root export |
|
|
14
14
|
| `@nexrall/code-core/agent` | `runAgentLoop` — the main agentic loop |
|
|
15
|
+
| `@nexrall/code-core/agent-types` | `loadAgentTypes`, `findAgentType` — sub-agent type definitions |
|
|
15
16
|
| `@nexrall/code-core/tools` | `executeTool` — built-in tool executor (bash, file I/O, search…) |
|
|
16
17
|
| `@nexrall/code-core/symbols` | `getSymbols`, `getWorkspaceSymbols` — regex-based LSP-lite scanner |
|
|
17
18
|
| `@nexrall/code-core/checkpoint` | `CheckpointManager` — persistent rewind / rollback |
|
|
@@ -33,7 +34,7 @@ const messages = [
|
|
|
33
34
|
];
|
|
34
35
|
|
|
35
36
|
const opts: AgentLoopOptions = {
|
|
36
|
-
model: '
|
|
37
|
+
model: 'claude-sonnet-5-5', // any catalogue model id; the tier aliases still work
|
|
37
38
|
workDir: process.cwd(),
|
|
38
39
|
env: { platform: 'node', cwd: process.cwd(), shell: 'bash' },
|
|
39
40
|
clientType: 'cli',
|
|
@@ -53,13 +54,13 @@ const history = await runAgentLoop(messages, opts);
|
|
|
53
54
|
console.log('Done —', history.length, 'messages');
|
|
54
55
|
```
|
|
55
56
|
|
|
56
|
-
> **Requires authentication.** The agent streams through the Nexrall API — users must be logged in via `
|
|
57
|
+
> **Requires authentication.** The agent streams through the Nexrall API — users must be logged in via `nex auth` (or set the `NEXRALL_TOKEN` env var).
|
|
57
58
|
|
|
58
59
|
## Agent loop options
|
|
59
60
|
|
|
60
61
|
```ts
|
|
61
62
|
interface AgentLoopOptions {
|
|
62
|
-
model
|
|
63
|
+
model?: string; // e.g. 'claude-sonnet-5-5', 'gpt-6-astra', 'deepseek-flash'
|
|
63
64
|
workDir: string;
|
|
64
65
|
env: EnvContext;
|
|
65
66
|
clientType?: 'cli' | 'vscode';
|
package/dist/index.d.ts
CHANGED
|
@@ -36,6 +36,7 @@ export * from './plugins/index';
|
|
|
36
36
|
export * from './plugins/installer';
|
|
37
37
|
export * from './plugins/sources';
|
|
38
38
|
export * from './plugins/data';
|
|
39
|
+
export * from './plugins/eval';
|
|
39
40
|
export * from './agent/trust';
|
|
40
41
|
export * from './agent/crossProcessLock';
|
|
41
42
|
export * from './agent/worktree';
|
package/dist/index.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA,cAAc,SAAS,CAAC;AACxB,cAAc,kBAAkB,CAAC;AACjC,cAAc,cAAc,CAAC;AAC7B,cAAc,cAAc,CAAC;AAC7B,cAAc,kBAAkB,CAAC;AACjC,cAAc,cAAc,CAAC;AAC7B,cAAc,uBAAuB,CAAC;AACtC,cAAc,wBAAwB,CAAC;AACvC,cAAc,uBAAuB,CAAC;AACtC,cAAc,0BAA0B,CAAC;AACzC,cAAc,sBAAsB,CAAC;AACrC,cAAc,mBAAmB,CAAC;AAClC,cAAc,eAAe,CAAC;AAC9B,cAAc,uBAAuB,CAAC;AACtC,cAAc,eAAe,CAAC;AAC9B,cAAc,gBAAgB,CAAC;AAC/B,cAAc,gBAAgB,CAAC;AAC/B,cAAc,4BAA4B,CAAC;AAC3C,cAAc,cAAc,CAAC;AAC7B,cAAc,kBAAkB,CAAC;AACjC,cAAc,oBAAoB,CAAC;AACnC,cAAc,eAAe,CAAC;AAC9B,cAAc,cAAc,CAAC;AAC7B,cAAc,aAAa,CAAC;AAC5B,cAAc,sBAAsB,CAAC;AACrC,cAAc,mBAAmB,CAAC;AAClC,cAAc,oBAAoB,CAAC;AACnC,cAAc,kBAAkB,CAAC;AACjC,cAAc,uBAAuB,CAAC;AACtC,cAAc,0BAA0B,CAAC;AACzC,cAAc,qBAAqB,CAAC;AACpC,cAAc,0BAA0B,CAAC;AACzC,cAAc,4BAA4B,CAAC;AAC3C,cAAc,2BAA2B,CAAC;AAC1C,cAAc,iBAAiB,CAAC;AAChC,cAAc,qBAAqB,CAAC;AACpC,cAAc,mBAAmB,CAAC;AAClC,cAAc,gBAAgB,CAAC;AAC/B,cAAc,eAAe,CAAC;AAC9B,cAAc,0BAA0B,CAAC;AACzC,cAAc,kBAAkB,CAAC;AACjC,cAAc,6BAA6B,CAAC;AAC5C,cAAc,qBAAqB,CAAC;AACpC,cAAc,oBAAoB,CAAC;AACnC,cAAc,sBAAsB,CAAC;AACrC,cAAc,uBAAuB,CAAC;AACtC,cAAc,iBAAiB,CAAC"}
|
|
1
|
+
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA,cAAc,SAAS,CAAC;AACxB,cAAc,kBAAkB,CAAC;AACjC,cAAc,cAAc,CAAC;AAC7B,cAAc,cAAc,CAAC;AAC7B,cAAc,kBAAkB,CAAC;AACjC,cAAc,cAAc,CAAC;AAC7B,cAAc,uBAAuB,CAAC;AACtC,cAAc,wBAAwB,CAAC;AACvC,cAAc,uBAAuB,CAAC;AACtC,cAAc,0BAA0B,CAAC;AACzC,cAAc,sBAAsB,CAAC;AACrC,cAAc,mBAAmB,CAAC;AAClC,cAAc,eAAe,CAAC;AAC9B,cAAc,uBAAuB,CAAC;AACtC,cAAc,eAAe,CAAC;AAC9B,cAAc,gBAAgB,CAAC;AAC/B,cAAc,gBAAgB,CAAC;AAC/B,cAAc,4BAA4B,CAAC;AAC3C,cAAc,cAAc,CAAC;AAC7B,cAAc,kBAAkB,CAAC;AACjC,cAAc,oBAAoB,CAAC;AACnC,cAAc,eAAe,CAAC;AAC9B,cAAc,cAAc,CAAC;AAC7B,cAAc,aAAa,CAAC;AAC5B,cAAc,sBAAsB,CAAC;AACrC,cAAc,mBAAmB,CAAC;AAClC,cAAc,oBAAoB,CAAC;AACnC,cAAc,kBAAkB,CAAC;AACjC,cAAc,uBAAuB,CAAC;AACtC,cAAc,0BAA0B,CAAC;AACzC,cAAc,qBAAqB,CAAC;AACpC,cAAc,0BAA0B,CAAC;AACzC,cAAc,4BAA4B,CAAC;AAC3C,cAAc,2BAA2B,CAAC;AAC1C,cAAc,iBAAiB,CAAC;AAChC,cAAc,qBAAqB,CAAC;AACpC,cAAc,mBAAmB,CAAC;AAClC,cAAc,gBAAgB,CAAC;AAC/B,cAAc,gBAAgB,CAAC;AAC/B,cAAc,eAAe,CAAC;AAC9B,cAAc,0BAA0B,CAAC;AACzC,cAAc,kBAAkB,CAAC;AACjC,cAAc,6BAA6B,CAAC;AAC5C,cAAc,qBAAqB,CAAC;AACpC,cAAc,oBAAoB,CAAC;AACnC,cAAc,sBAAsB,CAAC;AACrC,cAAc,uBAAuB,CAAC;AACtC,cAAc,iBAAiB,CAAC"}
|
package/dist/index.js
CHANGED
|
@@ -52,6 +52,7 @@ __exportStar(require("./plugins/index"), exports);
|
|
|
52
52
|
__exportStar(require("./plugins/installer"), exports);
|
|
53
53
|
__exportStar(require("./plugins/sources"), exports);
|
|
54
54
|
__exportStar(require("./plugins/data"), exports);
|
|
55
|
+
__exportStar(require("./plugins/eval"), exports);
|
|
55
56
|
__exportStar(require("./agent/trust"), exports);
|
|
56
57
|
__exportStar(require("./agent/crossProcessLock"), exports);
|
|
57
58
|
__exportStar(require("./agent/worktree"), exports);
|
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
export declare const EVAL_SCHEMA_VERSION = 1;
|
|
2
|
+
/** CC's default: each case runs three times. */
|
|
3
|
+
export declare const EVAL_DEFAULT_RUNS = 3;
|
|
4
|
+
export declare const EVAL_MAX_RUNS = 50;
|
|
5
|
+
export declare const EVAL_DEFAULT_THRESHOLD = 1;
|
|
6
|
+
export declare const EVAL_DEFAULT_MAX_TURNS = 10;
|
|
7
|
+
export declare const EVAL_DEFAULT_TIMEOUT_S = 300;
|
|
8
|
+
/** CC caps both; larger values are refused, not clamped silently. */
|
|
9
|
+
export declare const EVAL_MAX_TURNS_CEILING = 200;
|
|
10
|
+
export declare const EVAL_MAX_TIMEOUT_S = 3600;
|
|
11
|
+
export type GraderType = 'regex' | 'tool_used' | 'tool_order' | 'file_exists' | 'llm' | 'baseline';
|
|
12
|
+
/**
|
|
13
|
+
* Which arm a grader scores in.
|
|
14
|
+
* · 'both' (default) — counted for the plugin arm AND the no-plugin baseline.
|
|
15
|
+
* · 'with-only' — counted only when the plugin was loaded (CC's rule for the
|
|
16
|
+
* two-arm runs: a grader that can only pass WITH the plugin must not drag
|
|
17
|
+
* the baseline down).
|
|
18
|
+
*/
|
|
19
|
+
export type GraderArm = 'both' | 'with-only';
|
|
20
|
+
export interface EvalGrader {
|
|
21
|
+
/** File name inside graders/, for reporting. */
|
|
22
|
+
file: string;
|
|
23
|
+
type: GraderType;
|
|
24
|
+
weight: number;
|
|
25
|
+
arm: GraderArm;
|
|
26
|
+
pattern?: string;
|
|
27
|
+
flags?: string;
|
|
28
|
+
/** 'contains' (default) | 'not_contains' | 'count:<n>' */
|
|
29
|
+
match?: string;
|
|
30
|
+
/** 'last_message' (default) | 'trace' | 'file' */
|
|
31
|
+
target?: string;
|
|
32
|
+
/** target 'file': project-relative path or glob to read. */
|
|
33
|
+
pathPattern?: string;
|
|
34
|
+
tool?: string;
|
|
35
|
+
inputMatch?: string;
|
|
36
|
+
min?: number;
|
|
37
|
+
max?: number;
|
|
38
|
+
before?: string;
|
|
39
|
+
after?: string;
|
|
40
|
+
path?: string;
|
|
41
|
+
exists?: boolean;
|
|
42
|
+
criteria?: string;
|
|
43
|
+
/** llm: 'last_message' (default) | 'trace'. */
|
|
44
|
+
focus?: string;
|
|
45
|
+
/** baseline: the reference transcript file (case-relative). */
|
|
46
|
+
baselineFile?: string;
|
|
47
|
+
}
|
|
48
|
+
export interface EvalCase {
|
|
49
|
+
name: string;
|
|
50
|
+
dir: string;
|
|
51
|
+
prompt: string;
|
|
52
|
+
runs: number;
|
|
53
|
+
maxTurns: number;
|
|
54
|
+
timeoutSeconds: number;
|
|
55
|
+
allowedTools: string[];
|
|
56
|
+
tags: string[];
|
|
57
|
+
expectedOutcome?: string;
|
|
58
|
+
graders: EvalGrader[];
|
|
59
|
+
}
|
|
60
|
+
export interface DiscoveredSuite {
|
|
61
|
+
cases: EvalCase[];
|
|
62
|
+
/** Case dirs that failed to parse — reported, never silently dropped. */
|
|
63
|
+
errors: Array<{
|
|
64
|
+
dir: string;
|
|
65
|
+
error: string;
|
|
66
|
+
}>;
|
|
67
|
+
}
|
|
68
|
+
/** One tool call as the child's stream reported it. */
|
|
69
|
+
export interface EvalToolCall {
|
|
70
|
+
tool: string;
|
|
71
|
+
input: unknown;
|
|
72
|
+
}
|
|
73
|
+
/**
|
|
74
|
+
* What ONE run produced. `readFile` resolves a project-relative path or glob
|
|
75
|
+
* against the run's workspace — the caller owns the filesystem, so the graders
|
|
76
|
+
* stay pure.
|
|
77
|
+
*/
|
|
78
|
+
export interface EvalRunTrace {
|
|
79
|
+
lastMessage: string;
|
|
80
|
+
toolCalls: EvalToolCall[];
|
|
81
|
+
files: string[];
|
|
82
|
+
readFile: (relOrGlob: string) => string | null;
|
|
83
|
+
/** USD when the server reported it; null = unreported (never invented). */
|
|
84
|
+
costUsd: number | null;
|
|
85
|
+
tokens: number;
|
|
86
|
+
exitCode: number;
|
|
87
|
+
error?: string;
|
|
88
|
+
}
|
|
89
|
+
export interface GraderVerdict {
|
|
90
|
+
passed: boolean;
|
|
91
|
+
why: string;
|
|
92
|
+
}
|
|
93
|
+
/** The injected model call for llm/baseline graders. null = could not judge. */
|
|
94
|
+
export type EvalJudge = (args: {
|
|
95
|
+
criteria: string;
|
|
96
|
+
focusText: string;
|
|
97
|
+
baselineText?: string;
|
|
98
|
+
}) => Promise<boolean | null>;
|
|
99
|
+
export interface CaseRunOutcome {
|
|
100
|
+
arm: 'with' | 'without';
|
|
101
|
+
run: number;
|
|
102
|
+
score: number;
|
|
103
|
+
graderResults: Array<{
|
|
104
|
+
file: string;
|
|
105
|
+
type: GraderType;
|
|
106
|
+
weight: number;
|
|
107
|
+
passed: boolean;
|
|
108
|
+
why: string;
|
|
109
|
+
}>;
|
|
110
|
+
costUsd: number | null;
|
|
111
|
+
costReported: boolean;
|
|
112
|
+
durationMs: number;
|
|
113
|
+
error?: string;
|
|
114
|
+
}
|
|
115
|
+
export interface CaseOutcome {
|
|
116
|
+
name: string;
|
|
117
|
+
with: CaseRunOutcome[];
|
|
118
|
+
without: CaseRunOutcome[];
|
|
119
|
+
withScore: number | null;
|
|
120
|
+
withoutScore: number | null;
|
|
121
|
+
delta: number | null;
|
|
122
|
+
passed: boolean;
|
|
123
|
+
}
|
|
124
|
+
export interface EvalReport {
|
|
125
|
+
schemaVersion: number;
|
|
126
|
+
plugin: string;
|
|
127
|
+
startedAt: number;
|
|
128
|
+
durationMs: number;
|
|
129
|
+
threshold: number;
|
|
130
|
+
runs: number;
|
|
131
|
+
judgeModel?: string;
|
|
132
|
+
partial?: boolean;
|
|
133
|
+
partialReason?: string;
|
|
134
|
+
cases: CaseOutcome[];
|
|
135
|
+
errors: Array<{
|
|
136
|
+
dir: string;
|
|
137
|
+
error: string;
|
|
138
|
+
}>;
|
|
139
|
+
aggregates: {
|
|
140
|
+
casesPassed: number;
|
|
141
|
+
casesTotal: number;
|
|
142
|
+
overallScore: number;
|
|
143
|
+
meanDelta: number | null;
|
|
144
|
+
};
|
|
145
|
+
costUsd: number;
|
|
146
|
+
costMissing: number;
|
|
147
|
+
}
|
|
148
|
+
/** `[a, b]` / `a, b` / `a` → ['a','b'] — the two shapes real files use. */
|
|
149
|
+
export declare function parseList(value: string | undefined): string[];
|
|
150
|
+
/**
|
|
151
|
+
* One grader file → EvalGrader. Throws on anything that would make the grader
|
|
152
|
+
* silently unscorable (unknown type, missing pattern/tool/path, bad count).
|
|
153
|
+
*/
|
|
154
|
+
export declare function parseGraderFile(file: string, raw: string): EvalGrader;
|
|
155
|
+
/** One case directory → EvalCase. See the header for the layout. */
|
|
156
|
+
export declare function parseEvalCase(dir: string): EvalCase;
|
|
157
|
+
/** Every case under `<pluginDir>/evals` (the results dir is never a case). */
|
|
158
|
+
export declare function discoverEvalCases(pluginDir: string): DiscoveredSuite;
|
|
159
|
+
/** The text a grader's target names. Exposed for tests and the report's `why`. */
|
|
160
|
+
export declare function targetText(g: EvalGrader, trace: EvalRunTrace): string | null;
|
|
161
|
+
/**
|
|
162
|
+
* Score ONE grader over ONE run. Async because llm/baseline call the judge.
|
|
163
|
+
* A grader that cannot be evaluated FAILS (with the reason in `why`) — it must
|
|
164
|
+
* never pass by accident, or a broken suite would look green.
|
|
165
|
+
*/
|
|
166
|
+
export declare function evaluateGrader(g: EvalGrader, trace: EvalRunTrace, judge: EvalJudge | undefined): Promise<GraderVerdict>;
|
|
167
|
+
/** Which graders count in an arm: with-only graders never score the baseline. */
|
|
168
|
+
export declare function gradersForArm(graders: EvalGrader[], arm: 'with' | 'without'): EvalGrader[];
|
|
169
|
+
/** Score one run: weighted fraction of its graders that passed. */
|
|
170
|
+
export declare function scoreRun(graders: EvalGrader[], trace: EvalRunTrace, judge: EvalJudge | undefined, arm: 'with' | 'without'): Promise<{
|
|
171
|
+
score: number;
|
|
172
|
+
results: CaseRunOutcome['graderResults'];
|
|
173
|
+
}>;
|
|
174
|
+
/**
|
|
175
|
+
* Fold run outcomes into the report. Case score = mean of the plugin-arm run
|
|
176
|
+
* scores; `delta` = with − without (null when the baseline did not run).
|
|
177
|
+
* A case PASSES at ≥ threshold (CC's default: every grader, every run).
|
|
178
|
+
*/
|
|
179
|
+
export declare function aggregateReport(i: {
|
|
180
|
+
plugin: string;
|
|
181
|
+
startedAt: number;
|
|
182
|
+
durationMs: number;
|
|
183
|
+
threshold: number;
|
|
184
|
+
runs: number;
|
|
185
|
+
judgeModel?: string;
|
|
186
|
+
cases: Array<{
|
|
187
|
+
name: string;
|
|
188
|
+
with: CaseRunOutcome[];
|
|
189
|
+
without: CaseRunOutcome[];
|
|
190
|
+
}>;
|
|
191
|
+
errors: Array<{
|
|
192
|
+
dir: string;
|
|
193
|
+
error: string;
|
|
194
|
+
}>;
|
|
195
|
+
partial?: {
|
|
196
|
+
reason: string;
|
|
197
|
+
};
|
|
198
|
+
}): EvalReport;
|
|
199
|
+
//# sourceMappingURL=eval.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"eval.d.ts","sourceRoot":"","sources":["../../src/plugins/eval.ts"],"names":[],"mappings":"AAiCA,eAAO,MAAM,mBAAmB,IAAI,CAAC;AACrC,gDAAgD;AAChD,eAAO,MAAM,iBAAiB,IAAI,CAAC;AACnC,eAAO,MAAM,aAAa,KAAK,CAAC;AAChC,eAAO,MAAM,sBAAsB,IAAM,CAAC;AAC1C,eAAO,MAAM,sBAAsB,KAAK,CAAC;AACzC,eAAO,MAAM,sBAAsB,MAAM,CAAC;AAC1C,qEAAqE;AACrE,eAAO,MAAM,sBAAsB,MAAM,CAAC;AAC1C,eAAO,MAAM,kBAAkB,OAAO,CAAC;AAMvC,MAAM,MAAM,UAAU,GAAG,OAAO,GAAG,WAAW,GAAG,YAAY,GAAG,aAAa,GAAG,KAAK,GAAG,UAAU,CAAC;AAEnG;;;;;;GAMG;AACH,MAAM,MAAM,SAAS,GAAG,MAAM,GAAG,WAAW,CAAC;AAE7C,MAAM,WAAW,UAAU;IACzB,gDAAgD;IAChD,IAAI,EAAE,MAAM,CAAC;IACb,IAAI,EAAE,UAAU,CAAC;IACjB,MAAM,EAAE,MAAM,CAAC;IACf,GAAG,EAAE,SAAS,CAAC;IAEf,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,0DAA0D;IAC1D,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,kDAAkD;IAClD,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,4DAA4D;IAC5D,WAAW,CAAC,EAAE,MAAM,CAAC;IAErB,IAAI,CAAC,EAAE,MAAM,CAAC;IACd,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,GAAG,CAAC,EAAE,MAAM,CAAC;IACb,GAAG,CAAC,EAAE,MAAM,CAAC;IAEb,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,KAAK,CAAC,EAAE,MAAM,CAAC;IAEf,IAAI,CAAC,EAAE,MAAM,CAAC;IACd,MAAM,CAAC,EAAE,OAAO,CAAC;IAEjB,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,+CAA+C;IAC/C,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,+DAA+D;IAC/D,YAAY,CAAC,EAAE,MAAM,CAAC;CACvB;AAED,MAAM,WAAW,QAAQ;IACvB,IAAI,EAAE,MAAM,CAAC;IACb,GAAG,EAAE,MAAM,CAAC;IACZ,MAAM,EAAE,MAAM,CAAC;IACf,IAAI,EAAE,MAAM,CAAC;IACb,QAAQ,EAAE,MAAM,CAAC;IACjB,cAAc,EAAE,MAAM,CAAC;IACvB,YAAY,EAAE,MAAM,EAAE,CAAC;IACvB,IAAI,EAAE,MAAM,EAAE,CAAC;IACf,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB,OAAO,EAAE,UAAU,EAAE,CAAC;CACvB;AAED,MAAM,WAAW,eAAe;IAC9B,KAAK,EAAE,QAAQ,EAAE,CAAC;IAClB,yEAAyE;IACzE,MAAM,EAAE,KAAK,CAAC;QAAE,GAAG,EAAE,MAAM,CAAC;QAAC,KAAK,EAAE,MAAM,CAAA;KAAE,CAAC,CAAC;CAC/C;AAED,uDAAuD;AACvD,MAAM,WAAW,YAAY;IAC3B,IAAI,EAAE,MAAM,CAAC;IACb,KAAK,EAAE,OAAO,CAAC;CAChB;AAED;;;;GAIG;AACH,MAAM,WAAW,YAAY;IAC3B,WAAW,EAAE,MAAM,CAAC;IACpB,SAAS,EAAE,YAAY,EAAE,CAAC;IAC1B,KAAK,EAAE,MAAM,EAAE,CAAC;IAChB,QAAQ,EAAE,CAAC,SAAS,EAAE,MAAM,KAAK,MAAM,GAAG,IAAI,CAAC;IAC/C,2EAA2E;IAC3E,OAAO,EAAE,MAAM,GAAG,IAAI,CAAC;IACvB,MAAM,EAAE,MAAM,CAAC;IACf,QAAQ,EAAE,MAAM,CAAC;IACjB,KAAK,CAAC,EAAE,MAAM,CAAC;CAChB;AAED,MAAM,WAAW,aAAa;IAC5B,MAAM,EAAE,OAAO,CAAC;IAChB,GAAG,EAAE,MAAM,CAAC;CACb;AAED,gFAAgF;AAChF,MAAM,MAAM,SAAS,GAAG,CAAC,IAAI,EAAE;IAC7B,QAAQ,EAAE,MAAM,CAAC;IACjB,SAAS,EAAE,MAAM,CAAC;IAClB,YAAY,CAAC,EAAE,MAAM,CAAC;CACvB,KAAK,OAAO,CAAC,OAAO,GAAG,IAAI,CAAC,CAAC;AAE9B,MAAM,WAAW,cAAc;IAC7B,GAAG,EAAE,MAAM,GAAG,SAAS,CAAC;IACxB,GAAG,EAAE,MAAM,CAAC;IACZ,KAAK,EAAE,MAAM,CAAC;IACd,aAAa,EAAE,KAAK,CAAC;QAAE,IAAI,EAAE,MAAM,CAAC;QAAC,IAAI,EAAE,UAAU,CAAC;QAAC,MAAM,EAAE,MAAM,CAAC;QAAC,MAAM,EAAE,OAAO,CAAC;QAAC,GAAG,EAAE,MAAM,CAAA;KAAE,CAAC,CAAC;IACvG,OAAO,EAAE,MAAM,GAAG,IAAI,CAAC;IACvB,YAAY,EAAE,OAAO,CAAC;IACtB,UAAU,EAAE,MAAM,CAAC;IACnB,KAAK,CAAC,EAAE,MAAM,CAAC;CAChB;AAED,MAAM,WAAW,WAAW;IAC1B,IAAI,EAAE,MAAM,CAAC;IACb,IAAI,EAAE,cAAc,EAAE,CAAC;IACvB,OAAO,EAAE,cAAc,EAAE,CAAC;IAC1B,SAAS,EAAE,MAAM,GAAG,IAAI,CAAC;IACzB,YAAY,EAAE,MAAM,GAAG,IAAI,CAAC;IAC5B,KAAK,EAAE,MAAM,GAAG,IAAI,CAAC;IACrB,MAAM,EAAE,OAAO,CAAC;CACjB;AAED,MAAM,WAAW,UAAU;IACzB,aAAa,EAAE,MAAM,CAAC;IACtB,MAAM,EAAE,MAAM,CAAC;IACf,SAAS,EAAE,MAAM,CAAC;IAClB,UAAU,EAAE,MAAM,CAAC;IACnB,SAAS,EAAE,MAAM,CAAC;IAClB,IAAI,EAAE,MAAM,CAAC;IACb,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,OAAO,CAAC,EAAE,OAAO,CAAC;IAClB,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB,KAAK,EAAE,WAAW,EAAE,CAAC;IACrB,MAAM,EAAE,KAAK,CAAC;QAAE,GAAG,EAAE,MAAM,CAAC;QAAC,KAAK,EAAE,MAAM,CAAA;KAAE,CAAC,CAAC;IAC9C,UAAU,EAAE;QACV,WAAW,EAAE,MAAM,CAAC;QACpB,UAAU,EAAE,MAAM,CAAC;QACnB,YAAY,EAAE,MAAM,CAAC;QACrB,SAAS,EAAE,MAAM,GAAG,IAAI,CAAC;KAC1B,CAAC;IACF,OAAO,EAAE,MAAM,CAAC;IAChB,WAAW,EAAE,MAAM,CAAC;CACrB;AAID,2EAA2E;AAC3E,wBAAgB,SAAS,CAAC,KAAK,EAAE,MAAM,GAAG,SAAS,GAAG,MAAM,EAAE,CAO7D;AAUD;;;GAGG;AACH,wBAAgB,eAAe,CAAC,IAAI,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,UAAU,CAoErE;AAED,oEAAoE;AACpE,wBAAgB,aAAa,CAAC,GAAG,EAAE,MAAM,GAAG,QAAQ,CAmDnD;AAED,8EAA8E;AAC9E,wBAAgB,iBAAiB,CAAC,SAAS,EAAE,MAAM,GAAG,eAAe,CA6BpE;AAID,kFAAkF;AAClF,wBAAgB,UAAU,CAAC,CAAC,EAAE,UAAU,EAAE,KAAK,EAAE,YAAY,GAAG,MAAM,GAAG,IAAI,CAY5E;AAkBD;;;;GAIG;AACH,wBAAsB,cAAc,CAClC,CAAC,EAAE,UAAU,EACb,KAAK,EAAE,YAAY,EACnB,KAAK,EAAE,SAAS,GAAG,SAAS,GAC3B,OAAO,CAAC,aAAa,CAAC,CAgFxB;AAED,iFAAiF;AACjF,wBAAgB,aAAa,CAAC,OAAO,EAAE,UAAU,EAAE,EAAE,GAAG,EAAE,MAAM,GAAG,SAAS,GAAG,UAAU,EAAE,CAE1F;AAED,mEAAmE;AACnE,wBAAsB,QAAQ,CAC5B,OAAO,EAAE,UAAU,EAAE,EACrB,KAAK,EAAE,YAAY,EACnB,KAAK,EAAE,SAAS,GAAG,SAAS,EAC5B,GAAG,EAAE,MAAM,GAAG,SAAS,GACtB,OAAO,CAAC;IAAE,KAAK,EAAE,MAAM,CAAC;IAAC,OAAO,EAAE,cAAc,CAAC,eAAe,CAAC,CAAA;CAAE,CAAC,CAatE;AAID;;;;GAIG;AACH,wBAAgB,eAAe,CAAC,CAAC,EAAE;IACjC,MAAM,EAAE,MAAM,CAAC;IACf,SAAS,EAAE,MAAM,CAAC;IAClB,UAAU,EAAE,MAAM,CAAC;IACnB,SAAS,EAAE,MAAM,CAAC;IAClB,IAAI,EAAE,MAAM,CAAC;IACb,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,KAAK,EAAE,KAAK,CAAC;QACX,IAAI,EAAE,MAAM,CAAC;QACb,IAAI,EAAE,cAAc,EAAE,CAAC;QACvB,OAAO,EAAE,cAAc,EAAE,CAAC;KAC3B,CAAC,CAAC;IACH,MAAM,EAAE,KAAK,CAAC;QAAE,GAAG,EAAE,MAAM,CAAC;QAAC,KAAK,EAAE,MAAM,CAAA;KAAE,CAAC,CAAC;IAC9C,OAAO,CAAC,EAAE;QAAE,MAAM,EAAE,MAAM,CAAA;KAAE,CAAC;CAC9B,GAAG,UAAU,CAsCb"}
|
|
@@ -0,0 +1,470 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
// S21 — `nex plugin eval`: run a plugin against a suite of test cases and score
|
|
3
|
+
// the results (Claude Code's `claude plugin eval`, shipped in their W37).
|
|
4
|
+
//
|
|
5
|
+
// This file is the PURE half: case discovery/parsing, the grader vocabulary,
|
|
6
|
+
// the trace-based graders, and the scoring/aggregation. The caller owns ALL
|
|
7
|
+
// I/O — spawning each run, reading the child's event log, and judging (the
|
|
8
|
+
// `llm`/`baseline` graders call a model through an injected `judge` function so
|
|
9
|
+
// this module stays unit-testable without a backend).
|
|
10
|
+
//
|
|
11
|
+
// Suite layout (CC's, minus what we do not support yet — see below):
|
|
12
|
+
//
|
|
13
|
+
// my-plugin/
|
|
14
|
+
// evals/
|
|
15
|
+
// first-case/
|
|
16
|
+
// prompt.md # frontmatter = case fields; body = the prompt
|
|
17
|
+
// graders/
|
|
18
|
+
// criteria.md # frontmatter = grader; body = llm criteria
|
|
19
|
+
// skill-fired.md
|
|
20
|
+
// results/ # written by each run; never discovered as a case
|
|
21
|
+
//
|
|
22
|
+
// Deliberately NOT supported yet (refused loudly, never silently skipped):
|
|
23
|
+
// · case.yaml — needs a real YAML parser, which this repo does not carry
|
|
24
|
+
// (the frontmatter reader is "not a full YAML parser" by design). A case
|
|
25
|
+
// that ships one is an ERROR, because silently ignoring its fields would
|
|
26
|
+
// run a DIFFERENT suite than the author wrote.
|
|
27
|
+
// · mock MCP servers (`evals/mocks/`) — a follow-up; a case using them is
|
|
28
|
+
// likewise refused rather than run without the mocks.
|
|
29
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
30
|
+
if (k2 === undefined) k2 = k;
|
|
31
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
32
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
33
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
34
|
+
}
|
|
35
|
+
Object.defineProperty(o, k2, desc);
|
|
36
|
+
}) : (function(o, m, k, k2) {
|
|
37
|
+
if (k2 === undefined) k2 = k;
|
|
38
|
+
o[k2] = m[k];
|
|
39
|
+
}));
|
|
40
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
41
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
42
|
+
}) : function(o, v) {
|
|
43
|
+
o["default"] = v;
|
|
44
|
+
});
|
|
45
|
+
var __importStar = (this && this.__importStar) || (function () {
|
|
46
|
+
var ownKeys = function(o) {
|
|
47
|
+
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
48
|
+
var ar = [];
|
|
49
|
+
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
50
|
+
return ar;
|
|
51
|
+
};
|
|
52
|
+
return ownKeys(o);
|
|
53
|
+
};
|
|
54
|
+
return function (mod) {
|
|
55
|
+
if (mod && mod.__esModule) return mod;
|
|
56
|
+
var result = {};
|
|
57
|
+
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
58
|
+
__setModuleDefault(result, mod);
|
|
59
|
+
return result;
|
|
60
|
+
};
|
|
61
|
+
})();
|
|
62
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
63
|
+
exports.EVAL_MAX_TIMEOUT_S = exports.EVAL_MAX_TURNS_CEILING = exports.EVAL_DEFAULT_TIMEOUT_S = exports.EVAL_DEFAULT_MAX_TURNS = exports.EVAL_DEFAULT_THRESHOLD = exports.EVAL_MAX_RUNS = exports.EVAL_DEFAULT_RUNS = exports.EVAL_SCHEMA_VERSION = void 0;
|
|
64
|
+
exports.parseList = parseList;
|
|
65
|
+
exports.parseGraderFile = parseGraderFile;
|
|
66
|
+
exports.parseEvalCase = parseEvalCase;
|
|
67
|
+
exports.discoverEvalCases = discoverEvalCases;
|
|
68
|
+
exports.targetText = targetText;
|
|
69
|
+
exports.evaluateGrader = evaluateGrader;
|
|
70
|
+
exports.gradersForArm = gradersForArm;
|
|
71
|
+
exports.scoreRun = scoreRun;
|
|
72
|
+
exports.aggregateReport = aggregateReport;
|
|
73
|
+
const fs = __importStar(require("fs"));
|
|
74
|
+
const path = __importStar(require("path"));
|
|
75
|
+
const frontmatter_1 = require("../util/frontmatter");
|
|
76
|
+
const nestedInstructions_1 = require("../agent/nestedInstructions");
|
|
77
|
+
exports.EVAL_SCHEMA_VERSION = 1;
|
|
78
|
+
/** CC's default: each case runs three times. */
|
|
79
|
+
exports.EVAL_DEFAULT_RUNS = 3;
|
|
80
|
+
exports.EVAL_MAX_RUNS = 50;
|
|
81
|
+
exports.EVAL_DEFAULT_THRESHOLD = 1.0;
|
|
82
|
+
exports.EVAL_DEFAULT_MAX_TURNS = 10;
|
|
83
|
+
exports.EVAL_DEFAULT_TIMEOUT_S = 300;
|
|
84
|
+
/** CC caps both; larger values are refused, not clamped silently. */
|
|
85
|
+
exports.EVAL_MAX_TURNS_CEILING = 200;
|
|
86
|
+
exports.EVAL_MAX_TIMEOUT_S = 3600;
|
|
87
|
+
/** `match: count:N` in a regex grader. */
|
|
88
|
+
const COUNT_RE = /^count:(\d+)$/;
|
|
89
|
+
// ─── Parsing helpers ─────────────────────────────────────────────────────────
|
|
90
|
+
/** `[a, b]` / `a, b` / `a` → ['a','b'] — the two shapes real files use. */
|
|
91
|
+
function parseList(value) {
|
|
92
|
+
if (value === undefined)
|
|
93
|
+
return [];
|
|
94
|
+
const inner = value.trim().replace(/^\[|\]$/g, '');
|
|
95
|
+
return inner
|
|
96
|
+
.split(',')
|
|
97
|
+
.map((s) => s.trim().replace(/^["']|["']$/g, ''))
|
|
98
|
+
.filter(Boolean);
|
|
99
|
+
}
|
|
100
|
+
function parsePositiveInt(value, dflt, ceiling, what) {
|
|
101
|
+
if (value === undefined || value.trim() === '')
|
|
102
|
+
return dflt;
|
|
103
|
+
const n = Number(value.trim());
|
|
104
|
+
if (!Number.isInteger(n) || n < 1)
|
|
105
|
+
throw new Error(`${what} must be a positive integer (got "${value}")`);
|
|
106
|
+
if (n > ceiling)
|
|
107
|
+
throw new Error(`${what} ${n} exceeds the ceiling of ${ceiling}`);
|
|
108
|
+
return n;
|
|
109
|
+
}
|
|
110
|
+
/**
|
|
111
|
+
* One grader file → EvalGrader. Throws on anything that would make the grader
|
|
112
|
+
* silently unscorable (unknown type, missing pattern/tool/path, bad count).
|
|
113
|
+
*/
|
|
114
|
+
function parseGraderFile(file, raw) {
|
|
115
|
+
const { meta, body } = (0, frontmatter_1.parseFrontmatter)(raw);
|
|
116
|
+
const type = (meta.type ?? '').trim();
|
|
117
|
+
if (!['regex', 'tool_used', 'tool_order', 'file_exists', 'llm', 'baseline'].includes(type)) {
|
|
118
|
+
throw new Error(`grader "${file}": type must be regex|tool_used|tool_order|file_exists|llm|baseline (got "${meta.type ?? ''}")`);
|
|
119
|
+
}
|
|
120
|
+
const weightRaw = (meta.weight ?? '1').trim();
|
|
121
|
+
const weight = Number(weightRaw);
|
|
122
|
+
if (!Number.isFinite(weight) || weight <= 0)
|
|
123
|
+
throw new Error(`grader "${file}": weight must be a positive number`);
|
|
124
|
+
const arm = (meta.arm ?? 'both').trim() === 'with-only' ? 'with-only' : 'both';
|
|
125
|
+
if (meta.arm !== undefined && !['both', 'with-only'].includes(meta.arm.trim())) {
|
|
126
|
+
throw new Error(`grader "${file}": arm must be both or with-only (got "${meta.arm}")`);
|
|
127
|
+
}
|
|
128
|
+
const g = { file, type, weight, arm };
|
|
129
|
+
if (type === 'regex') {
|
|
130
|
+
if (!meta.pattern)
|
|
131
|
+
throw new Error(`grader "${file}": regex grader needs a \`pattern\``);
|
|
132
|
+
g.pattern = meta.pattern;
|
|
133
|
+
g.flags = meta.flags?.trim() || '';
|
|
134
|
+
const match = (meta.match ?? 'contains').trim();
|
|
135
|
+
if (!(match === 'contains' || match === 'not_contains' || COUNT_RE.test(match))) {
|
|
136
|
+
throw new Error(`grader "${file}": match must be contains | not_contains | count:<n> (got "${match}")`);
|
|
137
|
+
}
|
|
138
|
+
g.match = match;
|
|
139
|
+
const target = (meta.target ?? 'last_message').trim();
|
|
140
|
+
if (!['last_message', 'trace', 'file'].includes(target)) {
|
|
141
|
+
throw new Error(`grader "${file}": target must be last_message | trace | file (got "${target}")`);
|
|
142
|
+
}
|
|
143
|
+
g.target = target;
|
|
144
|
+
if (target === 'file') {
|
|
145
|
+
if (!meta.path)
|
|
146
|
+
throw new Error(`grader "${file}": target "file" needs a \`path\``);
|
|
147
|
+
g.pathPattern = meta.path;
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
else if (type === 'tool_used') {
|
|
151
|
+
if (!meta.tool)
|
|
152
|
+
throw new Error(`grader "${file}": tool_used grader needs a \`tool\``);
|
|
153
|
+
g.tool = meta.tool.trim();
|
|
154
|
+
g.inputMatch = meta.input_match?.trim() || undefined;
|
|
155
|
+
g.min = parsePositiveInt(meta.min, 1, 10000, 'min');
|
|
156
|
+
g.max = meta.max !== undefined && meta.max.trim() !== '' ? parsePositiveInt(meta.max, 1, 10000, 'max') : undefined;
|
|
157
|
+
if (g.max !== undefined && g.max < (g.min ?? 1))
|
|
158
|
+
throw new Error(`grader "${file}": max < min`);
|
|
159
|
+
}
|
|
160
|
+
else if (type === 'tool_order') {
|
|
161
|
+
if (!meta.before || !meta.after)
|
|
162
|
+
throw new Error(`grader "${file}": tool_order needs \`before\` and \`after\``);
|
|
163
|
+
g.before = meta.before.trim();
|
|
164
|
+
g.after = meta.after.trim();
|
|
165
|
+
}
|
|
166
|
+
else if (type === 'file_exists') {
|
|
167
|
+
if (!meta.path)
|
|
168
|
+
throw new Error(`grader "${file}": file_exists needs a \`path\``);
|
|
169
|
+
g.path = meta.path.trim();
|
|
170
|
+
const ex = (meta.exists ?? 'true').trim().toLowerCase();
|
|
171
|
+
g.exists = !(ex === 'false' || ex === '0' || ex === 'no');
|
|
172
|
+
}
|
|
173
|
+
else {
|
|
174
|
+
// llm / baseline: criteria from the body (the file IS the criteria text).
|
|
175
|
+
const criteria = body.trim();
|
|
176
|
+
if (!criteria)
|
|
177
|
+
throw new Error(`grader "${file}": ${type} grader needs criteria in the file body`);
|
|
178
|
+
g.criteria = criteria;
|
|
179
|
+
if (type === 'llm') {
|
|
180
|
+
const focus = (meta.focus ?? 'last_message').trim();
|
|
181
|
+
if (!['last_message', 'trace'].includes(focus)) {
|
|
182
|
+
throw new Error(`grader "${file}": focus must be last_message | trace (got "${focus}")`);
|
|
183
|
+
}
|
|
184
|
+
g.focus = focus;
|
|
185
|
+
}
|
|
186
|
+
else {
|
|
187
|
+
if (!meta.baseline_file)
|
|
188
|
+
throw new Error(`grader "${file}": baseline grader needs \`baseline_file\``);
|
|
189
|
+
g.baselineFile = meta.baseline_file.trim();
|
|
190
|
+
}
|
|
191
|
+
}
|
|
192
|
+
return g;
|
|
193
|
+
}
|
|
194
|
+
/** One case directory → EvalCase. See the header for the layout. */
|
|
195
|
+
function parseEvalCase(dir) {
|
|
196
|
+
const promptFile = path.join(dir, 'prompt.md');
|
|
197
|
+
const caseYaml = path.join(dir, 'case.yaml');
|
|
198
|
+
if (fs.existsSync(caseYaml)) {
|
|
199
|
+
throw new Error('case.yaml is not supported yet (this repo carries no YAML parser) — move its fields into prompt.md frontmatter');
|
|
200
|
+
}
|
|
201
|
+
if (fs.existsSync(path.join(dir, 'mocks'))) {
|
|
202
|
+
throw new Error('mock MCP servers (mocks/) are not supported yet');
|
|
203
|
+
}
|
|
204
|
+
let raw;
|
|
205
|
+
try {
|
|
206
|
+
raw = fs.readFileSync(promptFile, 'utf-8');
|
|
207
|
+
}
|
|
208
|
+
catch {
|
|
209
|
+
throw new Error('no prompt.md (a case needs one; case.yaml alone is not supported yet)');
|
|
210
|
+
}
|
|
211
|
+
const { meta, body } = (0, frontmatter_1.parseFrontmatter)(raw);
|
|
212
|
+
const prompt = body.trim();
|
|
213
|
+
if (!prompt)
|
|
214
|
+
throw new Error('prompt.md has an empty prompt body');
|
|
215
|
+
const gradersDir = path.join(dir, 'graders');
|
|
216
|
+
let graderFiles;
|
|
217
|
+
try {
|
|
218
|
+
graderFiles = fs
|
|
219
|
+
.readdirSync(gradersDir)
|
|
220
|
+
.filter((f) => f.endsWith('.md'))
|
|
221
|
+
.sort();
|
|
222
|
+
}
|
|
223
|
+
catch {
|
|
224
|
+
throw new Error('no graders/ directory (at least one grader is required)');
|
|
225
|
+
}
|
|
226
|
+
if (!graderFiles.length)
|
|
227
|
+
throw new Error('graders/ holds no .md files (at least one grader is required)');
|
|
228
|
+
const graders = graderFiles.map((f) => parseGraderFile(f, fs.readFileSync(path.join(gradersDir, f), 'utf-8')));
|
|
229
|
+
return {
|
|
230
|
+
name: meta.name?.trim() || path.basename(dir),
|
|
231
|
+
dir,
|
|
232
|
+
prompt,
|
|
233
|
+
runs: parsePositiveInt(meta.runs, exports.EVAL_DEFAULT_RUNS, exports.EVAL_MAX_RUNS, 'runs'),
|
|
234
|
+
maxTurns: parsePositiveInt(meta.max_turns, exports.EVAL_DEFAULT_MAX_TURNS, exports.EVAL_MAX_TURNS_CEILING, 'max_turns'),
|
|
235
|
+
timeoutSeconds: parsePositiveInt(meta.timeout_seconds, exports.EVAL_DEFAULT_TIMEOUT_S, exports.EVAL_MAX_TIMEOUT_S, 'timeout_seconds'),
|
|
236
|
+
allowedTools: parseList(meta.allowed_tools),
|
|
237
|
+
tags: parseList(meta.tags),
|
|
238
|
+
expectedOutcome: meta.expected_outcome?.trim() || undefined,
|
|
239
|
+
graders,
|
|
240
|
+
};
|
|
241
|
+
}
|
|
242
|
+
/** Every case under `<pluginDir>/evals` (the results dir is never a case). */
|
|
243
|
+
function discoverEvalCases(pluginDir) {
|
|
244
|
+
const evalsDir = path.join(pluginDir, 'evals');
|
|
245
|
+
const errors = [];
|
|
246
|
+
let names;
|
|
247
|
+
try {
|
|
248
|
+
names = fs
|
|
249
|
+
.readdirSync(evalsDir)
|
|
250
|
+
.filter((n) => n !== 'results' && !n.startsWith('.'))
|
|
251
|
+
.sort();
|
|
252
|
+
}
|
|
253
|
+
catch {
|
|
254
|
+
return { cases: [], errors: [] };
|
|
255
|
+
}
|
|
256
|
+
const cases = [];
|
|
257
|
+
for (const n of names) {
|
|
258
|
+
const dir = path.join(evalsDir, n);
|
|
259
|
+
let isDir;
|
|
260
|
+
try {
|
|
261
|
+
isDir = fs.statSync(dir).isDirectory();
|
|
262
|
+
}
|
|
263
|
+
catch {
|
|
264
|
+
isDir = false;
|
|
265
|
+
}
|
|
266
|
+
if (!isDir)
|
|
267
|
+
continue;
|
|
268
|
+
try {
|
|
269
|
+
cases.push(parseEvalCase(dir));
|
|
270
|
+
}
|
|
271
|
+
catch (e) {
|
|
272
|
+
errors.push({ dir, error: e.message });
|
|
273
|
+
}
|
|
274
|
+
}
|
|
275
|
+
return { cases, errors };
|
|
276
|
+
}
|
|
277
|
+
// ─── The graders ─────────────────────────────────────────────────────────────
|
|
278
|
+
/** The text a grader's target names. Exposed for tests and the report's `why`. */
|
|
279
|
+
function targetText(g, trace) {
|
|
280
|
+
if (g.type === 'regex' && g.target === 'trace') {
|
|
281
|
+
return trace.toolCalls.map((c) => `${c.tool} ${JSON.stringify(c.input)}`).join('\n');
|
|
282
|
+
}
|
|
283
|
+
if (g.type === 'regex' && g.target === 'file') {
|
|
284
|
+
const content = trace.readFile(g.pathPattern ?? '');
|
|
285
|
+
return content; // null = no matching file; the grader fails with that reason
|
|
286
|
+
}
|
|
287
|
+
if (g.type === 'llm' && g.focus === 'trace') {
|
|
288
|
+
return trace.toolCalls.map((c) => `${c.tool} ${JSON.stringify(c.input)}`).join('\n');
|
|
289
|
+
}
|
|
290
|
+
return trace.lastMessage;
|
|
291
|
+
}
|
|
292
|
+
/** null = the pattern does not compile; the grader FAILS with that reason. */
|
|
293
|
+
function compileRegex(g) {
|
|
294
|
+
try {
|
|
295
|
+
return new RegExp(g.pattern ?? '', g.flags ?? '');
|
|
296
|
+
}
|
|
297
|
+
catch {
|
|
298
|
+
return null;
|
|
299
|
+
}
|
|
300
|
+
}
|
|
301
|
+
function countMatches(re, text) {
|
|
302
|
+
// A global clone: a non-global regex .match() returns the first hit only, and
|
|
303
|
+
// count:N must count every occurrence.
|
|
304
|
+
const flags = re.flags.includes('g') ? re.flags : re.flags + 'g';
|
|
305
|
+
return (text.match(new RegExp(re.source, flags)) ?? []).length;
|
|
306
|
+
}
|
|
307
|
+
/**
|
|
308
|
+
* Score ONE grader over ONE run. Async because llm/baseline call the judge.
|
|
309
|
+
* A grader that cannot be evaluated FAILS (with the reason in `why`) — it must
|
|
310
|
+
* never pass by accident, or a broken suite would look green.
|
|
311
|
+
*/
|
|
312
|
+
async function evaluateGrader(g, trace, judge) {
|
|
313
|
+
try {
|
|
314
|
+
switch (g.type) {
|
|
315
|
+
case 'regex': {
|
|
316
|
+
const text = targetText(g, trace);
|
|
317
|
+
if (text === null)
|
|
318
|
+
return { passed: false, why: `no file matches ${g.pathPattern} (regex ${g.pattern})` };
|
|
319
|
+
const re = compileRegex(g);
|
|
320
|
+
if (!re)
|
|
321
|
+
return { passed: false, why: `invalid pattern: ${g.pattern}` };
|
|
322
|
+
const match = g.match ?? 'contains';
|
|
323
|
+
if (match === 'contains') {
|
|
324
|
+
return { passed: re.test(text), why: re.test(text) ? 'found' : `not found in ${g.target ?? 'last_message'}` };
|
|
325
|
+
}
|
|
326
|
+
if (match === 'not_contains') {
|
|
327
|
+
return {
|
|
328
|
+
passed: !re.test(text),
|
|
329
|
+
why: re.test(text) ? `unexpectedly found in ${g.target ?? 'last_message'}` : 'absent, as required',
|
|
330
|
+
};
|
|
331
|
+
}
|
|
332
|
+
const cm = COUNT_RE.exec(match);
|
|
333
|
+
const want = Number(cm[1]);
|
|
334
|
+
const got = countMatches(re, text);
|
|
335
|
+
return { passed: got === want, why: `count ${got}, wanted ${want}` };
|
|
336
|
+
}
|
|
337
|
+
case 'tool_used': {
|
|
338
|
+
const n = trace.toolCalls.filter((c) => {
|
|
339
|
+
if (c.tool !== g.tool)
|
|
340
|
+
return false;
|
|
341
|
+
if (!g.inputMatch)
|
|
342
|
+
return true;
|
|
343
|
+
try {
|
|
344
|
+
return new RegExp(g.inputMatch).test(JSON.stringify(c.input));
|
|
345
|
+
}
|
|
346
|
+
catch {
|
|
347
|
+
return false;
|
|
348
|
+
}
|
|
349
|
+
}).length;
|
|
350
|
+
const min = g.min ?? 1;
|
|
351
|
+
const ok = n >= min && (g.max === undefined || n <= g.max);
|
|
352
|
+
return {
|
|
353
|
+
passed: ok,
|
|
354
|
+
why: `${g.tool} used ${n} time(s)` +
|
|
355
|
+
(ok ? '' : ` (wanted ${g.max === undefined ? `>=${min}` : `${min}..${g.max}`})`),
|
|
356
|
+
};
|
|
357
|
+
}
|
|
358
|
+
case 'tool_order': {
|
|
359
|
+
const iBefore = trace.toolCalls.findIndex((c) => c.tool === g.before);
|
|
360
|
+
const iAfter = trace.toolCalls.findIndex((c) => c.tool === g.after);
|
|
361
|
+
if (iBefore === -1)
|
|
362
|
+
return { passed: false, why: `${g.before} never called` };
|
|
363
|
+
if (iAfter === -1)
|
|
364
|
+
return { passed: false, why: `${g.after} never called` };
|
|
365
|
+
return iBefore < iAfter
|
|
366
|
+
? { passed: true, why: `${g.before} before ${g.after}` }
|
|
367
|
+
: { passed: false, why: `${g.after} came before ${g.before}` };
|
|
368
|
+
}
|
|
369
|
+
case 'file_exists': {
|
|
370
|
+
const re = (0, nestedInstructions_1.pathGlobToRegExp)(g.path ?? '');
|
|
371
|
+
const hit = re ? trace.files.some((f) => re.test(f)) : false;
|
|
372
|
+
const want = g.exists !== false;
|
|
373
|
+
return {
|
|
374
|
+
passed: hit === want,
|
|
375
|
+
why: want ? (hit ? 'present' : 'missing') : hit ? 'present, wanted absent' : 'absent, as required',
|
|
376
|
+
};
|
|
377
|
+
}
|
|
378
|
+
case 'llm': {
|
|
379
|
+
if (!judge)
|
|
380
|
+
return { passed: false, why: 'no judge available (pass --judge-model or provide NEXRALL_EVAL_JUDGE_MODEL)' };
|
|
381
|
+
const focusText = targetText(g, trace) ?? '';
|
|
382
|
+
const verdict = await judge({ criteria: g.criteria ?? '', focusText });
|
|
383
|
+
if (verdict === null)
|
|
384
|
+
return { passed: false, why: 'judge could not be reached' };
|
|
385
|
+
return { passed: verdict, why: verdict ? 'judge: pass' : 'judge: fail' };
|
|
386
|
+
}
|
|
387
|
+
case 'baseline': {
|
|
388
|
+
if (!judge)
|
|
389
|
+
return { passed: false, why: 'no judge available (pass --judge-model)' };
|
|
390
|
+
const baselineText = trace.readFile(g.baselineFile ?? '');
|
|
391
|
+
if (baselineText === null)
|
|
392
|
+
return { passed: false, why: `baseline file ${g.baselineFile} not found` };
|
|
393
|
+
const verdict = await judge({ criteria: g.criteria ?? '', focusText: trace.lastMessage, baselineText });
|
|
394
|
+
if (verdict === null)
|
|
395
|
+
return { passed: false, why: 'judge could not be reached' };
|
|
396
|
+
return { passed: verdict, why: verdict ? 'at least as good as the baseline' : 'worse than the baseline' };
|
|
397
|
+
}
|
|
398
|
+
}
|
|
399
|
+
}
|
|
400
|
+
catch (e) {
|
|
401
|
+
return { passed: false, why: e.message };
|
|
402
|
+
}
|
|
403
|
+
}
|
|
404
|
+
/** Which graders count in an arm: with-only graders never score the baseline. */
|
|
405
|
+
function gradersForArm(graders, arm) {
|
|
406
|
+
return arm === 'with' ? graders : graders.filter((g) => g.arm !== 'with-only');
|
|
407
|
+
}
|
|
408
|
+
/** Score one run: weighted fraction of its graders that passed. */
|
|
409
|
+
async function scoreRun(graders, trace, judge, arm) {
|
|
410
|
+
const active = gradersForArm(graders, arm);
|
|
411
|
+
if (!active.length)
|
|
412
|
+
return { score: 0, results: [] };
|
|
413
|
+
const results = [];
|
|
414
|
+
let passedWeight = 0;
|
|
415
|
+
let totalWeight = 0;
|
|
416
|
+
for (const g of active) {
|
|
417
|
+
const v = await evaluateGrader(g, trace, judge);
|
|
418
|
+
totalWeight += g.weight;
|
|
419
|
+
if (v.passed)
|
|
420
|
+
passedWeight += g.weight;
|
|
421
|
+
results.push({ file: g.file, type: g.type, weight: g.weight, passed: v.passed, why: v.why });
|
|
422
|
+
}
|
|
423
|
+
return { score: totalWeight > 0 ? passedWeight / totalWeight : 0, results };
|
|
424
|
+
}
|
|
425
|
+
const mean = (xs) => (xs.length ? xs.reduce((a, b) => a + b, 0) / xs.length : null);
|
|
426
|
+
/**
|
|
427
|
+
* Fold run outcomes into the report. Case score = mean of the plugin-arm run
|
|
428
|
+
* scores; `delta` = with − without (null when the baseline did not run).
|
|
429
|
+
* A case PASSES at ≥ threshold (CC's default: every grader, every run).
|
|
430
|
+
*/
|
|
431
|
+
function aggregateReport(i) {
|
|
432
|
+
const cases = i.cases.map((c) => {
|
|
433
|
+
const withScore = mean(c.with.map((r) => r.score));
|
|
434
|
+
const withoutScore = mean(c.without.map((r) => r.score));
|
|
435
|
+
return {
|
|
436
|
+
name: c.name,
|
|
437
|
+
with: c.with,
|
|
438
|
+
without: c.without,
|
|
439
|
+
withScore,
|
|
440
|
+
withoutScore,
|
|
441
|
+
delta: withScore !== null && withoutScore !== null ? withScore - withoutScore : null,
|
|
442
|
+
passed: withScore !== null && withScore >= i.threshold,
|
|
443
|
+
};
|
|
444
|
+
});
|
|
445
|
+
const scores = cases.map((c) => c.withScore).filter((s) => s !== null);
|
|
446
|
+
const deltas = cases.map((c) => c.delta).filter((d) => d !== null);
|
|
447
|
+
const allRuns = cases.flatMap((c) => [...c.with, ...c.without]);
|
|
448
|
+
const reported = allRuns.filter((r) => r.costReported && r.costUsd !== null);
|
|
449
|
+
return {
|
|
450
|
+
schemaVersion: exports.EVAL_SCHEMA_VERSION,
|
|
451
|
+
plugin: i.plugin,
|
|
452
|
+
startedAt: i.startedAt,
|
|
453
|
+
durationMs: i.durationMs,
|
|
454
|
+
threshold: i.threshold,
|
|
455
|
+
runs: i.runs,
|
|
456
|
+
...(i.judgeModel ? { judgeModel: i.judgeModel } : {}),
|
|
457
|
+
...(i.partial ? { partial: true, partialReason: i.partial.reason } : {}),
|
|
458
|
+
cases,
|
|
459
|
+
errors: i.errors,
|
|
460
|
+
aggregates: {
|
|
461
|
+
casesPassed: cases.filter((c) => c.passed).length,
|
|
462
|
+
casesTotal: cases.length,
|
|
463
|
+
overallScore: scores.length ? scores.reduce((a, b) => a + b, 0) / scores.length : 0,
|
|
464
|
+
meanDelta: deltas.length ? deltas.reduce((a, b) => a + b, 0) / deltas.length : null,
|
|
465
|
+
},
|
|
466
|
+
costUsd: reported.reduce((a, r) => a + (r.costUsd ?? 0), 0),
|
|
467
|
+
costMissing: allRuns.length - reported.length,
|
|
468
|
+
};
|
|
469
|
+
}
|
|
470
|
+
//# sourceMappingURL=eval.js.map
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@nexrall/code-core",
|
|
3
|
-
"version": "1.4.
|
|
3
|
+
"version": "1.4.75",
|
|
4
4
|
"description": "Core agent loop, tools, and extension primitives for Nexrall Code — embed an AI coding agent in any Node.js application.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "Nexrall <support@nexrall.com> (https://nexrall.com)",
|