@b4run/evals 0.8.28
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +69 -0
- package/dist/define-eval.d.ts +3 -0
- package/dist/define-eval.d.ts.map +1 -0
- package/dist/define-eval.js +15 -0
- package/dist/gate.d.ts +13 -0
- package/dist/gate.d.ts.map +1 -0
- package/dist/gate.js +64 -0
- package/dist/index.d.ts +9 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +7 -0
- package/dist/llm-judge.d.ts +16 -0
- package/dist/llm-judge.d.ts.map +1 -0
- package/dist/llm-judge.js +57 -0
- package/dist/regex-safety.d.ts +2 -0
- package/dist/regex-safety.d.ts.map +1 -0
- package/dist/regex-safety.js +66 -0
- package/dist/resolve-dataset.d.ts +3 -0
- package/dist/resolve-dataset.d.ts.map +1 -0
- package/dist/resolve-dataset.js +38 -0
- package/dist/run-eval.d.ts +10 -0
- package/dist/run-eval.d.ts.map +1 -0
- package/dist/run-eval.js +56 -0
- package/dist/score.d.ts +8 -0
- package/dist/score.d.ts.map +1 -0
- package/dist/score.js +24 -0
- package/dist/scorers.d.ts +50 -0
- package/dist/scorers.d.ts.map +1 -0
- package/dist/scorers.js +141 -0
- package/dist/tsconfig.tsbuildinfo +1 -0
- package/dist/types.d.ts +74 -0
- package/dist/types.d.ts.map +1 -0
- package/dist/types.js +2 -0
- package/package.json +55 -0
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) Brian Love
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/README.md
ADDED
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
<p align="center">
|
|
2
|
+
<img src="https://raw.githubusercontent.com/cacheplane/b4run/main/docs/brand/b4-logo-horizontal-black-on-white.png" alt="B4.run" width="180" />
|
|
3
|
+
</p>
|
|
4
|
+
|
|
5
|
+
# @b4run/evals
|
|
6
|
+
|
|
7
|
+
Supported evaluation definitions, datasets, scorers, reports, and release gates for B4.run application behavior.
|
|
8
|
+
|
|
9
|
+
**Use this when:** You are defining repeatable evaluations, scorers, or release gates for a B4.run application.
|
|
10
|
+
|
|
11
|
+
## Install
|
|
12
|
+
|
|
13
|
+
```bash
|
|
14
|
+
pnpm add -D @b4run/evals @b4run/testing
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
## Example
|
|
18
|
+
|
|
19
|
+
```ts
|
|
20
|
+
import { contains, defineEval, gate, runEval } from "@b4run/evals"
|
|
21
|
+
import { createAgentHarness, script } from "@b4run/testing"
|
|
22
|
+
|
|
23
|
+
const evaluation = defineEval({
|
|
24
|
+
name: "support replies",
|
|
25
|
+
route: "/support#agent",
|
|
26
|
+
dataset: [
|
|
27
|
+
{
|
|
28
|
+
input: "Where is my order?",
|
|
29
|
+
fixtures: script()
|
|
30
|
+
.user("Where is my order?")
|
|
31
|
+
.replies("Your order is in transit."),
|
|
32
|
+
},
|
|
33
|
+
],
|
|
34
|
+
scorers: [contains("order", { threshold: 1 })],
|
|
35
|
+
gate: gate.perScorer(),
|
|
36
|
+
})
|
|
37
|
+
|
|
38
|
+
await using harness = await createAgentHarness({
|
|
39
|
+
appRoot: process.cwd(),
|
|
40
|
+
route: "/support#agent",
|
|
41
|
+
})
|
|
42
|
+
const report = await runEval(evaluation, {
|
|
43
|
+
runCase: async (testCase) => {
|
|
44
|
+
if (typeof testCase.input !== "string") throw new TypeError("Expected string input")
|
|
45
|
+
return harness.run({
|
|
46
|
+
input: testCase.input,
|
|
47
|
+
...(testCase.fixtures !== undefined ? { fixtures: testCase.fixtures } : {}),
|
|
48
|
+
})
|
|
49
|
+
},
|
|
50
|
+
})
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
## Runtime and stability
|
|
54
|
+
|
|
55
|
+
`@b4run/evals` is a supported node-only testing surface because it resolves JSON and JSONL datasets from disk. Deterministic agent runs come from `@b4run/testing`; scorer code still executes in replay, record, and live modes. `llmJudge()` needs a fixture, model credentials, or an injected fetch implementation.
|
|
56
|
+
|
|
57
|
+
## Related
|
|
58
|
+
|
|
59
|
+
- [Evals API reference](https://b4.run/docs/api/evals) — exact scorer, runner, and gate semantics.
|
|
60
|
+
- [Evals](https://b4.run/docs/evals) — the application evaluation workflow.
|
|
61
|
+
- [`@b4run/testing`](https://www.npmjs.com/package/@b4run/testing) and [Fixtures and Recording](https://b4.run/docs/testing-agents/fixtures) — deterministic model calls.
|
|
62
|
+
|
|
63
|
+
## Maturity and support
|
|
64
|
+
|
|
65
|
+
B4.run is pre-1.0, and its public surface can change. All publishable B4.run packages release together as a fixed group; review the [`@b4run/evals` changelog](https://github.com/cacheplane/b4run/blob/main/packages/evals/CHANGELOG.md) and [upgrading guide](https://b4.run/docs/upgrading) before upgrading. For support, use [GitHub Discussions](https://github.com/cacheplane/b4run/discussions); report defects in [GitHub Issues](https://github.com/cacheplane/b4run/issues).
|
|
66
|
+
|
|
67
|
+
## License
|
|
68
|
+
|
|
69
|
+
MIT. See the [repository license](https://github.com/cacheplane/b4run/blob/main/LICENSE).
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"define-eval.d.ts","sourceRoot":"","sources":["../src/define-eval.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,YAAY,CAAA;AAEhD,wBAAgB,UAAU,CAAC,GAAG,EAAE,cAAc,GAAG,cAAc,CAc9D"}
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
export function defineEval(def) {
|
|
2
|
+
if (!def.name || def.name.trim() === "") {
|
|
3
|
+
throw new Error("defineEval: `name` is required");
|
|
4
|
+
}
|
|
5
|
+
if (!def.scorers || def.scorers.length === 0) {
|
|
6
|
+
throw new Error(`defineEval("${def.name}"): at least one scorer is required`);
|
|
7
|
+
}
|
|
8
|
+
if (Array.isArray(def.dataset) && def.dataset.length === 0) {
|
|
9
|
+
throw new Error(`defineEval("${def.name}"): inline dataset is empty`);
|
|
10
|
+
}
|
|
11
|
+
if (def.dataset === undefined || def.dataset === null) {
|
|
12
|
+
throw new Error(`defineEval("${def.name}"): dataset is required`);
|
|
13
|
+
}
|
|
14
|
+
return def;
|
|
15
|
+
}
|
package/dist/gate.d.ts
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
import { DEFAULT_CASE_BAR, type EvalDefinition, type GatePolicy } from "./types.js";
|
|
2
|
+
export declare const gate: {
|
|
3
|
+
mean(n: number): GatePolicy;
|
|
4
|
+
passRate(n: number): GatePolicy;
|
|
5
|
+
everyCase(n: number): GatePolicy;
|
|
6
|
+
perScorer(): GatePolicy;
|
|
7
|
+
all(...policies: GatePolicy[]): GatePolicy;
|
|
8
|
+
any(...policies: GatePolicy[]): GatePolicy;
|
|
9
|
+
};
|
|
10
|
+
/** gate wins; else threshold → mean(threshold); else informational (always passes). */
|
|
11
|
+
export declare function resolveGate(def: Pick<EvalDefinition, "gate" | "threshold">): GatePolicy;
|
|
12
|
+
export { DEFAULT_CASE_BAR };
|
|
13
|
+
//# sourceMappingURL=gate.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"gate.d.ts","sourceRoot":"","sources":["../src/gate.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,gBAAgB,EAAE,KAAK,cAAc,EAAE,KAAK,UAAU,EAAmB,MAAM,YAAY,CAAA;AASpG,eAAO,MAAM,IAAI;IACf,IAAI,IAAI,MAAM,GAAG,UAAU;IAG3B,QAAQ,IAAI,MAAM,GAAG,UAAU;IAO/B,SAAS,IAAI,MAAM,GAAG,UAAU;IAMhC,SAAS,IAAI,UAAU;IAQvB,GAAG,cAAc,UAAU,EAAE,GAAG,UAAU;IAS1C,GAAG,cAAc,UAAU,EAAE,GAAG,UAAU;CAW3C,CAAA;AAED,uFAAuF;AACvF,wBAAgB,WAAW,CAAC,GAAG,EAAE,IAAI,CAAC,cAAc,EAAE,MAAM,GAAG,WAAW,CAAC,GAAG,UAAU,CAIvF;AAED,OAAO,EAAE,gBAAgB,EAAE,CAAA"}
|
package/dist/gate.js
ADDED
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
import { DEFAULT_CASE_BAR } from "./types.js";
|
|
2
|
+
function pass(reason) {
|
|
3
|
+
return reason !== undefined ? { passed: true, reason } : { passed: true };
|
|
4
|
+
}
|
|
5
|
+
function fail(reason) {
|
|
6
|
+
return { passed: false, reason };
|
|
7
|
+
}
|
|
8
|
+
export const gate = {
|
|
9
|
+
mean(n) {
|
|
10
|
+
return (r) => (r.mean >= n ? pass() : fail(`mean ${r.mean.toFixed(2)} < ${n}`));
|
|
11
|
+
},
|
|
12
|
+
passRate(n) {
|
|
13
|
+
return (r) => {
|
|
14
|
+
const rate = r.cases.length === 0 ? 1 : r.cases.filter((c) => c.passed).length / r.cases.length;
|
|
15
|
+
return rate >= n ? pass() : fail(`pass-rate ${rate.toFixed(2)} < ${n}`);
|
|
16
|
+
};
|
|
17
|
+
},
|
|
18
|
+
everyCase(n) {
|
|
19
|
+
return (r) => {
|
|
20
|
+
const bad = r.cases.find((c) => c.mean < n);
|
|
21
|
+
return bad ? fail(`case "${bad.name}" mean ${bad.mean.toFixed(2)} < ${n}`) : pass();
|
|
22
|
+
};
|
|
23
|
+
},
|
|
24
|
+
perScorer() {
|
|
25
|
+
return (r) => {
|
|
26
|
+
const bad = r.byScorer.find((s) => s.threshold !== undefined && s.mean < s.threshold);
|
|
27
|
+
return bad
|
|
28
|
+
? fail(`scorer "${bad.scorer}" mean ${bad.mean.toFixed(2)} < ${bad.threshold}`)
|
|
29
|
+
: pass();
|
|
30
|
+
};
|
|
31
|
+
},
|
|
32
|
+
all(...policies) {
|
|
33
|
+
return (r) => {
|
|
34
|
+
for (const p of policies) {
|
|
35
|
+
const res = p(r);
|
|
36
|
+
if (!res.passed)
|
|
37
|
+
return res;
|
|
38
|
+
}
|
|
39
|
+
return pass();
|
|
40
|
+
};
|
|
41
|
+
},
|
|
42
|
+
any(...policies) {
|
|
43
|
+
return (r) => {
|
|
44
|
+
const reasons = [];
|
|
45
|
+
for (const p of policies) {
|
|
46
|
+
const res = p(r);
|
|
47
|
+
if (res.passed)
|
|
48
|
+
return pass();
|
|
49
|
+
if (res.reason)
|
|
50
|
+
reasons.push(res.reason);
|
|
51
|
+
}
|
|
52
|
+
return fail(`no policy passed: ${reasons.join("; ")}`);
|
|
53
|
+
};
|
|
54
|
+
},
|
|
55
|
+
};
|
|
56
|
+
/** gate wins; else threshold → mean(threshold); else informational (always passes). */
|
|
57
|
+
export function resolveGate(def) {
|
|
58
|
+
if (def.gate)
|
|
59
|
+
return def.gate;
|
|
60
|
+
if (def.threshold !== undefined)
|
|
61
|
+
return gate.mean(def.threshold);
|
|
62
|
+
return () => pass("informational (no gate)");
|
|
63
|
+
}
|
|
64
|
+
export { DEFAULT_CASE_BAR };
|
package/dist/index.d.ts
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
export { defineEval } from "./define-eval.js";
|
|
2
|
+
export { gate, resolveGate } from "./gate.js";
|
|
3
|
+
export { type LlmJudgeOptions, llmJudge } from "./llm-judge.js";
|
|
4
|
+
export { resolveDataset } from "./resolve-dataset.js";
|
|
5
|
+
export { type RunEvalOptions, runEval } from "./run-eval.js";
|
|
6
|
+
export { type NormalizedScore, normalizeScore } from "./score.js";
|
|
7
|
+
export { contains, custom, exactMatch, jsonEquals, memoryFresh, memoryIsolated, memoryRecalled, regex, tokensUnder, toolCalled, } from "./scorers.js";
|
|
8
|
+
export type { CaseResult, CaseScore, Dataset, EvalCase, EvalDefinition, EvalReport, GatePolicy, GateResult, Score, ScoredReport, Scorer, ScorerAggregate, } from "./types.js";
|
|
9
|
+
//# sourceMappingURL=index.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,UAAU,EAAE,MAAM,kBAAkB,CAAA;AAC7C,OAAO,EAAE,IAAI,EAAE,WAAW,EAAE,MAAM,WAAW,CAAA;AAC7C,OAAO,EAAE,KAAK,eAAe,EAAE,QAAQ,EAAE,MAAM,gBAAgB,CAAA;AAC/D,OAAO,EAAE,cAAc,EAAE,MAAM,sBAAsB,CAAA;AACrD,OAAO,EAAE,KAAK,cAAc,EAAE,OAAO,EAAE,MAAM,eAAe,CAAA;AAC5D,OAAO,EAAE,KAAK,eAAe,EAAE,cAAc,EAAE,MAAM,YAAY,CAAA;AACjE,OAAO,EACL,QAAQ,EACR,MAAM,EACN,UAAU,EACV,UAAU,EACV,WAAW,EACX,cAAc,EACd,cAAc,EACd,KAAK,EACL,WAAW,EACX,UAAU,GACX,MAAM,cAAc,CAAA;AACrB,YAAY,EACV,UAAU,EACV,SAAS,EACT,OAAO,EACP,QAAQ,EACR,cAAc,EACd,UAAU,EACV,UAAU,EACV,UAAU,EACV,KAAK,EACL,YAAY,EACZ,MAAM,EACN,eAAe,GAChB,MAAM,YAAY,CAAA"}
|
package/dist/index.js
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
export { defineEval } from "./define-eval.js";
|
|
2
|
+
export { gate, resolveGate } from "./gate.js";
|
|
3
|
+
export { llmJudge } from "./llm-judge.js";
|
|
4
|
+
export { resolveDataset } from "./resolve-dataset.js";
|
|
5
|
+
export { runEval } from "./run-eval.js";
|
|
6
|
+
export { normalizeScore } from "./score.js";
|
|
7
|
+
export { contains, custom, exactMatch, jsonEquals, memoryFresh, memoryIsolated, memoryRecalled, regex, tokensUnder, toolCalled, } from "./scorers.js";
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import type { Scorer } from "./types.js";
|
|
2
|
+
type FetchImpl = (input: string, init: RequestInit) => Promise<Response>;
|
|
3
|
+
export interface LlmJudgeOptions {
|
|
4
|
+
/** Criteria template; supports {{input}}, {{expected}}, {{output}} interpolation. */
|
|
5
|
+
readonly criteria: string;
|
|
6
|
+
readonly model?: string;
|
|
7
|
+
readonly threshold?: number;
|
|
8
|
+
readonly name?: string;
|
|
9
|
+
/** Overrides for testing; default to env + global fetch. */
|
|
10
|
+
readonly baseUrl?: string;
|
|
11
|
+
readonly apiKey?: string;
|
|
12
|
+
readonly fetchImpl?: FetchImpl;
|
|
13
|
+
}
|
|
14
|
+
export declare function llmJudge(opts: LlmJudgeOptions): Scorer;
|
|
15
|
+
export {};
|
|
16
|
+
//# sourceMappingURL=llm-judge.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"llm-judge.d.ts","sourceRoot":"","sources":["../src/llm-judge.ts"],"names":[],"mappings":"AACA,OAAO,KAAK,EAAY,MAAM,EAAE,MAAM,YAAY,CAAA;AAElD,KAAK,SAAS,GAAG,CAAC,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,WAAW,KAAK,OAAO,CAAC,QAAQ,CAAC,CAAA;AAExE,MAAM,WAAW,eAAe;IAC9B,qFAAqF;IACrF,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAA;IACzB,QAAQ,CAAC,KAAK,CAAC,EAAE,MAAM,CAAA;IACvB,QAAQ,CAAC,SAAS,CAAC,EAAE,MAAM,CAAA;IAC3B,QAAQ,CAAC,IAAI,CAAC,EAAE,MAAM,CAAA;IACtB,4DAA4D;IAC5D,QAAQ,CAAC,OAAO,CAAC,EAAE,MAAM,CAAA;IACzB,QAAQ,CAAC,MAAM,CAAC,EAAE,MAAM,CAAA;IACxB,QAAQ,CAAC,SAAS,CAAC,EAAE,SAAS,CAAA;CAC/B;AAMD,wBAAgB,QAAQ,CAAC,IAAI,EAAE,eAAe,GAAG,MAAM,CAmDtD"}
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
function interpolate(template, vars) {
|
|
2
|
+
return template.replace(/\{\{(\w+)\}\}/g, (_m, key) => vars[key] ?? "");
|
|
3
|
+
}
|
|
4
|
+
export function llmJudge(opts) {
|
|
5
|
+
const model = opts.model ?? "gpt-5-mini";
|
|
6
|
+
return {
|
|
7
|
+
name: opts.name ?? "llmJudge",
|
|
8
|
+
...(opts.threshold !== undefined ? { threshold: opts.threshold } : {}),
|
|
9
|
+
score: async (run, testCase) => {
|
|
10
|
+
const baseUrl = opts.baseUrl ?? process.env.OPENAI_BASE_URL ?? "https://api.openai.com/v1";
|
|
11
|
+
const apiKey = opts.apiKey ?? process.env.OPENAI_API_KEY ?? "";
|
|
12
|
+
const fetchImpl = opts.fetchImpl ?? ((i, init) => fetch(i, init));
|
|
13
|
+
const criteria = interpolate(opts.criteria, {
|
|
14
|
+
input: String(testCase.input ?? ""),
|
|
15
|
+
expected: JSON.stringify(testCase.expected ?? ""),
|
|
16
|
+
output: run.finalMessage,
|
|
17
|
+
});
|
|
18
|
+
const user = [
|
|
19
|
+
`Criteria: ${criteria}`,
|
|
20
|
+
`Agent output: ${run.finalMessage}`,
|
|
21
|
+
`Respond ONLY with JSON: {"score": <0..1>, "reason": "<short>"}.`,
|
|
22
|
+
].join("\n");
|
|
23
|
+
let content;
|
|
24
|
+
try {
|
|
25
|
+
const res = await fetchImpl(`${baseUrl}/chat/completions`, {
|
|
26
|
+
method: "POST",
|
|
27
|
+
headers: { "content-type": "application/json", authorization: `Bearer ${apiKey}` },
|
|
28
|
+
body: JSON.stringify({
|
|
29
|
+
model,
|
|
30
|
+
messages: [
|
|
31
|
+
{
|
|
32
|
+
role: "system",
|
|
33
|
+
content: "You are a strict grader. Output only the requested JSON.",
|
|
34
|
+
},
|
|
35
|
+
{ role: "user", content: user },
|
|
36
|
+
],
|
|
37
|
+
}),
|
|
38
|
+
});
|
|
39
|
+
const json = (await res.json());
|
|
40
|
+
content = json.choices?.[0]?.message?.content ?? "";
|
|
41
|
+
}
|
|
42
|
+
catch (err) {
|
|
43
|
+
return {
|
|
44
|
+
score: 0,
|
|
45
|
+
reason: `judge request failed: ${err instanceof Error ? err.message : String(err)}`,
|
|
46
|
+
};
|
|
47
|
+
}
|
|
48
|
+
try {
|
|
49
|
+
const parsed = JSON.parse(content);
|
|
50
|
+
return { score: parsed.score, reason: parsed.reason ?? "" };
|
|
51
|
+
}
|
|
52
|
+
catch {
|
|
53
|
+
return { score: 0, reason: `could not parse judge verdict: ${content.slice(0, 120)}` };
|
|
54
|
+
}
|
|
55
|
+
},
|
|
56
|
+
};
|
|
57
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"regex-safety.d.ts","sourceRoot":"","sources":["../src/regex-safety.ts"],"names":[],"mappings":"AAmDA,wBAAgB,qBAAqB,CAAC,UAAU,EAAE,MAAM,GAAG,CAAC,KAAK,EAAE,MAAM,KAAK,OAAO,CA8BpF"}
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
import { Script } from "node:vm";
|
|
2
|
+
const MAX_REGEX_SOURCE_CODE_UNITS = 4_096;
|
|
3
|
+
const MAX_REGEX_INPUT_CODE_UNITS = 65_536;
|
|
4
|
+
const REGEX_EXECUTION_TIMEOUT_MS = 100;
|
|
5
|
+
const OVERSIZED_REGEX_SOURCE_MESSAGE = "Regular expression source exceeds 4096 UTF-16 code units";
|
|
6
|
+
const OVERSIZED_REGEX_INPUT_MESSAGE = "Regular expression input exceeds 65536 UTF-16 code units";
|
|
7
|
+
const REGEX_TIMEOUT_MESSAGE = "Regular expression evaluation exceeded 100ms execution limit";
|
|
8
|
+
const REGEX_TEST_SCRIPT = new Script("new RegExp(source, flags).test(input)", {
|
|
9
|
+
filename: "b4-regex-evaluation.vm",
|
|
10
|
+
});
|
|
11
|
+
function intrinsicRegExpGetter(property) {
|
|
12
|
+
const getter = Object.getOwnPropertyDescriptor(RegExp.prototype, property)?.get;
|
|
13
|
+
if (getter === undefined) {
|
|
14
|
+
throw new Error(`RegExp.prototype.${property} is unavailable`);
|
|
15
|
+
}
|
|
16
|
+
return getter;
|
|
17
|
+
}
|
|
18
|
+
const REGEXP_SOURCE_GETTER = intrinsicRegExpGetter("source");
|
|
19
|
+
const REGEXP_FLAG_GETTERS = [
|
|
20
|
+
[intrinsicRegExpGetter("hasIndices"), "d"],
|
|
21
|
+
[intrinsicRegExpGetter("global"), "g"],
|
|
22
|
+
[intrinsicRegExpGetter("ignoreCase"), "i"],
|
|
23
|
+
[intrinsicRegExpGetter("multiline"), "m"],
|
|
24
|
+
[intrinsicRegExpGetter("dotAll"), "s"],
|
|
25
|
+
[intrinsicRegExpGetter("unicode"), "u"],
|
|
26
|
+
[intrinsicRegExpGetter("unicodeSets"), "v"],
|
|
27
|
+
[intrinsicRegExpGetter("sticky"), "y"],
|
|
28
|
+
];
|
|
29
|
+
function snapshotFlags(expression) {
|
|
30
|
+
let flags = "";
|
|
31
|
+
for (const [getter, flag] of REGEXP_FLAG_GETTERS) {
|
|
32
|
+
if (Reflect.apply(getter, expression, []) === true)
|
|
33
|
+
flags += flag;
|
|
34
|
+
}
|
|
35
|
+
return flags;
|
|
36
|
+
}
|
|
37
|
+
function isScriptExecutionTimeout(error) {
|
|
38
|
+
return (typeof error === "object" &&
|
|
39
|
+
error !== null &&
|
|
40
|
+
"code" in error &&
|
|
41
|
+
error.code === "ERR_SCRIPT_EXECUTION_TIMEOUT");
|
|
42
|
+
}
|
|
43
|
+
export function createSafeRegexTester(expression) {
|
|
44
|
+
const source = Reflect.apply(REGEXP_SOURCE_GETTER, expression, []);
|
|
45
|
+
const flags = snapshotFlags(expression);
|
|
46
|
+
if (source.length > MAX_REGEX_SOURCE_CODE_UNITS) {
|
|
47
|
+
throw new RangeError(OVERSIZED_REGEX_SOURCE_MESSAGE);
|
|
48
|
+
}
|
|
49
|
+
return (input) => {
|
|
50
|
+
if (input.length > MAX_REGEX_INPUT_CODE_UNITS) {
|
|
51
|
+
throw new RangeError(OVERSIZED_REGEX_INPUT_MESSAGE);
|
|
52
|
+
}
|
|
53
|
+
try {
|
|
54
|
+
return (REGEX_TEST_SCRIPT.runInNewContext({ flags, input, source }, {
|
|
55
|
+
contextCodeGeneration: { strings: false, wasm: false },
|
|
56
|
+
timeout: REGEX_EXECUTION_TIMEOUT_MS,
|
|
57
|
+
}) === true);
|
|
58
|
+
}
|
|
59
|
+
catch (error) {
|
|
60
|
+
if (isScriptExecutionTimeout(error)) {
|
|
61
|
+
throw new RangeError(REGEX_TIMEOUT_MESSAGE);
|
|
62
|
+
}
|
|
63
|
+
throw error;
|
|
64
|
+
}
|
|
65
|
+
};
|
|
66
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"resolve-dataset.d.ts","sourceRoot":"","sources":["../src/resolve-dataset.ts"],"names":[],"mappings":"AAEA,OAAO,KAAK,EAAE,OAAO,EAAE,QAAQ,EAAE,MAAM,YAAY,CAAA;AAEnD,wBAAsB,cAAc,CAAC,OAAO,EAAE,OAAO,EAAE,OAAO,EAAE,MAAM,GAAG,OAAO,CAAC,QAAQ,EAAE,CAAC,CAiC3F"}
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
import { readFile } from "node:fs/promises";
|
|
2
|
+
import { isAbsolute, resolve } from "node:path";
|
|
3
|
+
export async function resolveDataset(dataset, baseDir) {
|
|
4
|
+
if (Array.isArray(dataset))
|
|
5
|
+
return [...dataset];
|
|
6
|
+
if (typeof dataset === "function")
|
|
7
|
+
return [...(await dataset())];
|
|
8
|
+
if (typeof dataset === "string") {
|
|
9
|
+
const path = isAbsolute(dataset) ? dataset : resolve(baseDir, dataset);
|
|
10
|
+
let raw;
|
|
11
|
+
try {
|
|
12
|
+
raw = await readFile(path, "utf8");
|
|
13
|
+
}
|
|
14
|
+
catch (err) {
|
|
15
|
+
throw new Error(`resolveDataset: cannot read dataset file "${path}": ${err instanceof Error ? err.message : String(err)}`);
|
|
16
|
+
}
|
|
17
|
+
if (path.endsWith(".jsonl")) {
|
|
18
|
+
return raw
|
|
19
|
+
.split("\n")
|
|
20
|
+
.map((l) => l.trim())
|
|
21
|
+
.filter(Boolean)
|
|
22
|
+
.map((line, i) => {
|
|
23
|
+
try {
|
|
24
|
+
return JSON.parse(line);
|
|
25
|
+
}
|
|
26
|
+
catch {
|
|
27
|
+
throw new Error(`resolveDataset: invalid JSONL at line ${i + 1} in "${path}"`);
|
|
28
|
+
}
|
|
29
|
+
});
|
|
30
|
+
}
|
|
31
|
+
const parsed = JSON.parse(raw);
|
|
32
|
+
if (!Array.isArray(parsed)) {
|
|
33
|
+
throw new Error(`resolveDataset: "${path}" must contain a JSON array of cases`);
|
|
34
|
+
}
|
|
35
|
+
return parsed;
|
|
36
|
+
}
|
|
37
|
+
throw new Error("resolveDataset: dataset must be an array, a path string, or a function");
|
|
38
|
+
}
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import type { AgentRunResult } from "@b4run/testing";
|
|
2
|
+
import { type EvalCase, type EvalDefinition, type EvalReport } from "./types.js";
|
|
3
|
+
export interface RunEvalOptions {
|
|
4
|
+
/** Executes one case and returns its run result (replay or live; injected by the CLI). */
|
|
5
|
+
readonly runCase: (testCase: EvalCase) => Promise<AgentRunResult>;
|
|
6
|
+
/** Base dir for resolving a string dataset path (the eval file's directory). */
|
|
7
|
+
readonly baseDir?: string;
|
|
8
|
+
}
|
|
9
|
+
export declare function runEval(def: EvalDefinition, options: RunEvalOptions): Promise<EvalReport>;
|
|
10
|
+
//# sourceMappingURL=run-eval.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"run-eval.d.ts","sourceRoot":"","sources":["../src/run-eval.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,gBAAgB,CAAA;AAIpD,OAAO,EAIL,KAAK,QAAQ,EACb,KAAK,cAAc,EACnB,KAAK,UAAU,EAEhB,MAAM,YAAY,CAAA;AAEnB,MAAM,WAAW,cAAc;IAC7B,0FAA0F;IAC1F,QAAQ,CAAC,OAAO,EAAE,CAAC,QAAQ,EAAE,QAAQ,KAAK,OAAO,CAAC,cAAc,CAAC,CAAA;IACjE,gFAAgF;IAChF,QAAQ,CAAC,OAAO,CAAC,EAAE,MAAM,CAAA;CAC1B;AAMD,wBAAsB,OAAO,CAAC,GAAG,EAAE,cAAc,EAAE,OAAO,EAAE,cAAc,GAAG,OAAO,CAAC,UAAU,CAAC,CAqD/F"}
|
package/dist/run-eval.js
ADDED
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
import { resolveGate } from "./gate.js";
|
|
2
|
+
import { resolveDataset } from "./resolve-dataset.js";
|
|
3
|
+
import { normalizeScore } from "./score.js";
|
|
4
|
+
import { DEFAULT_CASE_BAR, } from "./types.js";
|
|
5
|
+
function mean(nums) {
|
|
6
|
+
return nums.length === 0 ? 0 : nums.reduce((a, b) => a + b, 0) / nums.length;
|
|
7
|
+
}
|
|
8
|
+
export async function runEval(def, options) {
|
|
9
|
+
const cases = await resolveDataset(def.dataset, options.baseDir ?? process.cwd());
|
|
10
|
+
const thresholdOf = new Map(def.scorers.map((s) => [s.name, s.threshold]));
|
|
11
|
+
const caseResults = [];
|
|
12
|
+
for (const [index, testCase] of cases.entries()) {
|
|
13
|
+
const run = await options.runCase(testCase);
|
|
14
|
+
const scores = [];
|
|
15
|
+
for (const scorer of def.scorers) {
|
|
16
|
+
let normalized;
|
|
17
|
+
try {
|
|
18
|
+
normalized = normalizeScore(await scorer.score(run, testCase));
|
|
19
|
+
}
|
|
20
|
+
catch (err) {
|
|
21
|
+
normalized = { score: 0, reason: err instanceof Error ? err.message : String(err) };
|
|
22
|
+
}
|
|
23
|
+
scores.push({
|
|
24
|
+
scorer: scorer.name,
|
|
25
|
+
score: normalized.score,
|
|
26
|
+
...(normalized.label !== undefined ? { label: normalized.label } : {}),
|
|
27
|
+
...(normalized.reason !== undefined ? { reason: normalized.reason } : {}),
|
|
28
|
+
});
|
|
29
|
+
}
|
|
30
|
+
const passed = scores.every((s) => s.score >= (thresholdOf.get(s.scorer) ?? DEFAULT_CASE_BAR));
|
|
31
|
+
caseResults.push({
|
|
32
|
+
name: testCase.name ?? `case ${index + 1}`,
|
|
33
|
+
scores,
|
|
34
|
+
mean: mean(scores.map((s) => s.score)),
|
|
35
|
+
passed,
|
|
36
|
+
});
|
|
37
|
+
}
|
|
38
|
+
const byScorer = def.scorers.map((scorer) => {
|
|
39
|
+
const scorerScores = caseResults.flatMap((c) => c.scores.filter((s) => s.scorer === scorer.name).map((s) => s.score));
|
|
40
|
+
return {
|
|
41
|
+
scorer: scorer.name,
|
|
42
|
+
mean: mean(scorerScores),
|
|
43
|
+
...(scorer.threshold !== undefined ? { threshold: scorer.threshold } : {}),
|
|
44
|
+
};
|
|
45
|
+
});
|
|
46
|
+
const overallMean = mean(caseResults.flatMap((c) => c.scores.map((s) => s.score)));
|
|
47
|
+
const scored = { name: def.name, cases: caseResults, byScorer, mean: overallMean };
|
|
48
|
+
const gated = def.gate !== undefined || def.threshold !== undefined;
|
|
49
|
+
const result = resolveGate(def)(scored);
|
|
50
|
+
return {
|
|
51
|
+
...scored,
|
|
52
|
+
gated,
|
|
53
|
+
passed: result.passed,
|
|
54
|
+
...(result.reason !== undefined ? { reason: result.reason } : {}),
|
|
55
|
+
};
|
|
56
|
+
}
|
package/dist/score.d.ts
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
import type { Score } from "./types.js";
|
|
2
|
+
export interface NormalizedScore {
|
|
3
|
+
readonly score: number;
|
|
4
|
+
readonly label?: string;
|
|
5
|
+
readonly reason?: string;
|
|
6
|
+
}
|
|
7
|
+
export declare function normalizeScore(raw: Score): NormalizedScore;
|
|
8
|
+
//# sourceMappingURL=score.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"score.d.ts","sourceRoot":"","sources":["../src/score.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,KAAK,EAAE,MAAM,YAAY,CAAA;AAEvC,MAAM,WAAW,eAAe;IAC9B,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAA;IACtB,QAAQ,CAAC,KAAK,CAAC,EAAE,MAAM,CAAA;IACvB,QAAQ,CAAC,MAAM,CAAC,EAAE,MAAM,CAAA;CACzB;AAUD,wBAAgB,cAAc,CAAC,GAAG,EAAE,KAAK,GAAG,eAAe,CAU1D"}
|
package/dist/score.js
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
function clamp01(n) {
|
|
2
|
+
const x = Number(n);
|
|
3
|
+
if (!Number.isFinite(x))
|
|
4
|
+
return 0;
|
|
5
|
+
if (x < 0)
|
|
6
|
+
return 0;
|
|
7
|
+
if (x > 1)
|
|
8
|
+
return 1;
|
|
9
|
+
return x;
|
|
10
|
+
}
|
|
11
|
+
export function normalizeScore(raw) {
|
|
12
|
+
if (typeof raw === "boolean")
|
|
13
|
+
return { score: raw ? 1 : 0 };
|
|
14
|
+
if (typeof raw === "number")
|
|
15
|
+
return { score: clamp01(raw) };
|
|
16
|
+
if (raw === null || typeof raw !== "object")
|
|
17
|
+
return { score: 0 };
|
|
18
|
+
const out = { score: clamp01(raw.score) };
|
|
19
|
+
return {
|
|
20
|
+
...out,
|
|
21
|
+
...(raw.label !== undefined ? { label: raw.label } : {}),
|
|
22
|
+
...(raw.reason !== undefined ? { reason: raw.reason } : {}),
|
|
23
|
+
};
|
|
24
|
+
}
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
import type { AgentRunResult } from "@b4run/testing";
|
|
2
|
+
import type { EvalCase, Score, Scorer } from "./types.js";
|
|
3
|
+
/** finalMessage === case.expected (string compare). */
|
|
4
|
+
export declare function exactMatch(opts?: {
|
|
5
|
+
threshold?: number;
|
|
6
|
+
}): Scorer;
|
|
7
|
+
export declare function contains(substring: string, opts?: {
|
|
8
|
+
threshold?: number;
|
|
9
|
+
}): Scorer;
|
|
10
|
+
export declare function regex(re: RegExp, opts?: {
|
|
11
|
+
threshold?: number;
|
|
12
|
+
}): Scorer;
|
|
13
|
+
/** Deep-equals case.expected against parsed finalMessage (default) or a selector. */
|
|
14
|
+
export declare function jsonEquals(opts?: {
|
|
15
|
+
threshold?: number;
|
|
16
|
+
select?: (run: AgentRunResult) => unknown;
|
|
17
|
+
}): Scorer;
|
|
18
|
+
export declare function toolCalled(name: string, opts?: {
|
|
19
|
+
withArgs?: Record<string, unknown>;
|
|
20
|
+
threshold?: number;
|
|
21
|
+
}): Scorer;
|
|
22
|
+
export declare function tokensUnder(budget: number, opts?: {
|
|
23
|
+
threshold?: number;
|
|
24
|
+
}): Scorer;
|
|
25
|
+
export declare function custom(fn: (run: AgentRunResult, testCase: EvalCase) => Score | Promise<Score>, opts?: {
|
|
26
|
+
name?: string;
|
|
27
|
+
threshold?: number;
|
|
28
|
+
}): Scorer;
|
|
29
|
+
/**
|
|
30
|
+
* Score 1 if every id in `expectedIds` appears in at least one `recall` tool
|
|
31
|
+
* result string; score 0 with a reason listing missing ids.
|
|
32
|
+
*/
|
|
33
|
+
export declare function memoryRecalled(expectedIds: string[], opts?: {
|
|
34
|
+
threshold?: number;
|
|
35
|
+
}): Scorer;
|
|
36
|
+
/**
|
|
37
|
+
* Score 1 if `run.finalMessage` contains `expectedValue` (freshness check —
|
|
38
|
+
* the newer value surfaced in the response); else 0.
|
|
39
|
+
*/
|
|
40
|
+
export declare function memoryFresh(expectedValue: string, opts?: {
|
|
41
|
+
threshold?: number;
|
|
42
|
+
}): Scorer;
|
|
43
|
+
/**
|
|
44
|
+
* Score 1 if `forbidden` does NOT appear in any recall tool output or
|
|
45
|
+
* finalMessage (no cross-namespace leak); score 0 if it leaks.
|
|
46
|
+
*/
|
|
47
|
+
export declare function memoryIsolated(forbidden: string, opts?: {
|
|
48
|
+
threshold?: number;
|
|
49
|
+
}): Scorer;
|
|
50
|
+
//# sourceMappingURL=scorers.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"scorers.d.ts","sourceRoot":"","sources":["../src/scorers.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,gBAAgB,CAAA;AAEpD,OAAO,KAAK,EAAE,QAAQ,EAAE,KAAK,EAAE,MAAM,EAAE,MAAM,YAAY,CAAA;AAMzD,uDAAuD;AACvD,wBAAgB,UAAU,CAAC,IAAI,CAAC,EAAE;IAAE,SAAS,CAAC,EAAE,MAAM,CAAA;CAAE,GAAG,MAAM,CAMhE;AAED,wBAAgB,QAAQ,CAAC,SAAS,EAAE,MAAM,EAAE,IAAI,CAAC,EAAE;IAAE,SAAS,CAAC,EAAE,MAAM,CAAA;CAAE,GAAG,MAAM,CAMjF;AAED,wBAAgB,KAAK,CAAC,EAAE,EAAE,MAAM,EAAE,IAAI,CAAC,EAAE;IAAE,SAAS,CAAC,EAAE,MAAM,CAAA;CAAE,GAAG,MAAM,CAQvE;AAED,qFAAqF;AACrF,wBAAgB,UAAU,CAAC,IAAI,CAAC,EAAE;IAChC,SAAS,CAAC,EAAE,MAAM,CAAA;IAClB,MAAM,CAAC,EAAE,CAAC,GAAG,EAAE,cAAc,KAAK,OAAO,CAAA;CAC1C,GAAG,MAAM,CAkBT;AAED,wBAAgB,UAAU,CACxB,IAAI,EAAE,MAAM,EACZ,IAAI,CAAC,EAAE;IAAE,QAAQ,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;IAAC,SAAS,CAAC,EAAE,MAAM,CAAA;CAAE,GAChE,MAAM,CAgBR;AAED,wBAAgB,WAAW,CAAC,MAAM,EAAE,MAAM,EAAE,IAAI,CAAC,EAAE;IAAE,SAAS,CAAC,EAAE,MAAM,CAAA;CAAE,GAAG,MAAM,CAMjF;AAED,wBAAgB,MAAM,CACpB,EAAE,EAAE,CAAC,GAAG,EAAE,cAAc,EAAE,QAAQ,EAAE,QAAQ,KAAK,KAAK,GAAG,OAAO,CAAC,KAAK,CAAC,EACvE,IAAI,CAAC,EAAE;IAAE,IAAI,CAAC,EAAE,MAAM,CAAC;IAAC,SAAS,CAAC,EAAE,MAAM,CAAA;CAAE,GAC3C,MAAM,CAMR;AAaD;;;GAGG;AACH,wBAAgB,cAAc,CAAC,WAAW,EAAE,MAAM,EAAE,EAAE,IAAI,CAAC,EAAE;IAAE,SAAS,CAAC,EAAE,MAAM,CAAA;CAAE,GAAG,MAAM,CAY3F;AAED;;;GAGG;AACH,wBAAgB,WAAW,CAAC,aAAa,EAAE,MAAM,EAAE,IAAI,CAAC,EAAE;IAAE,SAAS,CAAC,EAAE,MAAM,CAAA;CAAE,GAAG,MAAM,CASxF;AAED;;;GAGG;AACH,wBAAgB,cAAc,CAAC,SAAS,EAAE,MAAM,EAAE,IAAI,CAAC,EAAE;IAAE,SAAS,CAAC,EAAE,MAAM,CAAA;CAAE,GAAG,MAAM,CAcvF"}
|