@tangle-network/agent-bench 0.1.0 → 0.3.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +17 -0
- package/HARNESS.md +302 -0
- package/README.md +26 -1
- package/fixtures/aec-bench.json +18 -0
- package/fixtures/agentbench-dbbench.json +22 -0
- package/fixtures/bfcl.json +45 -0
- package/fixtures/commit0.json +72 -0
- package/fixtures/crag.json +10 -0
- package/fixtures/dabstep.json +22 -0
- package/fixtures/enterpriseops-gym.json +103 -0
- package/fixtures/finresearchbench.json +21 -0
- package/fixtures/finsearchcomp.json +66 -0
- package/fixtures/frames.json +26 -0
- package/fixtures/hotpotqa.json +182 -0
- package/fixtures/nomiracl.json +26 -0
- package/fixtures/open-rag-bench.json +16 -0
- package/fixtures/pier-agent/no-model-task/environment/Dockerfile +16 -0
- package/fixtures/pier-agent/no-model-task/environment/seed/src/status.txt +1 -0
- package/fixtures/pier-agent/no-model-task/instruction.md +6 -0
- package/fixtures/pier-agent/no-model-task/pre_artifacts.sh +6 -0
- package/fixtures/pier-agent/no-model-task/task.toml +35 -0
- package/fixtures/pier-agent/no-model-task/tests/Dockerfile +17 -0
- package/fixtures/pier-agent/no-model-task/tests/seed/src/status.txt +1 -0
- package/fixtures/pier-agent/no-model-task/tests/test.sh +19 -0
- package/fixtures/programbench.json +17 -0
- package/fixtures/ragbench.json +21 -0
- package/fixtures/simpleqa.json +121 -0
- package/fixtures/t2-ragbench.json +13 -0
- package/fixtures/tau2-bench.json +16 -0
- package/fixtures/tau3-banking.json +16 -0
- package/fixtures/toollm.json +28 -0
- package/fixtures/webarena-verified.json +20 -0
- package/package.json +39 -15
- package/pier_agents/__init__.py +18 -0
- package/pier_agents/candidate_contract.py +755 -0
- package/pier_agents/process_boundary.py +321 -0
- package/pier_agents/tangle_candidate.py +907 -0
- package/pier_agents/workspace_boundary.py +368 -0
- package/scripts/appworld_driver.py +359 -0
- package/scripts/cadbench_prepare.py +22 -0
- package/scripts/cadgenbench_hard_parts.py +48 -0
- package/scripts/clbench_codebase_judge.py +73 -0
- package/scripts/commit0_judge.py +170 -0
- package/scripts/dabstep_judge.py +42 -0
- package/scripts/enterpriseops_gym_judge.py +281 -0
- package/scripts/programbench_judge.py +120 -0
- package/scripts/render-gate-chart.mjs +176 -0
- package/scripts/run-package-tests.mjs +56 -0
- package/scripts/terminate-pier-trial.mts +66 -0
- package/scripts/trata-hedge/README.md +56 -0
- package/scripts/trata-hedge/run.sh +60 -0
- package/scripts/trata-hedge/solve.py +83 -0
- package/scripts/verify-packed-consumer.mjs +224 -0
- package/scripts/verify-pier-agent.mts +715 -0
- package/scripts/verify-pier-pair.mts +74 -0
- package/scripts/verify-pier-recovery.mts +139 -0
- package/src/adapters.ts +26 -0
- package/src/benchmarks/_harness.test.mts +178 -0
- package/src/benchmarks/_harness.ts +239 -16
- package/src/benchmarks/agentbench.ts +163 -0
- package/src/benchmarks/appworld.test.mts +15 -9
- package/src/benchmarks/bfcl.ts +346 -0
- package/src/benchmarks/crag.ts +137 -0
- package/src/benchmarks/dabstep.test.mts +70 -0
- package/src/benchmarks/dabstep.ts +212 -0
- package/src/benchmarks/external-adapters.test.mts +150 -0
- package/src/benchmarks/finresearchbench.ts +269 -0
- package/src/benchmarks/humaneval.ts +20 -8
- package/src/benchmarks/nomiracl.ts +180 -0
- package/src/benchmarks/open-rag-bench.ts +153 -0
- package/src/benchmarks/rag-benchmarks.test.mts +138 -0
- package/src/benchmarks/rag-shared.ts +327 -0
- package/src/benchmarks/ragbench.ts +171 -0
- package/src/benchmarks/swe-bench.test.mts +61 -0
- package/src/benchmarks/swe-bench.ts +201 -19
- package/src/benchmarks/t2-ragbench.ts +166 -0
- package/src/benchmarks/tau-bench-shared.ts +214 -0
- package/src/benchmarks/tau2-bench.ts +30 -0
- package/src/benchmarks/tau3-banking.ts +29 -0
- package/src/benchmarks/terminal-bench.test.mts +33 -0
- package/src/benchmarks/terminal-bench.ts +23 -8
- package/src/benchmarks/toollm.ts +254 -0
- package/src/benchmarks/types.ts +42 -0
- package/src/benchmarks/webarena-verified.ts +200 -0
- package/src/commit0-prereqs.sh +0 -0
- package/src/coordination-mcp-container-reach.mts +181 -0
- package/src/decoder-live.mts +1 -1
- package/src/examples/README.md +103 -39
- package/src/examples/benchmark-matrix.mts +101 -0
- package/src/examples/lean-proof-gate.README.md +77 -0
- package/src/examples/lean-proof-gate.mts +162 -0
- package/src/examples/lean-verify.ts +95 -0
- package/src/examples/lean.Dockerfile +12 -0
- package/src/examples/math-demo.mts +9 -7
- package/src/examples/strategy-demo.mts +10 -12
- package/src/gate.ts +3 -2
- package/src/hev-eval.mts +69 -0
- package/src/hev-improve.mts +169 -0
- package/src/hev-structural.mts +688 -0
- package/src/index.ts +73 -0
- package/src/mbpp-structural.mts +662 -0
- package/src/pier-agent.test-fixtures.mts +19 -0
- package/src/pier-agent.test.mts +363 -0
- package/src/pier-agent.ts +657 -0
- package/src/pier-result-grader.mjs +30 -0
- package/src/pier-result-grader.test.mts +62 -0
- package/src/pier-result-grader.ts +108 -0
- package/src/pier-task-outcome.test.mts +117 -0
- package/src/pier-task-outcome.ts +240 -0
- package/src/pier-trial-controller.test.mts +412 -0
- package/src/pier-trial-controller.ts +858 -0
- package/src/pier-trial-supervisor.mjs +352 -0
- package/src/resolve-client.ts +25 -2
- package/src/run-benchmarks-cli.mts +72 -0
- package/src/run-benchmarks-report.ts +66 -0
- package/src/run-benchmarks.test.mts +231 -0
- package/src/run-benchmarks.ts +589 -0
- package/src/smoke-structural-rollout.mts +393 -0
- package/src/swe-bench-env.test.ts +207 -0
- package/src/swe-bench-env.ts +554 -0
- package/src/swe-jail.ts +293 -0
- package/src/swe-self-improve.mts +84 -0
- package/src/swe-structural-judge-policy.test.ts +117 -0
- package/src/swe-structural-judge-policy.ts +133 -0
- package/src/swe-structural-policy.test.ts +124 -0
- package/src/swe-structural-policy.ts +132 -0
- package/src/swe-structural-provenance.test.ts +93 -0
- package/src/swe-structural-provenance.ts +138 -0
- package/src/swe-structural.mts +1260 -0
- package/src/swe-temp.ts +14 -0
- package/src/tb-container-executor.mts +234 -0
- package/src/tb-container-executor.test.mts +99 -0
- package/src/tb-supervisor-sidecar.mts +222 -0
- package/src/trata-gepa.mts +1 -1
- package/steerers/eops-itsm-population.json +1 -0
- package/tb_agents/opencode_refine_agent.py +117 -0
- package/tb_agents/opencode_router_agent.py +406 -0
- package/tb_agents/opencode_supervisor_agent.py +239 -0
- package/tb_agents/script_agent.py +66 -0
|
@@ -0,0 +1,346 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Berkeley Function Calling Leaderboard adapter.
|
|
3
|
+
*
|
|
4
|
+
* Scope: deterministic function-call ground-truth categories from the official
|
|
5
|
+
* BFCL data files. This is NOT the full live BFCL leaderboard evaluator: agentic
|
|
6
|
+
* web-search/memory categories and BFCL's own model-response harness remain
|
|
7
|
+
* upstream responsibilities. The adapter loads official JSONL rows plus their
|
|
8
|
+
* `possible_answer` file and scores structured function-call artifacts against
|
|
9
|
+
* allowed function/argument values.
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
import { readFile, stat } from 'node:fs/promises'
|
|
13
|
+
import { join } from 'node:path'
|
|
14
|
+
import type { OutputAdapter } from '@tangle-network/agent-runtime/loops'
|
|
15
|
+
import { benchRoot } from './_harness'
|
|
16
|
+
import type { BenchmarkAdapter, BenchScore, BenchTask, LoadOptions } from './types'
|
|
17
|
+
|
|
18
|
+
const FIXTURES = join(benchRoot, 'fixtures', 'bfcl.json')
|
|
19
|
+
const DEFAULT_CATEGORY = 'BFCL_v4_simple_python'
|
|
20
|
+
|
|
21
|
+
interface BfclRow {
|
|
22
|
+
id: string
|
|
23
|
+
question: unknown
|
|
24
|
+
function: unknown
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
interface BfclAnswerRow {
|
|
28
|
+
id: string
|
|
29
|
+
ground_truth: unknown
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
interface BfclAllowedCall {
|
|
33
|
+
name: string
|
|
34
|
+
arguments: Record<string, unknown[]>
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
interface BfclMeta {
|
|
38
|
+
rowId: string
|
|
39
|
+
category: string
|
|
40
|
+
functions: unknown
|
|
41
|
+
expected: BfclAllowedCall[]
|
|
42
|
+
scoring: 'bfcl-ground-truth-subset'
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
interface BfclActualCall {
|
|
46
|
+
name: string
|
|
47
|
+
arguments: Record<string, unknown>
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
const bfclDir = (): string | undefined => process.env.BFCL_DIR
|
|
51
|
+
const bfclCategory = (): string => process.env.BFCL_CATEGORY ?? DEFAULT_CATEGORY
|
|
52
|
+
|
|
53
|
+
export const bfclOutput: OutputAdapter<string> = {
|
|
54
|
+
parse(events) {
|
|
55
|
+
let text = ''
|
|
56
|
+
for (const ev of events) {
|
|
57
|
+
const d = (ev as { data?: Record<string, unknown> })?.data
|
|
58
|
+
const t = d?.finalText ?? d?.text ?? d?.result
|
|
59
|
+
if (typeof t === 'string' && t.length > 0) text = t
|
|
60
|
+
}
|
|
61
|
+
return text.trim()
|
|
62
|
+
},
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
async function assertPath(path: string, label: string): Promise<void> {
|
|
66
|
+
try {
|
|
67
|
+
await stat(path)
|
|
68
|
+
} catch (err) {
|
|
69
|
+
throw new Error(`BFCL: missing ${label} at ${path} (${err instanceof Error ? err.message : err})`)
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
function readJsonl(raw: string): unknown[] {
|
|
74
|
+
return raw
|
|
75
|
+
.split(/\r?\n/)
|
|
76
|
+
.map((line) => line.trim())
|
|
77
|
+
.filter((line) => line.length > 0)
|
|
78
|
+
.map((line) => JSON.parse(line) as unknown)
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
function taskFile(dir: string, category: string): string {
|
|
82
|
+
return process.env.BFCL_DATA_FILE ?? join(dir, 'bfcl_eval', 'data', `${category}.json`)
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
function answerFile(dir: string, category: string): string {
|
|
86
|
+
return process.env.BFCL_ANSWER_FILE ?? join(dir, 'bfcl_eval', 'data', 'possible_answer', `${category}.json`)
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
function questionText(question: unknown): string {
|
|
90
|
+
if (typeof question === 'string') return question
|
|
91
|
+
if (!Array.isArray(question)) return JSON.stringify(question, null, 2)
|
|
92
|
+
const turns: string[] = []
|
|
93
|
+
for (const item of question.flat(3)) {
|
|
94
|
+
if (item && typeof item === 'object') {
|
|
95
|
+
const role = typeof (item as { role?: unknown }).role === 'string' ? (item as { role: string }).role : 'user'
|
|
96
|
+
const content = (item as { content?: unknown }).content
|
|
97
|
+
if (typeof content === 'string') turns.push(`${role}: ${content}`)
|
|
98
|
+
}
|
|
99
|
+
}
|
|
100
|
+
return turns.length > 0 ? turns.join('\n') : JSON.stringify(question, null, 2)
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
function normalizeScalar(value: unknown): string {
|
|
104
|
+
if (typeof value === 'number') return Number.isInteger(value) ? String(value) : String(Number(value.toFixed(8)))
|
|
105
|
+
if (typeof value === 'string') return value.trim().toLowerCase()
|
|
106
|
+
return JSON.stringify(value)
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
function normalizeAllowed(value: unknown): unknown[] {
|
|
110
|
+
return Array.isArray(value) ? value : [value]
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
function groundTruthToCalls(value: unknown): BfclAllowedCall[] {
|
|
114
|
+
if (!Array.isArray(value)) return []
|
|
115
|
+
const out: BfclAllowedCall[] = []
|
|
116
|
+
for (const item of value) {
|
|
117
|
+
if (!item || typeof item !== 'object') continue
|
|
118
|
+
for (const [name, args] of Object.entries(item as Record<string, unknown>)) {
|
|
119
|
+
if (!args || typeof args !== 'object' || Array.isArray(args)) continue
|
|
120
|
+
const normalized: Record<string, unknown[]> = {}
|
|
121
|
+
for (const [argName, allowed] of Object.entries(args as Record<string, unknown>)) {
|
|
122
|
+
normalized[argName] = normalizeAllowed(allowed)
|
|
123
|
+
}
|
|
124
|
+
out.push({ name, arguments: normalized })
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
return out
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
function rowToTask(row: BfclRow, answer: BfclAnswerRow, category: string): BenchTask {
|
|
131
|
+
const expected = groundTruthToCalls(answer.ground_truth)
|
|
132
|
+
if (expected.length === 0) {
|
|
133
|
+
throw new Error(`BFCL ${row.id}: possible_answer has no deterministic ground_truth calls`)
|
|
134
|
+
}
|
|
135
|
+
const meta: BfclMeta = {
|
|
136
|
+
rowId: row.id,
|
|
137
|
+
category,
|
|
138
|
+
functions: row.function,
|
|
139
|
+
expected,
|
|
140
|
+
scoring: 'bfcl-ground-truth-subset',
|
|
141
|
+
}
|
|
142
|
+
return {
|
|
143
|
+
id: row.id,
|
|
144
|
+
split: category,
|
|
145
|
+
prompt: [
|
|
146
|
+
'Solve this BFCL function-calling task.',
|
|
147
|
+
'Return only JSON: {"function_calls":[{"name":"...","arguments":{...}}]}.',
|
|
148
|
+
'',
|
|
149
|
+
questionText(row.question),
|
|
150
|
+
'',
|
|
151
|
+
`Available functions: ${JSON.stringify(row.function, null, 2)}`,
|
|
152
|
+
].join('\n'),
|
|
153
|
+
metadata: meta as unknown as Record<string, unknown>,
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
function readMeta(task: BenchTask): BfclMeta {
|
|
158
|
+
const md = task.metadata
|
|
159
|
+
if (!md || !Array.isArray(md.expected)) {
|
|
160
|
+
throw new Error(`BFCL task ${task.id} missing expected calls — loadTasks did not populate metadata`)
|
|
161
|
+
}
|
|
162
|
+
return md as unknown as BfclMeta
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
function selectTasks(tasks: BenchTask[], opts: LoadOptions): BenchTask[] {
|
|
166
|
+
let out = tasks
|
|
167
|
+
if (opts.split) out = out.filter((task) => task.split === opts.split)
|
|
168
|
+
if (opts.ids) {
|
|
169
|
+
const want = new Set(opts.ids)
|
|
170
|
+
out = out.filter((task) => want.has(task.id))
|
|
171
|
+
} else if (opts.limit !== undefined) {
|
|
172
|
+
out = out.slice(0, opts.limit)
|
|
173
|
+
}
|
|
174
|
+
if (out.length === 0) throw new Error(`BFCL: no tasks matched ${JSON.stringify(opts)}`)
|
|
175
|
+
return out
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
function buildTasks(rows: BfclRow[], answers: BfclAnswerRow[], category: string, opts: LoadOptions): BenchTask[] {
|
|
179
|
+
const byId = new Map(answers.map((answer) => [answer.id, answer]))
|
|
180
|
+
const tasks = rows.map((row) => {
|
|
181
|
+
const answer = byId.get(row.id)
|
|
182
|
+
if (!answer) throw new Error(`BFCL ${category}: missing possible_answer for ${row.id}`)
|
|
183
|
+
return rowToTask(row, answer, category)
|
|
184
|
+
})
|
|
185
|
+
return selectTasks(tasks, opts)
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
async function loadFixtures(opts: LoadOptions): Promise<BenchTask[]> {
|
|
189
|
+
const fixture = JSON.parse(await readFile(FIXTURES, 'utf8')) as { rows: BfclRow[]; answers: BfclAnswerRow[]; category: string }
|
|
190
|
+
console.warn(`[bfcl] BFCL_FIXTURES=1 — loading ${fixture.rows.length} adapter fixtures`)
|
|
191
|
+
return buildTasks(fixture.rows, fixture.answers, fixture.category, opts)
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
async function loadOfficialTasks(dir: string, opts: LoadOptions): Promise<BenchTask[]> {
|
|
195
|
+
const category = bfclCategory()
|
|
196
|
+
const dataPath = taskFile(dir, category)
|
|
197
|
+
const answersPath = answerFile(dir, category)
|
|
198
|
+
const [rawRows, rawAnswers] = await Promise.all([readFile(dataPath, 'utf8'), readFile(answersPath, 'utf8')])
|
|
199
|
+
const rows = readJsonl(rawRows) as BfclRow[]
|
|
200
|
+
const answers = readJsonl(rawAnswers) as BfclAnswerRow[]
|
|
201
|
+
return buildTasks(rows, answers, category, opts)
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
function extractJsonBlock(text: string): unknown {
|
|
205
|
+
const fences = [...text.matchAll(/```(?:json)?\s*([\s\S]*?)```/g)]
|
|
206
|
+
const raw = (fences.at(-1)?.[1] ?? text).trim()
|
|
207
|
+
try {
|
|
208
|
+
return JSON.parse(raw)
|
|
209
|
+
} catch {
|
|
210
|
+
return undefined
|
|
211
|
+
}
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
function normalizeArguments(value: unknown): Record<string, unknown> {
|
|
215
|
+
if (typeof value === 'string') {
|
|
216
|
+
try {
|
|
217
|
+
return normalizeArguments(JSON.parse(value))
|
|
218
|
+
} catch {
|
|
219
|
+
return {}
|
|
220
|
+
}
|
|
221
|
+
}
|
|
222
|
+
if (!value || typeof value !== 'object' || Array.isArray(value)) return {}
|
|
223
|
+
return value as Record<string, unknown>
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
function callFromObject(value: unknown): BfclActualCall | undefined {
|
|
227
|
+
if (!value || typeof value !== 'object') return undefined
|
|
228
|
+
const raw = value as Record<string, unknown>
|
|
229
|
+
if (raw.function && typeof raw.function === 'object') {
|
|
230
|
+
const fn = raw.function as Record<string, unknown>
|
|
231
|
+
if (typeof fn.name === 'string') return { name: fn.name, arguments: normalizeArguments(fn.arguments) }
|
|
232
|
+
}
|
|
233
|
+
const name =
|
|
234
|
+
typeof raw.name === 'string' ? raw.name
|
|
235
|
+
: typeof raw.function_name === 'string' ? raw.function_name
|
|
236
|
+
: typeof raw.tool_name === 'string' ? raw.tool_name
|
|
237
|
+
: undefined
|
|
238
|
+
if (!name) return undefined
|
|
239
|
+
return { name, arguments: normalizeArguments(raw.arguments ?? raw.args ?? raw.parameters) }
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
function extractActualCalls(artifact: string): BfclActualCall[] {
|
|
243
|
+
const parsed = extractJsonBlock(artifact)
|
|
244
|
+
if (Array.isArray(parsed)) return parsed.map(callFromObject).filter((call): call is BfclActualCall => Boolean(call))
|
|
245
|
+
if (parsed && typeof parsed === 'object') {
|
|
246
|
+
const raw = parsed as Record<string, unknown>
|
|
247
|
+
for (const key of ['function_calls', 'tool_calls', 'calls']) {
|
|
248
|
+
if (Array.isArray(raw[key])) {
|
|
249
|
+
return raw[key].map(callFromObject).filter((call): call is BfclActualCall => Boolean(call))
|
|
250
|
+
}
|
|
251
|
+
}
|
|
252
|
+
const single = callFromObject(raw)
|
|
253
|
+
if (single) return [single]
|
|
254
|
+
}
|
|
255
|
+
return []
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
function argMatches(expectedValues: unknown[], actual: unknown): boolean {
|
|
259
|
+
return expectedValues.some((expected) => normalizeScalar(expected) === normalizeScalar(actual))
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
function callMatches(expected: BfclAllowedCall, actual: BfclActualCall): boolean {
|
|
263
|
+
if (expected.name !== actual.name) return false
|
|
264
|
+
for (const [argName, allowed] of Object.entries(expected.arguments)) {
|
|
265
|
+
if (!(argName in actual.arguments)) {
|
|
266
|
+
if (allowed.some((value) => value === '' || value === null || value === undefined)) continue
|
|
267
|
+
return false
|
|
268
|
+
}
|
|
269
|
+
if (!argMatches(allowed, actual.arguments[argName])) return false
|
|
270
|
+
}
|
|
271
|
+
return true
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
function scoreCalls(task: BenchTask, artifact: string): BenchScore {
|
|
275
|
+
const meta = readMeta(task)
|
|
276
|
+
const actual = extractActualCalls(artifact)
|
|
277
|
+
const used = new Set<number>()
|
|
278
|
+
let matched = 0
|
|
279
|
+
for (const expected of meta.expected) {
|
|
280
|
+
const index = actual.findIndex((call, i) => !used.has(i) && callMatches(expected, call))
|
|
281
|
+
if (index >= 0) {
|
|
282
|
+
used.add(index)
|
|
283
|
+
matched += 1
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
const recall = meta.expected.length === 0 ? 0 : matched / meta.expected.length
|
|
287
|
+
const precision = actual.length === 0 ? 0 : matched / actual.length
|
|
288
|
+
const score = recall
|
|
289
|
+
return {
|
|
290
|
+
resolved: recall === 1 && precision === 1,
|
|
291
|
+
score,
|
|
292
|
+
detail: JSON.stringify({
|
|
293
|
+
scoring: meta.scoring,
|
|
294
|
+
category: meta.category,
|
|
295
|
+
expected: meta.expected,
|
|
296
|
+
actual,
|
|
297
|
+
precision,
|
|
298
|
+
recall,
|
|
299
|
+
fullBfclLeaderboardScore: null,
|
|
300
|
+
}),
|
|
301
|
+
}
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
export function createBfclAdapter(): BenchmarkAdapter {
|
|
305
|
+
const fixturesMode = process.env.BFCL_FIXTURES === '1'
|
|
306
|
+
|
|
307
|
+
return {
|
|
308
|
+
name: 'bfcl',
|
|
309
|
+
output: bfclOutput,
|
|
310
|
+
|
|
311
|
+
async preflight() {
|
|
312
|
+
if (fixturesMode) return
|
|
313
|
+
const dir = bfclDir()
|
|
314
|
+
if (!dir) {
|
|
315
|
+
throw new Error(
|
|
316
|
+
'BFCL_DIR is required. Fix: clone https://github.com/ShishirPatil/gorilla, set BFCL_DIR=/path/to/gorilla/berkeley-function-call-leaderboard, and optionally set BFCL_CATEGORY=BFCL_v4_simple_python.',
|
|
317
|
+
)
|
|
318
|
+
}
|
|
319
|
+
const category = bfclCategory()
|
|
320
|
+
await assertPath(taskFile(dir, category), 'BFCL task JSONL')
|
|
321
|
+
await assertPath(answerFile(dir, category), 'BFCL possible_answer JSONL')
|
|
322
|
+
await loadOfficialTasks(dir, { limit: 1 })
|
|
323
|
+
},
|
|
324
|
+
|
|
325
|
+
async loadTasks(opts: LoadOptions = {}) {
|
|
326
|
+
if (fixturesMode) return loadFixtures(opts)
|
|
327
|
+
const dir = bfclDir()
|
|
328
|
+
if (!dir) throw new Error('BFCL_DIR is required to load official BFCL rows')
|
|
329
|
+
return loadOfficialTasks(dir, opts)
|
|
330
|
+
},
|
|
331
|
+
|
|
332
|
+
async goldArtifact(task: BenchTask) {
|
|
333
|
+
const meta = readMeta(task)
|
|
334
|
+
return JSON.stringify({
|
|
335
|
+
function_calls: meta.expected.map((call) => ({
|
|
336
|
+
name: call.name,
|
|
337
|
+
arguments: Object.fromEntries(Object.entries(call.arguments).map(([key, values]) => [key, values[0]])),
|
|
338
|
+
})),
|
|
339
|
+
}, null, 2)
|
|
340
|
+
},
|
|
341
|
+
|
|
342
|
+
async judge(task: BenchTask, artifact: string): Promise<BenchScore> {
|
|
343
|
+
return scoreCalls(task, artifact)
|
|
344
|
+
},
|
|
345
|
+
}
|
|
346
|
+
}
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* CRAG adapter (Comprehensive RAG Benchmark).
|
|
3
|
+
*
|
|
4
|
+
* Live mode expects an official or compatible CRAG JSON/JSONL export. The
|
|
5
|
+
* adapter preserves CRAG domain/type/dynamism tags in metadata and scores final
|
|
6
|
+
* answers deterministically against the provided gold answer list.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import { readFile } from 'node:fs/promises'
|
|
10
|
+
import { join } from 'node:path'
|
|
11
|
+
import { benchRoot } from './_harness'
|
|
12
|
+
import type { BenchmarkAdapter, BenchScore, BenchTask, LoadOptions } from './types'
|
|
13
|
+
import {
|
|
14
|
+
FINAL_ANSWER_SENTINEL,
|
|
15
|
+
allStrings,
|
|
16
|
+
answerScoreToBenchScore,
|
|
17
|
+
firstString,
|
|
18
|
+
isObject,
|
|
19
|
+
ragAnswerOutput,
|
|
20
|
+
readJsonRows,
|
|
21
|
+
scoreAnswerArtifact,
|
|
22
|
+
selectTasks,
|
|
23
|
+
stringFrom,
|
|
24
|
+
} from './rag-shared'
|
|
25
|
+
|
|
26
|
+
const FIXTURES = join(benchRoot, 'fixtures', 'crag.json')
|
|
27
|
+
|
|
28
|
+
interface CragMeta {
|
|
29
|
+
benchmark: 'crag'
|
|
30
|
+
query: string
|
|
31
|
+
goldAnswers: string[]
|
|
32
|
+
domain: string
|
|
33
|
+
questionType: string
|
|
34
|
+
dynamism: string
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
const dataFile = (): string | undefined => process.env.CRAG_DATA_FILE
|
|
38
|
+
|
|
39
|
+
function rowToTask(raw: unknown, index: number): BenchTask {
|
|
40
|
+
if (!isObject(raw)) throw new Error(`CRAG row ${index} must be an object`)
|
|
41
|
+
const query = firstString(raw, ['query', 'question', 'prompt'])
|
|
42
|
+
const goldAnswers = allStrings(raw, ['answer', 'answers', 'gold', 'gold_answer', 'expected_answer'])
|
|
43
|
+
if (!query) throw new Error(`CRAG row ${index} missing query`)
|
|
44
|
+
if (goldAnswers.length === 0) throw new Error(`CRAG row ${index} missing gold answer`)
|
|
45
|
+
const domain = stringFrom(raw.domain) ?? 'unknown'
|
|
46
|
+
const questionType = stringFrom(raw.question_type) ?? stringFrom(raw.questionType) ?? 'unknown'
|
|
47
|
+
const dynamism = stringFrom(raw.static_or_dynamic) ?? stringFrom(raw.dynamism) ?? 'unknown'
|
|
48
|
+
const id = stringFrom(raw.id) ?? stringFrom(raw.query_id) ?? `crag-${index}`
|
|
49
|
+
const meta: CragMeta = {
|
|
50
|
+
benchmark: 'crag',
|
|
51
|
+
query,
|
|
52
|
+
goldAnswers,
|
|
53
|
+
domain,
|
|
54
|
+
questionType,
|
|
55
|
+
dynamism,
|
|
56
|
+
}
|
|
57
|
+
return {
|
|
58
|
+
id,
|
|
59
|
+
split: stringFrom(raw.split) ?? domain,
|
|
60
|
+
prompt: [
|
|
61
|
+
'Answer this CRAG factual question.',
|
|
62
|
+
'Return a concise answer and do not guess when the evidence is insufficient.',
|
|
63
|
+
'End with a single final line: `FINAL ANSWER: <answer>`.',
|
|
64
|
+
'',
|
|
65
|
+
`Question: ${query}`,
|
|
66
|
+
`Domain: ${domain}`,
|
|
67
|
+
`Question type: ${questionType}`,
|
|
68
|
+
`Dynamism: ${dynamism}`,
|
|
69
|
+
].join('\n'),
|
|
70
|
+
metadata: meta as unknown as Record<string, unknown>,
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
function readMeta(task: BenchTask): CragMeta {
|
|
75
|
+
const md = task.metadata
|
|
76
|
+
if (!md || !Array.isArray(md.goldAnswers)) {
|
|
77
|
+
throw new Error(`CRAG task ${task.id} missing metadata — loadTasks did not populate it`)
|
|
78
|
+
}
|
|
79
|
+
return md as unknown as CragMeta
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
async function loadRows(path: string): Promise<unknown[]> {
|
|
83
|
+
const rows = await readJsonRows(path)
|
|
84
|
+
if (rows.length === 0) throw new Error(`CRAG: no rows in ${path}`)
|
|
85
|
+
return rows
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
async function loadFixtures(opts: LoadOptions): Promise<BenchTask[]> {
|
|
89
|
+
const rows = JSON.parse(await readFile(FIXTURES, 'utf8')) as unknown[]
|
|
90
|
+
console.warn(`[crag] CRAG_FIXTURES=1 — loading ${rows.length} adapter fixtures`)
|
|
91
|
+
return selectTasks(rows.map(rowToTask), opts, 'CRAG')
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
export function createCragAdapter(): BenchmarkAdapter {
|
|
95
|
+
const fixturesMode = process.env.CRAG_FIXTURES === '1'
|
|
96
|
+
|
|
97
|
+
return {
|
|
98
|
+
name: 'crag',
|
|
99
|
+
output: ragAnswerOutput,
|
|
100
|
+
|
|
101
|
+
async preflight() {
|
|
102
|
+
if (fixturesMode) {
|
|
103
|
+
await readFile(FIXTURES, 'utf8')
|
|
104
|
+
return
|
|
105
|
+
}
|
|
106
|
+
const path = dataFile()
|
|
107
|
+
if (!path) {
|
|
108
|
+
throw new Error(
|
|
109
|
+
'CRAG_DATA_FILE is required. Fix: export facebookresearch/CRAG rows to JSONL and set CRAG_DATA_FILE=/path/to/crag.jsonl, or set CRAG_FIXTURES=1 for adapter plumbing.',
|
|
110
|
+
)
|
|
111
|
+
}
|
|
112
|
+
await loadRows(path)
|
|
113
|
+
},
|
|
114
|
+
|
|
115
|
+
async loadTasks(opts: LoadOptions = {}) {
|
|
116
|
+
if (fixturesMode) return loadFixtures(opts)
|
|
117
|
+
const path = dataFile()
|
|
118
|
+
if (!path) throw new Error('CRAG_DATA_FILE is required to load CRAG tasks')
|
|
119
|
+
return selectTasks((await loadRows(path)).map(rowToTask), opts, 'CRAG')
|
|
120
|
+
},
|
|
121
|
+
|
|
122
|
+
async goldArtifact(task: BenchTask) {
|
|
123
|
+
return `${FINAL_ANSWER_SENTINEL} ${readMeta(task).goldAnswers[0] ?? ''}`
|
|
124
|
+
},
|
|
125
|
+
|
|
126
|
+
async judge(task: BenchTask, artifact: string): Promise<BenchScore> {
|
|
127
|
+
const meta = readMeta(task)
|
|
128
|
+
const score = scoreAnswerArtifact(artifact, meta.goldAnswers)
|
|
129
|
+
return answerScoreToBenchScore(score, {
|
|
130
|
+
benchmark: meta.benchmark,
|
|
131
|
+
domain: meta.domain,
|
|
132
|
+
questionType: meta.questionType,
|
|
133
|
+
dynamism: meta.dynamism,
|
|
134
|
+
})
|
|
135
|
+
},
|
|
136
|
+
}
|
|
137
|
+
}
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Offline DABStep adapter test. Official live tasks need a DABStep checkout with
|
|
3
|
+
* the released dataset.csv. Fixture mode only exercises adapter plumbing; it
|
|
4
|
+
* never scores benchmark rows without the official grade.py.
|
|
5
|
+
*/
|
|
6
|
+
import assert from 'node:assert/strict'
|
|
7
|
+
import { mkdtemp, rm } from 'node:fs/promises'
|
|
8
|
+
import { tmpdir } from 'node:os'
|
|
9
|
+
import { join } from 'node:path'
|
|
10
|
+
import { test } from 'node:test'
|
|
11
|
+
import { createDabstepAdapter, dabstepAnswerOutput } from './dabstep'
|
|
12
|
+
|
|
13
|
+
process.env.DABSTEP_FIXTURES = '1'
|
|
14
|
+
|
|
15
|
+
type Events = Parameters<typeof dabstepAnswerOutput.parse>[0]
|
|
16
|
+
const stream = (text: string): Events => [{ data: { finalText: text } }] as unknown as Events
|
|
17
|
+
|
|
18
|
+
test('loadTasks fixtures expose DABStep prompt and resource metadata shape', async () => {
|
|
19
|
+
const adapter = createDabstepAdapter()
|
|
20
|
+
const tasks = await adapter.loadTasks({ ids: ['1'] })
|
|
21
|
+
assert.equal(tasks.length, 1)
|
|
22
|
+
assert.equal(tasks[0].id, '1')
|
|
23
|
+
assert.match(tasks[0].prompt, /DABStep data-analysis task/)
|
|
24
|
+
const meta = tasks[0].metadata as Record<string, unknown>
|
|
25
|
+
assert.equal(meta.taskId, 1)
|
|
26
|
+
assert.equal(Array.isArray(meta.golds), true)
|
|
27
|
+
})
|
|
28
|
+
|
|
29
|
+
test('answer OutputAdapter extracts final fenced answer when present', () => {
|
|
30
|
+
assert.equal(dabstepAnswerOutput.parse(stream('work\n```answer\n42\n```')), '42')
|
|
31
|
+
assert.equal(dabstepAnswerOutput.parse(stream('Final Answer: 42')), 'Final Answer: 42')
|
|
32
|
+
})
|
|
33
|
+
|
|
34
|
+
test('goldArtifact exposes the fixture oracle without scoring it as a benchmark result', async () => {
|
|
35
|
+
const adapter = createDabstepAdapter()
|
|
36
|
+
const [task] = await adapter.loadTasks({ ids: ['1'] })
|
|
37
|
+
const gold = await adapter.goldArtifact(task)
|
|
38
|
+
assert.equal(gold, '42')
|
|
39
|
+
})
|
|
40
|
+
|
|
41
|
+
test('judge fails loud without an official DABSTEP_DIR/grade.py', async () => {
|
|
42
|
+
const adapter = createDabstepAdapter()
|
|
43
|
+
const [task] = await adapter.loadTasks({ ids: ['1'] })
|
|
44
|
+
delete process.env.DABSTEP_DIR
|
|
45
|
+
await assert.rejects(adapter.judge(task, '42'), /DABSTEP_DIR is required/)
|
|
46
|
+
})
|
|
47
|
+
|
|
48
|
+
test('preflight is fixture-safe and live mode fails loud without DABSTEP_DIR', async () => {
|
|
49
|
+
const fixtureAdapter = createDabstepAdapter()
|
|
50
|
+
await fixtureAdapter.preflight()
|
|
51
|
+
delete process.env.DABSTEP_FIXTURES
|
|
52
|
+
delete process.env.DABSTEP_DIR
|
|
53
|
+
const liveAdapter = createDabstepAdapter()
|
|
54
|
+
await assert.rejects(liveAdapter.preflight(), /DABSTEP_DIR is required/)
|
|
55
|
+
process.env.DABSTEP_FIXTURES = '1'
|
|
56
|
+
})
|
|
57
|
+
|
|
58
|
+
test('live preflight fails loud when the checkout is missing released dataset.csv', async () => {
|
|
59
|
+
delete process.env.DABSTEP_FIXTURES
|
|
60
|
+
const dir = await mkdtemp(join(tmpdir(), 'dabstep-missing-dataset-'))
|
|
61
|
+
process.env.DABSTEP_DIR = dir
|
|
62
|
+
try {
|
|
63
|
+
const adapter = createDabstepAdapter()
|
|
64
|
+
await assert.rejects(adapter.preflight(), /released dataset\.csv/)
|
|
65
|
+
} finally {
|
|
66
|
+
process.env.DABSTEP_FIXTURES = '1'
|
|
67
|
+
delete process.env.DABSTEP_DIR
|
|
68
|
+
await rm(dir, { recursive: true, force: true })
|
|
69
|
+
}
|
|
70
|
+
})
|