@tangle-network/agent-eval 0.116.0 → 0.117.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +31 -0
- package/dist/analyst/index.d.ts +18 -11
- package/dist/analyst/index.js +10 -7
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyst-CFBc14Wc.d.ts → analyst-C8HHvfJp.d.ts} +1 -1
- package/dist/{analyze-runs-0rz_m29H.d.ts → analyze-runs--2x39HZ7.d.ts} +3 -3
- package/dist/{baseline-DsNteOgR.d.ts → baseline-DKq3gJpP.d.ts} +6 -3
- package/dist/belief-state/index.d.ts +6 -6
- package/dist/belief-state/index.js +1 -1
- package/dist/benchmarks/index.d.ts +11 -8
- package/dist/benchmarks/index.js +11 -10
- package/dist/builder-eval/index.d.ts +4 -4
- package/dist/builder-eval/index.js +1 -1
- package/dist/{calibration-Dz8TQV4y.d.ts → calibration-C8MTS7cw.d.ts} +2 -2
- package/dist/campaign/index.d.ts +53 -29
- package/dist/campaign/index.js +18 -13
- package/dist/chunk-3YYRZDON.js +45 -0
- package/dist/chunk-3YYRZDON.js.map +1 -0
- package/dist/{chunk-RPDDVKI7.js → chunk-4JLWXDYA.js} +2 -2
- package/dist/{chunk-NBSS5NDZ.js → chunk-CCZIVI3F.js} +54 -115
- package/dist/chunk-CCZIVI3F.js.map +1 -0
- package/dist/{chunk-J6P6PK2R.js → chunk-FQNLDL4D.js} +3 -3
- package/dist/{chunk-ONM6PEAE.js → chunk-GQCZRZ7L.js} +2 -2
- package/dist/chunk-HHWE3POT.js +94 -0
- package/dist/chunk-HHWE3POT.js.map +1 -0
- package/dist/{chunk-3274WNK7.js → chunk-HQPHZGL6.js} +687 -44
- package/dist/chunk-HQPHZGL6.js.map +1 -0
- package/dist/{chunk-FAOEFFRT.js → chunk-IDZTTFRR.js} +390 -78
- package/dist/chunk-IDZTTFRR.js.map +1 -0
- package/dist/{chunk-3LXTCTWL.js → chunk-JSDVRFAP.js} +2 -2
- package/dist/{chunk-MHNQWM4I.js → chunk-LQUTGLOZ.js} +5 -1
- package/dist/chunk-LQUTGLOZ.js.map +1 -0
- package/dist/{chunk-4D5RVB3W.js → chunk-LTVG32KX.js} +30 -5
- package/dist/chunk-LTVG32KX.js.map +1 -0
- package/dist/{chunk-CIUOICJT.js → chunk-MGEHEHSN.js} +62 -15
- package/dist/chunk-MGEHEHSN.js.map +1 -0
- package/dist/{chunk-GY4SYVPJ.js → chunk-NJC7U437.js} +97 -25
- package/dist/chunk-NJC7U437.js.map +1 -0
- package/dist/{chunk-NYFUT3B3.js → chunk-ODVOOEWQ.js} +31 -10
- package/dist/chunk-ODVOOEWQ.js.map +1 -0
- package/dist/{chunk-LNQEP766.js → chunk-S2F4J57L.js} +44 -4
- package/dist/chunk-S2F4J57L.js.map +1 -0
- package/dist/chunk-VCTY3W6J.js +798 -0
- package/dist/chunk-VCTY3W6J.js.map +1 -0
- package/dist/chunk-VF3XSYTI.js +545 -0
- package/dist/chunk-VF3XSYTI.js.map +1 -0
- package/dist/{chunk-TLDB7WRY.js → chunk-YZPO4UHR.js} +28 -31
- package/dist/chunk-YZPO4UHR.js.map +1 -0
- package/dist/{chunk-GSW3OBHK.js → chunk-ZUXV7UWZ.js} +350 -724
- package/dist/chunk-ZUXV7UWZ.js.map +1 -0
- package/dist/cli.js +4 -2
- package/dist/cli.js.map +1 -1
- package/dist/{code-agent-session-CdxteG0y.d.ts → code-agent-session-CjZsVd19.d.ts} +1 -1
- package/dist/contract/index.d.ts +43 -29
- package/dist/contract/index.js +56 -19
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-DbcDxouY.d.ts → control-6vuGfmDH.d.ts} +5 -5
- package/dist/control.d.ts +6 -6
- package/dist/cost-ledger-DWy3XdJc.d.ts +183 -0
- package/dist/{default-registry-DDfv22MQ.d.ts → default-registry-DaK8b3fv.d.ts} +2 -2
- package/dist/{emitter-BRchAAAx.d.ts → emitter-CjD7vUwv.d.ts} +2 -2
- package/dist/{failure-cluster-C48PiReX.d.ts → failure-cluster-DOAcSJ87.d.ts} +2 -2
- package/dist/{feedback-trajectory-pDcz1lQ1.d.ts → feedback-trajectory-BUnM58xL.d.ts} +3 -3
- package/dist/fuzz.d.ts +8 -16
- package/dist/fuzz.js +72 -42
- package/dist/fuzz.js.map +1 -1
- package/dist/{gepa-CQelRtuC.d.ts → gepa-eESocoDi.d.ts} +56 -6
- package/dist/hosted/index.d.ts +13 -10
- package/dist/{index-DbCXJfZ1.d.ts → index-PdX4VnPA.d.ts} +3 -3
- package/dist/index.d.ts +102 -57
- package/dist/index.js +328 -235
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-oMVxDTxl.d.ts → insight-report-DY4nDW9Q.d.ts} +1 -1
- package/dist/{integrity-C6PZ73iC.d.ts → integrity-DqlBiLyK.d.ts} +2 -2
- package/dist/{kind-factory-DWOvXjR_.d.ts → kind-factory-ClZmO25A.d.ts} +2 -2
- package/dist/llm-client-qoDd18Qz.d.ts +289 -0
- package/dist/meta-eval/index.d.ts +8 -7
- package/dist/meta-eval/index.js +1 -1
- package/dist/multishot/index.d.ts +9 -6
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +16 -6
- package/dist/pipelines/index.js +119 -23
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{policy-edit-Clb2v6Oa.d.ts → policy-edit-wG9uFEFm.d.ts} +13 -266
- package/dist/{pre-registration--vU0mMtD.d.ts → pre-registration-BWQhJ3vz.d.ts} +24 -5
- package/dist/{provenance-BbVagC68.d.ts → provenance-DpjwyseI.d.ts} +6 -6
- package/dist/{query-Ck190MOd.d.ts → query-CF7PG61p.d.ts} +5 -3
- package/dist/raw-provider-sink-C46HDghv.d.ts +132 -0
- package/dist/{release-report-CamNDe90.d.ts → release-report-C8G2i5Xi.d.ts} +2 -2
- package/dist/reporting.d.ts +10 -9
- package/dist/{researcher-Dwbo_Fxx.d.ts → researcher-C8XyxQsu.d.ts} +8 -8
- package/dist/rl.d.ts +18 -15
- package/dist/rl.js +2 -2
- package/dist/{rubric-predictive-validity-BIdf9h4R.d.ts → rubric-predictive-validity-p49lLVrE.d.ts} +1 -1
- package/dist/{run-campaign-UADIM77S.js → run-campaign-IM26A6PD.js} +4 -2
- package/dist/{run-record-CZmcpWPo.d.ts → run-record-BDH49H2E.d.ts} +1 -1
- package/dist/{runtime-trajectory-CC0jx9ql.d.ts → runtime-trajectory-DGBIUt4B.d.ts} +1 -1
- package/dist/{schema-SGWcK9wa.d.ts → schema-B3Q3l9Z_.d.ts} +2 -0
- package/dist/{semantic-concept-judge-CKjePUMh.d.ts → semantic-concept-judge-CXnPEJbf.d.ts} +24 -6
- package/dist/{statistics-oUbOJe-S.d.ts → statistics-KUnG73jH.d.ts} +1 -1
- package/dist/{storage-Dw_f7WMt.d.ts → storage-DrX3v_5B.d.ts} +12 -1
- package/dist/{store-9cAScOcb.d.ts → store-C1YxJDEK.d.ts} +1 -132
- package/dist/{store-BsVi7ncX.d.ts → store-DGqD0Pyo.d.ts} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-DTNgQycC.d.ts → summary-report-C5bKFfm-.d.ts} +2 -2
- package/dist/{test-graded-scenario-mzYBKspu.d.ts → test-graded-scenario-B0ybnPY7.d.ts} +3 -3
- package/dist/traces.d.ts +25 -14
- package/dist/traces.js +16 -4
- package/dist/{types-Ca_63YSD.d.ts → types-BSw1rOUB.d.ts} +41 -39
- package/dist/{types-C7DGg5ex.d.ts → types-BkfcQnxV.d.ts} +15 -0
- package/dist/wire/index.d.ts +28 -19
- package/dist/wire/index.js +4 -2
- package/docs/distributed-driver.md +1 -1
- package/package.json +3 -3
- package/dist/chunk-3274WNK7.js.map +0 -1
- package/dist/chunk-4D5RVB3W.js.map +0 -1
- package/dist/chunk-7GKEAIAD.js +0 -205
- package/dist/chunk-7GKEAIAD.js.map +0 -1
- package/dist/chunk-CIUOICJT.js.map +0 -1
- package/dist/chunk-FAOEFFRT.js.map +0 -1
- package/dist/chunk-GSW3OBHK.js.map +0 -1
- package/dist/chunk-GY4SYVPJ.js.map +0 -1
- package/dist/chunk-LNQEP766.js.map +0 -1
- package/dist/chunk-MHNQWM4I.js.map +0 -1
- package/dist/chunk-MPHTT5HE.js +0 -74
- package/dist/chunk-MPHTT5HE.js.map +0 -1
- package/dist/chunk-NBSS5NDZ.js.map +0 -1
- package/dist/chunk-NYFUT3B3.js.map +0 -1
- package/dist/chunk-TLDB7WRY.js.map +0 -1
- package/dist/cost-ledger-DuSqlw5B.d.ts +0 -113
- /package/dist/{chunk-RPDDVKI7.js.map → chunk-4JLWXDYA.js.map} +0 -0
- /package/dist/{chunk-J6P6PK2R.js.map → chunk-FQNLDL4D.js.map} +0 -0
- /package/dist/{chunk-ONM6PEAE.js.map → chunk-GQCZRZ7L.js.map} +0 -0
- /package/dist/{chunk-3LXTCTWL.js.map → chunk-JSDVRFAP.js.map} +0 -0
- /package/dist/{run-campaign-UADIM77S.js.map → run-campaign-IM26A6PD.js.map} +0 -0
package/dist/cli.js
CHANGED
|
@@ -5,8 +5,10 @@ import {
|
|
|
5
5
|
runRpcBatch,
|
|
6
6
|
runRpcOnce,
|
|
7
7
|
startServer
|
|
8
|
-
} from "./chunk-
|
|
9
|
-
import "./chunk-
|
|
8
|
+
} from "./chunk-LTVG32KX.js";
|
|
9
|
+
import "./chunk-NJC7U437.js";
|
|
10
|
+
import "./chunk-VCTY3W6J.js";
|
|
11
|
+
import "./chunk-VI2UW6B6.js";
|
|
10
12
|
import "./chunk-PC4UYEBM.js";
|
|
11
13
|
import "./chunk-ONWEPEDO.js";
|
|
12
14
|
import "./chunk-PZ5AY32C.js";
|
package/dist/cli.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"sources":["../src/cli.ts"],"sourcesContent":["#!/usr/bin/env node\n/**\n * agent-eval CLI.\n *\n * agent-eval serve [--port 5005] [--host 127.0.0.1]\n * agent-eval rpc <method> # one request from stdin → one response on stdout\n * agent-eval rpc-batch <method> # JSONL stdin → JSONL stdout\n * agent-eval openapi [--out path] # write OpenAPI spec\n * agent-eval version\n *\n * <method> is one of: judge, listRubrics, version. When omitted, the\n * stdin payload must be a full {method, params} envelope.\n */\nimport { writeFileSync } from 'node:fs'\nimport { handleVersion } from './wire/handlers'\nimport { buildOpenApi } from './wire/openapi'\nimport { runRpcBatch, runRpcOnce } from './wire/rpc'\nimport { startServer } from './wire/server'\n\ninterface Args {\n command: string\n positional: string[]\n flags: Record<string, string>\n}\n\nfunction parseArgs(argv: string[]): Args {\n const [command, ...rest] = argv\n const positional: string[] = []\n const flags: Record<string, string> = {}\n for (let i = 0; i < rest.length; i++) {\n const tok = rest[i]!\n if (tok.startsWith('--')) {\n const key = tok.slice(2)\n const next = rest[i + 1]\n if (next != null && !next.startsWith('--')) {\n flags[key] = next\n i++\n } else {\n flags[key] = 'true'\n }\n } else {\n positional.push(tok)\n }\n }\n return { command: command ?? 'help', positional, flags }\n}\n\nconst HELP = `agent-eval — wire-protocol entry point.\n\nCommands:\n serve [--port 5005] [--host 127.0.0.1]\n Start the HTTP server. POST /v1/judge, GET /v1/rubrics, GET /v1/version, GET /openapi.json.\n rpc <method>\n Read one JSON object from stdin (the params for <method>), write one\n JSON object to stdout. Method ∈ {judge, listRubrics, version}.\n rpc-batch <method>\n Like 'rpc' but JSONL in / JSONL out.\n openapi [--out openapi.json]\n Write the OpenAPI 3.1 spec.\n version\n Print server + wire-protocol version JSON.\n\nWithout arguments, prints this help.`\n\nasync function main(): Promise<number> {\n const { command, positional, flags } = parseArgs(process.argv.slice(2))\n\n switch (command) {\n case 'serve': {\n const port = Number(flags.port ?? 5005)\n const host = flags.host ?? '127.0.0.1'\n const server = startServer({ port, host })\n // Keep process alive on SIGINT/SIGTERM\n const shutdown = (sig: string) => {\n // eslint-disable-next-line no-console\n console.log(`[agent-eval] received ${sig}, shutting down`)\n server.close(() => process.exit(0))\n // Force exit after 5s if close hangs\n setTimeout(() => process.exit(1), 5000).unref()\n }\n process.on('SIGINT', () => shutdown('SIGINT'))\n process.on('SIGTERM', () => shutdown('SIGTERM'))\n // Block forever\n await new Promise(() => {})\n return 0\n }\n case 'rpc': {\n const [method] = positional\n return await runRpcOnce(method)\n }\n case 'rpc-batch': {\n const [method] = positional\n return await runRpcBatch(method)\n }\n case 'openapi': {\n const out = flags.out ?? 'openapi.json'\n const spec = buildOpenApi(handleVersion().version)\n writeFileSync(out, `${JSON.stringify(spec, null, 2)}\\n`, 'utf-8')\n // eslint-disable-next-line no-console\n console.log(`[agent-eval] wrote OpenAPI 3.1 spec to ${out}`)\n return 0\n }\n case 'version': {\n process.stdout.write(`${JSON.stringify(handleVersion(), null, 2)}\\n`)\n return 0\n }\n case 'help':\n case '--help':\n case '-h':\n case '':\n process.stdout.write(`${HELP}\\n`)\n return 0\n default:\n process.stderr.write(`unknown command: ${command}\\n${HELP}\\n`)\n return 1\n }\n}\n\nmain()\n .then((code) => process.exit(code))\n .catch((err) => {\n // eslint-disable-next-line no-console\n console.error('[agent-eval] cli error:', err)\n process.exit(1)\n })\n"],"mappings":"
|
|
1
|
+
{"version":3,"sources":["../src/cli.ts"],"sourcesContent":["#!/usr/bin/env node\n/**\n * agent-eval CLI.\n *\n * agent-eval serve [--port 5005] [--host 127.0.0.1]\n * agent-eval rpc <method> # one request from stdin → one response on stdout\n * agent-eval rpc-batch <method> # JSONL stdin → JSONL stdout\n * agent-eval openapi [--out path] # write OpenAPI spec\n * agent-eval version\n *\n * <method> is one of: judge, listRubrics, version. When omitted, the\n * stdin payload must be a full {method, params} envelope.\n */\nimport { writeFileSync } from 'node:fs'\nimport { handleVersion } from './wire/handlers'\nimport { buildOpenApi } from './wire/openapi'\nimport { runRpcBatch, runRpcOnce } from './wire/rpc'\nimport { startServer } from './wire/server'\n\ninterface Args {\n command: string\n positional: string[]\n flags: Record<string, string>\n}\n\nfunction parseArgs(argv: string[]): Args {\n const [command, ...rest] = argv\n const positional: string[] = []\n const flags: Record<string, string> = {}\n for (let i = 0; i < rest.length; i++) {\n const tok = rest[i]!\n if (tok.startsWith('--')) {\n const key = tok.slice(2)\n const next = rest[i + 1]\n if (next != null && !next.startsWith('--')) {\n flags[key] = next\n i++\n } else {\n flags[key] = 'true'\n }\n } else {\n positional.push(tok)\n }\n }\n return { command: command ?? 'help', positional, flags }\n}\n\nconst HELP = `agent-eval — wire-protocol entry point.\n\nCommands:\n serve [--port 5005] [--host 127.0.0.1]\n Start the HTTP server. POST /v1/judge, GET /v1/rubrics, GET /v1/version, GET /openapi.json.\n rpc <method>\n Read one JSON object from stdin (the params for <method>), write one\n JSON object to stdout. Method ∈ {judge, listRubrics, version}.\n rpc-batch <method>\n Like 'rpc' but JSONL in / JSONL out.\n openapi [--out openapi.json]\n Write the OpenAPI 3.1 spec.\n version\n Print server + wire-protocol version JSON.\n\nWithout arguments, prints this help.`\n\nasync function main(): Promise<number> {\n const { command, positional, flags } = parseArgs(process.argv.slice(2))\n\n switch (command) {\n case 'serve': {\n const port = Number(flags.port ?? 5005)\n const host = flags.host ?? '127.0.0.1'\n const server = startServer({ port, host })\n // Keep process alive on SIGINT/SIGTERM\n const shutdown = (sig: string) => {\n // eslint-disable-next-line no-console\n console.log(`[agent-eval] received ${sig}, shutting down`)\n server.close(() => process.exit(0))\n // Force exit after 5s if close hangs\n setTimeout(() => process.exit(1), 5000).unref()\n }\n process.on('SIGINT', () => shutdown('SIGINT'))\n process.on('SIGTERM', () => shutdown('SIGTERM'))\n // Block forever\n await new Promise(() => {})\n return 0\n }\n case 'rpc': {\n const [method] = positional\n return await runRpcOnce(method)\n }\n case 'rpc-batch': {\n const [method] = positional\n return await runRpcBatch(method)\n }\n case 'openapi': {\n const out = flags.out ?? 'openapi.json'\n const spec = buildOpenApi(handleVersion().version)\n writeFileSync(out, `${JSON.stringify(spec, null, 2)}\\n`, 'utf-8')\n // eslint-disable-next-line no-console\n console.log(`[agent-eval] wrote OpenAPI 3.1 spec to ${out}`)\n return 0\n }\n case 'version': {\n process.stdout.write(`${JSON.stringify(handleVersion(), null, 2)}\\n`)\n return 0\n }\n case 'help':\n case '--help':\n case '-h':\n case '':\n process.stdout.write(`${HELP}\\n`)\n return 0\n default:\n process.stderr.write(`unknown command: ${command}\\n${HELP}\\n`)\n return 1\n }\n}\n\nmain()\n .then((code) => process.exit(code))\n .catch((err) => {\n // eslint-disable-next-line no-console\n console.error('[agent-eval] cli error:', err)\n process.exit(1)\n })\n"],"mappings":";;;;;;;;;;;;;;;;AAaA,SAAS,qBAAqB;AAY9B,SAAS,UAAU,MAAsB;AACvC,QAAM,CAAC,SAAS,GAAG,IAAI,IAAI;AAC3B,QAAM,aAAuB,CAAC;AAC9B,QAAM,QAAgC,CAAC;AACvC,WAAS,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;AACpC,UAAM,MAAM,KAAK,CAAC;AAClB,QAAI,IAAI,WAAW,IAAI,GAAG;AACxB,YAAM,MAAM,IAAI,MAAM,CAAC;AACvB,YAAM,OAAO,KAAK,IAAI,CAAC;AACvB,UAAI,QAAQ,QAAQ,CAAC,KAAK,WAAW,IAAI,GAAG;AAC1C,cAAM,GAAG,IAAI;AACb;AAAA,MACF,OAAO;AACL,cAAM,GAAG,IAAI;AAAA,MACf;AAAA,IACF,OAAO;AACL,iBAAW,KAAK,GAAG;AAAA,IACrB;AAAA,EACF;AACA,SAAO,EAAE,SAAS,WAAW,QAAQ,YAAY,MAAM;AACzD;AAEA,IAAM,OAAO;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAiBb,eAAe,OAAwB;AACrC,QAAM,EAAE,SAAS,YAAY,MAAM,IAAI,UAAU,QAAQ,KAAK,MAAM,CAAC,CAAC;AAEtE,UAAQ,SAAS;AAAA,IACf,KAAK,SAAS;AACZ,YAAM,OAAO,OAAO,MAAM,QAAQ,IAAI;AACtC,YAAM,OAAO,MAAM,QAAQ;AAC3B,YAAM,SAAS,YAAY,EAAE,MAAM,KAAK,CAAC;AAEzC,YAAM,WAAW,CAAC,QAAgB;AAEhC,gBAAQ,IAAI,yBAAyB,GAAG,iBAAiB;AACzD,eAAO,MAAM,MAAM,QAAQ,KAAK,CAAC,CAAC;AAElC,mBAAW,MAAM,QAAQ,KAAK,CAAC,GAAG,GAAI,EAAE,MAAM;AAAA,MAChD;AACA,cAAQ,GAAG,UAAU,MAAM,SAAS,QAAQ,CAAC;AAC7C,cAAQ,GAAG,WAAW,MAAM,SAAS,SAAS,CAAC;AAE/C,YAAM,IAAI,QAAQ,MAAM;AAAA,MAAC,CAAC;AAC1B,aAAO;AAAA,IACT;AAAA,IACA,KAAK,OAAO;AACV,YAAM,CAAC,MAAM,IAAI;AACjB,aAAO,MAAM,WAAW,MAAM;AAAA,IAChC;AAAA,IACA,KAAK,aAAa;AAChB,YAAM,CAAC,MAAM,IAAI;AACjB,aAAO,MAAM,YAAY,MAAM;AAAA,IACjC;AAAA,IACA,KAAK,WAAW;AACd,YAAM,MAAM,MAAM,OAAO;AACzB,YAAM,OAAO,aAAa,cAAc,EAAE,OAAO;AACjD,oBAAc,KAAK,GAAG,KAAK,UAAU,MAAM,MAAM,CAAC,CAAC;AAAA,GAAM,OAAO;AAEhE,cAAQ,IAAI,0CAA0C,GAAG,EAAE;AAC3D,aAAO;AAAA,IACT;AAAA,IACA,KAAK,WAAW;AACd,cAAQ,OAAO,MAAM,GAAG,KAAK,UAAU,cAAc,GAAG,MAAM,CAAC,CAAC;AAAA,CAAI;AACpE,aAAO;AAAA,IACT;AAAA,IACA,KAAK;AAAA,IACL,KAAK;AAAA,IACL,KAAK;AAAA,IACL,KAAK;AACH,cAAQ,OAAO,MAAM,GAAG,IAAI;AAAA,CAAI;AAChC,aAAO;AAAA,IACT;AACE,cAAQ,OAAO,MAAM,oBAAoB,OAAO;AAAA,EAAK,IAAI;AAAA,CAAI;AAC7D,aAAO;AAAA,EACX;AACF;AAEA,KAAK,EACF,KAAK,CAAC,SAAS,QAAQ,KAAK,IAAI,CAAC,EACjC,MAAM,CAAC,QAAQ;AAEd,UAAQ,MAAM,2BAA2B,GAAG;AAC5C,UAAQ,KAAK,CAAC;AAChB,CAAC;","names":[]}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { a as RunSplitTag, b as RunCostProvenance, R as RunRecord } from './run-record-
|
|
1
|
+
import { a as RunSplitTag, b as RunCostProvenance, R as RunRecord } from './run-record-BDH49H2E.js';
|
|
2
2
|
|
|
3
3
|
type CodeAgentSessionSource = 'codex' | 'claude-code' | 'opencode' | 'kimi-code' | 'pi';
|
|
4
4
|
interface ParsedCodeAgentJsonl {
|
package/dist/contract/index.d.ts
CHANGED
|
@@ -1,35 +1,38 @@
|
|
|
1
|
-
import { S as Scenario, M as MutableSurface, D as DispatchContext, b as JudgeConfig,
|
|
2
|
-
export {
|
|
3
|
-
import { L as LoopProvenanceRecord, P as PowerPreflight, R as RunEvalOptions } from '../provenance-
|
|
4
|
-
export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, D as DefaultProductionGateOptions, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, O as ObjectiveSource, c as ParetoSignificanceGateOptions, d as PromotionObjective, e as PromotionPolicy, f as buildEvidenceVector, g as composeGate, h as defaultProductionGate, i as evolutionaryProposer, j as heldOutGate, p as paretoPolicy, k as paretoSignificanceGate, r as runEval } from '../provenance-
|
|
5
|
-
import { R as RunOptimizationOptions, a as RunImprovementLoopResult } from '../gepa-
|
|
6
|
-
export { G as GepaProposerOptions, b as
|
|
7
|
-
|
|
8
|
-
|
|
1
|
+
import { S as Scenario, M as MutableSurface, D as DispatchContext, b as JudgeConfig, c as SurfaceProposer, G as Gate, L as LabeledScenarioStore, C as CampaignResult, d as GateDecision } from '../types-BSw1rOUB.js';
|
|
2
|
+
export { e as CampaignAggregates, f as CampaignArtifactWriter, g as CampaignCellResult, h as CampaignCostMeter, i as CampaignTraceWriter, j as CodeSurface, k as Dispatch, l as GateContext, m as GateResult, n as GenerationCandidate, o as GenerationRecord, a as JudgeDimension, J as JudgeScore, p as Mutator, O as OptimizationProposer, q as OptimizerConfig, r as SessionScript } from '../types-BSw1rOUB.js';
|
|
3
|
+
import { L as LoopProvenanceRecord, P as PowerPreflight, R as RunEvalOptions } from '../provenance-DpjwyseI.js';
|
|
4
|
+
export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, D as DefaultProductionGateOptions, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, O as ObjectiveSource, c as ParetoSignificanceGateOptions, d as PromotionObjective, e as PromotionPolicy, f as buildEvidenceVector, g as composeGate, h as defaultProductionGate, i as evolutionaryProposer, j as heldOutGate, p as paretoPolicy, k as paretoSignificanceGate, r as runEval } from '../provenance-DpjwyseI.js';
|
|
5
|
+
import { R as RunOptimizationOptions, a as RunImprovementLoopResult } from '../gepa-eESocoDi.js';
|
|
6
|
+
export { G as GepaProposerOptions, b as REFERENCE_EQUIVALENCE_INPUT_LIMITS, c as REFERENCE_EQUIVALENCE_JUDGE_VERSION, d as ReferenceEquivalenceJudgeInput, e as ReferenceEquivalenceJudgeOptions, f as ReferenceEquivalenceJudgeResult, g as ReferenceEquivalenceScenario, h as RunCampaignOptions, i as RunImprovementLoopOptions, j as createReferenceEquivalenceJudge, k as gepaProposer, r as runCampaign, l as runImprovementLoop, m as runReferenceEquivalenceJudge } from '../gepa-eESocoDi.js';
|
|
7
|
+
export { c as AnalystFinding, C as ChatClient, p as CreateChatClientOpts, V as createChatClient } from '../policy-edit-wG9uFEFm.js';
|
|
8
|
+
import { C as CampaignStorage } from '../storage-DrX3v_5B.js';
|
|
9
|
+
export { f as fsCampaignStorage, i as inMemoryCampaignStorage } from '../storage-DrX3v_5B.js';
|
|
9
10
|
export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore, b as OutcomeStore } from '../outcome-store-rnXLEqSn.js';
|
|
10
11
|
import { HostedTenant, EvalRunCellScore, EvalRunGenerationSnapshot, EvalRunEvent, TraceSpanEvent } from '../hosted/index.js';
|
|
11
|
-
import {
|
|
12
|
-
import {
|
|
13
|
-
|
|
14
|
-
export {
|
|
15
|
-
export {
|
|
16
|
-
import { A as AnalyzeRunsOptions } from '../analyze-runs
|
|
17
|
-
export { a as analyzeRuns } from '../analyze-runs
|
|
18
|
-
export { C as CodeAgentSessionDiagnostic, a as CodeAgentSessionIntakeOptions, b as CodeAgentSessionIntakeResult, c as CodeAgentSessionMetrics, d as CodeAgentSessionSource, P as ParsedCodeAgentJsonl, f as fromClaudeCodeSession, e as fromCodexSession, g as fromKimiCodeSession, h as fromOpenCodeSession, i as fromPiSession, j as fromPigraphSession, p as parseCodeAgentJsonl } from '../code-agent-session-
|
|
19
|
-
import '../
|
|
12
|
+
import { b as CostLedgerSummary, d as CostReceipt, C as CostLedger } from '../cost-ledger-DWy3XdJc.js';
|
|
13
|
+
import { R as RunRecord, a as RunSplitTag } from '../run-record-BDH49H2E.js';
|
|
14
|
+
import { I as InsightReport } from '../insight-report-DY4nDW9Q.js';
|
|
15
|
+
export { C as CostProvenanceSummary, F as FailureClusterInsight, a as InterRaterInsight, J as JudgeInsight, L as LiftInsight, O as OutcomeCorrelationInsight, R as Recommendation, b as ReleaseSummary, S as ScalarDistribution } from '../insight-report-DY4nDW9Q.js';
|
|
16
|
+
export { D as DefaultAnalystRegistryOptions, c as buildDefaultAnalystRegistry } from '../default-registry-DaK8b3fv.js';
|
|
17
|
+
import { A as AnalyzeRunsOptions } from '../analyze-runs--2x39HZ7.js';
|
|
18
|
+
export { a as analyzeRuns } from '../analyze-runs--2x39HZ7.js';
|
|
19
|
+
export { C as CodeAgentSessionDiagnostic, a as CodeAgentSessionIntakeOptions, b as CodeAgentSessionIntakeResult, c as CodeAgentSessionMetrics, d as CodeAgentSessionSource, P as ParsedCodeAgentJsonl, f as fromClaudeCodeSession, e as fromCodexSession, g as fromKimiCodeSession, h as fromOpenCodeSession, i as fromPiSession, j as fromPigraphSession, p as parseCodeAgentJsonl } from '../code-agent-session-CjZsVd19.js';
|
|
20
|
+
import '../llm-client-qoDd18Qz.js';
|
|
21
|
+
import '../errors-oeQrLqXC.js';
|
|
22
|
+
import '../raw-provider-sink-C46HDghv.js';
|
|
23
|
+
import '../statistics-KUnG73jH.js';
|
|
20
24
|
import '../judge-calibration-7C-IDmKr.js';
|
|
21
|
-
import '../types-
|
|
25
|
+
import '../types-BkfcQnxV.js';
|
|
22
26
|
import '@tangle-network/tcloud';
|
|
23
27
|
import '../dataset-NENEzRgk.js';
|
|
24
|
-
import '../
|
|
25
|
-
import '../
|
|
26
|
-
import '../
|
|
28
|
+
import '../store-DGqD0Pyo.js';
|
|
29
|
+
import '../schema-B3Q3l9Z_.js';
|
|
30
|
+
import '../store-C1YxJDEK.js';
|
|
27
31
|
import '@tangle-network/agent-interface';
|
|
28
|
-
import '../summary-report-
|
|
29
|
-
import '../failure-cluster-
|
|
32
|
+
import '../summary-report-C5bKFfm-.js';
|
|
33
|
+
import '../failure-cluster-DOAcSJ87.js';
|
|
30
34
|
import '@ax-llm/ax';
|
|
31
|
-
import '../kind-factory-
|
|
32
|
-
import '../store-9cAScOcb.js';
|
|
35
|
+
import '../kind-factory-ClZmO25A.js';
|
|
33
36
|
import 'zod';
|
|
34
37
|
|
|
35
38
|
/**
|
|
@@ -57,8 +60,8 @@ import 'zod';
|
|
|
57
60
|
*/
|
|
58
61
|
|
|
59
62
|
interface SelfImproveBudget {
|
|
60
|
-
/** Hard
|
|
61
|
-
*
|
|
63
|
+
/** Hard spend cap across the full run. Each paid call reserves its enforced
|
|
64
|
+
* maximum before dispatch, so completed spend cannot cross this amount. */
|
|
62
65
|
dollars?: number;
|
|
63
66
|
/** How many improvement generations to explore. Default 3. Set 0 to
|
|
64
67
|
* skip improvement entirely (selfImprove becomes a baseline-only run). */
|
|
@@ -282,8 +285,13 @@ interface SelfImproveResult<TScenario extends Scenario, TArtifact> {
|
|
|
282
285
|
generationsExplored: number;
|
|
283
286
|
/** Wall-clock total. */
|
|
284
287
|
durationMs: number;
|
|
285
|
-
/** Total cost across
|
|
288
|
+
/** Total newly observed cost across the full run. */
|
|
286
289
|
totalCostUsd: number;
|
|
290
|
+
/** Canonical run-wide spend summary. */
|
|
291
|
+
cost: CostLedgerSummary;
|
|
292
|
+
/** Run-wide receipts across proposal, search, holdout, judging, analysis,
|
|
293
|
+
* and promotion work, with phase and actor attribution. */
|
|
294
|
+
receipts: CostReceipt[];
|
|
287
295
|
/**
|
|
288
296
|
* Rigor packet: distributional summary, paired-bootstrap lift CI,
|
|
289
297
|
* judge stats, contamination check, recommendations. Wired through
|
|
@@ -303,6 +311,12 @@ interface SelfImproveResult<TScenario extends Scenario, TArtifact> {
|
|
|
303
311
|
*/
|
|
304
312
|
raw: RunImprovementLoopResult<TArtifact, TScenario>;
|
|
305
313
|
}
|
|
314
|
+
/** Failed self-improvement run with an immutable receipt snapshot. */
|
|
315
|
+
declare class SelfImproveRunError extends Error {
|
|
316
|
+
readonly cost: CostLedgerSummary;
|
|
317
|
+
readonly receipts: CostReceipt[];
|
|
318
|
+
constructor(cause: unknown, ledger: CostLedger);
|
|
319
|
+
}
|
|
306
320
|
/**
|
|
307
321
|
* One-shot self-improvement loop. See module docstring for defaults +
|
|
308
322
|
* extension points.
|
|
@@ -841,4 +855,4 @@ interface FromOtelSpansOptions {
|
|
|
841
855
|
}
|
|
842
856
|
declare function fromOtelSpans(opts: FromOtelSpansOptions): RunRecord[];
|
|
843
857
|
|
|
844
|
-
export { type AgentEvalAgent, type AgentEvalEvaluateOptions, type AgentEvalImproveOptions, type AgentTraceContributor, type AgentTraceContributorType, type AgentTraceConversation, type AgentTraceFile, type AgentTraceIndex, type AgentTraceRange, type AgentTraceRecord, AnalyzeRunsOptions, type AuthoringProvenance, CampaignResult, CampaignStorage, type DefineAgentEvalOptions, type DefinedAgentEval, DispatchContext, type EvalCellScoreDelta, type EvalDimensionDelta, type EvalGenerationDiff, type EvalReportingSuiteInput, type EvalReportingSuiteOptions, type EvalReportingSuiteResult, type EvalRunDiff, type FeedbackTableMeta, type FeedbackTableRow, type FromFeedbackTableOptions, type FromFeedbackTableResult, type FromOtelSpansOptions, type FromRunRecordDirOptions, type FromRunRecordDirResult, Gate, GateDecision, HostedTenant, InsightReport, JudgeConfig, MutableSurface, type PartitionByAuthoringModelResult, RunEvalOptions, RunImprovementLoopResult, type RunRecordRejection, Scenario, type SelfImproveBudget, type SelfImproveLlm, type SelfImproveOptions, type SelfImproveProgressEvent, type SelfImproveResult, SurfaceProposer, defineAgentEval, diffGenerations, diffRunBaselineToWinner, diffRuns, evalReportingSuite, fromFeedbackTable, fromOtelSpans, fromRunRecordDir, parseAgentTrace, partitionRunsByAuthoringModel, selfImprove };
|
|
858
|
+
export { type AgentEvalAgent, type AgentEvalEvaluateOptions, type AgentEvalImproveOptions, type AgentTraceContributor, type AgentTraceContributorType, type AgentTraceConversation, type AgentTraceFile, type AgentTraceIndex, type AgentTraceRange, type AgentTraceRecord, AnalyzeRunsOptions, type AuthoringProvenance, CampaignResult, CampaignStorage, type DefineAgentEvalOptions, type DefinedAgentEval, DispatchContext, type EvalCellScoreDelta, type EvalDimensionDelta, type EvalGenerationDiff, type EvalReportingSuiteInput, type EvalReportingSuiteOptions, type EvalReportingSuiteResult, type EvalRunDiff, type FeedbackTableMeta, type FeedbackTableRow, type FromFeedbackTableOptions, type FromFeedbackTableResult, type FromOtelSpansOptions, type FromRunRecordDirOptions, type FromRunRecordDirResult, Gate, GateDecision, HostedTenant, InsightReport, JudgeConfig, MutableSurface, type PartitionByAuthoringModelResult, RunEvalOptions, RunImprovementLoopResult, type RunRecordRejection, Scenario, type SelfImproveBudget, type SelfImproveLlm, type SelfImproveOptions, type SelfImproveProgressEvent, type SelfImproveResult, SelfImproveRunError, SurfaceProposer, defineAgentEval, diffGenerations, diffRunBaselineToWinner, diffRuns, evalReportingSuite, fromFeedbackTable, fromOtelSpans, fromRunRecordDir, parseAgentTrace, partitionRunsByAuthoringModel, selfImprove };
|
package/dist/contract/index.js
CHANGED
|
@@ -12,11 +12,14 @@ import {
|
|
|
12
12
|
} from "../chunk-ZZUXHH3R.js";
|
|
13
13
|
import {
|
|
14
14
|
analyzeRuns
|
|
15
|
-
} from "../chunk-
|
|
15
|
+
} from "../chunk-FQNLDL4D.js";
|
|
16
16
|
import {
|
|
17
|
+
REFERENCE_EQUIVALENCE_INPUT_LIMITS,
|
|
18
|
+
REFERENCE_EQUIVALENCE_JUDGE_VERSION,
|
|
17
19
|
buildEvidenceVector,
|
|
18
20
|
campaignMeanComposite,
|
|
19
21
|
composeGate,
|
|
22
|
+
createReferenceEquivalenceJudge,
|
|
20
23
|
defaultProductionGate,
|
|
21
24
|
emitLoopProvenance,
|
|
22
25
|
evolutionaryProposer,
|
|
@@ -27,19 +30,23 @@ import {
|
|
|
27
30
|
powerPreflight,
|
|
28
31
|
runEval,
|
|
29
32
|
runImprovementLoop,
|
|
33
|
+
runReferenceEquivalenceJudge,
|
|
30
34
|
surfaceContentHash,
|
|
31
35
|
surfaceHash
|
|
32
|
-
} from "../chunk-
|
|
33
|
-
import "../chunk-VI2UW6B6.js";
|
|
36
|
+
} from "../chunk-HQPHZGL6.js";
|
|
34
37
|
import {
|
|
38
|
+
createRunCostLedger,
|
|
35
39
|
fsCampaignStorage,
|
|
36
40
|
inMemoryCampaignStorage,
|
|
41
|
+
resolveRunDir,
|
|
37
42
|
runCampaign
|
|
38
|
-
} from "../chunk-
|
|
43
|
+
} from "../chunk-IDZTTFRR.js";
|
|
39
44
|
import {
|
|
40
|
-
buildDefaultAnalystRegistry
|
|
41
|
-
|
|
42
|
-
|
|
45
|
+
buildDefaultAnalystRegistry,
|
|
46
|
+
createChatClient
|
|
47
|
+
} from "../chunk-VF3XSYTI.js";
|
|
48
|
+
import "../chunk-HHWE3POT.js";
|
|
49
|
+
import "../chunk-MGEHEHSN.js";
|
|
43
50
|
import {
|
|
44
51
|
FileSystemOutcomeStore,
|
|
45
52
|
InMemoryOutcomeStore
|
|
@@ -47,25 +54,38 @@ import {
|
|
|
47
54
|
import "../chunk-ARU2PZFM.js";
|
|
48
55
|
import "../chunk-DPZAEKA6.js";
|
|
49
56
|
import "../chunk-PJQFMIOX.js";
|
|
50
|
-
import "../chunk-
|
|
57
|
+
import "../chunk-LQUTGLOZ.js";
|
|
51
58
|
import "../chunk-GGE4NNQT.js";
|
|
52
59
|
import {
|
|
53
60
|
LLM_INPUT_TOKEN_ATTR_KEYS,
|
|
54
61
|
LLM_MODEL_ATTR_KEYS,
|
|
55
62
|
LLM_OUTPUT_TOKEN_ATTR_KEYS
|
|
56
|
-
} from "../chunk-
|
|
63
|
+
} from "../chunk-S2F4J57L.js";
|
|
57
64
|
import {
|
|
58
65
|
parseRunRecordSafe
|
|
59
66
|
} from "../chunk-5UF54T55.js";
|
|
60
67
|
import "../chunk-XJYR7XFV.js";
|
|
61
68
|
import "../chunk-VSMTAMNK.js";
|
|
62
|
-
import "../chunk-
|
|
69
|
+
import "../chunk-NJC7U437.js";
|
|
70
|
+
import "../chunk-VCTY3W6J.js";
|
|
71
|
+
import "../chunk-VI2UW6B6.js";
|
|
63
72
|
import "../chunk-PC4UYEBM.js";
|
|
64
73
|
import "../chunk-ONWEPEDO.js";
|
|
65
74
|
import "../chunk-PZ5AY32C.js";
|
|
66
75
|
|
|
67
76
|
// src/contract/self-improve.ts
|
|
68
77
|
import { createHash } from "crypto";
|
|
78
|
+
var SelfImproveRunError = class extends Error {
|
|
79
|
+
cost;
|
|
80
|
+
receipts;
|
|
81
|
+
constructor(cause, ledger) {
|
|
82
|
+
const original = cause instanceof Error ? cause : new Error(String(cause));
|
|
83
|
+
super(original.message, { cause: original });
|
|
84
|
+
this.name = "SelfImproveRunError";
|
|
85
|
+
this.cost = ledger.summary();
|
|
86
|
+
this.receipts = ledger.list();
|
|
87
|
+
}
|
|
88
|
+
};
|
|
69
89
|
function splitTrainHoldout(scenarios, fraction) {
|
|
70
90
|
function hash(s) {
|
|
71
91
|
let h = 2166136261 >>> 0;
|
|
@@ -96,12 +116,26 @@ function meanComposite(byScenario) {
|
|
|
96
116
|
}
|
|
97
117
|
async function selfImprove(opts) {
|
|
98
118
|
const startedAt = Date.now();
|
|
119
|
+
const requestedRunDir = opts.runDir ?? `mem://selfImprove-${startedAt}`;
|
|
120
|
+
const runDir = resolveRunDir(requestedRunDir);
|
|
121
|
+
const storage = opts.storage ?? (runDir.startsWith("mem://") ? inMemoryCampaignStorage() : fsCampaignStorage());
|
|
122
|
+
const costLedger = createRunCostLedger({
|
|
123
|
+
storage,
|
|
124
|
+
runDir,
|
|
125
|
+
costCeilingUsd: opts.budget?.dollars
|
|
126
|
+
});
|
|
127
|
+
try {
|
|
128
|
+
return await runSelfImprove(opts, costLedger, startedAt, runDir, storage);
|
|
129
|
+
} catch (error) {
|
|
130
|
+
throw new SelfImproveRunError(error, costLedger);
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
async function runSelfImprove(opts, costLedger, startedAt, runDir, storage) {
|
|
99
134
|
const budget = opts.budget ?? {};
|
|
100
135
|
const generations = budget.generations ?? 3;
|
|
101
136
|
const populationSize = budget.populationSize ?? 2;
|
|
102
137
|
const maxConcurrency = budget.maxConcurrency ?? 2;
|
|
103
138
|
const holdoutFraction = budget.holdoutFraction ?? 0.25;
|
|
104
|
-
const costCeiling = budget.dollars;
|
|
105
139
|
const expectUsage = opts.expectUsage ?? "assert";
|
|
106
140
|
const explicitHoldout = budget.holdoutScenarios;
|
|
107
141
|
const { train, holdout } = explicitHoldout ? {
|
|
@@ -131,9 +165,6 @@ async function selfImprove(opts) {
|
|
|
131
165
|
holdoutScenarios: holdout,
|
|
132
166
|
deltaThreshold: 0.05
|
|
133
167
|
});
|
|
134
|
-
const runDir = opts.runDir ?? `mem://selfImprove-${startedAt}`;
|
|
135
|
-
const isMemRunDir = runDir.startsWith("mem://");
|
|
136
|
-
const storage = opts.storage ?? (isMemRunDir ? inMemoryCampaignStorage() : fsCampaignStorage());
|
|
137
168
|
if (opts.onProgress) {
|
|
138
169
|
opts.onProgress({ kind: "baseline.started", scenarios: opts.scenarios.length });
|
|
139
170
|
}
|
|
@@ -157,7 +188,7 @@ async function selfImprove(opts) {
|
|
|
157
188
|
runDir,
|
|
158
189
|
maxConcurrency,
|
|
159
190
|
cellPlacement: opts.cellPlacement,
|
|
160
|
-
|
|
191
|
+
costLedger,
|
|
161
192
|
expectUsage,
|
|
162
193
|
labeledStore: opts.labeledStore,
|
|
163
194
|
captureSource: opts.captureSource,
|
|
@@ -201,10 +232,8 @@ async function selfImprove(opts) {
|
|
|
201
232
|
lift: winnerStats.compositeMean - baseline.compositeMean
|
|
202
233
|
});
|
|
203
234
|
}
|
|
204
|
-
const
|
|
205
|
-
|
|
206
|
-
0
|
|
207
|
-
);
|
|
235
|
+
const cost = result.cost;
|
|
236
|
+
const totalCost = cost.totalCostUsd;
|
|
208
237
|
const insight = await analyzeRuns({
|
|
209
238
|
runs: [
|
|
210
239
|
...cellsToRunRecords(result.baselineCampaign.cells, "baseline", runDir, opts.baselineSurface),
|
|
@@ -256,6 +285,8 @@ async function selfImprove(opts) {
|
|
|
256
285
|
generationsExplored: result.generations.length,
|
|
257
286
|
durationMs,
|
|
258
287
|
totalCostUsd: totalCost,
|
|
288
|
+
cost,
|
|
289
|
+
receipts: costLedger.list(),
|
|
259
290
|
insight,
|
|
260
291
|
...power ? { power } : {},
|
|
261
292
|
raw: result
|
|
@@ -988,10 +1019,15 @@ function collectNumericAttrs(spans) {
|
|
|
988
1019
|
export {
|
|
989
1020
|
FileSystemOutcomeStore,
|
|
990
1021
|
InMemoryOutcomeStore,
|
|
1022
|
+
REFERENCE_EQUIVALENCE_INPUT_LIMITS,
|
|
1023
|
+
REFERENCE_EQUIVALENCE_JUDGE_VERSION,
|
|
1024
|
+
SelfImproveRunError,
|
|
991
1025
|
analyzeRuns,
|
|
992
1026
|
buildDefaultAnalystRegistry,
|
|
993
1027
|
buildEvidenceVector,
|
|
994
1028
|
composeGate,
|
|
1029
|
+
createChatClient,
|
|
1030
|
+
createReferenceEquivalenceJudge,
|
|
995
1031
|
defaultProductionGate,
|
|
996
1032
|
defineAgentEval,
|
|
997
1033
|
diffGenerations,
|
|
@@ -1020,6 +1056,7 @@ export {
|
|
|
1020
1056
|
runCampaign,
|
|
1021
1057
|
runEval,
|
|
1022
1058
|
runImprovementLoop,
|
|
1059
|
+
runReferenceEquivalenceJudge,
|
|
1023
1060
|
selfImprove
|
|
1024
1061
|
};
|
|
1025
1062
|
//# sourceMappingURL=index.js.map
|