@tangle-network/agent-eval 0.115.3 → 0.117.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +55 -0
- package/dist/analyst/index.d.ts +16 -11
- package/dist/analyst/index.js +33 -25
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-BYHg6Irm.d.ts → analyze-runs--2x39HZ7.d.ts} +3 -3
- package/dist/{baseline-DsNteOgR.d.ts → baseline-DKq3gJpP.d.ts} +6 -3
- package/dist/belief-state/index.d.ts +6 -6
- package/dist/belief-state/index.js +1 -1
- package/dist/benchmarks/index.d.ts +12 -5
- package/dist/benchmarks/index.js +11 -10
- package/dist/builder-eval/index.d.ts +4 -4
- package/dist/builder-eval/index.js +1 -1
- package/dist/{calibration-Dz8TQV4y.d.ts → calibration-C8MTS7cw.d.ts} +2 -2
- package/dist/campaign/index.d.ts +247 -34
- package/dist/campaign/index.js +33 -13
- package/dist/chunk-3YYRZDON.js +45 -0
- package/dist/chunk-3YYRZDON.js.map +1 -0
- package/dist/{chunk-RPDDVKI7.js → chunk-4JLWXDYA.js} +2 -2
- package/dist/{chunk-WSBUZMBU.js → chunk-CCZIVI3F.js} +54 -115
- package/dist/chunk-CCZIVI3F.js.map +1 -0
- package/dist/{chunk-J6P6PK2R.js → chunk-FQNLDL4D.js} +3 -3
- package/dist/{chunk-ONM6PEAE.js → chunk-GQCZRZ7L.js} +2 -2
- package/dist/chunk-HHWE3POT.js +94 -0
- package/dist/chunk-HHWE3POT.js.map +1 -0
- package/dist/{chunk-ADYLPOSX.js → chunk-HQPHZGL6.js} +1112 -135
- package/dist/chunk-HQPHZGL6.js.map +1 -0
- package/dist/{chunk-FAOEFFRT.js → chunk-IDZTTFRR.js} +390 -78
- package/dist/chunk-IDZTTFRR.js.map +1 -0
- package/dist/{chunk-3LXTCTWL.js → chunk-JSDVRFAP.js} +2 -2
- package/dist/{chunk-MHNQWM4I.js → chunk-LQUTGLOZ.js} +5 -1
- package/dist/chunk-LQUTGLOZ.js.map +1 -0
- package/dist/{chunk-4D5RVB3W.js → chunk-LTVG32KX.js} +30 -5
- package/dist/chunk-LTVG32KX.js.map +1 -0
- package/dist/{chunk-5S5NJ63F.js → chunk-MGEHEHSN.js} +807 -15
- package/dist/chunk-MGEHEHSN.js.map +1 -0
- package/dist/{chunk-GY4SYVPJ.js → chunk-NJC7U437.js} +97 -25
- package/dist/chunk-NJC7U437.js.map +1 -0
- package/dist/{chunk-NYFUT3B3.js → chunk-ODVOOEWQ.js} +31 -10
- package/dist/chunk-ODVOOEWQ.js.map +1 -0
- package/dist/{chunk-LNQEP766.js → chunk-S2F4J57L.js} +44 -4
- package/dist/chunk-S2F4J57L.js.map +1 -0
- package/dist/chunk-VCTY3W6J.js +798 -0
- package/dist/chunk-VCTY3W6J.js.map +1 -0
- package/dist/chunk-VF3XSYTI.js +545 -0
- package/dist/chunk-VF3XSYTI.js.map +1 -0
- package/dist/{chunk-TLDB7WRY.js → chunk-YZPO4UHR.js} +28 -31
- package/dist/chunk-YZPO4UHR.js.map +1 -0
- package/dist/{chunk-KG4TD7EQ.js → chunk-ZUXV7UWZ.js} +1425 -697
- package/dist/chunk-ZUXV7UWZ.js.map +1 -0
- package/dist/cli.js +4 -2
- package/dist/cli.js.map +1 -1
- package/dist/{code-agent-session-D-g04tcy.d.ts → code-agent-session-CjZsVd19.d.ts} +1 -1
- package/dist/contract/index.d.ts +45 -31
- package/dist/contract/index.js +58 -19
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-CcBiAEnn.d.ts → control-6vuGfmDH.d.ts} +5 -5
- package/dist/control.d.ts +6 -6
- package/dist/cost-ledger-DWy3XdJc.d.ts +183 -0
- package/dist/{default-registry-DltpYR5u.d.ts → default-registry-DaK8b3fv.d.ts} +2 -1
- package/dist/{emitter-BRchAAAx.d.ts → emitter-CjD7vUwv.d.ts} +2 -2
- package/dist/{failure-cluster-C48PiReX.d.ts → failure-cluster-DOAcSJ87.d.ts} +2 -2
- package/dist/{feedback-trajectory-pDcz1lQ1.d.ts → feedback-trajectory-BUnM58xL.d.ts} +3 -3
- package/dist/fuzz.d.ts +8 -16
- package/dist/fuzz.js +72 -42
- package/dist/fuzz.js.map +1 -1
- package/dist/{gepa-dne9JDPL.d.ts → gepa-eESocoDi.d.ts} +64 -12
- package/dist/hosted/index.d.ts +14 -7
- package/dist/{index-BTEpx9He.d.ts → index-PdX4VnPA.d.ts} +3 -3
- package/dist/index.d.ts +97 -55
- package/dist/index.js +343 -244
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-IwwvqZZv.d.ts → insight-report-DY4nDW9Q.d.ts} +1 -1
- package/dist/{integrity-qemeBAyx.d.ts → integrity-DqlBiLyK.d.ts} +1 -1
- package/dist/kind-factory-ClZmO25A.d.ts +171 -0
- package/dist/{llm-client-DyqEH4jH.d.ts → llm-client-qoDd18Qz.d.ts} +27 -3
- package/dist/meta-eval/index.d.ts +8 -7
- package/dist/meta-eval/index.js +1 -1
- package/dist/multishot/index.d.ts +10 -3
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +16 -6
- package/dist/pipelines/index.js +119 -23
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{kind-factory-DcNg13sZ.d.ts → policy-edit-wG9uFEFm.d.ts} +114 -167
- package/dist/{pre-registration-D8h7ZxNL.d.ts → pre-registration-BWQhJ3vz.d.ts} +23 -4
- package/dist/{provenance-Bibyg1U9.d.ts → provenance-DpjwyseI.d.ts} +28 -16
- package/dist/{query-Ck190MOd.d.ts → query-CF7PG61p.d.ts} +5 -3
- package/dist/{release-report-CCtzajxP.d.ts → release-report-C8G2i5Xi.d.ts} +2 -2
- package/dist/reporting.d.ts +10 -9
- package/dist/{researcher-Dq-EtpbE.d.ts → researcher-C8XyxQsu.d.ts} +7 -7
- package/dist/rl.d.ts +17 -12
- package/dist/rl.js +2 -2
- package/dist/{rubric-predictive-validity-DYTLjGWu.d.ts → rubric-predictive-validity-p49lLVrE.d.ts} +1 -1
- package/dist/{run-campaign-UADIM77S.js → run-campaign-IM26A6PD.js} +4 -2
- package/dist/{run-record-B7RTi_ix.d.ts → run-record-BDH49H2E.d.ts} +2 -2
- package/dist/{runtime-trajectory-Dws7Kpgi.d.ts → runtime-trajectory-DGBIUt4B.d.ts} +1 -1
- package/dist/{schema-SGWcK9wa.d.ts → schema-B3Q3l9Z_.d.ts} +2 -0
- package/dist/{semantic-concept-judge-DxJmRkyJ.d.ts → semantic-concept-judge-CXnPEJbf.d.ts} +23 -5
- package/dist/{statistics-oUbOJe-S.d.ts → statistics-KUnG73jH.d.ts} +1 -1
- package/dist/{storage-Dw_f7WMt.d.ts → storage-DrX3v_5B.d.ts} +12 -1
- package/dist/{store-BsVi7ncX.d.ts → store-DGqD0Pyo.d.ts} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-BJ5aNwZ1.d.ts → summary-report-C5bKFfm-.d.ts} +2 -2
- package/dist/{test-graded-scenario-mzYBKspu.d.ts → test-graded-scenario-B0ybnPY7.d.ts} +3 -3
- package/dist/traces.d.ts +19 -10
- package/dist/traces.js +16 -4
- package/dist/{types-C5gJrOVT.d.ts → types-BSw1rOUB.d.ts} +97 -38
- package/dist/{types-C7DGg5ex.d.ts → types-BkfcQnxV.d.ts} +15 -0
- package/dist/wire/index.d.ts +28 -19
- package/dist/wire/index.js +4 -2
- package/docs/design/loop-taxonomy.md +1 -2
- package/docs/distributed-driver.md +1 -1
- package/package.json +3 -3
- package/dist/chunk-4D5RVB3W.js.map +0 -1
- package/dist/chunk-5S5NJ63F.js.map +0 -1
- package/dist/chunk-ADYLPOSX.js.map +0 -1
- package/dist/chunk-FAOEFFRT.js.map +0 -1
- package/dist/chunk-GY4SYVPJ.js.map +0 -1
- package/dist/chunk-I6LVHOV3.js +0 -205
- package/dist/chunk-I6LVHOV3.js.map +0 -1
- package/dist/chunk-KG4TD7EQ.js.map +0 -1
- package/dist/chunk-LNQEP766.js.map +0 -1
- package/dist/chunk-MHNQWM4I.js.map +0 -1
- package/dist/chunk-NYFUT3B3.js.map +0 -1
- package/dist/chunk-QMXXSNC4.js +0 -761
- package/dist/chunk-QMXXSNC4.js.map +0 -1
- package/dist/chunk-TLDB7WRY.js.map +0 -1
- package/dist/chunk-WSBUZMBU.js.map +0 -1
- package/dist/cost-ledger-DuSqlw5B.d.ts +0 -113
- package/dist/policy-edit-RLn8GWof.d.ts +0 -103
- /package/dist/{chunk-RPDDVKI7.js.map → chunk-4JLWXDYA.js.map} +0 -0
- /package/dist/{chunk-J6P6PK2R.js.map → chunk-FQNLDL4D.js.map} +0 -0
- /package/dist/{chunk-ONM6PEAE.js.map → chunk-GQCZRZ7L.js.map} +0 -0
- /package/dist/{chunk-3LXTCTWL.js.map → chunk-JSDVRFAP.js.map} +0 -0
- /package/dist/{run-campaign-UADIM77S.js.map → run-campaign-IM26A6PD.js.map} +0 -0
package/dist/cli.js
CHANGED
|
@@ -5,8 +5,10 @@ import {
|
|
|
5
5
|
runRpcBatch,
|
|
6
6
|
runRpcOnce,
|
|
7
7
|
startServer
|
|
8
|
-
} from "./chunk-
|
|
9
|
-
import "./chunk-
|
|
8
|
+
} from "./chunk-LTVG32KX.js";
|
|
9
|
+
import "./chunk-NJC7U437.js";
|
|
10
|
+
import "./chunk-VCTY3W6J.js";
|
|
11
|
+
import "./chunk-VI2UW6B6.js";
|
|
10
12
|
import "./chunk-PC4UYEBM.js";
|
|
11
13
|
import "./chunk-ONWEPEDO.js";
|
|
12
14
|
import "./chunk-PZ5AY32C.js";
|
package/dist/cli.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"sources":["../src/cli.ts"],"sourcesContent":["#!/usr/bin/env node\n/**\n * agent-eval CLI.\n *\n * agent-eval serve [--port 5005] [--host 127.0.0.1]\n * agent-eval rpc <method> # one request from stdin → one response on stdout\n * agent-eval rpc-batch <method> # JSONL stdin → JSONL stdout\n * agent-eval openapi [--out path] # write OpenAPI spec\n * agent-eval version\n *\n * <method> is one of: judge, listRubrics, version. When omitted, the\n * stdin payload must be a full {method, params} envelope.\n */\nimport { writeFileSync } from 'node:fs'\nimport { handleVersion } from './wire/handlers'\nimport { buildOpenApi } from './wire/openapi'\nimport { runRpcBatch, runRpcOnce } from './wire/rpc'\nimport { startServer } from './wire/server'\n\ninterface Args {\n command: string\n positional: string[]\n flags: Record<string, string>\n}\n\nfunction parseArgs(argv: string[]): Args {\n const [command, ...rest] = argv\n const positional: string[] = []\n const flags: Record<string, string> = {}\n for (let i = 0; i < rest.length; i++) {\n const tok = rest[i]!\n if (tok.startsWith('--')) {\n const key = tok.slice(2)\n const next = rest[i + 1]\n if (next != null && !next.startsWith('--')) {\n flags[key] = next\n i++\n } else {\n flags[key] = 'true'\n }\n } else {\n positional.push(tok)\n }\n }\n return { command: command ?? 'help', positional, flags }\n}\n\nconst HELP = `agent-eval — wire-protocol entry point.\n\nCommands:\n serve [--port 5005] [--host 127.0.0.1]\n Start the HTTP server. POST /v1/judge, GET /v1/rubrics, GET /v1/version, GET /openapi.json.\n rpc <method>\n Read one JSON object from stdin (the params for <method>), write one\n JSON object to stdout. Method ∈ {judge, listRubrics, version}.\n rpc-batch <method>\n Like 'rpc' but JSONL in / JSONL out.\n openapi [--out openapi.json]\n Write the OpenAPI 3.1 spec.\n version\n Print server + wire-protocol version JSON.\n\nWithout arguments, prints this help.`\n\nasync function main(): Promise<number> {\n const { command, positional, flags } = parseArgs(process.argv.slice(2))\n\n switch (command) {\n case 'serve': {\n const port = Number(flags.port ?? 5005)\n const host = flags.host ?? '127.0.0.1'\n const server = startServer({ port, host })\n // Keep process alive on SIGINT/SIGTERM\n const shutdown = (sig: string) => {\n // eslint-disable-next-line no-console\n console.log(`[agent-eval] received ${sig}, shutting down`)\n server.close(() => process.exit(0))\n // Force exit after 5s if close hangs\n setTimeout(() => process.exit(1), 5000).unref()\n }\n process.on('SIGINT', () => shutdown('SIGINT'))\n process.on('SIGTERM', () => shutdown('SIGTERM'))\n // Block forever\n await new Promise(() => {})\n return 0\n }\n case 'rpc': {\n const [method] = positional\n return await runRpcOnce(method)\n }\n case 'rpc-batch': {\n const [method] = positional\n return await runRpcBatch(method)\n }\n case 'openapi': {\n const out = flags.out ?? 'openapi.json'\n const spec = buildOpenApi(handleVersion().version)\n writeFileSync(out, `${JSON.stringify(spec, null, 2)}\\n`, 'utf-8')\n // eslint-disable-next-line no-console\n console.log(`[agent-eval] wrote OpenAPI 3.1 spec to ${out}`)\n return 0\n }\n case 'version': {\n process.stdout.write(`${JSON.stringify(handleVersion(), null, 2)}\\n`)\n return 0\n }\n case 'help':\n case '--help':\n case '-h':\n case '':\n process.stdout.write(`${HELP}\\n`)\n return 0\n default:\n process.stderr.write(`unknown command: ${command}\\n${HELP}\\n`)\n return 1\n }\n}\n\nmain()\n .then((code) => process.exit(code))\n .catch((err) => {\n // eslint-disable-next-line no-console\n console.error('[agent-eval] cli error:', err)\n process.exit(1)\n })\n"],"mappings":"
|
|
1
|
+
{"version":3,"sources":["../src/cli.ts"],"sourcesContent":["#!/usr/bin/env node\n/**\n * agent-eval CLI.\n *\n * agent-eval serve [--port 5005] [--host 127.0.0.1]\n * agent-eval rpc <method> # one request from stdin → one response on stdout\n * agent-eval rpc-batch <method> # JSONL stdin → JSONL stdout\n * agent-eval openapi [--out path] # write OpenAPI spec\n * agent-eval version\n *\n * <method> is one of: judge, listRubrics, version. When omitted, the\n * stdin payload must be a full {method, params} envelope.\n */\nimport { writeFileSync } from 'node:fs'\nimport { handleVersion } from './wire/handlers'\nimport { buildOpenApi } from './wire/openapi'\nimport { runRpcBatch, runRpcOnce } from './wire/rpc'\nimport { startServer } from './wire/server'\n\ninterface Args {\n command: string\n positional: string[]\n flags: Record<string, string>\n}\n\nfunction parseArgs(argv: string[]): Args {\n const [command, ...rest] = argv\n const positional: string[] = []\n const flags: Record<string, string> = {}\n for (let i = 0; i < rest.length; i++) {\n const tok = rest[i]!\n if (tok.startsWith('--')) {\n const key = tok.slice(2)\n const next = rest[i + 1]\n if (next != null && !next.startsWith('--')) {\n flags[key] = next\n i++\n } else {\n flags[key] = 'true'\n }\n } else {\n positional.push(tok)\n }\n }\n return { command: command ?? 'help', positional, flags }\n}\n\nconst HELP = `agent-eval — wire-protocol entry point.\n\nCommands:\n serve [--port 5005] [--host 127.0.0.1]\n Start the HTTP server. POST /v1/judge, GET /v1/rubrics, GET /v1/version, GET /openapi.json.\n rpc <method>\n Read one JSON object from stdin (the params for <method>), write one\n JSON object to stdout. Method ∈ {judge, listRubrics, version}.\n rpc-batch <method>\n Like 'rpc' but JSONL in / JSONL out.\n openapi [--out openapi.json]\n Write the OpenAPI 3.1 spec.\n version\n Print server + wire-protocol version JSON.\n\nWithout arguments, prints this help.`\n\nasync function main(): Promise<number> {\n const { command, positional, flags } = parseArgs(process.argv.slice(2))\n\n switch (command) {\n case 'serve': {\n const port = Number(flags.port ?? 5005)\n const host = flags.host ?? '127.0.0.1'\n const server = startServer({ port, host })\n // Keep process alive on SIGINT/SIGTERM\n const shutdown = (sig: string) => {\n // eslint-disable-next-line no-console\n console.log(`[agent-eval] received ${sig}, shutting down`)\n server.close(() => process.exit(0))\n // Force exit after 5s if close hangs\n setTimeout(() => process.exit(1), 5000).unref()\n }\n process.on('SIGINT', () => shutdown('SIGINT'))\n process.on('SIGTERM', () => shutdown('SIGTERM'))\n // Block forever\n await new Promise(() => {})\n return 0\n }\n case 'rpc': {\n const [method] = positional\n return await runRpcOnce(method)\n }\n case 'rpc-batch': {\n const [method] = positional\n return await runRpcBatch(method)\n }\n case 'openapi': {\n const out = flags.out ?? 'openapi.json'\n const spec = buildOpenApi(handleVersion().version)\n writeFileSync(out, `${JSON.stringify(spec, null, 2)}\\n`, 'utf-8')\n // eslint-disable-next-line no-console\n console.log(`[agent-eval] wrote OpenAPI 3.1 spec to ${out}`)\n return 0\n }\n case 'version': {\n process.stdout.write(`${JSON.stringify(handleVersion(), null, 2)}\\n`)\n return 0\n }\n case 'help':\n case '--help':\n case '-h':\n case '':\n process.stdout.write(`${HELP}\\n`)\n return 0\n default:\n process.stderr.write(`unknown command: ${command}\\n${HELP}\\n`)\n return 1\n }\n}\n\nmain()\n .then((code) => process.exit(code))\n .catch((err) => {\n // eslint-disable-next-line no-console\n console.error('[agent-eval] cli error:', err)\n process.exit(1)\n })\n"],"mappings":";;;;;;;;;;;;;;;;AAaA,SAAS,qBAAqB;AAY9B,SAAS,UAAU,MAAsB;AACvC,QAAM,CAAC,SAAS,GAAG,IAAI,IAAI;AAC3B,QAAM,aAAuB,CAAC;AAC9B,QAAM,QAAgC,CAAC;AACvC,WAAS,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;AACpC,UAAM,MAAM,KAAK,CAAC;AAClB,QAAI,IAAI,WAAW,IAAI,GAAG;AACxB,YAAM,MAAM,IAAI,MAAM,CAAC;AACvB,YAAM,OAAO,KAAK,IAAI,CAAC;AACvB,UAAI,QAAQ,QAAQ,CAAC,KAAK,WAAW,IAAI,GAAG;AAC1C,cAAM,GAAG,IAAI;AACb;AAAA,MACF,OAAO;AACL,cAAM,GAAG,IAAI;AAAA,MACf;AAAA,IACF,OAAO;AACL,iBAAW,KAAK,GAAG;AAAA,IACrB;AAAA,EACF;AACA,SAAO,EAAE,SAAS,WAAW,QAAQ,YAAY,MAAM;AACzD;AAEA,IAAM,OAAO;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAiBb,eAAe,OAAwB;AACrC,QAAM,EAAE,SAAS,YAAY,MAAM,IAAI,UAAU,QAAQ,KAAK,MAAM,CAAC,CAAC;AAEtE,UAAQ,SAAS;AAAA,IACf,KAAK,SAAS;AACZ,YAAM,OAAO,OAAO,MAAM,QAAQ,IAAI;AACtC,YAAM,OAAO,MAAM,QAAQ;AAC3B,YAAM,SAAS,YAAY,EAAE,MAAM,KAAK,CAAC;AAEzC,YAAM,WAAW,CAAC,QAAgB;AAEhC,gBAAQ,IAAI,yBAAyB,GAAG,iBAAiB;AACzD,eAAO,MAAM,MAAM,QAAQ,KAAK,CAAC,CAAC;AAElC,mBAAW,MAAM,QAAQ,KAAK,CAAC,GAAG,GAAI,EAAE,MAAM;AAAA,MAChD;AACA,cAAQ,GAAG,UAAU,MAAM,SAAS,QAAQ,CAAC;AAC7C,cAAQ,GAAG,WAAW,MAAM,SAAS,SAAS,CAAC;AAE/C,YAAM,IAAI,QAAQ,MAAM;AAAA,MAAC,CAAC;AAC1B,aAAO;AAAA,IACT;AAAA,IACA,KAAK,OAAO;AACV,YAAM,CAAC,MAAM,IAAI;AACjB,aAAO,MAAM,WAAW,MAAM;AAAA,IAChC;AAAA,IACA,KAAK,aAAa;AAChB,YAAM,CAAC,MAAM,IAAI;AACjB,aAAO,MAAM,YAAY,MAAM;AAAA,IACjC;AAAA,IACA,KAAK,WAAW;AACd,YAAM,MAAM,MAAM,OAAO;AACzB,YAAM,OAAO,aAAa,cAAc,EAAE,OAAO;AACjD,oBAAc,KAAK,GAAG,KAAK,UAAU,MAAM,MAAM,CAAC,CAAC;AAAA,GAAM,OAAO;AAEhE,cAAQ,IAAI,0CAA0C,GAAG,EAAE;AAC3D,aAAO;AAAA,IACT;AAAA,IACA,KAAK,WAAW;AACd,cAAQ,OAAO,MAAM,GAAG,KAAK,UAAU,cAAc,GAAG,MAAM,CAAC,CAAC;AAAA,CAAI;AACpE,aAAO;AAAA,IACT;AAAA,IACA,KAAK;AAAA,IACL,KAAK;AAAA,IACL,KAAK;AAAA,IACL,KAAK;AACH,cAAQ,OAAO,MAAM,GAAG,IAAI;AAAA,CAAI;AAChC,aAAO;AAAA,IACT;AACE,cAAQ,OAAO,MAAM,oBAAoB,OAAO;AAAA,EAAK,IAAI;AAAA,CAAI;AAC7D,aAAO;AAAA,EACX;AACF;AAEA,KAAK,EACF,KAAK,CAAC,SAAS,QAAQ,KAAK,IAAI,CAAC,EACjC,MAAM,CAAC,QAAQ;AAEd,UAAQ,MAAM,2BAA2B,GAAG;AAC5C,UAAQ,KAAK,CAAC;AAChB,CAAC;","names":[]}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { a as RunSplitTag, b as RunCostProvenance, R as RunRecord } from './run-record-BDH49H2E.js';
|
|
2
2
|
|
|
3
3
|
type CodeAgentSessionSource = 'codex' | 'claude-code' | 'opencode' | 'kimi-code' | 'pi';
|
|
4
4
|
interface ParsedCodeAgentJsonl {
|
package/dist/contract/index.d.ts
CHANGED
|
@@ -1,36 +1,38 @@
|
|
|
1
|
-
import { S as Scenario, M as MutableSurface, D as DispatchContext, b as JudgeConfig,
|
|
2
|
-
export {
|
|
3
|
-
import { L as LoopProvenanceRecord, P as PowerPreflight, R as RunEvalOptions } from '../provenance-
|
|
4
|
-
export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, D as DefaultProductionGateOptions, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, O as ObjectiveSource, c as ParetoSignificanceGateOptions, d as PromotionObjective, e as PromotionPolicy, f as buildEvidenceVector, g as composeGate, h as defaultProductionGate, i as evolutionaryProposer, j as heldOutGate, p as paretoPolicy, k as paretoSignificanceGate, r as runEval } from '../provenance-
|
|
5
|
-
import { R as RunOptimizationOptions, a as RunImprovementLoopResult } from '../gepa-
|
|
6
|
-
export { G as GepaProposerOptions, b as
|
|
7
|
-
|
|
8
|
-
|
|
1
|
+
import { S as Scenario, M as MutableSurface, D as DispatchContext, b as JudgeConfig, c as SurfaceProposer, G as Gate, L as LabeledScenarioStore, C as CampaignResult, d as GateDecision } from '../types-BSw1rOUB.js';
|
|
2
|
+
export { e as CampaignAggregates, f as CampaignArtifactWriter, g as CampaignCellResult, h as CampaignCostMeter, i as CampaignTraceWriter, j as CodeSurface, k as Dispatch, l as GateContext, m as GateResult, n as GenerationCandidate, o as GenerationRecord, a as JudgeDimension, J as JudgeScore, p as Mutator, O as OptimizationProposer, q as OptimizerConfig, r as SessionScript } from '../types-BSw1rOUB.js';
|
|
3
|
+
import { L as LoopProvenanceRecord, P as PowerPreflight, R as RunEvalOptions } from '../provenance-DpjwyseI.js';
|
|
4
|
+
export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, D as DefaultProductionGateOptions, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, O as ObjectiveSource, c as ParetoSignificanceGateOptions, d as PromotionObjective, e as PromotionPolicy, f as buildEvidenceVector, g as composeGate, h as defaultProductionGate, i as evolutionaryProposer, j as heldOutGate, p as paretoPolicy, k as paretoSignificanceGate, r as runEval } from '../provenance-DpjwyseI.js';
|
|
5
|
+
import { R as RunOptimizationOptions, a as RunImprovementLoopResult } from '../gepa-eESocoDi.js';
|
|
6
|
+
export { G as GepaProposerOptions, b as REFERENCE_EQUIVALENCE_INPUT_LIMITS, c as REFERENCE_EQUIVALENCE_JUDGE_VERSION, d as ReferenceEquivalenceJudgeInput, e as ReferenceEquivalenceJudgeOptions, f as ReferenceEquivalenceJudgeResult, g as ReferenceEquivalenceScenario, h as RunCampaignOptions, i as RunImprovementLoopOptions, j as createReferenceEquivalenceJudge, k as gepaProposer, r as runCampaign, l as runImprovementLoop, m as runReferenceEquivalenceJudge } from '../gepa-eESocoDi.js';
|
|
7
|
+
export { c as AnalystFinding, C as ChatClient, p as CreateChatClientOpts, V as createChatClient } from '../policy-edit-wG9uFEFm.js';
|
|
8
|
+
import { C as CampaignStorage } from '../storage-DrX3v_5B.js';
|
|
9
|
+
export { f as fsCampaignStorage, i as inMemoryCampaignStorage } from '../storage-DrX3v_5B.js';
|
|
9
10
|
export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore, b as OutcomeStore } from '../outcome-store-rnXLEqSn.js';
|
|
10
11
|
import { HostedTenant, EvalRunCellScore, EvalRunGenerationSnapshot, EvalRunEvent, TraceSpanEvent } from '../hosted/index.js';
|
|
11
|
-
import {
|
|
12
|
-
import {
|
|
13
|
-
|
|
14
|
-
export {
|
|
15
|
-
export {
|
|
16
|
-
import { A as AnalyzeRunsOptions } from '../analyze-runs
|
|
17
|
-
export { a as analyzeRuns } from '../analyze-runs
|
|
18
|
-
export { C as CodeAgentSessionDiagnostic, a as CodeAgentSessionIntakeOptions, b as CodeAgentSessionIntakeResult, c as CodeAgentSessionMetrics, d as CodeAgentSessionSource, P as ParsedCodeAgentJsonl, f as fromClaudeCodeSession, e as fromCodexSession, g as fromKimiCodeSession, h as fromOpenCodeSession, i as fromPiSession, j as fromPigraphSession, p as parseCodeAgentJsonl } from '../code-agent-session-
|
|
19
|
-
import '../
|
|
12
|
+
import { b as CostLedgerSummary, d as CostReceipt, C as CostLedger } from '../cost-ledger-DWy3XdJc.js';
|
|
13
|
+
import { R as RunRecord, a as RunSplitTag } from '../run-record-BDH49H2E.js';
|
|
14
|
+
import { I as InsightReport } from '../insight-report-DY4nDW9Q.js';
|
|
15
|
+
export { C as CostProvenanceSummary, F as FailureClusterInsight, a as InterRaterInsight, J as JudgeInsight, L as LiftInsight, O as OutcomeCorrelationInsight, R as Recommendation, b as ReleaseSummary, S as ScalarDistribution } from '../insight-report-DY4nDW9Q.js';
|
|
16
|
+
export { D as DefaultAnalystRegistryOptions, c as buildDefaultAnalystRegistry } from '../default-registry-DaK8b3fv.js';
|
|
17
|
+
import { A as AnalyzeRunsOptions } from '../analyze-runs--2x39HZ7.js';
|
|
18
|
+
export { a as analyzeRuns } from '../analyze-runs--2x39HZ7.js';
|
|
19
|
+
export { C as CodeAgentSessionDiagnostic, a as CodeAgentSessionIntakeOptions, b as CodeAgentSessionIntakeResult, c as CodeAgentSessionMetrics, d as CodeAgentSessionSource, P as ParsedCodeAgentJsonl, f as fromClaudeCodeSession, e as fromCodexSession, g as fromKimiCodeSession, h as fromOpenCodeSession, i as fromPiSession, j as fromPigraphSession, p as parseCodeAgentJsonl } from '../code-agent-session-CjZsVd19.js';
|
|
20
|
+
import '../llm-client-qoDd18Qz.js';
|
|
21
|
+
import '../errors-oeQrLqXC.js';
|
|
22
|
+
import '../raw-provider-sink-C46HDghv.js';
|
|
23
|
+
import '../statistics-KUnG73jH.js';
|
|
20
24
|
import '../judge-calibration-7C-IDmKr.js';
|
|
21
|
-
import '../types-
|
|
25
|
+
import '../types-BkfcQnxV.js';
|
|
22
26
|
import '@tangle-network/tcloud';
|
|
23
27
|
import '../dataset-NENEzRgk.js';
|
|
24
|
-
import '../
|
|
25
|
-
import '../
|
|
26
|
-
import '../
|
|
27
|
-
import '../llm-client-DyqEH4jH.js';
|
|
28
|
-
import '../raw-provider-sink-C46HDghv.js';
|
|
28
|
+
import '../store-DGqD0Pyo.js';
|
|
29
|
+
import '../schema-B3Q3l9Z_.js';
|
|
30
|
+
import '../store-C1YxJDEK.js';
|
|
29
31
|
import '@tangle-network/agent-interface';
|
|
30
|
-
import '../summary-report-
|
|
31
|
-
import '../failure-cluster-
|
|
32
|
+
import '../summary-report-C5bKFfm-.js';
|
|
33
|
+
import '../failure-cluster-DOAcSJ87.js';
|
|
32
34
|
import '@ax-llm/ax';
|
|
33
|
-
import '../
|
|
35
|
+
import '../kind-factory-ClZmO25A.js';
|
|
34
36
|
import 'zod';
|
|
35
37
|
|
|
36
38
|
/**
|
|
@@ -58,8 +60,8 @@ import 'zod';
|
|
|
58
60
|
*/
|
|
59
61
|
|
|
60
62
|
interface SelfImproveBudget {
|
|
61
|
-
/** Hard
|
|
62
|
-
*
|
|
63
|
+
/** Hard spend cap across the full run. Each paid call reserves its enforced
|
|
64
|
+
* maximum before dispatch, so completed spend cannot cross this amount. */
|
|
63
65
|
dollars?: number;
|
|
64
66
|
/** How many improvement generations to explore. Default 3. Set 0 to
|
|
65
67
|
* skip improvement entirely (selfImprove becomes a baseline-only run). */
|
|
@@ -75,7 +77,8 @@ interface SelfImproveBudget {
|
|
|
75
77
|
holdoutScenarios?: Scenario[];
|
|
76
78
|
/** Per-scenario replicates per cell — raises bootstrap-CI tightness. Default 1. */
|
|
77
79
|
reps?: number;
|
|
78
|
-
/**
|
|
80
|
+
/** @deprecated Must be 1 when supplied. The loop promotes only a candidate
|
|
81
|
+
* that replaces its single global incumbent. */
|
|
79
82
|
promoteTopK?: number;
|
|
80
83
|
}
|
|
81
84
|
interface SelfImproveLlm {
|
|
@@ -282,8 +285,13 @@ interface SelfImproveResult<TScenario extends Scenario, TArtifact> {
|
|
|
282
285
|
generationsExplored: number;
|
|
283
286
|
/** Wall-clock total. */
|
|
284
287
|
durationMs: number;
|
|
285
|
-
/** Total cost across
|
|
288
|
+
/** Total newly observed cost across the full run. */
|
|
286
289
|
totalCostUsd: number;
|
|
290
|
+
/** Canonical run-wide spend summary. */
|
|
291
|
+
cost: CostLedgerSummary;
|
|
292
|
+
/** Run-wide receipts across proposal, search, holdout, judging, analysis,
|
|
293
|
+
* and promotion work, with phase and actor attribution. */
|
|
294
|
+
receipts: CostReceipt[];
|
|
287
295
|
/**
|
|
288
296
|
* Rigor packet: distributional summary, paired-bootstrap lift CI,
|
|
289
297
|
* judge stats, contamination check, recommendations. Wired through
|
|
@@ -303,6 +311,12 @@ interface SelfImproveResult<TScenario extends Scenario, TArtifact> {
|
|
|
303
311
|
*/
|
|
304
312
|
raw: RunImprovementLoopResult<TArtifact, TScenario>;
|
|
305
313
|
}
|
|
314
|
+
/** Failed self-improvement run with an immutable receipt snapshot. */
|
|
315
|
+
declare class SelfImproveRunError extends Error {
|
|
316
|
+
readonly cost: CostLedgerSummary;
|
|
317
|
+
readonly receipts: CostReceipt[];
|
|
318
|
+
constructor(cause: unknown, ledger: CostLedger);
|
|
319
|
+
}
|
|
306
320
|
/**
|
|
307
321
|
* One-shot self-improvement loop. See module docstring for defaults +
|
|
308
322
|
* extension points.
|
|
@@ -841,4 +855,4 @@ interface FromOtelSpansOptions {
|
|
|
841
855
|
}
|
|
842
856
|
declare function fromOtelSpans(opts: FromOtelSpansOptions): RunRecord[];
|
|
843
857
|
|
|
844
|
-
export { type AgentEvalAgent, type AgentEvalEvaluateOptions, type AgentEvalImproveOptions, type AgentTraceContributor, type AgentTraceContributorType, type AgentTraceConversation, type AgentTraceFile, type AgentTraceIndex, type AgentTraceRange, type AgentTraceRecord, AnalyzeRunsOptions, type AuthoringProvenance, CampaignResult, CampaignStorage, type DefineAgentEvalOptions, type DefinedAgentEval, DispatchContext, type EvalCellScoreDelta, type EvalDimensionDelta, type EvalGenerationDiff, type EvalReportingSuiteInput, type EvalReportingSuiteOptions, type EvalReportingSuiteResult, type EvalRunDiff, type FeedbackTableMeta, type FeedbackTableRow, type FromFeedbackTableOptions, type FromFeedbackTableResult, type FromOtelSpansOptions, type FromRunRecordDirOptions, type FromRunRecordDirResult, Gate, GateDecision, HostedTenant, InsightReport, JudgeConfig, MutableSurface, type PartitionByAuthoringModelResult, RunEvalOptions, RunImprovementLoopResult, type RunRecordRejection, Scenario, type SelfImproveBudget, type SelfImproveLlm, type SelfImproveOptions, type SelfImproveProgressEvent, type SelfImproveResult, SurfaceProposer, defineAgentEval, diffGenerations, diffRunBaselineToWinner, diffRuns, evalReportingSuite, fromFeedbackTable, fromOtelSpans, fromRunRecordDir, parseAgentTrace, partitionRunsByAuthoringModel, selfImprove };
|
|
858
|
+
export { type AgentEvalAgent, type AgentEvalEvaluateOptions, type AgentEvalImproveOptions, type AgentTraceContributor, type AgentTraceContributorType, type AgentTraceConversation, type AgentTraceFile, type AgentTraceIndex, type AgentTraceRange, type AgentTraceRecord, AnalyzeRunsOptions, type AuthoringProvenance, CampaignResult, CampaignStorage, type DefineAgentEvalOptions, type DefinedAgentEval, DispatchContext, type EvalCellScoreDelta, type EvalDimensionDelta, type EvalGenerationDiff, type EvalReportingSuiteInput, type EvalReportingSuiteOptions, type EvalReportingSuiteResult, type EvalRunDiff, type FeedbackTableMeta, type FeedbackTableRow, type FromFeedbackTableOptions, type FromFeedbackTableResult, type FromOtelSpansOptions, type FromRunRecordDirOptions, type FromRunRecordDirResult, Gate, GateDecision, HostedTenant, InsightReport, JudgeConfig, MutableSurface, type PartitionByAuthoringModelResult, RunEvalOptions, RunImprovementLoopResult, type RunRecordRejection, Scenario, type SelfImproveBudget, type SelfImproveLlm, type SelfImproveOptions, type SelfImproveProgressEvent, type SelfImproveResult, SelfImproveRunError, SurfaceProposer, defineAgentEval, diffGenerations, diffRunBaselineToWinner, diffRuns, evalReportingSuite, fromFeedbackTable, fromOtelSpans, fromRunRecordDir, parseAgentTrace, partitionRunsByAuthoringModel, selfImprove };
|
package/dist/contract/index.js
CHANGED
|
@@ -12,10 +12,14 @@ import {
|
|
|
12
12
|
} from "../chunk-ZZUXHH3R.js";
|
|
13
13
|
import {
|
|
14
14
|
analyzeRuns
|
|
15
|
-
} from "../chunk-
|
|
15
|
+
} from "../chunk-FQNLDL4D.js";
|
|
16
16
|
import {
|
|
17
|
+
REFERENCE_EQUIVALENCE_INPUT_LIMITS,
|
|
18
|
+
REFERENCE_EQUIVALENCE_JUDGE_VERSION,
|
|
17
19
|
buildEvidenceVector,
|
|
20
|
+
campaignMeanComposite,
|
|
18
21
|
composeGate,
|
|
22
|
+
createReferenceEquivalenceJudge,
|
|
19
23
|
defaultProductionGate,
|
|
20
24
|
emitLoopProvenance,
|
|
21
25
|
evolutionaryProposer,
|
|
@@ -26,19 +30,23 @@ import {
|
|
|
26
30
|
powerPreflight,
|
|
27
31
|
runEval,
|
|
28
32
|
runImprovementLoop,
|
|
33
|
+
runReferenceEquivalenceJudge,
|
|
29
34
|
surfaceContentHash,
|
|
30
35
|
surfaceHash
|
|
31
|
-
} from "../chunk-
|
|
32
|
-
import "../chunk-VI2UW6B6.js";
|
|
36
|
+
} from "../chunk-HQPHZGL6.js";
|
|
33
37
|
import {
|
|
38
|
+
createRunCostLedger,
|
|
34
39
|
fsCampaignStorage,
|
|
35
40
|
inMemoryCampaignStorage,
|
|
41
|
+
resolveRunDir,
|
|
36
42
|
runCampaign
|
|
37
|
-
} from "../chunk-
|
|
43
|
+
} from "../chunk-IDZTTFRR.js";
|
|
38
44
|
import {
|
|
39
|
-
buildDefaultAnalystRegistry
|
|
40
|
-
|
|
41
|
-
|
|
45
|
+
buildDefaultAnalystRegistry,
|
|
46
|
+
createChatClient
|
|
47
|
+
} from "../chunk-VF3XSYTI.js";
|
|
48
|
+
import "../chunk-HHWE3POT.js";
|
|
49
|
+
import "../chunk-MGEHEHSN.js";
|
|
42
50
|
import {
|
|
43
51
|
FileSystemOutcomeStore,
|
|
44
52
|
InMemoryOutcomeStore
|
|
@@ -46,25 +54,38 @@ import {
|
|
|
46
54
|
import "../chunk-ARU2PZFM.js";
|
|
47
55
|
import "../chunk-DPZAEKA6.js";
|
|
48
56
|
import "../chunk-PJQFMIOX.js";
|
|
49
|
-
import "../chunk-
|
|
57
|
+
import "../chunk-LQUTGLOZ.js";
|
|
50
58
|
import "../chunk-GGE4NNQT.js";
|
|
51
59
|
import {
|
|
52
60
|
LLM_INPUT_TOKEN_ATTR_KEYS,
|
|
53
61
|
LLM_MODEL_ATTR_KEYS,
|
|
54
62
|
LLM_OUTPUT_TOKEN_ATTR_KEYS
|
|
55
|
-
} from "../chunk-
|
|
63
|
+
} from "../chunk-S2F4J57L.js";
|
|
56
64
|
import {
|
|
57
65
|
parseRunRecordSafe
|
|
58
66
|
} from "../chunk-5UF54T55.js";
|
|
59
67
|
import "../chunk-XJYR7XFV.js";
|
|
60
68
|
import "../chunk-VSMTAMNK.js";
|
|
61
|
-
import "../chunk-
|
|
69
|
+
import "../chunk-NJC7U437.js";
|
|
70
|
+
import "../chunk-VCTY3W6J.js";
|
|
71
|
+
import "../chunk-VI2UW6B6.js";
|
|
62
72
|
import "../chunk-PC4UYEBM.js";
|
|
63
73
|
import "../chunk-ONWEPEDO.js";
|
|
64
74
|
import "../chunk-PZ5AY32C.js";
|
|
65
75
|
|
|
66
76
|
// src/contract/self-improve.ts
|
|
67
77
|
import { createHash } from "crypto";
|
|
78
|
+
var SelfImproveRunError = class extends Error {
|
|
79
|
+
cost;
|
|
80
|
+
receipts;
|
|
81
|
+
constructor(cause, ledger) {
|
|
82
|
+
const original = cause instanceof Error ? cause : new Error(String(cause));
|
|
83
|
+
super(original.message, { cause: original });
|
|
84
|
+
this.name = "SelfImproveRunError";
|
|
85
|
+
this.cost = ledger.summary();
|
|
86
|
+
this.receipts = ledger.list();
|
|
87
|
+
}
|
|
88
|
+
};
|
|
68
89
|
function splitTrainHoldout(scenarios, fraction) {
|
|
69
90
|
function hash(s) {
|
|
70
91
|
let h = 2166136261 >>> 0;
|
|
@@ -95,12 +116,26 @@ function meanComposite(byScenario) {
|
|
|
95
116
|
}
|
|
96
117
|
async function selfImprove(opts) {
|
|
97
118
|
const startedAt = Date.now();
|
|
119
|
+
const requestedRunDir = opts.runDir ?? `mem://selfImprove-${startedAt}`;
|
|
120
|
+
const runDir = resolveRunDir(requestedRunDir);
|
|
121
|
+
const storage = opts.storage ?? (runDir.startsWith("mem://") ? inMemoryCampaignStorage() : fsCampaignStorage());
|
|
122
|
+
const costLedger = createRunCostLedger({
|
|
123
|
+
storage,
|
|
124
|
+
runDir,
|
|
125
|
+
costCeilingUsd: opts.budget?.dollars
|
|
126
|
+
});
|
|
127
|
+
try {
|
|
128
|
+
return await runSelfImprove(opts, costLedger, startedAt, runDir, storage);
|
|
129
|
+
} catch (error) {
|
|
130
|
+
throw new SelfImproveRunError(error, costLedger);
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
async function runSelfImprove(opts, costLedger, startedAt, runDir, storage) {
|
|
98
134
|
const budget = opts.budget ?? {};
|
|
99
135
|
const generations = budget.generations ?? 3;
|
|
100
136
|
const populationSize = budget.populationSize ?? 2;
|
|
101
137
|
const maxConcurrency = budget.maxConcurrency ?? 2;
|
|
102
138
|
const holdoutFraction = budget.holdoutFraction ?? 0.25;
|
|
103
|
-
const costCeiling = budget.dollars;
|
|
104
139
|
const expectUsage = opts.expectUsage ?? "assert";
|
|
105
140
|
const explicitHoldout = budget.holdoutScenarios;
|
|
106
141
|
const { train, holdout } = explicitHoldout ? {
|
|
@@ -130,9 +165,6 @@ async function selfImprove(opts) {
|
|
|
130
165
|
holdoutScenarios: holdout,
|
|
131
166
|
deltaThreshold: 0.05
|
|
132
167
|
});
|
|
133
|
-
const runDir = opts.runDir ?? `mem://selfImprove-${startedAt}`;
|
|
134
|
-
const isMemRunDir = runDir.startsWith("mem://");
|
|
135
|
-
const storage = opts.storage ?? (isMemRunDir ? inMemoryCampaignStorage() : fsCampaignStorage());
|
|
136
168
|
if (opts.onProgress) {
|
|
137
169
|
opts.onProgress({ kind: "baseline.started", scenarios: opts.scenarios.length });
|
|
138
170
|
}
|
|
@@ -156,7 +188,7 @@ async function selfImprove(opts) {
|
|
|
156
188
|
runDir,
|
|
157
189
|
maxConcurrency,
|
|
158
190
|
cellPlacement: opts.cellPlacement,
|
|
159
|
-
|
|
191
|
+
costLedger,
|
|
160
192
|
expectUsage,
|
|
161
193
|
labeledStore: opts.labeledStore,
|
|
162
194
|
captureSource: opts.captureSource,
|
|
@@ -200,10 +232,8 @@ async function selfImprove(opts) {
|
|
|
200
232
|
lift: winnerStats.compositeMean - baseline.compositeMean
|
|
201
233
|
});
|
|
202
234
|
}
|
|
203
|
-
const
|
|
204
|
-
|
|
205
|
-
0
|
|
206
|
-
);
|
|
235
|
+
const cost = result.cost;
|
|
236
|
+
const totalCost = cost.totalCostUsd;
|
|
207
237
|
const insight = await analyzeRuns({
|
|
208
238
|
runs: [
|
|
209
239
|
...cellsToRunRecords(result.baselineCampaign.cells, "baseline", runDir, opts.baselineSurface),
|
|
@@ -223,6 +253,7 @@ async function selfImprove(opts) {
|
|
|
223
253
|
winnerLabel: result.winnerLabel,
|
|
224
254
|
winnerRationale: result.winnerRationale,
|
|
225
255
|
diff: result.promotedDiff,
|
|
256
|
+
baselineSearchComposite: campaignMeanComposite(result.baselineCampaign),
|
|
226
257
|
generations: result.generations.map((g) => ({
|
|
227
258
|
generationIndex: g.record.generationIndex,
|
|
228
259
|
candidates: g.record.candidates,
|
|
@@ -254,6 +285,8 @@ async function selfImprove(opts) {
|
|
|
254
285
|
generationsExplored: result.generations.length,
|
|
255
286
|
durationMs,
|
|
256
287
|
totalCostUsd: totalCost,
|
|
288
|
+
cost,
|
|
289
|
+
receipts: costLedger.list(),
|
|
257
290
|
insight,
|
|
258
291
|
...power ? { power } : {},
|
|
259
292
|
raw: result
|
|
@@ -986,10 +1019,15 @@ function collectNumericAttrs(spans) {
|
|
|
986
1019
|
export {
|
|
987
1020
|
FileSystemOutcomeStore,
|
|
988
1021
|
InMemoryOutcomeStore,
|
|
1022
|
+
REFERENCE_EQUIVALENCE_INPUT_LIMITS,
|
|
1023
|
+
REFERENCE_EQUIVALENCE_JUDGE_VERSION,
|
|
1024
|
+
SelfImproveRunError,
|
|
989
1025
|
analyzeRuns,
|
|
990
1026
|
buildDefaultAnalystRegistry,
|
|
991
1027
|
buildEvidenceVector,
|
|
992
1028
|
composeGate,
|
|
1029
|
+
createChatClient,
|
|
1030
|
+
createReferenceEquivalenceJudge,
|
|
993
1031
|
defaultProductionGate,
|
|
994
1032
|
defineAgentEval,
|
|
995
1033
|
diffGenerations,
|
|
@@ -1018,6 +1056,7 @@ export {
|
|
|
1018
1056
|
runCampaign,
|
|
1019
1057
|
runEval,
|
|
1020
1058
|
runImprovementLoop,
|
|
1059
|
+
runReferenceEquivalenceJudge,
|
|
1021
1060
|
selfImprove
|
|
1022
1061
|
};
|
|
1023
1062
|
//# sourceMappingURL=index.js.map
|