@tangle-network/agent-eval 0.173.2 → 0.174.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +40 -0
- package/dist/{proposal-findings-bko3GGy-.js → abort-signal-CtzAM_sJ.js} +11 -11
- package/dist/abort-signal-CtzAM_sJ.js.map +1 -0
- package/dist/adapters/http.d.ts +2 -2
- package/dist/agent-profile-_xPxqVJt.d.ts +488 -0
- package/dist/agent-profile-_xPxqVJt.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +7 -9
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +8 -8
- package/dist/{benchmark-C4wk_Sjr.js → benchmark-DQKzykkO.js} +2 -2
- package/dist/{benchmark-C4wk_Sjr.js.map → benchmark-DQKzykkO.js.map} +1 -1
- package/dist/{benchmark-command-CS6gVHVq.js → benchmark-command-mZIlR-ra.js} +13 -13
- package/dist/{benchmark-command-CS6gVHVq.js.map → benchmark-command-mZIlR-ra.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +3 -4
- package/dist/benchmarks/index.d.ts.map +1 -1
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +5 -9
- package/dist/campaign/index.js +7 -7
- package/dist/{campaign-B3kPMU8S.js → campaign-BzMSCejE.js} +8 -8
- package/dist/{campaign-B3kPMU8S.js.map → campaign-BzMSCejE.js.map} +1 -1
- package/dist/cli.js +1 -1
- package/dist/{client-DlqdbM7n.d.ts → client-vyYQg3bm.d.ts} +2 -2
- package/dist/{client-DlqdbM7n.d.ts.map → client-vyYQg3bm.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +9 -10
- package/dist/contract/index.js +8 -8
- package/dist/{default-registry-B0bKikCb.js → default-registry-CrAp0pYq.js} +4 -4
- package/dist/{default-registry-B0bKikCb.js.map → default-registry-CrAp0pYq.js.map} +1 -1
- package/dist/{default-registry-BKwc8bN5.d.ts → default-registry-FfNzaUHV.d.ts} +3 -3
- package/dist/{default-registry-BKwc8bN5.d.ts.map → default-registry-FfNzaUHV.d.ts.map} +1 -1
- package/dist/{define-agent-eval-CY6qdlGV.d.ts → define-agent-eval-V1jQyCDR.d.ts} +102 -11
- package/dist/define-agent-eval-V1jQyCDR.d.ts.map +1 -0
- package/dist/{define-agent-eval-8h3lXXee.js → define-agent-eval-ox5McL6e.js} +331 -144
- package/dist/define-agent-eval-ox5McL6e.js.map +1 -0
- package/dist/{dspy-rlm-engine-CF0t2ITD.js → dspy-rlm-engine-Caz2pl4L.js} +3 -3
- package/dist/{dspy-rlm-engine-CF0t2ITD.js.map → dspy-rlm-engine-Caz2pl4L.js.map} +1 -1
- package/dist/{engine-DhFir3Ys.d.ts → engine-CvW_I72-.d.ts} +2 -2
- package/dist/{engine-DhFir3Ys.d.ts.map → engine-CvW_I72-.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +1 -4
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/{external-optimizer-process-BwITA9Jp.js → external-optimizer-process-CxnFL1hd.js} +2 -2
- package/dist/{external-optimizer-process-BwITA9Jp.js.map → external-optimizer-process-CxnFL1hd.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-wBWeoG6A.js → external-optimizer-subprocess-CQi27uEI.js} +2 -2
- package/dist/{external-optimizer-subprocess-wBWeoG6A.js.map → external-optimizer-subprocess-CQi27uEI.js.map} +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/fuzz.js.map +1 -1
- package/dist/hosted/index.d.ts +1 -1
- package/dist/{index-BQqOjerE.d.ts → index-BTrx5s8m.d.ts} +8 -9
- package/dist/index-BTrx5s8m.d.ts.map +1 -0
- package/dist/{index-D0Db5X-4.d.ts → index-Bn-nlnSV.d.ts} +4 -4
- package/dist/{index-D0Db5X-4.d.ts.map → index-Bn-nlnSV.d.ts.map} +1 -1
- package/dist/index-DKXuBPXf.d.ts +3840 -0
- package/dist/index-DKXuBPXf.d.ts.map +1 -0
- package/dist/index.d.ts +11 -13
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +10 -10
- package/dist/{kind-factory-gP6lDySe.js → kind-factory-BLvL-E44.js} +2 -2
- package/dist/{kind-factory-gP6lDySe.js.map → kind-factory-BLvL-E44.js.map} +1 -1
- package/dist/{llm-judge-BfqMFo4h.js → llm-judge-DmNaBrXB.js} +2541 -2435
- package/dist/llm-judge-DmNaBrXB.js.map +1 -0
- package/dist/{matrix-DGu8KhSs.d.ts → matrix-CJtXz1ky.d.ts} +2 -2
- package/dist/{matrix-DGu8KhSs.d.ts.map → matrix-CJtXz1ky.d.ts.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{produced-state-D91uDvQw.js → produced-state-B8mw6zj9.js} +2 -2
- package/dist/{produced-state-D91uDvQw.js.map → produced-state-B8mw6zj9.js.map} +1 -1
- package/dist/rl.d.ts +1 -1
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js.map +1 -1
- package/dist/{semantic-concept-judge-Dok7_35a.js → semantic-concept-judge-E3s_fEjB.js} +3 -3
- package/dist/{semantic-concept-judge-Dok7_35a.js.map → semantic-concept-judge-E3s_fEjB.js.map} +1 -1
- package/dist/{skillopt-optimization-method-DDw3v3gA.js → skillopt-optimization-method-f7399oGb.js} +5 -5
- package/dist/{skillopt-optimization-method-DDw3v3gA.js.map → skillopt-optimization-method-f7399oGb.js.map} +1 -1
- package/dist/statistical-heldout-Cqb73yE9.d.ts +1127 -0
- package/dist/statistical-heldout-Cqb73yE9.d.ts.map +1 -0
- package/dist/{store-otlp-Dow0pk_5.js → store-otlp-DV_H2HDu.js} +2 -2
- package/dist/{store-otlp-Dow0pk_5.js.map → store-otlp-DV_H2HDu.js.map} +1 -1
- package/dist/{store-tool-spans-CCZNsihA.d.ts → store-tool-spans-4o55ABER.d.ts} +3 -3
- package/dist/{store-tool-spans-CCZNsihA.d.ts.map → store-tool-spans-4o55ABER.d.ts.map} +1 -1
- package/dist/{store-tool-spans-CeNj_m2L.js → store-tool-spans-B9tjys_h.js} +3 -3
- package/dist/{store-tool-spans-CeNj_m2L.js.map → store-tool-spans-B9tjys_h.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts.map +1 -1
- package/dist/supervisor-run/index.js +25 -7
- package/dist/supervisor-run/index.js.map +1 -1
- package/dist/{task-failure-attributes-CZjZeBsY.js → task-failure-attributes-CUy9mkIY.js} +2 -2
- package/dist/{task-failure-attributes-CZjZeBsY.js.map → task-failure-attributes-CUy9mkIY.js.map} +1 -1
- package/dist/{tool-groups-Cp4Xdzrp.d.ts → tool-groups-DAe1t6zb.d.ts} +2 -2
- package/dist/tool-groups-DAe1t6zb.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +1 -1
- package/dist/traces.d.ts +2 -2
- package/dist/traces.js +4 -4
- package/dist/{types-Ba5UQyVD.d.ts → types-BJz2CPTM.d.ts} +2 -2
- package/dist/{types-Ba5UQyVD.d.ts.map → types-BJz2CPTM.d.ts.map} +1 -1
- package/dist/{types-CiWITkGo.js → types-DQ0e2E7y.js} +2 -2
- package/dist/types-DQ0e2E7y.js.map +1 -0
- package/dist/{types-BDV4PiMR.d.ts → types-Dd1ejaeI.d.ts} +2 -2
- package/dist/{types-BDV4PiMR.d.ts.map → types-Dd1ejaeI.d.ts.map} +1 -1
- package/docs/campaign-proposers.md +42 -0
- package/package.json +1 -1
- package/dist/agent-profile-B9_GGsG8.d.ts +0 -84
- package/dist/agent-profile-B9_GGsG8.d.ts.map +0 -1
- package/dist/backend-integrity-CeuTgqsd.d.ts +0 -280
- package/dist/backend-integrity-CeuTgqsd.d.ts.map +0 -1
- package/dist/benchmark-BjLGkfnN.d.ts +0 -236
- package/dist/benchmark-BjLGkfnN.d.ts.map +0 -1
- package/dist/define-agent-eval-8h3lXXee.js.map +0 -1
- package/dist/define-agent-eval-CY6qdlGV.d.ts.map +0 -1
- package/dist/external-optimizer-contracts-CQCpyrIL.d.ts +0 -172
- package/dist/external-optimizer-contracts-CQCpyrIL.d.ts.map +0 -1
- package/dist/heldout-gate-Df5hsqmm.d.ts +0 -453
- package/dist/heldout-gate-Df5hsqmm.d.ts.map +0 -1
- package/dist/index-BQqOjerE.d.ts.map +0 -1
- package/dist/index-CFDffsKz.d.ts +0 -1135
- package/dist/index-CFDffsKz.d.ts.map +0 -1
- package/dist/llm-judge-BfqMFo4h.js.map +0 -1
- package/dist/power-preflight-Ptse_Kq7.d.ts +0 -117
- package/dist/power-preflight-Ptse_Kq7.d.ts.map +0 -1
- package/dist/pre-registration-BoI4ucR3.d.ts +0 -592
- package/dist/pre-registration-BoI4ucR3.d.ts.map +0 -1
- package/dist/promotion-policy-CvMda3kU.d.ts +0 -134
- package/dist/promotion-policy-CvMda3kU.d.ts.map +0 -1
- package/dist/proposal-findings-bko3GGy-.js.map +0 -1
- package/dist/provenance-CRY67X50.d.ts +0 -1995
- package/dist/provenance-CRY67X50.d.ts.map +0 -1
- package/dist/statistical-heldout-DTyB_6-1.d.ts +0 -295
- package/dist/statistical-heldout-DTyB_6-1.d.ts.map +0 -1
- package/dist/tool-groups-Cp4Xdzrp.d.ts.map +0 -1
- package/dist/types-CiWITkGo.js.map +0 -1
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,46 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
|
|
|
4
4
|
|
|
5
5
|
---
|
|
6
6
|
|
|
7
|
+
## [0.174.0] — 2026-09-05
|
|
8
|
+
|
|
9
|
+
### Changed
|
|
10
|
+
|
|
11
|
+
- `selfImprove({ method })` executes the method directly and measures its selected surface on final cases.
|
|
12
|
+
It returns `SelfImproveMethodResult` with `mode: 'method'`, actual `raw.method` evidence, and `tangle.method-improvement` provenance.
|
|
13
|
+
It does not expose native `raw.generations` or `generationsExplored`.
|
|
14
|
+
- `selfImprove({ proposer })` returns `SelfImproveProposerResult` with `mode: 'proposer'` and the existing native history.
|
|
15
|
+
`SelfImproveResult` is the union; consumers must narrow by `mode` before reading fields specific to one mode.
|
|
16
|
+
- Method results with deferred holdout return `baseline: null`, `winner.compositeMean: null`, and no lift.
|
|
17
|
+
Method `cost` includes reported search and final spending; `ledgerCost` retains the actual receipt breakdown.
|
|
18
|
+
- Premeasured native baselines must match the evaluator manifest.
|
|
19
|
+
Use `surfaceDispatchRef(baselineSurface, dispatchRef)` when creating the baseline campaign.
|
|
20
|
+
See [the result and cache contracts](docs/campaign-proposers.md#read-an-improvement-result).
|
|
21
|
+
|
|
22
|
+
### Fixed
|
|
23
|
+
|
|
24
|
+
- Complete methods can select the unchanged baseline without triggering native duplicate-candidate rejection.
|
|
25
|
+
- Native ranking cannot override a complete method's selected winner or repeat its train and selection evaluations.
|
|
26
|
+
- Method-reported spend and incomplete accounting remain in the total without duplicating metered costs.
|
|
27
|
+
Both improvement and method comparison reconcile each method's report against its attributed ledger receipts.
|
|
28
|
+
- Final method comparisons reject missing replicas before averaging surviving scores.
|
|
29
|
+
- Native search and final caches bind the candidate's surface content to execution identity.
|
|
30
|
+
- Premeasured baselines from a different judge revision are refused before candidate execution.
|
|
31
|
+
- Native candidate history reports `ci95: null` when uncertainty was not estimated.
|
|
32
|
+
- An unchanged baseline's shared campaign contributes once to result analysis, including execution count, tokens, and cost.
|
|
33
|
+
- `BehaviorExplorer` uses observed scores when assigning the next round's evaluations by behavior cell.
|
|
34
|
+
Scenario records retain their individual identities; allocation uses the pooled cell identity.
|
|
35
|
+
|
|
36
|
+
---
|
|
37
|
+
|
|
38
|
+
## [0.173.3] — 2026-09-04
|
|
39
|
+
|
|
40
|
+
### Fixed
|
|
41
|
+
|
|
42
|
+
- The Runtime supervisor reader now treats `spawned.ownedTreeRoot` as the exclusive nested tree when present.
|
|
43
|
+
It rejects an additional child-id tree and any descendant spawn outside its parent's owned tree.
|
|
44
|
+
|
|
45
|
+
---
|
|
46
|
+
|
|
7
47
|
## [0.173.2] — 2026-09-04
|
|
8
48
|
|
|
9
49
|
### Fixed
|
|
@@ -1,13 +1,4 @@
|
|
|
1
1
|
import { z } from "zod";
|
|
2
|
-
//#region src/abort-signal.ts
|
|
3
|
-
/** Combine active cancellation sources without wrapping a single source. */
|
|
4
|
-
function combineAbortSignals(...signals) {
|
|
5
|
-
const active = [...new Set(signals.filter((signal) => signal !== void 0))];
|
|
6
|
-
if (active.length === 0) return void 0;
|
|
7
|
-
if (active.length === 1) return active[0];
|
|
8
|
-
return AbortSignal.any(active);
|
|
9
|
-
}
|
|
10
|
-
//#endregion
|
|
11
2
|
//#region src/analyst/proposal-findings.ts
|
|
12
3
|
const ProposalFindingSchema = z.object({
|
|
13
4
|
schema_version: z.literal("1.0.0"),
|
|
@@ -64,6 +55,15 @@ function findingLabel(finding, index) {
|
|
|
64
55
|
return typeof id === "string" && id.length > 0 ? id : `index ${index}`;
|
|
65
56
|
}
|
|
66
57
|
//#endregion
|
|
67
|
-
|
|
58
|
+
//#region src/abort-signal.ts
|
|
59
|
+
/** Combine active cancellation sources without wrapping a single source. */
|
|
60
|
+
function combineAbortSignals(...signals) {
|
|
61
|
+
const active = [...new Set(signals.filter((signal) => signal !== void 0))];
|
|
62
|
+
if (active.length === 0) return void 0;
|
|
63
|
+
if (active.length === 1) return active[0];
|
|
64
|
+
return AbortSignal.any(active);
|
|
65
|
+
}
|
|
66
|
+
//#endregion
|
|
67
|
+
export { assertProposalFindings as n, isProposalFinding as r, combineAbortSignals as t };
|
|
68
68
|
|
|
69
|
-
//# sourceMappingURL=
|
|
69
|
+
//# sourceMappingURL=abort-signal-CtzAM_sJ.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"abort-signal-CtzAM_sJ.js","names":[],"sources":["../src/analyst/proposal-findings.ts","../src/abort-signal.ts"],"sourcesContent":["import { z } from 'zod'\nimport type { ProposalFinding } from './types'\n\nconst ProposalFindingSchema = z\n .object({\n schema_version: z.literal('1.0.0'),\n finding_id: z.string().min(1),\n analyst_id: z.string().min(1),\n produced_at: z.string().min(1),\n severity: z.enum(['critical', 'high', 'medium', 'low', 'info']),\n area: z.string().min(1),\n claim: z.string().min(1),\n rationale: z.string().optional(),\n evidence_refs: z.array(\n z\n .object({\n kind: z.enum(['span', 'event', 'artifact', 'finding', 'metric']),\n uri: z.string().min(1),\n excerpt: z.string().optional(),\n })\n .strict(),\n ),\n recommended_action: z.string().optional(),\n validation_plan: z.string().optional(),\n confidence: z.number().min(0).max(1),\n subject: z.string().optional(),\n derived_from_judge: z.boolean().optional(),\n metadata: z.record(z.string(), z.unknown()).optional(),\n proposal_origin: z.enum(['search', 'production']),\n })\n .strict() satisfies z.ZodType<ProposalFinding>\n\n/** True when a finding names a source candidate generation may learn from. */\nexport function isProposalFinding(finding: unknown): finding is ProposalFinding {\n return ProposalFindingSchema.safeParse(finding).success\n}\n\n/**\n * Reject findings whose source has not been explicitly admitted for candidate\n * generation. Search feedback and observed production behavior are allowed;\n * final evaluation data has no allowed origin.\n */\nexport function assertProposalFindings(\n findings: unknown,\n context = 'proposal findings',\n): ReadonlyArray<ProposalFinding> {\n if (!Array.isArray(findings)) {\n throw new TypeError(`${context}: expected an array`)\n }\n const rejected = findings.flatMap((finding, index) =>\n isProposalFinding(finding) ? [] : [findingLabel(finding, index)],\n )\n if (rejected.length > 0) {\n throw new Error(\n `${context}: every finding must match AnalystFinding and declare ` +\n `proposal_origin as search or production. ` +\n `Rejected findings: [${rejected.join(', ')}].`,\n )\n }\n return findings as ReadonlyArray<ProposalFinding>\n}\n\nfunction findingLabel(finding: unknown, index: number): string {\n if (typeof finding !== 'object' || finding === null) return `index ${index}`\n const id = (finding as { finding_id?: unknown }).finding_id\n return typeof id === 'string' && id.length > 0 ? id : `index ${index}`\n}\n","/** Combine active cancellation sources without wrapping a single source. */\nexport function combineAbortSignals(\n ...signals: Array<AbortSignal | undefined>\n): AbortSignal | undefined {\n const active = [\n ...new Set(signals.filter((signal): signal is AbortSignal => signal !== undefined)),\n ]\n if (active.length === 0) return undefined\n if (active.length === 1) return active[0]\n return AbortSignal.any(active)\n}\n"],"mappings":";;AAGA,MAAM,wBAAwB,EAC3B,OAAO;CACN,gBAAgB,EAAE,QAAQ,OAAO;CACjC,YAAY,EAAE,OAAO,CAAC,CAAC,IAAI,CAAC;CAC5B,YAAY,EAAE,OAAO,CAAC,CAAC,IAAI,CAAC;CAC5B,aAAa,EAAE,OAAO,CAAC,CAAC,IAAI,CAAC;CAC7B,UAAU,EAAE,KAAK;EAAC;EAAY;EAAQ;EAAU;EAAO;CAAM,CAAC;CAC9D,MAAM,EAAE,OAAO,CAAC,CAAC,IAAI,CAAC;CACtB,OAAO,EAAE,OAAO,CAAC,CAAC,IAAI,CAAC;CACvB,WAAW,EAAE,OAAO,CAAC,CAAC,SAAS;CAC/B,eAAe,EAAE,MACf,EACG,OAAO;EACN,MAAM,EAAE,KAAK;GAAC;GAAQ;GAAS;GAAY;GAAW;EAAQ,CAAC;EAC/D,KAAK,EAAE,OAAO,CAAC,CAAC,IAAI,CAAC;EACrB,SAAS,EAAE,OAAO,CAAC,CAAC,SAAS;CAC/B,CAAC,CAAC,CACD,OAAO,CACZ;CACA,oBAAoB,EAAE,OAAO,CAAC,CAAC,SAAS;CACxC,iBAAiB,EAAE,OAAO,CAAC,CAAC,SAAS;CACrC,YAAY,EAAE,OAAO,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,IAAI,CAAC;CACnC,SAAS,EAAE,OAAO,CAAC,CAAC,SAAS;CAC7B,oBAAoB,EAAE,QAAQ,CAAC,CAAC,SAAS;CACzC,UAAU,EAAE,OAAO,EAAE,OAAO,GAAG,EAAE,QAAQ,CAAC,CAAC,CAAC,SAAS;CACrD,iBAAiB,EAAE,KAAK,CAAC,UAAU,YAAY,CAAC;AAClD,CAAC,CAAC,CACD,OAAO;;AAGV,SAAgB,kBAAkB,SAA8C;CAC9E,OAAO,sBAAsB,UAAU,OAAO,CAAC,CAAC;AAClD;;;;;;AAOA,SAAgB,uBACd,UACA,UAAU,qBACsB;CAChC,IAAI,CAAC,MAAM,QAAQ,QAAQ,GACzB,MAAM,IAAI,UAAU,GAAG,QAAQ,oBAAoB;CAErD,MAAM,WAAW,SAAS,SAAS,SAAS,UAC1C,kBAAkB,OAAO,IAAI,CAAC,IAAI,CAAC,aAAa,SAAS,KAAK,CAAC,CACjE;CACA,IAAI,SAAS,SAAS,GACpB,MAAM,IAAI,MACR,GAAG,QAAQ,qHAEc,SAAS,KAAK,IAAI,EAAE,GAC/C;CAEF,OAAO;AACT;AAEA,SAAS,aAAa,SAAkB,OAAuB;CAC7D,IAAI,OAAO,YAAY,YAAY,YAAY,MAAM,OAAO,SAAS;CACrE,MAAM,KAAM,QAAqC;CACjD,OAAO,OAAO,OAAO,YAAY,GAAG,SAAS,IAAI,KAAK,SAAS;AACjE;;;;ACjEA,SAAgB,oBACd,GAAG,SACsB;CACzB,MAAM,SAAS,CACb,GAAG,IAAI,IAAI,QAAQ,QAAQ,WAAkC,WAAW,KAAA,CAAS,CAAC,CACpF;CACA,IAAI,OAAO,WAAW,GAAG,OAAO,KAAA;CAChC,IAAI,OAAO,WAAW,GAAG,OAAO,OAAO;CACvC,OAAO,YAAY,IAAI,MAAM;AAC/B"}
|
package/dist/adapters/http.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { R as Scenario, d as DispatchContext, f as DispatchFn } from "../types-
|
|
2
|
-
import "../index-
|
|
1
|
+
import { R as Scenario, d as DispatchContext, f as DispatchFn } from "../types-BJz2CPTM.js";
|
|
2
|
+
import "../index-BTrx5s8m.js";
|
|
3
3
|
//#region src/adapters/http.d.ts
|
|
4
4
|
interface HttpDispatchOptions<TScenario extends Scenario, _TArtifact> {
|
|
5
5
|
/** Static endpoint URL. Mutually exclusive with `resolveUrl`. */
|
|
@@ -0,0 +1,488 @@
|
|
|
1
|
+
import { b as CustomTokenPricing, g as CostReceiptInput } from "./cost-ledger-DbQdN3nO.js";
|
|
2
|
+
import { c as AnalystRunInputs, f as AnalystUsageReceipt, i as AnalystFinding, p as EvidenceRef, w as TraceAnalysisStore } from "./types-DN2WdT5S.js";
|
|
3
|
+
import { h as ChatResponse, m as ChatRequest } from "./types-gvRsyJLh.js";
|
|
4
|
+
import { c as RegistryRunOpts, n as AnalystRegistry } from "./registry-7pOUBrtX.js";
|
|
5
|
+
import { AgentProfile, AgentProfile as AgentProfile$1, HarnessType, HarnessType as HarnessType$1 } from "@tangle-network/agent-interface";
|
|
6
|
+
//#region src/campaign/external-optimizer-contracts.d.ts
|
|
7
|
+
interface ExternalOptimizerProcessLimits {
|
|
8
|
+
/** Maximum serialized input written for the child process. */
|
|
9
|
+
maxInputBytes: number;
|
|
10
|
+
/** Maximum JSON result read from the child process. */
|
|
11
|
+
maxResultBytes: number;
|
|
12
|
+
/** Maximum stdout or stderr characters retained for diagnostics. */
|
|
13
|
+
maxOutputChars: number;
|
|
14
|
+
}
|
|
15
|
+
declare const DEFAULT_EXTERNAL_OPTIMIZER_PROCESS_LIMITS: Readonly<ExternalOptimizerProcessLimits>;
|
|
16
|
+
interface ExternalOptimizerCallbackLimits {
|
|
17
|
+
/** Maximum serialized request accepted by the loopback callback. */
|
|
18
|
+
maxRequestBytes: number;
|
|
19
|
+
/** Maximum serialized response returned by the loopback callback. */
|
|
20
|
+
maxResponseBytes: number;
|
|
21
|
+
}
|
|
22
|
+
declare const DEFAULT_EXTERNAL_OPTIMIZER_CALLBACK_LIMITS: Readonly<ExternalOptimizerCallbackLimits>;
|
|
23
|
+
interface ExternalOptimizerRunnerCommand {
|
|
24
|
+
command?: string;
|
|
25
|
+
args?: readonly string[];
|
|
26
|
+
env?: NodeJS.ProcessEnv;
|
|
27
|
+
/** Child-process resource limits. Omitted fields use finite defaults. */
|
|
28
|
+
limits?: Partial<ExternalOptimizerProcessLimits>;
|
|
29
|
+
}
|
|
30
|
+
type ExternalOptimizerResumeMode = 'never' | 'if-compatible' | 'required';
|
|
31
|
+
type ExternalTextCandidate = string | Record<string, string>;
|
|
32
|
+
interface ExternalTextEvaluationRequest {
|
|
33
|
+
candidate: ExternalTextCandidate;
|
|
34
|
+
exampleId: string;
|
|
35
|
+
}
|
|
36
|
+
declare function resolveExternalOptimizerProcessLimits(value: Partial<ExternalOptimizerProcessLimits> | undefined, label?: string): ExternalOptimizerProcessLimits;
|
|
37
|
+
declare function resolveExternalOptimizerCallbackLimits(value: Partial<ExternalOptimizerCallbackLimits> | undefined, label?: string): ExternalOptimizerCallbackLimits;
|
|
38
|
+
type DeepReadonly<T> = T extends ((...args: never[]) => unknown) ? T : T extends readonly (infer U)[] ? readonly DeepReadonly<U>[] : T extends object ? { readonly [K in keyof T]: DeepReadonly<T[K]>; } : T;
|
|
39
|
+
/** Provider-neutral request parsed once from the optimizer's loopback protocol. */
|
|
40
|
+
type ExternalOptimizerChatRequest = DeepReadonly<Omit<ChatRequest, 'model'> & {
|
|
41
|
+
model: string;
|
|
42
|
+
}>;
|
|
43
|
+
type ExternalOptimizerEndpointFormat = 'chat-completions' | 'responses' | 'anthropic-messages';
|
|
44
|
+
/** One exact model request admitted by the loopback proxy. */
|
|
45
|
+
interface ExternalOptimizerModelCallRequest {
|
|
46
|
+
/** Stable identity allocated by the cost ledger for this paid call. */
|
|
47
|
+
readonly callId: string;
|
|
48
|
+
/** Deeply immutable canonical request; HTTP protocol fields never cross this boundary. */
|
|
49
|
+
readonly request: ExternalOptimizerChatRequest;
|
|
50
|
+
/** Child response shape, when the execution owner needs to retain it as evidence. */
|
|
51
|
+
readonly endpointFormat?: ExternalOptimizerEndpointFormat;
|
|
52
|
+
readonly signal: AbortSignal;
|
|
53
|
+
}
|
|
54
|
+
/** Runtime-owned result for one admitted optimizer-model call. */
|
|
55
|
+
type ExternalOptimizerModelCallResult = {
|
|
56
|
+
readonly succeeded: true;
|
|
57
|
+
/** Canonical response encoded back into the child's protocol by Agent Eval. */
|
|
58
|
+
readonly response: ChatResponse;
|
|
59
|
+
/**
|
|
60
|
+
* Canonical measured usage/cost input retained by Agent Eval's cost ledger.
|
|
61
|
+
* `inputTokens` is the non-cached portion, `cachedTokens` is the cache-read
|
|
62
|
+
* portion, and response `promptTokens` equals their sum. Cache-write tokens
|
|
63
|
+
* remain a separately billed class and are never silently added to that total.
|
|
64
|
+
*/
|
|
65
|
+
readonly receipt: CostReceiptInput;
|
|
66
|
+
/** Opaque, finite JSON proof of the exact execution retained in provenance. */
|
|
67
|
+
readonly execution: unknown;
|
|
68
|
+
} | {
|
|
69
|
+
readonly succeeded: false;
|
|
70
|
+
/** Public failure text safe to retain and return to the child process. */
|
|
71
|
+
readonly error: string;
|
|
72
|
+
/** Usage/cost state for the failed Runtime call. Unknown values stay unknown. */
|
|
73
|
+
readonly receipt: CostReceiptInput;
|
|
74
|
+
/** Opaque, finite JSON proof of the failed exact execution. */
|
|
75
|
+
readonly execution: unknown;
|
|
76
|
+
};
|
|
77
|
+
/**
|
|
78
|
+
* Execution-neutral model-call seam for an external optimizer.
|
|
79
|
+
*
|
|
80
|
+
* The package that owns execution implements this with its exact execution
|
|
81
|
+
* path. For Discovery that owner is Runtime and the identity is an
|
|
82
|
+
* AgentProfile. The loopback proxy owns request validation, limits, response
|
|
83
|
+
* bounds, and cost-ledger recording. Once invoked, the callback must resolve
|
|
84
|
+
* with one success/failure result. Rejecting loses the execution record and
|
|
85
|
+
* therefore fails the optimizer attempt.
|
|
86
|
+
*/
|
|
87
|
+
type ExternalOptimizerModelCall = (request: ExternalOptimizerModelCallRequest) => Promise<ExternalOptimizerModelCallResult>;
|
|
88
|
+
/** One opaque Runtime execution record retained for one admitted model call. */
|
|
89
|
+
type ExternalOptimizerModelExecutionObservation = {
|
|
90
|
+
readonly sequence: number;
|
|
91
|
+
readonly callId: string;
|
|
92
|
+
readonly callRef: string;
|
|
93
|
+
readonly path: '/v1/chat/completions' | '/v1/responses' | '/v1/messages';
|
|
94
|
+
readonly model: string;
|
|
95
|
+
readonly succeeded: true;
|
|
96
|
+
readonly responseStatus: number;
|
|
97
|
+
readonly execution: unknown;
|
|
98
|
+
} | {
|
|
99
|
+
readonly sequence: number;
|
|
100
|
+
readonly callId: string;
|
|
101
|
+
readonly callRef: string;
|
|
102
|
+
readonly path: '/v1/chat/completions' | '/v1/responses' | '/v1/messages';
|
|
103
|
+
readonly model: string;
|
|
104
|
+
readonly succeeded: false;
|
|
105
|
+
readonly error: string;
|
|
106
|
+
readonly execution: unknown;
|
|
107
|
+
};
|
|
108
|
+
type ExternalOptimizerEvaluationRefusalReason = 'invalid-request' | 'evaluation-limit' | 'evaluation-failed';
|
|
109
|
+
/**
|
|
110
|
+
* Durable callback-side record of every candidate submitted for scoring,
|
|
111
|
+
* scored task, and callback refusal. Optimizer-internal proposals that never
|
|
112
|
+
* reach this callback are outside this record's scope.
|
|
113
|
+
*/
|
|
114
|
+
type ExternalOptimizerEvaluationObservation = {
|
|
115
|
+
readonly kind: 'proposal';
|
|
116
|
+
readonly sequence: number;
|
|
117
|
+
readonly candidate: ExternalTextCandidate;
|
|
118
|
+
readonly candidateHash: string;
|
|
119
|
+
} | {
|
|
120
|
+
readonly kind: 'evaluation';
|
|
121
|
+
readonly sequence: number;
|
|
122
|
+
readonly candidate: ExternalTextCandidate;
|
|
123
|
+
readonly candidateHash: string;
|
|
124
|
+
readonly exampleId: string;
|
|
125
|
+
/** One-based accepted evaluation number for this optimizer attempt. */
|
|
126
|
+
readonly evaluationNumber: number;
|
|
127
|
+
readonly response: unknown;
|
|
128
|
+
} | {
|
|
129
|
+
readonly kind: 'refusal';
|
|
130
|
+
readonly sequence: number;
|
|
131
|
+
readonly reason: ExternalOptimizerEvaluationRefusalReason;
|
|
132
|
+
readonly candidate?: ExternalTextCandidate;
|
|
133
|
+
readonly candidateHash?: string;
|
|
134
|
+
readonly exampleId?: string;
|
|
135
|
+
};
|
|
136
|
+
interface ExternalOptimizerModelBudget {
|
|
137
|
+
/** Optional optimizer-model spend ceiling, independent of task-evaluation spend. */
|
|
138
|
+
maxCostUsd?: number;
|
|
139
|
+
/** Maximum calls into the execution owner; owner-internal retries are reported there. */
|
|
140
|
+
maxRequests: number;
|
|
141
|
+
/** Reject a request body above this byte count. */
|
|
142
|
+
maxRequestBytes: number;
|
|
143
|
+
/** Reject a provider response above this byte count. */
|
|
144
|
+
maxResponseBytes: number;
|
|
145
|
+
/** Reject a request asking the provider for more output tokens. */
|
|
146
|
+
maxOutputTokensPerRequest: number;
|
|
147
|
+
/**
|
|
148
|
+
* Reasoning tokens a single response may bill beyond its completion limit.
|
|
149
|
+
*
|
|
150
|
+
* A reasoning model bounds only the completion by `max_tokens` and bills
|
|
151
|
+
* thinking on top, so a reservation sized to the completion alone is always
|
|
152
|
+
* too small and the ledger refuses the real charge. Callers that route a
|
|
153
|
+
* reasoning model declare its thinking budget here; the reservation covers
|
|
154
|
+
* it and a response exceeding it still fails loudly. Default: 0.
|
|
155
|
+
*/
|
|
156
|
+
maxReasoningTokensPerRequest?: number;
|
|
157
|
+
/**
|
|
158
|
+
* Optional rates used to estimate cost when the provider omits a valid
|
|
159
|
+
* `usage.cost`. Omit when billed USD is unknown; catalog estimates are not
|
|
160
|
+
* enforcement evidence.
|
|
161
|
+
*/
|
|
162
|
+
pricing?: CustomTokenPricing;
|
|
163
|
+
/** Per-provider-request deadline. Default: 300,000 ms. */
|
|
164
|
+
requestTimeoutMs?: number;
|
|
165
|
+
}
|
|
166
|
+
/** Per-wire request accounting for the loopback proxy. */
|
|
167
|
+
interface ExternalOptimizerWireCounts {
|
|
168
|
+
/** Execution-owner calls admitted on this wire, including failures. */
|
|
169
|
+
requestAttempts: number;
|
|
170
|
+
/** Successful 2xx responses recorded on this wire. */
|
|
171
|
+
successfulCompletions: number;
|
|
172
|
+
}
|
|
173
|
+
//#endregion
|
|
174
|
+
//#region src/analyst/benchmark-scoring.d.ts
|
|
175
|
+
declare function scoreAnalystFindings(testCase: Pick<AnalystBenchmarkCase, 'id' | 'expectedIssues' | 'labeledEvidence'>, findings: readonly AnalystFinding[]): AnalystFindingScore;
|
|
176
|
+
//#endregion
|
|
177
|
+
//#region src/analyst/benchmark.d.ts
|
|
178
|
+
interface AnalystEvidenceExpectation {
|
|
179
|
+
uri: string;
|
|
180
|
+
kind?: EvidenceRef['kind'];
|
|
181
|
+
}
|
|
182
|
+
interface AnalystIssueExpectation {
|
|
183
|
+
id: string;
|
|
184
|
+
findingIds?: readonly string[];
|
|
185
|
+
areas?: readonly string[];
|
|
186
|
+
subjects?: readonly string[];
|
|
187
|
+
evidence?: readonly AnalystEvidenceExpectation[];
|
|
188
|
+
evidenceMode?: 'any' | 'all';
|
|
189
|
+
/** Exact evidence location for the first unrecoverable or causal step. */
|
|
190
|
+
criticalEvidence?: readonly AnalystEvidenceExpectation[];
|
|
191
|
+
}
|
|
192
|
+
type AnalystBenchmarkLabelState = 'positive' | 'trusted-negative' | 'unlabeled';
|
|
193
|
+
interface AnalystBenchmarkCase<TInput = unknown> {
|
|
194
|
+
id: string;
|
|
195
|
+
/** Independent source unit used for resampling, such as a task or incident. */
|
|
196
|
+
clusterId: string;
|
|
197
|
+
/** Whether labels prove an issue, prove no issue, or leave the outcome unknown. */
|
|
198
|
+
labelState: AnalystBenchmarkLabelState;
|
|
199
|
+
input: TInput;
|
|
200
|
+
expectedIssues: readonly AnalystIssueExpectation[];
|
|
201
|
+
/** Complete set of labeled locations used to measure label-location agreement. */
|
|
202
|
+
labeledEvidence?: readonly AnalystEvidenceExpectation[];
|
|
203
|
+
tags?: readonly string[];
|
|
204
|
+
metadata?: Record<string, unknown>;
|
|
205
|
+
}
|
|
206
|
+
interface AnalystFindingScore {
|
|
207
|
+
expectedIssueCount: number;
|
|
208
|
+
matchedIssueIds: string[];
|
|
209
|
+
missedIssueIds: string[];
|
|
210
|
+
supportedFindingIndexes: number[];
|
|
211
|
+
unsupportedFindingIndexes: number[];
|
|
212
|
+
unlabeledEvidence: EvidenceRef[];
|
|
213
|
+
issueRecall: number;
|
|
214
|
+
findingPrecision: number;
|
|
215
|
+
f1: number;
|
|
216
|
+
criticalStepAccuracy: number | null;
|
|
217
|
+
/** Share of findings that cite at least one evidence location. */
|
|
218
|
+
citationCoverage: number | null;
|
|
219
|
+
/** Share of citations that include a non-empty source excerpt. */
|
|
220
|
+
citationExcerptCoverage: number | null;
|
|
221
|
+
/** Share of citations that agree with a labeled case location. */
|
|
222
|
+
citationLabelAgreement: number | null;
|
|
223
|
+
predictionOnLabelEmptyCase: boolean;
|
|
224
|
+
}
|
|
225
|
+
interface AnalystEvidenceResolutionError {
|
|
226
|
+
evidence: EvidenceRef;
|
|
227
|
+
class: string;
|
|
228
|
+
message: string;
|
|
229
|
+
}
|
|
230
|
+
interface AnalystEvidenceResolution {
|
|
231
|
+
checked: number;
|
|
232
|
+
resolved: number;
|
|
233
|
+
unresolvedEvidence: EvidenceRef[];
|
|
234
|
+
errors: AnalystEvidenceResolutionError[];
|
|
235
|
+
/** Null when no citations were checked or any resolution attempt failed. */
|
|
236
|
+
validity: number | null;
|
|
237
|
+
}
|
|
238
|
+
type AnalystEvidenceResolver<TInput = unknown> = (input: {
|
|
239
|
+
caseId: string;
|
|
240
|
+
caseInput: TInput;
|
|
241
|
+
evidence: EvidenceRef;
|
|
242
|
+
signal?: AbortSignal;
|
|
243
|
+
}) => boolean | Promise<boolean>;
|
|
244
|
+
/**
|
|
245
|
+
* Resolve canonical `trace://<trace>/span/<span>` evidence against a trace store.
|
|
246
|
+
* Other evidence kinds and URI schemes require a caller-supplied resolver.
|
|
247
|
+
*/
|
|
248
|
+
declare function traceStoreEvidenceResolver<TInput>(getStore: (input: TInput) => TraceAnalysisStore): AnalystEvidenceResolver<TInput>;
|
|
249
|
+
interface AnalystBenchmarkOutput {
|
|
250
|
+
findings: readonly AnalystFinding[];
|
|
251
|
+
usage?: AnalystUsageReceipt;
|
|
252
|
+
metadata?: Record<string, unknown>;
|
|
253
|
+
/**
|
|
254
|
+
* End-to-end duration measured by an external runner before import.
|
|
255
|
+
* Use null when the source explicitly did not capture duration.
|
|
256
|
+
*/
|
|
257
|
+
observedLatencyMs?: number | null;
|
|
258
|
+
/** Marks a completed transport as a failed analyst run while retaining usage and metadata. */
|
|
259
|
+
error?: AnalystBenchmarkError;
|
|
260
|
+
}
|
|
261
|
+
interface AnalystBenchmarkError {
|
|
262
|
+
class: string;
|
|
263
|
+
message: string;
|
|
264
|
+
code?: string;
|
|
265
|
+
status?: number;
|
|
266
|
+
}
|
|
267
|
+
interface AnalystBenchmarkRunner<TInput = unknown> {
|
|
268
|
+
id: string;
|
|
269
|
+
analyze(input: TInput, context: {
|
|
270
|
+
caseId: string;
|
|
271
|
+
repetition: number;
|
|
272
|
+
signal?: AbortSignal;
|
|
273
|
+
}): AnalystBenchmarkOutput | Promise<AnalystBenchmarkOutput>;
|
|
274
|
+
}
|
|
275
|
+
interface AnalystBenchmarkObservation {
|
|
276
|
+
runnerId: string;
|
|
277
|
+
caseId: string;
|
|
278
|
+
clusterId: string;
|
|
279
|
+
labelState: AnalystBenchmarkLabelState;
|
|
280
|
+
repetition: number;
|
|
281
|
+
executionIndex: number;
|
|
282
|
+
latencyMs: number | null;
|
|
283
|
+
latencySource: 'benchmark-clock' | 'runner-reported' | 'uncaptured';
|
|
284
|
+
findings: readonly AnalystFinding[];
|
|
285
|
+
score: AnalystFindingScore;
|
|
286
|
+
evidenceResolution?: AnalystEvidenceResolution;
|
|
287
|
+
caseTags: readonly string[];
|
|
288
|
+
caseMetadata?: Record<string, unknown>;
|
|
289
|
+
usage?: AnalystUsageReceipt;
|
|
290
|
+
runnerMetadata?: Record<string, unknown>;
|
|
291
|
+
error?: AnalystBenchmarkError;
|
|
292
|
+
}
|
|
293
|
+
interface AnalystLatencyDistribution {
|
|
294
|
+
min: number;
|
|
295
|
+
mean: number;
|
|
296
|
+
p50: number;
|
|
297
|
+
p95: number;
|
|
298
|
+
max: number;
|
|
299
|
+
}
|
|
300
|
+
interface AnalystBenchmarkSummary {
|
|
301
|
+
runnerId: string;
|
|
302
|
+
plannedRuns: number;
|
|
303
|
+
completedRuns: number;
|
|
304
|
+
failedRuns: number;
|
|
305
|
+
issueBearingRuns: number;
|
|
306
|
+
trustedNegativeRuns: number;
|
|
307
|
+
unlabeledRuns: number;
|
|
308
|
+
/** Pooled across all labeled issues and findings. */
|
|
309
|
+
issueRecall: number | null;
|
|
310
|
+
/** Pooled across all labeled issues and findings. */
|
|
311
|
+
findingPrecision: number | null;
|
|
312
|
+
/** Harmonic mean of the pooled precision and recall. */
|
|
313
|
+
f1: number | null;
|
|
314
|
+
/** Mean of per-case recall over issue-bearing runs. */
|
|
315
|
+
macroIssueRecall: number | null;
|
|
316
|
+
/** Mean of per-case precision over issue-bearing runs. */
|
|
317
|
+
macroFindingPrecision: number | null;
|
|
318
|
+
/** Mean of per-case F1 over issue-bearing runs. */
|
|
319
|
+
macroF1: number | null;
|
|
320
|
+
criticalStepAccuracy: number | null;
|
|
321
|
+
citationCoverage: number | null;
|
|
322
|
+
citationExcerptCoverage: number | null;
|
|
323
|
+
citationLabelAgreement: number | null;
|
|
324
|
+
citationResolution: number | null;
|
|
325
|
+
citationResolutionUnknownRuns: number;
|
|
326
|
+
unresolvedCitations: number;
|
|
327
|
+
citationResolutionErrors: number;
|
|
328
|
+
trustedNegativeFalsePositiveRate: number | null;
|
|
329
|
+
trustedNegativeFailureRate: number | null;
|
|
330
|
+
unlabeledPredictionRate: number | null;
|
|
331
|
+
unlabeledFailureRate: number | null;
|
|
332
|
+
/** Primary repeatability measure over complete finding identity and evidence. */
|
|
333
|
+
predictionAgreement: number | null;
|
|
334
|
+
/** Repeated cases contributing equally to predictionAgreement. */
|
|
335
|
+
predictionAgreementCases: number;
|
|
336
|
+
/** Secondary repeatability detail over matched expected labels. */
|
|
337
|
+
matchedLabelAgreement: number | null;
|
|
338
|
+
/** Positive repeated cases contributing equally to matchedLabelAgreement. */
|
|
339
|
+
matchedLabelAgreementCases: number;
|
|
340
|
+
latencyMs: AnalystLatencyDistribution | null;
|
|
341
|
+
benchmarkClockLatencyRuns: number;
|
|
342
|
+
runnerReportedLatencyRuns: number;
|
|
343
|
+
latencyUnknownRuns: number;
|
|
344
|
+
calls: number;
|
|
345
|
+
callsUnknownRuns: number;
|
|
346
|
+
inputTokens: number;
|
|
347
|
+
outputTokens: number;
|
|
348
|
+
reasoningTokens: number;
|
|
349
|
+
cachedTokens: number;
|
|
350
|
+
cacheWriteTokens: number;
|
|
351
|
+
tokenUsageUnknownRuns: number;
|
|
352
|
+
reasoningTokenUsageUnknownRuns: number;
|
|
353
|
+
cachedTokenUsageUnknownRuns: number;
|
|
354
|
+
cacheWriteTokenUsageUnknownRuns: number;
|
|
355
|
+
knownCostUsd: number;
|
|
356
|
+
costUnknownRuns: number;
|
|
357
|
+
}
|
|
358
|
+
interface AnalystBenchmarkDatasetRef {
|
|
359
|
+
id: string;
|
|
360
|
+
revision: string;
|
|
361
|
+
split?: string;
|
|
362
|
+
}
|
|
363
|
+
interface AnalystBenchmarkDescriptor {
|
|
364
|
+
id?: string;
|
|
365
|
+
dataset?: AnalystBenchmarkDatasetRef;
|
|
366
|
+
command?: string;
|
|
367
|
+
environment?: Record<string, string>;
|
|
368
|
+
metadata?: Record<string, unknown>;
|
|
369
|
+
}
|
|
370
|
+
interface AnalystBenchmarkProvenance extends AnalystBenchmarkDescriptor {
|
|
371
|
+
startedAt: string;
|
|
372
|
+
endedAt: string;
|
|
373
|
+
caseCount: number;
|
|
374
|
+
runnerIds: string[];
|
|
375
|
+
repetitions: number;
|
|
376
|
+
maxConcurrency: number;
|
|
377
|
+
runnerOrderSeed: number;
|
|
378
|
+
}
|
|
379
|
+
interface AnalystBenchmarkResult {
|
|
380
|
+
provenance: AnalystBenchmarkProvenance;
|
|
381
|
+
observations: AnalystBenchmarkObservation[];
|
|
382
|
+
summaries: AnalystBenchmarkSummary[];
|
|
383
|
+
}
|
|
384
|
+
interface RunAnalystBenchmarkOptions<TInput> {
|
|
385
|
+
cases: readonly AnalystBenchmarkCase<TInput>[];
|
|
386
|
+
runners: readonly AnalystBenchmarkRunner<TInput>[];
|
|
387
|
+
repetitions?: number;
|
|
388
|
+
maxConcurrency?: number;
|
|
389
|
+
runnerOrderSeed?: number;
|
|
390
|
+
resolveEvidence?: AnalystEvidenceResolver<TInput>;
|
|
391
|
+
benchmark?: AnalystBenchmarkDescriptor;
|
|
392
|
+
/** Previously persisted rows. Exact case, runner, repetition, and execution identities are required. */
|
|
393
|
+
initialObservations?: readonly AnalystBenchmarkObservation[];
|
|
394
|
+
onObservation?: (observation: AnalystBenchmarkObservation) => void | Promise<void>;
|
|
395
|
+
signal?: AbortSignal;
|
|
396
|
+
}
|
|
397
|
+
declare function runAnalystBenchmark<TInput>(options: RunAnalystBenchmarkOptions<TInput>): Promise<AnalystBenchmarkResult>;
|
|
398
|
+
declare function registryBenchmarkRunner(options: {
|
|
399
|
+
id: string;
|
|
400
|
+
registry: AnalystRegistry;
|
|
401
|
+
runOptions?: Omit<RegistryRunOpts, 'signal'>;
|
|
402
|
+
/** Count any selected analyst failure as a failed benchmark run. */
|
|
403
|
+
failOnAnalystFailure?: boolean;
|
|
404
|
+
}): AnalystBenchmarkRunner<AnalystRunInputs>;
|
|
405
|
+
//#endregion
|
|
406
|
+
//#region src/agent-profile.d.ts
|
|
407
|
+
/**
|
|
408
|
+
* The agentic coding harnesses an eval sweeps by default — the ones we care about
|
|
409
|
+
* ranking. This is the SINGLE source of that list; consumers import it instead of
|
|
410
|
+
* re-declaring their own (a re-declared list is how the fleet drifts). Pass an
|
|
411
|
+
* explicit `harnesses` (e.g. `harnessTypeSchema.options` for literally every known
|
|
412
|
+
* harness) to widen beyond these.
|
|
413
|
+
*/
|
|
414
|
+
declare const CODING_HARNESSES: readonly HarnessType[];
|
|
415
|
+
interface ProfileAxisSpec {
|
|
416
|
+
/** The domain profile to sweep. Its prompt/tools/skills are held fixed; only the
|
|
417
|
+
* harness and model vary. `model.default` is the fallback model. */
|
|
418
|
+
base: AgentProfile;
|
|
419
|
+
/** Harnesses to cross. Default: {@link CODING_HARNESSES}. */
|
|
420
|
+
harnesses?: readonly HarnessType[];
|
|
421
|
+
/** Models to cross. Default: `[base.model.default]` — one model, i.e. today's
|
|
422
|
+
* single-model behaviour, so omitting this never changes an existing run. */
|
|
423
|
+
models?: readonly string[];
|
|
424
|
+
/** Force every (harness, model) pair verbatim, even ones the harness can't run —
|
|
425
|
+
* for deliberately testing failure modes. Default (false): SNAP instead — a
|
|
426
|
+
* vendor-locked harness runs only the swept models in its family, or its native
|
|
427
|
+
* default when it supports none, so no harness is dropped and none gets a
|
|
428
|
+
* guaranteed-failing foreign-model cell. */
|
|
429
|
+
keepIncompatible?: boolean;
|
|
430
|
+
}
|
|
431
|
+
/** Model sentinel for a vendor-locked harness that supports none of the swept models:
|
|
432
|
+
* it carries no provider prefix, so `harnessSupportsModel` accepts it and the harness
|
|
433
|
+
* resolves it to its own native default model at runtime (e.g. kimi-code → its Kimi
|
|
434
|
+
* model). Lets `expandProfileAxes` snap-instead-of-drop without a per-harness flagship
|
|
435
|
+
* table that would rot as router catalogs change. */
|
|
436
|
+
declare const HARNESS_NATIVE_MODEL = "default";
|
|
437
|
+
/**
|
|
438
|
+
* Expand a base profile across the harness × model matrix into the `AgentProfile[]`
|
|
439
|
+
* that `runProfileMatrix` / `selfImprove` score — the ONE place "which harnesses ×
|
|
440
|
+
* which models do we evaluate" lives, so no product hand-rolls its own harness list
|
|
441
|
+
* or column→profile mapping (the pattern that let those copies drift and silently
|
|
442
|
+
* break the harness pivot).
|
|
443
|
+
*
|
|
444
|
+
* Each cell clones `base`, sets the canonical top-level `harness` and `model.default`,
|
|
445
|
+
* and stamps `metadata.harness` + `metadata.harnessModel` for matrix grouping. Both
|
|
446
|
+
* metadata fields are hash-bearing, so every cell gets a distinct `agentProfileId` row
|
|
447
|
+
* and results join back by harness/model via {@link harnessAxisOf} with no
|
|
448
|
+
* hand-recomputed key. A vendor-locked harness snaps to its family's swept models — or
|
|
449
|
+
* its native default ({@link HARNESS_NATIVE_MODEL}) when it supports none — so every
|
|
450
|
+
* requested harness runs; `keepIncompatible` forces every pair verbatim.
|
|
451
|
+
*
|
|
452
|
+
* Omit `harnesses`/`models` to sweep the full default set — the "turn it on for
|
|
453
|
+
* everything we care about" switch, identical in shape whether one harness or all.
|
|
454
|
+
*/
|
|
455
|
+
declare function expandProfileAxes(spec: ProfileAxisSpec): AgentProfile[];
|
|
456
|
+
/**
|
|
457
|
+
* Read the (harness, model) a matrix cell ran under, off a profile or a result row's
|
|
458
|
+
* profile — the join-back for a `byHarness` pivot. Returns undefined when the profile
|
|
459
|
+
* wasn't produced by {@link expandProfileAxes}. Callers group `result.byProfile` by
|
|
460
|
+
* this instead of recomputing an id (recomputing the wrong key is what broke the pivot
|
|
461
|
+
* in the hand-rolled copies).
|
|
462
|
+
*/
|
|
463
|
+
declare function harnessAxisOf(profile: Pick<AgentProfile, 'metadata'>): {
|
|
464
|
+
harness: HarnessType;
|
|
465
|
+
model: string;
|
|
466
|
+
} | undefined;
|
|
467
|
+
/**
|
|
468
|
+
* Collision-resistant, path-safe, human-readable profile id for eval artifacts.
|
|
469
|
+
* Scorecard joins still use `agentProfileHash`; this id is for run ids, matrix
|
|
470
|
+
* keys, and directory names where two profiles must not collapse onto one row.
|
|
471
|
+
* The suffix is the first 64 bits of the behaviour hash, enough for ordinary
|
|
472
|
+
* eval matrices while keeping filenames readable.
|
|
473
|
+
*/
|
|
474
|
+
declare function agentProfileId(profile: AgentProfile): string;
|
|
475
|
+
/**
|
|
476
|
+
* Deterministic behaviour identity for the canonical
|
|
477
|
+
* `@tangle-network/agent-interface` AgentProfile.
|
|
478
|
+
*
|
|
479
|
+
* `name` and `description` are labels and do not affect the hash. Profile
|
|
480
|
+
* `version`, prompt, model hints, tools, resources, hooks, modes, permissions,
|
|
481
|
+
* and extensions do affect the hash. Resource array order is hash-bearing
|
|
482
|
+
* because mount order can change agent behaviour. Undefined fields are treated
|
|
483
|
+
* as absent; explicit `null` fields remain hash-bearing.
|
|
484
|
+
*/
|
|
485
|
+
declare function agentProfileHash(profile: AgentProfile): string;
|
|
486
|
+
//#endregion
|
|
487
|
+
export { runAnalystBenchmark as A, ExternalOptimizerModelBudget as B, AnalystEvidenceResolutionError as C, AnalystLatencyDistribution as D, AnalystIssueExpectation as E, ExternalOptimizerCallbackLimits as F, ExternalOptimizerProcessLimits as G, ExternalOptimizerModelCallRequest as H, ExternalOptimizerChatRequest as I, ExternalOptimizerWireCounts as J, ExternalOptimizerResumeMode as K, ExternalOptimizerEndpointFormat as L, scoreAnalystFindings as M, DEFAULT_EXTERNAL_OPTIMIZER_CALLBACK_LIMITS as N, RunAnalystBenchmarkOptions as O, DEFAULT_EXTERNAL_OPTIMIZER_PROCESS_LIMITS as P, resolveExternalOptimizerProcessLimits as Q, ExternalOptimizerEvaluationObservation as R, AnalystEvidenceResolution as S, AnalystFindingScore as T, ExternalOptimizerModelCallResult as U, ExternalOptimizerModelCall as V, ExternalOptimizerModelExecutionObservation as W, ExternalTextEvaluationRequest as X, ExternalTextCandidate as Y, resolveExternalOptimizerCallbackLimits as Z, AnalystBenchmarkProvenance as _, ProfileAxisSpec as a, AnalystBenchmarkSummary as b, expandProfileAxes as c, AnalystBenchmarkDatasetRef as d, AnalystBenchmarkDescriptor as f, AnalystBenchmarkOutput as g, AnalystBenchmarkObservation as h, HarnessType$1 as i, traceStoreEvidenceResolver as j, registryBenchmarkRunner as k, harnessAxisOf as l, AnalystBenchmarkLabelState as m, CODING_HARNESSES as n, agentProfileHash as o, AnalystBenchmarkError as p, ExternalOptimizerRunnerCommand as q, HARNESS_NATIVE_MODEL as r, agentProfileId as s, AgentProfile$1 as t, AnalystBenchmarkCase as u, AnalystBenchmarkResult as v, AnalystEvidenceResolver as w, AnalystEvidenceExpectation as x, AnalystBenchmarkRunner as y, ExternalOptimizerEvaluationRefusalReason as z };
|
|
488
|
+
//# sourceMappingURL=agent-profile-_xPxqVJt.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"agent-profile-_xPxqVJt.d.ts","names":[],"sources":["../src/campaign/external-optimizer-contracts.ts","../src/analyst/benchmark-scoring.ts","../src/analyst/benchmark.ts","../src/agent-profile.ts"],"mappings":";;;;;;UAWiB;;EAEf;;EAEA;;EAEA;;cAGW,2CAA2C,SAAS;UAOhD;;EAEf;;EAEA;;cAGW,4CAA4C,SAAS;UAMjD;EACf;EACA;EACA,MAAM,OAAO;;EAEb,SAAS,QAAQ;;KAGP;KAEA,iCAAiC;UAE5B;EACf,WAAW;EACX;;iBAUc,sCACd,OAAO,QAAQ,6CACf,iBACC;iBAmBa,uCACd,OAAO,QAAQ,8CACf,iBACC;KAkBE,aAAa,KAAK,eAAc,6BACjC,IACA,0BAA0B,gBACf,aAAa,OACtB,+BACc,WAAW,IAAI,aAAa,EAAE,SAC1C;;KAGI,+BAA+B,aACzC,KAAK;EAA0B;;KAGrB;;UAMK;;WAEN;;WAEA,SAAS;;WAET,iBAAiB;WACjB,QAAQ;;;KAIP;WAEG;;WAEA,UAAU;;;;;;;WAOV,SAPU;;WASV;;WAGA;;WAEA;;WAEA,SATkC;;WAWlC;;;;;;;;;;;;KAaH,8BACV,SAAS,sCACN,QAAQ;;KAGD;WAEG;WACA;WACA;WACA;WACA;WACA;WACA;WACA;;WAGA;WACA;WACA;WACA;WACA;WACA;WACA;WACA;;KAGH;;;;;;KAUA;WAEG;WACA;WACA,WAAW;WACX;;WAGA;WACA;WACA,WAAW;WACX;WACA;;WAEA;WACA;;WAGA;WACA;WACA,QAAQ;WACR,YAAY;WACZ;WACA;;UAGE;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;EAUA;;;;;;EAMA,UAAU;;EAEV;;;UAIe;;EAEf;;EAEA;;;;iBCtQc,qBACd,UAAU,KAAK,oEACf,mBAAmB,mBAClB;;;UCIc;EACf;EACA,OAAO;;UAGQ;EACf;EACA;EACA;EACA;EACA,oBAAoB;EACpB;;EAEA,4BAA4B;;KAGlB;UAEK,qBAAqB;EACpC;;EAEA;;EAEA,YAAY;EACZ,OAAO;EACP,yBAAyB;;EAEzB,2BAA2B;EAC3B;EACA,WAAW;;UAGI;EACf;EACA;EACA;EACA;EACA;EACA,mBAAmB;EACnB;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;EACA;;UAGe;EACf,UAAU;EACV;EACA;;UAGe;EACf;EACA;EACA,oBAAoB;EACpB,QAAQ;;EAER;;KAGU,wBAAwB,qBAAqB;EACvD;EACA,WAAW;EACX,UAAU;EACV,SAAS;gBACK;;;;;iBAMA,2BAA2B,QACzC,WAAW,OAAO,WAAW,qBAC5B,wBAAwB;UAoBV;EACf,mBAAmB;EACnB,QAAQ;EACR,WAAW;;;;;EAKX;;EAEA,QAAQ;;UAGO;EACf;EACA;EACA;EACA;;UAGe,uBAAuB;EACtC;EACA,QACE,OAAO,QACP;IAAW;IAAgB;IAAoB,SAAS;MACvD,yBAAyB,QAAQ;;UAGrB;EACf;EACA;EACA;EACA,YAAY;EACZ;EACA;EACA;EACA;EACA,mBAAmB;EACnB,OAAO;EACP,qBAAqB;EACrB;EACA,eAAe;EACf,QAAQ;EACR,iBAAiB;EACjB,QAAQ;;UAGO;EACf;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;EACA,WAAW;EACX;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;;UAGe;EACf;EACA,UAAU;EACV;EACA,cAAc;EACd,WAAW;;UAGI,mCAAmC;EAClD;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf,YAAY;EACZ,cAAc;EACd,WAAW;;UAGI,2BAA2B;EAC1C,gBAAgB,qBAAqB;EACrC,kBAAkB,uBAAuB;EACzC;EACA;EACA;EACA,kBAAkB,wBAAwB;EAC1C,YAAY;;EAEZ,+BAA+B;EAC/B,iBAAiB,aAAa,uCAAuC;EACrE,SAAS;;iBAGW,oBAAoB,QACxC,SAAS,2BAA2B,UACnC,QAAQ;iBAyDK,wBAAwB;EACtC;EACA,UAAU;EACV,aAAa,KAAK;;EAElB;IACE,uBAAuB;;;;;;;;;;cCpUd,2BAA2B;UAOvB;;;EAGf,MAAM;;EAEN,qBAAqB;;;EAGrB;;;;;;EAMA;;;;;;;cAQW;;;;;;;;;;;;;;;;;;;iBAoBG,kBAAkB,MAAM,kBAAkB;;;;;;;;iBAwD1C,cACd,SAAS,KAAK;EACX,SAAS;EAAa;;;;;;;;;iBAiBX,eAAe,SAAS;;;;;;;;;;;iBAmDxB,iBAAiB,SAAS"}
|
package/dist/analyst/index.d.ts
CHANGED
|
@@ -1,15 +1,13 @@
|
|
|
1
1
|
import { b as CustomTokenPricing, c as CostLedgerHandle } from "../cost-ledger-DbQdN3nO.js";
|
|
2
2
|
import { I as TraceAnalystSpan, _ as ProposalFinding, a as AnalystInputKind, b as makeFinding, c as AnalystRunInputs, d as AnalystSeverity, f as AnalystUsageReceipt, g as ExecutionProbeRequest, h as ExecutionProbeOutcome, i as AnalystFinding, l as AnalystRunResult, m as ExecutionProbe, n as AnalystContext, o as AnalystRequirements, p as EvidenceRef, r as AnalystCost, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary, v as ProposalFindingOrigin, w as TraceAnalysisStore, x as makeProposalFinding, y as computeFindingId } from "../types-DN2WdT5S.js";
|
|
3
3
|
import { S as createChatClient, _ as CreateChatClientOpts, b as OpenAiCompatibleTransportOpts, f as ChatCallOpts, g as ChatTransport, h as ChatResponse, m as ChatRequest, p as ChatClient, v as CustomTransportOpts, x as SandboxSdkTransportOpts, y as MockTransportOpts } from "../types-gvRsyJLh.js";
|
|
4
|
-
import {
|
|
5
|
-
import { a as createTraceAnalyst, c as BehavioralAnalystOptions, i as TraceAnalystDefinition, l as behavioralAnalyst, n as buildDefaultAnalystRegistry, o as renderPriorFindings, r as CreateTraceAnalystOptions, s as runTraceAnalyst, t as DefaultAnalystRegistryOptions, u as deriveEfficiencyFindings } from "../default-registry-BKwc8bN5.js";
|
|
4
|
+
import { A as runAnalystBenchmark, C as AnalystEvidenceResolutionError, D as AnalystLatencyDistribution, E as AnalystIssueExpectation, M as scoreAnalystFindings, O as RunAnalystBenchmarkOptions, S as AnalystEvidenceResolution, T as AnalystFindingScore, V as ExternalOptimizerModelCall, W as ExternalOptimizerModelExecutionObservation, _ as AnalystBenchmarkProvenance, b as AnalystBenchmarkSummary, d as AnalystBenchmarkDatasetRef, f as AnalystBenchmarkDescriptor, g as AnalystBenchmarkOutput, h as AnalystBenchmarkObservation, j as traceStoreEvidenceResolver, k as registryBenchmarkRunner, m as AnalystBenchmarkLabelState, p as AnalystBenchmarkError, q as ExternalOptimizerRunnerCommand, t as AgentProfile, u as AnalystBenchmarkCase, v as AnalystBenchmarkResult, w as AnalystEvidenceResolver, x as AnalystEvidenceExpectation, y as AnalystBenchmarkRunner } from "../agent-profile-_xPxqVJt.js";
|
|
6
5
|
import { a as ExactAnalystRunPolicySnapshot, c as ExactAnalystSnapshot, d as ExactExecutionComponentSnapshot, i as ExactAnalystRunEvent, l as ExactCapableAnalyst, n as ExactAnalystExecutionPlanSnapshot, o as ExactAnalystRunResult, r as ExactAnalystRunCompletion, s as ExactAnalystRunSummary, t as ExactAnalystBudgetSnapshot, u as ExactExecutionComponentIdentity } from "../exact-types-BKOEILRP.js";
|
|
7
6
|
import { a as ExactAnalystBudgetPolicy, c as RegistryRunOpts, i as BudgetPolicy, n as AnalystRegistry, o as ExactAnalystRunExecutionError, r as AnalystRegistryOptions, s as ExactRegistryRunOpts, t as AnalystHooks } from "../registry-7pOUBrtX.js";
|
|
8
|
-
import { a as
|
|
9
|
-
import { n as
|
|
10
|
-
import {
|
|
11
|
-
import { C as
|
|
12
|
-
import { t as AgentProfile } from "../agent-profile-B9_GGsG8.js";
|
|
7
|
+
import { a as createTraceAnalyst, c as BehavioralAnalystOptions, i as TraceAnalystDefinition, l as behavioralAnalyst, n as buildDefaultAnalystRegistry, o as renderPriorFindings, r as CreateTraceAnalystOptions, s as runTraceAnalyst, t as DefaultAnalystRegistryOptions, u as deriveEfficiencyFindings } from "../default-registry-FfNzaUHV.js";
|
|
8
|
+
import { a as TraceAnalystLimits, c as RawAnalystEvidence, d as evidenceRefsFromRawFinding, f as parseRawFinding, i as TraceAnalysisEngineResult, l as RawAnalystFinding, n as TraceAnalysisEngine, o as resolveTraceAnalystLimits, r as TraceAnalysisEngineRequest, s as RAW_FINDING_SCHEMA_PROMPT, t as DEFAULT_TRACE_ANALYST_OUTPUT_TOKENS, u as RawAnalystFindingSchema } from "../engine-CvW_I72-.js";
|
|
9
|
+
import { n as buildTraceToolsForGroup, t as TraceToolGroupName } from "../tool-groups-DAe1t6zb.js";
|
|
10
|
+
import { A as SemanticConceptJudgeInput, C as DspyRlmTraceEngineOptions, E as createChatTraceEngine, S as renderFindingSubject, T as ChatTraceEngineOptions, _ as FindingSubject, a as FAILURE_MODE_KIND_SPEC, b as findingSubjectGrammarPromptFor, c as emitControlIntegrityFindings, d as FindingsStore, f as PersistedFinding, g as FINDING_SUBJECT_SYNTAX, h as FINDING_SUBJECT_KINDS, i as IMPROVEMENT_KIND_SPEC, j as SemanticConceptJudgeOptions, l as DiffPolicy, m as diffFindings, n as KNOWLEDGE_POISONING_KIND_SPEC, o as CONTROL_INTEGRITY_ANALYST, p as defaultIsMaterial, r as KNOWLEDGE_GAP_KIND_SPEC, s as ControlIntegrityAnalyst, t as DEFAULT_TRACE_ANALYST_KINDS, u as FindingsDiff, v as FindingSubjectKind, w as createDspyRlmTraceEngine, x as parseFindingSubject, y as KIND_EXPECTED_SUBJECTS } from "../index-Bn-nlnSV.js";
|
|
13
11
|
import { i as nodeHttpPrimeBridgeTransport, n as PrimeBridgeTransportRequest, r as PrimeBridgeTransportResult, t as PrimeBridgeTransport } from "../prime-bridge-transport-6feEglLf.js";
|
|
14
12
|
import { z } from "zod";
|
|
15
13
|
//#region src/analyst/adapters.d.ts
|
|
@@ -1267,11 +1265,11 @@ declare function runAnalystBenchmarkCommand(argv: readonly string[], env?: NodeJ
|
|
|
1267
1265
|
declare const ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM = "sha256-canonical-source-manifest";
|
|
1268
1266
|
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM = "sha256-canonical-file-manifest";
|
|
1269
1267
|
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES: readonly string[];
|
|
1270
|
-
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "
|
|
1268
|
+
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "10c77bd9ce4d896811395f8b623216f51bd03dab9ea209cb58e1d8d7b78acb7a";
|
|
1271
1269
|
declare const ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 = "1e03f2daed356d60316aabefb407ec1e437ac94d408d61eea4ae096e9c6fbb5b";
|
|
1272
1270
|
declare const ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 = "4dba263b6256a30d56c7fdb2d992d3a953c0035d731f359b704db806f68f75ac";
|
|
1273
1271
|
declare const ANALYST_BENCHMARK_IMPLEMENTATION_FILES: readonly string[];
|
|
1274
|
-
declare const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "
|
|
1272
|
+
declare const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "ffe86598b87f4fd8e8ba432c5bad5a4a86bba1c4969cfde3b9a28f8c72770d64";
|
|
1275
1273
|
declare function analystBenchmarkImplementationDigest(): string;
|
|
1276
1274
|
declare function analystBenchmarkDependencyLockDigest(): string;
|
|
1277
1275
|
//#endregion
|