@tangle-network/agent-eval 0.128.2 → 0.129.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +265 -0
- package/README.md +18 -0
- package/dist/analyst/index.d.ts +107 -165
- package/dist/analyst/index.js +5 -9
- package/dist/analyst/index.js.map +1 -1
- package/dist/belief-state/index.d.ts +2 -19
- package/dist/belief-state/index.js +30 -31
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +5 -8
- package/dist/benchmarks/index.js +12 -11
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +30 -39
- package/dist/campaign/index.js +11 -10
- package/dist/{chunk-NKAGIDE2.js → chunk-2QU3YOPR.js} +15 -274
- package/dist/chunk-2QU3YOPR.js.map +1 -0
- package/dist/{chunk-EJGRPCO3.js → chunk-3OCR4R5I.js} +245 -134
- package/dist/chunk-3OCR4R5I.js.map +1 -0
- package/dist/{chunk-2JX3CFMB.js → chunk-56TAVBOK.js} +5 -2
- package/dist/chunk-56TAVBOK.js.map +1 -0
- package/dist/{chunk-DPUHNQLN.js → chunk-7FO3TNPI.js} +2 -2
- package/dist/{chunk-DJKY2TSY.js → chunk-BSO5JDQH.js} +27 -120
- package/dist/chunk-BSO5JDQH.js.map +1 -0
- package/dist/{chunk-EZJEIH2R.js → chunk-C6LXANRU.js} +11 -20
- package/dist/chunk-C6LXANRU.js.map +1 -0
- package/dist/{chunk-ZUUWPZCV.js → chunk-DODXQREJ.js} +4 -4
- package/dist/{chunk-2MKQIFS4.js → chunk-E7QXT7SX.js} +2 -2
- package/dist/{chunk-NYLOYM6N.js → chunk-EG66UGL4.js} +37 -28
- package/dist/chunk-EG66UGL4.js.map +1 -0
- package/dist/{chunk-P5W7RQKK.js → chunk-FXTVJPYD.js} +2 -2
- package/dist/{chunk-VZSRQ272.js → chunk-G7MGMCZD.js} +6 -2
- package/dist/chunk-G7MGMCZD.js.map +1 -0
- package/dist/{chunk-IHQDPH7D.js → chunk-H23X7XKK.js} +85 -75
- package/dist/chunk-H23X7XKK.js.map +1 -0
- package/dist/{chunk-VBQ3CRKH.js → chunk-HPWUNB47.js} +4 -6
- package/dist/chunk-HPWUNB47.js.map +1 -0
- package/dist/{chunk-XPRT64IE.js → chunk-IYCLP2N2.js} +3 -3
- package/dist/chunk-IYCLP2N2.js.map +1 -0
- package/dist/{chunk-XDWDC2MP.js → chunk-JQSF5DQT.js} +11 -5
- package/dist/chunk-JQSF5DQT.js.map +1 -0
- package/dist/{chunk-NACAGYSY.js → chunk-M4YBQKIJ.js} +11 -11
- package/dist/chunk-M4YBQKIJ.js.map +1 -0
- package/dist/{chunk-YJBNWCAA.js → chunk-NY44NC4A.js} +3 -3
- package/dist/chunk-OIUOT4QD.js +44 -0
- package/dist/chunk-OIUOT4QD.js.map +1 -0
- package/dist/chunk-OWN5NPMC.js +152 -0
- package/dist/chunk-OWN5NPMC.js.map +1 -0
- package/dist/chunk-PC5DOSM7.js +579 -0
- package/dist/chunk-PC5DOSM7.js.map +1 -0
- package/dist/{chunk-UB2LOJ6Q.js → chunk-QB6BDBP2.js} +23 -20
- package/dist/chunk-QB6BDBP2.js.map +1 -0
- package/dist/chunk-RXHCETDZ.js +536 -0
- package/dist/chunk-RXHCETDZ.js.map +1 -0
- package/dist/{chunk-PBE2LOSS.js → chunk-SFLLL76A.js} +7 -7
- package/dist/chunk-SFLLL76A.js.map +1 -0
- package/dist/{chunk-VGRCHJON.js → chunk-T6RLYGAD.js} +3 -8
- package/dist/chunk-T6RLYGAD.js.map +1 -0
- package/dist/{chunk-VLOATJQ2.js → chunk-TJVT4QFF.js} +21 -18
- package/dist/chunk-TJVT4QFF.js.map +1 -0
- package/dist/{chunk-S5YLIBFX.js → chunk-TQ7LNKZ3.js} +2 -2
- package/dist/{chunk-EOSZT7PL.js → chunk-U4L7JRPZ.js} +2 -297
- package/dist/chunk-U4L7JRPZ.js.map +1 -0
- package/dist/chunk-U4PHLT2N.js +419 -0
- package/dist/chunk-U4PHLT2N.js.map +1 -0
- package/dist/{chunk-WS3NZZQQ.js → chunk-VCZ5FQYW.js} +3 -4
- package/dist/chunk-VCZ5FQYW.js.map +1 -0
- package/dist/{chunk-BYT7ELPS.js → chunk-WVATSFCP.js} +2 -2
- package/dist/{chunk-TSN7JT6D.js → chunk-X4YIBDER.js} +21 -5
- package/dist/{chunk-TSN7JT6D.js.map → chunk-X4YIBDER.js.map} +1 -1
- package/dist/{chunk-TBL77AUT.js → chunk-YQN4ICPP.js} +5 -5
- package/dist/{chunk-MHELPNRP.js → chunk-ZHTZ4EYI.js} +1 -1
- package/dist/chunk-ZHTZ4EYI.js.map +1 -0
- package/dist/cli.js +6 -5
- package/dist/cli.js.map +1 -1
- package/dist/contract/index.d.ts +47 -87
- package/dist/contract/index.js +14 -13
- package/dist/contract/index.js.map +1 -1
- package/dist/control.js +3 -2
- package/dist/fuzz.js +3 -2
- package/dist/fuzz.js.map +1 -1
- package/dist/index.d.ts +659 -203
- package/dist/index.js +145 -117
- package/dist/index.js.map +1 -1
- package/dist/meta-eval/index.js +2 -2
- package/dist/multishot/index.d.ts +3 -4
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.js +5 -5
- package/dist/reporting.d.ts +14 -0
- package/dist/reporting.js +7 -6
- package/dist/rl.d.ts +652 -82
- package/dist/rl.js +415 -171
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1071 -32
- package/dist/rollout/index.js +68 -10
- package/dist/run-campaign-OJJ7CZF4.js +18 -0
- package/dist/supervisor-run/index.d.ts +114 -4
- package/dist/supervisor-run/index.js +4 -3
- package/dist/traces.d.ts +1 -1
- package/dist/traces.js +6 -5
- package/dist/wire/index.d.ts +10 -11
- package/dist/wire/index.js +3 -3
- package/docs/feature-guide.md +1 -1
- package/docs/rollout.md +116 -2
- package/package.json +4 -4
- package/dist/chunk-2JX3CFMB.js.map +0 -1
- package/dist/chunk-DJKY2TSY.js.map +0 -1
- package/dist/chunk-EJGRPCO3.js.map +0 -1
- package/dist/chunk-EOSZT7PL.js.map +0 -1
- package/dist/chunk-EZJEIH2R.js.map +0 -1
- package/dist/chunk-IHQDPH7D.js.map +0 -1
- package/dist/chunk-MHELPNRP.js.map +0 -1
- package/dist/chunk-NACAGYSY.js.map +0 -1
- package/dist/chunk-NKAGIDE2.js.map +0 -1
- package/dist/chunk-NYLOYM6N.js.map +0 -1
- package/dist/chunk-PBE2LOSS.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-UB2LOJ6Q.js.map +0 -1
- package/dist/chunk-UWZZKKU7.js +0 -237
- package/dist/chunk-UWZZKKU7.js.map +0 -1
- package/dist/chunk-VBQ3CRKH.js.map +0 -1
- package/dist/chunk-VGRCHJON.js.map +0 -1
- package/dist/chunk-VLOATJQ2.js.map +0 -1
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WS3NZZQQ.js.map +0 -1
- package/dist/chunk-XDWDC2MP.js.map +0 -1
- package/dist/chunk-XPRT64IE.js.map +0 -1
- package/dist/run-campaign-ISHFZ7FJ.js +0 -17
- /package/dist/{chunk-DPUHNQLN.js.map → chunk-7FO3TNPI.js.map} +0 -0
- /package/dist/{chunk-ZUUWPZCV.js.map → chunk-DODXQREJ.js.map} +0 -0
- /package/dist/{chunk-2MKQIFS4.js.map → chunk-E7QXT7SX.js.map} +0 -0
- /package/dist/{chunk-P5W7RQKK.js.map → chunk-FXTVJPYD.js.map} +0 -0
- /package/dist/{chunk-YJBNWCAA.js.map → chunk-NY44NC4A.js.map} +0 -0
- /package/dist/{chunk-S5YLIBFX.js.map → chunk-TQ7LNKZ3.js.map} +0 -0
- /package/dist/{chunk-BYT7ELPS.js.map → chunk-WVATSFCP.js.map} +0 -0
- /package/dist/{chunk-TBL77AUT.js.map → chunk-YQN4ICPP.js.map} +0 -0
- /package/dist/{run-campaign-ISHFZ7FJ.js.map → run-campaign-OJJ7CZF4.js.map} +0 -0
package/dist/contract/index.d.ts
CHANGED
|
@@ -216,9 +216,8 @@ type CostLedgerHandle = Pick<CostLedger, Exclude<keyof CostLedger, 'listPending'
|
|
|
216
216
|
* { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
|
|
217
217
|
* )
|
|
218
218
|
*
|
|
219
|
-
*
|
|
220
|
-
*
|
|
221
|
-
* that need free-form text use `callLlm` and parse output themselves.
|
|
219
|
+
* `createChatClient` wraps this implementation for provider-neutral package
|
|
220
|
+
* entry points. Direct callers can use `callLlm` or `callLlmJson`.
|
|
222
221
|
*/
|
|
223
222
|
|
|
224
223
|
interface LlmMessage {
|
|
@@ -652,7 +651,7 @@ interface JudgeConfig<TArtifact, TScenario extends Scenario$1 = Scenario$1> {
|
|
|
652
651
|
/** The canonical judge verdict shape — one declaration, shared by campaign
|
|
653
652
|
* judges and the multishot judge runner (which re-exports this type).
|
|
654
653
|
*
|
|
655
|
-
* Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the
|
|
654
|
+
* Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the
|
|
656
655
|
* multishot runner emits 0-10. Cross-scale comparison must go through
|
|
657
656
|
* `detectScale` (src/campaign/gates/statistical-heldout.ts, used by
|
|
658
657
|
* promotion-policy) — never renormalize a producer's values in place, as
|
|
@@ -872,9 +871,6 @@ interface SurfaceProposer<TFindings = unknown> {
|
|
|
872
871
|
reason?: string;
|
|
873
872
|
};
|
|
874
873
|
}
|
|
875
|
-
/** Optional vocabulary alias. The loop is the optimizer; this object is the
|
|
876
|
-
* proposer inside that loop. */
|
|
877
|
-
type OptimizationProposer<TFindings = unknown> = SurfaceProposer<TFindings>;
|
|
878
874
|
interface OptimizerConfigBase {
|
|
879
875
|
populationSize: number;
|
|
880
876
|
maxGenerations: number;
|
|
@@ -1152,8 +1148,6 @@ interface CampaignAggregates {
|
|
|
1152
1148
|
byScenario: Record<string, ScenarioAggregate>;
|
|
1153
1149
|
/** Canonical campaign accounting, including worker and judge calls. */
|
|
1154
1150
|
cost: CostLedgerSummary;
|
|
1155
|
-
/** Compatibility alias of `cost.totalCostUsd`. */
|
|
1156
|
-
totalCostUsd: number;
|
|
1157
1151
|
/** Cells whose dispatch completed, including cells whose later judge failed. */
|
|
1158
1152
|
cellsExecuted: number;
|
|
1159
1153
|
cellsSkipped: number;
|
|
@@ -1164,7 +1158,7 @@ interface CampaignAggregates {
|
|
|
1164
1158
|
cellsDispatchFailed?: number;
|
|
1165
1159
|
/** Present on results that record failure stages. */
|
|
1166
1160
|
cellsJudgeFailed?: number;
|
|
1167
|
-
/**
|
|
1161
|
+
/** Failures whose stage could not be classified. */
|
|
1168
1162
|
cellsUnclassifiedFailed?: number;
|
|
1169
1163
|
}
|
|
1170
1164
|
interface CampaignResult<TArtifact = unknown, TScenario extends Scenario$1 = Scenario$1> {
|
|
@@ -1222,7 +1216,7 @@ interface CampaignStorage {
|
|
|
1222
1216
|
write(path: string, content: string | Uint8Array): void;
|
|
1223
1217
|
/** Append only when the current UTF-8 byte length matches `expectedBytes`.
|
|
1224
1218
|
* Returns the new length, or undefined when another writer won. */
|
|
1225
|
-
append
|
|
1219
|
+
append(path: string, content: string, expectedBytes: number): number | undefined;
|
|
1226
1220
|
}
|
|
1227
1221
|
/** Node-filesystem storage — the default. Lazily requires `node:fs` so the
|
|
1228
1222
|
* module imports cleanly in non-Node runtimes (where the caller passes
|
|
@@ -1358,9 +1352,9 @@ interface RunCampaignOptions<TScenario extends Scenario$1, TArtifact> {
|
|
|
1358
1352
|
}) => string | undefined;
|
|
1359
1353
|
}
|
|
1360
1354
|
/** Durable `<cell>/failure-receipt.json` written before a failed cell can
|
|
1361
|
-
* trigger campaign-wide cancellation. The cell
|
|
1362
|
-
*
|
|
1363
|
-
*
|
|
1355
|
+
* trigger campaign-wide cancellation. The cell records dispatch measurements;
|
|
1356
|
+
* `cost` covers every settled agent and judge call attributed to this exact run
|
|
1357
|
+
* attempt. */
|
|
1364
1358
|
interface CampaignCellFailureReceipt<TArtifact = unknown> {
|
|
1365
1359
|
schemaVersion: 1;
|
|
1366
1360
|
runAttemptId: string;
|
|
@@ -1632,42 +1626,27 @@ interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario$1> exte
|
|
|
1632
1626
|
declare function runImprovementLoop<TScenario extends Scenario$1, TArtifact>(opts: RunImprovementLoopOptions<TScenario, TArtifact>): Promise<RunImprovementLoopResult<TArtifact, TScenario>>;
|
|
1633
1627
|
|
|
1634
1628
|
/**
|
|
1635
|
-
*
|
|
1636
|
-
*
|
|
1637
|
-
*
|
|
1638
|
-
*
|
|
1639
|
-
*
|
|
1640
|
-
* couples analyst code to runtime concerns (cli-bridge vs router vs
|
|
1641
|
-
* sandbox-sdk) it shouldn't know about.
|
|
1642
|
-
*
|
|
1643
|
-
* `ChatClient` is one interface every analyst takes via `AnalystContext.chat`.
|
|
1644
|
-
* The operator decides at the registry boundary which transport binds
|
|
1645
|
-
* to it. Analyst code stays transport-agnostic; swapping production
|
|
1646
|
-
* (sandbox-sdk) for local dev (cli-bridge) or tests (mock) is a one-
|
|
1647
|
-
* line factory call.
|
|
1648
|
-
*
|
|
1649
|
-
* Designed to coexist: existing `LlmClient` callers and existing
|
|
1650
|
-
* `TCloud`-based judges keep working untouched. New analyst code uses
|
|
1651
|
-
* `ChatClient`. When old call sites migrate, they pick up budgeting,
|
|
1652
|
-
* cancellation, and unified telemetry for free.
|
|
1629
|
+
* Provider-neutral chat contract for every model call made by agent-eval.
|
|
1630
|
+
*
|
|
1631
|
+
* Callers choose the transport at the package boundary with `createChatClient`.
|
|
1632
|
+
* Evaluation code receives canonical requests and results without importing a
|
|
1633
|
+
* provider SDK.
|
|
1653
1634
|
*/
|
|
1654
1635
|
|
|
1655
1636
|
/**
|
|
1656
|
-
* Unified chat interface
|
|
1657
|
-
* compatible mental model stays. Two methods: a one-shot `chat()` and
|
|
1658
|
-
* an `streamChat()` for future agentic loops (not yet exposed).
|
|
1637
|
+
* Unified chat interface using the package's canonical LLM request and result.
|
|
1659
1638
|
*/
|
|
1660
1639
|
interface ChatClient {
|
|
1661
|
-
/** Display name of the bound transport
|
|
1640
|
+
/** Display name of the bound transport, included in telemetry. */
|
|
1662
1641
|
readonly transport: ChatTransport;
|
|
1663
|
-
/** Default model when caller omits
|
|
1642
|
+
/** Default model when the caller omits one. */
|
|
1664
1643
|
readonly defaultModel?: string;
|
|
1665
1644
|
/** Total provider attempts this transport can make for one chat call. */
|
|
1666
1645
|
readonly maximumAttempts?: number;
|
|
1667
1646
|
/** Implementations must enforce `req.maxTokens` when it is present. */
|
|
1668
1647
|
chat(req: ChatRequest, opts?: ChatCallOpts): Promise<ChatResponse>;
|
|
1669
1648
|
}
|
|
1670
|
-
type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'mock';
|
|
1649
|
+
type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'custom' | 'mock';
|
|
1671
1650
|
interface ChatRequest extends Omit<LlmCallRequest, 'model'> {
|
|
1672
1651
|
/** Optional — falls back to ChatClient.defaultModel. */
|
|
1673
1652
|
model?: string;
|
|
@@ -1683,7 +1662,7 @@ interface ChatCallOpts {
|
|
|
1683
1662
|
/** Stable provider idempotency key for retries/redrives of one paid call. */
|
|
1684
1663
|
idempotencyKey?: string;
|
|
1685
1664
|
}
|
|
1686
|
-
type CreateChatClientOpts = RouterTransportOpts | CliBridgeTransportOpts | DirectProviderTransportOpts | SandboxSdkTransportOpts | MockTransportOpts;
|
|
1665
|
+
type CreateChatClientOpts = RouterTransportOpts | CliBridgeTransportOpts | DirectProviderTransportOpts | SandboxSdkTransportOpts | CustomTransportOpts | MockTransportOpts;
|
|
1687
1666
|
interface BaseTransportOpts {
|
|
1688
1667
|
defaultModel?: string;
|
|
1689
1668
|
/** Total provider attempts. Required for opaque transports used in capped runs. */
|
|
@@ -1705,15 +1684,18 @@ interface DirectProviderTransportOpts extends BaseTransportOpts {
|
|
|
1705
1684
|
apiKey: string;
|
|
1706
1685
|
}
|
|
1707
1686
|
/**
|
|
1708
|
-
* Sandbox-SDK transport.
|
|
1709
|
-
*
|
|
1710
|
-
* configured Sandbox handle. We don't import the SDK here to keep
|
|
1711
|
-
* agent-eval dep-free of @tangle-network/sandbox.
|
|
1687
|
+
* Sandbox-SDK transport. The caller supplies a canonical chat function for an
|
|
1688
|
+
* already-configured Sandbox handle, so agent-eval does not import the SDK.
|
|
1712
1689
|
*/
|
|
1713
1690
|
interface SandboxSdkTransportOpts extends BaseTransportOpts {
|
|
1714
1691
|
transport: 'sandbox-sdk';
|
|
1715
1692
|
chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
|
|
1716
1693
|
}
|
|
1694
|
+
/** Caller-adapted SDK or transport returning the canonical ChatResponse shape. */
|
|
1695
|
+
interface CustomTransportOpts extends BaseTransportOpts {
|
|
1696
|
+
transport: 'custom';
|
|
1697
|
+
chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
|
|
1698
|
+
}
|
|
1717
1699
|
/**
|
|
1718
1700
|
* Mock transport for tests. The handler receives the request and returns
|
|
1719
1701
|
* whatever the test wants. No retries, no JSON-schema degrade.
|
|
@@ -2550,6 +2532,18 @@ interface VerifiableRewardExtractionOptions {
|
|
|
2550
2532
|
* doesn't report one. Default `0.7`.
|
|
2551
2533
|
*/
|
|
2552
2534
|
judgeConfidenceFloor?: number;
|
|
2535
|
+
/**
|
|
2536
|
+
* Whether the anti-Goodhart realness gate applies. Default `true`, and the
|
|
2537
|
+
* default is the one every training path must keep.
|
|
2538
|
+
*
|
|
2539
|
+
* Set `false` ONLY for detection and analysis. `rl/reward-hacking.ts` does,
|
|
2540
|
+
* for the same reason it reads `observedScore` for its proxy: it measures the
|
|
2541
|
+
* DIVERGENCE between the judge signal and the deterministic one, and a
|
|
2542
|
+
* deterministic reward that another gate already forced to 0 manufactures
|
|
2543
|
+
* exactly that divergence on exactly the gamed population. The detector would
|
|
2544
|
+
* then be re-reporting a verdict it was supposed to reach independently.
|
|
2545
|
+
*/
|
|
2546
|
+
applyRealnessGate?: boolean;
|
|
2553
2547
|
}
|
|
2554
2548
|
|
|
2555
2549
|
/**
|
|
@@ -2835,8 +2829,6 @@ interface JudgeInput {
|
|
|
2835
2829
|
costPhase?: string;
|
|
2836
2830
|
costTags?: Record<string, string>;
|
|
2837
2831
|
signal?: AbortSignal;
|
|
2838
|
-
/** Exact maximum provider attempts configured on the supplied TCloud client. */
|
|
2839
|
-
tcloudMaximumAttempts?: number;
|
|
2840
2832
|
}
|
|
2841
2833
|
|
|
2842
2834
|
interface PairedBootstrapResult {
|
|
@@ -4747,13 +4739,8 @@ interface AnalystRunSummary {
|
|
|
4747
4739
|
reason?: string;
|
|
4748
4740
|
findings_count: number;
|
|
4749
4741
|
latency_ms: number;
|
|
4750
|
-
|
|
4751
|
-
|
|
4752
|
-
* Additive receipt for model usage. Registry-produced summaries populate it
|
|
4753
|
-
* even when the analyst emits no findings. `cost_usd` remains the legacy
|
|
4754
|
-
* numeric field; inspect `usage.cost` before treating zero as observed.
|
|
4755
|
-
*/
|
|
4756
|
-
usage?: AnalystUsageReceipt;
|
|
4742
|
+
/** Additive model usage and cost provenance for this analyst. */
|
|
4743
|
+
usage: AnalystUsageReceipt;
|
|
4757
4744
|
/** When `status='failed'`: the error class + message, never the full stack. */
|
|
4758
4745
|
error?: {
|
|
4759
4746
|
class: string;
|
|
@@ -4815,12 +4802,9 @@ type AnalystRunEvent = {
|
|
|
4815
4802
|
/**
|
|
4816
4803
|
* Typed Ax output for analyst findings.
|
|
4817
4804
|
*
|
|
4818
|
-
*
|
|
4819
|
-
*
|
|
4820
|
-
*
|
|
4821
|
-
* native structured output; at the kind-factory boundary we Zod-validate
|
|
4822
|
-
* each emitted finding so malformed rows fail loud instead of being
|
|
4823
|
-
* silently lifted with default severity.
|
|
4805
|
+
* Ax binds the field as `findings:json[]` so the provider emits native
|
|
4806
|
+
* structured output. At the kind-factory boundary every row is validated
|
|
4807
|
+
* before it becomes an `AnalystFinding`.
|
|
4824
4808
|
*
|
|
4825
4809
|
* Why not `f.object().array()` directly in the signature? The Ax
|
|
4826
4810
|
* signature string `question:string -> findings:json[]` already lets
|
|
@@ -4829,31 +4813,7 @@ type AnalystRunEvent = {
|
|
|
4829
4813
|
* validation surface independent of which Ax version is installed.
|
|
4830
4814
|
*/
|
|
4831
4815
|
|
|
4832
|
-
/** Original public schema retained for stored rows and callback contracts. */
|
|
4833
4816
|
declare const RawAnalystFindingSchema: z.ZodObject<{
|
|
4834
|
-
evidence_uri: z.ZodString;
|
|
4835
|
-
evidence_excerpt: z.ZodOptional<z.ZodString>;
|
|
4836
|
-
severity: z.ZodEnum<{
|
|
4837
|
-
low: "low";
|
|
4838
|
-
high: "high";
|
|
4839
|
-
medium: "medium";
|
|
4840
|
-
critical: "critical";
|
|
4841
|
-
info: "info";
|
|
4842
|
-
}>;
|
|
4843
|
-
claim: z.ZodString;
|
|
4844
|
-
subject: z.ZodOptional<z.ZodString>;
|
|
4845
|
-
confidence: z.ZodNumber;
|
|
4846
|
-
rationale: z.ZodOptional<z.ZodString>;
|
|
4847
|
-
recommended_action: z.ZodOptional<z.ZodString>;
|
|
4848
|
-
}, z.core.$strict>;
|
|
4849
|
-
type RawAnalystFinding = z.infer<typeof RawAnalystFindingSchema>;
|
|
4850
|
-
/**
|
|
4851
|
-
* Canonical plural-evidence contract. The preprocessor accepts the original
|
|
4852
|
-
* `evidence_uri` / `evidence_excerpt` pair and normalizes it into one evidence
|
|
4853
|
-
* item so persisted rows and older model fixtures remain readable. New output
|
|
4854
|
-
* always receives the plural shape.
|
|
4855
|
-
*/
|
|
4856
|
-
declare const CanonicalRawAnalystFindingSchema: z.ZodPreprocess<z.ZodObject<{
|
|
4857
4817
|
evidence: z.ZodArray<z.ZodObject<{
|
|
4858
4818
|
uri: z.ZodString;
|
|
4859
4819
|
excerpt: z.ZodOptional<z.ZodString>;
|
|
@@ -4870,8 +4830,8 @@ declare const CanonicalRawAnalystFindingSchema: z.ZodPreprocess<z.ZodObject<{
|
|
|
4870
4830
|
confidence: z.ZodNumber;
|
|
4871
4831
|
rationale: z.ZodOptional<z.ZodString>;
|
|
4872
4832
|
recommended_action: z.ZodOptional<z.ZodString>;
|
|
4873
|
-
}, z.core.$strict
|
|
4874
|
-
type
|
|
4833
|
+
}, z.core.$strict>;
|
|
4834
|
+
type RawAnalystFinding = z.infer<typeof RawAnalystFindingSchema>;
|
|
4875
4835
|
|
|
4876
4836
|
/**
|
|
4877
4837
|
* Analyst-kind factory — the typed way to define trace analysts.
|
|
@@ -4880,7 +4840,7 @@ type CanonicalRawAnalystFinding = z.infer<typeof CanonicalRawAnalystFindingSchem
|
|
|
4880
4840
|
* and bounded Ax subqueries target one failure-mode lens (failure-mode
|
|
4881
4841
|
* classification, knowledge gap discovery, knowledge poisoning,
|
|
4882
4842
|
* self-improvement, ...). Kinds emit findings in the typed
|
|
4883
|
-
* `
|
|
4843
|
+
* `RawAnalystFinding` shape via a JSON-array Ax output; the factory
|
|
4884
4844
|
* validates each row with Zod and lifts it into `AnalystFinding[]`.
|
|
4885
4845
|
*
|
|
4886
4846
|
* Composition rules:
|
|
@@ -4943,7 +4903,7 @@ interface TraceAnalystKindSpec {
|
|
|
4943
4903
|
*/
|
|
4944
4904
|
interface TraceAnalystGolden {
|
|
4945
4905
|
question: string;
|
|
4946
|
-
expected: ReadonlyArray<Omit<
|
|
4906
|
+
expected: ReadonlyArray<Omit<RawAnalystFinding, 'confidence'>>;
|
|
4947
4907
|
}
|
|
4948
4908
|
|
|
4949
4909
|
/**
|
|
@@ -5806,4 +5766,4 @@ interface FromOtelSpansOptions {
|
|
|
5806
5766
|
}
|
|
5807
5767
|
declare function fromOtelSpans(opts: FromOtelSpansOptions): RunRecord[];
|
|
5808
5768
|
|
|
5809
|
-
export { type AgentEvalAgent, type AgentEvalEvaluateOptions, type AgentEvalImproveOptions, type AgentTraceContributor, type AgentTraceContributorType, type AgentTraceConversation, type AgentTraceFile, type AgentTraceIndex, type AgentTraceRange, type AgentTraceRecord, type AnalystFinding, type AnalyzeRunsOptions, type AuthoringProvenance, type AxisEvidence, type AxisVerdict, type BuildEvidenceVectorOptions, type CampaignAggregates, type CampaignArtifactWriter, type CampaignCellFailureReceipt, type CampaignCellResult, type CampaignCostMeter, type CampaignResult, type CampaignStorage, type CampaignTraceWriter, type CandidateExperimentExecutionInput, type ChatClient, type CodeAgentSessionAction, type CodeAgentSessionActionKind, type CodeAgentSessionActionStatus, type CodeAgentSessionActionSurface, type CodeAgentSessionDiagnostic, type CodeAgentSessionExecutionReceipt, type CodeAgentSessionIntakeOptions, type CodeAgentSessionIntakeResult, type CodeAgentSessionMetrics, type CodeAgentSessionObservation, type CodeAgentSessionSource, type CodeAgentSessionTerminalStatus, type CodeSurface, type CompareCandidateExperimentOptions, type CompareOptimizationMethodsOptions, type ComparisonCost, type CostLedgerHandle, type CostProvenanceSummary, type CreateChatClientOpts, type DefaultAnalystRegistryOptions, type DefaultProductionGateCheck, type DefaultProductionGateOptions, type DefaultProductionRewardHackingOptions, type DefineAgentEvalOptions, type DefinedAgentEval, type DeploymentOutcome, type DispatchFn as Dispatch, type DispatchContext, type EvalCellScoreDelta, type EvalDimensionDelta, type EvalGenerationDiff, type EvalReportingSuiteInput, type EvalReportingSuiteOptions, type EvalReportingSuiteResult, type EvalRunDiff, type EvaluatePairedMeasurementsOptions, type EvidenceVector, type ExecutionErrorOutcomeCell, type ExecutionInsight, type ExecutionReport, type ExternalOptimizationExample, type ExternalTextEvaluationResponse, type ExternalTextOptimizationMethodConfig, type ExternalTextOptimizerContext, type ExternalTextOptimizerResult, type FailureClassTally, type FailureClusterInsight, type FeedbackTableMeta, type FeedbackTableRow, FileSystemOutcomeStore, type FileSystemOutcomeStoreOptions, type FromFeedbackTableOptions, type FromFeedbackTableResult, type FromOtelSpansOptions, type FromRunRecordDirOptions, type FromRunRecordDirResult, type Gate, type GateCheckStatus, type GateContext, type GateContribution, type GateDecision, type GateResult, type GenerationCandidate, type GenerationRecord, type GepaAdaptiveEngineRun, type GepaEngineOptions, type GepaEngineRun, type GepaOptimizationMethodConfig, type GepaOptimizationRecipe, type GepaRunnerCommand, type HeldOutGateOptions, type HostedTenant, InMemoryOutcomeStore, type InsightReport, type InterRaterInsight, type JudgeConfig, type JudgeDimension, type JudgeInsight, type JudgeScore, type LiftInsight, type LlmJudgeDimension, type LlmJudgeOptions, type MutableSurface, type ObjectiveSource, type OpenAICompatibleOptimizerModel, type OptimizationMethod, type OptimizationMethodComparison, type OptimizationMethodInput, type OptimizationMethodProvenance, type OptimizationMethodResult, type OptimizationPackageSource, type
|
|
5769
|
+
export { type AgentEvalAgent, type AgentEvalEvaluateOptions, type AgentEvalImproveOptions, type AgentTraceContributor, type AgentTraceContributorType, type AgentTraceConversation, type AgentTraceFile, type AgentTraceIndex, type AgentTraceRange, type AgentTraceRecord, type AnalystFinding, type AnalyzeRunsOptions, type AuthoringProvenance, type AxisEvidence, type AxisVerdict, type BuildEvidenceVectorOptions, type CampaignAggregates, type CampaignArtifactWriter, type CampaignCellFailureReceipt, type CampaignCellResult, type CampaignCostMeter, type CampaignResult, type CampaignStorage, type CampaignTraceWriter, type CandidateExperimentExecutionInput, type ChatClient, type CodeAgentSessionAction, type CodeAgentSessionActionKind, type CodeAgentSessionActionStatus, type CodeAgentSessionActionSurface, type CodeAgentSessionDiagnostic, type CodeAgentSessionExecutionReceipt, type CodeAgentSessionIntakeOptions, type CodeAgentSessionIntakeResult, type CodeAgentSessionMetrics, type CodeAgentSessionObservation, type CodeAgentSessionSource, type CodeAgentSessionTerminalStatus, type CodeSurface, type CompareCandidateExperimentOptions, type CompareOptimizationMethodsOptions, type ComparisonCost, type CostLedgerHandle, type CostProvenanceSummary, type CreateChatClientOpts, type DefaultAnalystRegistryOptions, type DefaultProductionGateCheck, type DefaultProductionGateOptions, type DefaultProductionRewardHackingOptions, type DefineAgentEvalOptions, type DefinedAgentEval, type DeploymentOutcome, type DispatchFn as Dispatch, type DispatchContext, type EvalCellScoreDelta, type EvalDimensionDelta, type EvalGenerationDiff, type EvalReportingSuiteInput, type EvalReportingSuiteOptions, type EvalReportingSuiteResult, type EvalRunDiff, type EvaluatePairedMeasurementsOptions, type EvidenceVector, type ExecutionErrorOutcomeCell, type ExecutionInsight, type ExecutionReport, type ExternalOptimizationExample, type ExternalTextEvaluationResponse, type ExternalTextOptimizationMethodConfig, type ExternalTextOptimizerContext, type ExternalTextOptimizerResult, type FailureClassTally, type FailureClusterInsight, type FeedbackTableMeta, type FeedbackTableRow, FileSystemOutcomeStore, type FileSystemOutcomeStoreOptions, type FromFeedbackTableOptions, type FromFeedbackTableResult, type FromOtelSpansOptions, type FromRunRecordDirOptions, type FromRunRecordDirResult, type Gate, type GateCheckStatus, type GateContext, type GateContribution, type GateDecision, type GateResult, type GenerationCandidate, type GenerationRecord, type GepaAdaptiveEngineRun, type GepaEngineOptions, type GepaEngineRun, type GepaOptimizationMethodConfig, type GepaOptimizationRecipe, type GepaRunnerCommand, type HeldOutGateOptions, type HostedTenant, InMemoryOutcomeStore, type InsightReport, type InterRaterInsight, type JudgeConfig, type JudgeDimension, type JudgeInsight, type JudgeScore, type LiftInsight, type LlmJudgeDimension, type LlmJudgeOptions, type MutableSurface, type ObjectiveSource, type OpenAICompatibleOptimizerModel, type OptimizationMethod, type OptimizationMethodComparison, type OptimizationMethodInput, type OptimizationMethodProvenance, type OptimizationMethodResult, type OptimizationPackageSource, type OptimizationTokenUsage, type OptimizerConfig, type OptimizerModelBudget, type OutcomeCorrelationInsight, type OutcomeStore, type PairedMeasurement, type PairedMeasurementAdapter, type PairedMeasurementEvaluation, type ParetoSignificanceGateOptions, type ParsedCodeAgentJsonl, type PartitionByAuthoringModelResult, type PromotionObjective, type PromotionPolicy, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, type Recommendation, type ReferenceEquivalenceJudgeInput, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceJudgeResult, type ReferenceEquivalenceScenario, type ReleaseSummary, type RunCampaignOptions, type RunCandidateExperimentOptions, type RunEvalOptions, type RunImprovementLoopOptions, type RunImprovementLoopResult, type RunRecordRejection, type ScalarDistribution, type Scenario$1 as Scenario, type SealCandidateBenchmarkSuiteOptions, type SelfImproveBudget, type SelfImproveOptions, type SelfImproveProgressEvent, type SelfImproveResult, SelfImproveRunError, type SessionScript, type SkillOptOptimizationMethodConfig, type SkillOptRunnerCommand, type SkillOptTrainerConfig, type SummarizeExecutionOptions, type SurfaceProposer, type TokenUsageInsight, analyzeRuns, buildDefaultAnalystRegistry, buildEvidenceVector, campaignSplitDigest, compareOptimizationMethods, composeGate, createChatClient, createReferenceEquivalenceJudge, defaultProductionGate, defineAgentEval, diffGenerations, diffRunBaselineToWinner, diffRuns, evalReportingSuite, evaluatePairedMeasurements, externalTextOptimizationMethod, fromClaudeCodeSession, fromCodexSession, fromFeedbackTable, fromKimiCodeSession, fromOpenCodeSession, fromOtelSpans, fromPiSession, fromPigraphSession, fromRunRecordDir, fsCampaignStorage, gepaOptimizationMethod, heldOutGate, inMemoryCampaignStorage, llmJudge, measuredComparisonFromCandidateExperiment, observeCodeAgentSession, paretoPolicy, paretoSignificanceGate, parseAgentTrace, parseCodeAgentJsonl, partitionRunsByAuthoringModel, runCampaign, runCandidateExperiment, runEval, runImprovementLoop, runReferenceEquivalenceJudge, sealCandidateBenchmarkSuite, sealCandidateBenchmarkTask, sealCandidateExperiment, selfImprove, skillOptOptimizationMethod, summarizeExecution, verifyCandidateBenchmarkSuite, verifyCandidateBenchmarkSuiteInputs, verifyCandidateBenchmarkTask, verifyCandidateExperiment, verifyCandidateExperimentComparison };
|
package/dist/contract/index.js
CHANGED
|
@@ -14,7 +14,7 @@ import {
|
|
|
14
14
|
import {
|
|
15
15
|
analyzeRuns,
|
|
16
16
|
summarizeExecution
|
|
17
|
-
} from "../chunk-
|
|
17
|
+
} from "../chunk-M4YBQKIJ.js";
|
|
18
18
|
import {
|
|
19
19
|
REFERENCE_EQUIVALENCE_INPUT_LIMITS,
|
|
20
20
|
REFERENCE_EQUIVALENCE_JUDGE_VERSION,
|
|
@@ -40,7 +40,7 @@ import {
|
|
|
40
40
|
skillOptOptimizationMethod,
|
|
41
41
|
surfaceContentHash,
|
|
42
42
|
surfaceHash
|
|
43
|
-
} from "../chunk-
|
|
43
|
+
} from "../chunk-2QU3YOPR.js";
|
|
44
44
|
import {
|
|
45
45
|
campaignSplitDigest,
|
|
46
46
|
createRunCostLedger,
|
|
@@ -48,31 +48,31 @@ import {
|
|
|
48
48
|
inMemoryCampaignStorage,
|
|
49
49
|
resolveRunDir,
|
|
50
50
|
runCampaign
|
|
51
|
-
} from "../chunk-
|
|
51
|
+
} from "../chunk-C6LXANRU.js";
|
|
52
52
|
import {
|
|
53
53
|
buildDefaultAnalystRegistry,
|
|
54
54
|
createChatClient
|
|
55
|
-
} from "../chunk-
|
|
55
|
+
} from "../chunk-BSO5JDQH.js";
|
|
56
56
|
import "../chunk-HHWE3POT.js";
|
|
57
57
|
import "../chunk-WGXIEX7P.js";
|
|
58
58
|
import {
|
|
59
59
|
FileSystemOutcomeStore,
|
|
60
60
|
InMemoryOutcomeStore
|
|
61
61
|
} from "../chunk-3RF76KTD.js";
|
|
62
|
-
import "../chunk-
|
|
62
|
+
import "../chunk-EG66UGL4.js";
|
|
63
63
|
import {
|
|
64
64
|
campaignCellExecutionEvidence,
|
|
65
65
|
campaignCellJudgeDimensions,
|
|
66
66
|
campaignCellTaskScore,
|
|
67
67
|
campaignCellToRunRecord
|
|
68
|
-
} from "../chunk-
|
|
69
|
-
import "../chunk-
|
|
70
|
-
import "../chunk-
|
|
71
|
-
import "../chunk-
|
|
68
|
+
} from "../chunk-E7QXT7SX.js";
|
|
69
|
+
import "../chunk-SFLLL76A.js";
|
|
70
|
+
import "../chunk-TJVT4QFF.js";
|
|
71
|
+
import "../chunk-7FO3TNPI.js";
|
|
72
72
|
import {
|
|
73
73
|
pairedBootstrap
|
|
74
|
-
} from "../chunk-
|
|
75
|
-
import "../chunk-
|
|
74
|
+
} from "../chunk-ZHTZ4EYI.js";
|
|
75
|
+
import "../chunk-VCZ5FQYW.js";
|
|
76
76
|
import "../chunk-VI2UW6B6.js";
|
|
77
77
|
import {
|
|
78
78
|
readTaskFailureLabels,
|
|
@@ -91,8 +91,9 @@ import "../chunk-PC4UYEBM.js";
|
|
|
91
91
|
import {
|
|
92
92
|
modelHasSnapshot,
|
|
93
93
|
parseRunRecordSafe
|
|
94
|
-
} from "../chunk-
|
|
94
|
+
} from "../chunk-56TAVBOK.js";
|
|
95
95
|
import "../chunk-MA6HLL3S.js";
|
|
96
|
+
import "../chunk-OIUOT4QD.js";
|
|
96
97
|
import {
|
|
97
98
|
ValidationError
|
|
98
99
|
} from "../chunk-ONWEPEDO.js";
|
|
@@ -512,7 +513,7 @@ async function shipEvalRunToHosted(tenant, opts, summary, raw, runDir) {
|
|
|
512
513
|
surface,
|
|
513
514
|
cells,
|
|
514
515
|
compositeMean,
|
|
515
|
-
costUsd: campaign.aggregates.totalCostUsd,
|
|
516
|
+
costUsd: campaign.aggregates.cost.totalCostUsd,
|
|
516
517
|
durationMs
|
|
517
518
|
};
|
|
518
519
|
}
|