@tangle-network/agent-eval 0.128.1 → 0.129.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +271 -0
- package/README.md +18 -0
- package/dist/analyst/index.d.ts +107 -165
- package/dist/analyst/index.js +5 -9
- package/dist/analyst/index.js.map +1 -1
- package/dist/belief-state/index.d.ts +2 -19
- package/dist/belief-state/index.js +30 -31
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +5 -8
- package/dist/benchmarks/index.js +12 -11
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +30 -39
- package/dist/campaign/index.js +11 -10
- package/dist/{chunk-NKAGIDE2.js → chunk-2QU3YOPR.js} +15 -274
- package/dist/chunk-2QU3YOPR.js.map +1 -0
- package/dist/{chunk-EJGRPCO3.js → chunk-3OCR4R5I.js} +245 -134
- package/dist/chunk-3OCR4R5I.js.map +1 -0
- package/dist/{chunk-2JX3CFMB.js → chunk-56TAVBOK.js} +5 -2
- package/dist/chunk-56TAVBOK.js.map +1 -0
- package/dist/{chunk-DPUHNQLN.js → chunk-7FO3TNPI.js} +2 -2
- package/dist/{chunk-DJKY2TSY.js → chunk-BSO5JDQH.js} +27 -120
- package/dist/chunk-BSO5JDQH.js.map +1 -0
- package/dist/{chunk-EZJEIH2R.js → chunk-C6LXANRU.js} +11 -20
- package/dist/chunk-C6LXANRU.js.map +1 -0
- package/dist/{chunk-ZUUWPZCV.js → chunk-DODXQREJ.js} +4 -4
- package/dist/{chunk-2MKQIFS4.js → chunk-E7QXT7SX.js} +2 -2
- package/dist/{chunk-NYLOYM6N.js → chunk-EG66UGL4.js} +37 -28
- package/dist/chunk-EG66UGL4.js.map +1 -0
- package/dist/{chunk-P5W7RQKK.js → chunk-FXTVJPYD.js} +2 -2
- package/dist/{chunk-VZSRQ272.js → chunk-G7MGMCZD.js} +6 -2
- package/dist/chunk-G7MGMCZD.js.map +1 -0
- package/dist/{chunk-IHQDPH7D.js → chunk-H23X7XKK.js} +85 -75
- package/dist/chunk-H23X7XKK.js.map +1 -0
- package/dist/{chunk-VBQ3CRKH.js → chunk-HPWUNB47.js} +4 -6
- package/dist/chunk-HPWUNB47.js.map +1 -0
- package/dist/{chunk-XPRT64IE.js → chunk-IYCLP2N2.js} +3 -3
- package/dist/chunk-IYCLP2N2.js.map +1 -0
- package/dist/{chunk-XDWDC2MP.js → chunk-JQSF5DQT.js} +11 -5
- package/dist/chunk-JQSF5DQT.js.map +1 -0
- package/dist/{chunk-NACAGYSY.js → chunk-M4YBQKIJ.js} +11 -11
- package/dist/chunk-M4YBQKIJ.js.map +1 -0
- package/dist/{chunk-YJBNWCAA.js → chunk-NY44NC4A.js} +3 -3
- package/dist/chunk-OIUOT4QD.js +44 -0
- package/dist/chunk-OIUOT4QD.js.map +1 -0
- package/dist/chunk-OWN5NPMC.js +152 -0
- package/dist/chunk-OWN5NPMC.js.map +1 -0
- package/dist/chunk-PC5DOSM7.js +579 -0
- package/dist/chunk-PC5DOSM7.js.map +1 -0
- package/dist/{chunk-UB2LOJ6Q.js → chunk-QB6BDBP2.js} +23 -20
- package/dist/chunk-QB6BDBP2.js.map +1 -0
- package/dist/chunk-RXHCETDZ.js +536 -0
- package/dist/chunk-RXHCETDZ.js.map +1 -0
- package/dist/{chunk-PBE2LOSS.js → chunk-SFLLL76A.js} +7 -7
- package/dist/chunk-SFLLL76A.js.map +1 -0
- package/dist/{chunk-VGRCHJON.js → chunk-T6RLYGAD.js} +3 -8
- package/dist/chunk-T6RLYGAD.js.map +1 -0
- package/dist/{chunk-VLOATJQ2.js → chunk-TJVT4QFF.js} +21 -18
- package/dist/chunk-TJVT4QFF.js.map +1 -0
- package/dist/{chunk-S5YLIBFX.js → chunk-TQ7LNKZ3.js} +2 -2
- package/dist/{chunk-EOSZT7PL.js → chunk-U4L7JRPZ.js} +2 -297
- package/dist/chunk-U4L7JRPZ.js.map +1 -0
- package/dist/chunk-U4PHLT2N.js +419 -0
- package/dist/chunk-U4PHLT2N.js.map +1 -0
- package/dist/{chunk-WS3NZZQQ.js → chunk-VCZ5FQYW.js} +3 -4
- package/dist/chunk-VCZ5FQYW.js.map +1 -0
- package/dist/{chunk-BYT7ELPS.js → chunk-WVATSFCP.js} +2 -2
- package/dist/{chunk-TSN7JT6D.js → chunk-X4YIBDER.js} +21 -5
- package/dist/{chunk-TSN7JT6D.js.map → chunk-X4YIBDER.js.map} +1 -1
- package/dist/{chunk-TBL77AUT.js → chunk-YQN4ICPP.js} +5 -5
- package/dist/{chunk-MHELPNRP.js → chunk-ZHTZ4EYI.js} +1 -1
- package/dist/chunk-ZHTZ4EYI.js.map +1 -0
- package/dist/cli.js +6 -5
- package/dist/cli.js.map +1 -1
- package/dist/contract/index.d.ts +47 -87
- package/dist/contract/index.js +14 -13
- package/dist/contract/index.js.map +1 -1
- package/dist/control.js +3 -2
- package/dist/fuzz.js +3 -2
- package/dist/fuzz.js.map +1 -1
- package/dist/index.d.ts +659 -203
- package/dist/index.js +145 -117
- package/dist/index.js.map +1 -1
- package/dist/meta-eval/index.js +2 -2
- package/dist/multishot/index.d.ts +3 -4
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.js +5 -5
- package/dist/reporting.d.ts +14 -0
- package/dist/reporting.js +7 -6
- package/dist/rl.d.ts +652 -82
- package/dist/rl.js +415 -171
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1071 -32
- package/dist/rollout/index.js +68 -10
- package/dist/run-campaign-OJJ7CZF4.js +18 -0
- package/dist/supervisor-run/index.d.ts +114 -4
- package/dist/supervisor-run/index.js +4 -3
- package/dist/traces.d.ts +1 -1
- package/dist/traces.js +6 -5
- package/dist/wire/index.d.ts +10 -11
- package/dist/wire/index.js +3 -3
- package/docs/feature-guide.md +1 -1
- package/docs/rollout.md +116 -2
- package/package.json +4 -4
- package/dist/chunk-2JX3CFMB.js.map +0 -1
- package/dist/chunk-DJKY2TSY.js.map +0 -1
- package/dist/chunk-EJGRPCO3.js.map +0 -1
- package/dist/chunk-EOSZT7PL.js.map +0 -1
- package/dist/chunk-EZJEIH2R.js.map +0 -1
- package/dist/chunk-IHQDPH7D.js.map +0 -1
- package/dist/chunk-MHELPNRP.js.map +0 -1
- package/dist/chunk-NACAGYSY.js.map +0 -1
- package/dist/chunk-NKAGIDE2.js.map +0 -1
- package/dist/chunk-NYLOYM6N.js.map +0 -1
- package/dist/chunk-PBE2LOSS.js.map +0 -1
- package/dist/chunk-TT4KNT67.js +0 -124
- package/dist/chunk-TT4KNT67.js.map +0 -1
- package/dist/chunk-UB2LOJ6Q.js.map +0 -1
- package/dist/chunk-UWZZKKU7.js +0 -237
- package/dist/chunk-UWZZKKU7.js.map +0 -1
- package/dist/chunk-VBQ3CRKH.js.map +0 -1
- package/dist/chunk-VGRCHJON.js.map +0 -1
- package/dist/chunk-VLOATJQ2.js.map +0 -1
- package/dist/chunk-VZSRQ272.js.map +0 -1
- package/dist/chunk-WS3NZZQQ.js.map +0 -1
- package/dist/chunk-XDWDC2MP.js.map +0 -1
- package/dist/chunk-XPRT64IE.js.map +0 -1
- package/dist/run-campaign-ISHFZ7FJ.js +0 -17
- /package/dist/{chunk-DPUHNQLN.js.map → chunk-7FO3TNPI.js.map} +0 -0
- /package/dist/{chunk-ZUUWPZCV.js.map → chunk-DODXQREJ.js.map} +0 -0
- /package/dist/{chunk-2MKQIFS4.js.map → chunk-E7QXT7SX.js.map} +0 -0
- /package/dist/{chunk-P5W7RQKK.js.map → chunk-FXTVJPYD.js.map} +0 -0
- /package/dist/{chunk-YJBNWCAA.js.map → chunk-NY44NC4A.js.map} +0 -0
- /package/dist/{chunk-S5YLIBFX.js.map → chunk-TQ7LNKZ3.js.map} +0 -0
- /package/dist/{chunk-BYT7ELPS.js.map → chunk-WVATSFCP.js.map} +0 -0
- /package/dist/{chunk-TBL77AUT.js.map → chunk-YQN4ICPP.js.map} +0 -0
- /package/dist/{run-campaign-ISHFZ7FJ.js.map → run-campaign-OJJ7CZF4.js.map} +0 -0
package/dist/index.d.ts
CHANGED
|
@@ -2,7 +2,6 @@ import { AgentProfile, HarnessType } from '@tangle-network/agent-interface';
|
|
|
2
2
|
export { AgentProfile, HarnessType } from '@tangle-network/agent-interface';
|
|
3
3
|
import { AxAIArgs, AxAIService, AxFunction, AxAgentActorTurnCallbackArgs } from '@ax-llm/ax';
|
|
4
4
|
import { z } from 'zod';
|
|
5
|
-
import { TCloud } from '@tangle-network/tcloud';
|
|
6
5
|
|
|
7
6
|
/**
|
|
8
7
|
* TraceSchema v1 — the canonical data model for agent-eval.
|
|
@@ -1166,8 +1165,6 @@ interface CostReceipt extends CostCallBase, CostUsage {
|
|
|
1166
1165
|
actualCostUsd?: number;
|
|
1167
1166
|
error?: string;
|
|
1168
1167
|
}
|
|
1169
|
-
/** @deprecated Read-only compatibility shape. New paid work uses `runPaidCall`. */
|
|
1170
|
-
type CostLedgerEntry = Omit<CostReceipt, 'status' | 'callId' | 'phase' | 'actor' | 'maximumCostUsd' | 'usageUnknown' | 'pricing' | 'error'>;
|
|
1171
1168
|
interface CostReceiptInput extends CostUsage {
|
|
1172
1169
|
model: string;
|
|
1173
1170
|
/** Caller-supplied rates for a local estimate when the provider does not report billed cost. */
|
|
@@ -1506,9 +1503,8 @@ declare function providerFromBaseUrl(baseUrl: string): string;
|
|
|
1506
1503
|
* { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
|
|
1507
1504
|
* )
|
|
1508
1505
|
*
|
|
1509
|
-
*
|
|
1510
|
-
*
|
|
1511
|
-
* that need free-form text use `callLlm` and parse output themselves.
|
|
1506
|
+
* `createChatClient` wraps this implementation for provider-neutral package
|
|
1507
|
+
* entry points. Direct callers can use `callLlm` or `callLlmJson`.
|
|
1512
1508
|
*/
|
|
1513
1509
|
|
|
1514
1510
|
interface LlmMessage {
|
|
@@ -1644,8 +1640,8 @@ interface LlmClientOptions {
|
|
|
1644
1640
|
* total attempts × `timeoutMs`.
|
|
1645
1641
|
*/
|
|
1646
1642
|
deadlineMs?: number;
|
|
1647
|
-
/** Total provider attempts.
|
|
1648
|
-
|
|
1643
|
+
/** Total provider attempts. Default 3. */
|
|
1644
|
+
maximumAttempts?: number;
|
|
1649
1645
|
/** Token rates used when the provider omits cost or package pricing does not cover the model. */
|
|
1650
1646
|
customTokenPricing?: CustomTokenPricing;
|
|
1651
1647
|
/**
|
|
@@ -1693,10 +1689,9 @@ interface LlmClientOptions {
|
|
|
1693
1689
|
* name/message/code, then recurses into `error.cause` — undici nests the
|
|
1694
1690
|
* real socket fault one or more levels under `.cause`.
|
|
1695
1691
|
*
|
|
1696
|
-
* This is
|
|
1697
|
-
* `withJudgeRetry` both route through it, so
|
|
1698
|
-
*
|
|
1699
|
-
* TCloud-backed judge.
|
|
1692
|
+
* This is the retry classifier for the package: `callLlm` and
|
|
1693
|
+
* `withJudgeRetry` both route through it, so connection failures are treated
|
|
1694
|
+
* consistently across transports.
|
|
1700
1695
|
*/
|
|
1701
1696
|
declare function isTransientLlmError(err: unknown): boolean;
|
|
1702
1697
|
/** Exponential backoff: 500ms, 1s, 2s, 4s, ... capped at 16s. Attempt is 0-indexed. */
|
|
@@ -1797,42 +1792,27 @@ declare class LlmClient {
|
|
|
1797
1792
|
}
|
|
1798
1793
|
|
|
1799
1794
|
/**
|
|
1800
|
-
*
|
|
1801
|
-
*
|
|
1802
|
-
* agent-eval already ships an `LlmClient` (OpenAI-compatible, retry,
|
|
1803
|
-
* graceful JSON-schema degrade) and judges that talk to `TCloud`. Two
|
|
1804
|
-
* mixed patterns force every analyst author to pick a transport, which
|
|
1805
|
-
* couples analyst code to runtime concerns (cli-bridge vs router vs
|
|
1806
|
-
* sandbox-sdk) it shouldn't know about.
|
|
1807
|
-
*
|
|
1808
|
-
* `ChatClient` is one interface every analyst takes via `AnalystContext.chat`.
|
|
1809
|
-
* The operator decides at the registry boundary which transport binds
|
|
1810
|
-
* to it. Analyst code stays transport-agnostic; swapping production
|
|
1811
|
-
* (sandbox-sdk) for local dev (cli-bridge) or tests (mock) is a one-
|
|
1812
|
-
* line factory call.
|
|
1795
|
+
* Provider-neutral chat contract for every model call made by agent-eval.
|
|
1813
1796
|
*
|
|
1814
|
-
*
|
|
1815
|
-
*
|
|
1816
|
-
*
|
|
1817
|
-
* cancellation, and unified telemetry for free.
|
|
1797
|
+
* Callers choose the transport at the package boundary with `createChatClient`.
|
|
1798
|
+
* Evaluation code receives canonical requests and results without importing a
|
|
1799
|
+
* provider SDK.
|
|
1818
1800
|
*/
|
|
1819
1801
|
|
|
1820
1802
|
/**
|
|
1821
|
-
* Unified chat interface
|
|
1822
|
-
* compatible mental model stays. Two methods: a one-shot `chat()` and
|
|
1823
|
-
* an `streamChat()` for future agentic loops (not yet exposed).
|
|
1803
|
+
* Unified chat interface using the package's canonical LLM request and result.
|
|
1824
1804
|
*/
|
|
1825
1805
|
interface ChatClient {
|
|
1826
|
-
/** Display name of the bound transport
|
|
1806
|
+
/** Display name of the bound transport, included in telemetry. */
|
|
1827
1807
|
readonly transport: ChatTransport;
|
|
1828
|
-
/** Default model when caller omits
|
|
1808
|
+
/** Default model when the caller omits one. */
|
|
1829
1809
|
readonly defaultModel?: string;
|
|
1830
1810
|
/** Total provider attempts this transport can make for one chat call. */
|
|
1831
1811
|
readonly maximumAttempts?: number;
|
|
1832
1812
|
/** Implementations must enforce `req.maxTokens` when it is present. */
|
|
1833
1813
|
chat(req: ChatRequest, opts?: ChatCallOpts): Promise<ChatResponse>;
|
|
1834
1814
|
}
|
|
1835
|
-
type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'mock';
|
|
1815
|
+
type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'custom' | 'mock';
|
|
1836
1816
|
interface ChatRequest extends Omit<LlmCallRequest, 'model'> {
|
|
1837
1817
|
/** Optional — falls back to ChatClient.defaultModel. */
|
|
1838
1818
|
model?: string;
|
|
@@ -1848,7 +1828,7 @@ interface ChatCallOpts {
|
|
|
1848
1828
|
/** Stable provider idempotency key for retries/redrives of one paid call. */
|
|
1849
1829
|
idempotencyKey?: string;
|
|
1850
1830
|
}
|
|
1851
|
-
type CreateChatClientOpts = RouterTransportOpts | CliBridgeTransportOpts | DirectProviderTransportOpts | SandboxSdkTransportOpts | MockTransportOpts;
|
|
1831
|
+
type CreateChatClientOpts = RouterTransportOpts | CliBridgeTransportOpts | DirectProviderTransportOpts | SandboxSdkTransportOpts | CustomTransportOpts | MockTransportOpts;
|
|
1852
1832
|
interface BaseTransportOpts {
|
|
1853
1833
|
defaultModel?: string;
|
|
1854
1834
|
/** Total provider attempts. Required for opaque transports used in capped runs. */
|
|
@@ -1870,15 +1850,18 @@ interface DirectProviderTransportOpts extends BaseTransportOpts {
|
|
|
1870
1850
|
apiKey: string;
|
|
1871
1851
|
}
|
|
1872
1852
|
/**
|
|
1873
|
-
* Sandbox-SDK transport.
|
|
1874
|
-
*
|
|
1875
|
-
* configured Sandbox handle. We don't import the SDK here to keep
|
|
1876
|
-
* agent-eval dep-free of @tangle-network/sandbox.
|
|
1853
|
+
* Sandbox-SDK transport. The caller supplies a canonical chat function for an
|
|
1854
|
+
* already-configured Sandbox handle, so agent-eval does not import the SDK.
|
|
1877
1855
|
*/
|
|
1878
1856
|
interface SandboxSdkTransportOpts extends BaseTransportOpts {
|
|
1879
1857
|
transport: 'sandbox-sdk';
|
|
1880
1858
|
chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
|
|
1881
1859
|
}
|
|
1860
|
+
/** Caller-adapted SDK or transport returning the canonical ChatResponse shape. */
|
|
1861
|
+
interface CustomTransportOpts extends BaseTransportOpts {
|
|
1862
|
+
transport: 'custom';
|
|
1863
|
+
chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
|
|
1864
|
+
}
|
|
1882
1865
|
/**
|
|
1883
1866
|
* Mock transport for tests. The handler receives the request and returns
|
|
1884
1867
|
* whatever the test wants. No retries, no JSON-schema degrade.
|
|
@@ -2370,7 +2353,14 @@ type RunTaskFailure = {
|
|
|
2370
2353
|
failureClass: Exclude<FailureClass, 'success'>;
|
|
2371
2354
|
failureMode?: string;
|
|
2372
2355
|
};
|
|
2373
|
-
/**
|
|
2356
|
+
/**
|
|
2357
|
+
* Return task quality, preferring held-out evidence when both scores exist.
|
|
2358
|
+
*
|
|
2359
|
+
* RAW: no realness protection is applied. Built on `observedScore` rather
|
|
2360
|
+
* than repeating the split derivation, so only `rollout/reward.ts` reads the
|
|
2361
|
+
* raw fields. Anything that becomes training data must use `trainingScore` or
|
|
2362
|
+
* `trainingReward` instead.
|
|
2363
|
+
*/
|
|
2374
2364
|
declare function runTaskScore(record: RunRecord): number | undefined;
|
|
2375
2365
|
declare class RunRecordValidationError extends ValidationError {
|
|
2376
2366
|
readonly path: string;
|
|
@@ -2677,8 +2667,6 @@ interface BenchmarkRunnerConfig {
|
|
|
2677
2667
|
promptVersion?: string;
|
|
2678
2668
|
/** Shared ledger for agent and judge calls made by the benchmark. */
|
|
2679
2669
|
costLedger?: CostLedgerHandle;
|
|
2680
|
-
/** Exact maximum provider attempts configured on the supplied TCloud client. */
|
|
2681
|
-
tcloudMaximumAttempts?: number;
|
|
2682
2670
|
}
|
|
2683
2671
|
interface JudgeInput {
|
|
2684
2672
|
scenario: Scenario$1;
|
|
@@ -2689,11 +2677,8 @@ interface JudgeInput {
|
|
|
2689
2677
|
costPhase?: string;
|
|
2690
2678
|
costTags?: Record<string, string>;
|
|
2691
2679
|
signal?: AbortSignal;
|
|
2692
|
-
/** Exact maximum provider attempts configured on the supplied TCloud client. */
|
|
2693
|
-
tcloudMaximumAttempts?: number;
|
|
2694
2680
|
}
|
|
2695
|
-
type JudgeFn = (
|
|
2696
|
-
|
|
2681
|
+
type JudgeFn = (chat: ChatClient, input: JudgeInput) => Promise<JudgeScore$1[]>;
|
|
2697
2682
|
interface TestResult {
|
|
2698
2683
|
name: string;
|
|
2699
2684
|
passed: boolean;
|
|
@@ -2935,13 +2920,8 @@ interface AnalystRunSummary {
|
|
|
2935
2920
|
reason?: string;
|
|
2936
2921
|
findings_count: number;
|
|
2937
2922
|
latency_ms: number;
|
|
2938
|
-
|
|
2939
|
-
|
|
2940
|
-
* Additive receipt for model usage. Registry-produced summaries populate it
|
|
2941
|
-
* even when the analyst emits no findings. `cost_usd` remains the legacy
|
|
2942
|
-
* numeric field; inspect `usage.cost` before treating zero as observed.
|
|
2943
|
-
*/
|
|
2944
|
-
usage?: AnalystUsageReceipt;
|
|
2923
|
+
/** Additive model usage and cost provenance for this analyst. */
|
|
2924
|
+
usage: AnalystUsageReceipt;
|
|
2945
2925
|
/** When `status='failed'`: the error class + message, never the full stack. */
|
|
2946
2926
|
error?: {
|
|
2947
2927
|
class: string;
|
|
@@ -3003,12 +2983,9 @@ type AnalystRunEvent = {
|
|
|
3003
2983
|
/**
|
|
3004
2984
|
* Typed Ax output for analyst findings.
|
|
3005
2985
|
*
|
|
3006
|
-
*
|
|
3007
|
-
*
|
|
3008
|
-
*
|
|
3009
|
-
* native structured output; at the kind-factory boundary we Zod-validate
|
|
3010
|
-
* each emitted finding so malformed rows fail loud instead of being
|
|
3011
|
-
* silently lifted with default severity.
|
|
2986
|
+
* Ax binds the field as `findings:json[]` so the provider emits native
|
|
2987
|
+
* structured output. At the kind-factory boundary every row is validated
|
|
2988
|
+
* before it becomes an `AnalystFinding`.
|
|
3012
2989
|
*
|
|
3013
2990
|
* Why not `f.object().array()` directly in the signature? The Ax
|
|
3014
2991
|
* signature string `question:string -> findings:json[]` already lets
|
|
@@ -3022,31 +2999,7 @@ declare const RawAnalystEvidenceSchema: z.ZodObject<{
|
|
|
3022
2999
|
excerpt: z.ZodOptional<z.ZodString>;
|
|
3023
3000
|
}, z.core.$strict>;
|
|
3024
3001
|
type RawAnalystEvidence = z.infer<typeof RawAnalystEvidenceSchema>;
|
|
3025
|
-
/** Original public schema retained for stored rows and callback contracts. */
|
|
3026
3002
|
declare const RawAnalystFindingSchema: z.ZodObject<{
|
|
3027
|
-
evidence_uri: z.ZodString;
|
|
3028
|
-
evidence_excerpt: z.ZodOptional<z.ZodString>;
|
|
3029
|
-
severity: z.ZodEnum<{
|
|
3030
|
-
info: "info";
|
|
3031
|
-
critical: "critical";
|
|
3032
|
-
medium: "medium";
|
|
3033
|
-
low: "low";
|
|
3034
|
-
high: "high";
|
|
3035
|
-
}>;
|
|
3036
|
-
claim: z.ZodString;
|
|
3037
|
-
subject: z.ZodOptional<z.ZodString>;
|
|
3038
|
-
confidence: z.ZodNumber;
|
|
3039
|
-
rationale: z.ZodOptional<z.ZodString>;
|
|
3040
|
-
recommended_action: z.ZodOptional<z.ZodString>;
|
|
3041
|
-
}, z.core.$strict>;
|
|
3042
|
-
type RawAnalystFinding = z.infer<typeof RawAnalystFindingSchema>;
|
|
3043
|
-
/**
|
|
3044
|
-
* Canonical plural-evidence contract. The preprocessor accepts the original
|
|
3045
|
-
* `evidence_uri` / `evidence_excerpt` pair and normalizes it into one evidence
|
|
3046
|
-
* item so persisted rows and older model fixtures remain readable. New output
|
|
3047
|
-
* always receives the plural shape.
|
|
3048
|
-
*/
|
|
3049
|
-
declare const CanonicalRawAnalystFindingSchema: z.ZodPreprocess<z.ZodObject<{
|
|
3050
3003
|
evidence: z.ZodArray<z.ZodObject<{
|
|
3051
3004
|
uri: z.ZodString;
|
|
3052
3005
|
excerpt: z.ZodOptional<z.ZodString>;
|
|
@@ -3063,8 +3016,8 @@ declare const CanonicalRawAnalystFindingSchema: z.ZodPreprocess<z.ZodObject<{
|
|
|
3063
3016
|
confidence: z.ZodNumber;
|
|
3064
3017
|
rationale: z.ZodOptional<z.ZodString>;
|
|
3065
3018
|
recommended_action: z.ZodOptional<z.ZodString>;
|
|
3066
|
-
}, z.core.$strict
|
|
3067
|
-
type
|
|
3019
|
+
}, z.core.$strict>;
|
|
3020
|
+
type RawAnalystFinding = z.infer<typeof RawAnalystFindingSchema>;
|
|
3068
3021
|
|
|
3069
3022
|
/**
|
|
3070
3023
|
* Analyst-kind factory — the typed way to define trace analysts.
|
|
@@ -3073,7 +3026,7 @@ type CanonicalRawAnalystFinding = z.infer<typeof CanonicalRawAnalystFindingSchem
|
|
|
3073
3026
|
* and bounded Ax subqueries target one failure-mode lens (failure-mode
|
|
3074
3027
|
* classification, knowledge gap discovery, knowledge poisoning,
|
|
3075
3028
|
* self-improvement, ...). Kinds emit findings in the typed
|
|
3076
|
-
* `
|
|
3029
|
+
* `RawAnalystFinding` shape via a JSON-array Ax output; the factory
|
|
3077
3030
|
* validates each row with Zod and lifts it into `AnalystFinding[]`.
|
|
3078
3031
|
*
|
|
3079
3032
|
* Composition rules:
|
|
@@ -3136,7 +3089,7 @@ interface TraceAnalystKindSpec {
|
|
|
3136
3089
|
*/
|
|
3137
3090
|
interface TraceAnalystGolden {
|
|
3138
3091
|
question: string;
|
|
3139
|
-
expected: ReadonlyArray<Omit<
|
|
3092
|
+
expected: ReadonlyArray<Omit<RawAnalystFinding, 'confidence'>>;
|
|
3140
3093
|
}
|
|
3141
3094
|
interface CreateTraceAnalystKindOpts {
|
|
3142
3095
|
/** AxAIService bound at registration time. */
|
|
@@ -3831,9 +3784,9 @@ declare function ghCliClient(opts?: GhCliClientOptions): AutoPrClient;
|
|
|
3831
3784
|
* Domain-agnostic. Each agent provides its own scenarios, judges, and system prompt.
|
|
3832
3785
|
*/
|
|
3833
3786
|
declare class BenchmarkRunner {
|
|
3834
|
-
private
|
|
3787
|
+
private chat;
|
|
3835
3788
|
private config;
|
|
3836
|
-
constructor(
|
|
3789
|
+
constructor(chat: ChatClient, config: BenchmarkRunnerConfig);
|
|
3837
3790
|
run(scenarios?: Scenario$1[]): Promise<BenchmarkReport$1>;
|
|
3838
3791
|
}
|
|
3839
3792
|
|
|
@@ -4210,8 +4163,6 @@ interface AgentDriverConfig {
|
|
|
4210
4163
|
productContext?: string;
|
|
4211
4164
|
/** Shared account for driver-model calls. */
|
|
4212
4165
|
costLedger?: CostLedgerHandle;
|
|
4213
|
-
/** Exact provider attempt count, required when costLedger has a cap. */
|
|
4214
|
-
tcloudMaximumAttempts?: number;
|
|
4215
4166
|
}
|
|
4216
4167
|
/**
|
|
4217
4168
|
* AgentDriver — meta-agent that plays a persona against the real product.
|
|
@@ -4221,13 +4172,12 @@ interface AgentDriverConfig {
|
|
|
4221
4172
|
* the next realistic user message.
|
|
4222
4173
|
*/
|
|
4223
4174
|
declare class AgentDriver {
|
|
4224
|
-
private
|
|
4175
|
+
private chat;
|
|
4225
4176
|
private client;
|
|
4226
4177
|
private driverModel;
|
|
4227
4178
|
private productContext;
|
|
4228
4179
|
private costLedger;
|
|
4229
|
-
|
|
4230
|
-
constructor(tc: TCloud, config: AgentDriverConfig);
|
|
4180
|
+
constructor(chat: ChatClient, config: AgentDriverConfig);
|
|
4231
4181
|
/**
|
|
4232
4182
|
* Run a persona through the product.
|
|
4233
4183
|
*
|
|
@@ -4295,8 +4245,6 @@ interface DecideNextUserTurnOpts {
|
|
|
4295
4245
|
costLedger?: CostLedgerHandle;
|
|
4296
4246
|
/** Attribution tags merged into the paid-call receipt. */
|
|
4297
4247
|
costTags?: Record<string, string>;
|
|
4298
|
-
/** Exact provider attempt count, required when costLedger has a cap. */
|
|
4299
|
-
tcloudMaximumAttempts?: number;
|
|
4300
4248
|
}
|
|
4301
4249
|
/**
|
|
4302
4250
|
* Decide the simulated user's next turn — the reactive, adversarial
|
|
@@ -4305,7 +4253,7 @@ interface DecideNextUserTurnOpts {
|
|
|
4305
4253
|
* workspace machinery. Returns the next user message, or the literal "DONE"
|
|
4306
4254
|
* when the simulated professional would sign off.
|
|
4307
4255
|
*/
|
|
4308
|
-
declare function decideNextUserTurn(
|
|
4256
|
+
declare function decideNextUserTurn(chat: ChatClient, opts: DecideNextUserTurnOpts): Promise<string>;
|
|
4309
4257
|
|
|
4310
4258
|
interface ExecutorConfig {
|
|
4311
4259
|
/** System prompt for the agent under test */
|
|
@@ -4318,8 +4266,6 @@ interface ExecutorConfig {
|
|
|
4318
4266
|
costLedger?: CostLedgerHandle;
|
|
4319
4267
|
costPhase?: string;
|
|
4320
4268
|
costTags?: Record<string, string>;
|
|
4321
|
-
/** Exact maximum provider attempts configured on the supplied TCloud client. */
|
|
4322
|
-
tcloudMaximumAttempts?: number;
|
|
4323
4269
|
signal?: AbortSignal;
|
|
4324
4270
|
/** Regex patterns for detecting tool/API calls in responses */
|
|
4325
4271
|
toolCallPatterns?: RegExp[];
|
|
@@ -4338,11 +4284,11 @@ interface ExecutorConfig {
|
|
|
4338
4284
|
sleep?: (ms: number) => Promise<void>;
|
|
4339
4285
|
}
|
|
4340
4286
|
/**
|
|
4341
|
-
* Execute a scenario against an
|
|
4287
|
+
* Execute a scenario against an injected chat client.
|
|
4342
4288
|
*
|
|
4343
4289
|
* Runs multi-turn conversation, extracts artifacts, runs judges.
|
|
4344
4290
|
*/
|
|
4345
|
-
declare function executeScenario(
|
|
4291
|
+
declare function executeScenario(chat: ChatClient, scenario: Scenario$1, config: ExecutorConfig): Promise<ScenarioResult>;
|
|
4346
4292
|
|
|
4347
4293
|
/**
|
|
4348
4294
|
* Backend-integrity guard: distinguish "agent failed" from "eval ran against
|
|
@@ -4602,76 +4548,15 @@ type JudgeParseErrorOptions = {
|
|
|
4602
4548
|
llmCall?: LlmCallMetadata;
|
|
4603
4549
|
};
|
|
4604
4550
|
/**
|
|
4605
|
-
* A judge's
|
|
4606
|
-
*
|
|
4607
|
-
* row — a synthetic zero is indistinguishable from a real low score
|
|
4608
|
-
* downstream. Carries the raw response for forensics. Callers (executor,
|
|
4609
|
-
* ensemble wrappers) catch this per-judge and record a failed judge.
|
|
4551
|
+
* A judge's model response could not be parsed into scored dimensions.
|
|
4552
|
+
* Carries the raw response and paid-call metadata without fabricating a score.
|
|
4610
4553
|
*/
|
|
4611
4554
|
declare class JudgeParseError extends JudgeError {
|
|
4612
|
-
/** Name of the judge whose response failed to parse. */
|
|
4613
4555
|
readonly judgeName: string;
|
|
4614
|
-
/** The raw (truncated) model response that failed to parse. */
|
|
4615
4556
|
readonly raw: string;
|
|
4616
|
-
/** Paid-call metadata remains available even when the verdict is unusable. */
|
|
4617
4557
|
readonly llmCall?: LlmCallMetadata;
|
|
4618
4558
|
constructor(judgeName: string, raw: string, options?: JudgeParseErrorOptions);
|
|
4619
4559
|
}
|
|
4620
|
-
/**
|
|
4621
|
-
* Create a domain expert judge with a configurable domain.
|
|
4622
|
-
*
|
|
4623
|
-
* The judge evaluates professional accuracy and depth.
|
|
4624
|
-
*
|
|
4625
|
-
* @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
|
|
4626
|
-
* Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
|
|
4627
|
-
* multi-model panels via `ensembleJudge` (src/judge-panel.ts) — which are
|
|
4628
|
-
* pluggable, fail-loud, and drive the campaign/improvement-loop engines.
|
|
4629
|
-
*/
|
|
4630
|
-
declare function createDomainExpertJudge(domain: string): JudgeFn;
|
|
4631
|
-
/**
|
|
4632
|
-
* Code execution judge — evaluates whether code blocks are valid and runnable.
|
|
4633
|
-
*
|
|
4634
|
-
* @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
|
|
4635
|
-
* Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
|
|
4636
|
-
* multi-model panels via `ensembleJudge` (src/judge-panel.ts).
|
|
4637
|
-
*/
|
|
4638
|
-
declare const codeExecutionJudge: JudgeFn;
|
|
4639
|
-
/**
|
|
4640
|
-
* Coherence judge — evaluates multi-turn consistency and progression.
|
|
4641
|
-
*
|
|
4642
|
-
* @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
|
|
4643
|
-
* Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
|
|
4644
|
-
* multi-model panels via `ensembleJudge` (src/judge-panel.ts).
|
|
4645
|
-
*/
|
|
4646
|
-
declare const coherenceJudge: JudgeFn;
|
|
4647
|
-
/**
|
|
4648
|
-
* Adversarial judge — red-teams agent responses.
|
|
4649
|
-
*
|
|
4650
|
-
* @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
|
|
4651
|
-
* Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
|
|
4652
|
-
* multi-model panels via `ensembleJudge` (src/judge-panel.ts).
|
|
4653
|
-
*/
|
|
4654
|
-
declare const adversarialJudge: JudgeFn;
|
|
4655
|
-
/**
|
|
4656
|
-
* Create a custom judge with a fully custom prompt.
|
|
4657
|
-
*
|
|
4658
|
-
* @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
|
|
4659
|
-
* Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
|
|
4660
|
-
* multi-model panels via `ensembleJudge` (src/judge-panel.ts).
|
|
4661
|
-
*/
|
|
4662
|
-
declare function createCustomJudge(name: string, systemPrompt: string, opts?: {
|
|
4663
|
-
model?: string;
|
|
4664
|
-
temperature?: number;
|
|
4665
|
-
maxTokens?: number;
|
|
4666
|
-
}): JudgeFn;
|
|
4667
|
-
/**
|
|
4668
|
-
* Default judge set (domain must be provided for domain expert)
|
|
4669
|
-
*
|
|
4670
|
-
* @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
|
|
4671
|
-
* Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
|
|
4672
|
-
* multi-model panels via `ensembleJudge` (src/judge-panel.ts).
|
|
4673
|
-
*/
|
|
4674
|
-
declare function defaultJudges(domain: string): JudgeFn[];
|
|
4675
4560
|
|
|
4676
4561
|
type KnowledgeRequirementCategory = 'user_specific' | 'company_specific' | 'domain_specific' | 'codebase_specific' | 'market_specific' | 'regulatory' | 'tool_api' | 'credential_or_secret' | 'runtime_environment' | 'preference' | 'historical_context';
|
|
4677
4562
|
type KnowledgeAcquisitionMode = 'ask_user' | 'search_web' | 'query_connector' | 'inspect_repo' | 'run_command' | 'infer_low_confidence' | 'not_available';
|
|
@@ -4872,6 +4757,13 @@ interface GateEvidence {
|
|
|
4872
4757
|
/** Median per-task USD cost across the baseline runs, for
|
|
4873
4758
|
* symmetric reporting. */
|
|
4874
4759
|
medianBaselineCost: number | null;
|
|
4760
|
+
/**
|
|
4761
|
+
* Runs (candidate + baseline) dropped before pairing because the
|
|
4762
|
+
* authenticity gate flagged them as gamed. Surfaced rather than silent: a
|
|
4763
|
+
* promotion decision computed over a shrunken pool has to say by how much,
|
|
4764
|
+
* and a nonzero count here is itself the finding.
|
|
4765
|
+
*/
|
|
4766
|
+
realnessGatedRuns: number;
|
|
4875
4767
|
}
|
|
4876
4768
|
interface GateDecision$1 {
|
|
4877
4769
|
/** Final promote/no-promote verdict. */
|
|
@@ -5028,6 +4920,13 @@ interface ReleaseConfidenceMetrics {
|
|
|
5028
4920
|
domainCounts: Record<string, number>;
|
|
5029
4921
|
failureClassCounts: Partial<Record<FailureClass, number>>;
|
|
5030
4922
|
responsibleSurfaceCounts: Record<string, number>;
|
|
4923
|
+
/**
|
|
4924
|
+
* Runs excluded from `passRate` because the authenticity gate flagged them as
|
|
4925
|
+
* gamed. Surfaced, never silent: a release whose pass rate is computed over a
|
|
4926
|
+
* shrunken denominator has to say by how much, or the exclusion is just a
|
|
4927
|
+
* different way of hiding the same runs.
|
|
4928
|
+
*/
|
|
4929
|
+
realnessGatedRuns: number;
|
|
5031
4930
|
}
|
|
5032
4931
|
interface ReleaseConfidenceScorecard {
|
|
5033
4932
|
target: string;
|
|
@@ -5302,9 +5201,8 @@ interface ContinuousCalibrationResult extends CalibrationResult {
|
|
|
5302
5201
|
};
|
|
5303
5202
|
}
|
|
5304
5203
|
/**
|
|
5305
|
-
*
|
|
5306
|
-
*
|
|
5307
|
-
* are preserved unchanged so existing callers continue to work.
|
|
5204
|
+
* Extends `calibrateJudge` with continuous-value agreement metrics while
|
|
5205
|
+
* retaining its base calibration summary.
|
|
5308
5206
|
*/
|
|
5309
5207
|
declare function calibrateJudgeContinuous(golden: GoldenItem[], candidate: CandidateScore[], opts?: ContinuousAgreementOptions): ContinuousCalibrationResult;
|
|
5310
5208
|
|
|
@@ -6145,6 +6043,28 @@ declare function printDriverSummary(results: DriverResult[]): void;
|
|
|
6145
6043
|
* `outcome.reward` is THE single scalar (null = no verdict exists — a
|
|
6146
6044
|
* labeled gap, never 0). `outcome.realness_gated` is the anti-Goodhart
|
|
6147
6045
|
* flag: a gated line must never export as a positive training example.
|
|
6046
|
+
*
|
|
6047
|
+
* That last sentence is enforced here, by `validateRolloutLine`, not merely
|
|
6048
|
+
* documented. Validating `reward` and `realness_gated` independently — each a
|
|
6049
|
+
* well-typed field, their COMBINATION unchecked — is what let a line claiming
|
|
6050
|
+
* `{reward: 0.95, realness_gated: true}` validate clean and walk into every
|
|
6051
|
+
* training export. The relationship between the two IS the invariant, so it is
|
|
6052
|
+
* checked where every other structural claim about a line is checked.
|
|
6053
|
+
*
|
|
6054
|
+
* The invariant is about the OUTCOME, not about one field of it. Zeroing
|
|
6055
|
+
* `reward` while `outcome.metrics` still carried the per-layer scores that
|
|
6056
|
+
* reward was computed from exported the gamed signal anyway, in the dict the
|
|
6057
|
+
* verifiers format reads as its per-rubric scores. So `gateGamedOutcome`
|
|
6058
|
+
* transforms the whole outcome once, at `assertMinted` — the funnel every
|
|
6059
|
+
* minted line passes — and the reward-bearing components are relocated to
|
|
6060
|
+
* `provenance.gated_evidence`, which no exporter projects.
|
|
6061
|
+
*
|
|
6062
|
+
* WHICH checks each door applies is not decided in this file. `./gate-checks`
|
|
6063
|
+
* owns the canonical list and the total per-entry-point policy; the three doors
|
|
6064
|
+
* below (`validateRolloutLine`, `assertRewardGate`, `assertMinted`) each call
|
|
6065
|
+
* `gateErrors` with their declared policy, so a check added to that list applies
|
|
6066
|
+
* here without anyone editing this file, and a check deliberately skipped has to
|
|
6067
|
+
* name itself there.
|
|
6148
6068
|
*/
|
|
6149
6069
|
declare const ROLLOUT_SCHEMA = "tangle.rollout.v1";
|
|
6150
6070
|
/** `agent` = a solo evaluation run (no multi-agent topology). */
|
|
@@ -6173,6 +6093,14 @@ interface ChatMessage {
|
|
|
6173
6093
|
/** Required on role:"tool" — the ChatToolCall this result answers. */
|
|
6174
6094
|
tool_call_id?: string;
|
|
6175
6095
|
name?: string;
|
|
6096
|
+
/**
|
|
6097
|
+
* Harbor ATIF `is_copied_context` (RFC 0001 rule 7): this turn was COPIED IN
|
|
6098
|
+
* from another trajectory's context, not produced by the agent on this line.
|
|
6099
|
+
* The RFC makes excluding it from SFT a MUST, and `toSftRows` does — training
|
|
6100
|
+
* on it teaches the model to author text it never authored, and credits this
|
|
6101
|
+
* run for another one's work. Absent = false (authored here).
|
|
6102
|
+
*/
|
|
6103
|
+
is_copied_context?: boolean;
|
|
6176
6104
|
}
|
|
6177
6105
|
interface ToolDef {
|
|
6178
6106
|
type: 'function';
|
|
@@ -6196,6 +6124,21 @@ interface RolloutStep {
|
|
|
6196
6124
|
output?: string;
|
|
6197
6125
|
status?: 'ok' | 'error';
|
|
6198
6126
|
durationMs?: number;
|
|
6127
|
+
/**
|
|
6128
|
+
* LLM inferences this span represents. 0 = deterministic dispatch with no
|
|
6129
|
+
* model call — distinct from absent, which means the producer did not track it.
|
|
6130
|
+
*/
|
|
6131
|
+
llm_call_count?: number;
|
|
6132
|
+
/** Exact prompt tokenization. Removes the ambiguity of re-tokenizing text at train time. */
|
|
6133
|
+
prompt_token_ids?: number[];
|
|
6134
|
+
/** Exact completion tokenization; aligns index-wise with `logprobs`. */
|
|
6135
|
+
completion_token_ids?: number[];
|
|
6136
|
+
/**
|
|
6137
|
+
* Per-completion-token log probabilities under the sampling policy. Required
|
|
6138
|
+
* for off-policy correction (importance weighting) when the rollout was
|
|
6139
|
+
* generated by a policy other than the one being trained.
|
|
6140
|
+
*/
|
|
6141
|
+
logprobs?: number[];
|
|
6199
6142
|
}
|
|
6200
6143
|
interface RolloutTask {
|
|
6201
6144
|
/** Benchmark/suite id (e.g. "swe-bench-verified") or the experiment id. */
|
|
@@ -6240,11 +6183,39 @@ interface RolloutOutcome {
|
|
|
6240
6183
|
is_truncated: boolean;
|
|
6241
6184
|
error: string | null;
|
|
6242
6185
|
/**
|
|
6243
|
-
* Anti-Goodhart flag from `RunRecord.outcome.realness.gated`: the run
|
|
6244
|
-
*
|
|
6245
|
-
* line never qualifies for
|
|
6186
|
+
* Anti-Goodhart flag from `RunRecord.outcome.realness.gated`: the run faked
|
|
6187
|
+
* its success signal. `true` requires `reward` to be 0 or null — the
|
|
6188
|
+
* validator rejects the line otherwise — and the line never qualifies for
|
|
6189
|
+
* SFT. Required on the wire: a line that does not state the flag does not
|
|
6190
|
+
* validate, so no producer can dodge the gate by omitting it.
|
|
6191
|
+
*
|
|
6192
|
+
* `true` ALSO requires `metrics` to be empty and `verdict` to be null: the
|
|
6193
|
+
* numbers the reward was computed from are relocated to
|
|
6194
|
+
* `provenance.gated_evidence` by `gateGamedOutcome`. See that function for
|
|
6195
|
+
* why zeroing the scalar alone was not enough.
|
|
6246
6196
|
*/
|
|
6247
6197
|
realness_gated: boolean;
|
|
6198
|
+
/**
|
|
6199
|
+
* Whether an authenticity SCREEN ever RAN on this reward — a different claim
|
|
6200
|
+
* from `realness_gated`, which is the screen's VERDICT.
|
|
6201
|
+
*
|
|
6202
|
+
* `realness_gated: false` reads as "we looked and nothing fired". A producer
|
|
6203
|
+
* with no screen at all was emitting exactly that, so a never-screened reward
|
|
6204
|
+
* was indistinguishable on the wire from a screened-clean one, and the whole
|
|
6205
|
+
* anti-Goodhart apparatus silently treated the first as the second. The two
|
|
6206
|
+
* claims are now separable:
|
|
6207
|
+
*
|
|
6208
|
+
* - `true` — a screen ran; `realness_gated` is its verdict.
|
|
6209
|
+
* - `false` — the producer declares it HAS no screen (`unscreenedRewardFields`).
|
|
6210
|
+
* `assertMinted` REFUSES such a line when its reward is above
|
|
6211
|
+
* zero: an unscreened positive reward is precisely the signal
|
|
6212
|
+
* the gate exists to qualify, and nothing has qualified it.
|
|
6213
|
+
* - absent — not stated. Pre-unification ledgers land here, as does a
|
|
6214
|
+
* `RunRecord` carrying no `outcome.realness` at all. Absent is
|
|
6215
|
+
* read as "unknown", never as `false` (which would refuse most
|
|
6216
|
+
* of the existing corpus) and never as `true`.
|
|
6217
|
+
*/
|
|
6218
|
+
realness_screened?: boolean;
|
|
6248
6219
|
}
|
|
6249
6220
|
interface RolloutCostBlock {
|
|
6250
6221
|
usd: number | null;
|
|
@@ -6254,6 +6225,11 @@ interface RolloutCostBlock {
|
|
|
6254
6225
|
cache_read: number | null;
|
|
6255
6226
|
cache_write: number | null;
|
|
6256
6227
|
wall_s: number | null;
|
|
6228
|
+
/**
|
|
6229
|
+
* Total LLM inferences across the invocation (ATIF `llm_call_count`,
|
|
6230
|
+
* aggregated). Optional and additive: absent = not tracked, never 0.
|
|
6231
|
+
*/
|
|
6232
|
+
llm_call_count?: number | null;
|
|
6257
6233
|
}
|
|
6258
6234
|
interface RolloutArtifacts {
|
|
6259
6235
|
patch_path: string | null;
|
|
@@ -6261,11 +6237,43 @@ interface RolloutArtifacts {
|
|
|
6261
6237
|
/** Source-of-truth transcript pointer (session id / jsonl path) for audit. */
|
|
6262
6238
|
transcript_ref: string | null;
|
|
6263
6239
|
}
|
|
6240
|
+
/**
|
|
6241
|
+
* The reward-bearing half of a GATED line's outcome, moved off `outcome` and
|
|
6242
|
+
* parked here verbatim. Diagnostics, never training input — see
|
|
6243
|
+
* `gateGamedOutcome`.
|
|
6244
|
+
*/
|
|
6245
|
+
interface GatedEvidence {
|
|
6246
|
+
/** `outcome.metrics` exactly as the producer measured it. */
|
|
6247
|
+
metrics?: Record<string, unknown>;
|
|
6248
|
+
/** `outcome.verdict` verbatim — the judge record that claimed the success. */
|
|
6249
|
+
verdict?: unknown;
|
|
6250
|
+
/**
|
|
6251
|
+
* The per-step fields `tangle.rollout.v1` does not declare, parked here when
|
|
6252
|
+
* the gate projected `steps[]` down to the schema's own key set.
|
|
6253
|
+
*
|
|
6254
|
+
* A per-step reward is training signal exactly like the scalar, and `steps`
|
|
6255
|
+
* rides through `toRewardRows` verbatim — so a gated line was shipping its
|
|
6256
|
+
* step-level credit assignment at full value beside a `reward` of 0.
|
|
6257
|
+
*/
|
|
6258
|
+
steps?: unknown;
|
|
6259
|
+
}
|
|
6264
6260
|
interface RolloutProvenance {
|
|
6265
6261
|
captured_at: string;
|
|
6266
6262
|
capture: RolloutCapture;
|
|
6267
|
-
/**
|
|
6263
|
+
/**
|
|
6264
|
+
* Why this line is incomplete. Required when `messages` is empty (the
|
|
6265
|
+
* transcript could not be recovered); also set by interchange importers to
|
|
6266
|
+
* name a MISSING LABEL — an imported trajectory carries no verdict, so
|
|
6267
|
+
* `outcome.reward` is null and this says why.
|
|
6268
|
+
*/
|
|
6268
6269
|
gap?: string;
|
|
6270
|
+
/**
|
|
6271
|
+
* Present only on a realness-gated line: the outcome fields the gate
|
|
6272
|
+
* relocated, kept so an auditor can still see WHY the run was gated and what
|
|
6273
|
+
* it claimed. Deliberately OUTSIDE `outcome`, because every training exporter
|
|
6274
|
+
* reads `outcome` and none reads `provenance`.
|
|
6275
|
+
*/
|
|
6276
|
+
gated_evidence?: GatedEvidence;
|
|
6269
6277
|
}
|
|
6270
6278
|
interface RolloutLine {
|
|
6271
6279
|
schema: typeof ROLLOUT_SCHEMA;
|
|
@@ -6297,6 +6305,82 @@ interface RolloutLine {
|
|
|
6297
6305
|
declare function validateRolloutLine(value: unknown): string[];
|
|
6298
6306
|
declare function assertRolloutLine(value: unknown, context?: string): asserts value is RolloutLine;
|
|
6299
6307
|
declare function isRolloutLine(value: unknown): value is RolloutLine;
|
|
6308
|
+
/**
|
|
6309
|
+
* Phantom property. `declare const` means it exists only in the type system:
|
|
6310
|
+
* nothing is written at runtime, so a branded line still serializes to exactly
|
|
6311
|
+
* the same JSON as a plain one.
|
|
6312
|
+
*/
|
|
6313
|
+
declare const MINTED_ROLLOUT: unique symbol;
|
|
6314
|
+
/**
|
|
6315
|
+
* A minted outcome states the gate verdict — it is not allowed to stay silent —
|
|
6316
|
+
* and, when that verdict is `true`, carries nothing else the reward was derived
|
|
6317
|
+
* from (`gateGamedOutcome` has run).
|
|
6318
|
+
*/
|
|
6319
|
+
interface MintedRolloutOutcome extends RolloutOutcome {
|
|
6320
|
+
realness_gated: boolean;
|
|
6321
|
+
}
|
|
6322
|
+
/**
|
|
6323
|
+
* A `RolloutLine` whose reward has been checked against the anti-Goodhart
|
|
6324
|
+
* invariant. The type every training-data exporter takes.
|
|
6325
|
+
*
|
|
6326
|
+
* Why a brand and not just the interface: `RolloutLine` is structural, so any
|
|
6327
|
+
* hand-built object literal of the right shape IS one — which is how a line
|
|
6328
|
+
* declaring `{reward: 0.95, realness_gated: true}` reached the exporters
|
|
6329
|
+
* despite them "only accepting a minted line". The phantom symbol makes the
|
|
6330
|
+
* type nominal: it cannot be produced by writing an object literal, only by
|
|
6331
|
+
* `mintRolloutRows` (which applies the gate), `readRolloutLedger` (which
|
|
6332
|
+
* validates every line off disk), or an explicit, greppable `assertMinted`.
|
|
6333
|
+
*
|
|
6334
|
+
* Belt and braces on purpose. The brand closes first-party call sites at
|
|
6335
|
+
* COMPILE time; `validateRolloutLine` closes data arriving at RUNTIME (ledger
|
|
6336
|
+
* files, foreign imports, JSON from another process) where types are absent.
|
|
6337
|
+
* Neither alone is enough.
|
|
6338
|
+
*
|
|
6339
|
+
* Assignable to `RolloutLine` in one direction only: readers, analysis, and
|
|
6340
|
+
* the ledger writer keep taking the plain type.
|
|
6341
|
+
*/
|
|
6342
|
+
type MintedRolloutLine = Omit<RolloutLine, 'outcome'> & {
|
|
6343
|
+
readonly [MINTED_ROLLOUT]: true;
|
|
6344
|
+
outcome: MintedRolloutOutcome;
|
|
6345
|
+
};
|
|
6346
|
+
/**
|
|
6347
|
+
* Promote a line to the type the training exporters accept, applying the
|
|
6348
|
+
* anti-Goodhart gate to the WHOLE outcome on the way through. THE escape hatch
|
|
6349
|
+
* — grep `assertMinted` to enumerate every place a line enters the training
|
|
6350
|
+
* path without coming from mint or a ledger.
|
|
6351
|
+
*
|
|
6352
|
+
* The gate runs HERE, once, rather than at each producer, because this is the
|
|
6353
|
+
* single funnel every minted line passes: `mintRolloutRows` calls it,
|
|
6354
|
+
* `readRolloutLedger` calls it per line off disk, `scrubLines` calls it on the
|
|
6355
|
+
* way out of a release, and a hand-built line has no other door. One
|
|
6356
|
+
* transformation at the funnel means an already-published ledger holding a
|
|
6357
|
+
* gated line with populated `metrics` is RE-GATED when it is read, instead of
|
|
6358
|
+
* being rejected (which would make every such artifact unreadable) or trusted
|
|
6359
|
+
* (which is the leak). Three steps, in this order:
|
|
6360
|
+
*
|
|
6361
|
+
* 1. VALIDATE the schema.
|
|
6362
|
+
* 2. REFUSE every check `GATE_POLICIES.assertMinted` marks `enforce` — today
|
|
6363
|
+
* the reward relationship (which stays a REJECTION: a caller claiming
|
|
6364
|
+
* `{reward: 0.95, realness_gated: true}` is a producer defect and must fail
|
|
6365
|
+
* loudly, since laundering it into `reward: 0` here would hide the
|
|
6366
|
+
* producer) and a positive reward the producer declared it never screened.
|
|
6367
|
+
* 3. TRANSFORM the one check that policy marks `repair` — relocate the
|
|
6368
|
+
* reward's components off `outcome` (`gateGamedOutcome`), so no exporter
|
|
6369
|
+
* can leak them whichever field it reads.
|
|
6370
|
+
*
|
|
6371
|
+
* Step 2 enumerates nothing by hand: a check added to `GATE_CHECKS` is enforced
|
|
6372
|
+
* here the moment its disposition in that policy says so.
|
|
6373
|
+
*
|
|
6374
|
+
* Also normalizes the optional wire flag to an explicit boolean.
|
|
6375
|
+
* `realness_gated` is absent on pre-unification ledgers and absent means "not
|
|
6376
|
+
* flagged" per the schema, so filling it in states a claim the line was already
|
|
6377
|
+
* making, and makes the flag readable on every published row instead of most of
|
|
6378
|
+
* them. `realness_screened` is NOT filled in: absent means "unknown", and
|
|
6379
|
+
* inventing either value there would be the same overclaim this round removed.
|
|
6380
|
+
*/
|
|
6381
|
+
declare function assertMinted(value: unknown, context?: string): MintedRolloutLine;
|
|
6382
|
+
/** `assertMinted` over a batch, naming the offending index in the error. */
|
|
6383
|
+
declare function assertMintedLines(values: readonly unknown[], context?: string): MintedRolloutLine[];
|
|
6300
6384
|
|
|
6301
6385
|
/**
|
|
6302
6386
|
* Pure exporters over `tangle.rollout.v1` lines → the training-data shapes
|
|
@@ -6309,8 +6393,40 @@ declare function isRolloutLine(value: unknown): value is RolloutLine;
|
|
|
6309
6393
|
* All exporters are pure functions of the lines — filtering (never train on
|
|
6310
6394
|
* holdout, reward thresholds, the realness gate) happens HERE, on inline
|
|
6311
6395
|
* labels, no joins.
|
|
6396
|
+
*
|
|
6397
|
+
* Every exporter takes `MintedRolloutLine[]`, not `RolloutLine[]`: the reward
|
|
6398
|
+
* on a minted line has been checked against the anti-Goodhart invariant, and
|
|
6399
|
+
* the brand is what stops a hand-built object literal claiming a positive
|
|
6400
|
+
* reward on a gamed run from being handed to an exporter that copies it
|
|
6401
|
+
* verbatim into training data.
|
|
6312
6402
|
*/
|
|
6313
6403
|
|
|
6404
|
+
/**
|
|
6405
|
+
* The gate's two claims, which travel TOGETHER on every emitted row.
|
|
6406
|
+
*
|
|
6407
|
+
* `realness_gated` alone is ambiguous, and the ambiguity is exploitable:
|
|
6408
|
+
* `false` reads as "we screened it and nothing fired", so a producer that has no
|
|
6409
|
+
* screen at all emitted rows indistinguishable from screened-clean ones, and
|
|
6410
|
+
* every consumer of the published dataset read them as clean. The second field
|
|
6411
|
+
* is what separates the two claims, and it only removes the ambiguity if it
|
|
6412
|
+
* reaches the WIRE — for a round it existed on `RolloutOutcome` and on no
|
|
6413
|
+
* exported row shape at all, which left the published rows exactly as ambiguous
|
|
6414
|
+
* as before.
|
|
6415
|
+
*
|
|
6416
|
+
* So there is one helper and every row shape spreads it. A row that states one
|
|
6417
|
+
* claim without the other is not constructible by copying the pattern, and
|
|
6418
|
+
* `exporters.test.ts` walks every emitted shape to prove none does.
|
|
6419
|
+
*/
|
|
6420
|
+
interface RealnessLabels {
|
|
6421
|
+
/** The screen's VERDICT: the run faked its success signal. */
|
|
6422
|
+
realness_gated: boolean;
|
|
6423
|
+
/**
|
|
6424
|
+
* Whether a screen RAN at all. `true` = it ran, so `realness_gated` is its
|
|
6425
|
+
* verdict. `false` = the producer declares it has none. `null` = not stated
|
|
6426
|
+
* (pre-unification producers), which is "unknown" and never "clean".
|
|
6427
|
+
*/
|
|
6428
|
+
realness_screened: boolean | null;
|
|
6429
|
+
}
|
|
6314
6430
|
interface TrainingExportOptions {
|
|
6315
6431
|
/** Include held-out evaluation data in training output. Default false. */
|
|
6316
6432
|
allowHeldOutTrainingData?: boolean;
|
|
@@ -6326,15 +6442,24 @@ interface SftRow {
|
|
|
6326
6442
|
candidate_id: string | null;
|
|
6327
6443
|
instance_id: string;
|
|
6328
6444
|
reward: number;
|
|
6329
|
-
};
|
|
6445
|
+
} & RealnessLabels;
|
|
6330
6446
|
}
|
|
6331
6447
|
/**
|
|
6332
6448
|
* Supervised fine-tune rows: the completed conversation of each qualifying
|
|
6333
6449
|
* line. Fail-closed filters: trainable split only (never holdout/canary),
|
|
6334
|
-
*
|
|
6335
|
-
* no trainable content
|
|
6336
|
-
|
|
6337
|
-
|
|
6450
|
+
* reward strictly above `minimumQualityExclusive` (default 0), realness-gated
|
|
6451
|
+
* lines never qualify, gap lines carry no trainable content, and
|
|
6452
|
+
* copied-context turns are dropped from the transcript (Harbor ATIF RFC 0001
|
|
6453
|
+
* rule 7 — see `ChatMessage.is_copied_context`).
|
|
6454
|
+
*
|
|
6455
|
+
* `realness_gated` is therefore always `false` on an emitted row. It is carried
|
|
6456
|
+
* anyway: an SFT row is a pure imitation target, so the row states its realness
|
|
6457
|
+
* claims instead of making the reader know the format's policy, and carrying
|
|
6458
|
+
* both flags on all four shapes is what lets the release accounting measure
|
|
6459
|
+
* every config with one rule rather than skipping the one whose row shape
|
|
6460
|
+
* happened to omit the field.
|
|
6461
|
+
*/
|
|
6462
|
+
declare function toSftRows(lines: MintedRolloutLine[], options?: SftExportOptions): SftRow[];
|
|
6338
6463
|
interface RewardRow {
|
|
6339
6464
|
/** First user turn — the task prompt. */
|
|
6340
6465
|
prompt: string;
|
|
@@ -6346,14 +6471,307 @@ interface RewardRow {
|
|
|
6346
6471
|
candidate_id: string | null;
|
|
6347
6472
|
instance_id: string;
|
|
6348
6473
|
split: RolloutSplit;
|
|
6349
|
-
};
|
|
6474
|
+
} & RealnessLabels;
|
|
6350
6475
|
}
|
|
6351
6476
|
/**
|
|
6352
6477
|
* Reward-labeled rows for completed, positive-quality training runs.
|
|
6353
6478
|
*/
|
|
6354
|
-
declare function toRewardRows(lines:
|
|
6479
|
+
declare function toRewardRows(lines: MintedRolloutLine[], options?: TrainingExportOptions): RewardRow[];
|
|
6355
6480
|
declare function toJsonl(rows: ReadonlyArray<unknown>): string;
|
|
6356
6481
|
|
|
6482
|
+
/**
|
|
6483
|
+
* Harbor ATIF-v1.7 interchange — `tangle.rollout.v1` ⇄ Agent Trajectory
|
|
6484
|
+
* Interchange Format.
|
|
6485
|
+
*
|
|
6486
|
+
* ATIF is the portability format (spec:
|
|
6487
|
+
* https://www.harborframework.com/docs/agents/trajectory-format, normative
|
|
6488
|
+
* RFC: harbor-framework/harbor `rfcs/0001-trajectory-format.md`). It sits
|
|
6489
|
+
* BELOW the waist of the rollout hourglass in both directions — export reads
|
|
6490
|
+
* `RolloutLine[]`, import writes `RolloutLine[]` — and it is never a source
|
|
6491
|
+
* of training labels:
|
|
6492
|
+
*
|
|
6493
|
+
* ATIF models NO reward, NO judge verdict, NO task/split coordinates.
|
|
6494
|
+
*
|
|
6495
|
+
* Consequences, both deliberate:
|
|
6496
|
+
* - EXPORT drops `outcome.reward`, `outcome.reward_source` and
|
|
6497
|
+
* `outcome.verdict` entirely. They are not smuggled into `extra`: a
|
|
6498
|
+
* third-party reading our ATIF file must not be able to mistake an
|
|
6499
|
+
* agent-eval judge score for something ATIF sanctioned.
|
|
6500
|
+
* - IMPORT therefore mints UNLABELED lines: `reward: null` (the existing
|
|
6501
|
+
* "null reward is a labeled gap, never 0" semantics), `verdict: null`,
|
|
6502
|
+
* and a `provenance.gap` naming the missing label. An imported
|
|
6503
|
+
* trajectory is not a training example until a judge scores it.
|
|
6504
|
+
*
|
|
6505
|
+
* Everything else we own that ATIF has no field for travels in a namespaced
|
|
6506
|
+
* escrow at `extra.tangle.*`, so our own round-trip is exact while a foreign
|
|
6507
|
+
* reader can ignore it. Fields that neither ATIF nor the escrow can carry
|
|
6508
|
+
* come back explicitly null / fail-closed, never invented.
|
|
6509
|
+
*
|
|
6510
|
+
* THE ESCROW IS NAMESPACED, NOT AUTHENTICATED. Anyone can write
|
|
6511
|
+
* `extra.tangle.*` into a file. So the escrow may restore what a value IS, but
|
|
6512
|
+
* never what a line is ALLOWED to do: `task.split` is forced to `holdout` on
|
|
6513
|
+
* every import regardless of what the document claims, and promoting an
|
|
6514
|
+
* imported trajectory to a trainable split is an explicit, greppable act
|
|
6515
|
+
* (`relabelImportedSplit`) rather than a property of the file. The document
|
|
6516
|
+
* keeps its claim — the claim just is not authority.
|
|
6517
|
+
*
|
|
6518
|
+
* Multi-agent shape differs on purpose. ATIF EMBEDS children in
|
|
6519
|
+
* `subagent_trajectories`; we keep a flat ledger with a normalized
|
|
6520
|
+
* `parent_rollout_id` edge. Export assembles the tree, import flattens it.
|
|
6521
|
+
* `session_id` is RUN-scoped in ATIF, so it carries `run_id` — the coordinate
|
|
6522
|
+
* that is shared by every invocation of one run — not `rollout_id`, which
|
|
6523
|
+
* identifies a single invocation and would split one run across session ids.
|
|
6524
|
+
*
|
|
6525
|
+
* ROUND-TRIPPING IS IDEMPOTENT: `import(export(import(export(x))))` is
|
|
6526
|
+
* byte-identical to `import(export(x))`. Import composes `provenance.gap` as a
|
|
6527
|
+
* de-duplicated ordered set rather than appending, and it emits every
|
|
6528
|
+
* `ChatMessage` with keys in the canonical schema order (role, content,
|
|
6529
|
+
* reasoning_content, tool_calls, tool_call_id, name, is_copied_context), so a
|
|
6530
|
+
* ledger hashed on serialized bytes sees no diff across further passes. The
|
|
6531
|
+
* FIRST import may re-order a producer's keys — that is the canonicalization.
|
|
6532
|
+
*
|
|
6533
|
+
* NOT building a Letta converter. Letta's trajectory-v1 is a strict subset of
|
|
6534
|
+
* what we need from ATIF here — no per-step or aggregate cost, no
|
|
6535
|
+
* multi-agent/subagent structure, no token-id or logprob channel — so a Letta
|
|
6536
|
+
* sink would carry less than this one and add a second format to keep
|
|
6537
|
+
* correct. Decision recorded in docs/rollout.md; do not re-litigate without a
|
|
6538
|
+
* concrete consumer that reads Letta and cannot read ATIF.
|
|
6539
|
+
*/
|
|
6540
|
+
|
|
6541
|
+
declare const ATIF_SCHEMA_VERSION = "ATIF-v1.7";
|
|
6542
|
+
/** Gap note on every imported line — ATIF carries no verdict, so nothing is scored. */
|
|
6543
|
+
declare const HARBOR_IMPORT_GAP = "imported from Harbor ATIF; no verdict";
|
|
6544
|
+
type HarborStepSource = 'system' | 'user' | 'agent';
|
|
6545
|
+
interface HarborImageSource {
|
|
6546
|
+
media_type: string;
|
|
6547
|
+
path: string;
|
|
6548
|
+
}
|
|
6549
|
+
interface HarborContentPart {
|
|
6550
|
+
type: 'text' | 'image';
|
|
6551
|
+
text?: string;
|
|
6552
|
+
source?: HarborImageSource;
|
|
6553
|
+
}
|
|
6554
|
+
interface HarborToolCall {
|
|
6555
|
+
tool_call_id: string;
|
|
6556
|
+
function_name: string;
|
|
6557
|
+
/** ATIF requires a decoded JSON object here, unlike our raw argument string. */
|
|
6558
|
+
arguments: Record<string, unknown>;
|
|
6559
|
+
extra?: Record<string, unknown>;
|
|
6560
|
+
}
|
|
6561
|
+
interface HarborSubagentTrajectoryRef {
|
|
6562
|
+
trajectory_id?: string;
|
|
6563
|
+
trajectory_path?: string;
|
|
6564
|
+
/** Informational only since v1.7 — never a resolution key. */
|
|
6565
|
+
session_id?: string;
|
|
6566
|
+
extra?: Record<string, unknown>;
|
|
6567
|
+
}
|
|
6568
|
+
interface HarborObservationResult {
|
|
6569
|
+
source_call_id?: string;
|
|
6570
|
+
content?: string | HarborContentPart[];
|
|
6571
|
+
subagent_trajectory_ref?: HarborSubagentTrajectoryRef[];
|
|
6572
|
+
extra?: Record<string, unknown>;
|
|
6573
|
+
}
|
|
6574
|
+
interface HarborObservation {
|
|
6575
|
+
results: HarborObservationResult[];
|
|
6576
|
+
}
|
|
6577
|
+
interface HarborMetrics {
|
|
6578
|
+
prompt_tokens?: number;
|
|
6579
|
+
completion_tokens?: number;
|
|
6580
|
+
cached_tokens?: number;
|
|
6581
|
+
cost_usd?: number;
|
|
6582
|
+
prompt_token_ids?: number[];
|
|
6583
|
+
completion_token_ids?: number[];
|
|
6584
|
+
logprobs?: number[];
|
|
6585
|
+
extra?: Record<string, unknown>;
|
|
6586
|
+
}
|
|
6587
|
+
interface HarborStep {
|
|
6588
|
+
/** Ordinal, sequential from 1. */
|
|
6589
|
+
step_id: number;
|
|
6590
|
+
timestamp?: string;
|
|
6591
|
+
source: HarborStepSource;
|
|
6592
|
+
model_name?: string;
|
|
6593
|
+
reasoning_effort?: string | number;
|
|
6594
|
+
message: string | HarborContentPart[];
|
|
6595
|
+
reasoning_content?: string;
|
|
6596
|
+
tool_calls?: HarborToolCall[];
|
|
6597
|
+
observation?: HarborObservation;
|
|
6598
|
+
metrics?: HarborMetrics;
|
|
6599
|
+
llm_call_count?: number;
|
|
6600
|
+
is_copied_context?: boolean;
|
|
6601
|
+
extra?: Record<string, unknown>;
|
|
6602
|
+
}
|
|
6603
|
+
interface HarborAgent {
|
|
6604
|
+
name: string;
|
|
6605
|
+
version: string;
|
|
6606
|
+
model_name?: string;
|
|
6607
|
+
/** OpenAI function-calling schema — byte-identical to our `ToolDef`. */
|
|
6608
|
+
tool_definitions?: ToolDef[];
|
|
6609
|
+
extra?: Record<string, unknown>;
|
|
6610
|
+
}
|
|
6611
|
+
interface HarborFinalMetrics {
|
|
6612
|
+
total_prompt_tokens?: number;
|
|
6613
|
+
total_completion_tokens?: number;
|
|
6614
|
+
total_cached_tokens?: number;
|
|
6615
|
+
total_cost_usd?: number;
|
|
6616
|
+
total_steps?: number;
|
|
6617
|
+
extra?: Record<string, unknown>;
|
|
6618
|
+
}
|
|
6619
|
+
interface HarborTrajectory {
|
|
6620
|
+
schema_version: string;
|
|
6621
|
+
session_id?: string;
|
|
6622
|
+
/** Required on embedded subagents; we always set it so lines stay joinable. */
|
|
6623
|
+
trajectory_id?: string;
|
|
6624
|
+
agent: HarborAgent;
|
|
6625
|
+
steps: HarborStep[];
|
|
6626
|
+
notes?: string;
|
|
6627
|
+
final_metrics?: HarborFinalMetrics;
|
|
6628
|
+
continued_trajectory_ref?: string;
|
|
6629
|
+
subagent_trajectories?: HarborTrajectory[];
|
|
6630
|
+
extra?: Record<string, unknown>;
|
|
6631
|
+
}
|
|
6632
|
+
/**
|
|
6633
|
+
* Assemble one episode's flat lines into a single ATIF trajectory tree,
|
|
6634
|
+
* linked by `parent_rollout_id`.
|
|
6635
|
+
*
|
|
6636
|
+
* Reward, verdict and split are NOT emitted (ATIF models none of them); the
|
|
6637
|
+
* split and the rest of the task coordinates survive only in `extra.tangle`.
|
|
6638
|
+
*
|
|
6639
|
+
* We deliberately do NOT synthesize an `observation.subagent_trajectory_ref`
|
|
6640
|
+
* pointing at each child: our ledger records WHICH invocation spawned a
|
|
6641
|
+
* worker, not which STEP did, and attaching the ref to a guessed step would
|
|
6642
|
+
* fabricate a causal claim. Children are embedded in `subagent_trajectories`
|
|
6643
|
+
* (each with the `trajectory_id` the spec requires) and the edge is stated in
|
|
6644
|
+
* the child's escrowed `parent_rollout_id`.
|
|
6645
|
+
*
|
|
6646
|
+
* Throws when the lines are not one tree — use `toHarborTrajectories` for a forest.
|
|
6647
|
+
*/
|
|
6648
|
+
declare function toHarborTrajectory(lines: RolloutLine[]): HarborTrajectory;
|
|
6649
|
+
/** Every independent tree in the input, one ATIF document each. */
|
|
6650
|
+
declare function toHarborTrajectories(lines: RolloutLine[]): HarborTrajectory[];
|
|
6651
|
+
interface FromHarborOptions {
|
|
6652
|
+
/** Injected clock for deterministic output when the source carries no capture time. */
|
|
6653
|
+
now?: () => Date;
|
|
6654
|
+
}
|
|
6655
|
+
/**
|
|
6656
|
+
* Flatten an ATIF trajectory tree back into `tangle.rollout.v1` lines, parent
|
|
6657
|
+
* first, each child carrying `parent_rollout_id`.
|
|
6658
|
+
*
|
|
6659
|
+
* Every line comes back UNLABELED: `reward`, `reward_source` and `verdict` are
|
|
6660
|
+
* null and `provenance.gap` says why. ATIF models no verdict, so scoring an
|
|
6661
|
+
* imported trajectory is a judge's job, not this function's. Every line lands
|
|
6662
|
+
* on `holdout` whatever the document claims — see `relabelImportedSplit`.
|
|
6663
|
+
*/
|
|
6664
|
+
declare function fromHarborTrajectory(trajectory: HarborTrajectory, options?: FromHarborOptions): RolloutLine[];
|
|
6665
|
+
/**
|
|
6666
|
+
* THE explicit door out of `holdout` for imported lines.
|
|
6667
|
+
*
|
|
6668
|
+
* Import forces `holdout` because a document's own claim about its split is not
|
|
6669
|
+
* evidence — anyone can write `extra.tangle.task.split`. Promoting a file to a
|
|
6670
|
+
* trainable split is an operator's decision about provenance they verified, so
|
|
6671
|
+
* it is a separate, greppable call: `grep relabelImportedSplit` enumerates
|
|
6672
|
+
* every place foreign data was declared trainable, which is exactly the audit
|
|
6673
|
+
* the trusted-escrow version made impossible.
|
|
6674
|
+
*
|
|
6675
|
+
* Returns plain `RolloutLine`s. They still have to pass `assertMinted` (and its
|
|
6676
|
+
* anti-Goodhart check) to reach an exporter — re-labeling a split is not
|
|
6677
|
+
* minting a reward.
|
|
6678
|
+
*/
|
|
6679
|
+
declare function relabelImportedSplit(lines: readonly RolloutLine[], split: RolloutSplit): RolloutLine[];
|
|
6680
|
+
|
|
6681
|
+
/**
|
|
6682
|
+
* The two named score derivations every consumer must choose between.
|
|
6683
|
+
*
|
|
6684
|
+
* The anti-Goodhart gate (`outcome.realness.gated`) only holds if it is
|
|
6685
|
+
* impossible to read a run's score WITHOUT deciding whether the gate applies.
|
|
6686
|
+
* A bare `outcome.holdoutScore ?? outcome.searchScore` makes that decision
|
|
6687
|
+
* invisible — and silently answers "no gate", which is the wrong default on
|
|
6688
|
+
* every path that produces training data. So the expression lives here, once,
|
|
6689
|
+
* behind two names that force the caller to state the intent:
|
|
6690
|
+
*
|
|
6691
|
+
* - `trainingScore` / `trainingReward` — GATED. Anything that becomes
|
|
6692
|
+
* training data, or a reward a trainer consumes, uses these.
|
|
6693
|
+
* - `observedScore` — RAW. Analysis, reporting, and reward-hack DETECTION
|
|
6694
|
+
* need the ungated number; that is how a gamed run is visible at all.
|
|
6695
|
+
*
|
|
6696
|
+
* A leaf module on purpose: it imports only the `RunRecord` type, so gate and
|
|
6697
|
+
* reporting code can depend on it without pulling in the trace store that
|
|
6698
|
+
* `mint.ts` needs.
|
|
6699
|
+
*/
|
|
6700
|
+
|
|
6701
|
+
/**
|
|
6702
|
+
* Which split's score wins when a record carries both. `'holdout'` is the
|
|
6703
|
+
* canonical "real signal" default; `'search'` exists because some callers
|
|
6704
|
+
* deliberately score on the search split when both are present.
|
|
6705
|
+
*/
|
|
6706
|
+
type ScorePreference = 'holdout' | 'search';
|
|
6707
|
+
/** Only the outcome is read, so every accessor here accepts anything carrying one. */
|
|
6708
|
+
type Scored = Pick<RunRecord, 'outcome'>;
|
|
6709
|
+
/** True when the authenticity gate flagged the run as gamed (`realness.gated`). */
|
|
6710
|
+
declare function isRealnessGated(record: Scored): boolean;
|
|
6711
|
+
/**
|
|
6712
|
+
* The RAW score recorded on ONE split, with no cross-split fallback and no
|
|
6713
|
+
* anti-Goodhart gate.
|
|
6714
|
+
*
|
|
6715
|
+
* The narrowest of the three raw readers, and the one every split-scoped
|
|
6716
|
+
* consumer wants: a per-split report, a promotion gate, or a paired comparison
|
|
6717
|
+
* asks "what did this run score on the split I am summarising", and answering
|
|
6718
|
+
* it with the other split's number silently mixes populations. `undefined` =
|
|
6719
|
+
* that split was never scored.
|
|
6720
|
+
*
|
|
6721
|
+
* Same warning as `observedScore`: this INCLUDES runs flagged as gamed. Never
|
|
6722
|
+
* feed it into training data.
|
|
6723
|
+
*/
|
|
6724
|
+
declare function observedSplitScore(record: Scored, split: ScorePreference): number | undefined;
|
|
6725
|
+
/**
|
|
6726
|
+
* The RAW split score the run carries, with NO anti-Goodhart gate applied.
|
|
6727
|
+
*
|
|
6728
|
+
* INCLUDES RUNS FLAGGED AS GAMED (`outcome.realness.gated === true`); NEVER
|
|
6729
|
+
* feed this into training data — a fine-tune that sees it learns from gamed
|
|
6730
|
+
* successes. It is exported anyway because analysis, reporting, and
|
|
6731
|
+
* reward-hacking detection legitimately need the ungated number: forcing a
|
|
6732
|
+
* gamed run to 0 collapses the proxy signal toward ground truth and makes a
|
|
6733
|
+
* detector report "clean" on exactly the population that is being gamed.
|
|
6734
|
+
*
|
|
6735
|
+
* Returns `undefined` when the record carries neither score — an unscored run
|
|
6736
|
+
* is a labeled gap, not a measured zero, and each caller picks its own
|
|
6737
|
+
* sentinel (`?? 0`, `?? null`, skip, throw). Non-finite values are returned
|
|
6738
|
+
* as-is; callers that care keep their own `Number.isFinite` guard.
|
|
6739
|
+
*/
|
|
6740
|
+
declare function observedScore(record: Scored, prefer?: ScorePreference): number | undefined;
|
|
6741
|
+
/** Which split actually carried the score, or that none did. */
|
|
6742
|
+
type ScoreOrigin = 'holdout' | 'search' | 'unscored';
|
|
6743
|
+
/**
|
|
6744
|
+
* Where `observedScore` / `trainingScore` read their number from — the
|
|
6745
|
+
* provenance label a rollout line's `reward_source` is built from, and the
|
|
6746
|
+
* only supported way to ask "was this run scored at all" without respelling
|
|
6747
|
+
* the field access.
|
|
6748
|
+
*/
|
|
6749
|
+
declare function scoreOrigin(record: Scored, prefer?: ScorePreference): ScoreOrigin;
|
|
6750
|
+
/**
|
|
6751
|
+
* The GATED score — the only derivation allowed to reach training data.
|
|
6752
|
+
*
|
|
6753
|
+
* A realness-gated run scores 0 no matter what it claims, so a fine-tune
|
|
6754
|
+
* cannot learn from a gamed success. An unscored run stays `undefined` (a
|
|
6755
|
+
* labeled gap), keeping "we never measured this" distinct from "we measured
|
|
6756
|
+
* zero"; callers that need a number apply their own sentinel.
|
|
6757
|
+
*/
|
|
6758
|
+
declare function trainingScore(record: Scored, prefer?: ScorePreference): number | undefined;
|
|
6759
|
+
/**
|
|
6760
|
+
* `{reward, gated}` as written onto a minted `RolloutLine` — `trainingScore`
|
|
6761
|
+
* plus the flag itself, so the gate travels into the exported row and a
|
|
6762
|
+
* downstream filter can drop or down-weight the line.
|
|
6763
|
+
*
|
|
6764
|
+
* An unscored record yields `reward: null`, matching the schema's "no verdict
|
|
6765
|
+
* exists — a labeled gap, never 0" rule. It previously collapsed to 0, which
|
|
6766
|
+
* made a run nobody graded indistinguishable from one graded as a total
|
|
6767
|
+
* failure, and taught any trainer reading the row that the trajectory was bad.
|
|
6768
|
+
* A gated run still yields 0, because that IS a verdict: the gate decided.
|
|
6769
|
+
*/
|
|
6770
|
+
declare function trainingReward(record: Scored): {
|
|
6771
|
+
reward: number | null;
|
|
6772
|
+
gated: boolean;
|
|
6773
|
+
};
|
|
6774
|
+
|
|
6357
6775
|
/**
|
|
6358
6776
|
* Rollout minting — `tangle.rollout.v1` lines joined from the records the
|
|
6359
6777
|
* substrate ALREADY keeps. There is no separate rollout store: a rollout
|
|
@@ -6366,10 +6784,20 @@ declare function toJsonl(rows: ReadonlyArray<unknown>): string;
|
|
|
6366
6784
|
* - preference-pair export → `feedbackTrajectoryToOptimizerRow` (feedback-trajectory.ts)
|
|
6367
6785
|
* - PRM / reward-model → `reward-model-export.ts`
|
|
6368
6786
|
*
|
|
6369
|
-
* Anti-Goodhart invariant: a run whose `outcome.realness.gated` is true
|
|
6370
|
-
*
|
|
6371
|
-
* training data (`reward` forced
|
|
6372
|
-
*
|
|
6787
|
+
* Anti-Goodhart invariant: a run whose `outcome.realness.gated` is true is
|
|
6788
|
+
* never exported with a positive reward OR with any of the numbers that reward
|
|
6789
|
+
* was computed from. The gate travels into the training data (`reward` forced
|
|
6790
|
+
* to 0, `realness_gated: true`) and the whole outcome is transformed by
|
|
6791
|
+
* `gateGamedOutcome` inside `assertMinted` below, which relocates `metrics` and
|
|
6792
|
+
* `verdict` to `provenance.gated_evidence`. Mint returns
|
|
6793
|
+
* `MintedRolloutLine[]`: the brand the training exporters require, which only
|
|
6794
|
+
* this function, `readRolloutLedger`, and an explicit `assertMinted` can mint.
|
|
6795
|
+
*
|
|
6796
|
+
* A record carrying NEITHER split score is REJECTED (`ValidationError`), never
|
|
6797
|
+
* minted at 0 — "nobody graded this" is not the same claim as "graded a total
|
|
6798
|
+
* failure", and a trainer reading 0 learns the second. Lines that already
|
|
6799
|
+
* carry `reward: null` (interchange imports, existing ledgers) remain valid on
|
|
6800
|
+
* the wire; only the RunRecord→line door refuses.
|
|
6373
6801
|
*
|
|
6374
6802
|
* Records without spans become labeled GAP LINES (messages: [],
|
|
6375
6803
|
* provenance.gap) — present in the output AND surfaced in
|
|
@@ -6390,14 +6818,11 @@ interface MintRolloutOptions {
|
|
|
6390
6818
|
now?: () => Date;
|
|
6391
6819
|
}
|
|
6392
6820
|
interface MintRolloutResult {
|
|
6393
|
-
rows:
|
|
6821
|
+
rows: MintedRolloutLine[];
|
|
6394
6822
|
/** runIds that had a RunRecord but no spans — emitted as gap lines AND listed here. */
|
|
6395
6823
|
missingTraces: string[];
|
|
6396
6824
|
}
|
|
6397
|
-
|
|
6398
|
-
reward: number;
|
|
6399
|
-
gated: boolean;
|
|
6400
|
-
};
|
|
6825
|
+
|
|
6401
6826
|
/**
|
|
6402
6827
|
* Join RunRecords with their traces into canonical rollout lines. Records
|
|
6403
6828
|
* without spans are emitted as labeled gap lines and reported in
|
|
@@ -7370,7 +7795,7 @@ interface OtlpFlatLine {
|
|
|
7370
7795
|
}>;
|
|
7371
7796
|
}
|
|
7372
7797
|
interface FlattenOtlpOptions {
|
|
7373
|
-
/** `'openinference'` (default)
|
|
7798
|
+
/** `'openinference'` (default) maps source per-span attributes into the
|
|
7374
7799
|
* canonical OpenInference vocabulary the analyst readers consume. `'none'`
|
|
7375
7800
|
* passes attributes through untouched. */
|
|
7376
7801
|
attributeVocabulary?: 'openinference' | 'none';
|
|
@@ -8217,10 +8642,6 @@ interface LlmCorrectnessCheckerOpts {
|
|
|
8217
8642
|
costPhase?: string;
|
|
8218
8643
|
costTags?: Record<string, string>;
|
|
8219
8644
|
signal?: AbortSignal;
|
|
8220
|
-
/** Exact maximum provider attempts configured on the supplied TCloud client. */
|
|
8221
|
-
tcloudMaximumAttempts?: number;
|
|
8222
|
-
/** Usage/cost retained by a failed provider response; enables a safe retry. */
|
|
8223
|
-
receiptFromError?: (error: Error, attempt: number) => CostReceiptInput | undefined;
|
|
8224
8645
|
/** Max chars of artifact content sent to the checker. */
|
|
8225
8646
|
maxContentChars?: number;
|
|
8226
8647
|
/**
|
|
@@ -8253,7 +8674,7 @@ declare function parseCorrectnessResponse(raw: string): {
|
|
|
8253
8674
|
* only: a plan, a gesture, or a description of what should be done does not
|
|
8254
8675
|
* fulfil a requirement — the artifact must BE the deliverable.
|
|
8255
8676
|
*/
|
|
8256
|
-
declare function createLlmCorrectnessChecker(
|
|
8677
|
+
declare function createLlmCorrectnessChecker(chat: ChatClient, opts?: LlmCorrectnessCheckerOpts): CorrectnessChecker;
|
|
8257
8678
|
/**
|
|
8258
8679
|
* Deterministic `CorrectnessChecker` — the no-LLM counterpart to
|
|
8259
8680
|
* `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its
|
|
@@ -8477,7 +8898,7 @@ interface JudgeConfig<TArtifact, TScenario extends Scenario = Scenario> {
|
|
|
8477
8898
|
/** The canonical judge verdict shape — one declaration, shared by campaign
|
|
8478
8899
|
* judges and the multishot judge runner (which re-exports this type).
|
|
8479
8900
|
*
|
|
8480
|
-
* Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the
|
|
8901
|
+
* Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the
|
|
8481
8902
|
* multishot runner emits 0-10. Cross-scale comparison must go through
|
|
8482
8903
|
* `detectScale` (src/campaign/gates/statistical-heldout.ts, used by
|
|
8483
8904
|
* promotion-policy) — never renormalize a producer's values in place, as
|
|
@@ -8657,8 +9078,6 @@ interface CampaignAggregates {
|
|
|
8657
9078
|
byScenario: Record<string, ScenarioAggregate>;
|
|
8658
9079
|
/** Canonical campaign accounting, including worker and judge calls. */
|
|
8659
9080
|
cost: CostLedgerSummary;
|
|
8660
|
-
/** Compatibility alias of `cost.totalCostUsd`. */
|
|
8661
|
-
totalCostUsd: number;
|
|
8662
9081
|
/** Cells whose dispatch completed, including cells whose later judge failed. */
|
|
8663
9082
|
cellsExecuted: number;
|
|
8664
9083
|
cellsSkipped: number;
|
|
@@ -8669,7 +9088,7 @@ interface CampaignAggregates {
|
|
|
8669
9088
|
cellsDispatchFailed?: number;
|
|
8670
9089
|
/** Present on results that record failure stages. */
|
|
8671
9090
|
cellsJudgeFailed?: number;
|
|
8672
|
-
/**
|
|
9091
|
+
/** Failures whose stage could not be classified. */
|
|
8673
9092
|
cellsUnclassifiedFailed?: number;
|
|
8674
9093
|
}
|
|
8675
9094
|
interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scenario> {
|
|
@@ -11345,9 +11764,26 @@ interface CostSummary {
|
|
|
11345
11764
|
* for the analysis projection.
|
|
11346
11765
|
*/
|
|
11347
11766
|
|
|
11348
|
-
/**
|
|
11349
|
-
*
|
|
11350
|
-
*
|
|
11767
|
+
/**
|
|
11768
|
+
* The score the query/compare layer ranks on: holdout when present, else
|
|
11769
|
+
* search, with the anti-Goodhart gate applied — a run flagged as gamed reads 0
|
|
11770
|
+
* however high it claims to have scored.
|
|
11771
|
+
*
|
|
11772
|
+
* The gate is load-bearing here and this function did not previously apply it,
|
|
11773
|
+
* despite saying it did. `getBest` is few-shot exemplar selection: whatever it
|
|
11774
|
+
* returns is pasted into the next agent's prompt as an example to imitate.
|
|
11775
|
+
* Ranking on the raw number handed the highest-scoring gamed trajectory to
|
|
11776
|
+
* every subsequent run — propagation through the context window rather than
|
|
11777
|
+
* through a gradient, but propagation all the same.
|
|
11778
|
+
*
|
|
11779
|
+
* Analysis that needs to SEE the inflated number (reward-hack detection,
|
|
11780
|
+
* per-run reporting) reads `observedScore` from `rollout/reward.ts` directly.
|
|
11781
|
+
*
|
|
11782
|
+
* Returns `undefined` for an execution-only record (neither score present):
|
|
11783
|
+
* such rows are valid RunRecords, but cannot participate in score-ranked
|
|
11784
|
+
* queries. A gated run still reads 0 — the gate's verdict is a number, never
|
|
11785
|
+
* a gap.
|
|
11786
|
+
*/
|
|
11351
11787
|
declare function runScore(record: RunRecord): number | undefined;
|
|
11352
11788
|
interface RunRecordFilter {
|
|
11353
11789
|
experimentId?: string;
|
|
@@ -11382,6 +11818,17 @@ interface CandidateComparison {
|
|
|
11382
11818
|
bWins: number;
|
|
11383
11819
|
ties: number;
|
|
11384
11820
|
aWins: number;
|
|
11821
|
+
/**
|
|
11822
|
+
* Runs of either candidate excluded from the comparison because the
|
|
11823
|
+
* authenticity gate flagged them (`outcome.realness.gated`).
|
|
11824
|
+
*
|
|
11825
|
+
* Excluded rather than scored 0: `runScore` is gated, so leaving them in
|
|
11826
|
+
* would have entered a gamed run as a silent zero, which reads as "this
|
|
11827
|
+
* candidate failed the scenario" when what happened is "this candidate's
|
|
11828
|
+
* result is not evidence". A non-zero count here is itself the finding — a
|
|
11829
|
+
* comparison drawn over a shrunken scenario set has to say so.
|
|
11830
|
+
*/
|
|
11831
|
+
realnessGatedRuns: number;
|
|
11385
11832
|
}
|
|
11386
11833
|
/**
|
|
11387
11834
|
* Backing persistence for `EvalTraceStore`. The in-memory store is the default;
|
|
@@ -11420,6 +11867,12 @@ declare class EvalTraceStore {
|
|
|
11420
11867
|
* Highest-scoring run for a scenario (optionally restricted to a candidate).
|
|
11421
11868
|
* Returns null when no run matches. Ties resolve to the earliest-appended run
|
|
11422
11869
|
* so the result is stable.
|
|
11870
|
+
*
|
|
11871
|
+
* Runs flagged as gamed are DROPPED, not zeroed. The caller's use for this is
|
|
11872
|
+
* few-shot seeding — the returned trajectory becomes an example to copy — so
|
|
11873
|
+
* the same rule as SFT applies: a faked success must not be in the candidate
|
|
11874
|
+
* set at all. When every run for the scenario is gated the honest answer is
|
|
11875
|
+
* `null` (no exemplar), never the least-bad fake.
|
|
11423
11876
|
*/
|
|
11424
11877
|
getBest(scenarioId: string, opts?: {
|
|
11425
11878
|
candidateId?: string;
|
|
@@ -11430,6 +11883,9 @@ declare class EvalTraceStore {
|
|
|
11430
11883
|
* ran a scenario more than once, its best `runScore` for that scenario is
|
|
11431
11884
|
* used. Throws when there is no paired scenario — an unpaired "comparison" is
|
|
11432
11885
|
* not one.
|
|
11886
|
+
*
|
|
11887
|
+
* Realness-gated runs are excluded and counted in `realnessGatedRuns`, never
|
|
11888
|
+
* folded in as a zero.
|
|
11433
11889
|
*/
|
|
11434
11890
|
compareRuns(candidateA: string, candidateB: string): Promise<CandidateComparison>;
|
|
11435
11891
|
}
|
|
@@ -14937,7 +15393,7 @@ interface CampaignStorage {
|
|
|
14937
15393
|
write(path: string, content: string | Uint8Array): void;
|
|
14938
15394
|
/** Append only when the current UTF-8 byte length matches `expectedBytes`.
|
|
14939
15395
|
* Returns the new length, or undefined when another writer won. */
|
|
14940
|
-
append
|
|
15396
|
+
append(path: string, content: string, expectedBytes: number): number | undefined;
|
|
14941
15397
|
}
|
|
14942
15398
|
|
|
14943
15399
|
/**
|
|
@@ -17187,4 +17643,4 @@ type CachedJudge<TArtifact, TScenario extends Scenario = Scenario> = JudgeConfig
|
|
|
17187
17643
|
*/
|
|
17188
17644
|
declare function cachedJudge<TArtifact, TScenario extends Scenario = Scenario>(judge: JudgeConfig<TArtifact, TScenario>, store: VerdictCacheStore, options: CachedJudgeOptions): CachedJudge<TArtifact, TScenario>;
|
|
17189
17645
|
|
|
17190
|
-
export { AGENT_PROFILE_KINDS, ATTESTATION_ALGORITHM, type ActionExecutionPolicy, type ActionPolicyDecision, type ActionableSideInfo, type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, type AgentEvalErrorCode, type AgentInterfaceProfileLike, type AgentProfileCell, type AgentProfileCellInput, type AgentProfileCellSchemaVersion, AgentProfileCellValidationError, type AgentProfileDimensionValue, type AgentProfileHarness, type AgentProfileJson, type AgentProfileJsonObject, type AgentProfileKind, type AgentProfileRuntimeReceipt, type AgentProfileSource, type AgentProfileSourceInput, type AlignmentOp, type Analyst, type AnalystContext, type AnalystCost, type AnalystFinding, type AnalystHooks, type AnalystInputKind, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type AnalyzeTracesInput, type AnalyzeTracesOptions, type AnalyzeTracesResult, type AnalyzeTracesTurnSnapshot, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, type Artifact$1 as Artifact, type ArtifactCheck, type Artifact as ArtifactCheckArtifact, type ArtifactEventLike, type ArtifactResult, type ArtifactValidator, type AsiSeverity, type AssertCapabilityHeadroomOptions, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, type AttestationProvenance, type AttestationVerification, type AttestedReport, type AutoPrClient, AxGepaSteeringOptimizer, type AxSteeringOptimizerConfig, BENCHMARK_SPLIT_SEED, type BackendDescriptor, BackendIntegrityError, type BackendIntegrityReport, type BaselineOptions, type BaselineReport, BehaviorAssertion, type BehavioralMetrics, type BehavioralTokenSequence, type BenchmarkAdapter, type BenchmarkDatasetItem, type BenchmarkEvaluation, type BenchmarkFamily, type BenchmarkReport$1 as BenchmarkReport, type BenchmarkResponder, BenchmarkRunner, type BenchmarkRunnerConfig, type BenchmarkScenario, type BenchmarkSource, type BenchmarkTaskKind, type BisectOptions, type BisectResult, type BisectStep, type BlendWeights, type BootstrapOptions, type BootstrapResult, BudgetBreachError, BudgetGuard, type BudgetLedgerEntry, type BudgetPolicy, type BudgetSpec, CODING_HARNESSES, type CachedJudge, type CachedJudgeOptions, type CalibrationResult, CallExpectation, CallbackResearcher, type CallbackResearcherOptions, type CampaignFactoryParams, type CampaignIntegrityPolicy, type CampaignRunContext, type CampaignRunOutcome, type CampaignRunner, type CampaignScenario, type CampaignVariant, type CanaryAlert, type CanaryEvaluation, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateComparison, type CandidateScenario, type CandidateScore, type CanonicalRawAnalystFinding, type CapabilityHeadroomOptions, type CapabilityHeadroomResult, type CaptureFetchContext, type CaptureFetchOptions, CaptureIntegrityError, type CausalAttributionReport, type CellVerdict, type ChannelRollup, type ChatCallOpts, type ChatClient, type ChatMessage, type ChatRequest, type ChatResponse, type ChatToolCall, type ChatTransport, type CheckResult, type CliBridgeTransportOpts, type CliffsMagnitude, type ClusterBootstrapInterval, type ClusterSignFlipAlternative, type ClusterSignFlipResult, type ClusteredBinaryCluster, type ClusteredMatchedPair, type ClusteredPairedBinaryOptions, type ClusteredPairedBinaryResult, type ClusteredPairedBinaryStatistics, type CollectedArtifacts, type CommandRunner, type ComparePairedArmsOptions, type CompletionCriterion, type CompletionRequirement, type CompletionVerdict, type ConceptComplexity, type ConceptFinding, type ConceptSpec, type ConceptWeightStrategy, ConfigError, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContinuousAgreement, type ContinuousAgreementOptions, type ContinuousCalibrationResult, type ContractCheckResult, type ContractJudgeOptions, type ContractMetric, type ContractReport, type ContractRule, type ContractRuleKind, type ContractSpan, type ContractVerdict, type ContractViolation, type ControlActionFailureMode, type ControlActionOutcome, type ControlBudget, type ControlContext, type ControlDecision, type ControlEvalResult, type ControlRunResult, type ControlRunToRunRecordOptions, type ControlRuntimeConfig, type ControlRuntimeError, type ControlSeverity, type ControlStep, type ControlStopPolicies, ConvergenceTracker, type CorpusAgreementOptions, type CorpusAgreementPerDimension, type CorpusAgreementReport, type CorpusScoreRecord, type CorrectnessChecker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, type CostChannel, type CostEntry, CostLedger, type CostLedgerEntry, type CostLedgerFilter, type CostLedgerHandle, type CostLedgerOptions, type CostLedgerPersistence, CostLedgerPersistenceError, type CostLedgerSummary, type CostReceipt, CostReceiptCaptureError, type CostReceiptInput, type CostReport, CostReservationExceededError, type CostResult, type CostSummary, CostTracker, type CostUsage, type CounterfactualContext, type CounterfactualMutation, type CounterfactualResult, type CounterfactualRunner, type CreateAnalystAiConfig, type CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateExperimentInput, type CreateSandboxPoolOpts, type CreateTraceAnalystKindOpts, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, type CustomTokenPricing, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, type DataAcquisitionPlan, Dataset, type DatasetDifficulty, type DatasetManifest, type DatasetOverview, type DatasetProvenance, type DatasetScenario, type DatasetSplit, type DecideNextUserTurnOpts, type DefaultAnalystRegistryOptions, type DefaultVerdict, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DescriptionLengthCandidate, type DescriptionLengthConfig, type DescriptionLengthDecision, type DescriptionLengthEvidence, DescriptionLengthGate, type DescriptionLengthRejectionCode, type DetectorEvent, type DetectorSeverity, type DetectorSignal, type DiffPolicy, type DiffScorecardOptions, type DirEntry, type DirectProviderTransportOpts, type Direction, type DiscoverPersonasOptions, type DiscoveredPersona, DockerSandboxDriver, type DriverResult, type DriverState, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, type EProcess, type EProcessOptions, type EProcessState, type EProcessStep, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type ErrorCluster, type ErrorCountPattern, type ErrorStreakOptions, type EvalCampaignOptions, type EvalCampaignResult, type EvalResult, type EvalToolDef, EvalTraceStore, type EventFilter, type EventKind, type EvidenceRef, type EvolutionRound, type ExecutorConfig, type Expectation, type Experiment, type ExperimentPlan, type ExperimentProvenance, type ExperimentRep, type ExperimentResult, type ExperimentStats, type ExperimentStore, ExperimentTracker, type ExperimentTrackerOptions, type ExperimentVerdict, type ExportableSpan, type ExportedRewardModel, type ExtractOptions, type ExtractResult, type ExtractUsageFromSseOptions, type ExtractedUsage, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, type FactorContribution, type FactorialCell, type FailedRun, type FailureClass, type FailureClassification, type FailureContext, type FailureMode, type FailureRule, type FeedbackArtifactType, type FeedbackAttempt, type FeedbackLabel, type FeedbackLabelKind, type FeedbackLabelSource, type FeedbackOptimizerRow, type FeedbackOutcome, type FeedbackPattern, type FeedbackReplayAdapter, type FeedbackReplayResult, type FeedbackSeverity, type FeedbackSplitPolicy, type FeedbackTask, type FeedbackTrajectory, type FeedbackTrajectoryFilter, type FeedbackTrajectoryStore, type FieldDestination, type FileChange, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, type FileSystemRawProviderSinkOptions, FileSystemTraceStore, type FileSystemTraceStoreOptions, type Finding, type FindingSubject, type FindingSubjectKind, type FindingsDiff, FindingsStore, type FlattenOtlpOptions, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type GainDistributionBin, type GainDistributionFigureSpec, type GainDistributionOptions, type GateDecision$1 as GateDecision, type GateEvidence, type GenericSpan, type GhCliClientOptions, type GoldenItem, type GoldenSeverity, type GoldenSpec, HARNESS_NATIVE_MODEL, type HarnessAdapter, type HarnessConfig, type HarnessExperimentConfig, type HarnessExperimentResult, type HarnessIntervention, type HarnessRunRequest, type HarnessRunResult, type HarnessScenario, type HarnessSelection, type HarnessVariant, type HarnessVariantReport, type HeadroomClass, type HeadroomInput, HeldOutGate, type HeldOutGateConfig, type HeldOutGateRejectionCode, type HeldOutPartition, type HiddenCriteriaGrader, type HiddenGradeResult, type HiddenLeak, HoldoutAuditor, HoldoutLockedError, type HttpGithubClientOptions, type HypothesisManifest, type HypothesisResult, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, type ImageData, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, type InMemoryRawProviderSinkOptions, InMemoryTraceStore, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, type InterimReleaseConfidence, type InterimReleaseConfidenceInput, type JudgeConfig$1 as JudgeConfig, JudgeError, type JudgeFamily, type JudgeFleetOptions, type JudgeFn, type JudgeInput, JudgeParseError, type JudgeReplayGateArgs, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, type JudgeRubric, JudgeRunner, type JudgeScore$1 as JudgeScore, type JudgeScoreInput, type JudgeScoresRecord, type JudgeSpan, type JudgeVerdict, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type KnowledgeAcquisitionMode, type KnowledgeBundle, type KnowledgeFallbackPolicy, type KnowledgeFreshness, type KnowledgeImportance, type KnowledgeReadinessReport, type KnowledgeRecommendedAction, type KnowledgeRequirement, type KnowledgeRequirementCategory, type KnowledgeResponsibleSurface, type KnowledgeSensitivity, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, type Layer, type LayerResult, type LayerStatus, type LeaderboardOptions, type LeaderboardRow, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmCallError, type LlmCallMetadata, type LlmCallRequest, type LlmCallResult, LlmClient, type LlmClientOptions, type LlmCorrectnessCheckerOpts, type LlmJsonCall, type LlmJudgeDimension, type LlmJudgeOptions, type LlmMessage, LlmResponseError, type LlmReviewerConfig, LlmRouteAssertionError, type LlmRouteRequirements, type LlmSpan, type LlmSpanOtlpInput, type LlmUsage, LockedJsonlAppender, MODEL_PRICING, type MakeEvalToolsConfig, type MatchResult, type MatchedPair, type MatchedRunRecordPair, type MatcherResult, type MaximumCharge, type McNemarResult, type Measured, type MeasurementPolicy, type MergeOptions, type Message, type MetricSamples, type MetricVerdict, MetricsCollector, type MintRolloutOptions, type MintRolloutResult, type MockTransportOpts, type ModelCostRollup, type ModelPreflight, type ModelSeats, ModelsUnreachableError, type MuffledFinder, type MuffledFinding, MultiLayerVerifier, type MultiToolchainLayerConfig, type Mutator, Mutex, type NoLeakOptions, type NoProgressOptions, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, type Objective, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, type OtelExportConfig, type OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, type OtlpExport, OtlpFileTraceStore, type OtlpFileTraceStoreOptions, type OtlpFlatLine, type OtlpResourceSpans, type OtlpSpan, type OtlpSpanRole, type OtlpSpanRoleInput, type OtlpToRunRecordsOptions, type OtlpTraceRunRecord, type PaidCallResult, type PairArmsOptions, type PairArmsResult, type PairRunRecordsResult, type PairedArmRow, type PairedArmsComparison, type PairedBootstrapOptions, type PairedBootstrapResult, type PairedCorrectness, type PairedEvalueOptions, type PairedEvalueSequence, type PairedEvalueStep, type PairedMetricDelta, type PairedSignTestResult, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, type ParetoFigureSpec, type ParetoPoint, type ParetoResult, type PartitionHeldOutOptions, type PendingCostCall, type PendingCostCallView, type PersistedFinding, type PersonaConfig, type PersonaRigor, type Playbook, type PlaybookEntry, type PoolSlot, type PositionalBiasResult, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, type PreferenceMemoryEntry, type PreflightModelsOptions, type PreflightOutcome, type ProducedProposal, type ProducedState, type ProductBenchmarkArm, type ProductBenchmarkArtifactPaths, type ProductBenchmarkBudgets, type ProductBenchmarkExportOptions, type ProductBenchmarkExportResult, type ProductBenchmarkManifest, type ProductBenchmarkProfileRef, type ProductBenchmarkRecord, type ProductBenchmarkRepoRef, type ProductBenchmarkRunInput, type ProductBenchmarkScenario, type ProductBenchmarkSingleRunExportOptions, type ProductBenchmarkSplit, type ProductBenchmarkSubstrateVersions, type ProductBenchmarkValidationReport, ProductClient, type ProductClientConfig, type ProfileAxisSpec, type ProjectRuntimeTrajectoryEvidenceOptions, type ProjectedOtlpSpan, type PromptHandle, PromptRegistry, type ProportionInterval, type ProposalEventLike, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type ProposeFn, type ProposeInput, type ProposeOutput, type ProposeReviewConfig, type ProposeReviewControlAction, type ProposeReviewControlConfig, type ProposeReviewControlResult, type ProposeReviewControlState, type ProposeReviewReport, type ProposeReviewShot, type ProposedSideEffect, type ProvenanceReader, type ProviderRedactor, type QueryTracesPage, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, type RawAnalystEvidence, type RawAnalystFinding, type RawProviderDirection, type RawProviderEvent, type RawProviderSink, type RawProviderSinkFilter, type RecordRunsOptions, type RedTeamCase, type RedTeamCategory, type RedTeamFinding, type RedTeamPayload, type RedTeamReport, type RedactionReport, type RedactionRule, type ReferenceEquivalenceJudgeInput, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceJudgeResult, type ReferenceEquivalenceScenario, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, type RegistryRunOpts, type ReleaseConfidenceAxis, type ReleaseConfidenceAxisName, type ReleaseConfidenceInput, type ReleaseConfidenceIssue, type ReleaseConfidenceMetrics, type ReleaseConfidenceScorecard, type ReleaseConfidenceStatus, type ReleaseConfidenceThresholds, type ReleaseTraceEvidence, type RenderReleaseReportOptions, type RepeatedActionOptions, ReplayCache, type ReplayCacheEntry, ReplayCacheMissError, type ReplayCacheStats, ReplayError, type ReplayFetchOptions, type RepoRef, type RequirementCheck, type ResearchReport, type ResearchReportCandidate, type ResearchReportDecision, type ResearchReportMethodology, type ResearchReportOptions, type ResearchReportRecommendation, type Researcher, type RetrievalSpan, type Review, type ReviewFn, type ReviewInput, type ReviewMemoryEntry, type ReviewMemoryStore, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, type RewardRow, type RiskDifferenceResult, type RobustnessResult, type RolloutCapture, type RolloutLine, type RolloutRole, type RolloutScrubber, type RolloutSplit, type RolloutStep, type RouteMap, type RoutedField, type RouterTransportOpts, type RubricDimension, type Run, type RunCommandInput, type RunCommandResult, type RunCompleteHook, type RunCompleteHookContext, type RunCostProvenance, RunCritic, type RunCriticOptions, type RunEvidenceMetadata, type RunFilter, RunIntegrityError, type RunIntegrityExpectations, type RunIntegrityIssue, type RunIntegrityIssueCode, type RunIntegrityReport, type RunJudgeMetadata, type RunLayer, type RunOutcome, type RunPaidCallInput, type RunRecord, type RunRecordBackend, type RunRecordFilter, RunRecordValidationError, type RunScore, type RunScoreWeights, type RunSplitTag, type RunStatus, type RunTaskFailure, type RunTerminalOutcome, type RunTokenUsage, type RunTrace, type RuntimeEventLike, type RuntimeResolution, type RuntimeTrajectoryEvidenceProjection, type RuntimeTrajectoryEvidenceSummary, type RuntimeTrajectoryHookEvent, type RuntimeTrajectoryRecord, type RuntimeTrajectoryRunRecord, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_SCHEMA, type SandboxDriver, SandboxHarness, type SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type SandboxResult, type SandboxSdkTransportOpts, type SandboxSpan, type SatisfiedBy, type ScanOptions, type Scenario$1 as Scenario, type ScenarioCost, type ScenarioFile, ScenarioRegistry, type ScenarioResult, type ScoreKnowledgeReadinessOptions, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SearchSpanResult, type SearchTraceResult, type SeatName, type SeatPresetName, SeatUnsetError, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SelfPreferenceResult, type SemanticConceptJudgeInput, type SemanticConceptJudgeOptions, type SemanticConceptJudgeResult, type SequentialDecision, type SerializedRegex, type SeriesConvergenceOptions, type SeriesConvergenceResult, type Severity, type SftExportOptions, type SftRow, type SignTestAlternative, type SignedManifest, type SignedManifestAlgo, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, SkillUsageAnalyst, type SliceOptions, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, type SourceLimits, type Span, type SpanBase, type SpanFilter, type SpanHandle, type SpanKind, type SpanMatchRecord, SpanNotFoundError, type SpanPredicate, type SpanStatus, type SseUsageMode, type SteeringBundle, type SteeringChange, type SteeringDelta, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type SteeringRolePrompt, type StepAttribution, type StopDecision, type StreamingDetector, type SuboptimalCode, type SuboptimalSignal, SubprocessSandboxDriver, type SubprocessSandboxDriverOptions, type SummaryTable, type SummaryTableOptions, type SummaryTableRow, type SupervisorRunReader, type SupervisorRunReport, type SupervisorRunRollup, type SupervisorRunSources, type SupervisorRunTree, type SynthesisReason, type SynthesisTarget, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, type TaskGold, type TaskHeadroom, type TestGradedRunOptions, type TestGradedRunResult, type TestGradedScenario, type TestOutputParser, type TestResult, type TextMatcher, type ThresholdContract, TokenCounter, type TokenSpec, type ToolCallEventLike, type ToolDef, type ToolMatcher, type ToolSpan, type ToolSpanOtlpInput, type ToolStats, type ToolUseMetrics, type ToolUseOptions, type TraceAggregate, type TraceAnalysisStore, type TraceAnalystByteBudgets, type TraceAnalystFilters, type TraceAnalystGolden, type TraceAnalystHookOptions, type TraceAnalystKindSpec, type TraceAnalystSpan, type TraceAnalystSpanKind, type TraceAnalystSpanStatus, type TraceAnalystTraceSummary, type TraceContract, TraceContractBuilder, TraceEmitter, type TraceEmitterOptions, type TraceEvent, TraceFileMissingError, type TraceInsightContext, type TraceInsightFinding, type TraceInsightPanelRole, type TraceInsightPromptInput, type TraceInsightQualityGate, type TraceInsightQuestion, type TraceInsightReadiness, type TraceInsightSuite, type TraceInsightTask, TraceNotFoundError, type TraceStore, type TraceStoreSource, type TraceStoreToOtlpOptions, type TracedAnalystOptions, type TracedJudgeOptions, type TracesToOtlpResult, type Trajectory, type TrajectoryStep, type TreatmentClass, type TreatmentGate, type TreatmentGateInput, type TreatmentGateOptions, type TrialTrace, type Turn, type TurnMetrics, type TurnResult, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, type Unavailable, type UserQuestion, type ValidationContext, ValidationError, type ValidationIssue, type ValidationResult, type VerbosityBiasResult, type Verdict, type VerdictCacheStats, type VerdictCacheStore, type Verification, VerificationError, type VerificationReport, type VerifyContext, type VerifyFn, type VerifyOptions, type ViewSpansResult, type ViewTraceOversized, type ViewTraceResult, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, type WeightedCompositeInput, type WeightedCompositeResult, type WorkerDriverContext, type WorkflowTopology, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, acquisitionPlansForKnowledgeGaps, adversarialJudge, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, index$1 as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, buildWorkerDriverSystemPrompt, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, codeExecutionJudge, cohensD, coherenceJudge, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAnalystAi, createAntiSlopJudge, createChatClient, createCustomJudge, createDefaultReviewer, createDomainExpertJudge, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalystKind, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultJudges, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRetrievalSpan, isRolloutLine, isRunRecord, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalizeScores, notBlocked, objectiveEval, observeAll, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBootstrap, pairedCohensDz, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, index as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resolveModelPricing, resolveSeat, rolloutReward, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runsForScenario, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, typoMutator, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
|
|
17646
|
+
export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, type ActionExecutionPolicy, type ActionPolicyDecision, type ActionableSideInfo, type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, type AgentEvalErrorCode, type AgentInterfaceProfileLike, type AgentProfileCell, type AgentProfileCellInput, type AgentProfileCellSchemaVersion, AgentProfileCellValidationError, type AgentProfileDimensionValue, type AgentProfileHarness, type AgentProfileJson, type AgentProfileJsonObject, type AgentProfileKind, type AgentProfileRuntimeReceipt, type AgentProfileSource, type AgentProfileSourceInput, type AlignmentOp, type Analyst, type AnalystContext, type AnalystCost, type AnalystFinding, type AnalystHooks, type AnalystInputKind, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type AnalyzeTracesInput, type AnalyzeTracesOptions, type AnalyzeTracesResult, type AnalyzeTracesTurnSnapshot, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, type Artifact$1 as Artifact, type ArtifactCheck, type Artifact as ArtifactCheckArtifact, type ArtifactEventLike, type ArtifactResult, type ArtifactValidator, type AsiSeverity, type AssertCapabilityHeadroomOptions, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, type AttestationProvenance, type AttestationVerification, type AttestedReport, type AutoPrClient, AxGepaSteeringOptimizer, type AxSteeringOptimizerConfig, BENCHMARK_SPLIT_SEED, type BackendDescriptor, BackendIntegrityError, type BackendIntegrityReport, type BaselineOptions, type BaselineReport, BehaviorAssertion, type BehavioralMetrics, type BehavioralTokenSequence, type BenchmarkAdapter, type BenchmarkDatasetItem, type BenchmarkEvaluation, type BenchmarkFamily, type BenchmarkReport$1 as BenchmarkReport, type BenchmarkResponder, BenchmarkRunner, type BenchmarkRunnerConfig, type BenchmarkScenario, type BenchmarkSource, type BenchmarkTaskKind, type BisectOptions, type BisectResult, type BisectStep, type BlendWeights, type BootstrapOptions, type BootstrapResult, BudgetBreachError, BudgetGuard, type BudgetLedgerEntry, type BudgetPolicy, type BudgetSpec, CODING_HARNESSES, type CachedJudge, type CachedJudgeOptions, type CalibrationResult, CallExpectation, CallbackResearcher, type CallbackResearcherOptions, type CampaignFactoryParams, type CampaignIntegrityPolicy, type CampaignRunContext, type CampaignRunOutcome, type CampaignRunner, type CampaignScenario, type CampaignVariant, type CanaryAlert, type CanaryEvaluation, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateComparison, type CandidateScenario, type CandidateScore, type CapabilityHeadroomOptions, type CapabilityHeadroomResult, type CaptureFetchContext, type CaptureFetchOptions, CaptureIntegrityError, type CausalAttributionReport, type CellVerdict, type ChannelRollup, type ChatCallOpts, type ChatClient, type ChatMessage, type ChatRequest, type ChatResponse, type ChatToolCall, type ChatTransport, type CheckResult, type CliBridgeTransportOpts, type CliffsMagnitude, type ClusterBootstrapInterval, type ClusterSignFlipAlternative, type ClusterSignFlipResult, type ClusteredBinaryCluster, type ClusteredMatchedPair, type ClusteredPairedBinaryOptions, type ClusteredPairedBinaryResult, type ClusteredPairedBinaryStatistics, type CollectedArtifacts, type CommandRunner, type ComparePairedArmsOptions, type CompletionCriterion, type CompletionRequirement, type CompletionVerdict, type ConceptComplexity, type ConceptFinding, type ConceptSpec, type ConceptWeightStrategy, ConfigError, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContinuousAgreement, type ContinuousAgreementOptions, type ContinuousCalibrationResult, type ContractCheckResult, type ContractJudgeOptions, type ContractMetric, type ContractReport, type ContractRule, type ContractRuleKind, type ContractSpan, type ContractVerdict, type ContractViolation, type ControlActionFailureMode, type ControlActionOutcome, type ControlBudget, type ControlContext, type ControlDecision, type ControlEvalResult, type ControlRunResult, type ControlRunToRunRecordOptions, type ControlRuntimeConfig, type ControlRuntimeError, type ControlSeverity, type ControlStep, type ControlStopPolicies, ConvergenceTracker, type CorpusAgreementOptions, type CorpusAgreementPerDimension, type CorpusAgreementReport, type CorpusScoreRecord, type CorrectnessChecker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, type CostChannel, type CostEntry, CostLedger, type CostLedgerFilter, type CostLedgerHandle, type CostLedgerOptions, type CostLedgerPersistence, CostLedgerPersistenceError, type CostLedgerSummary, type CostReceipt, CostReceiptCaptureError, type CostReceiptInput, type CostReport, CostReservationExceededError, type CostResult, type CostSummary, CostTracker, type CostUsage, type CounterfactualContext, type CounterfactualMutation, type CounterfactualResult, type CounterfactualRunner, type CreateAnalystAiConfig, type CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateExperimentInput, type CreateSandboxPoolOpts, type CreateTraceAnalystKindOpts, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, type CustomTokenPricing, type CustomTransportOpts, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, type DataAcquisitionPlan, Dataset, type DatasetDifficulty, type DatasetManifest, type DatasetOverview, type DatasetProvenance, type DatasetScenario, type DatasetSplit, type DecideNextUserTurnOpts, type DefaultAnalystRegistryOptions, type DefaultVerdict, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DescriptionLengthCandidate, type DescriptionLengthConfig, type DescriptionLengthDecision, type DescriptionLengthEvidence, DescriptionLengthGate, type DescriptionLengthRejectionCode, type DetectorEvent, type DetectorSeverity, type DetectorSignal, type DiffPolicy, type DiffScorecardOptions, type DirEntry, type DirectProviderTransportOpts, type Direction, type DiscoverPersonasOptions, type DiscoveredPersona, DockerSandboxDriver, type DriverResult, type DriverState, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, type EProcess, type EProcessOptions, type EProcessState, type EProcessStep, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type ErrorCluster, type ErrorCountPattern, type ErrorStreakOptions, type EvalCampaignOptions, type EvalCampaignResult, type EvalResult, type EvalToolDef, EvalTraceStore, type EventFilter, type EventKind, type EvidenceRef, type EvolutionRound, type ExecutorConfig, type Expectation, type Experiment, type ExperimentPlan, type ExperimentProvenance, type ExperimentRep, type ExperimentResult, type ExperimentStats, type ExperimentStore, ExperimentTracker, type ExperimentTrackerOptions, type ExperimentVerdict, type ExportableSpan, type ExportedRewardModel, type ExtractOptions, type ExtractResult, type ExtractUsageFromSseOptions, type ExtractedUsage, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, type FactorContribution, type FactorialCell, type FailedRun, type FailureClass, type FailureClassification, type FailureContext, type FailureMode, type FailureRule, type FeedbackArtifactType, type FeedbackAttempt, type FeedbackLabel, type FeedbackLabelKind, type FeedbackLabelSource, type FeedbackOptimizerRow, type FeedbackOutcome, type FeedbackPattern, type FeedbackReplayAdapter, type FeedbackReplayResult, type FeedbackSeverity, type FeedbackSplitPolicy, type FeedbackTask, type FeedbackTrajectory, type FeedbackTrajectoryFilter, type FeedbackTrajectoryStore, type FieldDestination, type FileChange, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, type FileSystemRawProviderSinkOptions, FileSystemTraceStore, type FileSystemTraceStoreOptions, type Finding, type FindingSubject, type FindingSubjectKind, type FindingsDiff, FindingsStore, type FlattenOtlpOptions, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type FromHarborOptions, type GainDistributionBin, type GainDistributionFigureSpec, type GainDistributionOptions, type GateDecision$1 as GateDecision, type GateEvidence, type GenericSpan, type GhCliClientOptions, type GoldenItem, type GoldenSeverity, type GoldenSpec, HARBOR_IMPORT_GAP, HARNESS_NATIVE_MODEL, type HarborAgent, type HarborContentPart, type HarborFinalMetrics, type HarborImageSource, type HarborMetrics, type HarborObservation, type HarborObservationResult, type HarborStep, type HarborStepSource, type HarborSubagentTrajectoryRef, type HarborToolCall, type HarborTrajectory, type HarnessAdapter, type HarnessConfig, type HarnessExperimentConfig, type HarnessExperimentResult, type HarnessIntervention, type HarnessRunRequest, type HarnessRunResult, type HarnessScenario, type HarnessSelection, type HarnessVariant, type HarnessVariantReport, type HeadroomClass, type HeadroomInput, HeldOutGate, type HeldOutGateConfig, type HeldOutGateRejectionCode, type HeldOutPartition, type HiddenCriteriaGrader, type HiddenGradeResult, type HiddenLeak, HoldoutAuditor, HoldoutLockedError, type HttpGithubClientOptions, type HypothesisManifest, type HypothesisResult, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, type ImageData, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, type InMemoryRawProviderSinkOptions, InMemoryTraceStore, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, type InterimReleaseConfidence, type InterimReleaseConfidenceInput, type JudgeConfig$1 as JudgeConfig, JudgeError, type JudgeFamily, type JudgeFleetOptions, type JudgeFn, type JudgeInput, JudgeParseError, type JudgeReplayGateArgs, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, type JudgeRubric, JudgeRunner, type JudgeScore$1 as JudgeScore, type JudgeScoreInput, type JudgeScoresRecord, type JudgeSpan, type JudgeVerdict, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type KnowledgeAcquisitionMode, type KnowledgeBundle, type KnowledgeFallbackPolicy, type KnowledgeFreshness, type KnowledgeImportance, type KnowledgeReadinessReport, type KnowledgeRecommendedAction, type KnowledgeRequirement, type KnowledgeRequirementCategory, type KnowledgeResponsibleSurface, type KnowledgeSensitivity, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, type Layer, type LayerResult, type LayerStatus, type LeaderboardOptions, type LeaderboardRow, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmCallError, type LlmCallMetadata, type LlmCallRequest, type LlmCallResult, LlmClient, type LlmClientOptions, type LlmCorrectnessCheckerOpts, type LlmJsonCall, type LlmJudgeDimension, type LlmJudgeOptions, type LlmMessage, LlmResponseError, type LlmReviewerConfig, LlmRouteAssertionError, type LlmRouteRequirements, type LlmSpan, type LlmSpanOtlpInput, type LlmUsage, LockedJsonlAppender, MODEL_PRICING, type MakeEvalToolsConfig, type MatchResult, type MatchedPair, type MatchedRunRecordPair, type MatcherResult, type MaximumCharge, type McNemarResult, type Measured, type MeasurementPolicy, type MergeOptions, type Message, type MetricSamples, type MetricVerdict, MetricsCollector, type MintRolloutOptions, type MintRolloutResult, type MintedRolloutLine, type MintedRolloutOutcome, type MockTransportOpts, type ModelCostRollup, type ModelPreflight, type ModelSeats, ModelsUnreachableError, type MuffledFinder, type MuffledFinding, MultiLayerVerifier, type MultiToolchainLayerConfig, type Mutator, Mutex, type NoLeakOptions, type NoProgressOptions, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, type Objective, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, type OtelExportConfig, type OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, type OtlpExport, OtlpFileTraceStore, type OtlpFileTraceStoreOptions, type OtlpFlatLine, type OtlpResourceSpans, type OtlpSpan, type OtlpSpanRole, type OtlpSpanRoleInput, type OtlpToRunRecordsOptions, type OtlpTraceRunRecord, type PaidCallResult, type PairArmsOptions, type PairArmsResult, type PairRunRecordsResult, type PairedArmRow, type PairedArmsComparison, type PairedBootstrapOptions, type PairedBootstrapResult, type PairedCorrectness, type PairedEvalueOptions, type PairedEvalueSequence, type PairedEvalueStep, type PairedMetricDelta, type PairedSignTestResult, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, type ParetoFigureSpec, type ParetoPoint, type ParetoResult, type PartitionHeldOutOptions, type PendingCostCall, type PendingCostCallView, type PersistedFinding, type PersonaConfig, type PersonaRigor, type Playbook, type PlaybookEntry, type PoolSlot, type PositionalBiasResult, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, type PreferenceMemoryEntry, type PreflightModelsOptions, type PreflightOutcome, type ProducedProposal, type ProducedState, type ProductBenchmarkArm, type ProductBenchmarkArtifactPaths, type ProductBenchmarkBudgets, type ProductBenchmarkExportOptions, type ProductBenchmarkExportResult, type ProductBenchmarkManifest, type ProductBenchmarkProfileRef, type ProductBenchmarkRecord, type ProductBenchmarkRepoRef, type ProductBenchmarkRunInput, type ProductBenchmarkScenario, type ProductBenchmarkSingleRunExportOptions, type ProductBenchmarkSplit, type ProductBenchmarkSubstrateVersions, type ProductBenchmarkValidationReport, ProductClient, type ProductClientConfig, type ProfileAxisSpec, type ProjectRuntimeTrajectoryEvidenceOptions, type ProjectedOtlpSpan, type PromptHandle, PromptRegistry, type ProportionInterval, type ProposalEventLike, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type ProposeFn, type ProposeInput, type ProposeOutput, type ProposeReviewConfig, type ProposeReviewControlAction, type ProposeReviewControlConfig, type ProposeReviewControlResult, type ProposeReviewControlState, type ProposeReviewReport, type ProposeReviewShot, type ProposedSideEffect, type ProvenanceReader, type ProviderRedactor, type QueryTracesPage, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, type RawAnalystEvidence, type RawAnalystFinding, type RawProviderDirection, type RawProviderEvent, type RawProviderSink, type RawProviderSinkFilter, type RecordRunsOptions, type RedTeamCase, type RedTeamCategory, type RedTeamFinding, type RedTeamPayload, type RedTeamReport, type RedactionReport, type RedactionRule, type ReferenceEquivalenceJudgeInput, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceJudgeResult, type ReferenceEquivalenceScenario, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, type RegistryRunOpts, type ReleaseConfidenceAxis, type ReleaseConfidenceAxisName, type ReleaseConfidenceInput, type ReleaseConfidenceIssue, type ReleaseConfidenceMetrics, type ReleaseConfidenceScorecard, type ReleaseConfidenceStatus, type ReleaseConfidenceThresholds, type ReleaseTraceEvidence, type RenderReleaseReportOptions, type RepeatedActionOptions, ReplayCache, type ReplayCacheEntry, ReplayCacheMissError, type ReplayCacheStats, ReplayError, type ReplayFetchOptions, type RepoRef, type RequirementCheck, type ResearchReport, type ResearchReportCandidate, type ResearchReportDecision, type ResearchReportMethodology, type ResearchReportOptions, type ResearchReportRecommendation, type Researcher, type RetrievalSpan, type Review, type ReviewFn, type ReviewInput, type ReviewMemoryEntry, type ReviewMemoryStore, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, type RewardRow, type RiskDifferenceResult, type RobustnessResult, type RolloutCapture, type RolloutLine, type RolloutRole, type RolloutScrubber, type RolloutSplit, type RolloutStep, type RouteMap, type RoutedField, type RouterTransportOpts, type RubricDimension, type Run, type RunCommandInput, type RunCommandResult, type RunCompleteHook, type RunCompleteHookContext, type RunCostProvenance, RunCritic, type RunCriticOptions, type RunEvidenceMetadata, type RunFilter, RunIntegrityError, type RunIntegrityExpectations, type RunIntegrityIssue, type RunIntegrityIssueCode, type RunIntegrityReport, type RunJudgeMetadata, type RunLayer, type RunOutcome, type RunPaidCallInput, type RunRecord, type RunRecordBackend, type RunRecordFilter, RunRecordValidationError, type RunScore, type RunScoreWeights, type RunSplitTag, type RunStatus, type RunTaskFailure, type RunTerminalOutcome, type RunTokenUsage, type RunTrace, type RuntimeEventLike, type RuntimeResolution, type RuntimeTrajectoryEvidenceProjection, type RuntimeTrajectoryEvidenceSummary, type RuntimeTrajectoryHookEvent, type RuntimeTrajectoryRecord, type RuntimeTrajectoryRunRecord, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_SCHEMA, type SandboxDriver, SandboxHarness, type SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type SandboxResult, type SandboxSdkTransportOpts, type SandboxSpan, type SatisfiedBy, type ScanOptions, type Scenario$1 as Scenario, type ScenarioCost, type ScenarioFile, ScenarioRegistry, type ScenarioResult, type ScoreKnowledgeReadinessOptions, type ScoreOrigin, type ScorePreference, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SearchSpanResult, type SearchTraceResult, type SeatName, type SeatPresetName, SeatUnsetError, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SelfPreferenceResult, type SemanticConceptJudgeInput, type SemanticConceptJudgeOptions, type SemanticConceptJudgeResult, type SequentialDecision, type SerializedRegex, type SeriesConvergenceOptions, type SeriesConvergenceResult, type Severity, type SftExportOptions, type SftRow, type SignTestAlternative, type SignedManifest, type SignedManifestAlgo, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, SkillUsageAnalyst, type SliceOptions, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, type SourceLimits, type Span, type SpanBase, type SpanFilter, type SpanHandle, type SpanKind, type SpanMatchRecord, SpanNotFoundError, type SpanPredicate, type SpanStatus, type SseUsageMode, type SteeringBundle, type SteeringChange, type SteeringDelta, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type SteeringRolePrompt, type StepAttribution, type StopDecision, type StreamingDetector, type SuboptimalCode, type SuboptimalSignal, SubprocessSandboxDriver, type SubprocessSandboxDriverOptions, type SummaryTable, type SummaryTableOptions, type SummaryTableRow, type SupervisorRunReader, type SupervisorRunReport, type SupervisorRunRollup, type SupervisorRunSources, type SupervisorRunTree, type SynthesisReason, type SynthesisTarget, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, type TaskGold, type TaskHeadroom, type TestGradedRunOptions, type TestGradedRunResult, type TestGradedScenario, type TestOutputParser, type TestResult, type TextMatcher, type ThresholdContract, TokenCounter, type TokenSpec, type ToolCallEventLike, type ToolDef, type ToolMatcher, type ToolSpan, type ToolSpanOtlpInput, type ToolStats, type ToolUseMetrics, type ToolUseOptions, type TraceAggregate, type TraceAnalysisStore, type TraceAnalystByteBudgets, type TraceAnalystFilters, type TraceAnalystGolden, type TraceAnalystHookOptions, type TraceAnalystKindSpec, type TraceAnalystSpan, type TraceAnalystSpanKind, type TraceAnalystSpanStatus, type TraceAnalystTraceSummary, type TraceContract, TraceContractBuilder, TraceEmitter, type TraceEmitterOptions, type TraceEvent, TraceFileMissingError, type TraceInsightContext, type TraceInsightFinding, type TraceInsightPanelRole, type TraceInsightPromptInput, type TraceInsightQualityGate, type TraceInsightQuestion, type TraceInsightReadiness, type TraceInsightSuite, type TraceInsightTask, TraceNotFoundError, type TraceStore, type TraceStoreSource, type TraceStoreToOtlpOptions, type TracedAnalystOptions, type TracedJudgeOptions, type TracesToOtlpResult, type Trajectory, type TrajectoryStep, type TreatmentClass, type TreatmentGate, type TreatmentGateInput, type TreatmentGateOptions, type TrialTrace, type Turn, type TurnMetrics, type TurnResult, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, type Unavailable, type UserQuestion, type ValidationContext, ValidationError, type ValidationIssue, type ValidationResult, type VerbosityBiasResult, type Verdict, type VerdictCacheStats, type VerdictCacheStore, type Verification, VerificationError, type VerificationReport, type VerifyContext, type VerifyFn, type VerifyOptions, type ViewSpansResult, type ViewTraceOversized, type ViewTraceResult, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, type WeightedCompositeInput, type WeightedCompositeResult, type WorkerDriverContext, type WorkflowTopology, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, index$1 as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, buildWorkerDriverSystemPrompt, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAnalystAi, createAntiSlopJudge, createChatClient, createDefaultReviewer, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalystKind, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBootstrap, pairedCohensDz, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, index as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resolveModelPricing, resolveSeat, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runsForScenario, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
|