@tangle-network/agent-eval 0.128.2 → 0.129.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (137) hide show
  1. package/CHANGELOG.md +265 -0
  2. package/README.md +18 -0
  3. package/dist/analyst/index.d.ts +107 -165
  4. package/dist/analyst/index.js +5 -9
  5. package/dist/analyst/index.js.map +1 -1
  6. package/dist/belief-state/index.d.ts +2 -19
  7. package/dist/belief-state/index.js +30 -31
  8. package/dist/belief-state/index.js.map +1 -1
  9. package/dist/benchmarks/index.d.ts +5 -8
  10. package/dist/benchmarks/index.js +12 -11
  11. package/dist/builder-eval/index.js +1 -1
  12. package/dist/campaign/index.d.ts +30 -39
  13. package/dist/campaign/index.js +11 -10
  14. package/dist/{chunk-NKAGIDE2.js → chunk-2QU3YOPR.js} +15 -274
  15. package/dist/chunk-2QU3YOPR.js.map +1 -0
  16. package/dist/{chunk-EJGRPCO3.js → chunk-3OCR4R5I.js} +245 -134
  17. package/dist/chunk-3OCR4R5I.js.map +1 -0
  18. package/dist/{chunk-2JX3CFMB.js → chunk-56TAVBOK.js} +5 -2
  19. package/dist/chunk-56TAVBOK.js.map +1 -0
  20. package/dist/{chunk-DPUHNQLN.js → chunk-7FO3TNPI.js} +2 -2
  21. package/dist/{chunk-DJKY2TSY.js → chunk-BSO5JDQH.js} +27 -120
  22. package/dist/chunk-BSO5JDQH.js.map +1 -0
  23. package/dist/{chunk-EZJEIH2R.js → chunk-C6LXANRU.js} +11 -20
  24. package/dist/chunk-C6LXANRU.js.map +1 -0
  25. package/dist/{chunk-ZUUWPZCV.js → chunk-DODXQREJ.js} +4 -4
  26. package/dist/{chunk-2MKQIFS4.js → chunk-E7QXT7SX.js} +2 -2
  27. package/dist/{chunk-NYLOYM6N.js → chunk-EG66UGL4.js} +37 -28
  28. package/dist/chunk-EG66UGL4.js.map +1 -0
  29. package/dist/{chunk-P5W7RQKK.js → chunk-FXTVJPYD.js} +2 -2
  30. package/dist/{chunk-VZSRQ272.js → chunk-G7MGMCZD.js} +6 -2
  31. package/dist/chunk-G7MGMCZD.js.map +1 -0
  32. package/dist/{chunk-IHQDPH7D.js → chunk-H23X7XKK.js} +85 -75
  33. package/dist/chunk-H23X7XKK.js.map +1 -0
  34. package/dist/{chunk-VBQ3CRKH.js → chunk-HPWUNB47.js} +4 -6
  35. package/dist/chunk-HPWUNB47.js.map +1 -0
  36. package/dist/{chunk-XPRT64IE.js → chunk-IYCLP2N2.js} +3 -3
  37. package/dist/chunk-IYCLP2N2.js.map +1 -0
  38. package/dist/{chunk-XDWDC2MP.js → chunk-JQSF5DQT.js} +11 -5
  39. package/dist/chunk-JQSF5DQT.js.map +1 -0
  40. package/dist/{chunk-NACAGYSY.js → chunk-M4YBQKIJ.js} +11 -11
  41. package/dist/chunk-M4YBQKIJ.js.map +1 -0
  42. package/dist/{chunk-YJBNWCAA.js → chunk-NY44NC4A.js} +3 -3
  43. package/dist/chunk-OIUOT4QD.js +44 -0
  44. package/dist/chunk-OIUOT4QD.js.map +1 -0
  45. package/dist/chunk-OWN5NPMC.js +152 -0
  46. package/dist/chunk-OWN5NPMC.js.map +1 -0
  47. package/dist/chunk-PC5DOSM7.js +579 -0
  48. package/dist/chunk-PC5DOSM7.js.map +1 -0
  49. package/dist/{chunk-UB2LOJ6Q.js → chunk-QB6BDBP2.js} +23 -20
  50. package/dist/chunk-QB6BDBP2.js.map +1 -0
  51. package/dist/chunk-RXHCETDZ.js +536 -0
  52. package/dist/chunk-RXHCETDZ.js.map +1 -0
  53. package/dist/{chunk-PBE2LOSS.js → chunk-SFLLL76A.js} +7 -7
  54. package/dist/chunk-SFLLL76A.js.map +1 -0
  55. package/dist/{chunk-VGRCHJON.js → chunk-T6RLYGAD.js} +3 -8
  56. package/dist/chunk-T6RLYGAD.js.map +1 -0
  57. package/dist/{chunk-VLOATJQ2.js → chunk-TJVT4QFF.js} +21 -18
  58. package/dist/chunk-TJVT4QFF.js.map +1 -0
  59. package/dist/{chunk-S5YLIBFX.js → chunk-TQ7LNKZ3.js} +2 -2
  60. package/dist/{chunk-EOSZT7PL.js → chunk-U4L7JRPZ.js} +2 -297
  61. package/dist/chunk-U4L7JRPZ.js.map +1 -0
  62. package/dist/chunk-U4PHLT2N.js +419 -0
  63. package/dist/chunk-U4PHLT2N.js.map +1 -0
  64. package/dist/{chunk-WS3NZZQQ.js → chunk-VCZ5FQYW.js} +3 -4
  65. package/dist/chunk-VCZ5FQYW.js.map +1 -0
  66. package/dist/{chunk-BYT7ELPS.js → chunk-WVATSFCP.js} +2 -2
  67. package/dist/{chunk-TSN7JT6D.js → chunk-X4YIBDER.js} +21 -5
  68. package/dist/{chunk-TSN7JT6D.js.map → chunk-X4YIBDER.js.map} +1 -1
  69. package/dist/{chunk-TBL77AUT.js → chunk-YQN4ICPP.js} +5 -5
  70. package/dist/{chunk-MHELPNRP.js → chunk-ZHTZ4EYI.js} +1 -1
  71. package/dist/chunk-ZHTZ4EYI.js.map +1 -0
  72. package/dist/cli.js +6 -5
  73. package/dist/cli.js.map +1 -1
  74. package/dist/contract/index.d.ts +47 -87
  75. package/dist/contract/index.js +14 -13
  76. package/dist/contract/index.js.map +1 -1
  77. package/dist/control.js +3 -2
  78. package/dist/fuzz.js +3 -2
  79. package/dist/fuzz.js.map +1 -1
  80. package/dist/index.d.ts +659 -203
  81. package/dist/index.js +145 -117
  82. package/dist/index.js.map +1 -1
  83. package/dist/meta-eval/index.js +2 -2
  84. package/dist/multishot/index.d.ts +3 -4
  85. package/dist/multishot/index.js.map +1 -1
  86. package/dist/openapi.json +1 -1
  87. package/dist/pipelines/index.js +5 -5
  88. package/dist/reporting.d.ts +14 -0
  89. package/dist/reporting.js +7 -6
  90. package/dist/rl.d.ts +652 -82
  91. package/dist/rl.js +415 -171
  92. package/dist/rl.js.map +1 -1
  93. package/dist/rollout/index.d.ts +1071 -32
  94. package/dist/rollout/index.js +68 -10
  95. package/dist/run-campaign-OJJ7CZF4.js +18 -0
  96. package/dist/supervisor-run/index.d.ts +114 -4
  97. package/dist/supervisor-run/index.js +4 -3
  98. package/dist/traces.d.ts +1 -1
  99. package/dist/traces.js +6 -5
  100. package/dist/wire/index.d.ts +10 -11
  101. package/dist/wire/index.js +3 -3
  102. package/docs/feature-guide.md +1 -1
  103. package/docs/rollout.md +116 -2
  104. package/package.json +4 -4
  105. package/dist/chunk-2JX3CFMB.js.map +0 -1
  106. package/dist/chunk-DJKY2TSY.js.map +0 -1
  107. package/dist/chunk-EJGRPCO3.js.map +0 -1
  108. package/dist/chunk-EOSZT7PL.js.map +0 -1
  109. package/dist/chunk-EZJEIH2R.js.map +0 -1
  110. package/dist/chunk-IHQDPH7D.js.map +0 -1
  111. package/dist/chunk-MHELPNRP.js.map +0 -1
  112. package/dist/chunk-NACAGYSY.js.map +0 -1
  113. package/dist/chunk-NKAGIDE2.js.map +0 -1
  114. package/dist/chunk-NYLOYM6N.js.map +0 -1
  115. package/dist/chunk-PBE2LOSS.js.map +0 -1
  116. package/dist/chunk-TT4KNT67.js +0 -124
  117. package/dist/chunk-TT4KNT67.js.map +0 -1
  118. package/dist/chunk-UB2LOJ6Q.js.map +0 -1
  119. package/dist/chunk-UWZZKKU7.js +0 -237
  120. package/dist/chunk-UWZZKKU7.js.map +0 -1
  121. package/dist/chunk-VBQ3CRKH.js.map +0 -1
  122. package/dist/chunk-VGRCHJON.js.map +0 -1
  123. package/dist/chunk-VLOATJQ2.js.map +0 -1
  124. package/dist/chunk-VZSRQ272.js.map +0 -1
  125. package/dist/chunk-WS3NZZQQ.js.map +0 -1
  126. package/dist/chunk-XDWDC2MP.js.map +0 -1
  127. package/dist/chunk-XPRT64IE.js.map +0 -1
  128. package/dist/run-campaign-ISHFZ7FJ.js +0 -17
  129. /package/dist/{chunk-DPUHNQLN.js.map → chunk-7FO3TNPI.js.map} +0 -0
  130. /package/dist/{chunk-ZUUWPZCV.js.map → chunk-DODXQREJ.js.map} +0 -0
  131. /package/dist/{chunk-2MKQIFS4.js.map → chunk-E7QXT7SX.js.map} +0 -0
  132. /package/dist/{chunk-P5W7RQKK.js.map → chunk-FXTVJPYD.js.map} +0 -0
  133. /package/dist/{chunk-YJBNWCAA.js.map → chunk-NY44NC4A.js.map} +0 -0
  134. /package/dist/{chunk-S5YLIBFX.js.map → chunk-TQ7LNKZ3.js.map} +0 -0
  135. /package/dist/{chunk-BYT7ELPS.js.map → chunk-WVATSFCP.js.map} +0 -0
  136. /package/dist/{chunk-TBL77AUT.js.map → chunk-YQN4ICPP.js.map} +0 -0
  137. /package/dist/{run-campaign-ISHFZ7FJ.js.map → run-campaign-OJJ7CZF4.js.map} +0 -0
package/dist/index.d.ts CHANGED
@@ -2,7 +2,6 @@ import { AgentProfile, HarnessType } from '@tangle-network/agent-interface';
2
2
  export { AgentProfile, HarnessType } from '@tangle-network/agent-interface';
3
3
  import { AxAIArgs, AxAIService, AxFunction, AxAgentActorTurnCallbackArgs } from '@ax-llm/ax';
4
4
  import { z } from 'zod';
5
- import { TCloud } from '@tangle-network/tcloud';
6
5
 
7
6
  /**
8
7
  * TraceSchema v1 — the canonical data model for agent-eval.
@@ -1166,8 +1165,6 @@ interface CostReceipt extends CostCallBase, CostUsage {
1166
1165
  actualCostUsd?: number;
1167
1166
  error?: string;
1168
1167
  }
1169
- /** @deprecated Read-only compatibility shape. New paid work uses `runPaidCall`. */
1170
- type CostLedgerEntry = Omit<CostReceipt, 'status' | 'callId' | 'phase' | 'actor' | 'maximumCostUsd' | 'usageUnknown' | 'pricing' | 'error'>;
1171
1168
  interface CostReceiptInput extends CostUsage {
1172
1169
  model: string;
1173
1170
  /** Caller-supplied rates for a local estimate when the provider does not report billed cost. */
@@ -1506,9 +1503,8 @@ declare function providerFromBaseUrl(baseUrl: string): string;
1506
1503
  * { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
1507
1504
  * )
1508
1505
  *
1509
- * This is THE llm-calling seam for agent-eval primitives that need structured
1510
- * output (semantic concept judge, reviewer directives, critic scores). Primitives
1511
- * that need free-form text use `callLlm` and parse output themselves.
1506
+ * `createChatClient` wraps this implementation for provider-neutral package
1507
+ * entry points. Direct callers can use `callLlm` or `callLlmJson`.
1512
1508
  */
1513
1509
 
1514
1510
  interface LlmMessage {
@@ -1644,8 +1640,8 @@ interface LlmClientOptions {
1644
1640
  * total attempts × `timeoutMs`.
1645
1641
  */
1646
1642
  deadlineMs?: number;
1647
- /** Total provider attempts. Legacy option name; default 3 (1 initial + 2 retries). */
1648
- maxRetries?: number;
1643
+ /** Total provider attempts. Default 3. */
1644
+ maximumAttempts?: number;
1649
1645
  /** Token rates used when the provider omits cost or package pricing does not cover the model. */
1650
1646
  customTokenPricing?: CustomTokenPricing;
1651
1647
  /**
@@ -1693,10 +1689,9 @@ interface LlmClientOptions {
1693
1689
  * name/message/code, then recurses into `error.cause` — undici nests the
1694
1690
  * real socket fault one or more levels under `.cause`.
1695
1691
  *
1696
- * This is THE retry classifier for the package: `callLlm` and
1697
- * `withJudgeRetry` both route through it, so a connection-class error is
1698
- * treated identically whether it surfaces in the HTTP client or a
1699
- * TCloud-backed judge.
1692
+ * This is the retry classifier for the package: `callLlm` and
1693
+ * `withJudgeRetry` both route through it, so connection failures are treated
1694
+ * consistently across transports.
1700
1695
  */
1701
1696
  declare function isTransientLlmError(err: unknown): boolean;
1702
1697
  /** Exponential backoff: 500ms, 1s, 2s, 4s, ... capped at 16s. Attempt is 0-indexed. */
@@ -1797,42 +1792,27 @@ declare class LlmClient {
1797
1792
  }
1798
1793
 
1799
1794
  /**
1800
- * ChatClient the single LLM abstraction analysts call.
1801
- *
1802
- * agent-eval already ships an `LlmClient` (OpenAI-compatible, retry,
1803
- * graceful JSON-schema degrade) and judges that talk to `TCloud`. Two
1804
- * mixed patterns force every analyst author to pick a transport, which
1805
- * couples analyst code to runtime concerns (cli-bridge vs router vs
1806
- * sandbox-sdk) it shouldn't know about.
1807
- *
1808
- * `ChatClient` is one interface every analyst takes via `AnalystContext.chat`.
1809
- * The operator decides at the registry boundary which transport binds
1810
- * to it. Analyst code stays transport-agnostic; swapping production
1811
- * (sandbox-sdk) for local dev (cli-bridge) or tests (mock) is a one-
1812
- * line factory call.
1795
+ * Provider-neutral chat contract for every model call made by agent-eval.
1813
1796
  *
1814
- * Designed to coexist: existing `LlmClient` callers and existing
1815
- * `TCloud`-based judges keep working untouched. New analyst code uses
1816
- * `ChatClient`. When old call sites migrate, they pick up budgeting,
1817
- * cancellation, and unified telemetry for free.
1797
+ * Callers choose the transport at the package boundary with `createChatClient`.
1798
+ * Evaluation code receives canonical requests and results without importing a
1799
+ * provider SDK.
1818
1800
  */
1819
1801
 
1820
1802
  /**
1821
- * Unified chat interface. Mirrors LlmCallRequest/Result so the OpenAI-
1822
- * compatible mental model stays. Two methods: a one-shot `chat()` and
1823
- * an `streamChat()` for future agentic loops (not yet exposed).
1803
+ * Unified chat interface using the package's canonical LLM request and result.
1824
1804
  */
1825
1805
  interface ChatClient {
1826
- /** Display name of the bound transport included in telemetry. */
1806
+ /** Display name of the bound transport, included in telemetry. */
1827
1807
  readonly transport: ChatTransport;
1828
- /** Default model when caller omits — operators bind this per environment. */
1808
+ /** Default model when the caller omits one. */
1829
1809
  readonly defaultModel?: string;
1830
1810
  /** Total provider attempts this transport can make for one chat call. */
1831
1811
  readonly maximumAttempts?: number;
1832
1812
  /** Implementations must enforce `req.maxTokens` when it is present. */
1833
1813
  chat(req: ChatRequest, opts?: ChatCallOpts): Promise<ChatResponse>;
1834
1814
  }
1835
- type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'mock';
1815
+ type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'custom' | 'mock';
1836
1816
  interface ChatRequest extends Omit<LlmCallRequest, 'model'> {
1837
1817
  /** Optional — falls back to ChatClient.defaultModel. */
1838
1818
  model?: string;
@@ -1848,7 +1828,7 @@ interface ChatCallOpts {
1848
1828
  /** Stable provider idempotency key for retries/redrives of one paid call. */
1849
1829
  idempotencyKey?: string;
1850
1830
  }
1851
- type CreateChatClientOpts = RouterTransportOpts | CliBridgeTransportOpts | DirectProviderTransportOpts | SandboxSdkTransportOpts | MockTransportOpts;
1831
+ type CreateChatClientOpts = RouterTransportOpts | CliBridgeTransportOpts | DirectProviderTransportOpts | SandboxSdkTransportOpts | CustomTransportOpts | MockTransportOpts;
1852
1832
  interface BaseTransportOpts {
1853
1833
  defaultModel?: string;
1854
1834
  /** Total provider attempts. Required for opaque transports used in capped runs. */
@@ -1870,15 +1850,18 @@ interface DirectProviderTransportOpts extends BaseTransportOpts {
1870
1850
  apiKey: string;
1871
1851
  }
1872
1852
  /**
1873
- * Sandbox-SDK transport. Provided as a thin pass-through: the caller
1874
- * supplies a callable that mimics LlmClient.chat() against an already-
1875
- * configured Sandbox handle. We don't import the SDK here to keep
1876
- * agent-eval dep-free of @tangle-network/sandbox.
1853
+ * Sandbox-SDK transport. The caller supplies a canonical chat function for an
1854
+ * already-configured Sandbox handle, so agent-eval does not import the SDK.
1877
1855
  */
1878
1856
  interface SandboxSdkTransportOpts extends BaseTransportOpts {
1879
1857
  transport: 'sandbox-sdk';
1880
1858
  chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
1881
1859
  }
1860
+ /** Caller-adapted SDK or transport returning the canonical ChatResponse shape. */
1861
+ interface CustomTransportOpts extends BaseTransportOpts {
1862
+ transport: 'custom';
1863
+ chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
1864
+ }
1882
1865
  /**
1883
1866
  * Mock transport for tests. The handler receives the request and returns
1884
1867
  * whatever the test wants. No retries, no JSON-schema degrade.
@@ -2370,7 +2353,14 @@ type RunTaskFailure = {
2370
2353
  failureClass: Exclude<FailureClass, 'success'>;
2371
2354
  failureMode?: string;
2372
2355
  };
2373
- /** Return task quality, preferring held-out evidence when both scores exist. */
2356
+ /**
2357
+ * Return task quality, preferring held-out evidence when both scores exist.
2358
+ *
2359
+ * RAW: no realness protection is applied. Built on `observedScore` rather
2360
+ * than repeating the split derivation, so only `rollout/reward.ts` reads the
2361
+ * raw fields. Anything that becomes training data must use `trainingScore` or
2362
+ * `trainingReward` instead.
2363
+ */
2374
2364
  declare function runTaskScore(record: RunRecord): number | undefined;
2375
2365
  declare class RunRecordValidationError extends ValidationError {
2376
2366
  readonly path: string;
@@ -2677,8 +2667,6 @@ interface BenchmarkRunnerConfig {
2677
2667
  promptVersion?: string;
2678
2668
  /** Shared ledger for agent and judge calls made by the benchmark. */
2679
2669
  costLedger?: CostLedgerHandle;
2680
- /** Exact maximum provider attempts configured on the supplied TCloud client. */
2681
- tcloudMaximumAttempts?: number;
2682
2670
  }
2683
2671
  interface JudgeInput {
2684
2672
  scenario: Scenario$1;
@@ -2689,11 +2677,8 @@ interface JudgeInput {
2689
2677
  costPhase?: string;
2690
2678
  costTags?: Record<string, string>;
2691
2679
  signal?: AbortSignal;
2692
- /** Exact maximum provider attempts configured on the supplied TCloud client. */
2693
- tcloudMaximumAttempts?: number;
2694
2680
  }
2695
- type JudgeFn = (tc: TCloud, input: JudgeInput) => Promise<JudgeScore$1[]>;
2696
-
2681
+ type JudgeFn = (chat: ChatClient, input: JudgeInput) => Promise<JudgeScore$1[]>;
2697
2682
  interface TestResult {
2698
2683
  name: string;
2699
2684
  passed: boolean;
@@ -2935,13 +2920,8 @@ interface AnalystRunSummary {
2935
2920
  reason?: string;
2936
2921
  findings_count: number;
2937
2922
  latency_ms: number;
2938
- cost_usd: number;
2939
- /**
2940
- * Additive receipt for model usage. Registry-produced summaries populate it
2941
- * even when the analyst emits no findings. `cost_usd` remains the legacy
2942
- * numeric field; inspect `usage.cost` before treating zero as observed.
2943
- */
2944
- usage?: AnalystUsageReceipt;
2923
+ /** Additive model usage and cost provenance for this analyst. */
2924
+ usage: AnalystUsageReceipt;
2945
2925
  /** When `status='failed'`: the error class + message, never the full stack. */
2946
2926
  error?: {
2947
2927
  class: string;
@@ -3003,12 +2983,9 @@ type AnalystRunEvent = {
3003
2983
  /**
3004
2984
  * Typed Ax output for analyst findings.
3005
2985
  *
3006
- * Replaces the legacy `findings:string[]` pattern (where every bullet
3007
- * became a flat-severity `AnalystFinding`) with a structured object
3008
- * array. Ax binds the field as `findings:json[]` so the provider emits
3009
- * native structured output; at the kind-factory boundary we Zod-validate
3010
- * each emitted finding so malformed rows fail loud instead of being
3011
- * silently lifted with default severity.
2986
+ * Ax binds the field as `findings:json[]` so the provider emits native
2987
+ * structured output. At the kind-factory boundary every row is validated
2988
+ * before it becomes an `AnalystFinding`.
3012
2989
  *
3013
2990
  * Why not `f.object().array()` directly in the signature? The Ax
3014
2991
  * signature string `question:string -> findings:json[]` already lets
@@ -3022,31 +2999,7 @@ declare const RawAnalystEvidenceSchema: z.ZodObject<{
3022
2999
  excerpt: z.ZodOptional<z.ZodString>;
3023
3000
  }, z.core.$strict>;
3024
3001
  type RawAnalystEvidence = z.infer<typeof RawAnalystEvidenceSchema>;
3025
- /** Original public schema retained for stored rows and callback contracts. */
3026
3002
  declare const RawAnalystFindingSchema: z.ZodObject<{
3027
- evidence_uri: z.ZodString;
3028
- evidence_excerpt: z.ZodOptional<z.ZodString>;
3029
- severity: z.ZodEnum<{
3030
- info: "info";
3031
- critical: "critical";
3032
- medium: "medium";
3033
- low: "low";
3034
- high: "high";
3035
- }>;
3036
- claim: z.ZodString;
3037
- subject: z.ZodOptional<z.ZodString>;
3038
- confidence: z.ZodNumber;
3039
- rationale: z.ZodOptional<z.ZodString>;
3040
- recommended_action: z.ZodOptional<z.ZodString>;
3041
- }, z.core.$strict>;
3042
- type RawAnalystFinding = z.infer<typeof RawAnalystFindingSchema>;
3043
- /**
3044
- * Canonical plural-evidence contract. The preprocessor accepts the original
3045
- * `evidence_uri` / `evidence_excerpt` pair and normalizes it into one evidence
3046
- * item so persisted rows and older model fixtures remain readable. New output
3047
- * always receives the plural shape.
3048
- */
3049
- declare const CanonicalRawAnalystFindingSchema: z.ZodPreprocess<z.ZodObject<{
3050
3003
  evidence: z.ZodArray<z.ZodObject<{
3051
3004
  uri: z.ZodString;
3052
3005
  excerpt: z.ZodOptional<z.ZodString>;
@@ -3063,8 +3016,8 @@ declare const CanonicalRawAnalystFindingSchema: z.ZodPreprocess<z.ZodObject<{
3063
3016
  confidence: z.ZodNumber;
3064
3017
  rationale: z.ZodOptional<z.ZodString>;
3065
3018
  recommended_action: z.ZodOptional<z.ZodString>;
3066
- }, z.core.$strict>>;
3067
- type CanonicalRawAnalystFinding = z.infer<typeof CanonicalRawAnalystFindingSchema>;
3019
+ }, z.core.$strict>;
3020
+ type RawAnalystFinding = z.infer<typeof RawAnalystFindingSchema>;
3068
3021
 
3069
3022
  /**
3070
3023
  * Analyst-kind factory — the typed way to define trace analysts.
@@ -3073,7 +3026,7 @@ type CanonicalRawAnalystFinding = z.infer<typeof CanonicalRawAnalystFindingSchem
3073
3026
  * and bounded Ax subqueries target one failure-mode lens (failure-mode
3074
3027
  * classification, knowledge gap discovery, knowledge poisoning,
3075
3028
  * self-improvement, ...). Kinds emit findings in the typed
3076
- * `CanonicalRawAnalystFinding` shape via a JSON-array Ax output; the factory
3029
+ * `RawAnalystFinding` shape via a JSON-array Ax output; the factory
3077
3030
  * validates each row with Zod and lifts it into `AnalystFinding[]`.
3078
3031
  *
3079
3032
  * Composition rules:
@@ -3136,7 +3089,7 @@ interface TraceAnalystKindSpec {
3136
3089
  */
3137
3090
  interface TraceAnalystGolden {
3138
3091
  question: string;
3139
- expected: ReadonlyArray<Omit<CanonicalRawAnalystFinding, 'confidence'>>;
3092
+ expected: ReadonlyArray<Omit<RawAnalystFinding, 'confidence'>>;
3140
3093
  }
3141
3094
  interface CreateTraceAnalystKindOpts {
3142
3095
  /** AxAIService bound at registration time. */
@@ -3831,9 +3784,9 @@ declare function ghCliClient(opts?: GhCliClientOptions): AutoPrClient;
3831
3784
  * Domain-agnostic. Each agent provides its own scenarios, judges, and system prompt.
3832
3785
  */
3833
3786
  declare class BenchmarkRunner {
3834
- private tc;
3787
+ private chat;
3835
3788
  private config;
3836
- constructor(tc: TCloud, config: BenchmarkRunnerConfig);
3789
+ constructor(chat: ChatClient, config: BenchmarkRunnerConfig);
3837
3790
  run(scenarios?: Scenario$1[]): Promise<BenchmarkReport$1>;
3838
3791
  }
3839
3792
 
@@ -4210,8 +4163,6 @@ interface AgentDriverConfig {
4210
4163
  productContext?: string;
4211
4164
  /** Shared account for driver-model calls. */
4212
4165
  costLedger?: CostLedgerHandle;
4213
- /** Exact provider attempt count, required when costLedger has a cap. */
4214
- tcloudMaximumAttempts?: number;
4215
4166
  }
4216
4167
  /**
4217
4168
  * AgentDriver — meta-agent that plays a persona against the real product.
@@ -4221,13 +4172,12 @@ interface AgentDriverConfig {
4221
4172
  * the next realistic user message.
4222
4173
  */
4223
4174
  declare class AgentDriver {
4224
- private tc;
4175
+ private chat;
4225
4176
  private client;
4226
4177
  private driverModel;
4227
4178
  private productContext;
4228
4179
  private costLedger;
4229
- private tcloudMaximumAttempts?;
4230
- constructor(tc: TCloud, config: AgentDriverConfig);
4180
+ constructor(chat: ChatClient, config: AgentDriverConfig);
4231
4181
  /**
4232
4182
  * Run a persona through the product.
4233
4183
  *
@@ -4295,8 +4245,6 @@ interface DecideNextUserTurnOpts {
4295
4245
  costLedger?: CostLedgerHandle;
4296
4246
  /** Attribution tags merged into the paid-call receipt. */
4297
4247
  costTags?: Record<string, string>;
4298
- /** Exact provider attempt count, required when costLedger has a cap. */
4299
- tcloudMaximumAttempts?: number;
4300
4248
  }
4301
4249
  /**
4302
4250
  * Decide the simulated user's next turn — the reactive, adversarial
@@ -4305,7 +4253,7 @@ interface DecideNextUserTurnOpts {
4305
4253
  * workspace machinery. Returns the next user message, or the literal "DONE"
4306
4254
  * when the simulated professional would sign off.
4307
4255
  */
4308
- declare function decideNextUserTurn(tc: TCloud, opts: DecideNextUserTurnOpts): Promise<string>;
4256
+ declare function decideNextUserTurn(chat: ChatClient, opts: DecideNextUserTurnOpts): Promise<string>;
4309
4257
 
4310
4258
  interface ExecutorConfig {
4311
4259
  /** System prompt for the agent under test */
@@ -4318,8 +4266,6 @@ interface ExecutorConfig {
4318
4266
  costLedger?: CostLedgerHandle;
4319
4267
  costPhase?: string;
4320
4268
  costTags?: Record<string, string>;
4321
- /** Exact maximum provider attempts configured on the supplied TCloud client. */
4322
- tcloudMaximumAttempts?: number;
4323
4269
  signal?: AbortSignal;
4324
4270
  /** Regex patterns for detecting tool/API calls in responses */
4325
4271
  toolCallPatterns?: RegExp[];
@@ -4338,11 +4284,11 @@ interface ExecutorConfig {
4338
4284
  sleep?: (ms: number) => Promise<void>;
4339
4285
  }
4340
4286
  /**
4341
- * Execute a scenario against an LLM via tcloud.
4287
+ * Execute a scenario against an injected chat client.
4342
4288
  *
4343
4289
  * Runs multi-turn conversation, extracts artifacts, runs judges.
4344
4290
  */
4345
- declare function executeScenario(tc: TCloud, scenario: Scenario$1, config: ExecutorConfig): Promise<ScenarioResult>;
4291
+ declare function executeScenario(chat: ChatClient, scenario: Scenario$1, config: ExecutorConfig): Promise<ScenarioResult>;
4346
4292
 
4347
4293
  /**
4348
4294
  * Backend-integrity guard: distinguish "agent failed" from "eval ran against
@@ -4602,76 +4548,15 @@ type JudgeParseErrorOptions = {
4602
4548
  llmCall?: LlmCallMetadata;
4603
4549
  };
4604
4550
  /**
4605
- * A judge's LLM response could not be parsed into scored dimensions.
4606
- * Thrown instead of fabricating a `{ dimension: 'parse_error', score: 0 }`
4607
- * row — a synthetic zero is indistinguishable from a real low score
4608
- * downstream. Carries the raw response for forensics. Callers (executor,
4609
- * ensemble wrappers) catch this per-judge and record a failed judge.
4551
+ * A judge's model response could not be parsed into scored dimensions.
4552
+ * Carries the raw response and paid-call metadata without fabricating a score.
4610
4553
  */
4611
4554
  declare class JudgeParseError extends JudgeError {
4612
- /** Name of the judge whose response failed to parse. */
4613
4555
  readonly judgeName: string;
4614
- /** The raw (truncated) model response that failed to parse. */
4615
4556
  readonly raw: string;
4616
- /** Paid-call metadata remains available even when the verdict is unusable. */
4617
4557
  readonly llmCall?: LlmCallMetadata;
4618
4558
  constructor(judgeName: string, raw: string, options?: JudgeParseErrorOptions);
4619
4559
  }
4620
- /**
4621
- * Create a domain expert judge with a configurable domain.
4622
- *
4623
- * The judge evaluates professional accuracy and depth.
4624
- *
4625
- * @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
4626
- * Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
4627
- * multi-model panels via `ensembleJudge` (src/judge-panel.ts) — which are
4628
- * pluggable, fail-loud, and drive the campaign/improvement-loop engines.
4629
- */
4630
- declare function createDomainExpertJudge(domain: string): JudgeFn;
4631
- /**
4632
- * Code execution judge — evaluates whether code blocks are valid and runnable.
4633
- *
4634
- * @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
4635
- * Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
4636
- * multi-model panels via `ensembleJudge` (src/judge-panel.ts).
4637
- */
4638
- declare const codeExecutionJudge: JudgeFn;
4639
- /**
4640
- * Coherence judge — evaluates multi-turn consistency and progression.
4641
- *
4642
- * @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
4643
- * Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
4644
- * multi-model panels via `ensembleJudge` (src/judge-panel.ts).
4645
- */
4646
- declare const coherenceJudge: JudgeFn;
4647
- /**
4648
- * Adversarial judge — red-teams agent responses.
4649
- *
4650
- * @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
4651
- * Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
4652
- * multi-model panels via `ensembleJudge` (src/judge-panel.ts).
4653
- */
4654
- declare const adversarialJudge: JudgeFn;
4655
- /**
4656
- * Create a custom judge with a fully custom prompt.
4657
- *
4658
- * @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
4659
- * Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
4660
- * multi-model panels via `ensembleJudge` (src/judge-panel.ts).
4661
- */
4662
- declare function createCustomJudge(name: string, systemPrompt: string, opts?: {
4663
- model?: string;
4664
- temperature?: number;
4665
- maxTokens?: number;
4666
- }): JudgeFn;
4667
- /**
4668
- * Default judge set (domain must be provided for domain expert)
4669
- *
4670
- * @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
4671
- * Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
4672
- * multi-model panels via `ensembleJudge` (src/judge-panel.ts).
4673
- */
4674
- declare function defaultJudges(domain: string): JudgeFn[];
4675
4560
 
4676
4561
  type KnowledgeRequirementCategory = 'user_specific' | 'company_specific' | 'domain_specific' | 'codebase_specific' | 'market_specific' | 'regulatory' | 'tool_api' | 'credential_or_secret' | 'runtime_environment' | 'preference' | 'historical_context';
4677
4562
  type KnowledgeAcquisitionMode = 'ask_user' | 'search_web' | 'query_connector' | 'inspect_repo' | 'run_command' | 'infer_low_confidence' | 'not_available';
@@ -4872,6 +4757,13 @@ interface GateEvidence {
4872
4757
  /** Median per-task USD cost across the baseline runs, for
4873
4758
  * symmetric reporting. */
4874
4759
  medianBaselineCost: number | null;
4760
+ /**
4761
+ * Runs (candidate + baseline) dropped before pairing because the
4762
+ * authenticity gate flagged them as gamed. Surfaced rather than silent: a
4763
+ * promotion decision computed over a shrunken pool has to say by how much,
4764
+ * and a nonzero count here is itself the finding.
4765
+ */
4766
+ realnessGatedRuns: number;
4875
4767
  }
4876
4768
  interface GateDecision$1 {
4877
4769
  /** Final promote/no-promote verdict. */
@@ -5028,6 +4920,13 @@ interface ReleaseConfidenceMetrics {
5028
4920
  domainCounts: Record<string, number>;
5029
4921
  failureClassCounts: Partial<Record<FailureClass, number>>;
5030
4922
  responsibleSurfaceCounts: Record<string, number>;
4923
+ /**
4924
+ * Runs excluded from `passRate` because the authenticity gate flagged them as
4925
+ * gamed. Surfaced, never silent: a release whose pass rate is computed over a
4926
+ * shrunken denominator has to say by how much, or the exclusion is just a
4927
+ * different way of hiding the same runs.
4928
+ */
4929
+ realnessGatedRuns: number;
5031
4930
  }
5032
4931
  interface ReleaseConfidenceScorecard {
5033
4932
  target: string;
@@ -5302,9 +5201,8 @@ interface ContinuousCalibrationResult extends CalibrationResult {
5302
5201
  };
5303
5202
  }
5304
5203
  /**
5305
- * Drop-in superset of `calibrateJudge` that adds continuous-value
5306
- * agreement metrics. The old fields (n, pearson, kappa, mae, worstItems)
5307
- * are preserved unchanged so existing callers continue to work.
5204
+ * Extends `calibrateJudge` with continuous-value agreement metrics while
5205
+ * retaining its base calibration summary.
5308
5206
  */
5309
5207
  declare function calibrateJudgeContinuous(golden: GoldenItem[], candidate: CandidateScore[], opts?: ContinuousAgreementOptions): ContinuousCalibrationResult;
5310
5208
 
@@ -6145,6 +6043,28 @@ declare function printDriverSummary(results: DriverResult[]): void;
6145
6043
  * `outcome.reward` is THE single scalar (null = no verdict exists — a
6146
6044
  * labeled gap, never 0). `outcome.realness_gated` is the anti-Goodhart
6147
6045
  * flag: a gated line must never export as a positive training example.
6046
+ *
6047
+ * That last sentence is enforced here, by `validateRolloutLine`, not merely
6048
+ * documented. Validating `reward` and `realness_gated` independently — each a
6049
+ * well-typed field, their COMBINATION unchecked — is what let a line claiming
6050
+ * `{reward: 0.95, realness_gated: true}` validate clean and walk into every
6051
+ * training export. The relationship between the two IS the invariant, so it is
6052
+ * checked where every other structural claim about a line is checked.
6053
+ *
6054
+ * The invariant is about the OUTCOME, not about one field of it. Zeroing
6055
+ * `reward` while `outcome.metrics` still carried the per-layer scores that
6056
+ * reward was computed from exported the gamed signal anyway, in the dict the
6057
+ * verifiers format reads as its per-rubric scores. So `gateGamedOutcome`
6058
+ * transforms the whole outcome once, at `assertMinted` — the funnel every
6059
+ * minted line passes — and the reward-bearing components are relocated to
6060
+ * `provenance.gated_evidence`, which no exporter projects.
6061
+ *
6062
+ * WHICH checks each door applies is not decided in this file. `./gate-checks`
6063
+ * owns the canonical list and the total per-entry-point policy; the three doors
6064
+ * below (`validateRolloutLine`, `assertRewardGate`, `assertMinted`) each call
6065
+ * `gateErrors` with their declared policy, so a check added to that list applies
6066
+ * here without anyone editing this file, and a check deliberately skipped has to
6067
+ * name itself there.
6148
6068
  */
6149
6069
  declare const ROLLOUT_SCHEMA = "tangle.rollout.v1";
6150
6070
  /** `agent` = a solo evaluation run (no multi-agent topology). */
@@ -6173,6 +6093,14 @@ interface ChatMessage {
6173
6093
  /** Required on role:"tool" — the ChatToolCall this result answers. */
6174
6094
  tool_call_id?: string;
6175
6095
  name?: string;
6096
+ /**
6097
+ * Harbor ATIF `is_copied_context` (RFC 0001 rule 7): this turn was COPIED IN
6098
+ * from another trajectory's context, not produced by the agent on this line.
6099
+ * The RFC makes excluding it from SFT a MUST, and `toSftRows` does — training
6100
+ * on it teaches the model to author text it never authored, and credits this
6101
+ * run for another one's work. Absent = false (authored here).
6102
+ */
6103
+ is_copied_context?: boolean;
6176
6104
  }
6177
6105
  interface ToolDef {
6178
6106
  type: 'function';
@@ -6196,6 +6124,21 @@ interface RolloutStep {
6196
6124
  output?: string;
6197
6125
  status?: 'ok' | 'error';
6198
6126
  durationMs?: number;
6127
+ /**
6128
+ * LLM inferences this span represents. 0 = deterministic dispatch with no
6129
+ * model call — distinct from absent, which means the producer did not track it.
6130
+ */
6131
+ llm_call_count?: number;
6132
+ /** Exact prompt tokenization. Removes the ambiguity of re-tokenizing text at train time. */
6133
+ prompt_token_ids?: number[];
6134
+ /** Exact completion tokenization; aligns index-wise with `logprobs`. */
6135
+ completion_token_ids?: number[];
6136
+ /**
6137
+ * Per-completion-token log probabilities under the sampling policy. Required
6138
+ * for off-policy correction (importance weighting) when the rollout was
6139
+ * generated by a policy other than the one being trained.
6140
+ */
6141
+ logprobs?: number[];
6199
6142
  }
6200
6143
  interface RolloutTask {
6201
6144
  /** Benchmark/suite id (e.g. "swe-bench-verified") or the experiment id. */
@@ -6240,11 +6183,39 @@ interface RolloutOutcome {
6240
6183
  is_truncated: boolean;
6241
6184
  error: string | null;
6242
6185
  /**
6243
- * Anti-Goodhart flag from `RunRecord.outcome.realness.gated`: the run
6244
- * faked its success signal. Reward is forced to 0 at mint time and the
6245
- * line never qualifies for SFT.
6186
+ * Anti-Goodhart flag from `RunRecord.outcome.realness.gated`: the run faked
6187
+ * its success signal. `true` requires `reward` to be 0 or null the
6188
+ * validator rejects the line otherwise — and the line never qualifies for
6189
+ * SFT. Required on the wire: a line that does not state the flag does not
6190
+ * validate, so no producer can dodge the gate by omitting it.
6191
+ *
6192
+ * `true` ALSO requires `metrics` to be empty and `verdict` to be null: the
6193
+ * numbers the reward was computed from are relocated to
6194
+ * `provenance.gated_evidence` by `gateGamedOutcome`. See that function for
6195
+ * why zeroing the scalar alone was not enough.
6246
6196
  */
6247
6197
  realness_gated: boolean;
6198
+ /**
6199
+ * Whether an authenticity SCREEN ever RAN on this reward — a different claim
6200
+ * from `realness_gated`, which is the screen's VERDICT.
6201
+ *
6202
+ * `realness_gated: false` reads as "we looked and nothing fired". A producer
6203
+ * with no screen at all was emitting exactly that, so a never-screened reward
6204
+ * was indistinguishable on the wire from a screened-clean one, and the whole
6205
+ * anti-Goodhart apparatus silently treated the first as the second. The two
6206
+ * claims are now separable:
6207
+ *
6208
+ * - `true` — a screen ran; `realness_gated` is its verdict.
6209
+ * - `false` — the producer declares it HAS no screen (`unscreenedRewardFields`).
6210
+ * `assertMinted` REFUSES such a line when its reward is above
6211
+ * zero: an unscreened positive reward is precisely the signal
6212
+ * the gate exists to qualify, and nothing has qualified it.
6213
+ * - absent — not stated. Pre-unification ledgers land here, as does a
6214
+ * `RunRecord` carrying no `outcome.realness` at all. Absent is
6215
+ * read as "unknown", never as `false` (which would refuse most
6216
+ * of the existing corpus) and never as `true`.
6217
+ */
6218
+ realness_screened?: boolean;
6248
6219
  }
6249
6220
  interface RolloutCostBlock {
6250
6221
  usd: number | null;
@@ -6254,6 +6225,11 @@ interface RolloutCostBlock {
6254
6225
  cache_read: number | null;
6255
6226
  cache_write: number | null;
6256
6227
  wall_s: number | null;
6228
+ /**
6229
+ * Total LLM inferences across the invocation (ATIF `llm_call_count`,
6230
+ * aggregated). Optional and additive: absent = not tracked, never 0.
6231
+ */
6232
+ llm_call_count?: number | null;
6257
6233
  }
6258
6234
  interface RolloutArtifacts {
6259
6235
  patch_path: string | null;
@@ -6261,11 +6237,43 @@ interface RolloutArtifacts {
6261
6237
  /** Source-of-truth transcript pointer (session id / jsonl path) for audit. */
6262
6238
  transcript_ref: string | null;
6263
6239
  }
6240
+ /**
6241
+ * The reward-bearing half of a GATED line's outcome, moved off `outcome` and
6242
+ * parked here verbatim. Diagnostics, never training input — see
6243
+ * `gateGamedOutcome`.
6244
+ */
6245
+ interface GatedEvidence {
6246
+ /** `outcome.metrics` exactly as the producer measured it. */
6247
+ metrics?: Record<string, unknown>;
6248
+ /** `outcome.verdict` verbatim — the judge record that claimed the success. */
6249
+ verdict?: unknown;
6250
+ /**
6251
+ * The per-step fields `tangle.rollout.v1` does not declare, parked here when
6252
+ * the gate projected `steps[]` down to the schema's own key set.
6253
+ *
6254
+ * A per-step reward is training signal exactly like the scalar, and `steps`
6255
+ * rides through `toRewardRows` verbatim — so a gated line was shipping its
6256
+ * step-level credit assignment at full value beside a `reward` of 0.
6257
+ */
6258
+ steps?: unknown;
6259
+ }
6264
6260
  interface RolloutProvenance {
6265
6261
  captured_at: string;
6266
6262
  capture: RolloutCapture;
6267
- /** Present on gap lines: why `messages` could not be recovered. */
6263
+ /**
6264
+ * Why this line is incomplete. Required when `messages` is empty (the
6265
+ * transcript could not be recovered); also set by interchange importers to
6266
+ * name a MISSING LABEL — an imported trajectory carries no verdict, so
6267
+ * `outcome.reward` is null and this says why.
6268
+ */
6268
6269
  gap?: string;
6270
+ /**
6271
+ * Present only on a realness-gated line: the outcome fields the gate
6272
+ * relocated, kept so an auditor can still see WHY the run was gated and what
6273
+ * it claimed. Deliberately OUTSIDE `outcome`, because every training exporter
6274
+ * reads `outcome` and none reads `provenance`.
6275
+ */
6276
+ gated_evidence?: GatedEvidence;
6269
6277
  }
6270
6278
  interface RolloutLine {
6271
6279
  schema: typeof ROLLOUT_SCHEMA;
@@ -6297,6 +6305,82 @@ interface RolloutLine {
6297
6305
  declare function validateRolloutLine(value: unknown): string[];
6298
6306
  declare function assertRolloutLine(value: unknown, context?: string): asserts value is RolloutLine;
6299
6307
  declare function isRolloutLine(value: unknown): value is RolloutLine;
6308
+ /**
6309
+ * Phantom property. `declare const` means it exists only in the type system:
6310
+ * nothing is written at runtime, so a branded line still serializes to exactly
6311
+ * the same JSON as a plain one.
6312
+ */
6313
+ declare const MINTED_ROLLOUT: unique symbol;
6314
+ /**
6315
+ * A minted outcome states the gate verdict — it is not allowed to stay silent —
6316
+ * and, when that verdict is `true`, carries nothing else the reward was derived
6317
+ * from (`gateGamedOutcome` has run).
6318
+ */
6319
+ interface MintedRolloutOutcome extends RolloutOutcome {
6320
+ realness_gated: boolean;
6321
+ }
6322
+ /**
6323
+ * A `RolloutLine` whose reward has been checked against the anti-Goodhart
6324
+ * invariant. The type every training-data exporter takes.
6325
+ *
6326
+ * Why a brand and not just the interface: `RolloutLine` is structural, so any
6327
+ * hand-built object literal of the right shape IS one — which is how a line
6328
+ * declaring `{reward: 0.95, realness_gated: true}` reached the exporters
6329
+ * despite them "only accepting a minted line". The phantom symbol makes the
6330
+ * type nominal: it cannot be produced by writing an object literal, only by
6331
+ * `mintRolloutRows` (which applies the gate), `readRolloutLedger` (which
6332
+ * validates every line off disk), or an explicit, greppable `assertMinted`.
6333
+ *
6334
+ * Belt and braces on purpose. The brand closes first-party call sites at
6335
+ * COMPILE time; `validateRolloutLine` closes data arriving at RUNTIME (ledger
6336
+ * files, foreign imports, JSON from another process) where types are absent.
6337
+ * Neither alone is enough.
6338
+ *
6339
+ * Assignable to `RolloutLine` in one direction only: readers, analysis, and
6340
+ * the ledger writer keep taking the plain type.
6341
+ */
6342
+ type MintedRolloutLine = Omit<RolloutLine, 'outcome'> & {
6343
+ readonly [MINTED_ROLLOUT]: true;
6344
+ outcome: MintedRolloutOutcome;
6345
+ };
6346
+ /**
6347
+ * Promote a line to the type the training exporters accept, applying the
6348
+ * anti-Goodhart gate to the WHOLE outcome on the way through. THE escape hatch
6349
+ * — grep `assertMinted` to enumerate every place a line enters the training
6350
+ * path without coming from mint or a ledger.
6351
+ *
6352
+ * The gate runs HERE, once, rather than at each producer, because this is the
6353
+ * single funnel every minted line passes: `mintRolloutRows` calls it,
6354
+ * `readRolloutLedger` calls it per line off disk, `scrubLines` calls it on the
6355
+ * way out of a release, and a hand-built line has no other door. One
6356
+ * transformation at the funnel means an already-published ledger holding a
6357
+ * gated line with populated `metrics` is RE-GATED when it is read, instead of
6358
+ * being rejected (which would make every such artifact unreadable) or trusted
6359
+ * (which is the leak). Three steps, in this order:
6360
+ *
6361
+ * 1. VALIDATE the schema.
6362
+ * 2. REFUSE every check `GATE_POLICIES.assertMinted` marks `enforce` — today
6363
+ * the reward relationship (which stays a REJECTION: a caller claiming
6364
+ * `{reward: 0.95, realness_gated: true}` is a producer defect and must fail
6365
+ * loudly, since laundering it into `reward: 0` here would hide the
6366
+ * producer) and a positive reward the producer declared it never screened.
6367
+ * 3. TRANSFORM the one check that policy marks `repair` — relocate the
6368
+ * reward's components off `outcome` (`gateGamedOutcome`), so no exporter
6369
+ * can leak them whichever field it reads.
6370
+ *
6371
+ * Step 2 enumerates nothing by hand: a check added to `GATE_CHECKS` is enforced
6372
+ * here the moment its disposition in that policy says so.
6373
+ *
6374
+ * Also normalizes the optional wire flag to an explicit boolean.
6375
+ * `realness_gated` is absent on pre-unification ledgers and absent means "not
6376
+ * flagged" per the schema, so filling it in states a claim the line was already
6377
+ * making, and makes the flag readable on every published row instead of most of
6378
+ * them. `realness_screened` is NOT filled in: absent means "unknown", and
6379
+ * inventing either value there would be the same overclaim this round removed.
6380
+ */
6381
+ declare function assertMinted(value: unknown, context?: string): MintedRolloutLine;
6382
+ /** `assertMinted` over a batch, naming the offending index in the error. */
6383
+ declare function assertMintedLines(values: readonly unknown[], context?: string): MintedRolloutLine[];
6300
6384
 
6301
6385
  /**
6302
6386
  * Pure exporters over `tangle.rollout.v1` lines → the training-data shapes
@@ -6309,8 +6393,40 @@ declare function isRolloutLine(value: unknown): value is RolloutLine;
6309
6393
  * All exporters are pure functions of the lines — filtering (never train on
6310
6394
  * holdout, reward thresholds, the realness gate) happens HERE, on inline
6311
6395
  * labels, no joins.
6396
+ *
6397
+ * Every exporter takes `MintedRolloutLine[]`, not `RolloutLine[]`: the reward
6398
+ * on a minted line has been checked against the anti-Goodhart invariant, and
6399
+ * the brand is what stops a hand-built object literal claiming a positive
6400
+ * reward on a gamed run from being handed to an exporter that copies it
6401
+ * verbatim into training data.
6312
6402
  */
6313
6403
 
6404
+ /**
6405
+ * The gate's two claims, which travel TOGETHER on every emitted row.
6406
+ *
6407
+ * `realness_gated` alone is ambiguous, and the ambiguity is exploitable:
6408
+ * `false` reads as "we screened it and nothing fired", so a producer that has no
6409
+ * screen at all emitted rows indistinguishable from screened-clean ones, and
6410
+ * every consumer of the published dataset read them as clean. The second field
6411
+ * is what separates the two claims, and it only removes the ambiguity if it
6412
+ * reaches the WIRE — for a round it existed on `RolloutOutcome` and on no
6413
+ * exported row shape at all, which left the published rows exactly as ambiguous
6414
+ * as before.
6415
+ *
6416
+ * So there is one helper and every row shape spreads it. A row that states one
6417
+ * claim without the other is not constructible by copying the pattern, and
6418
+ * `exporters.test.ts` walks every emitted shape to prove none does.
6419
+ */
6420
+ interface RealnessLabels {
6421
+ /** The screen's VERDICT: the run faked its success signal. */
6422
+ realness_gated: boolean;
6423
+ /**
6424
+ * Whether a screen RAN at all. `true` = it ran, so `realness_gated` is its
6425
+ * verdict. `false` = the producer declares it has none. `null` = not stated
6426
+ * (pre-unification producers), which is "unknown" and never "clean".
6427
+ */
6428
+ realness_screened: boolean | null;
6429
+ }
6314
6430
  interface TrainingExportOptions {
6315
6431
  /** Include held-out evaluation data in training output. Default false. */
6316
6432
  allowHeldOutTrainingData?: boolean;
@@ -6326,15 +6442,24 @@ interface SftRow {
6326
6442
  candidate_id: string | null;
6327
6443
  instance_id: string;
6328
6444
  reward: number;
6329
- };
6445
+ } & RealnessLabels;
6330
6446
  }
6331
6447
  /**
6332
6448
  * Supervised fine-tune rows: the completed conversation of each qualifying
6333
6449
  * line. Fail-closed filters: trainable split only (never holdout/canary),
6334
- * positive reward, realness-gated lines never qualify, gap lines carry
6335
- * no trainable content.
6336
- */
6337
- declare function toSftRows(lines: RolloutLine[], options?: SftExportOptions): SftRow[];
6450
+ * reward strictly above `minimumQualityExclusive` (default 0), realness-gated
6451
+ * lines never qualify, gap lines carry no trainable content, and
6452
+ * copied-context turns are dropped from the transcript (Harbor ATIF RFC 0001
6453
+ * rule 7 see `ChatMessage.is_copied_context`).
6454
+ *
6455
+ * `realness_gated` is therefore always `false` on an emitted row. It is carried
6456
+ * anyway: an SFT row is a pure imitation target, so the row states its realness
6457
+ * claims instead of making the reader know the format's policy, and carrying
6458
+ * both flags on all four shapes is what lets the release accounting measure
6459
+ * every config with one rule rather than skipping the one whose row shape
6460
+ * happened to omit the field.
6461
+ */
6462
+ declare function toSftRows(lines: MintedRolloutLine[], options?: SftExportOptions): SftRow[];
6338
6463
  interface RewardRow {
6339
6464
  /** First user turn — the task prompt. */
6340
6465
  prompt: string;
@@ -6346,14 +6471,307 @@ interface RewardRow {
6346
6471
  candidate_id: string | null;
6347
6472
  instance_id: string;
6348
6473
  split: RolloutSplit;
6349
- };
6474
+ } & RealnessLabels;
6350
6475
  }
6351
6476
  /**
6352
6477
  * Reward-labeled rows for completed, positive-quality training runs.
6353
6478
  */
6354
- declare function toRewardRows(lines: RolloutLine[], options?: TrainingExportOptions): RewardRow[];
6479
+ declare function toRewardRows(lines: MintedRolloutLine[], options?: TrainingExportOptions): RewardRow[];
6355
6480
  declare function toJsonl(rows: ReadonlyArray<unknown>): string;
6356
6481
 
6482
+ /**
6483
+ * Harbor ATIF-v1.7 interchange — `tangle.rollout.v1` ⇄ Agent Trajectory
6484
+ * Interchange Format.
6485
+ *
6486
+ * ATIF is the portability format (spec:
6487
+ * https://www.harborframework.com/docs/agents/trajectory-format, normative
6488
+ * RFC: harbor-framework/harbor `rfcs/0001-trajectory-format.md`). It sits
6489
+ * BELOW the waist of the rollout hourglass in both directions — export reads
6490
+ * `RolloutLine[]`, import writes `RolloutLine[]` — and it is never a source
6491
+ * of training labels:
6492
+ *
6493
+ * ATIF models NO reward, NO judge verdict, NO task/split coordinates.
6494
+ *
6495
+ * Consequences, both deliberate:
6496
+ * - EXPORT drops `outcome.reward`, `outcome.reward_source` and
6497
+ * `outcome.verdict` entirely. They are not smuggled into `extra`: a
6498
+ * third-party reading our ATIF file must not be able to mistake an
6499
+ * agent-eval judge score for something ATIF sanctioned.
6500
+ * - IMPORT therefore mints UNLABELED lines: `reward: null` (the existing
6501
+ * "null reward is a labeled gap, never 0" semantics), `verdict: null`,
6502
+ * and a `provenance.gap` naming the missing label. An imported
6503
+ * trajectory is not a training example until a judge scores it.
6504
+ *
6505
+ * Everything else we own that ATIF has no field for travels in a namespaced
6506
+ * escrow at `extra.tangle.*`, so our own round-trip is exact while a foreign
6507
+ * reader can ignore it. Fields that neither ATIF nor the escrow can carry
6508
+ * come back explicitly null / fail-closed, never invented.
6509
+ *
6510
+ * THE ESCROW IS NAMESPACED, NOT AUTHENTICATED. Anyone can write
6511
+ * `extra.tangle.*` into a file. So the escrow may restore what a value IS, but
6512
+ * never what a line is ALLOWED to do: `task.split` is forced to `holdout` on
6513
+ * every import regardless of what the document claims, and promoting an
6514
+ * imported trajectory to a trainable split is an explicit, greppable act
6515
+ * (`relabelImportedSplit`) rather than a property of the file. The document
6516
+ * keeps its claim — the claim just is not authority.
6517
+ *
6518
+ * Multi-agent shape differs on purpose. ATIF EMBEDS children in
6519
+ * `subagent_trajectories`; we keep a flat ledger with a normalized
6520
+ * `parent_rollout_id` edge. Export assembles the tree, import flattens it.
6521
+ * `session_id` is RUN-scoped in ATIF, so it carries `run_id` — the coordinate
6522
+ * that is shared by every invocation of one run — not `rollout_id`, which
6523
+ * identifies a single invocation and would split one run across session ids.
6524
+ *
6525
+ * ROUND-TRIPPING IS IDEMPOTENT: `import(export(import(export(x))))` is
6526
+ * byte-identical to `import(export(x))`. Import composes `provenance.gap` as a
6527
+ * de-duplicated ordered set rather than appending, and it emits every
6528
+ * `ChatMessage` with keys in the canonical schema order (role, content,
6529
+ * reasoning_content, tool_calls, tool_call_id, name, is_copied_context), so a
6530
+ * ledger hashed on serialized bytes sees no diff across further passes. The
6531
+ * FIRST import may re-order a producer's keys — that is the canonicalization.
6532
+ *
6533
+ * NOT building a Letta converter. Letta's trajectory-v1 is a strict subset of
6534
+ * what we need from ATIF here — no per-step or aggregate cost, no
6535
+ * multi-agent/subagent structure, no token-id or logprob channel — so a Letta
6536
+ * sink would carry less than this one and add a second format to keep
6537
+ * correct. Decision recorded in docs/rollout.md; do not re-litigate without a
6538
+ * concrete consumer that reads Letta and cannot read ATIF.
6539
+ */
6540
+
6541
+ declare const ATIF_SCHEMA_VERSION = "ATIF-v1.7";
6542
+ /** Gap note on every imported line — ATIF carries no verdict, so nothing is scored. */
6543
+ declare const HARBOR_IMPORT_GAP = "imported from Harbor ATIF; no verdict";
6544
+ type HarborStepSource = 'system' | 'user' | 'agent';
6545
+ interface HarborImageSource {
6546
+ media_type: string;
6547
+ path: string;
6548
+ }
6549
+ interface HarborContentPart {
6550
+ type: 'text' | 'image';
6551
+ text?: string;
6552
+ source?: HarborImageSource;
6553
+ }
6554
+ interface HarborToolCall {
6555
+ tool_call_id: string;
6556
+ function_name: string;
6557
+ /** ATIF requires a decoded JSON object here, unlike our raw argument string. */
6558
+ arguments: Record<string, unknown>;
6559
+ extra?: Record<string, unknown>;
6560
+ }
6561
+ interface HarborSubagentTrajectoryRef {
6562
+ trajectory_id?: string;
6563
+ trajectory_path?: string;
6564
+ /** Informational only since v1.7 — never a resolution key. */
6565
+ session_id?: string;
6566
+ extra?: Record<string, unknown>;
6567
+ }
6568
+ interface HarborObservationResult {
6569
+ source_call_id?: string;
6570
+ content?: string | HarborContentPart[];
6571
+ subagent_trajectory_ref?: HarborSubagentTrajectoryRef[];
6572
+ extra?: Record<string, unknown>;
6573
+ }
6574
+ interface HarborObservation {
6575
+ results: HarborObservationResult[];
6576
+ }
6577
+ interface HarborMetrics {
6578
+ prompt_tokens?: number;
6579
+ completion_tokens?: number;
6580
+ cached_tokens?: number;
6581
+ cost_usd?: number;
6582
+ prompt_token_ids?: number[];
6583
+ completion_token_ids?: number[];
6584
+ logprobs?: number[];
6585
+ extra?: Record<string, unknown>;
6586
+ }
6587
+ interface HarborStep {
6588
+ /** Ordinal, sequential from 1. */
6589
+ step_id: number;
6590
+ timestamp?: string;
6591
+ source: HarborStepSource;
6592
+ model_name?: string;
6593
+ reasoning_effort?: string | number;
6594
+ message: string | HarborContentPart[];
6595
+ reasoning_content?: string;
6596
+ tool_calls?: HarborToolCall[];
6597
+ observation?: HarborObservation;
6598
+ metrics?: HarborMetrics;
6599
+ llm_call_count?: number;
6600
+ is_copied_context?: boolean;
6601
+ extra?: Record<string, unknown>;
6602
+ }
6603
+ interface HarborAgent {
6604
+ name: string;
6605
+ version: string;
6606
+ model_name?: string;
6607
+ /** OpenAI function-calling schema — byte-identical to our `ToolDef`. */
6608
+ tool_definitions?: ToolDef[];
6609
+ extra?: Record<string, unknown>;
6610
+ }
6611
+ interface HarborFinalMetrics {
6612
+ total_prompt_tokens?: number;
6613
+ total_completion_tokens?: number;
6614
+ total_cached_tokens?: number;
6615
+ total_cost_usd?: number;
6616
+ total_steps?: number;
6617
+ extra?: Record<string, unknown>;
6618
+ }
6619
+ interface HarborTrajectory {
6620
+ schema_version: string;
6621
+ session_id?: string;
6622
+ /** Required on embedded subagents; we always set it so lines stay joinable. */
6623
+ trajectory_id?: string;
6624
+ agent: HarborAgent;
6625
+ steps: HarborStep[];
6626
+ notes?: string;
6627
+ final_metrics?: HarborFinalMetrics;
6628
+ continued_trajectory_ref?: string;
6629
+ subagent_trajectories?: HarborTrajectory[];
6630
+ extra?: Record<string, unknown>;
6631
+ }
6632
+ /**
6633
+ * Assemble one episode's flat lines into a single ATIF trajectory tree,
6634
+ * linked by `parent_rollout_id`.
6635
+ *
6636
+ * Reward, verdict and split are NOT emitted (ATIF models none of them); the
6637
+ * split and the rest of the task coordinates survive only in `extra.tangle`.
6638
+ *
6639
+ * We deliberately do NOT synthesize an `observation.subagent_trajectory_ref`
6640
+ * pointing at each child: our ledger records WHICH invocation spawned a
6641
+ * worker, not which STEP did, and attaching the ref to a guessed step would
6642
+ * fabricate a causal claim. Children are embedded in `subagent_trajectories`
6643
+ * (each with the `trajectory_id` the spec requires) and the edge is stated in
6644
+ * the child's escrowed `parent_rollout_id`.
6645
+ *
6646
+ * Throws when the lines are not one tree — use `toHarborTrajectories` for a forest.
6647
+ */
6648
+ declare function toHarborTrajectory(lines: RolloutLine[]): HarborTrajectory;
6649
+ /** Every independent tree in the input, one ATIF document each. */
6650
+ declare function toHarborTrajectories(lines: RolloutLine[]): HarborTrajectory[];
6651
+ interface FromHarborOptions {
6652
+ /** Injected clock for deterministic output when the source carries no capture time. */
6653
+ now?: () => Date;
6654
+ }
6655
+ /**
6656
+ * Flatten an ATIF trajectory tree back into `tangle.rollout.v1` lines, parent
6657
+ * first, each child carrying `parent_rollout_id`.
6658
+ *
6659
+ * Every line comes back UNLABELED: `reward`, `reward_source` and `verdict` are
6660
+ * null and `provenance.gap` says why. ATIF models no verdict, so scoring an
6661
+ * imported trajectory is a judge's job, not this function's. Every line lands
6662
+ * on `holdout` whatever the document claims — see `relabelImportedSplit`.
6663
+ */
6664
+ declare function fromHarborTrajectory(trajectory: HarborTrajectory, options?: FromHarborOptions): RolloutLine[];
6665
+ /**
6666
+ * THE explicit door out of `holdout` for imported lines.
6667
+ *
6668
+ * Import forces `holdout` because a document's own claim about its split is not
6669
+ * evidence — anyone can write `extra.tangle.task.split`. Promoting a file to a
6670
+ * trainable split is an operator's decision about provenance they verified, so
6671
+ * it is a separate, greppable call: `grep relabelImportedSplit` enumerates
6672
+ * every place foreign data was declared trainable, which is exactly the audit
6673
+ * the trusted-escrow version made impossible.
6674
+ *
6675
+ * Returns plain `RolloutLine`s. They still have to pass `assertMinted` (and its
6676
+ * anti-Goodhart check) to reach an exporter — re-labeling a split is not
6677
+ * minting a reward.
6678
+ */
6679
+ declare function relabelImportedSplit(lines: readonly RolloutLine[], split: RolloutSplit): RolloutLine[];
6680
+
6681
+ /**
6682
+ * The two named score derivations every consumer must choose between.
6683
+ *
6684
+ * The anti-Goodhart gate (`outcome.realness.gated`) only holds if it is
6685
+ * impossible to read a run's score WITHOUT deciding whether the gate applies.
6686
+ * A bare `outcome.holdoutScore ?? outcome.searchScore` makes that decision
6687
+ * invisible — and silently answers "no gate", which is the wrong default on
6688
+ * every path that produces training data. So the expression lives here, once,
6689
+ * behind two names that force the caller to state the intent:
6690
+ *
6691
+ * - `trainingScore` / `trainingReward` — GATED. Anything that becomes
6692
+ * training data, or a reward a trainer consumes, uses these.
6693
+ * - `observedScore` — RAW. Analysis, reporting, and reward-hack DETECTION
6694
+ * need the ungated number; that is how a gamed run is visible at all.
6695
+ *
6696
+ * A leaf module on purpose: it imports only the `RunRecord` type, so gate and
6697
+ * reporting code can depend on it without pulling in the trace store that
6698
+ * `mint.ts` needs.
6699
+ */
6700
+
6701
+ /**
6702
+ * Which split's score wins when a record carries both. `'holdout'` is the
6703
+ * canonical "real signal" default; `'search'` exists because some callers
6704
+ * deliberately score on the search split when both are present.
6705
+ */
6706
+ type ScorePreference = 'holdout' | 'search';
6707
+ /** Only the outcome is read, so every accessor here accepts anything carrying one. */
6708
+ type Scored = Pick<RunRecord, 'outcome'>;
6709
+ /** True when the authenticity gate flagged the run as gamed (`realness.gated`). */
6710
+ declare function isRealnessGated(record: Scored): boolean;
6711
+ /**
6712
+ * The RAW score recorded on ONE split, with no cross-split fallback and no
6713
+ * anti-Goodhart gate.
6714
+ *
6715
+ * The narrowest of the three raw readers, and the one every split-scoped
6716
+ * consumer wants: a per-split report, a promotion gate, or a paired comparison
6717
+ * asks "what did this run score on the split I am summarising", and answering
6718
+ * it with the other split's number silently mixes populations. `undefined` =
6719
+ * that split was never scored.
6720
+ *
6721
+ * Same warning as `observedScore`: this INCLUDES runs flagged as gamed. Never
6722
+ * feed it into training data.
6723
+ */
6724
+ declare function observedSplitScore(record: Scored, split: ScorePreference): number | undefined;
6725
+ /**
6726
+ * The RAW split score the run carries, with NO anti-Goodhart gate applied.
6727
+ *
6728
+ * INCLUDES RUNS FLAGGED AS GAMED (`outcome.realness.gated === true`); NEVER
6729
+ * feed this into training data — a fine-tune that sees it learns from gamed
6730
+ * successes. It is exported anyway because analysis, reporting, and
6731
+ * reward-hacking detection legitimately need the ungated number: forcing a
6732
+ * gamed run to 0 collapses the proxy signal toward ground truth and makes a
6733
+ * detector report "clean" on exactly the population that is being gamed.
6734
+ *
6735
+ * Returns `undefined` when the record carries neither score — an unscored run
6736
+ * is a labeled gap, not a measured zero, and each caller picks its own
6737
+ * sentinel (`?? 0`, `?? null`, skip, throw). Non-finite values are returned
6738
+ * as-is; callers that care keep their own `Number.isFinite` guard.
6739
+ */
6740
+ declare function observedScore(record: Scored, prefer?: ScorePreference): number | undefined;
6741
+ /** Which split actually carried the score, or that none did. */
6742
+ type ScoreOrigin = 'holdout' | 'search' | 'unscored';
6743
+ /**
6744
+ * Where `observedScore` / `trainingScore` read their number from — the
6745
+ * provenance label a rollout line's `reward_source` is built from, and the
6746
+ * only supported way to ask "was this run scored at all" without respelling
6747
+ * the field access.
6748
+ */
6749
+ declare function scoreOrigin(record: Scored, prefer?: ScorePreference): ScoreOrigin;
6750
+ /**
6751
+ * The GATED score — the only derivation allowed to reach training data.
6752
+ *
6753
+ * A realness-gated run scores 0 no matter what it claims, so a fine-tune
6754
+ * cannot learn from a gamed success. An unscored run stays `undefined` (a
6755
+ * labeled gap), keeping "we never measured this" distinct from "we measured
6756
+ * zero"; callers that need a number apply their own sentinel.
6757
+ */
6758
+ declare function trainingScore(record: Scored, prefer?: ScorePreference): number | undefined;
6759
+ /**
6760
+ * `{reward, gated}` as written onto a minted `RolloutLine` — `trainingScore`
6761
+ * plus the flag itself, so the gate travels into the exported row and a
6762
+ * downstream filter can drop or down-weight the line.
6763
+ *
6764
+ * An unscored record yields `reward: null`, matching the schema's "no verdict
6765
+ * exists — a labeled gap, never 0" rule. It previously collapsed to 0, which
6766
+ * made a run nobody graded indistinguishable from one graded as a total
6767
+ * failure, and taught any trainer reading the row that the trajectory was bad.
6768
+ * A gated run still yields 0, because that IS a verdict: the gate decided.
6769
+ */
6770
+ declare function trainingReward(record: Scored): {
6771
+ reward: number | null;
6772
+ gated: boolean;
6773
+ };
6774
+
6357
6775
  /**
6358
6776
  * Rollout minting — `tangle.rollout.v1` lines joined from the records the
6359
6777
  * substrate ALREADY keeps. There is no separate rollout store: a rollout
@@ -6366,10 +6784,20 @@ declare function toJsonl(rows: ReadonlyArray<unknown>): string;
6366
6784
  * - preference-pair export → `feedbackTrajectoryToOptimizerRow` (feedback-trajectory.ts)
6367
6785
  * - PRM / reward-model → `reward-model-export.ts`
6368
6786
  *
6369
- * Anti-Goodhart invariant: a run whose `outcome.realness.gated` is true
6370
- * is never exported with a positive reward the gate travels into the
6371
- * training data (`reward` forced to 0, `realness_gated: true`), so a
6372
- * fine-tune cannot learn from gamed successes.
6787
+ * Anti-Goodhart invariant: a run whose `outcome.realness.gated` is true is
6788
+ * never exported with a positive reward OR with any of the numbers that reward
6789
+ * was computed from. The gate travels into the training data (`reward` forced
6790
+ * to 0, `realness_gated: true`) and the whole outcome is transformed by
6791
+ * `gateGamedOutcome` inside `assertMinted` below, which relocates `metrics` and
6792
+ * `verdict` to `provenance.gated_evidence`. Mint returns
6793
+ * `MintedRolloutLine[]`: the brand the training exporters require, which only
6794
+ * this function, `readRolloutLedger`, and an explicit `assertMinted` can mint.
6795
+ *
6796
+ * A record carrying NEITHER split score is REJECTED (`ValidationError`), never
6797
+ * minted at 0 — "nobody graded this" is not the same claim as "graded a total
6798
+ * failure", and a trainer reading 0 learns the second. Lines that already
6799
+ * carry `reward: null` (interchange imports, existing ledgers) remain valid on
6800
+ * the wire; only the RunRecord→line door refuses.
6373
6801
  *
6374
6802
  * Records without spans become labeled GAP LINES (messages: [],
6375
6803
  * provenance.gap) — present in the output AND surfaced in
@@ -6390,14 +6818,11 @@ interface MintRolloutOptions {
6390
6818
  now?: () => Date;
6391
6819
  }
6392
6820
  interface MintRolloutResult {
6393
- rows: RolloutLine[];
6821
+ rows: MintedRolloutLine[];
6394
6822
  /** runIds that had a RunRecord but no spans — emitted as gap lines AND listed here. */
6395
6823
  missingTraces: string[];
6396
6824
  }
6397
- declare function rolloutReward(record: RunRecord): {
6398
- reward: number;
6399
- gated: boolean;
6400
- };
6825
+
6401
6826
  /**
6402
6827
  * Join RunRecords with their traces into canonical rollout lines. Records
6403
6828
  * without spans are emitted as labeled gap lines and reported in
@@ -7370,7 +7795,7 @@ interface OtlpFlatLine {
7370
7795
  }>;
7371
7796
  }
7372
7797
  interface FlattenOtlpOptions {
7373
- /** `'openinference'` (default) mirrors legacy per-span attributes into the
7798
+ /** `'openinference'` (default) maps source per-span attributes into the
7374
7799
  * canonical OpenInference vocabulary the analyst readers consume. `'none'`
7375
7800
  * passes attributes through untouched. */
7376
7801
  attributeVocabulary?: 'openinference' | 'none';
@@ -8217,10 +8642,6 @@ interface LlmCorrectnessCheckerOpts {
8217
8642
  costPhase?: string;
8218
8643
  costTags?: Record<string, string>;
8219
8644
  signal?: AbortSignal;
8220
- /** Exact maximum provider attempts configured on the supplied TCloud client. */
8221
- tcloudMaximumAttempts?: number;
8222
- /** Usage/cost retained by a failed provider response; enables a safe retry. */
8223
- receiptFromError?: (error: Error, attempt: number) => CostReceiptInput | undefined;
8224
8645
  /** Max chars of artifact content sent to the checker. */
8225
8646
  maxContentChars?: number;
8226
8647
  /**
@@ -8253,7 +8674,7 @@ declare function parseCorrectnessResponse(raw: string): {
8253
8674
  * only: a plan, a gesture, or a description of what should be done does not
8254
8675
  * fulfil a requirement — the artifact must BE the deliverable.
8255
8676
  */
8256
- declare function createLlmCorrectnessChecker(tc: TCloud, opts?: LlmCorrectnessCheckerOpts): CorrectnessChecker;
8677
+ declare function createLlmCorrectnessChecker(chat: ChatClient, opts?: LlmCorrectnessCheckerOpts): CorrectnessChecker;
8257
8678
  /**
8258
8679
  * Deterministic `CorrectnessChecker` — the no-LLM counterpart to
8259
8680
  * `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its
@@ -8477,7 +8898,7 @@ interface JudgeConfig<TArtifact, TScenario extends Scenario = Scenario> {
8477
8898
  /** The canonical judge verdict shape — one declaration, shared by campaign
8478
8899
  * judges and the multishot judge runner (which re-exports this type).
8479
8900
  *
8480
- * Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the legacy
8901
+ * Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the
8481
8902
  * multishot runner emits 0-10. Cross-scale comparison must go through
8482
8903
  * `detectScale` (src/campaign/gates/statistical-heldout.ts, used by
8483
8904
  * promotion-policy) — never renormalize a producer's values in place, as
@@ -8657,8 +9078,6 @@ interface CampaignAggregates {
8657
9078
  byScenario: Record<string, ScenarioAggregate>;
8658
9079
  /** Canonical campaign accounting, including worker and judge calls. */
8659
9080
  cost: CostLedgerSummary;
8660
- /** Compatibility alias of `cost.totalCostUsd`. */
8661
- totalCostUsd: number;
8662
9081
  /** Cells whose dispatch completed, including cells whose later judge failed. */
8663
9082
  cellsExecuted: number;
8664
9083
  cellsSkipped: number;
@@ -8669,7 +9088,7 @@ interface CampaignAggregates {
8669
9088
  cellsDispatchFailed?: number;
8670
9089
  /** Present on results that record failure stages. */
8671
9090
  cellsJudgeFailed?: number;
8672
- /** Legacy failures whose stage was not recorded. */
9091
+ /** Failures whose stage could not be classified. */
8673
9092
  cellsUnclassifiedFailed?: number;
8674
9093
  }
8675
9094
  interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scenario> {
@@ -11345,9 +11764,26 @@ interface CostSummary {
11345
11764
  * for the analysis projection.
11346
11765
  */
11347
11766
 
11348
- /** The score the query/compare layer ranks on: holdout when present (the
11349
- * gated number), else search. Execution-only records are valid RunRecords,
11350
- * but cannot participate in score-ranked queries. */
11767
+ /**
11768
+ * The score the query/compare layer ranks on: holdout when present, else
11769
+ * search, with the anti-Goodhart gate applied — a run flagged as gamed reads 0
11770
+ * however high it claims to have scored.
11771
+ *
11772
+ * The gate is load-bearing here and this function did not previously apply it,
11773
+ * despite saying it did. `getBest` is few-shot exemplar selection: whatever it
11774
+ * returns is pasted into the next agent's prompt as an example to imitate.
11775
+ * Ranking on the raw number handed the highest-scoring gamed trajectory to
11776
+ * every subsequent run — propagation through the context window rather than
11777
+ * through a gradient, but propagation all the same.
11778
+ *
11779
+ * Analysis that needs to SEE the inflated number (reward-hack detection,
11780
+ * per-run reporting) reads `observedScore` from `rollout/reward.ts` directly.
11781
+ *
11782
+ * Returns `undefined` for an execution-only record (neither score present):
11783
+ * such rows are valid RunRecords, but cannot participate in score-ranked
11784
+ * queries. A gated run still reads 0 — the gate's verdict is a number, never
11785
+ * a gap.
11786
+ */
11351
11787
  declare function runScore(record: RunRecord): number | undefined;
11352
11788
  interface RunRecordFilter {
11353
11789
  experimentId?: string;
@@ -11382,6 +11818,17 @@ interface CandidateComparison {
11382
11818
  bWins: number;
11383
11819
  ties: number;
11384
11820
  aWins: number;
11821
+ /**
11822
+ * Runs of either candidate excluded from the comparison because the
11823
+ * authenticity gate flagged them (`outcome.realness.gated`).
11824
+ *
11825
+ * Excluded rather than scored 0: `runScore` is gated, so leaving them in
11826
+ * would have entered a gamed run as a silent zero, which reads as "this
11827
+ * candidate failed the scenario" when what happened is "this candidate's
11828
+ * result is not evidence". A non-zero count here is itself the finding — a
11829
+ * comparison drawn over a shrunken scenario set has to say so.
11830
+ */
11831
+ realnessGatedRuns: number;
11385
11832
  }
11386
11833
  /**
11387
11834
  * Backing persistence for `EvalTraceStore`. The in-memory store is the default;
@@ -11420,6 +11867,12 @@ declare class EvalTraceStore {
11420
11867
  * Highest-scoring run for a scenario (optionally restricted to a candidate).
11421
11868
  * Returns null when no run matches. Ties resolve to the earliest-appended run
11422
11869
  * so the result is stable.
11870
+ *
11871
+ * Runs flagged as gamed are DROPPED, not zeroed. The caller's use for this is
11872
+ * few-shot seeding — the returned trajectory becomes an example to copy — so
11873
+ * the same rule as SFT applies: a faked success must not be in the candidate
11874
+ * set at all. When every run for the scenario is gated the honest answer is
11875
+ * `null` (no exemplar), never the least-bad fake.
11423
11876
  */
11424
11877
  getBest(scenarioId: string, opts?: {
11425
11878
  candidateId?: string;
@@ -11430,6 +11883,9 @@ declare class EvalTraceStore {
11430
11883
  * ran a scenario more than once, its best `runScore` for that scenario is
11431
11884
  * used. Throws when there is no paired scenario — an unpaired "comparison" is
11432
11885
  * not one.
11886
+ *
11887
+ * Realness-gated runs are excluded and counted in `realnessGatedRuns`, never
11888
+ * folded in as a zero.
11433
11889
  */
11434
11890
  compareRuns(candidateA: string, candidateB: string): Promise<CandidateComparison>;
11435
11891
  }
@@ -14937,7 +15393,7 @@ interface CampaignStorage {
14937
15393
  write(path: string, content: string | Uint8Array): void;
14938
15394
  /** Append only when the current UTF-8 byte length matches `expectedBytes`.
14939
15395
  * Returns the new length, or undefined when another writer won. */
14940
- append?(path: string, content: string, expectedBytes: number): number | undefined;
15396
+ append(path: string, content: string, expectedBytes: number): number | undefined;
14941
15397
  }
14942
15398
 
14943
15399
  /**
@@ -17187,4 +17643,4 @@ type CachedJudge<TArtifact, TScenario extends Scenario = Scenario> = JudgeConfig
17187
17643
  */
17188
17644
  declare function cachedJudge<TArtifact, TScenario extends Scenario = Scenario>(judge: JudgeConfig<TArtifact, TScenario>, store: VerdictCacheStore, options: CachedJudgeOptions): CachedJudge<TArtifact, TScenario>;
17189
17645
 
17190
- export { AGENT_PROFILE_KINDS, ATTESTATION_ALGORITHM, type ActionExecutionPolicy, type ActionPolicyDecision, type ActionableSideInfo, type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, type AgentEvalErrorCode, type AgentInterfaceProfileLike, type AgentProfileCell, type AgentProfileCellInput, type AgentProfileCellSchemaVersion, AgentProfileCellValidationError, type AgentProfileDimensionValue, type AgentProfileHarness, type AgentProfileJson, type AgentProfileJsonObject, type AgentProfileKind, type AgentProfileRuntimeReceipt, type AgentProfileSource, type AgentProfileSourceInput, type AlignmentOp, type Analyst, type AnalystContext, type AnalystCost, type AnalystFinding, type AnalystHooks, type AnalystInputKind, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type AnalyzeTracesInput, type AnalyzeTracesOptions, type AnalyzeTracesResult, type AnalyzeTracesTurnSnapshot, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, type Artifact$1 as Artifact, type ArtifactCheck, type Artifact as ArtifactCheckArtifact, type ArtifactEventLike, type ArtifactResult, type ArtifactValidator, type AsiSeverity, type AssertCapabilityHeadroomOptions, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, type AttestationProvenance, type AttestationVerification, type AttestedReport, type AutoPrClient, AxGepaSteeringOptimizer, type AxSteeringOptimizerConfig, BENCHMARK_SPLIT_SEED, type BackendDescriptor, BackendIntegrityError, type BackendIntegrityReport, type BaselineOptions, type BaselineReport, BehaviorAssertion, type BehavioralMetrics, type BehavioralTokenSequence, type BenchmarkAdapter, type BenchmarkDatasetItem, type BenchmarkEvaluation, type BenchmarkFamily, type BenchmarkReport$1 as BenchmarkReport, type BenchmarkResponder, BenchmarkRunner, type BenchmarkRunnerConfig, type BenchmarkScenario, type BenchmarkSource, type BenchmarkTaskKind, type BisectOptions, type BisectResult, type BisectStep, type BlendWeights, type BootstrapOptions, type BootstrapResult, BudgetBreachError, BudgetGuard, type BudgetLedgerEntry, type BudgetPolicy, type BudgetSpec, CODING_HARNESSES, type CachedJudge, type CachedJudgeOptions, type CalibrationResult, CallExpectation, CallbackResearcher, type CallbackResearcherOptions, type CampaignFactoryParams, type CampaignIntegrityPolicy, type CampaignRunContext, type CampaignRunOutcome, type CampaignRunner, type CampaignScenario, type CampaignVariant, type CanaryAlert, type CanaryEvaluation, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateComparison, type CandidateScenario, type CandidateScore, type CanonicalRawAnalystFinding, type CapabilityHeadroomOptions, type CapabilityHeadroomResult, type CaptureFetchContext, type CaptureFetchOptions, CaptureIntegrityError, type CausalAttributionReport, type CellVerdict, type ChannelRollup, type ChatCallOpts, type ChatClient, type ChatMessage, type ChatRequest, type ChatResponse, type ChatToolCall, type ChatTransport, type CheckResult, type CliBridgeTransportOpts, type CliffsMagnitude, type ClusterBootstrapInterval, type ClusterSignFlipAlternative, type ClusterSignFlipResult, type ClusteredBinaryCluster, type ClusteredMatchedPair, type ClusteredPairedBinaryOptions, type ClusteredPairedBinaryResult, type ClusteredPairedBinaryStatistics, type CollectedArtifacts, type CommandRunner, type ComparePairedArmsOptions, type CompletionCriterion, type CompletionRequirement, type CompletionVerdict, type ConceptComplexity, type ConceptFinding, type ConceptSpec, type ConceptWeightStrategy, ConfigError, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContinuousAgreement, type ContinuousAgreementOptions, type ContinuousCalibrationResult, type ContractCheckResult, type ContractJudgeOptions, type ContractMetric, type ContractReport, type ContractRule, type ContractRuleKind, type ContractSpan, type ContractVerdict, type ContractViolation, type ControlActionFailureMode, type ControlActionOutcome, type ControlBudget, type ControlContext, type ControlDecision, type ControlEvalResult, type ControlRunResult, type ControlRunToRunRecordOptions, type ControlRuntimeConfig, type ControlRuntimeError, type ControlSeverity, type ControlStep, type ControlStopPolicies, ConvergenceTracker, type CorpusAgreementOptions, type CorpusAgreementPerDimension, type CorpusAgreementReport, type CorpusScoreRecord, type CorrectnessChecker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, type CostChannel, type CostEntry, CostLedger, type CostLedgerEntry, type CostLedgerFilter, type CostLedgerHandle, type CostLedgerOptions, type CostLedgerPersistence, CostLedgerPersistenceError, type CostLedgerSummary, type CostReceipt, CostReceiptCaptureError, type CostReceiptInput, type CostReport, CostReservationExceededError, type CostResult, type CostSummary, CostTracker, type CostUsage, type CounterfactualContext, type CounterfactualMutation, type CounterfactualResult, type CounterfactualRunner, type CreateAnalystAiConfig, type CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateExperimentInput, type CreateSandboxPoolOpts, type CreateTraceAnalystKindOpts, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, type CustomTokenPricing, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, type DataAcquisitionPlan, Dataset, type DatasetDifficulty, type DatasetManifest, type DatasetOverview, type DatasetProvenance, type DatasetScenario, type DatasetSplit, type DecideNextUserTurnOpts, type DefaultAnalystRegistryOptions, type DefaultVerdict, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DescriptionLengthCandidate, type DescriptionLengthConfig, type DescriptionLengthDecision, type DescriptionLengthEvidence, DescriptionLengthGate, type DescriptionLengthRejectionCode, type DetectorEvent, type DetectorSeverity, type DetectorSignal, type DiffPolicy, type DiffScorecardOptions, type DirEntry, type DirectProviderTransportOpts, type Direction, type DiscoverPersonasOptions, type DiscoveredPersona, DockerSandboxDriver, type DriverResult, type DriverState, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, type EProcess, type EProcessOptions, type EProcessState, type EProcessStep, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type ErrorCluster, type ErrorCountPattern, type ErrorStreakOptions, type EvalCampaignOptions, type EvalCampaignResult, type EvalResult, type EvalToolDef, EvalTraceStore, type EventFilter, type EventKind, type EvidenceRef, type EvolutionRound, type ExecutorConfig, type Expectation, type Experiment, type ExperimentPlan, type ExperimentProvenance, type ExperimentRep, type ExperimentResult, type ExperimentStats, type ExperimentStore, ExperimentTracker, type ExperimentTrackerOptions, type ExperimentVerdict, type ExportableSpan, type ExportedRewardModel, type ExtractOptions, type ExtractResult, type ExtractUsageFromSseOptions, type ExtractedUsage, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, type FactorContribution, type FactorialCell, type FailedRun, type FailureClass, type FailureClassification, type FailureContext, type FailureMode, type FailureRule, type FeedbackArtifactType, type FeedbackAttempt, type FeedbackLabel, type FeedbackLabelKind, type FeedbackLabelSource, type FeedbackOptimizerRow, type FeedbackOutcome, type FeedbackPattern, type FeedbackReplayAdapter, type FeedbackReplayResult, type FeedbackSeverity, type FeedbackSplitPolicy, type FeedbackTask, type FeedbackTrajectory, type FeedbackTrajectoryFilter, type FeedbackTrajectoryStore, type FieldDestination, type FileChange, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, type FileSystemRawProviderSinkOptions, FileSystemTraceStore, type FileSystemTraceStoreOptions, type Finding, type FindingSubject, type FindingSubjectKind, type FindingsDiff, FindingsStore, type FlattenOtlpOptions, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type GainDistributionBin, type GainDistributionFigureSpec, type GainDistributionOptions, type GateDecision$1 as GateDecision, type GateEvidence, type GenericSpan, type GhCliClientOptions, type GoldenItem, type GoldenSeverity, type GoldenSpec, HARNESS_NATIVE_MODEL, type HarnessAdapter, type HarnessConfig, type HarnessExperimentConfig, type HarnessExperimentResult, type HarnessIntervention, type HarnessRunRequest, type HarnessRunResult, type HarnessScenario, type HarnessSelection, type HarnessVariant, type HarnessVariantReport, type HeadroomClass, type HeadroomInput, HeldOutGate, type HeldOutGateConfig, type HeldOutGateRejectionCode, type HeldOutPartition, type HiddenCriteriaGrader, type HiddenGradeResult, type HiddenLeak, HoldoutAuditor, HoldoutLockedError, type HttpGithubClientOptions, type HypothesisManifest, type HypothesisResult, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, type ImageData, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, type InMemoryRawProviderSinkOptions, InMemoryTraceStore, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, type InterimReleaseConfidence, type InterimReleaseConfidenceInput, type JudgeConfig$1 as JudgeConfig, JudgeError, type JudgeFamily, type JudgeFleetOptions, type JudgeFn, type JudgeInput, JudgeParseError, type JudgeReplayGateArgs, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, type JudgeRubric, JudgeRunner, type JudgeScore$1 as JudgeScore, type JudgeScoreInput, type JudgeScoresRecord, type JudgeSpan, type JudgeVerdict, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type KnowledgeAcquisitionMode, type KnowledgeBundle, type KnowledgeFallbackPolicy, type KnowledgeFreshness, type KnowledgeImportance, type KnowledgeReadinessReport, type KnowledgeRecommendedAction, type KnowledgeRequirement, type KnowledgeRequirementCategory, type KnowledgeResponsibleSurface, type KnowledgeSensitivity, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, type Layer, type LayerResult, type LayerStatus, type LeaderboardOptions, type LeaderboardRow, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmCallError, type LlmCallMetadata, type LlmCallRequest, type LlmCallResult, LlmClient, type LlmClientOptions, type LlmCorrectnessCheckerOpts, type LlmJsonCall, type LlmJudgeDimension, type LlmJudgeOptions, type LlmMessage, LlmResponseError, type LlmReviewerConfig, LlmRouteAssertionError, type LlmRouteRequirements, type LlmSpan, type LlmSpanOtlpInput, type LlmUsage, LockedJsonlAppender, MODEL_PRICING, type MakeEvalToolsConfig, type MatchResult, type MatchedPair, type MatchedRunRecordPair, type MatcherResult, type MaximumCharge, type McNemarResult, type Measured, type MeasurementPolicy, type MergeOptions, type Message, type MetricSamples, type MetricVerdict, MetricsCollector, type MintRolloutOptions, type MintRolloutResult, type MockTransportOpts, type ModelCostRollup, type ModelPreflight, type ModelSeats, ModelsUnreachableError, type MuffledFinder, type MuffledFinding, MultiLayerVerifier, type MultiToolchainLayerConfig, type Mutator, Mutex, type NoLeakOptions, type NoProgressOptions, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, type Objective, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, type OtelExportConfig, type OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, type OtlpExport, OtlpFileTraceStore, type OtlpFileTraceStoreOptions, type OtlpFlatLine, type OtlpResourceSpans, type OtlpSpan, type OtlpSpanRole, type OtlpSpanRoleInput, type OtlpToRunRecordsOptions, type OtlpTraceRunRecord, type PaidCallResult, type PairArmsOptions, type PairArmsResult, type PairRunRecordsResult, type PairedArmRow, type PairedArmsComparison, type PairedBootstrapOptions, type PairedBootstrapResult, type PairedCorrectness, type PairedEvalueOptions, type PairedEvalueSequence, type PairedEvalueStep, type PairedMetricDelta, type PairedSignTestResult, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, type ParetoFigureSpec, type ParetoPoint, type ParetoResult, type PartitionHeldOutOptions, type PendingCostCall, type PendingCostCallView, type PersistedFinding, type PersonaConfig, type PersonaRigor, type Playbook, type PlaybookEntry, type PoolSlot, type PositionalBiasResult, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, type PreferenceMemoryEntry, type PreflightModelsOptions, type PreflightOutcome, type ProducedProposal, type ProducedState, type ProductBenchmarkArm, type ProductBenchmarkArtifactPaths, type ProductBenchmarkBudgets, type ProductBenchmarkExportOptions, type ProductBenchmarkExportResult, type ProductBenchmarkManifest, type ProductBenchmarkProfileRef, type ProductBenchmarkRecord, type ProductBenchmarkRepoRef, type ProductBenchmarkRunInput, type ProductBenchmarkScenario, type ProductBenchmarkSingleRunExportOptions, type ProductBenchmarkSplit, type ProductBenchmarkSubstrateVersions, type ProductBenchmarkValidationReport, ProductClient, type ProductClientConfig, type ProfileAxisSpec, type ProjectRuntimeTrajectoryEvidenceOptions, type ProjectedOtlpSpan, type PromptHandle, PromptRegistry, type ProportionInterval, type ProposalEventLike, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type ProposeFn, type ProposeInput, type ProposeOutput, type ProposeReviewConfig, type ProposeReviewControlAction, type ProposeReviewControlConfig, type ProposeReviewControlResult, type ProposeReviewControlState, type ProposeReviewReport, type ProposeReviewShot, type ProposedSideEffect, type ProvenanceReader, type ProviderRedactor, type QueryTracesPage, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, type RawAnalystEvidence, type RawAnalystFinding, type RawProviderDirection, type RawProviderEvent, type RawProviderSink, type RawProviderSinkFilter, type RecordRunsOptions, type RedTeamCase, type RedTeamCategory, type RedTeamFinding, type RedTeamPayload, type RedTeamReport, type RedactionReport, type RedactionRule, type ReferenceEquivalenceJudgeInput, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceJudgeResult, type ReferenceEquivalenceScenario, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, type RegistryRunOpts, type ReleaseConfidenceAxis, type ReleaseConfidenceAxisName, type ReleaseConfidenceInput, type ReleaseConfidenceIssue, type ReleaseConfidenceMetrics, type ReleaseConfidenceScorecard, type ReleaseConfidenceStatus, type ReleaseConfidenceThresholds, type ReleaseTraceEvidence, type RenderReleaseReportOptions, type RepeatedActionOptions, ReplayCache, type ReplayCacheEntry, ReplayCacheMissError, type ReplayCacheStats, ReplayError, type ReplayFetchOptions, type RepoRef, type RequirementCheck, type ResearchReport, type ResearchReportCandidate, type ResearchReportDecision, type ResearchReportMethodology, type ResearchReportOptions, type ResearchReportRecommendation, type Researcher, type RetrievalSpan, type Review, type ReviewFn, type ReviewInput, type ReviewMemoryEntry, type ReviewMemoryStore, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, type RewardRow, type RiskDifferenceResult, type RobustnessResult, type RolloutCapture, type RolloutLine, type RolloutRole, type RolloutScrubber, type RolloutSplit, type RolloutStep, type RouteMap, type RoutedField, type RouterTransportOpts, type RubricDimension, type Run, type RunCommandInput, type RunCommandResult, type RunCompleteHook, type RunCompleteHookContext, type RunCostProvenance, RunCritic, type RunCriticOptions, type RunEvidenceMetadata, type RunFilter, RunIntegrityError, type RunIntegrityExpectations, type RunIntegrityIssue, type RunIntegrityIssueCode, type RunIntegrityReport, type RunJudgeMetadata, type RunLayer, type RunOutcome, type RunPaidCallInput, type RunRecord, type RunRecordBackend, type RunRecordFilter, RunRecordValidationError, type RunScore, type RunScoreWeights, type RunSplitTag, type RunStatus, type RunTaskFailure, type RunTerminalOutcome, type RunTokenUsage, type RunTrace, type RuntimeEventLike, type RuntimeResolution, type RuntimeTrajectoryEvidenceProjection, type RuntimeTrajectoryEvidenceSummary, type RuntimeTrajectoryHookEvent, type RuntimeTrajectoryRecord, type RuntimeTrajectoryRunRecord, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_SCHEMA, type SandboxDriver, SandboxHarness, type SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type SandboxResult, type SandboxSdkTransportOpts, type SandboxSpan, type SatisfiedBy, type ScanOptions, type Scenario$1 as Scenario, type ScenarioCost, type ScenarioFile, ScenarioRegistry, type ScenarioResult, type ScoreKnowledgeReadinessOptions, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SearchSpanResult, type SearchTraceResult, type SeatName, type SeatPresetName, SeatUnsetError, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SelfPreferenceResult, type SemanticConceptJudgeInput, type SemanticConceptJudgeOptions, type SemanticConceptJudgeResult, type SequentialDecision, type SerializedRegex, type SeriesConvergenceOptions, type SeriesConvergenceResult, type Severity, type SftExportOptions, type SftRow, type SignTestAlternative, type SignedManifest, type SignedManifestAlgo, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, SkillUsageAnalyst, type SliceOptions, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, type SourceLimits, type Span, type SpanBase, type SpanFilter, type SpanHandle, type SpanKind, type SpanMatchRecord, SpanNotFoundError, type SpanPredicate, type SpanStatus, type SseUsageMode, type SteeringBundle, type SteeringChange, type SteeringDelta, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type SteeringRolePrompt, type StepAttribution, type StopDecision, type StreamingDetector, type SuboptimalCode, type SuboptimalSignal, SubprocessSandboxDriver, type SubprocessSandboxDriverOptions, type SummaryTable, type SummaryTableOptions, type SummaryTableRow, type SupervisorRunReader, type SupervisorRunReport, type SupervisorRunRollup, type SupervisorRunSources, type SupervisorRunTree, type SynthesisReason, type SynthesisTarget, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, type TaskGold, type TaskHeadroom, type TestGradedRunOptions, type TestGradedRunResult, type TestGradedScenario, type TestOutputParser, type TestResult, type TextMatcher, type ThresholdContract, TokenCounter, type TokenSpec, type ToolCallEventLike, type ToolDef, type ToolMatcher, type ToolSpan, type ToolSpanOtlpInput, type ToolStats, type ToolUseMetrics, type ToolUseOptions, type TraceAggregate, type TraceAnalysisStore, type TraceAnalystByteBudgets, type TraceAnalystFilters, type TraceAnalystGolden, type TraceAnalystHookOptions, type TraceAnalystKindSpec, type TraceAnalystSpan, type TraceAnalystSpanKind, type TraceAnalystSpanStatus, type TraceAnalystTraceSummary, type TraceContract, TraceContractBuilder, TraceEmitter, type TraceEmitterOptions, type TraceEvent, TraceFileMissingError, type TraceInsightContext, type TraceInsightFinding, type TraceInsightPanelRole, type TraceInsightPromptInput, type TraceInsightQualityGate, type TraceInsightQuestion, type TraceInsightReadiness, type TraceInsightSuite, type TraceInsightTask, TraceNotFoundError, type TraceStore, type TraceStoreSource, type TraceStoreToOtlpOptions, type TracedAnalystOptions, type TracedJudgeOptions, type TracesToOtlpResult, type Trajectory, type TrajectoryStep, type TreatmentClass, type TreatmentGate, type TreatmentGateInput, type TreatmentGateOptions, type TrialTrace, type Turn, type TurnMetrics, type TurnResult, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, type Unavailable, type UserQuestion, type ValidationContext, ValidationError, type ValidationIssue, type ValidationResult, type VerbosityBiasResult, type Verdict, type VerdictCacheStats, type VerdictCacheStore, type Verification, VerificationError, type VerificationReport, type VerifyContext, type VerifyFn, type VerifyOptions, type ViewSpansResult, type ViewTraceOversized, type ViewTraceResult, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, type WeightedCompositeInput, type WeightedCompositeResult, type WorkerDriverContext, type WorkflowTopology, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, acquisitionPlansForKnowledgeGaps, adversarialJudge, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, index$1 as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, buildWorkerDriverSystemPrompt, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, codeExecutionJudge, cohensD, coherenceJudge, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAnalystAi, createAntiSlopJudge, createChatClient, createCustomJudge, createDefaultReviewer, createDomainExpertJudge, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalystKind, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultJudges, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRetrievalSpan, isRolloutLine, isRunRecord, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalizeScores, notBlocked, objectiveEval, observeAll, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBootstrap, pairedCohensDz, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, index as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resolveModelPricing, resolveSeat, rolloutReward, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runsForScenario, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, typoMutator, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
17646
+ export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, type ActionExecutionPolicy, type ActionPolicyDecision, type ActionableSideInfo, type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, type AgentEvalErrorCode, type AgentInterfaceProfileLike, type AgentProfileCell, type AgentProfileCellInput, type AgentProfileCellSchemaVersion, AgentProfileCellValidationError, type AgentProfileDimensionValue, type AgentProfileHarness, type AgentProfileJson, type AgentProfileJsonObject, type AgentProfileKind, type AgentProfileRuntimeReceipt, type AgentProfileSource, type AgentProfileSourceInput, type AlignmentOp, type Analyst, type AnalystContext, type AnalystCost, type AnalystFinding, type AnalystHooks, type AnalystInputKind, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type AnalyzeTracesInput, type AnalyzeTracesOptions, type AnalyzeTracesResult, type AnalyzeTracesTurnSnapshot, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, type Artifact$1 as Artifact, type ArtifactCheck, type Artifact as ArtifactCheckArtifact, type ArtifactEventLike, type ArtifactResult, type ArtifactValidator, type AsiSeverity, type AssertCapabilityHeadroomOptions, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, type AttestationProvenance, type AttestationVerification, type AttestedReport, type AutoPrClient, AxGepaSteeringOptimizer, type AxSteeringOptimizerConfig, BENCHMARK_SPLIT_SEED, type BackendDescriptor, BackendIntegrityError, type BackendIntegrityReport, type BaselineOptions, type BaselineReport, BehaviorAssertion, type BehavioralMetrics, type BehavioralTokenSequence, type BenchmarkAdapter, type BenchmarkDatasetItem, type BenchmarkEvaluation, type BenchmarkFamily, type BenchmarkReport$1 as BenchmarkReport, type BenchmarkResponder, BenchmarkRunner, type BenchmarkRunnerConfig, type BenchmarkScenario, type BenchmarkSource, type BenchmarkTaskKind, type BisectOptions, type BisectResult, type BisectStep, type BlendWeights, type BootstrapOptions, type BootstrapResult, BudgetBreachError, BudgetGuard, type BudgetLedgerEntry, type BudgetPolicy, type BudgetSpec, CODING_HARNESSES, type CachedJudge, type CachedJudgeOptions, type CalibrationResult, CallExpectation, CallbackResearcher, type CallbackResearcherOptions, type CampaignFactoryParams, type CampaignIntegrityPolicy, type CampaignRunContext, type CampaignRunOutcome, type CampaignRunner, type CampaignScenario, type CampaignVariant, type CanaryAlert, type CanaryEvaluation, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateComparison, type CandidateScenario, type CandidateScore, type CapabilityHeadroomOptions, type CapabilityHeadroomResult, type CaptureFetchContext, type CaptureFetchOptions, CaptureIntegrityError, type CausalAttributionReport, type CellVerdict, type ChannelRollup, type ChatCallOpts, type ChatClient, type ChatMessage, type ChatRequest, type ChatResponse, type ChatToolCall, type ChatTransport, type CheckResult, type CliBridgeTransportOpts, type CliffsMagnitude, type ClusterBootstrapInterval, type ClusterSignFlipAlternative, type ClusterSignFlipResult, type ClusteredBinaryCluster, type ClusteredMatchedPair, type ClusteredPairedBinaryOptions, type ClusteredPairedBinaryResult, type ClusteredPairedBinaryStatistics, type CollectedArtifacts, type CommandRunner, type ComparePairedArmsOptions, type CompletionCriterion, type CompletionRequirement, type CompletionVerdict, type ConceptComplexity, type ConceptFinding, type ConceptSpec, type ConceptWeightStrategy, ConfigError, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContinuousAgreement, type ContinuousAgreementOptions, type ContinuousCalibrationResult, type ContractCheckResult, type ContractJudgeOptions, type ContractMetric, type ContractReport, type ContractRule, type ContractRuleKind, type ContractSpan, type ContractVerdict, type ContractViolation, type ControlActionFailureMode, type ControlActionOutcome, type ControlBudget, type ControlContext, type ControlDecision, type ControlEvalResult, type ControlRunResult, type ControlRunToRunRecordOptions, type ControlRuntimeConfig, type ControlRuntimeError, type ControlSeverity, type ControlStep, type ControlStopPolicies, ConvergenceTracker, type CorpusAgreementOptions, type CorpusAgreementPerDimension, type CorpusAgreementReport, type CorpusScoreRecord, type CorrectnessChecker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, type CostChannel, type CostEntry, CostLedger, type CostLedgerFilter, type CostLedgerHandle, type CostLedgerOptions, type CostLedgerPersistence, CostLedgerPersistenceError, type CostLedgerSummary, type CostReceipt, CostReceiptCaptureError, type CostReceiptInput, type CostReport, CostReservationExceededError, type CostResult, type CostSummary, CostTracker, type CostUsage, type CounterfactualContext, type CounterfactualMutation, type CounterfactualResult, type CounterfactualRunner, type CreateAnalystAiConfig, type CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateExperimentInput, type CreateSandboxPoolOpts, type CreateTraceAnalystKindOpts, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, type CustomTokenPricing, type CustomTransportOpts, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, type DataAcquisitionPlan, Dataset, type DatasetDifficulty, type DatasetManifest, type DatasetOverview, type DatasetProvenance, type DatasetScenario, type DatasetSplit, type DecideNextUserTurnOpts, type DefaultAnalystRegistryOptions, type DefaultVerdict, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DescriptionLengthCandidate, type DescriptionLengthConfig, type DescriptionLengthDecision, type DescriptionLengthEvidence, DescriptionLengthGate, type DescriptionLengthRejectionCode, type DetectorEvent, type DetectorSeverity, type DetectorSignal, type DiffPolicy, type DiffScorecardOptions, type DirEntry, type DirectProviderTransportOpts, type Direction, type DiscoverPersonasOptions, type DiscoveredPersona, DockerSandboxDriver, type DriverResult, type DriverState, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, type EProcess, type EProcessOptions, type EProcessState, type EProcessStep, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type ErrorCluster, type ErrorCountPattern, type ErrorStreakOptions, type EvalCampaignOptions, type EvalCampaignResult, type EvalResult, type EvalToolDef, EvalTraceStore, type EventFilter, type EventKind, type EvidenceRef, type EvolutionRound, type ExecutorConfig, type Expectation, type Experiment, type ExperimentPlan, type ExperimentProvenance, type ExperimentRep, type ExperimentResult, type ExperimentStats, type ExperimentStore, ExperimentTracker, type ExperimentTrackerOptions, type ExperimentVerdict, type ExportableSpan, type ExportedRewardModel, type ExtractOptions, type ExtractResult, type ExtractUsageFromSseOptions, type ExtractedUsage, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, type FactorContribution, type FactorialCell, type FailedRun, type FailureClass, type FailureClassification, type FailureContext, type FailureMode, type FailureRule, type FeedbackArtifactType, type FeedbackAttempt, type FeedbackLabel, type FeedbackLabelKind, type FeedbackLabelSource, type FeedbackOptimizerRow, type FeedbackOutcome, type FeedbackPattern, type FeedbackReplayAdapter, type FeedbackReplayResult, type FeedbackSeverity, type FeedbackSplitPolicy, type FeedbackTask, type FeedbackTrajectory, type FeedbackTrajectoryFilter, type FeedbackTrajectoryStore, type FieldDestination, type FileChange, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, type FileSystemRawProviderSinkOptions, FileSystemTraceStore, type FileSystemTraceStoreOptions, type Finding, type FindingSubject, type FindingSubjectKind, type FindingsDiff, FindingsStore, type FlattenOtlpOptions, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type FromHarborOptions, type GainDistributionBin, type GainDistributionFigureSpec, type GainDistributionOptions, type GateDecision$1 as GateDecision, type GateEvidence, type GenericSpan, type GhCliClientOptions, type GoldenItem, type GoldenSeverity, type GoldenSpec, HARBOR_IMPORT_GAP, HARNESS_NATIVE_MODEL, type HarborAgent, type HarborContentPart, type HarborFinalMetrics, type HarborImageSource, type HarborMetrics, type HarborObservation, type HarborObservationResult, type HarborStep, type HarborStepSource, type HarborSubagentTrajectoryRef, type HarborToolCall, type HarborTrajectory, type HarnessAdapter, type HarnessConfig, type HarnessExperimentConfig, type HarnessExperimentResult, type HarnessIntervention, type HarnessRunRequest, type HarnessRunResult, type HarnessScenario, type HarnessSelection, type HarnessVariant, type HarnessVariantReport, type HeadroomClass, type HeadroomInput, HeldOutGate, type HeldOutGateConfig, type HeldOutGateRejectionCode, type HeldOutPartition, type HiddenCriteriaGrader, type HiddenGradeResult, type HiddenLeak, HoldoutAuditor, HoldoutLockedError, type HttpGithubClientOptions, type HypothesisManifest, type HypothesisResult, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, type ImageData, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, type InMemoryRawProviderSinkOptions, InMemoryTraceStore, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, type InterimReleaseConfidence, type InterimReleaseConfidenceInput, type JudgeConfig$1 as JudgeConfig, JudgeError, type JudgeFamily, type JudgeFleetOptions, type JudgeFn, type JudgeInput, JudgeParseError, type JudgeReplayGateArgs, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, type JudgeRubric, JudgeRunner, type JudgeScore$1 as JudgeScore, type JudgeScoreInput, type JudgeScoresRecord, type JudgeSpan, type JudgeVerdict, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type KnowledgeAcquisitionMode, type KnowledgeBundle, type KnowledgeFallbackPolicy, type KnowledgeFreshness, type KnowledgeImportance, type KnowledgeReadinessReport, type KnowledgeRecommendedAction, type KnowledgeRequirement, type KnowledgeRequirementCategory, type KnowledgeResponsibleSurface, type KnowledgeSensitivity, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, type Layer, type LayerResult, type LayerStatus, type LeaderboardOptions, type LeaderboardRow, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmCallError, type LlmCallMetadata, type LlmCallRequest, type LlmCallResult, LlmClient, type LlmClientOptions, type LlmCorrectnessCheckerOpts, type LlmJsonCall, type LlmJudgeDimension, type LlmJudgeOptions, type LlmMessage, LlmResponseError, type LlmReviewerConfig, LlmRouteAssertionError, type LlmRouteRequirements, type LlmSpan, type LlmSpanOtlpInput, type LlmUsage, LockedJsonlAppender, MODEL_PRICING, type MakeEvalToolsConfig, type MatchResult, type MatchedPair, type MatchedRunRecordPair, type MatcherResult, type MaximumCharge, type McNemarResult, type Measured, type MeasurementPolicy, type MergeOptions, type Message, type MetricSamples, type MetricVerdict, MetricsCollector, type MintRolloutOptions, type MintRolloutResult, type MintedRolloutLine, type MintedRolloutOutcome, type MockTransportOpts, type ModelCostRollup, type ModelPreflight, type ModelSeats, ModelsUnreachableError, type MuffledFinder, type MuffledFinding, MultiLayerVerifier, type MultiToolchainLayerConfig, type Mutator, Mutex, type NoLeakOptions, type NoProgressOptions, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, type Objective, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, type OtelExportConfig, type OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, type OtlpExport, OtlpFileTraceStore, type OtlpFileTraceStoreOptions, type OtlpFlatLine, type OtlpResourceSpans, type OtlpSpan, type OtlpSpanRole, type OtlpSpanRoleInput, type OtlpToRunRecordsOptions, type OtlpTraceRunRecord, type PaidCallResult, type PairArmsOptions, type PairArmsResult, type PairRunRecordsResult, type PairedArmRow, type PairedArmsComparison, type PairedBootstrapOptions, type PairedBootstrapResult, type PairedCorrectness, type PairedEvalueOptions, type PairedEvalueSequence, type PairedEvalueStep, type PairedMetricDelta, type PairedSignTestResult, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, type ParetoFigureSpec, type ParetoPoint, type ParetoResult, type PartitionHeldOutOptions, type PendingCostCall, type PendingCostCallView, type PersistedFinding, type PersonaConfig, type PersonaRigor, type Playbook, type PlaybookEntry, type PoolSlot, type PositionalBiasResult, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, type PreferenceMemoryEntry, type PreflightModelsOptions, type PreflightOutcome, type ProducedProposal, type ProducedState, type ProductBenchmarkArm, type ProductBenchmarkArtifactPaths, type ProductBenchmarkBudgets, type ProductBenchmarkExportOptions, type ProductBenchmarkExportResult, type ProductBenchmarkManifest, type ProductBenchmarkProfileRef, type ProductBenchmarkRecord, type ProductBenchmarkRepoRef, type ProductBenchmarkRunInput, type ProductBenchmarkScenario, type ProductBenchmarkSingleRunExportOptions, type ProductBenchmarkSplit, type ProductBenchmarkSubstrateVersions, type ProductBenchmarkValidationReport, ProductClient, type ProductClientConfig, type ProfileAxisSpec, type ProjectRuntimeTrajectoryEvidenceOptions, type ProjectedOtlpSpan, type PromptHandle, PromptRegistry, type ProportionInterval, type ProposalEventLike, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type ProposeFn, type ProposeInput, type ProposeOutput, type ProposeReviewConfig, type ProposeReviewControlAction, type ProposeReviewControlConfig, type ProposeReviewControlResult, type ProposeReviewControlState, type ProposeReviewReport, type ProposeReviewShot, type ProposedSideEffect, type ProvenanceReader, type ProviderRedactor, type QueryTracesPage, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, type RawAnalystEvidence, type RawAnalystFinding, type RawProviderDirection, type RawProviderEvent, type RawProviderSink, type RawProviderSinkFilter, type RecordRunsOptions, type RedTeamCase, type RedTeamCategory, type RedTeamFinding, type RedTeamPayload, type RedTeamReport, type RedactionReport, type RedactionRule, type ReferenceEquivalenceJudgeInput, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceJudgeResult, type ReferenceEquivalenceScenario, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, type RegistryRunOpts, type ReleaseConfidenceAxis, type ReleaseConfidenceAxisName, type ReleaseConfidenceInput, type ReleaseConfidenceIssue, type ReleaseConfidenceMetrics, type ReleaseConfidenceScorecard, type ReleaseConfidenceStatus, type ReleaseConfidenceThresholds, type ReleaseTraceEvidence, type RenderReleaseReportOptions, type RepeatedActionOptions, ReplayCache, type ReplayCacheEntry, ReplayCacheMissError, type ReplayCacheStats, ReplayError, type ReplayFetchOptions, type RepoRef, type RequirementCheck, type ResearchReport, type ResearchReportCandidate, type ResearchReportDecision, type ResearchReportMethodology, type ResearchReportOptions, type ResearchReportRecommendation, type Researcher, type RetrievalSpan, type Review, type ReviewFn, type ReviewInput, type ReviewMemoryEntry, type ReviewMemoryStore, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, type RewardRow, type RiskDifferenceResult, type RobustnessResult, type RolloutCapture, type RolloutLine, type RolloutRole, type RolloutScrubber, type RolloutSplit, type RolloutStep, type RouteMap, type RoutedField, type RouterTransportOpts, type RubricDimension, type Run, type RunCommandInput, type RunCommandResult, type RunCompleteHook, type RunCompleteHookContext, type RunCostProvenance, RunCritic, type RunCriticOptions, type RunEvidenceMetadata, type RunFilter, RunIntegrityError, type RunIntegrityExpectations, type RunIntegrityIssue, type RunIntegrityIssueCode, type RunIntegrityReport, type RunJudgeMetadata, type RunLayer, type RunOutcome, type RunPaidCallInput, type RunRecord, type RunRecordBackend, type RunRecordFilter, RunRecordValidationError, type RunScore, type RunScoreWeights, type RunSplitTag, type RunStatus, type RunTaskFailure, type RunTerminalOutcome, type RunTokenUsage, type RunTrace, type RuntimeEventLike, type RuntimeResolution, type RuntimeTrajectoryEvidenceProjection, type RuntimeTrajectoryEvidenceSummary, type RuntimeTrajectoryHookEvent, type RuntimeTrajectoryRecord, type RuntimeTrajectoryRunRecord, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_SCHEMA, type SandboxDriver, SandboxHarness, type SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type SandboxResult, type SandboxSdkTransportOpts, type SandboxSpan, type SatisfiedBy, type ScanOptions, type Scenario$1 as Scenario, type ScenarioCost, type ScenarioFile, ScenarioRegistry, type ScenarioResult, type ScoreKnowledgeReadinessOptions, type ScoreOrigin, type ScorePreference, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SearchSpanResult, type SearchTraceResult, type SeatName, type SeatPresetName, SeatUnsetError, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SelfPreferenceResult, type SemanticConceptJudgeInput, type SemanticConceptJudgeOptions, type SemanticConceptJudgeResult, type SequentialDecision, type SerializedRegex, type SeriesConvergenceOptions, type SeriesConvergenceResult, type Severity, type SftExportOptions, type SftRow, type SignTestAlternative, type SignedManifest, type SignedManifestAlgo, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, SkillUsageAnalyst, type SliceOptions, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, type SourceLimits, type Span, type SpanBase, type SpanFilter, type SpanHandle, type SpanKind, type SpanMatchRecord, SpanNotFoundError, type SpanPredicate, type SpanStatus, type SseUsageMode, type SteeringBundle, type SteeringChange, type SteeringDelta, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type SteeringRolePrompt, type StepAttribution, type StopDecision, type StreamingDetector, type SuboptimalCode, type SuboptimalSignal, SubprocessSandboxDriver, type SubprocessSandboxDriverOptions, type SummaryTable, type SummaryTableOptions, type SummaryTableRow, type SupervisorRunReader, type SupervisorRunReport, type SupervisorRunRollup, type SupervisorRunSources, type SupervisorRunTree, type SynthesisReason, type SynthesisTarget, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, type TaskGold, type TaskHeadroom, type TestGradedRunOptions, type TestGradedRunResult, type TestGradedScenario, type TestOutputParser, type TestResult, type TextMatcher, type ThresholdContract, TokenCounter, type TokenSpec, type ToolCallEventLike, type ToolDef, type ToolMatcher, type ToolSpan, type ToolSpanOtlpInput, type ToolStats, type ToolUseMetrics, type ToolUseOptions, type TraceAggregate, type TraceAnalysisStore, type TraceAnalystByteBudgets, type TraceAnalystFilters, type TraceAnalystGolden, type TraceAnalystHookOptions, type TraceAnalystKindSpec, type TraceAnalystSpan, type TraceAnalystSpanKind, type TraceAnalystSpanStatus, type TraceAnalystTraceSummary, type TraceContract, TraceContractBuilder, TraceEmitter, type TraceEmitterOptions, type TraceEvent, TraceFileMissingError, type TraceInsightContext, type TraceInsightFinding, type TraceInsightPanelRole, type TraceInsightPromptInput, type TraceInsightQualityGate, type TraceInsightQuestion, type TraceInsightReadiness, type TraceInsightSuite, type TraceInsightTask, TraceNotFoundError, type TraceStore, type TraceStoreSource, type TraceStoreToOtlpOptions, type TracedAnalystOptions, type TracedJudgeOptions, type TracesToOtlpResult, type Trajectory, type TrajectoryStep, type TreatmentClass, type TreatmentGate, type TreatmentGateInput, type TreatmentGateOptions, type TrialTrace, type Turn, type TurnMetrics, type TurnResult, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, type Unavailable, type UserQuestion, type ValidationContext, ValidationError, type ValidationIssue, type ValidationResult, type VerbosityBiasResult, type Verdict, type VerdictCacheStats, type VerdictCacheStore, type Verification, VerificationError, type VerificationReport, type VerifyContext, type VerifyFn, type VerifyOptions, type ViewSpansResult, type ViewTraceOversized, type ViewTraceResult, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, type WeightedCompositeInput, type WeightedCompositeResult, type WorkerDriverContext, type WorkflowTopology, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, index$1 as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, buildWorkerDriverSystemPrompt, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAnalystAi, createAntiSlopJudge, createChatClient, createDefaultReviewer, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalystKind, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBootstrap, pairedCohensDz, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, index as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resolveModelPricing, resolveSeat, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runsForScenario, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };