@tangle-network/agent-eval 0.125.0 → 0.126.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. package/CHANGELOG.md +62 -35
  2. package/README.md +270 -189
  3. package/dist/analyst/index.d.ts +15 -145
  4. package/dist/analyst/index.js +33 -47
  5. package/dist/analyst/index.js.map +1 -1
  6. package/dist/benchmarks/index.d.ts +45 -162
  7. package/dist/benchmarks/index.js +8 -9
  8. package/dist/campaign/index.d.ts +3674 -5393
  9. package/dist/campaign/index.js +21 -95
  10. package/dist/{chunk-R226UZOI.js → chunk-474LBSOX.js} +2 -2
  11. package/dist/{chunk-HM6V7F3M.js → chunk-FO7HEH76.js} +3 -3
  12. package/dist/chunk-IILEIWGW.js +635 -0
  13. package/dist/chunk-IILEIWGW.js.map +1 -0
  14. package/dist/{chunk-EQUK3RFS.js → chunk-J5SQWP6Y.js} +8 -5
  15. package/dist/chunk-J5SQWP6Y.js.map +1 -0
  16. package/dist/{chunk-W5B3ZGP3.js → chunk-KE2VWPZX.js} +8 -6
  17. package/dist/{chunk-W5B3ZGP3.js.map → chunk-KE2VWPZX.js.map} +1 -1
  18. package/dist/{chunk-DT7OXY3C.js → chunk-LUNF2SEL.js} +538 -851
  19. package/dist/chunk-LUNF2SEL.js.map +1 -0
  20. package/dist/chunk-NGUYT5CI.js +4637 -0
  21. package/dist/chunk-NGUYT5CI.js.map +1 -0
  22. package/dist/{chunk-QFQZ3U3X.js → chunk-OCFJACJU.js} +2 -2
  23. package/dist/{chunk-GID26AN4.js → chunk-P22LJ3Y2.js} +4 -6
  24. package/dist/{chunk-GID26AN4.js.map → chunk-P22LJ3Y2.js.map} +1 -1
  25. package/dist/{chunk-SJT4OBVL.js → chunk-SDPM6554.js} +3 -3
  26. package/dist/{chunk-D5JZ7UDZ.js → chunk-UCLVDLCH.js} +136 -50
  27. package/dist/chunk-UCLVDLCH.js.map +1 -0
  28. package/dist/chunk-VMUENW6F.js +7274 -0
  29. package/dist/chunk-VMUENW6F.js.map +1 -0
  30. package/dist/{chunk-JKDNAOF5.js → chunk-W4L6C2XT.js} +2 -2
  31. package/dist/chunk-WGXIEX7P.js +116 -0
  32. package/dist/chunk-WGXIEX7P.js.map +1 -0
  33. package/dist/{chunk-GRCDRKII.js → chunk-WS3NZZQQ.js} +58 -20
  34. package/dist/chunk-WS3NZZQQ.js.map +1 -0
  35. package/dist/cli.js +3 -3
  36. package/dist/contract/index.d.ts +3220 -3094
  37. package/dist/contract/index.js +173 -42
  38. package/dist/contract/index.js.map +1 -1
  39. package/dist/control.js +2 -3
  40. package/dist/fuzz.d.ts +14 -1
  41. package/dist/fuzz.js +1 -1
  42. package/dist/hosted/index.d.ts +8 -1
  43. package/dist/index.d.ts +71 -687
  44. package/dist/index.js +178 -497
  45. package/dist/index.js.map +1 -1
  46. package/dist/openapi.json +1 -1
  47. package/dist/rl.d.ts +5 -100
  48. package/dist/rl.js +4 -5
  49. package/dist/rl.js.map +1 -1
  50. package/dist/{run-campaign-I3JXKVAK.js → run-campaign-LVFKZCEU.js} +3 -3
  51. package/dist/traces.js +2 -3
  52. package/dist/wire/index.d.ts +14 -1
  53. package/dist/wire/index.js +3 -3
  54. package/docs/campaign-proposers.md +363 -168
  55. package/docs/design/loop-taxonomy.md +142 -190
  56. package/docs/design.md +1 -1
  57. package/docs/distributed-driver.md +8 -11
  58. package/docs/feature-guide.md +20 -19
  59. package/docs/knowledge-readiness.md +2 -5
  60. package/docs/multi-shot-optimization.md +35 -27
  61. package/docs/rollout.md +5 -5
  62. package/package.json +4 -4
  63. package/dist/chunk-A62YMFWA.js +0 -9269
  64. package/dist/chunk-A62YMFWA.js.map +0 -1
  65. package/dist/chunk-A6GT67HT.js +0 -550
  66. package/dist/chunk-A6GT67HT.js.map +0 -1
  67. package/dist/chunk-D5JZ7UDZ.js.map +0 -1
  68. package/dist/chunk-DT7OXY3C.js.map +0 -1
  69. package/dist/chunk-EQUK3RFS.js.map +0 -1
  70. package/dist/chunk-GC4ATIKK.js +0 -317
  71. package/dist/chunk-GC4ATIKK.js.map +0 -1
  72. package/dist/chunk-GRCDRKII.js.map +0 -1
  73. package/dist/chunk-LOW3U7JZ.js +0 -328
  74. package/dist/chunk-LOW3U7JZ.js.map +0 -1
  75. package/dist/chunk-PMITBABE.js +0 -3841
  76. package/dist/chunk-PMITBABE.js.map +0 -1
  77. /package/dist/{chunk-R226UZOI.js.map → chunk-474LBSOX.js.map} +0 -0
  78. /package/dist/{chunk-HM6V7F3M.js.map → chunk-FO7HEH76.js.map} +0 -0
  79. /package/dist/{chunk-QFQZ3U3X.js.map → chunk-OCFJACJU.js.map} +0 -0
  80. /package/dist/{chunk-SJT4OBVL.js.map → chunk-SDPM6554.js.map} +0 -0
  81. /package/dist/{chunk-JKDNAOF5.js.map → chunk-W4L6C2XT.js.map} +0 -0
  82. /package/dist/{run-campaign-I3JXKVAK.js.map → run-campaign-LVFKZCEU.js.map} +0 -0
package/dist/index.d.ts CHANGED
@@ -1153,10 +1153,14 @@ interface CostReceipt extends CostCallBase, CostUsage {
1153
1153
  costUsd: number;
1154
1154
  costUnknown: boolean;
1155
1155
  usageUnknown?: boolean;
1156
+ /** Rates used to estimate cost locally. Absent when cost is provider-reported or unknown. */
1156
1157
  pricing?: {
1157
1158
  inputUsdPerThousand: number;
1159
+ cachedInputUsdPerThousand?: number;
1160
+ cacheWriteUsdPerThousand?: number;
1158
1161
  outputUsdPerThousand: number;
1159
1162
  };
1163
+ /** Cost reported by the provider, not a local token-price calculation. */
1160
1164
  actualCostUsd?: number;
1161
1165
  error?: string;
1162
1166
  }
@@ -1164,20 +1168,27 @@ interface CostReceipt extends CostCallBase, CostUsage {
1164
1168
  type CostLedgerEntry = Omit<CostReceipt, 'status' | 'callId' | 'phase' | 'actor' | 'maximumCostUsd' | 'usageUnknown' | 'pricing' | 'error'>;
1165
1169
  interface CostReceiptInput extends CostUsage {
1166
1170
  model: string;
1171
+ /** Caller-supplied rates for a local estimate when the provider does not report billed cost. */
1172
+ customTokenPricing?: CustomTokenPricing;
1167
1173
  actualCostUsd?: number;
1168
1174
  costUnknown?: boolean;
1169
1175
  usageUnknown?: boolean;
1170
1176
  }
1171
1177
  /** Per-million token rates for a model or endpoint not covered by package pricing. */
1172
1178
  interface CustomTokenPricing {
1179
+ /** Non-cached input tokens. */
1173
1180
  inputUsdPerMillion: number;
1181
+ /** Cache-read tokens. Falls back to the normal input rate when omitted. */
1182
+ cachedInputUsdPerMillion?: number;
1183
+ /** Cache-creation or cache-write tokens. Falls back to the normal input rate when omitted. */
1184
+ cacheWriteUsdPerMillion?: number;
1174
1185
  outputUsdPerMillion: number;
1175
1186
  }
1176
1187
  type MaximumCharge = {
1177
1188
  externallyEnforcedMaximumUsd: number;
1178
1189
  } | ({
1179
1190
  customTokenPricing: CustomTokenPricing;
1180
- } & Pick<CostUsage, 'inputTokens' | 'outputTokens'>) | ({
1191
+ } & Pick<CostUsage, 'inputTokens' | 'outputTokens' | 'cachedTokens' | 'cacheWriteTokens'>) | ({
1181
1192
  model: string;
1182
1193
  } & CostUsage);
1183
1194
  interface RunPaidCallInput<T> {
@@ -1245,6 +1256,8 @@ interface CostLedgerFilter {
1245
1256
  interface CostLedgerWaitOptions {
1246
1257
  /** Maximum time to wait for active provider calls. Default 5 seconds. */
1247
1258
  timeoutMs?: number;
1259
+ /** Wait only for calls matching this attribution filter. */
1260
+ filter?: CostLedgerFilter;
1248
1261
  }
1249
1262
  /** Append-only storage. `append` must atomically reject stale revisions. */
1250
1263
  interface CostLedgerPersistence {
@@ -1339,8 +1352,8 @@ interface CostResult {
1339
1352
  costUnknown: boolean;
1340
1353
  }
1341
1354
  declare function costForUsage(model: string, usage: CostUsage): CostResult;
1342
- /** Price input and output token counts with caller-supplied per-million rates. */
1343
- declare function costForTokenPricing(pricing: CustomTokenPricing, usage: Pick<CostUsage, 'inputTokens' | 'outputTokens'>): number;
1355
+ /** Price token counts with caller-supplied per-million rates. */
1356
+ declare function costForTokenPricing(pricing: CustomTokenPricing, usage: Pick<CostUsage, 'inputTokens' | 'outputTokens' | 'cachedTokens' | 'cacheWriteTokens'>): number;
1344
1357
 
1345
1358
  /**
1346
1359
  * RawProviderSink — first-class persistence for the actual HTTP-level
@@ -3680,112 +3693,6 @@ declare class SkillUsageAnalyst implements Analyst<SkillUsageReport> {
3680
3693
  }
3681
3694
  declare const SKILL_USAGE_ANALYST: SkillUsageAnalyst;
3682
3695
 
3683
- type PolicyEditSchemaVersion = 'policy-edit/v1';
3684
- declare const POLICY_EDIT_AXES: readonly ["carrier", "representation", "budget", "sampling", "output_contract", "tool_contract", "routing", "memory", "agent_profile", "deployment_target"];
3685
- type PolicyEditAxis = (typeof POLICY_EDIT_AXES)[number];
3686
- declare const POLICY_EDIT_TARGET_SURFACES: readonly ["prompt", "tool-contract", "runtime-config", "memory", "agent-profile", "code", "deployment"];
3687
- type PolicyEditTargetSurface = (typeof POLICY_EDIT_TARGET_SURFACES)[number];
3688
- type PolicyEditRisk = 'low' | 'medium' | 'high' | 'unknown';
3689
- type PolicyEditGainDirection = 'increase' | 'decrease';
3690
- type PolicyEditGainUnit = 'absolute' | 'relative' | 'percent' | 'score';
3691
- interface PolicyEditTarget {
3692
- surface: PolicyEditTargetSurface;
3693
- /** Stable path inside the target surface, for example `system-prompt:tools`
3694
- * or `budget.maxTurns`. */
3695
- path?: string;
3696
- /** Optional canonical deployment identity. Store the existing cell, not a
3697
- * local profile shape. */
3698
- agentProfileCell?: AgentProfileCell;
3699
- /** Human label when the path is not enough for a readable audit trail. */
3700
- label?: string;
3701
- }
3702
- type PolicyEditChange = {
3703
- kind: 'text';
3704
- mode: 'append' | 'prepend' | 'replace';
3705
- value: string;
3706
- /** Required when `mode === 'replace'`; exact match only. */
3707
- find?: string;
3708
- } | {
3709
- kind: 'json';
3710
- mode: 'set' | 'merge' | 'remove';
3711
- path: string;
3712
- value?: AgentProfileJson;
3713
- };
3714
- interface PolicyEditExpectedGain {
3715
- /** Metric this edit is expected to move, e.g. `holdout.composite`. */
3716
- metric: string;
3717
- direction: PolicyEditGainDirection;
3718
- /** Positive magnitude in the metric's native units. */
3719
- amount: number;
3720
- unit?: PolicyEditGainUnit;
3721
- rationale?: string;
3722
- }
3723
- interface PolicyEditSource {
3724
- findingIds: string[];
3725
- analystIds: string[];
3726
- evidenceRefs: EvidenceRef[];
3727
- /** Mirrors `AnalystFinding.derived_from_judge`; admission rejects it. */
3728
- derivedFromJudge?: boolean;
3729
- }
3730
- interface PolicyEdit {
3731
- schemaVersion: PolicyEditSchemaVersion;
3732
- editId: string;
3733
- axis: PolicyEditAxis;
3734
- target: PolicyEditTarget;
3735
- change: PolicyEditChange;
3736
- claim: string;
3737
- expectedGain: PolicyEditExpectedGain;
3738
- confidence: number;
3739
- risk: PolicyEditRisk;
3740
- source: PolicyEditSource;
3741
- rationale?: string;
3742
- validationPlan?: string;
3743
- metadata?: Record<string, unknown>;
3744
- }
3745
- declare const POLICY_EDIT_CANDIDATE_RECORD_SCHEMA: "tangle.policy-edit-candidate.v1";
3746
- /** JSON-safe attribution carried with a measured candidate and its scores. */
3747
- interface PolicyEditCandidateRecord {
3748
- schema: typeof POLICY_EDIT_CANDIDATE_RECORD_SCHEMA;
3749
- policyEdit: PolicyEdit;
3750
- }
3751
- type PolicyEditInit = Omit<PolicyEdit, 'schemaVersion' | 'editId'> & {
3752
- schemaVersion?: PolicyEditSchemaVersion;
3753
- editId?: string;
3754
- };
3755
- declare class PolicyEditValidationError extends ValidationError {
3756
- readonly path: string;
3757
- constructor(message: string, path?: string);
3758
- }
3759
- interface FindingToPolicyEditOptions {
3760
- expectedGain?: PolicyEditExpectedGain | ((finding: AnalystFinding) => PolicyEditExpectedGain | null | undefined);
3761
- risk?: PolicyEditRisk | ((finding: AnalystFinding) => PolicyEditRisk);
3762
- defaultAxis?: PolicyEditAxis;
3763
- defaultTargetSurface?: PolicyEditTargetSurface;
3764
- }
3765
- interface PolicyEditAdmissionOptions {
3766
- minScore?: number;
3767
- minExpectedGain?: number;
3768
- allowHighRisk?: boolean;
3769
- requireEvidence?: boolean;
3770
- }
3771
- interface PolicyEditAdmission {
3772
- edit: PolicyEdit;
3773
- decision: 'admit' | 'reject';
3774
- score: number;
3775
- reasons: string[];
3776
- }
3777
- declare function makePolicyEdit(init: PolicyEditInit): PolicyEdit;
3778
- declare function computePolicyEditId(edit: Omit<PolicyEdit, 'editId'> | PolicyEdit): string;
3779
- declare function validatePolicyEdit(input: unknown): PolicyEdit;
3780
- declare function makePolicyEditCandidateRecord(edit: PolicyEdit): PolicyEditCandidateRecord;
3781
- declare function validatePolicyEditCandidateRecord(input: unknown): PolicyEditCandidateRecord;
3782
- declare function isPolicyEdit(input: unknown): input is PolicyEdit;
3783
- declare function policyEditsFromFindings(findings: ReadonlyArray<AnalystFinding>, opts?: FindingToPolicyEditOptions): PolicyEdit[];
3784
- declare function policyEditFromFinding(finding: AnalystFinding, opts?: FindingToPolicyEditOptions): PolicyEdit | null;
3785
- declare function scorePolicyEditReadiness(edit: PolicyEdit, opts?: PolicyEditAdmissionOptions): number;
3786
- declare function admitPolicyEdit(edit: PolicyEdit, opts?: PolicyEditAdmissionOptions): PolicyEditAdmission;
3787
- declare function applyPolicyEditToSurface(surface: unknown, edit: PolicyEdit): unknown;
3788
-
3789
3696
  /**
3790
3697
  * Automated pull-request transports for the production loop.
3791
3698
  *
@@ -8516,98 +8423,8 @@ interface JudgeScore {
8516
8423
  /** Ensemble extras: each surviving judge's per-dimension scores. */
8517
8424
  perJudge?: Record<string, Record<string, number>>;
8518
8425
  }
8519
- /** A tier-4 code surface — a finalized candidate change to the agent's
8520
- * IMPLEMENTATION, not its prompt. Produced by autoresearch (reads codebase +
8521
- * trace findings → opens a worktree). `worktreeRef` locates the candidate;
8522
- * the exact commits, tree, and binary-patch digest identify it. See the
8523
- * improvement-tier table in `docs/design/loop-taxonomy.md`. */
8524
- interface CodeSurface {
8525
- readonly kind: 'code';
8526
- /** Worktree path or git ref holding the candidate code change. This is a
8527
- * mutable locator and is deliberately excluded from content hashes. */
8528
- readonly worktreeRef: string;
8529
- /** Human-readable ref the worktree was forked from. Not identity-bearing. */
8530
- readonly baseRef: string;
8531
- /** Exact commit the candidate was forked from. */
8532
- readonly baseCommit: string;
8533
- /** Exact tree object for `baseCommit`. */
8534
- readonly baseTree: string;
8535
- /** Exact finalized candidate commit. */
8536
- readonly candidateCommit: string;
8537
- /** Exact tree object for `candidateCommit`. */
8538
- readonly candidateTree: string;
8539
- /** Identity of the exact patch artifact. The deployable candidate bundle
8540
- * carries the same descriptor plus its base64-encoded content. */
8541
- readonly patch: {
8542
- readonly format: 'git-diff-binary';
8543
- readonly sha256: `sha256:${string}`;
8544
- readonly byteLength: number;
8545
- };
8546
- /** Human summary of what changed — rendered into the auto-PR body. */
8547
- readonly summary?: string;
8548
- }
8549
- /** The mutable surface a proposer changes. Tiers (see
8550
- * `docs/design/loop-taxonomy.md`):
8551
- * - `string` — tiers 1-2: system-prompt addendum / serialized tool
8552
- * config. Cheap, reversible, text-diffable.
8553
- * - `CodeSurface` — tier 4: an implementation change behind a worktree ref.
8554
- * Tier 3 (knowledge) is owned by agent-knowledge and rides its own adapter,
8555
- * not this type. */
8556
- type MutableSurface = string | CodeSurface;
8557
- /** A non-dominated parent on the GEPA Pareto frontier — a
8558
- * surface that, across the per-scenario objective vectors, no other tried
8559
- * surface beats on every scenario. A candidate worse on the mean composite
8560
- * but uniquely best on one hard scenario is non-dominated and survives here;
8561
- * the composite-best ranking would discard the lesson it carries. The loop
8562
- * computes the frontier across ALL generations and hands it to the proposer so
8563
- * a reflective proposer can combine complementary lessons (GEPA, Agrawal et
8564
- * al., arXiv:2507.19457). See `pareto.ts` (`paretoFrontier`). */
8565
- interface ParetoParent {
8566
- surface: MutableSurface;
8567
- surfaceHash: string;
8568
- /** The objective vector: per-scenario composite (higher is better). The
8569
- * axes the frontier is computed over. */
8570
- objectives: Record<string, number>;
8571
- /** Mean composite across the objective scenarios — the scalar summary used
8572
- * for ordering + display, NOT for dominance. */
8573
- composite: number;
8574
- /** Generation that produced this surface (`-1` for the baseline). */
8575
- generation: number;
8576
- label?: string;
8577
- rationale?: string;
8578
- }
8579
8426
  /** Five-valued verdict taxonomy (MOSS-paper alignment). */
8580
8427
  type GateDecision = 'ship' | 'hold' | 'need_more_work' | 'model_ceiling' | 'arch_ceiling';
8581
- interface GateContext<TArtifact, TScenario extends Scenario> {
8582
- candidateArtifacts: Map<string, TArtifact>;
8583
- baselineArtifacts?: Map<string, TArtifact>;
8584
- /** Candidate (winner) judge scores, keyed by cellId. */
8585
- judgeScores: Map<string, Record<string, JudgeScore>>;
8586
- /** Baseline judge scores, keyed by cellId. SEPARATE from `judgeScores` —
8587
- * baseline + candidate share cellIds (same scenarios), so a single map
8588
- * cannot represent both. A gate computing a holdout delta MUST read
8589
- * candidate from `judgeScores` and baseline from here. */
8590
- baselineJudgeScores?: Map<string, Record<string, JudgeScore>>;
8591
- /** Neutralized-arm judge scores, keyed by cellId — the winner surface with its
8592
- * content footprint-matched-blanked (via a `neutralize` fn). Same scenarios as
8593
- * `judgeScores`. Present ONLY when `runImprovementLoop` was given a `neutralize`
8594
- * function. A placebo gate (`neutralizationGate`) compares this arm's lift
8595
- * against the candidate's to reject decorative wins (lift from footprint, not
8596
- * content). Undefined otherwise. */
8597
- neutralizedJudgeScores?: Map<string, Record<string, JudgeScore>>;
8598
- /** Neutralized-arm artifacts, keyed by cellId. Present alongside
8599
- * `neutralizedJudgeScores`. */
8600
- neutralizedArtifacts?: Map<string, TArtifact>;
8601
- scenarios: TScenario[];
8602
- cost: {
8603
- candidate: number;
8604
- baseline: number;
8605
- };
8606
- /** Shared run spend account and receipt attribution phase. */
8607
- costLedger?: CostLedgerHandle;
8608
- costPhase?: string;
8609
- signal: AbortSignal;
8610
- }
8611
8428
  interface GateResult {
8612
8429
  decision: GateDecision;
8613
8430
  reasons: string[];
@@ -8618,11 +8435,6 @@ interface GateResult {
8618
8435
  }>;
8619
8436
  delta?: number;
8620
8437
  }
8621
- /** Composable promotion gate. */
8622
- interface Gate<TArtifact = unknown, TScenario extends Scenario = Scenario> {
8623
- name: string;
8624
- decide(ctx: GateContext<TArtifact, TScenario>): Promise<GateResult>;
8625
- }
8626
8438
  /** Scoped trace writer handed to each dispatch — every span
8627
8439
  * auto-tagged with the cellId so traces filter cleanly. */
8628
8440
  interface CampaignTraceWriter {
@@ -8753,8 +8565,6 @@ interface GenerationCandidate {
8753
8565
  * "because rationale Z" the audit requires to survive to the result.
8754
8566
  * Present when the proposer returned a `ProposedCandidate`. */
8755
8567
  rationale?: string;
8756
- /** Exact structured cause threaded from the proposer, when available. */
8757
- candidateRecord?: PolicyEditCandidateRecord;
8758
8568
  }
8759
8569
  interface CampaignAggregates {
8760
8570
  byJudge: Record<string, JudgeAggregate>;
@@ -10288,9 +10098,8 @@ declare function securityJudge(id: string, config: HarnessConfig): SandboxJudgeS
10288
10098
  * The `JudgeConfig` contract (src/campaign/types.ts) is deliberately a
10289
10099
  * function, not a fixed LLM-prompt shape: real consumers judge with
10290
10100
  * ensembles, deterministic checks, or one LLM call. `ensembleJudge`
10291
- * (src/judge-panel.ts) covers the multi-model case; `buildAgreementJudge`
10292
- * (src/campaign/distillation) covers the pure-comparator case. `llmJudge`
10293
- * covers the common single-call case the `JudgeConfig` doc-comment names:
10101
+ * (src/judge-panel.ts) covers the multi-model case. `llmJudge` covers the
10102
+ * common single-call case the `JudgeConfig` doc-comment names:
10294
10103
  * one model call against `prompt`, parsed into the canonical `JudgeScore`
10295
10104
  * (`{ dimensions, composite, notes }`) on the campaign [0,1] scale.
10296
10105
  *
@@ -15016,195 +14825,6 @@ interface CampaignStorage {
15016
14825
  append?(path: string, content: string, expectedBytes: number): number | undefined;
15017
14826
  }
15018
14827
 
15019
- /**
15020
- * `openAutoPr` — thin shell-out helper for the `runImprovementLoop` preset's
15021
- * `autoOnPromote: 'pr'` mode. Substitutes for the per-product PR-opening
15022
- * code consumers duplicated 4 times. The PR body includes the campaign's
15023
- * manifest hash, gate verdict, and scorecard summary so reviewers can see
15024
- * exactly what was promoted + why.
15025
- *
15026
- * NOT a deploy mechanism — this only OPENS a PR. The human reviews + merges.
15027
- * The Shape B (`autoOnPromote: 'config'`) live-runtime-mutation path is
15028
- * deferred to Pass B with the full shadow / canary / rollback stack.
15029
- */
15030
-
15031
- interface OpenAutoPrOptions<TArtifact, TScenario extends Scenario> {
15032
- /** Campaign result to attach to the PR. */
15033
- result: CampaignResult<TArtifact, TScenario>;
15034
- /** Gate verdict explaining the promotion. Substrate refuses to open a PR
15035
- * when `gate.decision !== 'ship'` — fails loud. */
15036
- gate: GateResult;
15037
- /** Promoted surface diff — typically the new system prompt addendum or
15038
- * full profile diff. Substrate writes it as the PR body. */
15039
- promotedDiff: string;
15040
- /** GH owner/repo target (e.g., `tangle-network/gtm-agent`). */
15041
- ghOwner: string;
15042
- ghRepo: string;
15043
- /** Branch name for the PR. Default `auto/<manifestHash[:12]>`. */
15044
- branch?: string;
15045
- /** PR title. Default includes manifest hash. */
15046
- title?: string;
15047
- /** Whether to actually open the PR or just dry-run. Default reads
15048
- * `GH_AUTO_PR_TOKEN` env — present = open, absent = dry-run. */
15049
- dryRun?: boolean;
15050
- /** Test seam — substitute `gh pr create` invocation. */
15051
- ghExec?: (args: string[]) => {
15052
- stdout: string;
15053
- stderr: string;
15054
- status: number;
15055
- };
15056
- }
15057
- interface OpenAutoPrResult {
15058
- opened: boolean;
15059
- prUrl?: string;
15060
- dryRun: boolean;
15061
- reason: string;
15062
- }
15063
- /**
15064
- * Open a GitHub PR for a gate-approved surface promotion, attaching the manifest hash, gate verdict, and diff as the PR body.
15065
- */
15066
- declare function openAutoPr<TArtifact, TScenario extends Scenario>(options: OpenAutoPrOptions<TArtifact, TScenario>): OpenAutoPrResult;
15067
-
15068
- /**
15069
- * `runOptimization` — the improvement loop body. Runs N generations: the
15070
- * `SurfaceProposer` proposes K candidate surfaces per generation, each
15071
- * candidate runs a campaign (the measurement), and only a candidate that beats
15072
- * the single global incumbent becomes the next generation's parent.
15073
- * Proposer-agnostic — the same loop runs an evolutionary population mutator
15074
- * (`evolutionaryProposer`) or any reflective / agentic proposer; they differ
15075
- * only in how `propose()` picks candidates.
15076
- *
15077
- * This is `runLoop`'s shape (plan → measure → decide) specialized to surface
15078
- * improvement: `proposer.propose` = plan, `runCampaign` = the measurement
15079
- * (which runs the worker behind `dispatch`), the mean-composite ranking = the
15080
- * validator, `proposer.decide` = the stop check.
15081
- *
15082
- * The gated-promotion shell (`runImprovementLoop`) wraps this with a holdout
15083
- * re-score + release gate + optional PR.
15084
- */
15085
-
15086
- interface RunOptimizationResult<TArtifact, TScenario extends Scenario> {
15087
- generations: Array<{
15088
- record: GenerationRecord;
15089
- surfaces: Array<{
15090
- surfaceHash: string;
15091
- surface: MutableSurface;
15092
- campaign: CampaignResult<TArtifact, TScenario>;
15093
- }>;
15094
- }>;
15095
- winnerSurface: MutableSurface;
15096
- winnerSurfaceHash: string;
15097
- /** Proposer label for the promoted surface. Present when the winning
15098
- * candidate came from a `ProposedCandidate` (a reflective proposer);
15099
- * absent when the winner is the baseline or a bare-surface mutator. */
15100
- winnerLabel?: string;
15101
- /** Proposer rationale for the promoted surface — the "because Z" that
15102
- * motivated the winning change. Survives to `SelfImproveResult` and the
15103
- * emitted provenance record. Absent when the winner is the baseline. */
15104
- winnerRationale?: string;
15105
- baselineCampaign: CampaignResult<TArtifact, TScenario>;
15106
- /** Run-wide spend, including agents, proposers, analysts, and judges. */
15107
- cost: CostLedgerSummary;
15108
- /** The GEPA Pareto frontier across every scored surface (baseline + all
15109
- * generations) by per-scenario objective vector — the non-dominated set.
15110
- * Each generation's `propose()` received the frontier-so-far as
15111
- * `ctx.paretoParents`; this is the final frontier. A surface here that is
15112
- * NOT the winner is uniquely best on some scenario the winner loses on. */
15113
- paretoFrontier: ParetoParent[];
15114
- }
15115
-
15116
- /**
15117
- * `runImprovementLoop` — the gated-promotion shell around the improvement
15118
- * loop body (`runOptimization`). Proposes candidate surfaces via the
15119
- * `SurfaceProposer`, re-scores the winner against the baseline on a
15120
- * holdout set, runs the release gate, and optionally opens a PR.
15121
- *
15122
- * Role vocabulary (see docs/design/loop-taxonomy.md):
15123
- * - PROPOSER = the `SurfaceProposer` (evolutionary GEPA mutator OR
15124
- * reflective analyst). Proposes candidate SURFACES — the
15125
- * worker's system prompt / tool config — NOT conversation
15126
- * turns.
15127
- * - MEASUREMENT= `runCampaign`. Scores one surface by running the worker
15128
- * (via `dispatch`) over scenarios and judging the output.
15129
- * - WORKER = the agent harness in the sandbox, invoked behind the
15130
- * topology-opaque `dispatch` seam — never referenced here.
15131
- *
15132
- * Distinct from `runLoop` in `@tangle-network/agent-runtime`, which is the
15133
- * INNER conversation loop (execution driver ↔ workers in a sandbox). `runImprovementLoop`
15134
- * is the OUTER loop: it improves the surface that those workers run.
15135
- *
15136
- * Hard-refuses unsafe configurations:
15137
- * - `tracing: 'off'` when a proposer is wired (improvement is unattributable)
15138
- * - `autoOnPromote: 'config'` — live mutation is unsupported without
15139
- * isolated deployment, rollback, and independent validation.
15140
- */
15141
-
15142
- interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario> extends RunOptimizationResult<TArtifact, TScenario> {
15143
- baselineOnHoldout: CampaignResult<TArtifact, TScenario>;
15144
- winnerOnHoldout: CampaignResult<TArtifact, TScenario>;
15145
- neutralizedOnHoldout?: CampaignResult<TArtifact, TScenario>;
15146
- neutralizedSurface?: MutableSurface;
15147
- gateResult: Awaited<ReturnType<Gate<TArtifact, TScenario>['decide']>>;
15148
- /** Present iff the loop ran with `holdout: 'deferred'`. When set,
15149
- * `baselineOnHoldout`/`winnerOnHoldout` are the shared EMPTY campaign (zero
15150
- * cells dispatched) and the gate verdict is the forced `'hold'`. */
15151
- holdout?: 'deferred';
15152
- /** Unified baseline→winner surface diff. Computed UNCONDITIONALLY (not only
15153
- * when `autoOnPromote === 'pr'`) so the diff that the gate decided on is
15154
- * always present on the result + in the emitted provenance record. Empty
15155
- * string when winner == baseline (no change to diff). */
15156
- promotedDiff: string;
15157
- prResult?: ReturnType<typeof openAutoPr>;
15158
- }
15159
-
15160
- /**
15161
- * `gepaProposer` — a reflective `SurfaceProposer` for prompt-tier surfaces.
15162
- * Each generation it reflects on the prior best candidate's per-scenario
15163
- * scores + weakest dimensions, asks an LLM to propose targeted rewrites of
15164
- * the current surface, and returns them as the next population.
15165
- *
15166
- * Maps onto the GEPA paper (Agrawal et al., arXiv:2507.19457):
15167
- * - *Reflection*: each generation reflects on the best parent's weakest
15168
- * dimensions + per-scenario top/bottom scores to propose targeted rewrites.
15169
- * - *Pareto frontier*: `runOptimization` maintains the non-dominated set of
15170
- * surfaces across generations (per-scenario objective vectors) and supplies
15171
- * it as `ctx.paretoParents`. A surface uniquely best on one hard scenario
15172
- * survives even when its mean composite is lower.
15173
- * - *Combine complementary lessons*: when the frontier has >1 member, the
15174
- * first population slot is a merge of those parents' strengths (one LLM
15175
- * call citing each parent's winning scenarios). Toggle via `combineParents`.
15176
- * Dominance is computed by the package-canonical `paretoFrontier` (`pareto.ts`).
15177
- *
15178
- * Optional `constraints` move structured-doc guards into the proposer
15179
- * (preserve H2 section headings, cap sentence-level edits) — useful when
15180
- * the surface IS a structured procedure like a SKILL.md / runbook /
15181
- * judge rubric. When `constraints` is omitted, behavior is unchanged.
15182
- *
15183
- * The proposer is surface-agnostic — any string surface in any consumer opts
15184
- * in by selecting it. Reuses the generic reflection primitive
15185
- * (`buildReflectionPrompt` / `parseReflectionResponse`) and the router client.
15186
- *
15187
- * Earns its keep where there is real per-instance signal (which the
15188
- * dimensional + per-scenario evidence + the `LabeledScenarioStore` flywheel
15189
- * now provide). For thin-signal surfaces it degrades to plain reflection.
15190
- * On generation 0 (no history) it reflects on the current surface against
15191
- * the mutation primitives alone.
15192
- */
15193
-
15194
- interface GepaProposerConstraints {
15195
- /** H2 section headings that MUST appear unchanged in every candidate.
15196
- * When set, the proposer auto-detects current H2s if this is empty AND
15197
- * rejects any candidate that drops or renames a preserved heading.
15198
- * Use when the surface is a structured doc (SKILL.md, runbook,
15199
- * sectioned system prompt, judge rubric). */
15200
- preserveSections?: string[];
15201
- /** Maximum sentence-level edits per candidate vs the parent surface.
15202
- * Rejection threshold = maxSentenceEdits × 2 (counts adds + removes).
15203
- * Inspired by SkillOpt's edit-budget as a "textual learning rate."
15204
- * Cap prevents an LLM rewrite from overwriting useful prior rules. */
15205
- maxSentenceEdits?: number;
15206
- }
15207
-
15208
14828
  interface BenchmarkRunOptions<TPayload = unknown, TArtifact = string> {
15209
14829
  adapter: BenchmarkAdapter<BenchmarkDatasetItem<TPayload>, TPayload, TArtifact>;
15210
14830
  respond: BenchmarkResponder<TPayload, TArtifact>;
@@ -16741,290 +16361,6 @@ declare function withOtelPipeline(opts?: OtelPipelineOptions): OtelPipelineHandl
16741
16361
  */
16742
16362
  declare function isOtelConfigured(): boolean;
16743
16363
 
16744
- /**
16745
- * Traced analyst wrapper — instruments `analyzeTraces` with spans so the
16746
- * analyst's internal model turns appear in the trace tree. Also wraps each
16747
- * actor turn callback with a span.
16748
- *
16749
- * The wrapper records the Ax turn loop at its public boundaries:
16750
- * 1. A parent span for the entire analyst run.
16751
- * 2. Per-turn child spans from the `onTurn` callback (captures code,
16752
- * output size, error status).
16753
- * 3. Summary attributes on the parent (total turns, usage, findings).
16754
- */
16755
-
16756
- interface TracedAnalystOptions {
16757
- /** TraceEmitter for span emission. */
16758
- emitter: TraceEmitter;
16759
- /** Parent span id. If omitted, uses emitter stack. */
16760
- parentSpanId?: string;
16761
- }
16762
- /**
16763
- * Run `analyzeTraces` wrapped in a parent span with per-turn child spans.
16764
- */
16765
- declare function tracedAnalyzeTraces(input: AnalyzeTracesInput, options: AnalyzeTracesOptions, traceOpts: TracedAnalystOptions): Promise<AnalyzeTracesResult>;
16766
-
16767
- /**
16768
- * Traced judge wrappers — instruments every LLM call inside the judge
16769
- * ensemble with child spans so OTEL sinks see per-judge latency, model,
16770
- * token counts, and score dimensions.
16771
- *
16772
- * The ensemble parent span groups all individual judge spans; each judge
16773
- * gets its own child span with model + score as attributes.
16774
- */
16775
-
16776
- interface TracedJudgeOptions {
16777
- /** TraceEmitter to emit spans into. */
16778
- emitter: TraceEmitter;
16779
- /** Parent span id for the ensemble. If omitted, uses the emitter stack. */
16780
- parentSpanId?: string;
16781
- }
16782
- /**
16783
- * Wrap a single JudgeFn so its LLM call emits a traced span.
16784
- */
16785
- declare function traceJudge(judge: JudgeFn, judgeName: string, opts: TracedJudgeOptions): JudgeFn;
16786
- /**
16787
- * Wrap an array of JudgeFns with tracing, running them inside an ensemble
16788
- * parent span. Returns a single function that calls all judges and merges
16789
- * their scores.
16790
- */
16791
- declare function traceJudgeEnsemble(judges: JudgeFn[], judgeNames: string[], opts: TracedJudgeOptions): JudgeFn;
16792
-
16793
- /**
16794
- * Gold scenarios for teacher→student distillation. The TEACHER is an
16795
- * expensive workflow (e.g. the 70-agent skill audit) whose verdicts are
16796
- * frozen as gold labels; the STUDENT is a cheap single-shot analyst whose
16797
- * prompt GEPA optimizes toward reproducing those labels.
16798
- *
16799
- * A `GoldScenario` is a `Scenario` (the substrate's input contract) carrying
16800
- * an OPAQUE `input` (what the student sees) and an OPAQUE `label` (the gold
16801
- * verdict the student's output is scored against). Both are typed `unknown`
16802
- * here: this module is domain-agnostic — it distills ANY analyst against ANY
16803
- * gold JSONL. The agreement comparator (see `agreement-judge.ts`) is what
16804
- * knows the label's shape.
16805
- *
16806
- * Loading + splitting are DETERMINISTIC and LLM-free: the gold set is the
16807
- * fixed ground truth, never regenerated here.
16808
- */
16809
-
16810
- /** A held gold record: opaque student-input + opaque gold-label, carried as a
16811
- * substrate `Scenario` so it flows through `runCampaign` unchanged. */
16812
- interface GoldScenario<TInput = unknown, TLabel = unknown> extends Scenario {
16813
- kind: 'gold';
16814
- /** What the student analyst is shown (rendered into its user prompt). */
16815
- input: TInput;
16816
- /** The teacher's gold verdict — the target the student's output is scored
16817
- * against by the agreement judge. NEVER shown to the student. */
16818
- label: TLabel;
16819
- }
16820
- /** Read a gold JSONL (one `{scenarioId|id, input, label, split?}` per line) into
16821
- * `GoldScenario[]`. Deterministic, no LLM. Blank lines are skipped; a line
16822
- * missing an id, `input`, or `label` throws (a silent skip would corrupt the
16823
- * split silently — fail loud on a malformed gold set). */
16824
- declare function loadGoldScenarios<TInput = unknown, TLabel = unknown>(jsonlPath: string): GoldScenario<TInput, TLabel>[];
16825
- /** Parse gold JSONL text directly (no fs). Exported so tests + in-memory
16826
- * callers exercise the same parse path as {@link loadGoldScenarios}. */
16827
- declare function parseGoldJsonl<TInput = unknown, TLabel = unknown>(text: string, sourceLabel?: string): GoldScenario<TInput, TLabel>[];
16828
- interface SplitGoldOptions {
16829
- /** Every Nth scenario (0-based index) goes to the TEST/holdout split; the
16830
- * rest train. Default 4 ⇒ a 25% holdout. Ignored for any scenario that
16831
- * carries an explicit `split:` tag (that is honored verbatim). */
16832
- testEveryNth?: number;
16833
- }
16834
- interface GoldSplit<TInput, TLabel> {
16835
- /** Training scenarios — the optimization pool the proposer searches over. */
16836
- train: GoldScenario<TInput, TLabel>[];
16837
- /** Held-out scenarios — kept OUT of training; scored only at the gate. */
16838
- test: GoldScenario<TInput, TLabel>[];
16839
- }
16840
- /** Deterministic train/test split. A scenario tagged `split:train|test` is
16841
- * routed by that tag; the rest fall to a modulo split (`index % testEveryNth
16842
- * === 0 ⇒ test`). Pure — same input always yields the same split, so a gold
16843
- * set's holdout is stable across runs (a shuffled split would let a lucky
16844
- * seed flatter the gate). */
16845
- declare function splitGold<TInput, TLabel>(scenarios: GoldScenario<TInput, TLabel>[], options?: SplitGoldOptions): GoldSplit<TInput, TLabel>;
16846
-
16847
- /**
16848
- * Agreement judge for teacher→student distillation. Scores a STUDENT artifact
16849
- * (the cheap analyst's produced label) against the GoldScenario's gold label
16850
- * (the teacher's verdict). The score IS the distillation objective: 1.0 means
16851
- * the student reproduced the teacher exactly, 0.0 means total disagreement.
16852
- *
16853
- * The comparison function is INJECTED (`compareLabels`) so the judge is
16854
- * domain-agnostic — distilling a skill-audit analyst, a triage analyst, or any
16855
- * other student is a one-line comparator swap. A default `fieldAgreement`
16856
- * comparator is provided for the common case: a flat verdict object with
16857
- * categorical and array fields.
16858
- *
16859
- * Everything here is PURE + unit-testable — no LLM. (The student spends tokens
16860
- * producing the artifact; scoring it against frozen gold does not.)
16861
- */
16862
-
16863
- /** What an injected comparator returns: a [0,1] composite plus the per-field
16864
- * (per-dimension) agreement breakdown the GEPA proposer reflects on to learn
16865
- * WHICH part of the verdict the student is getting wrong. */
16866
- interface AgreementResult {
16867
- /** Overall agreement in [0,1]. */
16868
- score: number;
16869
- /** Per-dimension agreement in [0,1] — keyed by field/aspect name. The
16870
- * reflective proposer surfaces the weakest of these as the lever to fix. */
16871
- dimensions: Record<string, number>;
16872
- }
16873
- /** Compare a produced label against a gold label → agreement. Injected so the
16874
- * judge is domain-agnostic. */
16875
- type CompareLabels<TProduced = unknown, TLabel = unknown> = (produced: TProduced, gold: TLabel) => AgreementResult;
16876
- interface BuildAgreementJudgeOptions<TProduced = unknown, TLabel = unknown> {
16877
- /** Judge name surfaced in `CampaignResult.aggregates.byJudge` + the gate. */
16878
- name?: string;
16879
- /** The agreement function — produced student label vs gold teacher label. */
16880
- compareLabels: CompareLabels<TProduced, TLabel>;
16881
- /** Dimension keys the judge declares up-front (for `JudgeConfig.dimensions`).
16882
- * When omitted, the dimensions present on the first scored result are used
16883
- * for display only; the composite is unaffected. */
16884
- dimensionKeys?: string[];
16885
- /** Only score `gold`-kind scenarios. Default true — a mixed campaign won't
16886
- * mis-apply the agreement judge to non-gold scenarios. */
16887
- goldOnly?: boolean;
16888
- }
16889
- /** Build a `JudgeConfig` that scores a produced student artifact against the
16890
- * scenario's gold label. Conforms to the substrate `JudgeConfig` contract:
16891
- * `score({artifact, scenario, signal}) => JudgeScore`. The `composite` is the
16892
- * comparator's `score`; `dimensions` carries its per-field breakdown plus the
16893
- * scalar `agreement` so a single-dimension consumer still sees the number. */
16894
- declare function buildAgreementJudge<TProduced, TInput, TLabel>(options: BuildAgreementJudgeOptions<TProduced, TLabel>): JudgeConfig<TProduced, GoldScenario<TInput, TLabel>>;
16895
- interface FieldAgreementSpec {
16896
- /** Categorical fields — scored exact-match (1 if equal, else 0). Compared
16897
- * with `===` after `JSON`-normalizing so `true`/`'high'`/`3` all work. */
16898
- categorical?: string[];
16899
- /** Array fields — scored by Jaccard overlap (|A∩B| / |A∪B|). Two empty
16900
- * arrays agree perfectly (1.0). Order-insensitive; elements compared by
16901
- * their `JSON.stringify`. */
16902
- array?: string[];
16903
- }
16904
- /** Default comparator: average per-field agreement over a flat verdict object.
16905
- * Categorical fields score exact-match; array fields score set-overlap
16906
- * (Jaccard). The composite is the unweighted mean across all declared fields,
16907
- * so missing a single boolean (e.g. `public_leak_risk`) costs `1/nFields` of
16908
- * the score — the leak-detection lever the audit cares about is a real,
16909
- * non-trivial fraction of the objective, not rounding noise.
16910
- *
16911
- * Pure. A field absent from BOTH produced + gold is treated as agreeing
16912
- * (both undefined ⇒ 1.0); a field present in only one side disagrees. */
16913
- declare function fieldAgreement<TProduced extends Record<string, unknown>, TLabel>(spec: FieldAgreementSpec): CompareLabels<TProduced, TLabel>;
16914
-
16915
- /**
16916
- * `runDistillation` — the teacher→student distillation loop. COMPOSES existing
16917
- * substrate primitives; reimplements none of them:
16918
- *
16919
- * - DRIVER = `gepaProposer` (reflective prompt optimizer)
16920
- * - LOOP = `runImprovementLoop` (outer: optimize → holdout re-score → gate)
16921
- * - MEASUREMENT = `runCampaign` (inside the loop) scoring the student
16922
- * - JUDGE = `buildAgreementJudge` — student label vs gold teacher label
16923
- * - GATE = caller-supplied (`heldOutGate` / `defaultProductionGate`)
16924
- * - STUDENT = a cheap single-shot analyst whose system prompt is the
16925
- * `MutableSurface` GEPA mutates; it calls the LLM through
16926
- * `createChatClient` and emits a JSON label.
16927
- *
16928
- * The surface IS the student's system prompt. Each generation GEPA rewrites it;
16929
- * `dispatchWithSurface` renders {surface + scenario.input} into a chat request,
16930
- * calls the (cheap) model, parses the produced JSON label, and returns it as
16931
- * the artifact. The agreement judge scores that label against the gold label.
16932
- *
16933
- * `autoOnPromote: 'none'` is FORCED — the loop never opens a PR; the caller
16934
- * (the `distill` CLI) decides what to do with the winning prompt.
16935
- */
16936
-
16937
- /** Render the student's prompt from {current surface, scenario input}. The
16938
- * surface is the system prompt; the scenario input is the user turn. Override
16939
- * to inject few-shot framing or a JSON-schema reminder. */
16940
- type RenderStudentPrompt<TInput> = (args: {
16941
- surface: string;
16942
- input: TInput;
16943
- scenarioId: string;
16944
- }) => ChatRequest['messages'];
16945
- /** Parse the model's raw text into a typed produced label. Throws on
16946
- * unparseable output — a thrown dispatch is recorded as a failed cell (never
16947
- * silently scored 0), which is the honest signal that the prompt isn't
16948
- * emitting valid JSON yet. */
16949
- type ParseStudentLabel<TProduced> = (rawContent: string, scenarioId: string) => TProduced;
16950
- interface RunDistillationOptions<TProduced, TInput, TLabel> {
16951
- /** The student analyst's INITIAL system prompt — the baseline surface GEPA
16952
- * searches from. */
16953
- baselinePrompt: string;
16954
- /** Training scenarios (the optimization pool). */
16955
- train: GoldScenario<TInput, TLabel>[];
16956
- /** Held-out scenarios — kept OUT of training; scored only at the gate. */
16957
- holdout: GoldScenario<TInput, TLabel>[];
16958
- /** Transport for BOTH the student (cheap model) and the GEPA reflection
16959
- * (the optimizer model). The student calls it via `createChatClient`. */
16960
- llm: CreateChatClientOpts;
16961
- /** Router transport the GEPA proposer reflects through. `gepaProposer` uses the
16962
- * package `LlmClient` directly (`LlmClientOptions`), not the ChatClient —
16963
- * pass the router creds here. A test may inject `fetch` to stub the
16964
- * reflection HTTP and exercise the wiring without real tokens. */
16965
- reflectionLlm: LlmClientOptions;
16966
- /** Cheap model the student runs (e.g. a small/fast model). */
16967
- studentModel: string;
16968
- /** Model GEPA uses to propose prompt rewrites (typically a stronger model). */
16969
- optimizerModel: string;
16970
- /** Agreement judge — produced student label vs gold teacher label. */
16971
- judge: JudgeConfig<TProduced, GoldScenario<TInput, TLabel>>;
16972
- /** Promotion gate. Default: `heldOutGate` over the holdout. Pass
16973
- * `defaultProductionGate({ holdoutScenarios: holdout, ... })` for the full
16974
- * red-team / reward-hacking / canary stack. */
16975
- gate?: Gate<TProduced, GoldScenario<TInput, TLabel>>;
16976
- /** GEPA population size (candidates per generation). Default 4. */
16977
- populationSize?: number;
16978
- /** GEPA generations. Default 3. */
16979
- maxGenerations?: number;
16980
- /** Campaign reps per scenario. Default 1 — raise for CI bands on a flaky
16981
- * student. */
16982
- reps?: number;
16983
- /** Where campaign artifacts + traces land. Default a temp dir under cwd. */
16984
- runDir?: string;
16985
- /** Levers offered to the GEPA reflection prompt. */
16986
- mutationPrimitives?: string[];
16987
- /** GEPA structured-doc constraints (preserve sections, edit budget). */
16988
- constraints?: GepaProposerConstraints;
16989
- /** Gate's minimum holdout-agreement delta to ship. Default 0.0 — a
16990
- * distillation run reports the lift; the caller decides the bar. Only used
16991
- * when `gate` is omitted (the default `heldOutGate`). */
16992
- deltaThreshold?: number;
16993
- /** Render the student prompt. Default: surface as system, JSON-stringified
16994
- * input as the user turn with a JSON-only instruction. */
16995
- renderStudentPrompt?: RenderStudentPrompt<TInput>;
16996
- /** Parse the model's text into a produced label. Default: strict JSON parse
16997
- * with fenced-block stripping. */
16998
- parseStudentLabel?: ParseStudentLabel<TProduced>;
16999
- /** Per-student-call sampling temperature. Default 0 (deterministic student;
17000
- * the optimization signal must come from the PROMPT, not sampling noise). */
17001
- studentTemperature?: number;
17002
- /** Per-student-call max tokens. Default 1024. */
17003
- studentMaxTokens?: number;
17004
- }
17005
- interface RunDistillationResult<TProduced, TInput, TLabel> extends RunImprovementLoopResult<TProduced, GoldScenario<TInput, TLabel>> {
17006
- /** The winning student prompt (a string surface). */
17007
- winnerPrompt: string;
17008
- /** Mean agreement on the HOLDOUT — baseline vs winner. The headline number:
17009
- * did distillation move the student closer to the teacher on UNSEEN gold? */
17010
- holdoutAgreement: {
17011
- baseline: number;
17012
- winner: number;
17013
- delta: number;
17014
- };
17015
- }
17016
- declare function runDistillation<TProduced, TInput, TLabel>(opts: RunDistillationOptions<TProduced, TInput, TLabel>): Promise<RunDistillationResult<TProduced, TInput, TLabel>>;
17017
- /** Default student prompt render: surface as system, JSON input as the user
17018
- * turn, with a JSON-only output instruction. */
17019
- declare function defaultRenderStudentPrompt<TInput>(args: {
17020
- surface: string;
17021
- input: TInput;
17022
- scenarioId: string;
17023
- }): ChatRequest['messages'];
17024
- /** Default label parse: strip a ```json fence if present, then `JSON.parse`.
17025
- * Throws on failure so the cell is recorded as failed, not silently zeroed. */
17026
- declare function defaultParseStudentLabel<TProduced>(rawContent: string, scenarioId: string): TProduced;
17027
-
17028
16364
  /**
17029
16365
  * Strong, generically-useful baseline ROLES — the top zone of an `AgentProfile`
17030
16366
  * before any domain layer. A product composes one of these with its own
@@ -17185,6 +16521,55 @@ declare namespace index {
17185
16521
  export { type index_AgentProfileSection as AgentProfileSection, index_BASELINE_ROLES as BASELINE_ROLES, type index_BaselineRoleKey as BaselineRoleKey, type index_ProfileSkill as ProfileSkill, type index_PromptProfile as PromptProfile, index_applyDomainPatch as applyDomainPatch, index_baselineProfile as baselineProfile, index_baselineProfileFromRole as baselineProfileFromRole, index_engineerRole as engineerRole, index_generalistRole as generalistRole, index_prodProfile as prodProfile, index_profileToSurface as profileToSurface, index_renderProfile as renderProfile, index_researcherRole as researcherRole, index_sectionHash as sectionHash };
17186
16522
  }
17187
16523
 
16524
+ /**
16525
+ * Traced analyst wrapper — instruments `analyzeTraces` with spans so the
16526
+ * analyst's internal model turns appear in the trace tree. Also wraps each
16527
+ * actor turn callback with a span.
16528
+ *
16529
+ * The wrapper records the Ax turn loop at its public boundaries:
16530
+ * 1. A parent span for the entire analyst run.
16531
+ * 2. Per-turn child spans from the `onTurn` callback (captures code,
16532
+ * output size, error status).
16533
+ * 3. Summary attributes on the parent (total turns, usage, findings).
16534
+ */
16535
+
16536
+ interface TracedAnalystOptions {
16537
+ /** TraceEmitter for span emission. */
16538
+ emitter: TraceEmitter;
16539
+ /** Parent span id. If omitted, uses emitter stack. */
16540
+ parentSpanId?: string;
16541
+ }
16542
+ /**
16543
+ * Run `analyzeTraces` wrapped in a parent span with per-turn child spans.
16544
+ */
16545
+ declare function tracedAnalyzeTraces(input: AnalyzeTracesInput, options: AnalyzeTracesOptions, traceOpts: TracedAnalystOptions): Promise<AnalyzeTracesResult>;
16546
+
16547
+ /**
16548
+ * Traced judge wrappers — instruments every LLM call inside the judge
16549
+ * ensemble with child spans so OTEL sinks see per-judge latency, model,
16550
+ * token counts, and score dimensions.
16551
+ *
16552
+ * The ensemble parent span groups all individual judge spans; each judge
16553
+ * gets its own child span with model + score as attributes.
16554
+ */
16555
+
16556
+ interface TracedJudgeOptions {
16557
+ /** TraceEmitter to emit spans into. */
16558
+ emitter: TraceEmitter;
16559
+ /** Parent span id for the ensemble. If omitted, uses the emitter stack. */
16560
+ parentSpanId?: string;
16561
+ }
16562
+ /**
16563
+ * Wrap a single JudgeFn so its LLM call emits a traced span.
16564
+ */
16565
+ declare function traceJudge(judge: JudgeFn, judgeName: string, opts: TracedJudgeOptions): JudgeFn;
16566
+ /**
16567
+ * Wrap an array of JudgeFns with tracing, running them inside an ensemble
16568
+ * parent span. Returns a single function that calls all judges and merges
16569
+ * their scores.
16570
+ */
16571
+ declare function traceJudgeEnsemble(judges: JudgeFn[], judgeNames: string[], opts: TracedJudgeOptions): JudgeFn;
16572
+
17188
16573
  /**
17189
16574
  * Program cost report — a thin projection over `CostLedger.summary()` that
17190
16575
  * adds the per-model rollup the summary lacks, plus `attachCostToReport`, the
@@ -17241,9 +16626,8 @@ declare function attachCostToReport<R extends object>(report: R, ledger: CostLed
17241
16626
  * - `judges` → `ensembleJudge({ models: seats.judges, … })` (src/judge-panel.ts)
17242
16627
  * and the `JudgeConfig`s handed to `makeEvalTools({ judges })`
17243
16628
  * (src/eval-tools.ts).
17244
- * - `reflection` → `selfImprove({ llm: { model: seats.reflection } })` — the
17245
- * `gepaProposer` reflection model (src/contract/self-improve.ts);
17246
- * same seat for any custom `SurfaceProposer`'s LLM.
16629
+ * - `reflection` → the model configured by a custom `SurfaceProposer` or
16630
+ * external optimization engine.
17247
16631
  * - `worker` → the dispatch model the agent itself calls — the model an
17248
16632
  * `AgentProfile` declares.
17249
16633
  * - `analyst` → the LLM behind `analyzeRuns` / analyst-registry kinds.
@@ -17262,7 +16646,7 @@ interface ModelSeats {
17262
16646
  judges?: string[];
17263
16647
  /** Analyst model — `analyzeRuns` / analyst-registry LLM calls. */
17264
16648
  analyst?: string;
17265
- /** Reflection/proposer model `gepaProposer` mutation proposals. */
16649
+ /** Reflection or candidate-generation model. */
17266
16650
  reflection?: string;
17267
16651
  /** Verifier model — completion/objective checking. */
17268
16652
  verifier?: string;
@@ -17681,4 +17065,4 @@ type CachedJudge<TArtifact, TScenario extends Scenario = Scenario> = JudgeConfig
17681
17065
  */
17682
17066
  declare function cachedJudge<TArtifact, TScenario extends Scenario = Scenario>(judge: JudgeConfig<TArtifact, TScenario>, store: VerdictCacheStore, options: CachedJudgeOptions): CachedJudge<TArtifact, TScenario>;
17683
17067
 
17684
- export { AGENT_PROFILE_KINDS, ATTESTATION_ALGORITHM, type ActionExecutionPolicy, type ActionPolicyDecision, type ActionableSideInfo, type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, type AgentEvalErrorCode, type AgentInterfaceProfileLike, type AgentProfileCell, type AgentProfileCellInput, type AgentProfileCellSchemaVersion, AgentProfileCellValidationError, type AgentProfileDimensionValue, type AgentProfileHarness, type AgentProfileJson, type AgentProfileJsonObject, type AgentProfileKind, type AgentProfileRuntimeReceipt, type AgentProfileSource, type AgentProfileSourceInput, type AgreementResult, type AlignmentOp, type Analyst, type AnalystContext, type AnalystCost, type AnalystFinding, type AnalystHooks, type AnalystInputKind, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type AnalyzeTracesInput, type AnalyzeTracesOptions, type AnalyzeTracesResult, type AnalyzeTracesTurnSnapshot, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, type Artifact$1 as Artifact, type ArtifactCheck, type Artifact as ArtifactCheckArtifact, type ArtifactEventLike, type ArtifactResult, type ArtifactValidator, type AsiSeverity, type AssertCapabilityHeadroomOptions, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, type AttestationProvenance, type AttestationVerification, type AttestedReport, type AutoPrClient, AxGepaSteeringOptimizer, type AxSteeringOptimizerConfig, BENCHMARK_SPLIT_SEED, type BackendDescriptor, BackendIntegrityError, type BackendIntegrityReport, type BaselineOptions, type BaselineReport, BehaviorAssertion, type BehavioralMetrics, type BehavioralTokenSequence, type BenchmarkAdapter, type BenchmarkDatasetItem, type BenchmarkEvaluation, type BenchmarkFamily, type BenchmarkReport$1 as BenchmarkReport, type BenchmarkResponder, BenchmarkRunner, type BenchmarkRunnerConfig, type BenchmarkScenario, type BenchmarkSource, type BenchmarkTaskKind, type BisectOptions, type BisectResult, type BisectStep, type BlendWeights, type BootstrapOptions, type BootstrapResult, BudgetBreachError, BudgetGuard, type BudgetLedgerEntry, type BudgetPolicy, type BudgetSpec, type BuildAgreementJudgeOptions, CODING_HARNESSES, type CachedJudge, type CachedJudgeOptions, type CalibrationResult, CallExpectation, CallbackResearcher, type CallbackResearcherOptions, type CampaignFactoryParams, type CampaignIntegrityPolicy, type CampaignRunContext, type CampaignRunOutcome, type CampaignRunner, type CampaignScenario, type CampaignVariant, type CanaryAlert, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateComparison, type CandidateScenario, type CandidateScore, type CanonicalRawAnalystFinding, type CapabilityHeadroomOptions, type CapabilityHeadroomResult, type CaptureFetchContext, type CaptureFetchOptions, CaptureIntegrityError, type CausalAttributionReport, type CellVerdict, type ChannelRollup, type ChatCallOpts, type ChatClient, type ChatMessage, type ChatRequest, type ChatResponse, type ChatToolCall, type ChatTransport, type CheckResult, type CliBridgeTransportOpts, type CliffsMagnitude, type ClusterBootstrapInterval, type ClusterSignFlipAlternative, type ClusterSignFlipResult, type ClusteredBinaryCluster, type ClusteredMatchedPair, type ClusteredPairedBinaryOptions, type ClusteredPairedBinaryResult, type ClusteredPairedBinaryStatistics, type CollectedArtifacts, type CommandRunner, type CompareLabels, type ComparePairedArmsOptions, type CompletionCriterion, type CompletionRequirement, type CompletionVerdict, type ConceptComplexity, type ConceptFinding, type ConceptSpec, type ConceptWeightStrategy, ConfigError, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContinuousAgreement, type ContinuousAgreementOptions, type ContinuousCalibrationResult, type ContractCheckResult, type ContractJudgeOptions, type ContractMetric, type ContractReport, type ContractRule, type ContractRuleKind, type ContractSpan, type ContractVerdict, type ContractViolation, type ControlActionFailureMode, type ControlActionOutcome, type ControlBudget, type ControlContext, type ControlDecision, type ControlEvalResult, type ControlRunResult, type ControlRunToRunRecordOptions, type ControlRuntimeConfig, type ControlRuntimeError, type ControlSeverity, type ControlStep, type ControlStopPolicies, ConvergenceTracker, type CorpusAgreementOptions, type CorpusAgreementPerDimension, type CorpusAgreementReport, type CorpusScoreRecord, type CorrectnessChecker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, type CostChannel, type CostEntry, CostLedger, type CostLedgerEntry, type CostLedgerFilter, type CostLedgerHandle, type CostLedgerOptions, type CostLedgerPersistence, CostLedgerPersistenceError, type CostLedgerSummary, type CostReceipt, CostReceiptCaptureError, type CostReceiptInput, type CostReport, CostReservationExceededError, type CostResult, type CostSummary, CostTracker, type CostUsage, type CounterfactualContext, type CounterfactualMutation, type CounterfactualResult, type CounterfactualRunner, type CreateAnalystAiConfig, type CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateExperimentInput, type CreateSandboxPoolOpts, type CreateTraceAnalystKindOpts, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, type CustomTokenPricing, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, type DataAcquisitionPlan, Dataset, type DatasetDifficulty, type DatasetManifest, type DatasetOverview, type DatasetProvenance, type DatasetScenario, type DatasetSplit, type DecideNextUserTurnOpts, type DefaultAnalystRegistryOptions, type DefaultVerdict, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DescriptionLengthCandidate, type DescriptionLengthConfig, type DescriptionLengthDecision, type DescriptionLengthEvidence, DescriptionLengthGate, type DescriptionLengthRejectionCode, type DetectorEvent, type DetectorSeverity, type DetectorSignal, type DiffPolicy, type DiffScorecardOptions, type DirEntry, type DirectProviderTransportOpts, type Direction, type DiscoverPersonasOptions, type DiscoveredPersona, DockerSandboxDriver, type DriverResult, type DriverState, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, type EProcess, type EProcessOptions, type EProcessState, type EProcessStep, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type ErrorCluster, type ErrorCountPattern, type ErrorStreakOptions, type EvalCampaignOptions, type EvalCampaignResult, type EvalResult, type EvalToolDef, EvalTraceStore, type EventFilter, type EventKind, type EvidenceRef, type EvolutionRound, type ExecutorConfig, type Expectation, type Experiment, type ExperimentPlan, type ExperimentProvenance, type ExperimentRep, type ExperimentResult, type ExperimentStats, type ExperimentStore, ExperimentTracker, type ExperimentTrackerOptions, type ExperimentVerdict, type ExportableSpan, type ExportedRewardModel, type ExtractOptions, type ExtractResult, type ExtractUsageFromSseOptions, type ExtractedUsage, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, type FactorContribution, type FactorialCell, type FailedRun, type FailureClass, type FailureClassification, type FailureContext, type FailureMode, type FailureRule, type FeedbackArtifactType, type FeedbackAttempt, type FeedbackLabel, type FeedbackLabelKind, type FeedbackLabelSource, type FeedbackOptimizerRow, type FeedbackOutcome, type FeedbackPattern, type FeedbackReplayAdapter, type FeedbackReplayResult, type FeedbackSeverity, type FeedbackSplitPolicy, type FeedbackTask, type FeedbackTrajectory, type FeedbackTrajectoryFilter, type FeedbackTrajectoryStore, type FieldAgreementSpec, type FieldDestination, type FileChange, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, type FileSystemRawProviderSinkOptions, FileSystemTraceStore, type FileSystemTraceStoreOptions, type Finding, type FindingSubject, type FindingSubjectKind, type FindingToPolicyEditOptions, type FindingsDiff, FindingsStore, type FlattenOtlpOptions, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type GainDistributionBin, type GainDistributionFigureSpec, type GainDistributionOptions, type GateDecision$1 as GateDecision, type GateEvidence, type GenericSpan, type GhCliClientOptions, type GoldScenario, type GoldSplit, type GoldenItem, type GoldenSeverity, type GoldenSpec, HARNESS_NATIVE_MODEL, type HarnessAdapter, type HarnessConfig, type HarnessExperimentConfig, type HarnessExperimentResult, type HarnessIntervention, type HarnessRunRequest, type HarnessRunResult, type HarnessScenario, type HarnessSelection, type HarnessVariant, type HarnessVariantReport, type HeadroomClass, type HeadroomInput, HeldOutGate, type HeldOutGateConfig, type HeldOutGateRejectionCode, type HeldOutPartition, type HiddenCriteriaGrader, type HiddenGradeResult, type HiddenLeak, HoldoutAuditor, HoldoutLockedError, type HttpGithubClientOptions, type HypothesisManifest, type HypothesisResult, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, type ImageData, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, type InMemoryRawProviderSinkOptions, InMemoryTraceStore, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, type InterimReleaseConfidence, type InterimReleaseConfidenceInput, type JudgeConfig$1 as JudgeConfig, JudgeError, type JudgeFamily, type JudgeFleetOptions, type JudgeFn, type JudgeInput, JudgeParseError, type JudgeReplayGateArgs, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, type JudgeRubric, JudgeRunner, type JudgeScore$1 as JudgeScore, type JudgeScoreInput, type JudgeScoresRecord, type JudgeSpan, type JudgeVerdict, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type KnowledgeAcquisitionMode, type KnowledgeBundle, type KnowledgeFallbackPolicy, type KnowledgeFreshness, type KnowledgeImportance, type KnowledgeReadinessReport, type KnowledgeRecommendedAction, type KnowledgeRequirement, type KnowledgeRequirementCategory, type KnowledgeResponsibleSurface, type KnowledgeSensitivity, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, type Layer, type LayerResult, type LayerStatus, type LeaderboardOptions, type LeaderboardRow, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmCallError, type LlmCallMetadata, type LlmCallRequest, type LlmCallResult, LlmClient, type LlmClientOptions, type LlmCorrectnessCheckerOpts, type LlmJsonCall, type LlmJudgeDimension, type LlmJudgeOptions, type LlmMessage, LlmResponseError, type LlmReviewerConfig, LlmRouteAssertionError, type LlmRouteRequirements, type LlmSpan, type LlmSpanOtlpInput, type LlmUsage, LockedJsonlAppender, MODEL_PRICING, type MakeEvalToolsConfig, type MatchResult, type MatchedPair, type MatcherResult, type MaximumCharge, type McNemarResult, type Measured, type MeasurementPolicy, type MergeOptions, type Message, type MetricSamples, type MetricVerdict, MetricsCollector, type MintRolloutOptions, type MintRolloutResult, type MockTransportOpts, type ModelCostRollup, type ModelPreflight, type ModelSeats, ModelsUnreachableError, type MuffledFinder, type MuffledFinding, MultiLayerVerifier, type MultiToolchainLayerConfig, type Mutator, Mutex, type NoLeakOptions, type NoProgressOptions, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, type Objective, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, type OtelExportConfig, type OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, type OtlpExport, OtlpFileTraceStore, type OtlpFileTraceStoreOptions, type OtlpFlatLine, type OtlpResourceSpans, type OtlpSpan, type OtlpToRunRecordsOptions, type OtlpTraceRunRecord, POLICY_EDIT_AXES, POLICY_EDIT_CANDIDATE_RECORD_SCHEMA, POLICY_EDIT_TARGET_SURFACES, type PaidCallResult, type PairArmsOptions, type PairArmsResult, type PairedArmRow, type PairedArmsComparison, type PairedBootstrapOptions, type PairedBootstrapResult, type PairedCorrectness, type PairedEvalueOptions, type PairedEvalueSequence, type PairedEvalueStep, type PairedMetricDelta, type PairedSignTestResult, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, type ParetoFigureSpec, type ParetoPoint, type ParetoResult, type ParseStudentLabel, type PartitionHeldOutOptions, type PendingCostCall, type PendingCostCallView, type PersistedFinding, type PersonaConfig, type PersonaRigor, type Playbook, type PlaybookEntry, type PolicyEdit, type PolicyEditAdmission, type PolicyEditAdmissionOptions, type PolicyEditAxis, type PolicyEditCandidateRecord, type PolicyEditChange, type PolicyEditExpectedGain, type PolicyEditGainDirection, type PolicyEditGainUnit, type PolicyEditInit, type PolicyEditRisk, type PolicyEditSchemaVersion, type PolicyEditSource, type PolicyEditTarget, type PolicyEditTargetSurface, PolicyEditValidationError, type PoolSlot, type PositionalBiasResult, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, type PreferenceMemoryEntry, type PreflightModelsOptions, type PreflightOutcome, type ProducedProposal, type ProducedState, type ProductBenchmarkArm, type ProductBenchmarkArtifactPaths, type ProductBenchmarkBudgets, type ProductBenchmarkExportOptions, type ProductBenchmarkExportResult, type ProductBenchmarkManifest, type ProductBenchmarkProfileRef, type ProductBenchmarkRecord, type ProductBenchmarkRepoRef, type ProductBenchmarkRunInput, type ProductBenchmarkScenario, type ProductBenchmarkSingleRunExportOptions, type ProductBenchmarkSplit, type ProductBenchmarkSubstrateVersions, type ProductBenchmarkValidationReport, ProductClient, type ProductClientConfig, type ProfileAxisSpec, type ProjectRuntimeTrajectoryEvidenceOptions, type ProjectedOtlpSpan, type PromptHandle, PromptRegistry, type ProportionInterval, type ProposalEventLike, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type ProposeFn, type ProposeInput, type ProposeOutput, type ProposeReviewConfig, type ProposeReviewControlAction, type ProposeReviewControlConfig, type ProposeReviewControlResult, type ProposeReviewControlState, type ProposeReviewReport, type ProposeReviewShot, type ProposedSideEffect, type ProvenanceReader, type ProviderRedactor, type QueryTracesPage, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_FORMAT, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, type RawAnalystEvidence, type RawAnalystFinding, type RawProviderDirection, type RawProviderEvent, type RawProviderSink, type RawProviderSinkFilter, type RecordRunsOptions, type RedTeamCase, type RedTeamCategory, type RedTeamFinding, type RedTeamPayload, type RedTeamReport, type RedactionReport, type RedactionRule, type ReferenceEquivalenceJudgeInput, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceJudgeResult, type ReferenceEquivalenceScenario, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, type RegistryRunOpts, type ReleaseConfidenceAxis, type ReleaseConfidenceAxisName, type ReleaseConfidenceInput, type ReleaseConfidenceIssue, type ReleaseConfidenceMetrics, type ReleaseConfidenceScorecard, type ReleaseConfidenceStatus, type ReleaseConfidenceThresholds, type ReleaseTraceEvidence, type RenderReleaseReportOptions, type RenderStudentPrompt, type RepeatedActionOptions, ReplayCache, type ReplayCacheEntry, ReplayCacheMissError, type ReplayCacheStats, ReplayError, type ReplayFetchOptions, type RepoRef, type RequirementCheck, type ResearchReport, type ResearchReportCandidate, type ResearchReportDecision, type ResearchReportMethodology, type ResearchReportOptions, type ResearchReportRecommendation, type Researcher, type RetrievalSpan, type Review, type ReviewFn, type ReviewInput, type ReviewMemoryEntry, type ReviewMemoryStore, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, type RewardRow, type RiskDifferenceResult, type RobustnessResult, type RolloutCapture, type RolloutLine, type RolloutRole, type RolloutScrubber, type RolloutSplit, type RolloutStep, type RouteMap, type RoutedField, type RouterTransportOpts, type RubricDimension, type Run, type RunCommandInput, type RunCommandResult, type RunCompleteHook, type RunCompleteHookContext, type RunCostProvenance, RunCritic, type RunCriticOptions, type RunDistillationOptions, type RunDistillationResult, type RunEvidenceMetadata, type RunFilter, RunIntegrityError, type RunIntegrityExpectations, type RunIntegrityIssue, type RunIntegrityIssueCode, type RunIntegrityReport, type RunJudgeMetadata, type RunLayer, type RunOutcome, type RunPaidCallInput, type RunRecord, type RunRecordBackend, type RunRecordFilter, RunRecordValidationError, type RunScore, type RunScoreWeights, type RunSplitTag, type RunStatus, type RunTokenUsage, type RunTrace, type RuntimeEventLike, type RuntimeResolution, type RuntimeTrajectoryEvidenceProjection, type RuntimeTrajectoryEvidenceSummary, type RuntimeTrajectoryHookEvent, type RuntimeTrajectoryRecord, type RuntimeTrajectoryRunRecord, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_SCHEMA, type SandboxDriver, SandboxHarness, type SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type SandboxResult, type SandboxSdkTransportOpts, type SandboxSpan, type SatisfiedBy, type ScanOptions, type Scenario$1 as Scenario, type ScenarioCost, type ScenarioFile, ScenarioRegistry, type ScenarioResult, type ScoreKnowledgeReadinessOptions, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SearchSpanResult, type SearchTraceResult, type SeatName, type SeatPresetName, SeatUnsetError, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SelfPreferenceResult, type SemanticConceptJudgeInput, type SemanticConceptJudgeOptions, type SemanticConceptJudgeResult, type SequentialDecision, type SerializedRegex, type SeriesConvergenceOptions, type SeriesConvergenceResult, type Severity, type SftExportOptions, type SftRow, type SignTestAlternative, type SignedManifest, type SignedManifestAlgo, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, SkillUsageAnalyst, type SliceOptions, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, type SourceLimits, type Span, type SpanBase, type SpanFilter, type SpanHandle, type SpanKind, type SpanMatchRecord, SpanNotFoundError, type SpanPredicate, type SpanStatus, type SplitGoldOptions, type SseUsageMode, type SteeringBundle, type SteeringChange, type SteeringDelta, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type SteeringRolePrompt, type StepAttribution, type StopDecision, type StreamingDetector, type SuboptimalCode, type SuboptimalSignal, SubprocessSandboxDriver, type SubprocessSandboxDriverOptions, type SummaryTable, type SummaryTableOptions, type SummaryTableRow, type SupervisorRunReader, type SupervisorRunReport, type SupervisorRunRollup, type SupervisorRunSources, type SupervisorRunTree, type SynthesisReason, type SynthesisTarget, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, type TaskGold, type TaskHeadroom, type TestGradedRunOptions, type TestGradedRunResult, type TestGradedScenario, type TestOutputParser, type TestResult, type TextMatcher, type ThresholdContract, TokenCounter, type TokenSpec, type ToolCallEventLike, type ToolDef, type ToolMatcher, type ToolSpan, type ToolSpanOtlpInput, type ToolStats, type ToolUseMetrics, type ToolUseOptions, type TraceAggregate, type TraceAnalysisStore, type TraceAnalystByteBudgets, type TraceAnalystFilters, type TraceAnalystGolden, type TraceAnalystHookOptions, type TraceAnalystKindSpec, type TraceAnalystSpan, type TraceAnalystSpanKind, type TraceAnalystSpanStatus, type TraceAnalystTraceSummary, type TraceContract, TraceContractBuilder, TraceEmitter, type TraceEmitterOptions, type TraceEvent, TraceFileMissingError, type TraceInsightContext, type TraceInsightFinding, type TraceInsightPanelRole, type TraceInsightPromptInput, type TraceInsightQualityGate, type TraceInsightQuestion, type TraceInsightReadiness, type TraceInsightSuite, type TraceInsightTask, TraceNotFoundError, type TraceStore, type TraceStoreSource, type TraceStoreToOtlpOptions, type TracedAnalystOptions, type TracedJudgeOptions, type TracesToOtlpResult, type Trajectory, type TrajectoryStep, type TreatmentClass, type TreatmentGate, type TreatmentGateInput, type TreatmentGateOptions, type TrialTrace, type Turn, type TurnMetrics, type TurnResult, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, type Unavailable, type UserQuestion, type ValidationContext, ValidationError, type ValidationIssue, type ValidationResult, type VerbosityBiasResult, type Verdict, type VerdictCacheStats, type VerdictCacheStore, type Verification, VerificationError, type VerificationReport, type VerifyContext, type VerifyFn, type VerifyOptions, type ViewSpansResult, type ViewTraceOversized, type ViewTraceResult, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, type WeightedCompositeInput, type WeightedCompositeResult, type WorkerDriverContext, type WorkflowTopology, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, acquisitionPlansForKnowledgeGaps, admitPolicyEdit, adversarialJudge, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyPolicyEditToSurface, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, index$1 as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildAgreementJudge, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, buildWorkerDriverSystemPrompt, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, codeExecutionJudge, cohensD, coherenceJudge, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computePolicyEditId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAnalystAi, createAntiSlopJudge, createChatClient, createCustomJudge, createDefaultReviewer, createDomainExpertJudge, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalystKind, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultJudges, defaultParseStudentLabel, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultRenderStudentPrompt, defaultTraceInsightPanel, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fieldAgreement, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isPolicyEdit, isRetrievalSpan, isRolloutLine, isRunRecord, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadGoldScenarios, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, makePolicyEdit, makePolicyEditCandidateRecord, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalizeScores, notBlocked, objectiveEval, observeAll, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairedBootstrap, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseGoldJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, policyEditFromFinding, policyEditsFromFindings, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, index as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredSampleSize, researchReport, resolveModelPricing, resolveRunCostProvenance, resolveSeat, rolloutReward, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runDistillation, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTestGradedScenario, runsForScenario, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scorePolicyEditReadiness, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, showMeasured, signManifest, spearmanR, splitGold, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, typoMutator, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validatePolicyEdit, validatePolicyEditCandidateRecord, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
17068
+ export { AGENT_PROFILE_KINDS, ATTESTATION_ALGORITHM, type ActionExecutionPolicy, type ActionPolicyDecision, type ActionableSideInfo, type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, type AgentEvalErrorCode, type AgentInterfaceProfileLike, type AgentProfileCell, type AgentProfileCellInput, type AgentProfileCellSchemaVersion, AgentProfileCellValidationError, type AgentProfileDimensionValue, type AgentProfileHarness, type AgentProfileJson, type AgentProfileJsonObject, type AgentProfileKind, type AgentProfileRuntimeReceipt, type AgentProfileSource, type AgentProfileSourceInput, type AlignmentOp, type Analyst, type AnalystContext, type AnalystCost, type AnalystFinding, type AnalystHooks, type AnalystInputKind, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type AnalyzeTracesInput, type AnalyzeTracesOptions, type AnalyzeTracesResult, type AnalyzeTracesTurnSnapshot, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, type Artifact$1 as Artifact, type ArtifactCheck, type Artifact as ArtifactCheckArtifact, type ArtifactEventLike, type ArtifactResult, type ArtifactValidator, type AsiSeverity, type AssertCapabilityHeadroomOptions, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, type AttestationProvenance, type AttestationVerification, type AttestedReport, type AutoPrClient, AxGepaSteeringOptimizer, type AxSteeringOptimizerConfig, BENCHMARK_SPLIT_SEED, type BackendDescriptor, BackendIntegrityError, type BackendIntegrityReport, type BaselineOptions, type BaselineReport, BehaviorAssertion, type BehavioralMetrics, type BehavioralTokenSequence, type BenchmarkAdapter, type BenchmarkDatasetItem, type BenchmarkEvaluation, type BenchmarkFamily, type BenchmarkReport$1 as BenchmarkReport, type BenchmarkResponder, BenchmarkRunner, type BenchmarkRunnerConfig, type BenchmarkScenario, type BenchmarkSource, type BenchmarkTaskKind, type BisectOptions, type BisectResult, type BisectStep, type BlendWeights, type BootstrapOptions, type BootstrapResult, BudgetBreachError, BudgetGuard, type BudgetLedgerEntry, type BudgetPolicy, type BudgetSpec, CODING_HARNESSES, type CachedJudge, type CachedJudgeOptions, type CalibrationResult, CallExpectation, CallbackResearcher, type CallbackResearcherOptions, type CampaignFactoryParams, type CampaignIntegrityPolicy, type CampaignRunContext, type CampaignRunOutcome, type CampaignRunner, type CampaignScenario, type CampaignVariant, type CanaryAlert, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateComparison, type CandidateScenario, type CandidateScore, type CanonicalRawAnalystFinding, type CapabilityHeadroomOptions, type CapabilityHeadroomResult, type CaptureFetchContext, type CaptureFetchOptions, CaptureIntegrityError, type CausalAttributionReport, type CellVerdict, type ChannelRollup, type ChatCallOpts, type ChatClient, type ChatMessage, type ChatRequest, type ChatResponse, type ChatToolCall, type ChatTransport, type CheckResult, type CliBridgeTransportOpts, type CliffsMagnitude, type ClusterBootstrapInterval, type ClusterSignFlipAlternative, type ClusterSignFlipResult, type ClusteredBinaryCluster, type ClusteredMatchedPair, type ClusteredPairedBinaryOptions, type ClusteredPairedBinaryResult, type ClusteredPairedBinaryStatistics, type CollectedArtifacts, type CommandRunner, type ComparePairedArmsOptions, type CompletionCriterion, type CompletionRequirement, type CompletionVerdict, type ConceptComplexity, type ConceptFinding, type ConceptSpec, type ConceptWeightStrategy, ConfigError, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContinuousAgreement, type ContinuousAgreementOptions, type ContinuousCalibrationResult, type ContractCheckResult, type ContractJudgeOptions, type ContractMetric, type ContractReport, type ContractRule, type ContractRuleKind, type ContractSpan, type ContractVerdict, type ContractViolation, type ControlActionFailureMode, type ControlActionOutcome, type ControlBudget, type ControlContext, type ControlDecision, type ControlEvalResult, type ControlRunResult, type ControlRunToRunRecordOptions, type ControlRuntimeConfig, type ControlRuntimeError, type ControlSeverity, type ControlStep, type ControlStopPolicies, ConvergenceTracker, type CorpusAgreementOptions, type CorpusAgreementPerDimension, type CorpusAgreementReport, type CorpusScoreRecord, type CorrectnessChecker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, type CostChannel, type CostEntry, CostLedger, type CostLedgerEntry, type CostLedgerFilter, type CostLedgerHandle, type CostLedgerOptions, type CostLedgerPersistence, CostLedgerPersistenceError, type CostLedgerSummary, type CostReceipt, CostReceiptCaptureError, type CostReceiptInput, type CostReport, CostReservationExceededError, type CostResult, type CostSummary, CostTracker, type CostUsage, type CounterfactualContext, type CounterfactualMutation, type CounterfactualResult, type CounterfactualRunner, type CreateAnalystAiConfig, type CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateExperimentInput, type CreateSandboxPoolOpts, type CreateTraceAnalystKindOpts, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, type CustomTokenPricing, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, type DataAcquisitionPlan, Dataset, type DatasetDifficulty, type DatasetManifest, type DatasetOverview, type DatasetProvenance, type DatasetScenario, type DatasetSplit, type DecideNextUserTurnOpts, type DefaultAnalystRegistryOptions, type DefaultVerdict, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DescriptionLengthCandidate, type DescriptionLengthConfig, type DescriptionLengthDecision, type DescriptionLengthEvidence, DescriptionLengthGate, type DescriptionLengthRejectionCode, type DetectorEvent, type DetectorSeverity, type DetectorSignal, type DiffPolicy, type DiffScorecardOptions, type DirEntry, type DirectProviderTransportOpts, type Direction, type DiscoverPersonasOptions, type DiscoveredPersona, DockerSandboxDriver, type DriverResult, type DriverState, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, type EProcess, type EProcessOptions, type EProcessState, type EProcessStep, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type ErrorCluster, type ErrorCountPattern, type ErrorStreakOptions, type EvalCampaignOptions, type EvalCampaignResult, type EvalResult, type EvalToolDef, EvalTraceStore, type EventFilter, type EventKind, type EvidenceRef, type EvolutionRound, type ExecutorConfig, type Expectation, type Experiment, type ExperimentPlan, type ExperimentProvenance, type ExperimentRep, type ExperimentResult, type ExperimentStats, type ExperimentStore, ExperimentTracker, type ExperimentTrackerOptions, type ExperimentVerdict, type ExportableSpan, type ExportedRewardModel, type ExtractOptions, type ExtractResult, type ExtractUsageFromSseOptions, type ExtractedUsage, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, type FactorContribution, type FactorialCell, type FailedRun, type FailureClass, type FailureClassification, type FailureContext, type FailureMode, type FailureRule, type FeedbackArtifactType, type FeedbackAttempt, type FeedbackLabel, type FeedbackLabelKind, type FeedbackLabelSource, type FeedbackOptimizerRow, type FeedbackOutcome, type FeedbackPattern, type FeedbackReplayAdapter, type FeedbackReplayResult, type FeedbackSeverity, type FeedbackSplitPolicy, type FeedbackTask, type FeedbackTrajectory, type FeedbackTrajectoryFilter, type FeedbackTrajectoryStore, type FieldDestination, type FileChange, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, type FileSystemRawProviderSinkOptions, FileSystemTraceStore, type FileSystemTraceStoreOptions, type Finding, type FindingSubject, type FindingSubjectKind, type FindingsDiff, FindingsStore, type FlattenOtlpOptions, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type GainDistributionBin, type GainDistributionFigureSpec, type GainDistributionOptions, type GateDecision$1 as GateDecision, type GateEvidence, type GenericSpan, type GhCliClientOptions, type GoldenItem, type GoldenSeverity, type GoldenSpec, HARNESS_NATIVE_MODEL, type HarnessAdapter, type HarnessConfig, type HarnessExperimentConfig, type HarnessExperimentResult, type HarnessIntervention, type HarnessRunRequest, type HarnessRunResult, type HarnessScenario, type HarnessSelection, type HarnessVariant, type HarnessVariantReport, type HeadroomClass, type HeadroomInput, HeldOutGate, type HeldOutGateConfig, type HeldOutGateRejectionCode, type HeldOutPartition, type HiddenCriteriaGrader, type HiddenGradeResult, type HiddenLeak, HoldoutAuditor, HoldoutLockedError, type HttpGithubClientOptions, type HypothesisManifest, type HypothesisResult, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, type ImageData, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, type InMemoryRawProviderSinkOptions, InMemoryTraceStore, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, type InterimReleaseConfidence, type InterimReleaseConfidenceInput, type JudgeConfig$1 as JudgeConfig, JudgeError, type JudgeFamily, type JudgeFleetOptions, type JudgeFn, type JudgeInput, JudgeParseError, type JudgeReplayGateArgs, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, type JudgeRubric, JudgeRunner, type JudgeScore$1 as JudgeScore, type JudgeScoreInput, type JudgeScoresRecord, type JudgeSpan, type JudgeVerdict, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type KnowledgeAcquisitionMode, type KnowledgeBundle, type KnowledgeFallbackPolicy, type KnowledgeFreshness, type KnowledgeImportance, type KnowledgeReadinessReport, type KnowledgeRecommendedAction, type KnowledgeRequirement, type KnowledgeRequirementCategory, type KnowledgeResponsibleSurface, type KnowledgeSensitivity, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, type Layer, type LayerResult, type LayerStatus, type LeaderboardOptions, type LeaderboardRow, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmCallError, type LlmCallMetadata, type LlmCallRequest, type LlmCallResult, LlmClient, type LlmClientOptions, type LlmCorrectnessCheckerOpts, type LlmJsonCall, type LlmJudgeDimension, type LlmJudgeOptions, type LlmMessage, LlmResponseError, type LlmReviewerConfig, LlmRouteAssertionError, type LlmRouteRequirements, type LlmSpan, type LlmSpanOtlpInput, type LlmUsage, LockedJsonlAppender, MODEL_PRICING, type MakeEvalToolsConfig, type MatchResult, type MatchedPair, type MatcherResult, type MaximumCharge, type McNemarResult, type Measured, type MeasurementPolicy, type MergeOptions, type Message, type MetricSamples, type MetricVerdict, MetricsCollector, type MintRolloutOptions, type MintRolloutResult, type MockTransportOpts, type ModelCostRollup, type ModelPreflight, type ModelSeats, ModelsUnreachableError, type MuffledFinder, type MuffledFinding, MultiLayerVerifier, type MultiToolchainLayerConfig, type Mutator, Mutex, type NoLeakOptions, type NoProgressOptions, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, type Objective, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, type OtelExportConfig, type OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, type OtlpExport, OtlpFileTraceStore, type OtlpFileTraceStoreOptions, type OtlpFlatLine, type OtlpResourceSpans, type OtlpSpan, type OtlpToRunRecordsOptions, type OtlpTraceRunRecord, type PaidCallResult, type PairArmsOptions, type PairArmsResult, type PairedArmRow, type PairedArmsComparison, type PairedBootstrapOptions, type PairedBootstrapResult, type PairedCorrectness, type PairedEvalueOptions, type PairedEvalueSequence, type PairedEvalueStep, type PairedMetricDelta, type PairedSignTestResult, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, type ParetoFigureSpec, type ParetoPoint, type ParetoResult, type PartitionHeldOutOptions, type PendingCostCall, type PendingCostCallView, type PersistedFinding, type PersonaConfig, type PersonaRigor, type Playbook, type PlaybookEntry, type PoolSlot, type PositionalBiasResult, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, type PreferenceMemoryEntry, type PreflightModelsOptions, type PreflightOutcome, type ProducedProposal, type ProducedState, type ProductBenchmarkArm, type ProductBenchmarkArtifactPaths, type ProductBenchmarkBudgets, type ProductBenchmarkExportOptions, type ProductBenchmarkExportResult, type ProductBenchmarkManifest, type ProductBenchmarkProfileRef, type ProductBenchmarkRecord, type ProductBenchmarkRepoRef, type ProductBenchmarkRunInput, type ProductBenchmarkScenario, type ProductBenchmarkSingleRunExportOptions, type ProductBenchmarkSplit, type ProductBenchmarkSubstrateVersions, type ProductBenchmarkValidationReport, ProductClient, type ProductClientConfig, type ProfileAxisSpec, type ProjectRuntimeTrajectoryEvidenceOptions, type ProjectedOtlpSpan, type PromptHandle, PromptRegistry, type ProportionInterval, type ProposalEventLike, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type ProposeFn, type ProposeInput, type ProposeOutput, type ProposeReviewConfig, type ProposeReviewControlAction, type ProposeReviewControlConfig, type ProposeReviewControlResult, type ProposeReviewControlState, type ProposeReviewReport, type ProposeReviewShot, type ProposedSideEffect, type ProvenanceReader, type ProviderRedactor, type QueryTracesPage, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_FORMAT, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, type RawAnalystEvidence, type RawAnalystFinding, type RawProviderDirection, type RawProviderEvent, type RawProviderSink, type RawProviderSinkFilter, type RecordRunsOptions, type RedTeamCase, type RedTeamCategory, type RedTeamFinding, type RedTeamPayload, type RedTeamReport, type RedactionReport, type RedactionRule, type ReferenceEquivalenceJudgeInput, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceJudgeResult, type ReferenceEquivalenceScenario, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, type RegistryRunOpts, type ReleaseConfidenceAxis, type ReleaseConfidenceAxisName, type ReleaseConfidenceInput, type ReleaseConfidenceIssue, type ReleaseConfidenceMetrics, type ReleaseConfidenceScorecard, type ReleaseConfidenceStatus, type ReleaseConfidenceThresholds, type ReleaseTraceEvidence, type RenderReleaseReportOptions, type RepeatedActionOptions, ReplayCache, type ReplayCacheEntry, ReplayCacheMissError, type ReplayCacheStats, ReplayError, type ReplayFetchOptions, type RepoRef, type RequirementCheck, type ResearchReport, type ResearchReportCandidate, type ResearchReportDecision, type ResearchReportMethodology, type ResearchReportOptions, type ResearchReportRecommendation, type Researcher, type RetrievalSpan, type Review, type ReviewFn, type ReviewInput, type ReviewMemoryEntry, type ReviewMemoryStore, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, type RewardRow, type RiskDifferenceResult, type RobustnessResult, type RolloutCapture, type RolloutLine, type RolloutRole, type RolloutScrubber, type RolloutSplit, type RolloutStep, type RouteMap, type RoutedField, type RouterTransportOpts, type RubricDimension, type Run, type RunCommandInput, type RunCommandResult, type RunCompleteHook, type RunCompleteHookContext, type RunCostProvenance, RunCritic, type RunCriticOptions, type RunEvidenceMetadata, type RunFilter, RunIntegrityError, type RunIntegrityExpectations, type RunIntegrityIssue, type RunIntegrityIssueCode, type RunIntegrityReport, type RunJudgeMetadata, type RunLayer, type RunOutcome, type RunPaidCallInput, type RunRecord, type RunRecordBackend, type RunRecordFilter, RunRecordValidationError, type RunScore, type RunScoreWeights, type RunSplitTag, type RunStatus, type RunTokenUsage, type RunTrace, type RuntimeEventLike, type RuntimeResolution, type RuntimeTrajectoryEvidenceProjection, type RuntimeTrajectoryEvidenceSummary, type RuntimeTrajectoryHookEvent, type RuntimeTrajectoryRecord, type RuntimeTrajectoryRunRecord, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_SCHEMA, type SandboxDriver, SandboxHarness, type SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type SandboxResult, type SandboxSdkTransportOpts, type SandboxSpan, type SatisfiedBy, type ScanOptions, type Scenario$1 as Scenario, type ScenarioCost, type ScenarioFile, ScenarioRegistry, type ScenarioResult, type ScoreKnowledgeReadinessOptions, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SearchSpanResult, type SearchTraceResult, type SeatName, type SeatPresetName, SeatUnsetError, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SelfPreferenceResult, type SemanticConceptJudgeInput, type SemanticConceptJudgeOptions, type SemanticConceptJudgeResult, type SequentialDecision, type SerializedRegex, type SeriesConvergenceOptions, type SeriesConvergenceResult, type Severity, type SftExportOptions, type SftRow, type SignTestAlternative, type SignedManifest, type SignedManifestAlgo, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, SkillUsageAnalyst, type SliceOptions, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, type SourceLimits, type Span, type SpanBase, type SpanFilter, type SpanHandle, type SpanKind, type SpanMatchRecord, SpanNotFoundError, type SpanPredicate, type SpanStatus, type SseUsageMode, type SteeringBundle, type SteeringChange, type SteeringDelta, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type SteeringRolePrompt, type StepAttribution, type StopDecision, type StreamingDetector, type SuboptimalCode, type SuboptimalSignal, SubprocessSandboxDriver, type SubprocessSandboxDriverOptions, type SummaryTable, type SummaryTableOptions, type SummaryTableRow, type SupervisorRunReader, type SupervisorRunReport, type SupervisorRunRollup, type SupervisorRunSources, type SupervisorRunTree, type SynthesisReason, type SynthesisTarget, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, type TaskGold, type TaskHeadroom, type TestGradedRunOptions, type TestGradedRunResult, type TestGradedScenario, type TestOutputParser, type TestResult, type TextMatcher, type ThresholdContract, TokenCounter, type TokenSpec, type ToolCallEventLike, type ToolDef, type ToolMatcher, type ToolSpan, type ToolSpanOtlpInput, type ToolStats, type ToolUseMetrics, type ToolUseOptions, type TraceAggregate, type TraceAnalysisStore, type TraceAnalystByteBudgets, type TraceAnalystFilters, type TraceAnalystGolden, type TraceAnalystHookOptions, type TraceAnalystKindSpec, type TraceAnalystSpan, type TraceAnalystSpanKind, type TraceAnalystSpanStatus, type TraceAnalystTraceSummary, type TraceContract, TraceContractBuilder, TraceEmitter, type TraceEmitterOptions, type TraceEvent, TraceFileMissingError, type TraceInsightContext, type TraceInsightFinding, type TraceInsightPanelRole, type TraceInsightPromptInput, type TraceInsightQualityGate, type TraceInsightQuestion, type TraceInsightReadiness, type TraceInsightSuite, type TraceInsightTask, TraceNotFoundError, type TraceStore, type TraceStoreSource, type TraceStoreToOtlpOptions, type TracedAnalystOptions, type TracedJudgeOptions, type TracesToOtlpResult, type Trajectory, type TrajectoryStep, type TreatmentClass, type TreatmentGate, type TreatmentGateInput, type TreatmentGateOptions, type TrialTrace, type Turn, type TurnMetrics, type TurnResult, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, type Unavailable, type UserQuestion, type ValidationContext, ValidationError, type ValidationIssue, type ValidationResult, type VerbosityBiasResult, type Verdict, type VerdictCacheStats, type VerdictCacheStore, type Verification, VerificationError, type VerificationReport, type VerifyContext, type VerifyFn, type VerifyOptions, type ViewSpansResult, type ViewTraceOversized, type ViewTraceResult, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, type WeightedCompositeInput, type WeightedCompositeResult, type WorkerDriverContext, type WorkflowTopology, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, acquisitionPlansForKnowledgeGaps, adversarialJudge, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, index$1 as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, buildWorkerDriverSystemPrompt, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, codeExecutionJudge, cohensD, coherenceJudge, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAnalystAi, createAntiSlopJudge, createChatClient, createCustomJudge, createDefaultReviewer, createDomainExpertJudge, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalystKind, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultJudges, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isRetrievalSpan, isRolloutLine, isRunRecord, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalizeScores, notBlocked, objectiveEval, observeAll, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairedBootstrap, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, index as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredSampleSize, researchReport, resolveModelPricing, resolveRunCostProvenance, resolveSeat, rolloutReward, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTestGradedScenario, runsForScenario, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, typoMutator, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };