@tangle-network/agent-eval 0.124.0 → 0.126.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (97) hide show
  1. package/CHANGELOG.md +60 -35
  2. package/README.md +270 -189
  3. package/dist/analyst/index.d.ts +15 -145
  4. package/dist/analyst/index.js +33 -47
  5. package/dist/analyst/index.js.map +1 -1
  6. package/dist/benchmarks/index.d.ts +45 -162
  7. package/dist/benchmarks/index.js +8 -9
  8. package/dist/campaign/index.d.ts +3655 -5365
  9. package/dist/campaign/index.js +21 -95
  10. package/dist/{chunk-R226UZOI.js → chunk-474LBSOX.js} +2 -2
  11. package/dist/{chunk-W5B3ZGP3.js → chunk-4B7ZZHPX.js} +8 -6
  12. package/dist/{chunk-W5B3ZGP3.js.map → chunk-4B7ZZHPX.js.map} +1 -1
  13. package/dist/{chunk-DT7OXY3C.js → chunk-CM4OILD2.js} +535 -846
  14. package/dist/chunk-CM4OILD2.js.map +1 -0
  15. package/dist/{chunk-HM6V7F3M.js → chunk-FO7HEH76.js} +3 -3
  16. package/dist/chunk-IILEIWGW.js +635 -0
  17. package/dist/chunk-IILEIWGW.js.map +1 -0
  18. package/dist/{chunk-EQUK3RFS.js → chunk-J5SQWP6Y.js} +8 -5
  19. package/dist/chunk-J5SQWP6Y.js.map +1 -0
  20. package/dist/chunk-KO2PZOGP.js +4637 -0
  21. package/dist/chunk-KO2PZOGP.js.map +1 -0
  22. package/dist/{chunk-4Y7AAATF.js → chunk-LKKT3IVV.js} +574 -81
  23. package/dist/chunk-LKKT3IVV.js.map +1 -0
  24. package/dist/chunk-M7AH34KV.js +155 -0
  25. package/dist/chunk-M7AH34KV.js.map +1 -0
  26. package/dist/chunk-NTOV7RU5.js +7152 -0
  27. package/dist/chunk-NTOV7RU5.js.map +1 -0
  28. package/dist/{chunk-QFQZ3U3X.js → chunk-OCFJACJU.js} +2 -2
  29. package/dist/{chunk-GID26AN4.js → chunk-P22LJ3Y2.js} +4 -6
  30. package/dist/{chunk-GID26AN4.js.map → chunk-P22LJ3Y2.js.map} +1 -1
  31. package/dist/{chunk-SJT4OBVL.js → chunk-SDPM6554.js} +3 -3
  32. package/dist/{chunk-D5JZ7UDZ.js → chunk-UCLVDLCH.js} +136 -50
  33. package/dist/chunk-UCLVDLCH.js.map +1 -0
  34. package/dist/chunk-UI4YMIN2.js +105 -0
  35. package/dist/chunk-UI4YMIN2.js.map +1 -0
  36. package/dist/chunk-VBQ3CRKH.js +291 -0
  37. package/dist/chunk-VBQ3CRKH.js.map +1 -0
  38. package/dist/{chunk-JKDNAOF5.js → chunk-W4L6C2XT.js} +2 -2
  39. package/dist/{chunk-GRCDRKII.js → chunk-WS3NZZQQ.js} +58 -20
  40. package/dist/chunk-WS3NZZQQ.js.map +1 -0
  41. package/dist/cli.js +3 -3
  42. package/dist/contract/index.d.ts +3221 -3094
  43. package/dist/contract/index.js +173 -42
  44. package/dist/contract/index.js.map +1 -1
  45. package/dist/control.js +2 -3
  46. package/dist/fuzz.d.ts +14 -1
  47. package/dist/fuzz.js +1 -1
  48. package/dist/hosted/index.d.ts +8 -1
  49. package/dist/index.d.ts +208 -690
  50. package/dist/index.js +185 -500
  51. package/dist/index.js.map +1 -1
  52. package/dist/openapi.json +1 -1
  53. package/dist/rl.d.ts +5 -100
  54. package/dist/rl.js +4 -5
  55. package/dist/rl.js.map +1 -1
  56. package/dist/rollout/index.d.ts +9 -1
  57. package/dist/rollout/index.js +6 -6
  58. package/dist/{run-campaign-I3JXKVAK.js → run-campaign-LVFKZCEU.js} +3 -3
  59. package/dist/supervisor-run/index.d.ts +156 -4
  60. package/dist/supervisor-run/index.js +14 -2
  61. package/dist/traces.js +2 -3
  62. package/dist/wire/index.d.ts +14 -1
  63. package/dist/wire/index.js +3 -3
  64. package/docs/campaign-proposers.md +363 -168
  65. package/docs/design/loop-taxonomy.md +142 -190
  66. package/docs/design.md +1 -1
  67. package/docs/distributed-driver.md +8 -11
  68. package/docs/feature-guide.md +20 -19
  69. package/docs/knowledge-readiness.md +2 -5
  70. package/docs/multi-shot-optimization.md +35 -27
  71. package/docs/rollout.md +5 -5
  72. package/package.json +4 -4
  73. package/dist/chunk-4Y7AAATF.js.map +0 -1
  74. package/dist/chunk-5PVZVCZB.js +0 -9190
  75. package/dist/chunk-5PVZVCZB.js.map +0 -1
  76. package/dist/chunk-A6GT67HT.js +0 -550
  77. package/dist/chunk-A6GT67HT.js.map +0 -1
  78. package/dist/chunk-D5JZ7UDZ.js.map +0 -1
  79. package/dist/chunk-DT7OXY3C.js.map +0 -1
  80. package/dist/chunk-EQUK3RFS.js.map +0 -1
  81. package/dist/chunk-GC4ATIKK.js +0 -317
  82. package/dist/chunk-GC4ATIKK.js.map +0 -1
  83. package/dist/chunk-GRCDRKII.js.map +0 -1
  84. package/dist/chunk-LOW3U7JZ.js +0 -328
  85. package/dist/chunk-LOW3U7JZ.js.map +0 -1
  86. package/dist/chunk-MGGFVCJ7.js +0 -288
  87. package/dist/chunk-MGGFVCJ7.js.map +0 -1
  88. package/dist/chunk-PMITBABE.js +0 -3841
  89. package/dist/chunk-PMITBABE.js.map +0 -1
  90. package/dist/chunk-R7ZRE2KV.js +0 -138
  91. package/dist/chunk-R7ZRE2KV.js.map +0 -1
  92. /package/dist/{chunk-R226UZOI.js.map → chunk-474LBSOX.js.map} +0 -0
  93. /package/dist/{chunk-HM6V7F3M.js.map → chunk-FO7HEH76.js.map} +0 -0
  94. /package/dist/{chunk-QFQZ3U3X.js.map → chunk-OCFJACJU.js.map} +0 -0
  95. /package/dist/{chunk-SJT4OBVL.js.map → chunk-SDPM6554.js.map} +0 -0
  96. /package/dist/{chunk-JKDNAOF5.js.map → chunk-W4L6C2XT.js.map} +0 -0
  97. /package/dist/{run-campaign-I3JXKVAK.js.map → run-campaign-LVFKZCEU.js.map} +0 -0
package/dist/index.d.ts CHANGED
@@ -1153,10 +1153,14 @@ interface CostReceipt extends CostCallBase, CostUsage {
1153
1153
  costUsd: number;
1154
1154
  costUnknown: boolean;
1155
1155
  usageUnknown?: boolean;
1156
+ /** Rates used to estimate cost locally. Absent when cost is provider-reported or unknown. */
1156
1157
  pricing?: {
1157
1158
  inputUsdPerThousand: number;
1159
+ cachedInputUsdPerThousand?: number;
1160
+ cacheWriteUsdPerThousand?: number;
1158
1161
  outputUsdPerThousand: number;
1159
1162
  };
1163
+ /** Cost reported by the provider, not a local token-price calculation. */
1160
1164
  actualCostUsd?: number;
1161
1165
  error?: string;
1162
1166
  }
@@ -1164,20 +1168,27 @@ interface CostReceipt extends CostCallBase, CostUsage {
1164
1168
  type CostLedgerEntry = Omit<CostReceipt, 'status' | 'callId' | 'phase' | 'actor' | 'maximumCostUsd' | 'usageUnknown' | 'pricing' | 'error'>;
1165
1169
  interface CostReceiptInput extends CostUsage {
1166
1170
  model: string;
1171
+ /** Caller-supplied rates for a local estimate when the provider does not report billed cost. */
1172
+ customTokenPricing?: CustomTokenPricing;
1167
1173
  actualCostUsd?: number;
1168
1174
  costUnknown?: boolean;
1169
1175
  usageUnknown?: boolean;
1170
1176
  }
1171
1177
  /** Per-million token rates for a model or endpoint not covered by package pricing. */
1172
1178
  interface CustomTokenPricing {
1179
+ /** Non-cached input tokens. */
1173
1180
  inputUsdPerMillion: number;
1181
+ /** Cache-read tokens. Falls back to the normal input rate when omitted. */
1182
+ cachedInputUsdPerMillion?: number;
1183
+ /** Cache-creation or cache-write tokens. Falls back to the normal input rate when omitted. */
1184
+ cacheWriteUsdPerMillion?: number;
1174
1185
  outputUsdPerMillion: number;
1175
1186
  }
1176
1187
  type MaximumCharge = {
1177
1188
  externallyEnforcedMaximumUsd: number;
1178
1189
  } | ({
1179
1190
  customTokenPricing: CustomTokenPricing;
1180
- } & Pick<CostUsage, 'inputTokens' | 'outputTokens'>) | ({
1191
+ } & Pick<CostUsage, 'inputTokens' | 'outputTokens' | 'cachedTokens' | 'cacheWriteTokens'>) | ({
1181
1192
  model: string;
1182
1193
  } & CostUsage);
1183
1194
  interface RunPaidCallInput<T> {
@@ -1245,6 +1256,8 @@ interface CostLedgerFilter {
1245
1256
  interface CostLedgerWaitOptions {
1246
1257
  /** Maximum time to wait for active provider calls. Default 5 seconds. */
1247
1258
  timeoutMs?: number;
1259
+ /** Wait only for calls matching this attribution filter. */
1260
+ filter?: CostLedgerFilter;
1248
1261
  }
1249
1262
  /** Append-only storage. `append` must atomically reject stale revisions. */
1250
1263
  interface CostLedgerPersistence {
@@ -1339,8 +1352,8 @@ interface CostResult {
1339
1352
  costUnknown: boolean;
1340
1353
  }
1341
1354
  declare function costForUsage(model: string, usage: CostUsage): CostResult;
1342
- /** Price input and output token counts with caller-supplied per-million rates. */
1343
- declare function costForTokenPricing(pricing: CustomTokenPricing, usage: Pick<CostUsage, 'inputTokens' | 'outputTokens'>): number;
1355
+ /** Price token counts with caller-supplied per-million rates. */
1356
+ declare function costForTokenPricing(pricing: CustomTokenPricing, usage: Pick<CostUsage, 'inputTokens' | 'outputTokens' | 'cachedTokens' | 'cacheWriteTokens'>): number;
1344
1357
 
1345
1358
  /**
1346
1359
  * RawProviderSink — first-class persistence for the actual HTTP-level
@@ -3680,112 +3693,6 @@ declare class SkillUsageAnalyst implements Analyst<SkillUsageReport> {
3680
3693
  }
3681
3694
  declare const SKILL_USAGE_ANALYST: SkillUsageAnalyst;
3682
3695
 
3683
- type PolicyEditSchemaVersion = 'policy-edit/v1';
3684
- declare const POLICY_EDIT_AXES: readonly ["carrier", "representation", "budget", "sampling", "output_contract", "tool_contract", "routing", "memory", "agent_profile", "deployment_target"];
3685
- type PolicyEditAxis = (typeof POLICY_EDIT_AXES)[number];
3686
- declare const POLICY_EDIT_TARGET_SURFACES: readonly ["prompt", "tool-contract", "runtime-config", "memory", "agent-profile", "code", "deployment"];
3687
- type PolicyEditTargetSurface = (typeof POLICY_EDIT_TARGET_SURFACES)[number];
3688
- type PolicyEditRisk = 'low' | 'medium' | 'high' | 'unknown';
3689
- type PolicyEditGainDirection = 'increase' | 'decrease';
3690
- type PolicyEditGainUnit = 'absolute' | 'relative' | 'percent' | 'score';
3691
- interface PolicyEditTarget {
3692
- surface: PolicyEditTargetSurface;
3693
- /** Stable path inside the target surface, for example `system-prompt:tools`
3694
- * or `budget.maxTurns`. */
3695
- path?: string;
3696
- /** Optional canonical deployment identity. Store the existing cell, not a
3697
- * local profile shape. */
3698
- agentProfileCell?: AgentProfileCell;
3699
- /** Human label when the path is not enough for a readable audit trail. */
3700
- label?: string;
3701
- }
3702
- type PolicyEditChange = {
3703
- kind: 'text';
3704
- mode: 'append' | 'prepend' | 'replace';
3705
- value: string;
3706
- /** Required when `mode === 'replace'`; exact match only. */
3707
- find?: string;
3708
- } | {
3709
- kind: 'json';
3710
- mode: 'set' | 'merge' | 'remove';
3711
- path: string;
3712
- value?: AgentProfileJson;
3713
- };
3714
- interface PolicyEditExpectedGain {
3715
- /** Metric this edit is expected to move, e.g. `holdout.composite`. */
3716
- metric: string;
3717
- direction: PolicyEditGainDirection;
3718
- /** Positive magnitude in the metric's native units. */
3719
- amount: number;
3720
- unit?: PolicyEditGainUnit;
3721
- rationale?: string;
3722
- }
3723
- interface PolicyEditSource {
3724
- findingIds: string[];
3725
- analystIds: string[];
3726
- evidenceRefs: EvidenceRef[];
3727
- /** Mirrors `AnalystFinding.derived_from_judge`; admission rejects it. */
3728
- derivedFromJudge?: boolean;
3729
- }
3730
- interface PolicyEdit {
3731
- schemaVersion: PolicyEditSchemaVersion;
3732
- editId: string;
3733
- axis: PolicyEditAxis;
3734
- target: PolicyEditTarget;
3735
- change: PolicyEditChange;
3736
- claim: string;
3737
- expectedGain: PolicyEditExpectedGain;
3738
- confidence: number;
3739
- risk: PolicyEditRisk;
3740
- source: PolicyEditSource;
3741
- rationale?: string;
3742
- validationPlan?: string;
3743
- metadata?: Record<string, unknown>;
3744
- }
3745
- declare const POLICY_EDIT_CANDIDATE_RECORD_SCHEMA: "tangle.policy-edit-candidate.v1";
3746
- /** JSON-safe attribution carried with a measured candidate and its scores. */
3747
- interface PolicyEditCandidateRecord {
3748
- schema: typeof POLICY_EDIT_CANDIDATE_RECORD_SCHEMA;
3749
- policyEdit: PolicyEdit;
3750
- }
3751
- type PolicyEditInit = Omit<PolicyEdit, 'schemaVersion' | 'editId'> & {
3752
- schemaVersion?: PolicyEditSchemaVersion;
3753
- editId?: string;
3754
- };
3755
- declare class PolicyEditValidationError extends ValidationError {
3756
- readonly path: string;
3757
- constructor(message: string, path?: string);
3758
- }
3759
- interface FindingToPolicyEditOptions {
3760
- expectedGain?: PolicyEditExpectedGain | ((finding: AnalystFinding) => PolicyEditExpectedGain | null | undefined);
3761
- risk?: PolicyEditRisk | ((finding: AnalystFinding) => PolicyEditRisk);
3762
- defaultAxis?: PolicyEditAxis;
3763
- defaultTargetSurface?: PolicyEditTargetSurface;
3764
- }
3765
- interface PolicyEditAdmissionOptions {
3766
- minScore?: number;
3767
- minExpectedGain?: number;
3768
- allowHighRisk?: boolean;
3769
- requireEvidence?: boolean;
3770
- }
3771
- interface PolicyEditAdmission {
3772
- edit: PolicyEdit;
3773
- decision: 'admit' | 'reject';
3774
- score: number;
3775
- reasons: string[];
3776
- }
3777
- declare function makePolicyEdit(init: PolicyEditInit): PolicyEdit;
3778
- declare function computePolicyEditId(edit: Omit<PolicyEdit, 'editId'> | PolicyEdit): string;
3779
- declare function validatePolicyEdit(input: unknown): PolicyEdit;
3780
- declare function makePolicyEditCandidateRecord(edit: PolicyEdit): PolicyEditCandidateRecord;
3781
- declare function validatePolicyEditCandidateRecord(input: unknown): PolicyEditCandidateRecord;
3782
- declare function isPolicyEdit(input: unknown): input is PolicyEdit;
3783
- declare function policyEditsFromFindings(findings: ReadonlyArray<AnalystFinding>, opts?: FindingToPolicyEditOptions): PolicyEdit[];
3784
- declare function policyEditFromFinding(finding: AnalystFinding, opts?: FindingToPolicyEditOptions): PolicyEdit | null;
3785
- declare function scorePolicyEditReadiness(edit: PolicyEdit, opts?: PolicyEditAdmissionOptions): number;
3786
- declare function admitPolicyEdit(edit: PolicyEdit, opts?: PolicyEditAdmissionOptions): PolicyEditAdmission;
3787
- declare function applyPolicyEditToSurface(surface: unknown, edit: PolicyEdit): unknown;
3788
-
3789
3696
  /**
3790
3697
  * Automated pull-request transports for the production loop.
3791
3698
  *
@@ -6521,6 +6428,34 @@ interface WorkerLogSource {
6521
6428
  readonly inbox: string | null;
6522
6429
  /** Worker patch byte length, or null when absent. */
6523
6430
  readonly patchBytes: number | null;
6431
+ /** Where this worker's transcript lives, for the rollout row. Null = no such artifact. */
6432
+ readonly transcriptRef?: string | null;
6433
+ /** Where this worker's delivered patch lives. Null = the store keeps no patch per worker. */
6434
+ readonly patchPath?: string | null;
6435
+ /** This worker's own inference tokens, when the store records them per worker. */
6436
+ readonly tokensIn?: number | null;
6437
+ readonly tokensOut?: number | null;
6438
+ readonly cacheRead?: number | null;
6439
+ readonly cacheWrite?: number | null;
6440
+ }
6441
+ /**
6442
+ * Facts a SOURCE structurally cannot express, each with the reason.
6443
+ *
6444
+ * The difference between "the artifact is missing" and "this store never
6445
+ * records that fact" is the difference between a run that spent $0 and a
6446
+ * harness that does not price inference — and the second harness is where a
6447
+ * loops-shaped assumption becomes a fabricated zero. A reader declares its
6448
+ * limits once; the analyzer reports `unavailable` for everything downstream.
6449
+ *
6450
+ * `null` on a field means the source DOES carry that fact.
6451
+ */
6452
+ interface SourceLimits {
6453
+ /** Reason inference spend has no price in this store (null = the store prices it). */
6454
+ readonly spendUsd: string | null;
6455
+ /** Reason workers carry no pass/fail verdict (null = verdicts are recorded). */
6456
+ readonly workerVerdicts: string | null;
6457
+ /** Reason no delivered artifact (patch/diff) is retained per worker (null = retained). */
6458
+ readonly deliverables: string | null;
6524
6459
  }
6525
6460
  /**
6526
6461
  * Everything the pure analyzer reads — already-read bytes, never paths. Each
@@ -6575,8 +6510,24 @@ interface SupervisorRunSources {
6575
6510
  sessions: number;
6576
6511
  input: number;
6577
6512
  output: number;
6513
+ /** Cached prompt tokens, when the store counts them separately. */
6514
+ cacheRead?: number;
6515
+ cacheWrite?: number;
6578
6516
  } | null;
6579
6517
  readonly harnessMissingReason: string | null;
6518
+ /** What this store structurally cannot record. See `SourceLimits`. */
6519
+ readonly limits: SourceLimits;
6520
+ /**
6521
+ * Where the ROOT invocation's transcript lives. Undefined lets the rollout
6522
+ * minter fall back to the loops layout (`<supRunDir>/journal.jsonl`); any
6523
+ * other store must say, or the row points at a path that never existed.
6524
+ */
6525
+ readonly rootTranscriptRef?: string | null;
6526
+ /**
6527
+ * The `traces` CLI command that covers this run's harness-session layer.
6528
+ * Null falls back to the analyzer's default (an opencode worker fleet).
6529
+ */
6530
+ readonly traceCommand: string | null;
6580
6531
  }
6581
6532
  /**
6582
6533
  * A source of supervisor-run bytes. Implementations own their storage layout;
@@ -6648,15 +6599,23 @@ interface DecisionMetrics {
6648
6599
  interface RoleSpend {
6649
6600
  readonly tokensIn: Measured<number>;
6650
6601
  readonly tokensOut: Measured<number>;
6602
+ /**
6603
+ * Cached prompt tokens read/written. On a harness that caches aggressively
6604
+ * these dwarf `tokensIn`, so a report that omits them understates the context
6605
+ * each invocation actually consumed. `unavailable` = the store has no such counter.
6606
+ */
6607
+ readonly cacheRead: Measured<number>;
6608
+ readonly cacheWrite: Measured<number>;
6651
6609
  readonly usd: Measured<number>;
6652
6610
  readonly source: string;
6653
6611
  }
6654
6612
  interface PerWorkerRow {
6655
6613
  readonly worker: string;
6656
6614
  readonly wallMs: number | null;
6657
- readonly tokensIn: number;
6658
- readonly tokensOut: number;
6659
- readonly usd: number;
6615
+ /** `null` = this store does not attribute tokens per worker (NOT "zero tokens"). */
6616
+ readonly tokensIn: number | null;
6617
+ readonly tokensOut: number | null;
6618
+ readonly usd: number | null;
6660
6619
  readonly patchBytes: number | null;
6661
6620
  readonly passed: boolean | null;
6662
6621
  }
@@ -6787,6 +6746,88 @@ declare function analyzeSupervisorRunSources(src: SupervisorRunSources, now?: ()
6787
6746
  */
6788
6747
  declare function rollupSupervisorRuns(reports: readonly SupervisorRunReport[]): SupervisorRunRollup;
6789
6748
 
6749
+ /**
6750
+ * Supervision-tree reader over a THIRD-PARTY harness: Claude Code.
6751
+ *
6752
+ * `loops-reader.ts` reads a supervisor we wrote, whose journal was designed
6753
+ * for this analysis. This reader reads a harness we do not control, whose
6754
+ * transcript was designed for replaying a chat — and recovers the same tree
6755
+ * from it. If both produce a `SupervisorRunSources`, the tree model is a
6756
+ * property of multi-agent runs, not of our journal format.
6757
+ *
6758
+ * ## Where the tree hides in a Claude Code transcript
6759
+ *
6760
+ * | Tree fact | Claude Code evidence |
6761
+ * |---|---|
6762
+ * | spawn | assistant `tool_use` (`Agent` / `Task`), answered by a `tool_result` whose `toolUseResult.agentId` names the child |
6763
+ * | settle | a `<task-notification>` block in a later user line: `<task-id>` = agentId, `<status>` |
6764
+ * | steer | assistant `tool_use` (`SendMessage`) with `input.to` = agentId — mid-task, to a LIVE child |
6765
+ * | delivered | that steer's `tool_result` carrying `success` / `resumedAgentId` |
6766
+ * | cancel | assistant `tool_use` (`TaskStop`) targeting an agentId |
6767
+ * | brain spend| `message.usage` on the main thread's assistant lines |
6768
+ * | worker spend| `message.usage` inside `<session>/subagents/agent-<id>.jsonl` |
6769
+ * | depth | a child transcript that itself contains `Agent` tool_use lines |
6770
+ *
6771
+ * Every one of those is read through `parseClaudeEntries` — the SAME line
6772
+ * parser `src/rollout/readers/claude-jsonl.ts` uses for solo rollouts. There
6773
+ * is no second transcript parser.
6774
+ *
6775
+ * ## What Claude Code cannot say
6776
+ *
6777
+ * It records tokens but never a price, runs no per-worker verify, and keeps no
6778
+ * per-worker patch. Those are declared once in `limits`, so the analyzer
6779
+ * reports `unavailable — <reason>` instead of the $0 / 0-accepted that summing
6780
+ * an empty field would produce. See `SourceLimits`.
6781
+ *
6782
+ * ## Metric coverage vs the loops journal
6783
+ *
6784
+ * Measured on a real 52-agent session (fixture:
6785
+ * `tests/fixtures/supervisor-run/claude-code-session-*`).
6786
+ *
6787
+ * | Metric | loops | Claude Code | Why |
6788
+ * |---|---|---|---|
6789
+ * | workersSpawned / Settled / Cancelled | full | full | spawn tool_use + task-notification + TaskStop |
6790
+ * | steers / steersDelivered / steersByWorker | full | full | `SendMessage`; delivery from its tool_result |
6791
+ * | waves / waveSizes / maxConcurrency | full | full | derived from spawn/settle instants |
6792
+ * | respawns / repeatedLabels | full | full | same derivation |
6793
+ * | delegationDepth | full | full | a child transcript's own spawn calls |
6794
+ * | timeToFirstSpawn / supervisorWall | full | full | transcript instants |
6795
+ * | idleMs / idlePct / workerUtilization | full | PARTIAL | an agent that never notifies is counted live to the end of the transcript |
6796
+ * | observeThenRespawn / respawnWithoutEvidence | full | full | ordering of spawn vs settle instants |
6797
+ * | workerEvidenceBytes | full | PARTIAL | the child's closing message; 0 for pruned transcripts |
6798
+ * | brain tokens in/out + cache | full | full | main-thread `message.usage` |
6799
+ * | worker tokens in/out + cache | via harness join | PARTIAL | only for retained subagent transcripts |
6800
+ * | perWorker wall | full | full | spawn → settle instants |
6801
+ * | accepted / rejected / emptyPass / settledVerdicts | full | NONE | no per-worker verify step exists |
6802
+ * | brain/worker/total usd, costPerAcceptedPatch | full | NONE | transcripts carry no price |
6803
+ * | patch stats, delivered, verifyPass/Rc | full | NONE | no diff is handed back |
6804
+ * | judgeResolved / Score / Passed / Total | full | NONE | no judge in the loop |
6805
+ * | driverSteerCalls, brainTruncations | full | NONE | no outer driver log, no per-call finish_reason tap |
6806
+ */
6807
+
6808
+ interface ClaudeCodeReaderOptions {
6809
+ /** The main session transcript: `~/.claude/projects/<slug>/<sessionId>.jsonl`. */
6810
+ readonly transcriptPath: string;
6811
+ /**
6812
+ * Directory of child transcripts. Defaults to `<transcript-dir>/<sessionId>/subagents`.
6813
+ * `null` skips the join, and every per-worker token count becomes unavailable.
6814
+ */
6815
+ readonly subagentsDir?: string | null;
6816
+ readonly runRef?: string;
6817
+ readonly instanceId?: string | null;
6818
+ readonly arm?: string | null;
6819
+ readonly spawnTools?: readonly string[];
6820
+ readonly steerTools?: readonly string[];
6821
+ readonly cancelTools?: readonly string[];
6822
+ }
6823
+ /**
6824
+ * Read a Claude Code session (plus its subagent transcripts) as supervision-tree
6825
+ * source bytes. Never throws on a missing artifact.
6826
+ */
6827
+ declare function readClaudeCodeSupervisorRun(opts: ClaudeCodeReaderOptions): Promise<SupervisorRunSources>;
6828
+ /** A `SupervisorRunReader` over a Claude Code session — the same contract loops implements. */
6829
+ declare function claudeCodeSupervisorRunReader(opts: ClaudeCodeReaderOptions): SupervisorRunReader;
6830
+
6790
6831
  /**
6791
6832
  * ONE implementation of `SupervisorRunReader`: the on-disk layout the loops
6792
6833
  * supervisor writes — `<runDir>/ws/.loops/supervisor/<id>/{journal.jsonl,
@@ -8382,98 +8423,8 @@ interface JudgeScore {
8382
8423
  /** Ensemble extras: each surviving judge's per-dimension scores. */
8383
8424
  perJudge?: Record<string, Record<string, number>>;
8384
8425
  }
8385
- /** A tier-4 code surface — a finalized candidate change to the agent's
8386
- * IMPLEMENTATION, not its prompt. Produced by autoresearch (reads codebase +
8387
- * trace findings → opens a worktree). `worktreeRef` locates the candidate;
8388
- * the exact commits, tree, and binary-patch digest identify it. See the
8389
- * improvement-tier table in `docs/design/loop-taxonomy.md`. */
8390
- interface CodeSurface {
8391
- readonly kind: 'code';
8392
- /** Worktree path or git ref holding the candidate code change. This is a
8393
- * mutable locator and is deliberately excluded from content hashes. */
8394
- readonly worktreeRef: string;
8395
- /** Human-readable ref the worktree was forked from. Not identity-bearing. */
8396
- readonly baseRef: string;
8397
- /** Exact commit the candidate was forked from. */
8398
- readonly baseCommit: string;
8399
- /** Exact tree object for `baseCommit`. */
8400
- readonly baseTree: string;
8401
- /** Exact finalized candidate commit. */
8402
- readonly candidateCommit: string;
8403
- /** Exact tree object for `candidateCommit`. */
8404
- readonly candidateTree: string;
8405
- /** Identity of the exact patch artifact. The deployable candidate bundle
8406
- * carries the same descriptor plus its base64-encoded content. */
8407
- readonly patch: {
8408
- readonly format: 'git-diff-binary';
8409
- readonly sha256: `sha256:${string}`;
8410
- readonly byteLength: number;
8411
- };
8412
- /** Human summary of what changed — rendered into the auto-PR body. */
8413
- readonly summary?: string;
8414
- }
8415
- /** The mutable surface a proposer changes. Tiers (see
8416
- * `docs/design/loop-taxonomy.md`):
8417
- * - `string` — tiers 1-2: system-prompt addendum / serialized tool
8418
- * config. Cheap, reversible, text-diffable.
8419
- * - `CodeSurface` — tier 4: an implementation change behind a worktree ref.
8420
- * Tier 3 (knowledge) is owned by agent-knowledge and rides its own adapter,
8421
- * not this type. */
8422
- type MutableSurface = string | CodeSurface;
8423
- /** A non-dominated parent on the GEPA Pareto frontier — a
8424
- * surface that, across the per-scenario objective vectors, no other tried
8425
- * surface beats on every scenario. A candidate worse on the mean composite
8426
- * but uniquely best on one hard scenario is non-dominated and survives here;
8427
- * the composite-best ranking would discard the lesson it carries. The loop
8428
- * computes the frontier across ALL generations and hands it to the proposer so
8429
- * a reflective proposer can combine complementary lessons (GEPA, Agrawal et
8430
- * al., arXiv:2507.19457). See `pareto.ts` (`paretoFrontier`). */
8431
- interface ParetoParent {
8432
- surface: MutableSurface;
8433
- surfaceHash: string;
8434
- /** The objective vector: per-scenario composite (higher is better). The
8435
- * axes the frontier is computed over. */
8436
- objectives: Record<string, number>;
8437
- /** Mean composite across the objective scenarios — the scalar summary used
8438
- * for ordering + display, NOT for dominance. */
8439
- composite: number;
8440
- /** Generation that produced this surface (`-1` for the baseline). */
8441
- generation: number;
8442
- label?: string;
8443
- rationale?: string;
8444
- }
8445
8426
  /** Five-valued verdict taxonomy (MOSS-paper alignment). */
8446
8427
  type GateDecision = 'ship' | 'hold' | 'need_more_work' | 'model_ceiling' | 'arch_ceiling';
8447
- interface GateContext<TArtifact, TScenario extends Scenario> {
8448
- candidateArtifacts: Map<string, TArtifact>;
8449
- baselineArtifacts?: Map<string, TArtifact>;
8450
- /** Candidate (winner) judge scores, keyed by cellId. */
8451
- judgeScores: Map<string, Record<string, JudgeScore>>;
8452
- /** Baseline judge scores, keyed by cellId. SEPARATE from `judgeScores` —
8453
- * baseline + candidate share cellIds (same scenarios), so a single map
8454
- * cannot represent both. A gate computing a holdout delta MUST read
8455
- * candidate from `judgeScores` and baseline from here. */
8456
- baselineJudgeScores?: Map<string, Record<string, JudgeScore>>;
8457
- /** Neutralized-arm judge scores, keyed by cellId — the winner surface with its
8458
- * content footprint-matched-blanked (via a `neutralize` fn). Same scenarios as
8459
- * `judgeScores`. Present ONLY when `runImprovementLoop` was given a `neutralize`
8460
- * function. A placebo gate (`neutralizationGate`) compares this arm's lift
8461
- * against the candidate's to reject decorative wins (lift from footprint, not
8462
- * content). Undefined otherwise. */
8463
- neutralizedJudgeScores?: Map<string, Record<string, JudgeScore>>;
8464
- /** Neutralized-arm artifacts, keyed by cellId. Present alongside
8465
- * `neutralizedJudgeScores`. */
8466
- neutralizedArtifacts?: Map<string, TArtifact>;
8467
- scenarios: TScenario[];
8468
- cost: {
8469
- candidate: number;
8470
- baseline: number;
8471
- };
8472
- /** Shared run spend account and receipt attribution phase. */
8473
- costLedger?: CostLedgerHandle;
8474
- costPhase?: string;
8475
- signal: AbortSignal;
8476
- }
8477
8428
  interface GateResult {
8478
8429
  decision: GateDecision;
8479
8430
  reasons: string[];
@@ -8484,11 +8435,6 @@ interface GateResult {
8484
8435
  }>;
8485
8436
  delta?: number;
8486
8437
  }
8487
- /** Composable promotion gate. */
8488
- interface Gate<TArtifact = unknown, TScenario extends Scenario = Scenario> {
8489
- name: string;
8490
- decide(ctx: GateContext<TArtifact, TScenario>): Promise<GateResult>;
8491
- }
8492
8438
  /** Scoped trace writer handed to each dispatch — every span
8493
8439
  * auto-tagged with the cellId so traces filter cleanly. */
8494
8440
  interface CampaignTraceWriter {
@@ -8619,8 +8565,6 @@ interface GenerationCandidate {
8619
8565
  * "because rationale Z" the audit requires to survive to the result.
8620
8566
  * Present when the proposer returned a `ProposedCandidate`. */
8621
8567
  rationale?: string;
8622
- /** Exact structured cause threaded from the proposer, when available. */
8623
- candidateRecord?: PolicyEditCandidateRecord;
8624
8568
  }
8625
8569
  interface CampaignAggregates {
8626
8570
  byJudge: Record<string, JudgeAggregate>;
@@ -10154,9 +10098,8 @@ declare function securityJudge(id: string, config: HarnessConfig): SandboxJudgeS
10154
10098
  * The `JudgeConfig` contract (src/campaign/types.ts) is deliberately a
10155
10099
  * function, not a fixed LLM-prompt shape: real consumers judge with
10156
10100
  * ensembles, deterministic checks, or one LLM call. `ensembleJudge`
10157
- * (src/judge-panel.ts) covers the multi-model case; `buildAgreementJudge`
10158
- * (src/campaign/distillation) covers the pure-comparator case. `llmJudge`
10159
- * covers the common single-call case the `JudgeConfig` doc-comment names:
10101
+ * (src/judge-panel.ts) covers the multi-model case. `llmJudge` covers the
10102
+ * common single-call case the `JudgeConfig` doc-comment names:
10160
10103
  * one model call against `prompt`, parsed into the canonical `JudgeScore`
10161
10104
  * (`{ dimensions, composite, notes }`) on the campaign [0,1] scale.
10162
10105
  *
@@ -14882,195 +14825,6 @@ interface CampaignStorage {
14882
14825
  append?(path: string, content: string, expectedBytes: number): number | undefined;
14883
14826
  }
14884
14827
 
14885
- /**
14886
- * `openAutoPr` — thin shell-out helper for the `runImprovementLoop` preset's
14887
- * `autoOnPromote: 'pr'` mode. Substitutes for the per-product PR-opening
14888
- * code consumers duplicated 4 times. The PR body includes the campaign's
14889
- * manifest hash, gate verdict, and scorecard summary so reviewers can see
14890
- * exactly what was promoted + why.
14891
- *
14892
- * NOT a deploy mechanism — this only OPENS a PR. The human reviews + merges.
14893
- * The Shape B (`autoOnPromote: 'config'`) live-runtime-mutation path is
14894
- * deferred to Pass B with the full shadow / canary / rollback stack.
14895
- */
14896
-
14897
- interface OpenAutoPrOptions<TArtifact, TScenario extends Scenario> {
14898
- /** Campaign result to attach to the PR. */
14899
- result: CampaignResult<TArtifact, TScenario>;
14900
- /** Gate verdict explaining the promotion. Substrate refuses to open a PR
14901
- * when `gate.decision !== 'ship'` — fails loud. */
14902
- gate: GateResult;
14903
- /** Promoted surface diff — typically the new system prompt addendum or
14904
- * full profile diff. Substrate writes it as the PR body. */
14905
- promotedDiff: string;
14906
- /** GH owner/repo target (e.g., `tangle-network/gtm-agent`). */
14907
- ghOwner: string;
14908
- ghRepo: string;
14909
- /** Branch name for the PR. Default `auto/<manifestHash[:12]>`. */
14910
- branch?: string;
14911
- /** PR title. Default includes manifest hash. */
14912
- title?: string;
14913
- /** Whether to actually open the PR or just dry-run. Default reads
14914
- * `GH_AUTO_PR_TOKEN` env — present = open, absent = dry-run. */
14915
- dryRun?: boolean;
14916
- /** Test seam — substitute `gh pr create` invocation. */
14917
- ghExec?: (args: string[]) => {
14918
- stdout: string;
14919
- stderr: string;
14920
- status: number;
14921
- };
14922
- }
14923
- interface OpenAutoPrResult {
14924
- opened: boolean;
14925
- prUrl?: string;
14926
- dryRun: boolean;
14927
- reason: string;
14928
- }
14929
- /**
14930
- * Open a GitHub PR for a gate-approved surface promotion, attaching the manifest hash, gate verdict, and diff as the PR body.
14931
- */
14932
- declare function openAutoPr<TArtifact, TScenario extends Scenario>(options: OpenAutoPrOptions<TArtifact, TScenario>): OpenAutoPrResult;
14933
-
14934
- /**
14935
- * `runOptimization` — the improvement loop body. Runs N generations: the
14936
- * `SurfaceProposer` proposes K candidate surfaces per generation, each
14937
- * candidate runs a campaign (the measurement), and only a candidate that beats
14938
- * the single global incumbent becomes the next generation's parent.
14939
- * Proposer-agnostic — the same loop runs an evolutionary population mutator
14940
- * (`evolutionaryProposer`) or any reflective / agentic proposer; they differ
14941
- * only in how `propose()` picks candidates.
14942
- *
14943
- * This is `runLoop`'s shape (plan → measure → decide) specialized to surface
14944
- * improvement: `proposer.propose` = plan, `runCampaign` = the measurement
14945
- * (which runs the worker behind `dispatch`), the mean-composite ranking = the
14946
- * validator, `proposer.decide` = the stop check.
14947
- *
14948
- * The gated-promotion shell (`runImprovementLoop`) wraps this with a holdout
14949
- * re-score + release gate + optional PR.
14950
- */
14951
-
14952
- interface RunOptimizationResult<TArtifact, TScenario extends Scenario> {
14953
- generations: Array<{
14954
- record: GenerationRecord;
14955
- surfaces: Array<{
14956
- surfaceHash: string;
14957
- surface: MutableSurface;
14958
- campaign: CampaignResult<TArtifact, TScenario>;
14959
- }>;
14960
- }>;
14961
- winnerSurface: MutableSurface;
14962
- winnerSurfaceHash: string;
14963
- /** Proposer label for the promoted surface. Present when the winning
14964
- * candidate came from a `ProposedCandidate` (a reflective proposer);
14965
- * absent when the winner is the baseline or a bare-surface mutator. */
14966
- winnerLabel?: string;
14967
- /** Proposer rationale for the promoted surface — the "because Z" that
14968
- * motivated the winning change. Survives to `SelfImproveResult` and the
14969
- * emitted provenance record. Absent when the winner is the baseline. */
14970
- winnerRationale?: string;
14971
- baselineCampaign: CampaignResult<TArtifact, TScenario>;
14972
- /** Run-wide spend, including agents, proposers, analysts, and judges. */
14973
- cost: CostLedgerSummary;
14974
- /** The GEPA Pareto frontier across every scored surface (baseline + all
14975
- * generations) by per-scenario objective vector — the non-dominated set.
14976
- * Each generation's `propose()` received the frontier-so-far as
14977
- * `ctx.paretoParents`; this is the final frontier. A surface here that is
14978
- * NOT the winner is uniquely best on some scenario the winner loses on. */
14979
- paretoFrontier: ParetoParent[];
14980
- }
14981
-
14982
- /**
14983
- * `runImprovementLoop` — the gated-promotion shell around the improvement
14984
- * loop body (`runOptimization`). Proposes candidate surfaces via the
14985
- * `SurfaceProposer`, re-scores the winner against the baseline on a
14986
- * holdout set, runs the release gate, and optionally opens a PR.
14987
- *
14988
- * Role vocabulary (see docs/design/loop-taxonomy.md):
14989
- * - PROPOSER = the `SurfaceProposer` (evolutionary GEPA mutator OR
14990
- * reflective analyst). Proposes candidate SURFACES — the
14991
- * worker's system prompt / tool config — NOT conversation
14992
- * turns.
14993
- * - MEASUREMENT= `runCampaign`. Scores one surface by running the worker
14994
- * (via `dispatch`) over scenarios and judging the output.
14995
- * - WORKER = the agent harness in the sandbox, invoked behind the
14996
- * topology-opaque `dispatch` seam — never referenced here.
14997
- *
14998
- * Distinct from `runLoop` in `@tangle-network/agent-runtime`, which is the
14999
- * INNER conversation loop (execution driver ↔ workers in a sandbox). `runImprovementLoop`
15000
- * is the OUTER loop: it improves the surface that those workers run.
15001
- *
15002
- * Hard-refuses unsafe configurations:
15003
- * - `tracing: 'off'` when a proposer is wired (improvement is unattributable)
15004
- * - `autoOnPromote: 'config'` — live mutation is unsupported without
15005
- * isolated deployment, rollback, and independent validation.
15006
- */
15007
-
15008
- interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario> extends RunOptimizationResult<TArtifact, TScenario> {
15009
- baselineOnHoldout: CampaignResult<TArtifact, TScenario>;
15010
- winnerOnHoldout: CampaignResult<TArtifact, TScenario>;
15011
- neutralizedOnHoldout?: CampaignResult<TArtifact, TScenario>;
15012
- neutralizedSurface?: MutableSurface;
15013
- gateResult: Awaited<ReturnType<Gate<TArtifact, TScenario>['decide']>>;
15014
- /** Present iff the loop ran with `holdout: 'deferred'`. When set,
15015
- * `baselineOnHoldout`/`winnerOnHoldout` are the shared EMPTY campaign (zero
15016
- * cells dispatched) and the gate verdict is the forced `'hold'`. */
15017
- holdout?: 'deferred';
15018
- /** Unified baseline→winner surface diff. Computed UNCONDITIONALLY (not only
15019
- * when `autoOnPromote === 'pr'`) so the diff that the gate decided on is
15020
- * always present on the result + in the emitted provenance record. Empty
15021
- * string when winner == baseline (no change to diff). */
15022
- promotedDiff: string;
15023
- prResult?: ReturnType<typeof openAutoPr>;
15024
- }
15025
-
15026
- /**
15027
- * `gepaProposer` — a reflective `SurfaceProposer` for prompt-tier surfaces.
15028
- * Each generation it reflects on the prior best candidate's per-scenario
15029
- * scores + weakest dimensions, asks an LLM to propose targeted rewrites of
15030
- * the current surface, and returns them as the next population.
15031
- *
15032
- * Maps onto the GEPA paper (Agrawal et al., arXiv:2507.19457):
15033
- * - *Reflection*: each generation reflects on the best parent's weakest
15034
- * dimensions + per-scenario top/bottom scores to propose targeted rewrites.
15035
- * - *Pareto frontier*: `runOptimization` maintains the non-dominated set of
15036
- * surfaces across generations (per-scenario objective vectors) and supplies
15037
- * it as `ctx.paretoParents`. A surface uniquely best on one hard scenario
15038
- * survives even when its mean composite is lower.
15039
- * - *Combine complementary lessons*: when the frontier has >1 member, the
15040
- * first population slot is a merge of those parents' strengths (one LLM
15041
- * call citing each parent's winning scenarios). Toggle via `combineParents`.
15042
- * Dominance is computed by the package-canonical `paretoFrontier` (`pareto.ts`).
15043
- *
15044
- * Optional `constraints` move structured-doc guards into the proposer
15045
- * (preserve H2 section headings, cap sentence-level edits) — useful when
15046
- * the surface IS a structured procedure like a SKILL.md / runbook /
15047
- * judge rubric. When `constraints` is omitted, behavior is unchanged.
15048
- *
15049
- * The proposer is surface-agnostic — any string surface in any consumer opts
15050
- * in by selecting it. Reuses the generic reflection primitive
15051
- * (`buildReflectionPrompt` / `parseReflectionResponse`) and the router client.
15052
- *
15053
- * Earns its keep where there is real per-instance signal (which the
15054
- * dimensional + per-scenario evidence + the `LabeledScenarioStore` flywheel
15055
- * now provide). For thin-signal surfaces it degrades to plain reflection.
15056
- * On generation 0 (no history) it reflects on the current surface against
15057
- * the mutation primitives alone.
15058
- */
15059
-
15060
- interface GepaProposerConstraints {
15061
- /** H2 section headings that MUST appear unchanged in every candidate.
15062
- * When set, the proposer auto-detects current H2s if this is empty AND
15063
- * rejects any candidate that drops or renames a preserved heading.
15064
- * Use when the surface is a structured doc (SKILL.md, runbook,
15065
- * sectioned system prompt, judge rubric). */
15066
- preserveSections?: string[];
15067
- /** Maximum sentence-level edits per candidate vs the parent surface.
15068
- * Rejection threshold = maxSentenceEdits × 2 (counts adds + removes).
15069
- * Inspired by SkillOpt's edit-budget as a "textual learning rate."
15070
- * Cap prevents an LLM rewrite from overwriting useful prior rules. */
15071
- maxSentenceEdits?: number;
15072
- }
15073
-
15074
14828
  interface BenchmarkRunOptions<TPayload = unknown, TArtifact = string> {
15075
14829
  adapter: BenchmarkAdapter<BenchmarkDatasetItem<TPayload>, TPayload, TArtifact>;
15076
14830
  respond: BenchmarkResponder<TPayload, TArtifact>;
@@ -16607,290 +16361,6 @@ declare function withOtelPipeline(opts?: OtelPipelineOptions): OtelPipelineHandl
16607
16361
  */
16608
16362
  declare function isOtelConfigured(): boolean;
16609
16363
 
16610
- /**
16611
- * Traced analyst wrapper — instruments `analyzeTraces` with spans so the
16612
- * analyst's internal model turns appear in the trace tree. Also wraps each
16613
- * actor turn callback with a span.
16614
- *
16615
- * The wrapper records the Ax turn loop at its public boundaries:
16616
- * 1. A parent span for the entire analyst run.
16617
- * 2. Per-turn child spans from the `onTurn` callback (captures code,
16618
- * output size, error status).
16619
- * 3. Summary attributes on the parent (total turns, usage, findings).
16620
- */
16621
-
16622
- interface TracedAnalystOptions {
16623
- /** TraceEmitter for span emission. */
16624
- emitter: TraceEmitter;
16625
- /** Parent span id. If omitted, uses emitter stack. */
16626
- parentSpanId?: string;
16627
- }
16628
- /**
16629
- * Run `analyzeTraces` wrapped in a parent span with per-turn child spans.
16630
- */
16631
- declare function tracedAnalyzeTraces(input: AnalyzeTracesInput, options: AnalyzeTracesOptions, traceOpts: TracedAnalystOptions): Promise<AnalyzeTracesResult>;
16632
-
16633
- /**
16634
- * Traced judge wrappers — instruments every LLM call inside the judge
16635
- * ensemble with child spans so OTEL sinks see per-judge latency, model,
16636
- * token counts, and score dimensions.
16637
- *
16638
- * The ensemble parent span groups all individual judge spans; each judge
16639
- * gets its own child span with model + score as attributes.
16640
- */
16641
-
16642
- interface TracedJudgeOptions {
16643
- /** TraceEmitter to emit spans into. */
16644
- emitter: TraceEmitter;
16645
- /** Parent span id for the ensemble. If omitted, uses the emitter stack. */
16646
- parentSpanId?: string;
16647
- }
16648
- /**
16649
- * Wrap a single JudgeFn so its LLM call emits a traced span.
16650
- */
16651
- declare function traceJudge(judge: JudgeFn, judgeName: string, opts: TracedJudgeOptions): JudgeFn;
16652
- /**
16653
- * Wrap an array of JudgeFns with tracing, running them inside an ensemble
16654
- * parent span. Returns a single function that calls all judges and merges
16655
- * their scores.
16656
- */
16657
- declare function traceJudgeEnsemble(judges: JudgeFn[], judgeNames: string[], opts: TracedJudgeOptions): JudgeFn;
16658
-
16659
- /**
16660
- * Gold scenarios for teacher→student distillation. The TEACHER is an
16661
- * expensive workflow (e.g. the 70-agent skill audit) whose verdicts are
16662
- * frozen as gold labels; the STUDENT is a cheap single-shot analyst whose
16663
- * prompt GEPA optimizes toward reproducing those labels.
16664
- *
16665
- * A `GoldScenario` is a `Scenario` (the substrate's input contract) carrying
16666
- * an OPAQUE `input` (what the student sees) and an OPAQUE `label` (the gold
16667
- * verdict the student's output is scored against). Both are typed `unknown`
16668
- * here: this module is domain-agnostic — it distills ANY analyst against ANY
16669
- * gold JSONL. The agreement comparator (see `agreement-judge.ts`) is what
16670
- * knows the label's shape.
16671
- *
16672
- * Loading + splitting are DETERMINISTIC and LLM-free: the gold set is the
16673
- * fixed ground truth, never regenerated here.
16674
- */
16675
-
16676
- /** A held gold record: opaque student-input + opaque gold-label, carried as a
16677
- * substrate `Scenario` so it flows through `runCampaign` unchanged. */
16678
- interface GoldScenario<TInput = unknown, TLabel = unknown> extends Scenario {
16679
- kind: 'gold';
16680
- /** What the student analyst is shown (rendered into its user prompt). */
16681
- input: TInput;
16682
- /** The teacher's gold verdict — the target the student's output is scored
16683
- * against by the agreement judge. NEVER shown to the student. */
16684
- label: TLabel;
16685
- }
16686
- /** Read a gold JSONL (one `{scenarioId|id, input, label, split?}` per line) into
16687
- * `GoldScenario[]`. Deterministic, no LLM. Blank lines are skipped; a line
16688
- * missing an id, `input`, or `label` throws (a silent skip would corrupt the
16689
- * split silently — fail loud on a malformed gold set). */
16690
- declare function loadGoldScenarios<TInput = unknown, TLabel = unknown>(jsonlPath: string): GoldScenario<TInput, TLabel>[];
16691
- /** Parse gold JSONL text directly (no fs). Exported so tests + in-memory
16692
- * callers exercise the same parse path as {@link loadGoldScenarios}. */
16693
- declare function parseGoldJsonl<TInput = unknown, TLabel = unknown>(text: string, sourceLabel?: string): GoldScenario<TInput, TLabel>[];
16694
- interface SplitGoldOptions {
16695
- /** Every Nth scenario (0-based index) goes to the TEST/holdout split; the
16696
- * rest train. Default 4 ⇒ a 25% holdout. Ignored for any scenario that
16697
- * carries an explicit `split:` tag (that is honored verbatim). */
16698
- testEveryNth?: number;
16699
- }
16700
- interface GoldSplit<TInput, TLabel> {
16701
- /** Training scenarios — the optimization pool the proposer searches over. */
16702
- train: GoldScenario<TInput, TLabel>[];
16703
- /** Held-out scenarios — kept OUT of training; scored only at the gate. */
16704
- test: GoldScenario<TInput, TLabel>[];
16705
- }
16706
- /** Deterministic train/test split. A scenario tagged `split:train|test` is
16707
- * routed by that tag; the rest fall to a modulo split (`index % testEveryNth
16708
- * === 0 ⇒ test`). Pure — same input always yields the same split, so a gold
16709
- * set's holdout is stable across runs (a shuffled split would let a lucky
16710
- * seed flatter the gate). */
16711
- declare function splitGold<TInput, TLabel>(scenarios: GoldScenario<TInput, TLabel>[], options?: SplitGoldOptions): GoldSplit<TInput, TLabel>;
16712
-
16713
- /**
16714
- * Agreement judge for teacher→student distillation. Scores a STUDENT artifact
16715
- * (the cheap analyst's produced label) against the GoldScenario's gold label
16716
- * (the teacher's verdict). The score IS the distillation objective: 1.0 means
16717
- * the student reproduced the teacher exactly, 0.0 means total disagreement.
16718
- *
16719
- * The comparison function is INJECTED (`compareLabels`) so the judge is
16720
- * domain-agnostic — distilling a skill-audit analyst, a triage analyst, or any
16721
- * other student is a one-line comparator swap. A default `fieldAgreement`
16722
- * comparator is provided for the common case: a flat verdict object with
16723
- * categorical and array fields.
16724
- *
16725
- * Everything here is PURE + unit-testable — no LLM. (The student spends tokens
16726
- * producing the artifact; scoring it against frozen gold does not.)
16727
- */
16728
-
16729
- /** What an injected comparator returns: a [0,1] composite plus the per-field
16730
- * (per-dimension) agreement breakdown the GEPA proposer reflects on to learn
16731
- * WHICH part of the verdict the student is getting wrong. */
16732
- interface AgreementResult {
16733
- /** Overall agreement in [0,1]. */
16734
- score: number;
16735
- /** Per-dimension agreement in [0,1] — keyed by field/aspect name. The
16736
- * reflective proposer surfaces the weakest of these as the lever to fix. */
16737
- dimensions: Record<string, number>;
16738
- }
16739
- /** Compare a produced label against a gold label → agreement. Injected so the
16740
- * judge is domain-agnostic. */
16741
- type CompareLabels<TProduced = unknown, TLabel = unknown> = (produced: TProduced, gold: TLabel) => AgreementResult;
16742
- interface BuildAgreementJudgeOptions<TProduced = unknown, TLabel = unknown> {
16743
- /** Judge name surfaced in `CampaignResult.aggregates.byJudge` + the gate. */
16744
- name?: string;
16745
- /** The agreement function — produced student label vs gold teacher label. */
16746
- compareLabels: CompareLabels<TProduced, TLabel>;
16747
- /** Dimension keys the judge declares up-front (for `JudgeConfig.dimensions`).
16748
- * When omitted, the dimensions present on the first scored result are used
16749
- * for display only; the composite is unaffected. */
16750
- dimensionKeys?: string[];
16751
- /** Only score `gold`-kind scenarios. Default true — a mixed campaign won't
16752
- * mis-apply the agreement judge to non-gold scenarios. */
16753
- goldOnly?: boolean;
16754
- }
16755
- /** Build a `JudgeConfig` that scores a produced student artifact against the
16756
- * scenario's gold label. Conforms to the substrate `JudgeConfig` contract:
16757
- * `score({artifact, scenario, signal}) => JudgeScore`. The `composite` is the
16758
- * comparator's `score`; `dimensions` carries its per-field breakdown plus the
16759
- * scalar `agreement` so a single-dimension consumer still sees the number. */
16760
- declare function buildAgreementJudge<TProduced, TInput, TLabel>(options: BuildAgreementJudgeOptions<TProduced, TLabel>): JudgeConfig<TProduced, GoldScenario<TInput, TLabel>>;
16761
- interface FieldAgreementSpec {
16762
- /** Categorical fields — scored exact-match (1 if equal, else 0). Compared
16763
- * with `===` after `JSON`-normalizing so `true`/`'high'`/`3` all work. */
16764
- categorical?: string[];
16765
- /** Array fields — scored by Jaccard overlap (|A∩B| / |A∪B|). Two empty
16766
- * arrays agree perfectly (1.0). Order-insensitive; elements compared by
16767
- * their `JSON.stringify`. */
16768
- array?: string[];
16769
- }
16770
- /** Default comparator: average per-field agreement over a flat verdict object.
16771
- * Categorical fields score exact-match; array fields score set-overlap
16772
- * (Jaccard). The composite is the unweighted mean across all declared fields,
16773
- * so missing a single boolean (e.g. `public_leak_risk`) costs `1/nFields` of
16774
- * the score — the leak-detection lever the audit cares about is a real,
16775
- * non-trivial fraction of the objective, not rounding noise.
16776
- *
16777
- * Pure. A field absent from BOTH produced + gold is treated as agreeing
16778
- * (both undefined ⇒ 1.0); a field present in only one side disagrees. */
16779
- declare function fieldAgreement<TProduced extends Record<string, unknown>, TLabel>(spec: FieldAgreementSpec): CompareLabels<TProduced, TLabel>;
16780
-
16781
- /**
16782
- * `runDistillation` — the teacher→student distillation loop. COMPOSES existing
16783
- * substrate primitives; reimplements none of them:
16784
- *
16785
- * - DRIVER = `gepaProposer` (reflective prompt optimizer)
16786
- * - LOOP = `runImprovementLoop` (outer: optimize → holdout re-score → gate)
16787
- * - MEASUREMENT = `runCampaign` (inside the loop) scoring the student
16788
- * - JUDGE = `buildAgreementJudge` — student label vs gold teacher label
16789
- * - GATE = caller-supplied (`heldOutGate` / `defaultProductionGate`)
16790
- * - STUDENT = a cheap single-shot analyst whose system prompt is the
16791
- * `MutableSurface` GEPA mutates; it calls the LLM through
16792
- * `createChatClient` and emits a JSON label.
16793
- *
16794
- * The surface IS the student's system prompt. Each generation GEPA rewrites it;
16795
- * `dispatchWithSurface` renders {surface + scenario.input} into a chat request,
16796
- * calls the (cheap) model, parses the produced JSON label, and returns it as
16797
- * the artifact. The agreement judge scores that label against the gold label.
16798
- *
16799
- * `autoOnPromote: 'none'` is FORCED — the loop never opens a PR; the caller
16800
- * (the `distill` CLI) decides what to do with the winning prompt.
16801
- */
16802
-
16803
- /** Render the student's prompt from {current surface, scenario input}. The
16804
- * surface is the system prompt; the scenario input is the user turn. Override
16805
- * to inject few-shot framing or a JSON-schema reminder. */
16806
- type RenderStudentPrompt<TInput> = (args: {
16807
- surface: string;
16808
- input: TInput;
16809
- scenarioId: string;
16810
- }) => ChatRequest['messages'];
16811
- /** Parse the model's raw text into a typed produced label. Throws on
16812
- * unparseable output — a thrown dispatch is recorded as a failed cell (never
16813
- * silently scored 0), which is the honest signal that the prompt isn't
16814
- * emitting valid JSON yet. */
16815
- type ParseStudentLabel<TProduced> = (rawContent: string, scenarioId: string) => TProduced;
16816
- interface RunDistillationOptions<TProduced, TInput, TLabel> {
16817
- /** The student analyst's INITIAL system prompt — the baseline surface GEPA
16818
- * searches from. */
16819
- baselinePrompt: string;
16820
- /** Training scenarios (the optimization pool). */
16821
- train: GoldScenario<TInput, TLabel>[];
16822
- /** Held-out scenarios — kept OUT of training; scored only at the gate. */
16823
- holdout: GoldScenario<TInput, TLabel>[];
16824
- /** Transport for BOTH the student (cheap model) and the GEPA reflection
16825
- * (the optimizer model). The student calls it via `createChatClient`. */
16826
- llm: CreateChatClientOpts;
16827
- /** Router transport the GEPA proposer reflects through. `gepaProposer` uses the
16828
- * package `LlmClient` directly (`LlmClientOptions`), not the ChatClient —
16829
- * pass the router creds here. A test may inject `fetch` to stub the
16830
- * reflection HTTP and exercise the wiring without real tokens. */
16831
- reflectionLlm: LlmClientOptions;
16832
- /** Cheap model the student runs (e.g. a small/fast model). */
16833
- studentModel: string;
16834
- /** Model GEPA uses to propose prompt rewrites (typically a stronger model). */
16835
- optimizerModel: string;
16836
- /** Agreement judge — produced student label vs gold teacher label. */
16837
- judge: JudgeConfig<TProduced, GoldScenario<TInput, TLabel>>;
16838
- /** Promotion gate. Default: `heldOutGate` over the holdout. Pass
16839
- * `defaultProductionGate({ holdoutScenarios: holdout, ... })` for the full
16840
- * red-team / reward-hacking / canary stack. */
16841
- gate?: Gate<TProduced, GoldScenario<TInput, TLabel>>;
16842
- /** GEPA population size (candidates per generation). Default 4. */
16843
- populationSize?: number;
16844
- /** GEPA generations. Default 3. */
16845
- maxGenerations?: number;
16846
- /** Campaign reps per scenario. Default 1 — raise for CI bands on a flaky
16847
- * student. */
16848
- reps?: number;
16849
- /** Where campaign artifacts + traces land. Default a temp dir under cwd. */
16850
- runDir?: string;
16851
- /** Levers offered to the GEPA reflection prompt. */
16852
- mutationPrimitives?: string[];
16853
- /** GEPA structured-doc constraints (preserve sections, edit budget). */
16854
- constraints?: GepaProposerConstraints;
16855
- /** Gate's minimum holdout-agreement delta to ship. Default 0.0 — a
16856
- * distillation run reports the lift; the caller decides the bar. Only used
16857
- * when `gate` is omitted (the default `heldOutGate`). */
16858
- deltaThreshold?: number;
16859
- /** Render the student prompt. Default: surface as system, JSON-stringified
16860
- * input as the user turn with a JSON-only instruction. */
16861
- renderStudentPrompt?: RenderStudentPrompt<TInput>;
16862
- /** Parse the model's text into a produced label. Default: strict JSON parse
16863
- * with fenced-block stripping. */
16864
- parseStudentLabel?: ParseStudentLabel<TProduced>;
16865
- /** Per-student-call sampling temperature. Default 0 (deterministic student;
16866
- * the optimization signal must come from the PROMPT, not sampling noise). */
16867
- studentTemperature?: number;
16868
- /** Per-student-call max tokens. Default 1024. */
16869
- studentMaxTokens?: number;
16870
- }
16871
- interface RunDistillationResult<TProduced, TInput, TLabel> extends RunImprovementLoopResult<TProduced, GoldScenario<TInput, TLabel>> {
16872
- /** The winning student prompt (a string surface). */
16873
- winnerPrompt: string;
16874
- /** Mean agreement on the HOLDOUT — baseline vs winner. The headline number:
16875
- * did distillation move the student closer to the teacher on UNSEEN gold? */
16876
- holdoutAgreement: {
16877
- baseline: number;
16878
- winner: number;
16879
- delta: number;
16880
- };
16881
- }
16882
- declare function runDistillation<TProduced, TInput, TLabel>(opts: RunDistillationOptions<TProduced, TInput, TLabel>): Promise<RunDistillationResult<TProduced, TInput, TLabel>>;
16883
- /** Default student prompt render: surface as system, JSON input as the user
16884
- * turn, with a JSON-only output instruction. */
16885
- declare function defaultRenderStudentPrompt<TInput>(args: {
16886
- surface: string;
16887
- input: TInput;
16888
- scenarioId: string;
16889
- }): ChatRequest['messages'];
16890
- /** Default label parse: strip a ```json fence if present, then `JSON.parse`.
16891
- * Throws on failure so the cell is recorded as failed, not silently zeroed. */
16892
- declare function defaultParseStudentLabel<TProduced>(rawContent: string, scenarioId: string): TProduced;
16893
-
16894
16364
  /**
16895
16365
  * Strong, generically-useful baseline ROLES — the top zone of an `AgentProfile`
16896
16366
  * before any domain layer. A product composes one of these with its own
@@ -17051,6 +16521,55 @@ declare namespace index {
17051
16521
  export { type index_AgentProfileSection as AgentProfileSection, index_BASELINE_ROLES as BASELINE_ROLES, type index_BaselineRoleKey as BaselineRoleKey, type index_ProfileSkill as ProfileSkill, type index_PromptProfile as PromptProfile, index_applyDomainPatch as applyDomainPatch, index_baselineProfile as baselineProfile, index_baselineProfileFromRole as baselineProfileFromRole, index_engineerRole as engineerRole, index_generalistRole as generalistRole, index_prodProfile as prodProfile, index_profileToSurface as profileToSurface, index_renderProfile as renderProfile, index_researcherRole as researcherRole, index_sectionHash as sectionHash };
17052
16522
  }
17053
16523
 
16524
+ /**
16525
+ * Traced analyst wrapper — instruments `analyzeTraces` with spans so the
16526
+ * analyst's internal model turns appear in the trace tree. Also wraps each
16527
+ * actor turn callback with a span.
16528
+ *
16529
+ * The wrapper records the Ax turn loop at its public boundaries:
16530
+ * 1. A parent span for the entire analyst run.
16531
+ * 2. Per-turn child spans from the `onTurn` callback (captures code,
16532
+ * output size, error status).
16533
+ * 3. Summary attributes on the parent (total turns, usage, findings).
16534
+ */
16535
+
16536
+ interface TracedAnalystOptions {
16537
+ /** TraceEmitter for span emission. */
16538
+ emitter: TraceEmitter;
16539
+ /** Parent span id. If omitted, uses emitter stack. */
16540
+ parentSpanId?: string;
16541
+ }
16542
+ /**
16543
+ * Run `analyzeTraces` wrapped in a parent span with per-turn child spans.
16544
+ */
16545
+ declare function tracedAnalyzeTraces(input: AnalyzeTracesInput, options: AnalyzeTracesOptions, traceOpts: TracedAnalystOptions): Promise<AnalyzeTracesResult>;
16546
+
16547
+ /**
16548
+ * Traced judge wrappers — instruments every LLM call inside the judge
16549
+ * ensemble with child spans so OTEL sinks see per-judge latency, model,
16550
+ * token counts, and score dimensions.
16551
+ *
16552
+ * The ensemble parent span groups all individual judge spans; each judge
16553
+ * gets its own child span with model + score as attributes.
16554
+ */
16555
+
16556
+ interface TracedJudgeOptions {
16557
+ /** TraceEmitter to emit spans into. */
16558
+ emitter: TraceEmitter;
16559
+ /** Parent span id for the ensemble. If omitted, uses the emitter stack. */
16560
+ parentSpanId?: string;
16561
+ }
16562
+ /**
16563
+ * Wrap a single JudgeFn so its LLM call emits a traced span.
16564
+ */
16565
+ declare function traceJudge(judge: JudgeFn, judgeName: string, opts: TracedJudgeOptions): JudgeFn;
16566
+ /**
16567
+ * Wrap an array of JudgeFns with tracing, running them inside an ensemble
16568
+ * parent span. Returns a single function that calls all judges and merges
16569
+ * their scores.
16570
+ */
16571
+ declare function traceJudgeEnsemble(judges: JudgeFn[], judgeNames: string[], opts: TracedJudgeOptions): JudgeFn;
16572
+
17054
16573
  /**
17055
16574
  * Program cost report — a thin projection over `CostLedger.summary()` that
17056
16575
  * adds the per-model rollup the summary lacks, plus `attachCostToReport`, the
@@ -17107,9 +16626,8 @@ declare function attachCostToReport<R extends object>(report: R, ledger: CostLed
17107
16626
  * - `judges` → `ensembleJudge({ models: seats.judges, … })` (src/judge-panel.ts)
17108
16627
  * and the `JudgeConfig`s handed to `makeEvalTools({ judges })`
17109
16628
  * (src/eval-tools.ts).
17110
- * - `reflection` → `selfImprove({ llm: { model: seats.reflection } })` — the
17111
- * `gepaProposer` reflection model (src/contract/self-improve.ts);
17112
- * same seat for any custom `SurfaceProposer`'s LLM.
16629
+ * - `reflection` → the model configured by a custom `SurfaceProposer` or
16630
+ * external optimization engine.
17113
16631
  * - `worker` → the dispatch model the agent itself calls — the model an
17114
16632
  * `AgentProfile` declares.
17115
16633
  * - `analyst` → the LLM behind `analyzeRuns` / analyst-registry kinds.
@@ -17128,7 +16646,7 @@ interface ModelSeats {
17128
16646
  judges?: string[];
17129
16647
  /** Analyst model — `analyzeRuns` / analyst-registry LLM calls. */
17130
16648
  analyst?: string;
17131
- /** Reflection/proposer model `gepaProposer` mutation proposals. */
16649
+ /** Reflection or candidate-generation model. */
17132
16650
  reflection?: string;
17133
16651
  /** Verifier model — completion/objective checking. */
17134
16652
  verifier?: string;
@@ -17547,4 +17065,4 @@ type CachedJudge<TArtifact, TScenario extends Scenario = Scenario> = JudgeConfig
17547
17065
  */
17548
17066
  declare function cachedJudge<TArtifact, TScenario extends Scenario = Scenario>(judge: JudgeConfig<TArtifact, TScenario>, store: VerdictCacheStore, options: CachedJudgeOptions): CachedJudge<TArtifact, TScenario>;
17549
17067
 
17550
- export { AGENT_PROFILE_KINDS, ATTESTATION_ALGORITHM, type ActionExecutionPolicy, type ActionPolicyDecision, type ActionableSideInfo, type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, type AgentEvalErrorCode, type AgentInterfaceProfileLike, type AgentProfileCell, type AgentProfileCellInput, type AgentProfileCellSchemaVersion, AgentProfileCellValidationError, type AgentProfileDimensionValue, type AgentProfileHarness, type AgentProfileJson, type AgentProfileJsonObject, type AgentProfileKind, type AgentProfileRuntimeReceipt, type AgentProfileSource, type AgentProfileSourceInput, type AgreementResult, type AlignmentOp, type Analyst, type AnalystContext, type AnalystCost, type AnalystFinding, type AnalystHooks, type AnalystInputKind, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type AnalyzeTracesInput, type AnalyzeTracesOptions, type AnalyzeTracesResult, type AnalyzeTracesTurnSnapshot, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, type Artifact$1 as Artifact, type ArtifactCheck, type Artifact as ArtifactCheckArtifact, type ArtifactEventLike, type ArtifactResult, type ArtifactValidator, type AsiSeverity, type AssertCapabilityHeadroomOptions, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, type AttestationProvenance, type AttestationVerification, type AttestedReport, type AutoPrClient, AxGepaSteeringOptimizer, type AxSteeringOptimizerConfig, BENCHMARK_SPLIT_SEED, type BackendDescriptor, BackendIntegrityError, type BackendIntegrityReport, type BaselineOptions, type BaselineReport, BehaviorAssertion, type BehavioralMetrics, type BehavioralTokenSequence, type BenchmarkAdapter, type BenchmarkDatasetItem, type BenchmarkEvaluation, type BenchmarkFamily, type BenchmarkReport$1 as BenchmarkReport, type BenchmarkResponder, BenchmarkRunner, type BenchmarkRunnerConfig, type BenchmarkScenario, type BenchmarkSource, type BenchmarkTaskKind, type BisectOptions, type BisectResult, type BisectStep, type BlendWeights, type BootstrapOptions, type BootstrapResult, BudgetBreachError, BudgetGuard, type BudgetLedgerEntry, type BudgetPolicy, type BudgetSpec, type BuildAgreementJudgeOptions, CODING_HARNESSES, type CachedJudge, type CachedJudgeOptions, type CalibrationResult, CallExpectation, CallbackResearcher, type CallbackResearcherOptions, type CampaignFactoryParams, type CampaignIntegrityPolicy, type CampaignRunContext, type CampaignRunOutcome, type CampaignRunner, type CampaignScenario, type CampaignVariant, type CanaryAlert, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateComparison, type CandidateScenario, type CandidateScore, type CanonicalRawAnalystFinding, type CapabilityHeadroomOptions, type CapabilityHeadroomResult, type CaptureFetchContext, type CaptureFetchOptions, CaptureIntegrityError, type CausalAttributionReport, type CellVerdict, type ChannelRollup, type ChatCallOpts, type ChatClient, type ChatMessage, type ChatRequest, type ChatResponse, type ChatToolCall, type ChatTransport, type CheckResult, type CliBridgeTransportOpts, type CliffsMagnitude, type ClusterBootstrapInterval, type ClusterSignFlipAlternative, type ClusterSignFlipResult, type ClusteredBinaryCluster, type ClusteredMatchedPair, type ClusteredPairedBinaryOptions, type ClusteredPairedBinaryResult, type ClusteredPairedBinaryStatistics, type CollectedArtifacts, type CommandRunner, type CompareLabels, type ComparePairedArmsOptions, type CompletionCriterion, type CompletionRequirement, type CompletionVerdict, type ConceptComplexity, type ConceptFinding, type ConceptSpec, type ConceptWeightStrategy, ConfigError, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContinuousAgreement, type ContinuousAgreementOptions, type ContinuousCalibrationResult, type ContractCheckResult, type ContractJudgeOptions, type ContractMetric, type ContractReport, type ContractRule, type ContractRuleKind, type ContractSpan, type ContractVerdict, type ContractViolation, type ControlActionFailureMode, type ControlActionOutcome, type ControlBudget, type ControlContext, type ControlDecision, type ControlEvalResult, type ControlRunResult, type ControlRunToRunRecordOptions, type ControlRuntimeConfig, type ControlRuntimeError, type ControlSeverity, type ControlStep, type ControlStopPolicies, ConvergenceTracker, type CorpusAgreementOptions, type CorpusAgreementPerDimension, type CorpusAgreementReport, type CorpusScoreRecord, type CorrectnessChecker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, type CostChannel, type CostEntry, CostLedger, type CostLedgerEntry, type CostLedgerFilter, type CostLedgerHandle, type CostLedgerOptions, type CostLedgerPersistence, CostLedgerPersistenceError, type CostLedgerSummary, type CostReceipt, CostReceiptCaptureError, type CostReceiptInput, type CostReport, CostReservationExceededError, type CostResult, type CostSummary, CostTracker, type CostUsage, type CounterfactualContext, type CounterfactualMutation, type CounterfactualResult, type CounterfactualRunner, type CreateAnalystAiConfig, type CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateExperimentInput, type CreateSandboxPoolOpts, type CreateTraceAnalystKindOpts, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, type CustomTokenPricing, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, type DataAcquisitionPlan, Dataset, type DatasetDifficulty, type DatasetManifest, type DatasetOverview, type DatasetProvenance, type DatasetScenario, type DatasetSplit, type DecideNextUserTurnOpts, type DefaultAnalystRegistryOptions, type DefaultVerdict, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DescriptionLengthCandidate, type DescriptionLengthConfig, type DescriptionLengthDecision, type DescriptionLengthEvidence, DescriptionLengthGate, type DescriptionLengthRejectionCode, type DetectorEvent, type DetectorSeverity, type DetectorSignal, type DiffPolicy, type DiffScorecardOptions, type DirEntry, type DirectProviderTransportOpts, type Direction, type DiscoverPersonasOptions, type DiscoveredPersona, DockerSandboxDriver, type DriverResult, type DriverState, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, type EProcess, type EProcessOptions, type EProcessState, type EProcessStep, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type ErrorCluster, type ErrorCountPattern, type ErrorStreakOptions, type EvalCampaignOptions, type EvalCampaignResult, type EvalResult, type EvalToolDef, EvalTraceStore, type EventFilter, type EventKind, type EvidenceRef, type EvolutionRound, type ExecutorConfig, type Expectation, type Experiment, type ExperimentPlan, type ExperimentProvenance, type ExperimentRep, type ExperimentResult, type ExperimentStats, type ExperimentStore, ExperimentTracker, type ExperimentTrackerOptions, type ExperimentVerdict, type ExportableSpan, type ExportedRewardModel, type ExtractOptions, type ExtractResult, type ExtractUsageFromSseOptions, type ExtractedUsage, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, type FactorContribution, type FactorialCell, type FailedRun, type FailureClass, type FailureClassification, type FailureContext, type FailureMode, type FailureRule, type FeedbackArtifactType, type FeedbackAttempt, type FeedbackLabel, type FeedbackLabelKind, type FeedbackLabelSource, type FeedbackOptimizerRow, type FeedbackOutcome, type FeedbackPattern, type FeedbackReplayAdapter, type FeedbackReplayResult, type FeedbackSeverity, type FeedbackSplitPolicy, type FeedbackTask, type FeedbackTrajectory, type FeedbackTrajectoryFilter, type FeedbackTrajectoryStore, type FieldAgreementSpec, type FieldDestination, type FileChange, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, type FileSystemRawProviderSinkOptions, FileSystemTraceStore, type FileSystemTraceStoreOptions, type Finding, type FindingSubject, type FindingSubjectKind, type FindingToPolicyEditOptions, type FindingsDiff, FindingsStore, type FlattenOtlpOptions, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type GainDistributionBin, type GainDistributionFigureSpec, type GainDistributionOptions, type GateDecision$1 as GateDecision, type GateEvidence, type GenericSpan, type GhCliClientOptions, type GoldScenario, type GoldSplit, type GoldenItem, type GoldenSeverity, type GoldenSpec, HARNESS_NATIVE_MODEL, type HarnessAdapter, type HarnessConfig, type HarnessExperimentConfig, type HarnessExperimentResult, type HarnessIntervention, type HarnessRunRequest, type HarnessRunResult, type HarnessScenario, type HarnessSelection, type HarnessVariant, type HarnessVariantReport, type HeadroomClass, type HeadroomInput, HeldOutGate, type HeldOutGateConfig, type HeldOutGateRejectionCode, type HeldOutPartition, type HiddenCriteriaGrader, type HiddenGradeResult, type HiddenLeak, HoldoutAuditor, HoldoutLockedError, type HttpGithubClientOptions, type HypothesisManifest, type HypothesisResult, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, type ImageData, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, type InMemoryRawProviderSinkOptions, InMemoryTraceStore, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, type InterimReleaseConfidence, type InterimReleaseConfidenceInput, type JudgeConfig$1 as JudgeConfig, JudgeError, type JudgeFamily, type JudgeFleetOptions, type JudgeFn, type JudgeInput, JudgeParseError, type JudgeReplayGateArgs, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, type JudgeRubric, JudgeRunner, type JudgeScore$1 as JudgeScore, type JudgeScoreInput, type JudgeScoresRecord, type JudgeSpan, type JudgeVerdict, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type KnowledgeAcquisitionMode, type KnowledgeBundle, type KnowledgeFallbackPolicy, type KnowledgeFreshness, type KnowledgeImportance, type KnowledgeReadinessReport, type KnowledgeRecommendedAction, type KnowledgeRequirement, type KnowledgeRequirementCategory, type KnowledgeResponsibleSurface, type KnowledgeSensitivity, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, type Layer, type LayerResult, type LayerStatus, type LeaderboardOptions, type LeaderboardRow, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmCallError, type LlmCallMetadata, type LlmCallRequest, type LlmCallResult, LlmClient, type LlmClientOptions, type LlmCorrectnessCheckerOpts, type LlmJsonCall, type LlmJudgeDimension, type LlmJudgeOptions, type LlmMessage, LlmResponseError, type LlmReviewerConfig, LlmRouteAssertionError, type LlmRouteRequirements, type LlmSpan, type LlmSpanOtlpInput, type LlmUsage, LockedJsonlAppender, MODEL_PRICING, type MakeEvalToolsConfig, type MatchResult, type MatchedPair, type MatcherResult, type MaximumCharge, type McNemarResult, type Measured, type MeasurementPolicy, type MergeOptions, type Message, type MetricSamples, type MetricVerdict, MetricsCollector, type MintRolloutOptions, type MintRolloutResult, type MockTransportOpts, type ModelCostRollup, type ModelPreflight, type ModelSeats, ModelsUnreachableError, type MuffledFinder, type MuffledFinding, MultiLayerVerifier, type MultiToolchainLayerConfig, type Mutator, Mutex, type NoLeakOptions, type NoProgressOptions, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, type Objective, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, type OtelExportConfig, type OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, type OtlpExport, OtlpFileTraceStore, type OtlpFileTraceStoreOptions, type OtlpFlatLine, type OtlpResourceSpans, type OtlpSpan, type OtlpToRunRecordsOptions, type OtlpTraceRunRecord, POLICY_EDIT_AXES, POLICY_EDIT_CANDIDATE_RECORD_SCHEMA, POLICY_EDIT_TARGET_SURFACES, type PaidCallResult, type PairArmsOptions, type PairArmsResult, type PairedArmRow, type PairedArmsComparison, type PairedBootstrapOptions, type PairedBootstrapResult, type PairedCorrectness, type PairedEvalueOptions, type PairedEvalueSequence, type PairedEvalueStep, type PairedMetricDelta, type PairedSignTestResult, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, type ParetoFigureSpec, type ParetoPoint, type ParetoResult, type ParseStudentLabel, type PartitionHeldOutOptions, type PendingCostCall, type PendingCostCallView, type PersistedFinding, type PersonaConfig, type PersonaRigor, type Playbook, type PlaybookEntry, type PolicyEdit, type PolicyEditAdmission, type PolicyEditAdmissionOptions, type PolicyEditAxis, type PolicyEditCandidateRecord, type PolicyEditChange, type PolicyEditExpectedGain, type PolicyEditGainDirection, type PolicyEditGainUnit, type PolicyEditInit, type PolicyEditRisk, type PolicyEditSchemaVersion, type PolicyEditSource, type PolicyEditTarget, type PolicyEditTargetSurface, PolicyEditValidationError, type PoolSlot, type PositionalBiasResult, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, type PreferenceMemoryEntry, type PreflightModelsOptions, type PreflightOutcome, type ProducedProposal, type ProducedState, type ProductBenchmarkArm, type ProductBenchmarkArtifactPaths, type ProductBenchmarkBudgets, type ProductBenchmarkExportOptions, type ProductBenchmarkExportResult, type ProductBenchmarkManifest, type ProductBenchmarkProfileRef, type ProductBenchmarkRecord, type ProductBenchmarkRepoRef, type ProductBenchmarkRunInput, type ProductBenchmarkScenario, type ProductBenchmarkSingleRunExportOptions, type ProductBenchmarkSplit, type ProductBenchmarkSubstrateVersions, type ProductBenchmarkValidationReport, ProductClient, type ProductClientConfig, type ProfileAxisSpec, type ProjectRuntimeTrajectoryEvidenceOptions, type ProjectedOtlpSpan, type PromptHandle, PromptRegistry, type ProportionInterval, type ProposalEventLike, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type ProposeFn, type ProposeInput, type ProposeOutput, type ProposeReviewConfig, type ProposeReviewControlAction, type ProposeReviewControlConfig, type ProposeReviewControlResult, type ProposeReviewControlState, type ProposeReviewReport, type ProposeReviewShot, type ProposedSideEffect, type ProvenanceReader, type ProviderRedactor, type QueryTracesPage, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_FORMAT, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, type RawAnalystEvidence, type RawAnalystFinding, type RawProviderDirection, type RawProviderEvent, type RawProviderSink, type RawProviderSinkFilter, type RecordRunsOptions, type RedTeamCase, type RedTeamCategory, type RedTeamFinding, type RedTeamPayload, type RedTeamReport, type RedactionReport, type RedactionRule, type ReferenceEquivalenceJudgeInput, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceJudgeResult, type ReferenceEquivalenceScenario, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, type RegistryRunOpts, type ReleaseConfidenceAxis, type ReleaseConfidenceAxisName, type ReleaseConfidenceInput, type ReleaseConfidenceIssue, type ReleaseConfidenceMetrics, type ReleaseConfidenceScorecard, type ReleaseConfidenceStatus, type ReleaseConfidenceThresholds, type ReleaseTraceEvidence, type RenderReleaseReportOptions, type RenderStudentPrompt, type RepeatedActionOptions, ReplayCache, type ReplayCacheEntry, ReplayCacheMissError, type ReplayCacheStats, ReplayError, type ReplayFetchOptions, type RepoRef, type RequirementCheck, type ResearchReport, type ResearchReportCandidate, type ResearchReportDecision, type ResearchReportMethodology, type ResearchReportOptions, type ResearchReportRecommendation, type Researcher, type RetrievalSpan, type Review, type ReviewFn, type ReviewInput, type ReviewMemoryEntry, type ReviewMemoryStore, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, type RewardRow, type RiskDifferenceResult, type RobustnessResult, type RolloutCapture, type RolloutLine, type RolloutRole, type RolloutScrubber, type RolloutSplit, type RolloutStep, type RouteMap, type RoutedField, type RouterTransportOpts, type RubricDimension, type Run, type RunCommandInput, type RunCommandResult, type RunCompleteHook, type RunCompleteHookContext, type RunCostProvenance, RunCritic, type RunCriticOptions, type RunDistillationOptions, type RunDistillationResult, type RunEvidenceMetadata, type RunFilter, RunIntegrityError, type RunIntegrityExpectations, type RunIntegrityIssue, type RunIntegrityIssueCode, type RunIntegrityReport, type RunJudgeMetadata, type RunLayer, type RunOutcome, type RunPaidCallInput, type RunRecord, type RunRecordBackend, type RunRecordFilter, RunRecordValidationError, type RunScore, type RunScoreWeights, type RunSplitTag, type RunStatus, type RunTokenUsage, type RunTrace, type RuntimeEventLike, type RuntimeResolution, type RuntimeTrajectoryEvidenceProjection, type RuntimeTrajectoryEvidenceSummary, type RuntimeTrajectoryHookEvent, type RuntimeTrajectoryRecord, type RuntimeTrajectoryRunRecord, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_SCHEMA, type SandboxDriver, SandboxHarness, type SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type SandboxResult, type SandboxSdkTransportOpts, type SandboxSpan, type SatisfiedBy, type ScanOptions, type Scenario$1 as Scenario, type ScenarioCost, type ScenarioFile, ScenarioRegistry, type ScenarioResult, type ScoreKnowledgeReadinessOptions, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SearchSpanResult, type SearchTraceResult, type SeatName, type SeatPresetName, SeatUnsetError, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SelfPreferenceResult, type SemanticConceptJudgeInput, type SemanticConceptJudgeOptions, type SemanticConceptJudgeResult, type SequentialDecision, type SerializedRegex, type SeriesConvergenceOptions, type SeriesConvergenceResult, type Severity, type SftExportOptions, type SftRow, type SignTestAlternative, type SignedManifest, type SignedManifestAlgo, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, SkillUsageAnalyst, type SliceOptions, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, type Span, type SpanBase, type SpanFilter, type SpanHandle, type SpanKind, type SpanMatchRecord, SpanNotFoundError, type SpanPredicate, type SpanStatus, type SplitGoldOptions, type SseUsageMode, type SteeringBundle, type SteeringChange, type SteeringDelta, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type SteeringRolePrompt, type StepAttribution, type StopDecision, type StreamingDetector, type SuboptimalCode, type SuboptimalSignal, SubprocessSandboxDriver, type SubprocessSandboxDriverOptions, type SummaryTable, type SummaryTableOptions, type SummaryTableRow, type SupervisorRunReader, type SupervisorRunReport, type SupervisorRunRollup, type SupervisorRunSources, type SupervisorRunTree, type SynthesisReason, type SynthesisTarget, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, type TaskGold, type TaskHeadroom, type TestGradedRunOptions, type TestGradedRunResult, type TestGradedScenario, type TestOutputParser, type TestResult, type TextMatcher, type ThresholdContract, TokenCounter, type TokenSpec, type ToolCallEventLike, type ToolDef, type ToolMatcher, type ToolSpan, type ToolSpanOtlpInput, type ToolStats, type ToolUseMetrics, type ToolUseOptions, type TraceAggregate, type TraceAnalysisStore, type TraceAnalystByteBudgets, type TraceAnalystFilters, type TraceAnalystGolden, type TraceAnalystHookOptions, type TraceAnalystKindSpec, type TraceAnalystSpan, type TraceAnalystSpanKind, type TraceAnalystSpanStatus, type TraceAnalystTraceSummary, type TraceContract, TraceContractBuilder, TraceEmitter, type TraceEmitterOptions, type TraceEvent, TraceFileMissingError, type TraceInsightContext, type TraceInsightFinding, type TraceInsightPanelRole, type TraceInsightPromptInput, type TraceInsightQualityGate, type TraceInsightQuestion, type TraceInsightReadiness, type TraceInsightSuite, type TraceInsightTask, TraceNotFoundError, type TraceStore, type TraceStoreSource, type TraceStoreToOtlpOptions, type TracedAnalystOptions, type TracedJudgeOptions, type TracesToOtlpResult, type Trajectory, type TrajectoryStep, type TreatmentClass, type TreatmentGate, type TreatmentGateInput, type TreatmentGateOptions, type TrialTrace, type Turn, type TurnMetrics, type TurnResult, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, type Unavailable, type UserQuestion, type ValidationContext, ValidationError, type ValidationIssue, type ValidationResult, type VerbosityBiasResult, type Verdict, type VerdictCacheStats, type VerdictCacheStore, type Verification, VerificationError, type VerificationReport, type VerifyContext, type VerifyFn, type VerifyOptions, type ViewSpansResult, type ViewTraceOversized, type ViewTraceResult, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, type WeightedCompositeInput, type WeightedCompositeResult, type WorkerDriverContext, type WorkflowTopology, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, acquisitionPlansForKnowledgeGaps, admitPolicyEdit, adversarialJudge, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyPolicyEditToSurface, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, index$1 as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildAgreementJudge, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, buildWorkerDriverSystemPrompt, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyTreatment, cliffsDelta, clusteredPairedBinary, codeExecutionJudge, cohensD, coherenceJudge, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computePolicyEditId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAnalystAi, createAntiSlopJudge, createChatClient, createCustomJudge, createDefaultReviewer, createDomainExpertJudge, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalystKind, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultJudges, defaultParseStudentLabel, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultRenderStudentPrompt, defaultTraceInsightPanel, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fieldAgreement, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isPolicyEdit, isRetrievalSpan, isRolloutLine, isRunRecord, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadGoldScenarios, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, makePolicyEdit, makePolicyEditCandidateRecord, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalizeScores, notBlocked, objectiveEval, observeAll, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairedBootstrap, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseGoldJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, policyEditFromFinding, policyEditsFromFindings, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, index as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredSampleSize, researchReport, resolveModelPricing, resolveRunCostProvenance, resolveSeat, rolloutReward, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runDistillation, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTestGradedScenario, runsForScenario, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scorePolicyEditReadiness, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, showMeasured, signManifest, spearmanR, splitGold, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, typoMutator, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validatePolicyEdit, validatePolicyEditCandidateRecord, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
17068
+ export { AGENT_PROFILE_KINDS, ATTESTATION_ALGORITHM, type ActionExecutionPolicy, type ActionPolicyDecision, type ActionableSideInfo, type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, type AgentEvalErrorCode, type AgentInterfaceProfileLike, type AgentProfileCell, type AgentProfileCellInput, type AgentProfileCellSchemaVersion, AgentProfileCellValidationError, type AgentProfileDimensionValue, type AgentProfileHarness, type AgentProfileJson, type AgentProfileJsonObject, type AgentProfileKind, type AgentProfileRuntimeReceipt, type AgentProfileSource, type AgentProfileSourceInput, type AlignmentOp, type Analyst, type AnalystContext, type AnalystCost, type AnalystFinding, type AnalystHooks, type AnalystInputKind, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type AnalyzeTracesInput, type AnalyzeTracesOptions, type AnalyzeTracesResult, type AnalyzeTracesTurnSnapshot, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, type Artifact$1 as Artifact, type ArtifactCheck, type Artifact as ArtifactCheckArtifact, type ArtifactEventLike, type ArtifactResult, type ArtifactValidator, type AsiSeverity, type AssertCapabilityHeadroomOptions, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, type AttestationProvenance, type AttestationVerification, type AttestedReport, type AutoPrClient, AxGepaSteeringOptimizer, type AxSteeringOptimizerConfig, BENCHMARK_SPLIT_SEED, type BackendDescriptor, BackendIntegrityError, type BackendIntegrityReport, type BaselineOptions, type BaselineReport, BehaviorAssertion, type BehavioralMetrics, type BehavioralTokenSequence, type BenchmarkAdapter, type BenchmarkDatasetItem, type BenchmarkEvaluation, type BenchmarkFamily, type BenchmarkReport$1 as BenchmarkReport, type BenchmarkResponder, BenchmarkRunner, type BenchmarkRunnerConfig, type BenchmarkScenario, type BenchmarkSource, type BenchmarkTaskKind, type BisectOptions, type BisectResult, type BisectStep, type BlendWeights, type BootstrapOptions, type BootstrapResult, BudgetBreachError, BudgetGuard, type BudgetLedgerEntry, type BudgetPolicy, type BudgetSpec, CODING_HARNESSES, type CachedJudge, type CachedJudgeOptions, type CalibrationResult, CallExpectation, CallbackResearcher, type CallbackResearcherOptions, type CampaignFactoryParams, type CampaignIntegrityPolicy, type CampaignRunContext, type CampaignRunOutcome, type CampaignRunner, type CampaignScenario, type CampaignVariant, type CanaryAlert, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateComparison, type CandidateScenario, type CandidateScore, type CanonicalRawAnalystFinding, type CapabilityHeadroomOptions, type CapabilityHeadroomResult, type CaptureFetchContext, type CaptureFetchOptions, CaptureIntegrityError, type CausalAttributionReport, type CellVerdict, type ChannelRollup, type ChatCallOpts, type ChatClient, type ChatMessage, type ChatRequest, type ChatResponse, type ChatToolCall, type ChatTransport, type CheckResult, type CliBridgeTransportOpts, type CliffsMagnitude, type ClusterBootstrapInterval, type ClusterSignFlipAlternative, type ClusterSignFlipResult, type ClusteredBinaryCluster, type ClusteredMatchedPair, type ClusteredPairedBinaryOptions, type ClusteredPairedBinaryResult, type ClusteredPairedBinaryStatistics, type CollectedArtifacts, type CommandRunner, type ComparePairedArmsOptions, type CompletionCriterion, type CompletionRequirement, type CompletionVerdict, type ConceptComplexity, type ConceptFinding, type ConceptSpec, type ConceptWeightStrategy, ConfigError, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContinuousAgreement, type ContinuousAgreementOptions, type ContinuousCalibrationResult, type ContractCheckResult, type ContractJudgeOptions, type ContractMetric, type ContractReport, type ContractRule, type ContractRuleKind, type ContractSpan, type ContractVerdict, type ContractViolation, type ControlActionFailureMode, type ControlActionOutcome, type ControlBudget, type ControlContext, type ControlDecision, type ControlEvalResult, type ControlRunResult, type ControlRunToRunRecordOptions, type ControlRuntimeConfig, type ControlRuntimeError, type ControlSeverity, type ControlStep, type ControlStopPolicies, ConvergenceTracker, type CorpusAgreementOptions, type CorpusAgreementPerDimension, type CorpusAgreementReport, type CorpusScoreRecord, type CorrectnessChecker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, type CostChannel, type CostEntry, CostLedger, type CostLedgerEntry, type CostLedgerFilter, type CostLedgerHandle, type CostLedgerOptions, type CostLedgerPersistence, CostLedgerPersistenceError, type CostLedgerSummary, type CostReceipt, CostReceiptCaptureError, type CostReceiptInput, type CostReport, CostReservationExceededError, type CostResult, type CostSummary, CostTracker, type CostUsage, type CounterfactualContext, type CounterfactualMutation, type CounterfactualResult, type CounterfactualRunner, type CreateAnalystAiConfig, type CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateExperimentInput, type CreateSandboxPoolOpts, type CreateTraceAnalystKindOpts, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, type CustomTokenPricing, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, type DataAcquisitionPlan, Dataset, type DatasetDifficulty, type DatasetManifest, type DatasetOverview, type DatasetProvenance, type DatasetScenario, type DatasetSplit, type DecideNextUserTurnOpts, type DefaultAnalystRegistryOptions, type DefaultVerdict, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DescriptionLengthCandidate, type DescriptionLengthConfig, type DescriptionLengthDecision, type DescriptionLengthEvidence, DescriptionLengthGate, type DescriptionLengthRejectionCode, type DetectorEvent, type DetectorSeverity, type DetectorSignal, type DiffPolicy, type DiffScorecardOptions, type DirEntry, type DirectProviderTransportOpts, type Direction, type DiscoverPersonasOptions, type DiscoveredPersona, DockerSandboxDriver, type DriverResult, type DriverState, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, type EProcess, type EProcessOptions, type EProcessState, type EProcessStep, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type ErrorCluster, type ErrorCountPattern, type ErrorStreakOptions, type EvalCampaignOptions, type EvalCampaignResult, type EvalResult, type EvalToolDef, EvalTraceStore, type EventFilter, type EventKind, type EvidenceRef, type EvolutionRound, type ExecutorConfig, type Expectation, type Experiment, type ExperimentPlan, type ExperimentProvenance, type ExperimentRep, type ExperimentResult, type ExperimentStats, type ExperimentStore, ExperimentTracker, type ExperimentTrackerOptions, type ExperimentVerdict, type ExportableSpan, type ExportedRewardModel, type ExtractOptions, type ExtractResult, type ExtractUsageFromSseOptions, type ExtractedUsage, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, type FactorContribution, type FactorialCell, type FailedRun, type FailureClass, type FailureClassification, type FailureContext, type FailureMode, type FailureRule, type FeedbackArtifactType, type FeedbackAttempt, type FeedbackLabel, type FeedbackLabelKind, type FeedbackLabelSource, type FeedbackOptimizerRow, type FeedbackOutcome, type FeedbackPattern, type FeedbackReplayAdapter, type FeedbackReplayResult, type FeedbackSeverity, type FeedbackSplitPolicy, type FeedbackTask, type FeedbackTrajectory, type FeedbackTrajectoryFilter, type FeedbackTrajectoryStore, type FieldDestination, type FileChange, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, type FileSystemRawProviderSinkOptions, FileSystemTraceStore, type FileSystemTraceStoreOptions, type Finding, type FindingSubject, type FindingSubjectKind, type FindingsDiff, FindingsStore, type FlattenOtlpOptions, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type GainDistributionBin, type GainDistributionFigureSpec, type GainDistributionOptions, type GateDecision$1 as GateDecision, type GateEvidence, type GenericSpan, type GhCliClientOptions, type GoldenItem, type GoldenSeverity, type GoldenSpec, HARNESS_NATIVE_MODEL, type HarnessAdapter, type HarnessConfig, type HarnessExperimentConfig, type HarnessExperimentResult, type HarnessIntervention, type HarnessRunRequest, type HarnessRunResult, type HarnessScenario, type HarnessSelection, type HarnessVariant, type HarnessVariantReport, type HeadroomClass, type HeadroomInput, HeldOutGate, type HeldOutGateConfig, type HeldOutGateRejectionCode, type HeldOutPartition, type HiddenCriteriaGrader, type HiddenGradeResult, type HiddenLeak, HoldoutAuditor, HoldoutLockedError, type HttpGithubClientOptions, type HypothesisManifest, type HypothesisResult, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, type ImageData, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, type InMemoryRawProviderSinkOptions, InMemoryTraceStore, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, type InterimReleaseConfidence, type InterimReleaseConfidenceInput, type JudgeConfig$1 as JudgeConfig, JudgeError, type JudgeFamily, type JudgeFleetOptions, type JudgeFn, type JudgeInput, JudgeParseError, type JudgeReplayGateArgs, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, type JudgeRubric, JudgeRunner, type JudgeScore$1 as JudgeScore, type JudgeScoreInput, type JudgeScoresRecord, type JudgeSpan, type JudgeVerdict, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type KnowledgeAcquisitionMode, type KnowledgeBundle, type KnowledgeFallbackPolicy, type KnowledgeFreshness, type KnowledgeImportance, type KnowledgeReadinessReport, type KnowledgeRecommendedAction, type KnowledgeRequirement, type KnowledgeRequirementCategory, type KnowledgeResponsibleSurface, type KnowledgeSensitivity, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, type Layer, type LayerResult, type LayerStatus, type LeaderboardOptions, type LeaderboardRow, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmCallError, type LlmCallMetadata, type LlmCallRequest, type LlmCallResult, LlmClient, type LlmClientOptions, type LlmCorrectnessCheckerOpts, type LlmJsonCall, type LlmJudgeDimension, type LlmJudgeOptions, type LlmMessage, LlmResponseError, type LlmReviewerConfig, LlmRouteAssertionError, type LlmRouteRequirements, type LlmSpan, type LlmSpanOtlpInput, type LlmUsage, LockedJsonlAppender, MODEL_PRICING, type MakeEvalToolsConfig, type MatchResult, type MatchedPair, type MatcherResult, type MaximumCharge, type McNemarResult, type Measured, type MeasurementPolicy, type MergeOptions, type Message, type MetricSamples, type MetricVerdict, MetricsCollector, type MintRolloutOptions, type MintRolloutResult, type MockTransportOpts, type ModelCostRollup, type ModelPreflight, type ModelSeats, ModelsUnreachableError, type MuffledFinder, type MuffledFinding, MultiLayerVerifier, type MultiToolchainLayerConfig, type Mutator, Mutex, type NoLeakOptions, type NoProgressOptions, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, type Objective, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, type OtelExportConfig, type OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, type OtlpExport, OtlpFileTraceStore, type OtlpFileTraceStoreOptions, type OtlpFlatLine, type OtlpResourceSpans, type OtlpSpan, type OtlpToRunRecordsOptions, type OtlpTraceRunRecord, type PaidCallResult, type PairArmsOptions, type PairArmsResult, type PairedArmRow, type PairedArmsComparison, type PairedBootstrapOptions, type PairedBootstrapResult, type PairedCorrectness, type PairedEvalueOptions, type PairedEvalueSequence, type PairedEvalueStep, type PairedMetricDelta, type PairedSignTestResult, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, type ParetoFigureSpec, type ParetoPoint, type ParetoResult, type PartitionHeldOutOptions, type PendingCostCall, type PendingCostCallView, type PersistedFinding, type PersonaConfig, type PersonaRigor, type Playbook, type PlaybookEntry, type PoolSlot, type PositionalBiasResult, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, type PreferenceMemoryEntry, type PreflightModelsOptions, type PreflightOutcome, type ProducedProposal, type ProducedState, type ProductBenchmarkArm, type ProductBenchmarkArtifactPaths, type ProductBenchmarkBudgets, type ProductBenchmarkExportOptions, type ProductBenchmarkExportResult, type ProductBenchmarkManifest, type ProductBenchmarkProfileRef, type ProductBenchmarkRecord, type ProductBenchmarkRepoRef, type ProductBenchmarkRunInput, type ProductBenchmarkScenario, type ProductBenchmarkSingleRunExportOptions, type ProductBenchmarkSplit, type ProductBenchmarkSubstrateVersions, type ProductBenchmarkValidationReport, ProductClient, type ProductClientConfig, type ProfileAxisSpec, type ProjectRuntimeTrajectoryEvidenceOptions, type ProjectedOtlpSpan, type PromptHandle, PromptRegistry, type ProportionInterval, type ProposalEventLike, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type ProposeFn, type ProposeInput, type ProposeOutput, type ProposeReviewConfig, type ProposeReviewControlAction, type ProposeReviewControlConfig, type ProposeReviewControlResult, type ProposeReviewControlState, type ProposeReviewReport, type ProposeReviewShot, type ProposedSideEffect, type ProvenanceReader, type ProviderRedactor, type QueryTracesPage, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_FORMAT, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, type RawAnalystEvidence, type RawAnalystFinding, type RawProviderDirection, type RawProviderEvent, type RawProviderSink, type RawProviderSinkFilter, type RecordRunsOptions, type RedTeamCase, type RedTeamCategory, type RedTeamFinding, type RedTeamPayload, type RedTeamReport, type RedactionReport, type RedactionRule, type ReferenceEquivalenceJudgeInput, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceJudgeResult, type ReferenceEquivalenceScenario, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, type RegistryRunOpts, type ReleaseConfidenceAxis, type ReleaseConfidenceAxisName, type ReleaseConfidenceInput, type ReleaseConfidenceIssue, type ReleaseConfidenceMetrics, type ReleaseConfidenceScorecard, type ReleaseConfidenceStatus, type ReleaseConfidenceThresholds, type ReleaseTraceEvidence, type RenderReleaseReportOptions, type RepeatedActionOptions, ReplayCache, type ReplayCacheEntry, ReplayCacheMissError, type ReplayCacheStats, ReplayError, type ReplayFetchOptions, type RepoRef, type RequirementCheck, type ResearchReport, type ResearchReportCandidate, type ResearchReportDecision, type ResearchReportMethodology, type ResearchReportOptions, type ResearchReportRecommendation, type Researcher, type RetrievalSpan, type Review, type ReviewFn, type ReviewInput, type ReviewMemoryEntry, type ReviewMemoryStore, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, type RewardRow, type RiskDifferenceResult, type RobustnessResult, type RolloutCapture, type RolloutLine, type RolloutRole, type RolloutScrubber, type RolloutSplit, type RolloutStep, type RouteMap, type RoutedField, type RouterTransportOpts, type RubricDimension, type Run, type RunCommandInput, type RunCommandResult, type RunCompleteHook, type RunCompleteHookContext, type RunCostProvenance, RunCritic, type RunCriticOptions, type RunEvidenceMetadata, type RunFilter, RunIntegrityError, type RunIntegrityExpectations, type RunIntegrityIssue, type RunIntegrityIssueCode, type RunIntegrityReport, type RunJudgeMetadata, type RunLayer, type RunOutcome, type RunPaidCallInput, type RunRecord, type RunRecordBackend, type RunRecordFilter, RunRecordValidationError, type RunScore, type RunScoreWeights, type RunSplitTag, type RunStatus, type RunTokenUsage, type RunTrace, type RuntimeEventLike, type RuntimeResolution, type RuntimeTrajectoryEvidenceProjection, type RuntimeTrajectoryEvidenceSummary, type RuntimeTrajectoryHookEvent, type RuntimeTrajectoryRecord, type RuntimeTrajectoryRunRecord, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_SCHEMA, type SandboxDriver, SandboxHarness, type SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type SandboxResult, type SandboxSdkTransportOpts, type SandboxSpan, type SatisfiedBy, type ScanOptions, type Scenario$1 as Scenario, type ScenarioCost, type ScenarioFile, ScenarioRegistry, type ScenarioResult, type ScoreKnowledgeReadinessOptions, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SearchSpanResult, type SearchTraceResult, type SeatName, type SeatPresetName, SeatUnsetError, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SelfPreferenceResult, type SemanticConceptJudgeInput, type SemanticConceptJudgeOptions, type SemanticConceptJudgeResult, type SequentialDecision, type SerializedRegex, type SeriesConvergenceOptions, type SeriesConvergenceResult, type Severity, type SftExportOptions, type SftRow, type SignTestAlternative, type SignedManifest, type SignedManifestAlgo, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, SkillUsageAnalyst, type SliceOptions, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, type SourceLimits, type Span, type SpanBase, type SpanFilter, type SpanHandle, type SpanKind, type SpanMatchRecord, SpanNotFoundError, type SpanPredicate, type SpanStatus, type SseUsageMode, type SteeringBundle, type SteeringChange, type SteeringDelta, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type SteeringRolePrompt, type StepAttribution, type StopDecision, type StreamingDetector, type SuboptimalCode, type SuboptimalSignal, SubprocessSandboxDriver, type SubprocessSandboxDriverOptions, type SummaryTable, type SummaryTableOptions, type SummaryTableRow, type SupervisorRunReader, type SupervisorRunReport, type SupervisorRunRollup, type SupervisorRunSources, type SupervisorRunTree, type SynthesisReason, type SynthesisTarget, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, type TaskGold, type TaskHeadroom, type TestGradedRunOptions, type TestGradedRunResult, type TestGradedScenario, type TestOutputParser, type TestResult, type TextMatcher, type ThresholdContract, TokenCounter, type TokenSpec, type ToolCallEventLike, type ToolDef, type ToolMatcher, type ToolSpan, type ToolSpanOtlpInput, type ToolStats, type ToolUseMetrics, type ToolUseOptions, type TraceAggregate, type TraceAnalysisStore, type TraceAnalystByteBudgets, type TraceAnalystFilters, type TraceAnalystGolden, type TraceAnalystHookOptions, type TraceAnalystKindSpec, type TraceAnalystSpan, type TraceAnalystSpanKind, type TraceAnalystSpanStatus, type TraceAnalystTraceSummary, type TraceContract, TraceContractBuilder, TraceEmitter, type TraceEmitterOptions, type TraceEvent, TraceFileMissingError, type TraceInsightContext, type TraceInsightFinding, type TraceInsightPanelRole, type TraceInsightPromptInput, type TraceInsightQualityGate, type TraceInsightQuestion, type TraceInsightReadiness, type TraceInsightSuite, type TraceInsightTask, TraceNotFoundError, type TraceStore, type TraceStoreSource, type TraceStoreToOtlpOptions, type TracedAnalystOptions, type TracedJudgeOptions, type TracesToOtlpResult, type Trajectory, type TrajectoryStep, type TreatmentClass, type TreatmentGate, type TreatmentGateInput, type TreatmentGateOptions, type TrialTrace, type Turn, type TurnMetrics, type TurnResult, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, type Unavailable, type UserQuestion, type ValidationContext, ValidationError, type ValidationIssue, type ValidationResult, type VerbosityBiasResult, type Verdict, type VerdictCacheStats, type VerdictCacheStore, type Verification, VerificationError, type VerificationReport, type VerifyContext, type VerifyFn, type VerifyOptions, type ViewSpansResult, type ViewTraceOversized, type ViewTraceResult, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, type WeightedCompositeInput, type WeightedCompositeResult, type WorkerDriverContext, type WorkflowTopology, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, acquisitionPlansForKnowledgeGaps, adversarialJudge, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, index$1 as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, buildWorkerDriverSystemPrompt, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, codeExecutionJudge, cohensD, coherenceJudge, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAnalystAi, createAntiSlopJudge, createChatClient, createCustomJudge, createDefaultReviewer, createDomainExpertJudge, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalystKind, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultJudges, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isRetrievalSpan, isRolloutLine, isRunRecord, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalizeScores, notBlocked, objectiveEval, observeAll, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairedBootstrap, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, index as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredSampleSize, researchReport, resolveModelPricing, resolveRunCostProvenance, resolveSeat, rolloutReward, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTestGradedScenario, runsForScenario, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, typoMutator, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };