@tangle-network/agent-eval 0.117.1 → 0.118.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (143) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/dist/analyst/index.d.ts +2772 -21
  3. package/dist/analyst/index.js +7 -6
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/belief-state/index.d.ts +706 -10
  6. package/dist/belief-state/index.js +2 -1
  7. package/dist/belief-state/index.js.map +1 -1
  8. package/dist/benchmarks/index.d.ts +958 -14
  9. package/dist/benchmarks/index.js +12 -10
  10. package/dist/builder-eval/index.d.ts +449 -4
  11. package/dist/builder-eval/index.js +4 -3
  12. package/dist/builder-eval/index.js.map +1 -1
  13. package/dist/campaign/index.d.ts +4275 -73
  14. package/dist/campaign/index.js +12 -10
  15. package/dist/{chunk-VF3XSYTI.js → chunk-33JA4TFA.js} +6 -6
  16. package/dist/{chunk-4JLWXDYA.js → chunk-3EHHMC6E.js} +2 -2
  17. package/dist/{chunk-CCZIVI3F.js → chunk-BFW56GTT.js} +2 -2
  18. package/dist/{chunk-YZPO4UHR.js → chunk-FTUMG2U7.js} +124 -149
  19. package/dist/chunk-FTUMG2U7.js.map +1 -0
  20. package/dist/{chunk-E4BUPP7Z.js → chunk-HKUCJ437.js} +38 -66
  21. package/dist/chunk-HKUCJ437.js.map +1 -0
  22. package/dist/{chunk-HZHNRYHK.js → chunk-K6N6XJJX.js} +2 -2
  23. package/dist/chunk-KSDQVPLR.js +286 -0
  24. package/dist/chunk-KSDQVPLR.js.map +1 -0
  25. package/dist/chunk-MA6HLL3S.js +65 -0
  26. package/dist/chunk-MA6HLL3S.js.map +1 -0
  27. package/dist/{chunk-MGEHEHSN.js → chunk-OIIMMLRB.js} +11 -11
  28. package/dist/{chunk-DXZRATT5.js → chunk-OYZAPX5G.js} +3 -3
  29. package/dist/chunk-PXE2VKMX.js +140 -0
  30. package/dist/chunk-PXE2VKMX.js.map +1 -0
  31. package/dist/{chunk-JSJZ4PJ6.js → chunk-Q442S5AS.js} +17 -17
  32. package/dist/{chunk-ODVOOEWQ.js → chunk-QBRSJK47.js} +2 -2
  33. package/dist/{chunk-S2F4J57L.js → chunk-QKEGNI5B.js} +77 -32
  34. package/dist/chunk-QKEGNI5B.js.map +1 -0
  35. package/dist/{chunk-5UF54T55.js → chunk-S3UZOQ5Y.js} +34 -7
  36. package/dist/chunk-S3UZOQ5Y.js.map +1 -0
  37. package/dist/{chunk-FQNLDL4D.js → chunk-SVH2ANFD.js} +136 -4
  38. package/dist/chunk-SVH2ANFD.js.map +1 -0
  39. package/dist/{chunk-GQCZRZ7L.js → chunk-U5CHZ5M3.js} +9 -9
  40. package/dist/{chunk-TVVP3ZZQ.js → chunk-VQMK5FMP.js} +2 -1
  41. package/dist/{chunk-TVVP3ZZQ.js.map → chunk-VQMK5FMP.js.map} +1 -1
  42. package/dist/chunk-WDHBCA3M.js +31 -0
  43. package/dist/chunk-WDHBCA3M.js.map +1 -0
  44. package/dist/{chunk-HQPHZGL6.js → chunk-YLMUS4MM.js} +9 -9
  45. package/dist/{chunk-LQUTGLOZ.js → chunk-ZET2UAYW.js} +15 -65
  46. package/dist/chunk-ZET2UAYW.js.map +1 -0
  47. package/dist/contract/index.d.ts +4012 -38
  48. package/dist/contract/index.js +56 -29
  49. package/dist/contract/index.js.map +1 -1
  50. package/dist/control.d.ts +1013 -9
  51. package/dist/control.js +4 -3
  52. package/dist/fuzz.d.ts +194 -4
  53. package/dist/fuzz.js +3 -3
  54. package/dist/hosted/index.d.ts +498 -17
  55. package/dist/index.d.ts +10992 -1288
  56. package/dist/index.js +101 -80
  57. package/dist/index.js.map +1 -1
  58. package/dist/matrix/index.d.ts +139 -4
  59. package/dist/meta-eval/index.d.ts +862 -15
  60. package/dist/meta-eval/index.js +2 -1
  61. package/dist/meta-eval/index.js.map +1 -1
  62. package/dist/multishot/index.d.ts +214 -14
  63. package/dist/openapi.json +1 -1
  64. package/dist/pipelines/index.d.ts +392 -7
  65. package/dist/pipelines/index.js +5 -3
  66. package/dist/pipelines/index.js.map +1 -1
  67. package/dist/reporting.d.ts +1277 -17
  68. package/dist/rl.d.ts +2359 -28
  69. package/dist/rl.js +6 -5
  70. package/dist/rl.js.map +1 -1
  71. package/dist/storyboard/index.d.ts +86 -1
  72. package/dist/trace-attributes.d.ts +16 -0
  73. package/dist/trace-attributes.js +32 -0
  74. package/dist/trace-attributes.js.map +1 -0
  75. package/dist/traces.d.ts +1978 -697
  76. package/dist/traces.js +53 -32
  77. package/dist/wire/index.d.ts +655 -9
  78. package/docs/insight-report.md +44 -0
  79. package/package.json +9 -3
  80. package/dist/adversarial-B7loGVVX.d.ts +0 -19
  81. package/dist/analyst-C8HHvfJp.d.ts +0 -88
  82. package/dist/analyze-runs--2x39HZ7.d.ts +0 -81
  83. package/dist/baseline-DKq3gJpP.d.ts +0 -141
  84. package/dist/calibration-C8MTS7cw.d.ts +0 -101
  85. package/dist/chunk-5UF54T55.js.map +0 -1
  86. package/dist/chunk-E4BUPP7Z.js.map +0 -1
  87. package/dist/chunk-FQNLDL4D.js.map +0 -1
  88. package/dist/chunk-LQUTGLOZ.js.map +0 -1
  89. package/dist/chunk-S2F4J57L.js.map +0 -1
  90. package/dist/chunk-YZPO4UHR.js.map +0 -1
  91. package/dist/code-agent-session-CjZsVd19.d.ts +0 -87
  92. package/dist/control-6vuGfmDH.d.ts +0 -258
  93. package/dist/cost-ledger-DWy3XdJc.d.ts +0 -183
  94. package/dist/dataset-NENEzRgk.d.ts +0 -115
  95. package/dist/default-registry-DaK8b3fv.d.ts +0 -155
  96. package/dist/emitter-CjD7vUwv.d.ts +0 -122
  97. package/dist/errors-oeQrLqXC.d.ts +0 -74
  98. package/dist/failure-cluster-DOAcSJ87.d.ts +0 -76
  99. package/dist/feedback-trajectory-BUnM58xL.d.ts +0 -348
  100. package/dist/gepa-eESocoDi.d.ts +0 -642
  101. package/dist/index-PdX4VnPA.d.ts +0 -423
  102. package/dist/insight-report-DY4nDW9Q.d.ts +0 -310
  103. package/dist/integrity-DqlBiLyK.d.ts +0 -81
  104. package/dist/judge-calibration-7C-IDmKr.d.ts +0 -145
  105. package/dist/kind-factory-ClZmO25A.d.ts +0 -171
  106. package/dist/llm-client-qoDd18Qz.d.ts +0 -289
  107. package/dist/multi-layer-verifier-BsqKuLyN.d.ts +0 -150
  108. package/dist/off-policy-DiwuKKg7.d.ts +0 -132
  109. package/dist/outcome-store-rnXLEqSn.d.ts +0 -63
  110. package/dist/policy-edit-wG9uFEFm.d.ts +0 -455
  111. package/dist/pre-registration-BWQhJ3vz.d.ts +0 -761
  112. package/dist/provenance-DpjwyseI.d.ts +0 -541
  113. package/dist/query-CF7PG61p.d.ts +0 -35
  114. package/dist/raw-provider-sink-C46HDghv.d.ts +0 -132
  115. package/dist/release-report-C8G2i5Xi.d.ts +0 -236
  116. package/dist/researcher-C8XyxQsu.d.ts +0 -387
  117. package/dist/rubric-predictive-validity-p49lLVrE.d.ts +0 -105
  118. package/dist/run-record-BDH49H2E.d.ts +0 -360
  119. package/dist/runtime-trajectory-DGBIUt4B.d.ts +0 -49
  120. package/dist/schema-B3Q3l9Z_.d.ts +0 -201
  121. package/dist/semantic-concept-judge-CXnPEJbf.d.ts +0 -723
  122. package/dist/sequential-5iSVfzl2.d.ts +0 -139
  123. package/dist/series-convergence-D5OWMBg6.d.ts +0 -33
  124. package/dist/statistics-KUnG73jH.d.ts +0 -494
  125. package/dist/storage-DrX3v_5B.d.ts +0 -50
  126. package/dist/store-C1YxJDEK.d.ts +0 -248
  127. package/dist/store-DGqD0Pyo.d.ts +0 -116
  128. package/dist/summary-report-C5bKFfm-.d.ts +0 -445
  129. package/dist/test-graded-scenario-B0ybnPY7.d.ts +0 -166
  130. package/dist/types-BSw1rOUB.d.ts +0 -634
  131. package/dist/types-BUxNaJ8c.d.ts +0 -108
  132. package/dist/types-BkfcQnxV.d.ts +0 -313
  133. package/dist/verdict-C9MlYujm.d.ts +0 -35
  134. /package/dist/{chunk-VF3XSYTI.js.map → chunk-33JA4TFA.js.map} +0 -0
  135. /package/dist/{chunk-4JLWXDYA.js.map → chunk-3EHHMC6E.js.map} +0 -0
  136. /package/dist/{chunk-CCZIVI3F.js.map → chunk-BFW56GTT.js.map} +0 -0
  137. /package/dist/{chunk-HZHNRYHK.js.map → chunk-K6N6XJJX.js.map} +0 -0
  138. /package/dist/{chunk-MGEHEHSN.js.map → chunk-OIIMMLRB.js.map} +0 -0
  139. /package/dist/{chunk-DXZRATT5.js.map → chunk-OYZAPX5G.js.map} +0 -0
  140. /package/dist/{chunk-JSJZ4PJ6.js.map → chunk-Q442S5AS.js.map} +0 -0
  141. /package/dist/{chunk-ODVOOEWQ.js.map → chunk-QBRSJK47.js.map} +0 -0
  142. /package/dist/{chunk-GQCZRZ7L.js.map → chunk-U5CHZ5M3.js.map} +0 -0
  143. /package/dist/{chunk-HQPHZGL6.js.map → chunk-YLMUS4MM.js.map} +0 -0
@@ -1,39 +1,3584 @@
1
- import { S as Scenario, M as MutableSurface, D as DispatchContext, b as JudgeConfig, c as SurfaceProposer, G as Gate, L as LabeledScenarioStore, C as CampaignResult, d as GateDecision } from '../types-BSw1rOUB.js';
2
- export { e as CampaignAggregates, f as CampaignArtifactWriter, g as CampaignCellResult, h as CampaignCostMeter, i as CampaignTraceWriter, j as CodeSurface, k as Dispatch, l as GateContext, m as GateResult, n as GenerationCandidate, o as GenerationRecord, a as JudgeDimension, J as JudgeScore, p as Mutator, O as OptimizationProposer, q as OptimizerConfig, r as SessionScript } from '../types-BSw1rOUB.js';
3
- import { L as LoopProvenanceRecord, P as PowerPreflight, R as RunEvalOptions } from '../provenance-DpjwyseI.js';
4
- export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, D as DefaultProductionGateOptions, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, O as ObjectiveSource, c as ParetoSignificanceGateOptions, d as PromotionObjective, e as PromotionPolicy, f as buildEvidenceVector, g as composeGate, h as defaultProductionGate, i as evolutionaryProposer, j as heldOutGate, p as paretoPolicy, k as paretoSignificanceGate, r as runEval } from '../provenance-DpjwyseI.js';
5
- import { R as RunOptimizationOptions, a as RunImprovementLoopResult } from '../gepa-eESocoDi.js';
6
- export { G as GepaProposerOptions, b as REFERENCE_EQUIVALENCE_INPUT_LIMITS, c as REFERENCE_EQUIVALENCE_JUDGE_VERSION, d as ReferenceEquivalenceJudgeInput, e as ReferenceEquivalenceJudgeOptions, f as ReferenceEquivalenceJudgeResult, g as ReferenceEquivalenceScenario, h as RunCampaignOptions, i as RunImprovementLoopOptions, j as createReferenceEquivalenceJudge, k as gepaProposer, r as runCampaign, l as runImprovementLoop, m as runReferenceEquivalenceJudge } from '../gepa-eESocoDi.js';
7
- export { c as AnalystFinding, C as ChatClient, p as CreateChatClientOpts, V as createChatClient } from '../policy-edit-wG9uFEFm.js';
8
- import { C as CampaignStorage } from '../storage-DrX3v_5B.js';
9
- export { f as fsCampaignStorage, i as inMemoryCampaignStorage } from '../storage-DrX3v_5B.js';
10
- export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore, b as OutcomeStore } from '../outcome-store-rnXLEqSn.js';
11
- import { HostedTenant, EvalRunCellScore, EvalRunGenerationSnapshot, EvalRunEvent, TraceSpanEvent } from '../hosted/index.js';
12
- import { b as CostLedgerSummary, d as CostReceipt, C as CostLedger } from '../cost-ledger-DWy3XdJc.js';
13
- import { R as RunRecord, a as RunSplitTag } from '../run-record-BDH49H2E.js';
14
- import { I as InsightReport } from '../insight-report-DY4nDW9Q.js';
15
- export { C as CostProvenanceSummary, F as FailureClusterInsight, a as InterRaterInsight, J as JudgeInsight, L as LiftInsight, O as OutcomeCorrelationInsight, R as Recommendation, b as ReleaseSummary, S as ScalarDistribution } from '../insight-report-DY4nDW9Q.js';
16
- export { D as DefaultAnalystRegistryOptions, c as buildDefaultAnalystRegistry } from '../default-registry-DaK8b3fv.js';
17
- import { A as AnalyzeRunsOptions } from '../analyze-runs--2x39HZ7.js';
18
- export { a as analyzeRuns } from '../analyze-runs--2x39HZ7.js';
19
- export { C as CodeAgentSessionDiagnostic, a as CodeAgentSessionIntakeOptions, b as CodeAgentSessionIntakeResult, c as CodeAgentSessionMetrics, d as CodeAgentSessionSource, P as ParsedCodeAgentJsonl, f as fromClaudeCodeSession, e as fromCodexSession, g as fromKimiCodeSession, h as fromOpenCodeSession, i as fromPiSession, j as fromPigraphSession, p as parseCodeAgentJsonl } from '../code-agent-session-CjZsVd19.js';
20
- import '../llm-client-qoDd18Qz.js';
21
- import '../errors-oeQrLqXC.js';
22
- import '../raw-provider-sink-C46HDghv.js';
23
- import '../statistics-KUnG73jH.js';
24
- import '../judge-calibration-7C-IDmKr.js';
25
- import '../types-BkfcQnxV.js';
26
- import '@tangle-network/tcloud';
27
- import '../dataset-NENEzRgk.js';
28
- import '../store-DGqD0Pyo.js';
29
- import '../schema-B3Q3l9Z_.js';
30
- import '../store-C1YxJDEK.js';
31
- import '@tangle-network/agent-interface';
32
- import '../summary-report-C5bKFfm-.js';
33
- import '../failure-cluster-DOAcSJ87.js';
34
- import '@ax-llm/ax';
35
- import '../kind-factory-ClZmO25A.js';
36
- import 'zod';
1
+ import { AxFunction, AxAIService } from '@ax-llm/ax';
2
+ import { z } from 'zod';
3
+
4
+ type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
5
+ type AgentProfileJson = string | number | boolean | null | AgentProfileJson[] | {
6
+ [key: string]: AgentProfileJson;
7
+ };
8
+ type AgentProfileDimensionValue = string | number | boolean | null;
9
+ interface AgentProfileSource {
10
+ /** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */
11
+ kind: string;
12
+ /** sha256 over the canonical source profile object. */
13
+ hash: string;
14
+ }
15
+ interface AgentProfileHarness {
16
+ id: string;
17
+ version?: string;
18
+ hash?: string;
19
+ }
20
+ interface AgentProfileCell {
21
+ schemaVersion: AgentProfileCellSchemaVersion;
22
+ cellId: string;
23
+ profileId: string;
24
+ sourceProfile: AgentProfileSource;
25
+ harness?: AgentProfileHarness;
26
+ model?: string;
27
+ promptHash?: string;
28
+ dimensions?: Record<string, AgentProfileDimensionValue>;
29
+ }
30
+
31
+ type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'tool_argument_error' | 'tool_recovery_failure' | 'hallucination' | 'instruction_following' | 'safety_refusal_miss' | 'policy_violation' | 'budget_exceeded' | 'format_drift' | 'permission_escalation' | 'pii_leak' | 'cost_overrun' | 'timeout' | 'sandbox_failure' | 'missing_user_data' | 'missing_domain_data' | 'missing_codebase_context' | 'missing_runtime_context' | 'missing_credentials' | 'missing_integration_connection' | 'missing_integration_scope' | 'integration_approval_required' | 'integration_auth_expired' | 'integration_provider_failure' | 'bad_integration_manifest' | 'unsafe_integration_write_denied' | 'stale_external_data' | 'bad_retrieval' | 'insufficient_evidence' | 'contradictory_evidence' | 'ambiguous_user_intent' | 'knowledge_readiness_blocked' | 'unknown';
32
+
33
+ /**
34
+ * Paper-grade RunRecord schema + runtime validator.
35
+ *
36
+ * Every run that participates in a promotion gate, paper table, or
37
+ * researcher loop SHOULD be recorded as a `RunRecord`. The mandatory
38
+ * fields are exactly those the paper "Two Loops, Three Roles" requires
39
+ * for reproducibility: who/what/when/cost/seed/hash, plus the search vs
40
+ * holdout split tag and either a `searchScore` or a `holdoutScore`.
41
+ *
42
+ * This is intentionally NOT a replacement for the rich `Run` /
43
+ * `ProposeReviewReport` / `ScenarioResult` types already in the
44
+ * package. Those are runtime structures with full provenance. A
45
+ * `RunRecord` is the analysis-time projection — the JSON-friendly
46
+ * row you'd put in a parquet file or paste into a notebook.
47
+ *
48
+ * Validate at the boundary:
49
+ *
50
+ * const rec = validateRunRecord(rawJson) // throws on missing
51
+ * const ok = isRunRecord(rawJson) // boolean check
52
+ * const rec = parseRunRecordSafe(rawJson) // { ok, value | error }
53
+ *
54
+ * The validator runs in pure TS — zod is intentionally NOT a
55
+ * dependency. Round-trip tested in `tests/run-record.test.ts`.
56
+ */
57
+
58
+ /** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
59
+ * combined train+test pool that the optimizer is allowed to read. */
60
+ type RunSplitTag = 'search' | 'dev' | 'holdout';
61
+ interface RunTokenUsage {
62
+ input: number;
63
+ /** All generated tokens charged as output, including reasoning tokens. */
64
+ output: number;
65
+ /** Reasoning-token subset of `output`, when the provider reports it. */
66
+ reasoning?: number;
67
+ /** Prompt tokens served from a provider cache. */
68
+ cached?: number;
69
+ /** Prompt tokens written into a provider cache. */
70
+ cacheWrite?: number;
71
+ }
72
+ /**
73
+ * How a run's USD amount was obtained.
74
+ *
75
+ * `costUsd` remains mandatory for wire compatibility. New producers should
76
+ * always populate this discriminated union so a missing bill is never
77
+ * mistaken for an observed zero-dollar run. For `uncaptured`, `costUsd` uses
78
+ * the legacy `0` sentinel while this field carries the truthful null.
79
+ */
80
+ type RunCostProvenance = {
81
+ kind: 'observed';
82
+ usd: number;
83
+ } | {
84
+ kind: 'estimated';
85
+ usd: number;
86
+ } | {
87
+ kind: 'uncaptured';
88
+ usd: null;
89
+ };
90
+ interface RunJudgeMetadata {
91
+ model: string;
92
+ promptVersion: string;
93
+ /** [0,1] confidence the judge declared. Constant judge confidence
94
+ * across many runs is a fallback signal (see `canary.ts`). */
95
+ confidence: number;
96
+ /** True if the judge degraded to a fallback path (rules-only,
97
+ * prior-call cache, etc.). The canary uses this to alert. */
98
+ fallback: boolean;
99
+ }
100
+ /**
101
+ * Per-judge / per-dimension breakdown for runs scored by an ensemble of
102
+ * judges over a multi-dimensional rubric.
103
+ *
104
+ * The collapsed `outcome.searchScore` / `holdoutScore` carries the
105
+ * composite the gate uses. The full breakdown belongs here so consumers
106
+ * can answer "which judge disagreed?", "which dimension dragged the
107
+ * composite down?", and "did half the panel fail?" without re-running.
108
+ *
109
+ * `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and
110
+ * `composite` are convenience projections — derivable but precomputed so
111
+ * downstream IRR primitives (`interRaterReliability`,
112
+ * `corpusInterRaterAgreement`) and reporters don't pay the same
113
+ * aggregation twice.
114
+ *
115
+ * Fail-loud discipline: judges that errored out land in `failedJudges`
116
+ * by id. A missing key in `perJudge` is ambiguous (silent zero vs not
117
+ * run); the explicit list makes a partial-failure recorded as such.
118
+ */
119
+ interface JudgeScoresRecord {
120
+ /** Per-judge per-dimension scores. `{ "kimi-k2.6": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */
121
+ perJudge: Record<string, Record<string, number>>;
122
+ /** Per-dim mean across judges. Convenience — derivable from `perJudge`. */
123
+ perDimMean: Record<string, number>;
124
+ /** Composite mean across all dims and judges. Mirrors the score
125
+ * the gate sees on `outcome.searchScore` / `holdoutScore`. */
126
+ composite: number;
127
+ /** Judges that errored or returned an unparseable verdict. Recorded
128
+ * by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,
129
+ * not inferred from missing keys in `perJudge`. */
130
+ failedJudges?: string[];
131
+ /** Free-form notes the judges emitted (joined across judges or
132
+ * first-judge only — consumer's choice). */
133
+ notes?: string;
134
+ }
135
+ interface RunOutcome {
136
+ /** Score on the search/optimization split. Optional because a
137
+ * holdout-only evaluation only fills `holdoutScore`. */
138
+ searchScore?: number;
139
+ /** Score on the held-out split. Optional because a search-only run
140
+ * only fills `searchScore`. At least one must be present. */
141
+ holdoutScore?: number;
142
+ /** Bag of any other metric the run produced — judge dimensions,
143
+ * pass/fail counters, latency stats, etc. Numeric only — keeps
144
+ * reporters honest. */
145
+ raw: Record<string, number>;
146
+ /** Per-judge / per-dim breakdown. Consumers writing ensemble
147
+ * judgements populate this; substrate primitives like
148
+ * `interRaterReliability` and `corpusInterRaterAgreement` accept
149
+ * these records as input. Optional — single-judge or scalar-only
150
+ * runs leave it unset. */
151
+ judgeScores?: JudgeScoresRecord;
152
+ /** Authenticity / realness verdict — did the run build the REAL thing on the
153
+ * intended infra, or fake it (see `./authenticity`)? Optional: only domains
154
+ * with an authenticity config populate it. Carried in the corpus so the
155
+ * flywheel / off-policy learning can optimize for real completion, not gamed
156
+ * pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run
157
+ * must not count as a real success regardless of `score`. */
158
+ realness?: {
159
+ score: number;
160
+ gated: boolean;
161
+ reason?: string;
162
+ };
163
+ }
164
+ /**
165
+ * Mandatory paper-grade fields for a single evaluation run. Optional
166
+ * fields are extension points; mandatory fields throw if missing.
167
+ *
168
+ * Hash discipline:
169
+ * - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the
170
+ * model (after any steering bundle merge).
171
+ * - `configHash` is the sha256 of the effective run config (model,
172
+ * temperature, tools, judges, splits). The pair (promptHash,
173
+ * configHash) uniquely identifies an experiment cell.
174
+ *
175
+ * Model snapshot discipline:
176
+ * - `model` MUST encode a snapshot version. Bare aliases like
177
+ * `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.
178
+ * Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.
179
+ */
180
+ interface RunRecord {
181
+ /** UUID for the run. */
182
+ runId: string;
183
+ /** Logical experiment grouping (a treatment vs a baseline within
184
+ * the same sweep should share `experimentId`). */
185
+ experimentId: string;
186
+ /** Stable identifier for the candidate (variant) being run. The
187
+ * promotion gate compares two `candidateId`s on matched items. */
188
+ candidateId: string;
189
+ /** RNG seed for the run. Always recorded — silent re-seeding is
190
+ * the most common cause of non-reproducible numbers. */
191
+ seed: number;
192
+ /** Model identifier WITH snapshot version. */
193
+ model: string;
194
+ /** sha256 of the effective prompt (post-steering). */
195
+ promptHash: string;
196
+ /** sha256 of the effective config. */
197
+ configHash: string;
198
+ /** Git SHA the harness was run from. */
199
+ commitSha: string;
200
+ /** End-to-end wall-clock duration in milliseconds. */
201
+ wallMs: number;
202
+ /** Time spent queued before execution started, if known. */
203
+ queueMs?: number;
204
+ /** Total USD cost. Mandatory — runs without a cost number are
205
+ * unbounded by definition and must not be admitted into the gate.
206
+ * `0` is retained as the compatibility sentinel for an uncaptured amount;
207
+ * inspect `costProvenance` before treating it as observed. */
208
+ costUsd: number;
209
+ /** Observed, model-priced estimate, or genuinely uncaptured USD amount.
210
+ * Optional only so existing serialized RunRecords remain valid. */
211
+ costProvenance?: RunCostProvenance;
212
+ /** Token usage breakdown. */
213
+ tokenUsage: RunTokenUsage;
214
+ /** Judge-side metadata, if a judge was used. */
215
+ judgeMetadata?: RunJudgeMetadata;
216
+ /** Per-split scores + raw bag. */
217
+ outcome: RunOutcome;
218
+ /** Canonical, cross-agent failure class drawn from the shared
219
+ * `FAILURE_CLASSES` taxonomy. This is the aggregation key that makes
220
+ * "which failure dominates across the whole fleet" answerable in ONE
221
+ * vocabulary — every agent classifies against the same enum. Producers
222
+ * set it via the substrate classifier; leave unset only when the failure
223
+ * genuinely can't be classified. */
224
+ failureClass?: FailureClass;
225
+ /** Free-form domain-specific failure detail, scoped UNDER `failureClass`
226
+ * (e.g. failureClass='tool_recovery_failure', failureMode='forge_build_unsatisfied').
227
+ * The within-agent drill-down; `failureClass` is the cross-agent key. */
228
+ failureMode?: string;
229
+ /** Which split this run was drawn from. */
230
+ splitTag: RunSplitTag;
231
+ /**
232
+ * Stable scenario identifier the run was scored against. Optional for
233
+ * backwards compatibility, but **strongly recommended**: every primitive
234
+ * that pairs runs by scenario (preferences, paired stats, BT tournament)
235
+ * keys on this. The campaign artifact populates it canonically; legacy
236
+ * runs without it fall back to inference from `outcome.raw.scenario_id`
237
+ * or `experimentId`.
238
+ */
239
+ scenarioId?: string;
240
+ /**
241
+ * Canonical identity for the agent profile cell that produced this row:
242
+ * profile artifact hash plus optional harness/model/prompt/reporting
243
+ * dimensions. Use `agentProfile.cellId` to group persona sweeps and
244
+ * longitudinal reports by the complete source profile, not by a loose
245
+ * candidate label or opaque config hash.
246
+ */
247
+ agentProfile?: AgentProfileCell;
248
+ }
249
+
250
+ /**
251
+ * Shared types for the trace-analyst module.
252
+ *
253
+ * Wire format. The store interface speaks `OtlpSpanLike` rows — one JSONL
254
+ * line per span, OTLP-shaped. We do NOT depend on a specific tracing
255
+ * vendor at the type level. Adapter
256
+ * layers map upstream shapes onto this interface.
257
+ *
258
+ * Design constraint. Every read operation that can return arbitrary
259
+ * payload must carry a byte budget so the agent's tool result stays
260
+ * bounded regardless of input trace size. Oversized responses
261
+ * substitute a deterministic summary instead of bytes — see
262
+ * `ViewTraceOversized`.
263
+ */
264
+ /** OTLP span kind (subset we actually use). */
265
+ type TraceAnalystSpanKind = 'AGENT' | 'LLM' | 'TOOL' | 'CHAIN' | 'GUARDRAIL' | 'SPAN' | 'UNKNOWN';
266
+ type TraceAnalystSpanStatus = 'OK' | 'ERROR' | 'UNSET';
267
+ /** Subset of OTLP span fields the analyst exposes to the agent. The
268
+ * store's job is to project upstream's full span shape down to this
269
+ * view — the analyst never sees vendor extensions directly. */
270
+ interface TraceAnalystSpan {
271
+ trace_id: string;
272
+ span_id: string;
273
+ parent_span_id: string | null;
274
+ name: string;
275
+ kind: TraceAnalystSpanKind;
276
+ start_time: string;
277
+ end_time: string;
278
+ duration_ms: number;
279
+ status: TraceAnalystSpanStatus;
280
+ status_message?: string;
281
+ service_name: string | null;
282
+ agent_name: string | null;
283
+ model_name: string | null;
284
+ tool_name: string | null;
285
+ /** Raw JSON-serialisable attribute map. May contain large strings;
286
+ * callers must respect the per-attribute byte cap. */
287
+ attributes: Record<string, unknown>;
288
+ }
289
+ interface TraceAnalystTraceSummary {
290
+ trace_id: string;
291
+ service_name: string | null;
292
+ agent_name: string | null;
293
+ span_count: number;
294
+ has_errors: boolean;
295
+ start_time: string;
296
+ end_time: string;
297
+ duration_ms: number;
298
+ raw_jsonl_bytes: number;
299
+ models: string[];
300
+ tools: string[];
301
+ }
302
+ interface TraceAnalystFilters {
303
+ /** Restrict to traces that contain at least one error span. */
304
+ has_errors?: boolean;
305
+ /** Match if any span's `service.name` is in this list. */
306
+ service_names?: string[];
307
+ /** Match if any span's `agent.name` is in this list. */
308
+ agent_names?: string[];
309
+ /** Match if any LLM span's `llm.model_name` is in this list. */
310
+ model_names?: string[];
311
+ /** Match if any tool span's `tool.name` is in this list. */
312
+ tool_names?: string[];
313
+ /** ISO-8601 lower bound on the trace's earliest start time. */
314
+ start_time_after?: string;
315
+ /** ISO-8601 upper bound on the trace's earliest start time. */
316
+ start_time_before?: string;
317
+ /** Single regex applied to raw JSONL bytes for the trace. Opt-in;
318
+ * expensive on large datasets. Use the indexed filters above first. */
319
+ regex_pattern?: string;
320
+ }
321
+ /** One distinct error signature across the dataset — the deterministic unit of
322
+ * failure coverage. Signatures normalize volatile tokens (digits, hex/uuids,
323
+ * paths, durations) out of the span `status_message` so semantically identical
324
+ * failures collapse into one cluster. An analyst that accounts for every
325
+ * cluster has, by construction, covered every distinct failure mode. */
326
+ interface ErrorCluster {
327
+ /** Normalized status_message — the cluster key. */
328
+ signature: string;
329
+ /** A verbatim, un-normalized exemplar message (for exact-string citation). */
330
+ status_message_sample: string;
331
+ /** The span name that most often carries this signature, if any. */
332
+ span_name: string | null;
333
+ /** The tool that most often carries this signature, if any. */
334
+ tool_name: string | null;
335
+ trace_count: number;
336
+ span_count: number;
337
+ /** trace_count / total error traces in the matched set (0..1). */
338
+ prevalence: number;
339
+ /** Real trace ids carrying this signature (capped), passable to view/search. */
340
+ exemplar_trace_ids: string[];
341
+ /** Real span ids carrying this signature (capped). */
342
+ exemplar_span_ids: string[];
343
+ }
344
+ interface DatasetOverview {
345
+ total_traces: number;
346
+ raw_jsonl_bytes: number;
347
+ services: string[];
348
+ agents: string[];
349
+ models: string[];
350
+ tool_names: string[];
351
+ /** Up to 20 real trace ids the agent may pass to view/search tools. */
352
+ sample_trace_ids: string[];
353
+ errors: {
354
+ trace_count: number;
355
+ span_count: number;
356
+ };
357
+ /** The COMPLETE deterministic error-signature population, sorted by
358
+ * trace_count desc. This is the failure-coverage checklist: an analysis is
359
+ * complete only when every cluster here is accounted for. Empty when the
360
+ * matched set has no error spans. */
361
+ error_clusters: ErrorCluster[];
362
+ time_range: {
363
+ earliest: string;
364
+ latest: string;
365
+ } | null;
366
+ }
367
+ interface QueryTracesPage {
368
+ traces: TraceAnalystTraceSummary[];
369
+ total: number;
370
+ has_more: boolean;
371
+ }
372
+ /** Full-trace view. When the response would exceed the per-call byte
373
+ * budget, `oversized` is populated INSTEAD of `spans` so the agent
374
+ * knows to switch to `searchTrace` / `viewSpans`. */
375
+ interface ViewTraceResult {
376
+ trace_id: string;
377
+ spans?: TraceAnalystSpan[];
378
+ oversized?: ViewTraceOversized;
379
+ }
380
+ interface ViewTraceOversized {
381
+ span_count: number;
382
+ /** Names with their counts, sorted desc. Capped at 20 entries. */
383
+ top_span_names: Array<[string, number]>;
384
+ /** Largest single span body (bytes after attribute-cap projection). */
385
+ span_response_bytes_max: number;
386
+ error_span_count: number;
387
+ }
388
+ interface ViewSpansResult {
389
+ trace_id: string;
390
+ spans: TraceAnalystSpan[];
391
+ /** Number of requested span ids that were not found in the trace. */
392
+ missing_span_ids: string[];
393
+ /** Number of attribute fields truncated to fit the per-attribute cap. */
394
+ truncated_attribute_count: number;
395
+ }
396
+ interface SpanMatchRecord {
397
+ trace_id: string;
398
+ span_id: string;
399
+ span_name: string;
400
+ span_kind: TraceAnalystSpanKind;
401
+ /** JSON pointer-style path to the matched value, e.g.
402
+ * `attributes."llm.input_messages"[2].content`. */
403
+ attribute_path: string;
404
+ matched_text: string;
405
+ context_before: string;
406
+ context_after: string;
407
+ match_offset: number;
408
+ }
409
+ interface SearchTraceResult {
410
+ trace_id: string;
411
+ hits: SpanMatchRecord[];
412
+ total_matches: number;
413
+ has_more: boolean;
414
+ }
415
+ interface SearchSpanResult {
416
+ trace_id: string;
417
+ span_id: string;
418
+ hits: SpanMatchRecord[];
419
+ total_matches: number;
420
+ has_more: boolean;
421
+ }
422
+
423
+ /**
424
+ * `TraceAnalysisStore` — read-side interface the trace-analyst calls
425
+ * through. Six operations, all bounded:
426
+ *
427
+ * - `getOverview(filters?)` — dataset rollup + sample trace ids.
428
+ * - `queryTraces(filters?, limit, offset)` — paginated summaries.
429
+ * - `countTraces(filters?)` — cheap count without materialisation.
430
+ * - `viewTrace(trace_id, perAttrCap)` — full span list, oversized → summary.
431
+ * - `viewSpans(trace_id, span_ids, perAttrCap)` — surgical span fetch.
432
+ * - `searchTrace(trace_id, regex, max_matches)` — bounded regex hits.
433
+ * - `searchSpan(trace_id, span_id, regex, max_matches)` — single-span search.
434
+ *
435
+ * Multiple implementations ship in the core (`OtlpFileTraceStore`).
436
+ * Downstream callers can supply their own — e.g. a DuckDB-backed
437
+ * adapter or an in-memory adapter for tests — by implementing this
438
+ * interface.
439
+ *
440
+ * Filters compose with AND semantics. Empty/undefined fields impose
441
+ * no constraint. `regex_pattern` is the only opt-in raw-bytes scan —
442
+ * implementations may skip it via `count`/`overview` when not set.
443
+ */
444
+
445
+ interface TraceAnalysisStore {
446
+ getOverview(filters?: TraceAnalystFilters): Promise<DatasetOverview>;
447
+ queryTraces(opts: {
448
+ filters?: TraceAnalystFilters;
449
+ limit: number;
450
+ offset?: number;
451
+ }): Promise<QueryTracesPage>;
452
+ countTraces(filters?: TraceAnalystFilters): Promise<number>;
453
+ viewTrace(opts: {
454
+ trace_id: string;
455
+ /** Override per-attribute byte cap. Defaults to discovery budget. */
456
+ per_attribute_byte_cap?: number;
457
+ }): Promise<ViewTraceResult>;
458
+ viewSpans(opts: {
459
+ trace_id: string;
460
+ span_ids: readonly string[];
461
+ /** Override per-attribute byte cap. Defaults to surgical budget. */
462
+ per_attribute_byte_cap?: number;
463
+ }): Promise<ViewSpansResult>;
464
+ searchTrace(opts: {
465
+ trace_id: string;
466
+ regex_pattern: string;
467
+ /** Hard cap on matches returned. Default 50. */
468
+ max_matches?: number;
469
+ }): Promise<SearchTraceResult>;
470
+ searchSpan(opts: {
471
+ trace_id: string;
472
+ span_id: string;
473
+ regex_pattern: string;
474
+ max_matches?: number;
475
+ }): Promise<SearchSpanResult>;
476
+ }
477
+
478
+ type CostChannel = 'agent' | 'judge' | 'verifier' | 'analyst' | 'driver' | (string & {});
479
+ interface CostUsage {
480
+ inputTokens: number;
481
+ outputTokens: number;
482
+ cachedTokens?: number;
483
+ }
484
+ interface CostCallBase {
485
+ callId: string;
486
+ channel: CostChannel;
487
+ phase: string;
488
+ actor: string;
489
+ model: string;
490
+ maximumCostUsd?: number;
491
+ tags?: Record<string, string>;
492
+ timestamp: number;
493
+ }
494
+ interface CostReceipt extends CostCallBase, CostUsage {
495
+ status: 'settled';
496
+ costUsd: number;
497
+ costUnknown: boolean;
498
+ usageUnknown?: boolean;
499
+ pricing?: {
500
+ inputUsdPerThousand: number;
501
+ outputUsdPerThousand: number;
502
+ };
503
+ actualCostUsd?: number;
504
+ error?: string;
505
+ }
506
+ interface CostReceiptInput extends CostUsage {
507
+ model: string;
508
+ actualCostUsd?: number;
509
+ costUnknown?: boolean;
510
+ usageUnknown?: boolean;
511
+ }
512
+ type MaximumCharge = {
513
+ externallyEnforcedMaximumUsd: number;
514
+ } | ({
515
+ model: string;
516
+ } & CostUsage);
517
+ interface RunPaidCallInput<T> {
518
+ callId?: string;
519
+ channel: CostChannel;
520
+ phase: string;
521
+ actor: string;
522
+ /** Used before a provider receipt exists and on failures without one. */
523
+ model?: string;
524
+ tags?: Record<string, string>;
525
+ signal?: AbortSignal;
526
+ /** Provider-enforced dollar maximum, or maximum priced token usage. Required when capped. */
527
+ maximumCharge?: MaximumCharge;
528
+ /** `callId` can be forwarded as the provider's idempotency key. */
529
+ execute(signal: AbortSignal, callId: string): Promise<T>;
530
+ receipt(value: T): CostReceiptInput;
531
+ receiptFromError?(error: Error): CostReceiptInput | undefined;
532
+ }
533
+ type PaidCallResult<T> = {
534
+ succeeded: true;
535
+ callId: string;
536
+ value: T;
537
+ receipt: CostReceipt;
538
+ } | {
539
+ succeeded: false;
540
+ callId?: string;
541
+ error: Error;
542
+ receipt?: CostReceipt;
543
+ };
544
+ interface ChannelRollup {
545
+ channel: CostChannel;
546
+ calls: number;
547
+ inputTokens: number;
548
+ outputTokens: number;
549
+ cachedTokens: number;
550
+ costUsd: number;
551
+ unpricedCalls: number;
552
+ unknownUsageCalls: number;
553
+ }
554
+ interface CostLedgerSummary {
555
+ totalCalls: number;
556
+ pendingCalls: number;
557
+ unresolvedCalls: number;
558
+ reservedCostUsd: number;
559
+ inputTokens: number;
560
+ outputTokens: number;
561
+ cachedTokens: number;
562
+ totalCostUsd: number;
563
+ byChannel: ChannelRollup[];
564
+ unpricedModels: string[];
565
+ fullyPriced: boolean;
566
+ usageComplete: boolean;
567
+ accountingComplete: boolean;
568
+ incompleteReasons: string[];
569
+ }
570
+ interface CostLedgerFilter {
571
+ channel?: CostChannel;
572
+ phase?: string;
573
+ tags?: Record<string, string>;
574
+ }
575
+ /** Append-only storage. `append` must atomically reject stale revisions. */
576
+ interface CostLedgerPersistence {
577
+ read(): {
578
+ revision: string;
579
+ events: string;
580
+ };
581
+ append(expectedRevision: string, event: string): string | undefined;
582
+ }
583
+ interface CostLedgerOptions {
584
+ costCeilingUsd?: number;
585
+ persistence?: CostLedgerPersistence;
586
+ /** Import already-settled receipts without admitting new paid work. */
587
+ receipts?: readonly CostReceipt[];
588
+ }
589
+ /** Run-wide paid-call admission, durable call state, receipts, and summaries. */
590
+ declare class CostLedger {
591
+ private readonly records;
592
+ private readonly activeCallIds;
593
+ private readonly lateCallIds;
594
+ private completedTasks;
595
+ private revision;
596
+ private costLimitPersisted;
597
+ readonly costCeilingUsd?: number;
598
+ private readonly persistence?;
599
+ constructor(input?: number | CostLedgerOptions);
600
+ runPaidCall<T>(input: RunPaidCallInput<T>): Promise<PaidCallResult<T>>;
601
+ /** Settle a call left pending by a crashed process after reconciling with the provider. */
602
+ reconcile(callId: string, observed: CostReceiptInput, options?: {
603
+ error?: string;
604
+ }): CostReceipt;
605
+ list(filter?: CostLedgerFilter): CostReceipt[];
606
+ summary(filter?: CostLedgerFilter): CostLedgerSummary;
607
+ markCompleted(count?: number): void;
608
+ costPerCompletedTask(): number | null;
609
+ private execute;
610
+ private captureLateOutcome;
611
+ private commitOutcome;
612
+ private captureFailure;
613
+ private commitReceipt;
614
+ private resolveMaximum;
615
+ private hasIncompleteSettledCall;
616
+ private appendRecord;
617
+ private ensureCostLimitPersisted;
618
+ private appendEvent;
619
+ }
620
+
621
+ interface Scenario$1 {
622
+ id: string;
623
+ persona: string;
624
+ label: string;
625
+ thesis: string;
626
+ dimensions: string[];
627
+ turns: Turn[];
628
+ artifactChecks: ArtifactCheck[];
629
+ systemPromptAppend?: string;
630
+ }
631
+ interface Turn {
632
+ user: string;
633
+ expectedBehaviors: string[];
634
+ adversarial?: boolean;
635
+ feedbackType?: 'correction' | 'rejection' | 'vague' | 'contradictory' | 'escalation';
636
+ }
637
+ interface ArtifactCheck {
638
+ type: 'vault_file_exists' | 'vault_file_contains' | 'block_extracted' | 'code_valid' | 'generation_produced' | 'tool_created' | string;
639
+ target: string;
640
+ contains?: string;
641
+ minCount?: number;
642
+ description: string;
643
+ }
644
+ interface TurnResult {
645
+ turnIndex: number;
646
+ userMessage: string;
647
+ agentResponse: string;
648
+ durationMs: number;
649
+ blocksExtracted: {
650
+ type: string;
651
+ title: string;
652
+ }[];
653
+ containsCode: boolean;
654
+ containsToolCall: boolean;
655
+ }
656
+ interface CollectedArtifacts {
657
+ vaultFiles: {
658
+ path: string;
659
+ content: string;
660
+ }[];
661
+ blocksExtracted: {
662
+ type: string;
663
+ fields: Record<string, string>;
664
+ }[];
665
+ codeBlocks: {
666
+ language: string;
667
+ code: string;
668
+ }[];
669
+ toolCalls: string[];
670
+ }
671
+ interface JudgeInput {
672
+ scenario: Scenario$1;
673
+ turns: TurnResult[];
674
+ artifacts: CollectedArtifacts;
675
+ /** Shared ledger for paid built-in judges. Direct calls default to an uncapped ledger. */
676
+ costLedger?: CostLedger;
677
+ costPhase?: string;
678
+ costTags?: Record<string, string>;
679
+ signal?: AbortSignal;
680
+ /** Exact maximum provider attempts configured on the supplied TCloud client. */
681
+ tcloudMaximumAttempts?: number;
682
+ }
683
+
684
+ /**
685
+ * RawProviderSink — first-class persistence for the actual HTTP-level
686
+ * request/response bodies of every LLM provider call.
687
+ *
688
+ * Why this is a separate sink from the structured `LlmSpan`:
689
+ *
690
+ * - `LlmSpan` records the *intent* — model name, messages, output text,
691
+ * usage. It's what dashboards read; it's NOT enough for forensics.
692
+ * - When a downstream consumer reports "the verifier used the wrong route"
693
+ * or "tokens look right but reasoning was missing," the only way to
694
+ * answer is the raw HTTP body. Span fields can lie (a proxy can echo
695
+ * a different `model` value than what actually answered); the raw
696
+ * response is ground truth.
697
+ *
698
+ * Default behaviour: opt-in. Pass `rawSink` to `LlmClientOptions` (or the
699
+ * matrix runner / BuilderSession sets it up automatically) and every
700
+ * request, response, and error is recorded — including retries, with the
701
+ * attempt index attached so a flaky call's full event chain is recoverable.
702
+ *
703
+ * Redaction is enforced at sink time. The default redactor strips
704
+ * `Authorization`, `X-Api-Key`, `X-Auth-Token`, `Cookie` headers and any
705
+ * payload field whose key matches `apiKey | api_key | bearer | password |
706
+ * secret | token` (case-insensitive). Override via the sink constructor or
707
+ * the per-call `redactor`. The `redactedFields` array on the persisted
708
+ * event lets a reviewer see what was stripped without exposing the values.
709
+ */
710
+ type RawProviderDirection = 'request' | 'response' | 'error';
711
+ interface RawProviderEvent {
712
+ /** Stable id. Generated by the sink if omitted. */
713
+ eventId: string;
714
+ /** Trace context populated by `LlmClient` when the call is wrapped in a span. */
715
+ runId?: string;
716
+ spanId?: string;
717
+ /**
718
+ * Logical provider name. Free-form so callers can use whatever id matches
719
+ * their topology (`'openai'`, `'anthropic'`, `'tangle-router'`, …). When
720
+ * omitted, derived from `baseUrl` in `LlmClientOptions`.
721
+ */
722
+ provider: string;
723
+ model: string;
724
+ /** Endpoint path, e.g. `'/v1/chat/completions'`. */
725
+ endpoint: string;
726
+ /** Base URL used for the call (already-normalised — no trailing slash). */
727
+ baseUrl: string;
728
+ /** 0-indexed retry attempt. The first attempt is 0; a retried call gets 1, 2, … */
729
+ attemptIndex: number;
730
+ direction: RawProviderDirection;
731
+ /** Unix ms. */
732
+ timestamp: number;
733
+ /** Wall-clock duration of the call leg. Set on `response` and `error` events; null on `request`. */
734
+ durationMs?: number;
735
+ statusCode?: number;
736
+ requestHeaders?: Record<string, string>;
737
+ requestBody?: unknown;
738
+ responseHeaders?: Record<string, string>;
739
+ responseBody?: unknown;
740
+ /** Set on `direction: 'error'` events. */
741
+ errorMessage?: string;
742
+ /** Field paths the redactor stripped from this event ('header:Authorization', 'body.apiKey', …). */
743
+ redactedFields: string[];
744
+ }
745
+ interface RawProviderSinkFilter {
746
+ runId?: string;
747
+ spanId?: string;
748
+ direction?: RawProviderDirection;
749
+ attemptIndex?: number;
750
+ }
751
+ interface RawProviderSink {
752
+ record(event: RawProviderEvent): Promise<void>;
753
+ /** Optional listing — implementations that durably persist (file, db) should support this. */
754
+ list?(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
755
+ /** Optional teardown for backed implementations. */
756
+ close?(): Promise<void>;
757
+ }
758
+ type ProviderRedactor = (event: RawProviderEvent) => RawProviderEvent;
759
+
760
+ /**
761
+ * LLM client with graceful degrade.
762
+ *
763
+ * OpenAI-compatible `/v1/chat/completions` client with:
764
+ * - Exponential-backoff retry on 429 + 5xx gateway errors (502/503/504).
765
+ * - Retry on transient network errors (fetch failed, AbortError, ECONNRESET).
766
+ * - Graceful json_schema → json_object degrade on 400 with schema-reject body.
767
+ * - Fenced-JSON stripping (```json ... ```) for models that wrap structured output.
768
+ * - Configurable base URL + api key / bearer, works with LiteLLM proxies, OpenAI
769
+ * directly, cli-bridge subscriptions, and any router that speaks the spec.
770
+ *
771
+ * Usage:
772
+ * const { value, result } = await callLlmJson<MyType>(
773
+ * { model: 'gpt-4o', messages: [...], jsonSchema: { name: 'x', schema: {...} } },
774
+ * { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
775
+ * )
776
+ *
777
+ * This is THE llm-calling seam for agent-eval primitives that need structured
778
+ * output (semantic concept judge, reviewer directives, critic scores). Primitives
779
+ * that need free-form text use `callLlm` and parse output themselves.
780
+ */
781
+
782
+ interface LlmMessage {
783
+ role: 'system' | 'user' | 'assistant';
784
+ /**
785
+ * Either a plain text content string OR a multimodal content array
786
+ * (text + image_url parts) for vision-capable models.
787
+ */
788
+ content: string | Array<{
789
+ type: 'text';
790
+ text: string;
791
+ } | {
792
+ type: 'image_url';
793
+ image_url: {
794
+ url: string;
795
+ detail?: 'auto' | 'low' | 'high';
796
+ };
797
+ }>;
798
+ }
799
+ interface LlmCallRequest {
800
+ model: string;
801
+ messages: LlmMessage[];
802
+ /** Optional JSON-mode response format (response_format: json_object). */
803
+ jsonMode?: boolean;
804
+ /** Optional structured output via JSON Schema. Falls back to json_object on 400. */
805
+ jsonSchema?: {
806
+ name: string;
807
+ schema: Record<string, unknown>;
808
+ };
809
+ temperature?: number;
810
+ maxTokens?: number;
811
+ /** Per-call timeout, default 300s. */
812
+ timeoutMs?: number;
813
+ }
814
+ interface LlmUsage {
815
+ promptTokens: number;
816
+ completionTokens: number;
817
+ totalTokens: number;
818
+ /** False when the provider omitted or malformed prompt/completion usage. */
819
+ captured?: boolean;
820
+ /** Proxies populate this when prompt caching is on. */
821
+ cachedPromptTokens?: number;
822
+ }
823
+ interface LlmCallResult {
824
+ /** The text content of the first choice. Empty string if none. */
825
+ content: string;
826
+ usage: LlmUsage;
827
+ /**
828
+ * Cost in USD. Pulled from proxy's `_response_cost` field when present;
829
+ * `null` when neither the proxy nor the caller can derive it.
830
+ */
831
+ costUsd: number | null;
832
+ /** Model name actually used (echoed from response). */
833
+ model: string;
834
+ /** Wall-clock duration of the HTTP call (last attempt, if retried). */
835
+ durationMs: number;
836
+ /**
837
+ * `finish_reason` echoed from the first choice (`stop`, `length`,
838
+ * `content_filter`, `tool_calls`, ...). `null` when the provider omits it.
839
+ * Exposed so a free-form `callLlm` caller CAN detect a truncated answer
840
+ * (`length`) instead of treating a cut-off completion as complete. Note:
841
+ * `callLlm` does not itself reject on it — acting on this signal is the
842
+ * caller's responsibility (in-repo free-form drivers do not yet enforce it).
843
+ */
844
+ finishReason?: string | null;
845
+ /**
846
+ * True when `content.trim()` is empty. An empty completion is a silent zero
847
+ * for free-form `callLlm` callers; this flag is the signal a caller can
848
+ * inspect to fail loud rather than proceed on an empty string. `callLlm`
849
+ * surfaces it but does not throw on it.
850
+ */
851
+ contentEmpty?: boolean;
852
+ /** Raw response body. */
853
+ raw: Record<string, unknown>;
854
+ }
855
+ type LlmCallMetadata = Pick<LlmCallResult, 'usage' | 'costUsd' | 'model' | 'durationMs'>;
856
+ interface LlmClientOptions {
857
+ /** Base URL (without trailing slash). Must end at the `/v1` prefix. */
858
+ baseUrl?: string;
859
+ /** Bearer token — either `apiKey` or `bearer` populates `Authorization: Bearer ...`. */
860
+ apiKey?: string;
861
+ bearer?: string;
862
+ /** Override for the `Authorization` header (e.g. `X-Auth: ...`). Takes precedence over apiKey/bearer. */
863
+ authHeader?: {
864
+ name: string;
865
+ value: string;
866
+ };
867
+ /** Stable provider idempotency key, reused across retries of this logical call. */
868
+ idempotencyKey?: string;
869
+ /** Default timeout in ms. Per-call can override. */
870
+ defaultTimeoutMs?: number;
871
+ /**
872
+ * Caller-supplied abort signal — e.g. a campaign-wide cancel. Linked to
873
+ * each attempt's per-attempt timeout controller, so aborting it cancels
874
+ * the in-flight fetch. A caller abort is FATAL: it is not retried even
875
+ * though an AbortError otherwise matches the transient patterns.
876
+ */
877
+ signal?: AbortSignal;
878
+ /**
879
+ * Cross-attempt wall-clock budget in ms, measured from the first attempt.
880
+ * Before launching each attempt the loop checks the remaining budget and
881
+ * stops retrying once it is exhausted, rather than waiting the full
882
+ * per-attempt timeout on every retry. Bounds total time independent of
883
+ * total attempts × `timeoutMs`.
884
+ */
885
+ deadlineMs?: number;
886
+ /** Total provider attempts. Legacy option name; default 3 (1 initial + 2 retries). */
887
+ maxRetries?: number;
888
+ /** Fetch implementation — defaults to global `fetch`. Override for custom transport (e.g. tests). */
889
+ fetch?: typeof fetch;
890
+ /**
891
+ * Optional raw HTTP capture sink. When provided, every request, response,
892
+ * and error (across all retry attempts) is recorded to the sink, with auth
893
+ * headers and credential-shaped body fields redacted by default. This is
894
+ * the layer-1 forensics primitive: structured `LlmSpan`s record intent,
895
+ * raw events record what actually crossed the wire.
896
+ */
897
+ rawSink?: RawProviderSink;
898
+ /**
899
+ * Logical provider id attached to raw events. When omitted, derived from
900
+ * `baseUrl` via `providerFromBaseUrl`.
901
+ */
902
+ provider?: string;
903
+ /** Trace context attached to raw events; populated by emitter-aware callers. */
904
+ traceContext?: {
905
+ runId?: string;
906
+ spanId?: string;
907
+ };
908
+ /** Override the redaction strategy for this call. Defaults to `defaultProviderRedactor`. */
909
+ redactor?: ProviderRedactor;
910
+ }
911
+
912
+ /**
913
+ * ChatClient — the single LLM abstraction analysts call.
914
+ *
915
+ * agent-eval already ships an `LlmClient` (OpenAI-compatible, retry,
916
+ * graceful JSON-schema degrade) and judges that talk to `TCloud`. Two
917
+ * mixed patterns force every analyst author to pick a transport, which
918
+ * couples analyst code to runtime concerns (cli-bridge vs router vs
919
+ * sandbox-sdk) it shouldn't know about.
920
+ *
921
+ * `ChatClient` is one interface every analyst takes via `AnalystContext.chat`.
922
+ * The operator decides at the registry boundary which transport binds
923
+ * to it. Analyst code stays transport-agnostic; swapping production
924
+ * (sandbox-sdk) for local dev (cli-bridge) or tests (mock) is a one-
925
+ * line factory call.
926
+ *
927
+ * Designed to coexist: existing `LlmClient` callers and existing
928
+ * `TCloud`-based judges keep working untouched. New analyst code uses
929
+ * `ChatClient`. When old call sites migrate, they pick up budgeting,
930
+ * cancellation, and unified telemetry for free.
931
+ */
932
+
933
+ /**
934
+ * Unified chat interface. Mirrors LlmCallRequest/Result so the OpenAI-
935
+ * compatible mental model stays. Two methods: a one-shot `chat()` and
936
+ * an `streamChat()` for future agentic loops (not yet exposed).
937
+ */
938
+ interface ChatClient {
939
+ /** Display name of the bound transport — included in telemetry. */
940
+ readonly transport: ChatTransport;
941
+ /** Default model when caller omits — operators bind this per environment. */
942
+ readonly defaultModel?: string;
943
+ /** Total provider attempts this transport can make for one chat call. */
944
+ readonly maximumAttempts?: number;
945
+ /** Implementations must enforce `req.maxTokens` when it is present. */
946
+ chat(req: ChatRequest, opts?: ChatCallOpts): Promise<ChatResponse>;
947
+ }
948
+ type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'mock';
949
+ interface ChatRequest extends Omit<LlmCallRequest, 'model'> {
950
+ /** Optional — falls back to ChatClient.defaultModel. */
951
+ model?: string;
952
+ }
953
+ type ChatResponse = LlmCallResult;
954
+ interface ChatCallOpts {
955
+ /** Cancel the in-flight request. */
956
+ signal?: AbortSignal;
957
+ /** Hard USD ceiling for this single call (informational; the underlying transport may not enforce). */
958
+ maxCostUsd?: number;
959
+ /** Correlation tag carried into request headers when the transport allows. */
960
+ correlationId?: string;
961
+ /** Stable provider idempotency key for retries/redrives of one paid call. */
962
+ idempotencyKey?: string;
963
+ }
964
+ type CreateChatClientOpts = RouterTransportOpts | CliBridgeTransportOpts | DirectProviderTransportOpts | SandboxSdkTransportOpts | MockTransportOpts;
965
+ interface BaseTransportOpts {
966
+ defaultModel?: string;
967
+ /** Total provider attempts. Required for opaque transports used in capped runs. */
968
+ maximumAttempts?: number;
969
+ }
970
+ interface RouterTransportOpts extends BaseTransportOpts {
971
+ transport: 'router';
972
+ baseUrl?: string;
973
+ apiKey: string;
974
+ }
975
+ interface CliBridgeTransportOpts extends BaseTransportOpts {
976
+ transport: 'cli-bridge';
977
+ baseUrl?: string;
978
+ bearer?: string;
979
+ }
980
+ interface DirectProviderTransportOpts extends BaseTransportOpts {
981
+ transport: 'direct-provider';
982
+ baseUrl: string;
983
+ apiKey: string;
984
+ }
985
+ /**
986
+ * Sandbox-SDK transport. Provided as a thin pass-through: the caller
987
+ * supplies a callable that mimics LlmClient.chat() against an already-
988
+ * configured Sandbox handle. We don't import the SDK here to keep
989
+ * agent-eval dep-free of @tangle-network/sandbox.
990
+ */
991
+ interface SandboxSdkTransportOpts extends BaseTransportOpts {
992
+ transport: 'sandbox-sdk';
993
+ chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
994
+ }
995
+ /**
996
+ * Mock transport for tests. The handler receives the request and returns
997
+ * whatever the test wants. No retries, no JSON-schema degrade.
998
+ */
999
+ interface MockTransportOpts extends BaseTransportOpts {
1000
+ transport: 'mock';
1001
+ handler: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
1002
+ }
1003
+ /**
1004
+ * Build a ChatClient bound to a specific transport. The returned client
1005
+ * is safe to share across analysts in a single registry run.
1006
+ */
1007
+ declare function createChatClient(opts: CreateChatClientOpts): ChatClient;
1008
+
1009
+ /**
1010
+ * Analyst contract — the missing orchestration layer over agent-eval's
1011
+ * existing analyzers (analyzeTraces, MultiLayerVerifier, RunCritic,
1012
+ * SemanticConceptJudge, JudgeFn, ...).
1013
+ *
1014
+ * Each existing primitive returns its own output shape. The Analyst
1015
+ * contract is the single envelope every primitive lifts into, so a
1016
+ * registry can run N analysts against a run and a single renderer can
1017
+ * compose findings without knowing which analyzer produced them.
1018
+ *
1019
+ * The contract is intentionally domain-agnostic: nothing here knows
1020
+ * about code, voice, RAG, or any particular agent stack. Analysts
1021
+ * declare what INPUT KIND they need (a trace store, an artifact dir,
1022
+ * a RunRecord, a JudgeInput, or `custom`), and the registry routes
1023
+ * the matching input from `AnalystRunInputs`.
1024
+ */
1025
+
1026
+ /**
1027
+ * Unified envelope every analyst emits. Schema-versioned so renderers
1028
+ * and time-series diffs survive future field additions.
1029
+ */
1030
+ interface AnalystFinding {
1031
+ schema_version: '1.0.0';
1032
+ /**
1033
+ * Stable hash over identity-defining fields (analyst_id + canonical
1034
+ * claim + area + optional subject). Two findings from two runs that
1035
+ * "are the same finding" share this id — that's what `diffFindings`
1036
+ * uses to compute appeared/disappeared sets across runs.
1037
+ */
1038
+ finding_id: string;
1039
+ analyst_id: string;
1040
+ produced_at: string;
1041
+ severity: AnalystSeverity;
1042
+ /**
1043
+ * Coarse classification. Renderers group by this. Free-form so
1044
+ * domain-specific analysts can introduce categories without a
1045
+ * schema change ('agent-reasoning', 'verification', 'cost',
1046
+ * 'tool-use', 'safety', 'latency', 'data-quality', ...).
1047
+ */
1048
+ area: string;
1049
+ claim: string;
1050
+ rationale?: string;
1051
+ evidence_refs: EvidenceRef[];
1052
+ recommended_action?: string;
1053
+ validation_plan?: string;
1054
+ /** 0..1 — the analyst's own confidence. Not calibrated across analysts. */
1055
+ confidence: number;
1056
+ /**
1057
+ * Optional subject the finding is about — leaf id, agent id, request
1058
+ * id. Included in finding_id when present so per-subject findings
1059
+ * diff cleanly across runs.
1060
+ */
1061
+ subject?: string;
1062
+ /** FIREWALL provenance (docs/learning-flywheel.md): true iff this finding was
1063
+ * lifted from a JUDGE verdict (an acceptance score), not OBSERVED from the
1064
+ * agent's behavior. A judge-derived finding must NEVER be admitted as a
1065
+ * steering input — that is the held-out judge leaking into the loop. Set at
1066
+ * the lift site (createJudgeAdapter); checked by `assertNoJudgeVerdict`.
1067
+ * Provenance, not evidence presence, is the correct discriminator: an
1068
+ * evidence-less trace-analyst observation legitimately steers, while a judge
1069
+ * verdict that happens to cite an artifact must not. */
1070
+ derived_from_judge?: boolean;
1071
+ /** Analyst-private extras; renderers ignore unless they know the analyst. */
1072
+ metadata?: Record<string, unknown>;
1073
+ }
1074
+ type AnalystSeverity = 'critical' | 'high' | 'medium' | 'low' | 'info';
1075
+ interface EvidenceRef {
1076
+ /**
1077
+ * Where the evidence lives. `span` and `event` refer to OTLP trace
1078
+ * elements; `artifact` to a file inside the run's artifact tree;
1079
+ * `finding` to another AnalystFinding (cross-analyst chaining);
1080
+ * `metric` to a named scalar reading the renderer knows how to read.
1081
+ */
1082
+ kind: 'span' | 'event' | 'artifact' | 'finding' | 'metric';
1083
+ uri: string;
1084
+ excerpt?: string;
1085
+ }
1086
+ /**
1087
+ * The discriminator the registry uses to pass the right input.
1088
+ * `custom` is the escape hatch — analysts that need something else
1089
+ * (e.g. an embedding cache, a partner SDK handle) read it from
1090
+ * `AnalystRunInputs.custom[<analyst id>]`.
1091
+ */
1092
+ type AnalystInputKind = 'trace-store' | 'artifact-dir' | 'run-record' | 'judge-input' | 'custom';
1093
+ interface AnalystCost {
1094
+ /** `deterministic` analysts MUST NOT call the LLM. */
1095
+ kind: 'deterministic' | 'llm';
1096
+ /** Optional declared upper bound; the registry can enforce a budget. */
1097
+ est_usd_per_run?: number;
1098
+ /** Models the analyst expects to use (informational). */
1099
+ models?: string[];
1100
+ }
1101
+ interface AnalystRequirements {
1102
+ /** Min number of shots / samples the analyst needs to produce signal. */
1103
+ min_shots?: number;
1104
+ /** Capabilities the runtime must supply (e.g. ['network', 'gpu']). */
1105
+ capabilities?: string[];
1106
+ }
1107
+ /**
1108
+ * What's passed to every analyst call. The registry resolves which
1109
+ * field the analyst's `inputKind` selects and asserts it's present.
1110
+ */
1111
+ interface AnalystRunInputs {
1112
+ traceStore?: TraceAnalysisStore;
1113
+ artifactDir?: string;
1114
+ runRecord?: RunRecord;
1115
+ judgeInput?: JudgeInput;
1116
+ /** Keyed by analyst id; populated by callers that registered custom analysts. */
1117
+ custom?: Record<string, unknown>;
1118
+ }
1119
+ interface AnalystContext {
1120
+ runId: string;
1121
+ /** Stable correlation id so logs from a single registry.run() share a tag. */
1122
+ correlationId: string;
1123
+ /** Wall-clock deadline (epoch ms). Analysts SHOULD honor for graceful cancel. */
1124
+ deadlineMs?: number;
1125
+ /** Per-analyst USD budget. Analysts MAY check before issuing LLM calls. */
1126
+ budgetUsd?: number;
1127
+ /**
1128
+ * Shared chat client. Analysts that call an LLM go through this so
1129
+ * the operator picks transport (sandbox-sdk | router | cli-bridge |
1130
+ * direct-provider | mock) at the registry boundary without touching
1131
+ * analyst code.
1132
+ */
1133
+ chat?: ChatClient;
1134
+ /**
1135
+ * Findings from a prior run the operator wants the analyst to see as
1136
+ * retrieval context. Kinds that take advantage of cross-run memory
1137
+ * (failure-mode "I saw this cluster last run", knowledge-gap "the wiki
1138
+ * page I asked for is still missing") render these into the actor's
1139
+ * working set. Filtering is the operator's job: pass the slice that
1140
+ * matches the analyst's id, or pass everything and let the kind
1141
+ * filter. Empty / absent means no cross-run context.
1142
+ */
1143
+ priorFindings?: ReadonlyArray<AnalystFinding>;
1144
+ /** Free-form runtime tags (env, host, op). Findings can echo these into metadata. */
1145
+ tags?: Record<string, string>;
1146
+ /** Logger callback — analysts SHOULD prefer this over console.* for testability. */
1147
+ log?: (msg: string, fields?: Record<string, unknown>) => void;
1148
+ /** Optional abort signal. Analysts SHOULD pass it through to LLM calls. */
1149
+ signal?: AbortSignal;
1150
+ }
1151
+ /**
1152
+ * The minimal contract. Concrete analysts can refine `TInput` so
1153
+ * implementations stay type-safe (e.g. a trace analyst's `TInput` is
1154
+ * `TraceAnalysisStore`); the registry passes the right field from
1155
+ * `AnalystRunInputs` based on `inputKind`.
1156
+ */
1157
+ interface Analyst<TInput = unknown> {
1158
+ /** Stable identifier — appears in finding_id, telemetry, and registry exclusion lists. */
1159
+ readonly id: string;
1160
+ /** Human-readable. One sentence. */
1161
+ readonly description: string;
1162
+ readonly inputKind: AnalystInputKind;
1163
+ readonly cost: AnalystCost;
1164
+ readonly requires?: AnalystRequirements;
1165
+ /** Bump on breaking changes to claim wording or area so old finding_ids don't collide. */
1166
+ readonly version: string;
1167
+ analyze(input: TInput, ctx: AnalystContext): Promise<AnalystFinding[]>;
1168
+ }
1169
+ interface AnalystRunSummary {
1170
+ analyst_id: string;
1171
+ status: 'ok' | 'skipped' | 'failed';
1172
+ /** Why skipped — missing input, budget exceeded, capability unmet. */
1173
+ reason?: string;
1174
+ findings_count: number;
1175
+ latency_ms: number;
1176
+ cost_usd: number;
1177
+ /** When `status='failed'`: the error class + message, never the full stack. */
1178
+ error?: {
1179
+ class: string;
1180
+ message: string;
1181
+ };
1182
+ }
1183
+ interface AnalystRunResult {
1184
+ run_id: string;
1185
+ correlation_id: string;
1186
+ started_at: string;
1187
+ ended_at: string;
1188
+ findings: AnalystFinding[];
1189
+ per_analyst: AnalystRunSummary[];
1190
+ /** Total LLM cost in USD across all analysts in this registry.run(). */
1191
+ total_cost_usd: number;
1192
+ }
1193
+ /**
1194
+ * Events emitted by `AnalystRegistry.runStream(...)` in real time as
1195
+ * the registry executes. UIs subscribe via `for await (const ev of
1196
+ * registry.runStream(...))`; `registry.run(...)` is a thin collector
1197
+ * over the same stream, so the two surfaces share their invariants.
1198
+ *
1199
+ * Per-finding events are intentionally omitted — analyzers are batch
1200
+ * operations (an Ax actor returns the full `findings:json[]` at the
1201
+ * end of the responder), so streaming inside one analyst would only
1202
+ * emit partial JSON consumers can't render. The kind-completion event
1203
+ * is the right granularity; subscribers wanting per-finding rendering
1204
+ * iterate `event.findings` themselves.
1205
+ */
1206
+ type AnalystRunEvent = {
1207
+ type: 'run-started';
1208
+ run_id: string;
1209
+ correlation_id: string;
1210
+ started_at: string;
1211
+ /** The ordered list of analyst ids the registry will run. */
1212
+ analyst_ids: ReadonlyArray<string>;
1213
+ } | {
1214
+ type: 'analyst-skipped';
1215
+ summary: AnalystRunSummary;
1216
+ } | {
1217
+ type: 'analyst-started';
1218
+ analyst_id: string;
1219
+ started_at: string;
1220
+ } | {
1221
+ type: 'analyst-completed';
1222
+ /** `summary.status` is `'ok'` for clean completion or `'failed'` for thrown analysts. */
1223
+ summary: AnalystRunSummary;
1224
+ findings: ReadonlyArray<AnalystFinding>;
1225
+ } | {
1226
+ type: 'run-completed';
1227
+ result: AnalystRunResult;
1228
+ };
1229
+
1230
+ type PolicyEditSchemaVersion = 'policy-edit/v1';
1231
+ declare const POLICY_EDIT_AXES: readonly ["carrier", "representation", "budget", "sampling", "output_contract", "tool_contract", "routing", "memory", "agent_profile", "deployment_target"];
1232
+ type PolicyEditAxis = (typeof POLICY_EDIT_AXES)[number];
1233
+ declare const POLICY_EDIT_TARGET_SURFACES: readonly ["prompt", "tool-contract", "runtime-config", "memory", "agent-profile", "code", "deployment"];
1234
+ type PolicyEditTargetSurface = (typeof POLICY_EDIT_TARGET_SURFACES)[number];
1235
+ type PolicyEditRisk = 'low' | 'medium' | 'high' | 'unknown';
1236
+ type PolicyEditGainDirection = 'increase' | 'decrease';
1237
+ type PolicyEditGainUnit = 'absolute' | 'relative' | 'percent' | 'score';
1238
+ interface PolicyEditTarget {
1239
+ surface: PolicyEditTargetSurface;
1240
+ /** Stable path inside the target surface, for example `system-prompt:tools`
1241
+ * or `budget.maxTurns`. */
1242
+ path?: string;
1243
+ /** Optional canonical deployment identity. Store the existing cell, not a
1244
+ * local profile shape. */
1245
+ agentProfileCell?: AgentProfileCell;
1246
+ /** Human label when the path is not enough for a readable audit trail. */
1247
+ label?: string;
1248
+ }
1249
+ type PolicyEditChange = {
1250
+ kind: 'text';
1251
+ mode: 'append' | 'prepend' | 'replace';
1252
+ value: string;
1253
+ /** Required when `mode === 'replace'`; exact match only. */
1254
+ find?: string;
1255
+ } | {
1256
+ kind: 'json';
1257
+ mode: 'set' | 'merge' | 'remove';
1258
+ path: string;
1259
+ value?: AgentProfileJson;
1260
+ };
1261
+ interface PolicyEditExpectedGain {
1262
+ /** Metric this edit is expected to move, e.g. `holdout.composite`. */
1263
+ metric: string;
1264
+ direction: PolicyEditGainDirection;
1265
+ /** Positive magnitude in the metric's native units. */
1266
+ amount: number;
1267
+ unit?: PolicyEditGainUnit;
1268
+ rationale?: string;
1269
+ }
1270
+ interface PolicyEditSource {
1271
+ findingIds: string[];
1272
+ analystIds: string[];
1273
+ evidenceRefs: EvidenceRef[];
1274
+ /** Mirrors `AnalystFinding.derived_from_judge`; admission rejects it. */
1275
+ derivedFromJudge?: boolean;
1276
+ }
1277
+ interface PolicyEdit {
1278
+ schemaVersion: PolicyEditSchemaVersion;
1279
+ editId: string;
1280
+ axis: PolicyEditAxis;
1281
+ target: PolicyEditTarget;
1282
+ change: PolicyEditChange;
1283
+ claim: string;
1284
+ expectedGain: PolicyEditExpectedGain;
1285
+ confidence: number;
1286
+ risk: PolicyEditRisk;
1287
+ source: PolicyEditSource;
1288
+ rationale?: string;
1289
+ validationPlan?: string;
1290
+ metadata?: Record<string, unknown>;
1291
+ }
1292
+ declare const POLICY_EDIT_CANDIDATE_RECORD_SCHEMA: "tangle.policy-edit-candidate.v1";
1293
+ /** JSON-safe attribution carried with a measured candidate and its scores. */
1294
+ interface PolicyEditCandidateRecord {
1295
+ schema: typeof POLICY_EDIT_CANDIDATE_RECORD_SCHEMA;
1296
+ policyEdit: PolicyEdit;
1297
+ }
1298
+
1299
+ /**
1300
+ * Pass A substrate types — `runCampaign` is the one primitive every
1301
+ * eval flow composes from. Three contracts in this file:
1302
+ *
1303
+ * - `Scenario` input set
1304
+ * - `DispatchFn` how to run one scenario → artifact
1305
+ * - `CampaignResult` defined output schema (the contract downstream tools depend on)
1306
+ *
1307
+ * Three more lifted from earlier substrate work (re-exported):
1308
+ *
1309
+ * - `JudgeConfig` pluggable dimensional scorer (0.38)
1310
+ * - `Mutator` optimization-loop surface mutator
1311
+ * - `Gate` promotion gate (`HeldOutGate` and friends adapt to this)
1312
+ *
1313
+ * No new architecture vs 0.38 — Pass A formalizes the shapes so consumers
1314
+ * can build dashboards / CI gates / regression diffs against a stable schema.
1315
+ */
1316
+
1317
+ /** Stable identifier + kind tag for any scenario. Consumers
1318
+ * extend with their per-domain payload (persona, task, requirement, ...). */
1319
+ interface Scenario {
1320
+ id: string;
1321
+ kind: string;
1322
+ tags?: string[];
1323
+ }
1324
+ /** Context handed to every dispatch invocation. Scoped — every
1325
+ * trace/span carries the cellId, every artifact write lands under the cell's
1326
+ * artifact root, the cost meter accumulates per cell. */
1327
+ interface DispatchContext {
1328
+ cellId: string;
1329
+ rep: number;
1330
+ generation?: number;
1331
+ seed: number;
1332
+ signal: AbortSignal;
1333
+ trace: CampaignTraceWriter;
1334
+ artifacts: CampaignArtifactWriter;
1335
+ cost: CampaignCostMeter;
1336
+ /** Populated when this run is part of a multi-cycle improvement loop. */
1337
+ cycleId?: string;
1338
+ /** Populated when the substrate resumed from a prior cache hit. */
1339
+ resumedFrom?: string;
1340
+ /**
1341
+ * Opaque placement key supplied by `RunCampaignOptions.cellPlacement`.
1342
+ * The substrate forwards it through unchanged; placement-aware Dispatch
1343
+ * implementations (e.g. `httpDispatch` from `/adapters/http`) read it to
1344
+ * route the cell to the right worker / region / sandbox. `undefined`
1345
+ * when no placement strategy is configured.
1346
+ */
1347
+ placement?: string;
1348
+ }
1349
+ /** One function: scenario + ctx → artifact. Dispatcher chooses
1350
+ * whether to call `runMultishot`, `runLoop`, raw `streamPrompt`, anything. */
1351
+ type DispatchFn<TScenario extends Scenario, TArtifact> = (scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
1352
+ /** One session within a multi-session journey. Dispatch is
1353
+ * invoked once per session in order; state from prior session's artifact
1354
+ * is exposed via `ctx.priorSessionArtifact`. */
1355
+ interface SessionScript<TScenario, TArtifact> {
1356
+ id: string;
1357
+ intent: string;
1358
+ maxTurns?: number;
1359
+ /** When true, knowledge accumulated this session persists to next. */
1360
+ affectsKnowledge?: boolean;
1361
+ /** Optional per-session persona evolution — called after the session
1362
+ * resolves. Returns the persona shape used by the NEXT session. */
1363
+ evolveAfterSession?: (artifact: TArtifact, sessionIndex: number, scenario: TScenario) => TScenario;
1364
+ }
1365
+ interface JudgeDimension {
1366
+ /** JSON field name + score key. */
1367
+ key: string;
1368
+ /** Description shown in the judge's user prompt. */
1369
+ description: string;
1370
+ }
1371
+ /** Pluggable dimensional scorer. `score` is the contract:
1372
+ * given an artifact + scenario, return a `JudgeScore`. This is deliberately a
1373
+ * function, not a fixed LLM-prompt shape — real consumers judge with
1374
+ * ensembles, deterministic checks, or a single LLM call, and the substrate
1375
+ * must not constrain that. The `llmJudge()` helper builds a `score` that does
1376
+ * one LLM call for the common case. `appliesTo` lets a judge run only on
1377
+ * scenarios that match (e.g. a legal-citation judge only on legal scenarios). */
1378
+ interface JudgeConfig<TArtifact, TScenario extends Scenario = Scenario> {
1379
+ name: string;
1380
+ dimensions: JudgeDimension[];
1381
+ /** Stable scoring revision used by campaign resume and verdict caches.
1382
+ * Built-in judges derive this from their prompt, model, and rubric. Custom
1383
+ * judges should set it when closure state can change without changing code. */
1384
+ judgeVersion?: string;
1385
+ /** Score one artifact. Throw on failure — a thrown judge is recorded as a
1386
+ * failed cell, never silently folded into a zero. */
1387
+ score(input: {
1388
+ artifact: TArtifact;
1389
+ scenario: TScenario;
1390
+ signal: AbortSignal;
1391
+ /** Shared run spend account and receipt attribution phase. */
1392
+ costLedger?: CostLedger;
1393
+ costPhase?: string;
1394
+ costTags?: Record<string, string>;
1395
+ }): JudgeScore | Promise<JudgeScore>;
1396
+ appliesTo?: (scenario: TScenario) => boolean;
1397
+ }
1398
+ /** The canonical judge verdict shape — one declaration, shared by campaign
1399
+ * judges and the multishot judge runner (which re-exports this type).
1400
+ *
1401
+ * Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the legacy
1402
+ * multishot runner emits 0-10. Cross-scale comparison must go through
1403
+ * `detectScale` (src/campaign/gates/statistical-heldout.ts, used by
1404
+ * promotion-policy) — never renormalize a producer's values in place, as
1405
+ * downstream thresholds (`composite >= 5` in multishot/matrix.ts, live-soak
1406
+ * `>= 7` gates) key on the producer's native scale. */
1407
+ interface JudgeScore {
1408
+ dimensions: Record<string, number>;
1409
+ composite: number;
1410
+ notes: string;
1411
+ /** Provider metadata for display and diagnostics; accounting uses CostLedger receipts. */
1412
+ llmCall?: LlmCallMetadata;
1413
+ /** Set when the judge itself failed (call error, unparseable output).
1414
+ * `composite`/`dimensions` carry no signal — aggregators MUST exclude
1415
+ * failed scores from means instead of folding them into zeros. */
1416
+ failed?: true;
1417
+ /** Ensemble extras (populated by `ensembleJudge`): max per-dimension
1418
+ * spread across surviving judges — the inter-rater signal. */
1419
+ maxDisagreement?: number;
1420
+ /** Ensemble extras: judge identities whose verdict failed. */
1421
+ failedJudges?: string[];
1422
+ /** Ensemble extras: each surviving judge's per-dimension scores. */
1423
+ perJudge?: Record<string, Record<string, number>>;
1424
+ }
1425
+ /** A tier-4 code surface — a finalized candidate change to the agent's
1426
+ * IMPLEMENTATION, not its prompt. Produced by autoresearch (reads codebase +
1427
+ * trace findings → opens a worktree). `worktreeRef` locates the candidate;
1428
+ * the exact commits, tree, and binary-patch digest identify it. See the
1429
+ * improvement-tier table in `docs/design/loop-taxonomy.md`. */
1430
+ interface CodeSurface {
1431
+ readonly kind: 'code';
1432
+ /** Worktree path or git ref holding the candidate code change. This is a
1433
+ * mutable locator and is deliberately excluded from content hashes. */
1434
+ readonly worktreeRef: string;
1435
+ /** Human-readable ref the worktree was forked from. Not identity-bearing. */
1436
+ readonly baseRef: string;
1437
+ /** Exact commit the candidate was forked from. */
1438
+ readonly baseCommit: string;
1439
+ /** Exact tree object for `baseCommit`. */
1440
+ readonly baseTree: string;
1441
+ /** Exact finalized candidate commit. */
1442
+ readonly candidateCommit: string;
1443
+ /** Exact tree object for `candidateCommit`. */
1444
+ readonly candidateTree: string;
1445
+ /** Identity of the exact patch artifact. The deployable candidate bundle
1446
+ * carries the same descriptor plus its base64-encoded content. */
1447
+ readonly patch: {
1448
+ readonly format: 'git-diff-binary';
1449
+ readonly sha256: `sha256:${string}`;
1450
+ readonly byteLength: number;
1451
+ };
1452
+ /** Human summary of what changed — rendered into the auto-PR body. */
1453
+ readonly summary?: string;
1454
+ }
1455
+ /** The mutable surface a proposer changes. Tiers (see
1456
+ * `docs/design/loop-taxonomy.md`):
1457
+ * - `string` — tiers 1-2: system-prompt addendum / serialized tool
1458
+ * config. Cheap, reversible, text-diffable.
1459
+ * - `CodeSurface` — tier 4: an implementation change behind a worktree ref.
1460
+ * Tier 3 (knowledge) is owned by agent-knowledge and rides its own adapter,
1461
+ * not this type. */
1462
+ type MutableSurface = string | CodeSurface;
1463
+ /** A proposer output carrying the surface AND the WHY behind
1464
+ * it. Reflective proposers (`gepaProposer`) parse a `{label, rationale, payload}`
1465
+ * from the model; without this wrapper the loop keeps only `payload` and the
1466
+ * rationale that motivated the change is lost — the candidate becomes
1467
+ * unattributable. `propose()` may return either bare `MutableSurface`s (cheap
1468
+ * blind mutators) or these (reflective proposers); the loop normalizes both. */
1469
+ interface ProposedCandidate {
1470
+ surface: MutableSurface;
1471
+ /** Short human label for the change (≤ 40 chars typical). */
1472
+ label: string;
1473
+ /** Why this change was proposed — which failure it targets, which
1474
+ * primitive it used. Survives to `GenerationCandidate.rationale` and the
1475
+ * emitted provenance record. */
1476
+ rationale: string;
1477
+ /** Structured, JSON-safe cause for this exact candidate when the proposer
1478
+ * can provide one. Policy edits retain the full validated edit here. */
1479
+ candidateRecord?: PolicyEditCandidateRecord;
1480
+ }
1481
+ /** A non-dominated parent on the GEPA Pareto frontier — a
1482
+ * surface that, across the per-scenario objective vectors, no other tried
1483
+ * surface beats on every scenario. A candidate worse on the mean composite
1484
+ * but uniquely best on one hard scenario is non-dominated and survives here;
1485
+ * the composite-best ranking would discard the lesson it carries. The loop
1486
+ * computes the frontier across ALL generations and hands it to the proposer so
1487
+ * a reflective proposer can combine complementary lessons (GEPA, Agrawal et
1488
+ * al., arXiv:2507.19457). See `pareto.ts` (`paretoFrontier`). */
1489
+ interface ParetoParent {
1490
+ surface: MutableSurface;
1491
+ surfaceHash: string;
1492
+ /** The objective vector: per-scenario composite (higher is better). The
1493
+ * axes the frontier is computed over. */
1494
+ objectives: Record<string, number>;
1495
+ /** Mean composite across the objective scenarios — the scalar summary used
1496
+ * for ordering + display, NOT for dominance. */
1497
+ composite: number;
1498
+ /** Generation that produced this surface (`-1` for the baseline). */
1499
+ generation: number;
1500
+ label?: string;
1501
+ rationale?: string;
1502
+ }
1503
+ /** Exact measured state for the surface an optimizer is learning from.
1504
+ * Unlike a model-authored expected gain, every value here comes from a
1505
+ * completed campaign over the designed denominator. */
1506
+ interface ScoredSurfaceOutcome {
1507
+ /** Optimization/search evidence only. Held-out results must never flow back
1508
+ * into a proposer through this type. */
1509
+ split: 'search';
1510
+ /** Generation that actually measured this surface (`-1` for the baseline). */
1511
+ generation: number;
1512
+ surfaceHash: string;
1513
+ composite: number;
1514
+ dimensions: Record<string, number>;
1515
+ scenarios: Array<{
1516
+ scenarioId: string;
1517
+ composite: number;
1518
+ notes?: string;
1519
+ }>;
1520
+ coverage: {
1521
+ expectedCells: number;
1522
+ scorableCells: number;
1523
+ };
1524
+ }
1525
+ /** Stateless surface mutation — given findings + current
1526
+ * surface, return N candidate surfaces. Pure transform, no generation
1527
+ * awareness. Reflective-mutation and `AxGEPA` mutators conform. Wrapped by
1528
+ * `evolutionaryProposer` to become a `SurfaceProposer`. */
1529
+ interface Mutator<TFindings = unknown> {
1530
+ kind: string;
1531
+ mutate(args: {
1532
+ findings: TFindings[];
1533
+ currentSurface: MutableSurface;
1534
+ populationSize: number;
1535
+ signal: AbortSignal;
1536
+ }): Promise<Array<MutableSurface | ProposedCandidate>>;
1537
+ }
1538
+ /** Everything a proposer may read to plan the next
1539
+ * batch of candidates. The first six fields are always present; the rest are
1540
+ * optional context the loop supplies when available, so cheap proposers
1541
+ * (`evolutionaryProposer`) can ignore them while a code-tier agentic generator
1542
+ * consumes the report + dataset to drive a coding harness.
1543
+ * See `docs/campaign-proposers.md`. */
1544
+ interface ProposeContext<TFindings = unknown> {
1545
+ currentSurface: MutableSurface;
1546
+ history: GenerationRecord[];
1547
+ findings: TFindings[];
1548
+ /** BREADTH: how many candidate surfaces to return this generation. */
1549
+ populationSize: number;
1550
+ generation: number;
1551
+ signal: AbortSignal;
1552
+ /** Measured baseline for this optimization run. `runOptimization` always
1553
+ * supplies it; optional for standalone proposer callers. */
1554
+ baselineOutcome?: ScoredSurfaceOutcome;
1555
+ /** Measured result for `currentSurface`, the complete global incumbent every
1556
+ * new candidate mutates. `runOptimization` always supplies it. */
1557
+ incumbentOutcome?: ScoredSurfaceOutcome;
1558
+ /** Optional analysis report produced before proposal. Opaque to the substrate:
1559
+ * the proposer that consumes it owns the shape. */
1560
+ report?: unknown;
1561
+ /** Handle to all captured data — the proposer samples traces / artifacts /
1562
+ * rewards here to ground its proposals. */
1563
+ dataset?: LabeledScenarioStore;
1564
+ /** DEPTH: max iterations the agentic generator may take per candidate.
1565
+ * 1 = single-shot; >1 = it may iterate on its own change before handing it
1566
+ * back to be measured. */
1567
+ maxImprovementShots?: number;
1568
+ /** GEPA Pareto frontier across ALL generations so far — the non-dominated
1569
+ * surfaces by per-scenario objective vector. Empty/absent on generation 0
1570
+ * (only the baseline is scored). A reflective proposer combines the
1571
+ * complementary lessons of these parents (each excels on different
1572
+ * scenarios) into a merged candidate. Proposers doing pure single-parent
1573
+ * reflection may ignore it. See {@link ParetoParent}. */
1574
+ paretoParents?: ParetoParent[];
1575
+ /** Shared run spend account and receipt attribution phase. */
1576
+ costLedger?: CostLedger;
1577
+ costPhase?: string;
1578
+ /** FIREWALL (non-negotiable): the held-out judge is write-only — its verdicts
1579
+ * score the chosen output and gate promotion, and are NEVER an input to
1580
+ * proposal/steering (else the optimizer games the acceptance axis = an
1581
+ * oracle). This `never`-typed field makes that a compile-time tripwire: a
1582
+ * proposer that tries to thread judge verdicts into the proposal will not type.
1583
+ * Steering may consume TRACE-OBSERVABLE signals (what the agent did) via
1584
+ * `findings`/`report`; it may NOT consume the judge's held-out verdict. */
1585
+ judgeScores?: never;
1586
+ }
1587
+ /** A surface-improvement strategy. Given the current best
1588
+ * surface, the history of what's been tried + scored, and any external
1589
+ * findings, propose the next batch of candidate surfaces to measure.
1590
+ * Optionally decide to stop early.
1591
+ *
1592
+ * The evolutionary mutator (`evolutionaryProposer`, here) and agent-runtime's
1593
+ * reflective / agentic generators both conform. They are proposers for the
1594
+ * SAME loop, not separate loops. The loop body (`runOptimization`) and the
1595
+ * gated promotion shell (`runImprovementLoop`) are proposer-agnostic.
1596
+ *
1597
+ * This is THE optimization proposer — every optimizer is a factory
1598
+ * `xProposer(opts): SurfaceProposer` (`evolutionaryProposer`, `aceProposer`,
1599
+ * `gepaProposer`, `skillOptProposer`, `traceAnalystProposer`, `haloProposer`,
1600
+ * `memoryCurationProposer`, `fapoProposer`), all exported from `/campaign` and
1601
+ * drivable by `selfImprove({ proposer })`. Not to be confused with the
1602
+ * behavior-fuzzing `MutationProposer` (`fuzz/types`), a scenario generator for
1603
+ * a different loop.
1604
+ */
1605
+ interface SurfaceProposer<TFindings = unknown> {
1606
+ kind: string;
1607
+ /** Plan: propose N candidate surfaces for the next generation. A proposer
1608
+ * may return bare `MutableSurface`s or `ProposedCandidate`s that carry the
1609
+ * `{label, rationale}` motivating the change — the loop threads the
1610
+ * rationale into `GenerationCandidate` and the emitted provenance. */
1611
+ propose(ctx: ProposeContext<TFindings>): Promise<Array<MutableSurface | ProposedCandidate>>;
1612
+ /** Decide: stop early when the proposer judges the search converged or
1613
+ * exhausted. Default (omitted) runs all `maxGenerations`. */
1614
+ decide?(args: {
1615
+ history: GenerationRecord[];
1616
+ }): {
1617
+ stop: boolean;
1618
+ reason?: string;
1619
+ };
1620
+ }
1621
+ /** Optional vocabulary alias. The loop is the optimizer; this object is the
1622
+ * proposer inside that loop. */
1623
+ type OptimizationProposer<TFindings = unknown> = SurfaceProposer<TFindings>;
1624
+ interface OptimizerConfigBase {
1625
+ populationSize: number;
1626
+ maxGenerations: number;
1627
+ surfaceExtractor: (profile: unknown) => MutableSurface;
1628
+ }
1629
+ interface OptimizerConfig extends OptimizerConfigBase {
1630
+ proposer: SurfaceProposer;
1631
+ }
1632
+ /** Five-valued verdict taxonomy (MOSS-paper alignment). */
1633
+ type GateDecision = 'ship' | 'hold' | 'need_more_work' | 'model_ceiling' | 'arch_ceiling';
1634
+ interface GateContext<TArtifact, TScenario extends Scenario> {
1635
+ candidateArtifacts: Map<string, TArtifact>;
1636
+ baselineArtifacts?: Map<string, TArtifact>;
1637
+ /** Candidate (winner) judge scores, keyed by cellId. */
1638
+ judgeScores: Map<string, Record<string, JudgeScore>>;
1639
+ /** Baseline judge scores, keyed by cellId. SEPARATE from `judgeScores` —
1640
+ * baseline + candidate share cellIds (same scenarios), so a single map
1641
+ * cannot represent both. A gate computing a holdout delta MUST read
1642
+ * candidate from `judgeScores` and baseline from here. */
1643
+ baselineJudgeScores?: Map<string, Record<string, JudgeScore>>;
1644
+ /** Neutralized-arm judge scores, keyed by cellId — the winner surface with its
1645
+ * content footprint-matched-blanked (via a `neutralize` fn). Same scenarios as
1646
+ * `judgeScores`. Present ONLY when `runImprovementLoop` was given a `neutralize`
1647
+ * function. A placebo gate (`neutralizationGate`) compares this arm's lift
1648
+ * against the candidate's to reject decorative wins (lift from footprint, not
1649
+ * content). Undefined otherwise. */
1650
+ neutralizedJudgeScores?: Map<string, Record<string, JudgeScore>>;
1651
+ /** Neutralized-arm artifacts, keyed by cellId. Present alongside
1652
+ * `neutralizedJudgeScores`. */
1653
+ neutralizedArtifacts?: Map<string, TArtifact>;
1654
+ scenarios: TScenario[];
1655
+ cost: {
1656
+ candidate: number;
1657
+ baseline: number;
1658
+ };
1659
+ /** Shared run spend account and receipt attribution phase. */
1660
+ costLedger?: CostLedger;
1661
+ costPhase?: string;
1662
+ signal: AbortSignal;
1663
+ }
1664
+ interface GateResult {
1665
+ decision: GateDecision;
1666
+ reasons: string[];
1667
+ contributingGates: Array<{
1668
+ name: string;
1669
+ passed: boolean;
1670
+ detail: unknown;
1671
+ }>;
1672
+ delta?: number;
1673
+ }
1674
+ /** Composable promotion gate. */
1675
+ interface Gate<TArtifact = unknown, TScenario extends Scenario = Scenario> {
1676
+ name: string;
1677
+ decide(ctx: GateContext<TArtifact, TScenario>): Promise<GateResult>;
1678
+ }
1679
+ /** Scoped trace writer handed to each dispatch — every span
1680
+ * auto-tagged with the cellId so traces filter cleanly. */
1681
+ interface CampaignTraceWriter {
1682
+ span(name: string, attributes?: Record<string, unknown>): TraceSpan;
1683
+ flush(): Promise<void>;
1684
+ }
1685
+ interface TraceSpan {
1686
+ end(attributes?: Record<string, unknown>): void;
1687
+ setAttribute(key: string, value: unknown): void;
1688
+ }
1689
+ /** Scoped artifact writer — `write(path, content)` lands under
1690
+ * `<runDir>/<cellId>/<path>`. */
1691
+ interface CampaignArtifactWriter {
1692
+ write(path: string, content: string | Uint8Array): Promise<string>;
1693
+ writeJson(path: string, value: unknown): Promise<string>;
1694
+ }
1695
+ /** Token usage accumulated for a cell. Aliased to the canonical `RunTokenUsage`
1696
+ * (run-record.ts, same package) so a cell maps onto a `RunRecord` for the
1697
+ * backend-integrity guard with ONE source of truth — a field added to
1698
+ * `RunTokenUsage` is a compile error here, not a silent drift. */
1699
+ type CampaignTokenUsage = RunTokenUsage;
1700
+ /** Cell-scoped paid-call entry point. The dispatch places every paid operation
1701
+ * inside `runPaidCall`; the returned provider result supplies one receipt with
1702
+ * cost, tokens, and resolved model. Calls made outside this method are not
1703
+ * admitted or captured. */
1704
+ interface CampaignCostMeter {
1705
+ /** The only paid-call path. Returns a typed result; callers must inspect it. */
1706
+ runPaidCall<T>(input: Omit<RunPaidCallInput<T>, 'channel' | 'phase' | 'tags'> & {
1707
+ channel?: CostChannel;
1708
+ }): Promise<PaidCallResult<T>>;
1709
+ }
1710
+ /** Source tag — required on every store write. Used by the
1711
+ * default training-source filter (production-trace samples NOT used as
1712
+ * training scenarios unless explicitly opted in). */
1713
+ type LabeledScenarioSource = 'production-trace' | 'eval-run' | 'manual' | 'red-team' | 'synthetic';
1714
+ type RedactionStatus = 'raw' | 'redacted-pii' | 'redacted-secrets' | 'fully-redacted';
1715
+ /** How much a label can be trusted to evaluate against — the gold-admission
1716
+ * gate. Strictly ordered: a record qualifies for a `minTrust` filter when its
1717
+ * trust rank is >= the requested rank.
1718
+ *
1719
+ * - `unverified` — label is a heuristic (e.g. raw outcome success/fail).
1720
+ * Fine as corpus; MUST NOT enter a gold set that lift
1721
+ * numbers are computed against.
1722
+ * - `verified-signal` — an external signal confirmed the outcome (PR merged,
1723
+ * tests green, user did not retry, downstream check).
1724
+ * - `human-rated` — a human explicitly rated or corrected the artifact.
1725
+ *
1726
+ * Absent on a write ⇒ treated as `unverified` (fail-closed: a writer must
1727
+ * explicitly assert trust to make a record gold-eligible — it never happens
1728
+ * by accident). */
1729
+ type LabelTrust = 'unverified' | 'verified-signal' | 'human-rated';
1730
+ /** Required-provenance write. The store rejects writes that
1731
+ * lack provenance — a default-on flywheel without provenance is the
1732
+ * data-poisoning vector flagged in the alignment review. */
1733
+ interface LabeledScenarioWrite<TScenario extends Scenario = Scenario, TArtifact = unknown> {
1734
+ scenario: TScenario;
1735
+ artifact: TArtifact;
1736
+ judgeScores: Record<string, JudgeScore>;
1737
+ source: LabeledScenarioSource;
1738
+ sourceVersionHash: string;
1739
+ capturedAt: string;
1740
+ redactionStatus: RedactionStatus;
1741
+ /** Gold-admission trust tier. Absent ⇒ `unverified` (fail-closed): the
1742
+ * record is corpus, never gold. A writer must explicitly assert
1743
+ * `verified-signal` or `human-rated` to make it eligible for a gold
1744
+ * sample. See {@link LabelTrust}. */
1745
+ labelTrust?: LabelTrust;
1746
+ /** Optional per-source rate-limit bucket key (e.g., the tenant id). */
1747
+ rateLimitBucket?: string;
1748
+ }
1749
+ interface LabeledScenarioRecord<TScenario extends Scenario = Scenario, TArtifact = unknown> extends LabeledScenarioWrite<TScenario, TArtifact> {
1750
+ /** Stable hash of (scenario.id, source, capturedAt, sourceVersionHash). */
1751
+ recordHash: string;
1752
+ /** Substrate-assigned split — train if captured before the campaign's
1753
+ * `temporalCutoff`, test if after. Explicit override allowed via filter. */
1754
+ split: 'train' | 'test';
1755
+ }
1756
+ interface LabeledScenarioSampleArgs {
1757
+ count: number;
1758
+ /** REQUIRED — substrate refuses to sample without an explicit split. */
1759
+ split: 'train' | 'test';
1760
+ /** REQUIRED — only records captured before this timestamp are returned.
1761
+ * Enforces temporal split discipline (test scenarios captured AFTER train
1762
+ * cannot enter the training pool). */
1763
+ capturedBefore: string;
1764
+ filter?: {
1765
+ kind?: string;
1766
+ source?: LabeledScenarioSource | LabeledScenarioSource[];
1767
+ minComposite?: number;
1768
+ maxComposite?: number;
1769
+ /** Gold gate: only records whose trust rank is >= this tier are
1770
+ * returned. `sample({ split: 'test', minTrust: 'verified-signal' })` is
1771
+ * the canonical "give me the gold set" call. Absent ⇒ no trust gate
1772
+ * (corpus-level read). */
1773
+ minTrust?: LabelTrust;
1774
+ };
1775
+ }
1776
+ interface LabeledScenarioStore {
1777
+ observe(write: LabeledScenarioWrite): Promise<void>;
1778
+ sample(args: LabeledScenarioSampleArgs): Promise<LabeledScenarioRecord[]>;
1779
+ size(): Promise<{
1780
+ train: number;
1781
+ test: number;
1782
+ bySource: Record<string, number>;
1783
+ /** Count by trust tier — tells the flywheel how much gold it has
1784
+ * accumulated vs. raw corpus. */
1785
+ byTrust: Record<LabelTrust, number>;
1786
+ }>;
1787
+ }
1788
+ interface CampaignCellResult<TArtifact> {
1789
+ /** Manifest that produced this cell. Resumability refuses to reuse a cell
1790
+ * whose manifest differs from the current run. */
1791
+ manifestHash?: string;
1792
+ cellId: string;
1793
+ scenarioId: string;
1794
+ rep: number;
1795
+ generation?: number;
1796
+ artifact: TArtifact;
1797
+ judgeScores: Record<string, JudgeScore>;
1798
+ costUsd: number;
1799
+ /** True when at least one priced receipt used the model table instead of a provider bill. */
1800
+ costEstimated?: boolean;
1801
+ /** Exact durable receipts required to reuse this cached result. */
1802
+ costCallIds?: string[];
1803
+ /** Agent-call token usage committed by `ctx.cost.runPaidCall`.
1804
+ * `{ input: 0, output: 0 }` when no paid agent call was recorded. */
1805
+ tokenUsage: CampaignTokenUsage;
1806
+ /** Concrete model from the latest committed agent receipt. Consumed by
1807
+ * `buildRunRecord` to pin the model when the declared profile uses a
1808
+ * runtime-resolved sentinel. */
1809
+ resolvedModel?: string;
1810
+ durationMs: number;
1811
+ seed: number;
1812
+ cached: boolean;
1813
+ error?: string;
1814
+ }
1815
+ interface JudgeAggregate {
1816
+ mean: number;
1817
+ stdev: number;
1818
+ ci95: [number, number];
1819
+ n: number;
1820
+ }
1821
+ interface ScenarioAggregate {
1822
+ meanComposite: number;
1823
+ ci95: [number, number];
1824
+ n: number;
1825
+ }
1826
+ interface GenerationRecord {
1827
+ generationIndex: number;
1828
+ candidates: GenerationCandidate[];
1829
+ promoted: string[];
1830
+ }
1831
+ /** One scored candidate surface in a generation. `dimensions` + `scenarios`
1832
+ * let a reflective proposer ground its next proposal on WHICH
1833
+ * dimensions the candidate is weakest on and WHICH scenarios it best/worst
1834
+ * handled — the evidence a blind `Mutator` cannot see. */
1835
+ interface GenerationCandidate {
1836
+ surfaceHash: string;
1837
+ composite: number;
1838
+ ci95: [number, number];
1839
+ /** Exact surface this candidate mutated. */
1840
+ parentSurfaceHash?: string;
1841
+ /** Measured search-split composite of the exact parent surface. */
1842
+ parentComposite?: number;
1843
+ /** Candidate composite minus its parent's composite. Present only when the
1844
+ * candidate completed the designed denominator. */
1845
+ observedDeltaFromParent?: number;
1846
+ /** Whether this candidate had a scorable result for every designed campaign
1847
+ * cell and was therefore eligible for ranking, promotion, and Pareto
1848
+ * selection. Older externally-authored records may omit this field; loop
1849
+ * records always populate it. */
1850
+ eligibleForPromotion?: boolean;
1851
+ /** Exact denominator receipt for selection eligibility. Scores stay
1852
+ * descriptive: an incomplete candidate is retained with its observed score
1853
+ * and errors instead of receiving an invented penalty. */
1854
+ coverage?: {
1855
+ expectedCells: number;
1856
+ scorableCells: number;
1857
+ unscorableCells: Array<{
1858
+ cellId: string;
1859
+ reason: string;
1860
+ }>;
1861
+ };
1862
+ /** Mean score per judge dimension across all cells (scenarios × reps ×
1863
+ * judges that reported the dimension). */
1864
+ dimensions: Record<string, number>;
1865
+ /** Per-scenario composite (mean over reps + judges), plus the judge's
1866
+ * free-form `notes` for that scenario — the "why it scored low" evidence a
1867
+ * reflective proposer grounds its next edit on. Keep `notes` GENERALIZABLE
1868
+ * (which checks/lines/dimensions failed and how), NOT case-specific ground
1869
+ * truth: leaking expected answers into the prompt is memorization, and the
1870
+ * held-out gate would reject it anyway. */
1871
+ scenarios: Array<{
1872
+ scenarioId: string;
1873
+ composite: number;
1874
+ notes?: string;
1875
+ }>;
1876
+ /** Proposer-supplied short label for the change. Present when the proposer
1877
+ * returned a `ProposedCandidate`; absent for bare-surface mutators. */
1878
+ label?: string;
1879
+ /** Proposer-supplied rationale — WHY this candidate was proposed. The
1880
+ * "because rationale Z" the audit requires to survive to the result.
1881
+ * Present when the proposer returned a `ProposedCandidate`. */
1882
+ rationale?: string;
1883
+ /** Exact structured cause threaded from the proposer, when available. */
1884
+ candidateRecord?: PolicyEditCandidateRecord;
1885
+ }
1886
+ interface CampaignAggregates {
1887
+ byJudge: Record<string, JudgeAggregate>;
1888
+ byScenario: Record<string, ScenarioAggregate>;
1889
+ /** Canonical campaign accounting, including worker and judge calls. */
1890
+ cost: CostLedgerSummary;
1891
+ /** Compatibility alias of `cost.totalCostUsd`. */
1892
+ totalCostUsd: number;
1893
+ cellsExecuted: number;
1894
+ cellsSkipped: number;
1895
+ cellsCached: number;
1896
+ cellsFailed: number;
1897
+ }
1898
+ interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scenario> {
1899
+ /** sha256(scenarios, judges, dispatch source ref, optimizer config, seed). Stable identity for reruns. */
1900
+ manifestHash: string;
1901
+ seed: number;
1902
+ startedAt: string;
1903
+ endedAt: string;
1904
+ durationMs: number;
1905
+ cells: Array<CampaignCellResult<TArtifact>>;
1906
+ aggregates: CampaignAggregates;
1907
+ optimization?: {
1908
+ generations: GenerationRecord[];
1909
+ winnerSurfaceHash?: string;
1910
+ };
1911
+ gate?: GateResult;
1912
+ prUrl?: string;
1913
+ runDir: string;
1914
+ artifactsByPath: Record<string, string>;
1915
+ /** Substrate strips the input scenarios to id+kind for the result manifest;
1916
+ * consumers needing full payload look it up via the original input. The
1917
+ * type parameter `TScenario` is propagated for downstream consumers that
1918
+ * want narrowed types when extending `CampaignResult`. */
1919
+ scenarios: Array<Pick<TScenario, 'id' | 'kind'>>;
1920
+ }
1921
+
1922
+ /**
1923
+ * `CampaignStorage` — the filesystem seam `runCampaign` writes through
1924
+ * (run/cell dirs, the resumability cache, per-cell artifacts, trace spans).
1925
+ *
1926
+ * The default (`fsCampaignStorage`) is the Node filesystem — identical
1927
+ * behavior to the inline `node:fs` calls it replaces, so existing CLI
1928
+ * consumers are unaffected. `inMemoryCampaignStorage` keeps everything in a
1929
+ * `Map`, so the substrate runs in environments WITHOUT a filesystem
1930
+ * (Cloudflare Workers, Deno Deploy, other edge runtimes) — the campaign
1931
+ * still produces its `CampaignResult` (cells + aggregates) in memory;
1932
+ * artifacts/traces simply aren't persisted to disk.
1933
+ *
1934
+ * Paths are opaque keys to the in-memory adapter — it does not parse them,
1935
+ * so the same `join(...)`-built paths work unchanged across both adapters.
1936
+ */
1937
+ interface CampaignStorage {
1938
+ /** Ensure a directory exists (recursive). No-op for in-memory. */
1939
+ ensureDir(dir: string): void;
1940
+ /** Does this path exist (as a written file or an ensured dir)? */
1941
+ exists(path: string): boolean;
1942
+ /** Read a UTF-8 file; `undefined` when missing or unreadable. */
1943
+ read(path: string): string | undefined;
1944
+ /** Write a file (string or bytes). Parent dir is assumed ensured. */
1945
+ write(path: string, content: string | Uint8Array): void;
1946
+ /** Append only when the current UTF-8 byte length matches `expectedBytes`.
1947
+ * Returns the new length, or undefined when another writer won. */
1948
+ append?(path: string, content: string, expectedBytes: number): number | undefined;
1949
+ }
1950
+ /** Node-filesystem storage — the default. Lazily requires `node:fs` so the
1951
+ * module imports cleanly in non-Node runtimes (where the caller passes
1952
+ * `inMemoryCampaignStorage` instead and never constructs this).
1953
+ *
1954
+ * `createRequire(import.meta.url)` is the ESM-native lazy require — a bare
1955
+ * `require` is a ReferenceError under `"type": "module"`, which is exactly
1956
+ * the shape this package publishes. */
1957
+ declare function fsCampaignStorage(): CampaignStorage;
1958
+ /** In-memory storage for filesystem-less runtimes. Artifacts + trace spans
1959
+ * live in a `Map` for the duration of the run; the `CampaignResult` is
1960
+ * fully populated, but nothing is persisted to disk. */
1961
+ declare function inMemoryCampaignStorage(): CampaignStorage;
1962
+
1963
+ /**
1964
+ * `runCampaign` — Pass A substrate primitive. ONE function that orchestrates
1965
+ * scenarios → dispatch → artifacts → judges → aggregates, with full
1966
+ * reproducibility (seed + manifest hash), cell-level resumability, bootstrap
1967
+ * CIs, and the `LabeledScenarioStore` capture flywheel.
1968
+ *
1969
+ * Improvement loops (optimizer / gate / autoOnPromote) ride on top of this
1970
+ * primitive but live in `presets/run-improvement-loop.ts`. This file keeps
1971
+ * the core orchestrator minimal — Phase 1 of the Pass A track.
1972
+ */
1973
+
1974
+ interface RunCampaignOptions<TScenario extends Scenario, TArtifact> {
1975
+ scenarios: TScenario[];
1976
+ dispatch: DispatchFn<TScenario, TArtifact>;
1977
+ /**
1978
+ * Stable identity for the dispatch behavior, included in the manifest/cache
1979
+ * key. Set this when the same function name can run different models,
1980
+ * prompts, tools, or external config.
1981
+ */
1982
+ dispatchRef?: string;
1983
+ judges?: JudgeConfig<TArtifact, TScenario>[];
1984
+ /** Required for reproducibility. Default 42. */
1985
+ seed?: number;
1986
+ /** Per-scenario replicates for CI bands. Default 1; raise to 5+ for
1987
+ * bootstrap-tight intervals on critical eval. */
1988
+ reps?: number;
1989
+ /** When true (default), completed cells are cached by
1990
+ * (manifestHash, scenarioId, rep, generation). Re-runs skip cached cells. */
1991
+ resumable?: boolean;
1992
+ /** Optional store — when present, every artifact + judge score is captured
1993
+ * with the configured `captureSource`. Capture is default ON; pass `'off'`
1994
+ * to disable. */
1995
+ labeledStore?: LabeledScenarioStore | 'off';
1996
+ captureSource?: 'production-trace' | 'eval-run' | 'manual' | 'red-team' | 'synthetic';
1997
+ captureSourceVersionHash?: string;
1998
+ /** Hard spend cap. Each paid call reserves its enforced maximum before dispatch. */
1999
+ costCeiling?: number;
2000
+ /** Shared spend account. Improvement loops pass one ledger through every
2001
+ * campaign so the ceiling and returned total are run-wide. */
2002
+ costLedger?: CostLedger;
2003
+ /** Attribution label for receipts recorded by this campaign. */
2004
+ costPhase?: string;
2005
+ /** Max concurrent cells. Default 2. */
2006
+ maxConcurrency?: number;
2007
+ /**
2008
+ * Per-cell dispatch deadline in ms. A `dispatch` that neither resolves nor
2009
+ * rejects within this window is a hang (a stalled model request, an
2010
+ * exhausted runtime resource, a backend that never closes its stream). When
2011
+ * set, the cell's `ctx.signal` is aborted and the cell is recorded as a LOUD
2012
+ * error (`dispatch exceeded <N>ms`) so the campaign proceeds and the failure
2013
+ * is visible — instead of one wedged cell silently hanging the whole run (and
2014
+ * every loop/CI job above it) forever. `undefined`/`0` = unbounded (legacy).
2015
+ */
2016
+ dispatchTimeoutMs?: number;
2017
+ /** Required: where artifacts + traces land. A bare name (not an absolute path)
2018
+ * resolves to the shared `~/.tangle/traces/<repo>/runs/<name>` root so run
2019
+ * bundles never pollute a repo working tree. Pass an absolute path to override. */
2020
+ runDir: string;
2021
+ /** Subject repo for the shared run-dir root (defaults to the CWD basename).
2022
+ * Only consulted when `runDir` is a bare name. */
2023
+ repo?: string;
2024
+ /** Tracing posture. Default is the substrate's `FileSystemTraceStore` rooted
2025
+ * at `<runDir>/traces/`. `'off'` disables capture entirely — substrate
2026
+ * refuses this when the caller wires `autoOnPromote !== 'none'`. */
2027
+ tracing?: 'on' | 'off';
2028
+ /**
2029
+ * Per-cell usage expectation — the early, fine-grained sibling of the
2030
+ * batch `assertRealBackend` guard. A cell that produced an artifact (no
2031
+ * error) but reported `costUsd === 0` AND zero tokens is a stub: the
2032
+ * dispatch never reported LLM activity via `ctx.cost`. Modes:
2033
+ * - `'warn'` (default) — log the offending cell loudly, keep going.
2034
+ * - `'assert'` — throw `BackendIntegrityError` on the first such cell
2035
+ * (fail-fast; recommended for CI campaigns expecting real LLM calls).
2036
+ * - `'off'` — no check (replay / deterministic-only / offline analysis).
2037
+ */
2038
+ expectUsage?: 'assert' | 'warn' | 'off';
2039
+ /** Test seam — override the wall clock for deterministic tests. */
2040
+ now?: () => Date;
2041
+ /** Test seam — override per-cell trace writer factory. */
2042
+ buildTraceWriter?: (cellId: string, dir: string) => CampaignTraceWriter;
2043
+ /** Storage backend for run/cell dirs, the resumability cache, artifacts,
2044
+ * and trace spans. Default: the Node filesystem (`fsCampaignStorage`).
2045
+ * Pass `inMemoryCampaignStorage()` to run in a filesystem-less runtime
2046
+ * (Cloudflare Workers, Deno, edge) — the `CampaignResult` is still
2047
+ * produced; artifacts/traces just aren't persisted to disk. */
2048
+ storage?: CampaignStorage;
2049
+ /**
2050
+ * Optional per-cell placement strategy. Returns an opaque string the
2051
+ * substrate forwards as `ctx.placement` to the Dispatch — placement-aware
2052
+ * Dispatches (e.g. `httpDispatch` from `/adapters/http`) use it to route
2053
+ * each cell to the right worker, region, or sandbox. When unset, every
2054
+ * cell receives `ctx.placement = undefined` and behaves identically to
2055
+ * the in-process case.
2056
+ *
2057
+ * @example
2058
+ * cellPlacement: ({ scenario }) => scenario.tags?.includes('eu') ? 'eu-west' : 'us-east'
2059
+ */
2060
+ cellPlacement?: (input: {
2061
+ scenario: TScenario;
2062
+ rep: number;
2063
+ generation?: number;
2064
+ }) => string | undefined;
2065
+ }
2066
+ /**
2067
+ * Core campaign orchestrator: fan scenarios through dispatch, score with judges, aggregate bootstrap CIs, and persist reproducible `CampaignResult` records.
2068
+ */
2069
+ declare function runCampaign<TScenario extends Scenario, TArtifact>(opts: RunCampaignOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
2070
+
2071
+ /**
2072
+ * `runEval` — the simplest preset over `runCampaign`. No optimizer, no
2073
+ * gate, no auto-PR. Just: run scenarios through dispatch, score with
2074
+ * judges, return CampaignResult.
2075
+ *
2076
+ * The 80% case for consumers who want a scorecard, not an improvement loop.
2077
+ */
2078
+
2079
+ interface RunEvalOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'runDir'> {
2080
+ runDir: string;
2081
+ }
2082
+ /**
2083
+ * Simplest evaluation preset: run scenarios through dispatch, score with judges, and return a `CampaignResult` — no optimizer, no gate, no PR.
2084
+ */
2085
+ declare function runEval<TScenario extends Scenario, TArtifact>(opts: RunEvalOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
2086
+
2087
+ /**
2088
+ * `openAutoPr` — thin shell-out helper for the `runImprovementLoop` preset's
2089
+ * `autoOnPromote: 'pr'` mode. Substitutes for the per-product PR-opening
2090
+ * code consumers duplicated 4 times. The PR body includes the campaign's
2091
+ * manifest hash, gate verdict, and scorecard summary so reviewers can see
2092
+ * exactly what was promoted + why.
2093
+ *
2094
+ * NOT a deploy mechanism — this only OPENS a PR. The human reviews + merges.
2095
+ * The Shape B (`autoOnPromote: 'config'`) live-runtime-mutation path is
2096
+ * deferred to Pass B with the full shadow / canary / rollback stack.
2097
+ */
2098
+
2099
+ interface OpenAutoPrOptions<TArtifact, TScenario extends Scenario> {
2100
+ /** Campaign result to attach to the PR. */
2101
+ result: CampaignResult<TArtifact, TScenario>;
2102
+ /** Gate verdict explaining the promotion. Substrate refuses to open a PR
2103
+ * when `gate.decision !== 'ship'` — fails loud. */
2104
+ gate: GateResult;
2105
+ /** Promoted surface diff — typically the new system prompt addendum or
2106
+ * full profile diff. Substrate writes it as the PR body. */
2107
+ promotedDiff: string;
2108
+ /** GH owner/repo target (e.g., `tangle-network/gtm-agent`). */
2109
+ ghOwner: string;
2110
+ ghRepo: string;
2111
+ /** Branch name for the PR. Default `auto/<manifestHash[:12]>`. */
2112
+ branch?: string;
2113
+ /** PR title. Default includes manifest hash. */
2114
+ title?: string;
2115
+ /** Whether to actually open the PR or just dry-run. Default reads
2116
+ * `GH_AUTO_PR_TOKEN` env — present = open, absent = dry-run. */
2117
+ dryRun?: boolean;
2118
+ /** Test seam — substitute `gh pr create` invocation. */
2119
+ ghExec?: (args: string[]) => {
2120
+ stdout: string;
2121
+ stderr: string;
2122
+ status: number;
2123
+ };
2124
+ }
2125
+ interface OpenAutoPrResult {
2126
+ opened: boolean;
2127
+ prUrl?: string;
2128
+ dryRun: boolean;
2129
+ reason: string;
2130
+ }
2131
+ /**
2132
+ * Open a GitHub PR for a gate-approved surface promotion, attaching the manifest hash, gate verdict, and diff as the PR body.
2133
+ */
2134
+ declare function openAutoPr<TArtifact, TScenario extends Scenario>(options: OpenAutoPrOptions<TArtifact, TScenario>): OpenAutoPrResult;
2135
+
2136
+ /**
2137
+ * `runOptimization` — the improvement loop body. Runs N generations: the
2138
+ * `SurfaceProposer` proposes K candidate surfaces per generation, each
2139
+ * candidate runs a campaign (the measurement), and only a candidate that beats
2140
+ * the single global incumbent becomes the next generation's parent.
2141
+ * Proposer-agnostic — the same loop runs an evolutionary population mutator
2142
+ * (`evolutionaryProposer`) or any reflective / agentic proposer; they differ
2143
+ * only in how `propose()` picks candidates.
2144
+ *
2145
+ * This is `runLoop`'s shape (plan → measure → decide) specialized to surface
2146
+ * improvement: `proposer.propose` = plan, `runCampaign` = the measurement
2147
+ * (which runs the worker behind `dispatch`), the mean-composite ranking = the
2148
+ * validator, `proposer.decide` = the stop check.
2149
+ *
2150
+ * The gated-promotion shell (`runImprovementLoop`) wraps this with a holdout
2151
+ * re-score + release gate + optional PR.
2152
+ */
2153
+
2154
+ interface RunOptimizationBaseOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'dispatch'> {
2155
+ /** Initial mutable surface (typically system prompt or addendum). */
2156
+ baselineSurface: MutableSurface;
2157
+ /** Dispatcher that takes the CURRENT surface + scenario → artifact. */
2158
+ dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: Parameters<RunCampaignOptions<TScenario, TArtifact>['dispatch']>[1]) => Promise<TArtifact>;
2159
+ /** The candidate-generation strategy. Wrap a population `Mutator` via
2160
+ * `evolutionaryProposer({ mutator })`, or pass any reflective / agentic
2161
+ * proposer that implements `SurfaceProposer`. */
2162
+ proposer: SurfaceProposer;
2163
+ populationSize: number;
2164
+ maxGenerations: number;
2165
+ /** @deprecated The loop has one global incumbent and can promote only the
2166
+ * single candidate that beats it. Retained for source compatibility. */
2167
+ promoteTopK?: number;
2168
+ /** DEPTH knob forwarded to the proposer's `propose()` — max iterations the
2169
+ * agentic generator may take per candidate. */
2170
+ maxImprovementShots?: number;
2171
+ /** Optional analysis report forwarded to `propose()`. Opaque here; the
2172
+ * proposer types it. */
2173
+ report?: unknown;
2174
+ /** Structured findings forwarded to `propose()` as `ctx.findings`. A
2175
+ * findings producer emits these from the
2176
+ * generation's traces; findings-grounded proposers consume them. Opaque here;
2177
+ * the proposer types its `TFindings`. Empty when no producer is wired. */
2178
+ findings?: unknown[];
2179
+ /** Per-generation findings producer. Runs once on the BASELINE campaign
2180
+ * (as `generation: -1`, the baseline convention) before generation 0
2181
+ * proposes — so even a single-generation run proposes with trace context —
2182
+ * and then after each generation's candidates are scored with that
2183
+ * generation's results; whatever it returns REPLACES `ctx.findings` for the
2184
+ * NEXT `propose()`, so the diagnosis is refreshed each round instead
2185
+ * of being a static one-shot. Generic by design: the substrate does not
2186
+ * import an analyst — the consumer plugs its trace-analyst registry / HALO
2187
+ * here (reading the per-candidate `runDir` traces). When absent, findings
2188
+ * stay the static `opts.findings`. */
2189
+ analyzeGeneration?: (input: {
2190
+ generation: number;
2191
+ runDir: string;
2192
+ candidates: Array<{
2193
+ surfaceHash: string;
2194
+ campaign: CampaignResult<TArtifact, TScenario>;
2195
+ composite: number;
2196
+ }>;
2197
+ history: GenerationRecord[];
2198
+ /** Shared run spend account and receipt attribution phase. */
2199
+ costLedger?: CostLedger;
2200
+ costPhase?: string;
2201
+ }) => Promise<unknown[]>;
2202
+ }
2203
+ type RunOptimizationOptions<TScenario extends Scenario, TArtifact> = RunOptimizationBaseOptions<TScenario, TArtifact>;
2204
+ interface RunOptimizationResult<TArtifact, TScenario extends Scenario> {
2205
+ generations: Array<{
2206
+ record: GenerationRecord;
2207
+ surfaces: Array<{
2208
+ surfaceHash: string;
2209
+ surface: MutableSurface;
2210
+ campaign: CampaignResult<TArtifact, TScenario>;
2211
+ }>;
2212
+ }>;
2213
+ winnerSurface: MutableSurface;
2214
+ winnerSurfaceHash: string;
2215
+ /** Proposer label for the promoted surface. Present when the winning
2216
+ * candidate came from a `ProposedCandidate` (a reflective proposer);
2217
+ * absent when the winner is the baseline or a bare-surface mutator. */
2218
+ winnerLabel?: string;
2219
+ /** Proposer rationale for the promoted surface — the "because Z" that
2220
+ * motivated the winning change. Survives to `SelfImproveResult` and the
2221
+ * emitted provenance record. Absent when the winner is the baseline. */
2222
+ winnerRationale?: string;
2223
+ baselineCampaign: CampaignResult<TArtifact, TScenario>;
2224
+ /** Run-wide spend, including agents, proposers, analysts, and judges. */
2225
+ cost: CostLedgerSummary;
2226
+ /** The GEPA Pareto frontier across every scored surface (baseline + all
2227
+ * generations) by per-scenario objective vector — the non-dominated set.
2228
+ * Each generation's `propose()` received the frontier-so-far as
2229
+ * `ctx.paretoParents`; this is the final frontier. A surface here that is
2230
+ * NOT the winner is uniquely best on some scenario the winner loses on. */
2231
+ paretoFrontier: ParetoParent[];
2232
+ }
2233
+
2234
+ /**
2235
+ * `runImprovementLoop` — the gated-promotion shell around the improvement
2236
+ * loop body (`runOptimization`). Proposes candidate surfaces via the
2237
+ * `SurfaceProposer`, re-scores the winner against the baseline on a
2238
+ * holdout set, runs the release gate, and optionally opens a PR.
2239
+ *
2240
+ * Role vocabulary (see docs/design/loop-taxonomy.md):
2241
+ * - PROPOSER = the `SurfaceProposer` (evolutionary GEPA mutator OR
2242
+ * reflective analyst). Proposes candidate SURFACES — the
2243
+ * worker's system prompt / tool config — NOT conversation
2244
+ * turns.
2245
+ * - MEASUREMENT= `runCampaign`. Scores one surface by running the worker
2246
+ * (via `dispatch`) over scenarios and judging the output.
2247
+ * - WORKER = the agent harness in the sandbox, invoked behind the
2248
+ * topology-opaque `dispatch` seam — never referenced here.
2249
+ *
2250
+ * Distinct from `runLoop` in `@tangle-network/agent-runtime`, which is the
2251
+ * INNER conversation loop (execution driver ↔ workers in a sandbox). `runImprovementLoop`
2252
+ * is the OUTER loop: it improves the surface that those workers run.
2253
+ *
2254
+ * Hard-refuses unsafe configurations:
2255
+ * - `tracing: 'off'` when a proposer is wired (improvement is unattributable)
2256
+ * - `autoOnPromote: 'config'` — DEFERRED to Pass B; v0.40 only ships
2257
+ * `'pr'` and `'none'`.
2258
+ */
2259
+
2260
+ type RunImprovementLoopOptions<TScenario extends Scenario, TArtifact> = RunOptimizationOptions<TScenario, TArtifact> & {
2261
+ /** Holdout scenarios kept OUT of the training optimization pool — used
2262
+ * ONLY to score baseline vs winner for the gate. */
2263
+ holdoutScenarios: TScenario[];
2264
+ /** Promotion gate. Substrate strongly recommends `defaultProductionGate`
2265
+ * for production wiring (composes red-team / reward-hacking / canary /
2266
+ * heldout). */
2267
+ gate: Gate<TArtifact, TScenario>;
2268
+ /** What to do when the gate ships:
2269
+ * - `'pr'`: open a PR via `openAutoPr`
2270
+ * - `'none'`: just report — caller decides what to do with the winner
2271
+ * v0.40 does NOT support `'config'` (live-runtime self-mutation) —
2272
+ * deferred to Pass B behind safety stack. */
2273
+ autoOnPromote: 'pr' | 'none';
2274
+ /** GH owner / repo for the auto-PR. Required when autoOnPromote === 'pr'. */
2275
+ ghOwner?: string;
2276
+ ghRepo?: string;
2277
+ /** Optional render override — substrate writes a diff-shaped surface; pass
2278
+ * a function to format the promoted surface differently. */
2279
+ renderPromotedDiff?: (winnerSurface: MutableSurface, baselineSurface: MutableSurface) => string;
2280
+ /** Placebo control. When supplied AND the winner differs from baseline, the
2281
+ * loop scores a THIRD holdout arm: the winner surface with its content
2282
+ * footprint-matched-blanked by this function (typically via `neutralizeText`).
2283
+ * Its scores are exposed to the gate as `ctx.neutralizedJudgeScores`, letting
2284
+ * a `neutralizationGate` reject a win whose lift survives blanking the content
2285
+ * (decorative — driven by footprint, not content). Costs one extra holdout
2286
+ * campaign; omit to skip. Return a byte/layout-matched blank of the winner. */
2287
+ neutralize?: (winnerSurface: MutableSurface, baselineSurface: MutableSurface) => MutableSurface;
2288
+ };
2289
+ interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario> extends RunOptimizationResult<TArtifact, TScenario> {
2290
+ baselineOnHoldout: CampaignResult<TArtifact, TScenario>;
2291
+ winnerOnHoldout: CampaignResult<TArtifact, TScenario>;
2292
+ gateResult: Awaited<ReturnType<Gate<TArtifact, TScenario>['decide']>>;
2293
+ /** Unified baseline→winner surface diff. Computed UNCONDITIONALLY (not only
2294
+ * when `autoOnPromote === 'pr'`) so the diff that the gate decided on is
2295
+ * always present on the result + in the emitted provenance record. Empty
2296
+ * string when winner == baseline (no change to diff). */
2297
+ promotedDiff: string;
2298
+ prResult?: ReturnType<typeof openAutoPr>;
2299
+ }
2300
+ /**
2301
+ * Gated-promotion shell over `runOptimization`: scores the winner against the baseline on a holdout set, runs the release gate, and optionally opens a PR.
2302
+ */
2303
+ declare function runImprovementLoop<TScenario extends Scenario, TArtifact>(opts: RunImprovementLoopOptions<TScenario, TArtifact>): Promise<RunImprovementLoopResult<TArtifact, TScenario>>;
2304
+
2305
+ declare const REFERENCE_EQUIVALENCE_JUDGE_VERSION = "reference-equivalence-judge-v1-2026-07-13";
2306
+ declare const REFERENCE_EQUIVALENCE_INPUT_LIMITS: {
2307
+ readonly userRequest: 8000;
2308
+ readonly expectedAnswer: 32000;
2309
+ readonly candidateOutput: 32000;
2310
+ };
2311
+ interface ReferenceEquivalenceScenario extends Scenario {
2312
+ userRequest: string;
2313
+ expectedAnswer: string;
2314
+ }
2315
+ interface ReferenceEquivalenceJudgeInput {
2316
+ userRequest: string;
2317
+ expectedAnswer: string;
2318
+ candidateOutput: string;
2319
+ }
2320
+ interface ReferenceEquivalenceJudgeOptions {
2321
+ /** Injected transport. No implicit provider or credentials are selected. */
2322
+ chat: ChatClient;
2323
+ /** Falls back to the ChatClient's default model. */
2324
+ model?: string;
2325
+ /** Used only by the direct-call adapter. */
2326
+ signal?: AbortSignal;
2327
+ /** Optional receipt destination for direct calls; campaigns supply their own. */
2328
+ costLedger?: CostLedger;
2329
+ }
2330
+ interface ReferenceEquivalenceJudgeResult extends LlmCallMetadata {
2331
+ kind: 'reference-equivalence';
2332
+ version: string;
2333
+ score: number;
2334
+ rationale: string;
2335
+ }
2336
+ /** Build the campaign-native expected-answer judge. */
2337
+ declare function createReferenceEquivalenceJudge(options: ReferenceEquivalenceJudgeOptions): JudgeConfig<string, ReferenceEquivalenceScenario>;
2338
+ /** Direct-call adapter over the campaign judge for product callers. */
2339
+ declare function runReferenceEquivalenceJudge(input: ReferenceEquivalenceJudgeInput, options: ReferenceEquivalenceJudgeOptions): Promise<ReferenceEquivalenceJudgeResult>;
2340
+
2341
+ /**
2342
+ * `evolutionaryProposer` — adapts a stateless `Mutator` (population mutation:
2343
+ * GEPA / AxGEPA / reflective-mutation) into a `SurfaceProposer`. This is
2344
+ * the evolutionary strategy: each generation, mutate the current best surface
2345
+ * into N candidates, measure, select. No generation memory beyond the current
2346
+ * surface; the loop body handles ranking + promotion.
2347
+ *
2348
+ * The reflective alternative is agent-runtime's runtime proposer with a
2349
+ * `reflectiveGenerator` / `agenticGenerator`: it reasons over the report +
2350
+ * trace findings to propose targeted edits rather than blind mutations. Both
2351
+ * conform to `SurfaceProposer`; the improvement loop is identical either way.
2352
+ */
2353
+
2354
+ interface EvolutionaryProposerOptions<TFindings = unknown> {
2355
+ mutator: Mutator<TFindings>;
2356
+ /** External findings fed to the mutator each generation. Default: []. */
2357
+ findings?: TFindings[];
2358
+ }
2359
+ /**
2360
+ * Wrap a stateless `Mutator` (GEPA, AxGEPA, reflective-mutation) as a `SurfaceProposer` that mutates the current best surface into N candidates each generation.
2361
+ */
2362
+ declare function evolutionaryProposer<TFindings = unknown>(opts: EvolutionaryProposerOptions<TFindings>): SurfaceProposer<TFindings>;
2363
+
2364
+ /**
2365
+ * `gepaProposer` — a reflective `SurfaceProposer` for prompt-tier surfaces.
2366
+ * Each generation it reflects on the prior best candidate's per-scenario
2367
+ * scores + weakest dimensions, asks an LLM to propose targeted rewrites of
2368
+ * the current surface, and returns them as the next population.
2369
+ *
2370
+ * Maps onto the GEPA paper (Agrawal et al., arXiv:2507.19457):
2371
+ * - *Reflection*: each generation reflects on the best parent's weakest
2372
+ * dimensions + per-scenario top/bottom scores to propose targeted rewrites.
2373
+ * - *Pareto frontier*: `runOptimization` maintains the non-dominated set of
2374
+ * surfaces across generations (per-scenario objective vectors) and supplies
2375
+ * it as `ctx.paretoParents`. A surface uniquely best on one hard scenario
2376
+ * survives even when its mean composite is lower.
2377
+ * - *Combine complementary lessons*: when the frontier has >1 member, the
2378
+ * first population slot is a merge of those parents' strengths (one LLM
2379
+ * call citing each parent's winning scenarios). Toggle via `combineParents`.
2380
+ * Dominance is computed by the package-canonical `paretoFrontier` (`pareto.ts`).
2381
+ *
2382
+ * Optional `constraints` move structured-doc guards into the proposer
2383
+ * (preserve H2 section headings, cap sentence-level edits) — useful when
2384
+ * the surface IS a structured procedure like a SKILL.md / runbook /
2385
+ * judge rubric. When `constraints` is omitted, behavior is unchanged.
2386
+ *
2387
+ * The proposer is surface-agnostic — any string surface in any consumer opts
2388
+ * in by selecting it. Reuses the generic reflection primitive
2389
+ * (`buildReflectionPrompt` / `parseReflectionResponse`) and the router client.
2390
+ *
2391
+ * Earns its keep where there is real per-instance signal (which the
2392
+ * dimensional + per-scenario evidence + the `LabeledScenarioStore` flywheel
2393
+ * now provide). For thin-signal surfaces it degrades to plain reflection.
2394
+ * On generation 0 (no history) it reflects on the current surface against
2395
+ * the mutation primitives alone.
2396
+ */
2397
+
2398
+ interface GepaProposerConstraints {
2399
+ /** H2 section headings that MUST appear unchanged in every candidate.
2400
+ * When set, the proposer auto-detects current H2s if this is empty AND
2401
+ * rejects any candidate that drops or renames a preserved heading.
2402
+ * Use when the surface is a structured doc (SKILL.md, runbook,
2403
+ * sectioned system prompt, judge rubric). */
2404
+ preserveSections?: string[];
2405
+ /** Maximum sentence-level edits per candidate vs the parent surface.
2406
+ * Rejection threshold = maxSentenceEdits × 2 (counts adds + removes).
2407
+ * Inspired by SkillOpt's edit-budget as a "textual learning rate."
2408
+ * Cap prevents an LLM rewrite from overwriting useful prior rules. */
2409
+ maxSentenceEdits?: number;
2410
+ }
2411
+ interface GepaProposerOptions {
2412
+ /** Router transport (apiKey/baseUrl). */
2413
+ llm: LlmClientOptions;
2414
+ /** Model that performs the reflection. */
2415
+ model: string;
2416
+ /** Optional ledger for direct proposer use. Campaign context takes precedence. */
2417
+ costLedger?: CostLedger;
2418
+ /** What is being optimized — appears in the reflection prompt for orientation. */
2419
+ target: string;
2420
+ /** Surface-specific mutation levers offered to the model. */
2421
+ mutationPrimitives?: string[];
2422
+ /** Top/bottom scenarios surfaced as evidence each generation. Default 3. */
2423
+ evidenceK?: number;
2424
+ /** Reflection sampling temperature. Default 0.7. */
2425
+ temperature?: number;
2426
+ /** Reflection max tokens. Default 6000. */
2427
+ maxTokens?: number;
2428
+ /** Structured-doc constraints. Candidates violating any are rejected
2429
+ * post-parse and dropped from the returned population. */
2430
+ constraints?: GepaProposerConstraints;
2431
+ /** GEPA combine-complementary-lessons: when the loop supplies a Pareto
2432
+ * frontier of >1 non-dominated parents (`ctx.paretoParents`), spend one
2433
+ * slot of the population on a merge of their strengths. Default `true` —
2434
+ * this is the GEPA-faithful behavior; the merge only fires once the
2435
+ * frontier has more than one member (generation ≥ 1). Set `false` for
2436
+ * pure single-parent reflection. */
2437
+ combineParents?: boolean;
2438
+ /** Cap on how many frontier parents feed one combine prompt (highest
2439
+ * composite first), to bound prompt size. Default 4. */
2440
+ combineMaxParents?: number;
2441
+ }
2442
+ /**
2443
+ * GEPA reflective proposer: each generation reflects on the weakest scenarios and dimensions to produce targeted prompt rewrites, optionally combining Pareto-frontier parents.
2444
+ */
2445
+ declare function gepaProposer(opts: GepaProposerOptions): SurfaceProposer;
2446
+
2447
+ /**
2448
+ * Compose multiple `Gate` implementations — every gate must pass for the
2449
+ * composite to ship. Closes the alignment reviewer's "default-only
2450
+ * heldOutGate + costGate would happily promote a reward-hacked prompt"
2451
+ * concern by making safety gates first-class composable defaults.
2452
+ */
2453
+
2454
+ /** Compose gates — all must `ship` for the composite to `ship`. First
2455
+ * non-ship verdict short-circuits the composite verdict, but ALL gates run
2456
+ * (so the result records every gate's reason — useful for diagnostics). */
2457
+ declare function composeGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(...gates: Array<Gate<TArtifact, TScenario>>): Gate<TArtifact, TScenario>;
2458
+
2459
+ /**
2460
+ * Dataset — versioned, sliceable, content-hashed scenario collection.
2461
+ *
2462
+ * Scenarios stop being ephemeral arrays and become first-class
2463
+ * artifacts. Every Dataset carries:
2464
+ * - content hash (sha256 over canonicalized scenario array)
2465
+ * - provenance (contributor, createdAt, sourceUrl)
2466
+ * - split labels (train | dev | test | holdout)
2467
+ * - difficulty tiers (easy | medium | hard | extreme)
2468
+ * - tags (free-form, per-scenario)
2469
+ *
2470
+ * `Dataset.slice({ difficulty, split, holdout, seed })` returns a
2471
+ * deterministic, reproducible subset. Holdout slices are locked: you
2472
+ * can read them but `mutate` throws, which prevents "oh I'll just
2473
+ * tweak that one scenario" contamination drift.
2474
+ */
2475
+ type DatasetSplit = 'train' | 'dev' | 'test' | 'holdout';
2476
+ type DatasetDifficulty = 'easy' | 'medium' | 'hard' | 'extreme';
2477
+ interface DatasetScenario {
2478
+ id: string;
2479
+ /** Arbitrary payload; the framework doesn't interpret it. */
2480
+ payload: unknown;
2481
+ split?: DatasetSplit;
2482
+ difficulty?: DatasetDifficulty;
2483
+ /** Canary token that MUST NOT round-trip through a correct agent output. */
2484
+ canary?: string;
2485
+ /**
2486
+ * Behavioral-canary forbidden pattern. A string OR a serialized regex
2487
+ * (`/.../flags`) that the agent under test MUST NOT emit. Used by
2488
+ * {@link import('./canary').checkBehavioralCanary | checkBehavioralCanary},
2489
+ * which inverts the contamination-style semantic: presence in the
2490
+ * agent output is a LEAK / failure, not a positive signal.
2491
+ *
2492
+ * Falls back to {@link canary} when omitted.
2493
+ */
2494
+ forbiddenPattern?: string;
2495
+ tags?: Record<string, string>;
2496
+ }
2497
+
2498
+ /**
2499
+ * Red-team battery — adversarial scenario corpus with per-category
2500
+ * scorers.
2501
+ *
2502
+ * Every scenario is a `DatasetScenario` with a `redTeam: {...}` payload
2503
+ * that declares: category, expected refusal, attack surface. Scorers
2504
+ * inspect the agent's output + trajectory and return a PASS/FAIL per
2505
+ * category. Aggregated via `redTeamReport`.
2506
+ *
2507
+ * The built-in corpus is intentionally conservative — 40 cases
2508
+ * covering the most-seen attack patterns. Extend via `extendCorpus`.
2509
+ */
2510
+
2511
+ type RedTeamCategory = 'prompt_injection_direct' | 'prompt_injection_indirect' | 'jailbreak_persona' | 'jailbreak_dan' | 'pii_leak' | 'permission_escalation' | 'data_exfil' | 'policy_override';
2512
+ interface RedTeamPayload {
2513
+ category: RedTeamCategory;
2514
+ /** The hostile input the agent sees. */
2515
+ input: string;
2516
+ /**
2517
+ * What a safe agent should do: "refuse" means explicit refusal,
2518
+ * "ignore" means process benignly without executing the attack.
2519
+ */
2520
+ expected: 'refuse' | 'ignore';
2521
+ /** Substring(s) that MUST NOT appear in output. */
2522
+ forbiddenStrings?: string[];
2523
+ /** Tool names the agent MUST NOT call for this scenario. */
2524
+ forbiddenTools?: string[];
2525
+ }
2526
+ interface RedTeamCase extends DatasetScenario {
2527
+ payload: RedTeamPayload;
2528
+ }
2529
+
2530
+ /**
2531
+ * `defaultProductionGate` — composes the substrate's existing safety
2532
+ * primitives (red-team / reward-hacking / canary / heldout) into a single
2533
+ * Gate.decide shape. Closes the alignment + Anthropic-SI reviewers' "safety
2534
+ * primitives are off the critical path" blocker.
2535
+ *
2536
+ * The composition is opinionated — when consumers wire `runImprovementLoop`,
2537
+ * THIS gate is the default. Consumers can still pass a custom gate to
2538
+ * override; the recommended pattern is to compose THIS gate with whatever
2539
+ * extra domain-specific gates they need (`composeGate(defaultProductionGate(...), customGate)`).
2540
+ */
2541
+
2542
+ interface DefaultProductionGateOptions {
2543
+ /** Required: scenarios held out from training; substrate compares
2544
+ * candidate-on-holdout vs baseline-on-holdout. */
2545
+ holdoutScenarios: Scenario[];
2546
+ /** Minimum held-out lift the **paired-bootstrap CI lower bound** must clear
2547
+ * to ship — NOT a point estimate. Default 0 ⇒ "confidently positive at the
2548
+ * confidence level". Interpreted in the judge's native composite scale (set
2549
+ * e.g. 2 for a 0-100 rubric to require a ≥2-point significant gain). */
2550
+ deltaThreshold?: number;
2551
+ /** Confidence level for the held-out + dimension bootstraps. Default 0.95. */
2552
+ confidence?: number;
2553
+ /** Bootstrap resamples. Default 2000. */
2554
+ bootstrapResamples?: number;
2555
+ /** Fixed bootstrap seed for a deterministic verdict. Default 1337. */
2556
+ bootstrapSeed?: number;
2557
+ /** Minimum paired holdout observations (scenarios × reps) before a
2558
+ * significance claim is allowed; below it the gate HOLDS with `few_runs`
2559
+ * rather than reading a degenerate CI. Default 3. */
2560
+ minProductiveRuns?: number;
2561
+ /** Ship statistic for the held-out significance test. Default `'mean'`
2562
+ * (tie-robust — see `heldoutSignificance`). Pass `'median'` for
2563
+ * outlier-robustness at the cost of tie-blindness. */
2564
+ heldoutStatistic?: 'mean' | 'median';
2565
+ /** Critical judge dimensions that must NOT significantly regress even when
2566
+ * the net composite rises (anti-Goodhart). The gate HOLDS if any listed
2567
+ * dimension's paired-delta CI lower bound < −`regressionTolerance`. E.g.
2568
+ * `['hallucination_free']` for a legal agent. */
2569
+ criticalDimensions?: string[];
2570
+ /** Tolerance for the per-dimension regression guard, in the dimension's
2571
+ * native scale. When omitted it auto-scales off observed magnitudes:
2572
+ * 0.05 on [0,1], 5 on 0-100. */
2573
+ regressionTolerance?: number;
2574
+ /** Total $ budget for ALL cells in this campaign — including baseline + candidate.
2575
+ * Composite verdict refuses to ship when spend exceeded budget. */
2576
+ budgetUsd?: number;
2577
+ /** Red-team cases to probe candidate outputs against. When omitted the
2578
+ * substrate uses `DEFAULT_RED_TEAM_CORPUS`. Provide a domain-specific
2579
+ * battery for tighter coverage. */
2580
+ redTeamBattery?: RedTeamCase[];
2581
+ /** Run records (oldest-first) needed for the reward-hacking detector.
2582
+ * Substrate populates from prior production-loop generations. */
2583
+ recentRuns?: RunRecord[];
2584
+ /** When true, the gate refuses to ship if the reward-hacking detector
2585
+ * fires at the `gaming` severity. Default true. */
2586
+ blockOnRewardHackingGaming?: boolean;
2587
+ }
2588
+ /**
2589
+ * Opinionated production gate composing held-out significance, red-team, reward-hacking, and canary checks into a single `Gate.decide` decision.
2590
+ */
2591
+ declare function defaultProductionGate<TArtifact, TScenario extends Scenario>(options: DefaultProductionGateOptions): Gate<TArtifact, TScenario>;
2592
+
2593
+ /**
2594
+ * @module
2595
+ * Composable held-out promotion gate backed by paired bootstrap confidence.
2596
+ *
2597
+ * Pair by full `scenario:rep` cellId, bootstrap the paired candidate-minus-
2598
+ * baseline delta, and ship only when CI.low strictly clears the threshold with
2599
+ * at least `minProductiveRuns` paired observations.
2600
+ *
2601
+ * Use when you want held-out significance as ONE of N composed gates instead
2602
+ * of the full `defaultProductionGate` stack (which adds critical-dimension
2603
+ * regression + reward-hacking guards on top).
2604
+ */
2605
+
2606
+ interface HeldOutGateOptions<TScenario extends Scenario = Scenario> {
2607
+ scenarios: TScenario[];
2608
+ /** Effect-size threshold the CI lower bound must clear, in the judge's native
2609
+ * scale. Default 0.5. Equality holds; CI.low must be greater than this value. */
2610
+ deltaThreshold?: number;
2611
+ /** Bootstrap CI confidence. Default 0.95. */
2612
+ confidence?: number;
2613
+ /** Minimum paired holdout observations to claim significance. Default 3. */
2614
+ minProductiveRuns?: number;
2615
+ /** Bootstrap resamples. Default 2000. */
2616
+ resamples?: number;
2617
+ /** Fixed bootstrap seed for deterministic verdicts. Default 1337. */
2618
+ bootstrapSeed?: number;
2619
+ }
2620
+ /**
2621
+ * Composable held-out gate: ships only when the PAIRED bootstrap CI lower bound
2622
+ * of the candidate-minus-baseline composite delta clears `deltaThreshold`.
2623
+ */
2624
+ declare function heldOutGate<TArtifact, TScenario extends Scenario>(options: HeldOutGateOptions<TScenario>): Gate<TArtifact, TScenario>;
2625
+
2626
+ /**
2627
+ * Pareto frontier — multi-objective optimization over candidate runs.
2628
+ *
2629
+ * Lifted from ADC pareto.ts and blueprint-agent frontier.ts. When you're
2630
+ * trading off (cost, latency, quality) or (passRate, tokenBudget,
2631
+ * ttfb), you rarely have a single "winner" — you have a set of
2632
+ * non-dominated candidates. This module exposes:
2633
+ *
2634
+ * - `paretoFrontier`: filter a set of candidates to the non-dominated ones
2635
+ * - `dominates`: does A dominate B across all objectives?
2636
+ *
2637
+ * Each objective is declared with a direction: 'maximize' (higher=better)
2638
+ * or 'minimize' (lower=better). Candidates are any object; pass an
2639
+ * `objective(candidate)` accessor.
2640
+ */
2641
+ type Direction = 'maximize' | 'minimize';
2642
+
2643
+ interface ContinuousAgreement {
2644
+ /** Cohen's κ_w with quadratic weights, computed on raw [0,1] scores. */
2645
+ weightedKappa: number;
2646
+ /** ICC(2,1): two-way random effects, absolute agreement, single rater. */
2647
+ icc: number;
2648
+ /** Pearson product-moment correlation (averaged over rater pairs if N>2). */
2649
+ pearson: number;
2650
+ /** Spearman rank correlation (averaged over rater pairs if N>2). */
2651
+ spearman: number;
2652
+ /** 95% bootstrap percentile CIs over items. */
2653
+ ci: {
2654
+ icc: [number, number];
2655
+ weightedKappa: [number, number];
2656
+ };
2657
+ /** Number of complete items (no NaN across raters). */
2658
+ n: number;
2659
+ /** Number of raters. */
2660
+ raters: number;
2661
+ }
2662
+
2663
+ interface PairedBootstrapResult {
2664
+ /** Number of paired observations. */
2665
+ n: number;
2666
+ /** Median of paired deltas (after − before). */
2667
+ median: number;
2668
+ /** Mean of paired deltas. */
2669
+ mean: number;
2670
+ /** Lower bound of the bootstrap CI on the chosen statistic. */
2671
+ low: number;
2672
+ /** Upper bound of the bootstrap CI on the chosen statistic. */
2673
+ high: number;
2674
+ /** Confidence level used (e.g. 0.95). */
2675
+ confidence: number;
2676
+ /** Number of bootstrap resamples used. */
2677
+ resamples: number;
2678
+ }
2679
+
2680
+ /**
2681
+ * Promotion policy over the evidence VECTOR — the substrate's answer to "never
2682
+ * collapse the multi-objective promotion decision into one scalar." A
2683
+ * `defaultProductionGate` is one opinionated composition; this module factors
2684
+ * the decision into two reusable pieces so MANY policies can compete over the
2685
+ * SAME evidence (the quant-desk pattern: one evidence bus, plural strategies):
2686
+ *
2687
+ * buildEvidenceVector(ctx, objectives, opts) -> EvidenceVector // the bus
2688
+ * PromotionPolicy = (ev: EvidenceVector) => GateResult // a strategy
2689
+ * paretoPolicy(ev) // the default strategy
2690
+ * paretoSignificanceGate(options): Gate // bus + policy as a Gate
2691
+ *
2692
+ * The Pareto policy is SYMMETRIC multi-objective: every objective is BOTH a
2693
+ * potential gain source AND a safety floor (unlike `defaultProductionGate`,
2694
+ * where only `composite` can win and `criticalDimensions` are pure floors). A
2695
+ * candidate ships iff it weakly DOMINATES the baseline at the confidence level —
2696
+ * no objective credibly worse (CI floor breach) AND at least one objective
2697
+ * credibly better (CI gain). Insufficient evidence on ANY axis -> need_more_work
2698
+ * (NOT folded into hold: "gather more reps" and "reject" are different actions).
2699
+ *
2700
+ * Cost/latency are NOT CI axes here — `GateContext` carries only an aggregate
2701
+ * per-side cost, no per-cell observation vector to bootstrap. Treat them as hard
2702
+ * constraints (compose with a budget gate via `composeGate`), not faked CIs.
2703
+ */
2704
+
2705
+ /** Where an objective's per-cell scalar comes from. `composite` reads the
2706
+ * judge's composite; `dimension` reads a named per-dimension score. */
2707
+ type ObjectiveSource = {
2708
+ kind: 'composite';
2709
+ } | {
2710
+ kind: 'dimension';
2711
+ dimension: string;
2712
+ };
2713
+ interface PromotionObjective {
2714
+ /** Stable label used in reports + `contributingGates`. */
2715
+ name: string;
2716
+ source: ObjectiveSource;
2717
+ /** 'maximize' (quality dims) or 'minimize' (error/risk/length dims). Orients
2718
+ * the paired delta so a positive bootstrap always means "candidate better". */
2719
+ direction: Direction;
2720
+ /** The good-direction paired-delta CI lower bound must EXCEED this to count
2721
+ * as a significant gain on this axis. Interpreted in the judge's native
2722
+ * scale. Default 0 (⇒ "confidently better"). */
2723
+ gainThreshold?: number;
2724
+ /** A floor breach (regression) is declared when the good-direction CI lower
2725
+ * bound is below −floorTolerance. When omitted it auto-scales off observed
2726
+ * magnitudes (0.05 on [0,1], 5 on 0-100), matching `dimensionRegressions`. */
2727
+ floorTolerance?: number;
2728
+ }
2729
+ /** Per-axis verdict from the good-direction paired bootstrap. */
2730
+ type AxisVerdict = 'improved' | 'regressed' | 'flat' | 'few_runs';
2731
+ interface AxisEvidence {
2732
+ name: string;
2733
+ source: ObjectiveSource;
2734
+ direction: Direction;
2735
+ /** Paired bootstrap on the GOOD-DIRECTION delta (oriented by `direction`):
2736
+ * a positive value means the candidate is better on this axis. */
2737
+ bootstrap: PairedBootstrapResult;
2738
+ /** Paired observations contributing to this axis. */
2739
+ n: number;
2740
+ gainThreshold: number;
2741
+ floorTolerance: number;
2742
+ verdict: AxisVerdict;
2743
+ }
2744
+ interface EvidenceVector {
2745
+ /** One entry per objective — NOTHING averaged across axes. */
2746
+ axes: AxisEvidence[];
2747
+ /** Smallest paired n across axes that produced observations — the binding
2748
+ * evidence-sufficiency constraint. 0 when no axis produced observations. */
2749
+ minN: number;
2750
+ /** Aggregate per-side cost from the gate context (a constraint input, not a
2751
+ * CI axis — see the module header). */
2752
+ cost: {
2753
+ candidate: number;
2754
+ baseline: number;
2755
+ };
2756
+ }
2757
+ /** A promotion strategy: a pure function from the evidence vector to a verdict.
2758
+ * Many policies can run over the same `EvidenceVector` and disagree — that's
2759
+ * the point (competing strategies, shared evidence). */
2760
+ type PromotionPolicy = (ev: EvidenceVector) => GateResult;
2761
+ interface BuildEvidenceVectorOptions {
2762
+ /** Minimum paired observations before an axis can claim significance; below
2763
+ * it the axis is `few_runs`. Default 3. */
2764
+ minProductiveRuns?: number;
2765
+ /** Confidence level for every axis bootstrap. Default 0.95. */
2766
+ confidence?: number;
2767
+ /** Bootstrap resamples. Default 2000. */
2768
+ resamples?: number;
2769
+ /** Fixed bootstrap seed for a deterministic, reproducible verdict. Default 1337. */
2770
+ seed?: number;
2771
+ }
2772
+ /**
2773
+ * The Evidence Bus. For each objective, pair candidate vs baseline by full
2774
+ * cellId and bootstrap a CI on the good-direction paired delta. Reuses the
2775
+ * exact `pairHoldout` + `pairedBootstrap` machinery the held-out gate uses, so
2776
+ * a single source of truth governs pairing granularity + scale handling.
2777
+ */
2778
+ declare function buildEvidenceVector<TArtifact, TScenario extends Scenario>(ctx: GateContext<TArtifact, TScenario>, objectives: PromotionObjective[], opts?: BuildEvidenceVectorOptions): EvidenceVector;
2779
+ /**
2780
+ * The default strategy: symmetric multi-objective Pareto significance. Ship iff
2781
+ * the candidate weakly dominates the baseline at the confidence level — no axis
2782
+ * credibly worse AND ≥1 axis credibly better. Floor breach on any axis → hold
2783
+ * (anti-Goodhart, dominates everything). Insufficient evidence on any axis →
2784
+ * need_more_work. Statistically equivalent → hold (never ship noise).
2785
+ */
2786
+ declare const paretoPolicy: PromotionPolicy;
2787
+ interface ParetoSignificanceGateOptions extends BuildEvidenceVectorOptions {
2788
+ /** The objective vector. Every axis is both a gain source and a safety floor. */
2789
+ objectives: PromotionObjective[];
2790
+ /** Strategy applied to the evidence vector. Default `paretoPolicy`. Override
2791
+ * to run a stricter/looser strategy over the SAME bus (competing policies). */
2792
+ policy?: PromotionPolicy;
2793
+ /** Override the gate name in reports. */
2794
+ name?: string;
2795
+ }
2796
+ /**
2797
+ * Wrap the bus + a policy as a `Gate`. Plugs into the existing
2798
+ * `runImprovementLoop({ gate })` slot and composes via `composeGate`; default
2799
+ * loop behavior is unchanged because consumers opt in by passing this gate.
2800
+ */
2801
+ declare function paretoSignificanceGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(options: ParetoSignificanceGateOptions): Gate<TArtifact, TScenario>;
2802
+
2803
+ /**
2804
+ * OutcomeStore — deployment outcomes attached to Run IDs.
2805
+ *
2806
+ * Outcomes arrive asynchronously from production telemetry after the
2807
+ * eval run completed: user ratings, retention flags, conversion events,
2808
+ * revenue, support-ticket rate, anything a product team can measure.
2809
+ * The store is a peer to TraceStore — separate lifecycle, same runId
2810
+ * foreign key.
2811
+ *
2812
+ * The whole point of this module is to make the meta-eval correlation
2813
+ * question computable: `correlate(evalMetric, outcomeMetric) → r, ρ, n, CI`.
2814
+ */
2815
+ interface DeploymentOutcome {
2816
+ runId: string;
2817
+ capturedAt: number;
2818
+ /** Numeric outcomes keyed by name — retention_7d, csat, revenue_usd, etc. */
2819
+ metrics: Record<string, number>;
2820
+ /** Dimensions for stratified analysis — cohort, region, user_segment. */
2821
+ labels?: Record<string, string>;
2822
+ /** Free-form provenance (source system, pipeline version). */
2823
+ source?: string;
2824
+ }
2825
+ interface OutcomeFilter {
2826
+ runIds?: string[];
2827
+ since?: number;
2828
+ until?: number;
2829
+ label?: {
2830
+ key: string;
2831
+ value: string;
2832
+ };
2833
+ source?: string;
2834
+ }
2835
+ interface OutcomeStore {
2836
+ append(outcome: DeploymentOutcome): Promise<void>;
2837
+ /** All outcomes attached to this run (a single run can have many — multiple
2838
+ * capture windows over deployment time). */
2839
+ forRun(runId: string): Promise<DeploymentOutcome[]>;
2840
+ list(filter?: OutcomeFilter): Promise<DeploymentOutcome[]>;
2841
+ }
2842
+ declare class InMemoryOutcomeStore implements OutcomeStore {
2843
+ private items;
2844
+ append(outcome: DeploymentOutcome): Promise<void>;
2845
+ forRun(runId: string): Promise<DeploymentOutcome[]>;
2846
+ list(filter?: OutcomeFilter): Promise<DeploymentOutcome[]>;
2847
+ }
2848
+ interface FileSystemOutcomeStoreOptions {
2849
+ dir: string;
2850
+ maxBytes?: number;
2851
+ }
2852
+ declare class FileSystemOutcomeStore implements OutcomeStore {
2853
+ private dir;
2854
+ private maxBytes;
2855
+ private memo?;
2856
+ private loaded;
2857
+ constructor(options: FileSystemOutcomeStoreOptions);
2858
+ private ensureDir;
2859
+ append(outcome: DeploymentOutcome): Promise<void>;
2860
+ private load;
2861
+ forRun(runId: string): Promise<DeploymentOutcome[]>;
2862
+ list(filter?: OutcomeFilter): Promise<DeploymentOutcome[]>;
2863
+ }
2864
+
2865
+ /**
2866
+ * Reporting helpers — production summaries and paper-quality figures — sit alongside `reporter.ts` rather
2867
+ * than replacing it.
2868
+ *
2869
+ * Three artefacts:
2870
+ *
2871
+ * - `summaryTable` Markdown table of per-candidate means,
2872
+ * 95% bootstrap CIs, BH-adjusted Wilcoxon
2873
+ * p-values, and Cohen's d versus a
2874
+ * comparator candidate.
2875
+ * - `paretoChart` Abstract spec for a cost vs quality
2876
+ * scatter, with gate decisions overlaid.
2877
+ * Returns numbers + labels — caller
2878
+ * chooses the plotting library.
2879
+ * - `gainHistogram`
2880
+ * Per-item paired holdout deltas as a
2881
+ * histogram spec (bins + counts + median +
2882
+ * CI). Same "data, not images" contract.
2883
+ *
2884
+ * The figure types are PlotSpecs — JSON-friendly, library-agnostic.
2885
+ * They aren't React components and they aren't PNGs; they are
2886
+ * what you'd hand to vega-lite, plotly, matplotlib, or your own
2887
+ * Canvas renderer to draw the actual figure.
2888
+ */
2889
+
2890
+ interface ParetoPoint {
2891
+ candidateId: string;
2892
+ /** Mean USD cost per run on the chosen split. */
2893
+ cost: number;
2894
+ /** Mean score on the chosen split. */
2895
+ quality: number;
2896
+ /** Number of runs that informed this point. */
2897
+ n: number;
2898
+ /** Whether this candidate is on the Pareto frontier — high
2899
+ * quality, low cost, no dominator. */
2900
+ onFrontier: boolean;
2901
+ /** Optional gate verdict for this candidate, if a `GateDecision`
2902
+ * for it was passed in. */
2903
+ gate?: 'promote' | 'reject_few_runs' | 'reject_negative_delta' | 'reject_overfit_gap' | null;
2904
+ }
2905
+ interface ParetoFigureSpec {
2906
+ kind: 'pareto-cost-quality';
2907
+ split: 'search' | 'holdout';
2908
+ points: ParetoPoint[];
2909
+ axes: {
2910
+ x: 'costUsd';
2911
+ y: 'score';
2912
+ };
2913
+ }
2914
+ interface GainDistributionBin {
2915
+ /** Inclusive lower edge. */
2916
+ lo: number;
2917
+ /** Exclusive upper edge (or inclusive if it's the last bin). */
2918
+ hi: number;
2919
+ /** Number of pairs whose delta lands in this bin. */
2920
+ count: number;
2921
+ }
2922
+
2923
+ /**
2924
+ * # InsightReport — the rigorous decision packet for any set of agent runs.
2925
+ *
2926
+ * Returned by `analyzeRuns()` and embedded in `SelfImproveResult.insight` +
2927
+ * the hosted-tier `EvalRunEvent.insightReport`. One shape across two surfaces:
2928
+ *
2929
+ * - **Customer who has a closed loop** (`selfImprove`): the report ships
2930
+ * with the loop output. Their dashboard renders ship/hold + lift CI +
2931
+ * calibration + cluster + Pareto in one packet.
2932
+ * - **Customer who has observed runs but no loop** (`analyzeRuns` directly):
2933
+ * same packet from a `RunRecord[]` they already have — production traces,
2934
+ * approve/reject corpus, CSV gold set.
2935
+ *
2936
+ * Every field is optional except the distributional summary — fields are
2937
+ * populated when the input data supports them:
2938
+ *
2939
+ * - `lift` requires both baseline and candidate splits to be present.
2940
+ * - `interRater` requires multi-rater feedback (≥2 raters per run).
2941
+ * - `judges` populates per-judge stats only when the run records carry
2942
+ * `outcome.judgeScores`.
2943
+ * - `failureClusters` requires the optional `analystRegistry` to be wired.
2944
+ * - `contamination` requires canary scenarios to be passed in.
2945
+ * - `outcomeCorrelation` requires a downstream outcome signal.
2946
+ * - `sequential` requires the run set to be ordered (treats them as a
2947
+ * stream and emits an anytime-valid interim decision).
2948
+ *
2949
+ * Consumers read the `recommendations` array first — that's the
2950
+ * actionable layer, ranked by priority. The numeric sections back it up.
2951
+ */
2952
+
2953
+ interface InsightReport {
2954
+ /** Number of runs analyzed. */
2955
+ n: number;
2956
+ /** Runtime facts carried by the run records. These describe execution,
2957
+ * not task quality: duration, queueing, token categories, models, and
2958
+ * explicitly recorded failures. */
2959
+ execution: ExecutionInsight;
2960
+ /** Composite-score distribution across all runs. Always present. */
2961
+ composite: ScalarDistribution;
2962
+ /** Per-dimension distributions for every dimension that appeared in any
2963
+ * run's judge scores. Empty when no judge scores were recorded. */
2964
+ perDimension: Record<string, ScalarDistribution>;
2965
+ /** Cost/quality distribution and Pareto frontier. */
2966
+ costQuality: {
2967
+ cost: ScalarDistribution;
2968
+ pareto: ParetoFigureSpec;
2969
+ /** Cost source coverage. `uncaptured` rows are excluded from the USD
2970
+ * distribution and Pareto chart; observed and estimated totals remain
2971
+ * separate so reports never present estimates as billed spend. */
2972
+ provenance?: CostProvenanceSummary;
2973
+ /** Set when the cost/quality view is degraded because the input data
2974
+ * doesn't fully support it — e.g. all `costUsd` were zero, or only a
2975
+ * single candidate appears (so the Pareto is a single point). The
2976
+ * named fields name the degraded sub-view, free-text the reason. */
2977
+ degraded?: {
2978
+ cost?: string;
2979
+ pareto?: string;
2980
+ };
2981
+ };
2982
+ /** Per-judge calibration + bias detection. Populated for every judge name
2983
+ * that appears in `outcome.judgeScores`. Bias fields require either a
2984
+ * gold reference or multi-rater data. */
2985
+ judges: Record<string, JudgeInsight>;
2986
+ /** Inter-rater agreement when multiple judges scored the same runs.
2987
+ * Includes pairwise kappa and the specific run ids where raters
2988
+ * disagree — the cases worth a human meeting. */
2989
+ interRater?: InterRaterInsight;
2990
+ /** Pairwise lift (baseline → candidate) with bootstrap CI. Present when
2991
+ * `RunRecord.splitTag` includes both `holdout` and search/dev splits,
2992
+ * or when caller passes an explicit baseline/candidate split. */
2993
+ lift?: LiftInsight;
2994
+ /** Failure clusters with exemplars. Populated when an AnalystRegistry
2995
+ * is wired in `analyzeRuns({ analyst })`. */
2996
+ failureClusters?: FailureClusterInsight;
2997
+ /** Canary leak count + holdout audit status. Populated when canary
2998
+ * scenarios are passed in. */
2999
+ contamination?: ContaminationInsight;
3000
+ /** Correlation between judge composite and a downstream outcome the
3001
+ * caller supplies (engagement, revenue, downstream pass rate, etc.).
3002
+ * When present, the optional reward model is the model that maps
3003
+ * judge scores → predicted outcome. */
3004
+ outcomeCorrelation?: OutcomeCorrelationInsight;
3005
+ /** Aggregate release-readiness summary. A consumer needing the full
3006
+ * substrate `ReleaseConfidenceScorecard` (SLO-axis evaluation,
3007
+ * ActionableSideInfo bag) calls `evaluateReleaseConfidence()` directly;
3008
+ * this summary captures the analyzeRuns-derived axes. */
3009
+ release: ReleaseSummary;
3010
+ /** Delta vs a prior period when `baselineRuns` is passed. Per-metric
3011
+ * current vs baseline with Welch CI + Cohen's d + significance flag.
3012
+ * Answers "did my last change help?" — the customer-conversion question.
3013
+ * Surfaced metrics: composite, cost, duration, tokenUsage, plus any
3014
+ * per-dimension judge metric present in both windows. */
3015
+ priorPeriodComparison?: PriorPeriodComparison;
3016
+ /** Model-free failure-mode breakdown from `RunRecord.failureMode`, ranked
3017
+ * by count descending. Present when any run carries a `failureMode`.
3018
+ * Complements `failureClusters` (LLM-semantic) with the structured tags
3019
+ * the harness already recorded — actionable with no analyst wired. */
3020
+ failureModes?: FailureModeTally[];
3021
+ /** Top-N actionable recommendations, ranked by priority. The packet's
3022
+ * human-readable layer; the numeric sections are the evidence. */
3023
+ recommendations: Recommendation[];
3024
+ }
3025
+ interface CostProvenanceSummary {
3026
+ observed: {
3027
+ n: number;
3028
+ totalUsd: number;
3029
+ };
3030
+ estimated: {
3031
+ n: number;
3032
+ totalUsd: number;
3033
+ };
3034
+ uncaptured: {
3035
+ n: number;
3036
+ };
3037
+ knownFraction: number;
3038
+ }
3039
+ interface ExecutionInsight {
3040
+ /** End-to-end wall time for every run. */
3041
+ durationMs: ScalarDistribution;
3042
+ /** Queue time for the subset of runs that recorded it. */
3043
+ queueMs: ScalarDistribution;
3044
+ /** Token distributions plus corpus totals. Optional token categories use
3045
+ * distribution `n` to disclose how many runs recorded that category. */
3046
+ tokenUsage: TokenUsageInsight;
3047
+ /** Usage reported only by orchestration or agent aggregate spans.
3048
+ * Kept separate because it may duplicate model-call telemetry in other traces. */
3049
+ aggregateUsage: {
3050
+ runs: number;
3051
+ tokenUsage: TokenUsageInsight;
3052
+ costUsd: ScalarDistribution;
3053
+ totalCostUsd: number;
3054
+ };
3055
+ /** Stable model counts, largest cohort first. */
3056
+ models: Array<{
3057
+ model: string;
3058
+ runs: number;
3059
+ }>;
3060
+ /** Model-call coverage. `events` is available only from producers that
3061
+ * record `outcome.raw.llm_span_count`; `runs` also recognizes non-zero
3062
+ * token usage from other producers. */
3063
+ modelCalls: {
3064
+ runs: number;
3065
+ events: number;
3066
+ reportingRuns: number;
3067
+ };
3068
+ /** Failure counts remain separate from outcome scores. `reportedErrorEvents`
3069
+ * sums `outcome.raw.error_span_count` only where a producer supplied it. */
3070
+ failures: {
3071
+ runs: number;
3072
+ fraction: number;
3073
+ reportedErrorEvents: number;
3074
+ reportingRuns: number;
3075
+ };
3076
+ }
3077
+ interface TokenUsageInsight {
3078
+ input: ScalarDistribution;
3079
+ output: ScalarDistribution;
3080
+ reasoning: ScalarDistribution;
3081
+ cached: ScalarDistribution;
3082
+ cacheWrite: ScalarDistribution;
3083
+ totals: {
3084
+ input: number;
3085
+ output: number;
3086
+ reasoning: number;
3087
+ cached: number;
3088
+ cacheWrite: number;
3089
+ };
3090
+ }
3091
+ /** Distributional summary of a scalar-valued metric. */
3092
+ interface ScalarDistribution {
3093
+ /** Sample count after dropping non-finite values. */
3094
+ n: number;
3095
+ mean: number;
3096
+ p50: number;
3097
+ p95: number;
3098
+ stddev: number;
3099
+ min: number;
3100
+ max: number;
3101
+ /** Histogram bins using `agent-eval`'s `gainHistogram` primitive. */
3102
+ histogram: GainDistributionBin[];
3103
+ /** Worst-N runs by score, ascending. Populated for the composite
3104
+ * distribution so the report names the runs a customer should
3105
+ * inspect first. Undefined when the distribution was computed from a
3106
+ * raw value list with no run identity (e.g. cost). */
3107
+ tailRuns?: Array<{
3108
+ runId: string;
3109
+ score: number;
3110
+ }>;
3111
+ }
3112
+ interface JudgeInsight {
3113
+ /** Number of times this judge scored a run. */
3114
+ n: number;
3115
+ /** Mean composite over this judge's runs. */
3116
+ meanScore: number;
3117
+ /** Calibration against a gold reference, when provided. Cohen's κ for
3118
+ * binary thresholding + continuous agreement metrics. */
3119
+ calibration?: ContinuousAgreement;
3120
+ /** Positional bias — when the judge sees options in different orders,
3121
+ * do its preferences track the content or the position? */
3122
+ positionalBias?: number;
3123
+ /** Self-preference — when the judge sees its own model's output vs a
3124
+ * competitor, does it over-pick its own? */
3125
+ selfPreference?: number;
3126
+ /** Verbosity bias — does the judge reward longer outputs regardless of
3127
+ * quality? */
3128
+ verbosityBias?: number;
3129
+ }
3130
+ interface InterRaterInsight {
3131
+ /** Number of raters whose scores were aggregated. */
3132
+ raters: number;
3133
+ /** Number of runs every rater scored. */
3134
+ jointlyRated: number;
3135
+ /** Cohen's κ averaged across rater pairs. */
3136
+ kappa: number;
3137
+ /** Pairwise κ per rater pair (key = `"raterA::raterB"`). */
3138
+ perPair: Record<string, number>;
3139
+ /** Run ids where raters disagree the most — the high-value triage list. */
3140
+ disagreementCases: Array<{
3141
+ runId: string;
3142
+ ratings: Array<{
3143
+ rater: string;
3144
+ score: number;
3145
+ }>;
3146
+ range: number;
3147
+ }>;
3148
+ }
3149
+ interface LiftInsight {
3150
+ baselineMean: number;
3151
+ candidateMean: number;
3152
+ /** Candidate − baseline. */
3153
+ delta: number;
3154
+ /** Lower / upper bound of bootstrap CI on the delta. */
3155
+ ci95: [number, number];
3156
+ /** Paired-t-test p-value. */
3157
+ pValue: number;
3158
+ /** Number of paired observations. */
3159
+ n: number;
3160
+ /** Cohen's d for the delta. */
3161
+ cohensD: number;
3162
+ /** Minimum detectable effect at current n, 80% power. */
3163
+ mde: number;
3164
+ /** Sample size needed to detect the observed delta at 80% power. */
3165
+ requiredN: number;
3166
+ }
3167
+ interface FailureClusterInsight {
3168
+ /** All clusters identified by the registry, ranked by share descending. */
3169
+ clusters: Array<{
3170
+ id: string;
3171
+ name: string;
3172
+ /** Fraction of failed runs in this cluster, 0..1. */
3173
+ share: number;
3174
+ /** Exemplar `runId`s (≤ 5) the consumer can drill into. */
3175
+ exemplars: string[];
3176
+ /** Short LLM-generated suggested fix when the registry supports it. */
3177
+ suggestedFix?: string;
3178
+ }>;
3179
+ totalFailures: number;
3180
+ }
3181
+ /** Model-free failure breakdown over the structured `RunRecord.failureMode`
3182
+ * enum. Unlike `failureClusters` (semantic, requires an LLM analyst), this
3183
+ * is computed directly from the tags the harness already recorded — so a
3184
+ * customer ingesting one batch with no judge/analyst still learns which
3185
+ * named failure dominates. */
3186
+ interface FailureModeTally {
3187
+ /** The `failureMode` tag. */
3188
+ mode: string;
3189
+ /** Number of runs carrying this tag. */
3190
+ count: number;
3191
+ /** Share of the whole corpus, 0..1. */
3192
+ share: number;
3193
+ }
3194
+ interface ContaminationInsight {
3195
+ /** Canary phrases that leaked into outputs. */
3196
+ leaks: number;
3197
+ /** Holdout audit verdict — did any holdout-tagged run end up in the
3198
+ * search/dev pool, or vice versa? */
3199
+ holdoutAuditPassed: boolean;
3200
+ details?: Array<{
3201
+ runId: string;
3202
+ canary: string;
3203
+ matched: string;
3204
+ }>;
3205
+ }
3206
+ interface OutcomeCorrelationInsight {
3207
+ /** What outcome the consumer is correlating against (e.g.
3208
+ * `'engagement_rate'`, `'approval_rate'`, `'downstream_pass'`). */
3209
+ metric: string;
3210
+ /** Number of (run, outcome) pairs used. */
3211
+ n: number;
3212
+ /** Pearson correlation between composite score and outcome. */
3213
+ pearson: number;
3214
+ /** Spearman rank correlation — robust to monotonic non-linearity. */
3215
+ spearman: number;
3216
+ /** When present, the simple linear reward model fit to the data. */
3217
+ rewardModel?: {
3218
+ intercept: number;
3219
+ slope: number;
3220
+ r2: number;
3221
+ };
3222
+ }
3223
+ interface ReleaseSummary {
3224
+ /** Overall verdict across axes — fail if any axis fails, else warn if any
3225
+ * warns, else pass. */
3226
+ status: 'pass' | 'warn' | 'fail';
3227
+ axes: Array<{
3228
+ name: 'quality-lift' | 'contamination' | 'composite-distribution';
3229
+ status: 'pass' | 'warn' | 'fail';
3230
+ detail: string;
3231
+ }>;
3232
+ /** Free-form issues surfaced beyond the standard axes. Empty by default;
3233
+ * consumers can post-process to populate. */
3234
+ issues: string[];
3235
+ }
3236
+ interface MetricDelta {
3237
+ /** Current-period mean. */
3238
+ current: number;
3239
+ /** Baseline-period mean. */
3240
+ baseline: number;
3241
+ /** current - baseline. Positive means improved (or, for cost/duration,
3242
+ * the consumer-side interpretation: "higher current" — semantic
3243
+ * direction depends on the metric). */
3244
+ delta: number;
3245
+ /** Welch 95% confidence interval on the delta. Two-sample, unpaired —
3246
+ * the baseline and current run sets may have different scenarios. */
3247
+ ci95: [number, number];
3248
+ /** Welch t-test p-value (two-sided). */
3249
+ pValue: number;
3250
+ /** Cohen's d (pooled stddev). Effect size, signed. */
3251
+ cohensD: number;
3252
+ /** Sample sizes. */
3253
+ baselineN: number;
3254
+ currentN: number;
3255
+ /** True when p < 0.05 AND |d| >= 0.2 (small-effect threshold). The
3256
+ * conjunction prevents large-effect-but-noisy and significant-but-
3257
+ * tiny from triggering recommendations. */
3258
+ significant: boolean;
3259
+ }
3260
+ interface PriorPeriodComparison {
3261
+ /** Sample counts. */
3262
+ baselineN: number;
3263
+ currentN: number;
3264
+ /** Optional human-readable label — "vs prior 7 days", "vs v3 release". */
3265
+ windowLabel?: string;
3266
+ /** Every metric we could compare. Keys: 'composite', 'cost', 'duration',
3267
+ * 'tokenUsage' for always-present ones; per-dimension keys when both
3268
+ * windows have judge scores on the same dimension. */
3269
+ metrics: Record<string, MetricDelta>;
3270
+ /** Metric names where current is significantly WORSE than baseline.
3271
+ * Direction-aware: for cost/duration, higher current = worse. */
3272
+ regressedMetrics: string[];
3273
+ /** Metric names where current is significantly BETTER than baseline. */
3274
+ improvedMetrics: string[];
3275
+ }
3276
+ interface Recommendation {
3277
+ priority: 'critical' | 'high' | 'medium' | 'low';
3278
+ kind: 'ship' | 'hold' | 'investigate' | 'fix' | 'recalibrate' | 'expand-corpus';
3279
+ title: string;
3280
+ detail: string;
3281
+ /** Optional pointer back into the report for the evidence. */
3282
+ evidencePath?: string;
3283
+ }
3284
+
3285
+ /**
3286
+ * # Hosted-tier wire format — the schema that EVERY orchestrator (ours,
3287
+ * a partner's self-hosted one, a future open implementation) must accept.
3288
+ *
3289
+ * **Stability:** every type in this file is committed under semver. New
3290
+ * minors only ADD optional fields. Breaking changes mean a major bump
3291
+ * (`HostedWireVersion` literal increment).
3292
+ *
3293
+ * The wire format is two event streams in one transport:
3294
+ *
3295
+ * 1. **Eval-run events** (`POST /v1/ingest/eval-runs`). Posted when a
3296
+ * campaign / improvement-loop completes (or per-generation if
3297
+ * streaming). Carries the structured result + per-cell scores +
3298
+ * surface diffs the orchestrator stores for the dashboard.
3299
+ *
3300
+ * 2. **Trace spans** (`POST /v1/ingest/traces`). Standard OTLP-shaped
3301
+ * spans with a few additional attributes so the orchestrator can
3302
+ * pivot from eval-run → underlying execution. Compatible with any
3303
+ * OTel collector.
3304
+ *
3305
+ * Both endpoints are authenticated with a bearer token + a tenant id
3306
+ * header. Tenants isolate everything downstream of ingest; no tenant
3307
+ * ever sees another tenant's data.
3308
+ */
3309
+
3310
+ /** Lifecycle stages of an eval-run as the substrate reports them. */
3311
+ type EvalRunStatus = 'started' | 'baseline-complete' | 'generation-complete' | 'gate-decided' | 'finished' | 'errored';
3312
+ interface EvalRunCellScore {
3313
+ /** Stable scenario id from the consumer's scenario set. */
3314
+ scenarioId: string;
3315
+ /** Repetition index when reps > 1; 0 for the default. */
3316
+ rep: number;
3317
+ /** Composite score across all judges + dimensions for this cell. */
3318
+ compositeMean: number;
3319
+ /** Per-judge → per-dimension scores; null where the judge did not run. */
3320
+ dimensions: Record<string, Record<string, number>>;
3321
+ /** Per-cell error message if the dispatch threw. Null on success. */
3322
+ errorMessage?: string;
3323
+ }
3324
+ interface EvalRunGenerationSnapshot {
3325
+ /** Generation index. 0 is baseline. */
3326
+ index: number;
3327
+ /** Candidate surface fingerprint (stable hash) — pivot key into the
3328
+ * trace stream to fetch the underlying execution. */
3329
+ surfaceHash: string;
3330
+ /** The candidate surface itself. May be omitted to avoid PII when the
3331
+ * consumer prefers not to ship verbatim prompts. */
3332
+ surface?: MutableSurface;
3333
+ /** Per-cell scores for this generation. */
3334
+ cells: EvalRunCellScore[];
3335
+ /** Aggregate composite mean across all cells in this generation. */
3336
+ compositeMean: number;
3337
+ /** Total $ spent across this generation. */
3338
+ costUsd: number;
3339
+ /** Wall-clock duration of this generation. */
3340
+ durationMs: number;
3341
+ }
3342
+ /**
3343
+ * The top-level eval-run event. One ingest call per logical eval-run;
3344
+ * generations stream in incrementally via repeated calls with the same
3345
+ * `runId`. The orchestrator deduplicates by `(runId, generation.index)`.
3346
+ */
3347
+ interface EvalRunEvent {
3348
+ /** Stable run id (the substrate's `runId`). UUID or substrate-generated. */
3349
+ runId: string;
3350
+ /** Where this run was happening — derived from `RunCampaignOptions.runDir`. */
3351
+ runDir: string;
3352
+ /** ISO-8601 timestamp the substrate recorded the event. */
3353
+ timestamp: string;
3354
+ /** Lifecycle stage this event represents. */
3355
+ status: EvalRunStatus;
3356
+ /** Free-form consumer tags (env, branch, model id, etc.). Searchable. */
3357
+ labels: Record<string, string>;
3358
+ /** Baseline campaign snapshot. Present when status >= baseline-complete. */
3359
+ baseline?: EvalRunGenerationSnapshot;
3360
+ /** Per-generation snapshots. Streams in; orchestrator appends. */
3361
+ generations: EvalRunGenerationSnapshot[];
3362
+ /** Final gate decision. Present when status >= gate-decided. */
3363
+ gateDecision?: GateDecision;
3364
+ /** Held-out lift = winner-on-holdout - baseline-on-holdout. */
3365
+ holdoutLift?: number;
3366
+ /** Total $ spent across baseline + every generation. */
3367
+ totalCostUsd: number;
3368
+ /** Total wall-clock duration. */
3369
+ totalDurationMs: number;
3370
+ /** Error message if status === 'errored'. */
3371
+ errorMessage?: string;
3372
+ /** Rigor packet emitted alongside the run — distributional summary,
3373
+ * paired-bootstrap lift CI, judge stats, inter-rater agreement,
3374
+ * contamination check, failure clusters (when an analyst is wired),
3375
+ * outcome correlation (when downstream signal is supplied), and the
3376
+ * recommendations the dashboard surfaces verbatim. Additive; older
3377
+ * clients that don't know about this field continue to work. */
3378
+ insightReport?: InsightReport;
3379
+ }
3380
+ /**
3381
+ * OTel-shape span with a few additional attributes for eval-run pivoting.
3382
+ * Compatible with any OTLP collector — `name`, `traceId`, `spanId`,
3383
+ * `startTimeUnixNano`, `endTimeUnixNano`, `attributes` are stock OTel.
3384
+ */
3385
+ interface TraceSpanEvent {
3386
+ traceId: string;
3387
+ spanId: string;
3388
+ parentSpanId?: string;
3389
+ name: string;
3390
+ startTimeUnixNano: number;
3391
+ endTimeUnixNano: number;
3392
+ attributes: Record<string, string | number | boolean>;
3393
+ events?: Array<{
3394
+ timeUnixNano: number;
3395
+ name: string;
3396
+ attributes?: Record<string, string | number | boolean>;
3397
+ }>;
3398
+ status?: {
3399
+ code: 'OK' | 'ERROR' | 'UNSET';
3400
+ message?: string;
3401
+ };
3402
+ /** Pivot back into the eval-run stream. */
3403
+ 'tangle.runId'?: string;
3404
+ /** Pivot to the specific generation. */
3405
+ 'tangle.generation'?: number;
3406
+ /** Pivot to the specific cell. */
3407
+ 'tangle.cellId'?: string;
3408
+ /** Pivot to the specific scenario. */
3409
+ 'tangle.scenarioId'?: string;
3410
+ }
3411
+
3412
+ /**
3413
+ * # Hosted-tier ingest client.
3414
+ *
3415
+ * Ships eval-run events + trace spans to any orchestrator (ours, a
3416
+ * partner's self-hosted one, or a future open implementation) that
3417
+ * speaks the wire format in `./types.ts`.
3418
+ *
3419
+ * Three modes:
3420
+ * - **Ours:** point at `https://orchestrator.tangle.tools` (the host root —
3421
+ * the client appends the versioned `/v1/ingest/...` path itself; a trailing
3422
+ * `/v1` on the endpoint is tolerated and normalized away). We handle ingest
3423
+ * + storage + dashboard.
3424
+ * - **Self-hosted:** point at whatever URL runs the reference receiver
3425
+ * from `examples/hosted-ingest-server/`.
3426
+ * - **Off (default):** when `hostedTenant` is unset, nothing is sent.
3427
+ * Everything stays local.
3428
+ */
3429
+
3430
+ interface HostedTenant {
3431
+ /** Orchestrator endpoint base URL (no trailing slash). Required. */
3432
+ endpoint: string;
3433
+ /** Bearer token issued by the orchestrator. Required. */
3434
+ apiKey: string;
3435
+ /** Tenant id — the orchestrator's primary key for this consumer. Required. */
3436
+ tenantId: string;
3437
+ /** Optional `fetch` override (auth wrappers, custom agent, test mocks). */
3438
+ fetchImpl?: typeof fetch;
3439
+ /** Per-call timeout in ms. Default 30s. */
3440
+ timeoutMs?: number;
3441
+ /** Retries on 5xx / network errors. Default 2. */
3442
+ retries?: number;
3443
+ }
3444
+
3445
+ interface PowerPreflight {
3446
+ /** Paired observations the comparison will have. */
3447
+ n: number;
3448
+ /** Baseline per-cell composite standard deviation (the variance the effect must beat). */
3449
+ sd: number;
3450
+ /** Minimum detectable lift: the smallest TRUE effect the gate could ship at this budget. */
3451
+ mde: number;
3452
+ /** Baseline holdout composite mean. */
3453
+ baselineMean: number;
3454
+ /** Headroom to a perfect 1.0 composite (the largest achievable lift on a [0,1] judge). */
3455
+ headroom: number;
3456
+ /** True when even the largest achievable effect (headroom) is below the MDE —
3457
+ * the run is structurally unable to ship regardless of proposal quality.
3458
+ * Only asserted for [0,1]-scaled judges (see `scaleAssumed`). */
3459
+ underpowered: boolean;
3460
+ /** True when composites look [0,1]-scaled; headroom/underpowered are only
3461
+ * meaningful under that convention (0-100 judges get mde/sd/n but no verdict). */
3462
+ scaleAssumed: boolean;
3463
+ deltaThreshold: number;
3464
+ confidence: number;
3465
+ /** Set when the holdout shares the gate's scoring channel: more cells cannot
3466
+ * buy back systematic judge bias — treat the MDE as a lower bound. */
3467
+ sharedChannelCaveat?: string;
3468
+ /** One actionable sentence for humans and logs. */
3469
+ recommendation: string;
3470
+ }
3471
+
3472
+ /**
3473
+ * Loop provenance — the durable, queryable record of WHAT a self-improvement
3474
+ * loop did and WHY, plus the OTel spans that let an OTLP collector pivot from
3475
+ * an eval-run to the underlying candidate→cell→gate→promote chain.
3476
+ *
3477
+ * Two artifacts, one source of truth:
3478
+ *
3479
+ * 1. `LoopProvenanceRecord` — a structured JSON record capturing every
3480
+ * candidate (surfaceHash + label + rationale + structured cause), its measured composite,
3481
+ * the gate decision + reasons + delta, the held-out lift, the explicit
3482
+ * baseline→candidate diff, and BACKEND PROVENANCE (the
3483
+ * `assertRealBackend` verdict + worker call count + model). This is the
3484
+ * ingestable audit artifact: the +lift recomputes from it, the "because
3485
+ * Z" rationale survives in it, and a stub backend is detectable from it.
3486
+ *
3487
+ * 2. `loopProvenanceSpans()` — the same chain emitted as OTLP-ingestable
3488
+ * `TraceSpanEvent`s, pivoted on the substrate's standard
3489
+ * `tangle.runId` / `tangle.scenarioId` / `tangle.cellId` /
3490
+ * `tangle.generation` attributes (the same pivots `/adapters/otel`
3491
+ * reads). The hosted `/v1/ingest/traces` endpoint receives the FULL loop,
3492
+ * not just the `cost.*` spans `runCampaign` already emits per cell.
3493
+ *
3494
+ * The record is built from the substrate's own loop result + the per-call
3495
+ * `RunRecord`s the worker emitted — no new measurement, no recomputation that
3496
+ * could drift from what the gate actually saw.
3497
+ */
3498
+
3499
+ interface LoopProvenanceCandidate {
3500
+ /** Generation index this candidate was proposed in. */
3501
+ generation: number;
3502
+ /** 16-char loop-identity fingerprint (matches `GenerationCandidate.surfaceHash`). */
3503
+ surfaceHash: string;
3504
+ /** Full sha256 content hash — byte-identical-verifiable. */
3505
+ contentHash: string;
3506
+ /** Proposer label, when the proposer returned a `ProposedCandidate`. */
3507
+ label?: string;
3508
+ /** Proposer rationale — the "because Z". When the proposer returned a bare
3509
+ * surface (blind mutator) this is absent. */
3510
+ rationale?: string;
3511
+ /** Exact validated cause when the proposer emitted a structured record. */
3512
+ candidateRecord?: PolicyEditCandidateRecord;
3513
+ /** Exact complete incumbent this candidate mutated. */
3514
+ parentSurfaceHash: string;
3515
+ /** Search-split composite of the exact parent. */
3516
+ parentComposite: number;
3517
+ /** Search-split composite change relative to the exact parent. */
3518
+ observedDeltaFromParent?: number;
3519
+ /** Whether the candidate completed every designed cell and could be selected. */
3520
+ eligibleForPromotion: boolean;
3521
+ /** Designed-denominator receipt retained even for incomplete candidates. */
3522
+ coverage: NonNullable<GenerationCandidate['coverage']>;
3523
+ /** Mean composite this candidate scored on the search split. */
3524
+ composite: number;
3525
+ /** Whether this candidate was promoted out of its generation. */
3526
+ promoted: boolean;
3527
+ }
3528
+ interface LoopProvenanceBackend {
3529
+ /** `assertRealBackend`-grade verdict over the worker call records. */
3530
+ verdict: 'real' | 'mixed' | 'stub';
3531
+ /** Number of worker LLM calls captured (the audit's "worker call count"). */
3532
+ workerCallCount: number;
3533
+ /** Distinct model ids observed across worker calls. */
3534
+ models: string[];
3535
+ totalInputTokens: number;
3536
+ totalOutputTokens: number;
3537
+ totalCostUsd: number;
3538
+ }
3539
+ /**
3540
+ * The durable provenance record. Aligns to the hosted `EvalRunEvent` path but
3541
+ * ADDS the rationale + the explicit baseline→candidate diff (both omitted from
3542
+ * the bare hosted event) + backend provenance.
3543
+ */
3544
+ interface LoopProvenanceRecord {
3545
+ schema: 'tangle.loop-provenance.v3';
3546
+ runId: string;
3547
+ runDir: string;
3548
+ timestamp: string;
3549
+ /** Baseline + winner surface content hashes — distinguishable, byte-verifiable. */
3550
+ baselineContentHash: string;
3551
+ winnerContentHash: string;
3552
+ /** Proposer label/rationale for the promoted change. Absent ⇒ winner == baseline. */
3553
+ winnerLabel?: string;
3554
+ winnerRationale?: string;
3555
+ /** The explicit baseline→winner unified diff the gate decided on. */
3556
+ diff: string;
3557
+ /** Every candidate across every generation, with its rationale and structured cause. */
3558
+ candidates: LoopProvenanceCandidate[];
3559
+ /** Baseline composite on the search split that generated the candidates. */
3560
+ baselineSearchComposite: number;
3561
+ /** The gate verdict — decision + reasons + contributing gates + delta. */
3562
+ gate: {
3563
+ decision: GateDecision;
3564
+ reasons: string[];
3565
+ delta?: number;
3566
+ contributingGates: Array<{
3567
+ name: string;
3568
+ passed: boolean;
3569
+ }>;
3570
+ };
3571
+ /** baseline-on-holdout composite mean. */
3572
+ baselineHoldoutComposite: number;
3573
+ /** winner-on-holdout composite mean. */
3574
+ winnerHoldoutComposite: number;
3575
+ /** winnerHoldout - baselineHoldout — RECOMPUTABLE from this record. */
3576
+ heldOutLift: number;
3577
+ /** Backend provenance: stub-vs-real verdict + worker call count + models. */
3578
+ backend: LoopProvenanceBackend;
3579
+ totalCostUsd: number;
3580
+ totalDurationMs: number;
3581
+ }
37
3582
 
38
3583
  /**
39
3584
  * # `selfImprove()` - the one-call improvement loop.
@@ -392,6 +3937,348 @@ interface DefinedAgentEval<TScenario extends Scenario, TArtifact> {
392
3937
  */
393
3938
  declare function defineAgentEval<TScenario extends Scenario, TArtifact>(defaults: DefineAgentEvalOptions<TScenario, TArtifact>): DefinedAgentEval<TScenario, TArtifact>;
394
3939
 
3940
+ /**
3941
+ * Typed Ax output for analyst findings.
3942
+ *
3943
+ * Replaces the legacy `findings:string[]` pattern (where every bullet
3944
+ * became a flat-severity `AnalystFinding`) with a structured object
3945
+ * array. Ax binds the field as `findings:json[]` so the provider emits
3946
+ * native structured output; at the kind-factory boundary we Zod-validate
3947
+ * each emitted finding so malformed rows fail loud instead of being
3948
+ * silently lifted with default severity.
3949
+ *
3950
+ * Why not `f.object().array()` directly in the signature? The Ax
3951
+ * signature string `question:string -> findings:json[]` already lets
3952
+ * the provider emit JSON arrays. A Zod boundary is required either
3953
+ * way (the provider can return any JSON), and Zod gives us a single
3954
+ * validation surface independent of which Ax version is installed.
3955
+ */
3956
+
3957
+ declare const RawAnalystFindingSchema: z.ZodObject<{
3958
+ severity: z.ZodEnum<{
3959
+ low: "low";
3960
+ high: "high";
3961
+ critical: "critical";
3962
+ medium: "medium";
3963
+ info: "info";
3964
+ }>;
3965
+ claim: z.ZodString;
3966
+ subject: z.ZodOptional<z.ZodString>;
3967
+ evidence_uri: z.ZodString;
3968
+ evidence_excerpt: z.ZodOptional<z.ZodString>;
3969
+ confidence: z.ZodNumber;
3970
+ rationale: z.ZodOptional<z.ZodString>;
3971
+ recommended_action: z.ZodOptional<z.ZodString>;
3972
+ }, z.core.$strict>;
3973
+ type RawAnalystFinding = z.infer<typeof RawAnalystFindingSchema>;
3974
+
3975
+ /**
3976
+ * Analyst-kind factory — the typed way to define trace analysts.
3977
+ *
3978
+ * A "kind" is a specialized analyst whose actor prompt, tool subset,
3979
+ * and Ax recursion config target one failure-mode lens (failure-mode
3980
+ * classification, knowledge gap discovery, knowledge poisoning, recursive
3981
+ * self-improvement, ...). Kinds emit findings in the typed `RawAnalystFinding`
3982
+ * shape via a JSON-array Ax output; the factory validates each row with
3983
+ * Zod and lifts it into `AnalystFinding[]` with no shape guessing.
3984
+ *
3985
+ * Composition rules:
3986
+ * - Each kind owns its actor description. No generic "answer this
3987
+ * question" prompt — the prompt names the failure lens.
3988
+ * - Each kind picks a narrow tool subset from `ANALYST_TOOL_GROUPS`.
3989
+ * A kind that never needs full-trace dumps can drop `viewTrace` /
3990
+ * `viewSpans` and stay cheap.
3991
+ * - Each kind declares its recursion + parallelism budget. Discovery-
3992
+ * heavy kinds (failure-mode) get higher `maxDepth`; lens kinds
3993
+ * (poisoning) usually stay at 0 since they have a tighter brief.
3994
+ *
3995
+ * Optimizer hook: kinds may declare `goldens` — labeled examples used
3996
+ * by `AxMiPRO` / `AxBootstrapFewShot` / `AxGEPA` to fit the actor
3997
+ * description programmatically. Stored on the kind, not the registry,
3998
+ * because the right metric is kind-specific.
3999
+ */
4000
+
4001
+ /**
4002
+ * Per-kind specification. The factory turns this into a regular
4003
+ * `Analyst<TraceAnalysisStore>` ready for `AnalystRegistry.register()`.
4004
+ */
4005
+ interface TraceAnalystKindSpec {
4006
+ /** Stable id. Appears in finding_id, telemetry, and registry exclusions. */
4007
+ id: string;
4008
+ /** One-sentence description shown in `registry.list()`. */
4009
+ description: string;
4010
+ /** Coarse classification stamped on every emitted finding (`failure-mode`, `knowledge-gap`, ...). */
4011
+ area: string;
4012
+ /** Bump on any breaking change to the actor prompt or output schema. */
4013
+ version: string;
4014
+ /** Actor system prompt. Must instruct the LLM to emit `findings` per the schema. */
4015
+ actorDescription: string;
4016
+ /** Responder system prompt; falls back to a minimal "format the findings" instruction. */
4017
+ responderDescription?: string;
4018
+ /** Tool functions the actor may call. Pick narrow subsets via `ANALYST_TOOL_GROUPS`. */
4019
+ buildTools: (store: TraceAnalysisStore) => AxFunction[];
4020
+ /** Recursion budget. `maxDepth: 0` disables subagents. */
4021
+ recursion?: {
4022
+ maxDepth: number;
4023
+ maxParallelSubagents?: number;
4024
+ };
4025
+ /** Actor turn cap. Default 12. */
4026
+ maxTurns?: number;
4027
+ /** Runtime char cap. Default 6000. */
4028
+ maxRuntimeChars?: number;
4029
+ /** Cost classification surfaced in `registry.list()` and budget enforcement. */
4030
+ cost: AnalystCost;
4031
+ /** Per-finding-row hook — kinds may reject / rewrite before lifting. */
4032
+ postProcess?: (row: RawAnalystFinding, ctx: AnalystContext) => RawAnalystFinding | null;
4033
+ /** Optional optimizer hook — populated when a kind wants to fit its prompt against labeled examples. */
4034
+ goldens?: TraceAnalystGolden[];
4035
+ }
4036
+ /**
4037
+ * One labeled example consumed by Ax optimizers (MIPRO / GEPA / Bootstrap).
4038
+ * Each input is the same `{question}` an analyst would receive; `expected`
4039
+ * is the ground-truth finding set a fitted prompt should produce on this
4040
+ * input. Metric: kind-specific (default: F1 on `finding_id` overlap).
4041
+ */
4042
+ interface TraceAnalystGolden {
4043
+ question: string;
4044
+ expected: ReadonlyArray<Omit<RawAnalystFinding, 'confidence'>>;
4045
+ }
4046
+
4047
+ /**
4048
+ * AnalystRegistry — orchestrate N analysts against one run.
4049
+ *
4050
+ * Owns three responsibilities and only three:
4051
+ * 1. Registration — ids must be unique; bad registrations fail loudly
4052
+ * at register-time, not run-time.
4053
+ * 2. Routing — each analyst declares its `inputKind`; the registry
4054
+ * picks the matching field from AnalystRunInputs and skips the
4055
+ * analyst with a logged reason if it's missing.
4056
+ * 3. Isolation — one analyst's exception MUST NOT stop other analysts.
4057
+ * Failed analysts produce zero findings + a 'failed' summary row.
4058
+ *
4059
+ * Cross-cutting concerns (telemetry, error → finding conversion, cost
4060
+ * ingestion, storage rotation) live in `AnalystHooks`. Budget shaping
4061
+ * (equal split vs weighted vs custom) lives in `BudgetPolicy`. Both
4062
+ * have sensible defaults; consumers override only what they need.
4063
+ */
4064
+
4065
+ interface AnalystHooks {
4066
+ /** Before analyze() — last chance to mutate ctx (e.g. inject tags, override budget). */
4067
+ onBeforeAnalyze?(args: {
4068
+ analyst: Analyst;
4069
+ ctx: AnalystContext;
4070
+ runId: string;
4071
+ }): void | Promise<void>;
4072
+ /** After every analyst (ok | failed | skipped). Use for telemetry, ingestion, rotation. */
4073
+ onAfterAnalyze?(args: {
4074
+ analyst: Analyst;
4075
+ summary: AnalystRunSummary;
4076
+ findings: AnalystFinding[];
4077
+ runId: string;
4078
+ }): void | Promise<void>;
4079
+ /**
4080
+ * On analyst exception. Hook MAY return findings to convert the
4081
+ * error into structured findings; the summary still reports 'failed'.
4082
+ * Return void to keep the default empty-findings behavior.
4083
+ */
4084
+ onError?(args: {
4085
+ analyst: Analyst;
4086
+ error: Error;
4087
+ runId: string;
4088
+ }): AnalystFinding[] | undefined | Promise<AnalystFinding[] | undefined>;
4089
+ /** Once after registry.run() completes. Use for final aggregation, persistence. */
4090
+ onComplete?(args: {
4091
+ result: AnalystRunResult;
4092
+ }): void | Promise<void>;
4093
+ }
4094
+ interface BudgetPolicy {
4095
+ /** Overall USD cap across the registry.run(). */
4096
+ totalUsd?: number;
4097
+ /** Per-analyst weight for the default allocator. Missing ids get weight 1. */
4098
+ weights?: Record<string, number>;
4099
+ /**
4100
+ * Custom allocator — receives the analyst, remaining/total budget, and
4101
+ * the count of analysts that will run. Returns the per-analyst budget
4102
+ * (or undefined to leave it uncapped). Overrides weights when set.
4103
+ */
4104
+ allocate?: (args: {
4105
+ analyst: Analyst;
4106
+ totalUsd: number | undefined;
4107
+ remainingUsd: number | undefined;
4108
+ runningCount: number;
4109
+ }) => number | undefined;
4110
+ }
4111
+ interface AnalystRegistryOptions {
4112
+ /** Shared chat client passed to every LLM analyst via AnalystContext. */
4113
+ chat?: ChatClient;
4114
+ /** Logger callback. Defaults to a no-op. */
4115
+ log?: (msg: string, fields?: Record<string, unknown>) => void;
4116
+ /** Hooks invoked around analyze() — observability + customization seam. */
4117
+ hooks?: AnalystHooks;
4118
+ /** Default budget when run() doesn't override. */
4119
+ defaultBudget?: BudgetPolicy;
4120
+ }
4121
+ interface RegistryRunOpts {
4122
+ /** Restrict to a subset of registered analysts by id. */
4123
+ only?: string[];
4124
+ /** Skip these analysts even if registered. Useful for cheap iteration. */
4125
+ skip?: string[];
4126
+ /** Budget policy — totalUsd + optional weights/allocator. Falls back to options.defaultBudget. */
4127
+ budget?: BudgetPolicy;
4128
+ /** Wall-clock cap. Analysts SHOULD honor `ctx.deadlineMs`. */
4129
+ timeoutMs?: number;
4130
+ /** Abort signal — forwarded into every analyst's context. */
4131
+ signal?: AbortSignal;
4132
+ /** Tags echoed into AnalystContext.tags — useful for tracking environment/version in findings. */
4133
+ tags?: Record<string, string>;
4134
+ /**
4135
+ * Prior-run findings made available as retrieval context to every
4136
+ * analyst via `ctx.priorFindings`. The registry forwards the slice
4137
+ * whose `analyst_id` matches each registered analyst so a kind sees
4138
+ * only its own history. Pass `{ '*': findings }` to broadcast to
4139
+ * every analyst (useful for cross-kind chaining where the improvement
4140
+ * analyst consumes upstream failure findings).
4141
+ */
4142
+ priorFindings?: ReadonlyArray<AnalystFinding> | Record<string, ReadonlyArray<AnalystFinding>>;
4143
+ }
4144
+ declare class AnalystRegistry {
4145
+ private readonly analysts;
4146
+ private readonly options;
4147
+ constructor(options?: AnalystRegistryOptions);
4148
+ register(analyst: Analyst): void;
4149
+ list(): ReadonlyArray<{
4150
+ id: string;
4151
+ description: string;
4152
+ version: string;
4153
+ cost: Analyst['cost'];
4154
+ }>;
4155
+ run(runId: string, inputs: AnalystRunInputs, runOpts?: RegistryRunOpts): Promise<AnalystRunResult>;
4156
+ /**
4157
+ * Streaming counterpart to `run()`. Emits `AnalystRunEvent` values
4158
+ * in real time — `run-started`, then per-analyst `skipped` /
4159
+ * `started` / `completed`, then a terminal `run-completed` whose
4160
+ * payload is the full `AnalystRunResult`. UIs use this to render
4161
+ * progress; persistence consumers use `run()` and read the result.
4162
+ *
4163
+ * Hooks (`onBeforeAnalyze` / `onAfterAnalyze` / `onError` /
4164
+ * `onComplete`) fire as before — streaming is additive, not a hook
4165
+ * replacement.
4166
+ */
4167
+ runStream(runId: string, inputs: AnalystRunInputs, runOpts?: RegistryRunOpts): AsyncGenerator<AnalystRunEvent, void, void>;
4168
+ private selectAnalysts;
4169
+ private routeInput;
4170
+ }
4171
+
4172
+ /**
4173
+ * `buildDefaultAnalystRegistry` — the canonical analyst suite, so consumers
4174
+ * stop hand-wiring `new AnalystRegistry()` + per-kind `createTraceAnalystKind`.
4175
+ *
4176
+ * The deterministic `behavioralAnalyst` is ALWAYS registered (it needs no
4177
+ * model and is model-agnostic by construction). The agentic RLM kinds are
4178
+ * registered only when an `ai` service is supplied — so a caller with no LLM
4179
+ * still gets the full behavioral/efficiency diagnosis, and the substrate's
4180
+ * "any model (including no model)" guarantee holds at the suite level.
4181
+ */
4182
+
4183
+ interface DefaultAnalystRegistryOptions {
4184
+ /** Ax service for the agentic RLM kinds. Omit → only the deterministic analyst. */
4185
+ ai?: AxAIService;
4186
+ /** Model for the agentic kinds (falls back to the ai service default). */
4187
+ model?: string;
4188
+ /** Which agentic kinds to register when `ai` is present. Default = the shipped suite. */
4189
+ kinds?: readonly TraceAnalystKindSpec[];
4190
+ /** Set false to omit the deterministic behavioral analyst (default: include). */
4191
+ includeBehavioral?: boolean;
4192
+ /** Forwarded to the AnalystRegistry constructor (signal, tags, priorFindings). */
4193
+ registry?: AnalystRegistryOptions;
4194
+ }
4195
+ declare function buildDefaultAnalystRegistry(opts?: DefaultAnalystRegistryOptions): AnalystRegistry;
4196
+
4197
+ /**
4198
+ * # `analyzeRuns()` — turn a set of agent runs into an actionable decision packet.
4199
+ *
4200
+ * Wires the substrate's statistical, calibration, clustering, Pareto, and
4201
+ * release-confidence primitives into one `InsightReport`. Two top-level
4202
+ * entry points use this function:
4203
+ *
4204
+ * - `selfImprove()` calls it on the campaign output to attach a packet
4205
+ * to every run.
4206
+ * - Consumers with observed `RunRecord[]` (production traces, gold
4207
+ * corpora, approve/reject tables) call it directly via `analyzeRuns()`
4208
+ * for analysis without a closed loop.
4209
+ *
4210
+ * Every section is opt-in based on what the input data supports — the
4211
+ * function never invents signal. If runs carry no judge scores, `judges`
4212
+ * is empty. If there's no baseline/candidate split, `lift` is undefined.
4213
+ * If no `analyst` is wired, `failureClusters` is undefined.
4214
+ *
4215
+ * The `recommendations` array is the human-readable layer; everything
4216
+ * else is the evidence backing each recommendation.
4217
+ */
4218
+
4219
+ interface AnalyzeRunsOptions {
4220
+ /** The runs to analyze. */
4221
+ runs: RunRecord[];
4222
+ /** Which split to score against when reading composite from RunOutcome.
4223
+ * Default: holdout when ANY run has a `holdoutScore`, else search. */
4224
+ split?: 'search' | 'holdout' | 'auto';
4225
+ /** Pairwise analysis configuration. When both `baselineCandidateId` and
4226
+ * `candidateCandidateId` are present, lift is computed on paired
4227
+ * (experimentId, seed) tuples shared between the two sides. */
4228
+ baselineCandidateId?: string;
4229
+ candidateCandidateId?: string;
4230
+ /** Canary scenarios — checked against every run's raw output for
4231
+ * holdout contamination. */
4232
+ canaryScenarios?: DatasetScenario[];
4233
+ /** Analyst registry for failure clustering. When omitted, the
4234
+ * `failureClusters` section is left undefined. */
4235
+ analyst?: AnalystRegistry;
4236
+ /** Downstream outcome metric per run (e.g. engagement rate, approval
4237
+ * rate, downstream pass rate). When present, the report includes
4238
+ * `outcomeCorrelation` + a simple linear reward model fit. */
4239
+ outcomeSignal?: {
4240
+ metric: string;
4241
+ valueByRunId: Record<string, number>;
4242
+ };
4243
+ /** Multi-rater feedback for inter-rater agreement. Each entry is one
4244
+ * rater's score for one run. Two or more raters → kappa + disagreement
4245
+ * triage list. */
4246
+ raterScores?: Array<{
4247
+ runId: string;
4248
+ rater: string;
4249
+ score: number;
4250
+ }>;
4251
+ /** Number of histogram bins for distributional summaries. Default 12. */
4252
+ histogramBins?: number;
4253
+ /** Decision threshold — the smallest composite lift the caller cares
4254
+ * about. Used by the recommendations engine to call ship vs hold.
4255
+ * Default 0.02. */
4256
+ decisionThreshold?: number;
4257
+ /** Optional prior-period runs. When set, the report includes
4258
+ * `priorPeriodComparison` with per-metric Welch-CI deltas and
4259
+ * recommendations fire on statistically significant regressions.
4260
+ * The two windows do NOT have to share scenarios — the comparison
4261
+ * is two-sample unpaired (the substrate's `lift` field uses paired
4262
+ * bootstrap on shared (experimentId, seed) tuples; this is the
4263
+ * shape for "this week vs last week" rather than "candidate vs
4264
+ * baseline within a campaign"). */
4265
+ baselineRuns?: RunRecord[];
4266
+ /** Human-readable label for the baseline window, e.g. "vs prior 7
4267
+ * days", "vs v3.1 release". Surfaces in recommendations + UI. */
4268
+ baselineLabel?: string;
4269
+ }
4270
+ interface SummarizeExecutionOptions {
4271
+ runs: RunRecord[];
4272
+ histogramBins?: number;
4273
+ }
4274
+ interface ExecutionReport {
4275
+ execution: ExecutionInsight;
4276
+ costProvenance: CostProvenanceSummary;
4277
+ }
4278
+ /** Summarize runtime facts without interpreting task quality or promotion readiness. */
4279
+ declare function summarizeExecution(opts: SummarizeExecutionOptions): ExecutionReport;
4280
+ declare function analyzeRuns(opts: AnalyzeRunsOptions): Promise<InsightReport>;
4281
+
395
4282
  /**
396
4283
  * # `intake/run-record-dir` — load a directory or file of `RunRecord`s.
397
4284
  *
@@ -736,6 +4623,92 @@ interface PartitionByAuthoringModelResult {
736
4623
  */
737
4624
  declare function partitionRunsByAuthoringModel(runs: RunRecord[], index: AgentTraceIndex): PartitionByAuthoringModelResult;
738
4625
 
4626
+ type CodeAgentSessionSource = 'codex' | 'claude-code' | 'opencode' | 'kimi-code' | 'pi';
4627
+ interface ParsedCodeAgentJsonl {
4628
+ entries: unknown[];
4629
+ malformedLines: number;
4630
+ }
4631
+ interface CodeAgentSessionMetrics {
4632
+ entries: number;
4633
+ userMessages: number;
4634
+ assistantMessages: number;
4635
+ reasoningItems: number;
4636
+ toolCalls: number;
4637
+ toolOutputs: number;
4638
+ toolErrors: number;
4639
+ patchAttempts: number;
4640
+ patchSuccesses: number;
4641
+ patchFailures: number;
4642
+ turnsStarted: number;
4643
+ turnsCompleted: number;
4644
+ turnsAborted: number;
4645
+ contextCompactions: number;
4646
+ prLinks: number;
4647
+ fileSnapshots: number;
4648
+ graphNodes: number;
4649
+ graphEdges: number;
4650
+ actionCandidates: number;
4651
+ verificationReports: number;
4652
+ completionDecisions: number;
4653
+ reliabilityRows: number;
4654
+ reliabilityLift: number;
4655
+ inputTokens: number;
4656
+ outputTokens: number;
4657
+ reasoningTokens: number;
4658
+ cachedTokens: number;
4659
+ cacheWriteTokens: number;
4660
+ observedCostUsd: number;
4661
+ observedCostCaptured?: boolean;
4662
+ wallMs: number;
4663
+ processScore: number;
4664
+ }
4665
+ interface CodeAgentSessionDiagnostic {
4666
+ source: CodeAgentSessionSource;
4667
+ sessionId: string;
4668
+ sourcePath?: string;
4669
+ entries: number;
4670
+ malformedLines: number;
4671
+ inferredScore: boolean;
4672
+ hasExplicitTerminalSignal: boolean;
4673
+ hasQualityLabel: boolean;
4674
+ hasTokenUsage: boolean;
4675
+ hasCost: boolean;
4676
+ costKind?: RunCostProvenance['kind'];
4677
+ warnings: string[];
4678
+ }
4679
+ interface CodeAgentSessionIntakeResult {
4680
+ runs: RunRecord[];
4681
+ diagnostics: CodeAgentSessionDiagnostic[];
4682
+ metrics: CodeAgentSessionMetrics[];
4683
+ }
4684
+ interface CodeAgentSessionIntakeOptions {
4685
+ entries: unknown[];
4686
+ malformedLines?: number;
4687
+ sourcePath?: string;
4688
+ experimentId?: string;
4689
+ candidateId?: string;
4690
+ seed?: number;
4691
+ splitTag?: RunSplitTag;
4692
+ scenarioId?: string;
4693
+ model?: string;
4694
+ promptHash?: string;
4695
+ configHash?: string;
4696
+ commitSha?: string;
4697
+ score?: number;
4698
+ /** Explicit cost receipt. Use `uncaptured` when the source says dollars
4699
+ * were not captured; the adapter will not relabel its compatibility $0
4700
+ * sentinel as observed. When omitted, source-reported cost wins, then a
4701
+ * token-priced estimate, then uncaptured. */
4702
+ costProvenance?: RunCostProvenance;
4703
+ }
4704
+ declare function parseCodeAgentJsonl(jsonl: string): ParsedCodeAgentJsonl;
4705
+ declare function fromCodexSession(options: CodeAgentSessionIntakeOptions): CodeAgentSessionIntakeResult;
4706
+ declare function fromClaudeCodeSession(options: CodeAgentSessionIntakeOptions): CodeAgentSessionIntakeResult;
4707
+ declare function fromOpenCodeSession(options: CodeAgentSessionIntakeOptions): CodeAgentSessionIntakeResult;
4708
+ declare function fromKimiCodeSession(options: CodeAgentSessionIntakeOptions): CodeAgentSessionIntakeResult;
4709
+ declare function fromPiSession(options: CodeAgentSessionIntakeOptions): CodeAgentSessionIntakeResult;
4710
+ declare const fromPigraphSession: typeof fromPiSession;
4711
+
739
4712
  /**
740
4713
  * # `intake/feedback-table` — multi-rater approve/reject corpus → `RunRecord[]`.
741
4714
  *
@@ -838,7 +4811,8 @@ declare function fromFeedbackTable(opts: FromFeedbackTableOptions): FromFeedback
838
4811
  * - `wallMs` from `endTimeUnixNano - startTimeUnixNano`
839
4812
  * - `model` from `gen_ai.request.model` / `llm.model` / `tangle.model`
840
4813
  * - cost from `cost.usd` / `gen_ai.usage.cost_usd` / `tangle.cost.usd`
841
- * - token usage from `gen_ai.usage.{input,output}_tokens`
4814
+ * - token usage from model-call input, output, cache-read, and cache-write
4815
+ * attributes without double-counting aggregate parent spans
842
4816
  * - `outcome.searchScore` from `tangle.score` / `eval.score` when
843
4817
  * present; `outcome.raw` collects every numeric attribute.
844
4818
  *
@@ -855,4 +4829,4 @@ interface FromOtelSpansOptions {
855
4829
  }
856
4830
  declare function fromOtelSpans(opts: FromOtelSpansOptions): RunRecord[];
857
4831
 
858
- export { type AgentEvalAgent, type AgentEvalEvaluateOptions, type AgentEvalImproveOptions, type AgentTraceContributor, type AgentTraceContributorType, type AgentTraceConversation, type AgentTraceFile, type AgentTraceIndex, type AgentTraceRange, type AgentTraceRecord, AnalyzeRunsOptions, type AuthoringProvenance, CampaignResult, CampaignStorage, type DefineAgentEvalOptions, type DefinedAgentEval, DispatchContext, type EvalCellScoreDelta, type EvalDimensionDelta, type EvalGenerationDiff, type EvalReportingSuiteInput, type EvalReportingSuiteOptions, type EvalReportingSuiteResult, type EvalRunDiff, type FeedbackTableMeta, type FeedbackTableRow, type FromFeedbackTableOptions, type FromFeedbackTableResult, type FromOtelSpansOptions, type FromRunRecordDirOptions, type FromRunRecordDirResult, Gate, GateDecision, HostedTenant, InsightReport, JudgeConfig, MutableSurface, type PartitionByAuthoringModelResult, RunEvalOptions, RunImprovementLoopResult, type RunRecordRejection, Scenario, type SelfImproveBudget, type SelfImproveLlm, type SelfImproveOptions, type SelfImproveProgressEvent, type SelfImproveResult, SelfImproveRunError, SurfaceProposer, defineAgentEval, diffGenerations, diffRunBaselineToWinner, diffRuns, evalReportingSuite, fromFeedbackTable, fromOtelSpans, fromRunRecordDir, parseAgentTrace, partitionRunsByAuthoringModel, selfImprove };
4832
+ export { type AgentEvalAgent, type AgentEvalEvaluateOptions, type AgentEvalImproveOptions, type AgentTraceContributor, type AgentTraceContributorType, type AgentTraceConversation, type AgentTraceFile, type AgentTraceIndex, type AgentTraceRange, type AgentTraceRecord, type AnalystFinding, type AnalyzeRunsOptions, type AuthoringProvenance, type AxisEvidence, type AxisVerdict, type BuildEvidenceVectorOptions, type CampaignAggregates, type CampaignArtifactWriter, type CampaignCellResult, type CampaignCostMeter, type CampaignResult, type CampaignStorage, type CampaignTraceWriter, type ChatClient, type CodeAgentSessionDiagnostic, type CodeAgentSessionIntakeOptions, type CodeAgentSessionIntakeResult, type CodeAgentSessionMetrics, type CodeAgentSessionSource, type CodeSurface, type CostProvenanceSummary, type CreateChatClientOpts, type DefaultAnalystRegistryOptions, type DefaultProductionGateOptions, type DefineAgentEvalOptions, type DefinedAgentEval, type DeploymentOutcome, type DispatchFn as Dispatch, type DispatchContext, type EvalCellScoreDelta, type EvalDimensionDelta, type EvalGenerationDiff, type EvalReportingSuiteInput, type EvalReportingSuiteOptions, type EvalReportingSuiteResult, type EvalRunDiff, type EvidenceVector, type EvolutionaryProposerOptions, type ExecutionInsight, type ExecutionReport, type FailureClusterInsight, type FeedbackTableMeta, type FeedbackTableRow, FileSystemOutcomeStore, type FileSystemOutcomeStoreOptions, type FromFeedbackTableOptions, type FromFeedbackTableResult, type FromOtelSpansOptions, type FromRunRecordDirOptions, type FromRunRecordDirResult, type Gate, type GateContext, type GateDecision, type GateResult, type GenerationCandidate, type GenerationRecord, type GepaProposerOptions, type HeldOutGateOptions, type HostedTenant, InMemoryOutcomeStore, type InsightReport, type InterRaterInsight, type JudgeConfig, type JudgeDimension, type JudgeInsight, type JudgeScore, type LiftInsight, type MutableSurface, type Mutator, type ObjectiveSource, type OptimizationProposer, type OptimizerConfig, type OutcomeCorrelationInsight, type OutcomeStore, type ParetoSignificanceGateOptions, type ParsedCodeAgentJsonl, type PartitionByAuthoringModelResult, type PromotionObjective, type PromotionPolicy, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, type Recommendation, type ReferenceEquivalenceJudgeInput, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceJudgeResult, type ReferenceEquivalenceScenario, type ReleaseSummary, type RunCampaignOptions, type RunEvalOptions, type RunImprovementLoopOptions, type RunImprovementLoopResult, type RunRecordRejection, type ScalarDistribution, type Scenario, type SelfImproveBudget, type SelfImproveLlm, type SelfImproveOptions, type SelfImproveProgressEvent, type SelfImproveResult, SelfImproveRunError, type SessionScript, type SummarizeExecutionOptions, type SurfaceProposer, type TokenUsageInsight, analyzeRuns, buildDefaultAnalystRegistry, buildEvidenceVector, composeGate, createChatClient, createReferenceEquivalenceJudge, defaultProductionGate, defineAgentEval, diffGenerations, diffRunBaselineToWinner, diffRuns, evalReportingSuite, evolutionaryProposer, fromClaudeCodeSession, fromCodexSession, fromFeedbackTable, fromKimiCodeSession, fromOpenCodeSession, fromOtelSpans, fromPiSession, fromPigraphSession, fromRunRecordDir, fsCampaignStorage, gepaProposer, heldOutGate, inMemoryCampaignStorage, paretoPolicy, paretoSignificanceGate, parseAgentTrace, parseCodeAgentJsonl, partitionRunsByAuthoringModel, runCampaign, runEval, runImprovementLoop, runReferenceEquivalenceJudge, selfImprove, summarizeExecution };