millforge 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. millforge/__init__.py +1174 -0
  2. millforge/_forge/LICENSE +21 -0
  3. millforge/_forge/PROVENANCE.json +295 -0
  4. millforge/_forge/UPDATE_POLICY.md +24 -0
  5. millforge/_forge/__init__.py +14 -0
  6. millforge/_forge/adapter.py +2232 -0
  7. millforge/_forge/base_runner.py +121 -0
  8. millforge/_forge/clients/__init__.py +10 -0
  9. millforge/_forge/clients/base.py +200 -0
  10. millforge/_forge/context/__init__.py +23 -0
  11. millforge/_forge/context/manager.py +178 -0
  12. millforge/_forge/context/strategies.py +335 -0
  13. millforge/_forge/core/__init__.py +16 -0
  14. millforge/_forge/core/inference.py +433 -0
  15. millforge/_forge/core/messages.py +119 -0
  16. millforge/_forge/core/runner.py +479 -0
  17. millforge/_forge/core/steps.py +108 -0
  18. millforge/_forge/core/workflow.py +400 -0
  19. millforge/_forge/errors.py +222 -0
  20. millforge/_forge/guardrails/__init__.py +21 -0
  21. millforge/_forge/guardrails/error_tracker.py +71 -0
  22. millforge/_forge/guardrails/guardrails.py +194 -0
  23. millforge/_forge/guardrails/nudge.py +47 -0
  24. millforge/_forge/guardrails/response_validator.py +119 -0
  25. millforge/_forge/guardrails/step_enforcer.py +183 -0
  26. millforge/_forge/prompts/__init__.py +16 -0
  27. millforge/_forge/prompts/nudges.py +95 -0
  28. millforge/_forge/prompts/templates.py +285 -0
  29. millforge/_version.py +3 -0
  30. millforge/artifacts.py +570 -0
  31. millforge/base/__init__.py +97 -0
  32. millforge/base/composition.py +402 -0
  33. millforge/base/context.py +285 -0
  34. millforge/base/harness.py +138 -0
  35. millforge/base/identity.py +465 -0
  36. millforge/base/options.py +34 -0
  37. millforge/base/platform.py +17 -0
  38. millforge/base/prompt.py +317 -0
  39. millforge/base/runner.py +546 -0
  40. millforge/compiled_plan.py +970 -0
  41. millforge/compiler/__init__.py +231 -0
  42. millforge/compiler/artifact_validation.py +257 -0
  43. millforge/compiler/canonicalization.py +169 -0
  44. millforge/compiler/capabilities.py +66 -0
  45. millforge/compiler/catalogs.py +500 -0
  46. millforge/compiler/diagnostics.py +491 -0
  47. millforge/compiler/graph.py +678 -0
  48. millforge/compiler/lowering.py +198 -0
  49. millforge/compiler/output.py +692 -0
  50. millforge/compiler/parsing.py +1424 -0
  51. millforge/compiler/requests.py +1180 -0
  52. millforge/compiler/schema_validation.py +272 -0
  53. millforge/compiler/semantic.py +490 -0
  54. millforge/compiler/service.py +448 -0
  55. millforge/compiler/source.py +375 -0
  56. millforge/compiler/validators.py +184 -0
  57. millforge/connectors/__init__.py +95 -0
  58. millforge/connectors/admission.py +801 -0
  59. millforge/connectors/broker.py +202 -0
  60. millforge/connectors/contracts.py +1159 -0
  61. millforge/connectors/diagnostics.py +189 -0
  62. millforge/connectors/fake.py +66 -0
  63. millforge/connectors/runtime.py +236 -0
  64. millforge/contracts.py +2860 -0
  65. millforge/custom_tools/__init__.py +67 -0
  66. millforge/custom_tools/compiler.py +724 -0
  67. millforge/custom_tools/contracts.py +1093 -0
  68. millforge/custom_tools/diagnostics.py +205 -0
  69. millforge/eval_artifacts.py +952 -0
  70. millforge/eval_boundary.py +2435 -0
  71. millforge/eval_fixtures/__init__.py +1 -0
  72. millforge/eval_fixtures/default_pack/__init__.py +1 -0
  73. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.bug_diagnosis.traceback.v1.json +52 -0
  74. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.direct_edit.import_sort.v1.json +52 -0
  75. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.evidence_discipline.no_source_change.v1.json +51 -0
  76. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.false_closure.visible_green.v1.json +52 -0
  77. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.multi_file.api_contract.v1.json +54 -0
  78. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.recovery.malformed_artifact.v1.json +54 -0
  79. millforge/eval_fixtures/default_pack/manifest.json +12 -0
  80. millforge/eval_modes.py +1282 -0
  81. millforge/eval_presets.py +1398 -0
  82. millforge/eval_reports.py +2517 -0
  83. millforge/eval_suite.py +2429 -0
  84. millforge/eval_trials.py +2632 -0
  85. millforge/eval_workflow.py +794 -0
  86. millforge/exceptions.py +122 -0
  87. millforge/model_backend.py +2098 -0
  88. millforge/protocols.py +340 -0
  89. millforge/py.typed +0 -0
  90. millforge/runtime.py +1791 -0
  91. millforge/testing/__init__.py +1089 -0
  92. millforge/tools/__init__.py +83 -0
  93. millforge/tools/builtin_runtime.py +1339 -0
  94. millforge/tools/builtins.py +773 -0
  95. millforge/tools/execution.py +1545 -0
  96. millforge/tools/path_policy.py +155 -0
  97. millforge/tools/pi_compat/PI_LICENSE +21 -0
  98. millforge/tools/pi_compat/PROVENANCE.json +55 -0
  99. millforge/tools/pi_compat/UPDATE_POLICY.md +36 -0
  100. millforge/tools/pi_compat/__init__.py +34 -0
  101. millforge/tools/pi_compat/contracts.py +49 -0
  102. millforge/tools/pi_compat/editing.py +390 -0
  103. millforge/tools/pi_compat/mutations.py +57 -0
  104. millforge/tools/pi_compat/operations.py +401 -0
  105. millforge/tools/pi_compat/paths.py +155 -0
  106. millforge/tools/pi_compat/process.py +1375 -0
  107. millforge/tools/pi_compat/search.py +738 -0
  108. millforge/tools/pi_compat/truncation.py +267 -0
  109. millforge/tools/pi_compat_catalog.py +396 -0
  110. millforge/tools/pi_compat_runtime.py +460 -0
  111. millforge/tools/registry.py +553 -0
  112. millforge/tools/results.py +533 -0
  113. millforge-0.1.0.dist-info/METADATA +844 -0
  114. millforge-0.1.0.dist-info/RECORD +116 -0
  115. millforge-0.1.0.dist-info/WHEEL +4 -0
  116. millforge-0.1.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,2517 @@
1
+ """Public 08C eval-report, budget, and live-admission contracts."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import json
7
+ import re
8
+ from collections import Counter
9
+ from collections.abc import Mapping, Sequence
10
+ from enum import Enum
11
+ from statistics import median
12
+ from typing import Any
13
+
14
+ from pydantic import BaseModel, ConfigDict, Field, StrictBool, StrictFloat, StrictInt
15
+ from pydantic import StrictStr, field_validator, model_validator
16
+
17
+ from millforge.eval_suite import (
18
+ EvalCampaignManifest,
19
+ EvalFailureTaxonomyLabel,
20
+ EvalSuiteExecutionMode,
21
+ EvalTaskCategory,
22
+ EvalTrialOutcome,
23
+ )
24
+ from millforge.eval_trials import (
25
+ EvalTrialArmId,
26
+ EvalTrialPlan,
27
+ EvalTrialRecord,
28
+ EvalTrialResumeIndex,
29
+ )
30
+
31
+ EVAL_REPORT_SCHEMA_VERSION = 1
32
+ EVAL_REPORT_HASH_KIND = "eval_report_sha256_v1"
33
+ EVAL_REPORT_JSON_HASH_KIND = "eval_report_json_sha256_v1"
34
+ EVAL_REPORT_MARKDOWN_HASH_KIND = "eval_report_markdown_sha256_v1"
35
+
36
+ _SHA256_RE = re.compile(r"^[0-9a-f]{64}$")
37
+ _UTC_TIMESTAMP_RE = re.compile(r"^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}Z$")
38
+ _ENDPOINT_URL = re.compile(r"https?://|localhost(?::|/|$)|127\.0\.0\.1|0\.0\.0\.0")
39
+ _WINDOWS_ABSOLUTE_PATH = re.compile(
40
+ r"(?:^|[\s\"'`([{<])(?:[A-Za-z]:[\\/]|\\\\)[^\s\"'`)\]}>,]*"
41
+ )
42
+ _POSIX_ABSOLUTE_PATH = re.compile(r"(?:^|[\s\"'`([{<])/(?!/)[^\s\"'`)\]}>,]*")
43
+ _USER_HOME_PATH = re.compile(r"(?:^|[\s\"'`([{<])~(?:/|\\)[^\s\"'`)\]}>,]*")
44
+ _CREDENTIAL_VALUE_PATTERNS = (
45
+ re.compile(r"\bsk-(?:live|proj|test)-[A-Za-z0-9_-]{10,}\b"),
46
+ re.compile(r"\b[rs]k_(?:live|test)_[A-Za-z0-9]{16,}\b"),
47
+ re.compile(r"\b(?:AKIA|ASIA)[0-9A-Z]{16}\b"),
48
+ re.compile(r"\bAIza[0-9A-Za-z_-]{20,}\b"),
49
+ re.compile(r"\bgh[opsu]_[A-Za-z0-9_]{20,}\b"),
50
+ )
51
+ _SECRET_FIELD_MARKERS = (
52
+ "api_key",
53
+ "apikey",
54
+ "auth_header",
55
+ "authorization",
56
+ "bearer",
57
+ "client_secret",
58
+ "credential",
59
+ "password",
60
+ "private_key",
61
+ "secret",
62
+ "access_token",
63
+ "auth_token",
64
+ "refresh_token",
65
+ )
66
+ _DENIED_TEXT_TOKENS = (
67
+ "api_key",
68
+ "authorization:",
69
+ "bearer ",
70
+ "credential",
71
+ "password",
72
+ "secret",
73
+ "access token",
74
+ "auth token",
75
+ "endpoint_url",
76
+ "endpoint url",
77
+ "millrace-agents",
78
+ ".millrace",
79
+ "daemon state",
80
+ "private workspace",
81
+ "private runtime",
82
+ "hidden scorer",
83
+ "hidden answer",
84
+ "hidden expected",
85
+ "expected output",
86
+ "scorer_rubric",
87
+ ".claude",
88
+ ".codex",
89
+ )
90
+
91
+
92
+ class _FrozenEvalReportDict(dict[Any, Any]):
93
+ """Dict-shaped immutable mapping that remains serializable by Pydantic."""
94
+
95
+ def __readonly(self, *args: Any, **kwargs: Any) -> None:
96
+ raise TypeError("eval-report mappings are immutable")
97
+
98
+ __setitem__ = __readonly
99
+ __delitem__ = __readonly
100
+ clear = __readonly
101
+ pop = __readonly
102
+ popitem = __readonly # type: ignore[assignment]
103
+ setdefault = __readonly
104
+ update = __readonly
105
+ __ior__ = __readonly # type: ignore[assignment]
106
+
107
+
108
+ class EvalReportContractModel(BaseModel):
109
+ """Closed, frozen base for public eval-report contracts."""
110
+
111
+ model_config = ConfigDict(extra="forbid", frozen=True, hide_input_in_errors=True)
112
+
113
+ @model_validator(mode="before")
114
+ @classmethod
115
+ def _reject_forbidden_payload(cls, data: Any) -> Any:
116
+ _reject_forbidden_material(data)
117
+ return data
118
+
119
+
120
+ class EvalReportPricingClass(str, Enum):
121
+ """Closed pricing classes for campaign budget accounting."""
122
+
123
+ OFFLINE_ZERO_COST = "offline_zero_cost"
124
+ FREE_TIER = "free_tier"
125
+ PROMOTIONAL_FREE_WINDOW = "promotional_free_window"
126
+ PAID_PROVIDER = "paid_provider"
127
+ LOCAL_METERED = "local_metered"
128
+
129
+
130
+ class EvalLiveAdmissionStatus(str, Enum):
131
+ """Live campaign admission states."""
132
+
133
+ ADMITTED = "admitted"
134
+ DENIED = "denied"
135
+
136
+
137
+ class EvalLiveAdmissionDiagnosticCode(str, Enum):
138
+ """Structured fail-closed live-admission diagnostic codes."""
139
+
140
+ PI_RUNTIME_UNAVAILABLE = "pi_runtime_unavailable"
141
+ MILLFORGE_LIVE_HARNESS_UNAVAILABLE = "millforge_live_harness_unavailable"
142
+ SHARED_BACKEND_CONFIGURATION_MISSING = "shared_backend_configuration_missing"
143
+ FIXTURE_WORKSPACE_LIFECYCLE_UNAVAILABLE = "fixture_workspace_lifecycle_unavailable"
144
+ RESOURCE_ENFORCEMENT_UNAVAILABLE = "resource_enforcement_unavailable"
145
+ BUDGET_POLICY_INVALID = "budget_policy_invalid"
146
+ APPEND_ONLY_STORE_SAFETY_UNPROVEN = "append_only_store_safety_unproven"
147
+ DETERMINISTIC_SCORER_UNAVAILABLE = "deterministic_scorer_unavailable"
148
+
149
+
150
+ class EvalBudgetDiagnosticCode(str, Enum):
151
+ """Structured budget validation diagnostic codes."""
152
+
153
+ MISSING_BUDGET_POLICY = "missing_budget_policy"
154
+ MISSING_LIVE_BUDGET_METADATA = "missing_live_budget_metadata"
155
+ MISSING_TOKEN_CEILING = "missing_token_ceiling"
156
+ MISSING_TRIAL_COUNT_CEILING = "missing_trial_count_ceiling"
157
+ INCOMPLETE_PROMOTIONAL_FREE_WINDOW = "incomplete_promotional_free_window"
158
+ UNFAIR_PAIRED_ARM_RATE_LIMIT = "unfair_paired_arm_rate_limit"
159
+ OFFLINE_POLICY_NOT_ZERO_COST = "offline_policy_not_zero_cost"
160
+ OFFLINE_POLICY_UNBOUNDED = "offline_policy_unbounded"
161
+ SPEND_CEILING_EXCEEDED = "spend_ceiling_exceeded"
162
+ PROMPT_TOKEN_CEILING_EXCEEDED = "prompt_token_ceiling_exceeded"
163
+ COMPLETION_TOKEN_CEILING_EXCEEDED = "completion_token_ceiling_exceeded"
164
+ MODEL_CALL_CEILING_EXCEEDED = "model_call_ceiling_exceeded"
165
+ RETRY_CEILING_EXCEEDED = "retry_ceiling_exceeded"
166
+ WALL_CLOCK_CEILING_EXCEEDED = "wall_clock_ceiling_exceeded"
167
+ TRIAL_COUNT_CEILING_EXCEEDED = "trial_count_ceiling_exceeded"
168
+
169
+
170
+ class EvalMetricDenominatorKind(str, Enum):
171
+ """Metric denominator sources pinned in reports."""
172
+
173
+ PLANNED_TRIALS = "planned_trials"
174
+ APPENDED_RECORDS = "appended_records"
175
+ VALID_RECORDS = "valid_records"
176
+ PAIRED_RECORDS = "paired_records"
177
+
178
+
179
+ class EvalReportMetricId(str, Enum):
180
+ """Closed primary and secondary metric IDs."""
181
+
182
+ VALID_COMPLETION = "valid_completion"
183
+ FALSE_CLOSURE = "false_closure"
184
+ FALSE_SUCCESS = "false_success"
185
+ ARTIFACT_COMPLETE = "artifact_complete"
186
+ CAPABILITY_VIOLATION = "capability_violation"
187
+ CORRECTLY_BLOCKED = "correctly_blocked"
188
+ FALSE_BLOCKED = "false_blocked"
189
+ RUNTIME_FAILURE = "runtime_failure"
190
+ PROVIDER_FAILURE = "provider_failure"
191
+ INVALID_TRIAL = "invalid_trial"
192
+ MISSING_PAIR = "missing_pair"
193
+ PENDING_TRIAL = "pending_trial"
194
+ INCOMPLETE_TRIAL = "incomplete_trial"
195
+ MODEL_CALLS = "model_calls"
196
+ PROMPT_TOKENS = "prompt_tokens"
197
+ COMPLETION_TOKENS = "completion_tokens"
198
+ ESTIMATED_COST = "estimated_cost"
199
+ WALL_CLOCK_SECONDS = "wall_clock_seconds"
200
+ RETRIES = "retries"
201
+ ARTIFACT_COUNT = "artifact_count"
202
+ ARTIFACT_BYTES = "artifact_bytes"
203
+ TURNS = "turns"
204
+ INVALID_TOOL_CALLS = "invalid_tool_calls"
205
+ MALFORMED_TOOL_CALLS = "malformed_tool_calls"
206
+ MALFORMED_ARGUMENTS = "malformed_arguments"
207
+ PREREQUISITE_VIOLATIONS = "prerequisite_violations"
208
+ PREMATURE_TERMINALS = "premature_terminals"
209
+ TOOL_RECOVERIES = "tool_recoveries"
210
+ COMPLETION_IMPROVEMENT = "completion_improvement"
211
+ COST_MULTIPLIER = "cost_multiplier"
212
+ LATENCY_MULTIPLIER = "latency_multiplier"
213
+
214
+
215
+ class EvalReportFailureTaxonomyCategory(str, Enum):
216
+ """Closed public report failure taxonomy."""
217
+
218
+ TASK_MISUNDERSTANDING = "task_misunderstanding"
219
+ WRONG_FILE = "wrong_file"
220
+ UNREAD_BEFORE_EDIT = "unread_before_edit"
221
+ INVALID_PATCH = "invalid_patch"
222
+ TEST_NOT_RUN = "test_not_run"
223
+ TEST_MISREAD = "test_misread"
224
+ MISSING_ARTIFACT = "missing_artifact"
225
+ UNSUPPORTED_SUCCESS_CLAIM = "unsupported_success_claim"
226
+ CHECKER_EVIDENCE_FAILURE = "checker_evidence_failure"
227
+ ARBITER_FALSE_CLOSURE = "arbiter_false_closure"
228
+ PREMATURE_TERMINAL = "premature_terminal"
229
+ TOOL_SCHEMA_FAILURE = "tool_schema_failure"
230
+ TOOL_RECOVERY_FAILURE = "tool_recovery_failure"
231
+ CONTEXT_LOSS = "context_loss"
232
+ BUDGET_EXHAUSTION = "budget_exhaustion"
233
+ PROVIDER_FAILURE = "provider_failure"
234
+ RUNNER_FAILURE = "runner_failure"
235
+ CAPABILITY_VIOLATION = "capability_violation"
236
+ INVALID_TRIAL_INFRASTRUCTURE = "invalid_trial_infrastructure"
237
+
238
+
239
+ class EvalReportConfoundId(str, Enum):
240
+ """Closed confounds that must remain visible in pilot reports."""
241
+
242
+ PI_PROMPT_TOOL_BEHAVIOR = "pi_prompt_tool_behavior"
243
+ MILLFORGE_PROMPT_TOOL_BEHAVIOR = "millforge_prompt_tool_behavior"
244
+ HARNESS_BEHAVIOR = "harness_behavior"
245
+ CONTEXT_PACKING = "context_packing"
246
+ PARSER_FALLBACK = "parser_fallback"
247
+ PROVIDER_NONDETERMINISM = "provider_nondeterminism"
248
+ RATE_LIMITING = "rate_limiting"
249
+ CACHED_PROVIDER_RESPONSES = "cached_provider_responses"
250
+ TOKEN_ACCOUNTING_DIFFERENCES = "token_accounting_differences"
251
+ SAMPLING_PARAMETER_MISMATCH = "sampling_parameter_mismatch"
252
+ OFFLINE_FAKE_LIMITATIONS = "offline_fake_limitations"
253
+
254
+
255
+ class EvalDecisionRuleStatus(str, Enum):
256
+ """Decision-rule evaluation states."""
257
+
258
+ PASSED = "passed"
259
+ FAILED = "failed"
260
+ DESCRIPTIVE_ONLY = "descriptive_only"
261
+ NOT_APPLICABLE = "not_applicable"
262
+
263
+
264
+ class EvalDecisionRuleKind(str, Enum):
265
+ """Closed pre-registered decision-rule contract classes."""
266
+
267
+ MAX_FALSE_CLOSURE_RATE = "max_false_closure_rate"
268
+ MIN_COMPLETION_IMPROVEMENT = "min_completion_improvement"
269
+ MAX_COST_MULTIPLIER = "max_cost_multiplier"
270
+ MAX_LATENCY_MULTIPLIER = "max_latency_multiplier"
271
+ ACCEPTABLE_FALSE_BLOCKED_TRADEOFF = "acceptable_false_blocked_tradeoff"
272
+ SEVERITY_ONE_ABORT_THRESHOLD = "severity_one_abort_threshold"
273
+
274
+
275
+ class EvalBudgetDiagnostic(EvalReportContractModel):
276
+ """One fail-closed budget diagnostic."""
277
+
278
+ diagnostic_code: EvalBudgetDiagnosticCode
279
+ rule_id: StrictStr
280
+ summary: StrictStr
281
+
282
+
283
+ class EvalLiveAdmissionDiagnostic(EvalReportContractModel):
284
+ """One structured live-admission diagnostic."""
285
+
286
+ diagnostic_code: EvalLiveAdmissionDiagnosticCode
287
+ rule_id: StrictStr
288
+ summary: StrictStr
289
+
290
+
291
+ class EvalPromotionalFreeWindow(EvalReportContractModel):
292
+ """Complete metadata required for promotional free execution."""
293
+
294
+ window_id: StrictStr
295
+ source_label: StrictStr
296
+ starts_at: StrictStr
297
+ ends_at: StrictStr
298
+ max_free_usd: StrictFloat = Field(ge=0.0)
299
+ terms_summary: StrictStr
300
+
301
+ @model_validator(mode="after")
302
+ def _window_valid(self) -> EvalPromotionalFreeWindow:
303
+ if not _UTC_TIMESTAMP_RE.fullmatch(self.starts_at):
304
+ raise ValueError("promotional window starts_at must be a UTC timestamp")
305
+ if not _UTC_TIMESTAMP_RE.fullmatch(self.ends_at):
306
+ raise ValueError("promotional window ends_at must be a UTC timestamp")
307
+ if self.ends_at <= self.starts_at:
308
+ raise ValueError("promotional window ends_at must follow starts_at")
309
+ return self
310
+
311
+
312
+ class EvalReportRateLimitPolicy(EvalReportContractModel):
313
+ """Public paired-arm rate-limit and retry/backoff budget policy."""
314
+
315
+ request_rate_per_window: StrictInt = Field(gt=0)
316
+ token_rate_per_window: StrictInt = Field(gt=0)
317
+ concurrent_request_limit: StrictInt = Field(gt=0)
318
+ window_seconds: StrictInt = Field(gt=0)
319
+ max_backoff_seconds: StrictInt = Field(ge=0)
320
+ max_retries_per_trial: StrictInt = Field(ge=0)
321
+
322
+
323
+ class EvalReportAbortThresholds(EvalReportContractModel):
324
+ """Severity-one abort thresholds for a campaign."""
325
+
326
+ max_false_closure_rate: StrictFloat = Field(ge=0.0, le=1.0)
327
+ max_capability_violation_rate: StrictFloat = Field(ge=0.0, le=1.0)
328
+ max_invalid_trial_rate: StrictFloat = Field(ge=0.0, le=1.0)
329
+
330
+
331
+ class EvalReportBudgetPolicy(EvalReportContractModel):
332
+ """Bounded campaign budget policy used by admission and reports."""
333
+
334
+ policy_id: StrictStr
335
+ pricing_class: EvalReportPricingClass
336
+ max_spend_usd: StrictFloat | None = Field(default=None, ge=0.0)
337
+ max_prompt_tokens: StrictInt | None = Field(default=None, ge=0)
338
+ max_completion_tokens: StrictInt | None = Field(default=None, ge=0)
339
+ max_model_calls: StrictInt | None = Field(default=None, ge=0)
340
+ max_retries_per_trial: StrictInt | None = Field(default=None, ge=0)
341
+ max_wall_clock_seconds: StrictInt | None = Field(default=None, ge=0)
342
+ max_trials_per_campaign: StrictInt | None = Field(default=None, ge=0)
343
+ promotional_free_window: EvalPromotionalFreeWindow | None = None
344
+ rate_limit_policy_by_arm: Mapping[EvalTrialArmId, EvalReportRateLimitPolicy] = (
345
+ Field(default_factory=dict)
346
+ )
347
+ abort_thresholds: EvalReportAbortThresholds
348
+ summary: StrictStr
349
+
350
+ @model_validator(mode="after")
351
+ def _policy_valid(self) -> EvalReportBudgetPolicy:
352
+ object.__setattr__(
353
+ self,
354
+ "rate_limit_policy_by_arm",
355
+ _freeze_eval_report_mapping(self.rate_limit_policy_by_arm),
356
+ )
357
+ return self
358
+
359
+
360
+ class EvalBudgetUsageEstimate(EvalReportContractModel):
361
+ """Deterministic campaign budget consumption estimate."""
362
+
363
+ estimated_spend_usd: StrictFloat = Field(ge=0.0)
364
+ prompt_tokens: StrictInt = Field(ge=0)
365
+ completion_tokens: StrictInt = Field(ge=0)
366
+ model_calls: StrictInt = Field(ge=0)
367
+ retries_per_trial: StrictInt = Field(ge=0)
368
+ wall_clock_seconds: StrictInt = Field(ge=0)
369
+ trial_count: StrictInt = Field(ge=0)
370
+
371
+
372
+ class EvalBudgetValidationResult(EvalReportContractModel):
373
+ """Fail-closed result for budget policy validation."""
374
+
375
+ valid: StrictBool
376
+ diagnostics: tuple[EvalBudgetDiagnostic, ...] = Field(default_factory=tuple)
377
+
378
+ @model_validator(mode="after")
379
+ def _result_valid(self) -> EvalBudgetValidationResult:
380
+ if self.valid and self.diagnostics:
381
+ raise ValueError("valid budget results must not include diagnostics")
382
+ if not self.valid and not self.diagnostics:
383
+ raise ValueError("invalid budget results require diagnostics")
384
+ return self
385
+
386
+
387
+ class EvalLiveAdmissionResult(EvalReportContractModel):
388
+ """Structured live or offline campaign admission result."""
389
+
390
+ status: EvalLiveAdmissionStatus
391
+ diagnostics: tuple[EvalLiveAdmissionDiagnostic, ...] = Field(default_factory=tuple)
392
+ budget_result: EvalBudgetValidationResult
393
+
394
+ @model_validator(mode="after")
395
+ def _admission_valid(self) -> EvalLiveAdmissionResult:
396
+ if self.status is EvalLiveAdmissionStatus.ADMITTED and self.diagnostics:
397
+ raise ValueError("admitted campaigns must not include denial diagnostics")
398
+ if self.status is EvalLiveAdmissionStatus.DENIED and not self.diagnostics:
399
+ raise ValueError("denied campaigns require diagnostics")
400
+ return self
401
+
402
+
403
+ class EvalMetricDenominator(EvalReportContractModel):
404
+ """Explicit metric denominator evidence."""
405
+
406
+ denominator_kind: EvalMetricDenominatorKind
407
+ count: StrictInt = Field(ge=0)
408
+ summary: StrictStr
409
+
410
+
411
+ class EvalMetricValue(EvalReportContractModel):
412
+ """One metric count/rate, with an optional total value, and denominator."""
413
+
414
+ metric_id: EvalReportMetricId
415
+ count: StrictInt = Field(ge=0)
416
+ denominator: EvalMetricDenominator
417
+ rate: StrictFloat = Field(ge=0.0, le=1.0)
418
+ value: StrictFloat | None = Field(default=None, ge=0.0)
419
+ severity_one: StrictBool = False
420
+
421
+ @model_validator(mode="after")
422
+ def _metric_valid(self) -> EvalMetricValue:
423
+ expected = (
424
+ 0.0 if self.denominator.count == 0 else self.count / self.denominator.count
425
+ )
426
+ if abs(self.rate - expected) > 0.000000001:
427
+ raise ValueError("metric rate must match count and denominator")
428
+ if self.count > self.denominator.count:
429
+ raise ValueError("metric count must not exceed denominator")
430
+ return self
431
+
432
+
433
+ class EvalPairedComparison(EvalReportContractModel):
434
+ """Per-metric paired comparison across the two admitted arms."""
435
+
436
+ metric_id: EvalReportMetricId
437
+ left_arm_id: EvalTrialArmId
438
+ right_arm_id: EvalTrialArmId
439
+ left_count: StrictInt = Field(ge=0)
440
+ right_count: StrictInt = Field(ge=0)
441
+ paired_denominator: StrictInt = Field(ge=0)
442
+ difference: StrictInt
443
+ missing_pair_count: StrictInt = Field(ge=0)
444
+
445
+
446
+ class EvalWilsonScoreInterval(EvalReportContractModel):
447
+ """Wilson score interval emitted only when sample-size rules allow it."""
448
+
449
+ successes: StrictInt = Field(ge=0)
450
+ total: StrictInt = Field(gt=0)
451
+ confidence_level: StrictFloat = Field(gt=0.0, lt=1.0)
452
+ lower: StrictFloat = Field(ge=0.0, le=1.0)
453
+ upper: StrictFloat = Field(ge=0.0, le=1.0)
454
+
455
+ @model_validator(mode="after")
456
+ def _interval_valid(self) -> EvalWilsonScoreInterval:
457
+ if self.successes > self.total:
458
+ raise ValueError("Wilson successes must not exceed total")
459
+ if self.lower > self.upper:
460
+ raise ValueError("Wilson interval lower must not exceed upper")
461
+ return self
462
+
463
+
464
+ class EvalDistributionSummary(EvalReportContractModel):
465
+ """Descriptive distribution summary for cost and latency-style values."""
466
+
467
+ statistic_id: StrictStr
468
+ sample_count: StrictInt = Field(ge=0)
469
+ raw_values: tuple[StrictFloat, ...] = Field(default_factory=tuple)
470
+ median: StrictFloat | None = None
471
+ p90: StrictFloat | None = None
472
+ p95: StrictFloat | None = None
473
+ descriptive_only: StrictBool = True
474
+
475
+ @model_validator(mode="after")
476
+ def _distribution_valid(self) -> EvalDistributionSummary:
477
+ if self.sample_count != len(self.raw_values):
478
+ raise ValueError("distribution sample_count must match raw_values")
479
+ if self.sample_count == 0 and any(
480
+ value is not None for value in (self.median, self.p90, self.p95)
481
+ ):
482
+ raise ValueError("empty distributions must not include percentiles")
483
+ return self
484
+
485
+
486
+ class EvalReportStatisticalSummary(EvalReportContractModel):
487
+ """Small-N descriptive statistics and eligibility diagnostics."""
488
+
489
+ metric_id: EvalReportMetricId
490
+ raw_count: StrictInt = Field(ge=0)
491
+ denominator_count: StrictInt = Field(ge=0)
492
+ rate: StrictFloat = Field(ge=0.0, le=1.0)
493
+ wilson_interval: EvalWilsonScoreInterval | None = None
494
+ paired_differences: tuple[StrictInt, ...] = Field(default_factory=tuple)
495
+ distributions: tuple[EvalDistributionSummary, ...] = Field(default_factory=tuple)
496
+ diagnostic: StrictStr
497
+ descriptive_only: StrictBool = True
498
+
499
+ @model_validator(mode="after")
500
+ def _statistical_summary_valid(self) -> EvalReportStatisticalSummary:
501
+ expected = (
502
+ 0.0
503
+ if self.denominator_count == 0
504
+ else self.raw_count / self.denominator_count
505
+ )
506
+ if abs(self.rate - expected) > 0.000000001:
507
+ raise ValueError("statistical summary rate must match raw counts")
508
+ if self.wilson_interval is not None and self.descriptive_only:
509
+ raise ValueError("Wilson-eligible summaries are not descriptive-only")
510
+ if not self.diagnostic.strip():
511
+ raise ValueError("statistical summaries require diagnostics")
512
+ return self
513
+
514
+
515
+ class EvalTaskSummary(EvalReportContractModel):
516
+ """Per-task report summary."""
517
+
518
+ fixture_id: StrictStr
519
+ trial_index: StrictInt = Field(ge=0)
520
+ category: StrictStr
521
+ metrics: tuple[EvalMetricValue, ...]
522
+
523
+
524
+ class EvalArmSummary(EvalReportContractModel):
525
+ """Per-arm report summary with explicit arm-local denominators."""
526
+
527
+ arm_id: EvalTrialArmId
528
+ metrics: tuple[EvalMetricValue, ...]
529
+
530
+
531
+ class EvalCategorySummary(EvalReportContractModel):
532
+ """Per-category aggregate report summary."""
533
+
534
+ category: EvalTaskCategory | StrictStr
535
+ metrics: tuple[EvalMetricValue, ...]
536
+
537
+
538
+ class EvalFailureTaxonomyAssignment(EvalReportContractModel):
539
+ """Optional manual taxonomy assignment that cannot affect scorer success."""
540
+
541
+ primary_category: EvalReportFailureTaxonomyCategory
542
+ contributing_categories: tuple[EvalReportFailureTaxonomyCategory, ...] = Field(
543
+ default_factory=tuple
544
+ )
545
+ explanation: StrictStr
546
+ category_explanations: Mapping[EvalReportFailureTaxonomyCategory, StrictStr] = (
547
+ Field(default_factory=dict)
548
+ )
549
+
550
+ @field_validator("explanation")
551
+ @classmethod
552
+ def _explanation_required(cls, value: str) -> str:
553
+ if not value.strip():
554
+ raise ValueError("manual taxonomy assignments require an explanation")
555
+ return value
556
+
557
+ @model_validator(mode="after")
558
+ def _manual_assignment_valid(self) -> EvalFailureTaxonomyAssignment:
559
+ categories = (self.primary_category, *self.contributing_categories)
560
+ if len(set(categories)) != len(categories):
561
+ raise ValueError("manual taxonomy categories must be unique")
562
+ explanations = dict(self.category_explanations)
563
+ if not explanations:
564
+ explanations = {category: self.explanation for category in categories}
565
+ missing = [category for category in categories if category not in explanations]
566
+ blank = [
567
+ category
568
+ for category, category_explanation in explanations.items()
569
+ if not category_explanation.strip()
570
+ ]
571
+ if missing or blank:
572
+ raise ValueError("each manual taxonomy category requires an explanation")
573
+ object.__setattr__(
574
+ self,
575
+ "category_explanations",
576
+ _freeze_eval_report_mapping(explanations),
577
+ )
578
+ return self
579
+
580
+
581
+ class EvalFailureTaxonomySummary(EvalReportContractModel):
582
+ """Closed taxonomy rollup for report data."""
583
+
584
+ category: EvalReportFailureTaxonomyCategory
585
+ count: StrictInt = Field(ge=0)
586
+ examples: tuple[StrictStr, ...] = Field(default_factory=tuple)
587
+
588
+
589
+ class EvalInvalidTrialSummary(EvalReportContractModel):
590
+ """Top-line invalid-trial visibility."""
591
+
592
+ invalid_trial_count: StrictInt = Field(ge=0)
593
+ appended_record_count: StrictInt = Field(ge=0)
594
+ invalid_trial_rate: StrictFloat = Field(ge=0.0, le=1.0)
595
+ diagnostics: tuple[StrictStr, ...] = Field(default_factory=tuple)
596
+
597
+ @model_validator(mode="after")
598
+ def _invalid_valid(self) -> EvalInvalidTrialSummary:
599
+ expected = (
600
+ 0.0
601
+ if self.appended_record_count == 0
602
+ else self.invalid_trial_count / self.appended_record_count
603
+ )
604
+ if abs(self.invalid_trial_rate - expected) > 0.000000001:
605
+ raise ValueError("invalid trial rate must match counts")
606
+ return self
607
+
608
+
609
+ class EvalConfoundEntry(EvalReportContractModel):
610
+ """Visible confound entry in JSON and Markdown reports."""
611
+
612
+ confound_id: EvalReportConfoundId
613
+ summary: StrictStr
614
+ affects_claims: StrictBool = True
615
+
616
+
617
+ class EvalDecisionRule(EvalReportContractModel):
618
+ """Pre-registered report decision rule."""
619
+
620
+ rule_id: StrictStr
621
+ rule_kind: EvalDecisionRuleKind
622
+ summary: StrictStr
623
+ metric_id: EvalReportMetricId
624
+ threshold: StrictFloat = Field(ge=0.0)
625
+ status: EvalDecisionRuleStatus
626
+ observed_value: StrictFloat | None = Field(default=None, ge=0.0)
627
+ diagnostic: StrictStr | None = None
628
+
629
+ @model_validator(mode="after")
630
+ def _rule_valid(self) -> EvalDecisionRule:
631
+ if (
632
+ self.status is EvalDecisionRuleStatus.DESCRIPTIVE_ONLY
633
+ and not self.diagnostic
634
+ ):
635
+ raise ValueError("descriptive-only decision rules require diagnostics")
636
+ return self
637
+
638
+
639
+ class EvalReportReproducibilityHashes(EvalReportContractModel):
640
+ """Hash references that make a report reproducible."""
641
+
642
+ campaign_manifest_hash: StrictStr
643
+ plan_hashes: tuple[StrictStr, ...]
644
+ resume_index_hash: StrictStr | None = None
645
+ record_hashes: tuple[StrictStr, ...] = Field(default_factory=tuple)
646
+ report_input_hash: StrictStr
647
+
648
+ @model_validator(mode="after")
649
+ def _hashes_valid(self) -> EvalReportReproducibilityHashes:
650
+ for digest in (
651
+ (self.campaign_manifest_hash, self.report_input_hash)
652
+ + self.plan_hashes
653
+ + self.record_hashes
654
+ ):
655
+ _validate_sha256(digest)
656
+ if self.resume_index_hash is not None:
657
+ _validate_sha256(self.resume_index_hash)
658
+ return self
659
+
660
+
661
+ class EvalReportPayload(EvalReportContractModel):
662
+ """Deterministic JSON report payload."""
663
+
664
+ schema_version: StrictInt = EVAL_REPORT_SCHEMA_VERSION
665
+ report_id: StrictStr
666
+ campaign_id: StrictStr
667
+ generated_at: StrictStr
668
+ admission: EvalLiveAdmissionResult
669
+ arms: tuple[EvalTrialArmId, EvalTrialArmId]
670
+ controlled_variables: Mapping[StrictStr, StrictStr] = Field(default_factory=dict)
671
+ budget_policy: EvalReportBudgetPolicy
672
+ budget_usage: EvalBudgetUsageEstimate
673
+ primary_metrics: tuple[EvalMetricValue, ...]
674
+ arm_summaries: tuple[EvalArmSummary, ...] = Field(default_factory=tuple)
675
+ paired_comparisons: tuple[EvalPairedComparison, ...] = Field(default_factory=tuple)
676
+ task_summaries: tuple[EvalTaskSummary, ...] = Field(default_factory=tuple)
677
+ category_summaries: tuple[EvalCategorySummary, ...] = Field(default_factory=tuple)
678
+ taxonomy_summaries: tuple[EvalFailureTaxonomySummary, ...] = Field(
679
+ default_factory=tuple
680
+ )
681
+ invalid_trials: EvalInvalidTrialSummary
682
+ statistical_summaries: tuple[EvalReportStatisticalSummary, ...] = Field(
683
+ default_factory=tuple
684
+ )
685
+ confounds: tuple[EvalConfoundEntry, ...]
686
+ decision_rules: tuple[EvalDecisionRule, ...]
687
+ reproducibility_hashes: EvalReportReproducibilityHashes
688
+ claim_boundaries: tuple[StrictStr, ...]
689
+ report_hash_kind: StrictStr = EVAL_REPORT_HASH_KIND
690
+ report_hash: StrictStr
691
+
692
+ @model_validator(mode="after")
693
+ def _payload_valid(self) -> EvalReportPayload:
694
+ if self.schema_version != EVAL_REPORT_SCHEMA_VERSION:
695
+ raise ValueError("unsupported eval-report schema_version")
696
+ if not _UTC_TIMESTAMP_RE.fullmatch(self.generated_at):
697
+ raise ValueError("report generated_at must be a UTC timestamp")
698
+ if len(set(self.arms)) != 2:
699
+ raise ValueError("reports require two distinct arms")
700
+ object.__setattr__(
701
+ self,
702
+ "controlled_variables",
703
+ _freeze_eval_report_mapping(self.controlled_variables),
704
+ )
705
+ if not self.claim_boundaries:
706
+ raise ValueError("reports require explicit claim boundaries")
707
+ if not self.confounds:
708
+ raise ValueError("reports require explicit confounds")
709
+ if self.report_hash_kind != EVAL_REPORT_HASH_KIND:
710
+ raise ValueError("unsupported report hash kind")
711
+ _validate_sha256(self.report_hash)
712
+ expected = calculate_eval_report_hash(self)
713
+ if self.report_hash != expected:
714
+ raise ValueError("report_hash does not match payload")
715
+ return self
716
+
717
+
718
+ class EvalMarkdownReport(EvalReportContractModel):
719
+ """Deterministic Markdown report content."""
720
+
721
+ schema_version: StrictInt = EVAL_REPORT_SCHEMA_VERSION
722
+ report_id: StrictStr
723
+ content: StrictStr
724
+ content_hash_kind: StrictStr = EVAL_REPORT_MARKDOWN_HASH_KIND
725
+ content_hash: StrictStr
726
+
727
+ @model_validator(mode="after")
728
+ def _markdown_valid(self) -> EvalMarkdownReport:
729
+ if self.schema_version != EVAL_REPORT_SCHEMA_VERSION:
730
+ raise ValueError("unsupported markdown report schema_version")
731
+ if self.content_hash_kind != EVAL_REPORT_MARKDOWN_HASH_KIND:
732
+ raise ValueError("unsupported markdown report hash kind")
733
+ _validate_sha256(self.content_hash)
734
+ if (
735
+ self.content_hash
736
+ != hashlib.sha256(self.content.encode("utf-8")).hexdigest()
737
+ ):
738
+ raise ValueError("markdown content hash does not match content")
739
+ return self
740
+
741
+
742
+ def validate_eval_budget_policy(
743
+ policy: EvalReportBudgetPolicy | Mapping[str, Any] | None,
744
+ *,
745
+ campaign_manifest: EvalCampaignManifest,
746
+ usage: EvalBudgetUsageEstimate | Mapping[str, Any] | None = None,
747
+ ) -> EvalBudgetValidationResult:
748
+ """Validate a campaign budget policy and fail closed with diagnostics."""
749
+ diagnostics: list[EvalBudgetDiagnostic] = []
750
+ if policy is None:
751
+ return EvalBudgetValidationResult(
752
+ valid=False,
753
+ diagnostics=(
754
+ _budget_diagnostic(
755
+ EvalBudgetDiagnosticCode.MISSING_BUDGET_POLICY,
756
+ "eval_reports.budget.required",
757
+ "Budget policy metadata is required for admission.",
758
+ ),
759
+ ),
760
+ )
761
+ try:
762
+ valid_policy = EvalReportBudgetPolicy.model_validate(policy)
763
+ except ValueError as exc:
764
+ return EvalBudgetValidationResult(
765
+ valid=False,
766
+ diagnostics=(
767
+ _budget_diagnostic(
768
+ EvalBudgetDiagnosticCode.MISSING_LIVE_BUDGET_METADATA,
769
+ "eval_reports.budget.model_validate",
770
+ f"Budget policy metadata is invalid: {exc}",
771
+ ),
772
+ ),
773
+ )
774
+ usage_estimate = (
775
+ EvalBudgetUsageEstimate.model_validate(usage)
776
+ if usage is not None
777
+ else EvalBudgetUsageEstimate(
778
+ estimated_spend_usd=0.0,
779
+ prompt_tokens=0,
780
+ completion_tokens=0,
781
+ model_calls=0,
782
+ retries_per_trial=0,
783
+ wall_clock_seconds=0,
784
+ trial_count=0,
785
+ )
786
+ )
787
+ if (
788
+ valid_policy.max_prompt_tokens is None
789
+ or valid_policy.max_completion_tokens is None
790
+ ):
791
+ diagnostics.append(
792
+ _budget_diagnostic(
793
+ EvalBudgetDiagnosticCode.MISSING_TOKEN_CEILING,
794
+ "eval_reports.budget.token_ceilings",
795
+ "Budget policy must declare prompt and completion token ceilings.",
796
+ )
797
+ )
798
+ if valid_policy.max_trials_per_campaign is None:
799
+ diagnostics.append(
800
+ _budget_diagnostic(
801
+ EvalBudgetDiagnosticCode.MISSING_TRIAL_COUNT_CEILING,
802
+ "eval_reports.budget.trial_ceiling",
803
+ "Budget policy must declare a trial-count ceiling.",
804
+ )
805
+ )
806
+ if campaign_manifest.execution_mode is EvalSuiteExecutionMode.OFFLINE_FAKE:
807
+ diagnostics.extend(_offline_budget_diagnostics(valid_policy))
808
+ else:
809
+ diagnostics.extend(_live_budget_metadata_diagnostics(valid_policy))
810
+ diagnostics.extend(_budget_usage_diagnostics(valid_policy, usage_estimate))
811
+ return EvalBudgetValidationResult(
812
+ valid=not diagnostics,
813
+ diagnostics=tuple(diagnostics),
814
+ )
815
+
816
+
817
+ def admit_eval_report_campaign(
818
+ campaign_manifest: EvalCampaignManifest,
819
+ *,
820
+ budget_policy: EvalReportBudgetPolicy | Mapping[str, Any] | None,
821
+ usage: EvalBudgetUsageEstimate | Mapping[str, Any] | None = None,
822
+ ) -> EvalLiveAdmissionResult:
823
+ """Return deterministic live/offline admission with structured diagnostics."""
824
+ budget_result = validate_eval_budget_policy(
825
+ budget_policy,
826
+ campaign_manifest=campaign_manifest,
827
+ usage=usage,
828
+ )
829
+ if campaign_manifest.execution_mode is EvalSuiteExecutionMode.OFFLINE_FAKE:
830
+ diagnostics = (
831
+ ()
832
+ if budget_result.valid
833
+ else (
834
+ EvalLiveAdmissionDiagnostic(
835
+ diagnostic_code=EvalLiveAdmissionDiagnosticCode.BUDGET_POLICY_INVALID,
836
+ rule_id="eval_reports.admission.offline_budget_policy",
837
+ summary="Offline fake admission requires a complete zero-cost bounded budget policy.",
838
+ ),
839
+ )
840
+ )
841
+ return EvalLiveAdmissionResult(
842
+ status=EvalLiveAdmissionStatus.ADMITTED
843
+ if budget_result.valid
844
+ else EvalLiveAdmissionStatus.DENIED,
845
+ diagnostics=diagnostics,
846
+ budget_result=budget_result,
847
+ )
848
+ return EvalLiveAdmissionResult(
849
+ status=EvalLiveAdmissionStatus.DENIED,
850
+ diagnostics=_live_unresolved_dependency_diagnostics(budget_result),
851
+ budget_result=budget_result,
852
+ )
853
+
854
+
855
+ def build_eval_report_payload(
856
+ *,
857
+ report_id: str,
858
+ campaign_manifest: EvalCampaignManifest,
859
+ plans: Sequence[EvalTrialPlan],
860
+ records: Sequence[EvalTrialRecord],
861
+ budget_policy: EvalReportBudgetPolicy,
862
+ usage: EvalBudgetUsageEstimate | None = None,
863
+ resume_index: EvalTrialResumeIndex | None = None,
864
+ generated_at: str = "1970-01-01T00:00:00Z",
865
+ decision_rules: Sequence[EvalDecisionRule] = (),
866
+ manual_taxonomy: Mapping[str, EvalFailureTaxonomyAssignment | Mapping[str, Any]]
867
+ | None = None,
868
+ ) -> EvalReportPayload:
869
+ """Build a deterministic pilot report from existing offline contracts."""
870
+ _validate_report_inputs(campaign_manifest, plans, records, resume_index)
871
+ usage = usage or _default_budget_usage_estimate(plans=plans, records=records)
872
+ admission = admit_eval_report_campaign(
873
+ campaign_manifest,
874
+ budget_policy=budget_policy,
875
+ usage=usage,
876
+ )
877
+ manual_taxonomy = _validate_manual_taxonomy_assignments(manual_taxonomy or {})
878
+ metrics = _primary_metrics(
879
+ plans=plans,
880
+ records=records,
881
+ resume_index=resume_index,
882
+ usage=usage,
883
+ )
884
+ input_hash = _report_input_hash(campaign_manifest, plans, records, resume_index)
885
+ payload = EvalReportPayload.model_construct(
886
+ schema_version=EVAL_REPORT_SCHEMA_VERSION,
887
+ report_id=report_id,
888
+ campaign_id=campaign_manifest.campaign_id,
889
+ generated_at=generated_at,
890
+ admission=admission,
891
+ arms=(EvalTrialArmId.EVAL_SMALL_PI, EvalTrialArmId.EVAL_SMALL_MILLFORGE),
892
+ controlled_variables={
893
+ "model_manifest_hash": campaign_manifest.model_manifest_hash,
894
+ "workflow_graph_hash": campaign_manifest.workflow_graph_hash,
895
+ "fixture_pack_hash": campaign_manifest.fixture_pack_hash,
896
+ "scorer_version": campaign_manifest.scorer_version,
897
+ },
898
+ budget_policy=budget_policy,
899
+ budget_usage=usage,
900
+ primary_metrics=metrics,
901
+ arm_summaries=_arm_summaries(plans=plans, records=records),
902
+ paired_comparisons=_paired_comparisons(plans=plans, records=records),
903
+ task_summaries=_task_summaries(plans=plans, records=records),
904
+ category_summaries=_category_summaries(plans=plans, records=records),
905
+ taxonomy_summaries=_taxonomy_summaries(records, manual_taxonomy),
906
+ invalid_trials=_invalid_trial_summary(records),
907
+ statistical_summaries=_statistical_summaries(
908
+ metrics=metrics,
909
+ paired_comparisons=_paired_comparisons(plans=plans, records=records),
910
+ usage=usage,
911
+ ),
912
+ confounds=default_eval_report_confounds(
913
+ offline_fake=campaign_manifest.execution_mode
914
+ is EvalSuiteExecutionMode.OFFLINE_FAKE
915
+ ),
916
+ decision_rules=tuple(decision_rules)
917
+ or default_eval_report_decision_rules(metrics),
918
+ reproducibility_hashes=EvalReportReproducibilityHashes(
919
+ campaign_manifest_hash=campaign_manifest.campaign_manifest_hash,
920
+ plan_hashes=tuple(plan.plan_hash for plan in plans),
921
+ resume_index_hash=resume_index.resume_index_hash
922
+ if resume_index is not None
923
+ else None,
924
+ record_hashes=tuple(record.record_hash for record in records),
925
+ report_input_hash=input_hash,
926
+ ),
927
+ claim_boundaries=(
928
+ "Offline fake reports are contract and harness-surface evidence only.",
929
+ "No Pi-vs-Millforge model-performance conclusion can be drawn.",
930
+ "Small pilot samples are descriptive unless a decision rule says otherwise.",
931
+ ),
932
+ report_hash_kind=EVAL_REPORT_HASH_KIND,
933
+ report_hash="0" * 64,
934
+ )
935
+ return EvalReportPayload.model_validate(
936
+ payload.model_copy(update={"report_hash": calculate_eval_report_hash(payload)})
937
+ )
938
+
939
+
940
+ def render_eval_markdown_report(payload: EvalReportPayload) -> EvalMarkdownReport:
941
+ """Render a deterministic human-readable Markdown report."""
942
+ primary_metric_ids = {
943
+ EvalReportMetricId.VALID_COMPLETION,
944
+ EvalReportMetricId.FALSE_CLOSURE,
945
+ EvalReportMetricId.FALSE_SUCCESS,
946
+ EvalReportMetricId.ARTIFACT_COMPLETE,
947
+ EvalReportMetricId.CAPABILITY_VIOLATION,
948
+ EvalReportMetricId.CORRECTLY_BLOCKED,
949
+ EvalReportMetricId.FALSE_BLOCKED,
950
+ EvalReportMetricId.RUNTIME_FAILURE,
951
+ EvalReportMetricId.PROVIDER_FAILURE,
952
+ EvalReportMetricId.INVALID_TRIAL,
953
+ }
954
+ primary_metrics = tuple(
955
+ metric
956
+ for metric in payload.primary_metrics
957
+ if metric.metric_id in primary_metric_ids
958
+ )
959
+ secondary_metrics = tuple(
960
+ metric
961
+ for metric in payload.primary_metrics
962
+ if metric.metric_id not in primary_metric_ids
963
+ )
964
+ lines = [
965
+ f"# Eval Report {payload.report_id}",
966
+ "",
967
+ f"- Campaign: {payload.campaign_id}",
968
+ f"- Admission status: {payload.admission.status.value}",
969
+ f"- Arms: {payload.arms[0].value}, {payload.arms[1].value}",
970
+ "",
971
+ "## Controlled Variables",
972
+ *[
973
+ f"- {key}: {payload.controlled_variables[key]}"
974
+ for key in sorted(payload.controlled_variables)
975
+ ],
976
+ "",
977
+ "## Budget Summary",
978
+ f"- Policy: {payload.budget_policy.policy_id}",
979
+ f"- Pricing class: {payload.budget_policy.pricing_class.value}",
980
+ f"- Estimated spend USD: {payload.budget_usage.estimated_spend_usd:.6f}",
981
+ f"- Prompt tokens: {payload.budget_usage.prompt_tokens}",
982
+ f"- Completion tokens: {payload.budget_usage.completion_tokens}",
983
+ f"- Model calls: {payload.budget_usage.model_calls}",
984
+ f"- Trial count: {payload.budget_usage.trial_count}",
985
+ "",
986
+ "## Claim Boundary",
987
+ *[f"- {boundary}" for boundary in payload.claim_boundaries],
988
+ "",
989
+ "## Primary Metrics",
990
+ *[_markdown_metric_line(metric) for metric in primary_metrics],
991
+ *(
992
+ (
993
+ "",
994
+ "## Secondary Metrics",
995
+ *[_markdown_metric_line(metric) for metric in secondary_metrics],
996
+ )
997
+ if secondary_metrics
998
+ else ()
999
+ ),
1000
+ "",
1001
+ "## Severity-One Outcomes",
1002
+ *[
1003
+ f"- {metric.metric_id.value}: {metric.count}"
1004
+ for metric in payload.primary_metrics
1005
+ if metric.severity_one
1006
+ ],
1007
+ "",
1008
+ "## Invalid Trials",
1009
+ f"- Invalid records: {payload.invalid_trials.invalid_trial_count}/"
1010
+ f"{payload.invalid_trials.appended_record_count}",
1011
+ "",
1012
+ "## Per-Category Summary",
1013
+ *[
1014
+ "- "
1015
+ + str(
1016
+ summary.category.value
1017
+ if hasattr(summary.category, "value")
1018
+ else summary.category
1019
+ )
1020
+ + ": "
1021
+ + ", ".join(_inline_metric_summary(metric) for metric in summary.metrics)
1022
+ for summary in payload.category_summaries
1023
+ ],
1024
+ "",
1025
+ "## Statistics",
1026
+ *[
1027
+ f"- {summary.metric_id.value}: {summary.raw_count}/"
1028
+ f"{summary.denominator_count} ({summary.rate:.6f}); "
1029
+ f"{summary.diagnostic}"
1030
+ for summary in payload.statistical_summaries
1031
+ ],
1032
+ "",
1033
+ "## Failure Taxonomy",
1034
+ *[
1035
+ f"- {summary.category.value}: {summary.count}"
1036
+ for summary in payload.taxonomy_summaries
1037
+ ],
1038
+ "",
1039
+ "## Confounds",
1040
+ *[
1041
+ f"- {entry.confound_id.value}: {entry.summary}"
1042
+ for entry in payload.confounds
1043
+ ],
1044
+ "",
1045
+ "## Decision Rules",
1046
+ *[
1047
+ f"- {rule.rule_id}: {rule.status.value}; threshold={rule.threshold:.6f}; "
1048
+ f"observed={rule.observed_value if rule.observed_value is not None else 'n/a'}"
1049
+ for rule in payload.decision_rules
1050
+ ],
1051
+ "",
1052
+ "## Reproducibility",
1053
+ f"- Campaign manifest: {payload.reproducibility_hashes.campaign_manifest_hash}",
1054
+ *[
1055
+ f"- Plan hash: {plan_hash}"
1056
+ for plan_hash in payload.reproducibility_hashes.plan_hashes
1057
+ ],
1058
+ *[
1059
+ f"- Record hash: {record_hash}"
1060
+ for record_hash in payload.reproducibility_hashes.record_hashes
1061
+ ],
1062
+ *(
1063
+ (f"- Resume index: {payload.reproducibility_hashes.resume_index_hash}",)
1064
+ if payload.reproducibility_hashes.resume_index_hash is not None
1065
+ else ()
1066
+ ),
1067
+ f"- Report input: {payload.reproducibility_hashes.report_input_hash}",
1068
+ f"- Report JSON hash: {calculate_eval_report_json_hash(payload)}",
1069
+ f"- Report hash: {payload.report_hash}",
1070
+ "",
1071
+ ]
1072
+ content = "\n".join(lines)
1073
+ _reject_forbidden_material(content)
1074
+ return EvalMarkdownReport(
1075
+ report_id=payload.report_id,
1076
+ content=content,
1077
+ content_hash=hashlib.sha256(content.encode("utf-8")).hexdigest(),
1078
+ )
1079
+
1080
+
1081
+ def _markdown_metric_line(metric: EvalMetricValue) -> str:
1082
+ return f"- {metric.metric_id.value}: {_inline_metric_summary(metric)}"
1083
+
1084
+
1085
+ def _inline_metric_summary(metric: EvalMetricValue) -> str:
1086
+ if metric.value is not None:
1087
+ return (
1088
+ f"{metric.value:.6f} across {metric.count}/"
1089
+ f"{metric.denominator.count} source records"
1090
+ )
1091
+ return f"{metric.count}/{metric.denominator.count} ({metric.rate:.6f})"
1092
+
1093
+
1094
+ def canonical_eval_report_bytes(value: BaseModel | Mapping[str, Any]) -> bytes:
1095
+ """Return canonical ASCII JSON bytes for an eval-report payload."""
1096
+ payload = (
1097
+ value.model_dump(mode="json") if isinstance(value, BaseModel) else dict(value)
1098
+ )
1099
+ _reject_forbidden_material(payload)
1100
+ return (
1101
+ json.dumps(payload, sort_keys=True, separators=(",", ":"), ensure_ascii=True)
1102
+ + "\n"
1103
+ ).encode("ascii")
1104
+
1105
+
1106
+ def calculate_eval_report_hash(report: EvalReportPayload) -> str:
1107
+ payload = report.model_dump(mode="json")
1108
+ payload.pop("report_hash", None)
1109
+ return hashlib.sha256(canonical_eval_report_bytes(payload)).hexdigest()
1110
+
1111
+
1112
+ def calculate_eval_report_json_hash(report: EvalReportPayload) -> str:
1113
+ return hashlib.sha256(canonical_eval_report_bytes(report)).hexdigest()
1114
+
1115
+
1116
+ def canonical_eval_report_json_bytes(report: EvalReportPayload) -> bytes:
1117
+ """Return deterministic report.json bytes for an eval-report payload."""
1118
+ return canonical_eval_report_bytes(report)
1119
+
1120
+
1121
+ def canonical_eval_markdown_report_bytes(
1122
+ report: EvalMarkdownReport | EvalReportPayload,
1123
+ ) -> bytes:
1124
+ """Return deterministic report.md bytes from a Markdown report or payload."""
1125
+ markdown = (
1126
+ render_eval_markdown_report(report)
1127
+ if isinstance(report, EvalReportPayload)
1128
+ else report
1129
+ )
1130
+ _reject_forbidden_material(markdown.content)
1131
+ return markdown.content.encode("utf-8")
1132
+
1133
+
1134
+ def build_eval_report_artifact_bytes(
1135
+ payload: EvalReportPayload,
1136
+ ) -> Mapping[str, bytes]:
1137
+ """Return deterministic report.json, report.md, and hash artifact bytes."""
1138
+ json_bytes = canonical_eval_report_json_bytes(payload)
1139
+ markdown_bytes = canonical_eval_markdown_report_bytes(payload)
1140
+ artifacts = {
1141
+ "report.json": json_bytes,
1142
+ "report.md": markdown_bytes,
1143
+ "report.sha256": (
1144
+ f"{EVAL_REPORT_HASH_KIND} {payload.report_hash}\n"
1145
+ f"{EVAL_REPORT_JSON_HASH_KIND} "
1146
+ f"{hashlib.sha256(json_bytes).hexdigest()}\n"
1147
+ f"{EVAL_REPORT_MARKDOWN_HASH_KIND} "
1148
+ f"{hashlib.sha256(markdown_bytes).hexdigest()}\n"
1149
+ ).encode("ascii"),
1150
+ }
1151
+ return _freeze_eval_report_mapping(artifacts)
1152
+
1153
+
1154
+ def default_eval_report_budget_policy() -> EvalReportBudgetPolicy:
1155
+ """Return the bounded zero-cost policy admitted for offline fake reports."""
1156
+ rate_limit = EvalReportRateLimitPolicy(
1157
+ request_rate_per_window=1,
1158
+ token_rate_per_window=1,
1159
+ concurrent_request_limit=1,
1160
+ window_seconds=1,
1161
+ max_backoff_seconds=0,
1162
+ max_retries_per_trial=0,
1163
+ )
1164
+ return EvalReportBudgetPolicy(
1165
+ policy_id="eval.08c.default.offline_zero_cost.v1",
1166
+ pricing_class=EvalReportPricingClass.OFFLINE_ZERO_COST,
1167
+ max_spend_usd=0.0,
1168
+ max_prompt_tokens=0,
1169
+ max_completion_tokens=0,
1170
+ max_model_calls=0,
1171
+ max_retries_per_trial=0,
1172
+ max_wall_clock_seconds=0,
1173
+ max_trials_per_campaign=64,
1174
+ rate_limit_policy_by_arm={
1175
+ EvalTrialArmId.EVAL_SMALL_PI: rate_limit,
1176
+ EvalTrialArmId.EVAL_SMALL_MILLFORGE: rate_limit,
1177
+ },
1178
+ abort_thresholds=EvalReportAbortThresholds(
1179
+ max_false_closure_rate=0.0,
1180
+ max_capability_violation_rate=0.0,
1181
+ max_invalid_trial_rate=0.0,
1182
+ ),
1183
+ summary="Bounded zero-cost offline fake report policy.",
1184
+ )
1185
+
1186
+
1187
+ def default_eval_report_confounds(
1188
+ *, offline_fake: bool = True
1189
+ ) -> tuple[EvalConfoundEntry, ...]:
1190
+ """Return the required public confound registry for pilot reports."""
1191
+ entries = tuple(
1192
+ EvalConfoundEntry(
1193
+ confound_id=confound_id,
1194
+ summary=_confound_summary(confound_id),
1195
+ affects_claims=True,
1196
+ )
1197
+ for confound_id in EvalReportConfoundId
1198
+ )
1199
+ if offline_fake:
1200
+ return entries
1201
+ return tuple(
1202
+ entry
1203
+ for entry in entries
1204
+ if entry.confound_id is not EvalReportConfoundId.OFFLINE_FAKE_LIMITATIONS
1205
+ )
1206
+
1207
+
1208
+ def default_eval_report_decision_rules(
1209
+ metrics: Sequence[EvalMetricValue],
1210
+ ) -> tuple[EvalDecisionRule, ...]:
1211
+ """Return pre-registered descriptive pilot decision rules."""
1212
+ by_id = {metric.metric_id: metric for metric in metrics}
1213
+ false_closure = by_id.get(EvalReportMetricId.FALSE_CLOSURE)
1214
+ valid_completion = by_id.get(EvalReportMetricId.VALID_COMPLETION)
1215
+ false_blocked = by_id.get(EvalReportMetricId.FALSE_BLOCKED)
1216
+ capability_violation = by_id.get(EvalReportMetricId.CAPABILITY_VIOLATION)
1217
+ invalid_trial = by_id.get(EvalReportMetricId.INVALID_TRIAL)
1218
+ return (
1219
+ EvalDecisionRule(
1220
+ rule_id="eval_reports.rules.max_false_closure_rate",
1221
+ rule_kind=EvalDecisionRuleKind.MAX_FALSE_CLOSURE_RATE,
1222
+ summary="Severity-one false-closure rate must remain at or below the threshold.",
1223
+ metric_id=EvalReportMetricId.FALSE_CLOSURE,
1224
+ threshold=0.0,
1225
+ status=_threshold_status(false_closure, 0.0),
1226
+ observed_value=false_closure.rate if false_closure else None,
1227
+ diagnostic="Pilot samples are descriptive unless confirmed later.",
1228
+ ),
1229
+ EvalDecisionRule(
1230
+ rule_id="eval_reports.rules.min_completion_improvement",
1231
+ rule_kind=EvalDecisionRuleKind.MIN_COMPLETION_IMPROVEMENT,
1232
+ summary="Minimum completion improvement worth pursuing must be positive in paired follow-up campaigns.",
1233
+ metric_id=EvalReportMetricId.COMPLETION_IMPROVEMENT,
1234
+ threshold=0.05,
1235
+ status=EvalDecisionRuleStatus.DESCRIPTIVE_ONLY,
1236
+ observed_value=valid_completion.rate if valid_completion else None,
1237
+ diagnostic="Pilot report records raw completion rates; improvement claims require powered paired follow-up.",
1238
+ ),
1239
+ EvalDecisionRule(
1240
+ rule_id="eval_reports.rules.max_cost_multiplier",
1241
+ rule_kind=EvalDecisionRuleKind.MAX_COST_MULTIPLIER,
1242
+ summary="Cost multiplier must stay within the pre-registered ceiling.",
1243
+ metric_id=EvalReportMetricId.COST_MULTIPLIER,
1244
+ threshold=1.5,
1245
+ status=EvalDecisionRuleStatus.DESCRIPTIVE_ONLY,
1246
+ observed_value=0.0,
1247
+ diagnostic="Offline fake cost is zero; live cost multipliers are descriptive until source-present usage exists.",
1248
+ ),
1249
+ EvalDecisionRule(
1250
+ rule_id="eval_reports.rules.max_latency_multiplier",
1251
+ rule_kind=EvalDecisionRuleKind.MAX_LATENCY_MULTIPLIER,
1252
+ summary="Latency multiplier must stay within the pre-registered ceiling.",
1253
+ metric_id=EvalReportMetricId.LATENCY_MULTIPLIER,
1254
+ threshold=1.5,
1255
+ status=EvalDecisionRuleStatus.DESCRIPTIVE_ONLY,
1256
+ observed_value=0.0,
1257
+ diagnostic="Offline fake latency is not a live performance measurement.",
1258
+ ),
1259
+ EvalDecisionRule(
1260
+ rule_id="eval_reports.rules.acceptable_false_blocked_tradeoff",
1261
+ rule_kind=EvalDecisionRuleKind.ACCEPTABLE_FALSE_BLOCKED_TRADEOFF,
1262
+ summary="False-blocked rate must be weighed against false-closure reduction.",
1263
+ metric_id=EvalReportMetricId.FALSE_BLOCKED,
1264
+ threshold=0.05,
1265
+ status=EvalDecisionRuleStatus.DESCRIPTIVE_ONLY,
1266
+ observed_value=false_blocked.rate if false_blocked else None,
1267
+ diagnostic="False-blocked tradeoff is descriptive in small pilot samples.",
1268
+ ),
1269
+ EvalDecisionRule(
1270
+ rule_id="eval_reports.rules.severity_one_abort_thresholds",
1271
+ rule_kind=EvalDecisionRuleKind.SEVERITY_ONE_ABORT_THRESHOLD,
1272
+ summary="Severity-one false closure, capability violation, and invalid trial rates remain abort-visible.",
1273
+ metric_id=EvalReportMetricId.INVALID_TRIAL,
1274
+ threshold=0.0,
1275
+ status=(
1276
+ EvalDecisionRuleStatus.FAILED
1277
+ if any(
1278
+ metric is not None and metric.rate > 0.0
1279
+ for metric in (false_closure, capability_violation, invalid_trial)
1280
+ )
1281
+ else EvalDecisionRuleStatus.PASSED
1282
+ ),
1283
+ observed_value=invalid_trial.rate if invalid_trial else None,
1284
+ diagnostic="Pilot samples are descriptive unless confirmed later.",
1285
+ ),
1286
+ )
1287
+
1288
+
1289
+ def wilson_score_interval(
1290
+ successes: int,
1291
+ total: int,
1292
+ *,
1293
+ z: float = 1.959963984540054,
1294
+ min_total: int = 30,
1295
+ ) -> tuple[float, float] | None:
1296
+ """Return a Wilson interval only when the sample is large enough."""
1297
+ if total < min_total:
1298
+ return None
1299
+ if successes < 0 or total < 0 or successes > total:
1300
+ raise ValueError("invalid Wilson interval counts")
1301
+ phat = successes / total
1302
+ denominator = 1.0 + z * z / total
1303
+ center = phat + z * z / (2 * total)
1304
+ spread = z * ((phat * (1.0 - phat) + z * z / (4 * total)) / total) ** 0.5
1305
+ return ((center - spread) / denominator, (center + spread) / denominator)
1306
+
1307
+
1308
+ def _offline_budget_diagnostics(
1309
+ policy: EvalReportBudgetPolicy,
1310
+ ) -> tuple[EvalBudgetDiagnostic, ...]:
1311
+ diagnostics: list[EvalBudgetDiagnostic] = []
1312
+ if policy.pricing_class is not EvalReportPricingClass.OFFLINE_ZERO_COST:
1313
+ diagnostics.append(
1314
+ _budget_diagnostic(
1315
+ EvalBudgetDiagnosticCode.OFFLINE_POLICY_NOT_ZERO_COST,
1316
+ "eval_reports.budget.offline_zero_cost",
1317
+ "Offline fake campaigns require the offline_zero_cost pricing class.",
1318
+ )
1319
+ )
1320
+ bounded_fields = (
1321
+ policy.max_prompt_tokens,
1322
+ policy.max_completion_tokens,
1323
+ policy.max_model_calls,
1324
+ policy.max_retries_per_trial,
1325
+ policy.max_wall_clock_seconds,
1326
+ policy.max_trials_per_campaign,
1327
+ )
1328
+ if any(value is None for value in bounded_fields):
1329
+ diagnostics.append(
1330
+ _budget_diagnostic(
1331
+ EvalBudgetDiagnosticCode.OFFLINE_POLICY_UNBOUNDED,
1332
+ "eval_reports.budget.offline_bounded",
1333
+ "Offline fake campaigns must declare token, model-call, retry, wall-clock, and trial ceilings.",
1334
+ )
1335
+ )
1336
+ if policy.max_spend_usd != 0.0:
1337
+ diagnostics.append(
1338
+ _budget_diagnostic(
1339
+ EvalBudgetDiagnosticCode.OFFLINE_POLICY_NOT_ZERO_COST,
1340
+ "eval_reports.budget.offline_spend",
1341
+ "Offline fake campaigns require a zero spend ceiling.",
1342
+ )
1343
+ )
1344
+ return tuple(diagnostics)
1345
+
1346
+
1347
+ def _live_budget_metadata_diagnostics(
1348
+ policy: EvalReportBudgetPolicy,
1349
+ ) -> tuple[EvalBudgetDiagnostic, ...]:
1350
+ diagnostics: list[EvalBudgetDiagnostic] = []
1351
+ required_live_ceilings = (
1352
+ policy.max_spend_usd,
1353
+ policy.max_model_calls,
1354
+ policy.max_retries_per_trial,
1355
+ policy.max_wall_clock_seconds,
1356
+ )
1357
+ if any(ceiling is None for ceiling in required_live_ceilings):
1358
+ diagnostics.append(
1359
+ _budget_diagnostic(
1360
+ EvalBudgetDiagnosticCode.MISSING_LIVE_BUDGET_METADATA,
1361
+ "eval_reports.budget.live_metadata",
1362
+ "Live campaigns require explicit spend, model-call, retry, and wall-clock ceilings.",
1363
+ )
1364
+ )
1365
+ if (
1366
+ policy.pricing_class is EvalReportPricingClass.PROMOTIONAL_FREE_WINDOW
1367
+ and policy.promotional_free_window is None
1368
+ ):
1369
+ diagnostics.append(
1370
+ _budget_diagnostic(
1371
+ EvalBudgetDiagnosticCode.INCOMPLETE_PROMOTIONAL_FREE_WINDOW,
1372
+ "eval_reports.budget.promotional_window",
1373
+ "Promotional free-window pricing requires complete window metadata.",
1374
+ )
1375
+ )
1376
+ arms = {
1377
+ EvalTrialArmId.EVAL_SMALL_PI,
1378
+ EvalTrialArmId.EVAL_SMALL_MILLFORGE,
1379
+ }
1380
+ if set(policy.rate_limit_policy_by_arm) != arms:
1381
+ diagnostics.append(
1382
+ _budget_diagnostic(
1383
+ EvalBudgetDiagnosticCode.UNFAIR_PAIRED_ARM_RATE_LIMIT,
1384
+ "eval_reports.budget.paired_rate_limits",
1385
+ "Paired arms require explicit rate-limit metadata for both arms.",
1386
+ )
1387
+ )
1388
+ elif (
1389
+ len(
1390
+ {
1391
+ canonical_eval_report_bytes(item)
1392
+ for item in policy.rate_limit_policy_by_arm.values()
1393
+ }
1394
+ )
1395
+ != 1
1396
+ ):
1397
+ diagnostics.append(
1398
+ _budget_diagnostic(
1399
+ EvalBudgetDiagnosticCode.UNFAIR_PAIRED_ARM_RATE_LIMIT,
1400
+ "eval_reports.budget.paired_rate_limit_parity",
1401
+ "Paired arm rate-limit metadata must be identical for fair admission.",
1402
+ )
1403
+ )
1404
+ return tuple(diagnostics)
1405
+
1406
+
1407
+ def _budget_usage_diagnostics(
1408
+ policy: EvalReportBudgetPolicy,
1409
+ usage: EvalBudgetUsageEstimate,
1410
+ ) -> tuple[EvalBudgetDiagnostic, ...]:
1411
+ checks = (
1412
+ (
1413
+ policy.max_spend_usd,
1414
+ usage.estimated_spend_usd,
1415
+ EvalBudgetDiagnosticCode.SPEND_CEILING_EXCEEDED,
1416
+ "spend",
1417
+ ),
1418
+ (
1419
+ policy.max_prompt_tokens,
1420
+ usage.prompt_tokens,
1421
+ EvalBudgetDiagnosticCode.PROMPT_TOKEN_CEILING_EXCEEDED,
1422
+ "prompt tokens",
1423
+ ),
1424
+ (
1425
+ policy.max_completion_tokens,
1426
+ usage.completion_tokens,
1427
+ EvalBudgetDiagnosticCode.COMPLETION_TOKEN_CEILING_EXCEEDED,
1428
+ "completion tokens",
1429
+ ),
1430
+ (
1431
+ policy.max_model_calls,
1432
+ usage.model_calls,
1433
+ EvalBudgetDiagnosticCode.MODEL_CALL_CEILING_EXCEEDED,
1434
+ "model calls",
1435
+ ),
1436
+ (
1437
+ policy.max_retries_per_trial,
1438
+ usage.retries_per_trial,
1439
+ EvalBudgetDiagnosticCode.RETRY_CEILING_EXCEEDED,
1440
+ "retries per trial",
1441
+ ),
1442
+ (
1443
+ policy.max_wall_clock_seconds,
1444
+ usage.wall_clock_seconds,
1445
+ EvalBudgetDiagnosticCode.WALL_CLOCK_CEILING_EXCEEDED,
1446
+ "wall-clock seconds",
1447
+ ),
1448
+ (
1449
+ policy.max_trials_per_campaign,
1450
+ usage.trial_count,
1451
+ EvalBudgetDiagnosticCode.TRIAL_COUNT_CEILING_EXCEEDED,
1452
+ "trials",
1453
+ ),
1454
+ )
1455
+ diagnostics: list[EvalBudgetDiagnostic] = []
1456
+ for ceiling, observed, code, label in checks:
1457
+ if ceiling is not None and observed > ceiling:
1458
+ diagnostics.append(
1459
+ _budget_diagnostic(
1460
+ code,
1461
+ f"eval_reports.budget.{code.value}",
1462
+ f"Campaign exceeds configured {label} ceiling.",
1463
+ )
1464
+ )
1465
+ return tuple(diagnostics)
1466
+
1467
+
1468
+ def _live_unresolved_dependency_diagnostics(
1469
+ budget_result: EvalBudgetValidationResult,
1470
+ ) -> tuple[EvalLiveAdmissionDiagnostic, ...]:
1471
+ diagnostics = [
1472
+ EvalLiveAdmissionDiagnostic(
1473
+ diagnostic_code=EvalLiveAdmissionDiagnosticCode.PI_RUNTIME_UNAVAILABLE,
1474
+ rule_id="eval_reports.live.pi_runtime",
1475
+ summary="Pi runtime support is not available for live eval campaigns.",
1476
+ ),
1477
+ EvalLiveAdmissionDiagnostic(
1478
+ diagnostic_code=EvalLiveAdmissionDiagnosticCode.MILLFORGE_LIVE_HARNESS_UNAVAILABLE,
1479
+ rule_id="eval_reports.live.millforge_harness",
1480
+ summary="Millforge live harness execution is not available for live eval campaigns.",
1481
+ ),
1482
+ EvalLiveAdmissionDiagnostic(
1483
+ diagnostic_code=EvalLiveAdmissionDiagnosticCode.SHARED_BACKEND_CONFIGURATION_MISSING,
1484
+ rule_id="eval_reports.live.shared_backend",
1485
+ summary="Shared backend configuration is unresolved.",
1486
+ ),
1487
+ EvalLiveAdmissionDiagnostic(
1488
+ diagnostic_code=EvalLiveAdmissionDiagnosticCode.FIXTURE_WORKSPACE_LIFECYCLE_UNAVAILABLE,
1489
+ rule_id="eval_reports.live.fixture_workspace",
1490
+ summary="Fixture workspace creation and reset lifecycle is unresolved.",
1491
+ ),
1492
+ EvalLiveAdmissionDiagnostic(
1493
+ diagnostic_code=EvalLiveAdmissionDiagnosticCode.RESOURCE_ENFORCEMENT_UNAVAILABLE,
1494
+ rule_id="eval_reports.live.resource_enforcement",
1495
+ summary="Resource ceiling enforcement is unresolved.",
1496
+ ),
1497
+ EvalLiveAdmissionDiagnostic(
1498
+ diagnostic_code=EvalLiveAdmissionDiagnosticCode.BUDGET_POLICY_INVALID,
1499
+ rule_id="eval_reports.live.budget_policy",
1500
+ summary="Live budget policy enforcement is unresolved.",
1501
+ ),
1502
+ EvalLiveAdmissionDiagnostic(
1503
+ diagnostic_code=EvalLiveAdmissionDiagnosticCode.APPEND_ONLY_STORE_SAFETY_UNPROVEN,
1504
+ rule_id="eval_reports.live.append_only_store",
1505
+ summary="Append-only store safety has not been proven for live runs.",
1506
+ ),
1507
+ EvalLiveAdmissionDiagnostic(
1508
+ diagnostic_code=EvalLiveAdmissionDiagnosticCode.DETERMINISTIC_SCORER_UNAVAILABLE,
1509
+ rule_id="eval_reports.live.deterministic_scorer",
1510
+ summary="Deterministic scorer availability is unresolved for live runs.",
1511
+ ),
1512
+ ]
1513
+ return tuple(diagnostics)
1514
+
1515
+
1516
+ def _primary_metrics(
1517
+ *,
1518
+ plans: Sequence[EvalTrialPlan],
1519
+ records: Sequence[EvalTrialRecord],
1520
+ resume_index: EvalTrialResumeIndex | None,
1521
+ usage: EvalBudgetUsageEstimate,
1522
+ ) -> tuple[EvalMetricValue, ...]:
1523
+ planned_denominator = EvalMetricDenominator(
1524
+ denominator_kind=EvalMetricDenominatorKind.PLANNED_TRIALS,
1525
+ count=len(plans) * 2,
1526
+ summary="All planned per-arm trial outcomes.",
1527
+ )
1528
+ record_denominator = EvalMetricDenominator(
1529
+ denominator_kind=EvalMetricDenominatorKind.APPENDED_RECORDS,
1530
+ count=len(records) * 2,
1531
+ summary="All appended per-arm trial records.",
1532
+ )
1533
+ planned_pair_denominator = EvalMetricDenominator(
1534
+ denominator_kind=EvalMetricDenominatorKind.PLANNED_TRIALS,
1535
+ count=len(plans),
1536
+ summary="All planned fixture/trial-index pairs.",
1537
+ )
1538
+ pending_trial_count = _pending_trial_count(
1539
+ plans=plans,
1540
+ records=records,
1541
+ resume_index=resume_index,
1542
+ )
1543
+ missing_pair_count = len(_missing_pair_keys(plans=plans, records=records))
1544
+ outcomes = [
1545
+ result.scorer_result.final_outcome
1546
+ for record in records
1547
+ for result in record.arm_results
1548
+ ]
1549
+ results = [
1550
+ result.scorer_result for record in records for result in record.arm_results
1551
+ ]
1552
+ return (
1553
+ _metric(
1554
+ EvalReportMetricId.VALID_COMPLETION,
1555
+ outcomes.count(EvalTrialOutcome.VALID_COMPLETION),
1556
+ record_denominator,
1557
+ ),
1558
+ _metric(
1559
+ EvalReportMetricId.FALSE_CLOSURE,
1560
+ outcomes.count(EvalTrialOutcome.FALSE_CLOSURE),
1561
+ record_denominator,
1562
+ severity_one=True,
1563
+ ),
1564
+ _metric(
1565
+ EvalReportMetricId.FALSE_SUCCESS,
1566
+ sum(result.false_success for result in results),
1567
+ record_denominator,
1568
+ ),
1569
+ _metric(
1570
+ EvalReportMetricId.ARTIFACT_COMPLETE,
1571
+ sum(result.artifact_complete for result in results),
1572
+ record_denominator,
1573
+ ),
1574
+ _metric(
1575
+ EvalReportMetricId.CAPABILITY_VIOLATION,
1576
+ sum(result.capability_violation for result in results),
1577
+ record_denominator,
1578
+ severity_one=True,
1579
+ ),
1580
+ _metric(
1581
+ EvalReportMetricId.CORRECTLY_BLOCKED,
1582
+ outcomes.count(EvalTrialOutcome.CORRECTLY_BLOCKED),
1583
+ record_denominator,
1584
+ ),
1585
+ _metric(
1586
+ EvalReportMetricId.FALSE_BLOCKED,
1587
+ outcomes.count(EvalTrialOutcome.FALSE_BLOCKED),
1588
+ record_denominator,
1589
+ ),
1590
+ _metric(
1591
+ EvalReportMetricId.RUNTIME_FAILURE,
1592
+ outcomes.count(EvalTrialOutcome.RUNTIME_FAILURE),
1593
+ planned_denominator,
1594
+ ),
1595
+ _metric(
1596
+ EvalReportMetricId.PROVIDER_FAILURE,
1597
+ outcomes.count(EvalTrialOutcome.PROVIDER_FAILURE),
1598
+ planned_denominator,
1599
+ ),
1600
+ _metric(
1601
+ EvalReportMetricId.INVALID_TRIAL,
1602
+ outcomes.count(EvalTrialOutcome.INVALID_TRIAL),
1603
+ record_denominator,
1604
+ severity_one=True,
1605
+ ),
1606
+ _metric(
1607
+ EvalReportMetricId.MISSING_PAIR,
1608
+ missing_pair_count,
1609
+ planned_pair_denominator,
1610
+ ),
1611
+ _metric(
1612
+ EvalReportMetricId.PENDING_TRIAL,
1613
+ pending_trial_count,
1614
+ planned_pair_denominator,
1615
+ ),
1616
+ _metric(
1617
+ EvalReportMetricId.INCOMPLETE_TRIAL,
1618
+ pending_trial_count,
1619
+ planned_pair_denominator,
1620
+ ),
1621
+ *_budget_usage_metrics(usage),
1622
+ *_resource_usage_metrics(
1623
+ records,
1624
+ summary_prefix=(
1625
+ "Total source-present resource usage in appended trial records."
1626
+ ),
1627
+ ),
1628
+ )
1629
+
1630
+
1631
+ def _value_metric(
1632
+ metric_id: EvalReportMetricId,
1633
+ value: float,
1634
+ denominator: EvalMetricDenominator,
1635
+ *,
1636
+ count: int | None = None,
1637
+ ) -> EvalMetricValue:
1638
+ source_count = denominator.count if count is None else count
1639
+ return EvalMetricValue(
1640
+ metric_id=metric_id,
1641
+ count=source_count,
1642
+ denominator=denominator,
1643
+ rate=0.0 if denominator.count == 0 else source_count / denominator.count,
1644
+ value=float(value),
1645
+ )
1646
+
1647
+
1648
+ def _metric(
1649
+ metric_id: EvalReportMetricId,
1650
+ count: int,
1651
+ denominator: EvalMetricDenominator,
1652
+ *,
1653
+ severity_one: bool = False,
1654
+ ) -> EvalMetricValue:
1655
+ return EvalMetricValue(
1656
+ metric_id=metric_id,
1657
+ count=count,
1658
+ denominator=denominator,
1659
+ rate=0.0 if denominator.count == 0 else count / denominator.count,
1660
+ severity_one=severity_one,
1661
+ )
1662
+
1663
+
1664
+ def _budget_usage_metrics(
1665
+ usage: EvalBudgetUsageEstimate,
1666
+ ) -> tuple[EvalMetricValue, ...]:
1667
+ denominator = EvalMetricDenominator(
1668
+ denominator_kind=EvalMetricDenominatorKind.PLANNED_TRIALS,
1669
+ count=usage.trial_count,
1670
+ summary=(
1671
+ "Campaign-level source-present budget usage covering planned trial pairs."
1672
+ ),
1673
+ )
1674
+ return (
1675
+ _value_metric(
1676
+ EvalReportMetricId.ESTIMATED_COST,
1677
+ usage.estimated_spend_usd,
1678
+ denominator,
1679
+ ),
1680
+ _value_metric(
1681
+ EvalReportMetricId.WALL_CLOCK_SECONDS,
1682
+ float(usage.wall_clock_seconds),
1683
+ denominator,
1684
+ ),
1685
+ _value_metric(
1686
+ EvalReportMetricId.RETRIES,
1687
+ float(usage.retries_per_trial),
1688
+ denominator,
1689
+ ),
1690
+ _value_metric(
1691
+ EvalReportMetricId.MODEL_CALLS,
1692
+ float(usage.model_calls),
1693
+ denominator,
1694
+ ),
1695
+ _value_metric(
1696
+ EvalReportMetricId.PROMPT_TOKENS,
1697
+ float(usage.prompt_tokens),
1698
+ denominator,
1699
+ ),
1700
+ _value_metric(
1701
+ EvalReportMetricId.COMPLETION_TOKENS,
1702
+ float(usage.completion_tokens),
1703
+ denominator,
1704
+ ),
1705
+ )
1706
+
1707
+
1708
+ def _default_budget_usage_estimate(
1709
+ *,
1710
+ plans: Sequence[EvalTrialPlan],
1711
+ records: Sequence[EvalTrialRecord],
1712
+ ) -> EvalBudgetUsageEstimate:
1713
+ return EvalBudgetUsageEstimate(
1714
+ estimated_spend_usd=0.0,
1715
+ prompt_tokens=sum(
1716
+ record.model_usage_summary.input_tokens for record in records
1717
+ ),
1718
+ completion_tokens=sum(
1719
+ record.model_usage_summary.output_tokens for record in records
1720
+ ),
1721
+ model_calls=sum(
1722
+ record.model_usage_summary.model_call_count for record in records
1723
+ ),
1724
+ retries_per_trial=0,
1725
+ wall_clock_seconds=0,
1726
+ trial_count=len(plans),
1727
+ )
1728
+
1729
+
1730
+ def _model_usage_metrics(
1731
+ records: Sequence[EvalTrialRecord],
1732
+ *,
1733
+ summary_prefix: str,
1734
+ ) -> tuple[EvalMetricValue, ...]:
1735
+ if not records:
1736
+ return ()
1737
+ usage_totals = (
1738
+ (
1739
+ EvalReportMetricId.MODEL_CALLS,
1740
+ sum(record.model_usage_summary.model_call_count for record in records),
1741
+ ),
1742
+ (
1743
+ EvalReportMetricId.PROMPT_TOKENS,
1744
+ sum(record.model_usage_summary.input_tokens for record in records),
1745
+ ),
1746
+ (
1747
+ EvalReportMetricId.COMPLETION_TOKENS,
1748
+ sum(record.model_usage_summary.output_tokens for record in records),
1749
+ ),
1750
+ )
1751
+ return tuple(
1752
+ _value_metric(
1753
+ metric_id,
1754
+ float(count),
1755
+ EvalMetricDenominator(
1756
+ denominator_kind=EvalMetricDenominatorKind.APPENDED_RECORDS,
1757
+ count=len(records),
1758
+ summary=summary_prefix,
1759
+ ),
1760
+ )
1761
+ for metric_id, count in usage_totals
1762
+ )
1763
+
1764
+
1765
+ def _resource_usage_metrics(
1766
+ records: Sequence[EvalTrialRecord],
1767
+ *,
1768
+ summary_prefix: str,
1769
+ ) -> tuple[EvalMetricValue, ...]:
1770
+ if not records:
1771
+ return ()
1772
+ usage_totals = (
1773
+ (
1774
+ EvalReportMetricId.ARTIFACT_COUNT,
1775
+ sum(record.resource_summary.artifact_count for record in records),
1776
+ ),
1777
+ (
1778
+ EvalReportMetricId.ARTIFACT_BYTES,
1779
+ sum(record.resource_summary.artifact_bytes for record in records),
1780
+ ),
1781
+ (
1782
+ EvalReportMetricId.TURNS,
1783
+ sum(record.resource_summary.turn_count for record in records),
1784
+ ),
1785
+ (
1786
+ EvalReportMetricId.INVALID_TOOL_CALLS,
1787
+ sum(record.resource_summary.invalid_tool_call_count for record in records),
1788
+ ),
1789
+ (
1790
+ EvalReportMetricId.MALFORMED_ARGUMENTS,
1791
+ sum(record.resource_summary.malformed_argument_count for record in records),
1792
+ ),
1793
+ (
1794
+ EvalReportMetricId.PREREQUISITE_VIOLATIONS,
1795
+ sum(
1796
+ record.resource_summary.prerequisite_violation_count
1797
+ for record in records
1798
+ ),
1799
+ ),
1800
+ (
1801
+ EvalReportMetricId.PREMATURE_TERMINALS,
1802
+ sum(record.resource_summary.premature_terminal_count for record in records),
1803
+ ),
1804
+ (
1805
+ EvalReportMetricId.TOOL_RECOVERIES,
1806
+ sum(record.resource_summary.tool_recovery_count for record in records),
1807
+ ),
1808
+ )
1809
+ return tuple(
1810
+ _value_metric(
1811
+ metric_id,
1812
+ float(count),
1813
+ EvalMetricDenominator(
1814
+ denominator_kind=EvalMetricDenominatorKind.APPENDED_RECORDS,
1815
+ count=len(records),
1816
+ summary=summary_prefix,
1817
+ ),
1818
+ )
1819
+ for metric_id, count in usage_totals
1820
+ )
1821
+
1822
+
1823
+ def _paired_comparisons(
1824
+ *, plans: Sequence[EvalTrialPlan], records: Sequence[EvalTrialRecord]
1825
+ ) -> tuple[EvalPairedComparison, ...]:
1826
+ record_by_trial = {record.trial_id: record for record in records}
1827
+ paired = [
1828
+ record
1829
+ for plan in plans
1830
+ if (record := record_by_trial.get(plan.trial_id)) is not None
1831
+ ]
1832
+ left_count = sum(
1833
+ result.scorer_result.primary_success
1834
+ for record in paired
1835
+ for result in record.arm_results
1836
+ if result.arm_id is EvalTrialArmId.EVAL_SMALL_PI
1837
+ )
1838
+ right_count = sum(
1839
+ result.scorer_result.primary_success
1840
+ for record in paired
1841
+ for result in record.arm_results
1842
+ if result.arm_id is EvalTrialArmId.EVAL_SMALL_MILLFORGE
1843
+ )
1844
+ return (
1845
+ EvalPairedComparison(
1846
+ metric_id=EvalReportMetricId.VALID_COMPLETION,
1847
+ left_arm_id=EvalTrialArmId.EVAL_SMALL_PI,
1848
+ right_arm_id=EvalTrialArmId.EVAL_SMALL_MILLFORGE,
1849
+ left_count=left_count,
1850
+ right_count=right_count,
1851
+ paired_denominator=len(paired),
1852
+ difference=right_count - left_count,
1853
+ missing_pair_count=len(_missing_pair_keys(plans=plans, records=records)),
1854
+ ),
1855
+ )
1856
+
1857
+
1858
+ def _arm_summaries(
1859
+ *, plans: Sequence[EvalTrialPlan], records: Sequence[EvalTrialRecord]
1860
+ ) -> tuple[EvalArmSummary, ...]:
1861
+ summaries: list[EvalArmSummary] = []
1862
+ planned_count = len(plans)
1863
+ for arm_id in (EvalTrialArmId.EVAL_SMALL_PI, EvalTrialArmId.EVAL_SMALL_MILLFORGE):
1864
+ arm_results = [
1865
+ result.scorer_result
1866
+ for record in records
1867
+ for result in record.arm_results
1868
+ if result.arm_id is arm_id
1869
+ ]
1870
+ outcomes = [result.final_outcome for result in arm_results]
1871
+ record_denominator = EvalMetricDenominator(
1872
+ denominator_kind=EvalMetricDenominatorKind.APPENDED_RECORDS,
1873
+ count=len(arm_results),
1874
+ summary=f"All valid and invalid appended records for {arm_id.value}.",
1875
+ )
1876
+ planned_denominator = EvalMetricDenominator(
1877
+ denominator_kind=EvalMetricDenominatorKind.PLANNED_TRIALS,
1878
+ count=planned_count,
1879
+ summary=f"All planned trials for {arm_id.value}.",
1880
+ )
1881
+ summaries.append(
1882
+ EvalArmSummary(
1883
+ arm_id=arm_id,
1884
+ metrics=(
1885
+ _metric(
1886
+ EvalReportMetricId.VALID_COMPLETION,
1887
+ outcomes.count(EvalTrialOutcome.VALID_COMPLETION),
1888
+ record_denominator,
1889
+ ),
1890
+ _metric(
1891
+ EvalReportMetricId.INVALID_TRIAL,
1892
+ outcomes.count(EvalTrialOutcome.INVALID_TRIAL),
1893
+ record_denominator,
1894
+ severity_one=True,
1895
+ ),
1896
+ _metric(
1897
+ EvalReportMetricId.RUNTIME_FAILURE,
1898
+ outcomes.count(EvalTrialOutcome.RUNTIME_FAILURE),
1899
+ planned_denominator,
1900
+ ),
1901
+ _metric(
1902
+ EvalReportMetricId.PROVIDER_FAILURE,
1903
+ outcomes.count(EvalTrialOutcome.PROVIDER_FAILURE),
1904
+ planned_denominator,
1905
+ ),
1906
+ _metric(
1907
+ EvalReportMetricId.ARTIFACT_COMPLETE,
1908
+ sum(result.artifact_complete for result in arm_results),
1909
+ record_denominator,
1910
+ ),
1911
+ ),
1912
+ )
1913
+ )
1914
+ return tuple(summaries)
1915
+
1916
+
1917
+ def _task_summaries(
1918
+ *, plans: Sequence[EvalTrialPlan], records: Sequence[EvalTrialRecord]
1919
+ ) -> tuple[EvalTaskSummary, ...]:
1920
+ record_by_trial = {record.trial_id: record for record in records}
1921
+ summaries: list[EvalTaskSummary] = []
1922
+ for plan in plans:
1923
+ record = record_by_trial.get(plan.trial_id)
1924
+ denominator = EvalMetricDenominator(
1925
+ denominator_kind=EvalMetricDenominatorKind.APPENDED_RECORDS,
1926
+ count=2 if record else 0,
1927
+ summary="Per-task appended arm records.",
1928
+ )
1929
+ success_count = (
1930
+ sum(result.scorer_result.primary_success for result in record.arm_results)
1931
+ if record
1932
+ else 0
1933
+ )
1934
+ invalid_count = (
1935
+ sum(
1936
+ result.scorer_result.final_outcome is EvalTrialOutcome.INVALID_TRIAL
1937
+ for result in record.arm_results
1938
+ )
1939
+ if record
1940
+ else 0
1941
+ )
1942
+ summaries.append(
1943
+ EvalTaskSummary(
1944
+ fixture_id=plan.fixture_instance.fixture_id,
1945
+ trial_index=plan.trial_index,
1946
+ category=plan.fixture_instance.public_projection.category.value,
1947
+ metrics=(
1948
+ _metric(
1949
+ EvalReportMetricId.VALID_COMPLETION,
1950
+ success_count,
1951
+ denominator,
1952
+ ),
1953
+ _metric(
1954
+ EvalReportMetricId.INVALID_TRIAL,
1955
+ invalid_count,
1956
+ denominator,
1957
+ severity_one=True,
1958
+ ),
1959
+ _metric(
1960
+ EvalReportMetricId.INCOMPLETE_TRIAL,
1961
+ 0 if record else 1,
1962
+ EvalMetricDenominator(
1963
+ denominator_kind=EvalMetricDenominatorKind.PLANNED_TRIALS,
1964
+ count=1,
1965
+ summary="One planned fixture/trial-index pair.",
1966
+ ),
1967
+ ),
1968
+ *_model_usage_metrics(
1969
+ (record,) if record else (),
1970
+ summary_prefix="Per-task source-present model usage.",
1971
+ ),
1972
+ *_resource_usage_metrics(
1973
+ (record,) if record else (),
1974
+ summary_prefix="Per-task source-present resource usage.",
1975
+ ),
1976
+ ),
1977
+ )
1978
+ )
1979
+ return tuple(summaries)
1980
+
1981
+
1982
+ def _category_summaries(
1983
+ *, plans: Sequence[EvalTrialPlan], records: Sequence[EvalTrialRecord]
1984
+ ) -> tuple[EvalCategorySummary, ...]:
1985
+ planned_by_category: Counter[str] = Counter(
1986
+ plan.fixture_instance.public_projection.category.value for plan in plans
1987
+ )
1988
+ by_category: dict[str, list[EvalTrialOutcome]] = {
1989
+ category: [] for category in planned_by_category
1990
+ }
1991
+ records_by_category: dict[str, list[EvalTrialRecord]] = {
1992
+ category: [] for category in planned_by_category
1993
+ }
1994
+ for record in records:
1995
+ records_by_category.setdefault(record.task_category, []).append(record)
1996
+ by_category.setdefault(record.task_category, []).extend(
1997
+ result.scorer_result.final_outcome for result in record.arm_results
1998
+ )
1999
+ summaries: list[EvalCategorySummary] = []
2000
+ for category in sorted(by_category):
2001
+ outcomes = by_category[category]
2002
+ denominator = EvalMetricDenominator(
2003
+ denominator_kind=EvalMetricDenominatorKind.APPENDED_RECORDS,
2004
+ count=len(outcomes),
2005
+ summary="Per-category appended arm records.",
2006
+ )
2007
+ summaries.append(
2008
+ EvalCategorySummary(
2009
+ category=category,
2010
+ metrics=(
2011
+ _metric(
2012
+ EvalReportMetricId.VALID_COMPLETION,
2013
+ outcomes.count(EvalTrialOutcome.VALID_COMPLETION),
2014
+ denominator,
2015
+ ),
2016
+ _metric(
2017
+ EvalReportMetricId.INVALID_TRIAL,
2018
+ outcomes.count(EvalTrialOutcome.INVALID_TRIAL),
2019
+ denominator,
2020
+ severity_one=True,
2021
+ ),
2022
+ _metric(
2023
+ EvalReportMetricId.INCOMPLETE_TRIAL,
2024
+ max(0, planned_by_category[category] - len(outcomes) // 2),
2025
+ EvalMetricDenominator(
2026
+ denominator_kind=EvalMetricDenominatorKind.PLANNED_TRIALS,
2027
+ count=planned_by_category[category],
2028
+ summary="Planned trial pairs for this category.",
2029
+ ),
2030
+ ),
2031
+ *_model_usage_metrics(
2032
+ records_by_category.get(category, ()),
2033
+ summary_prefix="Per-category source-present model usage.",
2034
+ ),
2035
+ *_resource_usage_metrics(
2036
+ records_by_category.get(category, ()),
2037
+ summary_prefix="Per-category source-present resource usage.",
2038
+ ),
2039
+ ),
2040
+ )
2041
+ )
2042
+ return tuple(summaries)
2043
+
2044
+
2045
+ def _missing_pair_keys(
2046
+ *, plans: Sequence[EvalTrialPlan], records: Sequence[EvalTrialRecord]
2047
+ ) -> set[tuple[str, int]]:
2048
+ recorded_pairs = {(record.fixture_id, record.trial_index) for record in records}
2049
+ return {
2050
+ (plan.fixture_instance.fixture_id, plan.trial_index)
2051
+ for plan in plans
2052
+ if (plan.fixture_instance.fixture_id, plan.trial_index) not in recorded_pairs
2053
+ }
2054
+
2055
+
2056
+ def _pending_trial_count(
2057
+ *,
2058
+ plans: Sequence[EvalTrialPlan],
2059
+ records: Sequence[EvalTrialRecord],
2060
+ resume_index: EvalTrialResumeIndex | None,
2061
+ ) -> int:
2062
+ if resume_index is None:
2063
+ completed_plan_hashes = {record.trial_plan_hash for record in records}
2064
+ return sum(plan.plan_hash not in completed_plan_hashes for plan in plans)
2065
+ plan_hashes = {plan.plan_hash for plan in plans}
2066
+ return sum(
2067
+ plan_hash in plan_hashes for plan_hash in resume_index.pending_trial_plan_hashes
2068
+ )
2069
+
2070
+
2071
+ def _validate_manual_taxonomy_assignments(
2072
+ manual_taxonomy: Mapping[str, EvalFailureTaxonomyAssignment | Mapping[str, Any]],
2073
+ ) -> Mapping[str, EvalFailureTaxonomyAssignment]:
2074
+ assignments: dict[str, EvalFailureTaxonomyAssignment] = {}
2075
+ for key, assignment in manual_taxonomy.items():
2076
+ if not key.strip():
2077
+ raise ValueError("manual taxonomy assignment keys must be non-empty")
2078
+ assignments[key] = EvalFailureTaxonomyAssignment.model_validate(assignment)
2079
+ return _freeze_eval_report_mapping(assignments)
2080
+
2081
+
2082
+ def _statistical_summaries(
2083
+ *,
2084
+ metrics: Sequence[EvalMetricValue],
2085
+ paired_comparisons: Sequence[EvalPairedComparison],
2086
+ usage: EvalBudgetUsageEstimate,
2087
+ ) -> tuple[EvalReportStatisticalSummary, ...]:
2088
+ paired_by_metric = {
2089
+ comparison.metric_id: comparison for comparison in paired_comparisons
2090
+ }
2091
+ summaries = [
2092
+ _metric_statistical_summary(
2093
+ metric,
2094
+ paired_comparison=paired_by_metric.get(metric.metric_id),
2095
+ )
2096
+ for metric in metrics
2097
+ if metric.metric_id
2098
+ in {
2099
+ EvalReportMetricId.VALID_COMPLETION,
2100
+ EvalReportMetricId.FALSE_CLOSURE,
2101
+ EvalReportMetricId.FALSE_BLOCKED,
2102
+ EvalReportMetricId.CAPABILITY_VIOLATION,
2103
+ EvalReportMetricId.PROVIDER_FAILURE,
2104
+ EvalReportMetricId.RUNTIME_FAILURE,
2105
+ EvalReportMetricId.INVALID_TRIAL,
2106
+ }
2107
+ ]
2108
+ summaries.append(
2109
+ EvalReportStatisticalSummary(
2110
+ metric_id=EvalReportMetricId.ESTIMATED_COST,
2111
+ raw_count=0,
2112
+ denominator_count=0,
2113
+ rate=0.0,
2114
+ distributions=(
2115
+ _distribution_summary(
2116
+ "estimated_cost_usd",
2117
+ (usage.estimated_spend_usd,),
2118
+ ),
2119
+ ),
2120
+ diagnostic="Cost summaries are descriptive and use source-present campaign usage only.",
2121
+ descriptive_only=True,
2122
+ )
2123
+ )
2124
+ summaries.append(
2125
+ EvalReportStatisticalSummary(
2126
+ metric_id=EvalReportMetricId.WALL_CLOCK_SECONDS,
2127
+ raw_count=0,
2128
+ denominator_count=0,
2129
+ rate=0.0,
2130
+ distributions=(
2131
+ _distribution_summary(
2132
+ "wall_clock_seconds",
2133
+ (float(usage.wall_clock_seconds),),
2134
+ ),
2135
+ ),
2136
+ diagnostic="Latency summaries are descriptive and use source-present campaign usage only.",
2137
+ descriptive_only=True,
2138
+ )
2139
+ )
2140
+ return tuple(summaries)
2141
+
2142
+
2143
+ def _metric_statistical_summary(
2144
+ metric: EvalMetricValue,
2145
+ *,
2146
+ paired_comparison: EvalPairedComparison | None,
2147
+ ) -> EvalReportStatisticalSummary:
2148
+ interval = wilson_score_interval(metric.count, metric.denominator.count)
2149
+ paired_differences = (
2150
+ (paired_comparison.difference,) if paired_comparison is not None else ()
2151
+ )
2152
+ if interval is None:
2153
+ return EvalReportStatisticalSummary(
2154
+ metric_id=metric.metric_id,
2155
+ raw_count=metric.count,
2156
+ denominator_count=metric.denominator.count,
2157
+ rate=metric.rate,
2158
+ paired_differences=paired_differences,
2159
+ diagnostic=(
2160
+ "Small-N pilot diagnostic only; Wilson interval omitted until "
2161
+ "the eligible denominator is at least 30."
2162
+ ),
2163
+ descriptive_only=True,
2164
+ )
2165
+ return EvalReportStatisticalSummary(
2166
+ metric_id=metric.metric_id,
2167
+ raw_count=metric.count,
2168
+ denominator_count=metric.denominator.count,
2169
+ rate=metric.rate,
2170
+ wilson_interval=EvalWilsonScoreInterval(
2171
+ successes=metric.count,
2172
+ total=metric.denominator.count,
2173
+ confidence_level=0.95,
2174
+ lower=interval[0],
2175
+ upper=interval[1],
2176
+ ),
2177
+ paired_differences=paired_differences,
2178
+ diagnostic="Wilson score interval emitted for eligible descriptive counts.",
2179
+ descriptive_only=False,
2180
+ )
2181
+
2182
+
2183
+ def _distribution_summary(
2184
+ statistic_id: str,
2185
+ values: Sequence[float],
2186
+ ) -> EvalDistributionSummary:
2187
+ raw_values = tuple(float(value) for value in values)
2188
+ if not raw_values:
2189
+ return EvalDistributionSummary(statistic_id=statistic_id, sample_count=0)
2190
+ ordered = tuple(sorted(raw_values))
2191
+ return EvalDistributionSummary(
2192
+ statistic_id=statistic_id,
2193
+ sample_count=len(raw_values),
2194
+ raw_values=raw_values,
2195
+ median=float(median(ordered)),
2196
+ p90=_nearest_rank_percentile(ordered, 0.90),
2197
+ p95=_nearest_rank_percentile(ordered, 0.95),
2198
+ )
2199
+
2200
+
2201
+ def _nearest_rank_percentile(values: Sequence[float], percentile: float) -> float:
2202
+ if not values:
2203
+ raise ValueError("percentile requires at least one value")
2204
+ index = max(0, min(len(values) - 1, int(round(percentile * len(values) + 0.5)) - 1))
2205
+ return float(values[index])
2206
+
2207
+
2208
+ def _taxonomy_summaries(
2209
+ records: Sequence[EvalTrialRecord],
2210
+ manual_taxonomy: Mapping[str, EvalFailureTaxonomyAssignment],
2211
+ ) -> tuple[EvalFailureTaxonomySummary, ...]:
2212
+ counts: Counter[EvalReportFailureTaxonomyCategory] = Counter()
2213
+ for assignment in manual_taxonomy.values():
2214
+ counts[assignment.primary_category] += 1
2215
+ counts.update(assignment.contributing_categories)
2216
+ for record in records:
2217
+ for result in record.arm_results:
2218
+ for label in result.scorer_result.failure_labels:
2219
+ category = _failure_label_to_report_category(label)
2220
+ if category is not None:
2221
+ counts[category] += 1
2222
+ return tuple(
2223
+ EvalFailureTaxonomySummary(category=category, count=counts[category])
2224
+ for category in sorted(counts, key=lambda item: item.value)
2225
+ )
2226
+
2227
+
2228
+ def _invalid_trial_summary(
2229
+ records: Sequence[EvalTrialRecord],
2230
+ ) -> EvalInvalidTrialSummary:
2231
+ explanations = tuple(
2232
+ explanation
2233
+ for record in records
2234
+ for explanation in record.invalid_trial_explanations.values()
2235
+ )
2236
+ invalid = sum(
2237
+ result.scorer_result.final_outcome is EvalTrialOutcome.INVALID_TRIAL
2238
+ for record in records
2239
+ for result in record.arm_results
2240
+ )
2241
+ appended = len(records) * 2
2242
+ return EvalInvalidTrialSummary(
2243
+ invalid_trial_count=invalid,
2244
+ appended_record_count=appended,
2245
+ invalid_trial_rate=0.0 if appended == 0 else invalid / appended,
2246
+ diagnostics=explanations,
2247
+ )
2248
+
2249
+
2250
+ def _validate_report_inputs(
2251
+ campaign_manifest: EvalCampaignManifest,
2252
+ plans: Sequence[EvalTrialPlan],
2253
+ records: Sequence[EvalTrialRecord],
2254
+ resume_index: EvalTrialResumeIndex | None,
2255
+ ) -> None:
2256
+ if not plans:
2257
+ raise ValueError("reports require at least one plan")
2258
+ if any(
2259
+ plan.campaign_manifest.campaign_id != campaign_manifest.campaign_id
2260
+ for plan in plans
2261
+ ):
2262
+ raise ValueError("plan campaign ID must match campaign manifest")
2263
+ if any(
2264
+ plan.campaign_manifest.campaign_manifest_hash
2265
+ != campaign_manifest.campaign_manifest_hash
2266
+ for plan in plans
2267
+ ):
2268
+ raise ValueError("plan campaign hash must match campaign manifest")
2269
+ plan_by_id = {plan.trial_id: plan for plan in plans}
2270
+ if len(plan_by_id) != len(plans):
2271
+ raise ValueError("reports reject duplicate planned trial IDs")
2272
+ plan_hashes = {plan.plan_hash for plan in plans}
2273
+ for record in records:
2274
+ plan = plan_by_id.get(record.trial_id)
2275
+ if plan is None:
2276
+ raise ValueError("record trial_id must be present in plans")
2277
+ checks = {
2278
+ "campaign ID": record.campaign_id == campaign_manifest.campaign_id,
2279
+ "campaign hash": record.campaign_manifest_hash
2280
+ == campaign_manifest.campaign_manifest_hash,
2281
+ "trial ID": record.trial_id == plan.trial_id,
2282
+ "trial index": record.trial_index == plan.trial_index,
2283
+ "trial hash": record.trial_plan_hash == plan.plan_hash,
2284
+ "fixture ID": record.fixture_id == plan.fixture_instance.fixture_id,
2285
+ "fixture instance ID": record.fixture_instance_id
2286
+ == plan.fixture_instance.fixture_instance_id,
2287
+ "fixture hash": record.fixture_hash == plan.fixture_instance.fixture_hash,
2288
+ "fixture snapshot hash": record.fixture_snapshot_hash
2289
+ == plan.fixture_instance.fixture_snapshot_hash,
2290
+ "model hash": record.model_manifest_hash
2291
+ == campaign_manifest.model_manifest_hash,
2292
+ "workflow hash": record.workflow_graph_hash
2293
+ == campaign_manifest.workflow_graph_hash,
2294
+ "fixture pack hash": record.fixture_pack_hash
2295
+ == campaign_manifest.fixture_pack_hash,
2296
+ }
2297
+ for label, valid in checks.items():
2298
+ if not valid:
2299
+ raise ValueError(f"record {label} must match report inputs")
2300
+ if tuple(record.arm_order) != tuple(plan.arm_order):
2301
+ raise ValueError("record arm order must match campaign plan")
2302
+ arm_plans = {arm_plan.arm_id: arm_plan for arm_plan in plan.arm_plans}
2303
+ for result in record.arm_results:
2304
+ arm_plan = arm_plans.get(result.arm_id)
2305
+ if arm_plan is None:
2306
+ raise ValueError("record arm must be present in campaign plan")
2307
+ result_checks = {
2308
+ "trial ID": result.trial_id == plan.trial_id,
2309
+ "trial hash": result.trial_plan_hash == plan.plan_hash,
2310
+ "fixture ID": result.scorer_result.fixture_id
2311
+ == plan.fixture_instance.fixture_id,
2312
+ "scorer input trial ID": result.scorer_input.trial_id == plan.trial_id,
2313
+ "scorer input fixture hash": result.scorer_input.fixture_hash
2314
+ == plan.fixture_instance.fixture_hash,
2315
+ "arm": result.arm_id == arm_plan.arm_id,
2316
+ }
2317
+ for label, valid in result_checks.items():
2318
+ if not valid:
2319
+ raise ValueError(f"scorer result {label} must match report inputs")
2320
+ if result.scorer_result.scorer_version != campaign_manifest.scorer_version:
2321
+ raise ValueError("scorer version must match campaign manifest")
2322
+ if resume_index is not None:
2323
+ if (
2324
+ resume_index.campaign_manifest_hash
2325
+ != campaign_manifest.campaign_manifest_hash
2326
+ ):
2327
+ raise ValueError("resume index campaign hash must match campaign manifest")
2328
+ record_hashes = {record.record_hash for record in records}
2329
+ completed_hashes = set(resume_index.completed_trial_record_hashes)
2330
+ pending_hashes = set(resume_index.pending_trial_plan_hashes)
2331
+ if not completed_hashes.issubset(record_hashes):
2332
+ raise ValueError("resume index completed records must be included")
2333
+ if not pending_hashes.issubset(plan_hashes):
2334
+ raise ValueError("resume index pending plans must be included")
2335
+ completed_plan_hashes = {record.trial_plan_hash for record in records}
2336
+ if completed_plan_hashes & pending_hashes:
2337
+ raise ValueError("resume index cannot mark completed plans as pending")
2338
+ if not completed_plan_hashes.issubset(plan_hashes):
2339
+ raise ValueError("resume index completed plans must be included")
2340
+
2341
+
2342
+ def _report_input_hash(
2343
+ campaign_manifest: EvalCampaignManifest,
2344
+ plans: Sequence[EvalTrialPlan],
2345
+ records: Sequence[EvalTrialRecord],
2346
+ resume_index: EvalTrialResumeIndex | None,
2347
+ ) -> str:
2348
+ payload = {
2349
+ "campaign_manifest_hash": campaign_manifest.campaign_manifest_hash,
2350
+ "plan_hashes": tuple(plan.plan_hash for plan in plans),
2351
+ "record_hashes": tuple(record.record_hash for record in records),
2352
+ "resume_index_hash": resume_index.resume_index_hash
2353
+ if resume_index is not None
2354
+ else None,
2355
+ }
2356
+ return hashlib.sha256(canonical_eval_report_bytes(payload)).hexdigest()
2357
+
2358
+
2359
+ def _threshold_status(
2360
+ metric: EvalMetricValue | None,
2361
+ threshold: float,
2362
+ ) -> EvalDecisionRuleStatus:
2363
+ if metric is None or metric.denominator.count < 30:
2364
+ return EvalDecisionRuleStatus.DESCRIPTIVE_ONLY
2365
+ return (
2366
+ EvalDecisionRuleStatus.PASSED
2367
+ if metric.rate <= threshold
2368
+ else EvalDecisionRuleStatus.FAILED
2369
+ )
2370
+
2371
+
2372
+ def _budget_diagnostic(
2373
+ code: EvalBudgetDiagnosticCode,
2374
+ rule_id: str,
2375
+ summary: str,
2376
+ ) -> EvalBudgetDiagnostic:
2377
+ return EvalBudgetDiagnostic(diagnostic_code=code, rule_id=rule_id, summary=summary)
2378
+
2379
+
2380
+ def _failure_label_to_report_category(
2381
+ label: EvalFailureTaxonomyLabel,
2382
+ ) -> EvalReportFailureTaxonomyCategory | None:
2383
+ return {
2384
+ EvalFailureTaxonomyLabel.REQUIRED_ARTIFACT_MISSING: (
2385
+ EvalReportFailureTaxonomyCategory.MISSING_ARTIFACT
2386
+ ),
2387
+ EvalFailureTaxonomyLabel.REQUIRED_ARTIFACT_MALFORMED: (
2388
+ EvalReportFailureTaxonomyCategory.INVALID_PATCH
2389
+ ),
2390
+ EvalFailureTaxonomyLabel.FALSE_SUCCESS_TERMINAL: (
2391
+ EvalReportFailureTaxonomyCategory.UNSUPPORTED_SUCCESS_CLAIM
2392
+ ),
2393
+ EvalFailureTaxonomyLabel.CAPABILITY_VIOLATION: (
2394
+ EvalReportFailureTaxonomyCategory.CAPABILITY_VIOLATION
2395
+ ),
2396
+ EvalFailureTaxonomyLabel.PROVIDER_DEFECT: (
2397
+ EvalReportFailureTaxonomyCategory.PROVIDER_FAILURE
2398
+ ),
2399
+ EvalFailureTaxonomyLabel.INFRASTRUCTURE_DEFECT: (
2400
+ EvalReportFailureTaxonomyCategory.INVALID_TRIAL_INFRASTRUCTURE
2401
+ ),
2402
+ }.get(label)
2403
+
2404
+
2405
+ def _confound_summary(confound_id: EvalReportConfoundId) -> str:
2406
+ return {
2407
+ EvalReportConfoundId.PI_PROMPT_TOOL_BEHAVIOR: "Pi prompt and tool behavior can differ from Millforge.",
2408
+ EvalReportConfoundId.MILLFORGE_PROMPT_TOOL_BEHAVIOR: "Millforge prompt and tool behavior can affect outcomes.",
2409
+ EvalReportConfoundId.HARNESS_BEHAVIOR: "Harness behavior can affect trial setup, execution, and evidence capture.",
2410
+ EvalReportConfoundId.CONTEXT_PACKING: "Context packing choices can affect task evidence.",
2411
+ EvalReportConfoundId.PARSER_FALLBACK: "Parser fallback behavior can affect terminal interpretation.",
2412
+ EvalReportConfoundId.PROVIDER_NONDETERMINISM: "Provider nondeterminism can affect live results.",
2413
+ EvalReportConfoundId.RATE_LIMITING: "Rate limiting can affect latency and retry behavior.",
2414
+ EvalReportConfoundId.CACHED_PROVIDER_RESPONSES: "Cached provider responses can affect cost and latency.",
2415
+ EvalReportConfoundId.TOKEN_ACCOUNTING_DIFFERENCES: "Token accounting can differ across backends.",
2416
+ EvalReportConfoundId.SAMPLING_PARAMETER_MISMATCH: "Sampling-parameter mismatches can affect comparability.",
2417
+ EvalReportConfoundId.OFFLINE_FAKE_LIMITATIONS: "Offline fake execution cannot support model-performance claims.",
2418
+ }[confound_id]
2419
+
2420
+
2421
+ def _freeze_eval_report_mapping(mapping: Mapping[Any, Any]) -> _FrozenEvalReportDict:
2422
+ return _FrozenEvalReportDict(mapping)
2423
+
2424
+
2425
+ def _validate_sha256(value: str) -> None:
2426
+ if not _SHA256_RE.fullmatch(value):
2427
+ raise ValueError("value must be a lowercase sha256 digest")
2428
+
2429
+
2430
+ def _reject_forbidden_material(value: Any, *, field_name: str | None = None) -> None:
2431
+ if field_name is not None:
2432
+ lowered = field_name.lower()
2433
+ if any(marker in lowered for marker in _SECRET_FIELD_MARKERS):
2434
+ raise ValueError("secret-like field names are forbidden in eval reports")
2435
+ if "endpoint" in lowered or "url" in lowered:
2436
+ raise ValueError("endpoint URLs are forbidden in eval reports")
2437
+ if isinstance(value, Mapping):
2438
+ for key, item in value.items():
2439
+ _reject_forbidden_material(item, field_name=str(key))
2440
+ return
2441
+ if isinstance(value, (tuple, list, set, frozenset)):
2442
+ for item in value:
2443
+ _reject_forbidden_material(item)
2444
+ return
2445
+ if isinstance(value, str):
2446
+ lowered = value.lower()
2447
+ if any(token in lowered for token in _DENIED_TEXT_TOKENS):
2448
+ raise ValueError("forbidden private material in eval report payload")
2449
+ if _ENDPOINT_URL.search(value):
2450
+ raise ValueError("endpoint URLs are forbidden in eval reports")
2451
+ if (
2452
+ _WINDOWS_ABSOLUTE_PATH.search(value)
2453
+ or _POSIX_ABSOLUTE_PATH.search(value)
2454
+ or _USER_HOME_PATH.search(value)
2455
+ ):
2456
+ raise ValueError("host absolute paths are forbidden in eval reports")
2457
+ if any(pattern.search(value) for pattern in _CREDENTIAL_VALUE_PATTERNS):
2458
+ raise ValueError("credential-like values are forbidden in eval reports")
2459
+
2460
+
2461
+ __all__ = [
2462
+ "EVAL_REPORT_HASH_KIND",
2463
+ "EVAL_REPORT_JSON_HASH_KIND",
2464
+ "EVAL_REPORT_MARKDOWN_HASH_KIND",
2465
+ "EVAL_REPORT_SCHEMA_VERSION",
2466
+ "EvalArmSummary",
2467
+ "EvalBudgetDiagnostic",
2468
+ "EvalBudgetDiagnosticCode",
2469
+ "EvalBudgetUsageEstimate",
2470
+ "EvalBudgetValidationResult",
2471
+ "EvalCategorySummary",
2472
+ "EvalConfoundEntry",
2473
+ "EvalDecisionRuleKind",
2474
+ "EvalDecisionRule",
2475
+ "EvalDecisionRuleStatus",
2476
+ "EvalDistributionSummary",
2477
+ "EvalFailureTaxonomyAssignment",
2478
+ "EvalFailureTaxonomySummary",
2479
+ "EvalInvalidTrialSummary",
2480
+ "EvalLiveAdmissionDiagnostic",
2481
+ "EvalLiveAdmissionDiagnosticCode",
2482
+ "EvalLiveAdmissionResult",
2483
+ "EvalLiveAdmissionStatus",
2484
+ "EvalMarkdownReport",
2485
+ "EvalMetricDenominator",
2486
+ "EvalMetricDenominatorKind",
2487
+ "EvalMetricValue",
2488
+ "EvalPairedComparison",
2489
+ "EvalPromotionalFreeWindow",
2490
+ "EvalReportAbortThresholds",
2491
+ "EvalReportBudgetPolicy",
2492
+ "EvalReportConfoundId",
2493
+ "EvalReportContractModel",
2494
+ "EvalReportFailureTaxonomyCategory",
2495
+ "EvalReportMetricId",
2496
+ "EvalReportPayload",
2497
+ "EvalReportPricingClass",
2498
+ "EvalReportRateLimitPolicy",
2499
+ "EvalReportReproducibilityHashes",
2500
+ "EvalReportStatisticalSummary",
2501
+ "EvalTaskSummary",
2502
+ "EvalWilsonScoreInterval",
2503
+ "admit_eval_report_campaign",
2504
+ "build_eval_report_artifact_bytes",
2505
+ "build_eval_report_payload",
2506
+ "calculate_eval_report_hash",
2507
+ "calculate_eval_report_json_hash",
2508
+ "canonical_eval_markdown_report_bytes",
2509
+ "canonical_eval_report_bytes",
2510
+ "canonical_eval_report_json_bytes",
2511
+ "default_eval_report_budget_policy",
2512
+ "default_eval_report_confounds",
2513
+ "default_eval_report_decision_rules",
2514
+ "render_eval_markdown_report",
2515
+ "validate_eval_budget_policy",
2516
+ "wilson_score_interval",
2517
+ ]