millforge 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- millforge/__init__.py +1174 -0
- millforge/_forge/LICENSE +21 -0
- millforge/_forge/PROVENANCE.json +295 -0
- millforge/_forge/UPDATE_POLICY.md +24 -0
- millforge/_forge/__init__.py +14 -0
- millforge/_forge/adapter.py +2232 -0
- millforge/_forge/base_runner.py +121 -0
- millforge/_forge/clients/__init__.py +10 -0
- millforge/_forge/clients/base.py +200 -0
- millforge/_forge/context/__init__.py +23 -0
- millforge/_forge/context/manager.py +178 -0
- millforge/_forge/context/strategies.py +335 -0
- millforge/_forge/core/__init__.py +16 -0
- millforge/_forge/core/inference.py +433 -0
- millforge/_forge/core/messages.py +119 -0
- millforge/_forge/core/runner.py +479 -0
- millforge/_forge/core/steps.py +108 -0
- millforge/_forge/core/workflow.py +400 -0
- millforge/_forge/errors.py +222 -0
- millforge/_forge/guardrails/__init__.py +21 -0
- millforge/_forge/guardrails/error_tracker.py +71 -0
- millforge/_forge/guardrails/guardrails.py +194 -0
- millforge/_forge/guardrails/nudge.py +47 -0
- millforge/_forge/guardrails/response_validator.py +119 -0
- millforge/_forge/guardrails/step_enforcer.py +183 -0
- millforge/_forge/prompts/__init__.py +16 -0
- millforge/_forge/prompts/nudges.py +95 -0
- millforge/_forge/prompts/templates.py +285 -0
- millforge/_version.py +3 -0
- millforge/artifacts.py +570 -0
- millforge/base/__init__.py +97 -0
- millforge/base/composition.py +402 -0
- millforge/base/context.py +285 -0
- millforge/base/harness.py +138 -0
- millforge/base/identity.py +465 -0
- millforge/base/options.py +34 -0
- millforge/base/platform.py +17 -0
- millforge/base/prompt.py +317 -0
- millforge/base/runner.py +546 -0
- millforge/compiled_plan.py +970 -0
- millforge/compiler/__init__.py +231 -0
- millforge/compiler/artifact_validation.py +257 -0
- millforge/compiler/canonicalization.py +169 -0
- millforge/compiler/capabilities.py +66 -0
- millforge/compiler/catalogs.py +500 -0
- millforge/compiler/diagnostics.py +491 -0
- millforge/compiler/graph.py +678 -0
- millforge/compiler/lowering.py +198 -0
- millforge/compiler/output.py +692 -0
- millforge/compiler/parsing.py +1424 -0
- millforge/compiler/requests.py +1180 -0
- millforge/compiler/schema_validation.py +272 -0
- millforge/compiler/semantic.py +490 -0
- millforge/compiler/service.py +448 -0
- millforge/compiler/source.py +375 -0
- millforge/compiler/validators.py +184 -0
- millforge/connectors/__init__.py +95 -0
- millforge/connectors/admission.py +801 -0
- millforge/connectors/broker.py +202 -0
- millforge/connectors/contracts.py +1159 -0
- millforge/connectors/diagnostics.py +189 -0
- millforge/connectors/fake.py +66 -0
- millforge/connectors/runtime.py +236 -0
- millforge/contracts.py +2860 -0
- millforge/custom_tools/__init__.py +67 -0
- millforge/custom_tools/compiler.py +724 -0
- millforge/custom_tools/contracts.py +1093 -0
- millforge/custom_tools/diagnostics.py +205 -0
- millforge/eval_artifacts.py +952 -0
- millforge/eval_boundary.py +2435 -0
- millforge/eval_fixtures/__init__.py +1 -0
- millforge/eval_fixtures/default_pack/__init__.py +1 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.bug_diagnosis.traceback.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.direct_edit.import_sort.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.evidence_discipline.no_source_change.v1.json +51 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.false_closure.visible_green.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.multi_file.api_contract.v1.json +54 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.recovery.malformed_artifact.v1.json +54 -0
- millforge/eval_fixtures/default_pack/manifest.json +12 -0
- millforge/eval_modes.py +1282 -0
- millforge/eval_presets.py +1398 -0
- millforge/eval_reports.py +2517 -0
- millforge/eval_suite.py +2429 -0
- millforge/eval_trials.py +2632 -0
- millforge/eval_workflow.py +794 -0
- millforge/exceptions.py +122 -0
- millforge/model_backend.py +2098 -0
- millforge/protocols.py +340 -0
- millforge/py.typed +0 -0
- millforge/runtime.py +1791 -0
- millforge/testing/__init__.py +1089 -0
- millforge/tools/__init__.py +83 -0
- millforge/tools/builtin_runtime.py +1339 -0
- millforge/tools/builtins.py +773 -0
- millforge/tools/execution.py +1545 -0
- millforge/tools/path_policy.py +155 -0
- millforge/tools/pi_compat/PI_LICENSE +21 -0
- millforge/tools/pi_compat/PROVENANCE.json +55 -0
- millforge/tools/pi_compat/UPDATE_POLICY.md +36 -0
- millforge/tools/pi_compat/__init__.py +34 -0
- millforge/tools/pi_compat/contracts.py +49 -0
- millforge/tools/pi_compat/editing.py +390 -0
- millforge/tools/pi_compat/mutations.py +57 -0
- millforge/tools/pi_compat/operations.py +401 -0
- millforge/tools/pi_compat/paths.py +155 -0
- millforge/tools/pi_compat/process.py +1375 -0
- millforge/tools/pi_compat/search.py +738 -0
- millforge/tools/pi_compat/truncation.py +267 -0
- millforge/tools/pi_compat_catalog.py +396 -0
- millforge/tools/pi_compat_runtime.py +460 -0
- millforge/tools/registry.py +553 -0
- millforge/tools/results.py +533 -0
- millforge-0.1.0.dist-info/METADATA +844 -0
- millforge-0.1.0.dist-info/RECORD +116 -0
- millforge-0.1.0.dist-info/WHEEL +4 -0
- millforge-0.1.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,2517 @@
|
|
|
1
|
+
"""Public 08C eval-report, budget, and live-admission contracts."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
import re
|
|
8
|
+
from collections import Counter
|
|
9
|
+
from collections.abc import Mapping, Sequence
|
|
10
|
+
from enum import Enum
|
|
11
|
+
from statistics import median
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
from pydantic import BaseModel, ConfigDict, Field, StrictBool, StrictFloat, StrictInt
|
|
15
|
+
from pydantic import StrictStr, field_validator, model_validator
|
|
16
|
+
|
|
17
|
+
from millforge.eval_suite import (
|
|
18
|
+
EvalCampaignManifest,
|
|
19
|
+
EvalFailureTaxonomyLabel,
|
|
20
|
+
EvalSuiteExecutionMode,
|
|
21
|
+
EvalTaskCategory,
|
|
22
|
+
EvalTrialOutcome,
|
|
23
|
+
)
|
|
24
|
+
from millforge.eval_trials import (
|
|
25
|
+
EvalTrialArmId,
|
|
26
|
+
EvalTrialPlan,
|
|
27
|
+
EvalTrialRecord,
|
|
28
|
+
EvalTrialResumeIndex,
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
EVAL_REPORT_SCHEMA_VERSION = 1
|
|
32
|
+
EVAL_REPORT_HASH_KIND = "eval_report_sha256_v1"
|
|
33
|
+
EVAL_REPORT_JSON_HASH_KIND = "eval_report_json_sha256_v1"
|
|
34
|
+
EVAL_REPORT_MARKDOWN_HASH_KIND = "eval_report_markdown_sha256_v1"
|
|
35
|
+
|
|
36
|
+
_SHA256_RE = re.compile(r"^[0-9a-f]{64}$")
|
|
37
|
+
_UTC_TIMESTAMP_RE = re.compile(r"^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}Z$")
|
|
38
|
+
_ENDPOINT_URL = re.compile(r"https?://|localhost(?::|/|$)|127\.0\.0\.1|0\.0\.0\.0")
|
|
39
|
+
_WINDOWS_ABSOLUTE_PATH = re.compile(
|
|
40
|
+
r"(?:^|[\s\"'`([{<])(?:[A-Za-z]:[\\/]|\\\\)[^\s\"'`)\]}>,]*"
|
|
41
|
+
)
|
|
42
|
+
_POSIX_ABSOLUTE_PATH = re.compile(r"(?:^|[\s\"'`([{<])/(?!/)[^\s\"'`)\]}>,]*")
|
|
43
|
+
_USER_HOME_PATH = re.compile(r"(?:^|[\s\"'`([{<])~(?:/|\\)[^\s\"'`)\]}>,]*")
|
|
44
|
+
_CREDENTIAL_VALUE_PATTERNS = (
|
|
45
|
+
re.compile(r"\bsk-(?:live|proj|test)-[A-Za-z0-9_-]{10,}\b"),
|
|
46
|
+
re.compile(r"\b[rs]k_(?:live|test)_[A-Za-z0-9]{16,}\b"),
|
|
47
|
+
re.compile(r"\b(?:AKIA|ASIA)[0-9A-Z]{16}\b"),
|
|
48
|
+
re.compile(r"\bAIza[0-9A-Za-z_-]{20,}\b"),
|
|
49
|
+
re.compile(r"\bgh[opsu]_[A-Za-z0-9_]{20,}\b"),
|
|
50
|
+
)
|
|
51
|
+
_SECRET_FIELD_MARKERS = (
|
|
52
|
+
"api_key",
|
|
53
|
+
"apikey",
|
|
54
|
+
"auth_header",
|
|
55
|
+
"authorization",
|
|
56
|
+
"bearer",
|
|
57
|
+
"client_secret",
|
|
58
|
+
"credential",
|
|
59
|
+
"password",
|
|
60
|
+
"private_key",
|
|
61
|
+
"secret",
|
|
62
|
+
"access_token",
|
|
63
|
+
"auth_token",
|
|
64
|
+
"refresh_token",
|
|
65
|
+
)
|
|
66
|
+
_DENIED_TEXT_TOKENS = (
|
|
67
|
+
"api_key",
|
|
68
|
+
"authorization:",
|
|
69
|
+
"bearer ",
|
|
70
|
+
"credential",
|
|
71
|
+
"password",
|
|
72
|
+
"secret",
|
|
73
|
+
"access token",
|
|
74
|
+
"auth token",
|
|
75
|
+
"endpoint_url",
|
|
76
|
+
"endpoint url",
|
|
77
|
+
"millrace-agents",
|
|
78
|
+
".millrace",
|
|
79
|
+
"daemon state",
|
|
80
|
+
"private workspace",
|
|
81
|
+
"private runtime",
|
|
82
|
+
"hidden scorer",
|
|
83
|
+
"hidden answer",
|
|
84
|
+
"hidden expected",
|
|
85
|
+
"expected output",
|
|
86
|
+
"scorer_rubric",
|
|
87
|
+
".claude",
|
|
88
|
+
".codex",
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
class _FrozenEvalReportDict(dict[Any, Any]):
|
|
93
|
+
"""Dict-shaped immutable mapping that remains serializable by Pydantic."""
|
|
94
|
+
|
|
95
|
+
def __readonly(self, *args: Any, **kwargs: Any) -> None:
|
|
96
|
+
raise TypeError("eval-report mappings are immutable")
|
|
97
|
+
|
|
98
|
+
__setitem__ = __readonly
|
|
99
|
+
__delitem__ = __readonly
|
|
100
|
+
clear = __readonly
|
|
101
|
+
pop = __readonly
|
|
102
|
+
popitem = __readonly # type: ignore[assignment]
|
|
103
|
+
setdefault = __readonly
|
|
104
|
+
update = __readonly
|
|
105
|
+
__ior__ = __readonly # type: ignore[assignment]
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
class EvalReportContractModel(BaseModel):
|
|
109
|
+
"""Closed, frozen base for public eval-report contracts."""
|
|
110
|
+
|
|
111
|
+
model_config = ConfigDict(extra="forbid", frozen=True, hide_input_in_errors=True)
|
|
112
|
+
|
|
113
|
+
@model_validator(mode="before")
|
|
114
|
+
@classmethod
|
|
115
|
+
def _reject_forbidden_payload(cls, data: Any) -> Any:
|
|
116
|
+
_reject_forbidden_material(data)
|
|
117
|
+
return data
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
class EvalReportPricingClass(str, Enum):
|
|
121
|
+
"""Closed pricing classes for campaign budget accounting."""
|
|
122
|
+
|
|
123
|
+
OFFLINE_ZERO_COST = "offline_zero_cost"
|
|
124
|
+
FREE_TIER = "free_tier"
|
|
125
|
+
PROMOTIONAL_FREE_WINDOW = "promotional_free_window"
|
|
126
|
+
PAID_PROVIDER = "paid_provider"
|
|
127
|
+
LOCAL_METERED = "local_metered"
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
class EvalLiveAdmissionStatus(str, Enum):
|
|
131
|
+
"""Live campaign admission states."""
|
|
132
|
+
|
|
133
|
+
ADMITTED = "admitted"
|
|
134
|
+
DENIED = "denied"
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
class EvalLiveAdmissionDiagnosticCode(str, Enum):
|
|
138
|
+
"""Structured fail-closed live-admission diagnostic codes."""
|
|
139
|
+
|
|
140
|
+
PI_RUNTIME_UNAVAILABLE = "pi_runtime_unavailable"
|
|
141
|
+
MILLFORGE_LIVE_HARNESS_UNAVAILABLE = "millforge_live_harness_unavailable"
|
|
142
|
+
SHARED_BACKEND_CONFIGURATION_MISSING = "shared_backend_configuration_missing"
|
|
143
|
+
FIXTURE_WORKSPACE_LIFECYCLE_UNAVAILABLE = "fixture_workspace_lifecycle_unavailable"
|
|
144
|
+
RESOURCE_ENFORCEMENT_UNAVAILABLE = "resource_enforcement_unavailable"
|
|
145
|
+
BUDGET_POLICY_INVALID = "budget_policy_invalid"
|
|
146
|
+
APPEND_ONLY_STORE_SAFETY_UNPROVEN = "append_only_store_safety_unproven"
|
|
147
|
+
DETERMINISTIC_SCORER_UNAVAILABLE = "deterministic_scorer_unavailable"
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
class EvalBudgetDiagnosticCode(str, Enum):
|
|
151
|
+
"""Structured budget validation diagnostic codes."""
|
|
152
|
+
|
|
153
|
+
MISSING_BUDGET_POLICY = "missing_budget_policy"
|
|
154
|
+
MISSING_LIVE_BUDGET_METADATA = "missing_live_budget_metadata"
|
|
155
|
+
MISSING_TOKEN_CEILING = "missing_token_ceiling"
|
|
156
|
+
MISSING_TRIAL_COUNT_CEILING = "missing_trial_count_ceiling"
|
|
157
|
+
INCOMPLETE_PROMOTIONAL_FREE_WINDOW = "incomplete_promotional_free_window"
|
|
158
|
+
UNFAIR_PAIRED_ARM_RATE_LIMIT = "unfair_paired_arm_rate_limit"
|
|
159
|
+
OFFLINE_POLICY_NOT_ZERO_COST = "offline_policy_not_zero_cost"
|
|
160
|
+
OFFLINE_POLICY_UNBOUNDED = "offline_policy_unbounded"
|
|
161
|
+
SPEND_CEILING_EXCEEDED = "spend_ceiling_exceeded"
|
|
162
|
+
PROMPT_TOKEN_CEILING_EXCEEDED = "prompt_token_ceiling_exceeded"
|
|
163
|
+
COMPLETION_TOKEN_CEILING_EXCEEDED = "completion_token_ceiling_exceeded"
|
|
164
|
+
MODEL_CALL_CEILING_EXCEEDED = "model_call_ceiling_exceeded"
|
|
165
|
+
RETRY_CEILING_EXCEEDED = "retry_ceiling_exceeded"
|
|
166
|
+
WALL_CLOCK_CEILING_EXCEEDED = "wall_clock_ceiling_exceeded"
|
|
167
|
+
TRIAL_COUNT_CEILING_EXCEEDED = "trial_count_ceiling_exceeded"
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
class EvalMetricDenominatorKind(str, Enum):
|
|
171
|
+
"""Metric denominator sources pinned in reports."""
|
|
172
|
+
|
|
173
|
+
PLANNED_TRIALS = "planned_trials"
|
|
174
|
+
APPENDED_RECORDS = "appended_records"
|
|
175
|
+
VALID_RECORDS = "valid_records"
|
|
176
|
+
PAIRED_RECORDS = "paired_records"
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
class EvalReportMetricId(str, Enum):
|
|
180
|
+
"""Closed primary and secondary metric IDs."""
|
|
181
|
+
|
|
182
|
+
VALID_COMPLETION = "valid_completion"
|
|
183
|
+
FALSE_CLOSURE = "false_closure"
|
|
184
|
+
FALSE_SUCCESS = "false_success"
|
|
185
|
+
ARTIFACT_COMPLETE = "artifact_complete"
|
|
186
|
+
CAPABILITY_VIOLATION = "capability_violation"
|
|
187
|
+
CORRECTLY_BLOCKED = "correctly_blocked"
|
|
188
|
+
FALSE_BLOCKED = "false_blocked"
|
|
189
|
+
RUNTIME_FAILURE = "runtime_failure"
|
|
190
|
+
PROVIDER_FAILURE = "provider_failure"
|
|
191
|
+
INVALID_TRIAL = "invalid_trial"
|
|
192
|
+
MISSING_PAIR = "missing_pair"
|
|
193
|
+
PENDING_TRIAL = "pending_trial"
|
|
194
|
+
INCOMPLETE_TRIAL = "incomplete_trial"
|
|
195
|
+
MODEL_CALLS = "model_calls"
|
|
196
|
+
PROMPT_TOKENS = "prompt_tokens"
|
|
197
|
+
COMPLETION_TOKENS = "completion_tokens"
|
|
198
|
+
ESTIMATED_COST = "estimated_cost"
|
|
199
|
+
WALL_CLOCK_SECONDS = "wall_clock_seconds"
|
|
200
|
+
RETRIES = "retries"
|
|
201
|
+
ARTIFACT_COUNT = "artifact_count"
|
|
202
|
+
ARTIFACT_BYTES = "artifact_bytes"
|
|
203
|
+
TURNS = "turns"
|
|
204
|
+
INVALID_TOOL_CALLS = "invalid_tool_calls"
|
|
205
|
+
MALFORMED_TOOL_CALLS = "malformed_tool_calls"
|
|
206
|
+
MALFORMED_ARGUMENTS = "malformed_arguments"
|
|
207
|
+
PREREQUISITE_VIOLATIONS = "prerequisite_violations"
|
|
208
|
+
PREMATURE_TERMINALS = "premature_terminals"
|
|
209
|
+
TOOL_RECOVERIES = "tool_recoveries"
|
|
210
|
+
COMPLETION_IMPROVEMENT = "completion_improvement"
|
|
211
|
+
COST_MULTIPLIER = "cost_multiplier"
|
|
212
|
+
LATENCY_MULTIPLIER = "latency_multiplier"
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
class EvalReportFailureTaxonomyCategory(str, Enum):
|
|
216
|
+
"""Closed public report failure taxonomy."""
|
|
217
|
+
|
|
218
|
+
TASK_MISUNDERSTANDING = "task_misunderstanding"
|
|
219
|
+
WRONG_FILE = "wrong_file"
|
|
220
|
+
UNREAD_BEFORE_EDIT = "unread_before_edit"
|
|
221
|
+
INVALID_PATCH = "invalid_patch"
|
|
222
|
+
TEST_NOT_RUN = "test_not_run"
|
|
223
|
+
TEST_MISREAD = "test_misread"
|
|
224
|
+
MISSING_ARTIFACT = "missing_artifact"
|
|
225
|
+
UNSUPPORTED_SUCCESS_CLAIM = "unsupported_success_claim"
|
|
226
|
+
CHECKER_EVIDENCE_FAILURE = "checker_evidence_failure"
|
|
227
|
+
ARBITER_FALSE_CLOSURE = "arbiter_false_closure"
|
|
228
|
+
PREMATURE_TERMINAL = "premature_terminal"
|
|
229
|
+
TOOL_SCHEMA_FAILURE = "tool_schema_failure"
|
|
230
|
+
TOOL_RECOVERY_FAILURE = "tool_recovery_failure"
|
|
231
|
+
CONTEXT_LOSS = "context_loss"
|
|
232
|
+
BUDGET_EXHAUSTION = "budget_exhaustion"
|
|
233
|
+
PROVIDER_FAILURE = "provider_failure"
|
|
234
|
+
RUNNER_FAILURE = "runner_failure"
|
|
235
|
+
CAPABILITY_VIOLATION = "capability_violation"
|
|
236
|
+
INVALID_TRIAL_INFRASTRUCTURE = "invalid_trial_infrastructure"
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
class EvalReportConfoundId(str, Enum):
|
|
240
|
+
"""Closed confounds that must remain visible in pilot reports."""
|
|
241
|
+
|
|
242
|
+
PI_PROMPT_TOOL_BEHAVIOR = "pi_prompt_tool_behavior"
|
|
243
|
+
MILLFORGE_PROMPT_TOOL_BEHAVIOR = "millforge_prompt_tool_behavior"
|
|
244
|
+
HARNESS_BEHAVIOR = "harness_behavior"
|
|
245
|
+
CONTEXT_PACKING = "context_packing"
|
|
246
|
+
PARSER_FALLBACK = "parser_fallback"
|
|
247
|
+
PROVIDER_NONDETERMINISM = "provider_nondeterminism"
|
|
248
|
+
RATE_LIMITING = "rate_limiting"
|
|
249
|
+
CACHED_PROVIDER_RESPONSES = "cached_provider_responses"
|
|
250
|
+
TOKEN_ACCOUNTING_DIFFERENCES = "token_accounting_differences"
|
|
251
|
+
SAMPLING_PARAMETER_MISMATCH = "sampling_parameter_mismatch"
|
|
252
|
+
OFFLINE_FAKE_LIMITATIONS = "offline_fake_limitations"
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
class EvalDecisionRuleStatus(str, Enum):
|
|
256
|
+
"""Decision-rule evaluation states."""
|
|
257
|
+
|
|
258
|
+
PASSED = "passed"
|
|
259
|
+
FAILED = "failed"
|
|
260
|
+
DESCRIPTIVE_ONLY = "descriptive_only"
|
|
261
|
+
NOT_APPLICABLE = "not_applicable"
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
class EvalDecisionRuleKind(str, Enum):
|
|
265
|
+
"""Closed pre-registered decision-rule contract classes."""
|
|
266
|
+
|
|
267
|
+
MAX_FALSE_CLOSURE_RATE = "max_false_closure_rate"
|
|
268
|
+
MIN_COMPLETION_IMPROVEMENT = "min_completion_improvement"
|
|
269
|
+
MAX_COST_MULTIPLIER = "max_cost_multiplier"
|
|
270
|
+
MAX_LATENCY_MULTIPLIER = "max_latency_multiplier"
|
|
271
|
+
ACCEPTABLE_FALSE_BLOCKED_TRADEOFF = "acceptable_false_blocked_tradeoff"
|
|
272
|
+
SEVERITY_ONE_ABORT_THRESHOLD = "severity_one_abort_threshold"
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
class EvalBudgetDiagnostic(EvalReportContractModel):
|
|
276
|
+
"""One fail-closed budget diagnostic."""
|
|
277
|
+
|
|
278
|
+
diagnostic_code: EvalBudgetDiagnosticCode
|
|
279
|
+
rule_id: StrictStr
|
|
280
|
+
summary: StrictStr
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
class EvalLiveAdmissionDiagnostic(EvalReportContractModel):
|
|
284
|
+
"""One structured live-admission diagnostic."""
|
|
285
|
+
|
|
286
|
+
diagnostic_code: EvalLiveAdmissionDiagnosticCode
|
|
287
|
+
rule_id: StrictStr
|
|
288
|
+
summary: StrictStr
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
class EvalPromotionalFreeWindow(EvalReportContractModel):
|
|
292
|
+
"""Complete metadata required for promotional free execution."""
|
|
293
|
+
|
|
294
|
+
window_id: StrictStr
|
|
295
|
+
source_label: StrictStr
|
|
296
|
+
starts_at: StrictStr
|
|
297
|
+
ends_at: StrictStr
|
|
298
|
+
max_free_usd: StrictFloat = Field(ge=0.0)
|
|
299
|
+
terms_summary: StrictStr
|
|
300
|
+
|
|
301
|
+
@model_validator(mode="after")
|
|
302
|
+
def _window_valid(self) -> EvalPromotionalFreeWindow:
|
|
303
|
+
if not _UTC_TIMESTAMP_RE.fullmatch(self.starts_at):
|
|
304
|
+
raise ValueError("promotional window starts_at must be a UTC timestamp")
|
|
305
|
+
if not _UTC_TIMESTAMP_RE.fullmatch(self.ends_at):
|
|
306
|
+
raise ValueError("promotional window ends_at must be a UTC timestamp")
|
|
307
|
+
if self.ends_at <= self.starts_at:
|
|
308
|
+
raise ValueError("promotional window ends_at must follow starts_at")
|
|
309
|
+
return self
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
class EvalReportRateLimitPolicy(EvalReportContractModel):
|
|
313
|
+
"""Public paired-arm rate-limit and retry/backoff budget policy."""
|
|
314
|
+
|
|
315
|
+
request_rate_per_window: StrictInt = Field(gt=0)
|
|
316
|
+
token_rate_per_window: StrictInt = Field(gt=0)
|
|
317
|
+
concurrent_request_limit: StrictInt = Field(gt=0)
|
|
318
|
+
window_seconds: StrictInt = Field(gt=0)
|
|
319
|
+
max_backoff_seconds: StrictInt = Field(ge=0)
|
|
320
|
+
max_retries_per_trial: StrictInt = Field(ge=0)
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
class EvalReportAbortThresholds(EvalReportContractModel):
|
|
324
|
+
"""Severity-one abort thresholds for a campaign."""
|
|
325
|
+
|
|
326
|
+
max_false_closure_rate: StrictFloat = Field(ge=0.0, le=1.0)
|
|
327
|
+
max_capability_violation_rate: StrictFloat = Field(ge=0.0, le=1.0)
|
|
328
|
+
max_invalid_trial_rate: StrictFloat = Field(ge=0.0, le=1.0)
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
class EvalReportBudgetPolicy(EvalReportContractModel):
|
|
332
|
+
"""Bounded campaign budget policy used by admission and reports."""
|
|
333
|
+
|
|
334
|
+
policy_id: StrictStr
|
|
335
|
+
pricing_class: EvalReportPricingClass
|
|
336
|
+
max_spend_usd: StrictFloat | None = Field(default=None, ge=0.0)
|
|
337
|
+
max_prompt_tokens: StrictInt | None = Field(default=None, ge=0)
|
|
338
|
+
max_completion_tokens: StrictInt | None = Field(default=None, ge=0)
|
|
339
|
+
max_model_calls: StrictInt | None = Field(default=None, ge=0)
|
|
340
|
+
max_retries_per_trial: StrictInt | None = Field(default=None, ge=0)
|
|
341
|
+
max_wall_clock_seconds: StrictInt | None = Field(default=None, ge=0)
|
|
342
|
+
max_trials_per_campaign: StrictInt | None = Field(default=None, ge=0)
|
|
343
|
+
promotional_free_window: EvalPromotionalFreeWindow | None = None
|
|
344
|
+
rate_limit_policy_by_arm: Mapping[EvalTrialArmId, EvalReportRateLimitPolicy] = (
|
|
345
|
+
Field(default_factory=dict)
|
|
346
|
+
)
|
|
347
|
+
abort_thresholds: EvalReportAbortThresholds
|
|
348
|
+
summary: StrictStr
|
|
349
|
+
|
|
350
|
+
@model_validator(mode="after")
|
|
351
|
+
def _policy_valid(self) -> EvalReportBudgetPolicy:
|
|
352
|
+
object.__setattr__(
|
|
353
|
+
self,
|
|
354
|
+
"rate_limit_policy_by_arm",
|
|
355
|
+
_freeze_eval_report_mapping(self.rate_limit_policy_by_arm),
|
|
356
|
+
)
|
|
357
|
+
return self
|
|
358
|
+
|
|
359
|
+
|
|
360
|
+
class EvalBudgetUsageEstimate(EvalReportContractModel):
|
|
361
|
+
"""Deterministic campaign budget consumption estimate."""
|
|
362
|
+
|
|
363
|
+
estimated_spend_usd: StrictFloat = Field(ge=0.0)
|
|
364
|
+
prompt_tokens: StrictInt = Field(ge=0)
|
|
365
|
+
completion_tokens: StrictInt = Field(ge=0)
|
|
366
|
+
model_calls: StrictInt = Field(ge=0)
|
|
367
|
+
retries_per_trial: StrictInt = Field(ge=0)
|
|
368
|
+
wall_clock_seconds: StrictInt = Field(ge=0)
|
|
369
|
+
trial_count: StrictInt = Field(ge=0)
|
|
370
|
+
|
|
371
|
+
|
|
372
|
+
class EvalBudgetValidationResult(EvalReportContractModel):
|
|
373
|
+
"""Fail-closed result for budget policy validation."""
|
|
374
|
+
|
|
375
|
+
valid: StrictBool
|
|
376
|
+
diagnostics: tuple[EvalBudgetDiagnostic, ...] = Field(default_factory=tuple)
|
|
377
|
+
|
|
378
|
+
@model_validator(mode="after")
|
|
379
|
+
def _result_valid(self) -> EvalBudgetValidationResult:
|
|
380
|
+
if self.valid and self.diagnostics:
|
|
381
|
+
raise ValueError("valid budget results must not include diagnostics")
|
|
382
|
+
if not self.valid and not self.diagnostics:
|
|
383
|
+
raise ValueError("invalid budget results require diagnostics")
|
|
384
|
+
return self
|
|
385
|
+
|
|
386
|
+
|
|
387
|
+
class EvalLiveAdmissionResult(EvalReportContractModel):
|
|
388
|
+
"""Structured live or offline campaign admission result."""
|
|
389
|
+
|
|
390
|
+
status: EvalLiveAdmissionStatus
|
|
391
|
+
diagnostics: tuple[EvalLiveAdmissionDiagnostic, ...] = Field(default_factory=tuple)
|
|
392
|
+
budget_result: EvalBudgetValidationResult
|
|
393
|
+
|
|
394
|
+
@model_validator(mode="after")
|
|
395
|
+
def _admission_valid(self) -> EvalLiveAdmissionResult:
|
|
396
|
+
if self.status is EvalLiveAdmissionStatus.ADMITTED and self.diagnostics:
|
|
397
|
+
raise ValueError("admitted campaigns must not include denial diagnostics")
|
|
398
|
+
if self.status is EvalLiveAdmissionStatus.DENIED and not self.diagnostics:
|
|
399
|
+
raise ValueError("denied campaigns require diagnostics")
|
|
400
|
+
return self
|
|
401
|
+
|
|
402
|
+
|
|
403
|
+
class EvalMetricDenominator(EvalReportContractModel):
|
|
404
|
+
"""Explicit metric denominator evidence."""
|
|
405
|
+
|
|
406
|
+
denominator_kind: EvalMetricDenominatorKind
|
|
407
|
+
count: StrictInt = Field(ge=0)
|
|
408
|
+
summary: StrictStr
|
|
409
|
+
|
|
410
|
+
|
|
411
|
+
class EvalMetricValue(EvalReportContractModel):
|
|
412
|
+
"""One metric count/rate, with an optional total value, and denominator."""
|
|
413
|
+
|
|
414
|
+
metric_id: EvalReportMetricId
|
|
415
|
+
count: StrictInt = Field(ge=0)
|
|
416
|
+
denominator: EvalMetricDenominator
|
|
417
|
+
rate: StrictFloat = Field(ge=0.0, le=1.0)
|
|
418
|
+
value: StrictFloat | None = Field(default=None, ge=0.0)
|
|
419
|
+
severity_one: StrictBool = False
|
|
420
|
+
|
|
421
|
+
@model_validator(mode="after")
|
|
422
|
+
def _metric_valid(self) -> EvalMetricValue:
|
|
423
|
+
expected = (
|
|
424
|
+
0.0 if self.denominator.count == 0 else self.count / self.denominator.count
|
|
425
|
+
)
|
|
426
|
+
if abs(self.rate - expected) > 0.000000001:
|
|
427
|
+
raise ValueError("metric rate must match count and denominator")
|
|
428
|
+
if self.count > self.denominator.count:
|
|
429
|
+
raise ValueError("metric count must not exceed denominator")
|
|
430
|
+
return self
|
|
431
|
+
|
|
432
|
+
|
|
433
|
+
class EvalPairedComparison(EvalReportContractModel):
|
|
434
|
+
"""Per-metric paired comparison across the two admitted arms."""
|
|
435
|
+
|
|
436
|
+
metric_id: EvalReportMetricId
|
|
437
|
+
left_arm_id: EvalTrialArmId
|
|
438
|
+
right_arm_id: EvalTrialArmId
|
|
439
|
+
left_count: StrictInt = Field(ge=0)
|
|
440
|
+
right_count: StrictInt = Field(ge=0)
|
|
441
|
+
paired_denominator: StrictInt = Field(ge=0)
|
|
442
|
+
difference: StrictInt
|
|
443
|
+
missing_pair_count: StrictInt = Field(ge=0)
|
|
444
|
+
|
|
445
|
+
|
|
446
|
+
class EvalWilsonScoreInterval(EvalReportContractModel):
|
|
447
|
+
"""Wilson score interval emitted only when sample-size rules allow it."""
|
|
448
|
+
|
|
449
|
+
successes: StrictInt = Field(ge=0)
|
|
450
|
+
total: StrictInt = Field(gt=0)
|
|
451
|
+
confidence_level: StrictFloat = Field(gt=0.0, lt=1.0)
|
|
452
|
+
lower: StrictFloat = Field(ge=0.0, le=1.0)
|
|
453
|
+
upper: StrictFloat = Field(ge=0.0, le=1.0)
|
|
454
|
+
|
|
455
|
+
@model_validator(mode="after")
|
|
456
|
+
def _interval_valid(self) -> EvalWilsonScoreInterval:
|
|
457
|
+
if self.successes > self.total:
|
|
458
|
+
raise ValueError("Wilson successes must not exceed total")
|
|
459
|
+
if self.lower > self.upper:
|
|
460
|
+
raise ValueError("Wilson interval lower must not exceed upper")
|
|
461
|
+
return self
|
|
462
|
+
|
|
463
|
+
|
|
464
|
+
class EvalDistributionSummary(EvalReportContractModel):
|
|
465
|
+
"""Descriptive distribution summary for cost and latency-style values."""
|
|
466
|
+
|
|
467
|
+
statistic_id: StrictStr
|
|
468
|
+
sample_count: StrictInt = Field(ge=0)
|
|
469
|
+
raw_values: tuple[StrictFloat, ...] = Field(default_factory=tuple)
|
|
470
|
+
median: StrictFloat | None = None
|
|
471
|
+
p90: StrictFloat | None = None
|
|
472
|
+
p95: StrictFloat | None = None
|
|
473
|
+
descriptive_only: StrictBool = True
|
|
474
|
+
|
|
475
|
+
@model_validator(mode="after")
|
|
476
|
+
def _distribution_valid(self) -> EvalDistributionSummary:
|
|
477
|
+
if self.sample_count != len(self.raw_values):
|
|
478
|
+
raise ValueError("distribution sample_count must match raw_values")
|
|
479
|
+
if self.sample_count == 0 and any(
|
|
480
|
+
value is not None for value in (self.median, self.p90, self.p95)
|
|
481
|
+
):
|
|
482
|
+
raise ValueError("empty distributions must not include percentiles")
|
|
483
|
+
return self
|
|
484
|
+
|
|
485
|
+
|
|
486
|
+
class EvalReportStatisticalSummary(EvalReportContractModel):
|
|
487
|
+
"""Small-N descriptive statistics and eligibility diagnostics."""
|
|
488
|
+
|
|
489
|
+
metric_id: EvalReportMetricId
|
|
490
|
+
raw_count: StrictInt = Field(ge=0)
|
|
491
|
+
denominator_count: StrictInt = Field(ge=0)
|
|
492
|
+
rate: StrictFloat = Field(ge=0.0, le=1.0)
|
|
493
|
+
wilson_interval: EvalWilsonScoreInterval | None = None
|
|
494
|
+
paired_differences: tuple[StrictInt, ...] = Field(default_factory=tuple)
|
|
495
|
+
distributions: tuple[EvalDistributionSummary, ...] = Field(default_factory=tuple)
|
|
496
|
+
diagnostic: StrictStr
|
|
497
|
+
descriptive_only: StrictBool = True
|
|
498
|
+
|
|
499
|
+
@model_validator(mode="after")
|
|
500
|
+
def _statistical_summary_valid(self) -> EvalReportStatisticalSummary:
|
|
501
|
+
expected = (
|
|
502
|
+
0.0
|
|
503
|
+
if self.denominator_count == 0
|
|
504
|
+
else self.raw_count / self.denominator_count
|
|
505
|
+
)
|
|
506
|
+
if abs(self.rate - expected) > 0.000000001:
|
|
507
|
+
raise ValueError("statistical summary rate must match raw counts")
|
|
508
|
+
if self.wilson_interval is not None and self.descriptive_only:
|
|
509
|
+
raise ValueError("Wilson-eligible summaries are not descriptive-only")
|
|
510
|
+
if not self.diagnostic.strip():
|
|
511
|
+
raise ValueError("statistical summaries require diagnostics")
|
|
512
|
+
return self
|
|
513
|
+
|
|
514
|
+
|
|
515
|
+
class EvalTaskSummary(EvalReportContractModel):
|
|
516
|
+
"""Per-task report summary."""
|
|
517
|
+
|
|
518
|
+
fixture_id: StrictStr
|
|
519
|
+
trial_index: StrictInt = Field(ge=0)
|
|
520
|
+
category: StrictStr
|
|
521
|
+
metrics: tuple[EvalMetricValue, ...]
|
|
522
|
+
|
|
523
|
+
|
|
524
|
+
class EvalArmSummary(EvalReportContractModel):
|
|
525
|
+
"""Per-arm report summary with explicit arm-local denominators."""
|
|
526
|
+
|
|
527
|
+
arm_id: EvalTrialArmId
|
|
528
|
+
metrics: tuple[EvalMetricValue, ...]
|
|
529
|
+
|
|
530
|
+
|
|
531
|
+
class EvalCategorySummary(EvalReportContractModel):
|
|
532
|
+
"""Per-category aggregate report summary."""
|
|
533
|
+
|
|
534
|
+
category: EvalTaskCategory | StrictStr
|
|
535
|
+
metrics: tuple[EvalMetricValue, ...]
|
|
536
|
+
|
|
537
|
+
|
|
538
|
+
class EvalFailureTaxonomyAssignment(EvalReportContractModel):
|
|
539
|
+
"""Optional manual taxonomy assignment that cannot affect scorer success."""
|
|
540
|
+
|
|
541
|
+
primary_category: EvalReportFailureTaxonomyCategory
|
|
542
|
+
contributing_categories: tuple[EvalReportFailureTaxonomyCategory, ...] = Field(
|
|
543
|
+
default_factory=tuple
|
|
544
|
+
)
|
|
545
|
+
explanation: StrictStr
|
|
546
|
+
category_explanations: Mapping[EvalReportFailureTaxonomyCategory, StrictStr] = (
|
|
547
|
+
Field(default_factory=dict)
|
|
548
|
+
)
|
|
549
|
+
|
|
550
|
+
@field_validator("explanation")
|
|
551
|
+
@classmethod
|
|
552
|
+
def _explanation_required(cls, value: str) -> str:
|
|
553
|
+
if not value.strip():
|
|
554
|
+
raise ValueError("manual taxonomy assignments require an explanation")
|
|
555
|
+
return value
|
|
556
|
+
|
|
557
|
+
@model_validator(mode="after")
|
|
558
|
+
def _manual_assignment_valid(self) -> EvalFailureTaxonomyAssignment:
|
|
559
|
+
categories = (self.primary_category, *self.contributing_categories)
|
|
560
|
+
if len(set(categories)) != len(categories):
|
|
561
|
+
raise ValueError("manual taxonomy categories must be unique")
|
|
562
|
+
explanations = dict(self.category_explanations)
|
|
563
|
+
if not explanations:
|
|
564
|
+
explanations = {category: self.explanation for category in categories}
|
|
565
|
+
missing = [category for category in categories if category not in explanations]
|
|
566
|
+
blank = [
|
|
567
|
+
category
|
|
568
|
+
for category, category_explanation in explanations.items()
|
|
569
|
+
if not category_explanation.strip()
|
|
570
|
+
]
|
|
571
|
+
if missing or blank:
|
|
572
|
+
raise ValueError("each manual taxonomy category requires an explanation")
|
|
573
|
+
object.__setattr__(
|
|
574
|
+
self,
|
|
575
|
+
"category_explanations",
|
|
576
|
+
_freeze_eval_report_mapping(explanations),
|
|
577
|
+
)
|
|
578
|
+
return self
|
|
579
|
+
|
|
580
|
+
|
|
581
|
+
class EvalFailureTaxonomySummary(EvalReportContractModel):
|
|
582
|
+
"""Closed taxonomy rollup for report data."""
|
|
583
|
+
|
|
584
|
+
category: EvalReportFailureTaxonomyCategory
|
|
585
|
+
count: StrictInt = Field(ge=0)
|
|
586
|
+
examples: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
587
|
+
|
|
588
|
+
|
|
589
|
+
class EvalInvalidTrialSummary(EvalReportContractModel):
|
|
590
|
+
"""Top-line invalid-trial visibility."""
|
|
591
|
+
|
|
592
|
+
invalid_trial_count: StrictInt = Field(ge=0)
|
|
593
|
+
appended_record_count: StrictInt = Field(ge=0)
|
|
594
|
+
invalid_trial_rate: StrictFloat = Field(ge=0.0, le=1.0)
|
|
595
|
+
diagnostics: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
596
|
+
|
|
597
|
+
@model_validator(mode="after")
|
|
598
|
+
def _invalid_valid(self) -> EvalInvalidTrialSummary:
|
|
599
|
+
expected = (
|
|
600
|
+
0.0
|
|
601
|
+
if self.appended_record_count == 0
|
|
602
|
+
else self.invalid_trial_count / self.appended_record_count
|
|
603
|
+
)
|
|
604
|
+
if abs(self.invalid_trial_rate - expected) > 0.000000001:
|
|
605
|
+
raise ValueError("invalid trial rate must match counts")
|
|
606
|
+
return self
|
|
607
|
+
|
|
608
|
+
|
|
609
|
+
class EvalConfoundEntry(EvalReportContractModel):
|
|
610
|
+
"""Visible confound entry in JSON and Markdown reports."""
|
|
611
|
+
|
|
612
|
+
confound_id: EvalReportConfoundId
|
|
613
|
+
summary: StrictStr
|
|
614
|
+
affects_claims: StrictBool = True
|
|
615
|
+
|
|
616
|
+
|
|
617
|
+
class EvalDecisionRule(EvalReportContractModel):
|
|
618
|
+
"""Pre-registered report decision rule."""
|
|
619
|
+
|
|
620
|
+
rule_id: StrictStr
|
|
621
|
+
rule_kind: EvalDecisionRuleKind
|
|
622
|
+
summary: StrictStr
|
|
623
|
+
metric_id: EvalReportMetricId
|
|
624
|
+
threshold: StrictFloat = Field(ge=0.0)
|
|
625
|
+
status: EvalDecisionRuleStatus
|
|
626
|
+
observed_value: StrictFloat | None = Field(default=None, ge=0.0)
|
|
627
|
+
diagnostic: StrictStr | None = None
|
|
628
|
+
|
|
629
|
+
@model_validator(mode="after")
|
|
630
|
+
def _rule_valid(self) -> EvalDecisionRule:
|
|
631
|
+
if (
|
|
632
|
+
self.status is EvalDecisionRuleStatus.DESCRIPTIVE_ONLY
|
|
633
|
+
and not self.diagnostic
|
|
634
|
+
):
|
|
635
|
+
raise ValueError("descriptive-only decision rules require diagnostics")
|
|
636
|
+
return self
|
|
637
|
+
|
|
638
|
+
|
|
639
|
+
class EvalReportReproducibilityHashes(EvalReportContractModel):
|
|
640
|
+
"""Hash references that make a report reproducible."""
|
|
641
|
+
|
|
642
|
+
campaign_manifest_hash: StrictStr
|
|
643
|
+
plan_hashes: tuple[StrictStr, ...]
|
|
644
|
+
resume_index_hash: StrictStr | None = None
|
|
645
|
+
record_hashes: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
646
|
+
report_input_hash: StrictStr
|
|
647
|
+
|
|
648
|
+
@model_validator(mode="after")
|
|
649
|
+
def _hashes_valid(self) -> EvalReportReproducibilityHashes:
|
|
650
|
+
for digest in (
|
|
651
|
+
(self.campaign_manifest_hash, self.report_input_hash)
|
|
652
|
+
+ self.plan_hashes
|
|
653
|
+
+ self.record_hashes
|
|
654
|
+
):
|
|
655
|
+
_validate_sha256(digest)
|
|
656
|
+
if self.resume_index_hash is not None:
|
|
657
|
+
_validate_sha256(self.resume_index_hash)
|
|
658
|
+
return self
|
|
659
|
+
|
|
660
|
+
|
|
661
|
+
class EvalReportPayload(EvalReportContractModel):
|
|
662
|
+
"""Deterministic JSON report payload."""
|
|
663
|
+
|
|
664
|
+
schema_version: StrictInt = EVAL_REPORT_SCHEMA_VERSION
|
|
665
|
+
report_id: StrictStr
|
|
666
|
+
campaign_id: StrictStr
|
|
667
|
+
generated_at: StrictStr
|
|
668
|
+
admission: EvalLiveAdmissionResult
|
|
669
|
+
arms: tuple[EvalTrialArmId, EvalTrialArmId]
|
|
670
|
+
controlled_variables: Mapping[StrictStr, StrictStr] = Field(default_factory=dict)
|
|
671
|
+
budget_policy: EvalReportBudgetPolicy
|
|
672
|
+
budget_usage: EvalBudgetUsageEstimate
|
|
673
|
+
primary_metrics: tuple[EvalMetricValue, ...]
|
|
674
|
+
arm_summaries: tuple[EvalArmSummary, ...] = Field(default_factory=tuple)
|
|
675
|
+
paired_comparisons: tuple[EvalPairedComparison, ...] = Field(default_factory=tuple)
|
|
676
|
+
task_summaries: tuple[EvalTaskSummary, ...] = Field(default_factory=tuple)
|
|
677
|
+
category_summaries: tuple[EvalCategorySummary, ...] = Field(default_factory=tuple)
|
|
678
|
+
taxonomy_summaries: tuple[EvalFailureTaxonomySummary, ...] = Field(
|
|
679
|
+
default_factory=tuple
|
|
680
|
+
)
|
|
681
|
+
invalid_trials: EvalInvalidTrialSummary
|
|
682
|
+
statistical_summaries: tuple[EvalReportStatisticalSummary, ...] = Field(
|
|
683
|
+
default_factory=tuple
|
|
684
|
+
)
|
|
685
|
+
confounds: tuple[EvalConfoundEntry, ...]
|
|
686
|
+
decision_rules: tuple[EvalDecisionRule, ...]
|
|
687
|
+
reproducibility_hashes: EvalReportReproducibilityHashes
|
|
688
|
+
claim_boundaries: tuple[StrictStr, ...]
|
|
689
|
+
report_hash_kind: StrictStr = EVAL_REPORT_HASH_KIND
|
|
690
|
+
report_hash: StrictStr
|
|
691
|
+
|
|
692
|
+
@model_validator(mode="after")
|
|
693
|
+
def _payload_valid(self) -> EvalReportPayload:
|
|
694
|
+
if self.schema_version != EVAL_REPORT_SCHEMA_VERSION:
|
|
695
|
+
raise ValueError("unsupported eval-report schema_version")
|
|
696
|
+
if not _UTC_TIMESTAMP_RE.fullmatch(self.generated_at):
|
|
697
|
+
raise ValueError("report generated_at must be a UTC timestamp")
|
|
698
|
+
if len(set(self.arms)) != 2:
|
|
699
|
+
raise ValueError("reports require two distinct arms")
|
|
700
|
+
object.__setattr__(
|
|
701
|
+
self,
|
|
702
|
+
"controlled_variables",
|
|
703
|
+
_freeze_eval_report_mapping(self.controlled_variables),
|
|
704
|
+
)
|
|
705
|
+
if not self.claim_boundaries:
|
|
706
|
+
raise ValueError("reports require explicit claim boundaries")
|
|
707
|
+
if not self.confounds:
|
|
708
|
+
raise ValueError("reports require explicit confounds")
|
|
709
|
+
if self.report_hash_kind != EVAL_REPORT_HASH_KIND:
|
|
710
|
+
raise ValueError("unsupported report hash kind")
|
|
711
|
+
_validate_sha256(self.report_hash)
|
|
712
|
+
expected = calculate_eval_report_hash(self)
|
|
713
|
+
if self.report_hash != expected:
|
|
714
|
+
raise ValueError("report_hash does not match payload")
|
|
715
|
+
return self
|
|
716
|
+
|
|
717
|
+
|
|
718
|
+
class EvalMarkdownReport(EvalReportContractModel):
|
|
719
|
+
"""Deterministic Markdown report content."""
|
|
720
|
+
|
|
721
|
+
schema_version: StrictInt = EVAL_REPORT_SCHEMA_VERSION
|
|
722
|
+
report_id: StrictStr
|
|
723
|
+
content: StrictStr
|
|
724
|
+
content_hash_kind: StrictStr = EVAL_REPORT_MARKDOWN_HASH_KIND
|
|
725
|
+
content_hash: StrictStr
|
|
726
|
+
|
|
727
|
+
@model_validator(mode="after")
|
|
728
|
+
def _markdown_valid(self) -> EvalMarkdownReport:
|
|
729
|
+
if self.schema_version != EVAL_REPORT_SCHEMA_VERSION:
|
|
730
|
+
raise ValueError("unsupported markdown report schema_version")
|
|
731
|
+
if self.content_hash_kind != EVAL_REPORT_MARKDOWN_HASH_KIND:
|
|
732
|
+
raise ValueError("unsupported markdown report hash kind")
|
|
733
|
+
_validate_sha256(self.content_hash)
|
|
734
|
+
if (
|
|
735
|
+
self.content_hash
|
|
736
|
+
!= hashlib.sha256(self.content.encode("utf-8")).hexdigest()
|
|
737
|
+
):
|
|
738
|
+
raise ValueError("markdown content hash does not match content")
|
|
739
|
+
return self
|
|
740
|
+
|
|
741
|
+
|
|
742
|
+
def validate_eval_budget_policy(
|
|
743
|
+
policy: EvalReportBudgetPolicy | Mapping[str, Any] | None,
|
|
744
|
+
*,
|
|
745
|
+
campaign_manifest: EvalCampaignManifest,
|
|
746
|
+
usage: EvalBudgetUsageEstimate | Mapping[str, Any] | None = None,
|
|
747
|
+
) -> EvalBudgetValidationResult:
|
|
748
|
+
"""Validate a campaign budget policy and fail closed with diagnostics."""
|
|
749
|
+
diagnostics: list[EvalBudgetDiagnostic] = []
|
|
750
|
+
if policy is None:
|
|
751
|
+
return EvalBudgetValidationResult(
|
|
752
|
+
valid=False,
|
|
753
|
+
diagnostics=(
|
|
754
|
+
_budget_diagnostic(
|
|
755
|
+
EvalBudgetDiagnosticCode.MISSING_BUDGET_POLICY,
|
|
756
|
+
"eval_reports.budget.required",
|
|
757
|
+
"Budget policy metadata is required for admission.",
|
|
758
|
+
),
|
|
759
|
+
),
|
|
760
|
+
)
|
|
761
|
+
try:
|
|
762
|
+
valid_policy = EvalReportBudgetPolicy.model_validate(policy)
|
|
763
|
+
except ValueError as exc:
|
|
764
|
+
return EvalBudgetValidationResult(
|
|
765
|
+
valid=False,
|
|
766
|
+
diagnostics=(
|
|
767
|
+
_budget_diagnostic(
|
|
768
|
+
EvalBudgetDiagnosticCode.MISSING_LIVE_BUDGET_METADATA,
|
|
769
|
+
"eval_reports.budget.model_validate",
|
|
770
|
+
f"Budget policy metadata is invalid: {exc}",
|
|
771
|
+
),
|
|
772
|
+
),
|
|
773
|
+
)
|
|
774
|
+
usage_estimate = (
|
|
775
|
+
EvalBudgetUsageEstimate.model_validate(usage)
|
|
776
|
+
if usage is not None
|
|
777
|
+
else EvalBudgetUsageEstimate(
|
|
778
|
+
estimated_spend_usd=0.0,
|
|
779
|
+
prompt_tokens=0,
|
|
780
|
+
completion_tokens=0,
|
|
781
|
+
model_calls=0,
|
|
782
|
+
retries_per_trial=0,
|
|
783
|
+
wall_clock_seconds=0,
|
|
784
|
+
trial_count=0,
|
|
785
|
+
)
|
|
786
|
+
)
|
|
787
|
+
if (
|
|
788
|
+
valid_policy.max_prompt_tokens is None
|
|
789
|
+
or valid_policy.max_completion_tokens is None
|
|
790
|
+
):
|
|
791
|
+
diagnostics.append(
|
|
792
|
+
_budget_diagnostic(
|
|
793
|
+
EvalBudgetDiagnosticCode.MISSING_TOKEN_CEILING,
|
|
794
|
+
"eval_reports.budget.token_ceilings",
|
|
795
|
+
"Budget policy must declare prompt and completion token ceilings.",
|
|
796
|
+
)
|
|
797
|
+
)
|
|
798
|
+
if valid_policy.max_trials_per_campaign is None:
|
|
799
|
+
diagnostics.append(
|
|
800
|
+
_budget_diagnostic(
|
|
801
|
+
EvalBudgetDiagnosticCode.MISSING_TRIAL_COUNT_CEILING,
|
|
802
|
+
"eval_reports.budget.trial_ceiling",
|
|
803
|
+
"Budget policy must declare a trial-count ceiling.",
|
|
804
|
+
)
|
|
805
|
+
)
|
|
806
|
+
if campaign_manifest.execution_mode is EvalSuiteExecutionMode.OFFLINE_FAKE:
|
|
807
|
+
diagnostics.extend(_offline_budget_diagnostics(valid_policy))
|
|
808
|
+
else:
|
|
809
|
+
diagnostics.extend(_live_budget_metadata_diagnostics(valid_policy))
|
|
810
|
+
diagnostics.extend(_budget_usage_diagnostics(valid_policy, usage_estimate))
|
|
811
|
+
return EvalBudgetValidationResult(
|
|
812
|
+
valid=not diagnostics,
|
|
813
|
+
diagnostics=tuple(diagnostics),
|
|
814
|
+
)
|
|
815
|
+
|
|
816
|
+
|
|
817
|
+
def admit_eval_report_campaign(
|
|
818
|
+
campaign_manifest: EvalCampaignManifest,
|
|
819
|
+
*,
|
|
820
|
+
budget_policy: EvalReportBudgetPolicy | Mapping[str, Any] | None,
|
|
821
|
+
usage: EvalBudgetUsageEstimate | Mapping[str, Any] | None = None,
|
|
822
|
+
) -> EvalLiveAdmissionResult:
|
|
823
|
+
"""Return deterministic live/offline admission with structured diagnostics."""
|
|
824
|
+
budget_result = validate_eval_budget_policy(
|
|
825
|
+
budget_policy,
|
|
826
|
+
campaign_manifest=campaign_manifest,
|
|
827
|
+
usage=usage,
|
|
828
|
+
)
|
|
829
|
+
if campaign_manifest.execution_mode is EvalSuiteExecutionMode.OFFLINE_FAKE:
|
|
830
|
+
diagnostics = (
|
|
831
|
+
()
|
|
832
|
+
if budget_result.valid
|
|
833
|
+
else (
|
|
834
|
+
EvalLiveAdmissionDiagnostic(
|
|
835
|
+
diagnostic_code=EvalLiveAdmissionDiagnosticCode.BUDGET_POLICY_INVALID,
|
|
836
|
+
rule_id="eval_reports.admission.offline_budget_policy",
|
|
837
|
+
summary="Offline fake admission requires a complete zero-cost bounded budget policy.",
|
|
838
|
+
),
|
|
839
|
+
)
|
|
840
|
+
)
|
|
841
|
+
return EvalLiveAdmissionResult(
|
|
842
|
+
status=EvalLiveAdmissionStatus.ADMITTED
|
|
843
|
+
if budget_result.valid
|
|
844
|
+
else EvalLiveAdmissionStatus.DENIED,
|
|
845
|
+
diagnostics=diagnostics,
|
|
846
|
+
budget_result=budget_result,
|
|
847
|
+
)
|
|
848
|
+
return EvalLiveAdmissionResult(
|
|
849
|
+
status=EvalLiveAdmissionStatus.DENIED,
|
|
850
|
+
diagnostics=_live_unresolved_dependency_diagnostics(budget_result),
|
|
851
|
+
budget_result=budget_result,
|
|
852
|
+
)
|
|
853
|
+
|
|
854
|
+
|
|
855
|
+
def build_eval_report_payload(
|
|
856
|
+
*,
|
|
857
|
+
report_id: str,
|
|
858
|
+
campaign_manifest: EvalCampaignManifest,
|
|
859
|
+
plans: Sequence[EvalTrialPlan],
|
|
860
|
+
records: Sequence[EvalTrialRecord],
|
|
861
|
+
budget_policy: EvalReportBudgetPolicy,
|
|
862
|
+
usage: EvalBudgetUsageEstimate | None = None,
|
|
863
|
+
resume_index: EvalTrialResumeIndex | None = None,
|
|
864
|
+
generated_at: str = "1970-01-01T00:00:00Z",
|
|
865
|
+
decision_rules: Sequence[EvalDecisionRule] = (),
|
|
866
|
+
manual_taxonomy: Mapping[str, EvalFailureTaxonomyAssignment | Mapping[str, Any]]
|
|
867
|
+
| None = None,
|
|
868
|
+
) -> EvalReportPayload:
|
|
869
|
+
"""Build a deterministic pilot report from existing offline contracts."""
|
|
870
|
+
_validate_report_inputs(campaign_manifest, plans, records, resume_index)
|
|
871
|
+
usage = usage or _default_budget_usage_estimate(plans=plans, records=records)
|
|
872
|
+
admission = admit_eval_report_campaign(
|
|
873
|
+
campaign_manifest,
|
|
874
|
+
budget_policy=budget_policy,
|
|
875
|
+
usage=usage,
|
|
876
|
+
)
|
|
877
|
+
manual_taxonomy = _validate_manual_taxonomy_assignments(manual_taxonomy or {})
|
|
878
|
+
metrics = _primary_metrics(
|
|
879
|
+
plans=plans,
|
|
880
|
+
records=records,
|
|
881
|
+
resume_index=resume_index,
|
|
882
|
+
usage=usage,
|
|
883
|
+
)
|
|
884
|
+
input_hash = _report_input_hash(campaign_manifest, plans, records, resume_index)
|
|
885
|
+
payload = EvalReportPayload.model_construct(
|
|
886
|
+
schema_version=EVAL_REPORT_SCHEMA_VERSION,
|
|
887
|
+
report_id=report_id,
|
|
888
|
+
campaign_id=campaign_manifest.campaign_id,
|
|
889
|
+
generated_at=generated_at,
|
|
890
|
+
admission=admission,
|
|
891
|
+
arms=(EvalTrialArmId.EVAL_SMALL_PI, EvalTrialArmId.EVAL_SMALL_MILLFORGE),
|
|
892
|
+
controlled_variables={
|
|
893
|
+
"model_manifest_hash": campaign_manifest.model_manifest_hash,
|
|
894
|
+
"workflow_graph_hash": campaign_manifest.workflow_graph_hash,
|
|
895
|
+
"fixture_pack_hash": campaign_manifest.fixture_pack_hash,
|
|
896
|
+
"scorer_version": campaign_manifest.scorer_version,
|
|
897
|
+
},
|
|
898
|
+
budget_policy=budget_policy,
|
|
899
|
+
budget_usage=usage,
|
|
900
|
+
primary_metrics=metrics,
|
|
901
|
+
arm_summaries=_arm_summaries(plans=plans, records=records),
|
|
902
|
+
paired_comparisons=_paired_comparisons(plans=plans, records=records),
|
|
903
|
+
task_summaries=_task_summaries(plans=plans, records=records),
|
|
904
|
+
category_summaries=_category_summaries(plans=plans, records=records),
|
|
905
|
+
taxonomy_summaries=_taxonomy_summaries(records, manual_taxonomy),
|
|
906
|
+
invalid_trials=_invalid_trial_summary(records),
|
|
907
|
+
statistical_summaries=_statistical_summaries(
|
|
908
|
+
metrics=metrics,
|
|
909
|
+
paired_comparisons=_paired_comparisons(plans=plans, records=records),
|
|
910
|
+
usage=usage,
|
|
911
|
+
),
|
|
912
|
+
confounds=default_eval_report_confounds(
|
|
913
|
+
offline_fake=campaign_manifest.execution_mode
|
|
914
|
+
is EvalSuiteExecutionMode.OFFLINE_FAKE
|
|
915
|
+
),
|
|
916
|
+
decision_rules=tuple(decision_rules)
|
|
917
|
+
or default_eval_report_decision_rules(metrics),
|
|
918
|
+
reproducibility_hashes=EvalReportReproducibilityHashes(
|
|
919
|
+
campaign_manifest_hash=campaign_manifest.campaign_manifest_hash,
|
|
920
|
+
plan_hashes=tuple(plan.plan_hash for plan in plans),
|
|
921
|
+
resume_index_hash=resume_index.resume_index_hash
|
|
922
|
+
if resume_index is not None
|
|
923
|
+
else None,
|
|
924
|
+
record_hashes=tuple(record.record_hash for record in records),
|
|
925
|
+
report_input_hash=input_hash,
|
|
926
|
+
),
|
|
927
|
+
claim_boundaries=(
|
|
928
|
+
"Offline fake reports are contract and harness-surface evidence only.",
|
|
929
|
+
"No Pi-vs-Millforge model-performance conclusion can be drawn.",
|
|
930
|
+
"Small pilot samples are descriptive unless a decision rule says otherwise.",
|
|
931
|
+
),
|
|
932
|
+
report_hash_kind=EVAL_REPORT_HASH_KIND,
|
|
933
|
+
report_hash="0" * 64,
|
|
934
|
+
)
|
|
935
|
+
return EvalReportPayload.model_validate(
|
|
936
|
+
payload.model_copy(update={"report_hash": calculate_eval_report_hash(payload)})
|
|
937
|
+
)
|
|
938
|
+
|
|
939
|
+
|
|
940
|
+
def render_eval_markdown_report(payload: EvalReportPayload) -> EvalMarkdownReport:
|
|
941
|
+
"""Render a deterministic human-readable Markdown report."""
|
|
942
|
+
primary_metric_ids = {
|
|
943
|
+
EvalReportMetricId.VALID_COMPLETION,
|
|
944
|
+
EvalReportMetricId.FALSE_CLOSURE,
|
|
945
|
+
EvalReportMetricId.FALSE_SUCCESS,
|
|
946
|
+
EvalReportMetricId.ARTIFACT_COMPLETE,
|
|
947
|
+
EvalReportMetricId.CAPABILITY_VIOLATION,
|
|
948
|
+
EvalReportMetricId.CORRECTLY_BLOCKED,
|
|
949
|
+
EvalReportMetricId.FALSE_BLOCKED,
|
|
950
|
+
EvalReportMetricId.RUNTIME_FAILURE,
|
|
951
|
+
EvalReportMetricId.PROVIDER_FAILURE,
|
|
952
|
+
EvalReportMetricId.INVALID_TRIAL,
|
|
953
|
+
}
|
|
954
|
+
primary_metrics = tuple(
|
|
955
|
+
metric
|
|
956
|
+
for metric in payload.primary_metrics
|
|
957
|
+
if metric.metric_id in primary_metric_ids
|
|
958
|
+
)
|
|
959
|
+
secondary_metrics = tuple(
|
|
960
|
+
metric
|
|
961
|
+
for metric in payload.primary_metrics
|
|
962
|
+
if metric.metric_id not in primary_metric_ids
|
|
963
|
+
)
|
|
964
|
+
lines = [
|
|
965
|
+
f"# Eval Report {payload.report_id}",
|
|
966
|
+
"",
|
|
967
|
+
f"- Campaign: {payload.campaign_id}",
|
|
968
|
+
f"- Admission status: {payload.admission.status.value}",
|
|
969
|
+
f"- Arms: {payload.arms[0].value}, {payload.arms[1].value}",
|
|
970
|
+
"",
|
|
971
|
+
"## Controlled Variables",
|
|
972
|
+
*[
|
|
973
|
+
f"- {key}: {payload.controlled_variables[key]}"
|
|
974
|
+
for key in sorted(payload.controlled_variables)
|
|
975
|
+
],
|
|
976
|
+
"",
|
|
977
|
+
"## Budget Summary",
|
|
978
|
+
f"- Policy: {payload.budget_policy.policy_id}",
|
|
979
|
+
f"- Pricing class: {payload.budget_policy.pricing_class.value}",
|
|
980
|
+
f"- Estimated spend USD: {payload.budget_usage.estimated_spend_usd:.6f}",
|
|
981
|
+
f"- Prompt tokens: {payload.budget_usage.prompt_tokens}",
|
|
982
|
+
f"- Completion tokens: {payload.budget_usage.completion_tokens}",
|
|
983
|
+
f"- Model calls: {payload.budget_usage.model_calls}",
|
|
984
|
+
f"- Trial count: {payload.budget_usage.trial_count}",
|
|
985
|
+
"",
|
|
986
|
+
"## Claim Boundary",
|
|
987
|
+
*[f"- {boundary}" for boundary in payload.claim_boundaries],
|
|
988
|
+
"",
|
|
989
|
+
"## Primary Metrics",
|
|
990
|
+
*[_markdown_metric_line(metric) for metric in primary_metrics],
|
|
991
|
+
*(
|
|
992
|
+
(
|
|
993
|
+
"",
|
|
994
|
+
"## Secondary Metrics",
|
|
995
|
+
*[_markdown_metric_line(metric) for metric in secondary_metrics],
|
|
996
|
+
)
|
|
997
|
+
if secondary_metrics
|
|
998
|
+
else ()
|
|
999
|
+
),
|
|
1000
|
+
"",
|
|
1001
|
+
"## Severity-One Outcomes",
|
|
1002
|
+
*[
|
|
1003
|
+
f"- {metric.metric_id.value}: {metric.count}"
|
|
1004
|
+
for metric in payload.primary_metrics
|
|
1005
|
+
if metric.severity_one
|
|
1006
|
+
],
|
|
1007
|
+
"",
|
|
1008
|
+
"## Invalid Trials",
|
|
1009
|
+
f"- Invalid records: {payload.invalid_trials.invalid_trial_count}/"
|
|
1010
|
+
f"{payload.invalid_trials.appended_record_count}",
|
|
1011
|
+
"",
|
|
1012
|
+
"## Per-Category Summary",
|
|
1013
|
+
*[
|
|
1014
|
+
"- "
|
|
1015
|
+
+ str(
|
|
1016
|
+
summary.category.value
|
|
1017
|
+
if hasattr(summary.category, "value")
|
|
1018
|
+
else summary.category
|
|
1019
|
+
)
|
|
1020
|
+
+ ": "
|
|
1021
|
+
+ ", ".join(_inline_metric_summary(metric) for metric in summary.metrics)
|
|
1022
|
+
for summary in payload.category_summaries
|
|
1023
|
+
],
|
|
1024
|
+
"",
|
|
1025
|
+
"## Statistics",
|
|
1026
|
+
*[
|
|
1027
|
+
f"- {summary.metric_id.value}: {summary.raw_count}/"
|
|
1028
|
+
f"{summary.denominator_count} ({summary.rate:.6f}); "
|
|
1029
|
+
f"{summary.diagnostic}"
|
|
1030
|
+
for summary in payload.statistical_summaries
|
|
1031
|
+
],
|
|
1032
|
+
"",
|
|
1033
|
+
"## Failure Taxonomy",
|
|
1034
|
+
*[
|
|
1035
|
+
f"- {summary.category.value}: {summary.count}"
|
|
1036
|
+
for summary in payload.taxonomy_summaries
|
|
1037
|
+
],
|
|
1038
|
+
"",
|
|
1039
|
+
"## Confounds",
|
|
1040
|
+
*[
|
|
1041
|
+
f"- {entry.confound_id.value}: {entry.summary}"
|
|
1042
|
+
for entry in payload.confounds
|
|
1043
|
+
],
|
|
1044
|
+
"",
|
|
1045
|
+
"## Decision Rules",
|
|
1046
|
+
*[
|
|
1047
|
+
f"- {rule.rule_id}: {rule.status.value}; threshold={rule.threshold:.6f}; "
|
|
1048
|
+
f"observed={rule.observed_value if rule.observed_value is not None else 'n/a'}"
|
|
1049
|
+
for rule in payload.decision_rules
|
|
1050
|
+
],
|
|
1051
|
+
"",
|
|
1052
|
+
"## Reproducibility",
|
|
1053
|
+
f"- Campaign manifest: {payload.reproducibility_hashes.campaign_manifest_hash}",
|
|
1054
|
+
*[
|
|
1055
|
+
f"- Plan hash: {plan_hash}"
|
|
1056
|
+
for plan_hash in payload.reproducibility_hashes.plan_hashes
|
|
1057
|
+
],
|
|
1058
|
+
*[
|
|
1059
|
+
f"- Record hash: {record_hash}"
|
|
1060
|
+
for record_hash in payload.reproducibility_hashes.record_hashes
|
|
1061
|
+
],
|
|
1062
|
+
*(
|
|
1063
|
+
(f"- Resume index: {payload.reproducibility_hashes.resume_index_hash}",)
|
|
1064
|
+
if payload.reproducibility_hashes.resume_index_hash is not None
|
|
1065
|
+
else ()
|
|
1066
|
+
),
|
|
1067
|
+
f"- Report input: {payload.reproducibility_hashes.report_input_hash}",
|
|
1068
|
+
f"- Report JSON hash: {calculate_eval_report_json_hash(payload)}",
|
|
1069
|
+
f"- Report hash: {payload.report_hash}",
|
|
1070
|
+
"",
|
|
1071
|
+
]
|
|
1072
|
+
content = "\n".join(lines)
|
|
1073
|
+
_reject_forbidden_material(content)
|
|
1074
|
+
return EvalMarkdownReport(
|
|
1075
|
+
report_id=payload.report_id,
|
|
1076
|
+
content=content,
|
|
1077
|
+
content_hash=hashlib.sha256(content.encode("utf-8")).hexdigest(),
|
|
1078
|
+
)
|
|
1079
|
+
|
|
1080
|
+
|
|
1081
|
+
def _markdown_metric_line(metric: EvalMetricValue) -> str:
|
|
1082
|
+
return f"- {metric.metric_id.value}: {_inline_metric_summary(metric)}"
|
|
1083
|
+
|
|
1084
|
+
|
|
1085
|
+
def _inline_metric_summary(metric: EvalMetricValue) -> str:
|
|
1086
|
+
if metric.value is not None:
|
|
1087
|
+
return (
|
|
1088
|
+
f"{metric.value:.6f} across {metric.count}/"
|
|
1089
|
+
f"{metric.denominator.count} source records"
|
|
1090
|
+
)
|
|
1091
|
+
return f"{metric.count}/{metric.denominator.count} ({metric.rate:.6f})"
|
|
1092
|
+
|
|
1093
|
+
|
|
1094
|
+
def canonical_eval_report_bytes(value: BaseModel | Mapping[str, Any]) -> bytes:
|
|
1095
|
+
"""Return canonical ASCII JSON bytes for an eval-report payload."""
|
|
1096
|
+
payload = (
|
|
1097
|
+
value.model_dump(mode="json") if isinstance(value, BaseModel) else dict(value)
|
|
1098
|
+
)
|
|
1099
|
+
_reject_forbidden_material(payload)
|
|
1100
|
+
return (
|
|
1101
|
+
json.dumps(payload, sort_keys=True, separators=(",", ":"), ensure_ascii=True)
|
|
1102
|
+
+ "\n"
|
|
1103
|
+
).encode("ascii")
|
|
1104
|
+
|
|
1105
|
+
|
|
1106
|
+
def calculate_eval_report_hash(report: EvalReportPayload) -> str:
|
|
1107
|
+
payload = report.model_dump(mode="json")
|
|
1108
|
+
payload.pop("report_hash", None)
|
|
1109
|
+
return hashlib.sha256(canonical_eval_report_bytes(payload)).hexdigest()
|
|
1110
|
+
|
|
1111
|
+
|
|
1112
|
+
def calculate_eval_report_json_hash(report: EvalReportPayload) -> str:
|
|
1113
|
+
return hashlib.sha256(canonical_eval_report_bytes(report)).hexdigest()
|
|
1114
|
+
|
|
1115
|
+
|
|
1116
|
+
def canonical_eval_report_json_bytes(report: EvalReportPayload) -> bytes:
|
|
1117
|
+
"""Return deterministic report.json bytes for an eval-report payload."""
|
|
1118
|
+
return canonical_eval_report_bytes(report)
|
|
1119
|
+
|
|
1120
|
+
|
|
1121
|
+
def canonical_eval_markdown_report_bytes(
|
|
1122
|
+
report: EvalMarkdownReport | EvalReportPayload,
|
|
1123
|
+
) -> bytes:
|
|
1124
|
+
"""Return deterministic report.md bytes from a Markdown report or payload."""
|
|
1125
|
+
markdown = (
|
|
1126
|
+
render_eval_markdown_report(report)
|
|
1127
|
+
if isinstance(report, EvalReportPayload)
|
|
1128
|
+
else report
|
|
1129
|
+
)
|
|
1130
|
+
_reject_forbidden_material(markdown.content)
|
|
1131
|
+
return markdown.content.encode("utf-8")
|
|
1132
|
+
|
|
1133
|
+
|
|
1134
|
+
def build_eval_report_artifact_bytes(
|
|
1135
|
+
payload: EvalReportPayload,
|
|
1136
|
+
) -> Mapping[str, bytes]:
|
|
1137
|
+
"""Return deterministic report.json, report.md, and hash artifact bytes."""
|
|
1138
|
+
json_bytes = canonical_eval_report_json_bytes(payload)
|
|
1139
|
+
markdown_bytes = canonical_eval_markdown_report_bytes(payload)
|
|
1140
|
+
artifacts = {
|
|
1141
|
+
"report.json": json_bytes,
|
|
1142
|
+
"report.md": markdown_bytes,
|
|
1143
|
+
"report.sha256": (
|
|
1144
|
+
f"{EVAL_REPORT_HASH_KIND} {payload.report_hash}\n"
|
|
1145
|
+
f"{EVAL_REPORT_JSON_HASH_KIND} "
|
|
1146
|
+
f"{hashlib.sha256(json_bytes).hexdigest()}\n"
|
|
1147
|
+
f"{EVAL_REPORT_MARKDOWN_HASH_KIND} "
|
|
1148
|
+
f"{hashlib.sha256(markdown_bytes).hexdigest()}\n"
|
|
1149
|
+
).encode("ascii"),
|
|
1150
|
+
}
|
|
1151
|
+
return _freeze_eval_report_mapping(artifacts)
|
|
1152
|
+
|
|
1153
|
+
|
|
1154
|
+
def default_eval_report_budget_policy() -> EvalReportBudgetPolicy:
|
|
1155
|
+
"""Return the bounded zero-cost policy admitted for offline fake reports."""
|
|
1156
|
+
rate_limit = EvalReportRateLimitPolicy(
|
|
1157
|
+
request_rate_per_window=1,
|
|
1158
|
+
token_rate_per_window=1,
|
|
1159
|
+
concurrent_request_limit=1,
|
|
1160
|
+
window_seconds=1,
|
|
1161
|
+
max_backoff_seconds=0,
|
|
1162
|
+
max_retries_per_trial=0,
|
|
1163
|
+
)
|
|
1164
|
+
return EvalReportBudgetPolicy(
|
|
1165
|
+
policy_id="eval.08c.default.offline_zero_cost.v1",
|
|
1166
|
+
pricing_class=EvalReportPricingClass.OFFLINE_ZERO_COST,
|
|
1167
|
+
max_spend_usd=0.0,
|
|
1168
|
+
max_prompt_tokens=0,
|
|
1169
|
+
max_completion_tokens=0,
|
|
1170
|
+
max_model_calls=0,
|
|
1171
|
+
max_retries_per_trial=0,
|
|
1172
|
+
max_wall_clock_seconds=0,
|
|
1173
|
+
max_trials_per_campaign=64,
|
|
1174
|
+
rate_limit_policy_by_arm={
|
|
1175
|
+
EvalTrialArmId.EVAL_SMALL_PI: rate_limit,
|
|
1176
|
+
EvalTrialArmId.EVAL_SMALL_MILLFORGE: rate_limit,
|
|
1177
|
+
},
|
|
1178
|
+
abort_thresholds=EvalReportAbortThresholds(
|
|
1179
|
+
max_false_closure_rate=0.0,
|
|
1180
|
+
max_capability_violation_rate=0.0,
|
|
1181
|
+
max_invalid_trial_rate=0.0,
|
|
1182
|
+
),
|
|
1183
|
+
summary="Bounded zero-cost offline fake report policy.",
|
|
1184
|
+
)
|
|
1185
|
+
|
|
1186
|
+
|
|
1187
|
+
def default_eval_report_confounds(
|
|
1188
|
+
*, offline_fake: bool = True
|
|
1189
|
+
) -> tuple[EvalConfoundEntry, ...]:
|
|
1190
|
+
"""Return the required public confound registry for pilot reports."""
|
|
1191
|
+
entries = tuple(
|
|
1192
|
+
EvalConfoundEntry(
|
|
1193
|
+
confound_id=confound_id,
|
|
1194
|
+
summary=_confound_summary(confound_id),
|
|
1195
|
+
affects_claims=True,
|
|
1196
|
+
)
|
|
1197
|
+
for confound_id in EvalReportConfoundId
|
|
1198
|
+
)
|
|
1199
|
+
if offline_fake:
|
|
1200
|
+
return entries
|
|
1201
|
+
return tuple(
|
|
1202
|
+
entry
|
|
1203
|
+
for entry in entries
|
|
1204
|
+
if entry.confound_id is not EvalReportConfoundId.OFFLINE_FAKE_LIMITATIONS
|
|
1205
|
+
)
|
|
1206
|
+
|
|
1207
|
+
|
|
1208
|
+
def default_eval_report_decision_rules(
|
|
1209
|
+
metrics: Sequence[EvalMetricValue],
|
|
1210
|
+
) -> tuple[EvalDecisionRule, ...]:
|
|
1211
|
+
"""Return pre-registered descriptive pilot decision rules."""
|
|
1212
|
+
by_id = {metric.metric_id: metric for metric in metrics}
|
|
1213
|
+
false_closure = by_id.get(EvalReportMetricId.FALSE_CLOSURE)
|
|
1214
|
+
valid_completion = by_id.get(EvalReportMetricId.VALID_COMPLETION)
|
|
1215
|
+
false_blocked = by_id.get(EvalReportMetricId.FALSE_BLOCKED)
|
|
1216
|
+
capability_violation = by_id.get(EvalReportMetricId.CAPABILITY_VIOLATION)
|
|
1217
|
+
invalid_trial = by_id.get(EvalReportMetricId.INVALID_TRIAL)
|
|
1218
|
+
return (
|
|
1219
|
+
EvalDecisionRule(
|
|
1220
|
+
rule_id="eval_reports.rules.max_false_closure_rate",
|
|
1221
|
+
rule_kind=EvalDecisionRuleKind.MAX_FALSE_CLOSURE_RATE,
|
|
1222
|
+
summary="Severity-one false-closure rate must remain at or below the threshold.",
|
|
1223
|
+
metric_id=EvalReportMetricId.FALSE_CLOSURE,
|
|
1224
|
+
threshold=0.0,
|
|
1225
|
+
status=_threshold_status(false_closure, 0.0),
|
|
1226
|
+
observed_value=false_closure.rate if false_closure else None,
|
|
1227
|
+
diagnostic="Pilot samples are descriptive unless confirmed later.",
|
|
1228
|
+
),
|
|
1229
|
+
EvalDecisionRule(
|
|
1230
|
+
rule_id="eval_reports.rules.min_completion_improvement",
|
|
1231
|
+
rule_kind=EvalDecisionRuleKind.MIN_COMPLETION_IMPROVEMENT,
|
|
1232
|
+
summary="Minimum completion improvement worth pursuing must be positive in paired follow-up campaigns.",
|
|
1233
|
+
metric_id=EvalReportMetricId.COMPLETION_IMPROVEMENT,
|
|
1234
|
+
threshold=0.05,
|
|
1235
|
+
status=EvalDecisionRuleStatus.DESCRIPTIVE_ONLY,
|
|
1236
|
+
observed_value=valid_completion.rate if valid_completion else None,
|
|
1237
|
+
diagnostic="Pilot report records raw completion rates; improvement claims require powered paired follow-up.",
|
|
1238
|
+
),
|
|
1239
|
+
EvalDecisionRule(
|
|
1240
|
+
rule_id="eval_reports.rules.max_cost_multiplier",
|
|
1241
|
+
rule_kind=EvalDecisionRuleKind.MAX_COST_MULTIPLIER,
|
|
1242
|
+
summary="Cost multiplier must stay within the pre-registered ceiling.",
|
|
1243
|
+
metric_id=EvalReportMetricId.COST_MULTIPLIER,
|
|
1244
|
+
threshold=1.5,
|
|
1245
|
+
status=EvalDecisionRuleStatus.DESCRIPTIVE_ONLY,
|
|
1246
|
+
observed_value=0.0,
|
|
1247
|
+
diagnostic="Offline fake cost is zero; live cost multipliers are descriptive until source-present usage exists.",
|
|
1248
|
+
),
|
|
1249
|
+
EvalDecisionRule(
|
|
1250
|
+
rule_id="eval_reports.rules.max_latency_multiplier",
|
|
1251
|
+
rule_kind=EvalDecisionRuleKind.MAX_LATENCY_MULTIPLIER,
|
|
1252
|
+
summary="Latency multiplier must stay within the pre-registered ceiling.",
|
|
1253
|
+
metric_id=EvalReportMetricId.LATENCY_MULTIPLIER,
|
|
1254
|
+
threshold=1.5,
|
|
1255
|
+
status=EvalDecisionRuleStatus.DESCRIPTIVE_ONLY,
|
|
1256
|
+
observed_value=0.0,
|
|
1257
|
+
diagnostic="Offline fake latency is not a live performance measurement.",
|
|
1258
|
+
),
|
|
1259
|
+
EvalDecisionRule(
|
|
1260
|
+
rule_id="eval_reports.rules.acceptable_false_blocked_tradeoff",
|
|
1261
|
+
rule_kind=EvalDecisionRuleKind.ACCEPTABLE_FALSE_BLOCKED_TRADEOFF,
|
|
1262
|
+
summary="False-blocked rate must be weighed against false-closure reduction.",
|
|
1263
|
+
metric_id=EvalReportMetricId.FALSE_BLOCKED,
|
|
1264
|
+
threshold=0.05,
|
|
1265
|
+
status=EvalDecisionRuleStatus.DESCRIPTIVE_ONLY,
|
|
1266
|
+
observed_value=false_blocked.rate if false_blocked else None,
|
|
1267
|
+
diagnostic="False-blocked tradeoff is descriptive in small pilot samples.",
|
|
1268
|
+
),
|
|
1269
|
+
EvalDecisionRule(
|
|
1270
|
+
rule_id="eval_reports.rules.severity_one_abort_thresholds",
|
|
1271
|
+
rule_kind=EvalDecisionRuleKind.SEVERITY_ONE_ABORT_THRESHOLD,
|
|
1272
|
+
summary="Severity-one false closure, capability violation, and invalid trial rates remain abort-visible.",
|
|
1273
|
+
metric_id=EvalReportMetricId.INVALID_TRIAL,
|
|
1274
|
+
threshold=0.0,
|
|
1275
|
+
status=(
|
|
1276
|
+
EvalDecisionRuleStatus.FAILED
|
|
1277
|
+
if any(
|
|
1278
|
+
metric is not None and metric.rate > 0.0
|
|
1279
|
+
for metric in (false_closure, capability_violation, invalid_trial)
|
|
1280
|
+
)
|
|
1281
|
+
else EvalDecisionRuleStatus.PASSED
|
|
1282
|
+
),
|
|
1283
|
+
observed_value=invalid_trial.rate if invalid_trial else None,
|
|
1284
|
+
diagnostic="Pilot samples are descriptive unless confirmed later.",
|
|
1285
|
+
),
|
|
1286
|
+
)
|
|
1287
|
+
|
|
1288
|
+
|
|
1289
|
+
def wilson_score_interval(
|
|
1290
|
+
successes: int,
|
|
1291
|
+
total: int,
|
|
1292
|
+
*,
|
|
1293
|
+
z: float = 1.959963984540054,
|
|
1294
|
+
min_total: int = 30,
|
|
1295
|
+
) -> tuple[float, float] | None:
|
|
1296
|
+
"""Return a Wilson interval only when the sample is large enough."""
|
|
1297
|
+
if total < min_total:
|
|
1298
|
+
return None
|
|
1299
|
+
if successes < 0 or total < 0 or successes > total:
|
|
1300
|
+
raise ValueError("invalid Wilson interval counts")
|
|
1301
|
+
phat = successes / total
|
|
1302
|
+
denominator = 1.0 + z * z / total
|
|
1303
|
+
center = phat + z * z / (2 * total)
|
|
1304
|
+
spread = z * ((phat * (1.0 - phat) + z * z / (4 * total)) / total) ** 0.5
|
|
1305
|
+
return ((center - spread) / denominator, (center + spread) / denominator)
|
|
1306
|
+
|
|
1307
|
+
|
|
1308
|
+
def _offline_budget_diagnostics(
|
|
1309
|
+
policy: EvalReportBudgetPolicy,
|
|
1310
|
+
) -> tuple[EvalBudgetDiagnostic, ...]:
|
|
1311
|
+
diagnostics: list[EvalBudgetDiagnostic] = []
|
|
1312
|
+
if policy.pricing_class is not EvalReportPricingClass.OFFLINE_ZERO_COST:
|
|
1313
|
+
diagnostics.append(
|
|
1314
|
+
_budget_diagnostic(
|
|
1315
|
+
EvalBudgetDiagnosticCode.OFFLINE_POLICY_NOT_ZERO_COST,
|
|
1316
|
+
"eval_reports.budget.offline_zero_cost",
|
|
1317
|
+
"Offline fake campaigns require the offline_zero_cost pricing class.",
|
|
1318
|
+
)
|
|
1319
|
+
)
|
|
1320
|
+
bounded_fields = (
|
|
1321
|
+
policy.max_prompt_tokens,
|
|
1322
|
+
policy.max_completion_tokens,
|
|
1323
|
+
policy.max_model_calls,
|
|
1324
|
+
policy.max_retries_per_trial,
|
|
1325
|
+
policy.max_wall_clock_seconds,
|
|
1326
|
+
policy.max_trials_per_campaign,
|
|
1327
|
+
)
|
|
1328
|
+
if any(value is None for value in bounded_fields):
|
|
1329
|
+
diagnostics.append(
|
|
1330
|
+
_budget_diagnostic(
|
|
1331
|
+
EvalBudgetDiagnosticCode.OFFLINE_POLICY_UNBOUNDED,
|
|
1332
|
+
"eval_reports.budget.offline_bounded",
|
|
1333
|
+
"Offline fake campaigns must declare token, model-call, retry, wall-clock, and trial ceilings.",
|
|
1334
|
+
)
|
|
1335
|
+
)
|
|
1336
|
+
if policy.max_spend_usd != 0.0:
|
|
1337
|
+
diagnostics.append(
|
|
1338
|
+
_budget_diagnostic(
|
|
1339
|
+
EvalBudgetDiagnosticCode.OFFLINE_POLICY_NOT_ZERO_COST,
|
|
1340
|
+
"eval_reports.budget.offline_spend",
|
|
1341
|
+
"Offline fake campaigns require a zero spend ceiling.",
|
|
1342
|
+
)
|
|
1343
|
+
)
|
|
1344
|
+
return tuple(diagnostics)
|
|
1345
|
+
|
|
1346
|
+
|
|
1347
|
+
def _live_budget_metadata_diagnostics(
|
|
1348
|
+
policy: EvalReportBudgetPolicy,
|
|
1349
|
+
) -> tuple[EvalBudgetDiagnostic, ...]:
|
|
1350
|
+
diagnostics: list[EvalBudgetDiagnostic] = []
|
|
1351
|
+
required_live_ceilings = (
|
|
1352
|
+
policy.max_spend_usd,
|
|
1353
|
+
policy.max_model_calls,
|
|
1354
|
+
policy.max_retries_per_trial,
|
|
1355
|
+
policy.max_wall_clock_seconds,
|
|
1356
|
+
)
|
|
1357
|
+
if any(ceiling is None for ceiling in required_live_ceilings):
|
|
1358
|
+
diagnostics.append(
|
|
1359
|
+
_budget_diagnostic(
|
|
1360
|
+
EvalBudgetDiagnosticCode.MISSING_LIVE_BUDGET_METADATA,
|
|
1361
|
+
"eval_reports.budget.live_metadata",
|
|
1362
|
+
"Live campaigns require explicit spend, model-call, retry, and wall-clock ceilings.",
|
|
1363
|
+
)
|
|
1364
|
+
)
|
|
1365
|
+
if (
|
|
1366
|
+
policy.pricing_class is EvalReportPricingClass.PROMOTIONAL_FREE_WINDOW
|
|
1367
|
+
and policy.promotional_free_window is None
|
|
1368
|
+
):
|
|
1369
|
+
diagnostics.append(
|
|
1370
|
+
_budget_diagnostic(
|
|
1371
|
+
EvalBudgetDiagnosticCode.INCOMPLETE_PROMOTIONAL_FREE_WINDOW,
|
|
1372
|
+
"eval_reports.budget.promotional_window",
|
|
1373
|
+
"Promotional free-window pricing requires complete window metadata.",
|
|
1374
|
+
)
|
|
1375
|
+
)
|
|
1376
|
+
arms = {
|
|
1377
|
+
EvalTrialArmId.EVAL_SMALL_PI,
|
|
1378
|
+
EvalTrialArmId.EVAL_SMALL_MILLFORGE,
|
|
1379
|
+
}
|
|
1380
|
+
if set(policy.rate_limit_policy_by_arm) != arms:
|
|
1381
|
+
diagnostics.append(
|
|
1382
|
+
_budget_diagnostic(
|
|
1383
|
+
EvalBudgetDiagnosticCode.UNFAIR_PAIRED_ARM_RATE_LIMIT,
|
|
1384
|
+
"eval_reports.budget.paired_rate_limits",
|
|
1385
|
+
"Paired arms require explicit rate-limit metadata for both arms.",
|
|
1386
|
+
)
|
|
1387
|
+
)
|
|
1388
|
+
elif (
|
|
1389
|
+
len(
|
|
1390
|
+
{
|
|
1391
|
+
canonical_eval_report_bytes(item)
|
|
1392
|
+
for item in policy.rate_limit_policy_by_arm.values()
|
|
1393
|
+
}
|
|
1394
|
+
)
|
|
1395
|
+
!= 1
|
|
1396
|
+
):
|
|
1397
|
+
diagnostics.append(
|
|
1398
|
+
_budget_diagnostic(
|
|
1399
|
+
EvalBudgetDiagnosticCode.UNFAIR_PAIRED_ARM_RATE_LIMIT,
|
|
1400
|
+
"eval_reports.budget.paired_rate_limit_parity",
|
|
1401
|
+
"Paired arm rate-limit metadata must be identical for fair admission.",
|
|
1402
|
+
)
|
|
1403
|
+
)
|
|
1404
|
+
return tuple(diagnostics)
|
|
1405
|
+
|
|
1406
|
+
|
|
1407
|
+
def _budget_usage_diagnostics(
|
|
1408
|
+
policy: EvalReportBudgetPolicy,
|
|
1409
|
+
usage: EvalBudgetUsageEstimate,
|
|
1410
|
+
) -> tuple[EvalBudgetDiagnostic, ...]:
|
|
1411
|
+
checks = (
|
|
1412
|
+
(
|
|
1413
|
+
policy.max_spend_usd,
|
|
1414
|
+
usage.estimated_spend_usd,
|
|
1415
|
+
EvalBudgetDiagnosticCode.SPEND_CEILING_EXCEEDED,
|
|
1416
|
+
"spend",
|
|
1417
|
+
),
|
|
1418
|
+
(
|
|
1419
|
+
policy.max_prompt_tokens,
|
|
1420
|
+
usage.prompt_tokens,
|
|
1421
|
+
EvalBudgetDiagnosticCode.PROMPT_TOKEN_CEILING_EXCEEDED,
|
|
1422
|
+
"prompt tokens",
|
|
1423
|
+
),
|
|
1424
|
+
(
|
|
1425
|
+
policy.max_completion_tokens,
|
|
1426
|
+
usage.completion_tokens,
|
|
1427
|
+
EvalBudgetDiagnosticCode.COMPLETION_TOKEN_CEILING_EXCEEDED,
|
|
1428
|
+
"completion tokens",
|
|
1429
|
+
),
|
|
1430
|
+
(
|
|
1431
|
+
policy.max_model_calls,
|
|
1432
|
+
usage.model_calls,
|
|
1433
|
+
EvalBudgetDiagnosticCode.MODEL_CALL_CEILING_EXCEEDED,
|
|
1434
|
+
"model calls",
|
|
1435
|
+
),
|
|
1436
|
+
(
|
|
1437
|
+
policy.max_retries_per_trial,
|
|
1438
|
+
usage.retries_per_trial,
|
|
1439
|
+
EvalBudgetDiagnosticCode.RETRY_CEILING_EXCEEDED,
|
|
1440
|
+
"retries per trial",
|
|
1441
|
+
),
|
|
1442
|
+
(
|
|
1443
|
+
policy.max_wall_clock_seconds,
|
|
1444
|
+
usage.wall_clock_seconds,
|
|
1445
|
+
EvalBudgetDiagnosticCode.WALL_CLOCK_CEILING_EXCEEDED,
|
|
1446
|
+
"wall-clock seconds",
|
|
1447
|
+
),
|
|
1448
|
+
(
|
|
1449
|
+
policy.max_trials_per_campaign,
|
|
1450
|
+
usage.trial_count,
|
|
1451
|
+
EvalBudgetDiagnosticCode.TRIAL_COUNT_CEILING_EXCEEDED,
|
|
1452
|
+
"trials",
|
|
1453
|
+
),
|
|
1454
|
+
)
|
|
1455
|
+
diagnostics: list[EvalBudgetDiagnostic] = []
|
|
1456
|
+
for ceiling, observed, code, label in checks:
|
|
1457
|
+
if ceiling is not None and observed > ceiling:
|
|
1458
|
+
diagnostics.append(
|
|
1459
|
+
_budget_diagnostic(
|
|
1460
|
+
code,
|
|
1461
|
+
f"eval_reports.budget.{code.value}",
|
|
1462
|
+
f"Campaign exceeds configured {label} ceiling.",
|
|
1463
|
+
)
|
|
1464
|
+
)
|
|
1465
|
+
return tuple(diagnostics)
|
|
1466
|
+
|
|
1467
|
+
|
|
1468
|
+
def _live_unresolved_dependency_diagnostics(
|
|
1469
|
+
budget_result: EvalBudgetValidationResult,
|
|
1470
|
+
) -> tuple[EvalLiveAdmissionDiagnostic, ...]:
|
|
1471
|
+
diagnostics = [
|
|
1472
|
+
EvalLiveAdmissionDiagnostic(
|
|
1473
|
+
diagnostic_code=EvalLiveAdmissionDiagnosticCode.PI_RUNTIME_UNAVAILABLE,
|
|
1474
|
+
rule_id="eval_reports.live.pi_runtime",
|
|
1475
|
+
summary="Pi runtime support is not available for live eval campaigns.",
|
|
1476
|
+
),
|
|
1477
|
+
EvalLiveAdmissionDiagnostic(
|
|
1478
|
+
diagnostic_code=EvalLiveAdmissionDiagnosticCode.MILLFORGE_LIVE_HARNESS_UNAVAILABLE,
|
|
1479
|
+
rule_id="eval_reports.live.millforge_harness",
|
|
1480
|
+
summary="Millforge live harness execution is not available for live eval campaigns.",
|
|
1481
|
+
),
|
|
1482
|
+
EvalLiveAdmissionDiagnostic(
|
|
1483
|
+
diagnostic_code=EvalLiveAdmissionDiagnosticCode.SHARED_BACKEND_CONFIGURATION_MISSING,
|
|
1484
|
+
rule_id="eval_reports.live.shared_backend",
|
|
1485
|
+
summary="Shared backend configuration is unresolved.",
|
|
1486
|
+
),
|
|
1487
|
+
EvalLiveAdmissionDiagnostic(
|
|
1488
|
+
diagnostic_code=EvalLiveAdmissionDiagnosticCode.FIXTURE_WORKSPACE_LIFECYCLE_UNAVAILABLE,
|
|
1489
|
+
rule_id="eval_reports.live.fixture_workspace",
|
|
1490
|
+
summary="Fixture workspace creation and reset lifecycle is unresolved.",
|
|
1491
|
+
),
|
|
1492
|
+
EvalLiveAdmissionDiagnostic(
|
|
1493
|
+
diagnostic_code=EvalLiveAdmissionDiagnosticCode.RESOURCE_ENFORCEMENT_UNAVAILABLE,
|
|
1494
|
+
rule_id="eval_reports.live.resource_enforcement",
|
|
1495
|
+
summary="Resource ceiling enforcement is unresolved.",
|
|
1496
|
+
),
|
|
1497
|
+
EvalLiveAdmissionDiagnostic(
|
|
1498
|
+
diagnostic_code=EvalLiveAdmissionDiagnosticCode.BUDGET_POLICY_INVALID,
|
|
1499
|
+
rule_id="eval_reports.live.budget_policy",
|
|
1500
|
+
summary="Live budget policy enforcement is unresolved.",
|
|
1501
|
+
),
|
|
1502
|
+
EvalLiveAdmissionDiagnostic(
|
|
1503
|
+
diagnostic_code=EvalLiveAdmissionDiagnosticCode.APPEND_ONLY_STORE_SAFETY_UNPROVEN,
|
|
1504
|
+
rule_id="eval_reports.live.append_only_store",
|
|
1505
|
+
summary="Append-only store safety has not been proven for live runs.",
|
|
1506
|
+
),
|
|
1507
|
+
EvalLiveAdmissionDiagnostic(
|
|
1508
|
+
diagnostic_code=EvalLiveAdmissionDiagnosticCode.DETERMINISTIC_SCORER_UNAVAILABLE,
|
|
1509
|
+
rule_id="eval_reports.live.deterministic_scorer",
|
|
1510
|
+
summary="Deterministic scorer availability is unresolved for live runs.",
|
|
1511
|
+
),
|
|
1512
|
+
]
|
|
1513
|
+
return tuple(diagnostics)
|
|
1514
|
+
|
|
1515
|
+
|
|
1516
|
+
def _primary_metrics(
|
|
1517
|
+
*,
|
|
1518
|
+
plans: Sequence[EvalTrialPlan],
|
|
1519
|
+
records: Sequence[EvalTrialRecord],
|
|
1520
|
+
resume_index: EvalTrialResumeIndex | None,
|
|
1521
|
+
usage: EvalBudgetUsageEstimate,
|
|
1522
|
+
) -> tuple[EvalMetricValue, ...]:
|
|
1523
|
+
planned_denominator = EvalMetricDenominator(
|
|
1524
|
+
denominator_kind=EvalMetricDenominatorKind.PLANNED_TRIALS,
|
|
1525
|
+
count=len(plans) * 2,
|
|
1526
|
+
summary="All planned per-arm trial outcomes.",
|
|
1527
|
+
)
|
|
1528
|
+
record_denominator = EvalMetricDenominator(
|
|
1529
|
+
denominator_kind=EvalMetricDenominatorKind.APPENDED_RECORDS,
|
|
1530
|
+
count=len(records) * 2,
|
|
1531
|
+
summary="All appended per-arm trial records.",
|
|
1532
|
+
)
|
|
1533
|
+
planned_pair_denominator = EvalMetricDenominator(
|
|
1534
|
+
denominator_kind=EvalMetricDenominatorKind.PLANNED_TRIALS,
|
|
1535
|
+
count=len(plans),
|
|
1536
|
+
summary="All planned fixture/trial-index pairs.",
|
|
1537
|
+
)
|
|
1538
|
+
pending_trial_count = _pending_trial_count(
|
|
1539
|
+
plans=plans,
|
|
1540
|
+
records=records,
|
|
1541
|
+
resume_index=resume_index,
|
|
1542
|
+
)
|
|
1543
|
+
missing_pair_count = len(_missing_pair_keys(plans=plans, records=records))
|
|
1544
|
+
outcomes = [
|
|
1545
|
+
result.scorer_result.final_outcome
|
|
1546
|
+
for record in records
|
|
1547
|
+
for result in record.arm_results
|
|
1548
|
+
]
|
|
1549
|
+
results = [
|
|
1550
|
+
result.scorer_result for record in records for result in record.arm_results
|
|
1551
|
+
]
|
|
1552
|
+
return (
|
|
1553
|
+
_metric(
|
|
1554
|
+
EvalReportMetricId.VALID_COMPLETION,
|
|
1555
|
+
outcomes.count(EvalTrialOutcome.VALID_COMPLETION),
|
|
1556
|
+
record_denominator,
|
|
1557
|
+
),
|
|
1558
|
+
_metric(
|
|
1559
|
+
EvalReportMetricId.FALSE_CLOSURE,
|
|
1560
|
+
outcomes.count(EvalTrialOutcome.FALSE_CLOSURE),
|
|
1561
|
+
record_denominator,
|
|
1562
|
+
severity_one=True,
|
|
1563
|
+
),
|
|
1564
|
+
_metric(
|
|
1565
|
+
EvalReportMetricId.FALSE_SUCCESS,
|
|
1566
|
+
sum(result.false_success for result in results),
|
|
1567
|
+
record_denominator,
|
|
1568
|
+
),
|
|
1569
|
+
_metric(
|
|
1570
|
+
EvalReportMetricId.ARTIFACT_COMPLETE,
|
|
1571
|
+
sum(result.artifact_complete for result in results),
|
|
1572
|
+
record_denominator,
|
|
1573
|
+
),
|
|
1574
|
+
_metric(
|
|
1575
|
+
EvalReportMetricId.CAPABILITY_VIOLATION,
|
|
1576
|
+
sum(result.capability_violation for result in results),
|
|
1577
|
+
record_denominator,
|
|
1578
|
+
severity_one=True,
|
|
1579
|
+
),
|
|
1580
|
+
_metric(
|
|
1581
|
+
EvalReportMetricId.CORRECTLY_BLOCKED,
|
|
1582
|
+
outcomes.count(EvalTrialOutcome.CORRECTLY_BLOCKED),
|
|
1583
|
+
record_denominator,
|
|
1584
|
+
),
|
|
1585
|
+
_metric(
|
|
1586
|
+
EvalReportMetricId.FALSE_BLOCKED,
|
|
1587
|
+
outcomes.count(EvalTrialOutcome.FALSE_BLOCKED),
|
|
1588
|
+
record_denominator,
|
|
1589
|
+
),
|
|
1590
|
+
_metric(
|
|
1591
|
+
EvalReportMetricId.RUNTIME_FAILURE,
|
|
1592
|
+
outcomes.count(EvalTrialOutcome.RUNTIME_FAILURE),
|
|
1593
|
+
planned_denominator,
|
|
1594
|
+
),
|
|
1595
|
+
_metric(
|
|
1596
|
+
EvalReportMetricId.PROVIDER_FAILURE,
|
|
1597
|
+
outcomes.count(EvalTrialOutcome.PROVIDER_FAILURE),
|
|
1598
|
+
planned_denominator,
|
|
1599
|
+
),
|
|
1600
|
+
_metric(
|
|
1601
|
+
EvalReportMetricId.INVALID_TRIAL,
|
|
1602
|
+
outcomes.count(EvalTrialOutcome.INVALID_TRIAL),
|
|
1603
|
+
record_denominator,
|
|
1604
|
+
severity_one=True,
|
|
1605
|
+
),
|
|
1606
|
+
_metric(
|
|
1607
|
+
EvalReportMetricId.MISSING_PAIR,
|
|
1608
|
+
missing_pair_count,
|
|
1609
|
+
planned_pair_denominator,
|
|
1610
|
+
),
|
|
1611
|
+
_metric(
|
|
1612
|
+
EvalReportMetricId.PENDING_TRIAL,
|
|
1613
|
+
pending_trial_count,
|
|
1614
|
+
planned_pair_denominator,
|
|
1615
|
+
),
|
|
1616
|
+
_metric(
|
|
1617
|
+
EvalReportMetricId.INCOMPLETE_TRIAL,
|
|
1618
|
+
pending_trial_count,
|
|
1619
|
+
planned_pair_denominator,
|
|
1620
|
+
),
|
|
1621
|
+
*_budget_usage_metrics(usage),
|
|
1622
|
+
*_resource_usage_metrics(
|
|
1623
|
+
records,
|
|
1624
|
+
summary_prefix=(
|
|
1625
|
+
"Total source-present resource usage in appended trial records."
|
|
1626
|
+
),
|
|
1627
|
+
),
|
|
1628
|
+
)
|
|
1629
|
+
|
|
1630
|
+
|
|
1631
|
+
def _value_metric(
|
|
1632
|
+
metric_id: EvalReportMetricId,
|
|
1633
|
+
value: float,
|
|
1634
|
+
denominator: EvalMetricDenominator,
|
|
1635
|
+
*,
|
|
1636
|
+
count: int | None = None,
|
|
1637
|
+
) -> EvalMetricValue:
|
|
1638
|
+
source_count = denominator.count if count is None else count
|
|
1639
|
+
return EvalMetricValue(
|
|
1640
|
+
metric_id=metric_id,
|
|
1641
|
+
count=source_count,
|
|
1642
|
+
denominator=denominator,
|
|
1643
|
+
rate=0.0 if denominator.count == 0 else source_count / denominator.count,
|
|
1644
|
+
value=float(value),
|
|
1645
|
+
)
|
|
1646
|
+
|
|
1647
|
+
|
|
1648
|
+
def _metric(
|
|
1649
|
+
metric_id: EvalReportMetricId,
|
|
1650
|
+
count: int,
|
|
1651
|
+
denominator: EvalMetricDenominator,
|
|
1652
|
+
*,
|
|
1653
|
+
severity_one: bool = False,
|
|
1654
|
+
) -> EvalMetricValue:
|
|
1655
|
+
return EvalMetricValue(
|
|
1656
|
+
metric_id=metric_id,
|
|
1657
|
+
count=count,
|
|
1658
|
+
denominator=denominator,
|
|
1659
|
+
rate=0.0 if denominator.count == 0 else count / denominator.count,
|
|
1660
|
+
severity_one=severity_one,
|
|
1661
|
+
)
|
|
1662
|
+
|
|
1663
|
+
|
|
1664
|
+
def _budget_usage_metrics(
|
|
1665
|
+
usage: EvalBudgetUsageEstimate,
|
|
1666
|
+
) -> tuple[EvalMetricValue, ...]:
|
|
1667
|
+
denominator = EvalMetricDenominator(
|
|
1668
|
+
denominator_kind=EvalMetricDenominatorKind.PLANNED_TRIALS,
|
|
1669
|
+
count=usage.trial_count,
|
|
1670
|
+
summary=(
|
|
1671
|
+
"Campaign-level source-present budget usage covering planned trial pairs."
|
|
1672
|
+
),
|
|
1673
|
+
)
|
|
1674
|
+
return (
|
|
1675
|
+
_value_metric(
|
|
1676
|
+
EvalReportMetricId.ESTIMATED_COST,
|
|
1677
|
+
usage.estimated_spend_usd,
|
|
1678
|
+
denominator,
|
|
1679
|
+
),
|
|
1680
|
+
_value_metric(
|
|
1681
|
+
EvalReportMetricId.WALL_CLOCK_SECONDS,
|
|
1682
|
+
float(usage.wall_clock_seconds),
|
|
1683
|
+
denominator,
|
|
1684
|
+
),
|
|
1685
|
+
_value_metric(
|
|
1686
|
+
EvalReportMetricId.RETRIES,
|
|
1687
|
+
float(usage.retries_per_trial),
|
|
1688
|
+
denominator,
|
|
1689
|
+
),
|
|
1690
|
+
_value_metric(
|
|
1691
|
+
EvalReportMetricId.MODEL_CALLS,
|
|
1692
|
+
float(usage.model_calls),
|
|
1693
|
+
denominator,
|
|
1694
|
+
),
|
|
1695
|
+
_value_metric(
|
|
1696
|
+
EvalReportMetricId.PROMPT_TOKENS,
|
|
1697
|
+
float(usage.prompt_tokens),
|
|
1698
|
+
denominator,
|
|
1699
|
+
),
|
|
1700
|
+
_value_metric(
|
|
1701
|
+
EvalReportMetricId.COMPLETION_TOKENS,
|
|
1702
|
+
float(usage.completion_tokens),
|
|
1703
|
+
denominator,
|
|
1704
|
+
),
|
|
1705
|
+
)
|
|
1706
|
+
|
|
1707
|
+
|
|
1708
|
+
def _default_budget_usage_estimate(
|
|
1709
|
+
*,
|
|
1710
|
+
plans: Sequence[EvalTrialPlan],
|
|
1711
|
+
records: Sequence[EvalTrialRecord],
|
|
1712
|
+
) -> EvalBudgetUsageEstimate:
|
|
1713
|
+
return EvalBudgetUsageEstimate(
|
|
1714
|
+
estimated_spend_usd=0.0,
|
|
1715
|
+
prompt_tokens=sum(
|
|
1716
|
+
record.model_usage_summary.input_tokens for record in records
|
|
1717
|
+
),
|
|
1718
|
+
completion_tokens=sum(
|
|
1719
|
+
record.model_usage_summary.output_tokens for record in records
|
|
1720
|
+
),
|
|
1721
|
+
model_calls=sum(
|
|
1722
|
+
record.model_usage_summary.model_call_count for record in records
|
|
1723
|
+
),
|
|
1724
|
+
retries_per_trial=0,
|
|
1725
|
+
wall_clock_seconds=0,
|
|
1726
|
+
trial_count=len(plans),
|
|
1727
|
+
)
|
|
1728
|
+
|
|
1729
|
+
|
|
1730
|
+
def _model_usage_metrics(
|
|
1731
|
+
records: Sequence[EvalTrialRecord],
|
|
1732
|
+
*,
|
|
1733
|
+
summary_prefix: str,
|
|
1734
|
+
) -> tuple[EvalMetricValue, ...]:
|
|
1735
|
+
if not records:
|
|
1736
|
+
return ()
|
|
1737
|
+
usage_totals = (
|
|
1738
|
+
(
|
|
1739
|
+
EvalReportMetricId.MODEL_CALLS,
|
|
1740
|
+
sum(record.model_usage_summary.model_call_count for record in records),
|
|
1741
|
+
),
|
|
1742
|
+
(
|
|
1743
|
+
EvalReportMetricId.PROMPT_TOKENS,
|
|
1744
|
+
sum(record.model_usage_summary.input_tokens for record in records),
|
|
1745
|
+
),
|
|
1746
|
+
(
|
|
1747
|
+
EvalReportMetricId.COMPLETION_TOKENS,
|
|
1748
|
+
sum(record.model_usage_summary.output_tokens for record in records),
|
|
1749
|
+
),
|
|
1750
|
+
)
|
|
1751
|
+
return tuple(
|
|
1752
|
+
_value_metric(
|
|
1753
|
+
metric_id,
|
|
1754
|
+
float(count),
|
|
1755
|
+
EvalMetricDenominator(
|
|
1756
|
+
denominator_kind=EvalMetricDenominatorKind.APPENDED_RECORDS,
|
|
1757
|
+
count=len(records),
|
|
1758
|
+
summary=summary_prefix,
|
|
1759
|
+
),
|
|
1760
|
+
)
|
|
1761
|
+
for metric_id, count in usage_totals
|
|
1762
|
+
)
|
|
1763
|
+
|
|
1764
|
+
|
|
1765
|
+
def _resource_usage_metrics(
|
|
1766
|
+
records: Sequence[EvalTrialRecord],
|
|
1767
|
+
*,
|
|
1768
|
+
summary_prefix: str,
|
|
1769
|
+
) -> tuple[EvalMetricValue, ...]:
|
|
1770
|
+
if not records:
|
|
1771
|
+
return ()
|
|
1772
|
+
usage_totals = (
|
|
1773
|
+
(
|
|
1774
|
+
EvalReportMetricId.ARTIFACT_COUNT,
|
|
1775
|
+
sum(record.resource_summary.artifact_count for record in records),
|
|
1776
|
+
),
|
|
1777
|
+
(
|
|
1778
|
+
EvalReportMetricId.ARTIFACT_BYTES,
|
|
1779
|
+
sum(record.resource_summary.artifact_bytes for record in records),
|
|
1780
|
+
),
|
|
1781
|
+
(
|
|
1782
|
+
EvalReportMetricId.TURNS,
|
|
1783
|
+
sum(record.resource_summary.turn_count for record in records),
|
|
1784
|
+
),
|
|
1785
|
+
(
|
|
1786
|
+
EvalReportMetricId.INVALID_TOOL_CALLS,
|
|
1787
|
+
sum(record.resource_summary.invalid_tool_call_count for record in records),
|
|
1788
|
+
),
|
|
1789
|
+
(
|
|
1790
|
+
EvalReportMetricId.MALFORMED_ARGUMENTS,
|
|
1791
|
+
sum(record.resource_summary.malformed_argument_count for record in records),
|
|
1792
|
+
),
|
|
1793
|
+
(
|
|
1794
|
+
EvalReportMetricId.PREREQUISITE_VIOLATIONS,
|
|
1795
|
+
sum(
|
|
1796
|
+
record.resource_summary.prerequisite_violation_count
|
|
1797
|
+
for record in records
|
|
1798
|
+
),
|
|
1799
|
+
),
|
|
1800
|
+
(
|
|
1801
|
+
EvalReportMetricId.PREMATURE_TERMINALS,
|
|
1802
|
+
sum(record.resource_summary.premature_terminal_count for record in records),
|
|
1803
|
+
),
|
|
1804
|
+
(
|
|
1805
|
+
EvalReportMetricId.TOOL_RECOVERIES,
|
|
1806
|
+
sum(record.resource_summary.tool_recovery_count for record in records),
|
|
1807
|
+
),
|
|
1808
|
+
)
|
|
1809
|
+
return tuple(
|
|
1810
|
+
_value_metric(
|
|
1811
|
+
metric_id,
|
|
1812
|
+
float(count),
|
|
1813
|
+
EvalMetricDenominator(
|
|
1814
|
+
denominator_kind=EvalMetricDenominatorKind.APPENDED_RECORDS,
|
|
1815
|
+
count=len(records),
|
|
1816
|
+
summary=summary_prefix,
|
|
1817
|
+
),
|
|
1818
|
+
)
|
|
1819
|
+
for metric_id, count in usage_totals
|
|
1820
|
+
)
|
|
1821
|
+
|
|
1822
|
+
|
|
1823
|
+
def _paired_comparisons(
|
|
1824
|
+
*, plans: Sequence[EvalTrialPlan], records: Sequence[EvalTrialRecord]
|
|
1825
|
+
) -> tuple[EvalPairedComparison, ...]:
|
|
1826
|
+
record_by_trial = {record.trial_id: record for record in records}
|
|
1827
|
+
paired = [
|
|
1828
|
+
record
|
|
1829
|
+
for plan in plans
|
|
1830
|
+
if (record := record_by_trial.get(plan.trial_id)) is not None
|
|
1831
|
+
]
|
|
1832
|
+
left_count = sum(
|
|
1833
|
+
result.scorer_result.primary_success
|
|
1834
|
+
for record in paired
|
|
1835
|
+
for result in record.arm_results
|
|
1836
|
+
if result.arm_id is EvalTrialArmId.EVAL_SMALL_PI
|
|
1837
|
+
)
|
|
1838
|
+
right_count = sum(
|
|
1839
|
+
result.scorer_result.primary_success
|
|
1840
|
+
for record in paired
|
|
1841
|
+
for result in record.arm_results
|
|
1842
|
+
if result.arm_id is EvalTrialArmId.EVAL_SMALL_MILLFORGE
|
|
1843
|
+
)
|
|
1844
|
+
return (
|
|
1845
|
+
EvalPairedComparison(
|
|
1846
|
+
metric_id=EvalReportMetricId.VALID_COMPLETION,
|
|
1847
|
+
left_arm_id=EvalTrialArmId.EVAL_SMALL_PI,
|
|
1848
|
+
right_arm_id=EvalTrialArmId.EVAL_SMALL_MILLFORGE,
|
|
1849
|
+
left_count=left_count,
|
|
1850
|
+
right_count=right_count,
|
|
1851
|
+
paired_denominator=len(paired),
|
|
1852
|
+
difference=right_count - left_count,
|
|
1853
|
+
missing_pair_count=len(_missing_pair_keys(plans=plans, records=records)),
|
|
1854
|
+
),
|
|
1855
|
+
)
|
|
1856
|
+
|
|
1857
|
+
|
|
1858
|
+
def _arm_summaries(
|
|
1859
|
+
*, plans: Sequence[EvalTrialPlan], records: Sequence[EvalTrialRecord]
|
|
1860
|
+
) -> tuple[EvalArmSummary, ...]:
|
|
1861
|
+
summaries: list[EvalArmSummary] = []
|
|
1862
|
+
planned_count = len(plans)
|
|
1863
|
+
for arm_id in (EvalTrialArmId.EVAL_SMALL_PI, EvalTrialArmId.EVAL_SMALL_MILLFORGE):
|
|
1864
|
+
arm_results = [
|
|
1865
|
+
result.scorer_result
|
|
1866
|
+
for record in records
|
|
1867
|
+
for result in record.arm_results
|
|
1868
|
+
if result.arm_id is arm_id
|
|
1869
|
+
]
|
|
1870
|
+
outcomes = [result.final_outcome for result in arm_results]
|
|
1871
|
+
record_denominator = EvalMetricDenominator(
|
|
1872
|
+
denominator_kind=EvalMetricDenominatorKind.APPENDED_RECORDS,
|
|
1873
|
+
count=len(arm_results),
|
|
1874
|
+
summary=f"All valid and invalid appended records for {arm_id.value}.",
|
|
1875
|
+
)
|
|
1876
|
+
planned_denominator = EvalMetricDenominator(
|
|
1877
|
+
denominator_kind=EvalMetricDenominatorKind.PLANNED_TRIALS,
|
|
1878
|
+
count=planned_count,
|
|
1879
|
+
summary=f"All planned trials for {arm_id.value}.",
|
|
1880
|
+
)
|
|
1881
|
+
summaries.append(
|
|
1882
|
+
EvalArmSummary(
|
|
1883
|
+
arm_id=arm_id,
|
|
1884
|
+
metrics=(
|
|
1885
|
+
_metric(
|
|
1886
|
+
EvalReportMetricId.VALID_COMPLETION,
|
|
1887
|
+
outcomes.count(EvalTrialOutcome.VALID_COMPLETION),
|
|
1888
|
+
record_denominator,
|
|
1889
|
+
),
|
|
1890
|
+
_metric(
|
|
1891
|
+
EvalReportMetricId.INVALID_TRIAL,
|
|
1892
|
+
outcomes.count(EvalTrialOutcome.INVALID_TRIAL),
|
|
1893
|
+
record_denominator,
|
|
1894
|
+
severity_one=True,
|
|
1895
|
+
),
|
|
1896
|
+
_metric(
|
|
1897
|
+
EvalReportMetricId.RUNTIME_FAILURE,
|
|
1898
|
+
outcomes.count(EvalTrialOutcome.RUNTIME_FAILURE),
|
|
1899
|
+
planned_denominator,
|
|
1900
|
+
),
|
|
1901
|
+
_metric(
|
|
1902
|
+
EvalReportMetricId.PROVIDER_FAILURE,
|
|
1903
|
+
outcomes.count(EvalTrialOutcome.PROVIDER_FAILURE),
|
|
1904
|
+
planned_denominator,
|
|
1905
|
+
),
|
|
1906
|
+
_metric(
|
|
1907
|
+
EvalReportMetricId.ARTIFACT_COMPLETE,
|
|
1908
|
+
sum(result.artifact_complete for result in arm_results),
|
|
1909
|
+
record_denominator,
|
|
1910
|
+
),
|
|
1911
|
+
),
|
|
1912
|
+
)
|
|
1913
|
+
)
|
|
1914
|
+
return tuple(summaries)
|
|
1915
|
+
|
|
1916
|
+
|
|
1917
|
+
def _task_summaries(
|
|
1918
|
+
*, plans: Sequence[EvalTrialPlan], records: Sequence[EvalTrialRecord]
|
|
1919
|
+
) -> tuple[EvalTaskSummary, ...]:
|
|
1920
|
+
record_by_trial = {record.trial_id: record for record in records}
|
|
1921
|
+
summaries: list[EvalTaskSummary] = []
|
|
1922
|
+
for plan in plans:
|
|
1923
|
+
record = record_by_trial.get(plan.trial_id)
|
|
1924
|
+
denominator = EvalMetricDenominator(
|
|
1925
|
+
denominator_kind=EvalMetricDenominatorKind.APPENDED_RECORDS,
|
|
1926
|
+
count=2 if record else 0,
|
|
1927
|
+
summary="Per-task appended arm records.",
|
|
1928
|
+
)
|
|
1929
|
+
success_count = (
|
|
1930
|
+
sum(result.scorer_result.primary_success for result in record.arm_results)
|
|
1931
|
+
if record
|
|
1932
|
+
else 0
|
|
1933
|
+
)
|
|
1934
|
+
invalid_count = (
|
|
1935
|
+
sum(
|
|
1936
|
+
result.scorer_result.final_outcome is EvalTrialOutcome.INVALID_TRIAL
|
|
1937
|
+
for result in record.arm_results
|
|
1938
|
+
)
|
|
1939
|
+
if record
|
|
1940
|
+
else 0
|
|
1941
|
+
)
|
|
1942
|
+
summaries.append(
|
|
1943
|
+
EvalTaskSummary(
|
|
1944
|
+
fixture_id=plan.fixture_instance.fixture_id,
|
|
1945
|
+
trial_index=plan.trial_index,
|
|
1946
|
+
category=plan.fixture_instance.public_projection.category.value,
|
|
1947
|
+
metrics=(
|
|
1948
|
+
_metric(
|
|
1949
|
+
EvalReportMetricId.VALID_COMPLETION,
|
|
1950
|
+
success_count,
|
|
1951
|
+
denominator,
|
|
1952
|
+
),
|
|
1953
|
+
_metric(
|
|
1954
|
+
EvalReportMetricId.INVALID_TRIAL,
|
|
1955
|
+
invalid_count,
|
|
1956
|
+
denominator,
|
|
1957
|
+
severity_one=True,
|
|
1958
|
+
),
|
|
1959
|
+
_metric(
|
|
1960
|
+
EvalReportMetricId.INCOMPLETE_TRIAL,
|
|
1961
|
+
0 if record else 1,
|
|
1962
|
+
EvalMetricDenominator(
|
|
1963
|
+
denominator_kind=EvalMetricDenominatorKind.PLANNED_TRIALS,
|
|
1964
|
+
count=1,
|
|
1965
|
+
summary="One planned fixture/trial-index pair.",
|
|
1966
|
+
),
|
|
1967
|
+
),
|
|
1968
|
+
*_model_usage_metrics(
|
|
1969
|
+
(record,) if record else (),
|
|
1970
|
+
summary_prefix="Per-task source-present model usage.",
|
|
1971
|
+
),
|
|
1972
|
+
*_resource_usage_metrics(
|
|
1973
|
+
(record,) if record else (),
|
|
1974
|
+
summary_prefix="Per-task source-present resource usage.",
|
|
1975
|
+
),
|
|
1976
|
+
),
|
|
1977
|
+
)
|
|
1978
|
+
)
|
|
1979
|
+
return tuple(summaries)
|
|
1980
|
+
|
|
1981
|
+
|
|
1982
|
+
def _category_summaries(
|
|
1983
|
+
*, plans: Sequence[EvalTrialPlan], records: Sequence[EvalTrialRecord]
|
|
1984
|
+
) -> tuple[EvalCategorySummary, ...]:
|
|
1985
|
+
planned_by_category: Counter[str] = Counter(
|
|
1986
|
+
plan.fixture_instance.public_projection.category.value for plan in plans
|
|
1987
|
+
)
|
|
1988
|
+
by_category: dict[str, list[EvalTrialOutcome]] = {
|
|
1989
|
+
category: [] for category in planned_by_category
|
|
1990
|
+
}
|
|
1991
|
+
records_by_category: dict[str, list[EvalTrialRecord]] = {
|
|
1992
|
+
category: [] for category in planned_by_category
|
|
1993
|
+
}
|
|
1994
|
+
for record in records:
|
|
1995
|
+
records_by_category.setdefault(record.task_category, []).append(record)
|
|
1996
|
+
by_category.setdefault(record.task_category, []).extend(
|
|
1997
|
+
result.scorer_result.final_outcome for result in record.arm_results
|
|
1998
|
+
)
|
|
1999
|
+
summaries: list[EvalCategorySummary] = []
|
|
2000
|
+
for category in sorted(by_category):
|
|
2001
|
+
outcomes = by_category[category]
|
|
2002
|
+
denominator = EvalMetricDenominator(
|
|
2003
|
+
denominator_kind=EvalMetricDenominatorKind.APPENDED_RECORDS,
|
|
2004
|
+
count=len(outcomes),
|
|
2005
|
+
summary="Per-category appended arm records.",
|
|
2006
|
+
)
|
|
2007
|
+
summaries.append(
|
|
2008
|
+
EvalCategorySummary(
|
|
2009
|
+
category=category,
|
|
2010
|
+
metrics=(
|
|
2011
|
+
_metric(
|
|
2012
|
+
EvalReportMetricId.VALID_COMPLETION,
|
|
2013
|
+
outcomes.count(EvalTrialOutcome.VALID_COMPLETION),
|
|
2014
|
+
denominator,
|
|
2015
|
+
),
|
|
2016
|
+
_metric(
|
|
2017
|
+
EvalReportMetricId.INVALID_TRIAL,
|
|
2018
|
+
outcomes.count(EvalTrialOutcome.INVALID_TRIAL),
|
|
2019
|
+
denominator,
|
|
2020
|
+
severity_one=True,
|
|
2021
|
+
),
|
|
2022
|
+
_metric(
|
|
2023
|
+
EvalReportMetricId.INCOMPLETE_TRIAL,
|
|
2024
|
+
max(0, planned_by_category[category] - len(outcomes) // 2),
|
|
2025
|
+
EvalMetricDenominator(
|
|
2026
|
+
denominator_kind=EvalMetricDenominatorKind.PLANNED_TRIALS,
|
|
2027
|
+
count=planned_by_category[category],
|
|
2028
|
+
summary="Planned trial pairs for this category.",
|
|
2029
|
+
),
|
|
2030
|
+
),
|
|
2031
|
+
*_model_usage_metrics(
|
|
2032
|
+
records_by_category.get(category, ()),
|
|
2033
|
+
summary_prefix="Per-category source-present model usage.",
|
|
2034
|
+
),
|
|
2035
|
+
*_resource_usage_metrics(
|
|
2036
|
+
records_by_category.get(category, ()),
|
|
2037
|
+
summary_prefix="Per-category source-present resource usage.",
|
|
2038
|
+
),
|
|
2039
|
+
),
|
|
2040
|
+
)
|
|
2041
|
+
)
|
|
2042
|
+
return tuple(summaries)
|
|
2043
|
+
|
|
2044
|
+
|
|
2045
|
+
def _missing_pair_keys(
|
|
2046
|
+
*, plans: Sequence[EvalTrialPlan], records: Sequence[EvalTrialRecord]
|
|
2047
|
+
) -> set[tuple[str, int]]:
|
|
2048
|
+
recorded_pairs = {(record.fixture_id, record.trial_index) for record in records}
|
|
2049
|
+
return {
|
|
2050
|
+
(plan.fixture_instance.fixture_id, plan.trial_index)
|
|
2051
|
+
for plan in plans
|
|
2052
|
+
if (plan.fixture_instance.fixture_id, plan.trial_index) not in recorded_pairs
|
|
2053
|
+
}
|
|
2054
|
+
|
|
2055
|
+
|
|
2056
|
+
def _pending_trial_count(
|
|
2057
|
+
*,
|
|
2058
|
+
plans: Sequence[EvalTrialPlan],
|
|
2059
|
+
records: Sequence[EvalTrialRecord],
|
|
2060
|
+
resume_index: EvalTrialResumeIndex | None,
|
|
2061
|
+
) -> int:
|
|
2062
|
+
if resume_index is None:
|
|
2063
|
+
completed_plan_hashes = {record.trial_plan_hash for record in records}
|
|
2064
|
+
return sum(plan.plan_hash not in completed_plan_hashes for plan in plans)
|
|
2065
|
+
plan_hashes = {plan.plan_hash for plan in plans}
|
|
2066
|
+
return sum(
|
|
2067
|
+
plan_hash in plan_hashes for plan_hash in resume_index.pending_trial_plan_hashes
|
|
2068
|
+
)
|
|
2069
|
+
|
|
2070
|
+
|
|
2071
|
+
def _validate_manual_taxonomy_assignments(
|
|
2072
|
+
manual_taxonomy: Mapping[str, EvalFailureTaxonomyAssignment | Mapping[str, Any]],
|
|
2073
|
+
) -> Mapping[str, EvalFailureTaxonomyAssignment]:
|
|
2074
|
+
assignments: dict[str, EvalFailureTaxonomyAssignment] = {}
|
|
2075
|
+
for key, assignment in manual_taxonomy.items():
|
|
2076
|
+
if not key.strip():
|
|
2077
|
+
raise ValueError("manual taxonomy assignment keys must be non-empty")
|
|
2078
|
+
assignments[key] = EvalFailureTaxonomyAssignment.model_validate(assignment)
|
|
2079
|
+
return _freeze_eval_report_mapping(assignments)
|
|
2080
|
+
|
|
2081
|
+
|
|
2082
|
+
def _statistical_summaries(
|
|
2083
|
+
*,
|
|
2084
|
+
metrics: Sequence[EvalMetricValue],
|
|
2085
|
+
paired_comparisons: Sequence[EvalPairedComparison],
|
|
2086
|
+
usage: EvalBudgetUsageEstimate,
|
|
2087
|
+
) -> tuple[EvalReportStatisticalSummary, ...]:
|
|
2088
|
+
paired_by_metric = {
|
|
2089
|
+
comparison.metric_id: comparison for comparison in paired_comparisons
|
|
2090
|
+
}
|
|
2091
|
+
summaries = [
|
|
2092
|
+
_metric_statistical_summary(
|
|
2093
|
+
metric,
|
|
2094
|
+
paired_comparison=paired_by_metric.get(metric.metric_id),
|
|
2095
|
+
)
|
|
2096
|
+
for metric in metrics
|
|
2097
|
+
if metric.metric_id
|
|
2098
|
+
in {
|
|
2099
|
+
EvalReportMetricId.VALID_COMPLETION,
|
|
2100
|
+
EvalReportMetricId.FALSE_CLOSURE,
|
|
2101
|
+
EvalReportMetricId.FALSE_BLOCKED,
|
|
2102
|
+
EvalReportMetricId.CAPABILITY_VIOLATION,
|
|
2103
|
+
EvalReportMetricId.PROVIDER_FAILURE,
|
|
2104
|
+
EvalReportMetricId.RUNTIME_FAILURE,
|
|
2105
|
+
EvalReportMetricId.INVALID_TRIAL,
|
|
2106
|
+
}
|
|
2107
|
+
]
|
|
2108
|
+
summaries.append(
|
|
2109
|
+
EvalReportStatisticalSummary(
|
|
2110
|
+
metric_id=EvalReportMetricId.ESTIMATED_COST,
|
|
2111
|
+
raw_count=0,
|
|
2112
|
+
denominator_count=0,
|
|
2113
|
+
rate=0.0,
|
|
2114
|
+
distributions=(
|
|
2115
|
+
_distribution_summary(
|
|
2116
|
+
"estimated_cost_usd",
|
|
2117
|
+
(usage.estimated_spend_usd,),
|
|
2118
|
+
),
|
|
2119
|
+
),
|
|
2120
|
+
diagnostic="Cost summaries are descriptive and use source-present campaign usage only.",
|
|
2121
|
+
descriptive_only=True,
|
|
2122
|
+
)
|
|
2123
|
+
)
|
|
2124
|
+
summaries.append(
|
|
2125
|
+
EvalReportStatisticalSummary(
|
|
2126
|
+
metric_id=EvalReportMetricId.WALL_CLOCK_SECONDS,
|
|
2127
|
+
raw_count=0,
|
|
2128
|
+
denominator_count=0,
|
|
2129
|
+
rate=0.0,
|
|
2130
|
+
distributions=(
|
|
2131
|
+
_distribution_summary(
|
|
2132
|
+
"wall_clock_seconds",
|
|
2133
|
+
(float(usage.wall_clock_seconds),),
|
|
2134
|
+
),
|
|
2135
|
+
),
|
|
2136
|
+
diagnostic="Latency summaries are descriptive and use source-present campaign usage only.",
|
|
2137
|
+
descriptive_only=True,
|
|
2138
|
+
)
|
|
2139
|
+
)
|
|
2140
|
+
return tuple(summaries)
|
|
2141
|
+
|
|
2142
|
+
|
|
2143
|
+
def _metric_statistical_summary(
|
|
2144
|
+
metric: EvalMetricValue,
|
|
2145
|
+
*,
|
|
2146
|
+
paired_comparison: EvalPairedComparison | None,
|
|
2147
|
+
) -> EvalReportStatisticalSummary:
|
|
2148
|
+
interval = wilson_score_interval(metric.count, metric.denominator.count)
|
|
2149
|
+
paired_differences = (
|
|
2150
|
+
(paired_comparison.difference,) if paired_comparison is not None else ()
|
|
2151
|
+
)
|
|
2152
|
+
if interval is None:
|
|
2153
|
+
return EvalReportStatisticalSummary(
|
|
2154
|
+
metric_id=metric.metric_id,
|
|
2155
|
+
raw_count=metric.count,
|
|
2156
|
+
denominator_count=metric.denominator.count,
|
|
2157
|
+
rate=metric.rate,
|
|
2158
|
+
paired_differences=paired_differences,
|
|
2159
|
+
diagnostic=(
|
|
2160
|
+
"Small-N pilot diagnostic only; Wilson interval omitted until "
|
|
2161
|
+
"the eligible denominator is at least 30."
|
|
2162
|
+
),
|
|
2163
|
+
descriptive_only=True,
|
|
2164
|
+
)
|
|
2165
|
+
return EvalReportStatisticalSummary(
|
|
2166
|
+
metric_id=metric.metric_id,
|
|
2167
|
+
raw_count=metric.count,
|
|
2168
|
+
denominator_count=metric.denominator.count,
|
|
2169
|
+
rate=metric.rate,
|
|
2170
|
+
wilson_interval=EvalWilsonScoreInterval(
|
|
2171
|
+
successes=metric.count,
|
|
2172
|
+
total=metric.denominator.count,
|
|
2173
|
+
confidence_level=0.95,
|
|
2174
|
+
lower=interval[0],
|
|
2175
|
+
upper=interval[1],
|
|
2176
|
+
),
|
|
2177
|
+
paired_differences=paired_differences,
|
|
2178
|
+
diagnostic="Wilson score interval emitted for eligible descriptive counts.",
|
|
2179
|
+
descriptive_only=False,
|
|
2180
|
+
)
|
|
2181
|
+
|
|
2182
|
+
|
|
2183
|
+
def _distribution_summary(
|
|
2184
|
+
statistic_id: str,
|
|
2185
|
+
values: Sequence[float],
|
|
2186
|
+
) -> EvalDistributionSummary:
|
|
2187
|
+
raw_values = tuple(float(value) for value in values)
|
|
2188
|
+
if not raw_values:
|
|
2189
|
+
return EvalDistributionSummary(statistic_id=statistic_id, sample_count=0)
|
|
2190
|
+
ordered = tuple(sorted(raw_values))
|
|
2191
|
+
return EvalDistributionSummary(
|
|
2192
|
+
statistic_id=statistic_id,
|
|
2193
|
+
sample_count=len(raw_values),
|
|
2194
|
+
raw_values=raw_values,
|
|
2195
|
+
median=float(median(ordered)),
|
|
2196
|
+
p90=_nearest_rank_percentile(ordered, 0.90),
|
|
2197
|
+
p95=_nearest_rank_percentile(ordered, 0.95),
|
|
2198
|
+
)
|
|
2199
|
+
|
|
2200
|
+
|
|
2201
|
+
def _nearest_rank_percentile(values: Sequence[float], percentile: float) -> float:
|
|
2202
|
+
if not values:
|
|
2203
|
+
raise ValueError("percentile requires at least one value")
|
|
2204
|
+
index = max(0, min(len(values) - 1, int(round(percentile * len(values) + 0.5)) - 1))
|
|
2205
|
+
return float(values[index])
|
|
2206
|
+
|
|
2207
|
+
|
|
2208
|
+
def _taxonomy_summaries(
|
|
2209
|
+
records: Sequence[EvalTrialRecord],
|
|
2210
|
+
manual_taxonomy: Mapping[str, EvalFailureTaxonomyAssignment],
|
|
2211
|
+
) -> tuple[EvalFailureTaxonomySummary, ...]:
|
|
2212
|
+
counts: Counter[EvalReportFailureTaxonomyCategory] = Counter()
|
|
2213
|
+
for assignment in manual_taxonomy.values():
|
|
2214
|
+
counts[assignment.primary_category] += 1
|
|
2215
|
+
counts.update(assignment.contributing_categories)
|
|
2216
|
+
for record in records:
|
|
2217
|
+
for result in record.arm_results:
|
|
2218
|
+
for label in result.scorer_result.failure_labels:
|
|
2219
|
+
category = _failure_label_to_report_category(label)
|
|
2220
|
+
if category is not None:
|
|
2221
|
+
counts[category] += 1
|
|
2222
|
+
return tuple(
|
|
2223
|
+
EvalFailureTaxonomySummary(category=category, count=counts[category])
|
|
2224
|
+
for category in sorted(counts, key=lambda item: item.value)
|
|
2225
|
+
)
|
|
2226
|
+
|
|
2227
|
+
|
|
2228
|
+
def _invalid_trial_summary(
|
|
2229
|
+
records: Sequence[EvalTrialRecord],
|
|
2230
|
+
) -> EvalInvalidTrialSummary:
|
|
2231
|
+
explanations = tuple(
|
|
2232
|
+
explanation
|
|
2233
|
+
for record in records
|
|
2234
|
+
for explanation in record.invalid_trial_explanations.values()
|
|
2235
|
+
)
|
|
2236
|
+
invalid = sum(
|
|
2237
|
+
result.scorer_result.final_outcome is EvalTrialOutcome.INVALID_TRIAL
|
|
2238
|
+
for record in records
|
|
2239
|
+
for result in record.arm_results
|
|
2240
|
+
)
|
|
2241
|
+
appended = len(records) * 2
|
|
2242
|
+
return EvalInvalidTrialSummary(
|
|
2243
|
+
invalid_trial_count=invalid,
|
|
2244
|
+
appended_record_count=appended,
|
|
2245
|
+
invalid_trial_rate=0.0 if appended == 0 else invalid / appended,
|
|
2246
|
+
diagnostics=explanations,
|
|
2247
|
+
)
|
|
2248
|
+
|
|
2249
|
+
|
|
2250
|
+
def _validate_report_inputs(
|
|
2251
|
+
campaign_manifest: EvalCampaignManifest,
|
|
2252
|
+
plans: Sequence[EvalTrialPlan],
|
|
2253
|
+
records: Sequence[EvalTrialRecord],
|
|
2254
|
+
resume_index: EvalTrialResumeIndex | None,
|
|
2255
|
+
) -> None:
|
|
2256
|
+
if not plans:
|
|
2257
|
+
raise ValueError("reports require at least one plan")
|
|
2258
|
+
if any(
|
|
2259
|
+
plan.campaign_manifest.campaign_id != campaign_manifest.campaign_id
|
|
2260
|
+
for plan in plans
|
|
2261
|
+
):
|
|
2262
|
+
raise ValueError("plan campaign ID must match campaign manifest")
|
|
2263
|
+
if any(
|
|
2264
|
+
plan.campaign_manifest.campaign_manifest_hash
|
|
2265
|
+
!= campaign_manifest.campaign_manifest_hash
|
|
2266
|
+
for plan in plans
|
|
2267
|
+
):
|
|
2268
|
+
raise ValueError("plan campaign hash must match campaign manifest")
|
|
2269
|
+
plan_by_id = {plan.trial_id: plan for plan in plans}
|
|
2270
|
+
if len(plan_by_id) != len(plans):
|
|
2271
|
+
raise ValueError("reports reject duplicate planned trial IDs")
|
|
2272
|
+
plan_hashes = {plan.plan_hash for plan in plans}
|
|
2273
|
+
for record in records:
|
|
2274
|
+
plan = plan_by_id.get(record.trial_id)
|
|
2275
|
+
if plan is None:
|
|
2276
|
+
raise ValueError("record trial_id must be present in plans")
|
|
2277
|
+
checks = {
|
|
2278
|
+
"campaign ID": record.campaign_id == campaign_manifest.campaign_id,
|
|
2279
|
+
"campaign hash": record.campaign_manifest_hash
|
|
2280
|
+
== campaign_manifest.campaign_manifest_hash,
|
|
2281
|
+
"trial ID": record.trial_id == plan.trial_id,
|
|
2282
|
+
"trial index": record.trial_index == plan.trial_index,
|
|
2283
|
+
"trial hash": record.trial_plan_hash == plan.plan_hash,
|
|
2284
|
+
"fixture ID": record.fixture_id == plan.fixture_instance.fixture_id,
|
|
2285
|
+
"fixture instance ID": record.fixture_instance_id
|
|
2286
|
+
== plan.fixture_instance.fixture_instance_id,
|
|
2287
|
+
"fixture hash": record.fixture_hash == plan.fixture_instance.fixture_hash,
|
|
2288
|
+
"fixture snapshot hash": record.fixture_snapshot_hash
|
|
2289
|
+
== plan.fixture_instance.fixture_snapshot_hash,
|
|
2290
|
+
"model hash": record.model_manifest_hash
|
|
2291
|
+
== campaign_manifest.model_manifest_hash,
|
|
2292
|
+
"workflow hash": record.workflow_graph_hash
|
|
2293
|
+
== campaign_manifest.workflow_graph_hash,
|
|
2294
|
+
"fixture pack hash": record.fixture_pack_hash
|
|
2295
|
+
== campaign_manifest.fixture_pack_hash,
|
|
2296
|
+
}
|
|
2297
|
+
for label, valid in checks.items():
|
|
2298
|
+
if not valid:
|
|
2299
|
+
raise ValueError(f"record {label} must match report inputs")
|
|
2300
|
+
if tuple(record.arm_order) != tuple(plan.arm_order):
|
|
2301
|
+
raise ValueError("record arm order must match campaign plan")
|
|
2302
|
+
arm_plans = {arm_plan.arm_id: arm_plan for arm_plan in plan.arm_plans}
|
|
2303
|
+
for result in record.arm_results:
|
|
2304
|
+
arm_plan = arm_plans.get(result.arm_id)
|
|
2305
|
+
if arm_plan is None:
|
|
2306
|
+
raise ValueError("record arm must be present in campaign plan")
|
|
2307
|
+
result_checks = {
|
|
2308
|
+
"trial ID": result.trial_id == plan.trial_id,
|
|
2309
|
+
"trial hash": result.trial_plan_hash == plan.plan_hash,
|
|
2310
|
+
"fixture ID": result.scorer_result.fixture_id
|
|
2311
|
+
== plan.fixture_instance.fixture_id,
|
|
2312
|
+
"scorer input trial ID": result.scorer_input.trial_id == plan.trial_id,
|
|
2313
|
+
"scorer input fixture hash": result.scorer_input.fixture_hash
|
|
2314
|
+
== plan.fixture_instance.fixture_hash,
|
|
2315
|
+
"arm": result.arm_id == arm_plan.arm_id,
|
|
2316
|
+
}
|
|
2317
|
+
for label, valid in result_checks.items():
|
|
2318
|
+
if not valid:
|
|
2319
|
+
raise ValueError(f"scorer result {label} must match report inputs")
|
|
2320
|
+
if result.scorer_result.scorer_version != campaign_manifest.scorer_version:
|
|
2321
|
+
raise ValueError("scorer version must match campaign manifest")
|
|
2322
|
+
if resume_index is not None:
|
|
2323
|
+
if (
|
|
2324
|
+
resume_index.campaign_manifest_hash
|
|
2325
|
+
!= campaign_manifest.campaign_manifest_hash
|
|
2326
|
+
):
|
|
2327
|
+
raise ValueError("resume index campaign hash must match campaign manifest")
|
|
2328
|
+
record_hashes = {record.record_hash for record in records}
|
|
2329
|
+
completed_hashes = set(resume_index.completed_trial_record_hashes)
|
|
2330
|
+
pending_hashes = set(resume_index.pending_trial_plan_hashes)
|
|
2331
|
+
if not completed_hashes.issubset(record_hashes):
|
|
2332
|
+
raise ValueError("resume index completed records must be included")
|
|
2333
|
+
if not pending_hashes.issubset(plan_hashes):
|
|
2334
|
+
raise ValueError("resume index pending plans must be included")
|
|
2335
|
+
completed_plan_hashes = {record.trial_plan_hash for record in records}
|
|
2336
|
+
if completed_plan_hashes & pending_hashes:
|
|
2337
|
+
raise ValueError("resume index cannot mark completed plans as pending")
|
|
2338
|
+
if not completed_plan_hashes.issubset(plan_hashes):
|
|
2339
|
+
raise ValueError("resume index completed plans must be included")
|
|
2340
|
+
|
|
2341
|
+
|
|
2342
|
+
def _report_input_hash(
|
|
2343
|
+
campaign_manifest: EvalCampaignManifest,
|
|
2344
|
+
plans: Sequence[EvalTrialPlan],
|
|
2345
|
+
records: Sequence[EvalTrialRecord],
|
|
2346
|
+
resume_index: EvalTrialResumeIndex | None,
|
|
2347
|
+
) -> str:
|
|
2348
|
+
payload = {
|
|
2349
|
+
"campaign_manifest_hash": campaign_manifest.campaign_manifest_hash,
|
|
2350
|
+
"plan_hashes": tuple(plan.plan_hash for plan in plans),
|
|
2351
|
+
"record_hashes": tuple(record.record_hash for record in records),
|
|
2352
|
+
"resume_index_hash": resume_index.resume_index_hash
|
|
2353
|
+
if resume_index is not None
|
|
2354
|
+
else None,
|
|
2355
|
+
}
|
|
2356
|
+
return hashlib.sha256(canonical_eval_report_bytes(payload)).hexdigest()
|
|
2357
|
+
|
|
2358
|
+
|
|
2359
|
+
def _threshold_status(
|
|
2360
|
+
metric: EvalMetricValue | None,
|
|
2361
|
+
threshold: float,
|
|
2362
|
+
) -> EvalDecisionRuleStatus:
|
|
2363
|
+
if metric is None or metric.denominator.count < 30:
|
|
2364
|
+
return EvalDecisionRuleStatus.DESCRIPTIVE_ONLY
|
|
2365
|
+
return (
|
|
2366
|
+
EvalDecisionRuleStatus.PASSED
|
|
2367
|
+
if metric.rate <= threshold
|
|
2368
|
+
else EvalDecisionRuleStatus.FAILED
|
|
2369
|
+
)
|
|
2370
|
+
|
|
2371
|
+
|
|
2372
|
+
def _budget_diagnostic(
|
|
2373
|
+
code: EvalBudgetDiagnosticCode,
|
|
2374
|
+
rule_id: str,
|
|
2375
|
+
summary: str,
|
|
2376
|
+
) -> EvalBudgetDiagnostic:
|
|
2377
|
+
return EvalBudgetDiagnostic(diagnostic_code=code, rule_id=rule_id, summary=summary)
|
|
2378
|
+
|
|
2379
|
+
|
|
2380
|
+
def _failure_label_to_report_category(
|
|
2381
|
+
label: EvalFailureTaxonomyLabel,
|
|
2382
|
+
) -> EvalReportFailureTaxonomyCategory | None:
|
|
2383
|
+
return {
|
|
2384
|
+
EvalFailureTaxonomyLabel.REQUIRED_ARTIFACT_MISSING: (
|
|
2385
|
+
EvalReportFailureTaxonomyCategory.MISSING_ARTIFACT
|
|
2386
|
+
),
|
|
2387
|
+
EvalFailureTaxonomyLabel.REQUIRED_ARTIFACT_MALFORMED: (
|
|
2388
|
+
EvalReportFailureTaxonomyCategory.INVALID_PATCH
|
|
2389
|
+
),
|
|
2390
|
+
EvalFailureTaxonomyLabel.FALSE_SUCCESS_TERMINAL: (
|
|
2391
|
+
EvalReportFailureTaxonomyCategory.UNSUPPORTED_SUCCESS_CLAIM
|
|
2392
|
+
),
|
|
2393
|
+
EvalFailureTaxonomyLabel.CAPABILITY_VIOLATION: (
|
|
2394
|
+
EvalReportFailureTaxonomyCategory.CAPABILITY_VIOLATION
|
|
2395
|
+
),
|
|
2396
|
+
EvalFailureTaxonomyLabel.PROVIDER_DEFECT: (
|
|
2397
|
+
EvalReportFailureTaxonomyCategory.PROVIDER_FAILURE
|
|
2398
|
+
),
|
|
2399
|
+
EvalFailureTaxonomyLabel.INFRASTRUCTURE_DEFECT: (
|
|
2400
|
+
EvalReportFailureTaxonomyCategory.INVALID_TRIAL_INFRASTRUCTURE
|
|
2401
|
+
),
|
|
2402
|
+
}.get(label)
|
|
2403
|
+
|
|
2404
|
+
|
|
2405
|
+
def _confound_summary(confound_id: EvalReportConfoundId) -> str:
|
|
2406
|
+
return {
|
|
2407
|
+
EvalReportConfoundId.PI_PROMPT_TOOL_BEHAVIOR: "Pi prompt and tool behavior can differ from Millforge.",
|
|
2408
|
+
EvalReportConfoundId.MILLFORGE_PROMPT_TOOL_BEHAVIOR: "Millforge prompt and tool behavior can affect outcomes.",
|
|
2409
|
+
EvalReportConfoundId.HARNESS_BEHAVIOR: "Harness behavior can affect trial setup, execution, and evidence capture.",
|
|
2410
|
+
EvalReportConfoundId.CONTEXT_PACKING: "Context packing choices can affect task evidence.",
|
|
2411
|
+
EvalReportConfoundId.PARSER_FALLBACK: "Parser fallback behavior can affect terminal interpretation.",
|
|
2412
|
+
EvalReportConfoundId.PROVIDER_NONDETERMINISM: "Provider nondeterminism can affect live results.",
|
|
2413
|
+
EvalReportConfoundId.RATE_LIMITING: "Rate limiting can affect latency and retry behavior.",
|
|
2414
|
+
EvalReportConfoundId.CACHED_PROVIDER_RESPONSES: "Cached provider responses can affect cost and latency.",
|
|
2415
|
+
EvalReportConfoundId.TOKEN_ACCOUNTING_DIFFERENCES: "Token accounting can differ across backends.",
|
|
2416
|
+
EvalReportConfoundId.SAMPLING_PARAMETER_MISMATCH: "Sampling-parameter mismatches can affect comparability.",
|
|
2417
|
+
EvalReportConfoundId.OFFLINE_FAKE_LIMITATIONS: "Offline fake execution cannot support model-performance claims.",
|
|
2418
|
+
}[confound_id]
|
|
2419
|
+
|
|
2420
|
+
|
|
2421
|
+
def _freeze_eval_report_mapping(mapping: Mapping[Any, Any]) -> _FrozenEvalReportDict:
|
|
2422
|
+
return _FrozenEvalReportDict(mapping)
|
|
2423
|
+
|
|
2424
|
+
|
|
2425
|
+
def _validate_sha256(value: str) -> None:
|
|
2426
|
+
if not _SHA256_RE.fullmatch(value):
|
|
2427
|
+
raise ValueError("value must be a lowercase sha256 digest")
|
|
2428
|
+
|
|
2429
|
+
|
|
2430
|
+
def _reject_forbidden_material(value: Any, *, field_name: str | None = None) -> None:
|
|
2431
|
+
if field_name is not None:
|
|
2432
|
+
lowered = field_name.lower()
|
|
2433
|
+
if any(marker in lowered for marker in _SECRET_FIELD_MARKERS):
|
|
2434
|
+
raise ValueError("secret-like field names are forbidden in eval reports")
|
|
2435
|
+
if "endpoint" in lowered or "url" in lowered:
|
|
2436
|
+
raise ValueError("endpoint URLs are forbidden in eval reports")
|
|
2437
|
+
if isinstance(value, Mapping):
|
|
2438
|
+
for key, item in value.items():
|
|
2439
|
+
_reject_forbidden_material(item, field_name=str(key))
|
|
2440
|
+
return
|
|
2441
|
+
if isinstance(value, (tuple, list, set, frozenset)):
|
|
2442
|
+
for item in value:
|
|
2443
|
+
_reject_forbidden_material(item)
|
|
2444
|
+
return
|
|
2445
|
+
if isinstance(value, str):
|
|
2446
|
+
lowered = value.lower()
|
|
2447
|
+
if any(token in lowered for token in _DENIED_TEXT_TOKENS):
|
|
2448
|
+
raise ValueError("forbidden private material in eval report payload")
|
|
2449
|
+
if _ENDPOINT_URL.search(value):
|
|
2450
|
+
raise ValueError("endpoint URLs are forbidden in eval reports")
|
|
2451
|
+
if (
|
|
2452
|
+
_WINDOWS_ABSOLUTE_PATH.search(value)
|
|
2453
|
+
or _POSIX_ABSOLUTE_PATH.search(value)
|
|
2454
|
+
or _USER_HOME_PATH.search(value)
|
|
2455
|
+
):
|
|
2456
|
+
raise ValueError("host absolute paths are forbidden in eval reports")
|
|
2457
|
+
if any(pattern.search(value) for pattern in _CREDENTIAL_VALUE_PATTERNS):
|
|
2458
|
+
raise ValueError("credential-like values are forbidden in eval reports")
|
|
2459
|
+
|
|
2460
|
+
|
|
2461
|
+
__all__ = [
|
|
2462
|
+
"EVAL_REPORT_HASH_KIND",
|
|
2463
|
+
"EVAL_REPORT_JSON_HASH_KIND",
|
|
2464
|
+
"EVAL_REPORT_MARKDOWN_HASH_KIND",
|
|
2465
|
+
"EVAL_REPORT_SCHEMA_VERSION",
|
|
2466
|
+
"EvalArmSummary",
|
|
2467
|
+
"EvalBudgetDiagnostic",
|
|
2468
|
+
"EvalBudgetDiagnosticCode",
|
|
2469
|
+
"EvalBudgetUsageEstimate",
|
|
2470
|
+
"EvalBudgetValidationResult",
|
|
2471
|
+
"EvalCategorySummary",
|
|
2472
|
+
"EvalConfoundEntry",
|
|
2473
|
+
"EvalDecisionRuleKind",
|
|
2474
|
+
"EvalDecisionRule",
|
|
2475
|
+
"EvalDecisionRuleStatus",
|
|
2476
|
+
"EvalDistributionSummary",
|
|
2477
|
+
"EvalFailureTaxonomyAssignment",
|
|
2478
|
+
"EvalFailureTaxonomySummary",
|
|
2479
|
+
"EvalInvalidTrialSummary",
|
|
2480
|
+
"EvalLiveAdmissionDiagnostic",
|
|
2481
|
+
"EvalLiveAdmissionDiagnosticCode",
|
|
2482
|
+
"EvalLiveAdmissionResult",
|
|
2483
|
+
"EvalLiveAdmissionStatus",
|
|
2484
|
+
"EvalMarkdownReport",
|
|
2485
|
+
"EvalMetricDenominator",
|
|
2486
|
+
"EvalMetricDenominatorKind",
|
|
2487
|
+
"EvalMetricValue",
|
|
2488
|
+
"EvalPairedComparison",
|
|
2489
|
+
"EvalPromotionalFreeWindow",
|
|
2490
|
+
"EvalReportAbortThresholds",
|
|
2491
|
+
"EvalReportBudgetPolicy",
|
|
2492
|
+
"EvalReportConfoundId",
|
|
2493
|
+
"EvalReportContractModel",
|
|
2494
|
+
"EvalReportFailureTaxonomyCategory",
|
|
2495
|
+
"EvalReportMetricId",
|
|
2496
|
+
"EvalReportPayload",
|
|
2497
|
+
"EvalReportPricingClass",
|
|
2498
|
+
"EvalReportRateLimitPolicy",
|
|
2499
|
+
"EvalReportReproducibilityHashes",
|
|
2500
|
+
"EvalReportStatisticalSummary",
|
|
2501
|
+
"EvalTaskSummary",
|
|
2502
|
+
"EvalWilsonScoreInterval",
|
|
2503
|
+
"admit_eval_report_campaign",
|
|
2504
|
+
"build_eval_report_artifact_bytes",
|
|
2505
|
+
"build_eval_report_payload",
|
|
2506
|
+
"calculate_eval_report_hash",
|
|
2507
|
+
"calculate_eval_report_json_hash",
|
|
2508
|
+
"canonical_eval_markdown_report_bytes",
|
|
2509
|
+
"canonical_eval_report_bytes",
|
|
2510
|
+
"canonical_eval_report_json_bytes",
|
|
2511
|
+
"default_eval_report_budget_policy",
|
|
2512
|
+
"default_eval_report_confounds",
|
|
2513
|
+
"default_eval_report_decision_rules",
|
|
2514
|
+
"render_eval_markdown_report",
|
|
2515
|
+
"validate_eval_budget_policy",
|
|
2516
|
+
"wilson_score_interval",
|
|
2517
|
+
]
|