millforge 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- millforge/__init__.py +1174 -0
- millforge/_forge/LICENSE +21 -0
- millforge/_forge/PROVENANCE.json +295 -0
- millforge/_forge/UPDATE_POLICY.md +24 -0
- millforge/_forge/__init__.py +14 -0
- millforge/_forge/adapter.py +2232 -0
- millforge/_forge/base_runner.py +121 -0
- millforge/_forge/clients/__init__.py +10 -0
- millforge/_forge/clients/base.py +200 -0
- millforge/_forge/context/__init__.py +23 -0
- millforge/_forge/context/manager.py +178 -0
- millforge/_forge/context/strategies.py +335 -0
- millforge/_forge/core/__init__.py +16 -0
- millforge/_forge/core/inference.py +433 -0
- millforge/_forge/core/messages.py +119 -0
- millforge/_forge/core/runner.py +479 -0
- millforge/_forge/core/steps.py +108 -0
- millforge/_forge/core/workflow.py +400 -0
- millforge/_forge/errors.py +222 -0
- millforge/_forge/guardrails/__init__.py +21 -0
- millforge/_forge/guardrails/error_tracker.py +71 -0
- millforge/_forge/guardrails/guardrails.py +194 -0
- millforge/_forge/guardrails/nudge.py +47 -0
- millforge/_forge/guardrails/response_validator.py +119 -0
- millforge/_forge/guardrails/step_enforcer.py +183 -0
- millforge/_forge/prompts/__init__.py +16 -0
- millforge/_forge/prompts/nudges.py +95 -0
- millforge/_forge/prompts/templates.py +285 -0
- millforge/_version.py +3 -0
- millforge/artifacts.py +570 -0
- millforge/base/__init__.py +97 -0
- millforge/base/composition.py +402 -0
- millforge/base/context.py +285 -0
- millforge/base/harness.py +138 -0
- millforge/base/identity.py +465 -0
- millforge/base/options.py +34 -0
- millforge/base/platform.py +17 -0
- millforge/base/prompt.py +317 -0
- millforge/base/runner.py +546 -0
- millforge/compiled_plan.py +970 -0
- millforge/compiler/__init__.py +231 -0
- millforge/compiler/artifact_validation.py +257 -0
- millforge/compiler/canonicalization.py +169 -0
- millforge/compiler/capabilities.py +66 -0
- millforge/compiler/catalogs.py +500 -0
- millforge/compiler/diagnostics.py +491 -0
- millforge/compiler/graph.py +678 -0
- millforge/compiler/lowering.py +198 -0
- millforge/compiler/output.py +692 -0
- millforge/compiler/parsing.py +1424 -0
- millforge/compiler/requests.py +1180 -0
- millforge/compiler/schema_validation.py +272 -0
- millforge/compiler/semantic.py +490 -0
- millforge/compiler/service.py +448 -0
- millforge/compiler/source.py +375 -0
- millforge/compiler/validators.py +184 -0
- millforge/connectors/__init__.py +95 -0
- millforge/connectors/admission.py +801 -0
- millforge/connectors/broker.py +202 -0
- millforge/connectors/contracts.py +1159 -0
- millforge/connectors/diagnostics.py +189 -0
- millforge/connectors/fake.py +66 -0
- millforge/connectors/runtime.py +236 -0
- millforge/contracts.py +2860 -0
- millforge/custom_tools/__init__.py +67 -0
- millforge/custom_tools/compiler.py +724 -0
- millforge/custom_tools/contracts.py +1093 -0
- millforge/custom_tools/diagnostics.py +205 -0
- millforge/eval_artifacts.py +952 -0
- millforge/eval_boundary.py +2435 -0
- millforge/eval_fixtures/__init__.py +1 -0
- millforge/eval_fixtures/default_pack/__init__.py +1 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.bug_diagnosis.traceback.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.direct_edit.import_sort.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.evidence_discipline.no_source_change.v1.json +51 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.false_closure.visible_green.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.multi_file.api_contract.v1.json +54 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.recovery.malformed_artifact.v1.json +54 -0
- millforge/eval_fixtures/default_pack/manifest.json +12 -0
- millforge/eval_modes.py +1282 -0
- millforge/eval_presets.py +1398 -0
- millforge/eval_reports.py +2517 -0
- millforge/eval_suite.py +2429 -0
- millforge/eval_trials.py +2632 -0
- millforge/eval_workflow.py +794 -0
- millforge/exceptions.py +122 -0
- millforge/model_backend.py +2098 -0
- millforge/protocols.py +340 -0
- millforge/py.typed +0 -0
- millforge/runtime.py +1791 -0
- millforge/testing/__init__.py +1089 -0
- millforge/tools/__init__.py +83 -0
- millforge/tools/builtin_runtime.py +1339 -0
- millforge/tools/builtins.py +773 -0
- millforge/tools/execution.py +1545 -0
- millforge/tools/path_policy.py +155 -0
- millforge/tools/pi_compat/PI_LICENSE +21 -0
- millforge/tools/pi_compat/PROVENANCE.json +55 -0
- millforge/tools/pi_compat/UPDATE_POLICY.md +36 -0
- millforge/tools/pi_compat/__init__.py +34 -0
- millforge/tools/pi_compat/contracts.py +49 -0
- millforge/tools/pi_compat/editing.py +390 -0
- millforge/tools/pi_compat/mutations.py +57 -0
- millforge/tools/pi_compat/operations.py +401 -0
- millforge/tools/pi_compat/paths.py +155 -0
- millforge/tools/pi_compat/process.py +1375 -0
- millforge/tools/pi_compat/search.py +738 -0
- millforge/tools/pi_compat/truncation.py +267 -0
- millforge/tools/pi_compat_catalog.py +396 -0
- millforge/tools/pi_compat_runtime.py +460 -0
- millforge/tools/registry.py +553 -0
- millforge/tools/results.py +533 -0
- millforge-0.1.0.dist-info/METADATA +844 -0
- millforge-0.1.0.dist-info/RECORD +116 -0
- millforge-0.1.0.dist-info/WHEEL +4 -0
- millforge-0.1.0.dist-info/licenses/LICENSE +201 -0
millforge/eval_suite.py
ADDED
|
@@ -0,0 +1,2429 @@
|
|
|
1
|
+
"""Public 08A offline eval-suite contract boundary."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
import re
|
|
8
|
+
import shlex
|
|
9
|
+
from importlib.resources import files
|
|
10
|
+
from collections.abc import Mapping
|
|
11
|
+
from enum import Enum
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
from pydantic import BaseModel, ConfigDict, Field, StrictBool, StrictFloat, StrictInt
|
|
16
|
+
from pydantic import StrictStr, field_validator, model_validator
|
|
17
|
+
|
|
18
|
+
from millforge.eval_modes import (
|
|
19
|
+
EVAL_SMALL_MILLFORGE_MODE_ID,
|
|
20
|
+
EVAL_SMALL_PI_MODE_ID,
|
|
21
|
+
EvalModelProfile,
|
|
22
|
+
default_eval_model_profile,
|
|
23
|
+
)
|
|
24
|
+
from millforge.eval_workflow import compact_eval_workflow_snapshot
|
|
25
|
+
|
|
26
|
+
EVAL_SUITE_SCHEMA_VERSION = 1
|
|
27
|
+
EVAL_SUITE_CAMPAIGN_MANIFEST_HASH_KIND = "eval_suite_campaign_manifest_sha256_v1"
|
|
28
|
+
EVAL_SUITE_MODEL_MANIFEST_HASH_KIND = "eval_suite_model_manifest_sha256_v1"
|
|
29
|
+
EVAL_SUITE_FIXTURE_HASH_KIND = "eval_suite_fixture_sha256_v1"
|
|
30
|
+
EVAL_SUITE_FIXTURE_PACK_HASH_KIND = "eval_suite_fixture_pack_sha256_v1"
|
|
31
|
+
EVAL_SUITE_SCORER_INPUT_HASH_KIND = "eval_suite_scorer_input_sha256_v1"
|
|
32
|
+
EVAL_SUITE_SCORER_RESULT_HASH_KIND = "eval_suite_scorer_result_sha256_v1"
|
|
33
|
+
EVAL_SUITE_DEFAULT_CAMPAIGN_ID = "eval.08a.default.offline.v1"
|
|
34
|
+
EVAL_SUITE_DEFAULT_CAMPAIGN_CREATED_AT = "1970-01-01T00:00:00Z"
|
|
35
|
+
EVAL_SUITE_DEFAULT_FIXTURE_PACK_ID = "pack.08a.offline.default.v1"
|
|
36
|
+
EVAL_SUITE_DEFAULT_SCORER_VERSION = "eval_suite.scorer.contract.v1"
|
|
37
|
+
EVAL_SUITE_OUTPUT_ROOT_HASH_KIND = "eval_suite_output_root_sha256_v1"
|
|
38
|
+
EVAL_SUITE_CLOSURE_EVIDENCE_HASH_KIND = "eval_suite_offline_closure_evidence_sha256_v1"
|
|
39
|
+
|
|
40
|
+
_EVAL_FIXTURE_PACK_PACKAGE = "millforge.eval_fixtures.default_pack"
|
|
41
|
+
_EVAL_FIXTURE_PACK_MANIFEST = "manifest.json"
|
|
42
|
+
|
|
43
|
+
_SHA256_RE = re.compile(r"^[0-9a-f]{64}$")
|
|
44
|
+
_UTC_TIMESTAMP_RE = re.compile(r"^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}Z$")
|
|
45
|
+
_WINDOWS_ABSOLUTE_PATH = re.compile(
|
|
46
|
+
r"(?:^|[\s\"'`([{<])(?:[A-Za-z]:[\\/]|\\\\)[^\s\"'`)\]}>,]*"
|
|
47
|
+
)
|
|
48
|
+
_POSIX_ABSOLUTE_PATH = re.compile(r"(?:^|[\s\"'`([{<])/(?!/)[^\s\"'`)\]}>,]*")
|
|
49
|
+
_USER_HOME_PATH = re.compile(r"(?:^|[\s\"'`([{<])~(?:/|\\)[^\s\"'`)\]}>,]*")
|
|
50
|
+
_ENDPOINT_URL = re.compile(r"https?://|localhost(?::|/|$)|127\.0\.0\.1|0\.0\.0\.0")
|
|
51
|
+
_CREDENTIAL_VALUE_PATTERNS = (
|
|
52
|
+
re.compile(r"\bsk-(?:live|proj|test)-[A-Za-z0-9_-]{10,}\b"),
|
|
53
|
+
re.compile(r"\b[rs]k_(?:live|test)_[A-Za-z0-9]{16,}\b"),
|
|
54
|
+
re.compile(r"\b(?:AKIA|ASIA)[0-9A-Z]{16}\b"),
|
|
55
|
+
re.compile(r"\bAIza[0-9A-Za-z_-]{20,}\b"),
|
|
56
|
+
re.compile(r"\bgh[opsu]_[A-Za-z0-9_]{20,}\b"),
|
|
57
|
+
)
|
|
58
|
+
_SECRET_FIELD_MARKERS = (
|
|
59
|
+
"api_key",
|
|
60
|
+
"apikey",
|
|
61
|
+
"auth_header",
|
|
62
|
+
"authorization",
|
|
63
|
+
"bearer",
|
|
64
|
+
"client_secret",
|
|
65
|
+
"credential",
|
|
66
|
+
"password",
|
|
67
|
+
"private_key",
|
|
68
|
+
"secret",
|
|
69
|
+
"access_token",
|
|
70
|
+
"auth_token",
|
|
71
|
+
"refresh_token",
|
|
72
|
+
)
|
|
73
|
+
_DENIED_TEXT_TOKENS = (
|
|
74
|
+
"api_key",
|
|
75
|
+
"authorization:",
|
|
76
|
+
"bearer ",
|
|
77
|
+
"credential",
|
|
78
|
+
"password",
|
|
79
|
+
"secret",
|
|
80
|
+
"access token",
|
|
81
|
+
"auth token",
|
|
82
|
+
"bearer token",
|
|
83
|
+
"millrace-agents",
|
|
84
|
+
".millrace",
|
|
85
|
+
"daemon state",
|
|
86
|
+
"endpoint_url",
|
|
87
|
+
"endpoint url",
|
|
88
|
+
"external service",
|
|
89
|
+
"hidden answer",
|
|
90
|
+
"local planning",
|
|
91
|
+
"private runtime",
|
|
92
|
+
"network access",
|
|
93
|
+
"package install",
|
|
94
|
+
"ref-forge",
|
|
95
|
+
".claude",
|
|
96
|
+
".codex",
|
|
97
|
+
)
|
|
98
|
+
_NETWORK_COMMANDS = frozenset(
|
|
99
|
+
{"curl", "ftp", "nc", "netcat", "scp", "sftp", "ssh", "telnet", "wget"}
|
|
100
|
+
)
|
|
101
|
+
_PACKAGE_COMMANDS = frozenset(
|
|
102
|
+
{"cargo", "gem", "npm", "pip", "pip3", "pnpm", "poetry", "uv", "yarn"}
|
|
103
|
+
)
|
|
104
|
+
_NONDETERMINISTIC_COMMAND_TOKENS = frozenset(
|
|
105
|
+
{"date", "random", "sleep", "time", "uuid"}
|
|
106
|
+
)
|
|
107
|
+
_NONDETERMINISTIC_PYTHON_PRIMITIVES = (
|
|
108
|
+
re.compile(r"\buuid\.uuid4\s*\("),
|
|
109
|
+
re.compile(r"\btime\.time\s*\("),
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
class _FrozenEvalSuiteDict(dict[Any, Any]):
|
|
114
|
+
"""Dict-shaped immutable mapping that remains serializable by Pydantic."""
|
|
115
|
+
|
|
116
|
+
def __readonly(self, *args: Any, **kwargs: Any) -> None:
|
|
117
|
+
raise TypeError("eval-suite mappings are immutable")
|
|
118
|
+
|
|
119
|
+
__setitem__ = __readonly
|
|
120
|
+
__delitem__ = __readonly
|
|
121
|
+
clear = __readonly
|
|
122
|
+
pop = __readonly
|
|
123
|
+
popitem = __readonly # type: ignore[assignment]
|
|
124
|
+
setdefault = __readonly
|
|
125
|
+
update = __readonly
|
|
126
|
+
__ior__ = __readonly # type: ignore[assignment]
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
class EvalSuiteContractModel(BaseModel):
|
|
130
|
+
"""Closed, frozen base for public eval-suite contracts."""
|
|
131
|
+
|
|
132
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
133
|
+
|
|
134
|
+
@model_validator(mode="before")
|
|
135
|
+
@classmethod
|
|
136
|
+
def _reject_forbidden_payload(cls, data: Any) -> Any:
|
|
137
|
+
_reject_forbidden_material(data)
|
|
138
|
+
return data
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
class EvalCampaignKind(str, Enum):
|
|
142
|
+
"""Closed campaign backend classes."""
|
|
143
|
+
|
|
144
|
+
HOSTED_API = "hosted_api"
|
|
145
|
+
LOCAL_OPENAI_COMPATIBLE = "local_openai_compatible"
|
|
146
|
+
LOCAL_NATIVE = "local_native"
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
class EvalSuiteExecutionMode(str, Enum):
|
|
150
|
+
"""Closed eval-suite execution modes."""
|
|
151
|
+
|
|
152
|
+
OFFLINE_FAKE = "offline_fake"
|
|
153
|
+
LIVE_RUNNER = "live_runner"
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
class EvalTaskCategory(str, Enum):
|
|
157
|
+
"""Closed 08A fixture task categories."""
|
|
158
|
+
|
|
159
|
+
DIRECT_EDIT = "direct_edit"
|
|
160
|
+
MULTI_FILE_CONSISTENCY = "multi_file_consistency"
|
|
161
|
+
BUG_DIAGNOSIS = "bug_diagnosis"
|
|
162
|
+
EVIDENCE_DISCIPLINE = "evidence_discipline"
|
|
163
|
+
RECOVERY = "recovery"
|
|
164
|
+
FALSE_CLOSURE_TRAP = "false_closure_trap"
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
class EvalDifficultyLevel(str, Enum):
|
|
168
|
+
"""Coarse public fixture difficulty labels."""
|
|
169
|
+
|
|
170
|
+
BASIC = "basic"
|
|
171
|
+
INTERMEDIATE = "intermediate"
|
|
172
|
+
ADVANCED = "advanced"
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
class EvalExpectedMutationKind(str, Enum):
|
|
176
|
+
"""Expected workspace mutation policy for a fixture."""
|
|
177
|
+
|
|
178
|
+
NO_SOURCE_CHANGE = "no_source_change"
|
|
179
|
+
SOURCE_CHANGE_REQUIRED = "source_change_required"
|
|
180
|
+
DOCUMENTATION_ONLY = "documentation_only"
|
|
181
|
+
TEST_ONLY = "test_only"
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
class EvalTrialOutcome(str, Enum):
|
|
185
|
+
"""Closed final scorer outcome classes."""
|
|
186
|
+
|
|
187
|
+
VALID_COMPLETION = "valid_completion"
|
|
188
|
+
CORRECTLY_BLOCKED = "correctly_blocked"
|
|
189
|
+
FALSE_CLOSURE = "false_closure"
|
|
190
|
+
FALSE_BLOCKED = "false_blocked"
|
|
191
|
+
RUNTIME_FAILURE = "runtime_failure"
|
|
192
|
+
PROVIDER_FAILURE = "provider_failure"
|
|
193
|
+
INVALID_TRIAL = "invalid_trial"
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
class EvalOfflineDryCampaignDiagnosticCode(str, Enum):
|
|
197
|
+
"""Fail-closed diagnostics for offline dry-campaign preflight."""
|
|
198
|
+
|
|
199
|
+
MISSING_BUDGET_POLICY = "missing_budget_policy"
|
|
200
|
+
INVALID_BUDGET_POLICY = "invalid_budget_policy"
|
|
201
|
+
LIVE_EXECUTION_UNAVAILABLE = "live_execution_unavailable"
|
|
202
|
+
UNSAFE_OUTPUT_ROOT = "unsafe_output_root"
|
|
203
|
+
FIXTURE_PACK_UNAVAILABLE = "fixture_pack_unavailable"
|
|
204
|
+
INVALID_TRIAL_COUNT = "invalid_trial_count"
|
|
205
|
+
MANIFEST_CONFLICT = "manifest_conflict"
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
class EvalFailureTaxonomyLabel(str, Enum):
|
|
209
|
+
"""Closed public failure taxonomy labels for scorer results."""
|
|
210
|
+
|
|
211
|
+
VISIBLE_CHECK_FAILED = "visible_check_failed"
|
|
212
|
+
HIDDEN_CHECK_FAILED = "hidden_check_failed"
|
|
213
|
+
REQUIRED_ARTIFACT_MISSING = "required_artifact_missing"
|
|
214
|
+
REQUIRED_ARTIFACT_MALFORMED = "required_artifact_malformed"
|
|
215
|
+
EXPECTED_MUTATION_ABSENT = "expected_mutation_absent"
|
|
216
|
+
UNAUTHORIZED_MUTATION = "unauthorized_mutation"
|
|
217
|
+
CAPABILITY_VIOLATION = "capability_violation"
|
|
218
|
+
FALSE_SUCCESS_TERMINAL = "false_success_terminal"
|
|
219
|
+
INFRASTRUCTURE_DEFECT = "infrastructure_defect"
|
|
220
|
+
PROVIDER_DEFECT = "provider_defect"
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
class EvalHashRecord(EvalSuiteContractModel):
|
|
224
|
+
"""Typed hash reference for eval-suite records."""
|
|
225
|
+
|
|
226
|
+
hash_kind: StrictStr
|
|
227
|
+
sha256: StrictStr
|
|
228
|
+
|
|
229
|
+
@field_validator("sha256")
|
|
230
|
+
@classmethod
|
|
231
|
+
def _sha256_valid(cls, value: str) -> str:
|
|
232
|
+
_validate_sha256(value)
|
|
233
|
+
return value
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
class EvalBudgetPolicyReference(EvalSuiteContractModel):
|
|
237
|
+
"""Public reference to a campaign budget policy."""
|
|
238
|
+
|
|
239
|
+
policy_id: StrictStr
|
|
240
|
+
summary: StrictStr
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
class EvalLiveDenialDiagnostic(EvalSuiteContractModel):
|
|
244
|
+
"""Fail-closed live execution denial diagnostic."""
|
|
245
|
+
|
|
246
|
+
diagnostic_code: StrictStr
|
|
247
|
+
summary: StrictStr
|
|
248
|
+
rule_id: StrictStr
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
class EvalOfflineDryCampaignDiagnostic(EvalSuiteContractModel):
|
|
252
|
+
"""Structured public diagnostic for dry-campaign validation failures."""
|
|
253
|
+
|
|
254
|
+
diagnostic_code: EvalOfflineDryCampaignDiagnosticCode
|
|
255
|
+
rule_id: StrictStr
|
|
256
|
+
summary: StrictStr
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
class EvalOfflineDryCampaignConfig(EvalSuiteContractModel):
|
|
260
|
+
"""Public offline dry-campaign configuration after preflight validation."""
|
|
261
|
+
|
|
262
|
+
schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
|
|
263
|
+
fixture_pack_id: StrictStr = EVAL_SUITE_DEFAULT_FIXTURE_PACK_ID
|
|
264
|
+
fixture_ids: tuple[StrictStr, ...]
|
|
265
|
+
deterministic_seed: StrictInt = Field(ge=0)
|
|
266
|
+
trial_count_per_fixture_per_arm: StrictInt = Field(gt=0)
|
|
267
|
+
output_root_hash_kind: StrictStr = EVAL_SUITE_OUTPUT_ROOT_HASH_KIND
|
|
268
|
+
output_root_hash: StrictStr
|
|
269
|
+
output_root_policy: StrictStr = "caller_provided_preflight_validated"
|
|
270
|
+
|
|
271
|
+
@model_validator(mode="after")
|
|
272
|
+
def _config_valid(self) -> EvalOfflineDryCampaignConfig:
|
|
273
|
+
if self.schema_version != EVAL_SUITE_SCHEMA_VERSION:
|
|
274
|
+
raise ValueError("unsupported eval-suite dry-campaign schema_version")
|
|
275
|
+
if self.fixture_pack_id != EVAL_SUITE_DEFAULT_FIXTURE_PACK_ID:
|
|
276
|
+
raise ValueError("only the default offline fixture pack is installed")
|
|
277
|
+
if not self.fixture_ids:
|
|
278
|
+
raise ValueError("dry campaigns require at least one fixture")
|
|
279
|
+
if len(set(self.fixture_ids)) != len(self.fixture_ids):
|
|
280
|
+
raise ValueError("dry-campaign fixture IDs must be unique")
|
|
281
|
+
if self.output_root_hash_kind != EVAL_SUITE_OUTPUT_ROOT_HASH_KIND:
|
|
282
|
+
raise ValueError("unsupported output root hash kind")
|
|
283
|
+
_validate_sha256(self.output_root_hash)
|
|
284
|
+
return self
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
class EvalOfflineDryCampaignPlan(EvalSuiteContractModel):
|
|
288
|
+
"""Public preflight result for a deterministic offline fake campaign."""
|
|
289
|
+
|
|
290
|
+
schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
|
|
291
|
+
config: EvalOfflineDryCampaignConfig
|
|
292
|
+
campaign_manifest: EvalCampaignManifest
|
|
293
|
+
fixture_pack_summary: EvalFixturePackSummary
|
|
294
|
+
admitted_arm_ids: tuple[StrictStr, StrictStr]
|
|
295
|
+
paired_plan_count: StrictInt = Field(ge=0)
|
|
296
|
+
planned_arm_trial_count: StrictInt = Field(ge=0)
|
|
297
|
+
trial_plan_hashes: tuple[StrictStr, ...]
|
|
298
|
+
campaign_store_root: StrictStr
|
|
299
|
+
manifest_relative_path: StrictStr
|
|
300
|
+
budget_policy_ref: EvalBudgetPolicyReference
|
|
301
|
+
live_execution_admitted: StrictBool = False
|
|
302
|
+
diagnostics: tuple[EvalOfflineDryCampaignDiagnostic, ...] = Field(
|
|
303
|
+
default_factory=tuple
|
|
304
|
+
)
|
|
305
|
+
|
|
306
|
+
@model_validator(mode="after")
|
|
307
|
+
def _dry_plan_valid(self) -> EvalOfflineDryCampaignPlan:
|
|
308
|
+
if self.schema_version != EVAL_SUITE_SCHEMA_VERSION:
|
|
309
|
+
raise ValueError("unsupported eval-suite dry-campaign schema_version")
|
|
310
|
+
if (
|
|
311
|
+
self.campaign_manifest.execution_mode
|
|
312
|
+
is not EvalSuiteExecutionMode.OFFLINE_FAKE
|
|
313
|
+
):
|
|
314
|
+
raise ValueError("dry campaigns must use offline fake execution")
|
|
315
|
+
if (
|
|
316
|
+
self.live_execution_admitted
|
|
317
|
+
or self.campaign_manifest.live_execution_admitted
|
|
318
|
+
):
|
|
319
|
+
raise ValueError("dry campaigns do not admit live execution")
|
|
320
|
+
if self.fixture_pack_summary.fixture_pack_id != self.config.fixture_pack_id:
|
|
321
|
+
raise ValueError("fixture pack summary must match dry-campaign config")
|
|
322
|
+
if self.fixture_pack_summary.fixture_pack_hash != (
|
|
323
|
+
self.campaign_manifest.fixture_pack_hash
|
|
324
|
+
):
|
|
325
|
+
raise ValueError("campaign manifest must reference fixture pack summary")
|
|
326
|
+
if self.paired_plan_count != len(self.trial_plan_hashes):
|
|
327
|
+
raise ValueError("paired_plan_count must match trial_plan_hashes")
|
|
328
|
+
if self.planned_arm_trial_count != self.paired_plan_count * len(
|
|
329
|
+
self.admitted_arm_ids
|
|
330
|
+
):
|
|
331
|
+
raise ValueError("planned_arm_trial_count must match admitted arms")
|
|
332
|
+
for digest in self.trial_plan_hashes:
|
|
333
|
+
_validate_sha256(digest)
|
|
334
|
+
_validate_relative_artifact_root(self.campaign_store_root)
|
|
335
|
+
_validate_relative_artifact_root(self.manifest_relative_path)
|
|
336
|
+
if self.diagnostics:
|
|
337
|
+
raise ValueError("valid dry-campaign plans must not carry diagnostics")
|
|
338
|
+
return self
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
class EvalOfflineDryCampaignRunResult(EvalSuiteContractModel):
|
|
342
|
+
"""Filesystem result from one deterministic offline fake campaign run."""
|
|
343
|
+
|
|
344
|
+
schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
|
|
345
|
+
dry_campaign_plan: EvalOfflineDryCampaignPlan
|
|
346
|
+
report_id: StrictStr
|
|
347
|
+
campaign_store_root: StrictStr
|
|
348
|
+
manifest_relative_path: StrictStr
|
|
349
|
+
plan_relative_path: StrictStr
|
|
350
|
+
trials_relative_path: StrictStr
|
|
351
|
+
index_relative_path: StrictStr
|
|
352
|
+
artifact_root_relative_path: StrictStr
|
|
353
|
+
report_relative_paths: Mapping[StrictStr, StrictStr]
|
|
354
|
+
completed_trial_ids: tuple[StrictStr, ...]
|
|
355
|
+
pending_trial_ids: tuple[StrictStr, ...]
|
|
356
|
+
appended_trial_ids: tuple[StrictStr, ...]
|
|
357
|
+
trial_plan_hashes: tuple[StrictStr, ...]
|
|
358
|
+
trial_record_hashes: tuple[StrictStr, ...]
|
|
359
|
+
resume_index_hash: StrictStr
|
|
360
|
+
report_hash: StrictStr
|
|
361
|
+
report_json_hash: StrictStr
|
|
362
|
+
report_markdown_hash: StrictStr
|
|
363
|
+
closure_evidence_relative_path: StrictStr
|
|
364
|
+
closure_evidence_hash: StrictStr
|
|
365
|
+
live_execution_admitted: StrictBool = False
|
|
366
|
+
|
|
367
|
+
@model_validator(mode="after")
|
|
368
|
+
def _run_result_valid(self) -> EvalOfflineDryCampaignRunResult:
|
|
369
|
+
if self.schema_version != EVAL_SUITE_SCHEMA_VERSION:
|
|
370
|
+
raise ValueError("unsupported eval-suite dry-campaign schema_version")
|
|
371
|
+
if self.live_execution_admitted:
|
|
372
|
+
raise ValueError("dry campaigns do not admit live execution")
|
|
373
|
+
if self.campaign_store_root != self.dry_campaign_plan.campaign_store_root:
|
|
374
|
+
raise ValueError("campaign store root must match dry-campaign plan")
|
|
375
|
+
for path_value in (
|
|
376
|
+
self.campaign_store_root,
|
|
377
|
+
self.manifest_relative_path,
|
|
378
|
+
self.plan_relative_path,
|
|
379
|
+
self.trials_relative_path,
|
|
380
|
+
self.index_relative_path,
|
|
381
|
+
self.artifact_root_relative_path,
|
|
382
|
+
self.closure_evidence_relative_path,
|
|
383
|
+
*self.report_relative_paths.values(),
|
|
384
|
+
):
|
|
385
|
+
_validate_relative_artifact_root(path_value)
|
|
386
|
+
expected_prefix = f"{self.campaign_store_root}/"
|
|
387
|
+
for path_value in (
|
|
388
|
+
self.manifest_relative_path,
|
|
389
|
+
self.plan_relative_path,
|
|
390
|
+
self.trials_relative_path,
|
|
391
|
+
self.index_relative_path,
|
|
392
|
+
self.artifact_root_relative_path,
|
|
393
|
+
self.closure_evidence_relative_path,
|
|
394
|
+
*self.report_relative_paths.values(),
|
|
395
|
+
):
|
|
396
|
+
if not path_value.startswith(expected_prefix):
|
|
397
|
+
raise ValueError("dry-campaign output paths must be campaign-relative")
|
|
398
|
+
for digest in (
|
|
399
|
+
*self.trial_plan_hashes,
|
|
400
|
+
*self.trial_record_hashes,
|
|
401
|
+
self.resume_index_hash,
|
|
402
|
+
self.report_hash,
|
|
403
|
+
self.report_json_hash,
|
|
404
|
+
self.report_markdown_hash,
|
|
405
|
+
self.closure_evidence_hash,
|
|
406
|
+
):
|
|
407
|
+
_validate_sha256(digest)
|
|
408
|
+
if set(self.report_relative_paths) != {
|
|
409
|
+
"report.json",
|
|
410
|
+
"report.md",
|
|
411
|
+
"report.sha256",
|
|
412
|
+
}:
|
|
413
|
+
raise ValueError("dry-campaign reports must include the public report set")
|
|
414
|
+
object.__setattr__(
|
|
415
|
+
self,
|
|
416
|
+
"report_relative_paths",
|
|
417
|
+
_freeze_eval_suite_mapping(self.report_relative_paths),
|
|
418
|
+
)
|
|
419
|
+
return self
|
|
420
|
+
|
|
421
|
+
|
|
422
|
+
class EvalOfflineDryCampaignClosureEvidence(EvalSuiteContractModel):
|
|
423
|
+
"""Compact public closure evidence for one offline dry-campaign run."""
|
|
424
|
+
|
|
425
|
+
schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
|
|
426
|
+
campaign_id: StrictStr
|
|
427
|
+
report_id: StrictStr
|
|
428
|
+
fixture_pack_hash: StrictStr
|
|
429
|
+
campaign_manifest_hash: StrictStr
|
|
430
|
+
model_manifest_hash: StrictStr
|
|
431
|
+
workflow_graph_hash: StrictStr
|
|
432
|
+
trial_plan_hashes: tuple[StrictStr, ...]
|
|
433
|
+
trial_record_hashes: tuple[StrictStr, ...]
|
|
434
|
+
resume_index_hash: StrictStr
|
|
435
|
+
report_hash: StrictStr
|
|
436
|
+
report_json_hash: StrictStr
|
|
437
|
+
report_markdown_hash: StrictStr
|
|
438
|
+
counts: Mapping[StrictStr, StrictInt]
|
|
439
|
+
unresolved_live_dependencies: tuple[StrictStr, ...]
|
|
440
|
+
live_denial_diagnostic_codes: tuple[StrictStr, ...]
|
|
441
|
+
live_denial_test_coverage: tuple[StrictStr, ...]
|
|
442
|
+
treatment_compiled_harness_hashes: Mapping[StrictStr, StrictStr]
|
|
443
|
+
public_hygiene_checks: Mapping[StrictStr, StrictBool]
|
|
444
|
+
claim_boundary: StrictStr
|
|
445
|
+
closure_evidence_hash_kind: StrictStr = EVAL_SUITE_CLOSURE_EVIDENCE_HASH_KIND
|
|
446
|
+
closure_evidence_hash: StrictStr
|
|
447
|
+
|
|
448
|
+
@model_validator(mode="after")
|
|
449
|
+
def _closure_evidence_valid(self) -> EvalOfflineDryCampaignClosureEvidence:
|
|
450
|
+
if self.schema_version != EVAL_SUITE_SCHEMA_VERSION:
|
|
451
|
+
raise ValueError("unsupported eval-suite closure-evidence schema_version")
|
|
452
|
+
for digest in (
|
|
453
|
+
self.fixture_pack_hash,
|
|
454
|
+
self.campaign_manifest_hash,
|
|
455
|
+
self.model_manifest_hash,
|
|
456
|
+
self.workflow_graph_hash,
|
|
457
|
+
*self.trial_plan_hashes,
|
|
458
|
+
*self.trial_record_hashes,
|
|
459
|
+
self.resume_index_hash,
|
|
460
|
+
self.report_hash,
|
|
461
|
+
self.report_json_hash,
|
|
462
|
+
self.report_markdown_hash,
|
|
463
|
+
*self.treatment_compiled_harness_hashes.values(),
|
|
464
|
+
self.closure_evidence_hash,
|
|
465
|
+
):
|
|
466
|
+
_validate_sha256(digest)
|
|
467
|
+
if self.closure_evidence_hash_kind != EVAL_SUITE_CLOSURE_EVIDENCE_HASH_KIND:
|
|
468
|
+
raise ValueError("unsupported closure evidence hash kind")
|
|
469
|
+
required_counts = {
|
|
470
|
+
"fixture_count",
|
|
471
|
+
"paired_plan_count",
|
|
472
|
+
"arm_trial_count",
|
|
473
|
+
"completed_trial_count",
|
|
474
|
+
"pending_trial_count",
|
|
475
|
+
"stored_trial_count",
|
|
476
|
+
"report_artifact_count",
|
|
477
|
+
}
|
|
478
|
+
if set(self.counts) != required_counts:
|
|
479
|
+
raise ValueError("closure evidence counts must use the compact count set")
|
|
480
|
+
if any(count < 0 for count in self.counts.values()):
|
|
481
|
+
raise ValueError("closure evidence counts must be non-negative")
|
|
482
|
+
required_hygiene = {
|
|
483
|
+
"absolute_paths_absent",
|
|
484
|
+
"home_paths_absent",
|
|
485
|
+
"urls_absent",
|
|
486
|
+
"auth_material_absent",
|
|
487
|
+
"runtime_state_absent",
|
|
488
|
+
"hidden_material_absent",
|
|
489
|
+
"raw_logs_absent",
|
|
490
|
+
}
|
|
491
|
+
if set(self.public_hygiene_checks) != required_hygiene:
|
|
492
|
+
raise ValueError("closure evidence must declare all public hygiene checks")
|
|
493
|
+
if not all(self.public_hygiene_checks.values()):
|
|
494
|
+
raise ValueError("closure evidence public hygiene checks must pass")
|
|
495
|
+
if not self.unresolved_live_dependencies:
|
|
496
|
+
raise ValueError("closure evidence must retain live dependency boundaries")
|
|
497
|
+
if not self.live_denial_diagnostic_codes:
|
|
498
|
+
raise ValueError("closure evidence must cite live denial diagnostics")
|
|
499
|
+
if not self.live_denial_test_coverage:
|
|
500
|
+
raise ValueError("closure evidence must cite denial regression coverage")
|
|
501
|
+
if not self.treatment_compiled_harness_hashes:
|
|
502
|
+
raise ValueError("closure evidence must include Spec 07E harness hashes")
|
|
503
|
+
object.__setattr__(self, "counts", _freeze_eval_suite_mapping(self.counts))
|
|
504
|
+
object.__setattr__(
|
|
505
|
+
self,
|
|
506
|
+
"treatment_compiled_harness_hashes",
|
|
507
|
+
_freeze_eval_suite_mapping(self.treatment_compiled_harness_hashes),
|
|
508
|
+
)
|
|
509
|
+
object.__setattr__(
|
|
510
|
+
self,
|
|
511
|
+
"public_hygiene_checks",
|
|
512
|
+
_freeze_eval_suite_mapping(self.public_hygiene_checks),
|
|
513
|
+
)
|
|
514
|
+
return self
|
|
515
|
+
|
|
516
|
+
|
|
517
|
+
class EvalModelPricingMetadata(EvalSuiteContractModel):
|
|
518
|
+
"""Public numeric model pricing metadata with labels kept separate."""
|
|
519
|
+
|
|
520
|
+
input_cost_per_million_tokens: StrictFloat = Field(ge=0.0)
|
|
521
|
+
output_cost_per_million_tokens: StrictFloat = Field(ge=0.0)
|
|
522
|
+
cached_input_cost_per_million_tokens: StrictFloat = Field(ge=0.0)
|
|
523
|
+
currency_label: StrictStr
|
|
524
|
+
source_label: StrictStr
|
|
525
|
+
|
|
526
|
+
|
|
527
|
+
class EvalModelRateLimitMetadata(EvalSuiteContractModel):
|
|
528
|
+
"""Public numeric model rate-limit metadata with labels kept separate."""
|
|
529
|
+
|
|
530
|
+
request_rate_per_window: StrictInt = Field(ge=0)
|
|
531
|
+
token_rate_per_window: StrictInt = Field(ge=0)
|
|
532
|
+
concurrent_request_limit: StrictInt = Field(ge=0)
|
|
533
|
+
window_seconds: StrictInt = Field(ge=0)
|
|
534
|
+
source_label: StrictStr
|
|
535
|
+
|
|
536
|
+
|
|
537
|
+
class EvalModelManifest(EvalSuiteContractModel):
|
|
538
|
+
"""Backend-neutral campaign model manifest without endpoints or secrets."""
|
|
539
|
+
|
|
540
|
+
schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
|
|
541
|
+
model_manifest_id: StrictStr
|
|
542
|
+
model_profile_id: StrictStr
|
|
543
|
+
model_profile_hash: StrictStr
|
|
544
|
+
provider_label: StrictStr
|
|
545
|
+
model_or_artifact_id: StrictStr
|
|
546
|
+
release_or_snapshot: StrictStr
|
|
547
|
+
serving_protocol: StrictStr
|
|
548
|
+
endpoint_class: EvalCampaignKind
|
|
549
|
+
temperature: StrictFloat = Field(ge=0.0, le=2.0)
|
|
550
|
+
top_p: StrictFloat = Field(gt=0.0, le=1.0)
|
|
551
|
+
max_prompt_tokens: StrictInt = Field(gt=0)
|
|
552
|
+
max_completion_tokens: StrictInt = Field(gt=0)
|
|
553
|
+
max_total_tokens: StrictInt = Field(gt=0)
|
|
554
|
+
tool_calling_mode: StrictStr
|
|
555
|
+
parser_id: StrictStr
|
|
556
|
+
seed_policy: StrictStr
|
|
557
|
+
context_window_tokens: StrictInt = Field(gt=0)
|
|
558
|
+
reasoning_controls: Mapping[StrictStr, StrictStr] = Field(default_factory=dict)
|
|
559
|
+
public_pricing: EvalModelPricingMetadata
|
|
560
|
+
public_rate_limits: EvalModelRateLimitMetadata
|
|
561
|
+
public_pricing_snapshot: Mapping[StrictStr, StrictStr] = Field(default_factory=dict)
|
|
562
|
+
public_rate_limit_snapshot: Mapping[StrictStr, StrictStr] = Field(
|
|
563
|
+
default_factory=dict
|
|
564
|
+
)
|
|
565
|
+
local_serving_snapshot: Mapping[StrictStr, StrictStr] = Field(default_factory=dict)
|
|
566
|
+
model_manifest_hash_kind: StrictStr = EVAL_SUITE_MODEL_MANIFEST_HASH_KIND
|
|
567
|
+
model_manifest_hash: StrictStr
|
|
568
|
+
|
|
569
|
+
@model_validator(mode="after")
|
|
570
|
+
def _model_manifest_valid(self) -> EvalModelManifest:
|
|
571
|
+
if self.schema_version != EVAL_SUITE_SCHEMA_VERSION:
|
|
572
|
+
raise ValueError("unsupported eval-suite model schema_version")
|
|
573
|
+
if self.max_total_tokens != self.max_prompt_tokens + self.max_completion_tokens:
|
|
574
|
+
raise ValueError(
|
|
575
|
+
"max_total_tokens must equal prompt plus completion tokens"
|
|
576
|
+
)
|
|
577
|
+
if self.context_window_tokens < self.max_total_tokens:
|
|
578
|
+
raise ValueError("context_window_tokens must cover max_total_tokens")
|
|
579
|
+
object.__setattr__(
|
|
580
|
+
self,
|
|
581
|
+
"reasoning_controls",
|
|
582
|
+
_freeze_eval_suite_mapping(self.reasoning_controls),
|
|
583
|
+
)
|
|
584
|
+
object.__setattr__(
|
|
585
|
+
self,
|
|
586
|
+
"public_pricing_snapshot",
|
|
587
|
+
_freeze_eval_suite_mapping(self.public_pricing_snapshot),
|
|
588
|
+
)
|
|
589
|
+
object.__setattr__(
|
|
590
|
+
self,
|
|
591
|
+
"public_rate_limit_snapshot",
|
|
592
|
+
_freeze_eval_suite_mapping(self.public_rate_limit_snapshot),
|
|
593
|
+
)
|
|
594
|
+
object.__setattr__(
|
|
595
|
+
self,
|
|
596
|
+
"local_serving_snapshot",
|
|
597
|
+
_freeze_eval_suite_mapping(self.local_serving_snapshot),
|
|
598
|
+
)
|
|
599
|
+
if self.model_manifest_hash_kind != EVAL_SUITE_MODEL_MANIFEST_HASH_KIND:
|
|
600
|
+
raise ValueError("unsupported eval-suite model hash kind")
|
|
601
|
+
_validate_sha256(self.model_profile_hash)
|
|
602
|
+
_validate_sha256(self.model_manifest_hash)
|
|
603
|
+
expected = calculate_eval_model_manifest_hash(self)
|
|
604
|
+
if self.model_manifest_hash != expected:
|
|
605
|
+
raise ValueError("model_manifest_hash does not match manifest payload")
|
|
606
|
+
return self
|
|
607
|
+
|
|
608
|
+
|
|
609
|
+
class EvalDifficultyMetadata(EvalSuiteContractModel):
|
|
610
|
+
"""Public fixture difficulty metadata."""
|
|
611
|
+
|
|
612
|
+
level: EvalDifficultyLevel
|
|
613
|
+
rationale: StrictStr
|
|
614
|
+
estimated_minutes: StrictInt = Field(gt=0)
|
|
615
|
+
|
|
616
|
+
|
|
617
|
+
class EvalVisibleCheck(EvalSuiteContractModel):
|
|
618
|
+
"""Runner-visible check descriptor."""
|
|
619
|
+
|
|
620
|
+
check_id: StrictStr
|
|
621
|
+
summary: StrictStr
|
|
622
|
+
command: StrictStr | None = None
|
|
623
|
+
|
|
624
|
+
@model_validator(mode="after")
|
|
625
|
+
def _visible_check_valid(self) -> EvalVisibleCheck:
|
|
626
|
+
if self.command is not None:
|
|
627
|
+
_validate_public_command(self.command)
|
|
628
|
+
return self
|
|
629
|
+
|
|
630
|
+
|
|
631
|
+
class EvalHiddenCheck(EvalSuiteContractModel):
|
|
632
|
+
"""Scorer-only hidden check descriptor."""
|
|
633
|
+
|
|
634
|
+
check_id: StrictStr
|
|
635
|
+
summary: StrictStr
|
|
636
|
+
scorer_rubric: StrictStr
|
|
637
|
+
|
|
638
|
+
|
|
639
|
+
class EvalExpectedMutationPolicy(EvalSuiteContractModel):
|
|
640
|
+
"""Expected workspace mutation policy for scorer use."""
|
|
641
|
+
|
|
642
|
+
mutation_kind: EvalExpectedMutationKind
|
|
643
|
+
allowed_paths: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
644
|
+
forbidden_paths: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
645
|
+
summary: StrictStr
|
|
646
|
+
|
|
647
|
+
@model_validator(mode="after")
|
|
648
|
+
def _mutation_policy_valid(self) -> EvalExpectedMutationPolicy:
|
|
649
|
+
for path in self.allowed_paths + self.forbidden_paths:
|
|
650
|
+
_validate_relative_path(path)
|
|
651
|
+
if (
|
|
652
|
+
self.mutation_kind == EvalExpectedMutationKind.NO_SOURCE_CHANGE
|
|
653
|
+
and self.allowed_paths
|
|
654
|
+
):
|
|
655
|
+
raise ValueError("no-source-change policies cannot allow mutation paths")
|
|
656
|
+
return self
|
|
657
|
+
|
|
658
|
+
|
|
659
|
+
class EvalTaskFixture(EvalSuiteContractModel):
|
|
660
|
+
"""Immutable scorer-owned fixture contract."""
|
|
661
|
+
|
|
662
|
+
schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
|
|
663
|
+
fixture_id: StrictStr
|
|
664
|
+
category: EvalTaskCategory
|
|
665
|
+
difficulty: EvalDifficultyMetadata
|
|
666
|
+
visible_prompt: StrictStr
|
|
667
|
+
visible_acceptance_criteria: tuple[StrictStr, ...]
|
|
668
|
+
file_allowlist: tuple[StrictStr, ...]
|
|
669
|
+
expected_mutation_policy: EvalExpectedMutationPolicy
|
|
670
|
+
file_manifest_hashes: tuple[EvalHashRecord, ...] = Field(default_factory=tuple)
|
|
671
|
+
visible_checks: tuple[EvalVisibleCheck, ...]
|
|
672
|
+
hidden_checks: tuple[EvalHiddenCheck, ...]
|
|
673
|
+
expected_final_outcome: EvalTrialOutcome
|
|
674
|
+
fixture_hash_kind: StrictStr = EVAL_SUITE_FIXTURE_HASH_KIND
|
|
675
|
+
fixture_hash: StrictStr
|
|
676
|
+
|
|
677
|
+
@model_validator(mode="after")
|
|
678
|
+
def _fixture_valid(self) -> EvalTaskFixture:
|
|
679
|
+
if self.schema_version != EVAL_SUITE_SCHEMA_VERSION:
|
|
680
|
+
raise ValueError("unsupported eval-suite fixture schema_version")
|
|
681
|
+
if not self.visible_acceptance_criteria:
|
|
682
|
+
raise ValueError("fixtures must include visible acceptance criteria")
|
|
683
|
+
if not self.visible_checks:
|
|
684
|
+
raise ValueError("fixtures must include visible checks")
|
|
685
|
+
if not self.hidden_checks:
|
|
686
|
+
raise ValueError("fixtures must include scorer-only hidden checks")
|
|
687
|
+
for path in self.file_allowlist:
|
|
688
|
+
_validate_relative_path(path)
|
|
689
|
+
if self.fixture_hash_kind != EVAL_SUITE_FIXTURE_HASH_KIND:
|
|
690
|
+
raise ValueError("unsupported fixture hash kind")
|
|
691
|
+
_validate_sha256(self.fixture_hash)
|
|
692
|
+
expected = calculate_eval_task_fixture_hash(self)
|
|
693
|
+
if self.fixture_hash != expected:
|
|
694
|
+
raise ValueError("fixture_hash does not match fixture payload")
|
|
695
|
+
return self
|
|
696
|
+
|
|
697
|
+
|
|
698
|
+
class EvalRunnerTaskProjection(EvalSuiteContractModel):
|
|
699
|
+
"""Runner-facing task projection without scorer-only fixture material."""
|
|
700
|
+
|
|
701
|
+
schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
|
|
702
|
+
fixture_id: StrictStr
|
|
703
|
+
category: EvalTaskCategory
|
|
704
|
+
difficulty: EvalDifficultyMetadata
|
|
705
|
+
visible_prompt: StrictStr
|
|
706
|
+
visible_acceptance_criteria: tuple[StrictStr, ...]
|
|
707
|
+
file_allowlist: tuple[StrictStr, ...]
|
|
708
|
+
visible_checks: tuple[EvalVisibleCheck, ...]
|
|
709
|
+
|
|
710
|
+
|
|
711
|
+
class EvalRunnerAcceptanceProjection(EvalSuiteContractModel):
|
|
712
|
+
"""Runner-facing acceptance projection with visible criteria and checks only."""
|
|
713
|
+
|
|
714
|
+
schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
|
|
715
|
+
fixture_id: StrictStr
|
|
716
|
+
visible_acceptance_criteria: tuple[StrictStr, ...]
|
|
717
|
+
visible_checks: tuple[EvalVisibleCheck, ...]
|
|
718
|
+
|
|
719
|
+
|
|
720
|
+
class EvalRunnerContextProjection(EvalSuiteContractModel):
|
|
721
|
+
"""Runner-facing context projection with only visible fixture context."""
|
|
722
|
+
|
|
723
|
+
schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
|
|
724
|
+
fixture_id: StrictStr
|
|
725
|
+
category: EvalTaskCategory
|
|
726
|
+
difficulty: EvalDifficultyMetadata
|
|
727
|
+
visible_prompt: StrictStr
|
|
728
|
+
visible_acceptance_criteria: tuple[StrictStr, ...]
|
|
729
|
+
file_allowlist: tuple[StrictStr, ...]
|
|
730
|
+
visible_checks: tuple[EvalVisibleCheck, ...]
|
|
731
|
+
|
|
732
|
+
|
|
733
|
+
class EvalPublicArtifactProjection(EvalSuiteContractModel):
|
|
734
|
+
"""Public artifact projection that excludes scorer-only fixture answers."""
|
|
735
|
+
|
|
736
|
+
schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
|
|
737
|
+
fixture_id: StrictStr
|
|
738
|
+
category: EvalTaskCategory
|
|
739
|
+
visible_acceptance_criteria: tuple[StrictStr, ...]
|
|
740
|
+
file_allowlist: tuple[StrictStr, ...]
|
|
741
|
+
visible_checks: tuple[EvalVisibleCheck, ...]
|
|
742
|
+
|
|
743
|
+
|
|
744
|
+
class EvalFixturePackSummary(EvalSuiteContractModel):
|
|
745
|
+
"""Hashable public summary of a fixture pack."""
|
|
746
|
+
|
|
747
|
+
schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
|
|
748
|
+
fixture_pack_id: StrictStr
|
|
749
|
+
fixture_ids: tuple[StrictStr, ...]
|
|
750
|
+
category_counts: Mapping[EvalTaskCategory, StrictInt]
|
|
751
|
+
fixture_hashes: tuple[EvalHashRecord, ...]
|
|
752
|
+
pack_summary: StrictStr
|
|
753
|
+
fixture_pack_hash_kind: StrictStr = EVAL_SUITE_FIXTURE_PACK_HASH_KIND
|
|
754
|
+
fixture_pack_hash: StrictStr
|
|
755
|
+
|
|
756
|
+
@model_validator(mode="after")
|
|
757
|
+
def _fixture_pack_valid(self) -> EvalFixturePackSummary:
|
|
758
|
+
if self.schema_version != EVAL_SUITE_SCHEMA_VERSION:
|
|
759
|
+
raise ValueError("unsupported eval-suite fixture pack schema_version")
|
|
760
|
+
if len(set(self.fixture_ids)) != len(self.fixture_ids):
|
|
761
|
+
raise ValueError("fixture pack fixture_ids must be unique")
|
|
762
|
+
if len(self.fixture_hashes) != len(self.fixture_ids):
|
|
763
|
+
raise ValueError("fixture pack hashes must match fixture_ids")
|
|
764
|
+
object.__setattr__(
|
|
765
|
+
self,
|
|
766
|
+
"category_counts",
|
|
767
|
+
_freeze_eval_suite_mapping(self.category_counts),
|
|
768
|
+
)
|
|
769
|
+
if self.fixture_pack_hash_kind != EVAL_SUITE_FIXTURE_PACK_HASH_KIND:
|
|
770
|
+
raise ValueError("unsupported fixture pack hash kind")
|
|
771
|
+
_validate_sha256(self.fixture_pack_hash)
|
|
772
|
+
expected = calculate_eval_fixture_pack_hash(self)
|
|
773
|
+
if self.fixture_pack_hash != expected:
|
|
774
|
+
raise ValueError("fixture_pack_hash does not match summary payload")
|
|
775
|
+
return self
|
|
776
|
+
|
|
777
|
+
|
|
778
|
+
class EvalCapabilityAuditSummary(EvalSuiteContractModel):
|
|
779
|
+
"""Scorer input summary of capability-envelope enforcement."""
|
|
780
|
+
|
|
781
|
+
capability_violation: StrictBool
|
|
782
|
+
denied_capability_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
783
|
+
summary: StrictStr
|
|
784
|
+
|
|
785
|
+
|
|
786
|
+
class EvalCheckResult(EvalSuiteContractModel):
|
|
787
|
+
"""Visible or hidden check result consumed by the scorer."""
|
|
788
|
+
|
|
789
|
+
check_id: StrictStr
|
|
790
|
+
passed: StrictBool
|
|
791
|
+
diagnostic: StrictStr | None = None
|
|
792
|
+
|
|
793
|
+
|
|
794
|
+
class EvalScorerInput(EvalSuiteContractModel):
|
|
795
|
+
"""Deterministic scorer input contract without live runner behavior."""
|
|
796
|
+
|
|
797
|
+
schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
|
|
798
|
+
trial_id: StrictStr
|
|
799
|
+
fixture_id: StrictStr
|
|
800
|
+
fixture_hash: StrictStr
|
|
801
|
+
final_workspace_hash: StrictStr | None = None
|
|
802
|
+
path_limited_workspace_hashes: tuple[EvalHashRecord, ...] = Field(
|
|
803
|
+
default_factory=tuple
|
|
804
|
+
)
|
|
805
|
+
public_artifact_hashes: tuple[EvalHashRecord, ...] = Field(default_factory=tuple)
|
|
806
|
+
required_public_artifact_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
807
|
+
provided_public_artifact_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
808
|
+
malformed_artifact_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
809
|
+
stage_terminal_results: tuple[StrictStr, ...]
|
|
810
|
+
capability_audit: EvalCapabilityAuditSummary
|
|
811
|
+
visible_check_results: tuple[EvalCheckResult, ...]
|
|
812
|
+
hidden_check_results: tuple[EvalCheckResult, ...]
|
|
813
|
+
claimed_mutation_present: StrictBool = True
|
|
814
|
+
unauthorized_mutation: StrictBool = False
|
|
815
|
+
checker_public_evidence_valid: StrictBool = True
|
|
816
|
+
runtime_failure: StrictBool = False
|
|
817
|
+
provider_failure: StrictBool = False
|
|
818
|
+
invalid_trial_explanation: StrictStr | None = None
|
|
819
|
+
scorer_input_hash_kind: StrictStr = EVAL_SUITE_SCORER_INPUT_HASH_KIND
|
|
820
|
+
scorer_input_hash: StrictStr
|
|
821
|
+
|
|
822
|
+
@model_validator(mode="after")
|
|
823
|
+
def _scorer_input_valid(self) -> EvalScorerInput:
|
|
824
|
+
if self.schema_version != EVAL_SUITE_SCHEMA_VERSION:
|
|
825
|
+
raise ValueError("unsupported eval-suite scorer input schema_version")
|
|
826
|
+
_validate_sha256(self.fixture_hash)
|
|
827
|
+
if self.final_workspace_hash is not None:
|
|
828
|
+
_validate_sha256(self.final_workspace_hash)
|
|
829
|
+
_validate_public_artifact_ids(self.required_public_artifact_ids)
|
|
830
|
+
_validate_public_artifact_ids(self.provided_public_artifact_ids)
|
|
831
|
+
_validate_public_artifact_ids(self.malformed_artifact_ids)
|
|
832
|
+
if len(set(self.required_public_artifact_ids)) != len(
|
|
833
|
+
self.required_public_artifact_ids
|
|
834
|
+
):
|
|
835
|
+
raise ValueError("required public artifact IDs must be unique")
|
|
836
|
+
if len(set(self.provided_public_artifact_ids)) != len(
|
|
837
|
+
self.provided_public_artifact_ids
|
|
838
|
+
):
|
|
839
|
+
raise ValueError("provided public artifact IDs must be unique")
|
|
840
|
+
if len(set(self.malformed_artifact_ids)) != len(self.malformed_artifact_ids):
|
|
841
|
+
raise ValueError("malformed public artifact IDs must be unique")
|
|
842
|
+
if self.scorer_input_hash_kind != EVAL_SUITE_SCORER_INPUT_HASH_KIND:
|
|
843
|
+
raise ValueError("unsupported scorer input hash kind")
|
|
844
|
+
_validate_sha256(self.scorer_input_hash)
|
|
845
|
+
expected = calculate_eval_scorer_input_hash(self)
|
|
846
|
+
if self.scorer_input_hash != expected:
|
|
847
|
+
raise ValueError("scorer_input_hash does not match input payload")
|
|
848
|
+
return self
|
|
849
|
+
|
|
850
|
+
|
|
851
|
+
class EvalScorerResult(EvalSuiteContractModel):
|
|
852
|
+
"""Deterministic scorer result contract."""
|
|
853
|
+
|
|
854
|
+
schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
|
|
855
|
+
trial_id: StrictStr
|
|
856
|
+
fixture_id: StrictStr
|
|
857
|
+
final_outcome: EvalTrialOutcome
|
|
858
|
+
primary_success: StrictBool
|
|
859
|
+
false_closure: StrictBool
|
|
860
|
+
false_success: StrictBool
|
|
861
|
+
correctly_blocked: StrictBool
|
|
862
|
+
capability_violation: StrictBool
|
|
863
|
+
artifact_complete: StrictBool
|
|
864
|
+
missing_artifact_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
865
|
+
malformed_artifact_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
866
|
+
failure_labels: tuple[EvalFailureTaxonomyLabel, ...] = Field(default_factory=tuple)
|
|
867
|
+
public_diagnostics: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
868
|
+
scorer_only_diagnostics: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
869
|
+
invalid_trial_explanation: StrictStr | None = None
|
|
870
|
+
scorer_version: StrictStr = EVAL_SUITE_DEFAULT_SCORER_VERSION
|
|
871
|
+
result_hash_kind: StrictStr = EVAL_SUITE_SCORER_RESULT_HASH_KIND
|
|
872
|
+
result_hash: StrictStr
|
|
873
|
+
|
|
874
|
+
@model_validator(mode="after")
|
|
875
|
+
def _scorer_result_valid(self) -> EvalScorerResult:
|
|
876
|
+
if self.schema_version != EVAL_SUITE_SCHEMA_VERSION:
|
|
877
|
+
raise ValueError("unsupported eval-suite scorer result schema_version")
|
|
878
|
+
if (
|
|
879
|
+
self.final_outcome == EvalTrialOutcome.INVALID_TRIAL
|
|
880
|
+
and not self.invalid_trial_explanation
|
|
881
|
+
):
|
|
882
|
+
raise ValueError("invalid_trial outcomes require an explanation")
|
|
883
|
+
if (
|
|
884
|
+
self.final_outcome != EvalTrialOutcome.INVALID_TRIAL
|
|
885
|
+
and self.invalid_trial_explanation
|
|
886
|
+
):
|
|
887
|
+
raise ValueError(
|
|
888
|
+
"invalid_trial_explanation is only valid for invalid_trial"
|
|
889
|
+
)
|
|
890
|
+
if self.final_outcome == EvalTrialOutcome.VALID_COMPLETION:
|
|
891
|
+
valid_completion_has_failure_evidence = (
|
|
892
|
+
self.false_closure
|
|
893
|
+
or self.false_success
|
|
894
|
+
or self.correctly_blocked
|
|
895
|
+
or self.capability_violation
|
|
896
|
+
or not self.artifact_complete
|
|
897
|
+
or self.missing_artifact_ids
|
|
898
|
+
or self.malformed_artifact_ids
|
|
899
|
+
or self.failure_labels
|
|
900
|
+
)
|
|
901
|
+
if self.primary_success and valid_completion_has_failure_evidence:
|
|
902
|
+
raise ValueError(
|
|
903
|
+
"valid_completion cannot include failure evidence or "
|
|
904
|
+
"inconsistent success flags"
|
|
905
|
+
)
|
|
906
|
+
if not self.primary_success and not valid_completion_has_failure_evidence:
|
|
907
|
+
raise ValueError(
|
|
908
|
+
"non-primary valid_completion requires failure evidence"
|
|
909
|
+
)
|
|
910
|
+
if self.false_closure:
|
|
911
|
+
raise ValueError("valid_completion cannot include false_closure")
|
|
912
|
+
if (
|
|
913
|
+
self.false_closure
|
|
914
|
+
and EvalFailureTaxonomyLabel.CAPABILITY_VIOLATION in self.failure_labels
|
|
915
|
+
):
|
|
916
|
+
if not self.capability_violation:
|
|
917
|
+
raise ValueError(
|
|
918
|
+
"capability violation label requires capability_violation"
|
|
919
|
+
)
|
|
920
|
+
if self.result_hash_kind != EVAL_SUITE_SCORER_RESULT_HASH_KIND:
|
|
921
|
+
raise ValueError("unsupported scorer result hash kind")
|
|
922
|
+
_validate_sha256(self.result_hash)
|
|
923
|
+
expected = calculate_eval_scorer_result_hash(self)
|
|
924
|
+
if self.result_hash != expected:
|
|
925
|
+
raise ValueError("result_hash does not match result payload")
|
|
926
|
+
return self
|
|
927
|
+
|
|
928
|
+
|
|
929
|
+
class EvalCampaignManifest(EvalSuiteContractModel):
|
|
930
|
+
"""Campaign contract tying modes, model, workflow, fixtures, and scorer."""
|
|
931
|
+
|
|
932
|
+
schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
|
|
933
|
+
campaign_id: StrictStr
|
|
934
|
+
campaign_kind: EvalCampaignKind
|
|
935
|
+
execution_mode: EvalSuiteExecutionMode
|
|
936
|
+
pi_eval_mode_id: StrictStr
|
|
937
|
+
millforge_eval_mode_id: StrictStr
|
|
938
|
+
model_manifest_ref: StrictStr
|
|
939
|
+
model_manifest_hash: StrictStr
|
|
940
|
+
workflow_graph_hash: StrictStr
|
|
941
|
+
fixture_pack_hash: StrictStr
|
|
942
|
+
scorer_version: StrictStr
|
|
943
|
+
created_at: StrictStr
|
|
944
|
+
budget_policy_ref: EvalBudgetPolicyReference
|
|
945
|
+
live_execution_admitted: StrictBool
|
|
946
|
+
live_denial_diagnostics: tuple[EvalLiveDenialDiagnostic, ...]
|
|
947
|
+
campaign_manifest_hash_kind: StrictStr = EVAL_SUITE_CAMPAIGN_MANIFEST_HASH_KIND
|
|
948
|
+
campaign_manifest_hash: StrictStr
|
|
949
|
+
|
|
950
|
+
@model_validator(mode="after")
|
|
951
|
+
def _campaign_manifest_valid(self) -> EvalCampaignManifest:
|
|
952
|
+
if self.schema_version != EVAL_SUITE_SCHEMA_VERSION:
|
|
953
|
+
raise ValueError("unsupported eval-suite campaign schema_version")
|
|
954
|
+
if not _UTC_TIMESTAMP_RE.fullmatch(self.created_at):
|
|
955
|
+
raise ValueError("campaign created_at must be a UTC timestamp")
|
|
956
|
+
for digest in (
|
|
957
|
+
self.model_manifest_hash,
|
|
958
|
+
self.workflow_graph_hash,
|
|
959
|
+
self.fixture_pack_hash,
|
|
960
|
+
self.campaign_manifest_hash,
|
|
961
|
+
):
|
|
962
|
+
_validate_sha256(digest)
|
|
963
|
+
if self.execution_mode == EvalSuiteExecutionMode.OFFLINE_FAKE:
|
|
964
|
+
if self.live_execution_admitted:
|
|
965
|
+
raise ValueError(
|
|
966
|
+
"offline eval-suite campaigns cannot admit live execution"
|
|
967
|
+
)
|
|
968
|
+
if not self.live_denial_diagnostics:
|
|
969
|
+
raise ValueError(
|
|
970
|
+
"offline campaigns must include live-denial diagnostics"
|
|
971
|
+
)
|
|
972
|
+
if self.execution_mode == EvalSuiteExecutionMode.LIVE_RUNNER:
|
|
973
|
+
raise ValueError(
|
|
974
|
+
"08A eval-suite campaign contracts do not admit live execution"
|
|
975
|
+
)
|
|
976
|
+
if self.campaign_manifest_hash_kind != EVAL_SUITE_CAMPAIGN_MANIFEST_HASH_KIND:
|
|
977
|
+
raise ValueError("unsupported campaign manifest hash kind")
|
|
978
|
+
expected = calculate_eval_campaign_manifest_hash(self)
|
|
979
|
+
if self.campaign_manifest_hash != expected:
|
|
980
|
+
raise ValueError("campaign_manifest_hash does not match manifest payload")
|
|
981
|
+
return self
|
|
982
|
+
|
|
983
|
+
|
|
984
|
+
def eval_model_manifest_from_profile(
|
|
985
|
+
profile: EvalModelProfile | None = None,
|
|
986
|
+
*,
|
|
987
|
+
model_manifest_id: str = "eval.08a.model.backend_neutral.default.v1",
|
|
988
|
+
) -> EvalModelManifest:
|
|
989
|
+
"""Build a campaign-grade manifest from the shared eval model profile."""
|
|
990
|
+
profile = profile or default_eval_model_profile()
|
|
991
|
+
manifest = EvalModelManifest.model_construct(
|
|
992
|
+
schema_version=EVAL_SUITE_SCHEMA_VERSION,
|
|
993
|
+
model_manifest_id=model_manifest_id,
|
|
994
|
+
model_profile_id=profile.profile_id,
|
|
995
|
+
model_profile_hash=profile.model_profile_hash,
|
|
996
|
+
provider_label=profile.provider_label,
|
|
997
|
+
model_or_artifact_id=profile.model_label,
|
|
998
|
+
release_or_snapshot="static-backend-neutral-profile",
|
|
999
|
+
serving_protocol=profile.serving_protocol,
|
|
1000
|
+
endpoint_class=EvalCampaignKind.LOCAL_OPENAI_COMPATIBLE,
|
|
1001
|
+
temperature=profile.temperature,
|
|
1002
|
+
top_p=profile.top_p,
|
|
1003
|
+
max_prompt_tokens=profile.max_prompt_tokens,
|
|
1004
|
+
max_completion_tokens=profile.max_completion_tokens,
|
|
1005
|
+
max_total_tokens=profile.max_total_tokens,
|
|
1006
|
+
tool_calling_mode=profile.tool_calling_mode,
|
|
1007
|
+
parser_id=profile.parser_id,
|
|
1008
|
+
seed_policy="no_seed",
|
|
1009
|
+
context_window_tokens=profile.max_total_tokens,
|
|
1010
|
+
reasoning_controls={"reasoning_effort": profile.reasoning_effort},
|
|
1011
|
+
public_pricing=EvalModelPricingMetadata(
|
|
1012
|
+
input_cost_per_million_tokens=0.0,
|
|
1013
|
+
output_cost_per_million_tokens=0.0,
|
|
1014
|
+
cached_input_cost_per_million_tokens=0.0,
|
|
1015
|
+
currency_label="none",
|
|
1016
|
+
source_label="static_descriptor",
|
|
1017
|
+
),
|
|
1018
|
+
public_rate_limits=EvalModelRateLimitMetadata(
|
|
1019
|
+
request_rate_per_window=0,
|
|
1020
|
+
token_rate_per_window=0,
|
|
1021
|
+
concurrent_request_limit=0,
|
|
1022
|
+
window_seconds=0,
|
|
1023
|
+
source_label="not_applicable",
|
|
1024
|
+
),
|
|
1025
|
+
public_pricing_snapshot=dict(profile.cost_accounting),
|
|
1026
|
+
public_rate_limit_snapshot={"rate_limit_source": "not_applicable"},
|
|
1027
|
+
local_serving_snapshot={"serving_snapshot": "not_applicable"},
|
|
1028
|
+
model_manifest_hash_kind=EVAL_SUITE_MODEL_MANIFEST_HASH_KIND,
|
|
1029
|
+
model_manifest_hash="0" * 64,
|
|
1030
|
+
)
|
|
1031
|
+
payload = manifest.model_dump(mode="json")
|
|
1032
|
+
payload["model_manifest_hash"] = calculate_eval_model_manifest_hash(manifest)
|
|
1033
|
+
return EvalModelManifest.model_validate(payload)
|
|
1034
|
+
|
|
1035
|
+
|
|
1036
|
+
def default_eval_suite_campaign_manifest(
|
|
1037
|
+
*,
|
|
1038
|
+
model_manifest: EvalModelManifest | None = None,
|
|
1039
|
+
fixture_pack_hash: str | None = None,
|
|
1040
|
+
created_at: str = EVAL_SUITE_DEFAULT_CAMPAIGN_CREATED_AT,
|
|
1041
|
+
) -> EvalCampaignManifest:
|
|
1042
|
+
"""Return the default 08A offline-only campaign manifest."""
|
|
1043
|
+
model_manifest = model_manifest or eval_model_manifest_from_profile()
|
|
1044
|
+
if fixture_pack_hash is None:
|
|
1045
|
+
fixture_pack_hash = load_eval_fixture_pack_summary().fixture_pack_hash
|
|
1046
|
+
manifest = EvalCampaignManifest.model_construct(
|
|
1047
|
+
schema_version=EVAL_SUITE_SCHEMA_VERSION,
|
|
1048
|
+
campaign_id=EVAL_SUITE_DEFAULT_CAMPAIGN_ID,
|
|
1049
|
+
campaign_kind=EvalCampaignKind.LOCAL_OPENAI_COMPATIBLE,
|
|
1050
|
+
execution_mode=EvalSuiteExecutionMode.OFFLINE_FAKE,
|
|
1051
|
+
pi_eval_mode_id=EVAL_SMALL_PI_MODE_ID,
|
|
1052
|
+
millforge_eval_mode_id=EVAL_SMALL_MILLFORGE_MODE_ID,
|
|
1053
|
+
model_manifest_ref=model_manifest.model_manifest_id,
|
|
1054
|
+
model_manifest_hash=model_manifest.model_manifest_hash,
|
|
1055
|
+
workflow_graph_hash=compact_eval_workflow_snapshot()["graph_sha256"],
|
|
1056
|
+
fixture_pack_hash=fixture_pack_hash,
|
|
1057
|
+
scorer_version=EVAL_SUITE_DEFAULT_SCORER_VERSION,
|
|
1058
|
+
created_at=created_at,
|
|
1059
|
+
budget_policy_ref=EvalBudgetPolicyReference(
|
|
1060
|
+
policy_id="eval.08a.default.offline_budget.v1",
|
|
1061
|
+
summary="Static offline fixture and scorer contract budget reference.",
|
|
1062
|
+
),
|
|
1063
|
+
live_execution_admitted=False,
|
|
1064
|
+
live_denial_diagnostics=(
|
|
1065
|
+
EvalLiveDenialDiagnostic(
|
|
1066
|
+
diagnostic_code="MF-EVAL-SUITE-001",
|
|
1067
|
+
summary="08A default campaign is offline-only and denies live execution.",
|
|
1068
|
+
rule_id="eval_suite.default_campaign.offline_only",
|
|
1069
|
+
),
|
|
1070
|
+
),
|
|
1071
|
+
campaign_manifest_hash_kind=EVAL_SUITE_CAMPAIGN_MANIFEST_HASH_KIND,
|
|
1072
|
+
campaign_manifest_hash="0" * 64,
|
|
1073
|
+
)
|
|
1074
|
+
payload = manifest.model_dump(mode="json")
|
|
1075
|
+
payload["campaign_manifest_hash"] = calculate_eval_campaign_manifest_hash(manifest)
|
|
1076
|
+
return EvalCampaignManifest.model_validate(payload)
|
|
1077
|
+
|
|
1078
|
+
|
|
1079
|
+
def configure_offline_fake_eval_campaign(
|
|
1080
|
+
*,
|
|
1081
|
+
output_root: str | Path,
|
|
1082
|
+
budget_policy: Any,
|
|
1083
|
+
fixture_pack_id: str = EVAL_SUITE_DEFAULT_FIXTURE_PACK_ID,
|
|
1084
|
+
fixture_ids: tuple[str, ...] | None = None,
|
|
1085
|
+
deterministic_seed: int = 0,
|
|
1086
|
+
trial_count_per_fixture_per_arm: int = 1,
|
|
1087
|
+
fake_runner_script: Any | None = None,
|
|
1088
|
+
allow_live_execution: bool = False,
|
|
1089
|
+
allow_live_model_call: bool = False,
|
|
1090
|
+
allow_pi_execution: bool = False,
|
|
1091
|
+
allow_millforge_harness_execution: bool = False,
|
|
1092
|
+
created_at: str = EVAL_SUITE_DEFAULT_CAMPAIGN_CREATED_AT,
|
|
1093
|
+
) -> EvalOfflineDryCampaignPlan:
|
|
1094
|
+
"""Preflight a deterministic offline fake campaign without writing records."""
|
|
1095
|
+
_reject_offline_dry_live_flags(
|
|
1096
|
+
allow_live_execution=allow_live_execution,
|
|
1097
|
+
allow_live_model_call=allow_live_model_call,
|
|
1098
|
+
allow_pi_execution=allow_pi_execution,
|
|
1099
|
+
allow_millforge_harness_execution=allow_millforge_harness_execution,
|
|
1100
|
+
)
|
|
1101
|
+
if fixture_pack_id != EVAL_SUITE_DEFAULT_FIXTURE_PACK_ID:
|
|
1102
|
+
raise ValueError(
|
|
1103
|
+
_offline_dry_diagnostic(
|
|
1104
|
+
EvalOfflineDryCampaignDiagnosticCode.FIXTURE_PACK_UNAVAILABLE,
|
|
1105
|
+
"eval_suite.dry_campaign.fixture_pack",
|
|
1106
|
+
"Only the installed default offline fixture pack is available.",
|
|
1107
|
+
).summary
|
|
1108
|
+
)
|
|
1109
|
+
if trial_count_per_fixture_per_arm <= 0:
|
|
1110
|
+
raise ValueError(
|
|
1111
|
+
_offline_dry_diagnostic(
|
|
1112
|
+
EvalOfflineDryCampaignDiagnosticCode.INVALID_TRIAL_COUNT,
|
|
1113
|
+
"eval_suite.dry_campaign.trial_count",
|
|
1114
|
+
"Dry campaigns require a positive trial count per fixture per arm.",
|
|
1115
|
+
).summary
|
|
1116
|
+
)
|
|
1117
|
+
_validate_offline_dry_output_root(output_root)
|
|
1118
|
+
|
|
1119
|
+
fixture_pack = load_eval_fixture_pack_summary()
|
|
1120
|
+
fixtures_by_id = {
|
|
1121
|
+
fixture.fixture_id: fixture for fixture in load_eval_task_fixtures()
|
|
1122
|
+
}
|
|
1123
|
+
selected_ids = fixture_ids or fixture_pack.fixture_ids
|
|
1124
|
+
missing_ids = tuple(
|
|
1125
|
+
fixture_id for fixture_id in selected_ids if fixture_id not in fixtures_by_id
|
|
1126
|
+
)
|
|
1127
|
+
if missing_ids:
|
|
1128
|
+
raise ValueError("unknown dry-campaign fixture ID")
|
|
1129
|
+
selected_fixtures = tuple(fixtures_by_id[fixture_id] for fixture_id in selected_ids)
|
|
1130
|
+
expanded_fixtures = tuple(
|
|
1131
|
+
fixture
|
|
1132
|
+
for fixture in selected_fixtures
|
|
1133
|
+
for _ in range(trial_count_per_fixture_per_arm)
|
|
1134
|
+
)
|
|
1135
|
+
trial_indexes = tuple(range(len(expanded_fixtures)))
|
|
1136
|
+
|
|
1137
|
+
campaign_manifest = default_eval_suite_campaign_manifest(
|
|
1138
|
+
fixture_pack_hash=fixture_pack.fixture_pack_hash,
|
|
1139
|
+
created_at=created_at,
|
|
1140
|
+
)
|
|
1141
|
+
from millforge.eval_reports import (
|
|
1142
|
+
EvalBudgetUsageEstimate,
|
|
1143
|
+
EvalReportBudgetPolicy,
|
|
1144
|
+
validate_eval_budget_policy,
|
|
1145
|
+
)
|
|
1146
|
+
from millforge.eval_trials import (
|
|
1147
|
+
EvalTrialArmId,
|
|
1148
|
+
canonical_eval_trial_store_manifest_bytes,
|
|
1149
|
+
plan_paired_eval_trials,
|
|
1150
|
+
)
|
|
1151
|
+
|
|
1152
|
+
if budget_policy is None:
|
|
1153
|
+
raise ValueError(
|
|
1154
|
+
_offline_dry_diagnostic(
|
|
1155
|
+
EvalOfflineDryCampaignDiagnosticCode.MISSING_BUDGET_POLICY,
|
|
1156
|
+
"eval_suite.dry_campaign.budget_policy",
|
|
1157
|
+
"Budget policy metadata is required for dry-campaign preflight.",
|
|
1158
|
+
).summary
|
|
1159
|
+
)
|
|
1160
|
+
budget_policy = EvalReportBudgetPolicy.model_validate(budget_policy)
|
|
1161
|
+
budget_result = validate_eval_budget_policy(
|
|
1162
|
+
budget_policy,
|
|
1163
|
+
campaign_manifest=campaign_manifest,
|
|
1164
|
+
usage=EvalBudgetUsageEstimate(
|
|
1165
|
+
estimated_spend_usd=0.0,
|
|
1166
|
+
prompt_tokens=0,
|
|
1167
|
+
completion_tokens=0,
|
|
1168
|
+
model_calls=0,
|
|
1169
|
+
retries_per_trial=0,
|
|
1170
|
+
wall_clock_seconds=0,
|
|
1171
|
+
trial_count=len(expanded_fixtures) * 2,
|
|
1172
|
+
),
|
|
1173
|
+
)
|
|
1174
|
+
if not budget_result.valid:
|
|
1175
|
+
raise ValueError(
|
|
1176
|
+
_offline_dry_diagnostic(
|
|
1177
|
+
EvalOfflineDryCampaignDiagnosticCode.INVALID_BUDGET_POLICY,
|
|
1178
|
+
"eval_suite.dry_campaign.budget_policy",
|
|
1179
|
+
budget_result.diagnostics[0].summary,
|
|
1180
|
+
).summary
|
|
1181
|
+
)
|
|
1182
|
+
|
|
1183
|
+
script = fake_runner_script or _default_offline_fake_runner_script()
|
|
1184
|
+
plans = plan_paired_eval_trials(
|
|
1185
|
+
fixtures=expanded_fixtures,
|
|
1186
|
+
fake_runner_script=script,
|
|
1187
|
+
seed=deterministic_seed,
|
|
1188
|
+
campaign_manifest=campaign_manifest,
|
|
1189
|
+
trial_indexes=trial_indexes,
|
|
1190
|
+
created_at=created_at,
|
|
1191
|
+
)
|
|
1192
|
+
store_manifest = _offline_dry_store_manifest(plans[0])
|
|
1193
|
+
manifest_bytes = canonical_eval_trial_store_manifest_bytes(store_manifest)
|
|
1194
|
+
manifest_path = Path(output_root) / plans[0].campaign_store_root / "manifest.json"
|
|
1195
|
+
if manifest_path.exists() and manifest_path.read_bytes() != manifest_bytes:
|
|
1196
|
+
raise ValueError(
|
|
1197
|
+
_offline_dry_diagnostic(
|
|
1198
|
+
EvalOfflineDryCampaignDiagnosticCode.MANIFEST_CONFLICT,
|
|
1199
|
+
"eval_suite.dry_campaign.manifest_conflict",
|
|
1200
|
+
"Existing campaign manifest differs from dry-campaign preflight.",
|
|
1201
|
+
).summary
|
|
1202
|
+
)
|
|
1203
|
+
|
|
1204
|
+
config = EvalOfflineDryCampaignConfig(
|
|
1205
|
+
fixture_pack_id=fixture_pack_id,
|
|
1206
|
+
fixture_ids=selected_ids,
|
|
1207
|
+
deterministic_seed=deterministic_seed,
|
|
1208
|
+
trial_count_per_fixture_per_arm=trial_count_per_fixture_per_arm,
|
|
1209
|
+
output_root_hash=hashlib.sha256(str(output_root).encode("utf-8")).hexdigest(),
|
|
1210
|
+
)
|
|
1211
|
+
return EvalOfflineDryCampaignPlan(
|
|
1212
|
+
config=config,
|
|
1213
|
+
campaign_manifest=campaign_manifest,
|
|
1214
|
+
fixture_pack_summary=fixture_pack,
|
|
1215
|
+
admitted_arm_ids=(
|
|
1216
|
+
EvalTrialArmId.EVAL_SMALL_PI.value,
|
|
1217
|
+
EvalTrialArmId.EVAL_SMALL_MILLFORGE.value,
|
|
1218
|
+
),
|
|
1219
|
+
paired_plan_count=len(plans),
|
|
1220
|
+
planned_arm_trial_count=len(plans) * 2,
|
|
1221
|
+
trial_plan_hashes=tuple(plan.plan_hash for plan in plans),
|
|
1222
|
+
campaign_store_root=plans[0].campaign_store_root,
|
|
1223
|
+
manifest_relative_path=f"{plans[0].campaign_store_root}/manifest.json",
|
|
1224
|
+
budget_policy_ref=EvalBudgetPolicyReference(
|
|
1225
|
+
policy_id=budget_policy.policy_id,
|
|
1226
|
+
summary=budget_policy.summary,
|
|
1227
|
+
),
|
|
1228
|
+
live_execution_admitted=False,
|
|
1229
|
+
diagnostics=(),
|
|
1230
|
+
)
|
|
1231
|
+
|
|
1232
|
+
|
|
1233
|
+
def run_offline_fake_eval_campaign(
|
|
1234
|
+
*,
|
|
1235
|
+
output_root: str | Path,
|
|
1236
|
+
budget_policy: Any,
|
|
1237
|
+
fixture_pack_id: str = EVAL_SUITE_DEFAULT_FIXTURE_PACK_ID,
|
|
1238
|
+
fixture_ids: tuple[str, ...] | None = None,
|
|
1239
|
+
deterministic_seed: int = 0,
|
|
1240
|
+
trial_count_per_fixture_per_arm: int = 1,
|
|
1241
|
+
fake_runner_script: Any | None = None,
|
|
1242
|
+
report_id: str | None = None,
|
|
1243
|
+
allow_live_execution: bool = False,
|
|
1244
|
+
allow_live_model_call: bool = False,
|
|
1245
|
+
allow_pi_execution: bool = False,
|
|
1246
|
+
allow_millforge_harness_execution: bool = False,
|
|
1247
|
+
created_at: str = EVAL_SUITE_DEFAULT_CAMPAIGN_CREATED_AT,
|
|
1248
|
+
) -> EvalOfflineDryCampaignRunResult:
|
|
1249
|
+
"""Run a deterministic offline fake campaign under a caller-selected root."""
|
|
1250
|
+
dry_plan = configure_offline_fake_eval_campaign(
|
|
1251
|
+
output_root=output_root,
|
|
1252
|
+
budget_policy=budget_policy,
|
|
1253
|
+
fixture_pack_id=fixture_pack_id,
|
|
1254
|
+
fixture_ids=fixture_ids,
|
|
1255
|
+
deterministic_seed=deterministic_seed,
|
|
1256
|
+
trial_count_per_fixture_per_arm=trial_count_per_fixture_per_arm,
|
|
1257
|
+
fake_runner_script=fake_runner_script,
|
|
1258
|
+
allow_live_execution=allow_live_execution,
|
|
1259
|
+
allow_live_model_call=allow_live_model_call,
|
|
1260
|
+
allow_pi_execution=allow_pi_execution,
|
|
1261
|
+
allow_millforge_harness_execution=allow_millforge_harness_execution,
|
|
1262
|
+
created_at=created_at,
|
|
1263
|
+
)
|
|
1264
|
+
|
|
1265
|
+
from millforge.eval_reports import (
|
|
1266
|
+
EvalReportBudgetPolicy,
|
|
1267
|
+
build_eval_report_artifact_bytes,
|
|
1268
|
+
build_eval_report_payload,
|
|
1269
|
+
)
|
|
1270
|
+
from millforge.eval_trials import (
|
|
1271
|
+
append_eval_trial_record_to_campaign_store,
|
|
1272
|
+
plan_paired_eval_trials,
|
|
1273
|
+
resume_eval_trial_campaign_store,
|
|
1274
|
+
run_offline_fake_eval_trial,
|
|
1275
|
+
)
|
|
1276
|
+
|
|
1277
|
+
budget = EvalReportBudgetPolicy.model_validate(budget_policy)
|
|
1278
|
+
fixtures_by_id = {
|
|
1279
|
+
fixture.fixture_id: fixture for fixture in load_eval_task_fixtures()
|
|
1280
|
+
}
|
|
1281
|
+
selected_fixtures = tuple(
|
|
1282
|
+
fixtures_by_id[fixture_id] for fixture_id in dry_plan.config.fixture_ids
|
|
1283
|
+
)
|
|
1284
|
+
expanded_fixtures = tuple(
|
|
1285
|
+
fixture
|
|
1286
|
+
for fixture in selected_fixtures
|
|
1287
|
+
for _ in range(trial_count_per_fixture_per_arm)
|
|
1288
|
+
)
|
|
1289
|
+
plans = plan_paired_eval_trials(
|
|
1290
|
+
fixtures=expanded_fixtures,
|
|
1291
|
+
fake_runner_script=fake_runner_script or _default_offline_fake_runner_script(),
|
|
1292
|
+
seed=deterministic_seed,
|
|
1293
|
+
campaign_manifest=dry_plan.campaign_manifest,
|
|
1294
|
+
trial_indexes=tuple(range(len(expanded_fixtures))),
|
|
1295
|
+
created_at=created_at,
|
|
1296
|
+
)
|
|
1297
|
+
if tuple(plan.plan_hash for plan in plans) != dry_plan.trial_plan_hashes:
|
|
1298
|
+
raise ValueError("dry-campaign plan hashes changed after preflight")
|
|
1299
|
+
|
|
1300
|
+
generated_records = {}
|
|
1301
|
+
for plan in plans:
|
|
1302
|
+
fixture = fixtures_by_id[plan.fixture_instance.fixture_id]
|
|
1303
|
+
generated_records[plan.trial_id] = run_offline_fake_eval_trial(
|
|
1304
|
+
plan,
|
|
1305
|
+
fixture=fixture,
|
|
1306
|
+
).trial_record
|
|
1307
|
+
existing_records = _read_offline_dry_campaign_record_summaries(
|
|
1308
|
+
output_root,
|
|
1309
|
+
plans[0],
|
|
1310
|
+
)
|
|
1311
|
+
_validate_offline_dry_record_summaries_match_plans(
|
|
1312
|
+
records=existing_records,
|
|
1313
|
+
plans=plans,
|
|
1314
|
+
generated_records=generated_records,
|
|
1315
|
+
)
|
|
1316
|
+
completed_ids = {record["trial_id"] for record in existing_records}
|
|
1317
|
+
appended_trial_ids: list[str] = []
|
|
1318
|
+
for plan in plans:
|
|
1319
|
+
if plan.trial_id in completed_ids:
|
|
1320
|
+
continue
|
|
1321
|
+
append_eval_trial_record_to_campaign_store(
|
|
1322
|
+
output_root,
|
|
1323
|
+
plan=plan,
|
|
1324
|
+
record=generated_records[plan.trial_id],
|
|
1325
|
+
plans=plans,
|
|
1326
|
+
)
|
|
1327
|
+
appended_trial_ids.append(plan.trial_id)
|
|
1328
|
+
completed_ids.add(plan.trial_id)
|
|
1329
|
+
|
|
1330
|
+
if appended_trial_ids:
|
|
1331
|
+
_offline_dry_output_path(
|
|
1332
|
+
output_root,
|
|
1333
|
+
f"{plans[0].campaign_store_root}/index.json",
|
|
1334
|
+
).unlink(missing_ok=True)
|
|
1335
|
+
final_resume = resume_eval_trial_campaign_store(
|
|
1336
|
+
output_root,
|
|
1337
|
+
plan=plans[0],
|
|
1338
|
+
plans=plans,
|
|
1339
|
+
)
|
|
1340
|
+
if final_resume.diagnostics:
|
|
1341
|
+
raise ValueError(final_resume.diagnostics[0].summary)
|
|
1342
|
+
if final_resume.resume_index is None:
|
|
1343
|
+
raise ValueError("dry-campaign resume index was not written")
|
|
1344
|
+
|
|
1345
|
+
records = tuple(generated_records[plan.trial_id] for plan in plans)
|
|
1346
|
+
payload = build_eval_report_payload(
|
|
1347
|
+
report_id=report_id or f"{dry_plan.campaign_manifest.campaign_id}.report.v1",
|
|
1348
|
+
campaign_manifest=dry_plan.campaign_manifest,
|
|
1349
|
+
plans=plans,
|
|
1350
|
+
records=records,
|
|
1351
|
+
resume_index=final_resume.resume_index,
|
|
1352
|
+
budget_policy=budget,
|
|
1353
|
+
generated_at=created_at,
|
|
1354
|
+
)
|
|
1355
|
+
report_artifacts = build_eval_report_artifact_bytes(payload)
|
|
1356
|
+
report_relative_paths = {
|
|
1357
|
+
name: f"{dry_plan.campaign_store_root}/reports/{name}"
|
|
1358
|
+
for name in sorted(report_artifacts)
|
|
1359
|
+
}
|
|
1360
|
+
for name, data in report_artifacts.items():
|
|
1361
|
+
path = _offline_dry_output_path(output_root, report_relative_paths[name])
|
|
1362
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
1363
|
+
path.write_bytes(data)
|
|
1364
|
+
closure_evidence = build_offline_dry_campaign_closure_evidence(
|
|
1365
|
+
run_plan=dry_plan,
|
|
1366
|
+
report_id=payload.report_id,
|
|
1367
|
+
completed_trial_ids=final_resume.completed_trial_ids,
|
|
1368
|
+
pending_trial_ids=final_resume.pending_trial_ids,
|
|
1369
|
+
trial_record_hashes=tuple(record.record_hash for record in records),
|
|
1370
|
+
resume_index_hash=final_resume.resume_index.resume_index_hash,
|
|
1371
|
+
report_hash=payload.report_hash,
|
|
1372
|
+
report_json_hash=hashlib.sha256(report_artifacts["report.json"]).hexdigest(),
|
|
1373
|
+
report_markdown_hash=hashlib.sha256(report_artifacts["report.md"]).hexdigest(),
|
|
1374
|
+
report_artifact_count=len(report_artifacts),
|
|
1375
|
+
treatment_compiled_harness_hashes=records[0].compiled_harness_hashes,
|
|
1376
|
+
)
|
|
1377
|
+
closure_evidence_relative_path = (
|
|
1378
|
+
f"{dry_plan.campaign_store_root}/closure_evidence.json"
|
|
1379
|
+
)
|
|
1380
|
+
_offline_dry_output_path(output_root, closure_evidence_relative_path).write_bytes(
|
|
1381
|
+
canonical_offline_dry_campaign_closure_evidence_bytes(closure_evidence)
|
|
1382
|
+
)
|
|
1383
|
+
|
|
1384
|
+
return EvalOfflineDryCampaignRunResult(
|
|
1385
|
+
dry_campaign_plan=dry_plan,
|
|
1386
|
+
report_id=payload.report_id,
|
|
1387
|
+
campaign_store_root=dry_plan.campaign_store_root,
|
|
1388
|
+
manifest_relative_path=dry_plan.manifest_relative_path,
|
|
1389
|
+
plan_relative_path=f"{dry_plan.campaign_store_root}/plan.json",
|
|
1390
|
+
trials_relative_path=f"{dry_plan.campaign_store_root}/trials.jsonl",
|
|
1391
|
+
index_relative_path=f"{dry_plan.campaign_store_root}/index.json",
|
|
1392
|
+
artifact_root_relative_path=f"{dry_plan.campaign_store_root}/artifacts",
|
|
1393
|
+
report_relative_paths=report_relative_paths,
|
|
1394
|
+
completed_trial_ids=final_resume.completed_trial_ids,
|
|
1395
|
+
pending_trial_ids=final_resume.pending_trial_ids,
|
|
1396
|
+
appended_trial_ids=tuple(appended_trial_ids),
|
|
1397
|
+
trial_plan_hashes=tuple(plan.plan_hash for plan in plans),
|
|
1398
|
+
trial_record_hashes=tuple(record.record_hash for record in records),
|
|
1399
|
+
resume_index_hash=final_resume.resume_index.resume_index_hash,
|
|
1400
|
+
report_hash=payload.report_hash,
|
|
1401
|
+
report_json_hash=hashlib.sha256(report_artifacts["report.json"]).hexdigest(),
|
|
1402
|
+
report_markdown_hash=hashlib.sha256(report_artifacts["report.md"]).hexdigest(),
|
|
1403
|
+
closure_evidence_relative_path=closure_evidence_relative_path,
|
|
1404
|
+
closure_evidence_hash=closure_evidence.closure_evidence_hash,
|
|
1405
|
+
live_execution_admitted=False,
|
|
1406
|
+
)
|
|
1407
|
+
|
|
1408
|
+
|
|
1409
|
+
def build_offline_dry_campaign_closure_evidence(
|
|
1410
|
+
*,
|
|
1411
|
+
run_plan: EvalOfflineDryCampaignPlan,
|
|
1412
|
+
report_id: str,
|
|
1413
|
+
completed_trial_ids: tuple[str, ...],
|
|
1414
|
+
pending_trial_ids: tuple[str, ...],
|
|
1415
|
+
trial_record_hashes: tuple[str, ...],
|
|
1416
|
+
resume_index_hash: str,
|
|
1417
|
+
report_hash: str,
|
|
1418
|
+
report_json_hash: str,
|
|
1419
|
+
report_markdown_hash: str,
|
|
1420
|
+
report_artifact_count: int,
|
|
1421
|
+
treatment_compiled_harness_hashes: Mapping[str, str],
|
|
1422
|
+
) -> EvalOfflineDryCampaignClosureEvidence:
|
|
1423
|
+
"""Return compact public closure evidence for a dry-campaign result."""
|
|
1424
|
+
evidence = EvalOfflineDryCampaignClosureEvidence.model_construct(
|
|
1425
|
+
schema_version=EVAL_SUITE_SCHEMA_VERSION,
|
|
1426
|
+
campaign_id=run_plan.campaign_manifest.campaign_id,
|
|
1427
|
+
report_id=report_id,
|
|
1428
|
+
fixture_pack_hash=run_plan.fixture_pack_summary.fixture_pack_hash,
|
|
1429
|
+
campaign_manifest_hash=run_plan.campaign_manifest.campaign_manifest_hash,
|
|
1430
|
+
model_manifest_hash=run_plan.campaign_manifest.model_manifest_hash,
|
|
1431
|
+
workflow_graph_hash=run_plan.campaign_manifest.workflow_graph_hash,
|
|
1432
|
+
trial_plan_hashes=run_plan.trial_plan_hashes,
|
|
1433
|
+
trial_record_hashes=trial_record_hashes,
|
|
1434
|
+
resume_index_hash=resume_index_hash,
|
|
1435
|
+
report_hash=report_hash,
|
|
1436
|
+
report_json_hash=report_json_hash,
|
|
1437
|
+
report_markdown_hash=report_markdown_hash,
|
|
1438
|
+
counts={
|
|
1439
|
+
"fixture_count": len(run_plan.config.fixture_ids),
|
|
1440
|
+
"paired_plan_count": run_plan.paired_plan_count,
|
|
1441
|
+
"arm_trial_count": run_plan.planned_arm_trial_count,
|
|
1442
|
+
"completed_trial_count": len(completed_trial_ids),
|
|
1443
|
+
"pending_trial_count": len(pending_trial_ids),
|
|
1444
|
+
"stored_trial_count": len(completed_trial_ids),
|
|
1445
|
+
"report_artifact_count": report_artifact_count,
|
|
1446
|
+
},
|
|
1447
|
+
unresolved_live_dependencies=(
|
|
1448
|
+
"pi_runtime",
|
|
1449
|
+
"millforge_live_harness_execution",
|
|
1450
|
+
"shared_model_backend_configuration",
|
|
1451
|
+
"fixture_workspace_lifecycle_reset",
|
|
1452
|
+
"resource_enforcement",
|
|
1453
|
+
),
|
|
1454
|
+
live_denial_diagnostic_codes=(
|
|
1455
|
+
"pi_runtime_unavailable",
|
|
1456
|
+
"millforge_live_harness_unavailable",
|
|
1457
|
+
"shared_backend_configuration_missing",
|
|
1458
|
+
"fixture_workspace_lifecycle_unavailable",
|
|
1459
|
+
"resource_enforcement_unavailable",
|
|
1460
|
+
),
|
|
1461
|
+
live_denial_test_coverage=(
|
|
1462
|
+
"tests/test_eval_reports.py::test_live_admission_returns_all_unresolved_dependency_diagnostics",
|
|
1463
|
+
"tests/test_eval_modes.py::test_live_admission_fails_closed_with_structured_deferred_dependencies",
|
|
1464
|
+
"tests/test_eval_trials.py::test_execution_result_rejects_live_admission",
|
|
1465
|
+
),
|
|
1466
|
+
treatment_compiled_harness_hashes=dict(
|
|
1467
|
+
sorted(treatment_compiled_harness_hashes.items())
|
|
1468
|
+
),
|
|
1469
|
+
public_hygiene_checks={
|
|
1470
|
+
"absolute_paths_absent": True,
|
|
1471
|
+
"home_paths_absent": True,
|
|
1472
|
+
"urls_absent": True,
|
|
1473
|
+
"auth_material_absent": True,
|
|
1474
|
+
"runtime_state_absent": True,
|
|
1475
|
+
"hidden_material_absent": True,
|
|
1476
|
+
"raw_logs_absent": True,
|
|
1477
|
+
},
|
|
1478
|
+
claim_boundary=(
|
|
1479
|
+
"Offline fake closure evidence proves deterministic public contract "
|
|
1480
|
+
"outputs only; live Pi-vs-Millforge remains denied."
|
|
1481
|
+
),
|
|
1482
|
+
closure_evidence_hash_kind=EVAL_SUITE_CLOSURE_EVIDENCE_HASH_KIND,
|
|
1483
|
+
closure_evidence_hash="0" * 64,
|
|
1484
|
+
)
|
|
1485
|
+
payload = evidence.model_dump(mode="json")
|
|
1486
|
+
payload["closure_evidence_hash"] = (
|
|
1487
|
+
calculate_offline_dry_campaign_closure_evidence_hash(evidence)
|
|
1488
|
+
)
|
|
1489
|
+
return EvalOfflineDryCampaignClosureEvidence.model_validate(payload)
|
|
1490
|
+
|
|
1491
|
+
|
|
1492
|
+
def eval_runner_task_projection(fixture: EvalTaskFixture) -> EvalRunnerTaskProjection:
|
|
1493
|
+
"""Return a runner-visible projection with scorer-only fields omitted."""
|
|
1494
|
+
return EvalRunnerTaskProjection(
|
|
1495
|
+
fixture_id=fixture.fixture_id,
|
|
1496
|
+
category=fixture.category,
|
|
1497
|
+
difficulty=fixture.difficulty,
|
|
1498
|
+
visible_prompt=fixture.visible_prompt,
|
|
1499
|
+
visible_acceptance_criteria=fixture.visible_acceptance_criteria,
|
|
1500
|
+
file_allowlist=fixture.file_allowlist,
|
|
1501
|
+
visible_checks=fixture.visible_checks,
|
|
1502
|
+
)
|
|
1503
|
+
|
|
1504
|
+
|
|
1505
|
+
def eval_runner_acceptance_projection(
|
|
1506
|
+
fixture: EvalTaskFixture,
|
|
1507
|
+
) -> EvalRunnerAcceptanceProjection:
|
|
1508
|
+
"""Return visible acceptance criteria and checks for runner consumption."""
|
|
1509
|
+
return EvalRunnerAcceptanceProjection(
|
|
1510
|
+
fixture_id=fixture.fixture_id,
|
|
1511
|
+
visible_acceptance_criteria=fixture.visible_acceptance_criteria,
|
|
1512
|
+
visible_checks=fixture.visible_checks,
|
|
1513
|
+
)
|
|
1514
|
+
|
|
1515
|
+
|
|
1516
|
+
def eval_runner_context_projection(
|
|
1517
|
+
fixture: EvalTaskFixture,
|
|
1518
|
+
) -> EvalRunnerContextProjection:
|
|
1519
|
+
"""Return visible fixture context for runner prompt assembly."""
|
|
1520
|
+
return EvalRunnerContextProjection(
|
|
1521
|
+
fixture_id=fixture.fixture_id,
|
|
1522
|
+
category=fixture.category,
|
|
1523
|
+
difficulty=fixture.difficulty,
|
|
1524
|
+
visible_prompt=fixture.visible_prompt,
|
|
1525
|
+
visible_acceptance_criteria=fixture.visible_acceptance_criteria,
|
|
1526
|
+
file_allowlist=fixture.file_allowlist,
|
|
1527
|
+
visible_checks=fixture.visible_checks,
|
|
1528
|
+
)
|
|
1529
|
+
|
|
1530
|
+
|
|
1531
|
+
def eval_public_artifact_projection(
|
|
1532
|
+
fixture: EvalTaskFixture,
|
|
1533
|
+
) -> EvalPublicArtifactProjection:
|
|
1534
|
+
"""Return public fixture metadata safe to serialize into trial artifacts."""
|
|
1535
|
+
return EvalPublicArtifactProjection(
|
|
1536
|
+
fixture_id=fixture.fixture_id,
|
|
1537
|
+
category=fixture.category,
|
|
1538
|
+
visible_acceptance_criteria=fixture.visible_acceptance_criteria,
|
|
1539
|
+
file_allowlist=fixture.file_allowlist,
|
|
1540
|
+
visible_checks=fixture.visible_checks,
|
|
1541
|
+
)
|
|
1542
|
+
|
|
1543
|
+
|
|
1544
|
+
def load_eval_task_fixtures() -> tuple[EvalTaskFixture, ...]:
|
|
1545
|
+
"""Load the built-in offline eval fixtures from package resources."""
|
|
1546
|
+
manifest = _load_default_eval_fixture_pack_manifest()
|
|
1547
|
+
fixture_ids = tuple(manifest["fixture_ids"])
|
|
1548
|
+
fixture_root = files(_EVAL_FIXTURE_PACK_PACKAGE).joinpath("fixtures")
|
|
1549
|
+
|
|
1550
|
+
fixtures = tuple(
|
|
1551
|
+
_load_eval_task_fixture_resource(
|
|
1552
|
+
fixture_root.joinpath(f"{fixture_id}.json"),
|
|
1553
|
+
)
|
|
1554
|
+
for fixture_id in fixture_ids
|
|
1555
|
+
)
|
|
1556
|
+
if tuple(fixture.fixture_id for fixture in fixtures) != fixture_ids:
|
|
1557
|
+
raise ValueError("fixture resource order does not match pack manifest")
|
|
1558
|
+
return fixtures
|
|
1559
|
+
|
|
1560
|
+
|
|
1561
|
+
def load_eval_task_fixture(fixture_id: str) -> EvalTaskFixture:
|
|
1562
|
+
"""Load a single built-in offline eval fixture by fixture ID."""
|
|
1563
|
+
fixtures_by_id = {
|
|
1564
|
+
fixture.fixture_id: fixture for fixture in load_eval_task_fixtures()
|
|
1565
|
+
}
|
|
1566
|
+
try:
|
|
1567
|
+
return fixtures_by_id[fixture_id]
|
|
1568
|
+
except KeyError as exc:
|
|
1569
|
+
raise KeyError(f"unknown eval-suite fixture_id: {fixture_id}") from exc
|
|
1570
|
+
|
|
1571
|
+
|
|
1572
|
+
def load_eval_fixture_pack_summary() -> EvalFixturePackSummary:
|
|
1573
|
+
"""Load the deterministic summary for the built-in offline fixture pack."""
|
|
1574
|
+
manifest = _load_default_eval_fixture_pack_manifest()
|
|
1575
|
+
fixtures = load_eval_task_fixtures()
|
|
1576
|
+
fixture_ids = tuple(fixture.fixture_id for fixture in fixtures)
|
|
1577
|
+
if fixture_ids != tuple(manifest["fixture_ids"]):
|
|
1578
|
+
raise ValueError("fixture pack manifest does not match loaded fixtures")
|
|
1579
|
+
|
|
1580
|
+
category_counts = {
|
|
1581
|
+
category: sum(1 for fixture in fixtures if fixture.category == category)
|
|
1582
|
+
for category in EvalTaskCategory
|
|
1583
|
+
if any(fixture.category == category for fixture in fixtures)
|
|
1584
|
+
}
|
|
1585
|
+
summary = EvalFixturePackSummary.model_construct(
|
|
1586
|
+
fixture_pack_id=str(manifest["fixture_pack_id"]),
|
|
1587
|
+
fixture_ids=fixture_ids,
|
|
1588
|
+
category_counts=category_counts,
|
|
1589
|
+
fixture_hashes=tuple(
|
|
1590
|
+
EvalHashRecord(
|
|
1591
|
+
hash_kind=EVAL_SUITE_FIXTURE_HASH_KIND,
|
|
1592
|
+
sha256=fixture.fixture_hash,
|
|
1593
|
+
)
|
|
1594
|
+
for fixture in fixtures
|
|
1595
|
+
),
|
|
1596
|
+
pack_summary=str(manifest["pack_summary"]),
|
|
1597
|
+
fixture_pack_hash_kind=EVAL_SUITE_FIXTURE_PACK_HASH_KIND,
|
|
1598
|
+
fixture_pack_hash="0" * 64,
|
|
1599
|
+
)
|
|
1600
|
+
payload = summary.model_dump(mode="json")
|
|
1601
|
+
payload["fixture_pack_hash"] = calculate_eval_fixture_pack_hash(summary)
|
|
1602
|
+
return EvalFixturePackSummary.model_validate(payload)
|
|
1603
|
+
|
|
1604
|
+
|
|
1605
|
+
_SUCCESS_TERMINAL_RESULTS = frozenset(
|
|
1606
|
+
{
|
|
1607
|
+
"PLAN_READY",
|
|
1608
|
+
"BUILDER_COMPLETE",
|
|
1609
|
+
"CHECKER_APPROVED",
|
|
1610
|
+
"ARBITER_CLOSED",
|
|
1611
|
+
}
|
|
1612
|
+
)
|
|
1613
|
+
_CLOSURE_TERMINAL_RESULTS = frozenset({"ARBITER_CLOSED"})
|
|
1614
|
+
_BLOCKED_TERMINAL_RESULTS = frozenset(
|
|
1615
|
+
{
|
|
1616
|
+
"PLAN_BLOCKED",
|
|
1617
|
+
"BUILDER_BLOCKED",
|
|
1618
|
+
"CHECKER_BLOCKED",
|
|
1619
|
+
"ARBITER_BLOCKED",
|
|
1620
|
+
}
|
|
1621
|
+
)
|
|
1622
|
+
|
|
1623
|
+
|
|
1624
|
+
def score_eval_trial(
|
|
1625
|
+
fixture: EvalTaskFixture,
|
|
1626
|
+
scorer_input: EvalScorerInput,
|
|
1627
|
+
) -> EvalScorerResult:
|
|
1628
|
+
"""Classify one offline eval trial with deterministic precedence."""
|
|
1629
|
+
if scorer_input.fixture_id != fixture.fixture_id:
|
|
1630
|
+
return _build_eval_scorer_result(
|
|
1631
|
+
scorer_input,
|
|
1632
|
+
final_outcome=EvalTrialOutcome.INVALID_TRIAL,
|
|
1633
|
+
primary_success=False,
|
|
1634
|
+
failure_labels=(EvalFailureTaxonomyLabel.INFRASTRUCTURE_DEFECT,),
|
|
1635
|
+
public_diagnostics=("Scorer input fixture_id does not match fixture.",),
|
|
1636
|
+
scorer_only_diagnostics=(
|
|
1637
|
+
f"expected fixture_id {fixture.fixture_id}; "
|
|
1638
|
+
f"got {scorer_input.fixture_id}",
|
|
1639
|
+
),
|
|
1640
|
+
invalid_trial_explanation="scorer input fixture_id does not match fixture",
|
|
1641
|
+
)
|
|
1642
|
+
if scorer_input.fixture_hash != fixture.fixture_hash:
|
|
1643
|
+
return _build_eval_scorer_result(
|
|
1644
|
+
scorer_input,
|
|
1645
|
+
final_outcome=EvalTrialOutcome.INVALID_TRIAL,
|
|
1646
|
+
primary_success=False,
|
|
1647
|
+
failure_labels=(EvalFailureTaxonomyLabel.INFRASTRUCTURE_DEFECT,),
|
|
1648
|
+
public_diagnostics=("Scorer input fixture hash does not match fixture.",),
|
|
1649
|
+
scorer_only_diagnostics=(
|
|
1650
|
+
f"expected fixture_hash {fixture.fixture_hash}; "
|
|
1651
|
+
f"got {scorer_input.fixture_hash}",
|
|
1652
|
+
),
|
|
1653
|
+
invalid_trial_explanation="scorer input fixture_hash does not match fixture",
|
|
1654
|
+
)
|
|
1655
|
+
if scorer_input.invalid_trial_explanation:
|
|
1656
|
+
return _build_eval_scorer_result(
|
|
1657
|
+
scorer_input,
|
|
1658
|
+
final_outcome=EvalTrialOutcome.INVALID_TRIAL,
|
|
1659
|
+
primary_success=False,
|
|
1660
|
+
failure_labels=(EvalFailureTaxonomyLabel.INFRASTRUCTURE_DEFECT,),
|
|
1661
|
+
public_diagnostics=("Evaluation infrastructure defect invalidated trial.",),
|
|
1662
|
+
scorer_only_diagnostics=(scorer_input.invalid_trial_explanation,),
|
|
1663
|
+
invalid_trial_explanation=scorer_input.invalid_trial_explanation,
|
|
1664
|
+
)
|
|
1665
|
+
|
|
1666
|
+
missing_artifact_ids = tuple(
|
|
1667
|
+
artifact_id
|
|
1668
|
+
for artifact_id in scorer_input.required_public_artifact_ids
|
|
1669
|
+
if artifact_id not in set(scorer_input.provided_public_artifact_ids)
|
|
1670
|
+
)
|
|
1671
|
+
malformed_artifact_ids = scorer_input.malformed_artifact_ids
|
|
1672
|
+
visible_failed = any(
|
|
1673
|
+
not result.passed for result in scorer_input.visible_check_results
|
|
1674
|
+
)
|
|
1675
|
+
hidden_failed = any(
|
|
1676
|
+
not result.passed for result in scorer_input.hidden_check_results
|
|
1677
|
+
)
|
|
1678
|
+
expected_mutation_absent = (
|
|
1679
|
+
fixture.expected_mutation_policy.mutation_kind
|
|
1680
|
+
!= EvalExpectedMutationKind.NO_SOURCE_CHANGE
|
|
1681
|
+
and not scorer_input.claimed_mutation_present
|
|
1682
|
+
)
|
|
1683
|
+
unauthorized_mutation = scorer_input.unauthorized_mutation
|
|
1684
|
+
artifact_complete = not missing_artifact_ids and not malformed_artifact_ids
|
|
1685
|
+
capability_violation = scorer_input.capability_audit.capability_violation
|
|
1686
|
+
invalid_public_evidence = not scorer_input.checker_public_evidence_valid
|
|
1687
|
+
evidence_defect = bool(
|
|
1688
|
+
visible_failed
|
|
1689
|
+
or hidden_failed
|
|
1690
|
+
or not artifact_complete
|
|
1691
|
+
or expected_mutation_absent
|
|
1692
|
+
or unauthorized_mutation
|
|
1693
|
+
or capability_violation
|
|
1694
|
+
or invalid_public_evidence
|
|
1695
|
+
)
|
|
1696
|
+
|
|
1697
|
+
failure_labels = _eval_failure_labels(
|
|
1698
|
+
visible_failed=visible_failed,
|
|
1699
|
+
hidden_failed=hidden_failed,
|
|
1700
|
+
missing_artifact_ids=missing_artifact_ids,
|
|
1701
|
+
malformed_artifact_ids=malformed_artifact_ids,
|
|
1702
|
+
expected_mutation_absent=expected_mutation_absent,
|
|
1703
|
+
unauthorized_mutation=unauthorized_mutation,
|
|
1704
|
+
capability_violation=capability_violation,
|
|
1705
|
+
success_terminal_unsupported=_success_terminal_unsupported(
|
|
1706
|
+
scorer_input,
|
|
1707
|
+
visible_failed=visible_failed,
|
|
1708
|
+
hidden_failed=hidden_failed,
|
|
1709
|
+
artifact_complete=artifact_complete,
|
|
1710
|
+
expected_mutation_absent=expected_mutation_absent,
|
|
1711
|
+
unauthorized_mutation=unauthorized_mutation,
|
|
1712
|
+
capability_violation=capability_violation,
|
|
1713
|
+
invalid_public_evidence=invalid_public_evidence,
|
|
1714
|
+
),
|
|
1715
|
+
invalid_public_evidence=invalid_public_evidence,
|
|
1716
|
+
provider_failure=scorer_input.provider_failure,
|
|
1717
|
+
)
|
|
1718
|
+
unsupported_success = (
|
|
1719
|
+
EvalFailureTaxonomyLabel.FALSE_SUCCESS_TERMINAL in failure_labels
|
|
1720
|
+
)
|
|
1721
|
+
final_closure_claimed = any(
|
|
1722
|
+
terminal in _CLOSURE_TERMINAL_RESULTS
|
|
1723
|
+
for terminal in scorer_input.stage_terminal_results
|
|
1724
|
+
)
|
|
1725
|
+
false_closure = final_closure_claimed and evidence_defect
|
|
1726
|
+
blocked = any(
|
|
1727
|
+
terminal in _BLOCKED_TERMINAL_RESULTS
|
|
1728
|
+
for terminal in scorer_input.stage_terminal_results
|
|
1729
|
+
)
|
|
1730
|
+
|
|
1731
|
+
if scorer_input.provider_failure:
|
|
1732
|
+
outcome = EvalTrialOutcome.PROVIDER_FAILURE
|
|
1733
|
+
elif scorer_input.runtime_failure:
|
|
1734
|
+
outcome = EvalTrialOutcome.RUNTIME_FAILURE
|
|
1735
|
+
elif false_closure:
|
|
1736
|
+
outcome = EvalTrialOutcome.FALSE_CLOSURE
|
|
1737
|
+
elif (
|
|
1738
|
+
blocked and fixture.expected_final_outcome == EvalTrialOutcome.CORRECTLY_BLOCKED
|
|
1739
|
+
):
|
|
1740
|
+
outcome = EvalTrialOutcome.CORRECTLY_BLOCKED
|
|
1741
|
+
elif blocked:
|
|
1742
|
+
outcome = EvalTrialOutcome.FALSE_BLOCKED
|
|
1743
|
+
else:
|
|
1744
|
+
outcome = EvalTrialOutcome.VALID_COMPLETION
|
|
1745
|
+
primary_success = (
|
|
1746
|
+
outcome == EvalTrialOutcome.VALID_COMPLETION
|
|
1747
|
+
and not evidence_defect
|
|
1748
|
+
and not unsupported_success
|
|
1749
|
+
)
|
|
1750
|
+
|
|
1751
|
+
return _build_eval_scorer_result(
|
|
1752
|
+
scorer_input,
|
|
1753
|
+
final_outcome=outcome,
|
|
1754
|
+
primary_success=primary_success,
|
|
1755
|
+
false_closure=false_closure,
|
|
1756
|
+
false_success=unsupported_success,
|
|
1757
|
+
correctly_blocked=outcome == EvalTrialOutcome.CORRECTLY_BLOCKED,
|
|
1758
|
+
capability_violation=capability_violation,
|
|
1759
|
+
artifact_complete=artifact_complete,
|
|
1760
|
+
missing_artifact_ids=missing_artifact_ids,
|
|
1761
|
+
malformed_artifact_ids=malformed_artifact_ids,
|
|
1762
|
+
failure_labels=failure_labels,
|
|
1763
|
+
public_diagnostics=_eval_public_diagnostics(
|
|
1764
|
+
outcome,
|
|
1765
|
+
missing_artifact_ids=missing_artifact_ids,
|
|
1766
|
+
malformed_artifact_ids=malformed_artifact_ids,
|
|
1767
|
+
capability_violation=capability_violation,
|
|
1768
|
+
visible_failed=visible_failed,
|
|
1769
|
+
unsupported_success=unsupported_success,
|
|
1770
|
+
),
|
|
1771
|
+
scorer_only_diagnostics=_eval_scorer_only_diagnostics(
|
|
1772
|
+
scorer_input,
|
|
1773
|
+
hidden_failed=hidden_failed,
|
|
1774
|
+
expected_mutation_absent=expected_mutation_absent,
|
|
1775
|
+
unauthorized_mutation=unauthorized_mutation,
|
|
1776
|
+
invalid_public_evidence=invalid_public_evidence,
|
|
1777
|
+
),
|
|
1778
|
+
)
|
|
1779
|
+
|
|
1780
|
+
|
|
1781
|
+
def canonical_eval_suite_bytes(value: BaseModel | Mapping[str, Any]) -> bytes:
|
|
1782
|
+
"""Return canonical ASCII JSON bytes for an eval-suite payload."""
|
|
1783
|
+
if isinstance(value, BaseModel):
|
|
1784
|
+
payload = value.model_dump(mode="json")
|
|
1785
|
+
else:
|
|
1786
|
+
payload = dict(value)
|
|
1787
|
+
return (
|
|
1788
|
+
json.dumps(payload, sort_keys=True, separators=(",", ":"), ensure_ascii=True)
|
|
1789
|
+
+ "\n"
|
|
1790
|
+
).encode("ascii")
|
|
1791
|
+
|
|
1792
|
+
|
|
1793
|
+
def canonical_offline_dry_campaign_closure_evidence_bytes(
|
|
1794
|
+
evidence: EvalOfflineDryCampaignClosureEvidence,
|
|
1795
|
+
) -> bytes:
|
|
1796
|
+
"""Return canonical ASCII JSON bytes for offline closure evidence."""
|
|
1797
|
+
return canonical_eval_suite_bytes(evidence)
|
|
1798
|
+
|
|
1799
|
+
|
|
1800
|
+
def calculate_offline_dry_campaign_closure_evidence_hash(
|
|
1801
|
+
evidence: EvalOfflineDryCampaignClosureEvidence,
|
|
1802
|
+
) -> str:
|
|
1803
|
+
"""Return the closure evidence hash with the self-hash field omitted."""
|
|
1804
|
+
payload = evidence.model_dump(mode="json")
|
|
1805
|
+
payload.pop("closure_evidence_hash", None)
|
|
1806
|
+
return hashlib.sha256(canonical_eval_suite_bytes(payload)).hexdigest()
|
|
1807
|
+
|
|
1808
|
+
|
|
1809
|
+
def calculate_eval_model_manifest_hash(manifest: EvalModelManifest) -> str:
|
|
1810
|
+
payload = manifest.model_dump(mode="json")
|
|
1811
|
+
payload.pop("model_manifest_hash", None)
|
|
1812
|
+
return hashlib.sha256(canonical_eval_suite_bytes(payload)).hexdigest()
|
|
1813
|
+
|
|
1814
|
+
|
|
1815
|
+
def calculate_eval_campaign_manifest_hash(manifest: EvalCampaignManifest) -> str:
|
|
1816
|
+
payload = manifest.model_dump(mode="json")
|
|
1817
|
+
payload.pop("campaign_manifest_hash", None)
|
|
1818
|
+
return hashlib.sha256(canonical_eval_suite_bytes(payload)).hexdigest()
|
|
1819
|
+
|
|
1820
|
+
|
|
1821
|
+
def calculate_eval_task_fixture_hash(fixture: EvalTaskFixture) -> str:
|
|
1822
|
+
payload = fixture.model_dump(mode="json")
|
|
1823
|
+
payload.pop("fixture_hash", None)
|
|
1824
|
+
return hashlib.sha256(canonical_eval_suite_bytes(payload)).hexdigest()
|
|
1825
|
+
|
|
1826
|
+
|
|
1827
|
+
def calculate_eval_fixture_pack_hash(summary: EvalFixturePackSummary) -> str:
|
|
1828
|
+
payload = summary.model_dump(mode="json")
|
|
1829
|
+
payload.pop("fixture_pack_hash", None)
|
|
1830
|
+
return hashlib.sha256(canonical_eval_suite_bytes(payload)).hexdigest()
|
|
1831
|
+
|
|
1832
|
+
|
|
1833
|
+
def calculate_eval_scorer_input_hash(scorer_input: EvalScorerInput) -> str:
|
|
1834
|
+
payload = scorer_input.model_dump(mode="json")
|
|
1835
|
+
payload.pop("scorer_input_hash", None)
|
|
1836
|
+
return hashlib.sha256(canonical_eval_suite_bytes(payload)).hexdigest()
|
|
1837
|
+
|
|
1838
|
+
|
|
1839
|
+
def calculate_eval_scorer_result_hash(result: EvalScorerResult) -> str:
|
|
1840
|
+
payload = result.model_dump(mode="json")
|
|
1841
|
+
payload.pop("result_hash", None)
|
|
1842
|
+
return hashlib.sha256(canonical_eval_suite_bytes(payload)).hexdigest()
|
|
1843
|
+
|
|
1844
|
+
|
|
1845
|
+
def _build_eval_scorer_result(
|
|
1846
|
+
scorer_input: EvalScorerInput,
|
|
1847
|
+
*,
|
|
1848
|
+
final_outcome: EvalTrialOutcome,
|
|
1849
|
+
primary_success: bool,
|
|
1850
|
+
false_closure: bool = False,
|
|
1851
|
+
false_success: bool = False,
|
|
1852
|
+
correctly_blocked: bool = False,
|
|
1853
|
+
capability_violation: bool | None = None,
|
|
1854
|
+
artifact_complete: bool = True,
|
|
1855
|
+
missing_artifact_ids: tuple[str, ...] = (),
|
|
1856
|
+
malformed_artifact_ids: tuple[str, ...] = (),
|
|
1857
|
+
failure_labels: tuple[EvalFailureTaxonomyLabel, ...] = (),
|
|
1858
|
+
public_diagnostics: tuple[str, ...] = (),
|
|
1859
|
+
scorer_only_diagnostics: tuple[str, ...] = (),
|
|
1860
|
+
invalid_trial_explanation: str | None = None,
|
|
1861
|
+
) -> EvalScorerResult:
|
|
1862
|
+
result = EvalScorerResult.model_construct(
|
|
1863
|
+
trial_id=scorer_input.trial_id,
|
|
1864
|
+
fixture_id=scorer_input.fixture_id,
|
|
1865
|
+
final_outcome=final_outcome,
|
|
1866
|
+
primary_success=primary_success,
|
|
1867
|
+
false_closure=false_closure,
|
|
1868
|
+
false_success=false_success,
|
|
1869
|
+
correctly_blocked=correctly_blocked,
|
|
1870
|
+
capability_violation=(
|
|
1871
|
+
scorer_input.capability_audit.capability_violation
|
|
1872
|
+
if capability_violation is None
|
|
1873
|
+
else capability_violation
|
|
1874
|
+
),
|
|
1875
|
+
artifact_complete=artifact_complete,
|
|
1876
|
+
missing_artifact_ids=missing_artifact_ids,
|
|
1877
|
+
malformed_artifact_ids=malformed_artifact_ids,
|
|
1878
|
+
failure_labels=failure_labels,
|
|
1879
|
+
public_diagnostics=public_diagnostics,
|
|
1880
|
+
scorer_only_diagnostics=scorer_only_diagnostics,
|
|
1881
|
+
invalid_trial_explanation=invalid_trial_explanation,
|
|
1882
|
+
scorer_version=EVAL_SUITE_DEFAULT_SCORER_VERSION,
|
|
1883
|
+
result_hash_kind=EVAL_SUITE_SCORER_RESULT_HASH_KIND,
|
|
1884
|
+
result_hash="0" * 64,
|
|
1885
|
+
)
|
|
1886
|
+
return EvalScorerResult.model_validate(
|
|
1887
|
+
result.model_copy(
|
|
1888
|
+
update={"result_hash": calculate_eval_scorer_result_hash(result)}
|
|
1889
|
+
)
|
|
1890
|
+
)
|
|
1891
|
+
|
|
1892
|
+
|
|
1893
|
+
def _success_terminal_unsupported(
|
|
1894
|
+
scorer_input: EvalScorerInput,
|
|
1895
|
+
*,
|
|
1896
|
+
visible_failed: bool,
|
|
1897
|
+
hidden_failed: bool,
|
|
1898
|
+
artifact_complete: bool,
|
|
1899
|
+
expected_mutation_absent: bool,
|
|
1900
|
+
unauthorized_mutation: bool,
|
|
1901
|
+
capability_violation: bool,
|
|
1902
|
+
invalid_public_evidence: bool,
|
|
1903
|
+
) -> bool:
|
|
1904
|
+
success_terminal_emitted = any(
|
|
1905
|
+
terminal in _SUCCESS_TERMINAL_RESULTS
|
|
1906
|
+
for terminal in scorer_input.stage_terminal_results
|
|
1907
|
+
)
|
|
1908
|
+
return success_terminal_emitted and bool(
|
|
1909
|
+
visible_failed
|
|
1910
|
+
or hidden_failed
|
|
1911
|
+
or not artifact_complete
|
|
1912
|
+
or expected_mutation_absent
|
|
1913
|
+
or unauthorized_mutation
|
|
1914
|
+
or capability_violation
|
|
1915
|
+
or invalid_public_evidence
|
|
1916
|
+
or scorer_input.runtime_failure
|
|
1917
|
+
or scorer_input.provider_failure
|
|
1918
|
+
)
|
|
1919
|
+
|
|
1920
|
+
|
|
1921
|
+
def _eval_failure_labels(
|
|
1922
|
+
*,
|
|
1923
|
+
visible_failed: bool,
|
|
1924
|
+
hidden_failed: bool,
|
|
1925
|
+
missing_artifact_ids: tuple[str, ...],
|
|
1926
|
+
malformed_artifact_ids: tuple[str, ...],
|
|
1927
|
+
expected_mutation_absent: bool,
|
|
1928
|
+
unauthorized_mutation: bool,
|
|
1929
|
+
capability_violation: bool,
|
|
1930
|
+
success_terminal_unsupported: bool,
|
|
1931
|
+
invalid_public_evidence: bool,
|
|
1932
|
+
provider_failure: bool,
|
|
1933
|
+
) -> tuple[EvalFailureTaxonomyLabel, ...]:
|
|
1934
|
+
labels: list[EvalFailureTaxonomyLabel] = []
|
|
1935
|
+
if visible_failed:
|
|
1936
|
+
labels.append(EvalFailureTaxonomyLabel.VISIBLE_CHECK_FAILED)
|
|
1937
|
+
if hidden_failed:
|
|
1938
|
+
labels.append(EvalFailureTaxonomyLabel.HIDDEN_CHECK_FAILED)
|
|
1939
|
+
if missing_artifact_ids:
|
|
1940
|
+
labels.append(EvalFailureTaxonomyLabel.REQUIRED_ARTIFACT_MISSING)
|
|
1941
|
+
if malformed_artifact_ids:
|
|
1942
|
+
labels.append(EvalFailureTaxonomyLabel.REQUIRED_ARTIFACT_MALFORMED)
|
|
1943
|
+
if expected_mutation_absent:
|
|
1944
|
+
labels.append(EvalFailureTaxonomyLabel.EXPECTED_MUTATION_ABSENT)
|
|
1945
|
+
if unauthorized_mutation:
|
|
1946
|
+
labels.append(EvalFailureTaxonomyLabel.UNAUTHORIZED_MUTATION)
|
|
1947
|
+
if capability_violation:
|
|
1948
|
+
labels.append(EvalFailureTaxonomyLabel.CAPABILITY_VIOLATION)
|
|
1949
|
+
if success_terminal_unsupported or invalid_public_evidence:
|
|
1950
|
+
labels.append(EvalFailureTaxonomyLabel.FALSE_SUCCESS_TERMINAL)
|
|
1951
|
+
if provider_failure:
|
|
1952
|
+
labels.append(EvalFailureTaxonomyLabel.PROVIDER_DEFECT)
|
|
1953
|
+
return tuple(dict.fromkeys(labels))
|
|
1954
|
+
|
|
1955
|
+
|
|
1956
|
+
def _eval_public_diagnostics(
|
|
1957
|
+
final_outcome: EvalTrialOutcome,
|
|
1958
|
+
*,
|
|
1959
|
+
missing_artifact_ids: tuple[str, ...],
|
|
1960
|
+
malformed_artifact_ids: tuple[str, ...],
|
|
1961
|
+
capability_violation: bool,
|
|
1962
|
+
visible_failed: bool,
|
|
1963
|
+
unsupported_success: bool,
|
|
1964
|
+
) -> tuple[str, ...]:
|
|
1965
|
+
diagnostics: list[str] = []
|
|
1966
|
+
if missing_artifact_ids:
|
|
1967
|
+
diagnostics.append("Required public artifacts are missing.")
|
|
1968
|
+
if malformed_artifact_ids:
|
|
1969
|
+
diagnostics.append("Required public artifacts are malformed.")
|
|
1970
|
+
if capability_violation:
|
|
1971
|
+
diagnostics.append("Capability envelope violation was observed.")
|
|
1972
|
+
if visible_failed:
|
|
1973
|
+
diagnostics.append("One or more visible checks failed.")
|
|
1974
|
+
if unsupported_success:
|
|
1975
|
+
diagnostics.append("A success terminal was unsupported by required evidence.")
|
|
1976
|
+
if final_outcome == EvalTrialOutcome.PROVIDER_FAILURE:
|
|
1977
|
+
diagnostics.append("Provider failure prevented a valid trial completion.")
|
|
1978
|
+
if final_outcome == EvalTrialOutcome.RUNTIME_FAILURE:
|
|
1979
|
+
diagnostics.append("Runtime failure prevented a valid trial completion.")
|
|
1980
|
+
return tuple(diagnostics)
|
|
1981
|
+
|
|
1982
|
+
|
|
1983
|
+
def _eval_scorer_only_diagnostics(
|
|
1984
|
+
scorer_input: EvalScorerInput,
|
|
1985
|
+
*,
|
|
1986
|
+
hidden_failed: bool,
|
|
1987
|
+
expected_mutation_absent: bool,
|
|
1988
|
+
unauthorized_mutation: bool,
|
|
1989
|
+
invalid_public_evidence: bool,
|
|
1990
|
+
) -> tuple[str, ...]:
|
|
1991
|
+
diagnostics: list[str] = []
|
|
1992
|
+
if hidden_failed:
|
|
1993
|
+
failed_ids = tuple(
|
|
1994
|
+
result.check_id
|
|
1995
|
+
for result in scorer_input.hidden_check_results
|
|
1996
|
+
if not result.passed
|
|
1997
|
+
)
|
|
1998
|
+
diagnostics.append(f"Hidden check failures: {', '.join(failed_ids)}.")
|
|
1999
|
+
if expected_mutation_absent:
|
|
2000
|
+
diagnostics.append("Expected workspace mutation was absent.")
|
|
2001
|
+
if unauthorized_mutation:
|
|
2002
|
+
diagnostics.append("Unauthorized workspace mutation was observed.")
|
|
2003
|
+
if invalid_public_evidence:
|
|
2004
|
+
diagnostics.append("Checker approved using invalid public evidence.")
|
|
2005
|
+
return tuple(diagnostics)
|
|
2006
|
+
|
|
2007
|
+
|
|
2008
|
+
def _validate_sha256(value: str) -> None:
|
|
2009
|
+
if not _SHA256_RE.fullmatch(value):
|
|
2010
|
+
raise ValueError("expected lowercase sha256 hex digest")
|
|
2011
|
+
|
|
2012
|
+
|
|
2013
|
+
def _validate_relative_path(value: str) -> None:
|
|
2014
|
+
if (
|
|
2015
|
+
not value
|
|
2016
|
+
or value.startswith(".")
|
|
2017
|
+
or "\\" in value
|
|
2018
|
+
or value.startswith("/")
|
|
2019
|
+
or ".." in value.split("/")
|
|
2020
|
+
):
|
|
2021
|
+
raise ValueError("fixture paths must be stable relative POSIX paths")
|
|
2022
|
+
_reject_forbidden_material(value)
|
|
2023
|
+
|
|
2024
|
+
|
|
2025
|
+
def _validate_relative_artifact_root(value: str) -> None:
|
|
2026
|
+
if (
|
|
2027
|
+
not value
|
|
2028
|
+
or value.startswith(".")
|
|
2029
|
+
or value.startswith("/")
|
|
2030
|
+
or "\\" in value
|
|
2031
|
+
or "//" in value
|
|
2032
|
+
or ".." in value.split("/")
|
|
2033
|
+
):
|
|
2034
|
+
raise ValueError("eval-suite artifact roots must be stable relative paths")
|
|
2035
|
+
_reject_forbidden_material(value)
|
|
2036
|
+
|
|
2037
|
+
|
|
2038
|
+
def _validate_public_artifact_ids(values: tuple[str, ...]) -> None:
|
|
2039
|
+
for value in values:
|
|
2040
|
+
if not value or "/" in value or "\\" in value or value.startswith("."):
|
|
2041
|
+
raise ValueError("public artifact IDs must be stable bare identifiers")
|
|
2042
|
+
_reject_forbidden_material(value)
|
|
2043
|
+
|
|
2044
|
+
|
|
2045
|
+
def _reject_offline_dry_live_flags(
|
|
2046
|
+
*,
|
|
2047
|
+
allow_live_execution: bool,
|
|
2048
|
+
allow_live_model_call: bool,
|
|
2049
|
+
allow_pi_execution: bool,
|
|
2050
|
+
allow_millforge_harness_execution: bool,
|
|
2051
|
+
) -> None:
|
|
2052
|
+
if any(
|
|
2053
|
+
(
|
|
2054
|
+
allow_live_execution,
|
|
2055
|
+
allow_live_model_call,
|
|
2056
|
+
allow_pi_execution,
|
|
2057
|
+
allow_millforge_harness_execution,
|
|
2058
|
+
)
|
|
2059
|
+
):
|
|
2060
|
+
raise ValueError(
|
|
2061
|
+
_offline_dry_diagnostic(
|
|
2062
|
+
EvalOfflineDryCampaignDiagnosticCode.LIVE_EXECUTION_UNAVAILABLE,
|
|
2063
|
+
"eval_suite.dry_campaign.live_execution",
|
|
2064
|
+
"Offline dry-campaign preflight rejects live execution flags.",
|
|
2065
|
+
).summary
|
|
2066
|
+
)
|
|
2067
|
+
|
|
2068
|
+
|
|
2069
|
+
def _validate_offline_dry_output_root(output_root: str | Path) -> None:
|
|
2070
|
+
root_text = str(output_root)
|
|
2071
|
+
if not root_text.strip():
|
|
2072
|
+
raise ValueError("output root is required")
|
|
2073
|
+
parts = tuple(part.lower() for part in Path(root_text).parts)
|
|
2074
|
+
unsafe_parts = {
|
|
2075
|
+
".claude",
|
|
2076
|
+
".codex",
|
|
2077
|
+
".eval-scratch",
|
|
2078
|
+
".millrace",
|
|
2079
|
+
".pytest_cache",
|
|
2080
|
+
".ruff_cache",
|
|
2081
|
+
"__pycache__",
|
|
2082
|
+
"ideas",
|
|
2083
|
+
"millrace-agents",
|
|
2084
|
+
"ref-forge",
|
|
2085
|
+
}
|
|
2086
|
+
if any(part in unsafe_parts for part in parts):
|
|
2087
|
+
raise ValueError(
|
|
2088
|
+
_offline_dry_diagnostic(
|
|
2089
|
+
EvalOfflineDryCampaignDiagnosticCode.UNSAFE_OUTPUT_ROOT,
|
|
2090
|
+
"eval_suite.dry_campaign.output_root",
|
|
2091
|
+
"Output root is under ignored control state.",
|
|
2092
|
+
).summary
|
|
2093
|
+
)
|
|
2094
|
+
|
|
2095
|
+
|
|
2096
|
+
def _offline_dry_output_path(output_root: str | Path, relative_path: str) -> Path:
|
|
2097
|
+
_validate_relative_artifact_root(relative_path)
|
|
2098
|
+
return Path(output_root).joinpath(*relative_path.split("/"))
|
|
2099
|
+
|
|
2100
|
+
|
|
2101
|
+
def _read_offline_dry_campaign_record_summaries(
|
|
2102
|
+
output_root: str | Path,
|
|
2103
|
+
plan: Any,
|
|
2104
|
+
) -> tuple[dict[str, str], ...]:
|
|
2105
|
+
trials_path = _offline_dry_output_path(
|
|
2106
|
+
output_root,
|
|
2107
|
+
f"{plan.campaign_store_root}/trials.jsonl",
|
|
2108
|
+
)
|
|
2109
|
+
if not trials_path.exists():
|
|
2110
|
+
return ()
|
|
2111
|
+
records: list[dict[str, str]] = []
|
|
2112
|
+
for line in trials_path.read_bytes().splitlines():
|
|
2113
|
+
if not line:
|
|
2114
|
+
continue
|
|
2115
|
+
payload = json.loads(line.decode("utf-8"))
|
|
2116
|
+
if not isinstance(payload, dict):
|
|
2117
|
+
raise ValueError("existing trial record must be a JSON object")
|
|
2118
|
+
for field_name in ("trial_id", "trial_plan_hash", "record_hash"):
|
|
2119
|
+
if not isinstance(payload.get(field_name), str):
|
|
2120
|
+
raise ValueError("existing trial record is missing public hash fields")
|
|
2121
|
+
record_hash = payload["record_hash"]
|
|
2122
|
+
hash_payload = dict(payload)
|
|
2123
|
+
hash_payload.pop("record_hash")
|
|
2124
|
+
expected = hashlib.sha256(canonical_eval_suite_bytes(hash_payload)).hexdigest()
|
|
2125
|
+
if record_hash != expected:
|
|
2126
|
+
raise ValueError("existing trial record hash does not match public payload")
|
|
2127
|
+
records.append(
|
|
2128
|
+
{
|
|
2129
|
+
"trial_id": payload["trial_id"],
|
|
2130
|
+
"trial_plan_hash": payload["trial_plan_hash"],
|
|
2131
|
+
"record_hash": record_hash,
|
|
2132
|
+
}
|
|
2133
|
+
)
|
|
2134
|
+
return tuple(records)
|
|
2135
|
+
|
|
2136
|
+
|
|
2137
|
+
def _validate_offline_dry_record_summaries_match_plans(
|
|
2138
|
+
*,
|
|
2139
|
+
records: tuple[dict[str, str], ...],
|
|
2140
|
+
plans: tuple[Any, ...],
|
|
2141
|
+
generated_records: Mapping[str, Any],
|
|
2142
|
+
) -> None:
|
|
2143
|
+
plans_by_trial_id = {plan.trial_id: plan for plan in plans}
|
|
2144
|
+
if len(plans_by_trial_id) != len(plans):
|
|
2145
|
+
raise ValueError("dry-campaign plans must have unique trial IDs")
|
|
2146
|
+
seen_trial_ids: set[str] = set()
|
|
2147
|
+
for record in records:
|
|
2148
|
+
trial_id = record["trial_id"]
|
|
2149
|
+
if trial_id in seen_trial_ids:
|
|
2150
|
+
raise ValueError("duplicate trial IDs are rejected by append-only stores")
|
|
2151
|
+
seen_trial_ids.add(trial_id)
|
|
2152
|
+
plan = plans_by_trial_id.get(trial_id)
|
|
2153
|
+
if plan is None:
|
|
2154
|
+
raise ValueError(
|
|
2155
|
+
"existing trial record is not present in dry-campaign plan"
|
|
2156
|
+
)
|
|
2157
|
+
if record["trial_plan_hash"] != plan.plan_hash:
|
|
2158
|
+
raise ValueError("existing trial record plan hash does not match plan")
|
|
2159
|
+
generated_record = generated_records.get(trial_id)
|
|
2160
|
+
if generated_record is None or record["record_hash"] != (
|
|
2161
|
+
generated_record.record_hash
|
|
2162
|
+
):
|
|
2163
|
+
raise ValueError("existing trial record hash does not match dry run")
|
|
2164
|
+
|
|
2165
|
+
|
|
2166
|
+
def _default_offline_fake_runner_script() -> Any:
|
|
2167
|
+
from millforge.eval_trials import (
|
|
2168
|
+
EvalFakeOutcomeScriptKind,
|
|
2169
|
+
EvalTrialFakeRunnerScript,
|
|
2170
|
+
)
|
|
2171
|
+
from millforge.eval_workflow import EvalStageId, EvalTerminalResult
|
|
2172
|
+
|
|
2173
|
+
return EvalTrialFakeRunnerScript(
|
|
2174
|
+
script_id="fake.valid_completion.v1",
|
|
2175
|
+
script_kind=EvalFakeOutcomeScriptKind.VALID_COMPLETION,
|
|
2176
|
+
terminal_results=(
|
|
2177
|
+
EvalTerminalResult.PLAN_READY,
|
|
2178
|
+
EvalTerminalResult.BUILDER_COMPLETE,
|
|
2179
|
+
EvalTerminalResult.CHECKER_APPROVED,
|
|
2180
|
+
EvalTerminalResult.ARBITER_CLOSED,
|
|
2181
|
+
),
|
|
2182
|
+
expected_outcome=EvalTrialOutcome.VALID_COMPLETION,
|
|
2183
|
+
stage_result_summaries={
|
|
2184
|
+
EvalStageId.PLANNER: "plan ready",
|
|
2185
|
+
EvalStageId.BUILDER: "builder complete",
|
|
2186
|
+
EvalStageId.CHECKER: "checker approved",
|
|
2187
|
+
EvalStageId.ARBITER: "arbiter closed",
|
|
2188
|
+
},
|
|
2189
|
+
)
|
|
2190
|
+
|
|
2191
|
+
|
|
2192
|
+
def _offline_dry_store_manifest(plan: Any) -> Any:
|
|
2193
|
+
from millforge.eval_trials import (
|
|
2194
|
+
EVAL_TRIAL_SCHEMA_VERSION,
|
|
2195
|
+
EVAL_TRIAL_STORE_MANIFEST_HASH_KIND,
|
|
2196
|
+
EvalTrialStoreManifest,
|
|
2197
|
+
calculate_eval_trial_store_manifest_hash,
|
|
2198
|
+
)
|
|
2199
|
+
|
|
2200
|
+
manifest = EvalTrialStoreManifest.model_construct(
|
|
2201
|
+
schema_version=EVAL_TRIAL_SCHEMA_VERSION,
|
|
2202
|
+
store_manifest_id=f"{plan.campaign_manifest.campaign_id}.store.v1",
|
|
2203
|
+
campaign_manifest_hash=plan.campaign_manifest.campaign_manifest_hash,
|
|
2204
|
+
record_hashes=(),
|
|
2205
|
+
append_only=True,
|
|
2206
|
+
store_manifest_hash_kind=EVAL_TRIAL_STORE_MANIFEST_HASH_KIND,
|
|
2207
|
+
store_manifest_hash="0" * 64,
|
|
2208
|
+
)
|
|
2209
|
+
return EvalTrialStoreManifest.model_validate(
|
|
2210
|
+
manifest.model_copy(
|
|
2211
|
+
update={
|
|
2212
|
+
"store_manifest_hash": calculate_eval_trial_store_manifest_hash(
|
|
2213
|
+
manifest
|
|
2214
|
+
)
|
|
2215
|
+
}
|
|
2216
|
+
)
|
|
2217
|
+
)
|
|
2218
|
+
|
|
2219
|
+
|
|
2220
|
+
def _offline_dry_diagnostic(
|
|
2221
|
+
code: EvalOfflineDryCampaignDiagnosticCode,
|
|
2222
|
+
rule_id: str,
|
|
2223
|
+
summary: str,
|
|
2224
|
+
) -> EvalOfflineDryCampaignDiagnostic:
|
|
2225
|
+
return EvalOfflineDryCampaignDiagnostic(
|
|
2226
|
+
diagnostic_code=code,
|
|
2227
|
+
rule_id=rule_id,
|
|
2228
|
+
summary=summary,
|
|
2229
|
+
)
|
|
2230
|
+
|
|
2231
|
+
|
|
2232
|
+
def _validate_public_command(command: str) -> None:
|
|
2233
|
+
stripped = command.strip()
|
|
2234
|
+
if not stripped:
|
|
2235
|
+
raise ValueError("commands must be non-empty")
|
|
2236
|
+
try:
|
|
2237
|
+
parts = shlex.split(stripped)
|
|
2238
|
+
except ValueError as exc:
|
|
2239
|
+
raise ValueError("commands must be shell-parseable") from exc
|
|
2240
|
+
if not parts:
|
|
2241
|
+
raise ValueError("commands must be non-empty")
|
|
2242
|
+
for token in _iter_public_command_tokens(parts):
|
|
2243
|
+
token_root = token.split(".", 1)[0]
|
|
2244
|
+
if token_root in _NETWORK_COMMANDS:
|
|
2245
|
+
raise ValueError("eval-suite checks must not require network commands")
|
|
2246
|
+
if token_root in _PACKAGE_COMMANDS:
|
|
2247
|
+
raise ValueError("eval-suite checks must not require package installation")
|
|
2248
|
+
if token in _NONDETERMINISTIC_COMMAND_TOKENS:
|
|
2249
|
+
raise ValueError("eval-suite checks must be deterministic")
|
|
2250
|
+
if any(pattern.search(stripped) for pattern in _NONDETERMINISTIC_PYTHON_PRIMITIVES):
|
|
2251
|
+
raise ValueError("eval-suite checks must be deterministic")
|
|
2252
|
+
_reject_forbidden_material(command)
|
|
2253
|
+
|
|
2254
|
+
|
|
2255
|
+
def _iter_public_command_tokens(parts: list[str]) -> tuple[str, ...]:
|
|
2256
|
+
tokens: list[str] = []
|
|
2257
|
+
for part in parts:
|
|
2258
|
+
for word in part.split():
|
|
2259
|
+
normalized = word.split("/")[-1].strip(" \t\r\n\"'`()[]{};,")
|
|
2260
|
+
if normalized.endswith(".exe"):
|
|
2261
|
+
normalized = normalized[:-4]
|
|
2262
|
+
if normalized:
|
|
2263
|
+
tokens.append(normalized.lower())
|
|
2264
|
+
return tuple(tokens)
|
|
2265
|
+
|
|
2266
|
+
|
|
2267
|
+
def _reject_forbidden_material(value: Any) -> None:
|
|
2268
|
+
if isinstance(value, Mapping):
|
|
2269
|
+
for key, child in value.items():
|
|
2270
|
+
key_text = str(key)
|
|
2271
|
+
_reject_secret_like_field_name(key_text)
|
|
2272
|
+
_reject_forbidden_material(key_text)
|
|
2273
|
+
_reject_forbidden_material(child)
|
|
2274
|
+
return
|
|
2275
|
+
if isinstance(value, (tuple, list, set, frozenset)):
|
|
2276
|
+
for child in value:
|
|
2277
|
+
_reject_forbidden_material(child)
|
|
2278
|
+
return
|
|
2279
|
+
if isinstance(value, Enum):
|
|
2280
|
+
_reject_forbidden_material(value.value)
|
|
2281
|
+
return
|
|
2282
|
+
if isinstance(value, str):
|
|
2283
|
+
lowered = value.lower()
|
|
2284
|
+
if any(token in lowered for token in _DENIED_TEXT_TOKENS):
|
|
2285
|
+
raise ValueError("eval-suite payload contains forbidden private material")
|
|
2286
|
+
if any(pattern.search(value) for pattern in _CREDENTIAL_VALUE_PATTERNS):
|
|
2287
|
+
raise ValueError("eval-suite payload contains credential-shaped API key")
|
|
2288
|
+
if _ENDPOINT_URL.search(value):
|
|
2289
|
+
raise ValueError("eval-suite payloads must not contain endpoint URLs")
|
|
2290
|
+
if (
|
|
2291
|
+
_WINDOWS_ABSOLUTE_PATH.search(value)
|
|
2292
|
+
or _POSIX_ABSOLUTE_PATH.search(value)
|
|
2293
|
+
or _USER_HOME_PATH.search(value)
|
|
2294
|
+
):
|
|
2295
|
+
raise ValueError("eval-suite payloads must not contain host paths")
|
|
2296
|
+
|
|
2297
|
+
|
|
2298
|
+
def _reject_secret_like_field_name(field_name: str) -> None:
|
|
2299
|
+
normalized = field_name.lower().replace("-", "_")
|
|
2300
|
+
if any(marker in normalized for marker in _SECRET_FIELD_MARKERS):
|
|
2301
|
+
raise ValueError("eval-suite payload contains secret-like field name")
|
|
2302
|
+
|
|
2303
|
+
|
|
2304
|
+
def _freeze_eval_suite_mapping(value: Mapping[Any, Any]) -> Mapping[Any, Any]:
|
|
2305
|
+
return _FrozenEvalSuiteDict(
|
|
2306
|
+
{key: _freeze_eval_suite_value(child) for key, child in value.items()}
|
|
2307
|
+
)
|
|
2308
|
+
|
|
2309
|
+
|
|
2310
|
+
def _freeze_eval_suite_value(value: Any) -> Any:
|
|
2311
|
+
if isinstance(value, Mapping):
|
|
2312
|
+
return _freeze_eval_suite_mapping(value)
|
|
2313
|
+
if isinstance(value, tuple):
|
|
2314
|
+
return tuple(_freeze_eval_suite_value(child) for child in value)
|
|
2315
|
+
if isinstance(value, list):
|
|
2316
|
+
return tuple(_freeze_eval_suite_value(child) for child in value)
|
|
2317
|
+
return value
|
|
2318
|
+
|
|
2319
|
+
|
|
2320
|
+
def _load_default_eval_fixture_pack_manifest() -> Mapping[str, Any]:
|
|
2321
|
+
manifest_resource = files(_EVAL_FIXTURE_PACK_PACKAGE).joinpath(
|
|
2322
|
+
_EVAL_FIXTURE_PACK_MANIFEST
|
|
2323
|
+
)
|
|
2324
|
+
manifest = json.loads(manifest_resource.read_text(encoding="utf-8"))
|
|
2325
|
+
if not isinstance(manifest, Mapping):
|
|
2326
|
+
raise ValueError("fixture pack manifest must be a JSON object")
|
|
2327
|
+
if manifest.get("fixture_pack_id") != EVAL_SUITE_DEFAULT_FIXTURE_PACK_ID:
|
|
2328
|
+
raise ValueError("fixture pack manifest has unexpected fixture_pack_id")
|
|
2329
|
+
fixture_ids = manifest.get("fixture_ids")
|
|
2330
|
+
if not isinstance(fixture_ids, list) or not fixture_ids:
|
|
2331
|
+
raise ValueError("fixture pack manifest must list fixture_ids")
|
|
2332
|
+
if not all(isinstance(fixture_id, str) for fixture_id in fixture_ids):
|
|
2333
|
+
raise ValueError("fixture pack manifest fixture_ids must be strings")
|
|
2334
|
+
return manifest
|
|
2335
|
+
|
|
2336
|
+
|
|
2337
|
+
def _load_eval_task_fixture_resource(resource: Any) -> EvalTaskFixture:
|
|
2338
|
+
payload = json.loads(resource.read_text(encoding="utf-8"))
|
|
2339
|
+
if not isinstance(payload, dict):
|
|
2340
|
+
raise ValueError("fixture resource must be a JSON object")
|
|
2341
|
+
payload["fixture_hash"] = _calculate_eval_suite_payload_hash(
|
|
2342
|
+
payload,
|
|
2343
|
+
hash_field="fixture_hash",
|
|
2344
|
+
)
|
|
2345
|
+
return EvalTaskFixture.model_validate(payload)
|
|
2346
|
+
|
|
2347
|
+
|
|
2348
|
+
def _calculate_eval_suite_payload_hash(
|
|
2349
|
+
payload: Mapping[str, Any],
|
|
2350
|
+
*,
|
|
2351
|
+
hash_field: str,
|
|
2352
|
+
) -> str:
|
|
2353
|
+
hash_payload = dict(payload)
|
|
2354
|
+
hash_payload.pop(hash_field, None)
|
|
2355
|
+
return hashlib.sha256(canonical_eval_suite_bytes(hash_payload)).hexdigest()
|
|
2356
|
+
|
|
2357
|
+
|
|
2358
|
+
__all__ = [
|
|
2359
|
+
"EVAL_SUITE_CAMPAIGN_MANIFEST_HASH_KIND",
|
|
2360
|
+
"EVAL_SUITE_CLOSURE_EVIDENCE_HASH_KIND",
|
|
2361
|
+
"EVAL_SUITE_DEFAULT_CAMPAIGN_CREATED_AT",
|
|
2362
|
+
"EVAL_SUITE_DEFAULT_CAMPAIGN_ID",
|
|
2363
|
+
"EVAL_SUITE_DEFAULT_FIXTURE_PACK_ID",
|
|
2364
|
+
"EVAL_SUITE_DEFAULT_SCORER_VERSION",
|
|
2365
|
+
"EVAL_SUITE_FIXTURE_HASH_KIND",
|
|
2366
|
+
"EVAL_SUITE_FIXTURE_PACK_HASH_KIND",
|
|
2367
|
+
"EVAL_SUITE_MODEL_MANIFEST_HASH_KIND",
|
|
2368
|
+
"EVAL_SUITE_OUTPUT_ROOT_HASH_KIND",
|
|
2369
|
+
"EVAL_SUITE_SCHEMA_VERSION",
|
|
2370
|
+
"EVAL_SUITE_SCORER_INPUT_HASH_KIND",
|
|
2371
|
+
"EVAL_SUITE_SCORER_RESULT_HASH_KIND",
|
|
2372
|
+
"EvalBudgetPolicyReference",
|
|
2373
|
+
"EvalCampaignKind",
|
|
2374
|
+
"EvalCampaignManifest",
|
|
2375
|
+
"EvalCapabilityAuditSummary",
|
|
2376
|
+
"EvalCheckResult",
|
|
2377
|
+
"EvalDifficultyLevel",
|
|
2378
|
+
"EvalDifficultyMetadata",
|
|
2379
|
+
"EvalExpectedMutationKind",
|
|
2380
|
+
"EvalExpectedMutationPolicy",
|
|
2381
|
+
"EvalFailureTaxonomyLabel",
|
|
2382
|
+
"EvalFixturePackSummary",
|
|
2383
|
+
"EvalHashRecord",
|
|
2384
|
+
"EvalHiddenCheck",
|
|
2385
|
+
"EvalLiveDenialDiagnostic",
|
|
2386
|
+
"EvalModelPricingMetadata",
|
|
2387
|
+
"EvalModelRateLimitMetadata",
|
|
2388
|
+
"EvalModelManifest",
|
|
2389
|
+
"EvalOfflineDryCampaignConfig",
|
|
2390
|
+
"EvalOfflineDryCampaignClosureEvidence",
|
|
2391
|
+
"EvalOfflineDryCampaignDiagnostic",
|
|
2392
|
+
"EvalOfflineDryCampaignDiagnosticCode",
|
|
2393
|
+
"EvalOfflineDryCampaignPlan",
|
|
2394
|
+
"EvalOfflineDryCampaignRunResult",
|
|
2395
|
+
"EvalPublicArtifactProjection",
|
|
2396
|
+
"EvalRunnerAcceptanceProjection",
|
|
2397
|
+
"EvalRunnerContextProjection",
|
|
2398
|
+
"EvalRunnerTaskProjection",
|
|
2399
|
+
"EvalScorerInput",
|
|
2400
|
+
"EvalScorerResult",
|
|
2401
|
+
"EvalSuiteContractModel",
|
|
2402
|
+
"EvalSuiteExecutionMode",
|
|
2403
|
+
"EvalTaskCategory",
|
|
2404
|
+
"EvalTaskFixture",
|
|
2405
|
+
"EvalTrialOutcome",
|
|
2406
|
+
"EvalVisibleCheck",
|
|
2407
|
+
"calculate_eval_campaign_manifest_hash",
|
|
2408
|
+
"calculate_eval_fixture_pack_hash",
|
|
2409
|
+
"calculate_eval_model_manifest_hash",
|
|
2410
|
+
"calculate_offline_dry_campaign_closure_evidence_hash",
|
|
2411
|
+
"calculate_eval_scorer_input_hash",
|
|
2412
|
+
"calculate_eval_scorer_result_hash",
|
|
2413
|
+
"calculate_eval_task_fixture_hash",
|
|
2414
|
+
"canonical_eval_suite_bytes",
|
|
2415
|
+
"canonical_offline_dry_campaign_closure_evidence_bytes",
|
|
2416
|
+
"configure_offline_fake_eval_campaign",
|
|
2417
|
+
"default_eval_suite_campaign_manifest",
|
|
2418
|
+
"build_offline_dry_campaign_closure_evidence",
|
|
2419
|
+
"eval_public_artifact_projection",
|
|
2420
|
+
"eval_model_manifest_from_profile",
|
|
2421
|
+
"eval_runner_acceptance_projection",
|
|
2422
|
+
"eval_runner_context_projection",
|
|
2423
|
+
"eval_runner_task_projection",
|
|
2424
|
+
"load_eval_fixture_pack_summary",
|
|
2425
|
+
"load_eval_task_fixture",
|
|
2426
|
+
"load_eval_task_fixtures",
|
|
2427
|
+
"run_offline_fake_eval_campaign",
|
|
2428
|
+
"score_eval_trial",
|
|
2429
|
+
]
|