millforge 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. millforge/__init__.py +1174 -0
  2. millforge/_forge/LICENSE +21 -0
  3. millforge/_forge/PROVENANCE.json +295 -0
  4. millforge/_forge/UPDATE_POLICY.md +24 -0
  5. millforge/_forge/__init__.py +14 -0
  6. millforge/_forge/adapter.py +2232 -0
  7. millforge/_forge/base_runner.py +121 -0
  8. millforge/_forge/clients/__init__.py +10 -0
  9. millforge/_forge/clients/base.py +200 -0
  10. millforge/_forge/context/__init__.py +23 -0
  11. millforge/_forge/context/manager.py +178 -0
  12. millforge/_forge/context/strategies.py +335 -0
  13. millforge/_forge/core/__init__.py +16 -0
  14. millforge/_forge/core/inference.py +433 -0
  15. millforge/_forge/core/messages.py +119 -0
  16. millforge/_forge/core/runner.py +479 -0
  17. millforge/_forge/core/steps.py +108 -0
  18. millforge/_forge/core/workflow.py +400 -0
  19. millforge/_forge/errors.py +222 -0
  20. millforge/_forge/guardrails/__init__.py +21 -0
  21. millforge/_forge/guardrails/error_tracker.py +71 -0
  22. millforge/_forge/guardrails/guardrails.py +194 -0
  23. millforge/_forge/guardrails/nudge.py +47 -0
  24. millforge/_forge/guardrails/response_validator.py +119 -0
  25. millforge/_forge/guardrails/step_enforcer.py +183 -0
  26. millforge/_forge/prompts/__init__.py +16 -0
  27. millforge/_forge/prompts/nudges.py +95 -0
  28. millforge/_forge/prompts/templates.py +285 -0
  29. millforge/_version.py +3 -0
  30. millforge/artifacts.py +570 -0
  31. millforge/base/__init__.py +97 -0
  32. millforge/base/composition.py +402 -0
  33. millforge/base/context.py +285 -0
  34. millforge/base/harness.py +138 -0
  35. millforge/base/identity.py +465 -0
  36. millforge/base/options.py +34 -0
  37. millforge/base/platform.py +17 -0
  38. millforge/base/prompt.py +317 -0
  39. millforge/base/runner.py +546 -0
  40. millforge/compiled_plan.py +970 -0
  41. millforge/compiler/__init__.py +231 -0
  42. millforge/compiler/artifact_validation.py +257 -0
  43. millforge/compiler/canonicalization.py +169 -0
  44. millforge/compiler/capabilities.py +66 -0
  45. millforge/compiler/catalogs.py +500 -0
  46. millforge/compiler/diagnostics.py +491 -0
  47. millforge/compiler/graph.py +678 -0
  48. millforge/compiler/lowering.py +198 -0
  49. millforge/compiler/output.py +692 -0
  50. millforge/compiler/parsing.py +1424 -0
  51. millforge/compiler/requests.py +1180 -0
  52. millforge/compiler/schema_validation.py +272 -0
  53. millforge/compiler/semantic.py +490 -0
  54. millforge/compiler/service.py +448 -0
  55. millforge/compiler/source.py +375 -0
  56. millforge/compiler/validators.py +184 -0
  57. millforge/connectors/__init__.py +95 -0
  58. millforge/connectors/admission.py +801 -0
  59. millforge/connectors/broker.py +202 -0
  60. millforge/connectors/contracts.py +1159 -0
  61. millforge/connectors/diagnostics.py +189 -0
  62. millforge/connectors/fake.py +66 -0
  63. millforge/connectors/runtime.py +236 -0
  64. millforge/contracts.py +2860 -0
  65. millforge/custom_tools/__init__.py +67 -0
  66. millforge/custom_tools/compiler.py +724 -0
  67. millforge/custom_tools/contracts.py +1093 -0
  68. millforge/custom_tools/diagnostics.py +205 -0
  69. millforge/eval_artifacts.py +952 -0
  70. millforge/eval_boundary.py +2435 -0
  71. millforge/eval_fixtures/__init__.py +1 -0
  72. millforge/eval_fixtures/default_pack/__init__.py +1 -0
  73. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.bug_diagnosis.traceback.v1.json +52 -0
  74. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.direct_edit.import_sort.v1.json +52 -0
  75. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.evidence_discipline.no_source_change.v1.json +51 -0
  76. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.false_closure.visible_green.v1.json +52 -0
  77. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.multi_file.api_contract.v1.json +54 -0
  78. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.recovery.malformed_artifact.v1.json +54 -0
  79. millforge/eval_fixtures/default_pack/manifest.json +12 -0
  80. millforge/eval_modes.py +1282 -0
  81. millforge/eval_presets.py +1398 -0
  82. millforge/eval_reports.py +2517 -0
  83. millforge/eval_suite.py +2429 -0
  84. millforge/eval_trials.py +2632 -0
  85. millforge/eval_workflow.py +794 -0
  86. millforge/exceptions.py +122 -0
  87. millforge/model_backend.py +2098 -0
  88. millforge/protocols.py +340 -0
  89. millforge/py.typed +0 -0
  90. millforge/runtime.py +1791 -0
  91. millforge/testing/__init__.py +1089 -0
  92. millforge/tools/__init__.py +83 -0
  93. millforge/tools/builtin_runtime.py +1339 -0
  94. millforge/tools/builtins.py +773 -0
  95. millforge/tools/execution.py +1545 -0
  96. millforge/tools/path_policy.py +155 -0
  97. millforge/tools/pi_compat/PI_LICENSE +21 -0
  98. millforge/tools/pi_compat/PROVENANCE.json +55 -0
  99. millforge/tools/pi_compat/UPDATE_POLICY.md +36 -0
  100. millforge/tools/pi_compat/__init__.py +34 -0
  101. millforge/tools/pi_compat/contracts.py +49 -0
  102. millforge/tools/pi_compat/editing.py +390 -0
  103. millforge/tools/pi_compat/mutations.py +57 -0
  104. millforge/tools/pi_compat/operations.py +401 -0
  105. millforge/tools/pi_compat/paths.py +155 -0
  106. millforge/tools/pi_compat/process.py +1375 -0
  107. millforge/tools/pi_compat/search.py +738 -0
  108. millforge/tools/pi_compat/truncation.py +267 -0
  109. millforge/tools/pi_compat_catalog.py +396 -0
  110. millforge/tools/pi_compat_runtime.py +460 -0
  111. millforge/tools/registry.py +553 -0
  112. millforge/tools/results.py +533 -0
  113. millforge-0.1.0.dist-info/METADATA +844 -0
  114. millforge-0.1.0.dist-info/RECORD +116 -0
  115. millforge-0.1.0.dist-info/WHEEL +4 -0
  116. millforge-0.1.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,2429 @@
1
+ """Public 08A offline eval-suite contract boundary."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import json
7
+ import re
8
+ import shlex
9
+ from importlib.resources import files
10
+ from collections.abc import Mapping
11
+ from enum import Enum
12
+ from pathlib import Path
13
+ from typing import Any
14
+
15
+ from pydantic import BaseModel, ConfigDict, Field, StrictBool, StrictFloat, StrictInt
16
+ from pydantic import StrictStr, field_validator, model_validator
17
+
18
+ from millforge.eval_modes import (
19
+ EVAL_SMALL_MILLFORGE_MODE_ID,
20
+ EVAL_SMALL_PI_MODE_ID,
21
+ EvalModelProfile,
22
+ default_eval_model_profile,
23
+ )
24
+ from millforge.eval_workflow import compact_eval_workflow_snapshot
25
+
26
+ EVAL_SUITE_SCHEMA_VERSION = 1
27
+ EVAL_SUITE_CAMPAIGN_MANIFEST_HASH_KIND = "eval_suite_campaign_manifest_sha256_v1"
28
+ EVAL_SUITE_MODEL_MANIFEST_HASH_KIND = "eval_suite_model_manifest_sha256_v1"
29
+ EVAL_SUITE_FIXTURE_HASH_KIND = "eval_suite_fixture_sha256_v1"
30
+ EVAL_SUITE_FIXTURE_PACK_HASH_KIND = "eval_suite_fixture_pack_sha256_v1"
31
+ EVAL_SUITE_SCORER_INPUT_HASH_KIND = "eval_suite_scorer_input_sha256_v1"
32
+ EVAL_SUITE_SCORER_RESULT_HASH_KIND = "eval_suite_scorer_result_sha256_v1"
33
+ EVAL_SUITE_DEFAULT_CAMPAIGN_ID = "eval.08a.default.offline.v1"
34
+ EVAL_SUITE_DEFAULT_CAMPAIGN_CREATED_AT = "1970-01-01T00:00:00Z"
35
+ EVAL_SUITE_DEFAULT_FIXTURE_PACK_ID = "pack.08a.offline.default.v1"
36
+ EVAL_SUITE_DEFAULT_SCORER_VERSION = "eval_suite.scorer.contract.v1"
37
+ EVAL_SUITE_OUTPUT_ROOT_HASH_KIND = "eval_suite_output_root_sha256_v1"
38
+ EVAL_SUITE_CLOSURE_EVIDENCE_HASH_KIND = "eval_suite_offline_closure_evidence_sha256_v1"
39
+
40
+ _EVAL_FIXTURE_PACK_PACKAGE = "millforge.eval_fixtures.default_pack"
41
+ _EVAL_FIXTURE_PACK_MANIFEST = "manifest.json"
42
+
43
+ _SHA256_RE = re.compile(r"^[0-9a-f]{64}$")
44
+ _UTC_TIMESTAMP_RE = re.compile(r"^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}Z$")
45
+ _WINDOWS_ABSOLUTE_PATH = re.compile(
46
+ r"(?:^|[\s\"'`([{<])(?:[A-Za-z]:[\\/]|\\\\)[^\s\"'`)\]}>,]*"
47
+ )
48
+ _POSIX_ABSOLUTE_PATH = re.compile(r"(?:^|[\s\"'`([{<])/(?!/)[^\s\"'`)\]}>,]*")
49
+ _USER_HOME_PATH = re.compile(r"(?:^|[\s\"'`([{<])~(?:/|\\)[^\s\"'`)\]}>,]*")
50
+ _ENDPOINT_URL = re.compile(r"https?://|localhost(?::|/|$)|127\.0\.0\.1|0\.0\.0\.0")
51
+ _CREDENTIAL_VALUE_PATTERNS = (
52
+ re.compile(r"\bsk-(?:live|proj|test)-[A-Za-z0-9_-]{10,}\b"),
53
+ re.compile(r"\b[rs]k_(?:live|test)_[A-Za-z0-9]{16,}\b"),
54
+ re.compile(r"\b(?:AKIA|ASIA)[0-9A-Z]{16}\b"),
55
+ re.compile(r"\bAIza[0-9A-Za-z_-]{20,}\b"),
56
+ re.compile(r"\bgh[opsu]_[A-Za-z0-9_]{20,}\b"),
57
+ )
58
+ _SECRET_FIELD_MARKERS = (
59
+ "api_key",
60
+ "apikey",
61
+ "auth_header",
62
+ "authorization",
63
+ "bearer",
64
+ "client_secret",
65
+ "credential",
66
+ "password",
67
+ "private_key",
68
+ "secret",
69
+ "access_token",
70
+ "auth_token",
71
+ "refresh_token",
72
+ )
73
+ _DENIED_TEXT_TOKENS = (
74
+ "api_key",
75
+ "authorization:",
76
+ "bearer ",
77
+ "credential",
78
+ "password",
79
+ "secret",
80
+ "access token",
81
+ "auth token",
82
+ "bearer token",
83
+ "millrace-agents",
84
+ ".millrace",
85
+ "daemon state",
86
+ "endpoint_url",
87
+ "endpoint url",
88
+ "external service",
89
+ "hidden answer",
90
+ "local planning",
91
+ "private runtime",
92
+ "network access",
93
+ "package install",
94
+ "ref-forge",
95
+ ".claude",
96
+ ".codex",
97
+ )
98
+ _NETWORK_COMMANDS = frozenset(
99
+ {"curl", "ftp", "nc", "netcat", "scp", "sftp", "ssh", "telnet", "wget"}
100
+ )
101
+ _PACKAGE_COMMANDS = frozenset(
102
+ {"cargo", "gem", "npm", "pip", "pip3", "pnpm", "poetry", "uv", "yarn"}
103
+ )
104
+ _NONDETERMINISTIC_COMMAND_TOKENS = frozenset(
105
+ {"date", "random", "sleep", "time", "uuid"}
106
+ )
107
+ _NONDETERMINISTIC_PYTHON_PRIMITIVES = (
108
+ re.compile(r"\buuid\.uuid4\s*\("),
109
+ re.compile(r"\btime\.time\s*\("),
110
+ )
111
+
112
+
113
+ class _FrozenEvalSuiteDict(dict[Any, Any]):
114
+ """Dict-shaped immutable mapping that remains serializable by Pydantic."""
115
+
116
+ def __readonly(self, *args: Any, **kwargs: Any) -> None:
117
+ raise TypeError("eval-suite mappings are immutable")
118
+
119
+ __setitem__ = __readonly
120
+ __delitem__ = __readonly
121
+ clear = __readonly
122
+ pop = __readonly
123
+ popitem = __readonly # type: ignore[assignment]
124
+ setdefault = __readonly
125
+ update = __readonly
126
+ __ior__ = __readonly # type: ignore[assignment]
127
+
128
+
129
+ class EvalSuiteContractModel(BaseModel):
130
+ """Closed, frozen base for public eval-suite contracts."""
131
+
132
+ model_config = ConfigDict(extra="forbid", frozen=True)
133
+
134
+ @model_validator(mode="before")
135
+ @classmethod
136
+ def _reject_forbidden_payload(cls, data: Any) -> Any:
137
+ _reject_forbidden_material(data)
138
+ return data
139
+
140
+
141
+ class EvalCampaignKind(str, Enum):
142
+ """Closed campaign backend classes."""
143
+
144
+ HOSTED_API = "hosted_api"
145
+ LOCAL_OPENAI_COMPATIBLE = "local_openai_compatible"
146
+ LOCAL_NATIVE = "local_native"
147
+
148
+
149
+ class EvalSuiteExecutionMode(str, Enum):
150
+ """Closed eval-suite execution modes."""
151
+
152
+ OFFLINE_FAKE = "offline_fake"
153
+ LIVE_RUNNER = "live_runner"
154
+
155
+
156
+ class EvalTaskCategory(str, Enum):
157
+ """Closed 08A fixture task categories."""
158
+
159
+ DIRECT_EDIT = "direct_edit"
160
+ MULTI_FILE_CONSISTENCY = "multi_file_consistency"
161
+ BUG_DIAGNOSIS = "bug_diagnosis"
162
+ EVIDENCE_DISCIPLINE = "evidence_discipline"
163
+ RECOVERY = "recovery"
164
+ FALSE_CLOSURE_TRAP = "false_closure_trap"
165
+
166
+
167
+ class EvalDifficultyLevel(str, Enum):
168
+ """Coarse public fixture difficulty labels."""
169
+
170
+ BASIC = "basic"
171
+ INTERMEDIATE = "intermediate"
172
+ ADVANCED = "advanced"
173
+
174
+
175
+ class EvalExpectedMutationKind(str, Enum):
176
+ """Expected workspace mutation policy for a fixture."""
177
+
178
+ NO_SOURCE_CHANGE = "no_source_change"
179
+ SOURCE_CHANGE_REQUIRED = "source_change_required"
180
+ DOCUMENTATION_ONLY = "documentation_only"
181
+ TEST_ONLY = "test_only"
182
+
183
+
184
+ class EvalTrialOutcome(str, Enum):
185
+ """Closed final scorer outcome classes."""
186
+
187
+ VALID_COMPLETION = "valid_completion"
188
+ CORRECTLY_BLOCKED = "correctly_blocked"
189
+ FALSE_CLOSURE = "false_closure"
190
+ FALSE_BLOCKED = "false_blocked"
191
+ RUNTIME_FAILURE = "runtime_failure"
192
+ PROVIDER_FAILURE = "provider_failure"
193
+ INVALID_TRIAL = "invalid_trial"
194
+
195
+
196
+ class EvalOfflineDryCampaignDiagnosticCode(str, Enum):
197
+ """Fail-closed diagnostics for offline dry-campaign preflight."""
198
+
199
+ MISSING_BUDGET_POLICY = "missing_budget_policy"
200
+ INVALID_BUDGET_POLICY = "invalid_budget_policy"
201
+ LIVE_EXECUTION_UNAVAILABLE = "live_execution_unavailable"
202
+ UNSAFE_OUTPUT_ROOT = "unsafe_output_root"
203
+ FIXTURE_PACK_UNAVAILABLE = "fixture_pack_unavailable"
204
+ INVALID_TRIAL_COUNT = "invalid_trial_count"
205
+ MANIFEST_CONFLICT = "manifest_conflict"
206
+
207
+
208
+ class EvalFailureTaxonomyLabel(str, Enum):
209
+ """Closed public failure taxonomy labels for scorer results."""
210
+
211
+ VISIBLE_CHECK_FAILED = "visible_check_failed"
212
+ HIDDEN_CHECK_FAILED = "hidden_check_failed"
213
+ REQUIRED_ARTIFACT_MISSING = "required_artifact_missing"
214
+ REQUIRED_ARTIFACT_MALFORMED = "required_artifact_malformed"
215
+ EXPECTED_MUTATION_ABSENT = "expected_mutation_absent"
216
+ UNAUTHORIZED_MUTATION = "unauthorized_mutation"
217
+ CAPABILITY_VIOLATION = "capability_violation"
218
+ FALSE_SUCCESS_TERMINAL = "false_success_terminal"
219
+ INFRASTRUCTURE_DEFECT = "infrastructure_defect"
220
+ PROVIDER_DEFECT = "provider_defect"
221
+
222
+
223
+ class EvalHashRecord(EvalSuiteContractModel):
224
+ """Typed hash reference for eval-suite records."""
225
+
226
+ hash_kind: StrictStr
227
+ sha256: StrictStr
228
+
229
+ @field_validator("sha256")
230
+ @classmethod
231
+ def _sha256_valid(cls, value: str) -> str:
232
+ _validate_sha256(value)
233
+ return value
234
+
235
+
236
+ class EvalBudgetPolicyReference(EvalSuiteContractModel):
237
+ """Public reference to a campaign budget policy."""
238
+
239
+ policy_id: StrictStr
240
+ summary: StrictStr
241
+
242
+
243
+ class EvalLiveDenialDiagnostic(EvalSuiteContractModel):
244
+ """Fail-closed live execution denial diagnostic."""
245
+
246
+ diagnostic_code: StrictStr
247
+ summary: StrictStr
248
+ rule_id: StrictStr
249
+
250
+
251
+ class EvalOfflineDryCampaignDiagnostic(EvalSuiteContractModel):
252
+ """Structured public diagnostic for dry-campaign validation failures."""
253
+
254
+ diagnostic_code: EvalOfflineDryCampaignDiagnosticCode
255
+ rule_id: StrictStr
256
+ summary: StrictStr
257
+
258
+
259
+ class EvalOfflineDryCampaignConfig(EvalSuiteContractModel):
260
+ """Public offline dry-campaign configuration after preflight validation."""
261
+
262
+ schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
263
+ fixture_pack_id: StrictStr = EVAL_SUITE_DEFAULT_FIXTURE_PACK_ID
264
+ fixture_ids: tuple[StrictStr, ...]
265
+ deterministic_seed: StrictInt = Field(ge=0)
266
+ trial_count_per_fixture_per_arm: StrictInt = Field(gt=0)
267
+ output_root_hash_kind: StrictStr = EVAL_SUITE_OUTPUT_ROOT_HASH_KIND
268
+ output_root_hash: StrictStr
269
+ output_root_policy: StrictStr = "caller_provided_preflight_validated"
270
+
271
+ @model_validator(mode="after")
272
+ def _config_valid(self) -> EvalOfflineDryCampaignConfig:
273
+ if self.schema_version != EVAL_SUITE_SCHEMA_VERSION:
274
+ raise ValueError("unsupported eval-suite dry-campaign schema_version")
275
+ if self.fixture_pack_id != EVAL_SUITE_DEFAULT_FIXTURE_PACK_ID:
276
+ raise ValueError("only the default offline fixture pack is installed")
277
+ if not self.fixture_ids:
278
+ raise ValueError("dry campaigns require at least one fixture")
279
+ if len(set(self.fixture_ids)) != len(self.fixture_ids):
280
+ raise ValueError("dry-campaign fixture IDs must be unique")
281
+ if self.output_root_hash_kind != EVAL_SUITE_OUTPUT_ROOT_HASH_KIND:
282
+ raise ValueError("unsupported output root hash kind")
283
+ _validate_sha256(self.output_root_hash)
284
+ return self
285
+
286
+
287
+ class EvalOfflineDryCampaignPlan(EvalSuiteContractModel):
288
+ """Public preflight result for a deterministic offline fake campaign."""
289
+
290
+ schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
291
+ config: EvalOfflineDryCampaignConfig
292
+ campaign_manifest: EvalCampaignManifest
293
+ fixture_pack_summary: EvalFixturePackSummary
294
+ admitted_arm_ids: tuple[StrictStr, StrictStr]
295
+ paired_plan_count: StrictInt = Field(ge=0)
296
+ planned_arm_trial_count: StrictInt = Field(ge=0)
297
+ trial_plan_hashes: tuple[StrictStr, ...]
298
+ campaign_store_root: StrictStr
299
+ manifest_relative_path: StrictStr
300
+ budget_policy_ref: EvalBudgetPolicyReference
301
+ live_execution_admitted: StrictBool = False
302
+ diagnostics: tuple[EvalOfflineDryCampaignDiagnostic, ...] = Field(
303
+ default_factory=tuple
304
+ )
305
+
306
+ @model_validator(mode="after")
307
+ def _dry_plan_valid(self) -> EvalOfflineDryCampaignPlan:
308
+ if self.schema_version != EVAL_SUITE_SCHEMA_VERSION:
309
+ raise ValueError("unsupported eval-suite dry-campaign schema_version")
310
+ if (
311
+ self.campaign_manifest.execution_mode
312
+ is not EvalSuiteExecutionMode.OFFLINE_FAKE
313
+ ):
314
+ raise ValueError("dry campaigns must use offline fake execution")
315
+ if (
316
+ self.live_execution_admitted
317
+ or self.campaign_manifest.live_execution_admitted
318
+ ):
319
+ raise ValueError("dry campaigns do not admit live execution")
320
+ if self.fixture_pack_summary.fixture_pack_id != self.config.fixture_pack_id:
321
+ raise ValueError("fixture pack summary must match dry-campaign config")
322
+ if self.fixture_pack_summary.fixture_pack_hash != (
323
+ self.campaign_manifest.fixture_pack_hash
324
+ ):
325
+ raise ValueError("campaign manifest must reference fixture pack summary")
326
+ if self.paired_plan_count != len(self.trial_plan_hashes):
327
+ raise ValueError("paired_plan_count must match trial_plan_hashes")
328
+ if self.planned_arm_trial_count != self.paired_plan_count * len(
329
+ self.admitted_arm_ids
330
+ ):
331
+ raise ValueError("planned_arm_trial_count must match admitted arms")
332
+ for digest in self.trial_plan_hashes:
333
+ _validate_sha256(digest)
334
+ _validate_relative_artifact_root(self.campaign_store_root)
335
+ _validate_relative_artifact_root(self.manifest_relative_path)
336
+ if self.diagnostics:
337
+ raise ValueError("valid dry-campaign plans must not carry diagnostics")
338
+ return self
339
+
340
+
341
+ class EvalOfflineDryCampaignRunResult(EvalSuiteContractModel):
342
+ """Filesystem result from one deterministic offline fake campaign run."""
343
+
344
+ schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
345
+ dry_campaign_plan: EvalOfflineDryCampaignPlan
346
+ report_id: StrictStr
347
+ campaign_store_root: StrictStr
348
+ manifest_relative_path: StrictStr
349
+ plan_relative_path: StrictStr
350
+ trials_relative_path: StrictStr
351
+ index_relative_path: StrictStr
352
+ artifact_root_relative_path: StrictStr
353
+ report_relative_paths: Mapping[StrictStr, StrictStr]
354
+ completed_trial_ids: tuple[StrictStr, ...]
355
+ pending_trial_ids: tuple[StrictStr, ...]
356
+ appended_trial_ids: tuple[StrictStr, ...]
357
+ trial_plan_hashes: tuple[StrictStr, ...]
358
+ trial_record_hashes: tuple[StrictStr, ...]
359
+ resume_index_hash: StrictStr
360
+ report_hash: StrictStr
361
+ report_json_hash: StrictStr
362
+ report_markdown_hash: StrictStr
363
+ closure_evidence_relative_path: StrictStr
364
+ closure_evidence_hash: StrictStr
365
+ live_execution_admitted: StrictBool = False
366
+
367
+ @model_validator(mode="after")
368
+ def _run_result_valid(self) -> EvalOfflineDryCampaignRunResult:
369
+ if self.schema_version != EVAL_SUITE_SCHEMA_VERSION:
370
+ raise ValueError("unsupported eval-suite dry-campaign schema_version")
371
+ if self.live_execution_admitted:
372
+ raise ValueError("dry campaigns do not admit live execution")
373
+ if self.campaign_store_root != self.dry_campaign_plan.campaign_store_root:
374
+ raise ValueError("campaign store root must match dry-campaign plan")
375
+ for path_value in (
376
+ self.campaign_store_root,
377
+ self.manifest_relative_path,
378
+ self.plan_relative_path,
379
+ self.trials_relative_path,
380
+ self.index_relative_path,
381
+ self.artifact_root_relative_path,
382
+ self.closure_evidence_relative_path,
383
+ *self.report_relative_paths.values(),
384
+ ):
385
+ _validate_relative_artifact_root(path_value)
386
+ expected_prefix = f"{self.campaign_store_root}/"
387
+ for path_value in (
388
+ self.manifest_relative_path,
389
+ self.plan_relative_path,
390
+ self.trials_relative_path,
391
+ self.index_relative_path,
392
+ self.artifact_root_relative_path,
393
+ self.closure_evidence_relative_path,
394
+ *self.report_relative_paths.values(),
395
+ ):
396
+ if not path_value.startswith(expected_prefix):
397
+ raise ValueError("dry-campaign output paths must be campaign-relative")
398
+ for digest in (
399
+ *self.trial_plan_hashes,
400
+ *self.trial_record_hashes,
401
+ self.resume_index_hash,
402
+ self.report_hash,
403
+ self.report_json_hash,
404
+ self.report_markdown_hash,
405
+ self.closure_evidence_hash,
406
+ ):
407
+ _validate_sha256(digest)
408
+ if set(self.report_relative_paths) != {
409
+ "report.json",
410
+ "report.md",
411
+ "report.sha256",
412
+ }:
413
+ raise ValueError("dry-campaign reports must include the public report set")
414
+ object.__setattr__(
415
+ self,
416
+ "report_relative_paths",
417
+ _freeze_eval_suite_mapping(self.report_relative_paths),
418
+ )
419
+ return self
420
+
421
+
422
+ class EvalOfflineDryCampaignClosureEvidence(EvalSuiteContractModel):
423
+ """Compact public closure evidence for one offline dry-campaign run."""
424
+
425
+ schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
426
+ campaign_id: StrictStr
427
+ report_id: StrictStr
428
+ fixture_pack_hash: StrictStr
429
+ campaign_manifest_hash: StrictStr
430
+ model_manifest_hash: StrictStr
431
+ workflow_graph_hash: StrictStr
432
+ trial_plan_hashes: tuple[StrictStr, ...]
433
+ trial_record_hashes: tuple[StrictStr, ...]
434
+ resume_index_hash: StrictStr
435
+ report_hash: StrictStr
436
+ report_json_hash: StrictStr
437
+ report_markdown_hash: StrictStr
438
+ counts: Mapping[StrictStr, StrictInt]
439
+ unresolved_live_dependencies: tuple[StrictStr, ...]
440
+ live_denial_diagnostic_codes: tuple[StrictStr, ...]
441
+ live_denial_test_coverage: tuple[StrictStr, ...]
442
+ treatment_compiled_harness_hashes: Mapping[StrictStr, StrictStr]
443
+ public_hygiene_checks: Mapping[StrictStr, StrictBool]
444
+ claim_boundary: StrictStr
445
+ closure_evidence_hash_kind: StrictStr = EVAL_SUITE_CLOSURE_EVIDENCE_HASH_KIND
446
+ closure_evidence_hash: StrictStr
447
+
448
+ @model_validator(mode="after")
449
+ def _closure_evidence_valid(self) -> EvalOfflineDryCampaignClosureEvidence:
450
+ if self.schema_version != EVAL_SUITE_SCHEMA_VERSION:
451
+ raise ValueError("unsupported eval-suite closure-evidence schema_version")
452
+ for digest in (
453
+ self.fixture_pack_hash,
454
+ self.campaign_manifest_hash,
455
+ self.model_manifest_hash,
456
+ self.workflow_graph_hash,
457
+ *self.trial_plan_hashes,
458
+ *self.trial_record_hashes,
459
+ self.resume_index_hash,
460
+ self.report_hash,
461
+ self.report_json_hash,
462
+ self.report_markdown_hash,
463
+ *self.treatment_compiled_harness_hashes.values(),
464
+ self.closure_evidence_hash,
465
+ ):
466
+ _validate_sha256(digest)
467
+ if self.closure_evidence_hash_kind != EVAL_SUITE_CLOSURE_EVIDENCE_HASH_KIND:
468
+ raise ValueError("unsupported closure evidence hash kind")
469
+ required_counts = {
470
+ "fixture_count",
471
+ "paired_plan_count",
472
+ "arm_trial_count",
473
+ "completed_trial_count",
474
+ "pending_trial_count",
475
+ "stored_trial_count",
476
+ "report_artifact_count",
477
+ }
478
+ if set(self.counts) != required_counts:
479
+ raise ValueError("closure evidence counts must use the compact count set")
480
+ if any(count < 0 for count in self.counts.values()):
481
+ raise ValueError("closure evidence counts must be non-negative")
482
+ required_hygiene = {
483
+ "absolute_paths_absent",
484
+ "home_paths_absent",
485
+ "urls_absent",
486
+ "auth_material_absent",
487
+ "runtime_state_absent",
488
+ "hidden_material_absent",
489
+ "raw_logs_absent",
490
+ }
491
+ if set(self.public_hygiene_checks) != required_hygiene:
492
+ raise ValueError("closure evidence must declare all public hygiene checks")
493
+ if not all(self.public_hygiene_checks.values()):
494
+ raise ValueError("closure evidence public hygiene checks must pass")
495
+ if not self.unresolved_live_dependencies:
496
+ raise ValueError("closure evidence must retain live dependency boundaries")
497
+ if not self.live_denial_diagnostic_codes:
498
+ raise ValueError("closure evidence must cite live denial diagnostics")
499
+ if not self.live_denial_test_coverage:
500
+ raise ValueError("closure evidence must cite denial regression coverage")
501
+ if not self.treatment_compiled_harness_hashes:
502
+ raise ValueError("closure evidence must include Spec 07E harness hashes")
503
+ object.__setattr__(self, "counts", _freeze_eval_suite_mapping(self.counts))
504
+ object.__setattr__(
505
+ self,
506
+ "treatment_compiled_harness_hashes",
507
+ _freeze_eval_suite_mapping(self.treatment_compiled_harness_hashes),
508
+ )
509
+ object.__setattr__(
510
+ self,
511
+ "public_hygiene_checks",
512
+ _freeze_eval_suite_mapping(self.public_hygiene_checks),
513
+ )
514
+ return self
515
+
516
+
517
+ class EvalModelPricingMetadata(EvalSuiteContractModel):
518
+ """Public numeric model pricing metadata with labels kept separate."""
519
+
520
+ input_cost_per_million_tokens: StrictFloat = Field(ge=0.0)
521
+ output_cost_per_million_tokens: StrictFloat = Field(ge=0.0)
522
+ cached_input_cost_per_million_tokens: StrictFloat = Field(ge=0.0)
523
+ currency_label: StrictStr
524
+ source_label: StrictStr
525
+
526
+
527
+ class EvalModelRateLimitMetadata(EvalSuiteContractModel):
528
+ """Public numeric model rate-limit metadata with labels kept separate."""
529
+
530
+ request_rate_per_window: StrictInt = Field(ge=0)
531
+ token_rate_per_window: StrictInt = Field(ge=0)
532
+ concurrent_request_limit: StrictInt = Field(ge=0)
533
+ window_seconds: StrictInt = Field(ge=0)
534
+ source_label: StrictStr
535
+
536
+
537
+ class EvalModelManifest(EvalSuiteContractModel):
538
+ """Backend-neutral campaign model manifest without endpoints or secrets."""
539
+
540
+ schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
541
+ model_manifest_id: StrictStr
542
+ model_profile_id: StrictStr
543
+ model_profile_hash: StrictStr
544
+ provider_label: StrictStr
545
+ model_or_artifact_id: StrictStr
546
+ release_or_snapshot: StrictStr
547
+ serving_protocol: StrictStr
548
+ endpoint_class: EvalCampaignKind
549
+ temperature: StrictFloat = Field(ge=0.0, le=2.0)
550
+ top_p: StrictFloat = Field(gt=0.0, le=1.0)
551
+ max_prompt_tokens: StrictInt = Field(gt=0)
552
+ max_completion_tokens: StrictInt = Field(gt=0)
553
+ max_total_tokens: StrictInt = Field(gt=0)
554
+ tool_calling_mode: StrictStr
555
+ parser_id: StrictStr
556
+ seed_policy: StrictStr
557
+ context_window_tokens: StrictInt = Field(gt=0)
558
+ reasoning_controls: Mapping[StrictStr, StrictStr] = Field(default_factory=dict)
559
+ public_pricing: EvalModelPricingMetadata
560
+ public_rate_limits: EvalModelRateLimitMetadata
561
+ public_pricing_snapshot: Mapping[StrictStr, StrictStr] = Field(default_factory=dict)
562
+ public_rate_limit_snapshot: Mapping[StrictStr, StrictStr] = Field(
563
+ default_factory=dict
564
+ )
565
+ local_serving_snapshot: Mapping[StrictStr, StrictStr] = Field(default_factory=dict)
566
+ model_manifest_hash_kind: StrictStr = EVAL_SUITE_MODEL_MANIFEST_HASH_KIND
567
+ model_manifest_hash: StrictStr
568
+
569
+ @model_validator(mode="after")
570
+ def _model_manifest_valid(self) -> EvalModelManifest:
571
+ if self.schema_version != EVAL_SUITE_SCHEMA_VERSION:
572
+ raise ValueError("unsupported eval-suite model schema_version")
573
+ if self.max_total_tokens != self.max_prompt_tokens + self.max_completion_tokens:
574
+ raise ValueError(
575
+ "max_total_tokens must equal prompt plus completion tokens"
576
+ )
577
+ if self.context_window_tokens < self.max_total_tokens:
578
+ raise ValueError("context_window_tokens must cover max_total_tokens")
579
+ object.__setattr__(
580
+ self,
581
+ "reasoning_controls",
582
+ _freeze_eval_suite_mapping(self.reasoning_controls),
583
+ )
584
+ object.__setattr__(
585
+ self,
586
+ "public_pricing_snapshot",
587
+ _freeze_eval_suite_mapping(self.public_pricing_snapshot),
588
+ )
589
+ object.__setattr__(
590
+ self,
591
+ "public_rate_limit_snapshot",
592
+ _freeze_eval_suite_mapping(self.public_rate_limit_snapshot),
593
+ )
594
+ object.__setattr__(
595
+ self,
596
+ "local_serving_snapshot",
597
+ _freeze_eval_suite_mapping(self.local_serving_snapshot),
598
+ )
599
+ if self.model_manifest_hash_kind != EVAL_SUITE_MODEL_MANIFEST_HASH_KIND:
600
+ raise ValueError("unsupported eval-suite model hash kind")
601
+ _validate_sha256(self.model_profile_hash)
602
+ _validate_sha256(self.model_manifest_hash)
603
+ expected = calculate_eval_model_manifest_hash(self)
604
+ if self.model_manifest_hash != expected:
605
+ raise ValueError("model_manifest_hash does not match manifest payload")
606
+ return self
607
+
608
+
609
+ class EvalDifficultyMetadata(EvalSuiteContractModel):
610
+ """Public fixture difficulty metadata."""
611
+
612
+ level: EvalDifficultyLevel
613
+ rationale: StrictStr
614
+ estimated_minutes: StrictInt = Field(gt=0)
615
+
616
+
617
+ class EvalVisibleCheck(EvalSuiteContractModel):
618
+ """Runner-visible check descriptor."""
619
+
620
+ check_id: StrictStr
621
+ summary: StrictStr
622
+ command: StrictStr | None = None
623
+
624
+ @model_validator(mode="after")
625
+ def _visible_check_valid(self) -> EvalVisibleCheck:
626
+ if self.command is not None:
627
+ _validate_public_command(self.command)
628
+ return self
629
+
630
+
631
+ class EvalHiddenCheck(EvalSuiteContractModel):
632
+ """Scorer-only hidden check descriptor."""
633
+
634
+ check_id: StrictStr
635
+ summary: StrictStr
636
+ scorer_rubric: StrictStr
637
+
638
+
639
+ class EvalExpectedMutationPolicy(EvalSuiteContractModel):
640
+ """Expected workspace mutation policy for scorer use."""
641
+
642
+ mutation_kind: EvalExpectedMutationKind
643
+ allowed_paths: tuple[StrictStr, ...] = Field(default_factory=tuple)
644
+ forbidden_paths: tuple[StrictStr, ...] = Field(default_factory=tuple)
645
+ summary: StrictStr
646
+
647
+ @model_validator(mode="after")
648
+ def _mutation_policy_valid(self) -> EvalExpectedMutationPolicy:
649
+ for path in self.allowed_paths + self.forbidden_paths:
650
+ _validate_relative_path(path)
651
+ if (
652
+ self.mutation_kind == EvalExpectedMutationKind.NO_SOURCE_CHANGE
653
+ and self.allowed_paths
654
+ ):
655
+ raise ValueError("no-source-change policies cannot allow mutation paths")
656
+ return self
657
+
658
+
659
+ class EvalTaskFixture(EvalSuiteContractModel):
660
+ """Immutable scorer-owned fixture contract."""
661
+
662
+ schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
663
+ fixture_id: StrictStr
664
+ category: EvalTaskCategory
665
+ difficulty: EvalDifficultyMetadata
666
+ visible_prompt: StrictStr
667
+ visible_acceptance_criteria: tuple[StrictStr, ...]
668
+ file_allowlist: tuple[StrictStr, ...]
669
+ expected_mutation_policy: EvalExpectedMutationPolicy
670
+ file_manifest_hashes: tuple[EvalHashRecord, ...] = Field(default_factory=tuple)
671
+ visible_checks: tuple[EvalVisibleCheck, ...]
672
+ hidden_checks: tuple[EvalHiddenCheck, ...]
673
+ expected_final_outcome: EvalTrialOutcome
674
+ fixture_hash_kind: StrictStr = EVAL_SUITE_FIXTURE_HASH_KIND
675
+ fixture_hash: StrictStr
676
+
677
+ @model_validator(mode="after")
678
+ def _fixture_valid(self) -> EvalTaskFixture:
679
+ if self.schema_version != EVAL_SUITE_SCHEMA_VERSION:
680
+ raise ValueError("unsupported eval-suite fixture schema_version")
681
+ if not self.visible_acceptance_criteria:
682
+ raise ValueError("fixtures must include visible acceptance criteria")
683
+ if not self.visible_checks:
684
+ raise ValueError("fixtures must include visible checks")
685
+ if not self.hidden_checks:
686
+ raise ValueError("fixtures must include scorer-only hidden checks")
687
+ for path in self.file_allowlist:
688
+ _validate_relative_path(path)
689
+ if self.fixture_hash_kind != EVAL_SUITE_FIXTURE_HASH_KIND:
690
+ raise ValueError("unsupported fixture hash kind")
691
+ _validate_sha256(self.fixture_hash)
692
+ expected = calculate_eval_task_fixture_hash(self)
693
+ if self.fixture_hash != expected:
694
+ raise ValueError("fixture_hash does not match fixture payload")
695
+ return self
696
+
697
+
698
+ class EvalRunnerTaskProjection(EvalSuiteContractModel):
699
+ """Runner-facing task projection without scorer-only fixture material."""
700
+
701
+ schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
702
+ fixture_id: StrictStr
703
+ category: EvalTaskCategory
704
+ difficulty: EvalDifficultyMetadata
705
+ visible_prompt: StrictStr
706
+ visible_acceptance_criteria: tuple[StrictStr, ...]
707
+ file_allowlist: tuple[StrictStr, ...]
708
+ visible_checks: tuple[EvalVisibleCheck, ...]
709
+
710
+
711
+ class EvalRunnerAcceptanceProjection(EvalSuiteContractModel):
712
+ """Runner-facing acceptance projection with visible criteria and checks only."""
713
+
714
+ schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
715
+ fixture_id: StrictStr
716
+ visible_acceptance_criteria: tuple[StrictStr, ...]
717
+ visible_checks: tuple[EvalVisibleCheck, ...]
718
+
719
+
720
+ class EvalRunnerContextProjection(EvalSuiteContractModel):
721
+ """Runner-facing context projection with only visible fixture context."""
722
+
723
+ schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
724
+ fixture_id: StrictStr
725
+ category: EvalTaskCategory
726
+ difficulty: EvalDifficultyMetadata
727
+ visible_prompt: StrictStr
728
+ visible_acceptance_criteria: tuple[StrictStr, ...]
729
+ file_allowlist: tuple[StrictStr, ...]
730
+ visible_checks: tuple[EvalVisibleCheck, ...]
731
+
732
+
733
+ class EvalPublicArtifactProjection(EvalSuiteContractModel):
734
+ """Public artifact projection that excludes scorer-only fixture answers."""
735
+
736
+ schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
737
+ fixture_id: StrictStr
738
+ category: EvalTaskCategory
739
+ visible_acceptance_criteria: tuple[StrictStr, ...]
740
+ file_allowlist: tuple[StrictStr, ...]
741
+ visible_checks: tuple[EvalVisibleCheck, ...]
742
+
743
+
744
+ class EvalFixturePackSummary(EvalSuiteContractModel):
745
+ """Hashable public summary of a fixture pack."""
746
+
747
+ schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
748
+ fixture_pack_id: StrictStr
749
+ fixture_ids: tuple[StrictStr, ...]
750
+ category_counts: Mapping[EvalTaskCategory, StrictInt]
751
+ fixture_hashes: tuple[EvalHashRecord, ...]
752
+ pack_summary: StrictStr
753
+ fixture_pack_hash_kind: StrictStr = EVAL_SUITE_FIXTURE_PACK_HASH_KIND
754
+ fixture_pack_hash: StrictStr
755
+
756
+ @model_validator(mode="after")
757
+ def _fixture_pack_valid(self) -> EvalFixturePackSummary:
758
+ if self.schema_version != EVAL_SUITE_SCHEMA_VERSION:
759
+ raise ValueError("unsupported eval-suite fixture pack schema_version")
760
+ if len(set(self.fixture_ids)) != len(self.fixture_ids):
761
+ raise ValueError("fixture pack fixture_ids must be unique")
762
+ if len(self.fixture_hashes) != len(self.fixture_ids):
763
+ raise ValueError("fixture pack hashes must match fixture_ids")
764
+ object.__setattr__(
765
+ self,
766
+ "category_counts",
767
+ _freeze_eval_suite_mapping(self.category_counts),
768
+ )
769
+ if self.fixture_pack_hash_kind != EVAL_SUITE_FIXTURE_PACK_HASH_KIND:
770
+ raise ValueError("unsupported fixture pack hash kind")
771
+ _validate_sha256(self.fixture_pack_hash)
772
+ expected = calculate_eval_fixture_pack_hash(self)
773
+ if self.fixture_pack_hash != expected:
774
+ raise ValueError("fixture_pack_hash does not match summary payload")
775
+ return self
776
+
777
+
778
+ class EvalCapabilityAuditSummary(EvalSuiteContractModel):
779
+ """Scorer input summary of capability-envelope enforcement."""
780
+
781
+ capability_violation: StrictBool
782
+ denied_capability_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
783
+ summary: StrictStr
784
+
785
+
786
+ class EvalCheckResult(EvalSuiteContractModel):
787
+ """Visible or hidden check result consumed by the scorer."""
788
+
789
+ check_id: StrictStr
790
+ passed: StrictBool
791
+ diagnostic: StrictStr | None = None
792
+
793
+
794
+ class EvalScorerInput(EvalSuiteContractModel):
795
+ """Deterministic scorer input contract without live runner behavior."""
796
+
797
+ schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
798
+ trial_id: StrictStr
799
+ fixture_id: StrictStr
800
+ fixture_hash: StrictStr
801
+ final_workspace_hash: StrictStr | None = None
802
+ path_limited_workspace_hashes: tuple[EvalHashRecord, ...] = Field(
803
+ default_factory=tuple
804
+ )
805
+ public_artifact_hashes: tuple[EvalHashRecord, ...] = Field(default_factory=tuple)
806
+ required_public_artifact_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
807
+ provided_public_artifact_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
808
+ malformed_artifact_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
809
+ stage_terminal_results: tuple[StrictStr, ...]
810
+ capability_audit: EvalCapabilityAuditSummary
811
+ visible_check_results: tuple[EvalCheckResult, ...]
812
+ hidden_check_results: tuple[EvalCheckResult, ...]
813
+ claimed_mutation_present: StrictBool = True
814
+ unauthorized_mutation: StrictBool = False
815
+ checker_public_evidence_valid: StrictBool = True
816
+ runtime_failure: StrictBool = False
817
+ provider_failure: StrictBool = False
818
+ invalid_trial_explanation: StrictStr | None = None
819
+ scorer_input_hash_kind: StrictStr = EVAL_SUITE_SCORER_INPUT_HASH_KIND
820
+ scorer_input_hash: StrictStr
821
+
822
+ @model_validator(mode="after")
823
+ def _scorer_input_valid(self) -> EvalScorerInput:
824
+ if self.schema_version != EVAL_SUITE_SCHEMA_VERSION:
825
+ raise ValueError("unsupported eval-suite scorer input schema_version")
826
+ _validate_sha256(self.fixture_hash)
827
+ if self.final_workspace_hash is not None:
828
+ _validate_sha256(self.final_workspace_hash)
829
+ _validate_public_artifact_ids(self.required_public_artifact_ids)
830
+ _validate_public_artifact_ids(self.provided_public_artifact_ids)
831
+ _validate_public_artifact_ids(self.malformed_artifact_ids)
832
+ if len(set(self.required_public_artifact_ids)) != len(
833
+ self.required_public_artifact_ids
834
+ ):
835
+ raise ValueError("required public artifact IDs must be unique")
836
+ if len(set(self.provided_public_artifact_ids)) != len(
837
+ self.provided_public_artifact_ids
838
+ ):
839
+ raise ValueError("provided public artifact IDs must be unique")
840
+ if len(set(self.malformed_artifact_ids)) != len(self.malformed_artifact_ids):
841
+ raise ValueError("malformed public artifact IDs must be unique")
842
+ if self.scorer_input_hash_kind != EVAL_SUITE_SCORER_INPUT_HASH_KIND:
843
+ raise ValueError("unsupported scorer input hash kind")
844
+ _validate_sha256(self.scorer_input_hash)
845
+ expected = calculate_eval_scorer_input_hash(self)
846
+ if self.scorer_input_hash != expected:
847
+ raise ValueError("scorer_input_hash does not match input payload")
848
+ return self
849
+
850
+
851
+ class EvalScorerResult(EvalSuiteContractModel):
852
+ """Deterministic scorer result contract."""
853
+
854
+ schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
855
+ trial_id: StrictStr
856
+ fixture_id: StrictStr
857
+ final_outcome: EvalTrialOutcome
858
+ primary_success: StrictBool
859
+ false_closure: StrictBool
860
+ false_success: StrictBool
861
+ correctly_blocked: StrictBool
862
+ capability_violation: StrictBool
863
+ artifact_complete: StrictBool
864
+ missing_artifact_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
865
+ malformed_artifact_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
866
+ failure_labels: tuple[EvalFailureTaxonomyLabel, ...] = Field(default_factory=tuple)
867
+ public_diagnostics: tuple[StrictStr, ...] = Field(default_factory=tuple)
868
+ scorer_only_diagnostics: tuple[StrictStr, ...] = Field(default_factory=tuple)
869
+ invalid_trial_explanation: StrictStr | None = None
870
+ scorer_version: StrictStr = EVAL_SUITE_DEFAULT_SCORER_VERSION
871
+ result_hash_kind: StrictStr = EVAL_SUITE_SCORER_RESULT_HASH_KIND
872
+ result_hash: StrictStr
873
+
874
+ @model_validator(mode="after")
875
+ def _scorer_result_valid(self) -> EvalScorerResult:
876
+ if self.schema_version != EVAL_SUITE_SCHEMA_VERSION:
877
+ raise ValueError("unsupported eval-suite scorer result schema_version")
878
+ if (
879
+ self.final_outcome == EvalTrialOutcome.INVALID_TRIAL
880
+ and not self.invalid_trial_explanation
881
+ ):
882
+ raise ValueError("invalid_trial outcomes require an explanation")
883
+ if (
884
+ self.final_outcome != EvalTrialOutcome.INVALID_TRIAL
885
+ and self.invalid_trial_explanation
886
+ ):
887
+ raise ValueError(
888
+ "invalid_trial_explanation is only valid for invalid_trial"
889
+ )
890
+ if self.final_outcome == EvalTrialOutcome.VALID_COMPLETION:
891
+ valid_completion_has_failure_evidence = (
892
+ self.false_closure
893
+ or self.false_success
894
+ or self.correctly_blocked
895
+ or self.capability_violation
896
+ or not self.artifact_complete
897
+ or self.missing_artifact_ids
898
+ or self.malformed_artifact_ids
899
+ or self.failure_labels
900
+ )
901
+ if self.primary_success and valid_completion_has_failure_evidence:
902
+ raise ValueError(
903
+ "valid_completion cannot include failure evidence or "
904
+ "inconsistent success flags"
905
+ )
906
+ if not self.primary_success and not valid_completion_has_failure_evidence:
907
+ raise ValueError(
908
+ "non-primary valid_completion requires failure evidence"
909
+ )
910
+ if self.false_closure:
911
+ raise ValueError("valid_completion cannot include false_closure")
912
+ if (
913
+ self.false_closure
914
+ and EvalFailureTaxonomyLabel.CAPABILITY_VIOLATION in self.failure_labels
915
+ ):
916
+ if not self.capability_violation:
917
+ raise ValueError(
918
+ "capability violation label requires capability_violation"
919
+ )
920
+ if self.result_hash_kind != EVAL_SUITE_SCORER_RESULT_HASH_KIND:
921
+ raise ValueError("unsupported scorer result hash kind")
922
+ _validate_sha256(self.result_hash)
923
+ expected = calculate_eval_scorer_result_hash(self)
924
+ if self.result_hash != expected:
925
+ raise ValueError("result_hash does not match result payload")
926
+ return self
927
+
928
+
929
+ class EvalCampaignManifest(EvalSuiteContractModel):
930
+ """Campaign contract tying modes, model, workflow, fixtures, and scorer."""
931
+
932
+ schema_version: StrictInt = EVAL_SUITE_SCHEMA_VERSION
933
+ campaign_id: StrictStr
934
+ campaign_kind: EvalCampaignKind
935
+ execution_mode: EvalSuiteExecutionMode
936
+ pi_eval_mode_id: StrictStr
937
+ millforge_eval_mode_id: StrictStr
938
+ model_manifest_ref: StrictStr
939
+ model_manifest_hash: StrictStr
940
+ workflow_graph_hash: StrictStr
941
+ fixture_pack_hash: StrictStr
942
+ scorer_version: StrictStr
943
+ created_at: StrictStr
944
+ budget_policy_ref: EvalBudgetPolicyReference
945
+ live_execution_admitted: StrictBool
946
+ live_denial_diagnostics: tuple[EvalLiveDenialDiagnostic, ...]
947
+ campaign_manifest_hash_kind: StrictStr = EVAL_SUITE_CAMPAIGN_MANIFEST_HASH_KIND
948
+ campaign_manifest_hash: StrictStr
949
+
950
+ @model_validator(mode="after")
951
+ def _campaign_manifest_valid(self) -> EvalCampaignManifest:
952
+ if self.schema_version != EVAL_SUITE_SCHEMA_VERSION:
953
+ raise ValueError("unsupported eval-suite campaign schema_version")
954
+ if not _UTC_TIMESTAMP_RE.fullmatch(self.created_at):
955
+ raise ValueError("campaign created_at must be a UTC timestamp")
956
+ for digest in (
957
+ self.model_manifest_hash,
958
+ self.workflow_graph_hash,
959
+ self.fixture_pack_hash,
960
+ self.campaign_manifest_hash,
961
+ ):
962
+ _validate_sha256(digest)
963
+ if self.execution_mode == EvalSuiteExecutionMode.OFFLINE_FAKE:
964
+ if self.live_execution_admitted:
965
+ raise ValueError(
966
+ "offline eval-suite campaigns cannot admit live execution"
967
+ )
968
+ if not self.live_denial_diagnostics:
969
+ raise ValueError(
970
+ "offline campaigns must include live-denial diagnostics"
971
+ )
972
+ if self.execution_mode == EvalSuiteExecutionMode.LIVE_RUNNER:
973
+ raise ValueError(
974
+ "08A eval-suite campaign contracts do not admit live execution"
975
+ )
976
+ if self.campaign_manifest_hash_kind != EVAL_SUITE_CAMPAIGN_MANIFEST_HASH_KIND:
977
+ raise ValueError("unsupported campaign manifest hash kind")
978
+ expected = calculate_eval_campaign_manifest_hash(self)
979
+ if self.campaign_manifest_hash != expected:
980
+ raise ValueError("campaign_manifest_hash does not match manifest payload")
981
+ return self
982
+
983
+
984
+ def eval_model_manifest_from_profile(
985
+ profile: EvalModelProfile | None = None,
986
+ *,
987
+ model_manifest_id: str = "eval.08a.model.backend_neutral.default.v1",
988
+ ) -> EvalModelManifest:
989
+ """Build a campaign-grade manifest from the shared eval model profile."""
990
+ profile = profile or default_eval_model_profile()
991
+ manifest = EvalModelManifest.model_construct(
992
+ schema_version=EVAL_SUITE_SCHEMA_VERSION,
993
+ model_manifest_id=model_manifest_id,
994
+ model_profile_id=profile.profile_id,
995
+ model_profile_hash=profile.model_profile_hash,
996
+ provider_label=profile.provider_label,
997
+ model_or_artifact_id=profile.model_label,
998
+ release_or_snapshot="static-backend-neutral-profile",
999
+ serving_protocol=profile.serving_protocol,
1000
+ endpoint_class=EvalCampaignKind.LOCAL_OPENAI_COMPATIBLE,
1001
+ temperature=profile.temperature,
1002
+ top_p=profile.top_p,
1003
+ max_prompt_tokens=profile.max_prompt_tokens,
1004
+ max_completion_tokens=profile.max_completion_tokens,
1005
+ max_total_tokens=profile.max_total_tokens,
1006
+ tool_calling_mode=profile.tool_calling_mode,
1007
+ parser_id=profile.parser_id,
1008
+ seed_policy="no_seed",
1009
+ context_window_tokens=profile.max_total_tokens,
1010
+ reasoning_controls={"reasoning_effort": profile.reasoning_effort},
1011
+ public_pricing=EvalModelPricingMetadata(
1012
+ input_cost_per_million_tokens=0.0,
1013
+ output_cost_per_million_tokens=0.0,
1014
+ cached_input_cost_per_million_tokens=0.0,
1015
+ currency_label="none",
1016
+ source_label="static_descriptor",
1017
+ ),
1018
+ public_rate_limits=EvalModelRateLimitMetadata(
1019
+ request_rate_per_window=0,
1020
+ token_rate_per_window=0,
1021
+ concurrent_request_limit=0,
1022
+ window_seconds=0,
1023
+ source_label="not_applicable",
1024
+ ),
1025
+ public_pricing_snapshot=dict(profile.cost_accounting),
1026
+ public_rate_limit_snapshot={"rate_limit_source": "not_applicable"},
1027
+ local_serving_snapshot={"serving_snapshot": "not_applicable"},
1028
+ model_manifest_hash_kind=EVAL_SUITE_MODEL_MANIFEST_HASH_KIND,
1029
+ model_manifest_hash="0" * 64,
1030
+ )
1031
+ payload = manifest.model_dump(mode="json")
1032
+ payload["model_manifest_hash"] = calculate_eval_model_manifest_hash(manifest)
1033
+ return EvalModelManifest.model_validate(payload)
1034
+
1035
+
1036
+ def default_eval_suite_campaign_manifest(
1037
+ *,
1038
+ model_manifest: EvalModelManifest | None = None,
1039
+ fixture_pack_hash: str | None = None,
1040
+ created_at: str = EVAL_SUITE_DEFAULT_CAMPAIGN_CREATED_AT,
1041
+ ) -> EvalCampaignManifest:
1042
+ """Return the default 08A offline-only campaign manifest."""
1043
+ model_manifest = model_manifest or eval_model_manifest_from_profile()
1044
+ if fixture_pack_hash is None:
1045
+ fixture_pack_hash = load_eval_fixture_pack_summary().fixture_pack_hash
1046
+ manifest = EvalCampaignManifest.model_construct(
1047
+ schema_version=EVAL_SUITE_SCHEMA_VERSION,
1048
+ campaign_id=EVAL_SUITE_DEFAULT_CAMPAIGN_ID,
1049
+ campaign_kind=EvalCampaignKind.LOCAL_OPENAI_COMPATIBLE,
1050
+ execution_mode=EvalSuiteExecutionMode.OFFLINE_FAKE,
1051
+ pi_eval_mode_id=EVAL_SMALL_PI_MODE_ID,
1052
+ millforge_eval_mode_id=EVAL_SMALL_MILLFORGE_MODE_ID,
1053
+ model_manifest_ref=model_manifest.model_manifest_id,
1054
+ model_manifest_hash=model_manifest.model_manifest_hash,
1055
+ workflow_graph_hash=compact_eval_workflow_snapshot()["graph_sha256"],
1056
+ fixture_pack_hash=fixture_pack_hash,
1057
+ scorer_version=EVAL_SUITE_DEFAULT_SCORER_VERSION,
1058
+ created_at=created_at,
1059
+ budget_policy_ref=EvalBudgetPolicyReference(
1060
+ policy_id="eval.08a.default.offline_budget.v1",
1061
+ summary="Static offline fixture and scorer contract budget reference.",
1062
+ ),
1063
+ live_execution_admitted=False,
1064
+ live_denial_diagnostics=(
1065
+ EvalLiveDenialDiagnostic(
1066
+ diagnostic_code="MF-EVAL-SUITE-001",
1067
+ summary="08A default campaign is offline-only and denies live execution.",
1068
+ rule_id="eval_suite.default_campaign.offline_only",
1069
+ ),
1070
+ ),
1071
+ campaign_manifest_hash_kind=EVAL_SUITE_CAMPAIGN_MANIFEST_HASH_KIND,
1072
+ campaign_manifest_hash="0" * 64,
1073
+ )
1074
+ payload = manifest.model_dump(mode="json")
1075
+ payload["campaign_manifest_hash"] = calculate_eval_campaign_manifest_hash(manifest)
1076
+ return EvalCampaignManifest.model_validate(payload)
1077
+
1078
+
1079
+ def configure_offline_fake_eval_campaign(
1080
+ *,
1081
+ output_root: str | Path,
1082
+ budget_policy: Any,
1083
+ fixture_pack_id: str = EVAL_SUITE_DEFAULT_FIXTURE_PACK_ID,
1084
+ fixture_ids: tuple[str, ...] | None = None,
1085
+ deterministic_seed: int = 0,
1086
+ trial_count_per_fixture_per_arm: int = 1,
1087
+ fake_runner_script: Any | None = None,
1088
+ allow_live_execution: bool = False,
1089
+ allow_live_model_call: bool = False,
1090
+ allow_pi_execution: bool = False,
1091
+ allow_millforge_harness_execution: bool = False,
1092
+ created_at: str = EVAL_SUITE_DEFAULT_CAMPAIGN_CREATED_AT,
1093
+ ) -> EvalOfflineDryCampaignPlan:
1094
+ """Preflight a deterministic offline fake campaign without writing records."""
1095
+ _reject_offline_dry_live_flags(
1096
+ allow_live_execution=allow_live_execution,
1097
+ allow_live_model_call=allow_live_model_call,
1098
+ allow_pi_execution=allow_pi_execution,
1099
+ allow_millforge_harness_execution=allow_millforge_harness_execution,
1100
+ )
1101
+ if fixture_pack_id != EVAL_SUITE_DEFAULT_FIXTURE_PACK_ID:
1102
+ raise ValueError(
1103
+ _offline_dry_diagnostic(
1104
+ EvalOfflineDryCampaignDiagnosticCode.FIXTURE_PACK_UNAVAILABLE,
1105
+ "eval_suite.dry_campaign.fixture_pack",
1106
+ "Only the installed default offline fixture pack is available.",
1107
+ ).summary
1108
+ )
1109
+ if trial_count_per_fixture_per_arm <= 0:
1110
+ raise ValueError(
1111
+ _offline_dry_diagnostic(
1112
+ EvalOfflineDryCampaignDiagnosticCode.INVALID_TRIAL_COUNT,
1113
+ "eval_suite.dry_campaign.trial_count",
1114
+ "Dry campaigns require a positive trial count per fixture per arm.",
1115
+ ).summary
1116
+ )
1117
+ _validate_offline_dry_output_root(output_root)
1118
+
1119
+ fixture_pack = load_eval_fixture_pack_summary()
1120
+ fixtures_by_id = {
1121
+ fixture.fixture_id: fixture for fixture in load_eval_task_fixtures()
1122
+ }
1123
+ selected_ids = fixture_ids or fixture_pack.fixture_ids
1124
+ missing_ids = tuple(
1125
+ fixture_id for fixture_id in selected_ids if fixture_id not in fixtures_by_id
1126
+ )
1127
+ if missing_ids:
1128
+ raise ValueError("unknown dry-campaign fixture ID")
1129
+ selected_fixtures = tuple(fixtures_by_id[fixture_id] for fixture_id in selected_ids)
1130
+ expanded_fixtures = tuple(
1131
+ fixture
1132
+ for fixture in selected_fixtures
1133
+ for _ in range(trial_count_per_fixture_per_arm)
1134
+ )
1135
+ trial_indexes = tuple(range(len(expanded_fixtures)))
1136
+
1137
+ campaign_manifest = default_eval_suite_campaign_manifest(
1138
+ fixture_pack_hash=fixture_pack.fixture_pack_hash,
1139
+ created_at=created_at,
1140
+ )
1141
+ from millforge.eval_reports import (
1142
+ EvalBudgetUsageEstimate,
1143
+ EvalReportBudgetPolicy,
1144
+ validate_eval_budget_policy,
1145
+ )
1146
+ from millforge.eval_trials import (
1147
+ EvalTrialArmId,
1148
+ canonical_eval_trial_store_manifest_bytes,
1149
+ plan_paired_eval_trials,
1150
+ )
1151
+
1152
+ if budget_policy is None:
1153
+ raise ValueError(
1154
+ _offline_dry_diagnostic(
1155
+ EvalOfflineDryCampaignDiagnosticCode.MISSING_BUDGET_POLICY,
1156
+ "eval_suite.dry_campaign.budget_policy",
1157
+ "Budget policy metadata is required for dry-campaign preflight.",
1158
+ ).summary
1159
+ )
1160
+ budget_policy = EvalReportBudgetPolicy.model_validate(budget_policy)
1161
+ budget_result = validate_eval_budget_policy(
1162
+ budget_policy,
1163
+ campaign_manifest=campaign_manifest,
1164
+ usage=EvalBudgetUsageEstimate(
1165
+ estimated_spend_usd=0.0,
1166
+ prompt_tokens=0,
1167
+ completion_tokens=0,
1168
+ model_calls=0,
1169
+ retries_per_trial=0,
1170
+ wall_clock_seconds=0,
1171
+ trial_count=len(expanded_fixtures) * 2,
1172
+ ),
1173
+ )
1174
+ if not budget_result.valid:
1175
+ raise ValueError(
1176
+ _offline_dry_diagnostic(
1177
+ EvalOfflineDryCampaignDiagnosticCode.INVALID_BUDGET_POLICY,
1178
+ "eval_suite.dry_campaign.budget_policy",
1179
+ budget_result.diagnostics[0].summary,
1180
+ ).summary
1181
+ )
1182
+
1183
+ script = fake_runner_script or _default_offline_fake_runner_script()
1184
+ plans = plan_paired_eval_trials(
1185
+ fixtures=expanded_fixtures,
1186
+ fake_runner_script=script,
1187
+ seed=deterministic_seed,
1188
+ campaign_manifest=campaign_manifest,
1189
+ trial_indexes=trial_indexes,
1190
+ created_at=created_at,
1191
+ )
1192
+ store_manifest = _offline_dry_store_manifest(plans[0])
1193
+ manifest_bytes = canonical_eval_trial_store_manifest_bytes(store_manifest)
1194
+ manifest_path = Path(output_root) / plans[0].campaign_store_root / "manifest.json"
1195
+ if manifest_path.exists() and manifest_path.read_bytes() != manifest_bytes:
1196
+ raise ValueError(
1197
+ _offline_dry_diagnostic(
1198
+ EvalOfflineDryCampaignDiagnosticCode.MANIFEST_CONFLICT,
1199
+ "eval_suite.dry_campaign.manifest_conflict",
1200
+ "Existing campaign manifest differs from dry-campaign preflight.",
1201
+ ).summary
1202
+ )
1203
+
1204
+ config = EvalOfflineDryCampaignConfig(
1205
+ fixture_pack_id=fixture_pack_id,
1206
+ fixture_ids=selected_ids,
1207
+ deterministic_seed=deterministic_seed,
1208
+ trial_count_per_fixture_per_arm=trial_count_per_fixture_per_arm,
1209
+ output_root_hash=hashlib.sha256(str(output_root).encode("utf-8")).hexdigest(),
1210
+ )
1211
+ return EvalOfflineDryCampaignPlan(
1212
+ config=config,
1213
+ campaign_manifest=campaign_manifest,
1214
+ fixture_pack_summary=fixture_pack,
1215
+ admitted_arm_ids=(
1216
+ EvalTrialArmId.EVAL_SMALL_PI.value,
1217
+ EvalTrialArmId.EVAL_SMALL_MILLFORGE.value,
1218
+ ),
1219
+ paired_plan_count=len(plans),
1220
+ planned_arm_trial_count=len(plans) * 2,
1221
+ trial_plan_hashes=tuple(plan.plan_hash for plan in plans),
1222
+ campaign_store_root=plans[0].campaign_store_root,
1223
+ manifest_relative_path=f"{plans[0].campaign_store_root}/manifest.json",
1224
+ budget_policy_ref=EvalBudgetPolicyReference(
1225
+ policy_id=budget_policy.policy_id,
1226
+ summary=budget_policy.summary,
1227
+ ),
1228
+ live_execution_admitted=False,
1229
+ diagnostics=(),
1230
+ )
1231
+
1232
+
1233
+ def run_offline_fake_eval_campaign(
1234
+ *,
1235
+ output_root: str | Path,
1236
+ budget_policy: Any,
1237
+ fixture_pack_id: str = EVAL_SUITE_DEFAULT_FIXTURE_PACK_ID,
1238
+ fixture_ids: tuple[str, ...] | None = None,
1239
+ deterministic_seed: int = 0,
1240
+ trial_count_per_fixture_per_arm: int = 1,
1241
+ fake_runner_script: Any | None = None,
1242
+ report_id: str | None = None,
1243
+ allow_live_execution: bool = False,
1244
+ allow_live_model_call: bool = False,
1245
+ allow_pi_execution: bool = False,
1246
+ allow_millforge_harness_execution: bool = False,
1247
+ created_at: str = EVAL_SUITE_DEFAULT_CAMPAIGN_CREATED_AT,
1248
+ ) -> EvalOfflineDryCampaignRunResult:
1249
+ """Run a deterministic offline fake campaign under a caller-selected root."""
1250
+ dry_plan = configure_offline_fake_eval_campaign(
1251
+ output_root=output_root,
1252
+ budget_policy=budget_policy,
1253
+ fixture_pack_id=fixture_pack_id,
1254
+ fixture_ids=fixture_ids,
1255
+ deterministic_seed=deterministic_seed,
1256
+ trial_count_per_fixture_per_arm=trial_count_per_fixture_per_arm,
1257
+ fake_runner_script=fake_runner_script,
1258
+ allow_live_execution=allow_live_execution,
1259
+ allow_live_model_call=allow_live_model_call,
1260
+ allow_pi_execution=allow_pi_execution,
1261
+ allow_millforge_harness_execution=allow_millforge_harness_execution,
1262
+ created_at=created_at,
1263
+ )
1264
+
1265
+ from millforge.eval_reports import (
1266
+ EvalReportBudgetPolicy,
1267
+ build_eval_report_artifact_bytes,
1268
+ build_eval_report_payload,
1269
+ )
1270
+ from millforge.eval_trials import (
1271
+ append_eval_trial_record_to_campaign_store,
1272
+ plan_paired_eval_trials,
1273
+ resume_eval_trial_campaign_store,
1274
+ run_offline_fake_eval_trial,
1275
+ )
1276
+
1277
+ budget = EvalReportBudgetPolicy.model_validate(budget_policy)
1278
+ fixtures_by_id = {
1279
+ fixture.fixture_id: fixture for fixture in load_eval_task_fixtures()
1280
+ }
1281
+ selected_fixtures = tuple(
1282
+ fixtures_by_id[fixture_id] for fixture_id in dry_plan.config.fixture_ids
1283
+ )
1284
+ expanded_fixtures = tuple(
1285
+ fixture
1286
+ for fixture in selected_fixtures
1287
+ for _ in range(trial_count_per_fixture_per_arm)
1288
+ )
1289
+ plans = plan_paired_eval_trials(
1290
+ fixtures=expanded_fixtures,
1291
+ fake_runner_script=fake_runner_script or _default_offline_fake_runner_script(),
1292
+ seed=deterministic_seed,
1293
+ campaign_manifest=dry_plan.campaign_manifest,
1294
+ trial_indexes=tuple(range(len(expanded_fixtures))),
1295
+ created_at=created_at,
1296
+ )
1297
+ if tuple(plan.plan_hash for plan in plans) != dry_plan.trial_plan_hashes:
1298
+ raise ValueError("dry-campaign plan hashes changed after preflight")
1299
+
1300
+ generated_records = {}
1301
+ for plan in plans:
1302
+ fixture = fixtures_by_id[plan.fixture_instance.fixture_id]
1303
+ generated_records[plan.trial_id] = run_offline_fake_eval_trial(
1304
+ plan,
1305
+ fixture=fixture,
1306
+ ).trial_record
1307
+ existing_records = _read_offline_dry_campaign_record_summaries(
1308
+ output_root,
1309
+ plans[0],
1310
+ )
1311
+ _validate_offline_dry_record_summaries_match_plans(
1312
+ records=existing_records,
1313
+ plans=plans,
1314
+ generated_records=generated_records,
1315
+ )
1316
+ completed_ids = {record["trial_id"] for record in existing_records}
1317
+ appended_trial_ids: list[str] = []
1318
+ for plan in plans:
1319
+ if plan.trial_id in completed_ids:
1320
+ continue
1321
+ append_eval_trial_record_to_campaign_store(
1322
+ output_root,
1323
+ plan=plan,
1324
+ record=generated_records[plan.trial_id],
1325
+ plans=plans,
1326
+ )
1327
+ appended_trial_ids.append(plan.trial_id)
1328
+ completed_ids.add(plan.trial_id)
1329
+
1330
+ if appended_trial_ids:
1331
+ _offline_dry_output_path(
1332
+ output_root,
1333
+ f"{plans[0].campaign_store_root}/index.json",
1334
+ ).unlink(missing_ok=True)
1335
+ final_resume = resume_eval_trial_campaign_store(
1336
+ output_root,
1337
+ plan=plans[0],
1338
+ plans=plans,
1339
+ )
1340
+ if final_resume.diagnostics:
1341
+ raise ValueError(final_resume.diagnostics[0].summary)
1342
+ if final_resume.resume_index is None:
1343
+ raise ValueError("dry-campaign resume index was not written")
1344
+
1345
+ records = tuple(generated_records[plan.trial_id] for plan in plans)
1346
+ payload = build_eval_report_payload(
1347
+ report_id=report_id or f"{dry_plan.campaign_manifest.campaign_id}.report.v1",
1348
+ campaign_manifest=dry_plan.campaign_manifest,
1349
+ plans=plans,
1350
+ records=records,
1351
+ resume_index=final_resume.resume_index,
1352
+ budget_policy=budget,
1353
+ generated_at=created_at,
1354
+ )
1355
+ report_artifacts = build_eval_report_artifact_bytes(payload)
1356
+ report_relative_paths = {
1357
+ name: f"{dry_plan.campaign_store_root}/reports/{name}"
1358
+ for name in sorted(report_artifacts)
1359
+ }
1360
+ for name, data in report_artifacts.items():
1361
+ path = _offline_dry_output_path(output_root, report_relative_paths[name])
1362
+ path.parent.mkdir(parents=True, exist_ok=True)
1363
+ path.write_bytes(data)
1364
+ closure_evidence = build_offline_dry_campaign_closure_evidence(
1365
+ run_plan=dry_plan,
1366
+ report_id=payload.report_id,
1367
+ completed_trial_ids=final_resume.completed_trial_ids,
1368
+ pending_trial_ids=final_resume.pending_trial_ids,
1369
+ trial_record_hashes=tuple(record.record_hash for record in records),
1370
+ resume_index_hash=final_resume.resume_index.resume_index_hash,
1371
+ report_hash=payload.report_hash,
1372
+ report_json_hash=hashlib.sha256(report_artifacts["report.json"]).hexdigest(),
1373
+ report_markdown_hash=hashlib.sha256(report_artifacts["report.md"]).hexdigest(),
1374
+ report_artifact_count=len(report_artifacts),
1375
+ treatment_compiled_harness_hashes=records[0].compiled_harness_hashes,
1376
+ )
1377
+ closure_evidence_relative_path = (
1378
+ f"{dry_plan.campaign_store_root}/closure_evidence.json"
1379
+ )
1380
+ _offline_dry_output_path(output_root, closure_evidence_relative_path).write_bytes(
1381
+ canonical_offline_dry_campaign_closure_evidence_bytes(closure_evidence)
1382
+ )
1383
+
1384
+ return EvalOfflineDryCampaignRunResult(
1385
+ dry_campaign_plan=dry_plan,
1386
+ report_id=payload.report_id,
1387
+ campaign_store_root=dry_plan.campaign_store_root,
1388
+ manifest_relative_path=dry_plan.manifest_relative_path,
1389
+ plan_relative_path=f"{dry_plan.campaign_store_root}/plan.json",
1390
+ trials_relative_path=f"{dry_plan.campaign_store_root}/trials.jsonl",
1391
+ index_relative_path=f"{dry_plan.campaign_store_root}/index.json",
1392
+ artifact_root_relative_path=f"{dry_plan.campaign_store_root}/artifacts",
1393
+ report_relative_paths=report_relative_paths,
1394
+ completed_trial_ids=final_resume.completed_trial_ids,
1395
+ pending_trial_ids=final_resume.pending_trial_ids,
1396
+ appended_trial_ids=tuple(appended_trial_ids),
1397
+ trial_plan_hashes=tuple(plan.plan_hash for plan in plans),
1398
+ trial_record_hashes=tuple(record.record_hash for record in records),
1399
+ resume_index_hash=final_resume.resume_index.resume_index_hash,
1400
+ report_hash=payload.report_hash,
1401
+ report_json_hash=hashlib.sha256(report_artifacts["report.json"]).hexdigest(),
1402
+ report_markdown_hash=hashlib.sha256(report_artifacts["report.md"]).hexdigest(),
1403
+ closure_evidence_relative_path=closure_evidence_relative_path,
1404
+ closure_evidence_hash=closure_evidence.closure_evidence_hash,
1405
+ live_execution_admitted=False,
1406
+ )
1407
+
1408
+
1409
+ def build_offline_dry_campaign_closure_evidence(
1410
+ *,
1411
+ run_plan: EvalOfflineDryCampaignPlan,
1412
+ report_id: str,
1413
+ completed_trial_ids: tuple[str, ...],
1414
+ pending_trial_ids: tuple[str, ...],
1415
+ trial_record_hashes: tuple[str, ...],
1416
+ resume_index_hash: str,
1417
+ report_hash: str,
1418
+ report_json_hash: str,
1419
+ report_markdown_hash: str,
1420
+ report_artifact_count: int,
1421
+ treatment_compiled_harness_hashes: Mapping[str, str],
1422
+ ) -> EvalOfflineDryCampaignClosureEvidence:
1423
+ """Return compact public closure evidence for a dry-campaign result."""
1424
+ evidence = EvalOfflineDryCampaignClosureEvidence.model_construct(
1425
+ schema_version=EVAL_SUITE_SCHEMA_VERSION,
1426
+ campaign_id=run_plan.campaign_manifest.campaign_id,
1427
+ report_id=report_id,
1428
+ fixture_pack_hash=run_plan.fixture_pack_summary.fixture_pack_hash,
1429
+ campaign_manifest_hash=run_plan.campaign_manifest.campaign_manifest_hash,
1430
+ model_manifest_hash=run_plan.campaign_manifest.model_manifest_hash,
1431
+ workflow_graph_hash=run_plan.campaign_manifest.workflow_graph_hash,
1432
+ trial_plan_hashes=run_plan.trial_plan_hashes,
1433
+ trial_record_hashes=trial_record_hashes,
1434
+ resume_index_hash=resume_index_hash,
1435
+ report_hash=report_hash,
1436
+ report_json_hash=report_json_hash,
1437
+ report_markdown_hash=report_markdown_hash,
1438
+ counts={
1439
+ "fixture_count": len(run_plan.config.fixture_ids),
1440
+ "paired_plan_count": run_plan.paired_plan_count,
1441
+ "arm_trial_count": run_plan.planned_arm_trial_count,
1442
+ "completed_trial_count": len(completed_trial_ids),
1443
+ "pending_trial_count": len(pending_trial_ids),
1444
+ "stored_trial_count": len(completed_trial_ids),
1445
+ "report_artifact_count": report_artifact_count,
1446
+ },
1447
+ unresolved_live_dependencies=(
1448
+ "pi_runtime",
1449
+ "millforge_live_harness_execution",
1450
+ "shared_model_backend_configuration",
1451
+ "fixture_workspace_lifecycle_reset",
1452
+ "resource_enforcement",
1453
+ ),
1454
+ live_denial_diagnostic_codes=(
1455
+ "pi_runtime_unavailable",
1456
+ "millforge_live_harness_unavailable",
1457
+ "shared_backend_configuration_missing",
1458
+ "fixture_workspace_lifecycle_unavailable",
1459
+ "resource_enforcement_unavailable",
1460
+ ),
1461
+ live_denial_test_coverage=(
1462
+ "tests/test_eval_reports.py::test_live_admission_returns_all_unresolved_dependency_diagnostics",
1463
+ "tests/test_eval_modes.py::test_live_admission_fails_closed_with_structured_deferred_dependencies",
1464
+ "tests/test_eval_trials.py::test_execution_result_rejects_live_admission",
1465
+ ),
1466
+ treatment_compiled_harness_hashes=dict(
1467
+ sorted(treatment_compiled_harness_hashes.items())
1468
+ ),
1469
+ public_hygiene_checks={
1470
+ "absolute_paths_absent": True,
1471
+ "home_paths_absent": True,
1472
+ "urls_absent": True,
1473
+ "auth_material_absent": True,
1474
+ "runtime_state_absent": True,
1475
+ "hidden_material_absent": True,
1476
+ "raw_logs_absent": True,
1477
+ },
1478
+ claim_boundary=(
1479
+ "Offline fake closure evidence proves deterministic public contract "
1480
+ "outputs only; live Pi-vs-Millforge remains denied."
1481
+ ),
1482
+ closure_evidence_hash_kind=EVAL_SUITE_CLOSURE_EVIDENCE_HASH_KIND,
1483
+ closure_evidence_hash="0" * 64,
1484
+ )
1485
+ payload = evidence.model_dump(mode="json")
1486
+ payload["closure_evidence_hash"] = (
1487
+ calculate_offline_dry_campaign_closure_evidence_hash(evidence)
1488
+ )
1489
+ return EvalOfflineDryCampaignClosureEvidence.model_validate(payload)
1490
+
1491
+
1492
+ def eval_runner_task_projection(fixture: EvalTaskFixture) -> EvalRunnerTaskProjection:
1493
+ """Return a runner-visible projection with scorer-only fields omitted."""
1494
+ return EvalRunnerTaskProjection(
1495
+ fixture_id=fixture.fixture_id,
1496
+ category=fixture.category,
1497
+ difficulty=fixture.difficulty,
1498
+ visible_prompt=fixture.visible_prompt,
1499
+ visible_acceptance_criteria=fixture.visible_acceptance_criteria,
1500
+ file_allowlist=fixture.file_allowlist,
1501
+ visible_checks=fixture.visible_checks,
1502
+ )
1503
+
1504
+
1505
+ def eval_runner_acceptance_projection(
1506
+ fixture: EvalTaskFixture,
1507
+ ) -> EvalRunnerAcceptanceProjection:
1508
+ """Return visible acceptance criteria and checks for runner consumption."""
1509
+ return EvalRunnerAcceptanceProjection(
1510
+ fixture_id=fixture.fixture_id,
1511
+ visible_acceptance_criteria=fixture.visible_acceptance_criteria,
1512
+ visible_checks=fixture.visible_checks,
1513
+ )
1514
+
1515
+
1516
+ def eval_runner_context_projection(
1517
+ fixture: EvalTaskFixture,
1518
+ ) -> EvalRunnerContextProjection:
1519
+ """Return visible fixture context for runner prompt assembly."""
1520
+ return EvalRunnerContextProjection(
1521
+ fixture_id=fixture.fixture_id,
1522
+ category=fixture.category,
1523
+ difficulty=fixture.difficulty,
1524
+ visible_prompt=fixture.visible_prompt,
1525
+ visible_acceptance_criteria=fixture.visible_acceptance_criteria,
1526
+ file_allowlist=fixture.file_allowlist,
1527
+ visible_checks=fixture.visible_checks,
1528
+ )
1529
+
1530
+
1531
+ def eval_public_artifact_projection(
1532
+ fixture: EvalTaskFixture,
1533
+ ) -> EvalPublicArtifactProjection:
1534
+ """Return public fixture metadata safe to serialize into trial artifacts."""
1535
+ return EvalPublicArtifactProjection(
1536
+ fixture_id=fixture.fixture_id,
1537
+ category=fixture.category,
1538
+ visible_acceptance_criteria=fixture.visible_acceptance_criteria,
1539
+ file_allowlist=fixture.file_allowlist,
1540
+ visible_checks=fixture.visible_checks,
1541
+ )
1542
+
1543
+
1544
+ def load_eval_task_fixtures() -> tuple[EvalTaskFixture, ...]:
1545
+ """Load the built-in offline eval fixtures from package resources."""
1546
+ manifest = _load_default_eval_fixture_pack_manifest()
1547
+ fixture_ids = tuple(manifest["fixture_ids"])
1548
+ fixture_root = files(_EVAL_FIXTURE_PACK_PACKAGE).joinpath("fixtures")
1549
+
1550
+ fixtures = tuple(
1551
+ _load_eval_task_fixture_resource(
1552
+ fixture_root.joinpath(f"{fixture_id}.json"),
1553
+ )
1554
+ for fixture_id in fixture_ids
1555
+ )
1556
+ if tuple(fixture.fixture_id for fixture in fixtures) != fixture_ids:
1557
+ raise ValueError("fixture resource order does not match pack manifest")
1558
+ return fixtures
1559
+
1560
+
1561
+ def load_eval_task_fixture(fixture_id: str) -> EvalTaskFixture:
1562
+ """Load a single built-in offline eval fixture by fixture ID."""
1563
+ fixtures_by_id = {
1564
+ fixture.fixture_id: fixture for fixture in load_eval_task_fixtures()
1565
+ }
1566
+ try:
1567
+ return fixtures_by_id[fixture_id]
1568
+ except KeyError as exc:
1569
+ raise KeyError(f"unknown eval-suite fixture_id: {fixture_id}") from exc
1570
+
1571
+
1572
+ def load_eval_fixture_pack_summary() -> EvalFixturePackSummary:
1573
+ """Load the deterministic summary for the built-in offline fixture pack."""
1574
+ manifest = _load_default_eval_fixture_pack_manifest()
1575
+ fixtures = load_eval_task_fixtures()
1576
+ fixture_ids = tuple(fixture.fixture_id for fixture in fixtures)
1577
+ if fixture_ids != tuple(manifest["fixture_ids"]):
1578
+ raise ValueError("fixture pack manifest does not match loaded fixtures")
1579
+
1580
+ category_counts = {
1581
+ category: sum(1 for fixture in fixtures if fixture.category == category)
1582
+ for category in EvalTaskCategory
1583
+ if any(fixture.category == category for fixture in fixtures)
1584
+ }
1585
+ summary = EvalFixturePackSummary.model_construct(
1586
+ fixture_pack_id=str(manifest["fixture_pack_id"]),
1587
+ fixture_ids=fixture_ids,
1588
+ category_counts=category_counts,
1589
+ fixture_hashes=tuple(
1590
+ EvalHashRecord(
1591
+ hash_kind=EVAL_SUITE_FIXTURE_HASH_KIND,
1592
+ sha256=fixture.fixture_hash,
1593
+ )
1594
+ for fixture in fixtures
1595
+ ),
1596
+ pack_summary=str(manifest["pack_summary"]),
1597
+ fixture_pack_hash_kind=EVAL_SUITE_FIXTURE_PACK_HASH_KIND,
1598
+ fixture_pack_hash="0" * 64,
1599
+ )
1600
+ payload = summary.model_dump(mode="json")
1601
+ payload["fixture_pack_hash"] = calculate_eval_fixture_pack_hash(summary)
1602
+ return EvalFixturePackSummary.model_validate(payload)
1603
+
1604
+
1605
+ _SUCCESS_TERMINAL_RESULTS = frozenset(
1606
+ {
1607
+ "PLAN_READY",
1608
+ "BUILDER_COMPLETE",
1609
+ "CHECKER_APPROVED",
1610
+ "ARBITER_CLOSED",
1611
+ }
1612
+ )
1613
+ _CLOSURE_TERMINAL_RESULTS = frozenset({"ARBITER_CLOSED"})
1614
+ _BLOCKED_TERMINAL_RESULTS = frozenset(
1615
+ {
1616
+ "PLAN_BLOCKED",
1617
+ "BUILDER_BLOCKED",
1618
+ "CHECKER_BLOCKED",
1619
+ "ARBITER_BLOCKED",
1620
+ }
1621
+ )
1622
+
1623
+
1624
+ def score_eval_trial(
1625
+ fixture: EvalTaskFixture,
1626
+ scorer_input: EvalScorerInput,
1627
+ ) -> EvalScorerResult:
1628
+ """Classify one offline eval trial with deterministic precedence."""
1629
+ if scorer_input.fixture_id != fixture.fixture_id:
1630
+ return _build_eval_scorer_result(
1631
+ scorer_input,
1632
+ final_outcome=EvalTrialOutcome.INVALID_TRIAL,
1633
+ primary_success=False,
1634
+ failure_labels=(EvalFailureTaxonomyLabel.INFRASTRUCTURE_DEFECT,),
1635
+ public_diagnostics=("Scorer input fixture_id does not match fixture.",),
1636
+ scorer_only_diagnostics=(
1637
+ f"expected fixture_id {fixture.fixture_id}; "
1638
+ f"got {scorer_input.fixture_id}",
1639
+ ),
1640
+ invalid_trial_explanation="scorer input fixture_id does not match fixture",
1641
+ )
1642
+ if scorer_input.fixture_hash != fixture.fixture_hash:
1643
+ return _build_eval_scorer_result(
1644
+ scorer_input,
1645
+ final_outcome=EvalTrialOutcome.INVALID_TRIAL,
1646
+ primary_success=False,
1647
+ failure_labels=(EvalFailureTaxonomyLabel.INFRASTRUCTURE_DEFECT,),
1648
+ public_diagnostics=("Scorer input fixture hash does not match fixture.",),
1649
+ scorer_only_diagnostics=(
1650
+ f"expected fixture_hash {fixture.fixture_hash}; "
1651
+ f"got {scorer_input.fixture_hash}",
1652
+ ),
1653
+ invalid_trial_explanation="scorer input fixture_hash does not match fixture",
1654
+ )
1655
+ if scorer_input.invalid_trial_explanation:
1656
+ return _build_eval_scorer_result(
1657
+ scorer_input,
1658
+ final_outcome=EvalTrialOutcome.INVALID_TRIAL,
1659
+ primary_success=False,
1660
+ failure_labels=(EvalFailureTaxonomyLabel.INFRASTRUCTURE_DEFECT,),
1661
+ public_diagnostics=("Evaluation infrastructure defect invalidated trial.",),
1662
+ scorer_only_diagnostics=(scorer_input.invalid_trial_explanation,),
1663
+ invalid_trial_explanation=scorer_input.invalid_trial_explanation,
1664
+ )
1665
+
1666
+ missing_artifact_ids = tuple(
1667
+ artifact_id
1668
+ for artifact_id in scorer_input.required_public_artifact_ids
1669
+ if artifact_id not in set(scorer_input.provided_public_artifact_ids)
1670
+ )
1671
+ malformed_artifact_ids = scorer_input.malformed_artifact_ids
1672
+ visible_failed = any(
1673
+ not result.passed for result in scorer_input.visible_check_results
1674
+ )
1675
+ hidden_failed = any(
1676
+ not result.passed for result in scorer_input.hidden_check_results
1677
+ )
1678
+ expected_mutation_absent = (
1679
+ fixture.expected_mutation_policy.mutation_kind
1680
+ != EvalExpectedMutationKind.NO_SOURCE_CHANGE
1681
+ and not scorer_input.claimed_mutation_present
1682
+ )
1683
+ unauthorized_mutation = scorer_input.unauthorized_mutation
1684
+ artifact_complete = not missing_artifact_ids and not malformed_artifact_ids
1685
+ capability_violation = scorer_input.capability_audit.capability_violation
1686
+ invalid_public_evidence = not scorer_input.checker_public_evidence_valid
1687
+ evidence_defect = bool(
1688
+ visible_failed
1689
+ or hidden_failed
1690
+ or not artifact_complete
1691
+ or expected_mutation_absent
1692
+ or unauthorized_mutation
1693
+ or capability_violation
1694
+ or invalid_public_evidence
1695
+ )
1696
+
1697
+ failure_labels = _eval_failure_labels(
1698
+ visible_failed=visible_failed,
1699
+ hidden_failed=hidden_failed,
1700
+ missing_artifact_ids=missing_artifact_ids,
1701
+ malformed_artifact_ids=malformed_artifact_ids,
1702
+ expected_mutation_absent=expected_mutation_absent,
1703
+ unauthorized_mutation=unauthorized_mutation,
1704
+ capability_violation=capability_violation,
1705
+ success_terminal_unsupported=_success_terminal_unsupported(
1706
+ scorer_input,
1707
+ visible_failed=visible_failed,
1708
+ hidden_failed=hidden_failed,
1709
+ artifact_complete=artifact_complete,
1710
+ expected_mutation_absent=expected_mutation_absent,
1711
+ unauthorized_mutation=unauthorized_mutation,
1712
+ capability_violation=capability_violation,
1713
+ invalid_public_evidence=invalid_public_evidence,
1714
+ ),
1715
+ invalid_public_evidence=invalid_public_evidence,
1716
+ provider_failure=scorer_input.provider_failure,
1717
+ )
1718
+ unsupported_success = (
1719
+ EvalFailureTaxonomyLabel.FALSE_SUCCESS_TERMINAL in failure_labels
1720
+ )
1721
+ final_closure_claimed = any(
1722
+ terminal in _CLOSURE_TERMINAL_RESULTS
1723
+ for terminal in scorer_input.stage_terminal_results
1724
+ )
1725
+ false_closure = final_closure_claimed and evidence_defect
1726
+ blocked = any(
1727
+ terminal in _BLOCKED_TERMINAL_RESULTS
1728
+ for terminal in scorer_input.stage_terminal_results
1729
+ )
1730
+
1731
+ if scorer_input.provider_failure:
1732
+ outcome = EvalTrialOutcome.PROVIDER_FAILURE
1733
+ elif scorer_input.runtime_failure:
1734
+ outcome = EvalTrialOutcome.RUNTIME_FAILURE
1735
+ elif false_closure:
1736
+ outcome = EvalTrialOutcome.FALSE_CLOSURE
1737
+ elif (
1738
+ blocked and fixture.expected_final_outcome == EvalTrialOutcome.CORRECTLY_BLOCKED
1739
+ ):
1740
+ outcome = EvalTrialOutcome.CORRECTLY_BLOCKED
1741
+ elif blocked:
1742
+ outcome = EvalTrialOutcome.FALSE_BLOCKED
1743
+ else:
1744
+ outcome = EvalTrialOutcome.VALID_COMPLETION
1745
+ primary_success = (
1746
+ outcome == EvalTrialOutcome.VALID_COMPLETION
1747
+ and not evidence_defect
1748
+ and not unsupported_success
1749
+ )
1750
+
1751
+ return _build_eval_scorer_result(
1752
+ scorer_input,
1753
+ final_outcome=outcome,
1754
+ primary_success=primary_success,
1755
+ false_closure=false_closure,
1756
+ false_success=unsupported_success,
1757
+ correctly_blocked=outcome == EvalTrialOutcome.CORRECTLY_BLOCKED,
1758
+ capability_violation=capability_violation,
1759
+ artifact_complete=artifact_complete,
1760
+ missing_artifact_ids=missing_artifact_ids,
1761
+ malformed_artifact_ids=malformed_artifact_ids,
1762
+ failure_labels=failure_labels,
1763
+ public_diagnostics=_eval_public_diagnostics(
1764
+ outcome,
1765
+ missing_artifact_ids=missing_artifact_ids,
1766
+ malformed_artifact_ids=malformed_artifact_ids,
1767
+ capability_violation=capability_violation,
1768
+ visible_failed=visible_failed,
1769
+ unsupported_success=unsupported_success,
1770
+ ),
1771
+ scorer_only_diagnostics=_eval_scorer_only_diagnostics(
1772
+ scorer_input,
1773
+ hidden_failed=hidden_failed,
1774
+ expected_mutation_absent=expected_mutation_absent,
1775
+ unauthorized_mutation=unauthorized_mutation,
1776
+ invalid_public_evidence=invalid_public_evidence,
1777
+ ),
1778
+ )
1779
+
1780
+
1781
+ def canonical_eval_suite_bytes(value: BaseModel | Mapping[str, Any]) -> bytes:
1782
+ """Return canonical ASCII JSON bytes for an eval-suite payload."""
1783
+ if isinstance(value, BaseModel):
1784
+ payload = value.model_dump(mode="json")
1785
+ else:
1786
+ payload = dict(value)
1787
+ return (
1788
+ json.dumps(payload, sort_keys=True, separators=(",", ":"), ensure_ascii=True)
1789
+ + "\n"
1790
+ ).encode("ascii")
1791
+
1792
+
1793
+ def canonical_offline_dry_campaign_closure_evidence_bytes(
1794
+ evidence: EvalOfflineDryCampaignClosureEvidence,
1795
+ ) -> bytes:
1796
+ """Return canonical ASCII JSON bytes for offline closure evidence."""
1797
+ return canonical_eval_suite_bytes(evidence)
1798
+
1799
+
1800
+ def calculate_offline_dry_campaign_closure_evidence_hash(
1801
+ evidence: EvalOfflineDryCampaignClosureEvidence,
1802
+ ) -> str:
1803
+ """Return the closure evidence hash with the self-hash field omitted."""
1804
+ payload = evidence.model_dump(mode="json")
1805
+ payload.pop("closure_evidence_hash", None)
1806
+ return hashlib.sha256(canonical_eval_suite_bytes(payload)).hexdigest()
1807
+
1808
+
1809
+ def calculate_eval_model_manifest_hash(manifest: EvalModelManifest) -> str:
1810
+ payload = manifest.model_dump(mode="json")
1811
+ payload.pop("model_manifest_hash", None)
1812
+ return hashlib.sha256(canonical_eval_suite_bytes(payload)).hexdigest()
1813
+
1814
+
1815
+ def calculate_eval_campaign_manifest_hash(manifest: EvalCampaignManifest) -> str:
1816
+ payload = manifest.model_dump(mode="json")
1817
+ payload.pop("campaign_manifest_hash", None)
1818
+ return hashlib.sha256(canonical_eval_suite_bytes(payload)).hexdigest()
1819
+
1820
+
1821
+ def calculate_eval_task_fixture_hash(fixture: EvalTaskFixture) -> str:
1822
+ payload = fixture.model_dump(mode="json")
1823
+ payload.pop("fixture_hash", None)
1824
+ return hashlib.sha256(canonical_eval_suite_bytes(payload)).hexdigest()
1825
+
1826
+
1827
+ def calculate_eval_fixture_pack_hash(summary: EvalFixturePackSummary) -> str:
1828
+ payload = summary.model_dump(mode="json")
1829
+ payload.pop("fixture_pack_hash", None)
1830
+ return hashlib.sha256(canonical_eval_suite_bytes(payload)).hexdigest()
1831
+
1832
+
1833
+ def calculate_eval_scorer_input_hash(scorer_input: EvalScorerInput) -> str:
1834
+ payload = scorer_input.model_dump(mode="json")
1835
+ payload.pop("scorer_input_hash", None)
1836
+ return hashlib.sha256(canonical_eval_suite_bytes(payload)).hexdigest()
1837
+
1838
+
1839
+ def calculate_eval_scorer_result_hash(result: EvalScorerResult) -> str:
1840
+ payload = result.model_dump(mode="json")
1841
+ payload.pop("result_hash", None)
1842
+ return hashlib.sha256(canonical_eval_suite_bytes(payload)).hexdigest()
1843
+
1844
+
1845
+ def _build_eval_scorer_result(
1846
+ scorer_input: EvalScorerInput,
1847
+ *,
1848
+ final_outcome: EvalTrialOutcome,
1849
+ primary_success: bool,
1850
+ false_closure: bool = False,
1851
+ false_success: bool = False,
1852
+ correctly_blocked: bool = False,
1853
+ capability_violation: bool | None = None,
1854
+ artifact_complete: bool = True,
1855
+ missing_artifact_ids: tuple[str, ...] = (),
1856
+ malformed_artifact_ids: tuple[str, ...] = (),
1857
+ failure_labels: tuple[EvalFailureTaxonomyLabel, ...] = (),
1858
+ public_diagnostics: tuple[str, ...] = (),
1859
+ scorer_only_diagnostics: tuple[str, ...] = (),
1860
+ invalid_trial_explanation: str | None = None,
1861
+ ) -> EvalScorerResult:
1862
+ result = EvalScorerResult.model_construct(
1863
+ trial_id=scorer_input.trial_id,
1864
+ fixture_id=scorer_input.fixture_id,
1865
+ final_outcome=final_outcome,
1866
+ primary_success=primary_success,
1867
+ false_closure=false_closure,
1868
+ false_success=false_success,
1869
+ correctly_blocked=correctly_blocked,
1870
+ capability_violation=(
1871
+ scorer_input.capability_audit.capability_violation
1872
+ if capability_violation is None
1873
+ else capability_violation
1874
+ ),
1875
+ artifact_complete=artifact_complete,
1876
+ missing_artifact_ids=missing_artifact_ids,
1877
+ malformed_artifact_ids=malformed_artifact_ids,
1878
+ failure_labels=failure_labels,
1879
+ public_diagnostics=public_diagnostics,
1880
+ scorer_only_diagnostics=scorer_only_diagnostics,
1881
+ invalid_trial_explanation=invalid_trial_explanation,
1882
+ scorer_version=EVAL_SUITE_DEFAULT_SCORER_VERSION,
1883
+ result_hash_kind=EVAL_SUITE_SCORER_RESULT_HASH_KIND,
1884
+ result_hash="0" * 64,
1885
+ )
1886
+ return EvalScorerResult.model_validate(
1887
+ result.model_copy(
1888
+ update={"result_hash": calculate_eval_scorer_result_hash(result)}
1889
+ )
1890
+ )
1891
+
1892
+
1893
+ def _success_terminal_unsupported(
1894
+ scorer_input: EvalScorerInput,
1895
+ *,
1896
+ visible_failed: bool,
1897
+ hidden_failed: bool,
1898
+ artifact_complete: bool,
1899
+ expected_mutation_absent: bool,
1900
+ unauthorized_mutation: bool,
1901
+ capability_violation: bool,
1902
+ invalid_public_evidence: bool,
1903
+ ) -> bool:
1904
+ success_terminal_emitted = any(
1905
+ terminal in _SUCCESS_TERMINAL_RESULTS
1906
+ for terminal in scorer_input.stage_terminal_results
1907
+ )
1908
+ return success_terminal_emitted and bool(
1909
+ visible_failed
1910
+ or hidden_failed
1911
+ or not artifact_complete
1912
+ or expected_mutation_absent
1913
+ or unauthorized_mutation
1914
+ or capability_violation
1915
+ or invalid_public_evidence
1916
+ or scorer_input.runtime_failure
1917
+ or scorer_input.provider_failure
1918
+ )
1919
+
1920
+
1921
+ def _eval_failure_labels(
1922
+ *,
1923
+ visible_failed: bool,
1924
+ hidden_failed: bool,
1925
+ missing_artifact_ids: tuple[str, ...],
1926
+ malformed_artifact_ids: tuple[str, ...],
1927
+ expected_mutation_absent: bool,
1928
+ unauthorized_mutation: bool,
1929
+ capability_violation: bool,
1930
+ success_terminal_unsupported: bool,
1931
+ invalid_public_evidence: bool,
1932
+ provider_failure: bool,
1933
+ ) -> tuple[EvalFailureTaxonomyLabel, ...]:
1934
+ labels: list[EvalFailureTaxonomyLabel] = []
1935
+ if visible_failed:
1936
+ labels.append(EvalFailureTaxonomyLabel.VISIBLE_CHECK_FAILED)
1937
+ if hidden_failed:
1938
+ labels.append(EvalFailureTaxonomyLabel.HIDDEN_CHECK_FAILED)
1939
+ if missing_artifact_ids:
1940
+ labels.append(EvalFailureTaxonomyLabel.REQUIRED_ARTIFACT_MISSING)
1941
+ if malformed_artifact_ids:
1942
+ labels.append(EvalFailureTaxonomyLabel.REQUIRED_ARTIFACT_MALFORMED)
1943
+ if expected_mutation_absent:
1944
+ labels.append(EvalFailureTaxonomyLabel.EXPECTED_MUTATION_ABSENT)
1945
+ if unauthorized_mutation:
1946
+ labels.append(EvalFailureTaxonomyLabel.UNAUTHORIZED_MUTATION)
1947
+ if capability_violation:
1948
+ labels.append(EvalFailureTaxonomyLabel.CAPABILITY_VIOLATION)
1949
+ if success_terminal_unsupported or invalid_public_evidence:
1950
+ labels.append(EvalFailureTaxonomyLabel.FALSE_SUCCESS_TERMINAL)
1951
+ if provider_failure:
1952
+ labels.append(EvalFailureTaxonomyLabel.PROVIDER_DEFECT)
1953
+ return tuple(dict.fromkeys(labels))
1954
+
1955
+
1956
+ def _eval_public_diagnostics(
1957
+ final_outcome: EvalTrialOutcome,
1958
+ *,
1959
+ missing_artifact_ids: tuple[str, ...],
1960
+ malformed_artifact_ids: tuple[str, ...],
1961
+ capability_violation: bool,
1962
+ visible_failed: bool,
1963
+ unsupported_success: bool,
1964
+ ) -> tuple[str, ...]:
1965
+ diagnostics: list[str] = []
1966
+ if missing_artifact_ids:
1967
+ diagnostics.append("Required public artifacts are missing.")
1968
+ if malformed_artifact_ids:
1969
+ diagnostics.append("Required public artifacts are malformed.")
1970
+ if capability_violation:
1971
+ diagnostics.append("Capability envelope violation was observed.")
1972
+ if visible_failed:
1973
+ diagnostics.append("One or more visible checks failed.")
1974
+ if unsupported_success:
1975
+ diagnostics.append("A success terminal was unsupported by required evidence.")
1976
+ if final_outcome == EvalTrialOutcome.PROVIDER_FAILURE:
1977
+ diagnostics.append("Provider failure prevented a valid trial completion.")
1978
+ if final_outcome == EvalTrialOutcome.RUNTIME_FAILURE:
1979
+ diagnostics.append("Runtime failure prevented a valid trial completion.")
1980
+ return tuple(diagnostics)
1981
+
1982
+
1983
+ def _eval_scorer_only_diagnostics(
1984
+ scorer_input: EvalScorerInput,
1985
+ *,
1986
+ hidden_failed: bool,
1987
+ expected_mutation_absent: bool,
1988
+ unauthorized_mutation: bool,
1989
+ invalid_public_evidence: bool,
1990
+ ) -> tuple[str, ...]:
1991
+ diagnostics: list[str] = []
1992
+ if hidden_failed:
1993
+ failed_ids = tuple(
1994
+ result.check_id
1995
+ for result in scorer_input.hidden_check_results
1996
+ if not result.passed
1997
+ )
1998
+ diagnostics.append(f"Hidden check failures: {', '.join(failed_ids)}.")
1999
+ if expected_mutation_absent:
2000
+ diagnostics.append("Expected workspace mutation was absent.")
2001
+ if unauthorized_mutation:
2002
+ diagnostics.append("Unauthorized workspace mutation was observed.")
2003
+ if invalid_public_evidence:
2004
+ diagnostics.append("Checker approved using invalid public evidence.")
2005
+ return tuple(diagnostics)
2006
+
2007
+
2008
+ def _validate_sha256(value: str) -> None:
2009
+ if not _SHA256_RE.fullmatch(value):
2010
+ raise ValueError("expected lowercase sha256 hex digest")
2011
+
2012
+
2013
+ def _validate_relative_path(value: str) -> None:
2014
+ if (
2015
+ not value
2016
+ or value.startswith(".")
2017
+ or "\\" in value
2018
+ or value.startswith("/")
2019
+ or ".." in value.split("/")
2020
+ ):
2021
+ raise ValueError("fixture paths must be stable relative POSIX paths")
2022
+ _reject_forbidden_material(value)
2023
+
2024
+
2025
+ def _validate_relative_artifact_root(value: str) -> None:
2026
+ if (
2027
+ not value
2028
+ or value.startswith(".")
2029
+ or value.startswith("/")
2030
+ or "\\" in value
2031
+ or "//" in value
2032
+ or ".." in value.split("/")
2033
+ ):
2034
+ raise ValueError("eval-suite artifact roots must be stable relative paths")
2035
+ _reject_forbidden_material(value)
2036
+
2037
+
2038
+ def _validate_public_artifact_ids(values: tuple[str, ...]) -> None:
2039
+ for value in values:
2040
+ if not value or "/" in value or "\\" in value or value.startswith("."):
2041
+ raise ValueError("public artifact IDs must be stable bare identifiers")
2042
+ _reject_forbidden_material(value)
2043
+
2044
+
2045
+ def _reject_offline_dry_live_flags(
2046
+ *,
2047
+ allow_live_execution: bool,
2048
+ allow_live_model_call: bool,
2049
+ allow_pi_execution: bool,
2050
+ allow_millforge_harness_execution: bool,
2051
+ ) -> None:
2052
+ if any(
2053
+ (
2054
+ allow_live_execution,
2055
+ allow_live_model_call,
2056
+ allow_pi_execution,
2057
+ allow_millforge_harness_execution,
2058
+ )
2059
+ ):
2060
+ raise ValueError(
2061
+ _offline_dry_diagnostic(
2062
+ EvalOfflineDryCampaignDiagnosticCode.LIVE_EXECUTION_UNAVAILABLE,
2063
+ "eval_suite.dry_campaign.live_execution",
2064
+ "Offline dry-campaign preflight rejects live execution flags.",
2065
+ ).summary
2066
+ )
2067
+
2068
+
2069
+ def _validate_offline_dry_output_root(output_root: str | Path) -> None:
2070
+ root_text = str(output_root)
2071
+ if not root_text.strip():
2072
+ raise ValueError("output root is required")
2073
+ parts = tuple(part.lower() for part in Path(root_text).parts)
2074
+ unsafe_parts = {
2075
+ ".claude",
2076
+ ".codex",
2077
+ ".eval-scratch",
2078
+ ".millrace",
2079
+ ".pytest_cache",
2080
+ ".ruff_cache",
2081
+ "__pycache__",
2082
+ "ideas",
2083
+ "millrace-agents",
2084
+ "ref-forge",
2085
+ }
2086
+ if any(part in unsafe_parts for part in parts):
2087
+ raise ValueError(
2088
+ _offline_dry_diagnostic(
2089
+ EvalOfflineDryCampaignDiagnosticCode.UNSAFE_OUTPUT_ROOT,
2090
+ "eval_suite.dry_campaign.output_root",
2091
+ "Output root is under ignored control state.",
2092
+ ).summary
2093
+ )
2094
+
2095
+
2096
+ def _offline_dry_output_path(output_root: str | Path, relative_path: str) -> Path:
2097
+ _validate_relative_artifact_root(relative_path)
2098
+ return Path(output_root).joinpath(*relative_path.split("/"))
2099
+
2100
+
2101
+ def _read_offline_dry_campaign_record_summaries(
2102
+ output_root: str | Path,
2103
+ plan: Any,
2104
+ ) -> tuple[dict[str, str], ...]:
2105
+ trials_path = _offline_dry_output_path(
2106
+ output_root,
2107
+ f"{plan.campaign_store_root}/trials.jsonl",
2108
+ )
2109
+ if not trials_path.exists():
2110
+ return ()
2111
+ records: list[dict[str, str]] = []
2112
+ for line in trials_path.read_bytes().splitlines():
2113
+ if not line:
2114
+ continue
2115
+ payload = json.loads(line.decode("utf-8"))
2116
+ if not isinstance(payload, dict):
2117
+ raise ValueError("existing trial record must be a JSON object")
2118
+ for field_name in ("trial_id", "trial_plan_hash", "record_hash"):
2119
+ if not isinstance(payload.get(field_name), str):
2120
+ raise ValueError("existing trial record is missing public hash fields")
2121
+ record_hash = payload["record_hash"]
2122
+ hash_payload = dict(payload)
2123
+ hash_payload.pop("record_hash")
2124
+ expected = hashlib.sha256(canonical_eval_suite_bytes(hash_payload)).hexdigest()
2125
+ if record_hash != expected:
2126
+ raise ValueError("existing trial record hash does not match public payload")
2127
+ records.append(
2128
+ {
2129
+ "trial_id": payload["trial_id"],
2130
+ "trial_plan_hash": payload["trial_plan_hash"],
2131
+ "record_hash": record_hash,
2132
+ }
2133
+ )
2134
+ return tuple(records)
2135
+
2136
+
2137
+ def _validate_offline_dry_record_summaries_match_plans(
2138
+ *,
2139
+ records: tuple[dict[str, str], ...],
2140
+ plans: tuple[Any, ...],
2141
+ generated_records: Mapping[str, Any],
2142
+ ) -> None:
2143
+ plans_by_trial_id = {plan.trial_id: plan for plan in plans}
2144
+ if len(plans_by_trial_id) != len(plans):
2145
+ raise ValueError("dry-campaign plans must have unique trial IDs")
2146
+ seen_trial_ids: set[str] = set()
2147
+ for record in records:
2148
+ trial_id = record["trial_id"]
2149
+ if trial_id in seen_trial_ids:
2150
+ raise ValueError("duplicate trial IDs are rejected by append-only stores")
2151
+ seen_trial_ids.add(trial_id)
2152
+ plan = plans_by_trial_id.get(trial_id)
2153
+ if plan is None:
2154
+ raise ValueError(
2155
+ "existing trial record is not present in dry-campaign plan"
2156
+ )
2157
+ if record["trial_plan_hash"] != plan.plan_hash:
2158
+ raise ValueError("existing trial record plan hash does not match plan")
2159
+ generated_record = generated_records.get(trial_id)
2160
+ if generated_record is None or record["record_hash"] != (
2161
+ generated_record.record_hash
2162
+ ):
2163
+ raise ValueError("existing trial record hash does not match dry run")
2164
+
2165
+
2166
+ def _default_offline_fake_runner_script() -> Any:
2167
+ from millforge.eval_trials import (
2168
+ EvalFakeOutcomeScriptKind,
2169
+ EvalTrialFakeRunnerScript,
2170
+ )
2171
+ from millforge.eval_workflow import EvalStageId, EvalTerminalResult
2172
+
2173
+ return EvalTrialFakeRunnerScript(
2174
+ script_id="fake.valid_completion.v1",
2175
+ script_kind=EvalFakeOutcomeScriptKind.VALID_COMPLETION,
2176
+ terminal_results=(
2177
+ EvalTerminalResult.PLAN_READY,
2178
+ EvalTerminalResult.BUILDER_COMPLETE,
2179
+ EvalTerminalResult.CHECKER_APPROVED,
2180
+ EvalTerminalResult.ARBITER_CLOSED,
2181
+ ),
2182
+ expected_outcome=EvalTrialOutcome.VALID_COMPLETION,
2183
+ stage_result_summaries={
2184
+ EvalStageId.PLANNER: "plan ready",
2185
+ EvalStageId.BUILDER: "builder complete",
2186
+ EvalStageId.CHECKER: "checker approved",
2187
+ EvalStageId.ARBITER: "arbiter closed",
2188
+ },
2189
+ )
2190
+
2191
+
2192
+ def _offline_dry_store_manifest(plan: Any) -> Any:
2193
+ from millforge.eval_trials import (
2194
+ EVAL_TRIAL_SCHEMA_VERSION,
2195
+ EVAL_TRIAL_STORE_MANIFEST_HASH_KIND,
2196
+ EvalTrialStoreManifest,
2197
+ calculate_eval_trial_store_manifest_hash,
2198
+ )
2199
+
2200
+ manifest = EvalTrialStoreManifest.model_construct(
2201
+ schema_version=EVAL_TRIAL_SCHEMA_VERSION,
2202
+ store_manifest_id=f"{plan.campaign_manifest.campaign_id}.store.v1",
2203
+ campaign_manifest_hash=plan.campaign_manifest.campaign_manifest_hash,
2204
+ record_hashes=(),
2205
+ append_only=True,
2206
+ store_manifest_hash_kind=EVAL_TRIAL_STORE_MANIFEST_HASH_KIND,
2207
+ store_manifest_hash="0" * 64,
2208
+ )
2209
+ return EvalTrialStoreManifest.model_validate(
2210
+ manifest.model_copy(
2211
+ update={
2212
+ "store_manifest_hash": calculate_eval_trial_store_manifest_hash(
2213
+ manifest
2214
+ )
2215
+ }
2216
+ )
2217
+ )
2218
+
2219
+
2220
+ def _offline_dry_diagnostic(
2221
+ code: EvalOfflineDryCampaignDiagnosticCode,
2222
+ rule_id: str,
2223
+ summary: str,
2224
+ ) -> EvalOfflineDryCampaignDiagnostic:
2225
+ return EvalOfflineDryCampaignDiagnostic(
2226
+ diagnostic_code=code,
2227
+ rule_id=rule_id,
2228
+ summary=summary,
2229
+ )
2230
+
2231
+
2232
+ def _validate_public_command(command: str) -> None:
2233
+ stripped = command.strip()
2234
+ if not stripped:
2235
+ raise ValueError("commands must be non-empty")
2236
+ try:
2237
+ parts = shlex.split(stripped)
2238
+ except ValueError as exc:
2239
+ raise ValueError("commands must be shell-parseable") from exc
2240
+ if not parts:
2241
+ raise ValueError("commands must be non-empty")
2242
+ for token in _iter_public_command_tokens(parts):
2243
+ token_root = token.split(".", 1)[0]
2244
+ if token_root in _NETWORK_COMMANDS:
2245
+ raise ValueError("eval-suite checks must not require network commands")
2246
+ if token_root in _PACKAGE_COMMANDS:
2247
+ raise ValueError("eval-suite checks must not require package installation")
2248
+ if token in _NONDETERMINISTIC_COMMAND_TOKENS:
2249
+ raise ValueError("eval-suite checks must be deterministic")
2250
+ if any(pattern.search(stripped) for pattern in _NONDETERMINISTIC_PYTHON_PRIMITIVES):
2251
+ raise ValueError("eval-suite checks must be deterministic")
2252
+ _reject_forbidden_material(command)
2253
+
2254
+
2255
+ def _iter_public_command_tokens(parts: list[str]) -> tuple[str, ...]:
2256
+ tokens: list[str] = []
2257
+ for part in parts:
2258
+ for word in part.split():
2259
+ normalized = word.split("/")[-1].strip(" \t\r\n\"'`()[]{};,")
2260
+ if normalized.endswith(".exe"):
2261
+ normalized = normalized[:-4]
2262
+ if normalized:
2263
+ tokens.append(normalized.lower())
2264
+ return tuple(tokens)
2265
+
2266
+
2267
+ def _reject_forbidden_material(value: Any) -> None:
2268
+ if isinstance(value, Mapping):
2269
+ for key, child in value.items():
2270
+ key_text = str(key)
2271
+ _reject_secret_like_field_name(key_text)
2272
+ _reject_forbidden_material(key_text)
2273
+ _reject_forbidden_material(child)
2274
+ return
2275
+ if isinstance(value, (tuple, list, set, frozenset)):
2276
+ for child in value:
2277
+ _reject_forbidden_material(child)
2278
+ return
2279
+ if isinstance(value, Enum):
2280
+ _reject_forbidden_material(value.value)
2281
+ return
2282
+ if isinstance(value, str):
2283
+ lowered = value.lower()
2284
+ if any(token in lowered for token in _DENIED_TEXT_TOKENS):
2285
+ raise ValueError("eval-suite payload contains forbidden private material")
2286
+ if any(pattern.search(value) for pattern in _CREDENTIAL_VALUE_PATTERNS):
2287
+ raise ValueError("eval-suite payload contains credential-shaped API key")
2288
+ if _ENDPOINT_URL.search(value):
2289
+ raise ValueError("eval-suite payloads must not contain endpoint URLs")
2290
+ if (
2291
+ _WINDOWS_ABSOLUTE_PATH.search(value)
2292
+ or _POSIX_ABSOLUTE_PATH.search(value)
2293
+ or _USER_HOME_PATH.search(value)
2294
+ ):
2295
+ raise ValueError("eval-suite payloads must not contain host paths")
2296
+
2297
+
2298
+ def _reject_secret_like_field_name(field_name: str) -> None:
2299
+ normalized = field_name.lower().replace("-", "_")
2300
+ if any(marker in normalized for marker in _SECRET_FIELD_MARKERS):
2301
+ raise ValueError("eval-suite payload contains secret-like field name")
2302
+
2303
+
2304
+ def _freeze_eval_suite_mapping(value: Mapping[Any, Any]) -> Mapping[Any, Any]:
2305
+ return _FrozenEvalSuiteDict(
2306
+ {key: _freeze_eval_suite_value(child) for key, child in value.items()}
2307
+ )
2308
+
2309
+
2310
+ def _freeze_eval_suite_value(value: Any) -> Any:
2311
+ if isinstance(value, Mapping):
2312
+ return _freeze_eval_suite_mapping(value)
2313
+ if isinstance(value, tuple):
2314
+ return tuple(_freeze_eval_suite_value(child) for child in value)
2315
+ if isinstance(value, list):
2316
+ return tuple(_freeze_eval_suite_value(child) for child in value)
2317
+ return value
2318
+
2319
+
2320
+ def _load_default_eval_fixture_pack_manifest() -> Mapping[str, Any]:
2321
+ manifest_resource = files(_EVAL_FIXTURE_PACK_PACKAGE).joinpath(
2322
+ _EVAL_FIXTURE_PACK_MANIFEST
2323
+ )
2324
+ manifest = json.loads(manifest_resource.read_text(encoding="utf-8"))
2325
+ if not isinstance(manifest, Mapping):
2326
+ raise ValueError("fixture pack manifest must be a JSON object")
2327
+ if manifest.get("fixture_pack_id") != EVAL_SUITE_DEFAULT_FIXTURE_PACK_ID:
2328
+ raise ValueError("fixture pack manifest has unexpected fixture_pack_id")
2329
+ fixture_ids = manifest.get("fixture_ids")
2330
+ if not isinstance(fixture_ids, list) or not fixture_ids:
2331
+ raise ValueError("fixture pack manifest must list fixture_ids")
2332
+ if not all(isinstance(fixture_id, str) for fixture_id in fixture_ids):
2333
+ raise ValueError("fixture pack manifest fixture_ids must be strings")
2334
+ return manifest
2335
+
2336
+
2337
+ def _load_eval_task_fixture_resource(resource: Any) -> EvalTaskFixture:
2338
+ payload = json.loads(resource.read_text(encoding="utf-8"))
2339
+ if not isinstance(payload, dict):
2340
+ raise ValueError("fixture resource must be a JSON object")
2341
+ payload["fixture_hash"] = _calculate_eval_suite_payload_hash(
2342
+ payload,
2343
+ hash_field="fixture_hash",
2344
+ )
2345
+ return EvalTaskFixture.model_validate(payload)
2346
+
2347
+
2348
+ def _calculate_eval_suite_payload_hash(
2349
+ payload: Mapping[str, Any],
2350
+ *,
2351
+ hash_field: str,
2352
+ ) -> str:
2353
+ hash_payload = dict(payload)
2354
+ hash_payload.pop(hash_field, None)
2355
+ return hashlib.sha256(canonical_eval_suite_bytes(hash_payload)).hexdigest()
2356
+
2357
+
2358
+ __all__ = [
2359
+ "EVAL_SUITE_CAMPAIGN_MANIFEST_HASH_KIND",
2360
+ "EVAL_SUITE_CLOSURE_EVIDENCE_HASH_KIND",
2361
+ "EVAL_SUITE_DEFAULT_CAMPAIGN_CREATED_AT",
2362
+ "EVAL_SUITE_DEFAULT_CAMPAIGN_ID",
2363
+ "EVAL_SUITE_DEFAULT_FIXTURE_PACK_ID",
2364
+ "EVAL_SUITE_DEFAULT_SCORER_VERSION",
2365
+ "EVAL_SUITE_FIXTURE_HASH_KIND",
2366
+ "EVAL_SUITE_FIXTURE_PACK_HASH_KIND",
2367
+ "EVAL_SUITE_MODEL_MANIFEST_HASH_KIND",
2368
+ "EVAL_SUITE_OUTPUT_ROOT_HASH_KIND",
2369
+ "EVAL_SUITE_SCHEMA_VERSION",
2370
+ "EVAL_SUITE_SCORER_INPUT_HASH_KIND",
2371
+ "EVAL_SUITE_SCORER_RESULT_HASH_KIND",
2372
+ "EvalBudgetPolicyReference",
2373
+ "EvalCampaignKind",
2374
+ "EvalCampaignManifest",
2375
+ "EvalCapabilityAuditSummary",
2376
+ "EvalCheckResult",
2377
+ "EvalDifficultyLevel",
2378
+ "EvalDifficultyMetadata",
2379
+ "EvalExpectedMutationKind",
2380
+ "EvalExpectedMutationPolicy",
2381
+ "EvalFailureTaxonomyLabel",
2382
+ "EvalFixturePackSummary",
2383
+ "EvalHashRecord",
2384
+ "EvalHiddenCheck",
2385
+ "EvalLiveDenialDiagnostic",
2386
+ "EvalModelPricingMetadata",
2387
+ "EvalModelRateLimitMetadata",
2388
+ "EvalModelManifest",
2389
+ "EvalOfflineDryCampaignConfig",
2390
+ "EvalOfflineDryCampaignClosureEvidence",
2391
+ "EvalOfflineDryCampaignDiagnostic",
2392
+ "EvalOfflineDryCampaignDiagnosticCode",
2393
+ "EvalOfflineDryCampaignPlan",
2394
+ "EvalOfflineDryCampaignRunResult",
2395
+ "EvalPublicArtifactProjection",
2396
+ "EvalRunnerAcceptanceProjection",
2397
+ "EvalRunnerContextProjection",
2398
+ "EvalRunnerTaskProjection",
2399
+ "EvalScorerInput",
2400
+ "EvalScorerResult",
2401
+ "EvalSuiteContractModel",
2402
+ "EvalSuiteExecutionMode",
2403
+ "EvalTaskCategory",
2404
+ "EvalTaskFixture",
2405
+ "EvalTrialOutcome",
2406
+ "EvalVisibleCheck",
2407
+ "calculate_eval_campaign_manifest_hash",
2408
+ "calculate_eval_fixture_pack_hash",
2409
+ "calculate_eval_model_manifest_hash",
2410
+ "calculate_offline_dry_campaign_closure_evidence_hash",
2411
+ "calculate_eval_scorer_input_hash",
2412
+ "calculate_eval_scorer_result_hash",
2413
+ "calculate_eval_task_fixture_hash",
2414
+ "canonical_eval_suite_bytes",
2415
+ "canonical_offline_dry_campaign_closure_evidence_bytes",
2416
+ "configure_offline_fake_eval_campaign",
2417
+ "default_eval_suite_campaign_manifest",
2418
+ "build_offline_dry_campaign_closure_evidence",
2419
+ "eval_public_artifact_projection",
2420
+ "eval_model_manifest_from_profile",
2421
+ "eval_runner_acceptance_projection",
2422
+ "eval_runner_context_projection",
2423
+ "eval_runner_task_projection",
2424
+ "load_eval_fixture_pack_summary",
2425
+ "load_eval_task_fixture",
2426
+ "load_eval_task_fixtures",
2427
+ "run_offline_fake_eval_campaign",
2428
+ "score_eval_trial",
2429
+ ]