millforge 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- millforge/__init__.py +1174 -0
- millforge/_forge/LICENSE +21 -0
- millforge/_forge/PROVENANCE.json +295 -0
- millforge/_forge/UPDATE_POLICY.md +24 -0
- millforge/_forge/__init__.py +14 -0
- millforge/_forge/adapter.py +2232 -0
- millforge/_forge/base_runner.py +121 -0
- millforge/_forge/clients/__init__.py +10 -0
- millforge/_forge/clients/base.py +200 -0
- millforge/_forge/context/__init__.py +23 -0
- millforge/_forge/context/manager.py +178 -0
- millforge/_forge/context/strategies.py +335 -0
- millforge/_forge/core/__init__.py +16 -0
- millforge/_forge/core/inference.py +433 -0
- millforge/_forge/core/messages.py +119 -0
- millforge/_forge/core/runner.py +479 -0
- millforge/_forge/core/steps.py +108 -0
- millforge/_forge/core/workflow.py +400 -0
- millforge/_forge/errors.py +222 -0
- millforge/_forge/guardrails/__init__.py +21 -0
- millforge/_forge/guardrails/error_tracker.py +71 -0
- millforge/_forge/guardrails/guardrails.py +194 -0
- millforge/_forge/guardrails/nudge.py +47 -0
- millforge/_forge/guardrails/response_validator.py +119 -0
- millforge/_forge/guardrails/step_enforcer.py +183 -0
- millforge/_forge/prompts/__init__.py +16 -0
- millforge/_forge/prompts/nudges.py +95 -0
- millforge/_forge/prompts/templates.py +285 -0
- millforge/_version.py +3 -0
- millforge/artifacts.py +570 -0
- millforge/base/__init__.py +97 -0
- millforge/base/composition.py +402 -0
- millforge/base/context.py +285 -0
- millforge/base/harness.py +138 -0
- millforge/base/identity.py +465 -0
- millforge/base/options.py +34 -0
- millforge/base/platform.py +17 -0
- millforge/base/prompt.py +317 -0
- millforge/base/runner.py +546 -0
- millforge/compiled_plan.py +970 -0
- millforge/compiler/__init__.py +231 -0
- millforge/compiler/artifact_validation.py +257 -0
- millforge/compiler/canonicalization.py +169 -0
- millforge/compiler/capabilities.py +66 -0
- millforge/compiler/catalogs.py +500 -0
- millforge/compiler/diagnostics.py +491 -0
- millforge/compiler/graph.py +678 -0
- millforge/compiler/lowering.py +198 -0
- millforge/compiler/output.py +692 -0
- millforge/compiler/parsing.py +1424 -0
- millforge/compiler/requests.py +1180 -0
- millforge/compiler/schema_validation.py +272 -0
- millforge/compiler/semantic.py +490 -0
- millforge/compiler/service.py +448 -0
- millforge/compiler/source.py +375 -0
- millforge/compiler/validators.py +184 -0
- millforge/connectors/__init__.py +95 -0
- millforge/connectors/admission.py +801 -0
- millforge/connectors/broker.py +202 -0
- millforge/connectors/contracts.py +1159 -0
- millforge/connectors/diagnostics.py +189 -0
- millforge/connectors/fake.py +66 -0
- millforge/connectors/runtime.py +236 -0
- millforge/contracts.py +2860 -0
- millforge/custom_tools/__init__.py +67 -0
- millforge/custom_tools/compiler.py +724 -0
- millforge/custom_tools/contracts.py +1093 -0
- millforge/custom_tools/diagnostics.py +205 -0
- millforge/eval_artifacts.py +952 -0
- millforge/eval_boundary.py +2435 -0
- millforge/eval_fixtures/__init__.py +1 -0
- millforge/eval_fixtures/default_pack/__init__.py +1 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.bug_diagnosis.traceback.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.direct_edit.import_sort.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.evidence_discipline.no_source_change.v1.json +51 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.false_closure.visible_green.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.multi_file.api_contract.v1.json +54 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.recovery.malformed_artifact.v1.json +54 -0
- millforge/eval_fixtures/default_pack/manifest.json +12 -0
- millforge/eval_modes.py +1282 -0
- millforge/eval_presets.py +1398 -0
- millforge/eval_reports.py +2517 -0
- millforge/eval_suite.py +2429 -0
- millforge/eval_trials.py +2632 -0
- millforge/eval_workflow.py +794 -0
- millforge/exceptions.py +122 -0
- millforge/model_backend.py +2098 -0
- millforge/protocols.py +340 -0
- millforge/py.typed +0 -0
- millforge/runtime.py +1791 -0
- millforge/testing/__init__.py +1089 -0
- millforge/tools/__init__.py +83 -0
- millforge/tools/builtin_runtime.py +1339 -0
- millforge/tools/builtins.py +773 -0
- millforge/tools/execution.py +1545 -0
- millforge/tools/path_policy.py +155 -0
- millforge/tools/pi_compat/PI_LICENSE +21 -0
- millforge/tools/pi_compat/PROVENANCE.json +55 -0
- millforge/tools/pi_compat/UPDATE_POLICY.md +36 -0
- millforge/tools/pi_compat/__init__.py +34 -0
- millforge/tools/pi_compat/contracts.py +49 -0
- millforge/tools/pi_compat/editing.py +390 -0
- millforge/tools/pi_compat/mutations.py +57 -0
- millforge/tools/pi_compat/operations.py +401 -0
- millforge/tools/pi_compat/paths.py +155 -0
- millforge/tools/pi_compat/process.py +1375 -0
- millforge/tools/pi_compat/search.py +738 -0
- millforge/tools/pi_compat/truncation.py +267 -0
- millforge/tools/pi_compat_catalog.py +396 -0
- millforge/tools/pi_compat_runtime.py +460 -0
- millforge/tools/registry.py +553 -0
- millforge/tools/results.py +533 -0
- millforge-0.1.0.dist-info/METADATA +844 -0
- millforge-0.1.0.dist-info/RECORD +116 -0
- millforge-0.1.0.dist-info/WHEEL +4 -0
- millforge-0.1.0.dist-info/licenses/LICENSE +201 -0
millforge/eval_modes.py
ADDED
|
@@ -0,0 +1,1282 @@
|
|
|
1
|
+
"""Static compact-eval mode descriptors for Pi and Millforge runners."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
import re
|
|
8
|
+
from collections.abc import Mapping
|
|
9
|
+
from enum import Enum
|
|
10
|
+
from types import MappingProxyType
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from pydantic import (
|
|
14
|
+
BaseModel,
|
|
15
|
+
ConfigDict,
|
|
16
|
+
Field,
|
|
17
|
+
StrictBool,
|
|
18
|
+
StrictFloat,
|
|
19
|
+
StrictInt,
|
|
20
|
+
StrictStr,
|
|
21
|
+
field_validator,
|
|
22
|
+
model_validator,
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
from millforge.eval_artifacts import (
|
|
26
|
+
EvalArtifactLayoutEntry,
|
|
27
|
+
EvalValidatorVisibilityRecord,
|
|
28
|
+
canonical_eval_artifact_layout,
|
|
29
|
+
)
|
|
30
|
+
from millforge.eval_boundary import (
|
|
31
|
+
EvalBoundaryBaseline,
|
|
32
|
+
EvalCapabilityEnvelope,
|
|
33
|
+
EvalContextTier,
|
|
34
|
+
EvalFixtureWorkspacePolicy,
|
|
35
|
+
EvalResourceCeiling,
|
|
36
|
+
EvalStageContextPolicy,
|
|
37
|
+
compact_eval_boundary_baseline,
|
|
38
|
+
default_eval_capability_envelopes,
|
|
39
|
+
default_eval_stage_context_policies,
|
|
40
|
+
default_eval_trial_resource_ceiling,
|
|
41
|
+
)
|
|
42
|
+
from millforge.eval_workflow import (
|
|
43
|
+
EvalStageContract,
|
|
44
|
+
EvalStageId,
|
|
45
|
+
EvalTerminalResult,
|
|
46
|
+
compact_eval_workflow_snapshot,
|
|
47
|
+
default_compact_eval_workflow_graph,
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
EVAL_MODE_SCHEMA_VERSION = 1
|
|
51
|
+
EVAL_MODE_FINGERPRINT_KIND = "eval_mode_descriptor_sha256_v1"
|
|
52
|
+
EVAL_MODE_FAIRNESS_FINGERPRINT_KIND = "eval_mode_fairness_sha256_v1"
|
|
53
|
+
EVAL_MODEL_PROFILE_HASH_KIND = "eval_model_profile_sha256_v1"
|
|
54
|
+
EVAL_SMALL_PI_MODE_ID = "eval_small_pi"
|
|
55
|
+
EVAL_SMALL_MILLFORGE_MODE_ID = "eval_small_millforge"
|
|
56
|
+
EVAL_DEFAULT_MODEL_PROFILE_ID = "eval.backend_neutral.default.v1"
|
|
57
|
+
EVAL_CLOSURE_BOUNDARY_ID = "millforge.eval_boundary.validate_eval_closure.v1"
|
|
58
|
+
EVAL_COMPARISON_ENGINEERING_SMOKE_ONLY = "engineering_smoke_only"
|
|
59
|
+
EVAL_SPEC_07_HARNESS_IDS: Mapping[EvalStageId, str] = {
|
|
60
|
+
EvalStageId.PLANNER: "millforge.eval.planner.single_task.v1",
|
|
61
|
+
EvalStageId.BUILDER: "millforge.eval.builder.code_patch.v1",
|
|
62
|
+
EvalStageId.CHECKER: "millforge.eval.checker.evidence_review.v1",
|
|
63
|
+
EvalStageId.ARBITER: "millforge.eval.arbiter.closure.v1",
|
|
64
|
+
}
|
|
65
|
+
EVAL_SPEC_07_HARNESS_IDS = MappingProxyType(dict(EVAL_SPEC_07_HARNESS_IDS))
|
|
66
|
+
_EVAL_SPEC_07_IMPLEMENTED_HARNESS_STAGES = (
|
|
67
|
+
EvalStageId.PLANNER,
|
|
68
|
+
EvalStageId.BUILDER,
|
|
69
|
+
EvalStageId.CHECKER,
|
|
70
|
+
EvalStageId.ARBITER,
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
_SERVING_CLASSES = frozenset({"hosted_api", "local_openai_compatible", "local_native"})
|
|
74
|
+
_SERVING_PROTOCOLS = frozenset(
|
|
75
|
+
{
|
|
76
|
+
"openai_compatible_chat",
|
|
77
|
+
"openai_compatible_responses",
|
|
78
|
+
"native_provider",
|
|
79
|
+
"offline_fake",
|
|
80
|
+
}
|
|
81
|
+
)
|
|
82
|
+
_TOOL_CALLING_MODES = frozenset({"native", "parser", "unsupported"})
|
|
83
|
+
_CONFOUND_SEVERITIES = frozenset({"info", "warning", "invalidating"})
|
|
84
|
+
_CONFOUND_KINDS = frozenset(
|
|
85
|
+
{
|
|
86
|
+
"runner_capability_enforcement",
|
|
87
|
+
"tool_calling",
|
|
88
|
+
"parser_fallback",
|
|
89
|
+
"context_packing",
|
|
90
|
+
"model_endpoint",
|
|
91
|
+
"token_accounting",
|
|
92
|
+
"wall_clock_measurement",
|
|
93
|
+
"deferred_millforge_harness",
|
|
94
|
+
"deferred_pi_runtime",
|
|
95
|
+
}
|
|
96
|
+
)
|
|
97
|
+
_DEPENDENCY_KINDS = frozenset(
|
|
98
|
+
{
|
|
99
|
+
"runner_runtime",
|
|
100
|
+
"runner_harness",
|
|
101
|
+
"model_backend",
|
|
102
|
+
"resource_enforcement",
|
|
103
|
+
"fixture_workspace",
|
|
104
|
+
}
|
|
105
|
+
)
|
|
106
|
+
_DEPENDENCY_AFFECTED_MODES = frozenset(
|
|
107
|
+
{
|
|
108
|
+
EVAL_SMALL_PI_MODE_ID,
|
|
109
|
+
EVAL_SMALL_MILLFORGE_MODE_ID,
|
|
110
|
+
"all_modes",
|
|
111
|
+
}
|
|
112
|
+
)
|
|
113
|
+
_LIVE_DEPENDENCY_IDS = (
|
|
114
|
+
"pi_live_runtime_support",
|
|
115
|
+
"spec_07_harness_presets",
|
|
116
|
+
"model_backend_configuration",
|
|
117
|
+
"resource_ceiling_enforcement",
|
|
118
|
+
"fixture_workspace_creation",
|
|
119
|
+
)
|
|
120
|
+
_ALLOWED_RUNNER_DIFFERENCE_FIELDS = (
|
|
121
|
+
"mode_id",
|
|
122
|
+
"runner_bindings",
|
|
123
|
+
"deferred_dependencies",
|
|
124
|
+
)
|
|
125
|
+
_DENIED_DESCRIPTOR_TOKENS = (
|
|
126
|
+
"api_key",
|
|
127
|
+
"credential",
|
|
128
|
+
"password",
|
|
129
|
+
"/mnt/f",
|
|
130
|
+
"f:\\",
|
|
131
|
+
"/home/",
|
|
132
|
+
"\\users\\",
|
|
133
|
+
"millrace-agents",
|
|
134
|
+
"ideas/",
|
|
135
|
+
"ref-forge/",
|
|
136
|
+
"hidden scorer",
|
|
137
|
+
"hidden_scorer",
|
|
138
|
+
)
|
|
139
|
+
_WINDOWS_ABSOLUTE_PATH = re.compile(
|
|
140
|
+
r"(?:^|[\s\"'`([{<])(?:[A-Za-z]:[\\/]|\\\\)[^\s\"'`)\]}>,]*"
|
|
141
|
+
)
|
|
142
|
+
_POSIX_ABSOLUTE_PATH = re.compile(r"(?:^|[\s\"'`([{<])/(?!/)[^\s\"'`)\]}>,]*")
|
|
143
|
+
_USER_HOME_PATH = re.compile(r"(?:^|[\s\"'`([{<])~(?:/|\\)[^\s\"'`)\]}>,]*")
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
class _FrozenDescriptorDict(dict[Any, Any]):
|
|
147
|
+
"""Dict-shaped immutable mapping that remains serializable by Pydantic."""
|
|
148
|
+
|
|
149
|
+
def __readonly(self, *args: Any, **kwargs: Any) -> None:
|
|
150
|
+
raise TypeError("eval descriptor mappings are immutable")
|
|
151
|
+
|
|
152
|
+
__setitem__ = __readonly
|
|
153
|
+
__delitem__ = __readonly
|
|
154
|
+
clear = __readonly
|
|
155
|
+
pop = __readonly
|
|
156
|
+
popitem = __readonly # type: ignore[assignment]
|
|
157
|
+
setdefault = __readonly
|
|
158
|
+
update = __readonly
|
|
159
|
+
__ior__ = __readonly # type: ignore[assignment]
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
class EvalRunnerKind(str, Enum):
|
|
163
|
+
"""Closed runner kinds for compact eval modes."""
|
|
164
|
+
|
|
165
|
+
PI = "pi"
|
|
166
|
+
MILLFORGE = "millforge"
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
class EvalRunnerBinding(BaseModel):
|
|
170
|
+
"""Closed binding from a compact eval stage to one runner kind."""
|
|
171
|
+
|
|
172
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
173
|
+
|
|
174
|
+
stage_id: EvalStageId
|
|
175
|
+
runner_kind: EvalRunnerKind
|
|
176
|
+
harness_id: StrictStr | None = None
|
|
177
|
+
|
|
178
|
+
@model_validator(mode="after")
|
|
179
|
+
def _binding_valid(self) -> EvalRunnerBinding:
|
|
180
|
+
if self.runner_kind == EvalRunnerKind.PI:
|
|
181
|
+
if self.harness_id is not None:
|
|
182
|
+
raise ValueError("pi runner bindings must not declare harness_id")
|
|
183
|
+
elif self.harness_id != EVAL_SPEC_07_HARNESS_IDS[self.stage_id]:
|
|
184
|
+
raise ValueError("millforge runner binding has an unknown harness_id")
|
|
185
|
+
return self
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
class EvalModelProfile(BaseModel):
|
|
189
|
+
"""Backend-neutral model profile included in compact eval descriptors."""
|
|
190
|
+
|
|
191
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
192
|
+
|
|
193
|
+
profile_id: StrictStr
|
|
194
|
+
provider_label: StrictStr
|
|
195
|
+
model_label: StrictStr
|
|
196
|
+
serving_class: StrictStr
|
|
197
|
+
serving_protocol: StrictStr
|
|
198
|
+
tool_calling_mode: StrictStr
|
|
199
|
+
parser_id: StrictStr
|
|
200
|
+
reasoning_effort: StrictStr
|
|
201
|
+
temperature: StrictFloat = Field(ge=0.0, le=2.0)
|
|
202
|
+
top_p: StrictFloat = Field(gt=0.0, le=1.0)
|
|
203
|
+
max_prompt_tokens: StrictInt = Field(gt=0)
|
|
204
|
+
max_completion_tokens: StrictInt = Field(gt=0)
|
|
205
|
+
max_total_tokens: StrictInt = Field(gt=0)
|
|
206
|
+
max_model_calls: StrictInt = Field(gt=0)
|
|
207
|
+
cost_accounting: Mapping[StrictStr, StrictStr]
|
|
208
|
+
model_profile_hash_kind: StrictStr = EVAL_MODEL_PROFILE_HASH_KIND
|
|
209
|
+
model_profile_hash: StrictStr
|
|
210
|
+
|
|
211
|
+
@field_validator(
|
|
212
|
+
"profile_id",
|
|
213
|
+
"provider_label",
|
|
214
|
+
"model_label",
|
|
215
|
+
"parser_id",
|
|
216
|
+
"reasoning_effort",
|
|
217
|
+
)
|
|
218
|
+
@classmethod
|
|
219
|
+
def _stable_text_valid(cls, value: str) -> str:
|
|
220
|
+
if not value.strip():
|
|
221
|
+
raise ValueError("model profile text fields must be non-empty")
|
|
222
|
+
_reject_descriptor_material_leaks(value)
|
|
223
|
+
return value
|
|
224
|
+
|
|
225
|
+
@field_validator("serving_class")
|
|
226
|
+
@classmethod
|
|
227
|
+
def _serving_class_valid(cls, value: str) -> str:
|
|
228
|
+
if value not in _SERVING_CLASSES:
|
|
229
|
+
raise ValueError("unsupported eval model serving_class")
|
|
230
|
+
return value
|
|
231
|
+
|
|
232
|
+
@field_validator("serving_protocol")
|
|
233
|
+
@classmethod
|
|
234
|
+
def _serving_protocol_valid(cls, value: str) -> str:
|
|
235
|
+
if value not in _SERVING_PROTOCOLS:
|
|
236
|
+
raise ValueError("unsupported eval model serving_protocol")
|
|
237
|
+
return value
|
|
238
|
+
|
|
239
|
+
@field_validator("tool_calling_mode")
|
|
240
|
+
@classmethod
|
|
241
|
+
def _tool_calling_mode_valid(cls, value: str) -> str:
|
|
242
|
+
if value not in _TOOL_CALLING_MODES:
|
|
243
|
+
raise ValueError("unsupported eval model tool_calling_mode")
|
|
244
|
+
return value
|
|
245
|
+
|
|
246
|
+
@model_validator(mode="after")
|
|
247
|
+
def _model_profile_valid(self) -> EvalModelProfile:
|
|
248
|
+
if self.max_total_tokens != self.max_prompt_tokens + self.max_completion_tokens:
|
|
249
|
+
raise ValueError(
|
|
250
|
+
"max_total_tokens must equal prompt plus completion token limits"
|
|
251
|
+
)
|
|
252
|
+
object.__setattr__(
|
|
253
|
+
self,
|
|
254
|
+
"cost_accounting",
|
|
255
|
+
_freeze_descriptor_mapping(self.cost_accounting),
|
|
256
|
+
)
|
|
257
|
+
if self.model_profile_hash_kind != EVAL_MODEL_PROFILE_HASH_KIND:
|
|
258
|
+
raise ValueError("unsupported model profile hash kind")
|
|
259
|
+
_validate_sha256(self.model_profile_hash)
|
|
260
|
+
expected = calculate_eval_model_profile_hash(self)
|
|
261
|
+
if self.model_profile_hash != expected:
|
|
262
|
+
raise ValueError("model_profile_hash does not match profile payload")
|
|
263
|
+
_reject_descriptor_material_leaks(self.model_dump(mode="json"))
|
|
264
|
+
return self
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
class EvalModeDeferredDependency(BaseModel):
|
|
268
|
+
"""Static descriptor reference to an intentionally deferred implementation."""
|
|
269
|
+
|
|
270
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
271
|
+
|
|
272
|
+
dependency_id: StrictStr
|
|
273
|
+
dependency_kind: StrictStr
|
|
274
|
+
affected_mode: StrictStr
|
|
275
|
+
affected_stage_id: EvalStageId | None = None
|
|
276
|
+
all_stage_scope: StrictBool = True
|
|
277
|
+
static_descriptor_admission_behavior: StrictStr
|
|
278
|
+
live_execution_behavior: StrictStr
|
|
279
|
+
summary: StrictStr
|
|
280
|
+
reference_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
281
|
+
required_before_live_execution: bool = True
|
|
282
|
+
|
|
283
|
+
@field_validator("dependency_kind")
|
|
284
|
+
@classmethod
|
|
285
|
+
def _dependency_kind_valid(cls, value: str) -> str:
|
|
286
|
+
if value not in _DEPENDENCY_KINDS:
|
|
287
|
+
raise ValueError("unsupported deferred dependency kind")
|
|
288
|
+
return value
|
|
289
|
+
|
|
290
|
+
@field_validator("affected_mode")
|
|
291
|
+
@classmethod
|
|
292
|
+
def _affected_mode_valid(cls, value: str) -> str:
|
|
293
|
+
if value not in _DEPENDENCY_AFFECTED_MODES:
|
|
294
|
+
raise ValueError("unsupported deferred dependency affected mode")
|
|
295
|
+
return value
|
|
296
|
+
|
|
297
|
+
@model_validator(mode="after")
|
|
298
|
+
def _dependency_valid(self) -> EvalModeDeferredDependency:
|
|
299
|
+
if not self.dependency_id.strip() or not self.summary.strip():
|
|
300
|
+
raise ValueError("deferred dependency fields must be non-empty")
|
|
301
|
+
if (
|
|
302
|
+
not self.static_descriptor_admission_behavior.strip()
|
|
303
|
+
or not self.live_execution_behavior.strip()
|
|
304
|
+
):
|
|
305
|
+
raise ValueError("deferred dependency behaviors must be non-empty")
|
|
306
|
+
if self.affected_stage_id is None and not self.all_stage_scope:
|
|
307
|
+
raise ValueError(
|
|
308
|
+
"deferred dependency must declare stage or all-stage scope"
|
|
309
|
+
)
|
|
310
|
+
if self.affected_stage_id is not None and self.all_stage_scope:
|
|
311
|
+
raise ValueError("deferred dependency cannot mix stage and all-stage scope")
|
|
312
|
+
_reject_descriptor_material_leaks(self.model_dump(mode="json"))
|
|
313
|
+
return self
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
class EvalModeConfound(BaseModel):
|
|
317
|
+
"""Structured comparability confound for eval mode reports."""
|
|
318
|
+
|
|
319
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
320
|
+
|
|
321
|
+
confound_id: StrictStr
|
|
322
|
+
kind: StrictStr
|
|
323
|
+
severity: StrictStr
|
|
324
|
+
summary: StrictStr
|
|
325
|
+
applies_to: tuple[StrictStr, ...]
|
|
326
|
+
evidence: tuple[StrictStr, ...]
|
|
327
|
+
comparison_effect: StrictStr
|
|
328
|
+
mitigation: StrictStr
|
|
329
|
+
|
|
330
|
+
@field_validator("kind")
|
|
331
|
+
@classmethod
|
|
332
|
+
def _kind_valid(cls, value: str) -> str:
|
|
333
|
+
if value not in _CONFOUND_KINDS:
|
|
334
|
+
raise ValueError("unsupported eval mode confound kind")
|
|
335
|
+
return value
|
|
336
|
+
|
|
337
|
+
@field_validator("severity")
|
|
338
|
+
@classmethod
|
|
339
|
+
def _severity_valid(cls, value: str) -> str:
|
|
340
|
+
if value not in _CONFOUND_SEVERITIES:
|
|
341
|
+
raise ValueError("unsupported eval mode confound severity")
|
|
342
|
+
return value
|
|
343
|
+
|
|
344
|
+
@model_validator(mode="after")
|
|
345
|
+
def _confound_valid(self) -> EvalModeConfound:
|
|
346
|
+
if not self.confound_id.strip() or not self.summary.strip():
|
|
347
|
+
raise ValueError("confound fields must be non-empty")
|
|
348
|
+
if not self.applies_to or any(not value.strip() for value in self.applies_to):
|
|
349
|
+
raise ValueError("confound applies_to must be non-empty")
|
|
350
|
+
if not self.evidence or any(not value.strip() for value in self.evidence):
|
|
351
|
+
raise ValueError("confound evidence must be non-empty")
|
|
352
|
+
if not self.comparison_effect.strip() or not self.mitigation.strip():
|
|
353
|
+
raise ValueError(
|
|
354
|
+
"confound comparison effect and mitigation must be non-empty"
|
|
355
|
+
)
|
|
356
|
+
_reject_descriptor_material_leaks(self.model_dump(mode="json"))
|
|
357
|
+
return self
|
|
358
|
+
|
|
359
|
+
|
|
360
|
+
class EvalModeComparison(BaseModel):
|
|
361
|
+
"""One allowed or disallowed difference between two eval descriptors."""
|
|
362
|
+
|
|
363
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
364
|
+
|
|
365
|
+
field_path: StrictStr
|
|
366
|
+
summary: StrictStr
|
|
367
|
+
left_value: Any
|
|
368
|
+
right_value: Any
|
|
369
|
+
|
|
370
|
+
@model_validator(mode="after")
|
|
371
|
+
def _comparison_valid(self) -> EvalModeComparison:
|
|
372
|
+
if not self.field_path.strip() or not self.summary.strip():
|
|
373
|
+
raise ValueError("comparison fields must be non-empty")
|
|
374
|
+
_reject_descriptor_material_leaks(self.model_dump(mode="json"))
|
|
375
|
+
return self
|
|
376
|
+
|
|
377
|
+
|
|
378
|
+
class EvalModeFairnessReport(BaseModel):
|
|
379
|
+
"""Structured fairness comparison report for two eval mode descriptors."""
|
|
380
|
+
|
|
381
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
382
|
+
|
|
383
|
+
left_mode_id: StrictStr
|
|
384
|
+
right_mode_id: StrictStr
|
|
385
|
+
comparable: StrictBool
|
|
386
|
+
classification: StrictStr
|
|
387
|
+
fairness_fingerprint_kind: StrictStr = EVAL_MODE_FAIRNESS_FINGERPRINT_KIND
|
|
388
|
+
shared_fairness_fingerprint: StrictStr | None
|
|
389
|
+
left_fairness_fingerprint: StrictStr
|
|
390
|
+
right_fairness_fingerprint: StrictStr
|
|
391
|
+
left_descriptor_fingerprint: StrictStr
|
|
392
|
+
right_descriptor_fingerprint: StrictStr
|
|
393
|
+
allowed_differences: tuple[EvalModeComparison, ...] = Field(default_factory=tuple)
|
|
394
|
+
disallowed_differences: tuple[EvalModeComparison, ...] = Field(
|
|
395
|
+
default_factory=tuple
|
|
396
|
+
)
|
|
397
|
+
confounds: tuple[EvalModeConfound, ...] = Field(default_factory=tuple)
|
|
398
|
+
deferred_dependencies: tuple[EvalModeDeferredDependency, ...] = Field(
|
|
399
|
+
default_factory=tuple
|
|
400
|
+
)
|
|
401
|
+
|
|
402
|
+
@model_validator(mode="after")
|
|
403
|
+
def _report_valid(self) -> EvalModeFairnessReport:
|
|
404
|
+
if self.fairness_fingerprint_kind != EVAL_MODE_FAIRNESS_FINGERPRINT_KIND:
|
|
405
|
+
raise ValueError("unsupported fairness fingerprint kind")
|
|
406
|
+
for fingerprint in (
|
|
407
|
+
self.left_fairness_fingerprint,
|
|
408
|
+
self.right_fairness_fingerprint,
|
|
409
|
+
self.left_descriptor_fingerprint,
|
|
410
|
+
self.right_descriptor_fingerprint,
|
|
411
|
+
):
|
|
412
|
+
_validate_sha256(fingerprint)
|
|
413
|
+
if self.shared_fairness_fingerprint is not None:
|
|
414
|
+
_validate_sha256(self.shared_fairness_fingerprint)
|
|
415
|
+
if self.comparable and self.disallowed_differences:
|
|
416
|
+
raise ValueError(
|
|
417
|
+
"comparable fairness reports cannot carry disallowed drift"
|
|
418
|
+
)
|
|
419
|
+
if self.classification != EVAL_COMPARISON_ENGINEERING_SMOKE_ONLY:
|
|
420
|
+
raise ValueError("unsupported eval mode comparison classification")
|
|
421
|
+
_reject_descriptor_material_leaks(self.model_dump(mode="json"))
|
|
422
|
+
return self
|
|
423
|
+
|
|
424
|
+
|
|
425
|
+
class EvalModeLiveAdmissionResult(BaseModel):
|
|
426
|
+
"""Fail-closed live execution admission diagnostic for an eval mode."""
|
|
427
|
+
|
|
428
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
429
|
+
|
|
430
|
+
mode_id: StrictStr
|
|
431
|
+
admitted: StrictBool
|
|
432
|
+
rule_id: StrictStr
|
|
433
|
+
diagnostic_code: StrictStr | None = None
|
|
434
|
+
diagnostic_summary: StrictStr | None = None
|
|
435
|
+
deferred_dependencies: tuple[EvalModeDeferredDependency, ...] = Field(
|
|
436
|
+
default_factory=tuple
|
|
437
|
+
)
|
|
438
|
+
|
|
439
|
+
@model_validator(mode="after")
|
|
440
|
+
def _admission_valid(self) -> EvalModeLiveAdmissionResult:
|
|
441
|
+
if self.admitted:
|
|
442
|
+
if (
|
|
443
|
+
self.diagnostic_code is not None
|
|
444
|
+
or self.diagnostic_summary is not None
|
|
445
|
+
or self.deferred_dependencies
|
|
446
|
+
):
|
|
447
|
+
raise ValueError("admitted eval modes must not carry diagnostics")
|
|
448
|
+
elif self.diagnostic_code is None or self.diagnostic_summary is None:
|
|
449
|
+
raise ValueError("denied eval modes must include diagnostics")
|
|
450
|
+
_reject_descriptor_material_leaks(self.model_dump(mode="json"))
|
|
451
|
+
return self
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
class EvalModeDescriptor(BaseModel):
|
|
455
|
+
"""Immutable static descriptor for one compact eval mode."""
|
|
456
|
+
|
|
457
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
458
|
+
|
|
459
|
+
schema_version: StrictInt = EVAL_MODE_SCHEMA_VERSION
|
|
460
|
+
mode_id: StrictStr
|
|
461
|
+
description: StrictStr
|
|
462
|
+
runner_bindings: tuple[EvalRunnerBinding, ...]
|
|
463
|
+
model_profile: EvalModelProfile
|
|
464
|
+
graph_id: StrictStr
|
|
465
|
+
graph_sha256: StrictStr
|
|
466
|
+
stage_ids: tuple[EvalStageId, ...]
|
|
467
|
+
stage_contracts: tuple[EvalStageContract, ...]
|
|
468
|
+
terminal_results: tuple[EvalTerminalResult, ...]
|
|
469
|
+
transition_semantics: tuple[Mapping[StrictStr, Any], ...]
|
|
470
|
+
boundary_baseline: EvalBoundaryBaseline
|
|
471
|
+
capability_envelopes: tuple[EvalCapabilityEnvelope, ...]
|
|
472
|
+
fixture_policy: EvalFixtureWorkspacePolicy
|
|
473
|
+
artifact_policy: tuple[EvalArtifactLayoutEntry, ...]
|
|
474
|
+
context_tier: EvalContextTier
|
|
475
|
+
stage_context_policies: tuple[EvalStageContextPolicy, ...]
|
|
476
|
+
redaction_categories: tuple[StrictStr, ...]
|
|
477
|
+
validator_visibility_policy: EvalValidatorVisibilityRecord
|
|
478
|
+
trial_resource_ceiling: EvalResourceCeiling
|
|
479
|
+
closure_boundary_id: StrictStr = EVAL_CLOSURE_BOUNDARY_ID
|
|
480
|
+
deferred_dependencies: tuple[EvalModeDeferredDependency, ...] = Field(
|
|
481
|
+
default_factory=tuple
|
|
482
|
+
)
|
|
483
|
+
declared_confounds: tuple[EvalModeConfound, ...] = Field(default_factory=tuple)
|
|
484
|
+
fairness_fingerprint_kind: StrictStr = EVAL_MODE_FAIRNESS_FINGERPRINT_KIND
|
|
485
|
+
fairness_fingerprint: StrictStr
|
|
486
|
+
descriptor_fingerprint_kind: StrictStr = EVAL_MODE_FINGERPRINT_KIND
|
|
487
|
+
descriptor_fingerprint: StrictStr
|
|
488
|
+
|
|
489
|
+
@model_validator(mode="after")
|
|
490
|
+
def _descriptor_valid(self) -> EvalModeDescriptor:
|
|
491
|
+
if self.schema_version != EVAL_MODE_SCHEMA_VERSION:
|
|
492
|
+
raise ValueError("unsupported eval mode schema_version")
|
|
493
|
+
if self.mode_id not in {EVAL_SMALL_PI_MODE_ID, EVAL_SMALL_MILLFORGE_MODE_ID}:
|
|
494
|
+
raise ValueError("unknown static eval mode id")
|
|
495
|
+
if not self.description.strip():
|
|
496
|
+
raise ValueError("descriptor description must be non-empty")
|
|
497
|
+
graph = default_compact_eval_workflow_graph()
|
|
498
|
+
workflow_snapshot = compact_eval_workflow_snapshot(graph)
|
|
499
|
+
if self.graph_id != graph.graph_id:
|
|
500
|
+
raise ValueError("descriptor graph_id does not match compact graph")
|
|
501
|
+
if self.graph_sha256 != workflow_snapshot["graph_sha256"]:
|
|
502
|
+
raise ValueError("descriptor graph_sha256 does not match compact graph")
|
|
503
|
+
if self.stage_ids != graph.stage_ids:
|
|
504
|
+
raise ValueError("descriptor stage_ids do not match compact graph")
|
|
505
|
+
if self.stage_contracts != graph.stages:
|
|
506
|
+
raise ValueError("descriptor stage contracts do not match compact graph")
|
|
507
|
+
if self.terminal_results != tuple(EvalTerminalResult):
|
|
508
|
+
raise ValueError("descriptor terminal results do not match compact graph")
|
|
509
|
+
if self.transition_semantics != tuple(workflow_snapshot["transitions"]):
|
|
510
|
+
raise ValueError("descriptor transitions do not match compact graph")
|
|
511
|
+
if self.boundary_baseline != compact_eval_boundary_baseline():
|
|
512
|
+
raise ValueError("descriptor boundary baseline does not match 06B")
|
|
513
|
+
if self.capability_envelopes != tuple(
|
|
514
|
+
default_eval_capability_envelopes()[stage_id] for stage_id in EvalStageId
|
|
515
|
+
):
|
|
516
|
+
raise ValueError("descriptor capability envelopes do not match 06B")
|
|
517
|
+
if self.fixture_policy != EvalFixtureWorkspacePolicy():
|
|
518
|
+
raise ValueError("descriptor fixture policy does not match 06B")
|
|
519
|
+
if self.artifact_policy != _default_artifact_policy_tuple():
|
|
520
|
+
raise ValueError("descriptor artifact policy does not match 06B")
|
|
521
|
+
if self.context_tier != EvalContextTier.COMPACT:
|
|
522
|
+
raise ValueError("descriptor context tier does not match 06B")
|
|
523
|
+
expected_context_policies = tuple(
|
|
524
|
+
default_eval_stage_context_policies()[stage_id] for stage_id in EvalStageId
|
|
525
|
+
)
|
|
526
|
+
if self.stage_context_policies != expected_context_policies:
|
|
527
|
+
raise ValueError("descriptor stage context policies do not match 06B")
|
|
528
|
+
expected_redactions = expected_context_policies[0].redaction.categories
|
|
529
|
+
if self.redaction_categories != expected_redactions:
|
|
530
|
+
raise ValueError("descriptor redaction policy does not match 06B")
|
|
531
|
+
if self.trial_resource_ceiling != default_eval_trial_resource_ceiling():
|
|
532
|
+
raise ValueError("descriptor trial resource ceiling does not match 06B")
|
|
533
|
+
if self.validator_visibility_policy != _default_validator_visibility_policy():
|
|
534
|
+
raise ValueError(
|
|
535
|
+
"descriptor validator visibility policy does not match 06B"
|
|
536
|
+
)
|
|
537
|
+
if self.closure_boundary_id != EVAL_CLOSURE_BOUNDARY_ID:
|
|
538
|
+
raise ValueError("descriptor closure boundary does not match 06B")
|
|
539
|
+
object.__setattr__(
|
|
540
|
+
self,
|
|
541
|
+
"transition_semantics",
|
|
542
|
+
tuple(
|
|
543
|
+
_freeze_descriptor_mapping(transition)
|
|
544
|
+
for transition in self.transition_semantics
|
|
545
|
+
),
|
|
546
|
+
)
|
|
547
|
+
_validate_runner_bindings(self.mode_id, self.runner_bindings)
|
|
548
|
+
if self.declared_confounds != _default_declared_confounds():
|
|
549
|
+
raise ValueError("descriptor declared confounds do not match defaults")
|
|
550
|
+
if self.fairness_fingerprint_kind != EVAL_MODE_FAIRNESS_FINGERPRINT_KIND:
|
|
551
|
+
raise ValueError("unsupported fairness fingerprint kind")
|
|
552
|
+
_validate_sha256(self.fairness_fingerprint)
|
|
553
|
+
expected_fairness_fingerprint = calculate_eval_mode_fairness_fingerprint(self)
|
|
554
|
+
if self.fairness_fingerprint != expected_fairness_fingerprint:
|
|
555
|
+
raise ValueError("fairness_fingerprint does not match fairness payload")
|
|
556
|
+
if self.descriptor_fingerprint_kind != EVAL_MODE_FINGERPRINT_KIND:
|
|
557
|
+
raise ValueError("unsupported descriptor fingerprint kind")
|
|
558
|
+
_validate_sha256(self.descriptor_fingerprint)
|
|
559
|
+
expected_fingerprint = calculate_eval_mode_fingerprint(self)
|
|
560
|
+
if self.descriptor_fingerprint != expected_fingerprint:
|
|
561
|
+
raise ValueError("descriptor_fingerprint does not match descriptor payload")
|
|
562
|
+
_reject_descriptor_material_leaks(self.model_dump(mode="json"))
|
|
563
|
+
return self
|
|
564
|
+
|
|
565
|
+
|
|
566
|
+
def default_eval_model_profile() -> EvalModelProfile:
|
|
567
|
+
"""Return the shared backend-neutral model profile for static eval modes."""
|
|
568
|
+
profile = EvalModelProfile.model_construct(
|
|
569
|
+
profile_id=EVAL_DEFAULT_MODEL_PROFILE_ID,
|
|
570
|
+
provider_label="provider-neutral",
|
|
571
|
+
model_label="eval-small-backend-neutral",
|
|
572
|
+
serving_class="local_openai_compatible",
|
|
573
|
+
serving_protocol="openai_compatible_responses",
|
|
574
|
+
tool_calling_mode="parser",
|
|
575
|
+
parser_id="millforge.eval.parser.compact_json.v1",
|
|
576
|
+
reasoning_effort="medium",
|
|
577
|
+
temperature=0.0,
|
|
578
|
+
top_p=1.0,
|
|
579
|
+
max_prompt_tokens=32_000,
|
|
580
|
+
max_completion_tokens=8_000,
|
|
581
|
+
max_total_tokens=40_000,
|
|
582
|
+
max_model_calls=2,
|
|
583
|
+
cost_accounting={
|
|
584
|
+
"currency": "none",
|
|
585
|
+
"unit": "not_applicable",
|
|
586
|
+
"rate_source": "static_descriptor",
|
|
587
|
+
},
|
|
588
|
+
model_profile_hash_kind=EVAL_MODEL_PROFILE_HASH_KIND,
|
|
589
|
+
model_profile_hash="0" * 64,
|
|
590
|
+
)
|
|
591
|
+
return EvalModelProfile.model_validate(
|
|
592
|
+
profile.model_copy(
|
|
593
|
+
update={"model_profile_hash": calculate_eval_model_profile_hash(profile)}
|
|
594
|
+
)
|
|
595
|
+
)
|
|
596
|
+
|
|
597
|
+
|
|
598
|
+
def default_eval_small_pi_mode() -> EvalModeDescriptor:
|
|
599
|
+
"""Return the default compact eval descriptor for the Pi runner."""
|
|
600
|
+
return _default_eval_mode(
|
|
601
|
+
mode_id=EVAL_SMALL_PI_MODE_ID,
|
|
602
|
+
runner_kind=EvalRunnerKind.PI,
|
|
603
|
+
)
|
|
604
|
+
|
|
605
|
+
|
|
606
|
+
def default_eval_small_millforge_mode(
|
|
607
|
+
*, spec_07_static_presets_ready: bool = False
|
|
608
|
+
) -> EvalModeDescriptor:
|
|
609
|
+
"""Return the default compact eval descriptor for the Millforge runner."""
|
|
610
|
+
return _default_eval_mode(
|
|
611
|
+
mode_id=EVAL_SMALL_MILLFORGE_MODE_ID,
|
|
612
|
+
runner_kind=EvalRunnerKind.MILLFORGE,
|
|
613
|
+
spec_07_static_presets_ready=spec_07_static_presets_ready,
|
|
614
|
+
)
|
|
615
|
+
|
|
616
|
+
|
|
617
|
+
def canonical_eval_mode_bytes(descriptor: EvalModeDescriptor) -> bytes:
|
|
618
|
+
"""Return canonical ASCII JSON bytes for a full eval mode descriptor."""
|
|
619
|
+
return _canonical_eval_json_bytes(descriptor.model_dump(mode="json"))
|
|
620
|
+
|
|
621
|
+
|
|
622
|
+
def canonical_eval_mode_fairness_bytes(descriptor: EvalModeDescriptor) -> bytes:
|
|
623
|
+
"""Return canonical ASCII JSON bytes for the fairness-relevant projection."""
|
|
624
|
+
return _canonical_eval_json_bytes(_fairness_payload(descriptor))
|
|
625
|
+
|
|
626
|
+
|
|
627
|
+
def calculate_eval_model_profile_hash(profile: EvalModelProfile) -> str:
|
|
628
|
+
"""Return the deterministic hash over backend-neutral model profile fields."""
|
|
629
|
+
payload = profile.model_dump(mode="json")
|
|
630
|
+
payload.pop("model_profile_hash", None)
|
|
631
|
+
return hashlib.sha256(_canonical_eval_json_bytes(payload)).hexdigest()
|
|
632
|
+
|
|
633
|
+
|
|
634
|
+
def calculate_eval_mode_fingerprint(descriptor: EvalModeDescriptor) -> str:
|
|
635
|
+
"""Return the full deterministic descriptor fingerprint."""
|
|
636
|
+
payload = descriptor.model_dump(mode="json")
|
|
637
|
+
payload.pop("descriptor_fingerprint", None)
|
|
638
|
+
payload.pop("fairness_fingerprint", None)
|
|
639
|
+
return hashlib.sha256(_canonical_eval_json_bytes(payload)).hexdigest()
|
|
640
|
+
|
|
641
|
+
|
|
642
|
+
def calculate_eval_mode_fairness_fingerprint(
|
|
643
|
+
descriptor: EvalModeDescriptor,
|
|
644
|
+
) -> str:
|
|
645
|
+
"""Return the deterministic descriptor fairness fingerprint."""
|
|
646
|
+
return hashlib.sha256(canonical_eval_mode_fairness_bytes(descriptor)).hexdigest()
|
|
647
|
+
|
|
648
|
+
|
|
649
|
+
def compare_eval_modes_for_fairness(
|
|
650
|
+
left: EvalModeDescriptor | None = None,
|
|
651
|
+
right: EvalModeDescriptor | None = None,
|
|
652
|
+
) -> EvalModeFairnessReport:
|
|
653
|
+
"""Compare two eval mode descriptors for controlled fairness comparability."""
|
|
654
|
+
left = left or default_eval_small_pi_mode()
|
|
655
|
+
right = right or default_eval_small_millforge_mode()
|
|
656
|
+
left_fairness = calculate_eval_mode_fairness_fingerprint(left)
|
|
657
|
+
right_fairness = calculate_eval_mode_fairness_fingerprint(right)
|
|
658
|
+
allowed = _allowed_runner_differences(left, right)
|
|
659
|
+
disallowed = _disallowed_fairness_differences(left, right)
|
|
660
|
+
deferred_dependencies = _unique_deferred_dependencies(
|
|
661
|
+
left.deferred_dependencies + right.deferred_dependencies
|
|
662
|
+
)
|
|
663
|
+
confounds = _default_confounds_for_dependencies(deferred_dependencies)
|
|
664
|
+
comparable = left_fairness == right_fairness and not disallowed
|
|
665
|
+
return EvalModeFairnessReport(
|
|
666
|
+
left_mode_id=left.mode_id,
|
|
667
|
+
right_mode_id=right.mode_id,
|
|
668
|
+
comparable=comparable,
|
|
669
|
+
classification=EVAL_COMPARISON_ENGINEERING_SMOKE_ONLY,
|
|
670
|
+
shared_fairness_fingerprint=left_fairness if comparable else None,
|
|
671
|
+
left_fairness_fingerprint=left_fairness,
|
|
672
|
+
right_fairness_fingerprint=right_fairness,
|
|
673
|
+
left_descriptor_fingerprint=left.descriptor_fingerprint,
|
|
674
|
+
right_descriptor_fingerprint=right.descriptor_fingerprint,
|
|
675
|
+
allowed_differences=allowed,
|
|
676
|
+
disallowed_differences=disallowed,
|
|
677
|
+
confounds=confounds,
|
|
678
|
+
deferred_dependencies=deferred_dependencies,
|
|
679
|
+
)
|
|
680
|
+
|
|
681
|
+
|
|
682
|
+
def admit_eval_mode_live_execution(
|
|
683
|
+
descriptor: EvalModeDescriptor,
|
|
684
|
+
*,
|
|
685
|
+
pi_live_runtime_available: bool = False,
|
|
686
|
+
spec_07_harness_presets_available: bool = False,
|
|
687
|
+
model_backend_configured: bool = False,
|
|
688
|
+
resource_ceilings_enforceable: bool = False,
|
|
689
|
+
fixture_workspace_creation_available: bool = False,
|
|
690
|
+
) -> EvalModeLiveAdmissionResult:
|
|
691
|
+
"""Return a fail-closed live execution admission result for an eval mode."""
|
|
692
|
+
deferred = list(descriptor.deferred_dependencies)
|
|
693
|
+
dependency_inputs = {
|
|
694
|
+
"model_backend_configuration": model_backend_configured,
|
|
695
|
+
"resource_ceiling_enforcement": resource_ceilings_enforceable,
|
|
696
|
+
"fixture_workspace_creation": fixture_workspace_creation_available,
|
|
697
|
+
}
|
|
698
|
+
if any(
|
|
699
|
+
binding.runner_kind == EvalRunnerKind.PI
|
|
700
|
+
for binding in descriptor.runner_bindings
|
|
701
|
+
):
|
|
702
|
+
dependency_inputs["pi_live_runtime_support"] = pi_live_runtime_available
|
|
703
|
+
if any(
|
|
704
|
+
binding.runner_kind == EvalRunnerKind.MILLFORGE
|
|
705
|
+
for binding in descriptor.runner_bindings
|
|
706
|
+
):
|
|
707
|
+
dependency_inputs["spec_07_harness_presets"] = spec_07_harness_presets_available
|
|
708
|
+
|
|
709
|
+
existing_ids = {dependency.dependency_id for dependency in deferred}
|
|
710
|
+
for dependency_id, available in dependency_inputs.items():
|
|
711
|
+
if not available and dependency_id not in existing_ids:
|
|
712
|
+
deferred.append(_deferred_dependency_for_id(dependency_id))
|
|
713
|
+
existing_ids.add(dependency_id)
|
|
714
|
+
|
|
715
|
+
unresolved = _unique_deferred_dependencies(tuple(deferred))
|
|
716
|
+
if unresolved:
|
|
717
|
+
return EvalModeLiveAdmissionResult(
|
|
718
|
+
mode_id=descriptor.mode_id,
|
|
719
|
+
admitted=False,
|
|
720
|
+
rule_id="eval.mode.live_admission.deferred_dependency",
|
|
721
|
+
diagnostic_code="MF-EVAL-M001",
|
|
722
|
+
diagnostic_summary="live eval mode execution has unresolved dependencies",
|
|
723
|
+
deferred_dependencies=unresolved,
|
|
724
|
+
)
|
|
725
|
+
return EvalModeLiveAdmissionResult(
|
|
726
|
+
mode_id=descriptor.mode_id,
|
|
727
|
+
admitted=True,
|
|
728
|
+
rule_id="eval.mode.live_admission.allowed",
|
|
729
|
+
)
|
|
730
|
+
|
|
731
|
+
|
|
732
|
+
def _default_eval_mode(
|
|
733
|
+
*,
|
|
734
|
+
mode_id: str,
|
|
735
|
+
runner_kind: EvalRunnerKind,
|
|
736
|
+
spec_07_static_presets_ready: bool = False,
|
|
737
|
+
) -> EvalModeDescriptor:
|
|
738
|
+
graph = default_compact_eval_workflow_graph()
|
|
739
|
+
workflow_snapshot = compact_eval_workflow_snapshot(graph)
|
|
740
|
+
context_policies = tuple(
|
|
741
|
+
default_eval_stage_context_policies()[stage_id] for stage_id in EvalStageId
|
|
742
|
+
)
|
|
743
|
+
descriptor = EvalModeDescriptor.model_construct(
|
|
744
|
+
schema_version=EVAL_MODE_SCHEMA_VERSION,
|
|
745
|
+
mode_id=mode_id,
|
|
746
|
+
description=_default_mode_description(mode_id),
|
|
747
|
+
runner_bindings=tuple(
|
|
748
|
+
EvalRunnerBinding(
|
|
749
|
+
stage_id=stage_id,
|
|
750
|
+
runner_kind=runner_kind,
|
|
751
|
+
harness_id=(
|
|
752
|
+
EVAL_SPEC_07_HARNESS_IDS[stage_id]
|
|
753
|
+
if runner_kind == EvalRunnerKind.MILLFORGE
|
|
754
|
+
else None
|
|
755
|
+
),
|
|
756
|
+
)
|
|
757
|
+
for stage_id in graph.stage_ids
|
|
758
|
+
),
|
|
759
|
+
model_profile=default_eval_model_profile(),
|
|
760
|
+
graph_id=graph.graph_id,
|
|
761
|
+
graph_sha256=workflow_snapshot["graph_sha256"],
|
|
762
|
+
stage_ids=graph.stage_ids,
|
|
763
|
+
stage_contracts=graph.stages,
|
|
764
|
+
terminal_results=tuple(EvalTerminalResult),
|
|
765
|
+
transition_semantics=tuple(workflow_snapshot["transitions"]),
|
|
766
|
+
boundary_baseline=compact_eval_boundary_baseline(),
|
|
767
|
+
capability_envelopes=tuple(
|
|
768
|
+
default_eval_capability_envelopes()[stage_id] for stage_id in EvalStageId
|
|
769
|
+
),
|
|
770
|
+
fixture_policy=EvalFixtureWorkspacePolicy(),
|
|
771
|
+
artifact_policy=_default_artifact_policy_tuple(),
|
|
772
|
+
context_tier=EvalContextTier.COMPACT,
|
|
773
|
+
stage_context_policies=context_policies,
|
|
774
|
+
redaction_categories=context_policies[0].redaction.categories,
|
|
775
|
+
validator_visibility_policy=_default_validator_visibility_policy(),
|
|
776
|
+
trial_resource_ceiling=default_eval_trial_resource_ceiling(),
|
|
777
|
+
closure_boundary_id=EVAL_CLOSURE_BOUNDARY_ID,
|
|
778
|
+
deferred_dependencies=_static_descriptor_deferred_dependencies(
|
|
779
|
+
runner_kind=runner_kind,
|
|
780
|
+
spec_07_static_presets_ready=spec_07_static_presets_ready,
|
|
781
|
+
),
|
|
782
|
+
declared_confounds=_default_declared_confounds(),
|
|
783
|
+
fairness_fingerprint_kind=EVAL_MODE_FAIRNESS_FINGERPRINT_KIND,
|
|
784
|
+
fairness_fingerprint="0" * 64,
|
|
785
|
+
descriptor_fingerprint_kind=EVAL_MODE_FINGERPRINT_KIND,
|
|
786
|
+
descriptor_fingerprint="0" * 64,
|
|
787
|
+
)
|
|
788
|
+
fairness_fingerprint = calculate_eval_mode_fairness_fingerprint(descriptor)
|
|
789
|
+
descriptor = descriptor.model_copy(
|
|
790
|
+
update={"fairness_fingerprint": fairness_fingerprint}
|
|
791
|
+
)
|
|
792
|
+
return EvalModeDescriptor.model_validate(
|
|
793
|
+
descriptor.model_copy(
|
|
794
|
+
update={
|
|
795
|
+
"descriptor_fingerprint": calculate_eval_mode_fingerprint(descriptor)
|
|
796
|
+
}
|
|
797
|
+
)
|
|
798
|
+
)
|
|
799
|
+
|
|
800
|
+
|
|
801
|
+
def _default_artifact_policy_tuple() -> tuple[EvalArtifactLayoutEntry, ...]:
|
|
802
|
+
layout = canonical_eval_artifact_layout()
|
|
803
|
+
return tuple(layout[artifact_id] for artifact_id in layout)
|
|
804
|
+
|
|
805
|
+
|
|
806
|
+
def _default_validator_visibility_policy() -> EvalValidatorVisibilityRecord:
|
|
807
|
+
return EvalValidatorVisibilityRecord(
|
|
808
|
+
visible_acceptance_check_ids=("public_acceptance_checks",)
|
|
809
|
+
)
|
|
810
|
+
|
|
811
|
+
|
|
812
|
+
def _default_mode_description(mode_id: str) -> str:
|
|
813
|
+
if mode_id == EVAL_SMALL_PI_MODE_ID:
|
|
814
|
+
return (
|
|
815
|
+
"Static compact eval descriptor for the Pi runner mode. The record "
|
|
816
|
+
"admits descriptor validation only and defers live Pi runtime support."
|
|
817
|
+
)
|
|
818
|
+
if mode_id == EVAL_SMALL_MILLFORGE_MODE_ID:
|
|
819
|
+
return (
|
|
820
|
+
"Static compact eval descriptor for the Millforge runner mode. The "
|
|
821
|
+
"record admits descriptor validation only and defers live Spec 07 "
|
|
822
|
+
"harness execution."
|
|
823
|
+
)
|
|
824
|
+
raise ValueError("unknown static eval mode id")
|
|
825
|
+
|
|
826
|
+
|
|
827
|
+
def _static_descriptor_deferred_dependencies(
|
|
828
|
+
*,
|
|
829
|
+
runner_kind: EvalRunnerKind,
|
|
830
|
+
spec_07_static_presets_ready: bool,
|
|
831
|
+
) -> tuple[EvalModeDeferredDependency, ...]:
|
|
832
|
+
if runner_kind == EvalRunnerKind.MILLFORGE:
|
|
833
|
+
if spec_07_static_presets_ready:
|
|
834
|
+
return ()
|
|
835
|
+
return (_deferred_dependency_for_id("spec_07_harness_presets"),)
|
|
836
|
+
if runner_kind == EvalRunnerKind.PI:
|
|
837
|
+
return (_deferred_dependency_for_id("pi_live_runtime_support"),)
|
|
838
|
+
return ()
|
|
839
|
+
|
|
840
|
+
|
|
841
|
+
def _fairness_payload(descriptor: EvalModeDescriptor) -> dict[str, Any]:
|
|
842
|
+
return {
|
|
843
|
+
"fairness_fingerprint_kind": EVAL_MODE_FAIRNESS_FINGERPRINT_KIND,
|
|
844
|
+
"schema_version": descriptor.schema_version,
|
|
845
|
+
"graph_id": descriptor.graph_id,
|
|
846
|
+
"graph_sha256": descriptor.graph_sha256,
|
|
847
|
+
"stage_ids": _json_value(descriptor.stage_ids),
|
|
848
|
+
"terminal_results": _json_value(descriptor.terminal_results),
|
|
849
|
+
"transition_semantics": _json_value(descriptor.transition_semantics),
|
|
850
|
+
"attempt_limits": tuple(
|
|
851
|
+
{
|
|
852
|
+
"stage_id": contract.stage_id.value,
|
|
853
|
+
"domain_attempt_limit": contract.domain_attempt_limit,
|
|
854
|
+
"infrastructure_retry_limit": contract.infrastructure_retry_limit,
|
|
855
|
+
"may_complete_workflow": contract.may_complete_workflow,
|
|
856
|
+
}
|
|
857
|
+
for contract in descriptor.stage_contracts
|
|
858
|
+
),
|
|
859
|
+
"capability_envelopes": _json_value(descriptor.capability_envelopes),
|
|
860
|
+
"fixture_policy": _json_value(descriptor.fixture_policy),
|
|
861
|
+
"artifact_policy": _json_value(descriptor.artifact_policy),
|
|
862
|
+
"context_tier": descriptor.context_tier.value,
|
|
863
|
+
"stage_context_policies": _json_value(descriptor.stage_context_policies),
|
|
864
|
+
"redaction_categories": _json_value(descriptor.redaction_categories),
|
|
865
|
+
"validator_visibility_policy": _json_value(
|
|
866
|
+
descriptor.validator_visibility_policy
|
|
867
|
+
),
|
|
868
|
+
"visible_acceptance_check_policy": _json_value(
|
|
869
|
+
descriptor.validator_visibility_policy.visible_acceptance_check_ids
|
|
870
|
+
),
|
|
871
|
+
"trial_resource_ceiling": _json_value(descriptor.trial_resource_ceiling),
|
|
872
|
+
"closure_boundary_id": descriptor.closure_boundary_id,
|
|
873
|
+
"model_profile_hash": descriptor.model_profile.model_profile_hash,
|
|
874
|
+
}
|
|
875
|
+
|
|
876
|
+
|
|
877
|
+
def _json_value(value: Any) -> Any:
|
|
878
|
+
if isinstance(value, BaseModel):
|
|
879
|
+
return value.model_dump(mode="json")
|
|
880
|
+
if isinstance(value, Enum):
|
|
881
|
+
return value.value
|
|
882
|
+
if isinstance(value, Mapping):
|
|
883
|
+
return {key: _json_value(child) for key, child in value.items()}
|
|
884
|
+
if isinstance(value, (tuple, list)):
|
|
885
|
+
return [_json_value(child) for child in value]
|
|
886
|
+
return value
|
|
887
|
+
|
|
888
|
+
|
|
889
|
+
def _allowed_runner_differences(
|
|
890
|
+
left: EvalModeDescriptor, right: EvalModeDescriptor
|
|
891
|
+
) -> tuple[EvalModeComparison, ...]:
|
|
892
|
+
left_payload = left.model_dump(mode="json")
|
|
893
|
+
right_payload = right.model_dump(mode="json")
|
|
894
|
+
differences: list[EvalModeComparison] = []
|
|
895
|
+
for field in _ALLOWED_RUNNER_DIFFERENCE_FIELDS:
|
|
896
|
+
if left_payload[field] != right_payload[field]:
|
|
897
|
+
differences.append(
|
|
898
|
+
EvalModeComparison(
|
|
899
|
+
field_path=field,
|
|
900
|
+
summary=_allowed_difference_summary(field),
|
|
901
|
+
left_value=left_payload[field],
|
|
902
|
+
right_value=right_payload[field],
|
|
903
|
+
)
|
|
904
|
+
)
|
|
905
|
+
return tuple(differences)
|
|
906
|
+
|
|
907
|
+
|
|
908
|
+
def _disallowed_fairness_differences(
|
|
909
|
+
left: EvalModeDescriptor, right: EvalModeDescriptor
|
|
910
|
+
) -> tuple[EvalModeComparison, ...]:
|
|
911
|
+
left_payload = _fairness_payload(left)
|
|
912
|
+
right_payload = _fairness_payload(right)
|
|
913
|
+
return tuple(
|
|
914
|
+
EvalModeComparison(
|
|
915
|
+
field_path=field,
|
|
916
|
+
summary=f"fairness-critical {field} differs",
|
|
917
|
+
left_value=left_payload[field],
|
|
918
|
+
right_value=right_payload[field],
|
|
919
|
+
)
|
|
920
|
+
for field in left_payload
|
|
921
|
+
if left_payload[field] != right_payload[field]
|
|
922
|
+
)
|
|
923
|
+
|
|
924
|
+
|
|
925
|
+
def _allowed_difference_summary(field: str) -> str:
|
|
926
|
+
if field == "runner_bindings":
|
|
927
|
+
return (
|
|
928
|
+
"runner kind, runner-specific binding ID, Millforge harness IDs, "
|
|
929
|
+
"Pi adapter descriptor, runner-internal prompts or templates, and "
|
|
930
|
+
"runner-internal tool-schema rendering are runner-specific"
|
|
931
|
+
)
|
|
932
|
+
if field == "deferred_dependencies":
|
|
933
|
+
return "runner-specific live dependencies are reported as deferred diagnostics"
|
|
934
|
+
return "mode identity is runner-specific"
|
|
935
|
+
|
|
936
|
+
|
|
937
|
+
def _default_confounds_for_dependencies(
|
|
938
|
+
dependencies: tuple[EvalModeDeferredDependency, ...],
|
|
939
|
+
) -> tuple[EvalModeConfound, ...]:
|
|
940
|
+
confounds: list[EvalModeConfound] = [
|
|
941
|
+
_confound_record(
|
|
942
|
+
confound_id="runner_capability_enforcement",
|
|
943
|
+
kind="runner_capability_enforcement",
|
|
944
|
+
severity="warning",
|
|
945
|
+
summary="Pi and Millforge live runners may enforce capabilities differently",
|
|
946
|
+
evidence=(
|
|
947
|
+
"static descriptors bind different runner kinds",
|
|
948
|
+
"capability envelopes are declared but no live runner is admitted",
|
|
949
|
+
),
|
|
950
|
+
comparison_effect="live capability enforcement parity is unproven",
|
|
951
|
+
mitigation="treat comparison as engineering smoke until live enforcement is measured",
|
|
952
|
+
),
|
|
953
|
+
_confound_record(
|
|
954
|
+
confound_id="tool_calling",
|
|
955
|
+
kind="tool_calling",
|
|
956
|
+
severity="warning",
|
|
957
|
+
summary="runner tool-call behavior is not measured by static descriptors",
|
|
958
|
+
evidence=(
|
|
959
|
+
"model profile declares parser-mediated tool calling",
|
|
960
|
+
"runner-internal tool schema rendering is runner-specific",
|
|
961
|
+
),
|
|
962
|
+
comparison_effect="tool-call success and repair behavior may differ by runner",
|
|
963
|
+
mitigation="compare live tool-call traces before controlled scoring",
|
|
964
|
+
),
|
|
965
|
+
_confound_record(
|
|
966
|
+
confound_id="parser_fallback",
|
|
967
|
+
kind="parser_fallback",
|
|
968
|
+
severity="info",
|
|
969
|
+
summary="parser fallback behavior is backend neutral but unmeasured",
|
|
970
|
+
evidence=(
|
|
971
|
+
"model profile uses millforge.eval.parser.compact_json.v1",
|
|
972
|
+
"no live malformed-output fallback evidence is present",
|
|
973
|
+
),
|
|
974
|
+
comparison_effect="fallback frequency could affect live outcomes",
|
|
975
|
+
mitigation="record parser fallback counts in live trial artifacts",
|
|
976
|
+
),
|
|
977
|
+
_confound_record(
|
|
978
|
+
confound_id="context_packing",
|
|
979
|
+
kind="context_packing",
|
|
980
|
+
severity="warning",
|
|
981
|
+
summary="live context packing differences remain unmeasured",
|
|
982
|
+
evidence=(
|
|
983
|
+
"compact context policies are statically equal",
|
|
984
|
+
"runner-specific prompt packing is outside the descriptor payload",
|
|
985
|
+
),
|
|
986
|
+
comparison_effect="prompt shape parity is unproven at execution time",
|
|
987
|
+
mitigation="capture and compare redacted context snapshots per stage",
|
|
988
|
+
),
|
|
989
|
+
_confound_record(
|
|
990
|
+
confound_id="model_endpoint",
|
|
991
|
+
kind="model_endpoint",
|
|
992
|
+
severity="invalidating",
|
|
993
|
+
summary="no live model endpoint configuration is admitted",
|
|
994
|
+
evidence=(
|
|
995
|
+
"model backend configuration is a live deferred dependency",
|
|
996
|
+
"descriptor contains no endpoint or private host material",
|
|
997
|
+
),
|
|
998
|
+
comparison_effect="live score comparison is invalid without a shared endpoint",
|
|
999
|
+
mitigation="admit live execution only after backend configuration is explicit",
|
|
1000
|
+
),
|
|
1001
|
+
_confound_record(
|
|
1002
|
+
confound_id="token_accounting",
|
|
1003
|
+
kind="token_accounting",
|
|
1004
|
+
severity="warning",
|
|
1005
|
+
summary="live token accounting is not measured by static descriptors",
|
|
1006
|
+
evidence=(
|
|
1007
|
+
"model profile declares token ceilings",
|
|
1008
|
+
"no runtime token usage records are included in static descriptors",
|
|
1009
|
+
),
|
|
1010
|
+
comparison_effect="cost and truncation behavior may differ in live runs",
|
|
1011
|
+
mitigation="compare model usage artifacts before controlled scoring",
|
|
1012
|
+
),
|
|
1013
|
+
_confound_record(
|
|
1014
|
+
confound_id="wall_clock_measurement",
|
|
1015
|
+
kind="wall_clock_measurement",
|
|
1016
|
+
severity="info",
|
|
1017
|
+
summary="wall-clock timing is outside static descriptor comparison",
|
|
1018
|
+
evidence=(
|
|
1019
|
+
"resource ceilings declare wall-clock limits",
|
|
1020
|
+
"static descriptors contain no measured durations",
|
|
1021
|
+
),
|
|
1022
|
+
comparison_effect="timing parity cannot be inferred from descriptors",
|
|
1023
|
+
mitigation="compare runtime resource usage artifacts for live trials",
|
|
1024
|
+
),
|
|
1025
|
+
]
|
|
1026
|
+
dependency_ids = {dependency.dependency_id for dependency in dependencies}
|
|
1027
|
+
if "spec_07_harness_presets" in dependency_ids:
|
|
1028
|
+
confounds.append(
|
|
1029
|
+
EvalModeConfound(
|
|
1030
|
+
confound_id="deferred_millforge_harness",
|
|
1031
|
+
kind="deferred_millforge_harness",
|
|
1032
|
+
severity="invalidating",
|
|
1033
|
+
summary="Spec 07 Millforge harness live execution is deferred",
|
|
1034
|
+
applies_to=(EVAL_SMALL_MILLFORGE_MODE_ID,),
|
|
1035
|
+
evidence=_spec_07_harness_readiness_evidence(),
|
|
1036
|
+
comparison_effect=(
|
|
1037
|
+
"Millforge live readiness is blocked until harness execution "
|
|
1038
|
+
"dependencies are admitted"
|
|
1039
|
+
),
|
|
1040
|
+
mitigation=(
|
|
1041
|
+
"admit Spec 07 harness execution dependencies before live runs"
|
|
1042
|
+
),
|
|
1043
|
+
)
|
|
1044
|
+
)
|
|
1045
|
+
if "pi_live_runtime_support" in dependency_ids:
|
|
1046
|
+
confounds.append(
|
|
1047
|
+
EvalModeConfound(
|
|
1048
|
+
confound_id="deferred_pi_runtime",
|
|
1049
|
+
kind="deferred_pi_runtime",
|
|
1050
|
+
severity="invalidating",
|
|
1051
|
+
summary="Pi live runtime support is deferred",
|
|
1052
|
+
applies_to=(EVAL_SMALL_PI_MODE_ID,),
|
|
1053
|
+
evidence=("pi_live_runtime_support deferred dependency is unresolved",),
|
|
1054
|
+
comparison_effect="Pi live readiness is blocked by missing runtime support",
|
|
1055
|
+
mitigation="implement Pi live runtime support before live runs",
|
|
1056
|
+
)
|
|
1057
|
+
)
|
|
1058
|
+
return tuple(confounds)
|
|
1059
|
+
|
|
1060
|
+
|
|
1061
|
+
def _default_declared_confounds() -> tuple[EvalModeConfound, ...]:
|
|
1062
|
+
return _default_confounds_for_dependencies(
|
|
1063
|
+
(
|
|
1064
|
+
_deferred_dependency_for_id("spec_07_harness_presets"),
|
|
1065
|
+
_deferred_dependency_for_id("pi_live_runtime_support"),
|
|
1066
|
+
)
|
|
1067
|
+
)
|
|
1068
|
+
|
|
1069
|
+
|
|
1070
|
+
def _confound_record(
|
|
1071
|
+
*,
|
|
1072
|
+
confound_id: str,
|
|
1073
|
+
kind: str,
|
|
1074
|
+
severity: str,
|
|
1075
|
+
summary: str,
|
|
1076
|
+
evidence: tuple[str, ...],
|
|
1077
|
+
comparison_effect: str,
|
|
1078
|
+
mitigation: str,
|
|
1079
|
+
) -> EvalModeConfound:
|
|
1080
|
+
return EvalModeConfound(
|
|
1081
|
+
confound_id=confound_id,
|
|
1082
|
+
kind=kind,
|
|
1083
|
+
severity=severity,
|
|
1084
|
+
summary=summary,
|
|
1085
|
+
applies_to=(EVAL_SMALL_PI_MODE_ID, EVAL_SMALL_MILLFORGE_MODE_ID),
|
|
1086
|
+
evidence=evidence,
|
|
1087
|
+
comparison_effect=comparison_effect,
|
|
1088
|
+
mitigation=mitigation,
|
|
1089
|
+
)
|
|
1090
|
+
|
|
1091
|
+
|
|
1092
|
+
def _unique_deferred_dependencies(
|
|
1093
|
+
dependencies: tuple[EvalModeDeferredDependency, ...],
|
|
1094
|
+
) -> tuple[EvalModeDeferredDependency, ...]:
|
|
1095
|
+
by_id: dict[str, EvalModeDeferredDependency] = {}
|
|
1096
|
+
for dependency in dependencies:
|
|
1097
|
+
by_id.setdefault(dependency.dependency_id, dependency)
|
|
1098
|
+
return tuple(by_id[dependency_id] for dependency_id in sorted(by_id))
|
|
1099
|
+
|
|
1100
|
+
|
|
1101
|
+
def _deferred_dependency_for_id(dependency_id: str) -> EvalModeDeferredDependency:
|
|
1102
|
+
records = {
|
|
1103
|
+
"pi_live_runtime_support": {
|
|
1104
|
+
"dependency_kind": "runner_runtime",
|
|
1105
|
+
"affected_mode": EVAL_SMALL_PI_MODE_ID,
|
|
1106
|
+
"summary": "Pi live runtime support is not implemented",
|
|
1107
|
+
"reference_ids": (),
|
|
1108
|
+
},
|
|
1109
|
+
"spec_07_harness_presets": {
|
|
1110
|
+
"dependency_kind": "runner_harness",
|
|
1111
|
+
"affected_mode": EVAL_SMALL_MILLFORGE_MODE_ID,
|
|
1112
|
+
"summary": (
|
|
1113
|
+
"Spec 07 harness source records are available; live harness "
|
|
1114
|
+
"execution is not admitted"
|
|
1115
|
+
),
|
|
1116
|
+
"reference_ids": _spec_07_implemented_harness_ids(),
|
|
1117
|
+
},
|
|
1118
|
+
"model_backend_configuration": {
|
|
1119
|
+
"dependency_kind": "model_backend",
|
|
1120
|
+
"affected_mode": "all_modes",
|
|
1121
|
+
"summary": "Live model backend configuration is missing",
|
|
1122
|
+
"reference_ids": (),
|
|
1123
|
+
},
|
|
1124
|
+
"resource_ceiling_enforcement": {
|
|
1125
|
+
"dependency_kind": "resource_enforcement",
|
|
1126
|
+
"affected_mode": "all_modes",
|
|
1127
|
+
"summary": "Resource ceilings cannot yet be enforced for live eval execution",
|
|
1128
|
+
"reference_ids": (),
|
|
1129
|
+
},
|
|
1130
|
+
"fixture_workspace_creation": {
|
|
1131
|
+
"dependency_kind": "fixture_workspace",
|
|
1132
|
+
"affected_mode": "all_modes",
|
|
1133
|
+
"summary": "Fixture workspace creation is unavailable",
|
|
1134
|
+
"reference_ids": (),
|
|
1135
|
+
},
|
|
1136
|
+
}
|
|
1137
|
+
if dependency_id not in _LIVE_DEPENDENCY_IDS:
|
|
1138
|
+
raise ValueError("unknown eval mode deferred dependency id")
|
|
1139
|
+
record = records[dependency_id]
|
|
1140
|
+
return EvalModeDeferredDependency(
|
|
1141
|
+
dependency_id=dependency_id,
|
|
1142
|
+
dependency_kind=str(record["dependency_kind"]),
|
|
1143
|
+
affected_mode=str(record["affected_mode"]),
|
|
1144
|
+
affected_stage_id=None,
|
|
1145
|
+
all_stage_scope=True,
|
|
1146
|
+
static_descriptor_admission_behavior="static descriptor validation passes",
|
|
1147
|
+
live_execution_behavior="live execution admission fails closed until resolved",
|
|
1148
|
+
summary=str(record["summary"]),
|
|
1149
|
+
reference_ids=tuple(record["reference_ids"]),
|
|
1150
|
+
)
|
|
1151
|
+
|
|
1152
|
+
|
|
1153
|
+
def _spec_07_implemented_harness_ids() -> tuple[str, ...]:
|
|
1154
|
+
return tuple(
|
|
1155
|
+
EVAL_SPEC_07_HARNESS_IDS[stage_id]
|
|
1156
|
+
for stage_id in _EVAL_SPEC_07_IMPLEMENTED_HARNESS_STAGES
|
|
1157
|
+
)
|
|
1158
|
+
|
|
1159
|
+
|
|
1160
|
+
def _spec_07_harness_readiness_evidence() -> tuple[str, ...]:
|
|
1161
|
+
planner_id, builder_id, checker_id, arbiter_id = _spec_07_implemented_harness_ids()
|
|
1162
|
+
return (
|
|
1163
|
+
f"Planner source record is implemented: {planner_id}",
|
|
1164
|
+
f"Builder source record is implemented: {builder_id}",
|
|
1165
|
+
f"Checker source record is implemented: {checker_id}",
|
|
1166
|
+
f"Arbiter source record is implemented: {arbiter_id}",
|
|
1167
|
+
"live Spec 07 harness execution remains unresolved",
|
|
1168
|
+
)
|
|
1169
|
+
|
|
1170
|
+
|
|
1171
|
+
def _validate_runner_bindings(
|
|
1172
|
+
mode_id: str, runner_bindings: tuple[EvalRunnerBinding, ...]
|
|
1173
|
+
) -> None:
|
|
1174
|
+
graph_stage_ids = default_compact_eval_workflow_graph().stage_ids
|
|
1175
|
+
if tuple(binding.stage_id for binding in runner_bindings) != graph_stage_ids:
|
|
1176
|
+
raise ValueError("runner bindings must cover exactly the compact eval stages")
|
|
1177
|
+
expected_kind = (
|
|
1178
|
+
EvalRunnerKind.PI
|
|
1179
|
+
if mode_id == EVAL_SMALL_PI_MODE_ID
|
|
1180
|
+
else EvalRunnerKind.MILLFORGE
|
|
1181
|
+
)
|
|
1182
|
+
for binding in runner_bindings:
|
|
1183
|
+
if binding.runner_kind != expected_kind:
|
|
1184
|
+
raise ValueError("runner binding kind does not match eval mode id")
|
|
1185
|
+
|
|
1186
|
+
|
|
1187
|
+
def _canonical_eval_json_bytes(value: Any) -> bytes:
|
|
1188
|
+
return (
|
|
1189
|
+
json.dumps(
|
|
1190
|
+
value,
|
|
1191
|
+
sort_keys=True,
|
|
1192
|
+
ensure_ascii=True,
|
|
1193
|
+
allow_nan=False,
|
|
1194
|
+
separators=(",", ":"),
|
|
1195
|
+
).replace("\r\n", "\n")
|
|
1196
|
+
+ "\n"
|
|
1197
|
+
).encode("ascii")
|
|
1198
|
+
|
|
1199
|
+
|
|
1200
|
+
def _freeze_descriptor_mapping(value: Mapping[Any, Any]) -> Mapping[Any, Any]:
|
|
1201
|
+
return _FrozenDescriptorDict(
|
|
1202
|
+
{key: _freeze_descriptor_value(child) for key, child in value.items()}
|
|
1203
|
+
)
|
|
1204
|
+
|
|
1205
|
+
|
|
1206
|
+
def _freeze_descriptor_value(value: Any) -> Any:
|
|
1207
|
+
if isinstance(value, Mapping):
|
|
1208
|
+
return _freeze_descriptor_mapping(value)
|
|
1209
|
+
if isinstance(value, tuple):
|
|
1210
|
+
return tuple(_freeze_descriptor_value(child) for child in value)
|
|
1211
|
+
if isinstance(value, list):
|
|
1212
|
+
return tuple(_freeze_descriptor_value(child) for child in value)
|
|
1213
|
+
return value
|
|
1214
|
+
|
|
1215
|
+
|
|
1216
|
+
def _descriptor_text_values(value: Any) -> tuple[str, ...]:
|
|
1217
|
+
if isinstance(value, str):
|
|
1218
|
+
return (value,)
|
|
1219
|
+
if isinstance(value, Mapping):
|
|
1220
|
+
return tuple(
|
|
1221
|
+
text
|
|
1222
|
+
for item in value.items()
|
|
1223
|
+
for child in item
|
|
1224
|
+
for text in _descriptor_text_values(child)
|
|
1225
|
+
)
|
|
1226
|
+
if isinstance(value, (tuple, list, set, frozenset)):
|
|
1227
|
+
return tuple(text for child in value for text in _descriptor_text_values(child))
|
|
1228
|
+
return ()
|
|
1229
|
+
|
|
1230
|
+
|
|
1231
|
+
def _reject_descriptor_material_leaks(value: Any) -> None:
|
|
1232
|
+
for text in _descriptor_text_values(value):
|
|
1233
|
+
lowered = text.lower()
|
|
1234
|
+
for token in _DENIED_DESCRIPTOR_TOKENS:
|
|
1235
|
+
if token in lowered:
|
|
1236
|
+
raise ValueError(
|
|
1237
|
+
"eval mode descriptors must not expose private material"
|
|
1238
|
+
)
|
|
1239
|
+
if (
|
|
1240
|
+
_WINDOWS_ABSOLUTE_PATH.search(text)
|
|
1241
|
+
or _POSIX_ABSOLUTE_PATH.search(text)
|
|
1242
|
+
or _USER_HOME_PATH.search(text)
|
|
1243
|
+
):
|
|
1244
|
+
raise ValueError("eval mode descriptors must not expose host paths")
|
|
1245
|
+
|
|
1246
|
+
|
|
1247
|
+
def _validate_sha256(value: str) -> None:
|
|
1248
|
+
if len(value) != 64 or any(
|
|
1249
|
+
character not in "0123456789abcdef" for character in value
|
|
1250
|
+
):
|
|
1251
|
+
raise ValueError("sha256 must be a lowercase hexadecimal SHA-256 digest")
|
|
1252
|
+
|
|
1253
|
+
|
|
1254
|
+
__all__ = [
|
|
1255
|
+
"EVAL_DEFAULT_MODEL_PROFILE_ID",
|
|
1256
|
+
"EVAL_MODE_FINGERPRINT_KIND",
|
|
1257
|
+
"EVAL_MODE_FAIRNESS_FINGERPRINT_KIND",
|
|
1258
|
+
"EVAL_MODE_SCHEMA_VERSION",
|
|
1259
|
+
"EVAL_MODEL_PROFILE_HASH_KIND",
|
|
1260
|
+
"EVAL_SMALL_MILLFORGE_MODE_ID",
|
|
1261
|
+
"EVAL_SMALL_PI_MODE_ID",
|
|
1262
|
+
"EVAL_SPEC_07_HARNESS_IDS",
|
|
1263
|
+
"EvalModeComparison",
|
|
1264
|
+
"EvalModeConfound",
|
|
1265
|
+
"EvalModeDeferredDependency",
|
|
1266
|
+
"EvalModeDescriptor",
|
|
1267
|
+
"EvalModeFairnessReport",
|
|
1268
|
+
"EvalModeLiveAdmissionResult",
|
|
1269
|
+
"EvalModelProfile",
|
|
1270
|
+
"EvalRunnerBinding",
|
|
1271
|
+
"EvalRunnerKind",
|
|
1272
|
+
"admit_eval_mode_live_execution",
|
|
1273
|
+
"calculate_eval_mode_fairness_fingerprint",
|
|
1274
|
+
"calculate_eval_mode_fingerprint",
|
|
1275
|
+
"calculate_eval_model_profile_hash",
|
|
1276
|
+
"canonical_eval_mode_fairness_bytes",
|
|
1277
|
+
"canonical_eval_mode_bytes",
|
|
1278
|
+
"compare_eval_modes_for_fairness",
|
|
1279
|
+
"default_eval_model_profile",
|
|
1280
|
+
"default_eval_small_millforge_mode",
|
|
1281
|
+
"default_eval_small_pi_mode",
|
|
1282
|
+
]
|