millforge 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- millforge/__init__.py +1174 -0
- millforge/_forge/LICENSE +21 -0
- millforge/_forge/PROVENANCE.json +295 -0
- millforge/_forge/UPDATE_POLICY.md +24 -0
- millforge/_forge/__init__.py +14 -0
- millforge/_forge/adapter.py +2232 -0
- millforge/_forge/base_runner.py +121 -0
- millforge/_forge/clients/__init__.py +10 -0
- millforge/_forge/clients/base.py +200 -0
- millforge/_forge/context/__init__.py +23 -0
- millforge/_forge/context/manager.py +178 -0
- millforge/_forge/context/strategies.py +335 -0
- millforge/_forge/core/__init__.py +16 -0
- millforge/_forge/core/inference.py +433 -0
- millforge/_forge/core/messages.py +119 -0
- millforge/_forge/core/runner.py +479 -0
- millforge/_forge/core/steps.py +108 -0
- millforge/_forge/core/workflow.py +400 -0
- millforge/_forge/errors.py +222 -0
- millforge/_forge/guardrails/__init__.py +21 -0
- millforge/_forge/guardrails/error_tracker.py +71 -0
- millforge/_forge/guardrails/guardrails.py +194 -0
- millforge/_forge/guardrails/nudge.py +47 -0
- millforge/_forge/guardrails/response_validator.py +119 -0
- millforge/_forge/guardrails/step_enforcer.py +183 -0
- millforge/_forge/prompts/__init__.py +16 -0
- millforge/_forge/prompts/nudges.py +95 -0
- millforge/_forge/prompts/templates.py +285 -0
- millforge/_version.py +3 -0
- millforge/artifacts.py +570 -0
- millforge/base/__init__.py +97 -0
- millforge/base/composition.py +402 -0
- millforge/base/context.py +285 -0
- millforge/base/harness.py +138 -0
- millforge/base/identity.py +465 -0
- millforge/base/options.py +34 -0
- millforge/base/platform.py +17 -0
- millforge/base/prompt.py +317 -0
- millforge/base/runner.py +546 -0
- millforge/compiled_plan.py +970 -0
- millforge/compiler/__init__.py +231 -0
- millforge/compiler/artifact_validation.py +257 -0
- millforge/compiler/canonicalization.py +169 -0
- millforge/compiler/capabilities.py +66 -0
- millforge/compiler/catalogs.py +500 -0
- millforge/compiler/diagnostics.py +491 -0
- millforge/compiler/graph.py +678 -0
- millforge/compiler/lowering.py +198 -0
- millforge/compiler/output.py +692 -0
- millforge/compiler/parsing.py +1424 -0
- millforge/compiler/requests.py +1180 -0
- millforge/compiler/schema_validation.py +272 -0
- millforge/compiler/semantic.py +490 -0
- millforge/compiler/service.py +448 -0
- millforge/compiler/source.py +375 -0
- millforge/compiler/validators.py +184 -0
- millforge/connectors/__init__.py +95 -0
- millforge/connectors/admission.py +801 -0
- millforge/connectors/broker.py +202 -0
- millforge/connectors/contracts.py +1159 -0
- millforge/connectors/diagnostics.py +189 -0
- millforge/connectors/fake.py +66 -0
- millforge/connectors/runtime.py +236 -0
- millforge/contracts.py +2860 -0
- millforge/custom_tools/__init__.py +67 -0
- millforge/custom_tools/compiler.py +724 -0
- millforge/custom_tools/contracts.py +1093 -0
- millforge/custom_tools/diagnostics.py +205 -0
- millforge/eval_artifacts.py +952 -0
- millforge/eval_boundary.py +2435 -0
- millforge/eval_fixtures/__init__.py +1 -0
- millforge/eval_fixtures/default_pack/__init__.py +1 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.bug_diagnosis.traceback.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.direct_edit.import_sort.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.evidence_discipline.no_source_change.v1.json +51 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.false_closure.visible_green.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.multi_file.api_contract.v1.json +54 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.recovery.malformed_artifact.v1.json +54 -0
- millforge/eval_fixtures/default_pack/manifest.json +12 -0
- millforge/eval_modes.py +1282 -0
- millforge/eval_presets.py +1398 -0
- millforge/eval_reports.py +2517 -0
- millforge/eval_suite.py +2429 -0
- millforge/eval_trials.py +2632 -0
- millforge/eval_workflow.py +794 -0
- millforge/exceptions.py +122 -0
- millforge/model_backend.py +2098 -0
- millforge/protocols.py +340 -0
- millforge/py.typed +0 -0
- millforge/runtime.py +1791 -0
- millforge/testing/__init__.py +1089 -0
- millforge/tools/__init__.py +83 -0
- millforge/tools/builtin_runtime.py +1339 -0
- millforge/tools/builtins.py +773 -0
- millforge/tools/execution.py +1545 -0
- millforge/tools/path_policy.py +155 -0
- millforge/tools/pi_compat/PI_LICENSE +21 -0
- millforge/tools/pi_compat/PROVENANCE.json +55 -0
- millforge/tools/pi_compat/UPDATE_POLICY.md +36 -0
- millforge/tools/pi_compat/__init__.py +34 -0
- millforge/tools/pi_compat/contracts.py +49 -0
- millforge/tools/pi_compat/editing.py +390 -0
- millforge/tools/pi_compat/mutations.py +57 -0
- millforge/tools/pi_compat/operations.py +401 -0
- millforge/tools/pi_compat/paths.py +155 -0
- millforge/tools/pi_compat/process.py +1375 -0
- millforge/tools/pi_compat/search.py +738 -0
- millforge/tools/pi_compat/truncation.py +267 -0
- millforge/tools/pi_compat_catalog.py +396 -0
- millforge/tools/pi_compat_runtime.py +460 -0
- millforge/tools/registry.py +553 -0
- millforge/tools/results.py +533 -0
- millforge-0.1.0.dist-info/METADATA +844 -0
- millforge-0.1.0.dist-info/RECORD +116 -0
- millforge-0.1.0.dist-info/WHEEL +4 -0
- millforge-0.1.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,2435 @@
|
|
|
1
|
+
"""Public 06B baseline boundary anchored to the compact eval workflow."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
from collections.abc import Mapping
|
|
8
|
+
from enum import Enum
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from pathlib import PurePosixPath
|
|
11
|
+
import re
|
|
12
|
+
from types import MappingProxyType
|
|
13
|
+
from typing import Any, cast
|
|
14
|
+
|
|
15
|
+
from pydantic import (
|
|
16
|
+
BaseModel,
|
|
17
|
+
ConfigDict,
|
|
18
|
+
Field,
|
|
19
|
+
StrictBool,
|
|
20
|
+
StrictInt,
|
|
21
|
+
StrictStr,
|
|
22
|
+
field_serializer,
|
|
23
|
+
field_validator,
|
|
24
|
+
model_validator,
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
from millforge.eval_workflow import (
|
|
28
|
+
EvalCandidateDisposition,
|
|
29
|
+
EvalStageId,
|
|
30
|
+
EvalTerminalResult,
|
|
31
|
+
EvalWorkflowOutcomeKind,
|
|
32
|
+
compact_eval_workflow_snapshot,
|
|
33
|
+
default_compact_eval_workflow_graph,
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
AUTHORITATIVE_COMPACT_EVAL_WORKFLOW_NAMES: tuple[str, ...] = (
|
|
37
|
+
"EvalStageId",
|
|
38
|
+
"EvalTerminalResult",
|
|
39
|
+
"EvalWorkflowOutcomeKind",
|
|
40
|
+
"EvalCandidateDisposition",
|
|
41
|
+
"default_compact_eval_workflow_graph",
|
|
42
|
+
"compact_eval_workflow_snapshot",
|
|
43
|
+
)
|
|
44
|
+
EVAL_BOUNDARY_MODULE_NAME = "millforge.eval_boundary"
|
|
45
|
+
EVAL_BOUNDARY_ARTIFACT_MODULE_REQUIRED = True
|
|
46
|
+
EVAL_BOUNDARY_ARTIFACT_MODULE_DECISION = (
|
|
47
|
+
"06B artifact schemas are defined in millforge.eval_artifacts so "
|
|
48
|
+
"millforge.eval_boundary remains focused on capability and fixture policy."
|
|
49
|
+
)
|
|
50
|
+
EVAL_DENIED_CAPABILITY_IDS: tuple[str, ...] = (
|
|
51
|
+
"network.access",
|
|
52
|
+
"package.install",
|
|
53
|
+
"git.mutate",
|
|
54
|
+
"runtime.control",
|
|
55
|
+
)
|
|
56
|
+
EVAL_BUILDER_DEFAULT_WRITE_ROOTS: tuple[str, ...] = (
|
|
57
|
+
"src",
|
|
58
|
+
"tests",
|
|
59
|
+
"README.md",
|
|
60
|
+
"ROADMAP.md",
|
|
61
|
+
)
|
|
62
|
+
EVAL_CHECKER_IGNORED_SCRATCH_ROOTS: tuple[str, ...] = (
|
|
63
|
+
".eval-scratch",
|
|
64
|
+
".pytest_cache",
|
|
65
|
+
)
|
|
66
|
+
EVAL_FIXTURE_IGNORED_GENERATED_ROOTS: tuple[str, ...] = (
|
|
67
|
+
".eval-scratch",
|
|
68
|
+
".mypy_cache",
|
|
69
|
+
".pytest_cache",
|
|
70
|
+
".ruff_cache",
|
|
71
|
+
"__pycache__",
|
|
72
|
+
)
|
|
73
|
+
EVAL_FIXTURE_IGNORED_GENERATED_SUFFIXES: tuple[str, ...] = (
|
|
74
|
+
".coverage",
|
|
75
|
+
".coverage.json",
|
|
76
|
+
".log",
|
|
77
|
+
".pyc",
|
|
78
|
+
".pyo",
|
|
79
|
+
)
|
|
80
|
+
EVAL_FIXTURE_MANIFEST_SCHEMA_VERSION = 1
|
|
81
|
+
EVAL_FIXTURE_DEFAULT_REVISION = "fixture-revision-1"
|
|
82
|
+
EVAL_FIXTURE_FILE_ROLES: tuple[str, ...] = (
|
|
83
|
+
"source",
|
|
84
|
+
"test",
|
|
85
|
+
"documentation",
|
|
86
|
+
"configuration",
|
|
87
|
+
"data",
|
|
88
|
+
)
|
|
89
|
+
EVAL_WORKSPACE_ISOLATION_CONTRACT = "fresh_copy"
|
|
90
|
+
EVAL_CONTEXT_FINGERPRINT_KIND = "eval_context_snapshot_sha256_v1"
|
|
91
|
+
EVAL_CONTEXT_DEFAULT_REDACTION_CATEGORIES: tuple[str, ...] = (
|
|
92
|
+
"scorer_only_material",
|
|
93
|
+
"secrets",
|
|
94
|
+
"daemon_state",
|
|
95
|
+
"host_paths",
|
|
96
|
+
"runtime_private_state",
|
|
97
|
+
"ideas_private_state",
|
|
98
|
+
"reference_private_state",
|
|
99
|
+
"git_history",
|
|
100
|
+
"private_conversations",
|
|
101
|
+
"unrelated_repository_outlines",
|
|
102
|
+
)
|
|
103
|
+
EVAL_CONTEXT_DEFAULT_REQUIRED_ARTIFACT_IDS: Mapping[EvalStageId, tuple[str, ...]] = (
|
|
104
|
+
MappingProxyType(
|
|
105
|
+
{
|
|
106
|
+
EvalStageId.PLANNER: ("task", "fixture_manifest", "acceptance_checks"),
|
|
107
|
+
EvalStageId.BUILDER: (
|
|
108
|
+
"task",
|
|
109
|
+
"fixture_manifest",
|
|
110
|
+
"acceptance_checks",
|
|
111
|
+
"plan",
|
|
112
|
+
),
|
|
113
|
+
EvalStageId.CHECKER: (
|
|
114
|
+
"fixture_manifest",
|
|
115
|
+
"acceptance_checks",
|
|
116
|
+
"workspace_diff",
|
|
117
|
+
"patch_summary",
|
|
118
|
+
"test_results",
|
|
119
|
+
),
|
|
120
|
+
EvalStageId.ARBITER: (
|
|
121
|
+
"acceptance_checks",
|
|
122
|
+
"workspace_diff",
|
|
123
|
+
"patch_summary",
|
|
124
|
+
"test_results",
|
|
125
|
+
"checker_verdict",
|
|
126
|
+
),
|
|
127
|
+
}
|
|
128
|
+
)
|
|
129
|
+
)
|
|
130
|
+
EVAL_CONTEXT_DEFAULT_ALLOWED_PATHS: Mapping[EvalStageId, tuple[str, ...]] = (
|
|
131
|
+
MappingProxyType(
|
|
132
|
+
{
|
|
133
|
+
EvalStageId.PLANNER: (),
|
|
134
|
+
EvalStageId.BUILDER: ("src", "tests", "README.md", "ROADMAP.md"),
|
|
135
|
+
EvalStageId.CHECKER: ("src", "tests", ".eval-scratch"),
|
|
136
|
+
EvalStageId.ARBITER: ("src", "tests"),
|
|
137
|
+
}
|
|
138
|
+
)
|
|
139
|
+
)
|
|
140
|
+
EVAL_STAGE_RESOURCE_CEILING_DEFAULTS: Mapping[EvalStageId, Mapping[str, int]] = (
|
|
141
|
+
MappingProxyType(
|
|
142
|
+
{
|
|
143
|
+
EvalStageId.PLANNER: MappingProxyType(
|
|
144
|
+
{
|
|
145
|
+
"prompt_tokens": 16_000,
|
|
146
|
+
"completion_tokens": 4_000,
|
|
147
|
+
"model_calls": 1,
|
|
148
|
+
"wall_clock_seconds": 600,
|
|
149
|
+
"shell_commands": 1,
|
|
150
|
+
"shell_command_seconds": 1,
|
|
151
|
+
"writable_bytes": 262_144,
|
|
152
|
+
"artifact_bytes": 262_144,
|
|
153
|
+
}
|
|
154
|
+
),
|
|
155
|
+
EvalStageId.BUILDER: MappingProxyType(
|
|
156
|
+
{
|
|
157
|
+
"prompt_tokens": 32_000,
|
|
158
|
+
"completion_tokens": 8_000,
|
|
159
|
+
"model_calls": 2,
|
|
160
|
+
"wall_clock_seconds": 1_800,
|
|
161
|
+
"shell_commands": 24,
|
|
162
|
+
"shell_command_seconds": 900,
|
|
163
|
+
"writable_bytes": 2_097_152,
|
|
164
|
+
"artifact_bytes": 524_288,
|
|
165
|
+
}
|
|
166
|
+
),
|
|
167
|
+
EvalStageId.CHECKER: MappingProxyType(
|
|
168
|
+
{
|
|
169
|
+
"prompt_tokens": 16_000,
|
|
170
|
+
"completion_tokens": 4_000,
|
|
171
|
+
"model_calls": 2,
|
|
172
|
+
"wall_clock_seconds": 900,
|
|
173
|
+
"shell_commands": 12,
|
|
174
|
+
"shell_command_seconds": 600,
|
|
175
|
+
"writable_bytes": 131_072,
|
|
176
|
+
"artifact_bytes": 524_288,
|
|
177
|
+
}
|
|
178
|
+
),
|
|
179
|
+
EvalStageId.ARBITER: MappingProxyType(
|
|
180
|
+
{
|
|
181
|
+
"prompt_tokens": 16_000,
|
|
182
|
+
"completion_tokens": 4_000,
|
|
183
|
+
"model_calls": 1,
|
|
184
|
+
"wall_clock_seconds": 600,
|
|
185
|
+
"shell_commands": 1,
|
|
186
|
+
"shell_command_seconds": 1,
|
|
187
|
+
"writable_bytes": 131_072,
|
|
188
|
+
"artifact_bytes": 262_144,
|
|
189
|
+
}
|
|
190
|
+
),
|
|
191
|
+
}
|
|
192
|
+
)
|
|
193
|
+
)
|
|
194
|
+
EVAL_TRIAL_RESOURCE_CEILING_DEFAULTS: Mapping[str, int] = MappingProxyType(
|
|
195
|
+
{
|
|
196
|
+
"prompt_tokens": 80_000,
|
|
197
|
+
"completion_tokens": 20_000,
|
|
198
|
+
"model_calls": 6,
|
|
199
|
+
"wall_clock_seconds": 3_900,
|
|
200
|
+
"shell_commands": 36,
|
|
201
|
+
"shell_command_seconds": 1_500,
|
|
202
|
+
"writable_bytes": 2_621_440,
|
|
203
|
+
"artifact_bytes": 1_572_864,
|
|
204
|
+
}
|
|
205
|
+
)
|
|
206
|
+
_WINDOWS_DRIVE_PREFIX = re.compile(r"^[A-Za-z]:")
|
|
207
|
+
_SECRET_ENV_MARKERS: tuple[str, ...] = (
|
|
208
|
+
"API_KEY",
|
|
209
|
+
"AUTH",
|
|
210
|
+
"CREDENTIAL",
|
|
211
|
+
"PASSWORD",
|
|
212
|
+
"SECRET",
|
|
213
|
+
"TOKEN",
|
|
214
|
+
)
|
|
215
|
+
_HIDDEN_CHECK_DENIED_TOKENS: tuple[str, ...] = (
|
|
216
|
+
"definition",
|
|
217
|
+
"expected",
|
|
218
|
+
"fixture",
|
|
219
|
+
"output",
|
|
220
|
+
"path",
|
|
221
|
+
"result",
|
|
222
|
+
"rubric",
|
|
223
|
+
"score",
|
|
224
|
+
"secret",
|
|
225
|
+
)
|
|
226
|
+
_PACKAGE_MANAGER_COMMANDS: frozenset[str] = frozenset(
|
|
227
|
+
{
|
|
228
|
+
"cargo",
|
|
229
|
+
"gem",
|
|
230
|
+
"npm",
|
|
231
|
+
"pip",
|
|
232
|
+
"pip3",
|
|
233
|
+
"pnpm",
|
|
234
|
+
"poetry",
|
|
235
|
+
"uv",
|
|
236
|
+
"yarn",
|
|
237
|
+
}
|
|
238
|
+
)
|
|
239
|
+
_NETWORK_COMMANDS: frozenset[str] = frozenset(
|
|
240
|
+
{
|
|
241
|
+
"curl",
|
|
242
|
+
"ftp",
|
|
243
|
+
"nc",
|
|
244
|
+
"netcat",
|
|
245
|
+
"scp",
|
|
246
|
+
"sftp",
|
|
247
|
+
"ssh",
|
|
248
|
+
"telnet",
|
|
249
|
+
"wget",
|
|
250
|
+
}
|
|
251
|
+
)
|
|
252
|
+
_RUNTIME_CONTROL_COMMANDS: frozenset[str] = frozenset(
|
|
253
|
+
{
|
|
254
|
+
"docker",
|
|
255
|
+
"docker-compose",
|
|
256
|
+
"millrace",
|
|
257
|
+
"podman",
|
|
258
|
+
"service",
|
|
259
|
+
"systemctl",
|
|
260
|
+
}
|
|
261
|
+
)
|
|
262
|
+
_SHELL_COMMAND_WRAPPERS: frozenset[str] = frozenset(
|
|
263
|
+
{
|
|
264
|
+
"bash",
|
|
265
|
+
"dash",
|
|
266
|
+
"fish",
|
|
267
|
+
"ksh",
|
|
268
|
+
"sh",
|
|
269
|
+
"zsh",
|
|
270
|
+
}
|
|
271
|
+
)
|
|
272
|
+
_SHELL_INTERPOLATION_TOKENS: tuple[str, ...] = (
|
|
273
|
+
"$",
|
|
274
|
+
"`",
|
|
275
|
+
"$(",
|
|
276
|
+
"${",
|
|
277
|
+
"{{",
|
|
278
|
+
"}}",
|
|
279
|
+
";",
|
|
280
|
+
"&&",
|
|
281
|
+
"||",
|
|
282
|
+
"|",
|
|
283
|
+
"<",
|
|
284
|
+
">",
|
|
285
|
+
)
|
|
286
|
+
_CONTEXT_LEAK_TOKENS: tuple[str, ...] = (
|
|
287
|
+
"F:\\",
|
|
288
|
+
"/mnt/f",
|
|
289
|
+
"millrace-agents",
|
|
290
|
+
"ideas/",
|
|
291
|
+
"ref-forge/",
|
|
292
|
+
"/home/",
|
|
293
|
+
"\\Users\\",
|
|
294
|
+
"API_KEY",
|
|
295
|
+
"DAEMON_STATE",
|
|
296
|
+
"daemon state",
|
|
297
|
+
"git history",
|
|
298
|
+
"git_history",
|
|
299
|
+
"hidden check",
|
|
300
|
+
"hidden checks",
|
|
301
|
+
"hidden_check",
|
|
302
|
+
"hidden_checks",
|
|
303
|
+
"hidden score",
|
|
304
|
+
"hidden_score",
|
|
305
|
+
"hidden_scores",
|
|
306
|
+
"scoring rubric",
|
|
307
|
+
"scoring_rubric",
|
|
308
|
+
"expected output",
|
|
309
|
+
"expected_output",
|
|
310
|
+
"private conversation",
|
|
311
|
+
"private conversations",
|
|
312
|
+
"private_conversation",
|
|
313
|
+
"private_conversations",
|
|
314
|
+
"private runtime",
|
|
315
|
+
"unrelated repository outline",
|
|
316
|
+
"unrelated repository outlines",
|
|
317
|
+
"unrelated_repository_outline",
|
|
318
|
+
"unrelated_repository_outlines",
|
|
319
|
+
)
|
|
320
|
+
_CONTEXT_WINDOWS_ABSOLUTE_PATH = re.compile(
|
|
321
|
+
r"(?:^|[\s\"'`([{<])(?:[A-Za-z]:[\\/]|\\\\)[^\s\"'`)\]}>,]*"
|
|
322
|
+
)
|
|
323
|
+
_CONTEXT_POSIX_ABSOLUTE_PATH = re.compile(r"(?:^|[\s\"'`([{<])/(?!/)[^\s\"'`)\]}>,]*")
|
|
324
|
+
_CONTEXT_USER_HOME_PATH = re.compile(r"(?:^|[\s\"'`([{<])~(?:/|\\)[^\s\"'`)\]}>,]*")
|
|
325
|
+
|
|
326
|
+
|
|
327
|
+
def _command_executable(argv: tuple[str, ...]) -> str:
|
|
328
|
+
return argv[0].split("/")[-1].lower()
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
def _is_python_executable(executable: str) -> bool:
|
|
332
|
+
return executable in {"py", "python", "python3"} or executable.startswith(
|
|
333
|
+
"python3."
|
|
334
|
+
)
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
def _python_module_name(argv: tuple[str, ...], executable: str) -> str | None:
|
|
338
|
+
if len(argv) < 3 or not _is_python_executable(executable) or argv[1] != "-m":
|
|
339
|
+
return None
|
|
340
|
+
return argv[2].split(".", maxsplit=1)[0].lower()
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
def _disallowed_argv_message(argv: tuple[str, ...]) -> str | None:
|
|
344
|
+
executable = _command_executable(argv)
|
|
345
|
+
if executable in _PACKAGE_MANAGER_COMMANDS:
|
|
346
|
+
return "package manager commands are not admitted"
|
|
347
|
+
if executable in _NETWORK_COMMANDS:
|
|
348
|
+
return "network commands are not admitted"
|
|
349
|
+
if executable in _RUNTIME_CONTROL_COMMANDS:
|
|
350
|
+
return "runtime control commands are not admitted"
|
|
351
|
+
if executable == "git":
|
|
352
|
+
return "git commands are not admitted in eval command descriptors"
|
|
353
|
+
if executable in _SHELL_COMMAND_WRAPPERS:
|
|
354
|
+
return "shell wrapper commands are not admitted"
|
|
355
|
+
module_name = _python_module_name(argv, executable)
|
|
356
|
+
if module_name in _PACKAGE_MANAGER_COMMANDS:
|
|
357
|
+
return "package manager module wrappers are not admitted"
|
|
358
|
+
return None
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
class EvalCapabilityId(str, Enum):
|
|
362
|
+
"""Closed capability IDs admitted by compact eval policy."""
|
|
363
|
+
|
|
364
|
+
ARTIFACT_READ = "artifact.read"
|
|
365
|
+
ARTIFACT_WRITE = "artifact.write"
|
|
366
|
+
EVIDENCE_EMIT = "evidence.emit"
|
|
367
|
+
RUNNER_INVOKE = "runner.invoke"
|
|
368
|
+
WORKSPACE_READ = "workspace.read"
|
|
369
|
+
WORKSPACE_WRITE = "workspace.write"
|
|
370
|
+
SHELL_RUN = "shell.run"
|
|
371
|
+
NETWORK_ACCESS = "network.access"
|
|
372
|
+
PACKAGE_INSTALL = "package.install"
|
|
373
|
+
GIT_MUTATE = "git.mutate"
|
|
374
|
+
RUNTIME_CONTROL = "runtime.control"
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
class EvalCapabilityEnvelope(BaseModel):
|
|
378
|
+
"""Immutable capability envelope for one compact eval stage."""
|
|
379
|
+
|
|
380
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
381
|
+
|
|
382
|
+
stage_id: EvalStageId
|
|
383
|
+
capability_ids: tuple[EvalCapabilityId, ...]
|
|
384
|
+
|
|
385
|
+
@field_validator("capability_ids")
|
|
386
|
+
@classmethod
|
|
387
|
+
def _capability_ids_valid(
|
|
388
|
+
cls, value: tuple[EvalCapabilityId, ...]
|
|
389
|
+
) -> tuple[EvalCapabilityId, ...]:
|
|
390
|
+
if len(set(value)) != len(value):
|
|
391
|
+
raise ValueError("capability_ids values must be unique")
|
|
392
|
+
return value
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
class EvalCapabilityValidationResult(BaseModel):
|
|
396
|
+
"""Structured result for one stage capability admission decision."""
|
|
397
|
+
|
|
398
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
399
|
+
|
|
400
|
+
stage_id: EvalStageId | StrictStr
|
|
401
|
+
capability_id: EvalCapabilityId | StrictStr
|
|
402
|
+
allowed: StrictBool
|
|
403
|
+
rule_id: StrictStr
|
|
404
|
+
diagnostic_code: StrictStr | None = None
|
|
405
|
+
diagnostic_summary: StrictStr | None = None
|
|
406
|
+
|
|
407
|
+
@model_validator(mode="after")
|
|
408
|
+
def _diagnostic_shape_valid(self) -> EvalCapabilityValidationResult:
|
|
409
|
+
if self.allowed:
|
|
410
|
+
if self.diagnostic_code is not None or self.diagnostic_summary is not None:
|
|
411
|
+
raise ValueError(
|
|
412
|
+
"allowed capability results must not include diagnostics"
|
|
413
|
+
)
|
|
414
|
+
elif self.diagnostic_code is None or self.diagnostic_summary is None:
|
|
415
|
+
raise ValueError("denied capability results must include diagnostics")
|
|
416
|
+
return self
|
|
417
|
+
|
|
418
|
+
|
|
419
|
+
class EvalCommandEnvironmentPolicy(BaseModel):
|
|
420
|
+
"""Closed environment policy for deterministic eval commands."""
|
|
421
|
+
|
|
422
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
423
|
+
|
|
424
|
+
inherit_environment: StrictBool = False
|
|
425
|
+
variables: Mapping[StrictStr, StrictStr] = Field(default_factory=dict)
|
|
426
|
+
|
|
427
|
+
@model_validator(mode="after")
|
|
428
|
+
def _environment_policy_valid(self) -> EvalCommandEnvironmentPolicy:
|
|
429
|
+
if self.inherit_environment:
|
|
430
|
+
raise ValueError("eval commands may not inherit ambient environment")
|
|
431
|
+
ordered: dict[str, str] = {}
|
|
432
|
+
for name, value in sorted(self.variables.items()):
|
|
433
|
+
if not name or not name.replace("_", "").isalnum() or name[0].isdigit():
|
|
434
|
+
raise ValueError(
|
|
435
|
+
"environment variable names must be stable identifiers"
|
|
436
|
+
)
|
|
437
|
+
if name == "*" or any(
|
|
438
|
+
marker in name.upper() for marker in _SECRET_ENV_MARKERS
|
|
439
|
+
):
|
|
440
|
+
raise ValueError("environment policy may not expose secrets")
|
|
441
|
+
if "\x00" in value:
|
|
442
|
+
raise ValueError(
|
|
443
|
+
"environment variable values must not contain NUL bytes"
|
|
444
|
+
)
|
|
445
|
+
ordered[name] = value
|
|
446
|
+
object.__setattr__(self, "variables", MappingProxyType(ordered))
|
|
447
|
+
return self
|
|
448
|
+
|
|
449
|
+
|
|
450
|
+
class EvalCommandDescriptor(BaseModel):
|
|
451
|
+
"""Deterministic descriptor for admitted Builder and Checker commands."""
|
|
452
|
+
|
|
453
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
454
|
+
|
|
455
|
+
command_id: StrictStr
|
|
456
|
+
argv: tuple[StrictStr, ...]
|
|
457
|
+
relative_working_directory: StrictStr = "."
|
|
458
|
+
admitted_read_roots: tuple[StrictStr, ...]
|
|
459
|
+
admitted_write_roots: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
460
|
+
timeout_seconds: StrictInt = Field(gt=0, le=600)
|
|
461
|
+
environment_policy: EvalCommandEnvironmentPolicy = Field(
|
|
462
|
+
default_factory=EvalCommandEnvironmentPolicy
|
|
463
|
+
)
|
|
464
|
+
expected_output_artifact_ids: tuple[StrictStr, ...]
|
|
465
|
+
|
|
466
|
+
@field_validator("command_id")
|
|
467
|
+
@classmethod
|
|
468
|
+
def _command_id_valid(cls, value: str) -> str:
|
|
469
|
+
if not value.strip() or any(
|
|
470
|
+
token in value for token in _SHELL_INTERPOLATION_TOKENS
|
|
471
|
+
):
|
|
472
|
+
raise ValueError("command_id must be a stable non-interpolated identifier")
|
|
473
|
+
return value
|
|
474
|
+
|
|
475
|
+
@field_validator("argv")
|
|
476
|
+
@classmethod
|
|
477
|
+
def _argv_valid(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
|
478
|
+
if not value:
|
|
479
|
+
raise ValueError("argv must not be empty")
|
|
480
|
+
if len(value) == 1 and any(character.isspace() for character in value[0]):
|
|
481
|
+
raise ValueError("argv entries must not require shell interpolation")
|
|
482
|
+
if disallowed_message := _disallowed_argv_message(value):
|
|
483
|
+
raise ValueError(disallowed_message)
|
|
484
|
+
for argument in value:
|
|
485
|
+
if not argument or "\x00" in argument:
|
|
486
|
+
raise ValueError("argv entries must be non-empty strings")
|
|
487
|
+
if any(token in argument for token in _SHELL_INTERPOLATION_TOKENS):
|
|
488
|
+
raise ValueError("argv entries must not require shell interpolation")
|
|
489
|
+
return value
|
|
490
|
+
|
|
491
|
+
@field_validator("admitted_read_roots", "admitted_write_roots")
|
|
492
|
+
@classmethod
|
|
493
|
+
def _roots_valid(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
|
494
|
+
if len(set(value)) != len(value):
|
|
495
|
+
raise ValueError("admitted roots must be unique")
|
|
496
|
+
for root in value:
|
|
497
|
+
_validate_relative_eval_path(root, allow_dot=False)
|
|
498
|
+
return value
|
|
499
|
+
|
|
500
|
+
@field_validator("expected_output_artifact_ids")
|
|
501
|
+
@classmethod
|
|
502
|
+
def _artifact_ids_valid(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
|
503
|
+
if not value:
|
|
504
|
+
raise ValueError("expected_output_artifact_ids must not be empty")
|
|
505
|
+
if len(set(value)) != len(value):
|
|
506
|
+
raise ValueError("expected_output_artifact_ids values must be unique")
|
|
507
|
+
for artifact_id in value:
|
|
508
|
+
if not artifact_id.strip() or "/" in artifact_id or "\\" in artifact_id:
|
|
509
|
+
raise ValueError("expected output artifact ids must be stable IDs")
|
|
510
|
+
return value
|
|
511
|
+
|
|
512
|
+
@model_validator(mode="after")
|
|
513
|
+
def _descriptor_shape_valid(self) -> EvalCommandDescriptor:
|
|
514
|
+
_validate_relative_eval_path(self.relative_working_directory, allow_dot=True)
|
|
515
|
+
if not self.admitted_read_roots:
|
|
516
|
+
raise ValueError("admitted_read_roots must not be empty")
|
|
517
|
+
if self.relative_working_directory != "." and not any(
|
|
518
|
+
_path_is_within_root(self.relative_working_directory, root)
|
|
519
|
+
for root in self.admitted_read_roots + self.admitted_write_roots
|
|
520
|
+
):
|
|
521
|
+
raise ValueError("relative_working_directory must be inside admitted roots")
|
|
522
|
+
return self
|
|
523
|
+
|
|
524
|
+
|
|
525
|
+
class EvalCommandAdmissionResult(BaseModel):
|
|
526
|
+
"""Structured command admission result for a compact eval stage."""
|
|
527
|
+
|
|
528
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
529
|
+
|
|
530
|
+
stage_id: EvalStageId | StrictStr
|
|
531
|
+
command_id: StrictStr
|
|
532
|
+
allowed: StrictBool
|
|
533
|
+
rule_id: StrictStr
|
|
534
|
+
diagnostic_code: StrictStr | None = None
|
|
535
|
+
diagnostic_summary: StrictStr | None = None
|
|
536
|
+
|
|
537
|
+
@model_validator(mode="after")
|
|
538
|
+
def _diagnostic_shape_valid(self) -> EvalCommandAdmissionResult:
|
|
539
|
+
if self.allowed:
|
|
540
|
+
if self.diagnostic_code is not None or self.diagnostic_summary is not None:
|
|
541
|
+
raise ValueError("allowed command results must not include diagnostics")
|
|
542
|
+
elif self.diagnostic_code is None or self.diagnostic_summary is None:
|
|
543
|
+
raise ValueError("denied command results must include diagnostics")
|
|
544
|
+
return self
|
|
545
|
+
|
|
546
|
+
|
|
547
|
+
class EvalBoundaryStageArtifacts(BaseModel):
|
|
548
|
+
"""Immutable logical artifact IDs for one compact eval stage."""
|
|
549
|
+
|
|
550
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
551
|
+
|
|
552
|
+
stage_id: EvalStageId
|
|
553
|
+
input_artifact_ids: tuple[StrictStr, ...]
|
|
554
|
+
output_artifact_ids: tuple[StrictStr, ...]
|
|
555
|
+
|
|
556
|
+
|
|
557
|
+
class EvalBoundaryBaseline(BaseModel):
|
|
558
|
+
"""Immutable 06B boundary baseline derived from the accepted 06A graph."""
|
|
559
|
+
|
|
560
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
561
|
+
|
|
562
|
+
schema_version: StrictInt = 1
|
|
563
|
+
module_name: StrictStr = EVAL_BOUNDARY_MODULE_NAME
|
|
564
|
+
artifact_module_required: StrictBool = EVAL_BOUNDARY_ARTIFACT_MODULE_REQUIRED
|
|
565
|
+
artifact_module_decision: StrictStr = EVAL_BOUNDARY_ARTIFACT_MODULE_DECISION
|
|
566
|
+
authoritative_public_names: tuple[StrictStr, ...] = Field(
|
|
567
|
+
default=AUTHORITATIVE_COMPACT_EVAL_WORKFLOW_NAMES
|
|
568
|
+
)
|
|
569
|
+
graph_id: StrictStr
|
|
570
|
+
graph_sha256: StrictStr
|
|
571
|
+
stage_ids: tuple[EvalStageId, ...]
|
|
572
|
+
terminal_results: tuple[EvalTerminalResult, ...]
|
|
573
|
+
outcome_kinds: tuple[EvalWorkflowOutcomeKind, ...]
|
|
574
|
+
candidate_dispositions: tuple[EvalCandidateDisposition, ...]
|
|
575
|
+
stage_artifacts: tuple[EvalBoundaryStageArtifacts, ...]
|
|
576
|
+
|
|
577
|
+
|
|
578
|
+
class EvalContextTier(str, Enum):
|
|
579
|
+
"""Closed context tiers for compact eval stage prompts."""
|
|
580
|
+
|
|
581
|
+
COMPACT = "compact"
|
|
582
|
+
VALIDATOR_VISIBLE = "validator_visible"
|
|
583
|
+
|
|
584
|
+
|
|
585
|
+
class EvalContextArtifactSummary(BaseModel):
|
|
586
|
+
"""Path-free summary of a model-visible artifact required by a stage."""
|
|
587
|
+
|
|
588
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
589
|
+
|
|
590
|
+
artifact_id: StrictStr
|
|
591
|
+
summary: StrictStr
|
|
592
|
+
|
|
593
|
+
@model_validator(mode="after")
|
|
594
|
+
def _summary_valid(self) -> EvalContextArtifactSummary:
|
|
595
|
+
_reject_context_material_leaks(self.model_dump(mode="json"))
|
|
596
|
+
return self
|
|
597
|
+
|
|
598
|
+
|
|
599
|
+
class EvalContextRedaction(BaseModel):
|
|
600
|
+
"""Deterministic summary of material omitted from compact context."""
|
|
601
|
+
|
|
602
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
603
|
+
|
|
604
|
+
categories: tuple[StrictStr, ...] = EVAL_CONTEXT_DEFAULT_REDACTION_CATEGORIES
|
|
605
|
+
summary: StrictStr = "scorer-only and private workspace material omitted"
|
|
606
|
+
redacted_item_count: StrictInt = Field(ge=0)
|
|
607
|
+
|
|
608
|
+
@model_validator(mode="after")
|
|
609
|
+
def _redaction_valid(self) -> EvalContextRedaction:
|
|
610
|
+
if len(set(self.categories)) != len(self.categories):
|
|
611
|
+
raise ValueError("redaction categories must be unique")
|
|
612
|
+
for category in self.categories:
|
|
613
|
+
if not category.strip() or "/" in category or "\\" in category:
|
|
614
|
+
raise ValueError("redaction categories must be stable identifiers")
|
|
615
|
+
_reject_context_material_leaks(self.summary)
|
|
616
|
+
return self
|
|
617
|
+
|
|
618
|
+
|
|
619
|
+
class EvalResourceCeiling(BaseModel):
|
|
620
|
+
"""Positive bounded resource ceiling for a compact eval trial or stage."""
|
|
621
|
+
|
|
622
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
623
|
+
|
|
624
|
+
scope: StrictStr
|
|
625
|
+
stage_id: EvalStageId | None = None
|
|
626
|
+
prompt_tokens: StrictInt = Field(gt=0, le=1_000_000)
|
|
627
|
+
completion_tokens: StrictInt = Field(gt=0, le=1_000_000)
|
|
628
|
+
model_calls: StrictInt = Field(gt=0, le=100)
|
|
629
|
+
wall_clock_seconds: StrictInt = Field(gt=0, le=86_400)
|
|
630
|
+
shell_commands: StrictInt = Field(gt=0, le=1_000)
|
|
631
|
+
shell_command_seconds: StrictInt = Field(gt=0, le=86_400)
|
|
632
|
+
writable_bytes: StrictInt = Field(gt=0, le=1_073_741_824)
|
|
633
|
+
artifact_bytes: StrictInt = Field(gt=0, le=1_073_741_824)
|
|
634
|
+
|
|
635
|
+
@model_validator(mode="after")
|
|
636
|
+
def _resource_ceiling_valid(self) -> EvalResourceCeiling:
|
|
637
|
+
if self.scope not in {"trial", "stage"}:
|
|
638
|
+
raise ValueError("resource ceiling scope must be trial or stage")
|
|
639
|
+
if self.scope == "trial" and self.stage_id is not None:
|
|
640
|
+
raise ValueError("trial resource ceilings must not declare stage_id")
|
|
641
|
+
if self.scope == "stage" and self.stage_id is None:
|
|
642
|
+
raise ValueError("stage resource ceilings must declare stage_id")
|
|
643
|
+
return self
|
|
644
|
+
|
|
645
|
+
|
|
646
|
+
class EvalStageContextPolicy(BaseModel):
|
|
647
|
+
"""Compact context assembly policy for one 06A stage."""
|
|
648
|
+
|
|
649
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
650
|
+
|
|
651
|
+
stage_id: EvalStageId
|
|
652
|
+
context_tier: EvalContextTier = EvalContextTier.COMPACT
|
|
653
|
+
allowed_capabilities: tuple[EvalCapabilityId, ...]
|
|
654
|
+
allowed_paths: tuple[StrictStr, ...]
|
|
655
|
+
required_artifact_ids: tuple[StrictStr, ...]
|
|
656
|
+
include_current_stage_contract: StrictBool = True
|
|
657
|
+
include_visible_acceptance_checks: StrictBool = True
|
|
658
|
+
redaction: EvalContextRedaction = Field(
|
|
659
|
+
default_factory=lambda: EvalContextRedaction(redacted_item_count=0)
|
|
660
|
+
)
|
|
661
|
+
resource_ceiling: EvalResourceCeiling
|
|
662
|
+
|
|
663
|
+
@field_validator("allowed_paths")
|
|
664
|
+
@classmethod
|
|
665
|
+
def _allowed_paths_valid(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
|
666
|
+
if len(set(value)) != len(value):
|
|
667
|
+
raise ValueError("allowed_paths values must be unique")
|
|
668
|
+
for path in value:
|
|
669
|
+
_validate_relative_eval_path(path, allow_dot=False)
|
|
670
|
+
_reject_context_material_leaks(path)
|
|
671
|
+
return value
|
|
672
|
+
|
|
673
|
+
@field_validator("required_artifact_ids")
|
|
674
|
+
@classmethod
|
|
675
|
+
def _required_artifacts_valid(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
|
676
|
+
if not value:
|
|
677
|
+
raise ValueError("required_artifact_ids must not be empty")
|
|
678
|
+
if len(set(value)) != len(value):
|
|
679
|
+
raise ValueError("required_artifact_ids values must be unique")
|
|
680
|
+
for artifact_id in value:
|
|
681
|
+
if not artifact_id.strip() or "/" in artifact_id or "\\" in artifact_id:
|
|
682
|
+
raise ValueError("required artifact ids must be stable IDs")
|
|
683
|
+
return value
|
|
684
|
+
|
|
685
|
+
@model_validator(mode="after")
|
|
686
|
+
def _context_policy_valid(self) -> EvalStageContextPolicy:
|
|
687
|
+
envelope = _EVAL_CAPABILITY_ENVELOPES[self.stage_id]
|
|
688
|
+
if self.allowed_capabilities != envelope.capability_ids:
|
|
689
|
+
raise ValueError("context policy capabilities must match stage envelope")
|
|
690
|
+
if self.resource_ceiling.scope != "stage":
|
|
691
|
+
raise ValueError("stage context policies require a stage resource ceiling")
|
|
692
|
+
if self.resource_ceiling.stage_id != self.stage_id:
|
|
693
|
+
raise ValueError("resource ceiling stage_id must match context policy")
|
|
694
|
+
return self
|
|
695
|
+
|
|
696
|
+
|
|
697
|
+
class EvalContextSnapshot(BaseModel):
|
|
698
|
+
"""Deterministic compact model-visible context snapshot."""
|
|
699
|
+
|
|
700
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
701
|
+
|
|
702
|
+
trial_id: StrictStr
|
|
703
|
+
stage_id: EvalStageId
|
|
704
|
+
context_tier: EvalContextTier
|
|
705
|
+
allowed_capabilities: tuple[StrictStr, ...]
|
|
706
|
+
allowed_paths: tuple[StrictStr, ...]
|
|
707
|
+
current_stage_contract: Mapping[StrictStr, Any]
|
|
708
|
+
required_artifact_summaries: tuple[EvalContextArtifactSummary, ...]
|
|
709
|
+
visible_acceptance_check_ids: tuple[StrictStr, ...]
|
|
710
|
+
redaction: EvalContextRedaction
|
|
711
|
+
byte_budget: StrictInt = Field(gt=0, le=1_073_741_824)
|
|
712
|
+
token_budget: StrictInt = Field(gt=0, le=1_000_000)
|
|
713
|
+
resource_ceiling: EvalResourceCeiling
|
|
714
|
+
fingerprint_kind: StrictStr = EVAL_CONTEXT_FINGERPRINT_KIND
|
|
715
|
+
fingerprint: StrictStr
|
|
716
|
+
|
|
717
|
+
@field_validator("trial_id", "fingerprint")
|
|
718
|
+
@classmethod
|
|
719
|
+
def _snapshot_text_valid(cls, value: str) -> str:
|
|
720
|
+
if not value.strip():
|
|
721
|
+
raise ValueError("context snapshot text fields must be non-empty")
|
|
722
|
+
_reject_context_material_leaks(value)
|
|
723
|
+
return value
|
|
724
|
+
|
|
725
|
+
@field_validator("allowed_paths")
|
|
726
|
+
@classmethod
|
|
727
|
+
def _snapshot_paths_valid(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
|
728
|
+
if len(set(value)) != len(value):
|
|
729
|
+
raise ValueError("allowed_paths values must be unique")
|
|
730
|
+
for path in value:
|
|
731
|
+
_validate_relative_eval_path(path, allow_dot=False)
|
|
732
|
+
_reject_context_material_leaks(path)
|
|
733
|
+
return value
|
|
734
|
+
|
|
735
|
+
@field_validator("visible_acceptance_check_ids")
|
|
736
|
+
@classmethod
|
|
737
|
+
def _visible_check_ids_valid(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
|
738
|
+
if len(set(value)) != len(value):
|
|
739
|
+
raise ValueError("visible_acceptance_check_ids values must be unique")
|
|
740
|
+
for check_id in value:
|
|
741
|
+
if not check_id.strip() or "/" in check_id or "\\" in check_id:
|
|
742
|
+
raise ValueError("visible acceptance check ids must be stable IDs")
|
|
743
|
+
_reject_context_material_leaks(check_id)
|
|
744
|
+
return value
|
|
745
|
+
|
|
746
|
+
@model_validator(mode="after")
|
|
747
|
+
def _snapshot_valid(self) -> EvalContextSnapshot:
|
|
748
|
+
_validate_sha256(self.fingerprint)
|
|
749
|
+
if self.fingerprint_kind != EVAL_CONTEXT_FINGERPRINT_KIND:
|
|
750
|
+
raise ValueError("unsupported context fingerprint kind")
|
|
751
|
+
if self.resource_ceiling.stage_id != self.stage_id:
|
|
752
|
+
raise ValueError("context snapshot resource ceiling must match stage")
|
|
753
|
+
_reject_context_material_leaks(self.model_dump(mode="json"))
|
|
754
|
+
return self
|
|
755
|
+
|
|
756
|
+
|
|
757
|
+
class EvalPathPolicyViolation(BaseModel):
|
|
758
|
+
"""Structured relative-path policy diagnostic without host path leakage."""
|
|
759
|
+
|
|
760
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
761
|
+
|
|
762
|
+
path: StrictStr
|
|
763
|
+
rule_id: StrictStr
|
|
764
|
+
diagnostic_code: StrictStr
|
|
765
|
+
diagnostic_summary: StrictStr
|
|
766
|
+
|
|
767
|
+
|
|
768
|
+
class EvalFixtureFile(BaseModel):
|
|
769
|
+
"""Declared fixture file identity."""
|
|
770
|
+
|
|
771
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
772
|
+
|
|
773
|
+
path: StrictStr
|
|
774
|
+
sha256: StrictStr
|
|
775
|
+
size_bytes: StrictInt = Field(ge=0)
|
|
776
|
+
role: StrictStr
|
|
777
|
+
model_readable: StrictBool
|
|
778
|
+
builder_mutable: StrictBool
|
|
779
|
+
|
|
780
|
+
@field_validator("path")
|
|
781
|
+
@classmethod
|
|
782
|
+
def _path_valid(cls, value: str) -> str:
|
|
783
|
+
_validate_relative_eval_path(value, allow_dot=False)
|
|
784
|
+
return value
|
|
785
|
+
|
|
786
|
+
@field_validator("role")
|
|
787
|
+
@classmethod
|
|
788
|
+
def _role_valid(cls, value: str) -> str:
|
|
789
|
+
if value not in EVAL_FIXTURE_FILE_ROLES:
|
|
790
|
+
raise ValueError("fixture file role is not in the closed role set")
|
|
791
|
+
return value
|
|
792
|
+
|
|
793
|
+
@field_validator("sha256")
|
|
794
|
+
@classmethod
|
|
795
|
+
def _sha256_valid(cls, value: str) -> str:
|
|
796
|
+
if len(value) != 64 or any(
|
|
797
|
+
character not in "0123456789abcdef" for character in value
|
|
798
|
+
):
|
|
799
|
+
raise ValueError("sha256 must be a lowercase hexadecimal SHA-256 digest")
|
|
800
|
+
return value
|
|
801
|
+
|
|
802
|
+
|
|
803
|
+
class EvalFixtureWorkspacePolicy(BaseModel):
|
|
804
|
+
"""Workspace isolation and stage write policy for one eval fixture."""
|
|
805
|
+
|
|
806
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
807
|
+
|
|
808
|
+
source_fixture_root_read_only: StrictBool = True
|
|
809
|
+
workspace_isolation: StrictStr = EVAL_WORKSPACE_ISOLATION_CONTRACT
|
|
810
|
+
stage_write_roots: Mapping[EvalStageId, tuple[StrictStr, ...]] = Field(
|
|
811
|
+
default_factory=lambda: {
|
|
812
|
+
EvalStageId.BUILDER: EVAL_BUILDER_DEFAULT_WRITE_ROOTS,
|
|
813
|
+
EvalStageId.CHECKER: (),
|
|
814
|
+
EvalStageId.ARBITER: (),
|
|
815
|
+
EvalStageId.PLANNER: (),
|
|
816
|
+
}
|
|
817
|
+
)
|
|
818
|
+
ignored_generated_roots: tuple[StrictStr, ...] = (
|
|
819
|
+
EVAL_FIXTURE_IGNORED_GENERATED_ROOTS
|
|
820
|
+
)
|
|
821
|
+
ignored_generated_suffixes: tuple[StrictStr, ...] = (
|
|
822
|
+
EVAL_FIXTURE_IGNORED_GENERATED_SUFFIXES
|
|
823
|
+
)
|
|
824
|
+
|
|
825
|
+
@model_validator(mode="after")
|
|
826
|
+
def _policy_valid(self) -> EvalFixtureWorkspacePolicy:
|
|
827
|
+
if not self.source_fixture_root_read_only:
|
|
828
|
+
raise ValueError("source fixture root must be read-only")
|
|
829
|
+
if self.workspace_isolation != EVAL_WORKSPACE_ISOLATION_CONTRACT:
|
|
830
|
+
raise ValueError("fixture workspaces must use a fresh copied workspace")
|
|
831
|
+
ordered_stage_write_roots: dict[EvalStageId, tuple[str, ...]] = {}
|
|
832
|
+
for stage_id in EvalStageId:
|
|
833
|
+
roots = tuple(self.stage_write_roots.get(stage_id, ()))
|
|
834
|
+
if len(set(roots)) != len(roots):
|
|
835
|
+
raise ValueError("stage write roots must be unique")
|
|
836
|
+
for root in roots:
|
|
837
|
+
_validate_relative_eval_path(root, allow_dot=False)
|
|
838
|
+
ordered_stage_write_roots[stage_id] = roots
|
|
839
|
+
if ordered_stage_write_roots[EvalStageId.CHECKER]:
|
|
840
|
+
raise ValueError("checker fixture workspace policy must be read-only")
|
|
841
|
+
if ordered_stage_write_roots[EvalStageId.ARBITER]:
|
|
842
|
+
raise ValueError("arbiter fixture workspace policy must be read-only")
|
|
843
|
+
if ordered_stage_write_roots[EvalStageId.PLANNER]:
|
|
844
|
+
raise ValueError("planner fixture workspace policy must be read-only")
|
|
845
|
+
for ignored_root in self.ignored_generated_roots:
|
|
846
|
+
_validate_relative_eval_path(ignored_root, allow_dot=False)
|
|
847
|
+
for suffix in self.ignored_generated_suffixes:
|
|
848
|
+
if not suffix or "/" in suffix or "\\" in suffix:
|
|
849
|
+
raise ValueError("ignored generated suffixes must be file suffixes")
|
|
850
|
+
object.__setattr__(
|
|
851
|
+
self, "stage_write_roots", MappingProxyType(ordered_stage_write_roots)
|
|
852
|
+
)
|
|
853
|
+
return self
|
|
854
|
+
|
|
855
|
+
@field_serializer("stage_write_roots")
|
|
856
|
+
def _serialize_stage_write_roots(
|
|
857
|
+
self, value: Mapping[EvalStageId, tuple[str, ...]]
|
|
858
|
+
) -> dict[str, list[str]]:
|
|
859
|
+
return {stage_id.value: list(value[stage_id]) for stage_id in EvalStageId}
|
|
860
|
+
|
|
861
|
+
|
|
862
|
+
class EvalFixtureManifest(BaseModel):
|
|
863
|
+
"""Deterministic fixture manifest with declared file hashes."""
|
|
864
|
+
|
|
865
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
866
|
+
|
|
867
|
+
schema_version: StrictInt
|
|
868
|
+
fixture_id: StrictStr
|
|
869
|
+
fixture_revision: StrictStr
|
|
870
|
+
task_id: StrictStr
|
|
871
|
+
source_root_label: StrictStr
|
|
872
|
+
allowed_read_paths: tuple[StrictStr, ...]
|
|
873
|
+
allowed_write_paths: tuple[StrictStr, ...]
|
|
874
|
+
allowed_command_roots: tuple[StrictStr, ...]
|
|
875
|
+
visible_acceptance_checks: tuple[StrictStr, ...]
|
|
876
|
+
hidden_check_ids: tuple[StrictStr, ...]
|
|
877
|
+
expected_mutation_paths: tuple[StrictStr, ...]
|
|
878
|
+
files: tuple[EvalFixtureFile, ...]
|
|
879
|
+
workspace_policy: EvalFixtureWorkspacePolicy = Field(
|
|
880
|
+
default_factory=EvalFixtureWorkspacePolicy
|
|
881
|
+
)
|
|
882
|
+
|
|
883
|
+
@field_validator("schema_version")
|
|
884
|
+
@classmethod
|
|
885
|
+
def _schema_version_valid(cls, value: int) -> int:
|
|
886
|
+
if value != EVAL_FIXTURE_MANIFEST_SCHEMA_VERSION:
|
|
887
|
+
raise ValueError("unsupported fixture manifest schema_version")
|
|
888
|
+
return value
|
|
889
|
+
|
|
890
|
+
@field_validator("fixture_id", "fixture_revision", "task_id", "source_root_label")
|
|
891
|
+
@classmethod
|
|
892
|
+
def _stable_text_valid(cls, value: str) -> str:
|
|
893
|
+
if not value.strip() or "/" in value or "\\" in value:
|
|
894
|
+
raise ValueError("fixture manifest text fields must be stable identifiers")
|
|
895
|
+
_reject_context_material_leaks(value)
|
|
896
|
+
return value
|
|
897
|
+
|
|
898
|
+
@field_validator(
|
|
899
|
+
"allowed_read_paths",
|
|
900
|
+
"allowed_write_paths",
|
|
901
|
+
"allowed_command_roots",
|
|
902
|
+
"expected_mutation_paths",
|
|
903
|
+
)
|
|
904
|
+
@classmethod
|
|
905
|
+
def _manifest_paths_valid(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
|
906
|
+
if len(set(value)) != len(value):
|
|
907
|
+
raise ValueError("fixture manifest paths must be unique")
|
|
908
|
+
for path in value:
|
|
909
|
+
try:
|
|
910
|
+
_validate_relative_eval_path(path, allow_dot=False)
|
|
911
|
+
except ValueError as exc:
|
|
912
|
+
raise ValueError(
|
|
913
|
+
"fixture manifest paths must be normalized relative POSIX paths"
|
|
914
|
+
) from exc
|
|
915
|
+
_reject_context_material_leaks(path)
|
|
916
|
+
return tuple(sorted(value))
|
|
917
|
+
|
|
918
|
+
@field_validator("visible_acceptance_checks")
|
|
919
|
+
@classmethod
|
|
920
|
+
def _visible_acceptance_checks_valid(
|
|
921
|
+
cls, value: tuple[str, ...]
|
|
922
|
+
) -> tuple[str, ...]:
|
|
923
|
+
if len(set(value)) != len(value):
|
|
924
|
+
raise ValueError("visible acceptance checks must be unique")
|
|
925
|
+
for check_id in value:
|
|
926
|
+
_validate_opaque_eval_id(check_id, field_name="visible acceptance check")
|
|
927
|
+
return value
|
|
928
|
+
|
|
929
|
+
@field_validator("hidden_check_ids")
|
|
930
|
+
@classmethod
|
|
931
|
+
def _hidden_check_ids_valid(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
|
932
|
+
if len(set(value)) != len(value):
|
|
933
|
+
raise ValueError("hidden_check_ids must be unique")
|
|
934
|
+
for check_id in value:
|
|
935
|
+
_validate_opaque_eval_id(check_id, field_name="hidden check id")
|
|
936
|
+
lowered = check_id.lower()
|
|
937
|
+
if any(token in lowered for token in _HIDDEN_CHECK_DENIED_TOKENS):
|
|
938
|
+
raise ValueError("hidden_check_ids must contain only opaque IDs")
|
|
939
|
+
return value
|
|
940
|
+
|
|
941
|
+
@model_validator(mode="after")
|
|
942
|
+
def _manifest_valid(self) -> EvalFixtureManifest:
|
|
943
|
+
paths = tuple(file.path for file in self.files)
|
|
944
|
+
if not paths:
|
|
945
|
+
raise ValueError("fixture manifests must declare at least one file")
|
|
946
|
+
if len(set(paths)) != len(paths):
|
|
947
|
+
raise ValueError("fixture file paths must be unique")
|
|
948
|
+
for mutation_path in self.expected_mutation_paths:
|
|
949
|
+
if mutation_path not in paths and not any(
|
|
950
|
+
_path_is_within_root(path, mutation_path) for path in paths
|
|
951
|
+
):
|
|
952
|
+
raise ValueError(
|
|
953
|
+
"expected_mutation_paths must reference declared fixture files or roots"
|
|
954
|
+
)
|
|
955
|
+
object.__setattr__(
|
|
956
|
+
self, "files", tuple(sorted(self.files, key=lambda file: file.path))
|
|
957
|
+
)
|
|
958
|
+
return self
|
|
959
|
+
|
|
960
|
+
|
|
961
|
+
class EvalFixtureWorkspaceSnapshot(BaseModel):
|
|
962
|
+
"""Deterministic comparison between a declared manifest and workspace state."""
|
|
963
|
+
|
|
964
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
965
|
+
|
|
966
|
+
fixture_id: StrictStr
|
|
967
|
+
fixture_manifest_sha256: StrictStr
|
|
968
|
+
files: tuple[EvalFixtureFile, ...]
|
|
969
|
+
added_paths: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
970
|
+
modified_paths: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
971
|
+
deleted_paths: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
972
|
+
unchanged_paths: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
973
|
+
ignored_generated_paths: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
974
|
+
unauthorized_mutation_paths: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
975
|
+
violations: tuple[EvalPathPolicyViolation, ...] = Field(default_factory=tuple)
|
|
976
|
+
|
|
977
|
+
@field_validator("fixture_manifest_sha256")
|
|
978
|
+
@classmethod
|
|
979
|
+
def _fixture_manifest_sha256_valid(cls, value: str) -> str:
|
|
980
|
+
_validate_sha256(value)
|
|
981
|
+
return value
|
|
982
|
+
|
|
983
|
+
|
|
984
|
+
class EvalClosureOutcomeKind(str, Enum):
|
|
985
|
+
"""Structured compact-eval closure validation outcomes."""
|
|
986
|
+
|
|
987
|
+
VALID_CLOSED_SUCCESS = "valid_closed_success"
|
|
988
|
+
VALID_CLOSED_REJECTION = "valid_closed_rejection"
|
|
989
|
+
VALID_BLOCKED_OUTCOME = "valid_blocked_outcome"
|
|
990
|
+
INVALID_ARTIFACT_BOUNDARY = "invalid_artifact_boundary"
|
|
991
|
+
INVALID_CAPABILITY_BOUNDARY = "invalid_capability_boundary"
|
|
992
|
+
INVALID_FIXTURE_BOUNDARY = "invalid_fixture_boundary"
|
|
993
|
+
INVALID_CONTEXT_BOUNDARY = "invalid_context_boundary"
|
|
994
|
+
|
|
995
|
+
|
|
996
|
+
class EvalClosureValidationResult(BaseModel):
|
|
997
|
+
"""Non-mutating aggregate validation result for compact eval closure evidence."""
|
|
998
|
+
|
|
999
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
1000
|
+
|
|
1001
|
+
valid: StrictBool
|
|
1002
|
+
outcome_kind: EvalClosureOutcomeKind
|
|
1003
|
+
terminal_result: EvalTerminalResult | None = None
|
|
1004
|
+
candidate_disposition: EvalCandidateDisposition | None = None
|
|
1005
|
+
evidence_artifact_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
1006
|
+
missing_artifact_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
1007
|
+
diagnostics: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
1008
|
+
|
|
1009
|
+
@model_validator(mode="after")
|
|
1010
|
+
def _closure_result_valid(self) -> EvalClosureValidationResult:
|
|
1011
|
+
if self.valid:
|
|
1012
|
+
if self.outcome_kind not in {
|
|
1013
|
+
EvalClosureOutcomeKind.VALID_CLOSED_SUCCESS,
|
|
1014
|
+
EvalClosureOutcomeKind.VALID_CLOSED_REJECTION,
|
|
1015
|
+
EvalClosureOutcomeKind.VALID_BLOCKED_OUTCOME,
|
|
1016
|
+
}:
|
|
1017
|
+
raise ValueError("valid closure results require a valid outcome kind")
|
|
1018
|
+
if self.diagnostics:
|
|
1019
|
+
raise ValueError("valid closure results must not include diagnostics")
|
|
1020
|
+
elif self.outcome_kind in {
|
|
1021
|
+
EvalClosureOutcomeKind.VALID_CLOSED_SUCCESS,
|
|
1022
|
+
EvalClosureOutcomeKind.VALID_CLOSED_REJECTION,
|
|
1023
|
+
EvalClosureOutcomeKind.VALID_BLOCKED_OUTCOME,
|
|
1024
|
+
}:
|
|
1025
|
+
raise ValueError("invalid closure results require an invalid outcome kind")
|
|
1026
|
+
return self
|
|
1027
|
+
|
|
1028
|
+
|
|
1029
|
+
class EvalCheckerApprovalValidationResult(BaseModel):
|
|
1030
|
+
"""Non-mutating validation result for Checker approval public evidence."""
|
|
1031
|
+
|
|
1032
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
1033
|
+
|
|
1034
|
+
valid: StrictBool
|
|
1035
|
+
evidence_artifact_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
1036
|
+
missing_artifact_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
1037
|
+
diagnostics: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
1038
|
+
|
|
1039
|
+
@model_validator(mode="after")
|
|
1040
|
+
def _checker_approval_result_valid(
|
|
1041
|
+
self,
|
|
1042
|
+
) -> EvalCheckerApprovalValidationResult:
|
|
1043
|
+
if self.valid:
|
|
1044
|
+
if self.missing_artifact_ids or self.diagnostics:
|
|
1045
|
+
raise ValueError(
|
|
1046
|
+
"valid checker approval results must not include diagnostics"
|
|
1047
|
+
)
|
|
1048
|
+
elif not self.missing_artifact_ids and not self.diagnostics:
|
|
1049
|
+
raise ValueError(
|
|
1050
|
+
"invalid checker approval results require diagnostics or missing IDs"
|
|
1051
|
+
)
|
|
1052
|
+
return self
|
|
1053
|
+
|
|
1054
|
+
|
|
1055
|
+
def _validate_relative_eval_path(value: str, *, allow_dot: bool) -> None:
|
|
1056
|
+
if not value:
|
|
1057
|
+
raise ValueError("eval paths must be non-empty relative POSIX paths")
|
|
1058
|
+
if value == ".":
|
|
1059
|
+
if allow_dot:
|
|
1060
|
+
return
|
|
1061
|
+
raise ValueError("eval root paths must not be '.'")
|
|
1062
|
+
posix_path = PurePosixPath(value)
|
|
1063
|
+
parts = posix_path.parts
|
|
1064
|
+
if (
|
|
1065
|
+
value.startswith("/")
|
|
1066
|
+
or value.startswith("\\")
|
|
1067
|
+
or "\\" in value
|
|
1068
|
+
or "//" in value
|
|
1069
|
+
or value.startswith("../")
|
|
1070
|
+
or value.endswith("/..")
|
|
1071
|
+
or "/../" in value
|
|
1072
|
+
or value in {"..", ""}
|
|
1073
|
+
or _WINDOWS_DRIVE_PREFIX.match(value)
|
|
1074
|
+
or "." in parts
|
|
1075
|
+
or posix_path.as_posix() != value
|
|
1076
|
+
):
|
|
1077
|
+
raise ValueError("eval paths must be normalized relative POSIX paths")
|
|
1078
|
+
|
|
1079
|
+
|
|
1080
|
+
def _canonical_eval_json_bytes(value: Any) -> bytes:
|
|
1081
|
+
return (
|
|
1082
|
+
json.dumps(
|
|
1083
|
+
value,
|
|
1084
|
+
sort_keys=True,
|
|
1085
|
+
ensure_ascii=True,
|
|
1086
|
+
allow_nan=False,
|
|
1087
|
+
separators=(",", ":"),
|
|
1088
|
+
).replace("\r\n", "\n")
|
|
1089
|
+
+ "\n"
|
|
1090
|
+
).encode("ascii")
|
|
1091
|
+
|
|
1092
|
+
|
|
1093
|
+
def _context_material_text_values(value: Any) -> tuple[str, ...]:
|
|
1094
|
+
if isinstance(value, str):
|
|
1095
|
+
return (value,)
|
|
1096
|
+
if isinstance(value, Mapping):
|
|
1097
|
+
return tuple(
|
|
1098
|
+
text
|
|
1099
|
+
for child in value.values()
|
|
1100
|
+
for text in _context_material_text_values(child)
|
|
1101
|
+
)
|
|
1102
|
+
if isinstance(value, (tuple, list, set, frozenset)):
|
|
1103
|
+
return tuple(
|
|
1104
|
+
text for child in value for text in _context_material_text_values(child)
|
|
1105
|
+
)
|
|
1106
|
+
return ()
|
|
1107
|
+
|
|
1108
|
+
|
|
1109
|
+
def _reject_context_material_leaks(value: Any) -> None:
|
|
1110
|
+
for text in _context_material_text_values(value):
|
|
1111
|
+
if text in EVAL_CONTEXT_DEFAULT_REDACTION_CATEGORIES:
|
|
1112
|
+
continue
|
|
1113
|
+
lowered = text.lower()
|
|
1114
|
+
for token in _CONTEXT_LEAK_TOKENS:
|
|
1115
|
+
if token.lower() in lowered:
|
|
1116
|
+
raise ValueError(
|
|
1117
|
+
"eval context snapshots must not expose private material"
|
|
1118
|
+
)
|
|
1119
|
+
if (
|
|
1120
|
+
_CONTEXT_WINDOWS_ABSOLUTE_PATH.search(text)
|
|
1121
|
+
or _CONTEXT_POSIX_ABSOLUTE_PATH.search(text)
|
|
1122
|
+
or _CONTEXT_USER_HOME_PATH.search(text)
|
|
1123
|
+
):
|
|
1124
|
+
raise ValueError("eval context snapshots must not expose host paths")
|
|
1125
|
+
|
|
1126
|
+
|
|
1127
|
+
def _validate_sha256(value: str) -> None:
|
|
1128
|
+
if len(value) != 64 or any(
|
|
1129
|
+
character not in "0123456789abcdef" for character in value
|
|
1130
|
+
):
|
|
1131
|
+
raise ValueError("sha256 must be a lowercase hexadecimal SHA-256 digest")
|
|
1132
|
+
|
|
1133
|
+
|
|
1134
|
+
def _validate_opaque_eval_id(value: str, *, field_name: str) -> None:
|
|
1135
|
+
if not value.strip():
|
|
1136
|
+
raise ValueError(f"{field_name} must be a non-empty opaque ID")
|
|
1137
|
+
if "/" in value or "\\" in value or _WINDOWS_DRIVE_PREFIX.match(value):
|
|
1138
|
+
raise ValueError(f"{field_name} must not contain paths")
|
|
1139
|
+
if any(character.isspace() for character in value):
|
|
1140
|
+
raise ValueError(f"{field_name} must not contain whitespace")
|
|
1141
|
+
_reject_context_material_leaks(value)
|
|
1142
|
+
|
|
1143
|
+
|
|
1144
|
+
def _context_fingerprint_payload(snapshot: EvalContextSnapshot) -> dict[str, Any]:
|
|
1145
|
+
payload = snapshot.model_dump(mode="json")
|
|
1146
|
+
payload.pop("fingerprint", None)
|
|
1147
|
+
return payload
|
|
1148
|
+
|
|
1149
|
+
|
|
1150
|
+
def calculate_eval_context_fingerprint(snapshot: EvalContextSnapshot) -> str:
|
|
1151
|
+
"""Return the deterministic fingerprint for a compact context snapshot."""
|
|
1152
|
+
return hashlib.sha256(
|
|
1153
|
+
_canonical_eval_json_bytes(_context_fingerprint_payload(snapshot))
|
|
1154
|
+
).hexdigest()
|
|
1155
|
+
|
|
1156
|
+
|
|
1157
|
+
def _path_policy_violation(
|
|
1158
|
+
path: str,
|
|
1159
|
+
*,
|
|
1160
|
+
rule_id: str,
|
|
1161
|
+
diagnostic_code: str,
|
|
1162
|
+
diagnostic_summary: str,
|
|
1163
|
+
) -> EvalPathPolicyViolation:
|
|
1164
|
+
return EvalPathPolicyViolation(
|
|
1165
|
+
path=_diagnostic_path(path),
|
|
1166
|
+
rule_id=rule_id,
|
|
1167
|
+
diagnostic_code=diagnostic_code,
|
|
1168
|
+
diagnostic_summary=diagnostic_summary,
|
|
1169
|
+
)
|
|
1170
|
+
|
|
1171
|
+
|
|
1172
|
+
def _diagnostic_path(path: str) -> str:
|
|
1173
|
+
if (
|
|
1174
|
+
path.startswith("/")
|
|
1175
|
+
or path.startswith("\\")
|
|
1176
|
+
or _WINDOWS_DRIVE_PREFIX.match(path)
|
|
1177
|
+
):
|
|
1178
|
+
return "<absolute-path>"
|
|
1179
|
+
return path
|
|
1180
|
+
|
|
1181
|
+
|
|
1182
|
+
def validate_eval_fixture_path(
|
|
1183
|
+
path: str,
|
|
1184
|
+
*,
|
|
1185
|
+
filesystem_root: Path | None = None,
|
|
1186
|
+
) -> EvalPathPolicyViolation | None:
|
|
1187
|
+
"""Return a structured violation for an invalid fixture path, else None."""
|
|
1188
|
+
try:
|
|
1189
|
+
_validate_relative_eval_path(path, allow_dot=False)
|
|
1190
|
+
except ValueError:
|
|
1191
|
+
return _path_policy_violation(
|
|
1192
|
+
path=path,
|
|
1193
|
+
rule_id="eval.fixture.path.invalid",
|
|
1194
|
+
diagnostic_code="MF-EVAL-F001",
|
|
1195
|
+
diagnostic_summary="fixture paths must be normalized relative POSIX paths",
|
|
1196
|
+
)
|
|
1197
|
+
if filesystem_root is None:
|
|
1198
|
+
return None
|
|
1199
|
+
root = filesystem_root.resolve()
|
|
1200
|
+
candidate = (root / Path(*PurePosixPath(path).parts)).resolve()
|
|
1201
|
+
try:
|
|
1202
|
+
candidate.relative_to(root)
|
|
1203
|
+
except ValueError:
|
|
1204
|
+
return _path_policy_violation(
|
|
1205
|
+
path=path,
|
|
1206
|
+
rule_id="eval.fixture.path.symlink_escape",
|
|
1207
|
+
diagnostic_code="MF-EVAL-F002",
|
|
1208
|
+
diagnostic_summary="fixture path resolves outside fixture root",
|
|
1209
|
+
)
|
|
1210
|
+
return None
|
|
1211
|
+
|
|
1212
|
+
|
|
1213
|
+
def eval_fixture_file_hash(path: Path) -> str:
|
|
1214
|
+
"""Return the SHA-256 hash for a fixture file."""
|
|
1215
|
+
digest = hashlib.sha256()
|
|
1216
|
+
with path.open("rb") as file:
|
|
1217
|
+
for chunk in iter(lambda: file.read(1024 * 1024), b""):
|
|
1218
|
+
digest.update(chunk)
|
|
1219
|
+
return digest.hexdigest()
|
|
1220
|
+
|
|
1221
|
+
|
|
1222
|
+
def _default_fixture_file_role(relative_path: str) -> str:
|
|
1223
|
+
if relative_path.startswith("tests/") or relative_path.startswith("test/"):
|
|
1224
|
+
return "test"
|
|
1225
|
+
if relative_path.startswith("src/"):
|
|
1226
|
+
return "source"
|
|
1227
|
+
if relative_path.endswith((".md", ".rst", ".txt")):
|
|
1228
|
+
return "documentation"
|
|
1229
|
+
if relative_path.endswith((".json", ".toml", ".yaml", ".yml", ".ini", ".cfg")):
|
|
1230
|
+
return "configuration"
|
|
1231
|
+
return "data"
|
|
1232
|
+
|
|
1233
|
+
|
|
1234
|
+
def eval_fixture_file_record(
|
|
1235
|
+
fixture_root: Path,
|
|
1236
|
+
relative_path: str,
|
|
1237
|
+
*,
|
|
1238
|
+
role: str | None = None,
|
|
1239
|
+
model_readable: bool = True,
|
|
1240
|
+
builder_mutable: bool = False,
|
|
1241
|
+
) -> EvalFixtureFile:
|
|
1242
|
+
"""Build a deterministic file record for one relative fixture path."""
|
|
1243
|
+
violation = validate_eval_fixture_path(relative_path, filesystem_root=fixture_root)
|
|
1244
|
+
if violation is not None:
|
|
1245
|
+
raise ValueError(violation.diagnostic_summary)
|
|
1246
|
+
path = fixture_root / Path(*PurePosixPath(relative_path).parts)
|
|
1247
|
+
return EvalFixtureFile(
|
|
1248
|
+
path=relative_path,
|
|
1249
|
+
sha256=eval_fixture_file_hash(path),
|
|
1250
|
+
size_bytes=path.stat().st_size,
|
|
1251
|
+
role=role or _default_fixture_file_role(relative_path),
|
|
1252
|
+
model_readable=model_readable,
|
|
1253
|
+
builder_mutable=builder_mutable,
|
|
1254
|
+
)
|
|
1255
|
+
|
|
1256
|
+
|
|
1257
|
+
def eval_fixture_manifest_sha256(manifest: EvalFixtureManifest) -> str:
|
|
1258
|
+
"""Return the deterministic SHA-256 of an expanded fixture manifest payload."""
|
|
1259
|
+
return hashlib.sha256(
|
|
1260
|
+
_canonical_eval_json_bytes(manifest.model_dump(mode="json"))
|
|
1261
|
+
).hexdigest()
|
|
1262
|
+
|
|
1263
|
+
|
|
1264
|
+
def eval_fixture_manifest_from_paths(
|
|
1265
|
+
fixture_id: str,
|
|
1266
|
+
fixture_root: Path,
|
|
1267
|
+
relative_paths: tuple[str, ...],
|
|
1268
|
+
*,
|
|
1269
|
+
task_id: str,
|
|
1270
|
+
schema_version: int = EVAL_FIXTURE_MANIFEST_SCHEMA_VERSION,
|
|
1271
|
+
fixture_revision: str = EVAL_FIXTURE_DEFAULT_REVISION,
|
|
1272
|
+
source_root_label: str = "fixture_workspace",
|
|
1273
|
+
allowed_read_paths: tuple[str, ...] | None = None,
|
|
1274
|
+
allowed_write_paths: tuple[str, ...] = EVAL_BUILDER_DEFAULT_WRITE_ROOTS,
|
|
1275
|
+
allowed_command_roots: tuple[str, ...] = ("src", "tests"),
|
|
1276
|
+
visible_acceptance_checks: tuple[str, ...] = (),
|
|
1277
|
+
hidden_check_ids: tuple[str, ...] = (),
|
|
1278
|
+
expected_mutation_paths: tuple[str, ...] | None = None,
|
|
1279
|
+
workspace_policy: EvalFixtureWorkspacePolicy | None = None,
|
|
1280
|
+
) -> EvalFixtureManifest:
|
|
1281
|
+
"""Build a deterministic manifest from declared fixture-relative files."""
|
|
1282
|
+
declared_paths = tuple(sorted(relative_paths))
|
|
1283
|
+
mutation_paths = expected_mutation_paths or ()
|
|
1284
|
+
return EvalFixtureManifest(
|
|
1285
|
+
schema_version=schema_version,
|
|
1286
|
+
fixture_id=fixture_id,
|
|
1287
|
+
fixture_revision=fixture_revision,
|
|
1288
|
+
task_id=task_id,
|
|
1289
|
+
source_root_label=source_root_label,
|
|
1290
|
+
allowed_read_paths=allowed_read_paths or declared_paths,
|
|
1291
|
+
allowed_write_paths=allowed_write_paths,
|
|
1292
|
+
allowed_command_roots=allowed_command_roots,
|
|
1293
|
+
visible_acceptance_checks=visible_acceptance_checks,
|
|
1294
|
+
hidden_check_ids=hidden_check_ids,
|
|
1295
|
+
expected_mutation_paths=mutation_paths,
|
|
1296
|
+
files=tuple(
|
|
1297
|
+
eval_fixture_file_record(
|
|
1298
|
+
fixture_root,
|
|
1299
|
+
relative_path,
|
|
1300
|
+
builder_mutable=relative_path in mutation_paths,
|
|
1301
|
+
)
|
|
1302
|
+
for relative_path in declared_paths
|
|
1303
|
+
),
|
|
1304
|
+
workspace_policy=workspace_policy or EvalFixtureWorkspacePolicy(),
|
|
1305
|
+
)
|
|
1306
|
+
|
|
1307
|
+
|
|
1308
|
+
def _is_ignored_generated_path(path: str, policy: EvalFixtureWorkspacePolicy) -> bool:
|
|
1309
|
+
parts = PurePosixPath(path).parts
|
|
1310
|
+
return any(
|
|
1311
|
+
root in parts or _path_is_within_root(path, root)
|
|
1312
|
+
for root in policy.ignored_generated_roots
|
|
1313
|
+
) or any(path.endswith(suffix) for suffix in policy.ignored_generated_suffixes)
|
|
1314
|
+
|
|
1315
|
+
|
|
1316
|
+
def _workspace_files(
|
|
1317
|
+
workspace_root: Path,
|
|
1318
|
+
) -> tuple[tuple[str, ...], tuple[EvalPathPolicyViolation, ...]]:
|
|
1319
|
+
paths: list[str] = []
|
|
1320
|
+
violations: list[EvalPathPolicyViolation] = []
|
|
1321
|
+
for file_path in workspace_root.rglob("*"):
|
|
1322
|
+
if file_path.is_file():
|
|
1323
|
+
relative_path = file_path.relative_to(workspace_root).as_posix()
|
|
1324
|
+
violation = validate_eval_fixture_path(
|
|
1325
|
+
relative_path, filesystem_root=workspace_root
|
|
1326
|
+
)
|
|
1327
|
+
if violation is None:
|
|
1328
|
+
paths.append(relative_path)
|
|
1329
|
+
else:
|
|
1330
|
+
violations.append(violation)
|
|
1331
|
+
return tuple(sorted(paths)), tuple(sorted(violations, key=lambda item: item.path))
|
|
1332
|
+
|
|
1333
|
+
|
|
1334
|
+
def _unauthorized_paths(
|
|
1335
|
+
*,
|
|
1336
|
+
stage_id: EvalStageId,
|
|
1337
|
+
added_paths: tuple[str, ...],
|
|
1338
|
+
modified_paths: tuple[str, ...],
|
|
1339
|
+
deleted_paths: tuple[str, ...],
|
|
1340
|
+
policy: EvalFixtureWorkspacePolicy,
|
|
1341
|
+
) -> tuple[str, ...]:
|
|
1342
|
+
allowed_roots = policy.stage_write_roots.get(stage_id, ())
|
|
1343
|
+
changed_paths = added_paths + modified_paths + deleted_paths
|
|
1344
|
+
if not allowed_roots:
|
|
1345
|
+
return tuple(sorted(changed_paths))
|
|
1346
|
+
return tuple(
|
|
1347
|
+
sorted(
|
|
1348
|
+
path
|
|
1349
|
+
for path in changed_paths
|
|
1350
|
+
if not any(_path_is_within_root(path, root) for root in allowed_roots)
|
|
1351
|
+
)
|
|
1352
|
+
)
|
|
1353
|
+
|
|
1354
|
+
|
|
1355
|
+
def eval_fixture_workspace_snapshot(
|
|
1356
|
+
manifest: EvalFixtureManifest,
|
|
1357
|
+
workspace_root: Path,
|
|
1358
|
+
*,
|
|
1359
|
+
stage_id: EvalStageId = EvalStageId.BUILDER,
|
|
1360
|
+
) -> EvalFixtureWorkspaceSnapshot:
|
|
1361
|
+
"""Compare a workspace with its fixture manifest using relative paths only."""
|
|
1362
|
+
declared = {file.path: file for file in manifest.files}
|
|
1363
|
+
current_files: dict[str, EvalFixtureFile] = {}
|
|
1364
|
+
ignored_generated_paths: list[str] = []
|
|
1365
|
+
workspace_files, workspace_violations = _workspace_files(workspace_root)
|
|
1366
|
+
violations: list[EvalPathPolicyViolation] = list(workspace_violations)
|
|
1367
|
+
for relative_path in workspace_files:
|
|
1368
|
+
if _is_ignored_generated_path(relative_path, manifest.workspace_policy):
|
|
1369
|
+
ignored_generated_paths.append(relative_path)
|
|
1370
|
+
continue
|
|
1371
|
+
current_files[relative_path] = eval_fixture_file_record(
|
|
1372
|
+
workspace_root, relative_path
|
|
1373
|
+
)
|
|
1374
|
+
|
|
1375
|
+
declared_paths = set(declared)
|
|
1376
|
+
current_paths = set(current_files)
|
|
1377
|
+
added_paths = tuple(sorted(current_paths - declared_paths))
|
|
1378
|
+
deleted_paths = tuple(sorted(declared_paths - current_paths))
|
|
1379
|
+
unchanged_paths = tuple(
|
|
1380
|
+
sorted(
|
|
1381
|
+
path
|
|
1382
|
+
for path in declared_paths & current_paths
|
|
1383
|
+
if declared[path].sha256 == current_files[path].sha256
|
|
1384
|
+
and declared[path].size_bytes == current_files[path].size_bytes
|
|
1385
|
+
)
|
|
1386
|
+
)
|
|
1387
|
+
modified_paths = tuple(
|
|
1388
|
+
sorted((declared_paths & current_paths) - set(unchanged_paths))
|
|
1389
|
+
)
|
|
1390
|
+
unauthorized_mutation_paths = _unauthorized_paths(
|
|
1391
|
+
stage_id=stage_id,
|
|
1392
|
+
added_paths=added_paths,
|
|
1393
|
+
modified_paths=modified_paths,
|
|
1394
|
+
deleted_paths=deleted_paths,
|
|
1395
|
+
policy=manifest.workspace_policy,
|
|
1396
|
+
)
|
|
1397
|
+
unauthorized_mutation_paths = tuple(
|
|
1398
|
+
sorted(
|
|
1399
|
+
set(unauthorized_mutation_paths)
|
|
1400
|
+
| {violation.path for violation in violations}
|
|
1401
|
+
)
|
|
1402
|
+
)
|
|
1403
|
+
return EvalFixtureWorkspaceSnapshot(
|
|
1404
|
+
fixture_id=manifest.fixture_id,
|
|
1405
|
+
fixture_manifest_sha256=eval_fixture_manifest_sha256(manifest),
|
|
1406
|
+
files=tuple(current_files[path] for path in sorted(current_files)),
|
|
1407
|
+
added_paths=added_paths,
|
|
1408
|
+
modified_paths=modified_paths,
|
|
1409
|
+
deleted_paths=deleted_paths,
|
|
1410
|
+
unchanged_paths=unchanged_paths,
|
|
1411
|
+
ignored_generated_paths=tuple(sorted(ignored_generated_paths)),
|
|
1412
|
+
unauthorized_mutation_paths=unauthorized_mutation_paths,
|
|
1413
|
+
violations=tuple(violations),
|
|
1414
|
+
)
|
|
1415
|
+
|
|
1416
|
+
|
|
1417
|
+
def _path_is_within_root(path: str, root: str) -> bool:
|
|
1418
|
+
return path == root or path.startswith(f"{root}/")
|
|
1419
|
+
|
|
1420
|
+
|
|
1421
|
+
def _coerce_stage_id(stage_id: EvalStageId | str) -> EvalStageId | str:
|
|
1422
|
+
if isinstance(stage_id, EvalStageId):
|
|
1423
|
+
return stage_id
|
|
1424
|
+
try:
|
|
1425
|
+
return EvalStageId(stage_id)
|
|
1426
|
+
except ValueError:
|
|
1427
|
+
return stage_id
|
|
1428
|
+
|
|
1429
|
+
|
|
1430
|
+
def _coerce_capability_id(capability: EvalCapabilityId | str) -> EvalCapabilityId | str:
|
|
1431
|
+
if isinstance(capability, EvalCapabilityId):
|
|
1432
|
+
return capability
|
|
1433
|
+
try:
|
|
1434
|
+
return EvalCapabilityId(capability)
|
|
1435
|
+
except ValueError:
|
|
1436
|
+
return capability
|
|
1437
|
+
|
|
1438
|
+
|
|
1439
|
+
_EVAL_CAPABILITY_ENVELOPES: Mapping[EvalStageId, EvalCapabilityEnvelope] = (
|
|
1440
|
+
MappingProxyType(
|
|
1441
|
+
{
|
|
1442
|
+
EvalStageId.PLANNER: EvalCapabilityEnvelope(
|
|
1443
|
+
stage_id=EvalStageId.PLANNER,
|
|
1444
|
+
capability_ids=(
|
|
1445
|
+
EvalCapabilityId.ARTIFACT_READ,
|
|
1446
|
+
EvalCapabilityId.ARTIFACT_WRITE,
|
|
1447
|
+
EvalCapabilityId.EVIDENCE_EMIT,
|
|
1448
|
+
EvalCapabilityId.RUNNER_INVOKE,
|
|
1449
|
+
),
|
|
1450
|
+
),
|
|
1451
|
+
EvalStageId.BUILDER: EvalCapabilityEnvelope(
|
|
1452
|
+
stage_id=EvalStageId.BUILDER,
|
|
1453
|
+
capability_ids=(
|
|
1454
|
+
EvalCapabilityId.ARTIFACT_READ,
|
|
1455
|
+
EvalCapabilityId.ARTIFACT_WRITE,
|
|
1456
|
+
EvalCapabilityId.EVIDENCE_EMIT,
|
|
1457
|
+
EvalCapabilityId.RUNNER_INVOKE,
|
|
1458
|
+
EvalCapabilityId.WORKSPACE_READ,
|
|
1459
|
+
EvalCapabilityId.WORKSPACE_WRITE,
|
|
1460
|
+
EvalCapabilityId.SHELL_RUN,
|
|
1461
|
+
),
|
|
1462
|
+
),
|
|
1463
|
+
EvalStageId.CHECKER: EvalCapabilityEnvelope(
|
|
1464
|
+
stage_id=EvalStageId.CHECKER,
|
|
1465
|
+
capability_ids=(
|
|
1466
|
+
EvalCapabilityId.WORKSPACE_READ,
|
|
1467
|
+
EvalCapabilityId.ARTIFACT_READ,
|
|
1468
|
+
EvalCapabilityId.ARTIFACT_WRITE,
|
|
1469
|
+
EvalCapabilityId.SHELL_RUN,
|
|
1470
|
+
EvalCapabilityId.EVIDENCE_EMIT,
|
|
1471
|
+
EvalCapabilityId.RUNNER_INVOKE,
|
|
1472
|
+
),
|
|
1473
|
+
),
|
|
1474
|
+
EvalStageId.ARBITER: EvalCapabilityEnvelope(
|
|
1475
|
+
stage_id=EvalStageId.ARBITER,
|
|
1476
|
+
capability_ids=(
|
|
1477
|
+
EvalCapabilityId.WORKSPACE_READ,
|
|
1478
|
+
EvalCapabilityId.ARTIFACT_READ,
|
|
1479
|
+
EvalCapabilityId.ARTIFACT_WRITE,
|
|
1480
|
+
EvalCapabilityId.EVIDENCE_EMIT,
|
|
1481
|
+
EvalCapabilityId.RUNNER_INVOKE,
|
|
1482
|
+
),
|
|
1483
|
+
),
|
|
1484
|
+
}
|
|
1485
|
+
)
|
|
1486
|
+
)
|
|
1487
|
+
|
|
1488
|
+
|
|
1489
|
+
def default_eval_stage_resource_ceiling(stage_id: EvalStageId) -> EvalResourceCeiling:
|
|
1490
|
+
"""Return the default positive bounded resource ceiling for one stage."""
|
|
1491
|
+
return EvalResourceCeiling(
|
|
1492
|
+
scope="stage",
|
|
1493
|
+
stage_id=stage_id,
|
|
1494
|
+
**dict(EVAL_STAGE_RESOURCE_CEILING_DEFAULTS[stage_id]),
|
|
1495
|
+
)
|
|
1496
|
+
|
|
1497
|
+
|
|
1498
|
+
def default_eval_trial_resource_ceiling() -> EvalResourceCeiling:
|
|
1499
|
+
"""Return the default positive bounded resource ceiling for one trial."""
|
|
1500
|
+
defaults = EVAL_TRIAL_RESOURCE_CEILING_DEFAULTS
|
|
1501
|
+
return EvalResourceCeiling(
|
|
1502
|
+
scope="trial",
|
|
1503
|
+
prompt_tokens=defaults["prompt_tokens"],
|
|
1504
|
+
completion_tokens=defaults["completion_tokens"],
|
|
1505
|
+
model_calls=defaults["model_calls"],
|
|
1506
|
+
wall_clock_seconds=defaults["wall_clock_seconds"],
|
|
1507
|
+
shell_commands=defaults["shell_commands"],
|
|
1508
|
+
shell_command_seconds=defaults["shell_command_seconds"],
|
|
1509
|
+
writable_bytes=defaults["writable_bytes"],
|
|
1510
|
+
artifact_bytes=defaults["artifact_bytes"],
|
|
1511
|
+
)
|
|
1512
|
+
|
|
1513
|
+
|
|
1514
|
+
def default_eval_stage_context_policy(
|
|
1515
|
+
stage_id: EvalStageId,
|
|
1516
|
+
*,
|
|
1517
|
+
context_tier: EvalContextTier = EvalContextTier.COMPACT,
|
|
1518
|
+
) -> EvalStageContextPolicy:
|
|
1519
|
+
"""Return the default compact context policy for one 06A stage."""
|
|
1520
|
+
return EvalStageContextPolicy(
|
|
1521
|
+
stage_id=stage_id,
|
|
1522
|
+
context_tier=context_tier,
|
|
1523
|
+
allowed_capabilities=_EVAL_CAPABILITY_ENVELOPES[stage_id].capability_ids,
|
|
1524
|
+
allowed_paths=EVAL_CONTEXT_DEFAULT_ALLOWED_PATHS[stage_id],
|
|
1525
|
+
required_artifact_ids=EVAL_CONTEXT_DEFAULT_REQUIRED_ARTIFACT_IDS[stage_id],
|
|
1526
|
+
resource_ceiling=default_eval_stage_resource_ceiling(stage_id),
|
|
1527
|
+
)
|
|
1528
|
+
|
|
1529
|
+
|
|
1530
|
+
def default_eval_stage_context_policies() -> Mapping[
|
|
1531
|
+
EvalStageId, EvalStageContextPolicy
|
|
1532
|
+
]:
|
|
1533
|
+
"""Return immutable compact context policies keyed by 06A stage ID."""
|
|
1534
|
+
return MappingProxyType(
|
|
1535
|
+
{
|
|
1536
|
+
stage_id: default_eval_stage_context_policy(stage_id)
|
|
1537
|
+
for stage_id in EvalStageId
|
|
1538
|
+
}
|
|
1539
|
+
)
|
|
1540
|
+
|
|
1541
|
+
|
|
1542
|
+
def build_eval_context_snapshot(
|
|
1543
|
+
*,
|
|
1544
|
+
trial_id: str,
|
|
1545
|
+
policy: EvalStageContextPolicy,
|
|
1546
|
+
required_artifact_summaries: tuple[EvalContextArtifactSummary, ...],
|
|
1547
|
+
visible_acceptance_check_ids: tuple[str, ...],
|
|
1548
|
+
byte_budget: int | None = None,
|
|
1549
|
+
token_budget: int | None = None,
|
|
1550
|
+
) -> EvalContextSnapshot:
|
|
1551
|
+
"""Build a deterministic compact model-visible context snapshot."""
|
|
1552
|
+
graph = default_compact_eval_workflow_graph()
|
|
1553
|
+
stage_contract = next(
|
|
1554
|
+
stage for stage in graph.stages if stage.stage_id == policy.stage_id
|
|
1555
|
+
)
|
|
1556
|
+
snapshot = EvalContextSnapshot(
|
|
1557
|
+
trial_id=trial_id,
|
|
1558
|
+
stage_id=policy.stage_id,
|
|
1559
|
+
context_tier=policy.context_tier,
|
|
1560
|
+
allowed_capabilities=tuple(
|
|
1561
|
+
capability.value for capability in policy.allowed_capabilities
|
|
1562
|
+
),
|
|
1563
|
+
allowed_paths=policy.allowed_paths,
|
|
1564
|
+
current_stage_contract=stage_contract.model_dump(mode="json"),
|
|
1565
|
+
required_artifact_summaries=required_artifact_summaries,
|
|
1566
|
+
visible_acceptance_check_ids=visible_acceptance_check_ids,
|
|
1567
|
+
redaction=policy.redaction,
|
|
1568
|
+
byte_budget=byte_budget or policy.resource_ceiling.artifact_bytes,
|
|
1569
|
+
token_budget=token_budget or policy.resource_ceiling.prompt_tokens,
|
|
1570
|
+
resource_ceiling=policy.resource_ceiling,
|
|
1571
|
+
fingerprint="0" * 64,
|
|
1572
|
+
)
|
|
1573
|
+
return snapshot.model_copy(
|
|
1574
|
+
update={"fingerprint": calculate_eval_context_fingerprint(snapshot)}
|
|
1575
|
+
)
|
|
1576
|
+
|
|
1577
|
+
|
|
1578
|
+
def compact_eval_boundary_baseline() -> EvalBoundaryBaseline:
|
|
1579
|
+
"""Return the 06B module-shape baseline without rewriting 06A graph rules."""
|
|
1580
|
+
graph = default_compact_eval_workflow_graph()
|
|
1581
|
+
snapshot = compact_eval_workflow_snapshot(graph)
|
|
1582
|
+
return EvalBoundaryBaseline(
|
|
1583
|
+
graph_id=graph.graph_id,
|
|
1584
|
+
graph_sha256=snapshot["graph_sha256"],
|
|
1585
|
+
stage_ids=graph.stage_ids,
|
|
1586
|
+
terminal_results=tuple(EvalTerminalResult),
|
|
1587
|
+
outcome_kinds=tuple(EvalWorkflowOutcomeKind),
|
|
1588
|
+
candidate_dispositions=tuple(EvalCandidateDisposition),
|
|
1589
|
+
stage_artifacts=tuple(
|
|
1590
|
+
EvalBoundaryStageArtifacts(
|
|
1591
|
+
stage_id=stage.stage_id,
|
|
1592
|
+
input_artifact_ids=stage.input_artifact_ids,
|
|
1593
|
+
output_artifact_ids=stage.output_artifact_ids,
|
|
1594
|
+
)
|
|
1595
|
+
for stage in graph.stages
|
|
1596
|
+
),
|
|
1597
|
+
)
|
|
1598
|
+
|
|
1599
|
+
|
|
1600
|
+
def compact_eval_boundary_baseline_snapshot() -> dict[str, Any]:
|
|
1601
|
+
"""Return a deterministic JSON-compatible baseline snapshot."""
|
|
1602
|
+
return compact_eval_boundary_baseline().model_dump(mode="json")
|
|
1603
|
+
|
|
1604
|
+
|
|
1605
|
+
def default_eval_capability_envelopes() -> Mapping[EvalStageId, EvalCapabilityEnvelope]:
|
|
1606
|
+
"""Return immutable compact eval capability envelopes keyed by 06A stage ID."""
|
|
1607
|
+
return _EVAL_CAPABILITY_ENVELOPES
|
|
1608
|
+
|
|
1609
|
+
|
|
1610
|
+
def eval_stage_capability_envelope(stage_id: EvalStageId) -> EvalCapabilityEnvelope:
|
|
1611
|
+
"""Return the immutable capability envelope for one compact eval stage."""
|
|
1612
|
+
return _EVAL_CAPABILITY_ENVELOPES[stage_id]
|
|
1613
|
+
|
|
1614
|
+
|
|
1615
|
+
def validate_eval_stage_capability(
|
|
1616
|
+
stage_id: EvalStageId | str, capability: EvalCapabilityId | str
|
|
1617
|
+
) -> EvalCapabilityValidationResult:
|
|
1618
|
+
"""Validate one capability request against deterministic compact eval policy."""
|
|
1619
|
+
resolved_stage_id = _coerce_stage_id(stage_id)
|
|
1620
|
+
resolved_capability = _coerce_capability_id(capability)
|
|
1621
|
+
if not isinstance(resolved_stage_id, EvalStageId):
|
|
1622
|
+
return EvalCapabilityValidationResult(
|
|
1623
|
+
stage_id=resolved_stage_id,
|
|
1624
|
+
capability_id=resolved_capability,
|
|
1625
|
+
allowed=False,
|
|
1626
|
+
rule_id="eval.capability.unknown_stage",
|
|
1627
|
+
diagnostic_code="MF-EVAL-C001",
|
|
1628
|
+
diagnostic_summary="unknown compact eval stage id",
|
|
1629
|
+
)
|
|
1630
|
+
if not isinstance(resolved_capability, EvalCapabilityId):
|
|
1631
|
+
return EvalCapabilityValidationResult(
|
|
1632
|
+
stage_id=resolved_stage_id,
|
|
1633
|
+
capability_id=resolved_capability,
|
|
1634
|
+
allowed=False,
|
|
1635
|
+
rule_id="eval.capability.unknown_capability",
|
|
1636
|
+
diagnostic_code="MF-EVAL-C002",
|
|
1637
|
+
diagnostic_summary="unknown compact eval capability id",
|
|
1638
|
+
)
|
|
1639
|
+
envelope = _EVAL_CAPABILITY_ENVELOPES[resolved_stage_id]
|
|
1640
|
+
if resolved_capability in envelope.capability_ids:
|
|
1641
|
+
return EvalCapabilityValidationResult(
|
|
1642
|
+
stage_id=resolved_stage_id,
|
|
1643
|
+
capability_id=resolved_capability,
|
|
1644
|
+
allowed=True,
|
|
1645
|
+
rule_id="eval.capability.allowed",
|
|
1646
|
+
)
|
|
1647
|
+
if resolved_capability.value in EVAL_DENIED_CAPABILITY_IDS:
|
|
1648
|
+
return EvalCapabilityValidationResult(
|
|
1649
|
+
stage_id=resolved_stage_id,
|
|
1650
|
+
capability_id=resolved_capability,
|
|
1651
|
+
allowed=False,
|
|
1652
|
+
rule_id="eval.capability.denied_dangerous_all_stages",
|
|
1653
|
+
diagnostic_code="MF-EVAL-C003",
|
|
1654
|
+
diagnostic_summary="capability is denied for every compact eval stage",
|
|
1655
|
+
)
|
|
1656
|
+
return EvalCapabilityValidationResult(
|
|
1657
|
+
stage_id=resolved_stage_id,
|
|
1658
|
+
capability_id=resolved_capability,
|
|
1659
|
+
allowed=False,
|
|
1660
|
+
rule_id=f"eval.capability.denied.{resolved_stage_id.value}",
|
|
1661
|
+
diagnostic_code="MF-EVAL-C004",
|
|
1662
|
+
diagnostic_summary="capability is not in the stage envelope",
|
|
1663
|
+
)
|
|
1664
|
+
|
|
1665
|
+
|
|
1666
|
+
def validate_eval_stage_command(
|
|
1667
|
+
stage_id: EvalStageId | str,
|
|
1668
|
+
descriptor: EvalCommandDescriptor,
|
|
1669
|
+
*,
|
|
1670
|
+
builder_allowed_write_roots: tuple[str, ...] = EVAL_BUILDER_DEFAULT_WRITE_ROOTS,
|
|
1671
|
+
) -> EvalCommandAdmissionResult:
|
|
1672
|
+
"""Validate a deterministic command descriptor for one compact eval stage."""
|
|
1673
|
+
resolved_stage_id = _coerce_stage_id(stage_id)
|
|
1674
|
+
if not isinstance(resolved_stage_id, EvalStageId):
|
|
1675
|
+
return EvalCommandAdmissionResult(
|
|
1676
|
+
stage_id=resolved_stage_id,
|
|
1677
|
+
command_id=descriptor.command_id,
|
|
1678
|
+
allowed=False,
|
|
1679
|
+
rule_id="eval.command.unknown_stage",
|
|
1680
|
+
diagnostic_code="MF-EVAL-D001",
|
|
1681
|
+
diagnostic_summary="unknown compact eval stage id",
|
|
1682
|
+
)
|
|
1683
|
+
shell_result = validate_eval_stage_capability(
|
|
1684
|
+
resolved_stage_id, EvalCapabilityId.SHELL_RUN
|
|
1685
|
+
)
|
|
1686
|
+
if not shell_result.allowed:
|
|
1687
|
+
return EvalCommandAdmissionResult(
|
|
1688
|
+
stage_id=resolved_stage_id,
|
|
1689
|
+
command_id=descriptor.command_id,
|
|
1690
|
+
allowed=False,
|
|
1691
|
+
rule_id="eval.command.stage_has_no_shell",
|
|
1692
|
+
diagnostic_code="MF-EVAL-D002",
|
|
1693
|
+
diagnostic_summary="stage cannot run shell commands",
|
|
1694
|
+
)
|
|
1695
|
+
if disallowed_message := _disallowed_argv_message(descriptor.argv):
|
|
1696
|
+
return EvalCommandAdmissionResult(
|
|
1697
|
+
stage_id=resolved_stage_id,
|
|
1698
|
+
command_id=descriptor.command_id,
|
|
1699
|
+
allowed=False,
|
|
1700
|
+
rule_id="eval.command.descriptor_unsafe",
|
|
1701
|
+
diagnostic_code="MF-EVAL-D005",
|
|
1702
|
+
diagnostic_summary=disallowed_message,
|
|
1703
|
+
)
|
|
1704
|
+
if resolved_stage_id == EvalStageId.CHECKER:
|
|
1705
|
+
invalid_checker_writes = tuple(
|
|
1706
|
+
root
|
|
1707
|
+
for root in descriptor.admitted_write_roots
|
|
1708
|
+
if not any(
|
|
1709
|
+
_path_is_within_root(root, scratch)
|
|
1710
|
+
for scratch in EVAL_CHECKER_IGNORED_SCRATCH_ROOTS
|
|
1711
|
+
)
|
|
1712
|
+
)
|
|
1713
|
+
if invalid_checker_writes:
|
|
1714
|
+
return EvalCommandAdmissionResult(
|
|
1715
|
+
stage_id=resolved_stage_id,
|
|
1716
|
+
command_id=descriptor.command_id,
|
|
1717
|
+
allowed=False,
|
|
1718
|
+
rule_id="eval.command.checker_write_denied",
|
|
1719
|
+
diagnostic_code="MF-EVAL-D003",
|
|
1720
|
+
diagnostic_summary="checker commands may write only ignored scratch outputs",
|
|
1721
|
+
)
|
|
1722
|
+
if resolved_stage_id == EvalStageId.BUILDER:
|
|
1723
|
+
for root in builder_allowed_write_roots:
|
|
1724
|
+
_validate_relative_eval_path(root, allow_dot=False)
|
|
1725
|
+
invalid_builder_writes = tuple(
|
|
1726
|
+
root
|
|
1727
|
+
for root in descriptor.admitted_write_roots
|
|
1728
|
+
if not any(
|
|
1729
|
+
_path_is_within_root(root, allowed_root)
|
|
1730
|
+
for allowed_root in builder_allowed_write_roots
|
|
1731
|
+
)
|
|
1732
|
+
)
|
|
1733
|
+
if invalid_builder_writes:
|
|
1734
|
+
return EvalCommandAdmissionResult(
|
|
1735
|
+
stage_id=resolved_stage_id,
|
|
1736
|
+
command_id=descriptor.command_id,
|
|
1737
|
+
allowed=False,
|
|
1738
|
+
rule_id="eval.command.builder_write_root_denied",
|
|
1739
|
+
diagnostic_code="MF-EVAL-D004",
|
|
1740
|
+
diagnostic_summary="builder command writes outside allowed roots",
|
|
1741
|
+
)
|
|
1742
|
+
return EvalCommandAdmissionResult(
|
|
1743
|
+
stage_id=resolved_stage_id,
|
|
1744
|
+
command_id=descriptor.command_id,
|
|
1745
|
+
allowed=True,
|
|
1746
|
+
rule_id="eval.command.allowed",
|
|
1747
|
+
)
|
|
1748
|
+
|
|
1749
|
+
|
|
1750
|
+
def _closure_result(
|
|
1751
|
+
outcome_kind: EvalClosureOutcomeKind,
|
|
1752
|
+
*,
|
|
1753
|
+
terminal_result: EvalTerminalResult | None = None,
|
|
1754
|
+
candidate_disposition: EvalCandidateDisposition | None = None,
|
|
1755
|
+
evidence_artifact_ids: tuple[str, ...] = (),
|
|
1756
|
+
missing_artifact_ids: tuple[str, ...] = (),
|
|
1757
|
+
diagnostics: tuple[str, ...] = (),
|
|
1758
|
+
) -> EvalClosureValidationResult:
|
|
1759
|
+
return EvalClosureValidationResult(
|
|
1760
|
+
valid=outcome_kind
|
|
1761
|
+
in {
|
|
1762
|
+
EvalClosureOutcomeKind.VALID_CLOSED_SUCCESS,
|
|
1763
|
+
EvalClosureOutcomeKind.VALID_CLOSED_REJECTION,
|
|
1764
|
+
EvalClosureOutcomeKind.VALID_BLOCKED_OUTCOME,
|
|
1765
|
+
},
|
|
1766
|
+
outcome_kind=outcome_kind,
|
|
1767
|
+
terminal_result=terminal_result,
|
|
1768
|
+
candidate_disposition=candidate_disposition,
|
|
1769
|
+
evidence_artifact_ids=evidence_artifact_ids,
|
|
1770
|
+
missing_artifact_ids=missing_artifact_ids,
|
|
1771
|
+
diagnostics=diagnostics,
|
|
1772
|
+
)
|
|
1773
|
+
|
|
1774
|
+
|
|
1775
|
+
def _artifact_bundle_items(
|
|
1776
|
+
artifact_bundle: Mapping[Any, Any],
|
|
1777
|
+
) -> dict[str, Any]:
|
|
1778
|
+
artifacts: dict[str, Any] = {}
|
|
1779
|
+
for raw_artifact_id, record in artifact_bundle.items():
|
|
1780
|
+
artifact_id = getattr(raw_artifact_id, "value", raw_artifact_id)
|
|
1781
|
+
artifacts[str(artifact_id)] = record
|
|
1782
|
+
return artifacts
|
|
1783
|
+
|
|
1784
|
+
|
|
1785
|
+
def _artifact_record_mapping(record: Any) -> Mapping[str, Any]:
|
|
1786
|
+
if isinstance(record, BaseModel):
|
|
1787
|
+
return record.model_dump(mode="json")
|
|
1788
|
+
if isinstance(record, Mapping):
|
|
1789
|
+
return record
|
|
1790
|
+
raise TypeError("artifact records must be mappings or Pydantic models")
|
|
1791
|
+
|
|
1792
|
+
|
|
1793
|
+
def _fixture_manifest_from_artifact_record(record: Any) -> EvalFixtureManifest:
|
|
1794
|
+
from millforge.eval_artifacts import EvalFixtureManifestArtifact
|
|
1795
|
+
|
|
1796
|
+
if isinstance(record, EvalFixtureManifest):
|
|
1797
|
+
return record
|
|
1798
|
+
if isinstance(record, EvalFixtureManifestArtifact):
|
|
1799
|
+
return record.fixture_manifest
|
|
1800
|
+
artifact = EvalFixtureManifestArtifact.model_validate(
|
|
1801
|
+
_artifact_record_mapping(record)
|
|
1802
|
+
)
|
|
1803
|
+
return artifact.fixture_manifest
|
|
1804
|
+
|
|
1805
|
+
|
|
1806
|
+
def _validate_present_closure_artifacts(
|
|
1807
|
+
artifact_bundle: Mapping[Any, Any],
|
|
1808
|
+
) -> tuple[dict[str, BaseModel], tuple[str, ...]]:
|
|
1809
|
+
from millforge.eval_artifacts import validate_eval_artifact_record
|
|
1810
|
+
|
|
1811
|
+
artifacts = _artifact_bundle_items(artifact_bundle)
|
|
1812
|
+
validated: dict[str, BaseModel] = {}
|
|
1813
|
+
diagnostics: list[str] = []
|
|
1814
|
+
for artifact_id, record in artifacts.items():
|
|
1815
|
+
try:
|
|
1816
|
+
validated[artifact_id] = validate_eval_artifact_record(
|
|
1817
|
+
artifact_id, _artifact_record_mapping(record)
|
|
1818
|
+
)
|
|
1819
|
+
except (TypeError, ValueError) as exc:
|
|
1820
|
+
diagnostics.append(f"{artifact_id}: {exc}")
|
|
1821
|
+
return validated, tuple(diagnostics)
|
|
1822
|
+
|
|
1823
|
+
|
|
1824
|
+
def _required_closure_artifact_ids(
|
|
1825
|
+
outcome_kind: EvalClosureOutcomeKind,
|
|
1826
|
+
) -> tuple[str, ...]:
|
|
1827
|
+
from millforge.eval_artifacts import EVAL_LOGICAL_06A_ARTIFACT_IDS
|
|
1828
|
+
|
|
1829
|
+
base_artifact_ids = (
|
|
1830
|
+
"task",
|
|
1831
|
+
"fixture_manifest",
|
|
1832
|
+
"acceptance_checks",
|
|
1833
|
+
"plan",
|
|
1834
|
+
"arbiter_verdict",
|
|
1835
|
+
"context_snapshot",
|
|
1836
|
+
)
|
|
1837
|
+
if outcome_kind == EvalClosureOutcomeKind.VALID_BLOCKED_OUTCOME:
|
|
1838
|
+
return base_artifact_ids
|
|
1839
|
+
return EVAL_LOGICAL_06A_ARTIFACT_IDS + ("context_snapshot",)
|
|
1840
|
+
|
|
1841
|
+
|
|
1842
|
+
def _capability_snapshot_results(
|
|
1843
|
+
capability_snapshot: Any,
|
|
1844
|
+
) -> tuple[EvalCapabilityValidationResult, ...]:
|
|
1845
|
+
if isinstance(capability_snapshot, Mapping) and "results" in capability_snapshot:
|
|
1846
|
+
return _capability_snapshot_results(capability_snapshot["results"])
|
|
1847
|
+
if isinstance(capability_snapshot, Mapping):
|
|
1848
|
+
results: list[EvalCapabilityValidationResult] = []
|
|
1849
|
+
for raw_stage_id, capabilities in capability_snapshot.items():
|
|
1850
|
+
stage_id = _coerce_stage_id(raw_stage_id)
|
|
1851
|
+
if isinstance(capabilities, EvalCapabilityEnvelope):
|
|
1852
|
+
capability_ids = capabilities.capability_ids
|
|
1853
|
+
elif isinstance(capabilities, Mapping) and "capability_ids" in capabilities:
|
|
1854
|
+
capability_ids = tuple(capabilities["capability_ids"])
|
|
1855
|
+
else:
|
|
1856
|
+
capability_ids = tuple(capabilities)
|
|
1857
|
+
for capability in capability_ids:
|
|
1858
|
+
results.append(validate_eval_stage_capability(stage_id, capability))
|
|
1859
|
+
return tuple(results)
|
|
1860
|
+
results = []
|
|
1861
|
+
for item in tuple(capability_snapshot):
|
|
1862
|
+
if isinstance(item, EvalCapabilityValidationResult):
|
|
1863
|
+
results.append(item)
|
|
1864
|
+
elif isinstance(item, EvalCapabilityEnvelope):
|
|
1865
|
+
results.extend(
|
|
1866
|
+
validate_eval_stage_capability(item.stage_id, capability)
|
|
1867
|
+
for capability in item.capability_ids
|
|
1868
|
+
)
|
|
1869
|
+
elif isinstance(item, Mapping) and "allowed" in item:
|
|
1870
|
+
results.append(EvalCapabilityValidationResult.model_validate(item))
|
|
1871
|
+
elif isinstance(item, Mapping) and "capability_ids" in item:
|
|
1872
|
+
stage_id = item["stage_id"]
|
|
1873
|
+
results.extend(
|
|
1874
|
+
validate_eval_stage_capability(stage_id, capability)
|
|
1875
|
+
for capability in item["capability_ids"]
|
|
1876
|
+
)
|
|
1877
|
+
else:
|
|
1878
|
+
raise TypeError("capability snapshot entries are not recognized")
|
|
1879
|
+
return tuple(results)
|
|
1880
|
+
|
|
1881
|
+
|
|
1882
|
+
def _capability_snapshot_diagnostics(capability_snapshot: Any) -> tuple[str, ...]:
|
|
1883
|
+
try:
|
|
1884
|
+
results = _capability_snapshot_results(capability_snapshot)
|
|
1885
|
+
except TypeError as exc:
|
|
1886
|
+
return (str(exc),)
|
|
1887
|
+
if not results:
|
|
1888
|
+
return ("capability snapshot must include admitted capability evidence",)
|
|
1889
|
+
result_stage_ids = {
|
|
1890
|
+
result.stage_id
|
|
1891
|
+
for result in results
|
|
1892
|
+
if isinstance(result.stage_id, EvalStageId)
|
|
1893
|
+
}
|
|
1894
|
+
missing_stage_ids = tuple(
|
|
1895
|
+
stage_id.value for stage_id in EvalStageId if stage_id not in result_stage_ids
|
|
1896
|
+
)
|
|
1897
|
+
diagnostics = [
|
|
1898
|
+
f"{result.stage_id}: {result.capability_id}: {result.rule_id}"
|
|
1899
|
+
for result in results
|
|
1900
|
+
if not result.allowed
|
|
1901
|
+
]
|
|
1902
|
+
if missing_stage_ids:
|
|
1903
|
+
diagnostics.append(
|
|
1904
|
+
"capability snapshot missing stages: " + ", ".join(missing_stage_ids)
|
|
1905
|
+
)
|
|
1906
|
+
return tuple(diagnostics)
|
|
1907
|
+
|
|
1908
|
+
|
|
1909
|
+
def _fixture_snapshot_diagnostics(
|
|
1910
|
+
fixture_snapshot: EvalFixtureWorkspaceSnapshot | Mapping[str, Any],
|
|
1911
|
+
artifact_bundle: Mapping[Any, Any],
|
|
1912
|
+
) -> tuple[str, ...]:
|
|
1913
|
+
snapshot = (
|
|
1914
|
+
fixture_snapshot
|
|
1915
|
+
if isinstance(fixture_snapshot, EvalFixtureWorkspaceSnapshot)
|
|
1916
|
+
else EvalFixtureWorkspaceSnapshot.model_validate(fixture_snapshot)
|
|
1917
|
+
)
|
|
1918
|
+
diagnostics: list[str] = []
|
|
1919
|
+
if snapshot.unauthorized_mutation_paths:
|
|
1920
|
+
diagnostics.append("fixture snapshot includes unauthorized mutations")
|
|
1921
|
+
if snapshot.violations:
|
|
1922
|
+
diagnostics.append("fixture snapshot includes path policy violations")
|
|
1923
|
+
|
|
1924
|
+
artifacts = _artifact_bundle_items(artifact_bundle)
|
|
1925
|
+
manifest_record = artifacts.get("fixture_manifest")
|
|
1926
|
+
if manifest_record is not None:
|
|
1927
|
+
manifest = _fixture_manifest_from_artifact_record(manifest_record)
|
|
1928
|
+
if snapshot.fixture_id != manifest.fixture_id:
|
|
1929
|
+
diagnostics.append("fixture snapshot fixture_id does not match manifest")
|
|
1930
|
+
if snapshot.fixture_manifest_sha256 != eval_fixture_manifest_sha256(manifest):
|
|
1931
|
+
diagnostics.append(
|
|
1932
|
+
"fixture snapshot manifest digest does not match expanded manifest"
|
|
1933
|
+
)
|
|
1934
|
+
return tuple(diagnostics)
|
|
1935
|
+
|
|
1936
|
+
|
|
1937
|
+
def _fixture_snapshot_mutation_paths(
|
|
1938
|
+
fixture_snapshot: EvalFixtureWorkspaceSnapshot | Mapping[str, Any],
|
|
1939
|
+
) -> tuple[str, ...]:
|
|
1940
|
+
snapshot = (
|
|
1941
|
+
fixture_snapshot
|
|
1942
|
+
if isinstance(fixture_snapshot, EvalFixtureWorkspaceSnapshot)
|
|
1943
|
+
else EvalFixtureWorkspaceSnapshot.model_validate(fixture_snapshot)
|
|
1944
|
+
)
|
|
1945
|
+
return snapshot.added_paths + snapshot.modified_paths + snapshot.deleted_paths
|
|
1946
|
+
|
|
1947
|
+
|
|
1948
|
+
def _context_boundary_diagnostics(
|
|
1949
|
+
validated_artifacts: Mapping[str, BaseModel],
|
|
1950
|
+
) -> tuple[str, ...]:
|
|
1951
|
+
acceptance = validated_artifacts.get("acceptance_checks")
|
|
1952
|
+
context = validated_artifacts.get("context_snapshot")
|
|
1953
|
+
if acceptance is None or context is None:
|
|
1954
|
+
return ("closure context validation requires acceptance and context artifacts",)
|
|
1955
|
+
|
|
1956
|
+
visible_checks = tuple(
|
|
1957
|
+
check.check_id for check in getattr(acceptance, "visible_acceptance_checks", ())
|
|
1958
|
+
)
|
|
1959
|
+
context_visible_checks = tuple(getattr(context, "visible_acceptance_check_ids", ()))
|
|
1960
|
+
if not visible_checks:
|
|
1961
|
+
return ("visible acceptance checks must be declared",)
|
|
1962
|
+
if set(context_visible_checks) != set(visible_checks):
|
|
1963
|
+
return (
|
|
1964
|
+
"context visible acceptance check IDs must match acceptance artifact IDs",
|
|
1965
|
+
)
|
|
1966
|
+
return ()
|
|
1967
|
+
|
|
1968
|
+
|
|
1969
|
+
def _closure_evidence_artifact_ids(
|
|
1970
|
+
validated_artifacts: Mapping[str, BaseModel],
|
|
1971
|
+
) -> tuple[str, ...]:
|
|
1972
|
+
arbiter_verdict = validated_artifacts["arbiter_verdict"]
|
|
1973
|
+
references = getattr(arbiter_verdict, "closure_evidence_references", ())
|
|
1974
|
+
return tuple(reference.artifact_id.value for reference in references)
|
|
1975
|
+
|
|
1976
|
+
|
|
1977
|
+
def _closure_outcome_kind(
|
|
1978
|
+
validated_artifacts: Mapping[str, BaseModel],
|
|
1979
|
+
) -> tuple[EvalClosureOutcomeKind, EvalTerminalResult, EvalCandidateDisposition]:
|
|
1980
|
+
from millforge.eval_artifacts import (
|
|
1981
|
+
EvalArbiterVerdictArtifact,
|
|
1982
|
+
EvalArbiterVerdictValue,
|
|
1983
|
+
)
|
|
1984
|
+
|
|
1985
|
+
arbiter_verdict = cast(
|
|
1986
|
+
EvalArbiterVerdictArtifact, validated_artifacts["arbiter_verdict"]
|
|
1987
|
+
)
|
|
1988
|
+
verdict = arbiter_verdict.verdict
|
|
1989
|
+
disposition = arbiter_verdict.candidate_disposition
|
|
1990
|
+
if verdict == EvalArbiterVerdictValue.BLOCKED:
|
|
1991
|
+
return (
|
|
1992
|
+
EvalClosureOutcomeKind.VALID_BLOCKED_OUTCOME,
|
|
1993
|
+
EvalTerminalResult.ARBITER_BLOCKED,
|
|
1994
|
+
EvalCandidateDisposition.BLOCKED,
|
|
1995
|
+
)
|
|
1996
|
+
if verdict == EvalArbiterVerdictValue.REJECTED:
|
|
1997
|
+
return (
|
|
1998
|
+
EvalClosureOutcomeKind.VALID_CLOSED_REJECTION,
|
|
1999
|
+
EvalTerminalResult.ARBITER_REJECTED,
|
|
2000
|
+
EvalCandidateDisposition.REJECTED,
|
|
2001
|
+
)
|
|
2002
|
+
if disposition == EvalCandidateDisposition.REJECTED:
|
|
2003
|
+
return (
|
|
2004
|
+
EvalClosureOutcomeKind.VALID_CLOSED_REJECTION,
|
|
2005
|
+
EvalTerminalResult.ARBITER_REJECTED,
|
|
2006
|
+
disposition,
|
|
2007
|
+
)
|
|
2008
|
+
return (
|
|
2009
|
+
EvalClosureOutcomeKind.VALID_CLOSED_SUCCESS,
|
|
2010
|
+
EvalTerminalResult.ARBITER_CLOSED,
|
|
2011
|
+
disposition,
|
|
2012
|
+
)
|
|
2013
|
+
|
|
2014
|
+
|
|
2015
|
+
def _terminal_path_diagnostics(
|
|
2016
|
+
validated_artifacts: Mapping[str, BaseModel],
|
|
2017
|
+
outcome_kind: EvalClosureOutcomeKind,
|
|
2018
|
+
) -> tuple[str, ...]:
|
|
2019
|
+
from millforge.eval_artifacts import (
|
|
2020
|
+
EvalArbiterVerdictArtifact,
|
|
2021
|
+
EvalArbiterVerdictValue,
|
|
2022
|
+
)
|
|
2023
|
+
|
|
2024
|
+
arbiter_verdict_base = validated_artifacts.get("arbiter_verdict")
|
|
2025
|
+
if arbiter_verdict_base is None:
|
|
2026
|
+
return ("arbiter verdict is required to prove terminal path",)
|
|
2027
|
+
arbiter_verdict = cast(EvalArbiterVerdictArtifact, arbiter_verdict_base)
|
|
2028
|
+
|
|
2029
|
+
verdict = arbiter_verdict.verdict
|
|
2030
|
+
disposition = arbiter_verdict.candidate_disposition
|
|
2031
|
+
if verdict == EvalArbiterVerdictValue.BLOCKED and (
|
|
2032
|
+
outcome_kind != EvalClosureOutcomeKind.VALID_BLOCKED_OUTCOME
|
|
2033
|
+
or disposition != EvalCandidateDisposition.BLOCKED
|
|
2034
|
+
):
|
|
2035
|
+
return ("blocked terminal path requires blocked candidate disposition",)
|
|
2036
|
+
if verdict != EvalArbiterVerdictValue.BLOCKED and (
|
|
2037
|
+
outcome_kind == EvalClosureOutcomeKind.VALID_BLOCKED_OUTCOME
|
|
2038
|
+
or disposition == EvalCandidateDisposition.BLOCKED
|
|
2039
|
+
):
|
|
2040
|
+
return ("blocked candidate disposition requires blocked arbiter verdict",)
|
|
2041
|
+
return ()
|
|
2042
|
+
|
|
2043
|
+
|
|
2044
|
+
def _checker_gate_diagnostics(
|
|
2045
|
+
validated_artifacts: Mapping[str, BaseModel],
|
|
2046
|
+
outcome_kind: EvalClosureOutcomeKind,
|
|
2047
|
+
) -> tuple[str, ...]:
|
|
2048
|
+
from millforge.eval_artifacts import (
|
|
2049
|
+
EvalCheckerVerdictArtifact,
|
|
2050
|
+
EvalCheckerVerdictValue,
|
|
2051
|
+
)
|
|
2052
|
+
|
|
2053
|
+
checker_base = validated_artifacts.get("checker_verdict")
|
|
2054
|
+
if (
|
|
2055
|
+
checker_base is None
|
|
2056
|
+
or outcome_kind == EvalClosureOutcomeKind.VALID_BLOCKED_OUTCOME
|
|
2057
|
+
):
|
|
2058
|
+
return ()
|
|
2059
|
+
checker = cast(EvalCheckerVerdictArtifact, checker_base)
|
|
2060
|
+
if outcome_kind == EvalClosureOutcomeKind.VALID_CLOSED_SUCCESS:
|
|
2061
|
+
if checker.verdict != EvalCheckerVerdictValue.APPROVED:
|
|
2062
|
+
return ("arbiter closure requires approved checker verdict",)
|
|
2063
|
+
elif checker.verdict == EvalCheckerVerdictValue.BLOCKED:
|
|
2064
|
+
return ("arbiter rejection requires non-blocked checker verdict",)
|
|
2065
|
+
return ()
|
|
2066
|
+
|
|
2067
|
+
|
|
2068
|
+
def _test_results_failed(test_results: BaseModel) -> bool:
|
|
2069
|
+
return (
|
|
2070
|
+
getattr(test_results, "exit_code", 0) != 0
|
|
2071
|
+
or getattr(test_results, "failed_count", 0) != 0
|
|
2072
|
+
or not getattr(test_results, "deterministic", True)
|
|
2073
|
+
or not getattr(test_results, "allowed_by_policy", True)
|
|
2074
|
+
)
|
|
2075
|
+
|
|
2076
|
+
|
|
2077
|
+
def _failed_command_outcomes(patch_summary: BaseModel) -> tuple[Any, ...]:
|
|
2078
|
+
return tuple(
|
|
2079
|
+
outcome
|
|
2080
|
+
for outcome in getattr(patch_summary, "command_outcomes", ())
|
|
2081
|
+
if outcome.exit_code != 0
|
|
2082
|
+
)
|
|
2083
|
+
|
|
2084
|
+
|
|
2085
|
+
def _command_status_diagnostics(
|
|
2086
|
+
validated_artifacts: Mapping[str, BaseModel],
|
|
2087
|
+
outcome_kind: EvalClosureOutcomeKind,
|
|
2088
|
+
) -> tuple[str, ...]:
|
|
2089
|
+
if outcome_kind != EvalClosureOutcomeKind.VALID_CLOSED_SUCCESS:
|
|
2090
|
+
return ()
|
|
2091
|
+
|
|
2092
|
+
diagnostics: list[str] = []
|
|
2093
|
+
test_results = validated_artifacts.get("test_results")
|
|
2094
|
+
if test_results is not None and _test_results_failed(test_results):
|
|
2095
|
+
diagnostics.append("arbiter closure cannot claim success with failed tests")
|
|
2096
|
+
|
|
2097
|
+
patch_summary = validated_artifacts.get("patch_summary")
|
|
2098
|
+
if patch_summary is not None:
|
|
2099
|
+
if _failed_command_outcomes(patch_summary):
|
|
2100
|
+
diagnostics.append(
|
|
2101
|
+
"arbiter closure cannot claim success with failed static checks"
|
|
2102
|
+
)
|
|
2103
|
+
if getattr(patch_summary, "unresolved_issues", ()):
|
|
2104
|
+
diagnostics.append(
|
|
2105
|
+
"arbiter closure cannot claim success with unresolved builder issues"
|
|
2106
|
+
)
|
|
2107
|
+
return tuple(diagnostics)
|
|
2108
|
+
|
|
2109
|
+
|
|
2110
|
+
def _mutation_evidence_diagnostics(
|
|
2111
|
+
validated_artifacts: Mapping[str, BaseModel],
|
|
2112
|
+
outcome_kind: EvalClosureOutcomeKind,
|
|
2113
|
+
) -> tuple[str, ...]:
|
|
2114
|
+
if outcome_kind != EvalClosureOutcomeKind.VALID_CLOSED_SUCCESS:
|
|
2115
|
+
return ()
|
|
2116
|
+
|
|
2117
|
+
manifest_base = validated_artifacts.get("fixture_manifest")
|
|
2118
|
+
workspace_diff = validated_artifacts.get("workspace_diff")
|
|
2119
|
+
if manifest_base is None:
|
|
2120
|
+
return ()
|
|
2121
|
+
manifest = _fixture_manifest_from_artifact_record(manifest_base)
|
|
2122
|
+
if not manifest.expected_mutation_paths:
|
|
2123
|
+
return ()
|
|
2124
|
+
if workspace_diff is None:
|
|
2125
|
+
return ("required mutation paths require workspace_diff artifact",)
|
|
2126
|
+
changed_paths = (
|
|
2127
|
+
tuple(getattr(workspace_diff, "added_paths", ()))
|
|
2128
|
+
+ tuple(getattr(workspace_diff, "modified_paths", ()))
|
|
2129
|
+
+ tuple(getattr(workspace_diff, "deleted_paths", ()))
|
|
2130
|
+
)
|
|
2131
|
+
if not changed_paths:
|
|
2132
|
+
return ("required mutation paths require non-empty workspace_diff changes",)
|
|
2133
|
+
return ()
|
|
2134
|
+
|
|
2135
|
+
|
|
2136
|
+
def _arbiter_acceptance_diagnostics(
|
|
2137
|
+
validated_artifacts: Mapping[str, BaseModel],
|
|
2138
|
+
outcome_kind: EvalClosureOutcomeKind,
|
|
2139
|
+
) -> tuple[str, ...]:
|
|
2140
|
+
if outcome_kind != EvalClosureOutcomeKind.VALID_CLOSED_SUCCESS:
|
|
2141
|
+
return ()
|
|
2142
|
+
arbiter_verdict = validated_artifacts.get("arbiter_verdict")
|
|
2143
|
+
if arbiter_verdict is None:
|
|
2144
|
+
return ()
|
|
2145
|
+
if getattr(arbiter_verdict, "open_acceptance_check_ids", ()):
|
|
2146
|
+
return ("arbiter closure cannot leave required acceptance checks open",)
|
|
2147
|
+
return ()
|
|
2148
|
+
|
|
2149
|
+
|
|
2150
|
+
def _arbiter_stage_result_diagnostics(
|
|
2151
|
+
validated_artifacts: Mapping[str, BaseModel],
|
|
2152
|
+
inferred_terminal_result: EvalTerminalResult,
|
|
2153
|
+
) -> tuple[str, ...]:
|
|
2154
|
+
stage_result = validated_artifacts.get("stage_result")
|
|
2155
|
+
if stage_result is None:
|
|
2156
|
+
return ()
|
|
2157
|
+
if getattr(stage_result, "stage_id", None) != EvalStageId.ARBITER:
|
|
2158
|
+
return ("closure stage_result must belong to eval_arbiter",)
|
|
2159
|
+
terminal_result = getattr(stage_result, "terminal_result", None)
|
|
2160
|
+
if (
|
|
2161
|
+
terminal_result
|
|
2162
|
+
not in default_compact_eval_workflow_graph()
|
|
2163
|
+
.stage_contracts[EvalStageId.ARBITER]
|
|
2164
|
+
.legal_terminal_results
|
|
2165
|
+
):
|
|
2166
|
+
return ("closure stage_result terminal is illegal for eval_arbiter",)
|
|
2167
|
+
if terminal_result != inferred_terminal_result:
|
|
2168
|
+
return ("closure stage_result terminal does not match arbiter verdict",)
|
|
2169
|
+
return ()
|
|
2170
|
+
|
|
2171
|
+
|
|
2172
|
+
def validate_eval_checker_approval(
|
|
2173
|
+
artifact_bundle: Mapping[Any, Any],
|
|
2174
|
+
) -> EvalCheckerApprovalValidationResult:
|
|
2175
|
+
"""Validate a Checker verdict against visible public execution evidence."""
|
|
2176
|
+
from millforge.eval_artifacts import (
|
|
2177
|
+
EvalCheckerVerdictArtifact,
|
|
2178
|
+
EvalCheckerVerdictValue,
|
|
2179
|
+
)
|
|
2180
|
+
|
|
2181
|
+
validated_artifacts, artifact_diagnostics = _validate_present_closure_artifacts(
|
|
2182
|
+
artifact_bundle
|
|
2183
|
+
)
|
|
2184
|
+
if artifact_diagnostics:
|
|
2185
|
+
return EvalCheckerApprovalValidationResult(
|
|
2186
|
+
valid=False,
|
|
2187
|
+
diagnostics=artifact_diagnostics,
|
|
2188
|
+
)
|
|
2189
|
+
|
|
2190
|
+
checker_base = validated_artifacts.get("checker_verdict")
|
|
2191
|
+
if checker_base is None:
|
|
2192
|
+
return EvalCheckerApprovalValidationResult(
|
|
2193
|
+
valid=False,
|
|
2194
|
+
missing_artifact_ids=("checker_verdict",),
|
|
2195
|
+
diagnostics=("checker approval validation requires checker_verdict",),
|
|
2196
|
+
)
|
|
2197
|
+
|
|
2198
|
+
checker = cast(EvalCheckerVerdictArtifact, checker_base)
|
|
2199
|
+
evidence_artifact_ids = tuple(
|
|
2200
|
+
reference.artifact_id.value for reference in checker.evidence_references
|
|
2201
|
+
)
|
|
2202
|
+
if checker.verdict != EvalCheckerVerdictValue.APPROVED:
|
|
2203
|
+
return EvalCheckerApprovalValidationResult(
|
|
2204
|
+
valid=True,
|
|
2205
|
+
evidence_artifact_ids=evidence_artifact_ids,
|
|
2206
|
+
)
|
|
2207
|
+
|
|
2208
|
+
diagnostics: list[str] = []
|
|
2209
|
+
test_results = validated_artifacts.get("test_results")
|
|
2210
|
+
if test_results is not None and _test_results_failed(test_results):
|
|
2211
|
+
diagnostics.append("checker approval cannot rely on failed tests")
|
|
2212
|
+
|
|
2213
|
+
patch_summary = validated_artifacts.get("patch_summary")
|
|
2214
|
+
if patch_summary is not None and _failed_command_outcomes(patch_summary):
|
|
2215
|
+
diagnostics.append("checker approval cannot rely on failed static checks")
|
|
2216
|
+
|
|
2217
|
+
if diagnostics:
|
|
2218
|
+
return EvalCheckerApprovalValidationResult(
|
|
2219
|
+
valid=False,
|
|
2220
|
+
evidence_artifact_ids=evidence_artifact_ids,
|
|
2221
|
+
diagnostics=tuple(diagnostics),
|
|
2222
|
+
)
|
|
2223
|
+
|
|
2224
|
+
return EvalCheckerApprovalValidationResult(
|
|
2225
|
+
valid=True,
|
|
2226
|
+
evidence_artifact_ids=evidence_artifact_ids,
|
|
2227
|
+
)
|
|
2228
|
+
|
|
2229
|
+
|
|
2230
|
+
def validate_eval_closure(
|
|
2231
|
+
artifact_bundle: Mapping[Any, Any],
|
|
2232
|
+
fixture_snapshot: EvalFixtureWorkspaceSnapshot | Mapping[str, Any],
|
|
2233
|
+
capability_snapshot: Any,
|
|
2234
|
+
) -> EvalClosureValidationResult:
|
|
2235
|
+
"""Validate compact eval closure evidence without mutating workspace state."""
|
|
2236
|
+
validated_artifacts, artifact_diagnostics = _validate_present_closure_artifacts(
|
|
2237
|
+
artifact_bundle
|
|
2238
|
+
)
|
|
2239
|
+
if artifact_diagnostics:
|
|
2240
|
+
return _closure_result(
|
|
2241
|
+
EvalClosureOutcomeKind.INVALID_ARTIFACT_BOUNDARY,
|
|
2242
|
+
diagnostics=artifact_diagnostics,
|
|
2243
|
+
)
|
|
2244
|
+
|
|
2245
|
+
base_required_artifact_ids = _required_closure_artifact_ids(
|
|
2246
|
+
EvalClosureOutcomeKind.VALID_BLOCKED_OUTCOME
|
|
2247
|
+
)
|
|
2248
|
+
base_missing_artifact_ids = tuple(
|
|
2249
|
+
artifact_id
|
|
2250
|
+
for artifact_id in base_required_artifact_ids
|
|
2251
|
+
if artifact_id not in validated_artifacts
|
|
2252
|
+
)
|
|
2253
|
+
if base_missing_artifact_ids:
|
|
2254
|
+
return _closure_result(
|
|
2255
|
+
EvalClosureOutcomeKind.INVALID_ARTIFACT_BOUNDARY,
|
|
2256
|
+
missing_artifact_ids=base_missing_artifact_ids,
|
|
2257
|
+
diagnostics=(
|
|
2258
|
+
"closure artifact bundle is missing required terminal-path artifacts",
|
|
2259
|
+
),
|
|
2260
|
+
)
|
|
2261
|
+
|
|
2262
|
+
outcome_kind, terminal_result, candidate_disposition = _closure_outcome_kind(
|
|
2263
|
+
validated_artifacts
|
|
2264
|
+
)
|
|
2265
|
+
terminal_path_diagnostics = _terminal_path_diagnostics(
|
|
2266
|
+
validated_artifacts, outcome_kind
|
|
2267
|
+
)
|
|
2268
|
+
if terminal_path_diagnostics:
|
|
2269
|
+
return _closure_result(
|
|
2270
|
+
EvalClosureOutcomeKind.INVALID_ARTIFACT_BOUNDARY,
|
|
2271
|
+
diagnostics=terminal_path_diagnostics,
|
|
2272
|
+
)
|
|
2273
|
+
|
|
2274
|
+
gate_diagnostics = (
|
|
2275
|
+
_checker_gate_diagnostics(validated_artifacts, outcome_kind)
|
|
2276
|
+
+ _command_status_diagnostics(validated_artifacts, outcome_kind)
|
|
2277
|
+
+ _mutation_evidence_diagnostics(validated_artifacts, outcome_kind)
|
|
2278
|
+
+ _arbiter_acceptance_diagnostics(validated_artifacts, outcome_kind)
|
|
2279
|
+
+ _arbiter_stage_result_diagnostics(validated_artifacts, terminal_result)
|
|
2280
|
+
)
|
|
2281
|
+
if gate_diagnostics:
|
|
2282
|
+
return _closure_result(
|
|
2283
|
+
EvalClosureOutcomeKind.INVALID_ARTIFACT_BOUNDARY,
|
|
2284
|
+
diagnostics=gate_diagnostics,
|
|
2285
|
+
)
|
|
2286
|
+
|
|
2287
|
+
required_artifact_ids = _required_closure_artifact_ids(outcome_kind)
|
|
2288
|
+
missing_artifact_ids = tuple(
|
|
2289
|
+
artifact_id
|
|
2290
|
+
for artifact_id in required_artifact_ids
|
|
2291
|
+
if artifact_id not in validated_artifacts
|
|
2292
|
+
)
|
|
2293
|
+
if missing_artifact_ids:
|
|
2294
|
+
return _closure_result(
|
|
2295
|
+
EvalClosureOutcomeKind.INVALID_ARTIFACT_BOUNDARY,
|
|
2296
|
+
missing_artifact_ids=missing_artifact_ids,
|
|
2297
|
+
diagnostics=(
|
|
2298
|
+
"closure artifact bundle is missing required terminal-path artifacts",
|
|
2299
|
+
),
|
|
2300
|
+
)
|
|
2301
|
+
|
|
2302
|
+
relaxed_blocked_artifact_ids = (
|
|
2303
|
+
"workspace_diff",
|
|
2304
|
+
"patch_summary",
|
|
2305
|
+
"test_results",
|
|
2306
|
+
"checker_verdict",
|
|
2307
|
+
)
|
|
2308
|
+
using_relaxed_blocked_artifacts = (
|
|
2309
|
+
outcome_kind == EvalClosureOutcomeKind.VALID_BLOCKED_OUTCOME
|
|
2310
|
+
and any(
|
|
2311
|
+
artifact_id not in validated_artifacts
|
|
2312
|
+
for artifact_id in relaxed_blocked_artifact_ids
|
|
2313
|
+
)
|
|
2314
|
+
)
|
|
2315
|
+
if using_relaxed_blocked_artifacts:
|
|
2316
|
+
try:
|
|
2317
|
+
mutation_paths = _fixture_snapshot_mutation_paths(fixture_snapshot)
|
|
2318
|
+
except (TypeError, ValueError) as exc:
|
|
2319
|
+
return _closure_result(
|
|
2320
|
+
EvalClosureOutcomeKind.INVALID_FIXTURE_BOUNDARY,
|
|
2321
|
+
diagnostics=(str(exc),),
|
|
2322
|
+
)
|
|
2323
|
+
if mutation_paths:
|
|
2324
|
+
return _closure_result(
|
|
2325
|
+
EvalClosureOutcomeKind.INVALID_ARTIFACT_BOUNDARY,
|
|
2326
|
+
diagnostics=(
|
|
2327
|
+
"shortened blocked terminal path requires unmodified fixture snapshot",
|
|
2328
|
+
),
|
|
2329
|
+
)
|
|
2330
|
+
|
|
2331
|
+
capability_diagnostics = _capability_snapshot_diagnostics(capability_snapshot)
|
|
2332
|
+
if capability_diagnostics:
|
|
2333
|
+
return _closure_result(
|
|
2334
|
+
EvalClosureOutcomeKind.INVALID_CAPABILITY_BOUNDARY,
|
|
2335
|
+
diagnostics=capability_diagnostics,
|
|
2336
|
+
)
|
|
2337
|
+
|
|
2338
|
+
try:
|
|
2339
|
+
fixture_diagnostics = _fixture_snapshot_diagnostics(
|
|
2340
|
+
fixture_snapshot, artifact_bundle
|
|
2341
|
+
)
|
|
2342
|
+
except (TypeError, ValueError) as exc:
|
|
2343
|
+
fixture_diagnostics = (str(exc),)
|
|
2344
|
+
if fixture_diagnostics:
|
|
2345
|
+
return _closure_result(
|
|
2346
|
+
EvalClosureOutcomeKind.INVALID_FIXTURE_BOUNDARY,
|
|
2347
|
+
diagnostics=fixture_diagnostics,
|
|
2348
|
+
)
|
|
2349
|
+
|
|
2350
|
+
context_diagnostics = _context_boundary_diagnostics(validated_artifacts)
|
|
2351
|
+
if context_diagnostics:
|
|
2352
|
+
return _closure_result(
|
|
2353
|
+
EvalClosureOutcomeKind.INVALID_CONTEXT_BOUNDARY,
|
|
2354
|
+
diagnostics=context_diagnostics,
|
|
2355
|
+
)
|
|
2356
|
+
|
|
2357
|
+
evidence_artifact_ids = _closure_evidence_artifact_ids(validated_artifacts)
|
|
2358
|
+
if not evidence_artifact_ids or any(
|
|
2359
|
+
artifact_id not in validated_artifacts for artifact_id in evidence_artifact_ids
|
|
2360
|
+
):
|
|
2361
|
+
return _closure_result(
|
|
2362
|
+
EvalClosureOutcomeKind.INVALID_ARTIFACT_BOUNDARY,
|
|
2363
|
+
diagnostics=(
|
|
2364
|
+
"arbiter closure evidence references must resolve inside artifact bundle",
|
|
2365
|
+
),
|
|
2366
|
+
)
|
|
2367
|
+
|
|
2368
|
+
return _closure_result(
|
|
2369
|
+
outcome_kind,
|
|
2370
|
+
terminal_result=terminal_result,
|
|
2371
|
+
candidate_disposition=candidate_disposition,
|
|
2372
|
+
evidence_artifact_ids=evidence_artifact_ids,
|
|
2373
|
+
)
|
|
2374
|
+
|
|
2375
|
+
|
|
2376
|
+
__all__ = [
|
|
2377
|
+
"AUTHORITATIVE_COMPACT_EVAL_WORKFLOW_NAMES",
|
|
2378
|
+
"EVAL_BOUNDARY_ARTIFACT_MODULE_DECISION",
|
|
2379
|
+
"EVAL_BOUNDARY_ARTIFACT_MODULE_REQUIRED",
|
|
2380
|
+
"EVAL_BOUNDARY_MODULE_NAME",
|
|
2381
|
+
"EVAL_BUILDER_DEFAULT_WRITE_ROOTS",
|
|
2382
|
+
"EVAL_CHECKER_IGNORED_SCRATCH_ROOTS",
|
|
2383
|
+
"EVAL_CONTEXT_DEFAULT_ALLOWED_PATHS",
|
|
2384
|
+
"EVAL_CONTEXT_DEFAULT_REDACTION_CATEGORIES",
|
|
2385
|
+
"EVAL_CONTEXT_DEFAULT_REQUIRED_ARTIFACT_IDS",
|
|
2386
|
+
"EVAL_CONTEXT_FINGERPRINT_KIND",
|
|
2387
|
+
"EVAL_DENIED_CAPABILITY_IDS",
|
|
2388
|
+
"EVAL_FIXTURE_IGNORED_GENERATED_ROOTS",
|
|
2389
|
+
"EVAL_FIXTURE_IGNORED_GENERATED_SUFFIXES",
|
|
2390
|
+
"EVAL_STAGE_RESOURCE_CEILING_DEFAULTS",
|
|
2391
|
+
"EVAL_TRIAL_RESOURCE_CEILING_DEFAULTS",
|
|
2392
|
+
"EVAL_WORKSPACE_ISOLATION_CONTRACT",
|
|
2393
|
+
"EvalBoundaryBaseline",
|
|
2394
|
+
"EvalBoundaryStageArtifacts",
|
|
2395
|
+
"EvalCapabilityEnvelope",
|
|
2396
|
+
"EvalCapabilityId",
|
|
2397
|
+
"EvalCapabilityValidationResult",
|
|
2398
|
+
"EvalCommandAdmissionResult",
|
|
2399
|
+
"EvalCommandDescriptor",
|
|
2400
|
+
"EvalCommandEnvironmentPolicy",
|
|
2401
|
+
"EvalCheckerApprovalValidationResult",
|
|
2402
|
+
"EvalContextArtifactSummary",
|
|
2403
|
+
"EvalContextRedaction",
|
|
2404
|
+
"EvalContextSnapshot",
|
|
2405
|
+
"EvalContextTier",
|
|
2406
|
+
"EvalClosureOutcomeKind",
|
|
2407
|
+
"EvalClosureValidationResult",
|
|
2408
|
+
"EvalFixtureFile",
|
|
2409
|
+
"EvalFixtureManifest",
|
|
2410
|
+
"EvalFixtureWorkspacePolicy",
|
|
2411
|
+
"EvalFixtureWorkspaceSnapshot",
|
|
2412
|
+
"EvalPathPolicyViolation",
|
|
2413
|
+
"EvalResourceCeiling",
|
|
2414
|
+
"EvalStageContextPolicy",
|
|
2415
|
+
"build_eval_context_snapshot",
|
|
2416
|
+
"calculate_eval_context_fingerprint",
|
|
2417
|
+
"compact_eval_boundary_baseline",
|
|
2418
|
+
"compact_eval_boundary_baseline_snapshot",
|
|
2419
|
+
"default_eval_capability_envelopes",
|
|
2420
|
+
"default_eval_stage_context_policies",
|
|
2421
|
+
"default_eval_stage_context_policy",
|
|
2422
|
+
"default_eval_stage_resource_ceiling",
|
|
2423
|
+
"default_eval_trial_resource_ceiling",
|
|
2424
|
+
"eval_fixture_file_hash",
|
|
2425
|
+
"eval_fixture_file_record",
|
|
2426
|
+
"eval_fixture_manifest_from_paths",
|
|
2427
|
+
"eval_fixture_manifest_sha256",
|
|
2428
|
+
"eval_fixture_workspace_snapshot",
|
|
2429
|
+
"eval_stage_capability_envelope",
|
|
2430
|
+
"validate_eval_checker_approval",
|
|
2431
|
+
"validate_eval_fixture_path",
|
|
2432
|
+
"validate_eval_closure",
|
|
2433
|
+
"validate_eval_stage_capability",
|
|
2434
|
+
"validate_eval_stage_command",
|
|
2435
|
+
]
|