millforge 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. millforge/__init__.py +1174 -0
  2. millforge/_forge/LICENSE +21 -0
  3. millforge/_forge/PROVENANCE.json +295 -0
  4. millforge/_forge/UPDATE_POLICY.md +24 -0
  5. millforge/_forge/__init__.py +14 -0
  6. millforge/_forge/adapter.py +2232 -0
  7. millforge/_forge/base_runner.py +121 -0
  8. millforge/_forge/clients/__init__.py +10 -0
  9. millforge/_forge/clients/base.py +200 -0
  10. millforge/_forge/context/__init__.py +23 -0
  11. millforge/_forge/context/manager.py +178 -0
  12. millforge/_forge/context/strategies.py +335 -0
  13. millforge/_forge/core/__init__.py +16 -0
  14. millforge/_forge/core/inference.py +433 -0
  15. millforge/_forge/core/messages.py +119 -0
  16. millforge/_forge/core/runner.py +479 -0
  17. millforge/_forge/core/steps.py +108 -0
  18. millforge/_forge/core/workflow.py +400 -0
  19. millforge/_forge/errors.py +222 -0
  20. millforge/_forge/guardrails/__init__.py +21 -0
  21. millforge/_forge/guardrails/error_tracker.py +71 -0
  22. millforge/_forge/guardrails/guardrails.py +194 -0
  23. millforge/_forge/guardrails/nudge.py +47 -0
  24. millforge/_forge/guardrails/response_validator.py +119 -0
  25. millforge/_forge/guardrails/step_enforcer.py +183 -0
  26. millforge/_forge/prompts/__init__.py +16 -0
  27. millforge/_forge/prompts/nudges.py +95 -0
  28. millforge/_forge/prompts/templates.py +285 -0
  29. millforge/_version.py +3 -0
  30. millforge/artifacts.py +570 -0
  31. millforge/base/__init__.py +97 -0
  32. millforge/base/composition.py +402 -0
  33. millforge/base/context.py +285 -0
  34. millforge/base/harness.py +138 -0
  35. millforge/base/identity.py +465 -0
  36. millforge/base/options.py +34 -0
  37. millforge/base/platform.py +17 -0
  38. millforge/base/prompt.py +317 -0
  39. millforge/base/runner.py +546 -0
  40. millforge/compiled_plan.py +970 -0
  41. millforge/compiler/__init__.py +231 -0
  42. millforge/compiler/artifact_validation.py +257 -0
  43. millforge/compiler/canonicalization.py +169 -0
  44. millforge/compiler/capabilities.py +66 -0
  45. millforge/compiler/catalogs.py +500 -0
  46. millforge/compiler/diagnostics.py +491 -0
  47. millforge/compiler/graph.py +678 -0
  48. millforge/compiler/lowering.py +198 -0
  49. millforge/compiler/output.py +692 -0
  50. millforge/compiler/parsing.py +1424 -0
  51. millforge/compiler/requests.py +1180 -0
  52. millforge/compiler/schema_validation.py +272 -0
  53. millforge/compiler/semantic.py +490 -0
  54. millforge/compiler/service.py +448 -0
  55. millforge/compiler/source.py +375 -0
  56. millforge/compiler/validators.py +184 -0
  57. millforge/connectors/__init__.py +95 -0
  58. millforge/connectors/admission.py +801 -0
  59. millforge/connectors/broker.py +202 -0
  60. millforge/connectors/contracts.py +1159 -0
  61. millforge/connectors/diagnostics.py +189 -0
  62. millforge/connectors/fake.py +66 -0
  63. millforge/connectors/runtime.py +236 -0
  64. millforge/contracts.py +2860 -0
  65. millforge/custom_tools/__init__.py +67 -0
  66. millforge/custom_tools/compiler.py +724 -0
  67. millforge/custom_tools/contracts.py +1093 -0
  68. millforge/custom_tools/diagnostics.py +205 -0
  69. millforge/eval_artifacts.py +952 -0
  70. millforge/eval_boundary.py +2435 -0
  71. millforge/eval_fixtures/__init__.py +1 -0
  72. millforge/eval_fixtures/default_pack/__init__.py +1 -0
  73. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.bug_diagnosis.traceback.v1.json +52 -0
  74. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.direct_edit.import_sort.v1.json +52 -0
  75. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.evidence_discipline.no_source_change.v1.json +51 -0
  76. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.false_closure.visible_green.v1.json +52 -0
  77. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.multi_file.api_contract.v1.json +54 -0
  78. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.recovery.malformed_artifact.v1.json +54 -0
  79. millforge/eval_fixtures/default_pack/manifest.json +12 -0
  80. millforge/eval_modes.py +1282 -0
  81. millforge/eval_presets.py +1398 -0
  82. millforge/eval_reports.py +2517 -0
  83. millforge/eval_suite.py +2429 -0
  84. millforge/eval_trials.py +2632 -0
  85. millforge/eval_workflow.py +794 -0
  86. millforge/exceptions.py +122 -0
  87. millforge/model_backend.py +2098 -0
  88. millforge/protocols.py +340 -0
  89. millforge/py.typed +0 -0
  90. millforge/runtime.py +1791 -0
  91. millforge/testing/__init__.py +1089 -0
  92. millforge/tools/__init__.py +83 -0
  93. millforge/tools/builtin_runtime.py +1339 -0
  94. millforge/tools/builtins.py +773 -0
  95. millforge/tools/execution.py +1545 -0
  96. millforge/tools/path_policy.py +155 -0
  97. millforge/tools/pi_compat/PI_LICENSE +21 -0
  98. millforge/tools/pi_compat/PROVENANCE.json +55 -0
  99. millforge/tools/pi_compat/UPDATE_POLICY.md +36 -0
  100. millforge/tools/pi_compat/__init__.py +34 -0
  101. millforge/tools/pi_compat/contracts.py +49 -0
  102. millforge/tools/pi_compat/editing.py +390 -0
  103. millforge/tools/pi_compat/mutations.py +57 -0
  104. millforge/tools/pi_compat/operations.py +401 -0
  105. millforge/tools/pi_compat/paths.py +155 -0
  106. millforge/tools/pi_compat/process.py +1375 -0
  107. millforge/tools/pi_compat/search.py +738 -0
  108. millforge/tools/pi_compat/truncation.py +267 -0
  109. millforge/tools/pi_compat_catalog.py +396 -0
  110. millforge/tools/pi_compat_runtime.py +460 -0
  111. millforge/tools/registry.py +553 -0
  112. millforge/tools/results.py +533 -0
  113. millforge-0.1.0.dist-info/METADATA +844 -0
  114. millforge-0.1.0.dist-info/RECORD +116 -0
  115. millforge-0.1.0.dist-info/WHEEL +4 -0
  116. millforge-0.1.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,2435 @@
1
+ """Public 06B baseline boundary anchored to the compact eval workflow."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import json
7
+ from collections.abc import Mapping
8
+ from enum import Enum
9
+ from pathlib import Path
10
+ from pathlib import PurePosixPath
11
+ import re
12
+ from types import MappingProxyType
13
+ from typing import Any, cast
14
+
15
+ from pydantic import (
16
+ BaseModel,
17
+ ConfigDict,
18
+ Field,
19
+ StrictBool,
20
+ StrictInt,
21
+ StrictStr,
22
+ field_serializer,
23
+ field_validator,
24
+ model_validator,
25
+ )
26
+
27
+ from millforge.eval_workflow import (
28
+ EvalCandidateDisposition,
29
+ EvalStageId,
30
+ EvalTerminalResult,
31
+ EvalWorkflowOutcomeKind,
32
+ compact_eval_workflow_snapshot,
33
+ default_compact_eval_workflow_graph,
34
+ )
35
+
36
+ AUTHORITATIVE_COMPACT_EVAL_WORKFLOW_NAMES: tuple[str, ...] = (
37
+ "EvalStageId",
38
+ "EvalTerminalResult",
39
+ "EvalWorkflowOutcomeKind",
40
+ "EvalCandidateDisposition",
41
+ "default_compact_eval_workflow_graph",
42
+ "compact_eval_workflow_snapshot",
43
+ )
44
+ EVAL_BOUNDARY_MODULE_NAME = "millforge.eval_boundary"
45
+ EVAL_BOUNDARY_ARTIFACT_MODULE_REQUIRED = True
46
+ EVAL_BOUNDARY_ARTIFACT_MODULE_DECISION = (
47
+ "06B artifact schemas are defined in millforge.eval_artifacts so "
48
+ "millforge.eval_boundary remains focused on capability and fixture policy."
49
+ )
50
+ EVAL_DENIED_CAPABILITY_IDS: tuple[str, ...] = (
51
+ "network.access",
52
+ "package.install",
53
+ "git.mutate",
54
+ "runtime.control",
55
+ )
56
+ EVAL_BUILDER_DEFAULT_WRITE_ROOTS: tuple[str, ...] = (
57
+ "src",
58
+ "tests",
59
+ "README.md",
60
+ "ROADMAP.md",
61
+ )
62
+ EVAL_CHECKER_IGNORED_SCRATCH_ROOTS: tuple[str, ...] = (
63
+ ".eval-scratch",
64
+ ".pytest_cache",
65
+ )
66
+ EVAL_FIXTURE_IGNORED_GENERATED_ROOTS: tuple[str, ...] = (
67
+ ".eval-scratch",
68
+ ".mypy_cache",
69
+ ".pytest_cache",
70
+ ".ruff_cache",
71
+ "__pycache__",
72
+ )
73
+ EVAL_FIXTURE_IGNORED_GENERATED_SUFFIXES: tuple[str, ...] = (
74
+ ".coverage",
75
+ ".coverage.json",
76
+ ".log",
77
+ ".pyc",
78
+ ".pyo",
79
+ )
80
+ EVAL_FIXTURE_MANIFEST_SCHEMA_VERSION = 1
81
+ EVAL_FIXTURE_DEFAULT_REVISION = "fixture-revision-1"
82
+ EVAL_FIXTURE_FILE_ROLES: tuple[str, ...] = (
83
+ "source",
84
+ "test",
85
+ "documentation",
86
+ "configuration",
87
+ "data",
88
+ )
89
+ EVAL_WORKSPACE_ISOLATION_CONTRACT = "fresh_copy"
90
+ EVAL_CONTEXT_FINGERPRINT_KIND = "eval_context_snapshot_sha256_v1"
91
+ EVAL_CONTEXT_DEFAULT_REDACTION_CATEGORIES: tuple[str, ...] = (
92
+ "scorer_only_material",
93
+ "secrets",
94
+ "daemon_state",
95
+ "host_paths",
96
+ "runtime_private_state",
97
+ "ideas_private_state",
98
+ "reference_private_state",
99
+ "git_history",
100
+ "private_conversations",
101
+ "unrelated_repository_outlines",
102
+ )
103
+ EVAL_CONTEXT_DEFAULT_REQUIRED_ARTIFACT_IDS: Mapping[EvalStageId, tuple[str, ...]] = (
104
+ MappingProxyType(
105
+ {
106
+ EvalStageId.PLANNER: ("task", "fixture_manifest", "acceptance_checks"),
107
+ EvalStageId.BUILDER: (
108
+ "task",
109
+ "fixture_manifest",
110
+ "acceptance_checks",
111
+ "plan",
112
+ ),
113
+ EvalStageId.CHECKER: (
114
+ "fixture_manifest",
115
+ "acceptance_checks",
116
+ "workspace_diff",
117
+ "patch_summary",
118
+ "test_results",
119
+ ),
120
+ EvalStageId.ARBITER: (
121
+ "acceptance_checks",
122
+ "workspace_diff",
123
+ "patch_summary",
124
+ "test_results",
125
+ "checker_verdict",
126
+ ),
127
+ }
128
+ )
129
+ )
130
+ EVAL_CONTEXT_DEFAULT_ALLOWED_PATHS: Mapping[EvalStageId, tuple[str, ...]] = (
131
+ MappingProxyType(
132
+ {
133
+ EvalStageId.PLANNER: (),
134
+ EvalStageId.BUILDER: ("src", "tests", "README.md", "ROADMAP.md"),
135
+ EvalStageId.CHECKER: ("src", "tests", ".eval-scratch"),
136
+ EvalStageId.ARBITER: ("src", "tests"),
137
+ }
138
+ )
139
+ )
140
+ EVAL_STAGE_RESOURCE_CEILING_DEFAULTS: Mapping[EvalStageId, Mapping[str, int]] = (
141
+ MappingProxyType(
142
+ {
143
+ EvalStageId.PLANNER: MappingProxyType(
144
+ {
145
+ "prompt_tokens": 16_000,
146
+ "completion_tokens": 4_000,
147
+ "model_calls": 1,
148
+ "wall_clock_seconds": 600,
149
+ "shell_commands": 1,
150
+ "shell_command_seconds": 1,
151
+ "writable_bytes": 262_144,
152
+ "artifact_bytes": 262_144,
153
+ }
154
+ ),
155
+ EvalStageId.BUILDER: MappingProxyType(
156
+ {
157
+ "prompt_tokens": 32_000,
158
+ "completion_tokens": 8_000,
159
+ "model_calls": 2,
160
+ "wall_clock_seconds": 1_800,
161
+ "shell_commands": 24,
162
+ "shell_command_seconds": 900,
163
+ "writable_bytes": 2_097_152,
164
+ "artifact_bytes": 524_288,
165
+ }
166
+ ),
167
+ EvalStageId.CHECKER: MappingProxyType(
168
+ {
169
+ "prompt_tokens": 16_000,
170
+ "completion_tokens": 4_000,
171
+ "model_calls": 2,
172
+ "wall_clock_seconds": 900,
173
+ "shell_commands": 12,
174
+ "shell_command_seconds": 600,
175
+ "writable_bytes": 131_072,
176
+ "artifact_bytes": 524_288,
177
+ }
178
+ ),
179
+ EvalStageId.ARBITER: MappingProxyType(
180
+ {
181
+ "prompt_tokens": 16_000,
182
+ "completion_tokens": 4_000,
183
+ "model_calls": 1,
184
+ "wall_clock_seconds": 600,
185
+ "shell_commands": 1,
186
+ "shell_command_seconds": 1,
187
+ "writable_bytes": 131_072,
188
+ "artifact_bytes": 262_144,
189
+ }
190
+ ),
191
+ }
192
+ )
193
+ )
194
+ EVAL_TRIAL_RESOURCE_CEILING_DEFAULTS: Mapping[str, int] = MappingProxyType(
195
+ {
196
+ "prompt_tokens": 80_000,
197
+ "completion_tokens": 20_000,
198
+ "model_calls": 6,
199
+ "wall_clock_seconds": 3_900,
200
+ "shell_commands": 36,
201
+ "shell_command_seconds": 1_500,
202
+ "writable_bytes": 2_621_440,
203
+ "artifact_bytes": 1_572_864,
204
+ }
205
+ )
206
+ _WINDOWS_DRIVE_PREFIX = re.compile(r"^[A-Za-z]:")
207
+ _SECRET_ENV_MARKERS: tuple[str, ...] = (
208
+ "API_KEY",
209
+ "AUTH",
210
+ "CREDENTIAL",
211
+ "PASSWORD",
212
+ "SECRET",
213
+ "TOKEN",
214
+ )
215
+ _HIDDEN_CHECK_DENIED_TOKENS: tuple[str, ...] = (
216
+ "definition",
217
+ "expected",
218
+ "fixture",
219
+ "output",
220
+ "path",
221
+ "result",
222
+ "rubric",
223
+ "score",
224
+ "secret",
225
+ )
226
+ _PACKAGE_MANAGER_COMMANDS: frozenset[str] = frozenset(
227
+ {
228
+ "cargo",
229
+ "gem",
230
+ "npm",
231
+ "pip",
232
+ "pip3",
233
+ "pnpm",
234
+ "poetry",
235
+ "uv",
236
+ "yarn",
237
+ }
238
+ )
239
+ _NETWORK_COMMANDS: frozenset[str] = frozenset(
240
+ {
241
+ "curl",
242
+ "ftp",
243
+ "nc",
244
+ "netcat",
245
+ "scp",
246
+ "sftp",
247
+ "ssh",
248
+ "telnet",
249
+ "wget",
250
+ }
251
+ )
252
+ _RUNTIME_CONTROL_COMMANDS: frozenset[str] = frozenset(
253
+ {
254
+ "docker",
255
+ "docker-compose",
256
+ "millrace",
257
+ "podman",
258
+ "service",
259
+ "systemctl",
260
+ }
261
+ )
262
+ _SHELL_COMMAND_WRAPPERS: frozenset[str] = frozenset(
263
+ {
264
+ "bash",
265
+ "dash",
266
+ "fish",
267
+ "ksh",
268
+ "sh",
269
+ "zsh",
270
+ }
271
+ )
272
+ _SHELL_INTERPOLATION_TOKENS: tuple[str, ...] = (
273
+ "$",
274
+ "`",
275
+ "$(",
276
+ "${",
277
+ "{{",
278
+ "}}",
279
+ ";",
280
+ "&&",
281
+ "||",
282
+ "|",
283
+ "<",
284
+ ">",
285
+ )
286
+ _CONTEXT_LEAK_TOKENS: tuple[str, ...] = (
287
+ "F:\\",
288
+ "/mnt/f",
289
+ "millrace-agents",
290
+ "ideas/",
291
+ "ref-forge/",
292
+ "/home/",
293
+ "\\Users\\",
294
+ "API_KEY",
295
+ "DAEMON_STATE",
296
+ "daemon state",
297
+ "git history",
298
+ "git_history",
299
+ "hidden check",
300
+ "hidden checks",
301
+ "hidden_check",
302
+ "hidden_checks",
303
+ "hidden score",
304
+ "hidden_score",
305
+ "hidden_scores",
306
+ "scoring rubric",
307
+ "scoring_rubric",
308
+ "expected output",
309
+ "expected_output",
310
+ "private conversation",
311
+ "private conversations",
312
+ "private_conversation",
313
+ "private_conversations",
314
+ "private runtime",
315
+ "unrelated repository outline",
316
+ "unrelated repository outlines",
317
+ "unrelated_repository_outline",
318
+ "unrelated_repository_outlines",
319
+ )
320
+ _CONTEXT_WINDOWS_ABSOLUTE_PATH = re.compile(
321
+ r"(?:^|[\s\"'`([{<])(?:[A-Za-z]:[\\/]|\\\\)[^\s\"'`)\]}>,]*"
322
+ )
323
+ _CONTEXT_POSIX_ABSOLUTE_PATH = re.compile(r"(?:^|[\s\"'`([{<])/(?!/)[^\s\"'`)\]}>,]*")
324
+ _CONTEXT_USER_HOME_PATH = re.compile(r"(?:^|[\s\"'`([{<])~(?:/|\\)[^\s\"'`)\]}>,]*")
325
+
326
+
327
+ def _command_executable(argv: tuple[str, ...]) -> str:
328
+ return argv[0].split("/")[-1].lower()
329
+
330
+
331
+ def _is_python_executable(executable: str) -> bool:
332
+ return executable in {"py", "python", "python3"} or executable.startswith(
333
+ "python3."
334
+ )
335
+
336
+
337
+ def _python_module_name(argv: tuple[str, ...], executable: str) -> str | None:
338
+ if len(argv) < 3 or not _is_python_executable(executable) or argv[1] != "-m":
339
+ return None
340
+ return argv[2].split(".", maxsplit=1)[0].lower()
341
+
342
+
343
+ def _disallowed_argv_message(argv: tuple[str, ...]) -> str | None:
344
+ executable = _command_executable(argv)
345
+ if executable in _PACKAGE_MANAGER_COMMANDS:
346
+ return "package manager commands are not admitted"
347
+ if executable in _NETWORK_COMMANDS:
348
+ return "network commands are not admitted"
349
+ if executable in _RUNTIME_CONTROL_COMMANDS:
350
+ return "runtime control commands are not admitted"
351
+ if executable == "git":
352
+ return "git commands are not admitted in eval command descriptors"
353
+ if executable in _SHELL_COMMAND_WRAPPERS:
354
+ return "shell wrapper commands are not admitted"
355
+ module_name = _python_module_name(argv, executable)
356
+ if module_name in _PACKAGE_MANAGER_COMMANDS:
357
+ return "package manager module wrappers are not admitted"
358
+ return None
359
+
360
+
361
+ class EvalCapabilityId(str, Enum):
362
+ """Closed capability IDs admitted by compact eval policy."""
363
+
364
+ ARTIFACT_READ = "artifact.read"
365
+ ARTIFACT_WRITE = "artifact.write"
366
+ EVIDENCE_EMIT = "evidence.emit"
367
+ RUNNER_INVOKE = "runner.invoke"
368
+ WORKSPACE_READ = "workspace.read"
369
+ WORKSPACE_WRITE = "workspace.write"
370
+ SHELL_RUN = "shell.run"
371
+ NETWORK_ACCESS = "network.access"
372
+ PACKAGE_INSTALL = "package.install"
373
+ GIT_MUTATE = "git.mutate"
374
+ RUNTIME_CONTROL = "runtime.control"
375
+
376
+
377
+ class EvalCapabilityEnvelope(BaseModel):
378
+ """Immutable capability envelope for one compact eval stage."""
379
+
380
+ model_config = ConfigDict(extra="forbid", frozen=True)
381
+
382
+ stage_id: EvalStageId
383
+ capability_ids: tuple[EvalCapabilityId, ...]
384
+
385
+ @field_validator("capability_ids")
386
+ @classmethod
387
+ def _capability_ids_valid(
388
+ cls, value: tuple[EvalCapabilityId, ...]
389
+ ) -> tuple[EvalCapabilityId, ...]:
390
+ if len(set(value)) != len(value):
391
+ raise ValueError("capability_ids values must be unique")
392
+ return value
393
+
394
+
395
+ class EvalCapabilityValidationResult(BaseModel):
396
+ """Structured result for one stage capability admission decision."""
397
+
398
+ model_config = ConfigDict(extra="forbid", frozen=True)
399
+
400
+ stage_id: EvalStageId | StrictStr
401
+ capability_id: EvalCapabilityId | StrictStr
402
+ allowed: StrictBool
403
+ rule_id: StrictStr
404
+ diagnostic_code: StrictStr | None = None
405
+ diagnostic_summary: StrictStr | None = None
406
+
407
+ @model_validator(mode="after")
408
+ def _diagnostic_shape_valid(self) -> EvalCapabilityValidationResult:
409
+ if self.allowed:
410
+ if self.diagnostic_code is not None or self.diagnostic_summary is not None:
411
+ raise ValueError(
412
+ "allowed capability results must not include diagnostics"
413
+ )
414
+ elif self.diagnostic_code is None or self.diagnostic_summary is None:
415
+ raise ValueError("denied capability results must include diagnostics")
416
+ return self
417
+
418
+
419
+ class EvalCommandEnvironmentPolicy(BaseModel):
420
+ """Closed environment policy for deterministic eval commands."""
421
+
422
+ model_config = ConfigDict(extra="forbid", frozen=True)
423
+
424
+ inherit_environment: StrictBool = False
425
+ variables: Mapping[StrictStr, StrictStr] = Field(default_factory=dict)
426
+
427
+ @model_validator(mode="after")
428
+ def _environment_policy_valid(self) -> EvalCommandEnvironmentPolicy:
429
+ if self.inherit_environment:
430
+ raise ValueError("eval commands may not inherit ambient environment")
431
+ ordered: dict[str, str] = {}
432
+ for name, value in sorted(self.variables.items()):
433
+ if not name or not name.replace("_", "").isalnum() or name[0].isdigit():
434
+ raise ValueError(
435
+ "environment variable names must be stable identifiers"
436
+ )
437
+ if name == "*" or any(
438
+ marker in name.upper() for marker in _SECRET_ENV_MARKERS
439
+ ):
440
+ raise ValueError("environment policy may not expose secrets")
441
+ if "\x00" in value:
442
+ raise ValueError(
443
+ "environment variable values must not contain NUL bytes"
444
+ )
445
+ ordered[name] = value
446
+ object.__setattr__(self, "variables", MappingProxyType(ordered))
447
+ return self
448
+
449
+
450
+ class EvalCommandDescriptor(BaseModel):
451
+ """Deterministic descriptor for admitted Builder and Checker commands."""
452
+
453
+ model_config = ConfigDict(extra="forbid", frozen=True)
454
+
455
+ command_id: StrictStr
456
+ argv: tuple[StrictStr, ...]
457
+ relative_working_directory: StrictStr = "."
458
+ admitted_read_roots: tuple[StrictStr, ...]
459
+ admitted_write_roots: tuple[StrictStr, ...] = Field(default_factory=tuple)
460
+ timeout_seconds: StrictInt = Field(gt=0, le=600)
461
+ environment_policy: EvalCommandEnvironmentPolicy = Field(
462
+ default_factory=EvalCommandEnvironmentPolicy
463
+ )
464
+ expected_output_artifact_ids: tuple[StrictStr, ...]
465
+
466
+ @field_validator("command_id")
467
+ @classmethod
468
+ def _command_id_valid(cls, value: str) -> str:
469
+ if not value.strip() or any(
470
+ token in value for token in _SHELL_INTERPOLATION_TOKENS
471
+ ):
472
+ raise ValueError("command_id must be a stable non-interpolated identifier")
473
+ return value
474
+
475
+ @field_validator("argv")
476
+ @classmethod
477
+ def _argv_valid(cls, value: tuple[str, ...]) -> tuple[str, ...]:
478
+ if not value:
479
+ raise ValueError("argv must not be empty")
480
+ if len(value) == 1 and any(character.isspace() for character in value[0]):
481
+ raise ValueError("argv entries must not require shell interpolation")
482
+ if disallowed_message := _disallowed_argv_message(value):
483
+ raise ValueError(disallowed_message)
484
+ for argument in value:
485
+ if not argument or "\x00" in argument:
486
+ raise ValueError("argv entries must be non-empty strings")
487
+ if any(token in argument for token in _SHELL_INTERPOLATION_TOKENS):
488
+ raise ValueError("argv entries must not require shell interpolation")
489
+ return value
490
+
491
+ @field_validator("admitted_read_roots", "admitted_write_roots")
492
+ @classmethod
493
+ def _roots_valid(cls, value: tuple[str, ...]) -> tuple[str, ...]:
494
+ if len(set(value)) != len(value):
495
+ raise ValueError("admitted roots must be unique")
496
+ for root in value:
497
+ _validate_relative_eval_path(root, allow_dot=False)
498
+ return value
499
+
500
+ @field_validator("expected_output_artifact_ids")
501
+ @classmethod
502
+ def _artifact_ids_valid(cls, value: tuple[str, ...]) -> tuple[str, ...]:
503
+ if not value:
504
+ raise ValueError("expected_output_artifact_ids must not be empty")
505
+ if len(set(value)) != len(value):
506
+ raise ValueError("expected_output_artifact_ids values must be unique")
507
+ for artifact_id in value:
508
+ if not artifact_id.strip() or "/" in artifact_id or "\\" in artifact_id:
509
+ raise ValueError("expected output artifact ids must be stable IDs")
510
+ return value
511
+
512
+ @model_validator(mode="after")
513
+ def _descriptor_shape_valid(self) -> EvalCommandDescriptor:
514
+ _validate_relative_eval_path(self.relative_working_directory, allow_dot=True)
515
+ if not self.admitted_read_roots:
516
+ raise ValueError("admitted_read_roots must not be empty")
517
+ if self.relative_working_directory != "." and not any(
518
+ _path_is_within_root(self.relative_working_directory, root)
519
+ for root in self.admitted_read_roots + self.admitted_write_roots
520
+ ):
521
+ raise ValueError("relative_working_directory must be inside admitted roots")
522
+ return self
523
+
524
+
525
+ class EvalCommandAdmissionResult(BaseModel):
526
+ """Structured command admission result for a compact eval stage."""
527
+
528
+ model_config = ConfigDict(extra="forbid", frozen=True)
529
+
530
+ stage_id: EvalStageId | StrictStr
531
+ command_id: StrictStr
532
+ allowed: StrictBool
533
+ rule_id: StrictStr
534
+ diagnostic_code: StrictStr | None = None
535
+ diagnostic_summary: StrictStr | None = None
536
+
537
+ @model_validator(mode="after")
538
+ def _diagnostic_shape_valid(self) -> EvalCommandAdmissionResult:
539
+ if self.allowed:
540
+ if self.diagnostic_code is not None or self.diagnostic_summary is not None:
541
+ raise ValueError("allowed command results must not include diagnostics")
542
+ elif self.diagnostic_code is None or self.diagnostic_summary is None:
543
+ raise ValueError("denied command results must include diagnostics")
544
+ return self
545
+
546
+
547
+ class EvalBoundaryStageArtifacts(BaseModel):
548
+ """Immutable logical artifact IDs for one compact eval stage."""
549
+
550
+ model_config = ConfigDict(extra="forbid", frozen=True)
551
+
552
+ stage_id: EvalStageId
553
+ input_artifact_ids: tuple[StrictStr, ...]
554
+ output_artifact_ids: tuple[StrictStr, ...]
555
+
556
+
557
+ class EvalBoundaryBaseline(BaseModel):
558
+ """Immutable 06B boundary baseline derived from the accepted 06A graph."""
559
+
560
+ model_config = ConfigDict(extra="forbid", frozen=True)
561
+
562
+ schema_version: StrictInt = 1
563
+ module_name: StrictStr = EVAL_BOUNDARY_MODULE_NAME
564
+ artifact_module_required: StrictBool = EVAL_BOUNDARY_ARTIFACT_MODULE_REQUIRED
565
+ artifact_module_decision: StrictStr = EVAL_BOUNDARY_ARTIFACT_MODULE_DECISION
566
+ authoritative_public_names: tuple[StrictStr, ...] = Field(
567
+ default=AUTHORITATIVE_COMPACT_EVAL_WORKFLOW_NAMES
568
+ )
569
+ graph_id: StrictStr
570
+ graph_sha256: StrictStr
571
+ stage_ids: tuple[EvalStageId, ...]
572
+ terminal_results: tuple[EvalTerminalResult, ...]
573
+ outcome_kinds: tuple[EvalWorkflowOutcomeKind, ...]
574
+ candidate_dispositions: tuple[EvalCandidateDisposition, ...]
575
+ stage_artifacts: tuple[EvalBoundaryStageArtifacts, ...]
576
+
577
+
578
+ class EvalContextTier(str, Enum):
579
+ """Closed context tiers for compact eval stage prompts."""
580
+
581
+ COMPACT = "compact"
582
+ VALIDATOR_VISIBLE = "validator_visible"
583
+
584
+
585
+ class EvalContextArtifactSummary(BaseModel):
586
+ """Path-free summary of a model-visible artifact required by a stage."""
587
+
588
+ model_config = ConfigDict(extra="forbid", frozen=True)
589
+
590
+ artifact_id: StrictStr
591
+ summary: StrictStr
592
+
593
+ @model_validator(mode="after")
594
+ def _summary_valid(self) -> EvalContextArtifactSummary:
595
+ _reject_context_material_leaks(self.model_dump(mode="json"))
596
+ return self
597
+
598
+
599
+ class EvalContextRedaction(BaseModel):
600
+ """Deterministic summary of material omitted from compact context."""
601
+
602
+ model_config = ConfigDict(extra="forbid", frozen=True)
603
+
604
+ categories: tuple[StrictStr, ...] = EVAL_CONTEXT_DEFAULT_REDACTION_CATEGORIES
605
+ summary: StrictStr = "scorer-only and private workspace material omitted"
606
+ redacted_item_count: StrictInt = Field(ge=0)
607
+
608
+ @model_validator(mode="after")
609
+ def _redaction_valid(self) -> EvalContextRedaction:
610
+ if len(set(self.categories)) != len(self.categories):
611
+ raise ValueError("redaction categories must be unique")
612
+ for category in self.categories:
613
+ if not category.strip() or "/" in category or "\\" in category:
614
+ raise ValueError("redaction categories must be stable identifiers")
615
+ _reject_context_material_leaks(self.summary)
616
+ return self
617
+
618
+
619
+ class EvalResourceCeiling(BaseModel):
620
+ """Positive bounded resource ceiling for a compact eval trial or stage."""
621
+
622
+ model_config = ConfigDict(extra="forbid", frozen=True)
623
+
624
+ scope: StrictStr
625
+ stage_id: EvalStageId | None = None
626
+ prompt_tokens: StrictInt = Field(gt=0, le=1_000_000)
627
+ completion_tokens: StrictInt = Field(gt=0, le=1_000_000)
628
+ model_calls: StrictInt = Field(gt=0, le=100)
629
+ wall_clock_seconds: StrictInt = Field(gt=0, le=86_400)
630
+ shell_commands: StrictInt = Field(gt=0, le=1_000)
631
+ shell_command_seconds: StrictInt = Field(gt=0, le=86_400)
632
+ writable_bytes: StrictInt = Field(gt=0, le=1_073_741_824)
633
+ artifact_bytes: StrictInt = Field(gt=0, le=1_073_741_824)
634
+
635
+ @model_validator(mode="after")
636
+ def _resource_ceiling_valid(self) -> EvalResourceCeiling:
637
+ if self.scope not in {"trial", "stage"}:
638
+ raise ValueError("resource ceiling scope must be trial or stage")
639
+ if self.scope == "trial" and self.stage_id is not None:
640
+ raise ValueError("trial resource ceilings must not declare stage_id")
641
+ if self.scope == "stage" and self.stage_id is None:
642
+ raise ValueError("stage resource ceilings must declare stage_id")
643
+ return self
644
+
645
+
646
+ class EvalStageContextPolicy(BaseModel):
647
+ """Compact context assembly policy for one 06A stage."""
648
+
649
+ model_config = ConfigDict(extra="forbid", frozen=True)
650
+
651
+ stage_id: EvalStageId
652
+ context_tier: EvalContextTier = EvalContextTier.COMPACT
653
+ allowed_capabilities: tuple[EvalCapabilityId, ...]
654
+ allowed_paths: tuple[StrictStr, ...]
655
+ required_artifact_ids: tuple[StrictStr, ...]
656
+ include_current_stage_contract: StrictBool = True
657
+ include_visible_acceptance_checks: StrictBool = True
658
+ redaction: EvalContextRedaction = Field(
659
+ default_factory=lambda: EvalContextRedaction(redacted_item_count=0)
660
+ )
661
+ resource_ceiling: EvalResourceCeiling
662
+
663
+ @field_validator("allowed_paths")
664
+ @classmethod
665
+ def _allowed_paths_valid(cls, value: tuple[str, ...]) -> tuple[str, ...]:
666
+ if len(set(value)) != len(value):
667
+ raise ValueError("allowed_paths values must be unique")
668
+ for path in value:
669
+ _validate_relative_eval_path(path, allow_dot=False)
670
+ _reject_context_material_leaks(path)
671
+ return value
672
+
673
+ @field_validator("required_artifact_ids")
674
+ @classmethod
675
+ def _required_artifacts_valid(cls, value: tuple[str, ...]) -> tuple[str, ...]:
676
+ if not value:
677
+ raise ValueError("required_artifact_ids must not be empty")
678
+ if len(set(value)) != len(value):
679
+ raise ValueError("required_artifact_ids values must be unique")
680
+ for artifact_id in value:
681
+ if not artifact_id.strip() or "/" in artifact_id or "\\" in artifact_id:
682
+ raise ValueError("required artifact ids must be stable IDs")
683
+ return value
684
+
685
+ @model_validator(mode="after")
686
+ def _context_policy_valid(self) -> EvalStageContextPolicy:
687
+ envelope = _EVAL_CAPABILITY_ENVELOPES[self.stage_id]
688
+ if self.allowed_capabilities != envelope.capability_ids:
689
+ raise ValueError("context policy capabilities must match stage envelope")
690
+ if self.resource_ceiling.scope != "stage":
691
+ raise ValueError("stage context policies require a stage resource ceiling")
692
+ if self.resource_ceiling.stage_id != self.stage_id:
693
+ raise ValueError("resource ceiling stage_id must match context policy")
694
+ return self
695
+
696
+
697
+ class EvalContextSnapshot(BaseModel):
698
+ """Deterministic compact model-visible context snapshot."""
699
+
700
+ model_config = ConfigDict(extra="forbid", frozen=True)
701
+
702
+ trial_id: StrictStr
703
+ stage_id: EvalStageId
704
+ context_tier: EvalContextTier
705
+ allowed_capabilities: tuple[StrictStr, ...]
706
+ allowed_paths: tuple[StrictStr, ...]
707
+ current_stage_contract: Mapping[StrictStr, Any]
708
+ required_artifact_summaries: tuple[EvalContextArtifactSummary, ...]
709
+ visible_acceptance_check_ids: tuple[StrictStr, ...]
710
+ redaction: EvalContextRedaction
711
+ byte_budget: StrictInt = Field(gt=0, le=1_073_741_824)
712
+ token_budget: StrictInt = Field(gt=0, le=1_000_000)
713
+ resource_ceiling: EvalResourceCeiling
714
+ fingerprint_kind: StrictStr = EVAL_CONTEXT_FINGERPRINT_KIND
715
+ fingerprint: StrictStr
716
+
717
+ @field_validator("trial_id", "fingerprint")
718
+ @classmethod
719
+ def _snapshot_text_valid(cls, value: str) -> str:
720
+ if not value.strip():
721
+ raise ValueError("context snapshot text fields must be non-empty")
722
+ _reject_context_material_leaks(value)
723
+ return value
724
+
725
+ @field_validator("allowed_paths")
726
+ @classmethod
727
+ def _snapshot_paths_valid(cls, value: tuple[str, ...]) -> tuple[str, ...]:
728
+ if len(set(value)) != len(value):
729
+ raise ValueError("allowed_paths values must be unique")
730
+ for path in value:
731
+ _validate_relative_eval_path(path, allow_dot=False)
732
+ _reject_context_material_leaks(path)
733
+ return value
734
+
735
+ @field_validator("visible_acceptance_check_ids")
736
+ @classmethod
737
+ def _visible_check_ids_valid(cls, value: tuple[str, ...]) -> tuple[str, ...]:
738
+ if len(set(value)) != len(value):
739
+ raise ValueError("visible_acceptance_check_ids values must be unique")
740
+ for check_id in value:
741
+ if not check_id.strip() or "/" in check_id or "\\" in check_id:
742
+ raise ValueError("visible acceptance check ids must be stable IDs")
743
+ _reject_context_material_leaks(check_id)
744
+ return value
745
+
746
+ @model_validator(mode="after")
747
+ def _snapshot_valid(self) -> EvalContextSnapshot:
748
+ _validate_sha256(self.fingerprint)
749
+ if self.fingerprint_kind != EVAL_CONTEXT_FINGERPRINT_KIND:
750
+ raise ValueError("unsupported context fingerprint kind")
751
+ if self.resource_ceiling.stage_id != self.stage_id:
752
+ raise ValueError("context snapshot resource ceiling must match stage")
753
+ _reject_context_material_leaks(self.model_dump(mode="json"))
754
+ return self
755
+
756
+
757
+ class EvalPathPolicyViolation(BaseModel):
758
+ """Structured relative-path policy diagnostic without host path leakage."""
759
+
760
+ model_config = ConfigDict(extra="forbid", frozen=True)
761
+
762
+ path: StrictStr
763
+ rule_id: StrictStr
764
+ diagnostic_code: StrictStr
765
+ diagnostic_summary: StrictStr
766
+
767
+
768
+ class EvalFixtureFile(BaseModel):
769
+ """Declared fixture file identity."""
770
+
771
+ model_config = ConfigDict(extra="forbid", frozen=True)
772
+
773
+ path: StrictStr
774
+ sha256: StrictStr
775
+ size_bytes: StrictInt = Field(ge=0)
776
+ role: StrictStr
777
+ model_readable: StrictBool
778
+ builder_mutable: StrictBool
779
+
780
+ @field_validator("path")
781
+ @classmethod
782
+ def _path_valid(cls, value: str) -> str:
783
+ _validate_relative_eval_path(value, allow_dot=False)
784
+ return value
785
+
786
+ @field_validator("role")
787
+ @classmethod
788
+ def _role_valid(cls, value: str) -> str:
789
+ if value not in EVAL_FIXTURE_FILE_ROLES:
790
+ raise ValueError("fixture file role is not in the closed role set")
791
+ return value
792
+
793
+ @field_validator("sha256")
794
+ @classmethod
795
+ def _sha256_valid(cls, value: str) -> str:
796
+ if len(value) != 64 or any(
797
+ character not in "0123456789abcdef" for character in value
798
+ ):
799
+ raise ValueError("sha256 must be a lowercase hexadecimal SHA-256 digest")
800
+ return value
801
+
802
+
803
+ class EvalFixtureWorkspacePolicy(BaseModel):
804
+ """Workspace isolation and stage write policy for one eval fixture."""
805
+
806
+ model_config = ConfigDict(extra="forbid", frozen=True)
807
+
808
+ source_fixture_root_read_only: StrictBool = True
809
+ workspace_isolation: StrictStr = EVAL_WORKSPACE_ISOLATION_CONTRACT
810
+ stage_write_roots: Mapping[EvalStageId, tuple[StrictStr, ...]] = Field(
811
+ default_factory=lambda: {
812
+ EvalStageId.BUILDER: EVAL_BUILDER_DEFAULT_WRITE_ROOTS,
813
+ EvalStageId.CHECKER: (),
814
+ EvalStageId.ARBITER: (),
815
+ EvalStageId.PLANNER: (),
816
+ }
817
+ )
818
+ ignored_generated_roots: tuple[StrictStr, ...] = (
819
+ EVAL_FIXTURE_IGNORED_GENERATED_ROOTS
820
+ )
821
+ ignored_generated_suffixes: tuple[StrictStr, ...] = (
822
+ EVAL_FIXTURE_IGNORED_GENERATED_SUFFIXES
823
+ )
824
+
825
+ @model_validator(mode="after")
826
+ def _policy_valid(self) -> EvalFixtureWorkspacePolicy:
827
+ if not self.source_fixture_root_read_only:
828
+ raise ValueError("source fixture root must be read-only")
829
+ if self.workspace_isolation != EVAL_WORKSPACE_ISOLATION_CONTRACT:
830
+ raise ValueError("fixture workspaces must use a fresh copied workspace")
831
+ ordered_stage_write_roots: dict[EvalStageId, tuple[str, ...]] = {}
832
+ for stage_id in EvalStageId:
833
+ roots = tuple(self.stage_write_roots.get(stage_id, ()))
834
+ if len(set(roots)) != len(roots):
835
+ raise ValueError("stage write roots must be unique")
836
+ for root in roots:
837
+ _validate_relative_eval_path(root, allow_dot=False)
838
+ ordered_stage_write_roots[stage_id] = roots
839
+ if ordered_stage_write_roots[EvalStageId.CHECKER]:
840
+ raise ValueError("checker fixture workspace policy must be read-only")
841
+ if ordered_stage_write_roots[EvalStageId.ARBITER]:
842
+ raise ValueError("arbiter fixture workspace policy must be read-only")
843
+ if ordered_stage_write_roots[EvalStageId.PLANNER]:
844
+ raise ValueError("planner fixture workspace policy must be read-only")
845
+ for ignored_root in self.ignored_generated_roots:
846
+ _validate_relative_eval_path(ignored_root, allow_dot=False)
847
+ for suffix in self.ignored_generated_suffixes:
848
+ if not suffix or "/" in suffix or "\\" in suffix:
849
+ raise ValueError("ignored generated suffixes must be file suffixes")
850
+ object.__setattr__(
851
+ self, "stage_write_roots", MappingProxyType(ordered_stage_write_roots)
852
+ )
853
+ return self
854
+
855
+ @field_serializer("stage_write_roots")
856
+ def _serialize_stage_write_roots(
857
+ self, value: Mapping[EvalStageId, tuple[str, ...]]
858
+ ) -> dict[str, list[str]]:
859
+ return {stage_id.value: list(value[stage_id]) for stage_id in EvalStageId}
860
+
861
+
862
+ class EvalFixtureManifest(BaseModel):
863
+ """Deterministic fixture manifest with declared file hashes."""
864
+
865
+ model_config = ConfigDict(extra="forbid", frozen=True)
866
+
867
+ schema_version: StrictInt
868
+ fixture_id: StrictStr
869
+ fixture_revision: StrictStr
870
+ task_id: StrictStr
871
+ source_root_label: StrictStr
872
+ allowed_read_paths: tuple[StrictStr, ...]
873
+ allowed_write_paths: tuple[StrictStr, ...]
874
+ allowed_command_roots: tuple[StrictStr, ...]
875
+ visible_acceptance_checks: tuple[StrictStr, ...]
876
+ hidden_check_ids: tuple[StrictStr, ...]
877
+ expected_mutation_paths: tuple[StrictStr, ...]
878
+ files: tuple[EvalFixtureFile, ...]
879
+ workspace_policy: EvalFixtureWorkspacePolicy = Field(
880
+ default_factory=EvalFixtureWorkspacePolicy
881
+ )
882
+
883
+ @field_validator("schema_version")
884
+ @classmethod
885
+ def _schema_version_valid(cls, value: int) -> int:
886
+ if value != EVAL_FIXTURE_MANIFEST_SCHEMA_VERSION:
887
+ raise ValueError("unsupported fixture manifest schema_version")
888
+ return value
889
+
890
+ @field_validator("fixture_id", "fixture_revision", "task_id", "source_root_label")
891
+ @classmethod
892
+ def _stable_text_valid(cls, value: str) -> str:
893
+ if not value.strip() or "/" in value or "\\" in value:
894
+ raise ValueError("fixture manifest text fields must be stable identifiers")
895
+ _reject_context_material_leaks(value)
896
+ return value
897
+
898
+ @field_validator(
899
+ "allowed_read_paths",
900
+ "allowed_write_paths",
901
+ "allowed_command_roots",
902
+ "expected_mutation_paths",
903
+ )
904
+ @classmethod
905
+ def _manifest_paths_valid(cls, value: tuple[str, ...]) -> tuple[str, ...]:
906
+ if len(set(value)) != len(value):
907
+ raise ValueError("fixture manifest paths must be unique")
908
+ for path in value:
909
+ try:
910
+ _validate_relative_eval_path(path, allow_dot=False)
911
+ except ValueError as exc:
912
+ raise ValueError(
913
+ "fixture manifest paths must be normalized relative POSIX paths"
914
+ ) from exc
915
+ _reject_context_material_leaks(path)
916
+ return tuple(sorted(value))
917
+
918
+ @field_validator("visible_acceptance_checks")
919
+ @classmethod
920
+ def _visible_acceptance_checks_valid(
921
+ cls, value: tuple[str, ...]
922
+ ) -> tuple[str, ...]:
923
+ if len(set(value)) != len(value):
924
+ raise ValueError("visible acceptance checks must be unique")
925
+ for check_id in value:
926
+ _validate_opaque_eval_id(check_id, field_name="visible acceptance check")
927
+ return value
928
+
929
+ @field_validator("hidden_check_ids")
930
+ @classmethod
931
+ def _hidden_check_ids_valid(cls, value: tuple[str, ...]) -> tuple[str, ...]:
932
+ if len(set(value)) != len(value):
933
+ raise ValueError("hidden_check_ids must be unique")
934
+ for check_id in value:
935
+ _validate_opaque_eval_id(check_id, field_name="hidden check id")
936
+ lowered = check_id.lower()
937
+ if any(token in lowered for token in _HIDDEN_CHECK_DENIED_TOKENS):
938
+ raise ValueError("hidden_check_ids must contain only opaque IDs")
939
+ return value
940
+
941
+ @model_validator(mode="after")
942
+ def _manifest_valid(self) -> EvalFixtureManifest:
943
+ paths = tuple(file.path for file in self.files)
944
+ if not paths:
945
+ raise ValueError("fixture manifests must declare at least one file")
946
+ if len(set(paths)) != len(paths):
947
+ raise ValueError("fixture file paths must be unique")
948
+ for mutation_path in self.expected_mutation_paths:
949
+ if mutation_path not in paths and not any(
950
+ _path_is_within_root(path, mutation_path) for path in paths
951
+ ):
952
+ raise ValueError(
953
+ "expected_mutation_paths must reference declared fixture files or roots"
954
+ )
955
+ object.__setattr__(
956
+ self, "files", tuple(sorted(self.files, key=lambda file: file.path))
957
+ )
958
+ return self
959
+
960
+
961
+ class EvalFixtureWorkspaceSnapshot(BaseModel):
962
+ """Deterministic comparison between a declared manifest and workspace state."""
963
+
964
+ model_config = ConfigDict(extra="forbid", frozen=True)
965
+
966
+ fixture_id: StrictStr
967
+ fixture_manifest_sha256: StrictStr
968
+ files: tuple[EvalFixtureFile, ...]
969
+ added_paths: tuple[StrictStr, ...] = Field(default_factory=tuple)
970
+ modified_paths: tuple[StrictStr, ...] = Field(default_factory=tuple)
971
+ deleted_paths: tuple[StrictStr, ...] = Field(default_factory=tuple)
972
+ unchanged_paths: tuple[StrictStr, ...] = Field(default_factory=tuple)
973
+ ignored_generated_paths: tuple[StrictStr, ...] = Field(default_factory=tuple)
974
+ unauthorized_mutation_paths: tuple[StrictStr, ...] = Field(default_factory=tuple)
975
+ violations: tuple[EvalPathPolicyViolation, ...] = Field(default_factory=tuple)
976
+
977
+ @field_validator("fixture_manifest_sha256")
978
+ @classmethod
979
+ def _fixture_manifest_sha256_valid(cls, value: str) -> str:
980
+ _validate_sha256(value)
981
+ return value
982
+
983
+
984
+ class EvalClosureOutcomeKind(str, Enum):
985
+ """Structured compact-eval closure validation outcomes."""
986
+
987
+ VALID_CLOSED_SUCCESS = "valid_closed_success"
988
+ VALID_CLOSED_REJECTION = "valid_closed_rejection"
989
+ VALID_BLOCKED_OUTCOME = "valid_blocked_outcome"
990
+ INVALID_ARTIFACT_BOUNDARY = "invalid_artifact_boundary"
991
+ INVALID_CAPABILITY_BOUNDARY = "invalid_capability_boundary"
992
+ INVALID_FIXTURE_BOUNDARY = "invalid_fixture_boundary"
993
+ INVALID_CONTEXT_BOUNDARY = "invalid_context_boundary"
994
+
995
+
996
+ class EvalClosureValidationResult(BaseModel):
997
+ """Non-mutating aggregate validation result for compact eval closure evidence."""
998
+
999
+ model_config = ConfigDict(extra="forbid", frozen=True)
1000
+
1001
+ valid: StrictBool
1002
+ outcome_kind: EvalClosureOutcomeKind
1003
+ terminal_result: EvalTerminalResult | None = None
1004
+ candidate_disposition: EvalCandidateDisposition | None = None
1005
+ evidence_artifact_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
1006
+ missing_artifact_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
1007
+ diagnostics: tuple[StrictStr, ...] = Field(default_factory=tuple)
1008
+
1009
+ @model_validator(mode="after")
1010
+ def _closure_result_valid(self) -> EvalClosureValidationResult:
1011
+ if self.valid:
1012
+ if self.outcome_kind not in {
1013
+ EvalClosureOutcomeKind.VALID_CLOSED_SUCCESS,
1014
+ EvalClosureOutcomeKind.VALID_CLOSED_REJECTION,
1015
+ EvalClosureOutcomeKind.VALID_BLOCKED_OUTCOME,
1016
+ }:
1017
+ raise ValueError("valid closure results require a valid outcome kind")
1018
+ if self.diagnostics:
1019
+ raise ValueError("valid closure results must not include diagnostics")
1020
+ elif self.outcome_kind in {
1021
+ EvalClosureOutcomeKind.VALID_CLOSED_SUCCESS,
1022
+ EvalClosureOutcomeKind.VALID_CLOSED_REJECTION,
1023
+ EvalClosureOutcomeKind.VALID_BLOCKED_OUTCOME,
1024
+ }:
1025
+ raise ValueError("invalid closure results require an invalid outcome kind")
1026
+ return self
1027
+
1028
+
1029
+ class EvalCheckerApprovalValidationResult(BaseModel):
1030
+ """Non-mutating validation result for Checker approval public evidence."""
1031
+
1032
+ model_config = ConfigDict(extra="forbid", frozen=True)
1033
+
1034
+ valid: StrictBool
1035
+ evidence_artifact_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
1036
+ missing_artifact_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
1037
+ diagnostics: tuple[StrictStr, ...] = Field(default_factory=tuple)
1038
+
1039
+ @model_validator(mode="after")
1040
+ def _checker_approval_result_valid(
1041
+ self,
1042
+ ) -> EvalCheckerApprovalValidationResult:
1043
+ if self.valid:
1044
+ if self.missing_artifact_ids or self.diagnostics:
1045
+ raise ValueError(
1046
+ "valid checker approval results must not include diagnostics"
1047
+ )
1048
+ elif not self.missing_artifact_ids and not self.diagnostics:
1049
+ raise ValueError(
1050
+ "invalid checker approval results require diagnostics or missing IDs"
1051
+ )
1052
+ return self
1053
+
1054
+
1055
+ def _validate_relative_eval_path(value: str, *, allow_dot: bool) -> None:
1056
+ if not value:
1057
+ raise ValueError("eval paths must be non-empty relative POSIX paths")
1058
+ if value == ".":
1059
+ if allow_dot:
1060
+ return
1061
+ raise ValueError("eval root paths must not be '.'")
1062
+ posix_path = PurePosixPath(value)
1063
+ parts = posix_path.parts
1064
+ if (
1065
+ value.startswith("/")
1066
+ or value.startswith("\\")
1067
+ or "\\" in value
1068
+ or "//" in value
1069
+ or value.startswith("../")
1070
+ or value.endswith("/..")
1071
+ or "/../" in value
1072
+ or value in {"..", ""}
1073
+ or _WINDOWS_DRIVE_PREFIX.match(value)
1074
+ or "." in parts
1075
+ or posix_path.as_posix() != value
1076
+ ):
1077
+ raise ValueError("eval paths must be normalized relative POSIX paths")
1078
+
1079
+
1080
+ def _canonical_eval_json_bytes(value: Any) -> bytes:
1081
+ return (
1082
+ json.dumps(
1083
+ value,
1084
+ sort_keys=True,
1085
+ ensure_ascii=True,
1086
+ allow_nan=False,
1087
+ separators=(",", ":"),
1088
+ ).replace("\r\n", "\n")
1089
+ + "\n"
1090
+ ).encode("ascii")
1091
+
1092
+
1093
+ def _context_material_text_values(value: Any) -> tuple[str, ...]:
1094
+ if isinstance(value, str):
1095
+ return (value,)
1096
+ if isinstance(value, Mapping):
1097
+ return tuple(
1098
+ text
1099
+ for child in value.values()
1100
+ for text in _context_material_text_values(child)
1101
+ )
1102
+ if isinstance(value, (tuple, list, set, frozenset)):
1103
+ return tuple(
1104
+ text for child in value for text in _context_material_text_values(child)
1105
+ )
1106
+ return ()
1107
+
1108
+
1109
+ def _reject_context_material_leaks(value: Any) -> None:
1110
+ for text in _context_material_text_values(value):
1111
+ if text in EVAL_CONTEXT_DEFAULT_REDACTION_CATEGORIES:
1112
+ continue
1113
+ lowered = text.lower()
1114
+ for token in _CONTEXT_LEAK_TOKENS:
1115
+ if token.lower() in lowered:
1116
+ raise ValueError(
1117
+ "eval context snapshots must not expose private material"
1118
+ )
1119
+ if (
1120
+ _CONTEXT_WINDOWS_ABSOLUTE_PATH.search(text)
1121
+ or _CONTEXT_POSIX_ABSOLUTE_PATH.search(text)
1122
+ or _CONTEXT_USER_HOME_PATH.search(text)
1123
+ ):
1124
+ raise ValueError("eval context snapshots must not expose host paths")
1125
+
1126
+
1127
+ def _validate_sha256(value: str) -> None:
1128
+ if len(value) != 64 or any(
1129
+ character not in "0123456789abcdef" for character in value
1130
+ ):
1131
+ raise ValueError("sha256 must be a lowercase hexadecimal SHA-256 digest")
1132
+
1133
+
1134
+ def _validate_opaque_eval_id(value: str, *, field_name: str) -> None:
1135
+ if not value.strip():
1136
+ raise ValueError(f"{field_name} must be a non-empty opaque ID")
1137
+ if "/" in value or "\\" in value or _WINDOWS_DRIVE_PREFIX.match(value):
1138
+ raise ValueError(f"{field_name} must not contain paths")
1139
+ if any(character.isspace() for character in value):
1140
+ raise ValueError(f"{field_name} must not contain whitespace")
1141
+ _reject_context_material_leaks(value)
1142
+
1143
+
1144
+ def _context_fingerprint_payload(snapshot: EvalContextSnapshot) -> dict[str, Any]:
1145
+ payload = snapshot.model_dump(mode="json")
1146
+ payload.pop("fingerprint", None)
1147
+ return payload
1148
+
1149
+
1150
+ def calculate_eval_context_fingerprint(snapshot: EvalContextSnapshot) -> str:
1151
+ """Return the deterministic fingerprint for a compact context snapshot."""
1152
+ return hashlib.sha256(
1153
+ _canonical_eval_json_bytes(_context_fingerprint_payload(snapshot))
1154
+ ).hexdigest()
1155
+
1156
+
1157
+ def _path_policy_violation(
1158
+ path: str,
1159
+ *,
1160
+ rule_id: str,
1161
+ diagnostic_code: str,
1162
+ diagnostic_summary: str,
1163
+ ) -> EvalPathPolicyViolation:
1164
+ return EvalPathPolicyViolation(
1165
+ path=_diagnostic_path(path),
1166
+ rule_id=rule_id,
1167
+ diagnostic_code=diagnostic_code,
1168
+ diagnostic_summary=diagnostic_summary,
1169
+ )
1170
+
1171
+
1172
+ def _diagnostic_path(path: str) -> str:
1173
+ if (
1174
+ path.startswith("/")
1175
+ or path.startswith("\\")
1176
+ or _WINDOWS_DRIVE_PREFIX.match(path)
1177
+ ):
1178
+ return "<absolute-path>"
1179
+ return path
1180
+
1181
+
1182
+ def validate_eval_fixture_path(
1183
+ path: str,
1184
+ *,
1185
+ filesystem_root: Path | None = None,
1186
+ ) -> EvalPathPolicyViolation | None:
1187
+ """Return a structured violation for an invalid fixture path, else None."""
1188
+ try:
1189
+ _validate_relative_eval_path(path, allow_dot=False)
1190
+ except ValueError:
1191
+ return _path_policy_violation(
1192
+ path=path,
1193
+ rule_id="eval.fixture.path.invalid",
1194
+ diagnostic_code="MF-EVAL-F001",
1195
+ diagnostic_summary="fixture paths must be normalized relative POSIX paths",
1196
+ )
1197
+ if filesystem_root is None:
1198
+ return None
1199
+ root = filesystem_root.resolve()
1200
+ candidate = (root / Path(*PurePosixPath(path).parts)).resolve()
1201
+ try:
1202
+ candidate.relative_to(root)
1203
+ except ValueError:
1204
+ return _path_policy_violation(
1205
+ path=path,
1206
+ rule_id="eval.fixture.path.symlink_escape",
1207
+ diagnostic_code="MF-EVAL-F002",
1208
+ diagnostic_summary="fixture path resolves outside fixture root",
1209
+ )
1210
+ return None
1211
+
1212
+
1213
+ def eval_fixture_file_hash(path: Path) -> str:
1214
+ """Return the SHA-256 hash for a fixture file."""
1215
+ digest = hashlib.sha256()
1216
+ with path.open("rb") as file:
1217
+ for chunk in iter(lambda: file.read(1024 * 1024), b""):
1218
+ digest.update(chunk)
1219
+ return digest.hexdigest()
1220
+
1221
+
1222
+ def _default_fixture_file_role(relative_path: str) -> str:
1223
+ if relative_path.startswith("tests/") or relative_path.startswith("test/"):
1224
+ return "test"
1225
+ if relative_path.startswith("src/"):
1226
+ return "source"
1227
+ if relative_path.endswith((".md", ".rst", ".txt")):
1228
+ return "documentation"
1229
+ if relative_path.endswith((".json", ".toml", ".yaml", ".yml", ".ini", ".cfg")):
1230
+ return "configuration"
1231
+ return "data"
1232
+
1233
+
1234
+ def eval_fixture_file_record(
1235
+ fixture_root: Path,
1236
+ relative_path: str,
1237
+ *,
1238
+ role: str | None = None,
1239
+ model_readable: bool = True,
1240
+ builder_mutable: bool = False,
1241
+ ) -> EvalFixtureFile:
1242
+ """Build a deterministic file record for one relative fixture path."""
1243
+ violation = validate_eval_fixture_path(relative_path, filesystem_root=fixture_root)
1244
+ if violation is not None:
1245
+ raise ValueError(violation.diagnostic_summary)
1246
+ path = fixture_root / Path(*PurePosixPath(relative_path).parts)
1247
+ return EvalFixtureFile(
1248
+ path=relative_path,
1249
+ sha256=eval_fixture_file_hash(path),
1250
+ size_bytes=path.stat().st_size,
1251
+ role=role or _default_fixture_file_role(relative_path),
1252
+ model_readable=model_readable,
1253
+ builder_mutable=builder_mutable,
1254
+ )
1255
+
1256
+
1257
+ def eval_fixture_manifest_sha256(manifest: EvalFixtureManifest) -> str:
1258
+ """Return the deterministic SHA-256 of an expanded fixture manifest payload."""
1259
+ return hashlib.sha256(
1260
+ _canonical_eval_json_bytes(manifest.model_dump(mode="json"))
1261
+ ).hexdigest()
1262
+
1263
+
1264
+ def eval_fixture_manifest_from_paths(
1265
+ fixture_id: str,
1266
+ fixture_root: Path,
1267
+ relative_paths: tuple[str, ...],
1268
+ *,
1269
+ task_id: str,
1270
+ schema_version: int = EVAL_FIXTURE_MANIFEST_SCHEMA_VERSION,
1271
+ fixture_revision: str = EVAL_FIXTURE_DEFAULT_REVISION,
1272
+ source_root_label: str = "fixture_workspace",
1273
+ allowed_read_paths: tuple[str, ...] | None = None,
1274
+ allowed_write_paths: tuple[str, ...] = EVAL_BUILDER_DEFAULT_WRITE_ROOTS,
1275
+ allowed_command_roots: tuple[str, ...] = ("src", "tests"),
1276
+ visible_acceptance_checks: tuple[str, ...] = (),
1277
+ hidden_check_ids: tuple[str, ...] = (),
1278
+ expected_mutation_paths: tuple[str, ...] | None = None,
1279
+ workspace_policy: EvalFixtureWorkspacePolicy | None = None,
1280
+ ) -> EvalFixtureManifest:
1281
+ """Build a deterministic manifest from declared fixture-relative files."""
1282
+ declared_paths = tuple(sorted(relative_paths))
1283
+ mutation_paths = expected_mutation_paths or ()
1284
+ return EvalFixtureManifest(
1285
+ schema_version=schema_version,
1286
+ fixture_id=fixture_id,
1287
+ fixture_revision=fixture_revision,
1288
+ task_id=task_id,
1289
+ source_root_label=source_root_label,
1290
+ allowed_read_paths=allowed_read_paths or declared_paths,
1291
+ allowed_write_paths=allowed_write_paths,
1292
+ allowed_command_roots=allowed_command_roots,
1293
+ visible_acceptance_checks=visible_acceptance_checks,
1294
+ hidden_check_ids=hidden_check_ids,
1295
+ expected_mutation_paths=mutation_paths,
1296
+ files=tuple(
1297
+ eval_fixture_file_record(
1298
+ fixture_root,
1299
+ relative_path,
1300
+ builder_mutable=relative_path in mutation_paths,
1301
+ )
1302
+ for relative_path in declared_paths
1303
+ ),
1304
+ workspace_policy=workspace_policy or EvalFixtureWorkspacePolicy(),
1305
+ )
1306
+
1307
+
1308
+ def _is_ignored_generated_path(path: str, policy: EvalFixtureWorkspacePolicy) -> bool:
1309
+ parts = PurePosixPath(path).parts
1310
+ return any(
1311
+ root in parts or _path_is_within_root(path, root)
1312
+ for root in policy.ignored_generated_roots
1313
+ ) or any(path.endswith(suffix) for suffix in policy.ignored_generated_suffixes)
1314
+
1315
+
1316
+ def _workspace_files(
1317
+ workspace_root: Path,
1318
+ ) -> tuple[tuple[str, ...], tuple[EvalPathPolicyViolation, ...]]:
1319
+ paths: list[str] = []
1320
+ violations: list[EvalPathPolicyViolation] = []
1321
+ for file_path in workspace_root.rglob("*"):
1322
+ if file_path.is_file():
1323
+ relative_path = file_path.relative_to(workspace_root).as_posix()
1324
+ violation = validate_eval_fixture_path(
1325
+ relative_path, filesystem_root=workspace_root
1326
+ )
1327
+ if violation is None:
1328
+ paths.append(relative_path)
1329
+ else:
1330
+ violations.append(violation)
1331
+ return tuple(sorted(paths)), tuple(sorted(violations, key=lambda item: item.path))
1332
+
1333
+
1334
+ def _unauthorized_paths(
1335
+ *,
1336
+ stage_id: EvalStageId,
1337
+ added_paths: tuple[str, ...],
1338
+ modified_paths: tuple[str, ...],
1339
+ deleted_paths: tuple[str, ...],
1340
+ policy: EvalFixtureWorkspacePolicy,
1341
+ ) -> tuple[str, ...]:
1342
+ allowed_roots = policy.stage_write_roots.get(stage_id, ())
1343
+ changed_paths = added_paths + modified_paths + deleted_paths
1344
+ if not allowed_roots:
1345
+ return tuple(sorted(changed_paths))
1346
+ return tuple(
1347
+ sorted(
1348
+ path
1349
+ for path in changed_paths
1350
+ if not any(_path_is_within_root(path, root) for root in allowed_roots)
1351
+ )
1352
+ )
1353
+
1354
+
1355
+ def eval_fixture_workspace_snapshot(
1356
+ manifest: EvalFixtureManifest,
1357
+ workspace_root: Path,
1358
+ *,
1359
+ stage_id: EvalStageId = EvalStageId.BUILDER,
1360
+ ) -> EvalFixtureWorkspaceSnapshot:
1361
+ """Compare a workspace with its fixture manifest using relative paths only."""
1362
+ declared = {file.path: file for file in manifest.files}
1363
+ current_files: dict[str, EvalFixtureFile] = {}
1364
+ ignored_generated_paths: list[str] = []
1365
+ workspace_files, workspace_violations = _workspace_files(workspace_root)
1366
+ violations: list[EvalPathPolicyViolation] = list(workspace_violations)
1367
+ for relative_path in workspace_files:
1368
+ if _is_ignored_generated_path(relative_path, manifest.workspace_policy):
1369
+ ignored_generated_paths.append(relative_path)
1370
+ continue
1371
+ current_files[relative_path] = eval_fixture_file_record(
1372
+ workspace_root, relative_path
1373
+ )
1374
+
1375
+ declared_paths = set(declared)
1376
+ current_paths = set(current_files)
1377
+ added_paths = tuple(sorted(current_paths - declared_paths))
1378
+ deleted_paths = tuple(sorted(declared_paths - current_paths))
1379
+ unchanged_paths = tuple(
1380
+ sorted(
1381
+ path
1382
+ for path in declared_paths & current_paths
1383
+ if declared[path].sha256 == current_files[path].sha256
1384
+ and declared[path].size_bytes == current_files[path].size_bytes
1385
+ )
1386
+ )
1387
+ modified_paths = tuple(
1388
+ sorted((declared_paths & current_paths) - set(unchanged_paths))
1389
+ )
1390
+ unauthorized_mutation_paths = _unauthorized_paths(
1391
+ stage_id=stage_id,
1392
+ added_paths=added_paths,
1393
+ modified_paths=modified_paths,
1394
+ deleted_paths=deleted_paths,
1395
+ policy=manifest.workspace_policy,
1396
+ )
1397
+ unauthorized_mutation_paths = tuple(
1398
+ sorted(
1399
+ set(unauthorized_mutation_paths)
1400
+ | {violation.path for violation in violations}
1401
+ )
1402
+ )
1403
+ return EvalFixtureWorkspaceSnapshot(
1404
+ fixture_id=manifest.fixture_id,
1405
+ fixture_manifest_sha256=eval_fixture_manifest_sha256(manifest),
1406
+ files=tuple(current_files[path] for path in sorted(current_files)),
1407
+ added_paths=added_paths,
1408
+ modified_paths=modified_paths,
1409
+ deleted_paths=deleted_paths,
1410
+ unchanged_paths=unchanged_paths,
1411
+ ignored_generated_paths=tuple(sorted(ignored_generated_paths)),
1412
+ unauthorized_mutation_paths=unauthorized_mutation_paths,
1413
+ violations=tuple(violations),
1414
+ )
1415
+
1416
+
1417
+ def _path_is_within_root(path: str, root: str) -> bool:
1418
+ return path == root or path.startswith(f"{root}/")
1419
+
1420
+
1421
+ def _coerce_stage_id(stage_id: EvalStageId | str) -> EvalStageId | str:
1422
+ if isinstance(stage_id, EvalStageId):
1423
+ return stage_id
1424
+ try:
1425
+ return EvalStageId(stage_id)
1426
+ except ValueError:
1427
+ return stage_id
1428
+
1429
+
1430
+ def _coerce_capability_id(capability: EvalCapabilityId | str) -> EvalCapabilityId | str:
1431
+ if isinstance(capability, EvalCapabilityId):
1432
+ return capability
1433
+ try:
1434
+ return EvalCapabilityId(capability)
1435
+ except ValueError:
1436
+ return capability
1437
+
1438
+
1439
+ _EVAL_CAPABILITY_ENVELOPES: Mapping[EvalStageId, EvalCapabilityEnvelope] = (
1440
+ MappingProxyType(
1441
+ {
1442
+ EvalStageId.PLANNER: EvalCapabilityEnvelope(
1443
+ stage_id=EvalStageId.PLANNER,
1444
+ capability_ids=(
1445
+ EvalCapabilityId.ARTIFACT_READ,
1446
+ EvalCapabilityId.ARTIFACT_WRITE,
1447
+ EvalCapabilityId.EVIDENCE_EMIT,
1448
+ EvalCapabilityId.RUNNER_INVOKE,
1449
+ ),
1450
+ ),
1451
+ EvalStageId.BUILDER: EvalCapabilityEnvelope(
1452
+ stage_id=EvalStageId.BUILDER,
1453
+ capability_ids=(
1454
+ EvalCapabilityId.ARTIFACT_READ,
1455
+ EvalCapabilityId.ARTIFACT_WRITE,
1456
+ EvalCapabilityId.EVIDENCE_EMIT,
1457
+ EvalCapabilityId.RUNNER_INVOKE,
1458
+ EvalCapabilityId.WORKSPACE_READ,
1459
+ EvalCapabilityId.WORKSPACE_WRITE,
1460
+ EvalCapabilityId.SHELL_RUN,
1461
+ ),
1462
+ ),
1463
+ EvalStageId.CHECKER: EvalCapabilityEnvelope(
1464
+ stage_id=EvalStageId.CHECKER,
1465
+ capability_ids=(
1466
+ EvalCapabilityId.WORKSPACE_READ,
1467
+ EvalCapabilityId.ARTIFACT_READ,
1468
+ EvalCapabilityId.ARTIFACT_WRITE,
1469
+ EvalCapabilityId.SHELL_RUN,
1470
+ EvalCapabilityId.EVIDENCE_EMIT,
1471
+ EvalCapabilityId.RUNNER_INVOKE,
1472
+ ),
1473
+ ),
1474
+ EvalStageId.ARBITER: EvalCapabilityEnvelope(
1475
+ stage_id=EvalStageId.ARBITER,
1476
+ capability_ids=(
1477
+ EvalCapabilityId.WORKSPACE_READ,
1478
+ EvalCapabilityId.ARTIFACT_READ,
1479
+ EvalCapabilityId.ARTIFACT_WRITE,
1480
+ EvalCapabilityId.EVIDENCE_EMIT,
1481
+ EvalCapabilityId.RUNNER_INVOKE,
1482
+ ),
1483
+ ),
1484
+ }
1485
+ )
1486
+ )
1487
+
1488
+
1489
+ def default_eval_stage_resource_ceiling(stage_id: EvalStageId) -> EvalResourceCeiling:
1490
+ """Return the default positive bounded resource ceiling for one stage."""
1491
+ return EvalResourceCeiling(
1492
+ scope="stage",
1493
+ stage_id=stage_id,
1494
+ **dict(EVAL_STAGE_RESOURCE_CEILING_DEFAULTS[stage_id]),
1495
+ )
1496
+
1497
+
1498
+ def default_eval_trial_resource_ceiling() -> EvalResourceCeiling:
1499
+ """Return the default positive bounded resource ceiling for one trial."""
1500
+ defaults = EVAL_TRIAL_RESOURCE_CEILING_DEFAULTS
1501
+ return EvalResourceCeiling(
1502
+ scope="trial",
1503
+ prompt_tokens=defaults["prompt_tokens"],
1504
+ completion_tokens=defaults["completion_tokens"],
1505
+ model_calls=defaults["model_calls"],
1506
+ wall_clock_seconds=defaults["wall_clock_seconds"],
1507
+ shell_commands=defaults["shell_commands"],
1508
+ shell_command_seconds=defaults["shell_command_seconds"],
1509
+ writable_bytes=defaults["writable_bytes"],
1510
+ artifact_bytes=defaults["artifact_bytes"],
1511
+ )
1512
+
1513
+
1514
+ def default_eval_stage_context_policy(
1515
+ stage_id: EvalStageId,
1516
+ *,
1517
+ context_tier: EvalContextTier = EvalContextTier.COMPACT,
1518
+ ) -> EvalStageContextPolicy:
1519
+ """Return the default compact context policy for one 06A stage."""
1520
+ return EvalStageContextPolicy(
1521
+ stage_id=stage_id,
1522
+ context_tier=context_tier,
1523
+ allowed_capabilities=_EVAL_CAPABILITY_ENVELOPES[stage_id].capability_ids,
1524
+ allowed_paths=EVAL_CONTEXT_DEFAULT_ALLOWED_PATHS[stage_id],
1525
+ required_artifact_ids=EVAL_CONTEXT_DEFAULT_REQUIRED_ARTIFACT_IDS[stage_id],
1526
+ resource_ceiling=default_eval_stage_resource_ceiling(stage_id),
1527
+ )
1528
+
1529
+
1530
+ def default_eval_stage_context_policies() -> Mapping[
1531
+ EvalStageId, EvalStageContextPolicy
1532
+ ]:
1533
+ """Return immutable compact context policies keyed by 06A stage ID."""
1534
+ return MappingProxyType(
1535
+ {
1536
+ stage_id: default_eval_stage_context_policy(stage_id)
1537
+ for stage_id in EvalStageId
1538
+ }
1539
+ )
1540
+
1541
+
1542
+ def build_eval_context_snapshot(
1543
+ *,
1544
+ trial_id: str,
1545
+ policy: EvalStageContextPolicy,
1546
+ required_artifact_summaries: tuple[EvalContextArtifactSummary, ...],
1547
+ visible_acceptance_check_ids: tuple[str, ...],
1548
+ byte_budget: int | None = None,
1549
+ token_budget: int | None = None,
1550
+ ) -> EvalContextSnapshot:
1551
+ """Build a deterministic compact model-visible context snapshot."""
1552
+ graph = default_compact_eval_workflow_graph()
1553
+ stage_contract = next(
1554
+ stage for stage in graph.stages if stage.stage_id == policy.stage_id
1555
+ )
1556
+ snapshot = EvalContextSnapshot(
1557
+ trial_id=trial_id,
1558
+ stage_id=policy.stage_id,
1559
+ context_tier=policy.context_tier,
1560
+ allowed_capabilities=tuple(
1561
+ capability.value for capability in policy.allowed_capabilities
1562
+ ),
1563
+ allowed_paths=policy.allowed_paths,
1564
+ current_stage_contract=stage_contract.model_dump(mode="json"),
1565
+ required_artifact_summaries=required_artifact_summaries,
1566
+ visible_acceptance_check_ids=visible_acceptance_check_ids,
1567
+ redaction=policy.redaction,
1568
+ byte_budget=byte_budget or policy.resource_ceiling.artifact_bytes,
1569
+ token_budget=token_budget or policy.resource_ceiling.prompt_tokens,
1570
+ resource_ceiling=policy.resource_ceiling,
1571
+ fingerprint="0" * 64,
1572
+ )
1573
+ return snapshot.model_copy(
1574
+ update={"fingerprint": calculate_eval_context_fingerprint(snapshot)}
1575
+ )
1576
+
1577
+
1578
+ def compact_eval_boundary_baseline() -> EvalBoundaryBaseline:
1579
+ """Return the 06B module-shape baseline without rewriting 06A graph rules."""
1580
+ graph = default_compact_eval_workflow_graph()
1581
+ snapshot = compact_eval_workflow_snapshot(graph)
1582
+ return EvalBoundaryBaseline(
1583
+ graph_id=graph.graph_id,
1584
+ graph_sha256=snapshot["graph_sha256"],
1585
+ stage_ids=graph.stage_ids,
1586
+ terminal_results=tuple(EvalTerminalResult),
1587
+ outcome_kinds=tuple(EvalWorkflowOutcomeKind),
1588
+ candidate_dispositions=tuple(EvalCandidateDisposition),
1589
+ stage_artifacts=tuple(
1590
+ EvalBoundaryStageArtifacts(
1591
+ stage_id=stage.stage_id,
1592
+ input_artifact_ids=stage.input_artifact_ids,
1593
+ output_artifact_ids=stage.output_artifact_ids,
1594
+ )
1595
+ for stage in graph.stages
1596
+ ),
1597
+ )
1598
+
1599
+
1600
+ def compact_eval_boundary_baseline_snapshot() -> dict[str, Any]:
1601
+ """Return a deterministic JSON-compatible baseline snapshot."""
1602
+ return compact_eval_boundary_baseline().model_dump(mode="json")
1603
+
1604
+
1605
+ def default_eval_capability_envelopes() -> Mapping[EvalStageId, EvalCapabilityEnvelope]:
1606
+ """Return immutable compact eval capability envelopes keyed by 06A stage ID."""
1607
+ return _EVAL_CAPABILITY_ENVELOPES
1608
+
1609
+
1610
+ def eval_stage_capability_envelope(stage_id: EvalStageId) -> EvalCapabilityEnvelope:
1611
+ """Return the immutable capability envelope for one compact eval stage."""
1612
+ return _EVAL_CAPABILITY_ENVELOPES[stage_id]
1613
+
1614
+
1615
+ def validate_eval_stage_capability(
1616
+ stage_id: EvalStageId | str, capability: EvalCapabilityId | str
1617
+ ) -> EvalCapabilityValidationResult:
1618
+ """Validate one capability request against deterministic compact eval policy."""
1619
+ resolved_stage_id = _coerce_stage_id(stage_id)
1620
+ resolved_capability = _coerce_capability_id(capability)
1621
+ if not isinstance(resolved_stage_id, EvalStageId):
1622
+ return EvalCapabilityValidationResult(
1623
+ stage_id=resolved_stage_id,
1624
+ capability_id=resolved_capability,
1625
+ allowed=False,
1626
+ rule_id="eval.capability.unknown_stage",
1627
+ diagnostic_code="MF-EVAL-C001",
1628
+ diagnostic_summary="unknown compact eval stage id",
1629
+ )
1630
+ if not isinstance(resolved_capability, EvalCapabilityId):
1631
+ return EvalCapabilityValidationResult(
1632
+ stage_id=resolved_stage_id,
1633
+ capability_id=resolved_capability,
1634
+ allowed=False,
1635
+ rule_id="eval.capability.unknown_capability",
1636
+ diagnostic_code="MF-EVAL-C002",
1637
+ diagnostic_summary="unknown compact eval capability id",
1638
+ )
1639
+ envelope = _EVAL_CAPABILITY_ENVELOPES[resolved_stage_id]
1640
+ if resolved_capability in envelope.capability_ids:
1641
+ return EvalCapabilityValidationResult(
1642
+ stage_id=resolved_stage_id,
1643
+ capability_id=resolved_capability,
1644
+ allowed=True,
1645
+ rule_id="eval.capability.allowed",
1646
+ )
1647
+ if resolved_capability.value in EVAL_DENIED_CAPABILITY_IDS:
1648
+ return EvalCapabilityValidationResult(
1649
+ stage_id=resolved_stage_id,
1650
+ capability_id=resolved_capability,
1651
+ allowed=False,
1652
+ rule_id="eval.capability.denied_dangerous_all_stages",
1653
+ diagnostic_code="MF-EVAL-C003",
1654
+ diagnostic_summary="capability is denied for every compact eval stage",
1655
+ )
1656
+ return EvalCapabilityValidationResult(
1657
+ stage_id=resolved_stage_id,
1658
+ capability_id=resolved_capability,
1659
+ allowed=False,
1660
+ rule_id=f"eval.capability.denied.{resolved_stage_id.value}",
1661
+ diagnostic_code="MF-EVAL-C004",
1662
+ diagnostic_summary="capability is not in the stage envelope",
1663
+ )
1664
+
1665
+
1666
+ def validate_eval_stage_command(
1667
+ stage_id: EvalStageId | str,
1668
+ descriptor: EvalCommandDescriptor,
1669
+ *,
1670
+ builder_allowed_write_roots: tuple[str, ...] = EVAL_BUILDER_DEFAULT_WRITE_ROOTS,
1671
+ ) -> EvalCommandAdmissionResult:
1672
+ """Validate a deterministic command descriptor for one compact eval stage."""
1673
+ resolved_stage_id = _coerce_stage_id(stage_id)
1674
+ if not isinstance(resolved_stage_id, EvalStageId):
1675
+ return EvalCommandAdmissionResult(
1676
+ stage_id=resolved_stage_id,
1677
+ command_id=descriptor.command_id,
1678
+ allowed=False,
1679
+ rule_id="eval.command.unknown_stage",
1680
+ diagnostic_code="MF-EVAL-D001",
1681
+ diagnostic_summary="unknown compact eval stage id",
1682
+ )
1683
+ shell_result = validate_eval_stage_capability(
1684
+ resolved_stage_id, EvalCapabilityId.SHELL_RUN
1685
+ )
1686
+ if not shell_result.allowed:
1687
+ return EvalCommandAdmissionResult(
1688
+ stage_id=resolved_stage_id,
1689
+ command_id=descriptor.command_id,
1690
+ allowed=False,
1691
+ rule_id="eval.command.stage_has_no_shell",
1692
+ diagnostic_code="MF-EVAL-D002",
1693
+ diagnostic_summary="stage cannot run shell commands",
1694
+ )
1695
+ if disallowed_message := _disallowed_argv_message(descriptor.argv):
1696
+ return EvalCommandAdmissionResult(
1697
+ stage_id=resolved_stage_id,
1698
+ command_id=descriptor.command_id,
1699
+ allowed=False,
1700
+ rule_id="eval.command.descriptor_unsafe",
1701
+ diagnostic_code="MF-EVAL-D005",
1702
+ diagnostic_summary=disallowed_message,
1703
+ )
1704
+ if resolved_stage_id == EvalStageId.CHECKER:
1705
+ invalid_checker_writes = tuple(
1706
+ root
1707
+ for root in descriptor.admitted_write_roots
1708
+ if not any(
1709
+ _path_is_within_root(root, scratch)
1710
+ for scratch in EVAL_CHECKER_IGNORED_SCRATCH_ROOTS
1711
+ )
1712
+ )
1713
+ if invalid_checker_writes:
1714
+ return EvalCommandAdmissionResult(
1715
+ stage_id=resolved_stage_id,
1716
+ command_id=descriptor.command_id,
1717
+ allowed=False,
1718
+ rule_id="eval.command.checker_write_denied",
1719
+ diagnostic_code="MF-EVAL-D003",
1720
+ diagnostic_summary="checker commands may write only ignored scratch outputs",
1721
+ )
1722
+ if resolved_stage_id == EvalStageId.BUILDER:
1723
+ for root in builder_allowed_write_roots:
1724
+ _validate_relative_eval_path(root, allow_dot=False)
1725
+ invalid_builder_writes = tuple(
1726
+ root
1727
+ for root in descriptor.admitted_write_roots
1728
+ if not any(
1729
+ _path_is_within_root(root, allowed_root)
1730
+ for allowed_root in builder_allowed_write_roots
1731
+ )
1732
+ )
1733
+ if invalid_builder_writes:
1734
+ return EvalCommandAdmissionResult(
1735
+ stage_id=resolved_stage_id,
1736
+ command_id=descriptor.command_id,
1737
+ allowed=False,
1738
+ rule_id="eval.command.builder_write_root_denied",
1739
+ diagnostic_code="MF-EVAL-D004",
1740
+ diagnostic_summary="builder command writes outside allowed roots",
1741
+ )
1742
+ return EvalCommandAdmissionResult(
1743
+ stage_id=resolved_stage_id,
1744
+ command_id=descriptor.command_id,
1745
+ allowed=True,
1746
+ rule_id="eval.command.allowed",
1747
+ )
1748
+
1749
+
1750
+ def _closure_result(
1751
+ outcome_kind: EvalClosureOutcomeKind,
1752
+ *,
1753
+ terminal_result: EvalTerminalResult | None = None,
1754
+ candidate_disposition: EvalCandidateDisposition | None = None,
1755
+ evidence_artifact_ids: tuple[str, ...] = (),
1756
+ missing_artifact_ids: tuple[str, ...] = (),
1757
+ diagnostics: tuple[str, ...] = (),
1758
+ ) -> EvalClosureValidationResult:
1759
+ return EvalClosureValidationResult(
1760
+ valid=outcome_kind
1761
+ in {
1762
+ EvalClosureOutcomeKind.VALID_CLOSED_SUCCESS,
1763
+ EvalClosureOutcomeKind.VALID_CLOSED_REJECTION,
1764
+ EvalClosureOutcomeKind.VALID_BLOCKED_OUTCOME,
1765
+ },
1766
+ outcome_kind=outcome_kind,
1767
+ terminal_result=terminal_result,
1768
+ candidate_disposition=candidate_disposition,
1769
+ evidence_artifact_ids=evidence_artifact_ids,
1770
+ missing_artifact_ids=missing_artifact_ids,
1771
+ diagnostics=diagnostics,
1772
+ )
1773
+
1774
+
1775
+ def _artifact_bundle_items(
1776
+ artifact_bundle: Mapping[Any, Any],
1777
+ ) -> dict[str, Any]:
1778
+ artifacts: dict[str, Any] = {}
1779
+ for raw_artifact_id, record in artifact_bundle.items():
1780
+ artifact_id = getattr(raw_artifact_id, "value", raw_artifact_id)
1781
+ artifacts[str(artifact_id)] = record
1782
+ return artifacts
1783
+
1784
+
1785
+ def _artifact_record_mapping(record: Any) -> Mapping[str, Any]:
1786
+ if isinstance(record, BaseModel):
1787
+ return record.model_dump(mode="json")
1788
+ if isinstance(record, Mapping):
1789
+ return record
1790
+ raise TypeError("artifact records must be mappings or Pydantic models")
1791
+
1792
+
1793
+ def _fixture_manifest_from_artifact_record(record: Any) -> EvalFixtureManifest:
1794
+ from millforge.eval_artifacts import EvalFixtureManifestArtifact
1795
+
1796
+ if isinstance(record, EvalFixtureManifest):
1797
+ return record
1798
+ if isinstance(record, EvalFixtureManifestArtifact):
1799
+ return record.fixture_manifest
1800
+ artifact = EvalFixtureManifestArtifact.model_validate(
1801
+ _artifact_record_mapping(record)
1802
+ )
1803
+ return artifact.fixture_manifest
1804
+
1805
+
1806
+ def _validate_present_closure_artifacts(
1807
+ artifact_bundle: Mapping[Any, Any],
1808
+ ) -> tuple[dict[str, BaseModel], tuple[str, ...]]:
1809
+ from millforge.eval_artifacts import validate_eval_artifact_record
1810
+
1811
+ artifacts = _artifact_bundle_items(artifact_bundle)
1812
+ validated: dict[str, BaseModel] = {}
1813
+ diagnostics: list[str] = []
1814
+ for artifact_id, record in artifacts.items():
1815
+ try:
1816
+ validated[artifact_id] = validate_eval_artifact_record(
1817
+ artifact_id, _artifact_record_mapping(record)
1818
+ )
1819
+ except (TypeError, ValueError) as exc:
1820
+ diagnostics.append(f"{artifact_id}: {exc}")
1821
+ return validated, tuple(diagnostics)
1822
+
1823
+
1824
+ def _required_closure_artifact_ids(
1825
+ outcome_kind: EvalClosureOutcomeKind,
1826
+ ) -> tuple[str, ...]:
1827
+ from millforge.eval_artifacts import EVAL_LOGICAL_06A_ARTIFACT_IDS
1828
+
1829
+ base_artifact_ids = (
1830
+ "task",
1831
+ "fixture_manifest",
1832
+ "acceptance_checks",
1833
+ "plan",
1834
+ "arbiter_verdict",
1835
+ "context_snapshot",
1836
+ )
1837
+ if outcome_kind == EvalClosureOutcomeKind.VALID_BLOCKED_OUTCOME:
1838
+ return base_artifact_ids
1839
+ return EVAL_LOGICAL_06A_ARTIFACT_IDS + ("context_snapshot",)
1840
+
1841
+
1842
+ def _capability_snapshot_results(
1843
+ capability_snapshot: Any,
1844
+ ) -> tuple[EvalCapabilityValidationResult, ...]:
1845
+ if isinstance(capability_snapshot, Mapping) and "results" in capability_snapshot:
1846
+ return _capability_snapshot_results(capability_snapshot["results"])
1847
+ if isinstance(capability_snapshot, Mapping):
1848
+ results: list[EvalCapabilityValidationResult] = []
1849
+ for raw_stage_id, capabilities in capability_snapshot.items():
1850
+ stage_id = _coerce_stage_id(raw_stage_id)
1851
+ if isinstance(capabilities, EvalCapabilityEnvelope):
1852
+ capability_ids = capabilities.capability_ids
1853
+ elif isinstance(capabilities, Mapping) and "capability_ids" in capabilities:
1854
+ capability_ids = tuple(capabilities["capability_ids"])
1855
+ else:
1856
+ capability_ids = tuple(capabilities)
1857
+ for capability in capability_ids:
1858
+ results.append(validate_eval_stage_capability(stage_id, capability))
1859
+ return tuple(results)
1860
+ results = []
1861
+ for item in tuple(capability_snapshot):
1862
+ if isinstance(item, EvalCapabilityValidationResult):
1863
+ results.append(item)
1864
+ elif isinstance(item, EvalCapabilityEnvelope):
1865
+ results.extend(
1866
+ validate_eval_stage_capability(item.stage_id, capability)
1867
+ for capability in item.capability_ids
1868
+ )
1869
+ elif isinstance(item, Mapping) and "allowed" in item:
1870
+ results.append(EvalCapabilityValidationResult.model_validate(item))
1871
+ elif isinstance(item, Mapping) and "capability_ids" in item:
1872
+ stage_id = item["stage_id"]
1873
+ results.extend(
1874
+ validate_eval_stage_capability(stage_id, capability)
1875
+ for capability in item["capability_ids"]
1876
+ )
1877
+ else:
1878
+ raise TypeError("capability snapshot entries are not recognized")
1879
+ return tuple(results)
1880
+
1881
+
1882
+ def _capability_snapshot_diagnostics(capability_snapshot: Any) -> tuple[str, ...]:
1883
+ try:
1884
+ results = _capability_snapshot_results(capability_snapshot)
1885
+ except TypeError as exc:
1886
+ return (str(exc),)
1887
+ if not results:
1888
+ return ("capability snapshot must include admitted capability evidence",)
1889
+ result_stage_ids = {
1890
+ result.stage_id
1891
+ for result in results
1892
+ if isinstance(result.stage_id, EvalStageId)
1893
+ }
1894
+ missing_stage_ids = tuple(
1895
+ stage_id.value for stage_id in EvalStageId if stage_id not in result_stage_ids
1896
+ )
1897
+ diagnostics = [
1898
+ f"{result.stage_id}: {result.capability_id}: {result.rule_id}"
1899
+ for result in results
1900
+ if not result.allowed
1901
+ ]
1902
+ if missing_stage_ids:
1903
+ diagnostics.append(
1904
+ "capability snapshot missing stages: " + ", ".join(missing_stage_ids)
1905
+ )
1906
+ return tuple(diagnostics)
1907
+
1908
+
1909
+ def _fixture_snapshot_diagnostics(
1910
+ fixture_snapshot: EvalFixtureWorkspaceSnapshot | Mapping[str, Any],
1911
+ artifact_bundle: Mapping[Any, Any],
1912
+ ) -> tuple[str, ...]:
1913
+ snapshot = (
1914
+ fixture_snapshot
1915
+ if isinstance(fixture_snapshot, EvalFixtureWorkspaceSnapshot)
1916
+ else EvalFixtureWorkspaceSnapshot.model_validate(fixture_snapshot)
1917
+ )
1918
+ diagnostics: list[str] = []
1919
+ if snapshot.unauthorized_mutation_paths:
1920
+ diagnostics.append("fixture snapshot includes unauthorized mutations")
1921
+ if snapshot.violations:
1922
+ diagnostics.append("fixture snapshot includes path policy violations")
1923
+
1924
+ artifacts = _artifact_bundle_items(artifact_bundle)
1925
+ manifest_record = artifacts.get("fixture_manifest")
1926
+ if manifest_record is not None:
1927
+ manifest = _fixture_manifest_from_artifact_record(manifest_record)
1928
+ if snapshot.fixture_id != manifest.fixture_id:
1929
+ diagnostics.append("fixture snapshot fixture_id does not match manifest")
1930
+ if snapshot.fixture_manifest_sha256 != eval_fixture_manifest_sha256(manifest):
1931
+ diagnostics.append(
1932
+ "fixture snapshot manifest digest does not match expanded manifest"
1933
+ )
1934
+ return tuple(diagnostics)
1935
+
1936
+
1937
+ def _fixture_snapshot_mutation_paths(
1938
+ fixture_snapshot: EvalFixtureWorkspaceSnapshot | Mapping[str, Any],
1939
+ ) -> tuple[str, ...]:
1940
+ snapshot = (
1941
+ fixture_snapshot
1942
+ if isinstance(fixture_snapshot, EvalFixtureWorkspaceSnapshot)
1943
+ else EvalFixtureWorkspaceSnapshot.model_validate(fixture_snapshot)
1944
+ )
1945
+ return snapshot.added_paths + snapshot.modified_paths + snapshot.deleted_paths
1946
+
1947
+
1948
+ def _context_boundary_diagnostics(
1949
+ validated_artifacts: Mapping[str, BaseModel],
1950
+ ) -> tuple[str, ...]:
1951
+ acceptance = validated_artifacts.get("acceptance_checks")
1952
+ context = validated_artifacts.get("context_snapshot")
1953
+ if acceptance is None or context is None:
1954
+ return ("closure context validation requires acceptance and context artifacts",)
1955
+
1956
+ visible_checks = tuple(
1957
+ check.check_id for check in getattr(acceptance, "visible_acceptance_checks", ())
1958
+ )
1959
+ context_visible_checks = tuple(getattr(context, "visible_acceptance_check_ids", ()))
1960
+ if not visible_checks:
1961
+ return ("visible acceptance checks must be declared",)
1962
+ if set(context_visible_checks) != set(visible_checks):
1963
+ return (
1964
+ "context visible acceptance check IDs must match acceptance artifact IDs",
1965
+ )
1966
+ return ()
1967
+
1968
+
1969
+ def _closure_evidence_artifact_ids(
1970
+ validated_artifacts: Mapping[str, BaseModel],
1971
+ ) -> tuple[str, ...]:
1972
+ arbiter_verdict = validated_artifacts["arbiter_verdict"]
1973
+ references = getattr(arbiter_verdict, "closure_evidence_references", ())
1974
+ return tuple(reference.artifact_id.value for reference in references)
1975
+
1976
+
1977
+ def _closure_outcome_kind(
1978
+ validated_artifacts: Mapping[str, BaseModel],
1979
+ ) -> tuple[EvalClosureOutcomeKind, EvalTerminalResult, EvalCandidateDisposition]:
1980
+ from millforge.eval_artifacts import (
1981
+ EvalArbiterVerdictArtifact,
1982
+ EvalArbiterVerdictValue,
1983
+ )
1984
+
1985
+ arbiter_verdict = cast(
1986
+ EvalArbiterVerdictArtifact, validated_artifacts["arbiter_verdict"]
1987
+ )
1988
+ verdict = arbiter_verdict.verdict
1989
+ disposition = arbiter_verdict.candidate_disposition
1990
+ if verdict == EvalArbiterVerdictValue.BLOCKED:
1991
+ return (
1992
+ EvalClosureOutcomeKind.VALID_BLOCKED_OUTCOME,
1993
+ EvalTerminalResult.ARBITER_BLOCKED,
1994
+ EvalCandidateDisposition.BLOCKED,
1995
+ )
1996
+ if verdict == EvalArbiterVerdictValue.REJECTED:
1997
+ return (
1998
+ EvalClosureOutcomeKind.VALID_CLOSED_REJECTION,
1999
+ EvalTerminalResult.ARBITER_REJECTED,
2000
+ EvalCandidateDisposition.REJECTED,
2001
+ )
2002
+ if disposition == EvalCandidateDisposition.REJECTED:
2003
+ return (
2004
+ EvalClosureOutcomeKind.VALID_CLOSED_REJECTION,
2005
+ EvalTerminalResult.ARBITER_REJECTED,
2006
+ disposition,
2007
+ )
2008
+ return (
2009
+ EvalClosureOutcomeKind.VALID_CLOSED_SUCCESS,
2010
+ EvalTerminalResult.ARBITER_CLOSED,
2011
+ disposition,
2012
+ )
2013
+
2014
+
2015
+ def _terminal_path_diagnostics(
2016
+ validated_artifacts: Mapping[str, BaseModel],
2017
+ outcome_kind: EvalClosureOutcomeKind,
2018
+ ) -> tuple[str, ...]:
2019
+ from millforge.eval_artifacts import (
2020
+ EvalArbiterVerdictArtifact,
2021
+ EvalArbiterVerdictValue,
2022
+ )
2023
+
2024
+ arbiter_verdict_base = validated_artifacts.get("arbiter_verdict")
2025
+ if arbiter_verdict_base is None:
2026
+ return ("arbiter verdict is required to prove terminal path",)
2027
+ arbiter_verdict = cast(EvalArbiterVerdictArtifact, arbiter_verdict_base)
2028
+
2029
+ verdict = arbiter_verdict.verdict
2030
+ disposition = arbiter_verdict.candidate_disposition
2031
+ if verdict == EvalArbiterVerdictValue.BLOCKED and (
2032
+ outcome_kind != EvalClosureOutcomeKind.VALID_BLOCKED_OUTCOME
2033
+ or disposition != EvalCandidateDisposition.BLOCKED
2034
+ ):
2035
+ return ("blocked terminal path requires blocked candidate disposition",)
2036
+ if verdict != EvalArbiterVerdictValue.BLOCKED and (
2037
+ outcome_kind == EvalClosureOutcomeKind.VALID_BLOCKED_OUTCOME
2038
+ or disposition == EvalCandidateDisposition.BLOCKED
2039
+ ):
2040
+ return ("blocked candidate disposition requires blocked arbiter verdict",)
2041
+ return ()
2042
+
2043
+
2044
+ def _checker_gate_diagnostics(
2045
+ validated_artifacts: Mapping[str, BaseModel],
2046
+ outcome_kind: EvalClosureOutcomeKind,
2047
+ ) -> tuple[str, ...]:
2048
+ from millforge.eval_artifacts import (
2049
+ EvalCheckerVerdictArtifact,
2050
+ EvalCheckerVerdictValue,
2051
+ )
2052
+
2053
+ checker_base = validated_artifacts.get("checker_verdict")
2054
+ if (
2055
+ checker_base is None
2056
+ or outcome_kind == EvalClosureOutcomeKind.VALID_BLOCKED_OUTCOME
2057
+ ):
2058
+ return ()
2059
+ checker = cast(EvalCheckerVerdictArtifact, checker_base)
2060
+ if outcome_kind == EvalClosureOutcomeKind.VALID_CLOSED_SUCCESS:
2061
+ if checker.verdict != EvalCheckerVerdictValue.APPROVED:
2062
+ return ("arbiter closure requires approved checker verdict",)
2063
+ elif checker.verdict == EvalCheckerVerdictValue.BLOCKED:
2064
+ return ("arbiter rejection requires non-blocked checker verdict",)
2065
+ return ()
2066
+
2067
+
2068
+ def _test_results_failed(test_results: BaseModel) -> bool:
2069
+ return (
2070
+ getattr(test_results, "exit_code", 0) != 0
2071
+ or getattr(test_results, "failed_count", 0) != 0
2072
+ or not getattr(test_results, "deterministic", True)
2073
+ or not getattr(test_results, "allowed_by_policy", True)
2074
+ )
2075
+
2076
+
2077
+ def _failed_command_outcomes(patch_summary: BaseModel) -> tuple[Any, ...]:
2078
+ return tuple(
2079
+ outcome
2080
+ for outcome in getattr(patch_summary, "command_outcomes", ())
2081
+ if outcome.exit_code != 0
2082
+ )
2083
+
2084
+
2085
+ def _command_status_diagnostics(
2086
+ validated_artifacts: Mapping[str, BaseModel],
2087
+ outcome_kind: EvalClosureOutcomeKind,
2088
+ ) -> tuple[str, ...]:
2089
+ if outcome_kind != EvalClosureOutcomeKind.VALID_CLOSED_SUCCESS:
2090
+ return ()
2091
+
2092
+ diagnostics: list[str] = []
2093
+ test_results = validated_artifacts.get("test_results")
2094
+ if test_results is not None and _test_results_failed(test_results):
2095
+ diagnostics.append("arbiter closure cannot claim success with failed tests")
2096
+
2097
+ patch_summary = validated_artifacts.get("patch_summary")
2098
+ if patch_summary is not None:
2099
+ if _failed_command_outcomes(patch_summary):
2100
+ diagnostics.append(
2101
+ "arbiter closure cannot claim success with failed static checks"
2102
+ )
2103
+ if getattr(patch_summary, "unresolved_issues", ()):
2104
+ diagnostics.append(
2105
+ "arbiter closure cannot claim success with unresolved builder issues"
2106
+ )
2107
+ return tuple(diagnostics)
2108
+
2109
+
2110
+ def _mutation_evidence_diagnostics(
2111
+ validated_artifacts: Mapping[str, BaseModel],
2112
+ outcome_kind: EvalClosureOutcomeKind,
2113
+ ) -> tuple[str, ...]:
2114
+ if outcome_kind != EvalClosureOutcomeKind.VALID_CLOSED_SUCCESS:
2115
+ return ()
2116
+
2117
+ manifest_base = validated_artifacts.get("fixture_manifest")
2118
+ workspace_diff = validated_artifacts.get("workspace_diff")
2119
+ if manifest_base is None:
2120
+ return ()
2121
+ manifest = _fixture_manifest_from_artifact_record(manifest_base)
2122
+ if not manifest.expected_mutation_paths:
2123
+ return ()
2124
+ if workspace_diff is None:
2125
+ return ("required mutation paths require workspace_diff artifact",)
2126
+ changed_paths = (
2127
+ tuple(getattr(workspace_diff, "added_paths", ()))
2128
+ + tuple(getattr(workspace_diff, "modified_paths", ()))
2129
+ + tuple(getattr(workspace_diff, "deleted_paths", ()))
2130
+ )
2131
+ if not changed_paths:
2132
+ return ("required mutation paths require non-empty workspace_diff changes",)
2133
+ return ()
2134
+
2135
+
2136
+ def _arbiter_acceptance_diagnostics(
2137
+ validated_artifacts: Mapping[str, BaseModel],
2138
+ outcome_kind: EvalClosureOutcomeKind,
2139
+ ) -> tuple[str, ...]:
2140
+ if outcome_kind != EvalClosureOutcomeKind.VALID_CLOSED_SUCCESS:
2141
+ return ()
2142
+ arbiter_verdict = validated_artifacts.get("arbiter_verdict")
2143
+ if arbiter_verdict is None:
2144
+ return ()
2145
+ if getattr(arbiter_verdict, "open_acceptance_check_ids", ()):
2146
+ return ("arbiter closure cannot leave required acceptance checks open",)
2147
+ return ()
2148
+
2149
+
2150
+ def _arbiter_stage_result_diagnostics(
2151
+ validated_artifacts: Mapping[str, BaseModel],
2152
+ inferred_terminal_result: EvalTerminalResult,
2153
+ ) -> tuple[str, ...]:
2154
+ stage_result = validated_artifacts.get("stage_result")
2155
+ if stage_result is None:
2156
+ return ()
2157
+ if getattr(stage_result, "stage_id", None) != EvalStageId.ARBITER:
2158
+ return ("closure stage_result must belong to eval_arbiter",)
2159
+ terminal_result = getattr(stage_result, "terminal_result", None)
2160
+ if (
2161
+ terminal_result
2162
+ not in default_compact_eval_workflow_graph()
2163
+ .stage_contracts[EvalStageId.ARBITER]
2164
+ .legal_terminal_results
2165
+ ):
2166
+ return ("closure stage_result terminal is illegal for eval_arbiter",)
2167
+ if terminal_result != inferred_terminal_result:
2168
+ return ("closure stage_result terminal does not match arbiter verdict",)
2169
+ return ()
2170
+
2171
+
2172
+ def validate_eval_checker_approval(
2173
+ artifact_bundle: Mapping[Any, Any],
2174
+ ) -> EvalCheckerApprovalValidationResult:
2175
+ """Validate a Checker verdict against visible public execution evidence."""
2176
+ from millforge.eval_artifacts import (
2177
+ EvalCheckerVerdictArtifact,
2178
+ EvalCheckerVerdictValue,
2179
+ )
2180
+
2181
+ validated_artifacts, artifact_diagnostics = _validate_present_closure_artifacts(
2182
+ artifact_bundle
2183
+ )
2184
+ if artifact_diagnostics:
2185
+ return EvalCheckerApprovalValidationResult(
2186
+ valid=False,
2187
+ diagnostics=artifact_diagnostics,
2188
+ )
2189
+
2190
+ checker_base = validated_artifacts.get("checker_verdict")
2191
+ if checker_base is None:
2192
+ return EvalCheckerApprovalValidationResult(
2193
+ valid=False,
2194
+ missing_artifact_ids=("checker_verdict",),
2195
+ diagnostics=("checker approval validation requires checker_verdict",),
2196
+ )
2197
+
2198
+ checker = cast(EvalCheckerVerdictArtifact, checker_base)
2199
+ evidence_artifact_ids = tuple(
2200
+ reference.artifact_id.value for reference in checker.evidence_references
2201
+ )
2202
+ if checker.verdict != EvalCheckerVerdictValue.APPROVED:
2203
+ return EvalCheckerApprovalValidationResult(
2204
+ valid=True,
2205
+ evidence_artifact_ids=evidence_artifact_ids,
2206
+ )
2207
+
2208
+ diagnostics: list[str] = []
2209
+ test_results = validated_artifacts.get("test_results")
2210
+ if test_results is not None and _test_results_failed(test_results):
2211
+ diagnostics.append("checker approval cannot rely on failed tests")
2212
+
2213
+ patch_summary = validated_artifacts.get("patch_summary")
2214
+ if patch_summary is not None and _failed_command_outcomes(patch_summary):
2215
+ diagnostics.append("checker approval cannot rely on failed static checks")
2216
+
2217
+ if diagnostics:
2218
+ return EvalCheckerApprovalValidationResult(
2219
+ valid=False,
2220
+ evidence_artifact_ids=evidence_artifact_ids,
2221
+ diagnostics=tuple(diagnostics),
2222
+ )
2223
+
2224
+ return EvalCheckerApprovalValidationResult(
2225
+ valid=True,
2226
+ evidence_artifact_ids=evidence_artifact_ids,
2227
+ )
2228
+
2229
+
2230
+ def validate_eval_closure(
2231
+ artifact_bundle: Mapping[Any, Any],
2232
+ fixture_snapshot: EvalFixtureWorkspaceSnapshot | Mapping[str, Any],
2233
+ capability_snapshot: Any,
2234
+ ) -> EvalClosureValidationResult:
2235
+ """Validate compact eval closure evidence without mutating workspace state."""
2236
+ validated_artifacts, artifact_diagnostics = _validate_present_closure_artifacts(
2237
+ artifact_bundle
2238
+ )
2239
+ if artifact_diagnostics:
2240
+ return _closure_result(
2241
+ EvalClosureOutcomeKind.INVALID_ARTIFACT_BOUNDARY,
2242
+ diagnostics=artifact_diagnostics,
2243
+ )
2244
+
2245
+ base_required_artifact_ids = _required_closure_artifact_ids(
2246
+ EvalClosureOutcomeKind.VALID_BLOCKED_OUTCOME
2247
+ )
2248
+ base_missing_artifact_ids = tuple(
2249
+ artifact_id
2250
+ for artifact_id in base_required_artifact_ids
2251
+ if artifact_id not in validated_artifacts
2252
+ )
2253
+ if base_missing_artifact_ids:
2254
+ return _closure_result(
2255
+ EvalClosureOutcomeKind.INVALID_ARTIFACT_BOUNDARY,
2256
+ missing_artifact_ids=base_missing_artifact_ids,
2257
+ diagnostics=(
2258
+ "closure artifact bundle is missing required terminal-path artifacts",
2259
+ ),
2260
+ )
2261
+
2262
+ outcome_kind, terminal_result, candidate_disposition = _closure_outcome_kind(
2263
+ validated_artifacts
2264
+ )
2265
+ terminal_path_diagnostics = _terminal_path_diagnostics(
2266
+ validated_artifacts, outcome_kind
2267
+ )
2268
+ if terminal_path_diagnostics:
2269
+ return _closure_result(
2270
+ EvalClosureOutcomeKind.INVALID_ARTIFACT_BOUNDARY,
2271
+ diagnostics=terminal_path_diagnostics,
2272
+ )
2273
+
2274
+ gate_diagnostics = (
2275
+ _checker_gate_diagnostics(validated_artifacts, outcome_kind)
2276
+ + _command_status_diagnostics(validated_artifacts, outcome_kind)
2277
+ + _mutation_evidence_diagnostics(validated_artifacts, outcome_kind)
2278
+ + _arbiter_acceptance_diagnostics(validated_artifacts, outcome_kind)
2279
+ + _arbiter_stage_result_diagnostics(validated_artifacts, terminal_result)
2280
+ )
2281
+ if gate_diagnostics:
2282
+ return _closure_result(
2283
+ EvalClosureOutcomeKind.INVALID_ARTIFACT_BOUNDARY,
2284
+ diagnostics=gate_diagnostics,
2285
+ )
2286
+
2287
+ required_artifact_ids = _required_closure_artifact_ids(outcome_kind)
2288
+ missing_artifact_ids = tuple(
2289
+ artifact_id
2290
+ for artifact_id in required_artifact_ids
2291
+ if artifact_id not in validated_artifacts
2292
+ )
2293
+ if missing_artifact_ids:
2294
+ return _closure_result(
2295
+ EvalClosureOutcomeKind.INVALID_ARTIFACT_BOUNDARY,
2296
+ missing_artifact_ids=missing_artifact_ids,
2297
+ diagnostics=(
2298
+ "closure artifact bundle is missing required terminal-path artifacts",
2299
+ ),
2300
+ )
2301
+
2302
+ relaxed_blocked_artifact_ids = (
2303
+ "workspace_diff",
2304
+ "patch_summary",
2305
+ "test_results",
2306
+ "checker_verdict",
2307
+ )
2308
+ using_relaxed_blocked_artifacts = (
2309
+ outcome_kind == EvalClosureOutcomeKind.VALID_BLOCKED_OUTCOME
2310
+ and any(
2311
+ artifact_id not in validated_artifacts
2312
+ for artifact_id in relaxed_blocked_artifact_ids
2313
+ )
2314
+ )
2315
+ if using_relaxed_blocked_artifacts:
2316
+ try:
2317
+ mutation_paths = _fixture_snapshot_mutation_paths(fixture_snapshot)
2318
+ except (TypeError, ValueError) as exc:
2319
+ return _closure_result(
2320
+ EvalClosureOutcomeKind.INVALID_FIXTURE_BOUNDARY,
2321
+ diagnostics=(str(exc),),
2322
+ )
2323
+ if mutation_paths:
2324
+ return _closure_result(
2325
+ EvalClosureOutcomeKind.INVALID_ARTIFACT_BOUNDARY,
2326
+ diagnostics=(
2327
+ "shortened blocked terminal path requires unmodified fixture snapshot",
2328
+ ),
2329
+ )
2330
+
2331
+ capability_diagnostics = _capability_snapshot_diagnostics(capability_snapshot)
2332
+ if capability_diagnostics:
2333
+ return _closure_result(
2334
+ EvalClosureOutcomeKind.INVALID_CAPABILITY_BOUNDARY,
2335
+ diagnostics=capability_diagnostics,
2336
+ )
2337
+
2338
+ try:
2339
+ fixture_diagnostics = _fixture_snapshot_diagnostics(
2340
+ fixture_snapshot, artifact_bundle
2341
+ )
2342
+ except (TypeError, ValueError) as exc:
2343
+ fixture_diagnostics = (str(exc),)
2344
+ if fixture_diagnostics:
2345
+ return _closure_result(
2346
+ EvalClosureOutcomeKind.INVALID_FIXTURE_BOUNDARY,
2347
+ diagnostics=fixture_diagnostics,
2348
+ )
2349
+
2350
+ context_diagnostics = _context_boundary_diagnostics(validated_artifacts)
2351
+ if context_diagnostics:
2352
+ return _closure_result(
2353
+ EvalClosureOutcomeKind.INVALID_CONTEXT_BOUNDARY,
2354
+ diagnostics=context_diagnostics,
2355
+ )
2356
+
2357
+ evidence_artifact_ids = _closure_evidence_artifact_ids(validated_artifacts)
2358
+ if not evidence_artifact_ids or any(
2359
+ artifact_id not in validated_artifacts for artifact_id in evidence_artifact_ids
2360
+ ):
2361
+ return _closure_result(
2362
+ EvalClosureOutcomeKind.INVALID_ARTIFACT_BOUNDARY,
2363
+ diagnostics=(
2364
+ "arbiter closure evidence references must resolve inside artifact bundle",
2365
+ ),
2366
+ )
2367
+
2368
+ return _closure_result(
2369
+ outcome_kind,
2370
+ terminal_result=terminal_result,
2371
+ candidate_disposition=candidate_disposition,
2372
+ evidence_artifact_ids=evidence_artifact_ids,
2373
+ )
2374
+
2375
+
2376
+ __all__ = [
2377
+ "AUTHORITATIVE_COMPACT_EVAL_WORKFLOW_NAMES",
2378
+ "EVAL_BOUNDARY_ARTIFACT_MODULE_DECISION",
2379
+ "EVAL_BOUNDARY_ARTIFACT_MODULE_REQUIRED",
2380
+ "EVAL_BOUNDARY_MODULE_NAME",
2381
+ "EVAL_BUILDER_DEFAULT_WRITE_ROOTS",
2382
+ "EVAL_CHECKER_IGNORED_SCRATCH_ROOTS",
2383
+ "EVAL_CONTEXT_DEFAULT_ALLOWED_PATHS",
2384
+ "EVAL_CONTEXT_DEFAULT_REDACTION_CATEGORIES",
2385
+ "EVAL_CONTEXT_DEFAULT_REQUIRED_ARTIFACT_IDS",
2386
+ "EVAL_CONTEXT_FINGERPRINT_KIND",
2387
+ "EVAL_DENIED_CAPABILITY_IDS",
2388
+ "EVAL_FIXTURE_IGNORED_GENERATED_ROOTS",
2389
+ "EVAL_FIXTURE_IGNORED_GENERATED_SUFFIXES",
2390
+ "EVAL_STAGE_RESOURCE_CEILING_DEFAULTS",
2391
+ "EVAL_TRIAL_RESOURCE_CEILING_DEFAULTS",
2392
+ "EVAL_WORKSPACE_ISOLATION_CONTRACT",
2393
+ "EvalBoundaryBaseline",
2394
+ "EvalBoundaryStageArtifacts",
2395
+ "EvalCapabilityEnvelope",
2396
+ "EvalCapabilityId",
2397
+ "EvalCapabilityValidationResult",
2398
+ "EvalCommandAdmissionResult",
2399
+ "EvalCommandDescriptor",
2400
+ "EvalCommandEnvironmentPolicy",
2401
+ "EvalCheckerApprovalValidationResult",
2402
+ "EvalContextArtifactSummary",
2403
+ "EvalContextRedaction",
2404
+ "EvalContextSnapshot",
2405
+ "EvalContextTier",
2406
+ "EvalClosureOutcomeKind",
2407
+ "EvalClosureValidationResult",
2408
+ "EvalFixtureFile",
2409
+ "EvalFixtureManifest",
2410
+ "EvalFixtureWorkspacePolicy",
2411
+ "EvalFixtureWorkspaceSnapshot",
2412
+ "EvalPathPolicyViolation",
2413
+ "EvalResourceCeiling",
2414
+ "EvalStageContextPolicy",
2415
+ "build_eval_context_snapshot",
2416
+ "calculate_eval_context_fingerprint",
2417
+ "compact_eval_boundary_baseline",
2418
+ "compact_eval_boundary_baseline_snapshot",
2419
+ "default_eval_capability_envelopes",
2420
+ "default_eval_stage_context_policies",
2421
+ "default_eval_stage_context_policy",
2422
+ "default_eval_stage_resource_ceiling",
2423
+ "default_eval_trial_resource_ceiling",
2424
+ "eval_fixture_file_hash",
2425
+ "eval_fixture_file_record",
2426
+ "eval_fixture_manifest_from_paths",
2427
+ "eval_fixture_manifest_sha256",
2428
+ "eval_fixture_workspace_snapshot",
2429
+ "eval_stage_capability_envelope",
2430
+ "validate_eval_checker_approval",
2431
+ "validate_eval_fixture_path",
2432
+ "validate_eval_closure",
2433
+ "validate_eval_stage_capability",
2434
+ "validate_eval_stage_command",
2435
+ ]