millforge 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- millforge/__init__.py +1174 -0
- millforge/_forge/LICENSE +21 -0
- millforge/_forge/PROVENANCE.json +295 -0
- millforge/_forge/UPDATE_POLICY.md +24 -0
- millforge/_forge/__init__.py +14 -0
- millforge/_forge/adapter.py +2232 -0
- millforge/_forge/base_runner.py +121 -0
- millforge/_forge/clients/__init__.py +10 -0
- millforge/_forge/clients/base.py +200 -0
- millforge/_forge/context/__init__.py +23 -0
- millforge/_forge/context/manager.py +178 -0
- millforge/_forge/context/strategies.py +335 -0
- millforge/_forge/core/__init__.py +16 -0
- millforge/_forge/core/inference.py +433 -0
- millforge/_forge/core/messages.py +119 -0
- millforge/_forge/core/runner.py +479 -0
- millforge/_forge/core/steps.py +108 -0
- millforge/_forge/core/workflow.py +400 -0
- millforge/_forge/errors.py +222 -0
- millforge/_forge/guardrails/__init__.py +21 -0
- millforge/_forge/guardrails/error_tracker.py +71 -0
- millforge/_forge/guardrails/guardrails.py +194 -0
- millforge/_forge/guardrails/nudge.py +47 -0
- millforge/_forge/guardrails/response_validator.py +119 -0
- millforge/_forge/guardrails/step_enforcer.py +183 -0
- millforge/_forge/prompts/__init__.py +16 -0
- millforge/_forge/prompts/nudges.py +95 -0
- millforge/_forge/prompts/templates.py +285 -0
- millforge/_version.py +3 -0
- millforge/artifacts.py +570 -0
- millforge/base/__init__.py +97 -0
- millforge/base/composition.py +402 -0
- millforge/base/context.py +285 -0
- millforge/base/harness.py +138 -0
- millforge/base/identity.py +465 -0
- millforge/base/options.py +34 -0
- millforge/base/platform.py +17 -0
- millforge/base/prompt.py +317 -0
- millforge/base/runner.py +546 -0
- millforge/compiled_plan.py +970 -0
- millforge/compiler/__init__.py +231 -0
- millforge/compiler/artifact_validation.py +257 -0
- millforge/compiler/canonicalization.py +169 -0
- millforge/compiler/capabilities.py +66 -0
- millforge/compiler/catalogs.py +500 -0
- millforge/compiler/diagnostics.py +491 -0
- millforge/compiler/graph.py +678 -0
- millforge/compiler/lowering.py +198 -0
- millforge/compiler/output.py +692 -0
- millforge/compiler/parsing.py +1424 -0
- millforge/compiler/requests.py +1180 -0
- millforge/compiler/schema_validation.py +272 -0
- millforge/compiler/semantic.py +490 -0
- millforge/compiler/service.py +448 -0
- millforge/compiler/source.py +375 -0
- millforge/compiler/validators.py +184 -0
- millforge/connectors/__init__.py +95 -0
- millforge/connectors/admission.py +801 -0
- millforge/connectors/broker.py +202 -0
- millforge/connectors/contracts.py +1159 -0
- millforge/connectors/diagnostics.py +189 -0
- millforge/connectors/fake.py +66 -0
- millforge/connectors/runtime.py +236 -0
- millforge/contracts.py +2860 -0
- millforge/custom_tools/__init__.py +67 -0
- millforge/custom_tools/compiler.py +724 -0
- millforge/custom_tools/contracts.py +1093 -0
- millforge/custom_tools/diagnostics.py +205 -0
- millforge/eval_artifacts.py +952 -0
- millforge/eval_boundary.py +2435 -0
- millforge/eval_fixtures/__init__.py +1 -0
- millforge/eval_fixtures/default_pack/__init__.py +1 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.bug_diagnosis.traceback.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.direct_edit.import_sort.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.evidence_discipline.no_source_change.v1.json +51 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.false_closure.visible_green.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.multi_file.api_contract.v1.json +54 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.recovery.malformed_artifact.v1.json +54 -0
- millforge/eval_fixtures/default_pack/manifest.json +12 -0
- millforge/eval_modes.py +1282 -0
- millforge/eval_presets.py +1398 -0
- millforge/eval_reports.py +2517 -0
- millforge/eval_suite.py +2429 -0
- millforge/eval_trials.py +2632 -0
- millforge/eval_workflow.py +794 -0
- millforge/exceptions.py +122 -0
- millforge/model_backend.py +2098 -0
- millforge/protocols.py +340 -0
- millforge/py.typed +0 -0
- millforge/runtime.py +1791 -0
- millforge/testing/__init__.py +1089 -0
- millforge/tools/__init__.py +83 -0
- millforge/tools/builtin_runtime.py +1339 -0
- millforge/tools/builtins.py +773 -0
- millforge/tools/execution.py +1545 -0
- millforge/tools/path_policy.py +155 -0
- millforge/tools/pi_compat/PI_LICENSE +21 -0
- millforge/tools/pi_compat/PROVENANCE.json +55 -0
- millforge/tools/pi_compat/UPDATE_POLICY.md +36 -0
- millforge/tools/pi_compat/__init__.py +34 -0
- millforge/tools/pi_compat/contracts.py +49 -0
- millforge/tools/pi_compat/editing.py +390 -0
- millforge/tools/pi_compat/mutations.py +57 -0
- millforge/tools/pi_compat/operations.py +401 -0
- millforge/tools/pi_compat/paths.py +155 -0
- millforge/tools/pi_compat/process.py +1375 -0
- millforge/tools/pi_compat/search.py +738 -0
- millforge/tools/pi_compat/truncation.py +267 -0
- millforge/tools/pi_compat_catalog.py +396 -0
- millforge/tools/pi_compat_runtime.py +460 -0
- millforge/tools/registry.py +553 -0
- millforge/tools/results.py +533 -0
- millforge-0.1.0.dist-info/METADATA +844 -0
- millforge-0.1.0.dist-info/RECORD +116 -0
- millforge-0.1.0.dist-info/WHEEL +4 -0
- millforge-0.1.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,952 @@
|
|
|
1
|
+
"""Typed compact-eval artifact schemas and canonical layout metadata."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
from collections.abc import Mapping
|
|
8
|
+
from enum import Enum
|
|
9
|
+
import re
|
|
10
|
+
from types import MappingProxyType
|
|
11
|
+
from typing import Any, Literal
|
|
12
|
+
|
|
13
|
+
from pydantic import BaseModel, ConfigDict, Field, StrictBool, StrictInt, StrictStr
|
|
14
|
+
from pydantic import field_validator, model_validator
|
|
15
|
+
|
|
16
|
+
from millforge.eval_boundary import (
|
|
17
|
+
EVAL_CONTEXT_DEFAULT_REDACTION_CATEGORIES,
|
|
18
|
+
EvalContextRedaction,
|
|
19
|
+
EvalContextTier,
|
|
20
|
+
EvalFixtureManifest,
|
|
21
|
+
EvalResourceCeiling,
|
|
22
|
+
)
|
|
23
|
+
from millforge.eval_workflow import (
|
|
24
|
+
EvalCandidateDisposition,
|
|
25
|
+
EvalStageId,
|
|
26
|
+
EvalTerminalResult,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
EVAL_ARTIFACT_SCHEMA_VERSION = 1
|
|
30
|
+
EVAL_ARTIFACT_FIXED_TIMESTAMP = "1970-01-01T00:00:00Z"
|
|
31
|
+
EVAL_ARTIFACT_LAYOUT_ROOT = "trial"
|
|
32
|
+
EVAL_ARTIFACT_MEDIA_TYPE_JSON = "application/json"
|
|
33
|
+
EVAL_ARTIFACT_MEDIA_TYPE_JSONL = "application/x-ndjson"
|
|
34
|
+
EVAL_PUBLIC_ARTIFACT_IDS: tuple[str, ...] = (
|
|
35
|
+
"task",
|
|
36
|
+
"fixture_manifest",
|
|
37
|
+
"acceptance_checks",
|
|
38
|
+
"plan",
|
|
39
|
+
"workspace_diff",
|
|
40
|
+
"patch_summary",
|
|
41
|
+
"test_results",
|
|
42
|
+
"checker_verdict",
|
|
43
|
+
"arbiter_verdict",
|
|
44
|
+
"stage_result",
|
|
45
|
+
"event_log",
|
|
46
|
+
"resource_usage",
|
|
47
|
+
"model_usage",
|
|
48
|
+
"validator_result",
|
|
49
|
+
"context_snapshot",
|
|
50
|
+
"artifact_manifest",
|
|
51
|
+
)
|
|
52
|
+
EVAL_LOGICAL_06A_ARTIFACT_IDS: tuple[str, ...] = (
|
|
53
|
+
"task",
|
|
54
|
+
"fixture_manifest",
|
|
55
|
+
"acceptance_checks",
|
|
56
|
+
"plan",
|
|
57
|
+
"workspace_diff",
|
|
58
|
+
"patch_summary",
|
|
59
|
+
"test_results",
|
|
60
|
+
"checker_verdict",
|
|
61
|
+
"arbiter_verdict",
|
|
62
|
+
)
|
|
63
|
+
EVAL_RUNTIME_MEASUREMENT_ARTIFACT_IDS: tuple[str, ...] = (
|
|
64
|
+
"stage_result",
|
|
65
|
+
"event_log",
|
|
66
|
+
"resource_usage",
|
|
67
|
+
"model_usage",
|
|
68
|
+
"validator_result",
|
|
69
|
+
"context_snapshot",
|
|
70
|
+
"artifact_manifest",
|
|
71
|
+
)
|
|
72
|
+
_PUBLIC_LEAK_TOKENS: tuple[str, ...] = (
|
|
73
|
+
"F:\\",
|
|
74
|
+
"/mnt/f",
|
|
75
|
+
"millrace-agents",
|
|
76
|
+
"ideas/",
|
|
77
|
+
"ref-forge/",
|
|
78
|
+
"/home/",
|
|
79
|
+
"\\Users\\",
|
|
80
|
+
"API_KEY",
|
|
81
|
+
"DAEMON_STATE",
|
|
82
|
+
"daemon state",
|
|
83
|
+
"hidden check",
|
|
84
|
+
"hidden checks",
|
|
85
|
+
"hidden_check",
|
|
86
|
+
"hidden_checks",
|
|
87
|
+
"hidden score",
|
|
88
|
+
"hidden_score",
|
|
89
|
+
"hidden_scores",
|
|
90
|
+
"scoring rubric",
|
|
91
|
+
"scoring_rubric",
|
|
92
|
+
"expected output",
|
|
93
|
+
"expected_output",
|
|
94
|
+
"private runtime",
|
|
95
|
+
)
|
|
96
|
+
_WINDOWS_ABSOLUTE_PATH = re.compile(
|
|
97
|
+
r"(?:^|[\s\"'`([{<])(?:[A-Za-z]:[\\/]|\\\\)[^\s\"'`)\]}>,]*"
|
|
98
|
+
)
|
|
99
|
+
_POSIX_ABSOLUTE_PATH = re.compile(r"(?:^|[\s\"'`([{<])/(?!/)[^\s\"'`)\]}>,]*")
|
|
100
|
+
_USER_HOME_PATH = re.compile(r"(?:^|[\s\"'`([{<])~(?:/|\\)[^\s\"'`)\]}>,]*")
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
class EvalArtifactId(str, Enum):
|
|
104
|
+
"""Closed public compact-eval artifact IDs."""
|
|
105
|
+
|
|
106
|
+
TASK = "task"
|
|
107
|
+
FIXTURE_MANIFEST = "fixture_manifest"
|
|
108
|
+
ACCEPTANCE_CHECKS = "acceptance_checks"
|
|
109
|
+
PLAN = "plan"
|
|
110
|
+
WORKSPACE_DIFF = "workspace_diff"
|
|
111
|
+
PATCH_SUMMARY = "patch_summary"
|
|
112
|
+
TEST_RESULTS = "test_results"
|
|
113
|
+
CHECKER_VERDICT = "checker_verdict"
|
|
114
|
+
ARBITER_VERDICT = "arbiter_verdict"
|
|
115
|
+
STAGE_RESULT = "stage_result"
|
|
116
|
+
EVENT_LOG = "event_log"
|
|
117
|
+
RESOURCE_USAGE = "resource_usage"
|
|
118
|
+
MODEL_USAGE = "model_usage"
|
|
119
|
+
VALIDATOR_RESULT = "validator_result"
|
|
120
|
+
CONTEXT_SNAPSHOT = "context_snapshot"
|
|
121
|
+
ARTIFACT_MANIFEST = "artifact_manifest"
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
class EvalArtifactSection(str, Enum):
|
|
125
|
+
"""Canonical artifact layout sections under the trial root."""
|
|
126
|
+
|
|
127
|
+
INPUT = "trial/input"
|
|
128
|
+
PLANNING = "trial/planning"
|
|
129
|
+
EXECUTION = "trial/execution"
|
|
130
|
+
CHECKING = "trial/checking"
|
|
131
|
+
CLOSURE = "trial/closure"
|
|
132
|
+
RUNTIME = "trial/runtime"
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
class EvalCheckerVerdictValue(str, Enum):
|
|
136
|
+
"""Closed Checker verdict values."""
|
|
137
|
+
|
|
138
|
+
APPROVED = "approved"
|
|
139
|
+
REJECTED = "rejected"
|
|
140
|
+
BLOCKED = "blocked"
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
class EvalArbiterVerdictValue(str, Enum):
|
|
144
|
+
"""Closed Arbiter verdict values."""
|
|
145
|
+
|
|
146
|
+
CLOSED = "closed"
|
|
147
|
+
REJECTED = "rejected"
|
|
148
|
+
BLOCKED = "blocked"
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
class EvalArtifactLayoutEntry(BaseModel):
|
|
152
|
+
"""Path-free canonical layout metadata for one public artifact."""
|
|
153
|
+
|
|
154
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
155
|
+
|
|
156
|
+
artifact_id: EvalArtifactId
|
|
157
|
+
section: EvalArtifactSection
|
|
158
|
+
canonical_filename: StrictStr
|
|
159
|
+
media_type: StrictStr = EVAL_ARTIFACT_MEDIA_TYPE_JSON
|
|
160
|
+
schema_id: StrictStr
|
|
161
|
+
model_visible: StrictBool = True
|
|
162
|
+
|
|
163
|
+
@field_validator("canonical_filename")
|
|
164
|
+
@classmethod
|
|
165
|
+
def _filename_valid(cls, value: str) -> str:
|
|
166
|
+
if (
|
|
167
|
+
not value
|
|
168
|
+
or "/" in value
|
|
169
|
+
or "\\" in value
|
|
170
|
+
or value.startswith(".")
|
|
171
|
+
or ".." in value
|
|
172
|
+
):
|
|
173
|
+
raise ValueError("canonical_filename must be a stable relative filename")
|
|
174
|
+
return value
|
|
175
|
+
|
|
176
|
+
@property
|
|
177
|
+
def layout_path(self) -> str:
|
|
178
|
+
"""Return the canonical trial-relative layout path."""
|
|
179
|
+
return f"{self.section.value}/{self.canonical_filename}"
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
class EvalArtifactReference(BaseModel):
|
|
183
|
+
"""Reference to another public artifact without host paths."""
|
|
184
|
+
|
|
185
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
186
|
+
|
|
187
|
+
artifact_id: EvalArtifactId
|
|
188
|
+
summary: StrictStr
|
|
189
|
+
|
|
190
|
+
@field_validator("summary")
|
|
191
|
+
@classmethod
|
|
192
|
+
def _summary_nonblank(cls, value: str) -> str:
|
|
193
|
+
if not value.strip():
|
|
194
|
+
raise ValueError("artifact reference summaries must be non-empty")
|
|
195
|
+
_reject_public_material_leaks(value)
|
|
196
|
+
return value
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
class EvalArtifactBase(BaseModel):
|
|
200
|
+
"""Shared closed-world fields required on every compact-eval artifact."""
|
|
201
|
+
|
|
202
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
203
|
+
|
|
204
|
+
schema_version: StrictInt = EVAL_ARTIFACT_SCHEMA_VERSION
|
|
205
|
+
artifact_id: EvalArtifactId
|
|
206
|
+
trial_id: StrictStr
|
|
207
|
+
stage_id: EvalStageId | None = None
|
|
208
|
+
created_by: StrictStr
|
|
209
|
+
created_at: StrictStr = EVAL_ARTIFACT_FIXED_TIMESTAMP
|
|
210
|
+
references: tuple[EvalArtifactReference, ...] = Field(default_factory=tuple)
|
|
211
|
+
summary: StrictStr
|
|
212
|
+
|
|
213
|
+
@field_validator("trial_id", "created_by", "created_at", "summary")
|
|
214
|
+
@classmethod
|
|
215
|
+
def _stable_text_valid(cls, value: str) -> str:
|
|
216
|
+
if not value.strip():
|
|
217
|
+
raise ValueError("artifact text fields must be non-empty")
|
|
218
|
+
_reject_public_material_leaks(value)
|
|
219
|
+
return value
|
|
220
|
+
|
|
221
|
+
@model_validator(mode="after")
|
|
222
|
+
def _base_artifact_valid(self) -> EvalArtifactBase:
|
|
223
|
+
if self.schema_version != EVAL_ARTIFACT_SCHEMA_VERSION:
|
|
224
|
+
raise ValueError("unsupported eval artifact schema_version")
|
|
225
|
+
_reject_public_material_leaks(self.model_dump(mode="json"))
|
|
226
|
+
return self
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
class EvalTaskArtifact(EvalArtifactBase):
|
|
230
|
+
"""Model-visible task artifact."""
|
|
231
|
+
|
|
232
|
+
artifact_id: Literal[EvalArtifactId.TASK] = EvalArtifactId.TASK
|
|
233
|
+
stage_id: None = None
|
|
234
|
+
task_id: StrictStr
|
|
235
|
+
prompt: StrictStr
|
|
236
|
+
fixture_id: StrictStr
|
|
237
|
+
acceptance_criteria: tuple[StrictStr, ...]
|
|
238
|
+
command_hints: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
239
|
+
required_output_artifact_ids: tuple[EvalArtifactId, ...]
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
class EvalAcceptanceCheck(BaseModel):
|
|
243
|
+
"""One model-visible public acceptance check."""
|
|
244
|
+
|
|
245
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
246
|
+
|
|
247
|
+
check_id: StrictStr
|
|
248
|
+
check_kind: StrictStr
|
|
249
|
+
descriptor: StrictStr
|
|
250
|
+
expected_success: StrictStr
|
|
251
|
+
public_rationale: StrictStr
|
|
252
|
+
|
|
253
|
+
@model_validator(mode="after")
|
|
254
|
+
def _visible_check_valid(self) -> EvalAcceptanceCheck:
|
|
255
|
+
_reject_public_material_leaks(self.model_dump(mode="json"))
|
|
256
|
+
return self
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
class EvalAcceptanceChecksArtifact(EvalArtifactBase):
|
|
260
|
+
"""Artifact containing only model-visible acceptance checks."""
|
|
261
|
+
|
|
262
|
+
artifact_id: Literal[EvalArtifactId.ACCEPTANCE_CHECKS] = (
|
|
263
|
+
EvalArtifactId.ACCEPTANCE_CHECKS
|
|
264
|
+
)
|
|
265
|
+
stage_id: None = None
|
|
266
|
+
visible_acceptance_checks: tuple[EvalAcceptanceCheck, ...]
|
|
267
|
+
|
|
268
|
+
@field_validator("visible_acceptance_checks")
|
|
269
|
+
@classmethod
|
|
270
|
+
def _checks_nonempty(
|
|
271
|
+
cls, value: tuple[EvalAcceptanceCheck, ...]
|
|
272
|
+
) -> tuple[EvalAcceptanceCheck, ...]:
|
|
273
|
+
if not value:
|
|
274
|
+
raise ValueError("acceptance_checks must include visible checks")
|
|
275
|
+
return value
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
class EvalFixtureManifestArtifact(EvalArtifactBase):
|
|
279
|
+
"""Public artifact wrapper for an expanded fixture manifest payload."""
|
|
280
|
+
|
|
281
|
+
artifact_id: Literal[EvalArtifactId.FIXTURE_MANIFEST] = (
|
|
282
|
+
EvalArtifactId.FIXTURE_MANIFEST
|
|
283
|
+
)
|
|
284
|
+
stage_id: None = None
|
|
285
|
+
fixture_manifest: EvalFixtureManifest
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
class EvalPlanArtifact(EvalArtifactBase):
|
|
289
|
+
"""Planner output artifact."""
|
|
290
|
+
|
|
291
|
+
artifact_id: Literal[EvalArtifactId.PLAN] = EvalArtifactId.PLAN
|
|
292
|
+
stage_id: Literal[EvalStageId.PLANNER] = EvalStageId.PLANNER
|
|
293
|
+
implementation_steps: tuple[StrictStr, ...]
|
|
294
|
+
expected_files_to_inspect: tuple[StrictStr, ...]
|
|
295
|
+
expected_files_to_mutate: tuple[StrictStr, ...]
|
|
296
|
+
checks_to_run: tuple[StrictStr, ...]
|
|
297
|
+
risk_notes: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
298
|
+
no_hidden_checks_known: StrictBool
|
|
299
|
+
|
|
300
|
+
@model_validator(mode="after")
|
|
301
|
+
def _plan_valid(self) -> EvalPlanArtifact:
|
|
302
|
+
if not self.no_hidden_checks_known:
|
|
303
|
+
raise ValueError("plan must explicitly claim no hidden checks are known")
|
|
304
|
+
return self
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
class EvalFileHashChange(BaseModel):
|
|
308
|
+
"""Per-file before/after hash record for workspace diffs."""
|
|
309
|
+
|
|
310
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
311
|
+
|
|
312
|
+
path: StrictStr
|
|
313
|
+
before_sha256: StrictStr | None = None
|
|
314
|
+
after_sha256: StrictStr | None = None
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
class EvalWorkspaceDiffArtifact(EvalArtifactBase):
|
|
318
|
+
"""Builder workspace diff artifact."""
|
|
319
|
+
|
|
320
|
+
artifact_id: Literal[EvalArtifactId.WORKSPACE_DIFF] = EvalArtifactId.WORKSPACE_DIFF
|
|
321
|
+
stage_id: Literal[EvalStageId.BUILDER] = EvalStageId.BUILDER
|
|
322
|
+
added_paths: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
323
|
+
modified_paths: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
324
|
+
deleted_paths: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
325
|
+
ignored_generated_paths: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
326
|
+
file_hashes: tuple[EvalFileHashChange, ...] = Field(default_factory=tuple)
|
|
327
|
+
unauthorized_mutation_diagnostics: tuple[StrictStr, ...] = Field(
|
|
328
|
+
default_factory=tuple
|
|
329
|
+
)
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
class EvalCommandOutcome(BaseModel):
|
|
333
|
+
"""Public command outcome summary."""
|
|
334
|
+
|
|
335
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
336
|
+
|
|
337
|
+
command_id: StrictStr
|
|
338
|
+
exit_code: StrictInt
|
|
339
|
+
summary: StrictStr
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
class EvalPatchSummaryArtifact(EvalArtifactBase):
|
|
343
|
+
"""Builder patch summary artifact."""
|
|
344
|
+
|
|
345
|
+
artifact_id: Literal[EvalArtifactId.PATCH_SUMMARY] = EvalArtifactId.PATCH_SUMMARY
|
|
346
|
+
stage_id: Literal[EvalStageId.BUILDER] = EvalStageId.BUILDER
|
|
347
|
+
changed_files: tuple[StrictStr, ...]
|
|
348
|
+
behavior_summary: StrictStr
|
|
349
|
+
commands_run: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
350
|
+
command_outcomes: tuple[EvalCommandOutcome, ...] = Field(default_factory=tuple)
|
|
351
|
+
unresolved_issues: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
class EvalTestResultsArtifact(EvalArtifactBase):
|
|
355
|
+
"""Builder test results artifact."""
|
|
356
|
+
|
|
357
|
+
artifact_id: Literal[EvalArtifactId.TEST_RESULTS] = EvalArtifactId.TEST_RESULTS
|
|
358
|
+
stage_id: Literal[EvalStageId.BUILDER] = EvalStageId.BUILDER
|
|
359
|
+
command: tuple[StrictStr, ...]
|
|
360
|
+
exit_code: StrictInt
|
|
361
|
+
duration_seconds: StrictInt = Field(ge=0)
|
|
362
|
+
output_summary: StrictStr
|
|
363
|
+
passed_count: StrictInt = Field(ge=0)
|
|
364
|
+
failed_count: StrictInt = Field(ge=0)
|
|
365
|
+
skipped_count: StrictInt = Field(ge=0)
|
|
366
|
+
deterministic: StrictBool
|
|
367
|
+
allowed_by_policy: StrictBool
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
class EvalCheckerVerdictArtifact(EvalArtifactBase):
|
|
371
|
+
"""Checker verdict artifact."""
|
|
372
|
+
|
|
373
|
+
artifact_id: Literal[EvalArtifactId.CHECKER_VERDICT] = (
|
|
374
|
+
EvalArtifactId.CHECKER_VERDICT
|
|
375
|
+
)
|
|
376
|
+
stage_id: Literal[EvalStageId.CHECKER] = EvalStageId.CHECKER
|
|
377
|
+
verdict: EvalCheckerVerdictValue
|
|
378
|
+
evidence_references: tuple[EvalArtifactReference, ...]
|
|
379
|
+
failed_public_checks: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
380
|
+
unresolved_required_issues: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
381
|
+
requested_remediation_summary: StrictStr | None = None
|
|
382
|
+
blocker_summary: StrictStr | None = None
|
|
383
|
+
|
|
384
|
+
@model_validator(mode="after")
|
|
385
|
+
def _checker_verdict_valid(self) -> EvalCheckerVerdictArtifact:
|
|
386
|
+
if not self.evidence_references:
|
|
387
|
+
raise ValueError("checker verdicts must cite public evidence")
|
|
388
|
+
if self.verdict == EvalCheckerVerdictValue.APPROVED:
|
|
389
|
+
if self.failed_public_checks or self.unresolved_required_issues:
|
|
390
|
+
raise ValueError(
|
|
391
|
+
"approved checker verdicts must not include failed checks or "
|
|
392
|
+
"unresolved required issues"
|
|
393
|
+
)
|
|
394
|
+
if self.requested_remediation_summary or self.blocker_summary:
|
|
395
|
+
raise ValueError(
|
|
396
|
+
"approved checker verdicts must not request remediation or blockers"
|
|
397
|
+
)
|
|
398
|
+
elif self.verdict == EvalCheckerVerdictValue.REJECTED:
|
|
399
|
+
if not self.unresolved_required_issues:
|
|
400
|
+
raise ValueError(
|
|
401
|
+
"rejected checker verdicts must include unresolved required issues"
|
|
402
|
+
)
|
|
403
|
+
if not self.requested_remediation_summary:
|
|
404
|
+
raise ValueError(
|
|
405
|
+
"rejected checker verdicts must include remediation summary"
|
|
406
|
+
)
|
|
407
|
+
elif not self.blocker_summary:
|
|
408
|
+
raise ValueError("blocked checker verdicts must include blocker summary")
|
|
409
|
+
return self
|
|
410
|
+
|
|
411
|
+
|
|
412
|
+
class EvalArbiterVerdictArtifact(EvalArtifactBase):
|
|
413
|
+
"""Arbiter verdict artifact."""
|
|
414
|
+
|
|
415
|
+
artifact_id: Literal[EvalArtifactId.ARBITER_VERDICT] = (
|
|
416
|
+
EvalArtifactId.ARBITER_VERDICT
|
|
417
|
+
)
|
|
418
|
+
stage_id: Literal[EvalStageId.ARBITER] = EvalStageId.ARBITER
|
|
419
|
+
verdict: EvalArbiterVerdictValue
|
|
420
|
+
candidate_disposition: EvalCandidateDisposition
|
|
421
|
+
closure_evidence_references: tuple[EvalArtifactReference, ...]
|
|
422
|
+
missing_artifact_diagnostics: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
423
|
+
unauthorized_mutation_diagnostics: tuple[StrictStr, ...] = Field(
|
|
424
|
+
default_factory=tuple
|
|
425
|
+
)
|
|
426
|
+
public_acceptance_status: StrictStr
|
|
427
|
+
open_acceptance_check_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
428
|
+
|
|
429
|
+
@model_validator(mode="after")
|
|
430
|
+
def _arbiter_verdict_valid(self) -> EvalArbiterVerdictArtifact:
|
|
431
|
+
if not self.closure_evidence_references:
|
|
432
|
+
raise ValueError("arbiter verdicts must cite public closure evidence")
|
|
433
|
+
if self.verdict == EvalArbiterVerdictValue.CLOSED:
|
|
434
|
+
if self.candidate_disposition != EvalCandidateDisposition.APPROVED:
|
|
435
|
+
raise ValueError(
|
|
436
|
+
"closed arbiter verdicts require approved candidate disposition"
|
|
437
|
+
)
|
|
438
|
+
if (
|
|
439
|
+
self.missing_artifact_diagnostics
|
|
440
|
+
or self.unauthorized_mutation_diagnostics
|
|
441
|
+
or self.open_acceptance_check_ids
|
|
442
|
+
):
|
|
443
|
+
raise ValueError(
|
|
444
|
+
"closed arbiter verdicts must not include unresolved diagnostics"
|
|
445
|
+
)
|
|
446
|
+
elif self.verdict == EvalArbiterVerdictValue.REJECTED:
|
|
447
|
+
if self.candidate_disposition != EvalCandidateDisposition.REJECTED:
|
|
448
|
+
raise ValueError(
|
|
449
|
+
"rejected arbiter verdicts require rejected candidate disposition"
|
|
450
|
+
)
|
|
451
|
+
if not (
|
|
452
|
+
self.missing_artifact_diagnostics
|
|
453
|
+
or self.unauthorized_mutation_diagnostics
|
|
454
|
+
or self.open_acceptance_check_ids
|
|
455
|
+
or self.public_acceptance_status.strip()
|
|
456
|
+
):
|
|
457
|
+
raise ValueError("rejected arbiter verdicts must include rationale")
|
|
458
|
+
elif self.candidate_disposition != EvalCandidateDisposition.BLOCKED:
|
|
459
|
+
raise ValueError(
|
|
460
|
+
"blocked arbiter verdicts require blocked candidate disposition"
|
|
461
|
+
)
|
|
462
|
+
return self
|
|
463
|
+
|
|
464
|
+
|
|
465
|
+
class EvalStageResultArtifact(EvalArtifactBase):
|
|
466
|
+
"""Runtime stage result artifact."""
|
|
467
|
+
|
|
468
|
+
artifact_id: Literal[EvalArtifactId.STAGE_RESULT] = EvalArtifactId.STAGE_RESULT
|
|
469
|
+
stage_id: EvalStageId
|
|
470
|
+
terminal_result: EvalTerminalResult
|
|
471
|
+
attempt_count: StrictInt = Field(ge=0)
|
|
472
|
+
infrastructure_retry_count: StrictInt = Field(ge=0, le=1)
|
|
473
|
+
duration_seconds: StrictInt = Field(ge=0)
|
|
474
|
+
output_artifact_refs: tuple[EvalArtifactReference, ...] = Field(
|
|
475
|
+
default_factory=tuple
|
|
476
|
+
)
|
|
477
|
+
diagnostics: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
478
|
+
|
|
479
|
+
|
|
480
|
+
class EvalEventRecord(BaseModel):
|
|
481
|
+
"""Append-only structured public event record."""
|
|
482
|
+
|
|
483
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
484
|
+
|
|
485
|
+
event_id: StrictStr
|
|
486
|
+
stage_id: EvalStageId
|
|
487
|
+
event_type: StrictStr
|
|
488
|
+
summary: StrictStr
|
|
489
|
+
|
|
490
|
+
@model_validator(mode="after")
|
|
491
|
+
def _event_valid(self) -> EvalEventRecord:
|
|
492
|
+
_reject_public_material_leaks(self.model_dump(mode="json"))
|
|
493
|
+
return self
|
|
494
|
+
|
|
495
|
+
|
|
496
|
+
class EvalEventLogArtifact(EvalArtifactBase):
|
|
497
|
+
"""Public event log artifact."""
|
|
498
|
+
|
|
499
|
+
artifact_id: Literal[EvalArtifactId.EVENT_LOG] = EvalArtifactId.EVENT_LOG
|
|
500
|
+
records: tuple[EvalEventRecord, ...]
|
|
501
|
+
|
|
502
|
+
|
|
503
|
+
class EvalResourceUsageArtifact(EvalArtifactBase):
|
|
504
|
+
"""Runtime resource usage artifact."""
|
|
505
|
+
|
|
506
|
+
artifact_id: Literal[EvalArtifactId.RESOURCE_USAGE] = EvalArtifactId.RESOURCE_USAGE
|
|
507
|
+
wall_clock_seconds: StrictInt = Field(ge=0)
|
|
508
|
+
shell_command_count: StrictInt = Field(ge=0)
|
|
509
|
+
shell_command_seconds: StrictInt = Field(ge=0)
|
|
510
|
+
writable_bytes: StrictInt = Field(ge=0)
|
|
511
|
+
artifact_bytes: StrictInt = Field(ge=0)
|
|
512
|
+
retry_count: StrictInt = Field(ge=0)
|
|
513
|
+
|
|
514
|
+
|
|
515
|
+
class EvalModelUsageArtifact(EvalArtifactBase):
|
|
516
|
+
"""Runtime model usage artifact with deterministic placeholder support."""
|
|
517
|
+
|
|
518
|
+
artifact_id: Literal[EvalArtifactId.MODEL_USAGE] = EvalArtifactId.MODEL_USAGE
|
|
519
|
+
provider: StrictStr | None = None
|
|
520
|
+
model: StrictStr | None = None
|
|
521
|
+
prompt_tokens: StrictInt = Field(ge=0)
|
|
522
|
+
completion_tokens: StrictInt = Field(ge=0)
|
|
523
|
+
total_tokens: StrictInt = Field(ge=0)
|
|
524
|
+
reasoning_tokens: StrictInt | None = Field(default=None, ge=0)
|
|
525
|
+
cached_tokens: StrictInt | None = Field(default=None, ge=0)
|
|
526
|
+
estimated_cost_micros: StrictInt = Field(ge=0)
|
|
527
|
+
wall_clock_seconds: StrictInt = Field(ge=0)
|
|
528
|
+
retry_count: StrictInt = Field(ge=0)
|
|
529
|
+
|
|
530
|
+
@model_validator(mode="after")
|
|
531
|
+
def _tokens_valid(self) -> EvalModelUsageArtifact:
|
|
532
|
+
if self.total_tokens != self.prompt_tokens + self.completion_tokens:
|
|
533
|
+
raise ValueError(
|
|
534
|
+
"total_tokens must equal prompt_tokens plus completion_tokens"
|
|
535
|
+
)
|
|
536
|
+
return self
|
|
537
|
+
|
|
538
|
+
|
|
539
|
+
class EvalValidatorResultArtifact(EvalArtifactBase):
|
|
540
|
+
"""Model-visible validator result artifact."""
|
|
541
|
+
|
|
542
|
+
artifact_id: Literal[EvalArtifactId.VALIDATOR_RESULT] = (
|
|
543
|
+
EvalArtifactId.VALIDATOR_RESULT
|
|
544
|
+
)
|
|
545
|
+
visible_check_results: tuple[EvalCommandOutcome, ...]
|
|
546
|
+
public_diagnostics: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
547
|
+
model_visible: StrictBool = True
|
|
548
|
+
|
|
549
|
+
@model_validator(mode="after")
|
|
550
|
+
def _validator_result_valid(self) -> EvalValidatorResultArtifact:
|
|
551
|
+
if not self.model_visible:
|
|
552
|
+
raise ValueError("validator_result public artifact must be model-visible")
|
|
553
|
+
return self
|
|
554
|
+
|
|
555
|
+
|
|
556
|
+
class EvalValidatorVisibilityRecord(BaseModel):
|
|
557
|
+
"""Model-facing visibility boundary for validator inputs and outputs."""
|
|
558
|
+
|
|
559
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
560
|
+
|
|
561
|
+
visible_acceptance_check_ids: tuple[StrictStr, ...]
|
|
562
|
+
model_visible_artifact_ids: tuple[EvalArtifactId, ...] = (
|
|
563
|
+
EvalArtifactId.ACCEPTANCE_CHECKS,
|
|
564
|
+
EvalArtifactId.VALIDATOR_RESULT,
|
|
565
|
+
)
|
|
566
|
+
visible_validator_filename: StrictStr = "validator_result.visible.json"
|
|
567
|
+
scorer_only_opaque_check_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
568
|
+
scorer_only_definitions_excluded: StrictBool = True
|
|
569
|
+
scorer_only_expected_outputs_excluded: StrictBool = True
|
|
570
|
+
scorer_only_rubrics_excluded: StrictBool = True
|
|
571
|
+
scorer_only_final_scores_excluded: StrictBool = True
|
|
572
|
+
|
|
573
|
+
@model_validator(mode="after")
|
|
574
|
+
def _visibility_record_valid(self) -> EvalValidatorVisibilityRecord:
|
|
575
|
+
if not self.visible_acceptance_check_ids:
|
|
576
|
+
raise ValueError("visibility records require visible check ids")
|
|
577
|
+
if len(set(self.visible_acceptance_check_ids)) != len(
|
|
578
|
+
self.visible_acceptance_check_ids
|
|
579
|
+
):
|
|
580
|
+
raise ValueError("visible acceptance check ids must be unique")
|
|
581
|
+
if len(set(self.scorer_only_opaque_check_ids)) != len(
|
|
582
|
+
self.scorer_only_opaque_check_ids
|
|
583
|
+
):
|
|
584
|
+
raise ValueError("opaque scorer-only check ids must be unique")
|
|
585
|
+
if set(self.visible_acceptance_check_ids) & set(
|
|
586
|
+
self.scorer_only_opaque_check_ids
|
|
587
|
+
):
|
|
588
|
+
raise ValueError("visible and scorer-only check ids must not overlap")
|
|
589
|
+
if EvalArtifactId.VALIDATOR_RESULT not in self.model_visible_artifact_ids:
|
|
590
|
+
raise ValueError("validator_result.visible.json must be model-visible")
|
|
591
|
+
if EvalArtifactId.ACCEPTANCE_CHECKS not in self.model_visible_artifact_ids:
|
|
592
|
+
raise ValueError("visible acceptance checks must be model-visible")
|
|
593
|
+
if self.visible_validator_filename != "validator_result.visible.json":
|
|
594
|
+
raise ValueError("visible validator result filename is fixed")
|
|
595
|
+
if not all(
|
|
596
|
+
(
|
|
597
|
+
self.scorer_only_definitions_excluded,
|
|
598
|
+
self.scorer_only_expected_outputs_excluded,
|
|
599
|
+
self.scorer_only_rubrics_excluded,
|
|
600
|
+
self.scorer_only_final_scores_excluded,
|
|
601
|
+
)
|
|
602
|
+
):
|
|
603
|
+
raise ValueError("scorer-only material must be structurally excluded")
|
|
604
|
+
_reject_public_material_leaks(self.model_dump(mode="json"))
|
|
605
|
+
return self
|
|
606
|
+
|
|
607
|
+
|
|
608
|
+
class EvalContextSnapshotArtifact(EvalArtifactBase):
|
|
609
|
+
"""Path-free model-visible context snapshot metadata."""
|
|
610
|
+
|
|
611
|
+
artifact_id: Literal[EvalArtifactId.CONTEXT_SNAPSHOT] = (
|
|
612
|
+
EvalArtifactId.CONTEXT_SNAPSHOT
|
|
613
|
+
)
|
|
614
|
+
stage_id: EvalStageId
|
|
615
|
+
context_tier: EvalContextTier
|
|
616
|
+
allowed_capabilities: tuple[StrictStr, ...]
|
|
617
|
+
allowed_paths: tuple[StrictStr, ...]
|
|
618
|
+
current_stage_contract: Mapping[StrictStr, Any]
|
|
619
|
+
required_artifact_summaries: tuple[EvalArtifactReference, ...]
|
|
620
|
+
visible_acceptance_check_ids: tuple[StrictStr, ...]
|
|
621
|
+
redaction: EvalContextRedaction
|
|
622
|
+
redaction_summary: StrictStr
|
|
623
|
+
byte_budget: StrictInt = Field(gt=0)
|
|
624
|
+
token_budget: StrictInt = Field(gt=0)
|
|
625
|
+
resource_ceiling: EvalResourceCeiling
|
|
626
|
+
fingerprint: StrictStr
|
|
627
|
+
|
|
628
|
+
@model_validator(mode="after")
|
|
629
|
+
def _context_snapshot_valid(self) -> EvalContextSnapshotArtifact:
|
|
630
|
+
_validate_sha256(self.fingerprint)
|
|
631
|
+
if self.resource_ceiling.stage_id != self.stage_id:
|
|
632
|
+
raise ValueError("context snapshot resource ceiling must match stage")
|
|
633
|
+
_reject_public_material_leaks(self.model_dump(mode="json"))
|
|
634
|
+
return self
|
|
635
|
+
|
|
636
|
+
|
|
637
|
+
class EvalArtifactManifestEntry(BaseModel):
|
|
638
|
+
"""Deterministic public artifact manifest entry."""
|
|
639
|
+
|
|
640
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
641
|
+
|
|
642
|
+
artifact_id: EvalArtifactId
|
|
643
|
+
layout_path: StrictStr
|
|
644
|
+
media_type: StrictStr
|
|
645
|
+
schema_id: StrictStr
|
|
646
|
+
byte_size: StrictInt = Field(ge=0)
|
|
647
|
+
sha256: StrictStr
|
|
648
|
+
producer: StrictStr
|
|
649
|
+
model_visible: StrictBool = True
|
|
650
|
+
|
|
651
|
+
@model_validator(mode="after")
|
|
652
|
+
def _manifest_entry_valid(self) -> EvalArtifactManifestEntry:
|
|
653
|
+
layout = eval_artifact_layout_entry(self.artifact_id)
|
|
654
|
+
if self.layout_path != layout.layout_path:
|
|
655
|
+
raise ValueError("manifest entry layout_path must match canonical layout")
|
|
656
|
+
if self.media_type != layout.media_type or self.schema_id != layout.schema_id:
|
|
657
|
+
raise ValueError("manifest entry metadata must match canonical layout")
|
|
658
|
+
_validate_sha256(self.sha256)
|
|
659
|
+
_reject_public_material_leaks(self.model_dump(mode="json"))
|
|
660
|
+
return self
|
|
661
|
+
|
|
662
|
+
|
|
663
|
+
class EvalArtifactManifestArtifact(EvalArtifactBase):
|
|
664
|
+
"""Deterministic manifest of declared public artifacts."""
|
|
665
|
+
|
|
666
|
+
artifact_id: Literal[EvalArtifactId.ARTIFACT_MANIFEST] = (
|
|
667
|
+
EvalArtifactId.ARTIFACT_MANIFEST
|
|
668
|
+
)
|
|
669
|
+
entries: tuple[EvalArtifactManifestEntry, ...]
|
|
670
|
+
|
|
671
|
+
@model_validator(mode="after")
|
|
672
|
+
def _manifest_valid(self) -> EvalArtifactManifestArtifact:
|
|
673
|
+
artifact_ids = tuple(entry.artifact_id for entry in self.entries)
|
|
674
|
+
if EvalArtifactId.ARTIFACT_MANIFEST in artifact_ids:
|
|
675
|
+
raise ValueError("artifact_manifest must not reference itself")
|
|
676
|
+
if len(set(artifact_ids)) != len(artifact_ids):
|
|
677
|
+
raise ValueError("artifact_manifest entries must be unique")
|
|
678
|
+
object.__setattr__(
|
|
679
|
+
self,
|
|
680
|
+
"entries",
|
|
681
|
+
tuple(
|
|
682
|
+
sorted(
|
|
683
|
+
self.entries,
|
|
684
|
+
key=lambda entry: tuple(EvalArtifactId).index(entry.artifact_id),
|
|
685
|
+
)
|
|
686
|
+
),
|
|
687
|
+
)
|
|
688
|
+
return self
|
|
689
|
+
|
|
690
|
+
|
|
691
|
+
_EVAL_ARTIFACT_LAYOUT: Mapping[EvalArtifactId, EvalArtifactLayoutEntry] = (
|
|
692
|
+
MappingProxyType(
|
|
693
|
+
{
|
|
694
|
+
EvalArtifactId.TASK: EvalArtifactLayoutEntry(
|
|
695
|
+
artifact_id=EvalArtifactId.TASK,
|
|
696
|
+
section=EvalArtifactSection.INPUT,
|
|
697
|
+
canonical_filename="task.json",
|
|
698
|
+
schema_id="eval_task_artifact_v1",
|
|
699
|
+
),
|
|
700
|
+
EvalArtifactId.FIXTURE_MANIFEST: EvalArtifactLayoutEntry(
|
|
701
|
+
artifact_id=EvalArtifactId.FIXTURE_MANIFEST,
|
|
702
|
+
section=EvalArtifactSection.INPUT,
|
|
703
|
+
canonical_filename="fixture_manifest.json",
|
|
704
|
+
schema_id="eval_fixture_manifest_artifact_v1",
|
|
705
|
+
),
|
|
706
|
+
EvalArtifactId.ACCEPTANCE_CHECKS: EvalArtifactLayoutEntry(
|
|
707
|
+
artifact_id=EvalArtifactId.ACCEPTANCE_CHECKS,
|
|
708
|
+
section=EvalArtifactSection.INPUT,
|
|
709
|
+
canonical_filename="acceptance_checks.json",
|
|
710
|
+
schema_id="eval_acceptance_checks_artifact_v1",
|
|
711
|
+
),
|
|
712
|
+
EvalArtifactId.PLAN: EvalArtifactLayoutEntry(
|
|
713
|
+
artifact_id=EvalArtifactId.PLAN,
|
|
714
|
+
section=EvalArtifactSection.PLANNING,
|
|
715
|
+
canonical_filename="plan.json",
|
|
716
|
+
schema_id="eval_plan_artifact_v1",
|
|
717
|
+
),
|
|
718
|
+
EvalArtifactId.WORKSPACE_DIFF: EvalArtifactLayoutEntry(
|
|
719
|
+
artifact_id=EvalArtifactId.WORKSPACE_DIFF,
|
|
720
|
+
section=EvalArtifactSection.EXECUTION,
|
|
721
|
+
canonical_filename="workspace_diff.json",
|
|
722
|
+
schema_id="eval_workspace_diff_artifact_v1",
|
|
723
|
+
),
|
|
724
|
+
EvalArtifactId.PATCH_SUMMARY: EvalArtifactLayoutEntry(
|
|
725
|
+
artifact_id=EvalArtifactId.PATCH_SUMMARY,
|
|
726
|
+
section=EvalArtifactSection.EXECUTION,
|
|
727
|
+
canonical_filename="patch_summary.json",
|
|
728
|
+
schema_id="eval_patch_summary_artifact_v1",
|
|
729
|
+
),
|
|
730
|
+
EvalArtifactId.TEST_RESULTS: EvalArtifactLayoutEntry(
|
|
731
|
+
artifact_id=EvalArtifactId.TEST_RESULTS,
|
|
732
|
+
section=EvalArtifactSection.EXECUTION,
|
|
733
|
+
canonical_filename="test_results.json",
|
|
734
|
+
schema_id="eval_test_results_artifact_v1",
|
|
735
|
+
),
|
|
736
|
+
EvalArtifactId.CHECKER_VERDICT: EvalArtifactLayoutEntry(
|
|
737
|
+
artifact_id=EvalArtifactId.CHECKER_VERDICT,
|
|
738
|
+
section=EvalArtifactSection.CHECKING,
|
|
739
|
+
canonical_filename="checker_verdict.json",
|
|
740
|
+
schema_id="eval_checker_verdict_artifact_v1",
|
|
741
|
+
),
|
|
742
|
+
EvalArtifactId.ARBITER_VERDICT: EvalArtifactLayoutEntry(
|
|
743
|
+
artifact_id=EvalArtifactId.ARBITER_VERDICT,
|
|
744
|
+
section=EvalArtifactSection.CLOSURE,
|
|
745
|
+
canonical_filename="arbiter_verdict.json",
|
|
746
|
+
schema_id="eval_arbiter_verdict_artifact_v1",
|
|
747
|
+
),
|
|
748
|
+
EvalArtifactId.STAGE_RESULT: EvalArtifactLayoutEntry(
|
|
749
|
+
artifact_id=EvalArtifactId.STAGE_RESULT,
|
|
750
|
+
section=EvalArtifactSection.RUNTIME,
|
|
751
|
+
canonical_filename="stage_result.json",
|
|
752
|
+
schema_id="eval_stage_result_artifact_v1",
|
|
753
|
+
),
|
|
754
|
+
EvalArtifactId.EVENT_LOG: EvalArtifactLayoutEntry(
|
|
755
|
+
artifact_id=EvalArtifactId.EVENT_LOG,
|
|
756
|
+
section=EvalArtifactSection.RUNTIME,
|
|
757
|
+
canonical_filename="event_log.jsonl",
|
|
758
|
+
media_type=EVAL_ARTIFACT_MEDIA_TYPE_JSONL,
|
|
759
|
+
schema_id="eval_event_log_artifact_v1",
|
|
760
|
+
),
|
|
761
|
+
EvalArtifactId.RESOURCE_USAGE: EvalArtifactLayoutEntry(
|
|
762
|
+
artifact_id=EvalArtifactId.RESOURCE_USAGE,
|
|
763
|
+
section=EvalArtifactSection.RUNTIME,
|
|
764
|
+
canonical_filename="resource_usage.json",
|
|
765
|
+
schema_id="eval_resource_usage_artifact_v1",
|
|
766
|
+
),
|
|
767
|
+
EvalArtifactId.MODEL_USAGE: EvalArtifactLayoutEntry(
|
|
768
|
+
artifact_id=EvalArtifactId.MODEL_USAGE,
|
|
769
|
+
section=EvalArtifactSection.RUNTIME,
|
|
770
|
+
canonical_filename="model_usage.json",
|
|
771
|
+
schema_id="eval_model_usage_artifact_v1",
|
|
772
|
+
),
|
|
773
|
+
EvalArtifactId.VALIDATOR_RESULT: EvalArtifactLayoutEntry(
|
|
774
|
+
artifact_id=EvalArtifactId.VALIDATOR_RESULT,
|
|
775
|
+
section=EvalArtifactSection.RUNTIME,
|
|
776
|
+
canonical_filename="validator_result.visible.json",
|
|
777
|
+
schema_id="eval_validator_result_visible_artifact_v1",
|
|
778
|
+
),
|
|
779
|
+
EvalArtifactId.CONTEXT_SNAPSHOT: EvalArtifactLayoutEntry(
|
|
780
|
+
artifact_id=EvalArtifactId.CONTEXT_SNAPSHOT,
|
|
781
|
+
section=EvalArtifactSection.RUNTIME,
|
|
782
|
+
canonical_filename="context_snapshot.<stage>.json",
|
|
783
|
+
schema_id="eval_context_snapshot_artifact_v1",
|
|
784
|
+
),
|
|
785
|
+
EvalArtifactId.ARTIFACT_MANIFEST: EvalArtifactLayoutEntry(
|
|
786
|
+
artifact_id=EvalArtifactId.ARTIFACT_MANIFEST,
|
|
787
|
+
section=EvalArtifactSection.RUNTIME,
|
|
788
|
+
canonical_filename="artifact_manifest.json",
|
|
789
|
+
schema_id="eval_artifact_manifest_artifact_v1",
|
|
790
|
+
),
|
|
791
|
+
}
|
|
792
|
+
)
|
|
793
|
+
)
|
|
794
|
+
|
|
795
|
+
EVAL_ARTIFACT_SCHEMAS: Mapping[EvalArtifactId, type[BaseModel]] = MappingProxyType(
|
|
796
|
+
{
|
|
797
|
+
EvalArtifactId.TASK: EvalTaskArtifact,
|
|
798
|
+
EvalArtifactId.FIXTURE_MANIFEST: EvalFixtureManifestArtifact,
|
|
799
|
+
EvalArtifactId.ACCEPTANCE_CHECKS: EvalAcceptanceChecksArtifact,
|
|
800
|
+
EvalArtifactId.PLAN: EvalPlanArtifact,
|
|
801
|
+
EvalArtifactId.WORKSPACE_DIFF: EvalWorkspaceDiffArtifact,
|
|
802
|
+
EvalArtifactId.PATCH_SUMMARY: EvalPatchSummaryArtifact,
|
|
803
|
+
EvalArtifactId.TEST_RESULTS: EvalTestResultsArtifact,
|
|
804
|
+
EvalArtifactId.CHECKER_VERDICT: EvalCheckerVerdictArtifact,
|
|
805
|
+
EvalArtifactId.ARBITER_VERDICT: EvalArbiterVerdictArtifact,
|
|
806
|
+
EvalArtifactId.STAGE_RESULT: EvalStageResultArtifact,
|
|
807
|
+
EvalArtifactId.EVENT_LOG: EvalEventLogArtifact,
|
|
808
|
+
EvalArtifactId.RESOURCE_USAGE: EvalResourceUsageArtifact,
|
|
809
|
+
EvalArtifactId.MODEL_USAGE: EvalModelUsageArtifact,
|
|
810
|
+
EvalArtifactId.VALIDATOR_RESULT: EvalValidatorResultArtifact,
|
|
811
|
+
EvalArtifactId.CONTEXT_SNAPSHOT: EvalContextSnapshotArtifact,
|
|
812
|
+
EvalArtifactId.ARTIFACT_MANIFEST: EvalArtifactManifestArtifact,
|
|
813
|
+
}
|
|
814
|
+
)
|
|
815
|
+
|
|
816
|
+
|
|
817
|
+
def canonical_eval_artifact_layout() -> Mapping[
|
|
818
|
+
EvalArtifactId, EvalArtifactLayoutEntry
|
|
819
|
+
]:
|
|
820
|
+
"""Return immutable canonical layout metadata keyed by artifact ID."""
|
|
821
|
+
return _EVAL_ARTIFACT_LAYOUT
|
|
822
|
+
|
|
823
|
+
|
|
824
|
+
def eval_artifact_layout_entry(
|
|
825
|
+
artifact_id: EvalArtifactId | str,
|
|
826
|
+
) -> EvalArtifactLayoutEntry:
|
|
827
|
+
"""Return canonical layout metadata for one declared artifact ID."""
|
|
828
|
+
return _EVAL_ARTIFACT_LAYOUT[EvalArtifactId(artifact_id)]
|
|
829
|
+
|
|
830
|
+
|
|
831
|
+
def canonical_eval_artifact_manifest_bytes(
|
|
832
|
+
manifest: EvalArtifactManifestArtifact,
|
|
833
|
+
) -> bytes:
|
|
834
|
+
"""Return deterministic ASCII JSON bytes for an artifact manifest."""
|
|
835
|
+
return _canonical_json_bytes(manifest.model_dump(mode="json"))
|
|
836
|
+
|
|
837
|
+
|
|
838
|
+
def calculate_eval_artifact_manifest_sha256(
|
|
839
|
+
manifest: EvalArtifactManifestArtifact,
|
|
840
|
+
) -> str:
|
|
841
|
+
"""Return the deterministic SHA-256 for a canonical manifest."""
|
|
842
|
+
return hashlib.sha256(canonical_eval_artifact_manifest_bytes(manifest)).hexdigest()
|
|
843
|
+
|
|
844
|
+
|
|
845
|
+
def validate_eval_artifact_record(
|
|
846
|
+
artifact_id: EvalArtifactId | str, record: Mapping[str, Any]
|
|
847
|
+
) -> BaseModel:
|
|
848
|
+
"""Validate one artifact record against the closed public schema registry."""
|
|
849
|
+
resolved_artifact_id = EvalArtifactId(artifact_id)
|
|
850
|
+
schema = EVAL_ARTIFACT_SCHEMAS[resolved_artifact_id]
|
|
851
|
+
return schema.model_validate(record)
|
|
852
|
+
|
|
853
|
+
|
|
854
|
+
def _canonical_json_bytes(value: Any) -> bytes:
|
|
855
|
+
return (
|
|
856
|
+
json.dumps(
|
|
857
|
+
value,
|
|
858
|
+
sort_keys=True,
|
|
859
|
+
ensure_ascii=True,
|
|
860
|
+
allow_nan=False,
|
|
861
|
+
separators=(",", ":"),
|
|
862
|
+
).replace("\r\n", "\n")
|
|
863
|
+
+ "\n"
|
|
864
|
+
).encode("ascii")
|
|
865
|
+
|
|
866
|
+
|
|
867
|
+
def _validate_sha256(value: str) -> None:
|
|
868
|
+
if len(value) != 64 or any(
|
|
869
|
+
character not in "0123456789abcdef" for character in value
|
|
870
|
+
):
|
|
871
|
+
raise ValueError("sha256 must be a lowercase hexadecimal SHA-256 digest")
|
|
872
|
+
|
|
873
|
+
|
|
874
|
+
def _reject_public_material_leaks(value: Any) -> None:
|
|
875
|
+
for text in _public_material_text_values(value):
|
|
876
|
+
if text in EVAL_CONTEXT_DEFAULT_REDACTION_CATEGORIES:
|
|
877
|
+
continue
|
|
878
|
+
lowered = text.lower()
|
|
879
|
+
for token in _PUBLIC_LEAK_TOKENS:
|
|
880
|
+
if token.lower() in lowered:
|
|
881
|
+
raise ValueError(
|
|
882
|
+
"public eval artifacts must not expose private material"
|
|
883
|
+
)
|
|
884
|
+
if (
|
|
885
|
+
_WINDOWS_ABSOLUTE_PATH.search(text)
|
|
886
|
+
or _POSIX_ABSOLUTE_PATH.search(text)
|
|
887
|
+
or _USER_HOME_PATH.search(text)
|
|
888
|
+
):
|
|
889
|
+
raise ValueError("public eval artifacts must not expose host paths")
|
|
890
|
+
|
|
891
|
+
|
|
892
|
+
def _public_material_text_values(value: Any) -> tuple[str, ...]:
|
|
893
|
+
if isinstance(value, str):
|
|
894
|
+
return (value,)
|
|
895
|
+
if isinstance(value, Mapping):
|
|
896
|
+
return tuple(
|
|
897
|
+
text
|
|
898
|
+
for child in value.values()
|
|
899
|
+
for text in _public_material_text_values(child)
|
|
900
|
+
)
|
|
901
|
+
if isinstance(value, (tuple, list, set, frozenset)):
|
|
902
|
+
return tuple(
|
|
903
|
+
text for child in value for text in _public_material_text_values(child)
|
|
904
|
+
)
|
|
905
|
+
return ()
|
|
906
|
+
|
|
907
|
+
|
|
908
|
+
__all__ = [
|
|
909
|
+
"EVAL_ARTIFACT_FIXED_TIMESTAMP",
|
|
910
|
+
"EVAL_ARTIFACT_LAYOUT_ROOT",
|
|
911
|
+
"EVAL_ARTIFACT_MEDIA_TYPE_JSON",
|
|
912
|
+
"EVAL_ARTIFACT_MEDIA_TYPE_JSONL",
|
|
913
|
+
"EVAL_ARTIFACT_SCHEMA_VERSION",
|
|
914
|
+
"EVAL_ARTIFACT_SCHEMAS",
|
|
915
|
+
"EVAL_LOGICAL_06A_ARTIFACT_IDS",
|
|
916
|
+
"EVAL_PUBLIC_ARTIFACT_IDS",
|
|
917
|
+
"EVAL_RUNTIME_MEASUREMENT_ARTIFACT_IDS",
|
|
918
|
+
"EvalAcceptanceCheck",
|
|
919
|
+
"EvalAcceptanceChecksArtifact",
|
|
920
|
+
"EvalArbiterVerdictArtifact",
|
|
921
|
+
"EvalArbiterVerdictValue",
|
|
922
|
+
"EvalArtifactBase",
|
|
923
|
+
"EvalArtifactId",
|
|
924
|
+
"EvalArtifactLayoutEntry",
|
|
925
|
+
"EvalArtifactManifestArtifact",
|
|
926
|
+
"EvalArtifactManifestEntry",
|
|
927
|
+
"EvalArtifactReference",
|
|
928
|
+
"EvalArtifactSection",
|
|
929
|
+
"EvalCheckerVerdictArtifact",
|
|
930
|
+
"EvalCheckerVerdictValue",
|
|
931
|
+
"EvalCommandOutcome",
|
|
932
|
+
"EvalContextSnapshotArtifact",
|
|
933
|
+
"EvalEventLogArtifact",
|
|
934
|
+
"EvalEventRecord",
|
|
935
|
+
"EvalFileHashChange",
|
|
936
|
+
"EvalFixtureManifestArtifact",
|
|
937
|
+
"EvalModelUsageArtifact",
|
|
938
|
+
"EvalPatchSummaryArtifact",
|
|
939
|
+
"EvalPlanArtifact",
|
|
940
|
+
"EvalResourceUsageArtifact",
|
|
941
|
+
"EvalStageResultArtifact",
|
|
942
|
+
"EvalTaskArtifact",
|
|
943
|
+
"EvalTestResultsArtifact",
|
|
944
|
+
"EvalValidatorResultArtifact",
|
|
945
|
+
"EvalValidatorVisibilityRecord",
|
|
946
|
+
"EvalWorkspaceDiffArtifact",
|
|
947
|
+
"calculate_eval_artifact_manifest_sha256",
|
|
948
|
+
"canonical_eval_artifact_layout",
|
|
949
|
+
"canonical_eval_artifact_manifest_bytes",
|
|
950
|
+
"eval_artifact_layout_entry",
|
|
951
|
+
"validate_eval_artifact_record",
|
|
952
|
+
]
|