millforge 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- millforge/__init__.py +1174 -0
- millforge/_forge/LICENSE +21 -0
- millforge/_forge/PROVENANCE.json +295 -0
- millforge/_forge/UPDATE_POLICY.md +24 -0
- millforge/_forge/__init__.py +14 -0
- millforge/_forge/adapter.py +2232 -0
- millforge/_forge/base_runner.py +121 -0
- millforge/_forge/clients/__init__.py +10 -0
- millforge/_forge/clients/base.py +200 -0
- millforge/_forge/context/__init__.py +23 -0
- millforge/_forge/context/manager.py +178 -0
- millforge/_forge/context/strategies.py +335 -0
- millforge/_forge/core/__init__.py +16 -0
- millforge/_forge/core/inference.py +433 -0
- millforge/_forge/core/messages.py +119 -0
- millforge/_forge/core/runner.py +479 -0
- millforge/_forge/core/steps.py +108 -0
- millforge/_forge/core/workflow.py +400 -0
- millforge/_forge/errors.py +222 -0
- millforge/_forge/guardrails/__init__.py +21 -0
- millforge/_forge/guardrails/error_tracker.py +71 -0
- millforge/_forge/guardrails/guardrails.py +194 -0
- millforge/_forge/guardrails/nudge.py +47 -0
- millforge/_forge/guardrails/response_validator.py +119 -0
- millforge/_forge/guardrails/step_enforcer.py +183 -0
- millforge/_forge/prompts/__init__.py +16 -0
- millforge/_forge/prompts/nudges.py +95 -0
- millforge/_forge/prompts/templates.py +285 -0
- millforge/_version.py +3 -0
- millforge/artifacts.py +570 -0
- millforge/base/__init__.py +97 -0
- millforge/base/composition.py +402 -0
- millforge/base/context.py +285 -0
- millforge/base/harness.py +138 -0
- millforge/base/identity.py +465 -0
- millforge/base/options.py +34 -0
- millforge/base/platform.py +17 -0
- millforge/base/prompt.py +317 -0
- millforge/base/runner.py +546 -0
- millforge/compiled_plan.py +970 -0
- millforge/compiler/__init__.py +231 -0
- millforge/compiler/artifact_validation.py +257 -0
- millforge/compiler/canonicalization.py +169 -0
- millforge/compiler/capabilities.py +66 -0
- millforge/compiler/catalogs.py +500 -0
- millforge/compiler/diagnostics.py +491 -0
- millforge/compiler/graph.py +678 -0
- millforge/compiler/lowering.py +198 -0
- millforge/compiler/output.py +692 -0
- millforge/compiler/parsing.py +1424 -0
- millforge/compiler/requests.py +1180 -0
- millforge/compiler/schema_validation.py +272 -0
- millforge/compiler/semantic.py +490 -0
- millforge/compiler/service.py +448 -0
- millforge/compiler/source.py +375 -0
- millforge/compiler/validators.py +184 -0
- millforge/connectors/__init__.py +95 -0
- millforge/connectors/admission.py +801 -0
- millforge/connectors/broker.py +202 -0
- millforge/connectors/contracts.py +1159 -0
- millforge/connectors/diagnostics.py +189 -0
- millforge/connectors/fake.py +66 -0
- millforge/connectors/runtime.py +236 -0
- millforge/contracts.py +2860 -0
- millforge/custom_tools/__init__.py +67 -0
- millforge/custom_tools/compiler.py +724 -0
- millforge/custom_tools/contracts.py +1093 -0
- millforge/custom_tools/diagnostics.py +205 -0
- millforge/eval_artifacts.py +952 -0
- millforge/eval_boundary.py +2435 -0
- millforge/eval_fixtures/__init__.py +1 -0
- millforge/eval_fixtures/default_pack/__init__.py +1 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.bug_diagnosis.traceback.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.direct_edit.import_sort.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.evidence_discipline.no_source_change.v1.json +51 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.false_closure.visible_green.v1.json +52 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.multi_file.api_contract.v1.json +54 -0
- millforge/eval_fixtures/default_pack/fixtures/fixture.08a.recovery.malformed_artifact.v1.json +54 -0
- millforge/eval_fixtures/default_pack/manifest.json +12 -0
- millforge/eval_modes.py +1282 -0
- millforge/eval_presets.py +1398 -0
- millforge/eval_reports.py +2517 -0
- millforge/eval_suite.py +2429 -0
- millforge/eval_trials.py +2632 -0
- millforge/eval_workflow.py +794 -0
- millforge/exceptions.py +122 -0
- millforge/model_backend.py +2098 -0
- millforge/protocols.py +340 -0
- millforge/py.typed +0 -0
- millforge/runtime.py +1791 -0
- millforge/testing/__init__.py +1089 -0
- millforge/tools/__init__.py +83 -0
- millforge/tools/builtin_runtime.py +1339 -0
- millforge/tools/builtins.py +773 -0
- millforge/tools/execution.py +1545 -0
- millforge/tools/path_policy.py +155 -0
- millforge/tools/pi_compat/PI_LICENSE +21 -0
- millforge/tools/pi_compat/PROVENANCE.json +55 -0
- millforge/tools/pi_compat/UPDATE_POLICY.md +36 -0
- millforge/tools/pi_compat/__init__.py +34 -0
- millforge/tools/pi_compat/contracts.py +49 -0
- millforge/tools/pi_compat/editing.py +390 -0
- millforge/tools/pi_compat/mutations.py +57 -0
- millforge/tools/pi_compat/operations.py +401 -0
- millforge/tools/pi_compat/paths.py +155 -0
- millforge/tools/pi_compat/process.py +1375 -0
- millforge/tools/pi_compat/search.py +738 -0
- millforge/tools/pi_compat/truncation.py +267 -0
- millforge/tools/pi_compat_catalog.py +396 -0
- millforge/tools/pi_compat_runtime.py +460 -0
- millforge/tools/registry.py +553 -0
- millforge/tools/results.py +533 -0
- millforge-0.1.0.dist-info/METADATA +844 -0
- millforge-0.1.0.dist-info/RECORD +116 -0
- millforge-0.1.0.dist-info/WHEEL +4 -0
- millforge-0.1.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,794 @@
|
|
|
1
|
+
"""Compact public eval workflow contracts for Millforge."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
from collections.abc import Mapping
|
|
8
|
+
from enum import Enum
|
|
9
|
+
from types import MappingProxyType
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from pydantic import BaseModel, ConfigDict, Field, StrictBool, StrictInt, StrictStr
|
|
13
|
+
from pydantic import field_serializer, field_validator, model_validator
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class EvalStageId(str, Enum):
|
|
17
|
+
"""Closed stage IDs for the compact eval workflow."""
|
|
18
|
+
|
|
19
|
+
PLANNER = "eval_planner"
|
|
20
|
+
BUILDER = "eval_builder"
|
|
21
|
+
CHECKER = "eval_checker"
|
|
22
|
+
ARBITER = "eval_arbiter"
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class EvalTerminalResult(str, Enum):
|
|
26
|
+
"""Closed terminal results emitted by compact eval stages."""
|
|
27
|
+
|
|
28
|
+
PLAN_READY = "PLAN_READY"
|
|
29
|
+
PLAN_BLOCKED = "PLAN_BLOCKED"
|
|
30
|
+
BUILDER_COMPLETE = "BUILDER_COMPLETE"
|
|
31
|
+
BUILDER_BLOCKED = "BUILDER_BLOCKED"
|
|
32
|
+
CHECKER_APPROVED = "CHECKER_APPROVED"
|
|
33
|
+
CHECKER_REJECTED = "CHECKER_REJECTED"
|
|
34
|
+
CHECKER_BLOCKED = "CHECKER_BLOCKED"
|
|
35
|
+
ARBITER_CLOSED = "ARBITER_CLOSED"
|
|
36
|
+
ARBITER_REJECTED = "ARBITER_REJECTED"
|
|
37
|
+
ARBITER_BLOCKED = "ARBITER_BLOCKED"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class EvalWorkflowOutcomeKind(str, Enum):
|
|
41
|
+
"""Closed workflow transition outcome kinds."""
|
|
42
|
+
|
|
43
|
+
CONTINUE = "continue"
|
|
44
|
+
COMPLETED = "completed"
|
|
45
|
+
BLOCKED = "blocked"
|
|
46
|
+
INVALID = "invalid"
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class EvalCandidateDisposition(str, Enum):
|
|
50
|
+
"""Closed candidate dispositions carried between compact eval stages."""
|
|
51
|
+
|
|
52
|
+
NONE = "none"
|
|
53
|
+
APPROVED = "approved"
|
|
54
|
+
REJECTED = "rejected"
|
|
55
|
+
BLOCKED = "blocked"
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
OMITTED_COMPACT_EVAL_STAGE_IDS: tuple[str, ...] = (
|
|
59
|
+
"manager",
|
|
60
|
+
"fixer",
|
|
61
|
+
"doublechecker",
|
|
62
|
+
"troubleshooter",
|
|
63
|
+
"consultant",
|
|
64
|
+
"mechanic",
|
|
65
|
+
"auditor",
|
|
66
|
+
"updater",
|
|
67
|
+
"librarian",
|
|
68
|
+
"analyst",
|
|
69
|
+
"professor",
|
|
70
|
+
"curator",
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
class EvalStageContract(BaseModel):
|
|
75
|
+
"""Immutable contract for one compact eval workflow stage."""
|
|
76
|
+
|
|
77
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
78
|
+
|
|
79
|
+
stage_id: EvalStageId
|
|
80
|
+
role_summary: StrictStr
|
|
81
|
+
input_artifact_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
82
|
+
output_artifact_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
|
|
83
|
+
legal_terminal_results: tuple[EvalTerminalResult, ...]
|
|
84
|
+
domain_attempt_limit: StrictInt = Field(gt=0)
|
|
85
|
+
infrastructure_retry_limit: StrictInt = Field(ge=0, le=1)
|
|
86
|
+
may_complete_workflow: StrictBool
|
|
87
|
+
|
|
88
|
+
@field_validator("role_summary")
|
|
89
|
+
@classmethod
|
|
90
|
+
def _role_summary_nonblank(cls, value: str) -> str:
|
|
91
|
+
if not value.strip():
|
|
92
|
+
raise ValueError("role_summary must be a non-empty string")
|
|
93
|
+
return value
|
|
94
|
+
|
|
95
|
+
@field_validator("input_artifact_ids", "output_artifact_ids")
|
|
96
|
+
@classmethod
|
|
97
|
+
def _artifact_ids_valid(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
|
98
|
+
if len(set(value)) != len(value):
|
|
99
|
+
raise ValueError("artifact ids must be unique")
|
|
100
|
+
for artifact_id in value:
|
|
101
|
+
if not artifact_id.strip():
|
|
102
|
+
raise ValueError("artifact ids must be non-empty strings")
|
|
103
|
+
return value
|
|
104
|
+
|
|
105
|
+
@field_validator("legal_terminal_results")
|
|
106
|
+
@classmethod
|
|
107
|
+
def _legal_terminal_results_valid(
|
|
108
|
+
cls, value: tuple[EvalTerminalResult, ...]
|
|
109
|
+
) -> tuple[EvalTerminalResult, ...]:
|
|
110
|
+
if not value:
|
|
111
|
+
raise ValueError("legal_terminal_results must not be empty")
|
|
112
|
+
if len(set(value)) != len(value):
|
|
113
|
+
raise ValueError("legal_terminal_results values must be unique")
|
|
114
|
+
return value
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
class EvalAttemptState(BaseModel):
|
|
118
|
+
"""Immutable domain-attempt and infrastructure-retry counters."""
|
|
119
|
+
|
|
120
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
121
|
+
|
|
122
|
+
planner_attempts: StrictInt = Field(default=0, ge=0)
|
|
123
|
+
builder_attempts: StrictInt = Field(default=0, ge=0)
|
|
124
|
+
checker_attempts: StrictInt = Field(default=0, ge=0)
|
|
125
|
+
arbiter_attempts: StrictInt = Field(default=0, ge=0)
|
|
126
|
+
infrastructure_retries: Mapping[EvalStageId, StrictInt] = Field(
|
|
127
|
+
default_factory=dict
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
@model_validator(mode="after")
|
|
131
|
+
def _freeze_retries(self) -> EvalAttemptState:
|
|
132
|
+
ordered: dict[EvalStageId, int] = {}
|
|
133
|
+
for stage_id in EvalStageId:
|
|
134
|
+
retry_count = self.infrastructure_retries.get(stage_id, 0)
|
|
135
|
+
if retry_count < 0:
|
|
136
|
+
raise ValueError("infrastructure retry counts must be non-negative")
|
|
137
|
+
if retry_count > 1:
|
|
138
|
+
raise ValueError("infrastructure retry counts may not exceed 1")
|
|
139
|
+
if retry_count:
|
|
140
|
+
ordered[stage_id] = retry_count
|
|
141
|
+
object.__setattr__(self, "infrastructure_retries", MappingProxyType(ordered))
|
|
142
|
+
return self
|
|
143
|
+
|
|
144
|
+
@field_serializer("infrastructure_retries")
|
|
145
|
+
def _serialize_retries(self, value: Mapping[EvalStageId, int]) -> dict[str, int]:
|
|
146
|
+
return {
|
|
147
|
+
stage_id.value: value[stage_id]
|
|
148
|
+
for stage_id in EvalStageId
|
|
149
|
+
if stage_id in value
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
class EvalTransitionDecision(BaseModel):
|
|
154
|
+
"""Immutable transition decision for a compact eval workflow result."""
|
|
155
|
+
|
|
156
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
157
|
+
|
|
158
|
+
outcome_kind: EvalWorkflowOutcomeKind
|
|
159
|
+
current_stage_id: EvalStageId | StrictStr
|
|
160
|
+
terminal_result: EvalTerminalResult | StrictStr
|
|
161
|
+
next_stage_id: EvalStageId | None = None
|
|
162
|
+
candidate_disposition: EvalCandidateDisposition = EvalCandidateDisposition.NONE
|
|
163
|
+
diagnostic_code: StrictStr | None = None
|
|
164
|
+
diagnostic_summary: StrictStr | None = None
|
|
165
|
+
|
|
166
|
+
@model_validator(mode="after")
|
|
167
|
+
def _decision_shape_valid(self) -> EvalTransitionDecision:
|
|
168
|
+
if self.outcome_kind == EvalWorkflowOutcomeKind.CONTINUE:
|
|
169
|
+
if self.next_stage_id is None:
|
|
170
|
+
raise ValueError("continue decisions must declare next_stage_id")
|
|
171
|
+
if self.diagnostic_code is not None or self.diagnostic_summary is not None:
|
|
172
|
+
raise ValueError("continue decisions must not include diagnostics")
|
|
173
|
+
else:
|
|
174
|
+
if self.next_stage_id is not None:
|
|
175
|
+
raise ValueError("terminal decisions must not declare next_stage_id")
|
|
176
|
+
if self.outcome_kind == EvalWorkflowOutcomeKind.INVALID:
|
|
177
|
+
if self.diagnostic_code is None or self.diagnostic_summary is None:
|
|
178
|
+
raise ValueError("invalid decisions must include diagnostics")
|
|
179
|
+
if (self.diagnostic_code is None) != (self.diagnostic_summary is None):
|
|
180
|
+
raise ValueError(
|
|
181
|
+
"diagnostic_code and diagnostic_summary must appear together"
|
|
182
|
+
)
|
|
183
|
+
return self
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
class CompactEvalWorkflowGraph(BaseModel):
|
|
187
|
+
"""Immutable static compact eval workflow graph."""
|
|
188
|
+
|
|
189
|
+
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
190
|
+
|
|
191
|
+
graph_id: StrictStr = "compact_eval_workflow.v1"
|
|
192
|
+
stages: tuple[EvalStageContract, ...]
|
|
193
|
+
|
|
194
|
+
@model_validator(mode="before")
|
|
195
|
+
@classmethod
|
|
196
|
+
def _reject_mapping_stages(cls, data: Any) -> Any:
|
|
197
|
+
if not isinstance(data, Mapping):
|
|
198
|
+
return data
|
|
199
|
+
values = dict(data)
|
|
200
|
+
if isinstance(values.get("stages"), Mapping):
|
|
201
|
+
raise ValueError("stages must be an ordered tuple or list")
|
|
202
|
+
return values
|
|
203
|
+
|
|
204
|
+
@model_validator(mode="after")
|
|
205
|
+
def _graph_shape_valid(self) -> CompactEvalWorkflowGraph:
|
|
206
|
+
expected_stage_ids = tuple(stage_id for stage_id in EvalStageId)
|
|
207
|
+
actual_stage_ids = tuple(stage.stage_id for stage in self.stages)
|
|
208
|
+
if actual_stage_ids != expected_stage_ids:
|
|
209
|
+
raise ValueError(
|
|
210
|
+
"compact eval workflow stages must be exactly "
|
|
211
|
+
"eval_planner, eval_builder, eval_checker, eval_arbiter"
|
|
212
|
+
)
|
|
213
|
+
expected_results = {
|
|
214
|
+
EvalStageId.PLANNER: (
|
|
215
|
+
EvalTerminalResult.PLAN_READY,
|
|
216
|
+
EvalTerminalResult.PLAN_BLOCKED,
|
|
217
|
+
),
|
|
218
|
+
EvalStageId.BUILDER: (
|
|
219
|
+
EvalTerminalResult.BUILDER_COMPLETE,
|
|
220
|
+
EvalTerminalResult.BUILDER_BLOCKED,
|
|
221
|
+
),
|
|
222
|
+
EvalStageId.CHECKER: (
|
|
223
|
+
EvalTerminalResult.CHECKER_APPROVED,
|
|
224
|
+
EvalTerminalResult.CHECKER_REJECTED,
|
|
225
|
+
EvalTerminalResult.CHECKER_BLOCKED,
|
|
226
|
+
),
|
|
227
|
+
EvalStageId.ARBITER: (
|
|
228
|
+
EvalTerminalResult.ARBITER_CLOSED,
|
|
229
|
+
EvalTerminalResult.ARBITER_REJECTED,
|
|
230
|
+
EvalTerminalResult.ARBITER_BLOCKED,
|
|
231
|
+
),
|
|
232
|
+
}
|
|
233
|
+
expected_domain_limits = {
|
|
234
|
+
EvalStageId.PLANNER: 1,
|
|
235
|
+
EvalStageId.BUILDER: 2,
|
|
236
|
+
EvalStageId.CHECKER: 2,
|
|
237
|
+
EvalStageId.ARBITER: 1,
|
|
238
|
+
}
|
|
239
|
+
for stage in self.stages:
|
|
240
|
+
if stage.legal_terminal_results != expected_results[stage.stage_id]:
|
|
241
|
+
raise ValueError(
|
|
242
|
+
f"{stage.stage_id.value} legal_terminal_results are invalid"
|
|
243
|
+
)
|
|
244
|
+
if stage.domain_attempt_limit != expected_domain_limits[stage.stage_id]:
|
|
245
|
+
raise ValueError(
|
|
246
|
+
f"{stage.stage_id.value} domain_attempt_limit is invalid"
|
|
247
|
+
)
|
|
248
|
+
if stage.infrastructure_retry_limit > 1:
|
|
249
|
+
raise ValueError(
|
|
250
|
+
f"{stage.stage_id.value} infrastructure_retry_limit is invalid"
|
|
251
|
+
)
|
|
252
|
+
expected_completion = stage.stage_id == EvalStageId.ARBITER
|
|
253
|
+
if stage.may_complete_workflow is not expected_completion:
|
|
254
|
+
raise ValueError(
|
|
255
|
+
f"{stage.stage_id.value} completion permission is invalid"
|
|
256
|
+
)
|
|
257
|
+
return self
|
|
258
|
+
|
|
259
|
+
@property
|
|
260
|
+
def stage_ids(self) -> tuple[EvalStageId, ...]:
|
|
261
|
+
"""Return stage IDs in graph order."""
|
|
262
|
+
return tuple(stage.stage_id for stage in self.stages)
|
|
263
|
+
|
|
264
|
+
@property
|
|
265
|
+
def stage_contracts(self) -> Mapping[EvalStageId, EvalStageContract]:
|
|
266
|
+
"""Return stage contracts keyed by stage ID."""
|
|
267
|
+
return MappingProxyType({stage.stage_id: stage for stage in self.stages})
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def default_compact_eval_workflow_graph() -> CompactEvalWorkflowGraph:
|
|
271
|
+
"""Return the static compact eval workflow graph contract."""
|
|
272
|
+
return CompactEvalWorkflowGraph(
|
|
273
|
+
stages=(
|
|
274
|
+
EvalStageContract(
|
|
275
|
+
stage_id=EvalStageId.PLANNER,
|
|
276
|
+
role_summary="Plan a compact eval candidate workflow.",
|
|
277
|
+
input_artifact_ids=("task", "fixture_manifest", "acceptance_checks"),
|
|
278
|
+
output_artifact_ids=("plan",),
|
|
279
|
+
legal_terminal_results=(
|
|
280
|
+
EvalTerminalResult.PLAN_READY,
|
|
281
|
+
EvalTerminalResult.PLAN_BLOCKED,
|
|
282
|
+
),
|
|
283
|
+
domain_attempt_limit=1,
|
|
284
|
+
infrastructure_retry_limit=1,
|
|
285
|
+
may_complete_workflow=False,
|
|
286
|
+
),
|
|
287
|
+
EvalStageContract(
|
|
288
|
+
stage_id=EvalStageId.BUILDER,
|
|
289
|
+
role_summary="Build the candidate implementation under eval.",
|
|
290
|
+
input_artifact_ids=(
|
|
291
|
+
"task",
|
|
292
|
+
"fixture_manifest",
|
|
293
|
+
"plan",
|
|
294
|
+
"checker_verdict",
|
|
295
|
+
),
|
|
296
|
+
output_artifact_ids=(
|
|
297
|
+
"workspace_diff",
|
|
298
|
+
"patch_summary",
|
|
299
|
+
"test_results",
|
|
300
|
+
),
|
|
301
|
+
legal_terminal_results=(
|
|
302
|
+
EvalTerminalResult.BUILDER_COMPLETE,
|
|
303
|
+
EvalTerminalResult.BUILDER_BLOCKED,
|
|
304
|
+
),
|
|
305
|
+
domain_attempt_limit=2,
|
|
306
|
+
infrastructure_retry_limit=1,
|
|
307
|
+
may_complete_workflow=False,
|
|
308
|
+
),
|
|
309
|
+
EvalStageContract(
|
|
310
|
+
stage_id=EvalStageId.CHECKER,
|
|
311
|
+
role_summary="Check the candidate implementation against the eval plan.",
|
|
312
|
+
input_artifact_ids=(
|
|
313
|
+
"task",
|
|
314
|
+
"fixture_manifest",
|
|
315
|
+
"plan",
|
|
316
|
+
"workspace_diff",
|
|
317
|
+
"patch_summary",
|
|
318
|
+
"test_results",
|
|
319
|
+
),
|
|
320
|
+
output_artifact_ids=("checker_verdict",),
|
|
321
|
+
legal_terminal_results=(
|
|
322
|
+
EvalTerminalResult.CHECKER_APPROVED,
|
|
323
|
+
EvalTerminalResult.CHECKER_REJECTED,
|
|
324
|
+
EvalTerminalResult.CHECKER_BLOCKED,
|
|
325
|
+
),
|
|
326
|
+
domain_attempt_limit=2,
|
|
327
|
+
infrastructure_retry_limit=1,
|
|
328
|
+
may_complete_workflow=False,
|
|
329
|
+
),
|
|
330
|
+
EvalStageContract(
|
|
331
|
+
stage_id=EvalStageId.ARBITER,
|
|
332
|
+
role_summary="Settle the compact eval candidate outcome.",
|
|
333
|
+
input_artifact_ids=(
|
|
334
|
+
"task",
|
|
335
|
+
"fixture_manifest",
|
|
336
|
+
"plan",
|
|
337
|
+
"workspace_diff",
|
|
338
|
+
"patch_summary",
|
|
339
|
+
"test_results",
|
|
340
|
+
"checker_verdict",
|
|
341
|
+
),
|
|
342
|
+
output_artifact_ids=("arbiter_verdict",),
|
|
343
|
+
legal_terminal_results=(
|
|
344
|
+
EvalTerminalResult.ARBITER_CLOSED,
|
|
345
|
+
EvalTerminalResult.ARBITER_REJECTED,
|
|
346
|
+
EvalTerminalResult.ARBITER_BLOCKED,
|
|
347
|
+
),
|
|
348
|
+
domain_attempt_limit=1,
|
|
349
|
+
infrastructure_retry_limit=1,
|
|
350
|
+
may_complete_workflow=True,
|
|
351
|
+
),
|
|
352
|
+
)
|
|
353
|
+
)
|
|
354
|
+
|
|
355
|
+
|
|
356
|
+
def _canonical_json_serialize(obj: Any) -> str:
|
|
357
|
+
return (
|
|
358
|
+
json.dumps(
|
|
359
|
+
obj,
|
|
360
|
+
sort_keys=True,
|
|
361
|
+
ensure_ascii=True,
|
|
362
|
+
allow_nan=False,
|
|
363
|
+
separators=(",", ":"),
|
|
364
|
+
).replace("\r\n", "\n")
|
|
365
|
+
+ "\n"
|
|
366
|
+
)
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
def _diagnostic_decision(
|
|
370
|
+
*,
|
|
371
|
+
code: str,
|
|
372
|
+
summary: str,
|
|
373
|
+
current_stage_id: EvalStageId | str,
|
|
374
|
+
terminal_result: EvalTerminalResult | str,
|
|
375
|
+
) -> EvalTransitionDecision:
|
|
376
|
+
return EvalTransitionDecision(
|
|
377
|
+
outcome_kind=EvalWorkflowOutcomeKind.INVALID,
|
|
378
|
+
current_stage_id=current_stage_id,
|
|
379
|
+
terminal_result=terminal_result,
|
|
380
|
+
diagnostic_code=code,
|
|
381
|
+
diagnostic_summary=summary,
|
|
382
|
+
)
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
def _coerce_stage_id(value: EvalStageId | str) -> EvalStageId | None:
|
|
386
|
+
if isinstance(value, EvalStageId):
|
|
387
|
+
return value
|
|
388
|
+
try:
|
|
389
|
+
return EvalStageId(value)
|
|
390
|
+
except ValueError:
|
|
391
|
+
return None
|
|
392
|
+
|
|
393
|
+
|
|
394
|
+
def _coerce_terminal_result(
|
|
395
|
+
value: EvalTerminalResult | str,
|
|
396
|
+
) -> EvalTerminalResult | None:
|
|
397
|
+
if isinstance(value, EvalTerminalResult):
|
|
398
|
+
return value
|
|
399
|
+
try:
|
|
400
|
+
return EvalTerminalResult(value)
|
|
401
|
+
except ValueError:
|
|
402
|
+
return None
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
def _raw_attempt_mapping(
|
|
406
|
+
attempt_state: EvalAttemptState | Mapping[str, Any],
|
|
407
|
+
) -> Mapping[str, Any]:
|
|
408
|
+
if isinstance(attempt_state, EvalAttemptState):
|
|
409
|
+
return attempt_state.model_dump(mode="json")
|
|
410
|
+
return attempt_state
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
def _attempt_state_diagnostic(
|
|
414
|
+
*,
|
|
415
|
+
current_stage_id: EvalStageId,
|
|
416
|
+
terminal_result: EvalTerminalResult,
|
|
417
|
+
attempt_state: EvalAttemptState | Mapping[str, Any],
|
|
418
|
+
graph: CompactEvalWorkflowGraph,
|
|
419
|
+
) -> EvalTransitionDecision | None:
|
|
420
|
+
raw_attempts = _raw_attempt_mapping(attempt_state)
|
|
421
|
+
raw_retries = raw_attempts.get("infrastructure_retries", {})
|
|
422
|
+
if not isinstance(raw_retries, Mapping):
|
|
423
|
+
return _diagnostic_decision(
|
|
424
|
+
code="MF-EVAL-G004",
|
|
425
|
+
summary="infrastructure retry state is not a mapping",
|
|
426
|
+
current_stage_id=current_stage_id,
|
|
427
|
+
terminal_result=terminal_result,
|
|
428
|
+
)
|
|
429
|
+
for raw_stage_id, raw_retry_count in raw_retries.items():
|
|
430
|
+
stage_id = _coerce_stage_id(raw_stage_id)
|
|
431
|
+
if stage_id is None:
|
|
432
|
+
return _diagnostic_decision(
|
|
433
|
+
code="MF-EVAL-G004",
|
|
434
|
+
summary="infrastructure retry state contains an unknown stage",
|
|
435
|
+
current_stage_id=current_stage_id,
|
|
436
|
+
terminal_result=terminal_result,
|
|
437
|
+
)
|
|
438
|
+
if not isinstance(raw_retry_count, int) or isinstance(raw_retry_count, bool):
|
|
439
|
+
return _diagnostic_decision(
|
|
440
|
+
code="MF-EVAL-G004",
|
|
441
|
+
summary="infrastructure retry counts must be integers",
|
|
442
|
+
current_stage_id=current_stage_id,
|
|
443
|
+
terminal_result=terminal_result,
|
|
444
|
+
)
|
|
445
|
+
if raw_retry_count < 0 or raw_retry_count > 1:
|
|
446
|
+
return _diagnostic_decision(
|
|
447
|
+
code="MF-EVAL-G004",
|
|
448
|
+
summary="infrastructure retry count exceeds the compact graph limit",
|
|
449
|
+
current_stage_id=current_stage_id,
|
|
450
|
+
terminal_result=terminal_result,
|
|
451
|
+
)
|
|
452
|
+
|
|
453
|
+
limits = {stage.stage_id: stage.domain_attempt_limit for stage in graph.stages}
|
|
454
|
+
count_fields = {
|
|
455
|
+
EvalStageId.PLANNER: "planner_attempts",
|
|
456
|
+
EvalStageId.BUILDER: "builder_attempts",
|
|
457
|
+
EvalStageId.CHECKER: "checker_attempts",
|
|
458
|
+
EvalStageId.ARBITER: "arbiter_attempts",
|
|
459
|
+
}
|
|
460
|
+
counts: dict[EvalStageId, int] = {}
|
|
461
|
+
for stage_id, field_name in count_fields.items():
|
|
462
|
+
raw_count = raw_attempts.get(field_name, 0)
|
|
463
|
+
if not isinstance(raw_count, int) or isinstance(raw_count, bool):
|
|
464
|
+
return _diagnostic_decision(
|
|
465
|
+
code="MF-EVAL-G003",
|
|
466
|
+
summary="domain attempt counts must be integers",
|
|
467
|
+
current_stage_id=current_stage_id,
|
|
468
|
+
terminal_result=terminal_result,
|
|
469
|
+
)
|
|
470
|
+
if raw_count < 0 or raw_count > limits[stage_id]:
|
|
471
|
+
return _diagnostic_decision(
|
|
472
|
+
code="MF-EVAL-G003",
|
|
473
|
+
summary="domain attempt count exceeds the compact graph limit",
|
|
474
|
+
current_stage_id=current_stage_id,
|
|
475
|
+
terminal_result=terminal_result,
|
|
476
|
+
)
|
|
477
|
+
counts[stage_id] = raw_count
|
|
478
|
+
|
|
479
|
+
if counts[current_stage_id] < 1:
|
|
480
|
+
return _diagnostic_decision(
|
|
481
|
+
code="MF-EVAL-G003",
|
|
482
|
+
summary="resolving stage must have at least one post-terminal attempt",
|
|
483
|
+
current_stage_id=current_stage_id,
|
|
484
|
+
terminal_result=terminal_result,
|
|
485
|
+
)
|
|
486
|
+
if current_stage_id != EvalStageId.PLANNER and counts[EvalStageId.PLANNER] != 1:
|
|
487
|
+
return _diagnostic_decision(
|
|
488
|
+
code="MF-EVAL-G005",
|
|
489
|
+
summary="transition omits the required planner stage",
|
|
490
|
+
current_stage_id=current_stage_id,
|
|
491
|
+
terminal_result=terminal_result,
|
|
492
|
+
)
|
|
493
|
+
if current_stage_id == EvalStageId.CHECKER and counts[EvalStageId.BUILDER] < 1:
|
|
494
|
+
return _diagnostic_decision(
|
|
495
|
+
code="MF-EVAL-G005",
|
|
496
|
+
summary="transition omits the required builder stage",
|
|
497
|
+
current_stage_id=current_stage_id,
|
|
498
|
+
terminal_result=terminal_result,
|
|
499
|
+
)
|
|
500
|
+
if current_stage_id == EvalStageId.ARBITER:
|
|
501
|
+
if counts[EvalStageId.BUILDER] < 1 and counts[EvalStageId.CHECKER] < 1:
|
|
502
|
+
return _diagnostic_decision(
|
|
503
|
+
code="MF-EVAL-G005",
|
|
504
|
+
summary="transition omits candidate evidence before arbiter",
|
|
505
|
+
current_stage_id=current_stage_id,
|
|
506
|
+
terminal_result=terminal_result,
|
|
507
|
+
)
|
|
508
|
+
if current_stage_id != EvalStageId.ARBITER and counts[EvalStageId.ARBITER] > 0:
|
|
509
|
+
return _diagnostic_decision(
|
|
510
|
+
code="MF-EVAL-G003",
|
|
511
|
+
summary="arbiter attempts cannot precede a non-arbiter transition",
|
|
512
|
+
current_stage_id=current_stage_id,
|
|
513
|
+
terminal_result=terminal_result,
|
|
514
|
+
)
|
|
515
|
+
return None
|
|
516
|
+
|
|
517
|
+
|
|
518
|
+
def _continue_decision(
|
|
519
|
+
*,
|
|
520
|
+
current_stage_id: EvalStageId,
|
|
521
|
+
terminal_result: EvalTerminalResult,
|
|
522
|
+
next_stage_id: EvalStageId,
|
|
523
|
+
candidate_disposition: EvalCandidateDisposition = EvalCandidateDisposition.NONE,
|
|
524
|
+
graph: CompactEvalWorkflowGraph,
|
|
525
|
+
) -> EvalTransitionDecision:
|
|
526
|
+
if next_stage_id not in graph.stage_contracts:
|
|
527
|
+
return _diagnostic_decision(
|
|
528
|
+
code="MF-EVAL-G005",
|
|
529
|
+
summary="transition routes to a stage omitted from the compact graph",
|
|
530
|
+
current_stage_id=current_stage_id,
|
|
531
|
+
terminal_result=terminal_result,
|
|
532
|
+
)
|
|
533
|
+
return EvalTransitionDecision(
|
|
534
|
+
outcome_kind=EvalWorkflowOutcomeKind.CONTINUE,
|
|
535
|
+
current_stage_id=current_stage_id,
|
|
536
|
+
terminal_result=terminal_result,
|
|
537
|
+
next_stage_id=next_stage_id,
|
|
538
|
+
candidate_disposition=candidate_disposition,
|
|
539
|
+
)
|
|
540
|
+
|
|
541
|
+
|
|
542
|
+
def resolve_eval_transition(
|
|
543
|
+
current_stage_id: EvalStageId | str,
|
|
544
|
+
terminal_result: EvalTerminalResult | str,
|
|
545
|
+
attempt_state: EvalAttemptState | Mapping[str, Any] | None = None,
|
|
546
|
+
*,
|
|
547
|
+
graph: CompactEvalWorkflowGraph | None = None,
|
|
548
|
+
) -> EvalTransitionDecision:
|
|
549
|
+
"""Resolve a compact eval terminal result into the next workflow decision."""
|
|
550
|
+
graph = graph or default_compact_eval_workflow_graph()
|
|
551
|
+
attempt_state = attempt_state or EvalAttemptState()
|
|
552
|
+
stage_id = _coerce_stage_id(current_stage_id)
|
|
553
|
+
if stage_id is None:
|
|
554
|
+
return _diagnostic_decision(
|
|
555
|
+
code="MF-EVAL-G001",
|
|
556
|
+
summary="unknown compact eval stage id",
|
|
557
|
+
current_stage_id=current_stage_id,
|
|
558
|
+
terminal_result=terminal_result,
|
|
559
|
+
)
|
|
560
|
+
result = _coerce_terminal_result(terminal_result)
|
|
561
|
+
if result is None:
|
|
562
|
+
return _diagnostic_decision(
|
|
563
|
+
code="MF-EVAL-G002",
|
|
564
|
+
summary="unknown compact eval terminal result",
|
|
565
|
+
current_stage_id=stage_id,
|
|
566
|
+
terminal_result=terminal_result,
|
|
567
|
+
)
|
|
568
|
+
if stage_id != EvalStageId.ARBITER and result == EvalTerminalResult.ARBITER_CLOSED:
|
|
569
|
+
return _diagnostic_decision(
|
|
570
|
+
code="MF-EVAL-G006",
|
|
571
|
+
summary="only arbiter may complete the compact eval workflow",
|
|
572
|
+
current_stage_id=stage_id,
|
|
573
|
+
terminal_result=result,
|
|
574
|
+
)
|
|
575
|
+
if result not in graph.stage_contracts[stage_id].legal_terminal_results:
|
|
576
|
+
return _diagnostic_decision(
|
|
577
|
+
code="MF-EVAL-G002",
|
|
578
|
+
summary="terminal result is not legal for the resolving stage",
|
|
579
|
+
current_stage_id=stage_id,
|
|
580
|
+
terminal_result=result,
|
|
581
|
+
)
|
|
582
|
+
|
|
583
|
+
invalid_attempts = _attempt_state_diagnostic(
|
|
584
|
+
current_stage_id=stage_id,
|
|
585
|
+
terminal_result=result,
|
|
586
|
+
attempt_state=attempt_state,
|
|
587
|
+
graph=graph,
|
|
588
|
+
)
|
|
589
|
+
if invalid_attempts is not None:
|
|
590
|
+
return invalid_attempts
|
|
591
|
+
|
|
592
|
+
counts = EvalAttemptState.model_validate(attempt_state).model_dump(mode="json")
|
|
593
|
+
builder_attempts = counts["builder_attempts"]
|
|
594
|
+
checker_attempts = counts["checker_attempts"]
|
|
595
|
+
|
|
596
|
+
if stage_id == EvalStageId.PLANNER:
|
|
597
|
+
if result == EvalTerminalResult.PLAN_READY:
|
|
598
|
+
return _continue_decision(
|
|
599
|
+
current_stage_id=stage_id,
|
|
600
|
+
terminal_result=result,
|
|
601
|
+
next_stage_id=EvalStageId.BUILDER,
|
|
602
|
+
graph=graph,
|
|
603
|
+
)
|
|
604
|
+
return EvalTransitionDecision(
|
|
605
|
+
outcome_kind=EvalWorkflowOutcomeKind.BLOCKED,
|
|
606
|
+
current_stage_id=stage_id,
|
|
607
|
+
terminal_result=result,
|
|
608
|
+
)
|
|
609
|
+
|
|
610
|
+
if stage_id == EvalStageId.BUILDER:
|
|
611
|
+
if result == EvalTerminalResult.BUILDER_COMPLETE:
|
|
612
|
+
return _continue_decision(
|
|
613
|
+
current_stage_id=stage_id,
|
|
614
|
+
terminal_result=result,
|
|
615
|
+
next_stage_id=EvalStageId.CHECKER,
|
|
616
|
+
graph=graph,
|
|
617
|
+
)
|
|
618
|
+
return _continue_decision(
|
|
619
|
+
current_stage_id=stage_id,
|
|
620
|
+
terminal_result=result,
|
|
621
|
+
next_stage_id=EvalStageId.ARBITER,
|
|
622
|
+
candidate_disposition=EvalCandidateDisposition.BLOCKED,
|
|
623
|
+
graph=graph,
|
|
624
|
+
)
|
|
625
|
+
|
|
626
|
+
if stage_id == EvalStageId.CHECKER:
|
|
627
|
+
if result == EvalTerminalResult.CHECKER_APPROVED:
|
|
628
|
+
return _continue_decision(
|
|
629
|
+
current_stage_id=stage_id,
|
|
630
|
+
terminal_result=result,
|
|
631
|
+
next_stage_id=EvalStageId.ARBITER,
|
|
632
|
+
candidate_disposition=EvalCandidateDisposition.APPROVED,
|
|
633
|
+
graph=graph,
|
|
634
|
+
)
|
|
635
|
+
if result == EvalTerminalResult.CHECKER_BLOCKED:
|
|
636
|
+
return _continue_decision(
|
|
637
|
+
current_stage_id=stage_id,
|
|
638
|
+
terminal_result=result,
|
|
639
|
+
next_stage_id=EvalStageId.ARBITER,
|
|
640
|
+
candidate_disposition=EvalCandidateDisposition.BLOCKED,
|
|
641
|
+
graph=graph,
|
|
642
|
+
)
|
|
643
|
+
if builder_attempts < 2 and checker_attempts < 2:
|
|
644
|
+
return _continue_decision(
|
|
645
|
+
current_stage_id=stage_id,
|
|
646
|
+
terminal_result=result,
|
|
647
|
+
next_stage_id=EvalStageId.BUILDER,
|
|
648
|
+
candidate_disposition=EvalCandidateDisposition.REJECTED,
|
|
649
|
+
graph=graph,
|
|
650
|
+
)
|
|
651
|
+
return _continue_decision(
|
|
652
|
+
current_stage_id=stage_id,
|
|
653
|
+
terminal_result=result,
|
|
654
|
+
next_stage_id=EvalStageId.ARBITER,
|
|
655
|
+
candidate_disposition=EvalCandidateDisposition.REJECTED,
|
|
656
|
+
graph=graph,
|
|
657
|
+
)
|
|
658
|
+
|
|
659
|
+
if result == EvalTerminalResult.ARBITER_CLOSED:
|
|
660
|
+
contract = graph.stage_contracts[stage_id]
|
|
661
|
+
if not contract.may_complete_workflow:
|
|
662
|
+
return _diagnostic_decision(
|
|
663
|
+
code="MF-EVAL-G006",
|
|
664
|
+
summary="only arbiter may complete the compact eval workflow",
|
|
665
|
+
current_stage_id=stage_id,
|
|
666
|
+
terminal_result=result,
|
|
667
|
+
)
|
|
668
|
+
return EvalTransitionDecision(
|
|
669
|
+
outcome_kind=EvalWorkflowOutcomeKind.COMPLETED,
|
|
670
|
+
current_stage_id=stage_id,
|
|
671
|
+
terminal_result=result,
|
|
672
|
+
)
|
|
673
|
+
disposition = (
|
|
674
|
+
EvalCandidateDisposition.REJECTED
|
|
675
|
+
if result == EvalTerminalResult.ARBITER_REJECTED
|
|
676
|
+
else EvalCandidateDisposition.BLOCKED
|
|
677
|
+
)
|
|
678
|
+
return EvalTransitionDecision(
|
|
679
|
+
outcome_kind=EvalWorkflowOutcomeKind.BLOCKED,
|
|
680
|
+
current_stage_id=stage_id,
|
|
681
|
+
terminal_result=result,
|
|
682
|
+
candidate_disposition=disposition,
|
|
683
|
+
)
|
|
684
|
+
|
|
685
|
+
|
|
686
|
+
def compact_eval_workflow_snapshot(
|
|
687
|
+
graph: CompactEvalWorkflowGraph | None = None,
|
|
688
|
+
) -> dict[str, Any]:
|
|
689
|
+
"""Return the deterministic public snapshot for the compact eval graph."""
|
|
690
|
+
graph = graph or default_compact_eval_workflow_graph()
|
|
691
|
+
snapshot: dict[str, Any] = {
|
|
692
|
+
"graph_id": graph.graph_id,
|
|
693
|
+
"omitted_stage_ids": list(OMITTED_COMPACT_EVAL_STAGE_IDS),
|
|
694
|
+
"schema_version": 1,
|
|
695
|
+
"stages": [stage.model_dump(mode="json") for stage in graph.stages],
|
|
696
|
+
"transitions": [
|
|
697
|
+
{
|
|
698
|
+
"current_stage_id": "eval_planner",
|
|
699
|
+
"terminal_result": "PLAN_READY",
|
|
700
|
+
"outcome_kind": "continue",
|
|
701
|
+
"next_stage_id": "eval_builder",
|
|
702
|
+
"candidate_disposition": "none",
|
|
703
|
+
},
|
|
704
|
+
{
|
|
705
|
+
"current_stage_id": "eval_planner",
|
|
706
|
+
"terminal_result": "PLAN_BLOCKED",
|
|
707
|
+
"outcome_kind": "blocked",
|
|
708
|
+
"candidate_disposition": "none",
|
|
709
|
+
},
|
|
710
|
+
{
|
|
711
|
+
"current_stage_id": "eval_builder",
|
|
712
|
+
"terminal_result": "BUILDER_COMPLETE",
|
|
713
|
+
"outcome_kind": "continue",
|
|
714
|
+
"next_stage_id": "eval_checker",
|
|
715
|
+
"candidate_disposition": "none",
|
|
716
|
+
},
|
|
717
|
+
{
|
|
718
|
+
"current_stage_id": "eval_builder",
|
|
719
|
+
"terminal_result": "BUILDER_BLOCKED",
|
|
720
|
+
"outcome_kind": "continue",
|
|
721
|
+
"next_stage_id": "eval_arbiter",
|
|
722
|
+
"candidate_disposition": "blocked",
|
|
723
|
+
},
|
|
724
|
+
{
|
|
725
|
+
"current_stage_id": "eval_checker",
|
|
726
|
+
"terminal_result": "CHECKER_APPROVED",
|
|
727
|
+
"outcome_kind": "continue",
|
|
728
|
+
"next_stage_id": "eval_arbiter",
|
|
729
|
+
"candidate_disposition": "approved",
|
|
730
|
+
},
|
|
731
|
+
{
|
|
732
|
+
"current_stage_id": "eval_checker",
|
|
733
|
+
"terminal_result": "CHECKER_REJECTED",
|
|
734
|
+
"condition": "builder_attempts < 2 and checker_attempts < 2",
|
|
735
|
+
"outcome_kind": "continue",
|
|
736
|
+
"next_stage_id": "eval_builder",
|
|
737
|
+
"candidate_disposition": "rejected",
|
|
738
|
+
},
|
|
739
|
+
{
|
|
740
|
+
"current_stage_id": "eval_checker",
|
|
741
|
+
"terminal_result": "CHECKER_REJECTED",
|
|
742
|
+
"condition": "builder_attempts >= 2 or checker_attempts >= 2",
|
|
743
|
+
"outcome_kind": "continue",
|
|
744
|
+
"next_stage_id": "eval_arbiter",
|
|
745
|
+
"candidate_disposition": "rejected",
|
|
746
|
+
},
|
|
747
|
+
{
|
|
748
|
+
"current_stage_id": "eval_checker",
|
|
749
|
+
"terminal_result": "CHECKER_BLOCKED",
|
|
750
|
+
"outcome_kind": "continue",
|
|
751
|
+
"next_stage_id": "eval_arbiter",
|
|
752
|
+
"candidate_disposition": "blocked",
|
|
753
|
+
},
|
|
754
|
+
{
|
|
755
|
+
"current_stage_id": "eval_arbiter",
|
|
756
|
+
"terminal_result": "ARBITER_CLOSED",
|
|
757
|
+
"outcome_kind": "completed",
|
|
758
|
+
"candidate_disposition": "none",
|
|
759
|
+
},
|
|
760
|
+
{
|
|
761
|
+
"current_stage_id": "eval_arbiter",
|
|
762
|
+
"terminal_result": "ARBITER_REJECTED",
|
|
763
|
+
"outcome_kind": "blocked",
|
|
764
|
+
"candidate_disposition": "rejected",
|
|
765
|
+
},
|
|
766
|
+
{
|
|
767
|
+
"current_stage_id": "eval_arbiter",
|
|
768
|
+
"terminal_result": "ARBITER_BLOCKED",
|
|
769
|
+
"outcome_kind": "blocked",
|
|
770
|
+
"candidate_disposition": "blocked",
|
|
771
|
+
},
|
|
772
|
+
],
|
|
773
|
+
}
|
|
774
|
+
snapshot["graph_sha256"] = calculate_compact_eval_workflow_sha256(snapshot)
|
|
775
|
+
return snapshot
|
|
776
|
+
|
|
777
|
+
|
|
778
|
+
def calculate_compact_eval_workflow_sha256(snapshot: Mapping[str, Any]) -> str:
|
|
779
|
+
"""Hash a compact eval snapshot without its ``graph_sha256`` field."""
|
|
780
|
+
body = dict(snapshot)
|
|
781
|
+
body.pop("graph_sha256", None)
|
|
782
|
+
return hashlib.sha256(_canonical_json_serialize(body).encode("utf-8")).hexdigest()
|
|
783
|
+
|
|
784
|
+
|
|
785
|
+
def canonical_compact_eval_workflow_bytes(
|
|
786
|
+
graph: CompactEvalWorkflowGraph | None = None,
|
|
787
|
+
) -> bytes:
|
|
788
|
+
"""Serialize the compact eval graph snapshot as canonical UTF-8 JSON bytes."""
|
|
789
|
+
snapshot = compact_eval_workflow_snapshot(graph)
|
|
790
|
+
expected = snapshot["graph_sha256"]
|
|
791
|
+
computed = calculate_compact_eval_workflow_sha256(snapshot)
|
|
792
|
+
if computed != expected:
|
|
793
|
+
raise ValueError("compact eval graph fingerprint verification failed")
|
|
794
|
+
return _canonical_json_serialize(snapshot).encode("utf-8")
|