millforge 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. millforge/__init__.py +1174 -0
  2. millforge/_forge/LICENSE +21 -0
  3. millforge/_forge/PROVENANCE.json +295 -0
  4. millforge/_forge/UPDATE_POLICY.md +24 -0
  5. millforge/_forge/__init__.py +14 -0
  6. millforge/_forge/adapter.py +2232 -0
  7. millforge/_forge/base_runner.py +121 -0
  8. millforge/_forge/clients/__init__.py +10 -0
  9. millforge/_forge/clients/base.py +200 -0
  10. millforge/_forge/context/__init__.py +23 -0
  11. millforge/_forge/context/manager.py +178 -0
  12. millforge/_forge/context/strategies.py +335 -0
  13. millforge/_forge/core/__init__.py +16 -0
  14. millforge/_forge/core/inference.py +433 -0
  15. millforge/_forge/core/messages.py +119 -0
  16. millforge/_forge/core/runner.py +479 -0
  17. millforge/_forge/core/steps.py +108 -0
  18. millforge/_forge/core/workflow.py +400 -0
  19. millforge/_forge/errors.py +222 -0
  20. millforge/_forge/guardrails/__init__.py +21 -0
  21. millforge/_forge/guardrails/error_tracker.py +71 -0
  22. millforge/_forge/guardrails/guardrails.py +194 -0
  23. millforge/_forge/guardrails/nudge.py +47 -0
  24. millforge/_forge/guardrails/response_validator.py +119 -0
  25. millforge/_forge/guardrails/step_enforcer.py +183 -0
  26. millforge/_forge/prompts/__init__.py +16 -0
  27. millforge/_forge/prompts/nudges.py +95 -0
  28. millforge/_forge/prompts/templates.py +285 -0
  29. millforge/_version.py +3 -0
  30. millforge/artifacts.py +570 -0
  31. millforge/base/__init__.py +97 -0
  32. millforge/base/composition.py +402 -0
  33. millforge/base/context.py +285 -0
  34. millforge/base/harness.py +138 -0
  35. millforge/base/identity.py +465 -0
  36. millforge/base/options.py +34 -0
  37. millforge/base/platform.py +17 -0
  38. millforge/base/prompt.py +317 -0
  39. millforge/base/runner.py +546 -0
  40. millforge/compiled_plan.py +970 -0
  41. millforge/compiler/__init__.py +231 -0
  42. millforge/compiler/artifact_validation.py +257 -0
  43. millforge/compiler/canonicalization.py +169 -0
  44. millforge/compiler/capabilities.py +66 -0
  45. millforge/compiler/catalogs.py +500 -0
  46. millforge/compiler/diagnostics.py +491 -0
  47. millforge/compiler/graph.py +678 -0
  48. millforge/compiler/lowering.py +198 -0
  49. millforge/compiler/output.py +692 -0
  50. millforge/compiler/parsing.py +1424 -0
  51. millforge/compiler/requests.py +1180 -0
  52. millforge/compiler/schema_validation.py +272 -0
  53. millforge/compiler/semantic.py +490 -0
  54. millforge/compiler/service.py +448 -0
  55. millforge/compiler/source.py +375 -0
  56. millforge/compiler/validators.py +184 -0
  57. millforge/connectors/__init__.py +95 -0
  58. millforge/connectors/admission.py +801 -0
  59. millforge/connectors/broker.py +202 -0
  60. millforge/connectors/contracts.py +1159 -0
  61. millforge/connectors/diagnostics.py +189 -0
  62. millforge/connectors/fake.py +66 -0
  63. millforge/connectors/runtime.py +236 -0
  64. millforge/contracts.py +2860 -0
  65. millforge/custom_tools/__init__.py +67 -0
  66. millforge/custom_tools/compiler.py +724 -0
  67. millforge/custom_tools/contracts.py +1093 -0
  68. millforge/custom_tools/diagnostics.py +205 -0
  69. millforge/eval_artifacts.py +952 -0
  70. millforge/eval_boundary.py +2435 -0
  71. millforge/eval_fixtures/__init__.py +1 -0
  72. millforge/eval_fixtures/default_pack/__init__.py +1 -0
  73. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.bug_diagnosis.traceback.v1.json +52 -0
  74. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.direct_edit.import_sort.v1.json +52 -0
  75. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.evidence_discipline.no_source_change.v1.json +51 -0
  76. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.false_closure.visible_green.v1.json +52 -0
  77. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.multi_file.api_contract.v1.json +54 -0
  78. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.recovery.malformed_artifact.v1.json +54 -0
  79. millforge/eval_fixtures/default_pack/manifest.json +12 -0
  80. millforge/eval_modes.py +1282 -0
  81. millforge/eval_presets.py +1398 -0
  82. millforge/eval_reports.py +2517 -0
  83. millforge/eval_suite.py +2429 -0
  84. millforge/eval_trials.py +2632 -0
  85. millforge/eval_workflow.py +794 -0
  86. millforge/exceptions.py +122 -0
  87. millforge/model_backend.py +2098 -0
  88. millforge/protocols.py +340 -0
  89. millforge/py.typed +0 -0
  90. millforge/runtime.py +1791 -0
  91. millforge/testing/__init__.py +1089 -0
  92. millforge/tools/__init__.py +83 -0
  93. millforge/tools/builtin_runtime.py +1339 -0
  94. millforge/tools/builtins.py +773 -0
  95. millforge/tools/execution.py +1545 -0
  96. millforge/tools/path_policy.py +155 -0
  97. millforge/tools/pi_compat/PI_LICENSE +21 -0
  98. millforge/tools/pi_compat/PROVENANCE.json +55 -0
  99. millforge/tools/pi_compat/UPDATE_POLICY.md +36 -0
  100. millforge/tools/pi_compat/__init__.py +34 -0
  101. millforge/tools/pi_compat/contracts.py +49 -0
  102. millforge/tools/pi_compat/editing.py +390 -0
  103. millforge/tools/pi_compat/mutations.py +57 -0
  104. millforge/tools/pi_compat/operations.py +401 -0
  105. millforge/tools/pi_compat/paths.py +155 -0
  106. millforge/tools/pi_compat/process.py +1375 -0
  107. millforge/tools/pi_compat/search.py +738 -0
  108. millforge/tools/pi_compat/truncation.py +267 -0
  109. millforge/tools/pi_compat_catalog.py +396 -0
  110. millforge/tools/pi_compat_runtime.py +460 -0
  111. millforge/tools/registry.py +553 -0
  112. millforge/tools/results.py +533 -0
  113. millforge-0.1.0.dist-info/METADATA +844 -0
  114. millforge-0.1.0.dist-info/RECORD +116 -0
  115. millforge-0.1.0.dist-info/WHEEL +4 -0
  116. millforge-0.1.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,794 @@
1
+ """Compact public eval workflow contracts for Millforge."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import json
7
+ from collections.abc import Mapping
8
+ from enum import Enum
9
+ from types import MappingProxyType
10
+ from typing import Any
11
+
12
+ from pydantic import BaseModel, ConfigDict, Field, StrictBool, StrictInt, StrictStr
13
+ from pydantic import field_serializer, field_validator, model_validator
14
+
15
+
16
+ class EvalStageId(str, Enum):
17
+ """Closed stage IDs for the compact eval workflow."""
18
+
19
+ PLANNER = "eval_planner"
20
+ BUILDER = "eval_builder"
21
+ CHECKER = "eval_checker"
22
+ ARBITER = "eval_arbiter"
23
+
24
+
25
+ class EvalTerminalResult(str, Enum):
26
+ """Closed terminal results emitted by compact eval stages."""
27
+
28
+ PLAN_READY = "PLAN_READY"
29
+ PLAN_BLOCKED = "PLAN_BLOCKED"
30
+ BUILDER_COMPLETE = "BUILDER_COMPLETE"
31
+ BUILDER_BLOCKED = "BUILDER_BLOCKED"
32
+ CHECKER_APPROVED = "CHECKER_APPROVED"
33
+ CHECKER_REJECTED = "CHECKER_REJECTED"
34
+ CHECKER_BLOCKED = "CHECKER_BLOCKED"
35
+ ARBITER_CLOSED = "ARBITER_CLOSED"
36
+ ARBITER_REJECTED = "ARBITER_REJECTED"
37
+ ARBITER_BLOCKED = "ARBITER_BLOCKED"
38
+
39
+
40
+ class EvalWorkflowOutcomeKind(str, Enum):
41
+ """Closed workflow transition outcome kinds."""
42
+
43
+ CONTINUE = "continue"
44
+ COMPLETED = "completed"
45
+ BLOCKED = "blocked"
46
+ INVALID = "invalid"
47
+
48
+
49
+ class EvalCandidateDisposition(str, Enum):
50
+ """Closed candidate dispositions carried between compact eval stages."""
51
+
52
+ NONE = "none"
53
+ APPROVED = "approved"
54
+ REJECTED = "rejected"
55
+ BLOCKED = "blocked"
56
+
57
+
58
+ OMITTED_COMPACT_EVAL_STAGE_IDS: tuple[str, ...] = (
59
+ "manager",
60
+ "fixer",
61
+ "doublechecker",
62
+ "troubleshooter",
63
+ "consultant",
64
+ "mechanic",
65
+ "auditor",
66
+ "updater",
67
+ "librarian",
68
+ "analyst",
69
+ "professor",
70
+ "curator",
71
+ )
72
+
73
+
74
+ class EvalStageContract(BaseModel):
75
+ """Immutable contract for one compact eval workflow stage."""
76
+
77
+ model_config = ConfigDict(extra="forbid", frozen=True)
78
+
79
+ stage_id: EvalStageId
80
+ role_summary: StrictStr
81
+ input_artifact_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
82
+ output_artifact_ids: tuple[StrictStr, ...] = Field(default_factory=tuple)
83
+ legal_terminal_results: tuple[EvalTerminalResult, ...]
84
+ domain_attempt_limit: StrictInt = Field(gt=0)
85
+ infrastructure_retry_limit: StrictInt = Field(ge=0, le=1)
86
+ may_complete_workflow: StrictBool
87
+
88
+ @field_validator("role_summary")
89
+ @classmethod
90
+ def _role_summary_nonblank(cls, value: str) -> str:
91
+ if not value.strip():
92
+ raise ValueError("role_summary must be a non-empty string")
93
+ return value
94
+
95
+ @field_validator("input_artifact_ids", "output_artifact_ids")
96
+ @classmethod
97
+ def _artifact_ids_valid(cls, value: tuple[str, ...]) -> tuple[str, ...]:
98
+ if len(set(value)) != len(value):
99
+ raise ValueError("artifact ids must be unique")
100
+ for artifact_id in value:
101
+ if not artifact_id.strip():
102
+ raise ValueError("artifact ids must be non-empty strings")
103
+ return value
104
+
105
+ @field_validator("legal_terminal_results")
106
+ @classmethod
107
+ def _legal_terminal_results_valid(
108
+ cls, value: tuple[EvalTerminalResult, ...]
109
+ ) -> tuple[EvalTerminalResult, ...]:
110
+ if not value:
111
+ raise ValueError("legal_terminal_results must not be empty")
112
+ if len(set(value)) != len(value):
113
+ raise ValueError("legal_terminal_results values must be unique")
114
+ return value
115
+
116
+
117
+ class EvalAttemptState(BaseModel):
118
+ """Immutable domain-attempt and infrastructure-retry counters."""
119
+
120
+ model_config = ConfigDict(extra="forbid", frozen=True)
121
+
122
+ planner_attempts: StrictInt = Field(default=0, ge=0)
123
+ builder_attempts: StrictInt = Field(default=0, ge=0)
124
+ checker_attempts: StrictInt = Field(default=0, ge=0)
125
+ arbiter_attempts: StrictInt = Field(default=0, ge=0)
126
+ infrastructure_retries: Mapping[EvalStageId, StrictInt] = Field(
127
+ default_factory=dict
128
+ )
129
+
130
+ @model_validator(mode="after")
131
+ def _freeze_retries(self) -> EvalAttemptState:
132
+ ordered: dict[EvalStageId, int] = {}
133
+ for stage_id in EvalStageId:
134
+ retry_count = self.infrastructure_retries.get(stage_id, 0)
135
+ if retry_count < 0:
136
+ raise ValueError("infrastructure retry counts must be non-negative")
137
+ if retry_count > 1:
138
+ raise ValueError("infrastructure retry counts may not exceed 1")
139
+ if retry_count:
140
+ ordered[stage_id] = retry_count
141
+ object.__setattr__(self, "infrastructure_retries", MappingProxyType(ordered))
142
+ return self
143
+
144
+ @field_serializer("infrastructure_retries")
145
+ def _serialize_retries(self, value: Mapping[EvalStageId, int]) -> dict[str, int]:
146
+ return {
147
+ stage_id.value: value[stage_id]
148
+ for stage_id in EvalStageId
149
+ if stage_id in value
150
+ }
151
+
152
+
153
+ class EvalTransitionDecision(BaseModel):
154
+ """Immutable transition decision for a compact eval workflow result."""
155
+
156
+ model_config = ConfigDict(extra="forbid", frozen=True)
157
+
158
+ outcome_kind: EvalWorkflowOutcomeKind
159
+ current_stage_id: EvalStageId | StrictStr
160
+ terminal_result: EvalTerminalResult | StrictStr
161
+ next_stage_id: EvalStageId | None = None
162
+ candidate_disposition: EvalCandidateDisposition = EvalCandidateDisposition.NONE
163
+ diagnostic_code: StrictStr | None = None
164
+ diagnostic_summary: StrictStr | None = None
165
+
166
+ @model_validator(mode="after")
167
+ def _decision_shape_valid(self) -> EvalTransitionDecision:
168
+ if self.outcome_kind == EvalWorkflowOutcomeKind.CONTINUE:
169
+ if self.next_stage_id is None:
170
+ raise ValueError("continue decisions must declare next_stage_id")
171
+ if self.diagnostic_code is not None or self.diagnostic_summary is not None:
172
+ raise ValueError("continue decisions must not include diagnostics")
173
+ else:
174
+ if self.next_stage_id is not None:
175
+ raise ValueError("terminal decisions must not declare next_stage_id")
176
+ if self.outcome_kind == EvalWorkflowOutcomeKind.INVALID:
177
+ if self.diagnostic_code is None or self.diagnostic_summary is None:
178
+ raise ValueError("invalid decisions must include diagnostics")
179
+ if (self.diagnostic_code is None) != (self.diagnostic_summary is None):
180
+ raise ValueError(
181
+ "diagnostic_code and diagnostic_summary must appear together"
182
+ )
183
+ return self
184
+
185
+
186
+ class CompactEvalWorkflowGraph(BaseModel):
187
+ """Immutable static compact eval workflow graph."""
188
+
189
+ model_config = ConfigDict(extra="forbid", frozen=True)
190
+
191
+ graph_id: StrictStr = "compact_eval_workflow.v1"
192
+ stages: tuple[EvalStageContract, ...]
193
+
194
+ @model_validator(mode="before")
195
+ @classmethod
196
+ def _reject_mapping_stages(cls, data: Any) -> Any:
197
+ if not isinstance(data, Mapping):
198
+ return data
199
+ values = dict(data)
200
+ if isinstance(values.get("stages"), Mapping):
201
+ raise ValueError("stages must be an ordered tuple or list")
202
+ return values
203
+
204
+ @model_validator(mode="after")
205
+ def _graph_shape_valid(self) -> CompactEvalWorkflowGraph:
206
+ expected_stage_ids = tuple(stage_id for stage_id in EvalStageId)
207
+ actual_stage_ids = tuple(stage.stage_id for stage in self.stages)
208
+ if actual_stage_ids != expected_stage_ids:
209
+ raise ValueError(
210
+ "compact eval workflow stages must be exactly "
211
+ "eval_planner, eval_builder, eval_checker, eval_arbiter"
212
+ )
213
+ expected_results = {
214
+ EvalStageId.PLANNER: (
215
+ EvalTerminalResult.PLAN_READY,
216
+ EvalTerminalResult.PLAN_BLOCKED,
217
+ ),
218
+ EvalStageId.BUILDER: (
219
+ EvalTerminalResult.BUILDER_COMPLETE,
220
+ EvalTerminalResult.BUILDER_BLOCKED,
221
+ ),
222
+ EvalStageId.CHECKER: (
223
+ EvalTerminalResult.CHECKER_APPROVED,
224
+ EvalTerminalResult.CHECKER_REJECTED,
225
+ EvalTerminalResult.CHECKER_BLOCKED,
226
+ ),
227
+ EvalStageId.ARBITER: (
228
+ EvalTerminalResult.ARBITER_CLOSED,
229
+ EvalTerminalResult.ARBITER_REJECTED,
230
+ EvalTerminalResult.ARBITER_BLOCKED,
231
+ ),
232
+ }
233
+ expected_domain_limits = {
234
+ EvalStageId.PLANNER: 1,
235
+ EvalStageId.BUILDER: 2,
236
+ EvalStageId.CHECKER: 2,
237
+ EvalStageId.ARBITER: 1,
238
+ }
239
+ for stage in self.stages:
240
+ if stage.legal_terminal_results != expected_results[stage.stage_id]:
241
+ raise ValueError(
242
+ f"{stage.stage_id.value} legal_terminal_results are invalid"
243
+ )
244
+ if stage.domain_attempt_limit != expected_domain_limits[stage.stage_id]:
245
+ raise ValueError(
246
+ f"{stage.stage_id.value} domain_attempt_limit is invalid"
247
+ )
248
+ if stage.infrastructure_retry_limit > 1:
249
+ raise ValueError(
250
+ f"{stage.stage_id.value} infrastructure_retry_limit is invalid"
251
+ )
252
+ expected_completion = stage.stage_id == EvalStageId.ARBITER
253
+ if stage.may_complete_workflow is not expected_completion:
254
+ raise ValueError(
255
+ f"{stage.stage_id.value} completion permission is invalid"
256
+ )
257
+ return self
258
+
259
+ @property
260
+ def stage_ids(self) -> tuple[EvalStageId, ...]:
261
+ """Return stage IDs in graph order."""
262
+ return tuple(stage.stage_id for stage in self.stages)
263
+
264
+ @property
265
+ def stage_contracts(self) -> Mapping[EvalStageId, EvalStageContract]:
266
+ """Return stage contracts keyed by stage ID."""
267
+ return MappingProxyType({stage.stage_id: stage for stage in self.stages})
268
+
269
+
270
+ def default_compact_eval_workflow_graph() -> CompactEvalWorkflowGraph:
271
+ """Return the static compact eval workflow graph contract."""
272
+ return CompactEvalWorkflowGraph(
273
+ stages=(
274
+ EvalStageContract(
275
+ stage_id=EvalStageId.PLANNER,
276
+ role_summary="Plan a compact eval candidate workflow.",
277
+ input_artifact_ids=("task", "fixture_manifest", "acceptance_checks"),
278
+ output_artifact_ids=("plan",),
279
+ legal_terminal_results=(
280
+ EvalTerminalResult.PLAN_READY,
281
+ EvalTerminalResult.PLAN_BLOCKED,
282
+ ),
283
+ domain_attempt_limit=1,
284
+ infrastructure_retry_limit=1,
285
+ may_complete_workflow=False,
286
+ ),
287
+ EvalStageContract(
288
+ stage_id=EvalStageId.BUILDER,
289
+ role_summary="Build the candidate implementation under eval.",
290
+ input_artifact_ids=(
291
+ "task",
292
+ "fixture_manifest",
293
+ "plan",
294
+ "checker_verdict",
295
+ ),
296
+ output_artifact_ids=(
297
+ "workspace_diff",
298
+ "patch_summary",
299
+ "test_results",
300
+ ),
301
+ legal_terminal_results=(
302
+ EvalTerminalResult.BUILDER_COMPLETE,
303
+ EvalTerminalResult.BUILDER_BLOCKED,
304
+ ),
305
+ domain_attempt_limit=2,
306
+ infrastructure_retry_limit=1,
307
+ may_complete_workflow=False,
308
+ ),
309
+ EvalStageContract(
310
+ stage_id=EvalStageId.CHECKER,
311
+ role_summary="Check the candidate implementation against the eval plan.",
312
+ input_artifact_ids=(
313
+ "task",
314
+ "fixture_manifest",
315
+ "plan",
316
+ "workspace_diff",
317
+ "patch_summary",
318
+ "test_results",
319
+ ),
320
+ output_artifact_ids=("checker_verdict",),
321
+ legal_terminal_results=(
322
+ EvalTerminalResult.CHECKER_APPROVED,
323
+ EvalTerminalResult.CHECKER_REJECTED,
324
+ EvalTerminalResult.CHECKER_BLOCKED,
325
+ ),
326
+ domain_attempt_limit=2,
327
+ infrastructure_retry_limit=1,
328
+ may_complete_workflow=False,
329
+ ),
330
+ EvalStageContract(
331
+ stage_id=EvalStageId.ARBITER,
332
+ role_summary="Settle the compact eval candidate outcome.",
333
+ input_artifact_ids=(
334
+ "task",
335
+ "fixture_manifest",
336
+ "plan",
337
+ "workspace_diff",
338
+ "patch_summary",
339
+ "test_results",
340
+ "checker_verdict",
341
+ ),
342
+ output_artifact_ids=("arbiter_verdict",),
343
+ legal_terminal_results=(
344
+ EvalTerminalResult.ARBITER_CLOSED,
345
+ EvalTerminalResult.ARBITER_REJECTED,
346
+ EvalTerminalResult.ARBITER_BLOCKED,
347
+ ),
348
+ domain_attempt_limit=1,
349
+ infrastructure_retry_limit=1,
350
+ may_complete_workflow=True,
351
+ ),
352
+ )
353
+ )
354
+
355
+
356
+ def _canonical_json_serialize(obj: Any) -> str:
357
+ return (
358
+ json.dumps(
359
+ obj,
360
+ sort_keys=True,
361
+ ensure_ascii=True,
362
+ allow_nan=False,
363
+ separators=(",", ":"),
364
+ ).replace("\r\n", "\n")
365
+ + "\n"
366
+ )
367
+
368
+
369
+ def _diagnostic_decision(
370
+ *,
371
+ code: str,
372
+ summary: str,
373
+ current_stage_id: EvalStageId | str,
374
+ terminal_result: EvalTerminalResult | str,
375
+ ) -> EvalTransitionDecision:
376
+ return EvalTransitionDecision(
377
+ outcome_kind=EvalWorkflowOutcomeKind.INVALID,
378
+ current_stage_id=current_stage_id,
379
+ terminal_result=terminal_result,
380
+ diagnostic_code=code,
381
+ diagnostic_summary=summary,
382
+ )
383
+
384
+
385
+ def _coerce_stage_id(value: EvalStageId | str) -> EvalStageId | None:
386
+ if isinstance(value, EvalStageId):
387
+ return value
388
+ try:
389
+ return EvalStageId(value)
390
+ except ValueError:
391
+ return None
392
+
393
+
394
+ def _coerce_terminal_result(
395
+ value: EvalTerminalResult | str,
396
+ ) -> EvalTerminalResult | None:
397
+ if isinstance(value, EvalTerminalResult):
398
+ return value
399
+ try:
400
+ return EvalTerminalResult(value)
401
+ except ValueError:
402
+ return None
403
+
404
+
405
+ def _raw_attempt_mapping(
406
+ attempt_state: EvalAttemptState | Mapping[str, Any],
407
+ ) -> Mapping[str, Any]:
408
+ if isinstance(attempt_state, EvalAttemptState):
409
+ return attempt_state.model_dump(mode="json")
410
+ return attempt_state
411
+
412
+
413
+ def _attempt_state_diagnostic(
414
+ *,
415
+ current_stage_id: EvalStageId,
416
+ terminal_result: EvalTerminalResult,
417
+ attempt_state: EvalAttemptState | Mapping[str, Any],
418
+ graph: CompactEvalWorkflowGraph,
419
+ ) -> EvalTransitionDecision | None:
420
+ raw_attempts = _raw_attempt_mapping(attempt_state)
421
+ raw_retries = raw_attempts.get("infrastructure_retries", {})
422
+ if not isinstance(raw_retries, Mapping):
423
+ return _diagnostic_decision(
424
+ code="MF-EVAL-G004",
425
+ summary="infrastructure retry state is not a mapping",
426
+ current_stage_id=current_stage_id,
427
+ terminal_result=terminal_result,
428
+ )
429
+ for raw_stage_id, raw_retry_count in raw_retries.items():
430
+ stage_id = _coerce_stage_id(raw_stage_id)
431
+ if stage_id is None:
432
+ return _diagnostic_decision(
433
+ code="MF-EVAL-G004",
434
+ summary="infrastructure retry state contains an unknown stage",
435
+ current_stage_id=current_stage_id,
436
+ terminal_result=terminal_result,
437
+ )
438
+ if not isinstance(raw_retry_count, int) or isinstance(raw_retry_count, bool):
439
+ return _diagnostic_decision(
440
+ code="MF-EVAL-G004",
441
+ summary="infrastructure retry counts must be integers",
442
+ current_stage_id=current_stage_id,
443
+ terminal_result=terminal_result,
444
+ )
445
+ if raw_retry_count < 0 or raw_retry_count > 1:
446
+ return _diagnostic_decision(
447
+ code="MF-EVAL-G004",
448
+ summary="infrastructure retry count exceeds the compact graph limit",
449
+ current_stage_id=current_stage_id,
450
+ terminal_result=terminal_result,
451
+ )
452
+
453
+ limits = {stage.stage_id: stage.domain_attempt_limit for stage in graph.stages}
454
+ count_fields = {
455
+ EvalStageId.PLANNER: "planner_attempts",
456
+ EvalStageId.BUILDER: "builder_attempts",
457
+ EvalStageId.CHECKER: "checker_attempts",
458
+ EvalStageId.ARBITER: "arbiter_attempts",
459
+ }
460
+ counts: dict[EvalStageId, int] = {}
461
+ for stage_id, field_name in count_fields.items():
462
+ raw_count = raw_attempts.get(field_name, 0)
463
+ if not isinstance(raw_count, int) or isinstance(raw_count, bool):
464
+ return _diagnostic_decision(
465
+ code="MF-EVAL-G003",
466
+ summary="domain attempt counts must be integers",
467
+ current_stage_id=current_stage_id,
468
+ terminal_result=terminal_result,
469
+ )
470
+ if raw_count < 0 or raw_count > limits[stage_id]:
471
+ return _diagnostic_decision(
472
+ code="MF-EVAL-G003",
473
+ summary="domain attempt count exceeds the compact graph limit",
474
+ current_stage_id=current_stage_id,
475
+ terminal_result=terminal_result,
476
+ )
477
+ counts[stage_id] = raw_count
478
+
479
+ if counts[current_stage_id] < 1:
480
+ return _diagnostic_decision(
481
+ code="MF-EVAL-G003",
482
+ summary="resolving stage must have at least one post-terminal attempt",
483
+ current_stage_id=current_stage_id,
484
+ terminal_result=terminal_result,
485
+ )
486
+ if current_stage_id != EvalStageId.PLANNER and counts[EvalStageId.PLANNER] != 1:
487
+ return _diagnostic_decision(
488
+ code="MF-EVAL-G005",
489
+ summary="transition omits the required planner stage",
490
+ current_stage_id=current_stage_id,
491
+ terminal_result=terminal_result,
492
+ )
493
+ if current_stage_id == EvalStageId.CHECKER and counts[EvalStageId.BUILDER] < 1:
494
+ return _diagnostic_decision(
495
+ code="MF-EVAL-G005",
496
+ summary="transition omits the required builder stage",
497
+ current_stage_id=current_stage_id,
498
+ terminal_result=terminal_result,
499
+ )
500
+ if current_stage_id == EvalStageId.ARBITER:
501
+ if counts[EvalStageId.BUILDER] < 1 and counts[EvalStageId.CHECKER] < 1:
502
+ return _diagnostic_decision(
503
+ code="MF-EVAL-G005",
504
+ summary="transition omits candidate evidence before arbiter",
505
+ current_stage_id=current_stage_id,
506
+ terminal_result=terminal_result,
507
+ )
508
+ if current_stage_id != EvalStageId.ARBITER and counts[EvalStageId.ARBITER] > 0:
509
+ return _diagnostic_decision(
510
+ code="MF-EVAL-G003",
511
+ summary="arbiter attempts cannot precede a non-arbiter transition",
512
+ current_stage_id=current_stage_id,
513
+ terminal_result=terminal_result,
514
+ )
515
+ return None
516
+
517
+
518
+ def _continue_decision(
519
+ *,
520
+ current_stage_id: EvalStageId,
521
+ terminal_result: EvalTerminalResult,
522
+ next_stage_id: EvalStageId,
523
+ candidate_disposition: EvalCandidateDisposition = EvalCandidateDisposition.NONE,
524
+ graph: CompactEvalWorkflowGraph,
525
+ ) -> EvalTransitionDecision:
526
+ if next_stage_id not in graph.stage_contracts:
527
+ return _diagnostic_decision(
528
+ code="MF-EVAL-G005",
529
+ summary="transition routes to a stage omitted from the compact graph",
530
+ current_stage_id=current_stage_id,
531
+ terminal_result=terminal_result,
532
+ )
533
+ return EvalTransitionDecision(
534
+ outcome_kind=EvalWorkflowOutcomeKind.CONTINUE,
535
+ current_stage_id=current_stage_id,
536
+ terminal_result=terminal_result,
537
+ next_stage_id=next_stage_id,
538
+ candidate_disposition=candidate_disposition,
539
+ )
540
+
541
+
542
+ def resolve_eval_transition(
543
+ current_stage_id: EvalStageId | str,
544
+ terminal_result: EvalTerminalResult | str,
545
+ attempt_state: EvalAttemptState | Mapping[str, Any] | None = None,
546
+ *,
547
+ graph: CompactEvalWorkflowGraph | None = None,
548
+ ) -> EvalTransitionDecision:
549
+ """Resolve a compact eval terminal result into the next workflow decision."""
550
+ graph = graph or default_compact_eval_workflow_graph()
551
+ attempt_state = attempt_state or EvalAttemptState()
552
+ stage_id = _coerce_stage_id(current_stage_id)
553
+ if stage_id is None:
554
+ return _diagnostic_decision(
555
+ code="MF-EVAL-G001",
556
+ summary="unknown compact eval stage id",
557
+ current_stage_id=current_stage_id,
558
+ terminal_result=terminal_result,
559
+ )
560
+ result = _coerce_terminal_result(terminal_result)
561
+ if result is None:
562
+ return _diagnostic_decision(
563
+ code="MF-EVAL-G002",
564
+ summary="unknown compact eval terminal result",
565
+ current_stage_id=stage_id,
566
+ terminal_result=terminal_result,
567
+ )
568
+ if stage_id != EvalStageId.ARBITER and result == EvalTerminalResult.ARBITER_CLOSED:
569
+ return _diagnostic_decision(
570
+ code="MF-EVAL-G006",
571
+ summary="only arbiter may complete the compact eval workflow",
572
+ current_stage_id=stage_id,
573
+ terminal_result=result,
574
+ )
575
+ if result not in graph.stage_contracts[stage_id].legal_terminal_results:
576
+ return _diagnostic_decision(
577
+ code="MF-EVAL-G002",
578
+ summary="terminal result is not legal for the resolving stage",
579
+ current_stage_id=stage_id,
580
+ terminal_result=result,
581
+ )
582
+
583
+ invalid_attempts = _attempt_state_diagnostic(
584
+ current_stage_id=stage_id,
585
+ terminal_result=result,
586
+ attempt_state=attempt_state,
587
+ graph=graph,
588
+ )
589
+ if invalid_attempts is not None:
590
+ return invalid_attempts
591
+
592
+ counts = EvalAttemptState.model_validate(attempt_state).model_dump(mode="json")
593
+ builder_attempts = counts["builder_attempts"]
594
+ checker_attempts = counts["checker_attempts"]
595
+
596
+ if stage_id == EvalStageId.PLANNER:
597
+ if result == EvalTerminalResult.PLAN_READY:
598
+ return _continue_decision(
599
+ current_stage_id=stage_id,
600
+ terminal_result=result,
601
+ next_stage_id=EvalStageId.BUILDER,
602
+ graph=graph,
603
+ )
604
+ return EvalTransitionDecision(
605
+ outcome_kind=EvalWorkflowOutcomeKind.BLOCKED,
606
+ current_stage_id=stage_id,
607
+ terminal_result=result,
608
+ )
609
+
610
+ if stage_id == EvalStageId.BUILDER:
611
+ if result == EvalTerminalResult.BUILDER_COMPLETE:
612
+ return _continue_decision(
613
+ current_stage_id=stage_id,
614
+ terminal_result=result,
615
+ next_stage_id=EvalStageId.CHECKER,
616
+ graph=graph,
617
+ )
618
+ return _continue_decision(
619
+ current_stage_id=stage_id,
620
+ terminal_result=result,
621
+ next_stage_id=EvalStageId.ARBITER,
622
+ candidate_disposition=EvalCandidateDisposition.BLOCKED,
623
+ graph=graph,
624
+ )
625
+
626
+ if stage_id == EvalStageId.CHECKER:
627
+ if result == EvalTerminalResult.CHECKER_APPROVED:
628
+ return _continue_decision(
629
+ current_stage_id=stage_id,
630
+ terminal_result=result,
631
+ next_stage_id=EvalStageId.ARBITER,
632
+ candidate_disposition=EvalCandidateDisposition.APPROVED,
633
+ graph=graph,
634
+ )
635
+ if result == EvalTerminalResult.CHECKER_BLOCKED:
636
+ return _continue_decision(
637
+ current_stage_id=stage_id,
638
+ terminal_result=result,
639
+ next_stage_id=EvalStageId.ARBITER,
640
+ candidate_disposition=EvalCandidateDisposition.BLOCKED,
641
+ graph=graph,
642
+ )
643
+ if builder_attempts < 2 and checker_attempts < 2:
644
+ return _continue_decision(
645
+ current_stage_id=stage_id,
646
+ terminal_result=result,
647
+ next_stage_id=EvalStageId.BUILDER,
648
+ candidate_disposition=EvalCandidateDisposition.REJECTED,
649
+ graph=graph,
650
+ )
651
+ return _continue_decision(
652
+ current_stage_id=stage_id,
653
+ terminal_result=result,
654
+ next_stage_id=EvalStageId.ARBITER,
655
+ candidate_disposition=EvalCandidateDisposition.REJECTED,
656
+ graph=graph,
657
+ )
658
+
659
+ if result == EvalTerminalResult.ARBITER_CLOSED:
660
+ contract = graph.stage_contracts[stage_id]
661
+ if not contract.may_complete_workflow:
662
+ return _diagnostic_decision(
663
+ code="MF-EVAL-G006",
664
+ summary="only arbiter may complete the compact eval workflow",
665
+ current_stage_id=stage_id,
666
+ terminal_result=result,
667
+ )
668
+ return EvalTransitionDecision(
669
+ outcome_kind=EvalWorkflowOutcomeKind.COMPLETED,
670
+ current_stage_id=stage_id,
671
+ terminal_result=result,
672
+ )
673
+ disposition = (
674
+ EvalCandidateDisposition.REJECTED
675
+ if result == EvalTerminalResult.ARBITER_REJECTED
676
+ else EvalCandidateDisposition.BLOCKED
677
+ )
678
+ return EvalTransitionDecision(
679
+ outcome_kind=EvalWorkflowOutcomeKind.BLOCKED,
680
+ current_stage_id=stage_id,
681
+ terminal_result=result,
682
+ candidate_disposition=disposition,
683
+ )
684
+
685
+
686
+ def compact_eval_workflow_snapshot(
687
+ graph: CompactEvalWorkflowGraph | None = None,
688
+ ) -> dict[str, Any]:
689
+ """Return the deterministic public snapshot for the compact eval graph."""
690
+ graph = graph or default_compact_eval_workflow_graph()
691
+ snapshot: dict[str, Any] = {
692
+ "graph_id": graph.graph_id,
693
+ "omitted_stage_ids": list(OMITTED_COMPACT_EVAL_STAGE_IDS),
694
+ "schema_version": 1,
695
+ "stages": [stage.model_dump(mode="json") for stage in graph.stages],
696
+ "transitions": [
697
+ {
698
+ "current_stage_id": "eval_planner",
699
+ "terminal_result": "PLAN_READY",
700
+ "outcome_kind": "continue",
701
+ "next_stage_id": "eval_builder",
702
+ "candidate_disposition": "none",
703
+ },
704
+ {
705
+ "current_stage_id": "eval_planner",
706
+ "terminal_result": "PLAN_BLOCKED",
707
+ "outcome_kind": "blocked",
708
+ "candidate_disposition": "none",
709
+ },
710
+ {
711
+ "current_stage_id": "eval_builder",
712
+ "terminal_result": "BUILDER_COMPLETE",
713
+ "outcome_kind": "continue",
714
+ "next_stage_id": "eval_checker",
715
+ "candidate_disposition": "none",
716
+ },
717
+ {
718
+ "current_stage_id": "eval_builder",
719
+ "terminal_result": "BUILDER_BLOCKED",
720
+ "outcome_kind": "continue",
721
+ "next_stage_id": "eval_arbiter",
722
+ "candidate_disposition": "blocked",
723
+ },
724
+ {
725
+ "current_stage_id": "eval_checker",
726
+ "terminal_result": "CHECKER_APPROVED",
727
+ "outcome_kind": "continue",
728
+ "next_stage_id": "eval_arbiter",
729
+ "candidate_disposition": "approved",
730
+ },
731
+ {
732
+ "current_stage_id": "eval_checker",
733
+ "terminal_result": "CHECKER_REJECTED",
734
+ "condition": "builder_attempts < 2 and checker_attempts < 2",
735
+ "outcome_kind": "continue",
736
+ "next_stage_id": "eval_builder",
737
+ "candidate_disposition": "rejected",
738
+ },
739
+ {
740
+ "current_stage_id": "eval_checker",
741
+ "terminal_result": "CHECKER_REJECTED",
742
+ "condition": "builder_attempts >= 2 or checker_attempts >= 2",
743
+ "outcome_kind": "continue",
744
+ "next_stage_id": "eval_arbiter",
745
+ "candidate_disposition": "rejected",
746
+ },
747
+ {
748
+ "current_stage_id": "eval_checker",
749
+ "terminal_result": "CHECKER_BLOCKED",
750
+ "outcome_kind": "continue",
751
+ "next_stage_id": "eval_arbiter",
752
+ "candidate_disposition": "blocked",
753
+ },
754
+ {
755
+ "current_stage_id": "eval_arbiter",
756
+ "terminal_result": "ARBITER_CLOSED",
757
+ "outcome_kind": "completed",
758
+ "candidate_disposition": "none",
759
+ },
760
+ {
761
+ "current_stage_id": "eval_arbiter",
762
+ "terminal_result": "ARBITER_REJECTED",
763
+ "outcome_kind": "blocked",
764
+ "candidate_disposition": "rejected",
765
+ },
766
+ {
767
+ "current_stage_id": "eval_arbiter",
768
+ "terminal_result": "ARBITER_BLOCKED",
769
+ "outcome_kind": "blocked",
770
+ "candidate_disposition": "blocked",
771
+ },
772
+ ],
773
+ }
774
+ snapshot["graph_sha256"] = calculate_compact_eval_workflow_sha256(snapshot)
775
+ return snapshot
776
+
777
+
778
+ def calculate_compact_eval_workflow_sha256(snapshot: Mapping[str, Any]) -> str:
779
+ """Hash a compact eval snapshot without its ``graph_sha256`` field."""
780
+ body = dict(snapshot)
781
+ body.pop("graph_sha256", None)
782
+ return hashlib.sha256(_canonical_json_serialize(body).encode("utf-8")).hexdigest()
783
+
784
+
785
+ def canonical_compact_eval_workflow_bytes(
786
+ graph: CompactEvalWorkflowGraph | None = None,
787
+ ) -> bytes:
788
+ """Serialize the compact eval graph snapshot as canonical UTF-8 JSON bytes."""
789
+ snapshot = compact_eval_workflow_snapshot(graph)
790
+ expected = snapshot["graph_sha256"]
791
+ computed = calculate_compact_eval_workflow_sha256(snapshot)
792
+ if computed != expected:
793
+ raise ValueError("compact eval graph fingerprint verification failed")
794
+ return _canonical_json_serialize(snapshot).encode("utf-8")