archforge-optimizer 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- archforge/__init__.py +76 -0
- archforge/__main__.py +10 -0
- archforge/architect.py +442 -0
- archforge/cli.py +881 -0
- archforge/config.py +140 -0
- archforge/config_init.py +150 -0
- archforge/diff.py +206 -0
- archforge/engine.py +444 -0
- archforge/gatekeeper.py +290 -0
- archforge/host/__init__.py +20 -0
- archforge/host/adapters/__init__.py +41 -0
- archforge/host/adapters/base.py +311 -0
- archforge/host/adapters/helpers.py +163 -0
- archforge/host/adapters/langgraph.py +726 -0
- archforge/host/base.py +105 -0
- archforge/host/fake.py +380 -0
- archforge/judge/__init__.py +20 -0
- archforge/judge/base.py +257 -0
- archforge/judge/scripted.py +145 -0
- archforge/lint.py +180 -0
- archforge/llm/__init__.py +65 -0
- archforge/llm/_common.py +94 -0
- archforge/llm/anthropic.py +90 -0
- archforge/llm/base.py +90 -0
- archforge/llm/gemini.py +112 -0
- archforge/llm/groq.py +63 -0
- archforge/llm/openai.py +63 -0
- archforge/llm/scripted.py +134 -0
- archforge/middleware.py +181 -0
- archforge/models.py +435 -0
- archforge/mutate.py +214 -0
- archforge/otel.py +613 -0
- archforge/runlog.py +103 -0
- archforge/runner.py +153 -0
- archforge/spec_builder.py +126 -0
- archforge/stores/__init__.py +22 -0
- archforge/stores/_jsonl.py +81 -0
- archforge/stores/attempt_store.py +161 -0
- archforge/stores/spec_store.py +188 -0
- archforge/stores/trace_store.py +42 -0
- archforge/suite.py +248 -0
- archforge/userconfig.py +144 -0
- archforge_optimizer-0.1.0.dist-info/METADATA +420 -0
- archforge_optimizer-0.1.0.dist-info/RECORD +47 -0
- archforge_optimizer-0.1.0.dist-info/WHEEL +4 -0
- archforge_optimizer-0.1.0.dist-info/entry_points.txt +2 -0
- archforge_optimizer-0.1.0.dist-info/licenses/LICENSE +21 -0
archforge/engine.py
ADDED
|
@@ -0,0 +1,444 @@
|
|
|
1
|
+
"""The evolve engine — orchestrates one P-E-C cycle and the loop.
|
|
2
|
+
|
|
3
|
+
This is the "nervous system" that wires the four organs (Architect, SuiteRunner,
|
|
4
|
+
Gatekeeper) plus the stores into a coherent run-over-run loop:
|
|
5
|
+
|
|
6
|
+
incumbent = SpecStore.active()
|
|
7
|
+
result = architect.next_attempt(incumbent, worst_task_scores, attempt_store)
|
|
8
|
+
if not result.proposed -> record, maybe plateau, continue
|
|
9
|
+
candidate_spec_id = spec_store.commit(result.proposal.candidate, parent=incumbent)
|
|
10
|
+
attempt = Attempt{candidate_spec_id, parent=incumbent, change, verdict=PROMOTED}
|
|
11
|
+
(PROMOTED is the *initial* verdict; the Gatekeeper flips it)
|
|
12
|
+
attempt_id = attempt_store.append(attempt)
|
|
13
|
+
inc_run = suite_runner.run_suite(incumbent_suite, R) # baseline, cacheable
|
|
14
|
+
cand_run = suite_runner.run_suite(candidate_spec, R)
|
|
15
|
+
decision = gatekeeper.decide(attempt_id, cand_run, inc_run)
|
|
16
|
+
applied = gatekeeper.apply_decision(decision)
|
|
17
|
+
record CycleResult
|
|
18
|
+
|
|
19
|
+
The engine is constructed with explicit components (host/judge/architect) +
|
|
20
|
+
stores, so tests inject fakes directly and a CLI shim wires the real or
|
|
21
|
+
scripted variants per `--provider` config. Departmental rules honored:
|
|
22
|
+
E3 budget caps abort cleanly (budget_guard raises CycleAborted; incumbent untouched)
|
|
23
|
+
E8 K consecutive no-promotion cycles -> plateau (loop stops, status surfaced)
|
|
24
|
+
I1 only the Gatekeeper moves `active`; the engine never touches it
|
|
25
|
+
I5 every run is stamped with (suite_id, rubric_id); the Gatekeeper checks it
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
from dataclasses import dataclass, field
|
|
31
|
+
from typing import Callable
|
|
32
|
+
|
|
33
|
+
from archforge import userconfig as ucfg
|
|
34
|
+
import archforge.models as m
|
|
35
|
+
from archforge.architect import ArchitectProtocol, ArchitectResult
|
|
36
|
+
from archforge.gatekeeper import Action, Decision, Gatekeeper
|
|
37
|
+
from archforge.judge.base import JudgeProtocol
|
|
38
|
+
from archforge.stores import AttemptStore, SpecStore, TraceStore
|
|
39
|
+
from archforge.suite import Suite, SuiteRunner, SuiteRun
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
# --------------------------------------------------------------------------- #
|
|
43
|
+
# Outcomes
|
|
44
|
+
# --------------------------------------------------------------------------- #
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class CycleAborted(Exception):
|
|
48
|
+
"""Raised when a budget cap is hit mid-cycle (E3). The incumbent is untouched."""
|
|
49
|
+
|
|
50
|
+
def __init__(self, reason: str, partial: "CycleResult | None" = None) -> None:
|
|
51
|
+
super().__init__(reason)
|
|
52
|
+
self.reason = reason
|
|
53
|
+
self.partial = partial
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
@dataclass
|
|
57
|
+
class CycleResult:
|
|
58
|
+
"""One cycle's outcome, for the report + plateau accounting."""
|
|
59
|
+
|
|
60
|
+
cycle: int
|
|
61
|
+
attempted: bool # did the Architect propose a candidate?
|
|
62
|
+
architect_status: ArchitectResult | None = None
|
|
63
|
+
decision: Decision | None = None
|
|
64
|
+
applied_attempt_id: str | None = None
|
|
65
|
+
incumbent_mean: float | None = None
|
|
66
|
+
candidate_mean: float | None = None
|
|
67
|
+
tokens: int = 0
|
|
68
|
+
latency_ms: float = 0.0 # summed wall-clock this cycle (inc + cand)
|
|
69
|
+
note: str = ""
|
|
70
|
+
|
|
71
|
+
@property
|
|
72
|
+
def promoted(self) -> bool:
|
|
73
|
+
return self.decision is not None and self.decision.action is Action.AUTO_PROMOTE
|
|
74
|
+
|
|
75
|
+
@property
|
|
76
|
+
def queued(self) -> bool:
|
|
77
|
+
return self.decision is not None and self.decision.action is Action.QUEUE_HUMAN
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
@dataclass
|
|
81
|
+
class LoopResult:
|
|
82
|
+
"""The aggregate outcome of `evolve_loop`."""
|
|
83
|
+
|
|
84
|
+
cycles_run: int = 0
|
|
85
|
+
promotions: int = 0
|
|
86
|
+
queued: int = 0
|
|
87
|
+
plateaued: bool = False
|
|
88
|
+
aborted: bool = False
|
|
89
|
+
abort_reason: str = ""
|
|
90
|
+
results: list[CycleResult] = field(default_factory=list)
|
|
91
|
+
final_incumbent_id: str | None = None
|
|
92
|
+
final_incumbent_mean: float | None = None
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
@dataclass
|
|
96
|
+
class CycleCtx:
|
|
97
|
+
"""Side-channel context the ``on_cycle`` hook needs but ``CycleResult`` lacks.
|
|
98
|
+
|
|
99
|
+
``CycleResult`` carries the decision + aggregate means + cost, NOT the two
|
|
100
|
+
Specs (needed for the mutation diff) nor the full ``SuiteRun`` objects (needed
|
|
101
|
+
for per-task + rubric-dim rendering) nor the ``Change`` record. All four live
|
|
102
|
+
in ``evolve_cycle``'s scope; this is the bag the hook reads. ``CycleResult`` is
|
|
103
|
+
passed alongside it (the hook gets ``(result, ctx)``) so the callback can see
|
|
104
|
+
the persisted attempt's identity + the verdict, not just the inputs.
|
|
105
|
+
"""
|
|
106
|
+
|
|
107
|
+
parent_spec: m.Spec
|
|
108
|
+
candidate_spec: m.Spec
|
|
109
|
+
cand_run: SuiteRun
|
|
110
|
+
inc_run: SuiteRun
|
|
111
|
+
change: m.Change
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
@dataclass
|
|
115
|
+
class DeployCtx:
|
|
116
|
+
"""Context for the ``on_deploy`` hook — what an AUTO_PROMOTE just shipped.
|
|
117
|
+
|
|
118
|
+
Richer than ``on_promote``'s single-Spec payload: the parent (for the
|
|
119
|
+
envelope's lineage + diff), the ``Decision`` (rule + margin), the two
|
|
120
|
+
``SuiteRun``s (the scores block), and the cycle index — everything a
|
|
121
|
+
``build_optimized_envelope`` needs. ``on_promote`` stays as-is for back-compat;
|
|
122
|
+
this is the new richer seam the CLI wires for the unified deploy artifact.
|
|
123
|
+
"""
|
|
124
|
+
|
|
125
|
+
parent: m.Spec
|
|
126
|
+
decision: Decision
|
|
127
|
+
cand_run: SuiteRun
|
|
128
|
+
inc_run: SuiteRun
|
|
129
|
+
promoted_at_cycle: int
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
# --------------------------------------------------------------------------- #
|
|
133
|
+
# The engine
|
|
134
|
+
# --------------------------------------------------------------------------- #
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
@dataclass
|
|
138
|
+
class EngineConfig:
|
|
139
|
+
"""Tunables for the loop (Beyond thresholds, which live in m.Thresholds)."""
|
|
140
|
+
|
|
141
|
+
# Tunable defaults resolve lazily from archforge.userconfig (the active config),
|
|
142
|
+
# so importing the engine — and even materializing an EngineConfig() default —
|
|
143
|
+
# works BEFORE `init` has run (no from-import at module load).
|
|
144
|
+
max_cycles: int = field(default_factory=lambda: ucfg.get("DEFAULT_MAX_CYCLES"))
|
|
145
|
+
max_tokens_per_cycle: int | None = field(default_factory=lambda: ucfg.get("DEFAULT_MAX_TOKENS_PER_CYCLE")) # E3 (None=∞)
|
|
146
|
+
max_tokens_total: int | None = field(default_factory=lambda: ucfg.get("DEFAULT_MAX_TOKENS_TOTAL"))
|
|
147
|
+
repeats: int = field(default_factory=lambda: ucfg.get("DEFAULT_REPEATS")) # R (adaptive raises it)
|
|
148
|
+
plateau_cycles: int = field(default_factory=lambda: ucfg.get("DEFAULT_PLATEAU_CYCLES")) # K (E8)
|
|
149
|
+
# per-cycle wall-clock cap (ms). Defaults to None (opt-in) — closing the budget
|
|
150
|
+
# hole for non-LLM-heavy pipelines whose cost is *time* not tokens (a retriever/
|
|
151
|
+
# tool/rule node costs ~0 tokens). Mirrors max_tokens_per_cycle's per-cycle
|
|
152
|
+
# shape + none-means-∞ contract; timed post-eval (same as the token cap), not
|
|
153
|
+
# mid-step. No total-wall cap (YAGNI; the loop is already bounded by max_cycles
|
|
154
|
+
# + this per-cycle cap).
|
|
155
|
+
max_wall_ms_per_cycle: float | None = field(default_factory=lambda: ucfg.get("DEFAULT_MAX_WALL_MS_PER_CYCLE"))
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
class Engine:
|
|
159
|
+
"""Runs Propose-Evaluate-Commit cycles and loops over them.
|
|
160
|
+
|
|
161
|
+
Holds the long-lived stores + judge + runner + gatekeeper; the host and
|
|
162
|
+
architect are supplied per construction (real or scripted).
|
|
163
|
+
"""
|
|
164
|
+
|
|
165
|
+
def __init__(
|
|
166
|
+
self,
|
|
167
|
+
*,
|
|
168
|
+
host: object, # HostMAS (typed loosely to avoid import cycle)
|
|
169
|
+
judge: JudgeProtocol,
|
|
170
|
+
architect: ArchitectProtocol,
|
|
171
|
+
spec_store: SpecStore,
|
|
172
|
+
attempt_store: AttemptStore,
|
|
173
|
+
trace_store: TraceStore,
|
|
174
|
+
suite: Suite,
|
|
175
|
+
thresholds: m.Thresholds | None = None,
|
|
176
|
+
config: EngineConfig | None = None,
|
|
177
|
+
on_promote: Callable[[m.Spec], None] | None = None,
|
|
178
|
+
on_cycle: Callable[[CycleResult, CycleCtx], None] | None = None,
|
|
179
|
+
on_deploy: Callable[[m.Spec, DeployCtx], None] | None = None,
|
|
180
|
+
) -> None:
|
|
181
|
+
self._host = host
|
|
182
|
+
self._judge = judge
|
|
183
|
+
self._architect = architect
|
|
184
|
+
self._specs = spec_store
|
|
185
|
+
self._attempts = attempt_store
|
|
186
|
+
self._traces = trace_store
|
|
187
|
+
self._suite = suite
|
|
188
|
+
self._th = thresholds or m.Thresholds()
|
|
189
|
+
self._cfg = config or EngineConfig()
|
|
190
|
+
self._on_promote = on_promote
|
|
191
|
+
# Opt-in per-cycle + deploy hooks (default None -> byte-identical when
|
|
192
|
+
# absent, so Engines built without them — existing tests, embedders that
|
|
193
|
+
# only use `on_promote` — are unchanged). `on_cycle` fires on every
|
|
194
|
+
# ATTEMPTED cycle (promote/queue/discard) AFTER the result is persisted,
|
|
195
|
+
# carrying the richer `CycleCtx` side channel (specs + SuiteRuns + Change).
|
|
196
|
+
# `on_deploy` fires ONLY on AUTO_PROMOTE alongside `on_promote` (kept as-is
|
|
197
|
+
# for back-compat) — the new richer seam the CLI wires to write the unified
|
|
198
|
+
# `optimized.json`. The CLI wires `on_deploy`, leaving `on_promote` to
|
|
199
|
+
# embedders/tests; the engine fires both when both are present.
|
|
200
|
+
self._on_cycle = on_cycle
|
|
201
|
+
self._on_deploy = on_deploy
|
|
202
|
+
self._runner = SuiteRunner(host, judge, trace_store,
|
|
203
|
+
epsilon=self._th.unrunnable_frac,
|
|
204
|
+
judge_retries=ucfg.get("DEFAULT_JUDGE_RETRIES"))
|
|
205
|
+
self._gatekeeper = Gatekeeper(spec_store, attempt_store, thresholds=self._th)
|
|
206
|
+
# baseline cache: incumbent_suite_run keyed by spec_id (stable until rubric changes)
|
|
207
|
+
self._baseline_cache: dict[str, SuiteRun] = {}
|
|
208
|
+
self._tokens_total = 0
|
|
209
|
+
# worst-task scores cache for the Architect (last incumbent's run scores)
|
|
210
|
+
self._last_incumbent_scores: list[m.RunScore] = []
|
|
211
|
+
|
|
212
|
+
# ----------------------------------------------------------------- one cycle
|
|
213
|
+
def evolve_cycle(self, cycle: int = 0) -> CycleResult:
|
|
214
|
+
"""Run exactly one P-E-C cycle from the active incumbent."""
|
|
215
|
+
incumbent = self._specs.active() # raises NoActiveSpecError if none
|
|
216
|
+
arch = self._architect.next_attempt(
|
|
217
|
+
incumbent, self._last_incumbent_scores, attempt_store=self._attempts,
|
|
218
|
+
)
|
|
219
|
+
|
|
220
|
+
if not arch.proposed:
|
|
221
|
+
return CycleResult(cycle=cycle, attempted=False, architect_status=arch,
|
|
222
|
+
note=arch.note or _status_note(arch))
|
|
223
|
+
|
|
224
|
+
proposal = arch.proposal
|
|
225
|
+
candidate = proposal.candidate
|
|
226
|
+
candidate_spec_id = self._specs.commit(candidate, parent_spec_id=incumbent.spec_id,
|
|
227
|
+
status=m.SpecStatus.CANDIDATE)
|
|
228
|
+
attempt = m.Attempt(
|
|
229
|
+
candidate_spec_id=candidate_spec_id, parent_spec_id=incumbent.spec_id,
|
|
230
|
+
change=proposal.change, verdict=m.Verdict.PROMOTED,
|
|
231
|
+
)
|
|
232
|
+
attempt_id = self._attempts.append(attempt)
|
|
233
|
+
|
|
234
|
+
# Evaluate both, same suite + rubric (I5). Baseline is cacheable.
|
|
235
|
+
inc_run = self._baseline_for(incumbent)
|
|
236
|
+
self._last_incumbent_scores = list(inc_run.scores)
|
|
237
|
+
candidate_spec = self._specs.get(candidate_spec_id)
|
|
238
|
+
cand_run = self._bounded_run(candidate_spec, tag="candidate")
|
|
239
|
+
|
|
240
|
+
self._check_budget_cycle(
|
|
241
|
+
cand_run.tokens + inc_run.tokens,
|
|
242
|
+
cand_run.latency_ms + inc_run.latency_ms,
|
|
243
|
+
)
|
|
244
|
+
decision = self._gatekeeper.decide(attempt_id, cand_run, inc_run)
|
|
245
|
+
applied = self._gatekeeper.apply_decision(decision)
|
|
246
|
+
|
|
247
|
+
# Persist the scored result so the human-facing surfaces (status, report,
|
|
248
|
+
# Approval Queue) show the real delta + cost without re-running the suite.
|
|
249
|
+
# `tokens` carries only the candidate-side marginal cost — the incumbent
|
|
250
|
+
# baseline is a shared, cacheable cost accounted for at the loop level
|
|
251
|
+
# (CycleResult.tokens), never double-counted onto a single attempt.
|
|
252
|
+
self._attempts.set_result(
|
|
253
|
+
attempt_id,
|
|
254
|
+
m.SuiteResult(
|
|
255
|
+
mean=cand_run.mean,
|
|
256
|
+
margin_vs_incumbent=decision.margin,
|
|
257
|
+
repeats=cand_run.repeats,
|
|
258
|
+
unrunnable=cand_run.unrunnable,
|
|
259
|
+
rubric_id=cand_run.rubric_id,
|
|
260
|
+
suite_id=cand_run.suite_id,
|
|
261
|
+
tokens=cand_run.tokens,
|
|
262
|
+
),
|
|
263
|
+
)
|
|
264
|
+
|
|
265
|
+
result = CycleResult(
|
|
266
|
+
cycle=cycle, attempted=True, architect_status=arch, decision=decision,
|
|
267
|
+
applied_attempt_id=applied.attempt_id,
|
|
268
|
+
incumbent_mean=inc_run.mean,
|
|
269
|
+
candidate_mean=cand_run.mean,
|
|
270
|
+
tokens=cand_run.tokens + inc_run.tokens,
|
|
271
|
+
latency_ms=cand_run.latency_ms + inc_run.latency_ms,
|
|
272
|
+
note=decision.reason,
|
|
273
|
+
)
|
|
274
|
+
# `on_cycle` — the per-cycle surface (improvements #2/#3/#5). Fires AFTER
|
|
275
|
+
# the result is persisted (the attempt + suite_result are visible) on every
|
|
276
|
+
# attempted cycle; the callback owns rendering (the CLI's card + run-log).
|
|
277
|
+
# `CycleCtx` carries the two Specs (for the mutation diff), the two
|
|
278
|
+
# SuiteRuns (rubric dims + per-task), and the Change record — `CycleResult`
|
|
279
|
+
# alone lacks them. Not fired for `attempted=false` cycles (nothing to
|
|
280
|
+
# diff). Fires BEFORE `on_promote`/`on_deploy` so the cycle narrative prints
|
|
281
|
+
# first, then the deploy notice — the human-readable order.
|
|
282
|
+
if self._on_cycle is not None:
|
|
283
|
+
self._on_cycle(
|
|
284
|
+
result,
|
|
285
|
+
CycleCtx(parent_spec=incumbent, candidate_spec=candidate_spec,
|
|
286
|
+
cand_run=cand_run, inc_run=inc_run, change=proposal.change),
|
|
287
|
+
)
|
|
288
|
+
# Deploy hook (Tier-2 sidecar auto-sync): an opt-in callback fired ONLY on
|
|
289
|
+
# an AUTO_PROMOTE — i.e. the candidate just became the active incumbent.
|
|
290
|
+
# The callback owns any side effect (e.g. export_spec_sidecar → a JSON file
|
|
291
|
+
# the MAS overlays onto its config consts so the win reaches production
|
|
292
|
+
# without the Forge on the hot path). Not fired for QUEUE_HUMAN (a human
|
|
293
|
+
# gate) or a discard. The engine does no I/O; the callback does.
|
|
294
|
+
if self._on_promote is not None and decision.action is Action.AUTO_PROMOTE:
|
|
295
|
+
self._on_promote(self._specs.get(candidate_spec_id))
|
|
296
|
+
# `on_deploy` — the richer deploy seam (improvement #4): fires on the same
|
|
297
|
+
# AUTO_PROMOTE as `on_promote` but carries the parent + Decision + both
|
|
298
|
+
# SuiteRuns + the cycle index, so the callback (the CLI's `_on_deploy`)
|
|
299
|
+
# can write the unified `optimized.json` envelope without re-reading the
|
|
300
|
+
# stores. `on_promote` stays for back-compat (embedders/tests); both fire
|
|
301
|
+
# when both are wired — the CLI wires only `on_deploy`.
|
|
302
|
+
if self._on_deploy is not None and decision.action is Action.AUTO_PROMOTE:
|
|
303
|
+
self._on_deploy(
|
|
304
|
+
self._specs.get(candidate_spec_id),
|
|
305
|
+
DeployCtx(parent=incumbent, decision=decision, cand_run=cand_run,
|
|
306
|
+
inc_run=inc_run, promoted_at_cycle=cycle),
|
|
307
|
+
)
|
|
308
|
+
# Seed the baseline cache with the just-promoted candidate's run so the
|
|
309
|
+
# NEXT cycle's ``_baseline_for(new_incumbent)`` is a cache HIT (not a fresh
|
|
310
|
+
# suite run). Without this, a candidate promoted at mean M is re-scored on
|
|
311
|
+
# the next cycle as the new incumbent — Judge run-to-run variance flips M
|
|
312
|
+
# (e.g. 1.000 -> 0.800), discarding the score the Gatekeeper promoted on
|
|
313
|
+
# and injecting noise into every margin thereafter. The promoted run is the
|
|
314
|
+
# authoritative baseline: it was scored under the SAME suite/rubric (I5),
|
|
315
|
+
# validated below by the ``(rubric_id, suite_id)`` guard ``_baseline_for``
|
|
316
|
+
# applies on read. Don't seed a QUEUE_HUMAN (the human may re-ask a fresh
|
|
317
|
+
# score on approval) or a discard (the candidate isn't the incumbent).
|
|
318
|
+
if decision.action is Action.AUTO_PROMOTE:
|
|
319
|
+
key = candidate_spec_id or candidate.compute_spec_id()
|
|
320
|
+
self._baseline_cache[key] = cand_run
|
|
321
|
+
return result
|
|
322
|
+
|
|
323
|
+
# ----------------------------------------------------------------- the loop
|
|
324
|
+
def evolve_loop(self) -> LoopResult:
|
|
325
|
+
"""Repeat `evolve_cycle` until budget cap or plateau (E3/E8)."""
|
|
326
|
+
out = LoopResult()
|
|
327
|
+
plateau_streak = 0
|
|
328
|
+
# spec_id -> mean, for each spec promoted THIS run. A just-promoted
|
|
329
|
+
# incumbent's baseline may not yet be in `_baseline_cache` (its run was the
|
|
330
|
+
# candidate's, cached iff a later cycle baselines it); this map lets the
|
|
331
|
+
# loop tail surface `final_incumbent_mean` from the promoting cycle's
|
|
332
|
+
# `candidate_mean` WITHOUT re-scoring. (Trivial-bug fix: it was never
|
|
333
|
+
# populated, so the summary always read `final_mean=-`.)
|
|
334
|
+
promoted_means: dict[str, float] = {}
|
|
335
|
+
for i in range(self._cfg.max_cycles):
|
|
336
|
+
if self._over_total_budget():
|
|
337
|
+
out.aborted = True
|
|
338
|
+
out.abort_reason = "total token budget reached"
|
|
339
|
+
break
|
|
340
|
+
try:
|
|
341
|
+
r = self.evolve_cycle(cycle=i)
|
|
342
|
+
except CycleAborted as exc:
|
|
343
|
+
out.aborted = True
|
|
344
|
+
out.abort_reason = exc.reason
|
|
345
|
+
if exc.partial is not None:
|
|
346
|
+
out.results.append(exc.partial)
|
|
347
|
+
break
|
|
348
|
+
out.results.append(r)
|
|
349
|
+
out.cycles_run += 1
|
|
350
|
+
if r.promoted:
|
|
351
|
+
out.promotions += 1
|
|
352
|
+
plateau_streak = 0
|
|
353
|
+
# The candidate just became the active incumbent. Record its mean
|
|
354
|
+
# so the loop tail can surface `final_incumbent_mean` WITHOUT
|
|
355
|
+
# re-scoring it: its `cand_run` was the candidate run, NOT cached
|
|
356
|
+
# as a baseline, so `_baseline_for(<new active>)` would re-score
|
|
357
|
+
# it (an unwanted extra suite run). This map is the zero-scoring
|
|
358
|
+
# path for the promoted case.
|
|
359
|
+
if r.candidate_mean is not None and self._specs.active_id() is not None:
|
|
360
|
+
promoted_means[self._specs.active_id()] = r.candidate_mean
|
|
361
|
+
elif r.queued:
|
|
362
|
+
out.queued += 1
|
|
363
|
+
plateau_streak = 0 # a queued change is forward progress
|
|
364
|
+
elif not r.attempted:
|
|
365
|
+
plateau_streak += 1
|
|
366
|
+
else:
|
|
367
|
+
plateau_streak += 1 # a discard also counts toward plateau
|
|
368
|
+
|
|
369
|
+
self._tokens_total += r.tokens
|
|
370
|
+
if plateau_streak >= self._cfg.plateau_cycles:
|
|
371
|
+
out.plateaued = True
|
|
372
|
+
break
|
|
373
|
+
out.final_incumbent_id = self._specs.active_id()
|
|
374
|
+
# `final_incumbent_mean` — trivial-bug fix (was never populated → always
|
|
375
|
+
# `-`). Prefer the just-promoted mean (zero scoring); else the baseline
|
|
376
|
+
# cache, but ONLY if already scored this run (never force a fresh suite
|
|
377
|
+
# run just to fill a summary field — a no-promotion plateau loop that
|
|
378
|
+
# never proposed has no scored mean, and `-` is the honest value then).
|
|
379
|
+
final_id = out.final_incumbent_id
|
|
380
|
+
if final_id is not None:
|
|
381
|
+
if final_id in promoted_means:
|
|
382
|
+
out.final_incumbent_mean = promoted_means[final_id]
|
|
383
|
+
else:
|
|
384
|
+
cached = self._baseline_cache.get(final_id)
|
|
385
|
+
if (cached is not None
|
|
386
|
+
and (cached.rubric_id, cached.suite_id)
|
|
387
|
+
== (self._suite.rubric_id, self._suite.suite_id)):
|
|
388
|
+
out.final_incumbent_mean = cached.mean
|
|
389
|
+
return out
|
|
390
|
+
|
|
391
|
+
# ----------------------------------------------------------------- helpers
|
|
392
|
+
def _baseline_for(self, incumbent: m.Spec) -> SuiteRun:
|
|
393
|
+
"""The incumbent's suite run, cached until the rubric/suite changes."""
|
|
394
|
+
key = incumbent.spec_id or incumbent.compute_spec_id()
|
|
395
|
+
if key in self._baseline_cache:
|
|
396
|
+
cached = self._baseline_cache[key]
|
|
397
|
+
# invalidate if rubric/suite drifted since cache (I5 — same geometry)
|
|
398
|
+
if (cached.rubric_id, cached.suite_id) == (self._suite.rubric_id, self._suite.suite_id):
|
|
399
|
+
return cached
|
|
400
|
+
run = self._bounded_run(incumbent, tag="incumbent")
|
|
401
|
+
self._baseline_cache[key] = run
|
|
402
|
+
return run
|
|
403
|
+
|
|
404
|
+
def _bounded_run(self, spec: m.Spec, *, tag: str) -> SuiteRun:
|
|
405
|
+
return self._runner.run_suite(spec, self._suite, R=self._cfg.repeats)
|
|
406
|
+
|
|
407
|
+
def _check_budget_cycle(self, cycle_tokens: int, cycle_latency_ms: float = 0.0) -> None:
|
|
408
|
+
cap = self._cfg.max_tokens_per_cycle
|
|
409
|
+
if cap is not None and cycle_tokens > cap:
|
|
410
|
+
raise CycleAborted(
|
|
411
|
+
f"per-cycle token cap exceeded ({cycle_tokens} > {cap}; tag budget)",
|
|
412
|
+
)
|
|
413
|
+
# Wall-clock cap: the cost fix for non-LLM-heavy pipelines (a retriever/
|
|
414
|
+
# tool/rule node costs ~0 tokens, so it sails past the token cap). Same
|
|
415
|
+
# post-eval timing + > contract as the token cap — incumbent untouched.
|
|
416
|
+
wall_cap = self._cfg.max_wall_ms_per_cycle
|
|
417
|
+
if wall_cap is not None and cycle_latency_ms > wall_cap:
|
|
418
|
+
raise CycleAborted(
|
|
419
|
+
f"per-cycle wall-clock cap exceeded "
|
|
420
|
+
f"({cycle_latency_ms:.1f}ms > {wall_cap}ms)",
|
|
421
|
+
)
|
|
422
|
+
|
|
423
|
+
def _over_total_budget(self) -> bool:
|
|
424
|
+
# A total token budget is a *ceiling*: stop at or before reaching it, not
|
|
425
|
+
# only after crossing it. `>=` makes a 0 budget mean "spend nothing" (the
|
|
426
|
+
# loop aborts before the first cycle) and keeps a real budget from being
|
|
427
|
+
# overrun by a single cycle (> would let one cycle blow straight past it).
|
|
428
|
+
cap = self._cfg.max_tokens_total
|
|
429
|
+
return cap is not None and self._tokens_total >= cap
|
|
430
|
+
|
|
431
|
+
|
|
432
|
+
def _status_note(arch: ArchitectResult) -> str:
|
|
433
|
+
if arch.status == "lint_rejected":
|
|
434
|
+
reasons = "; ".join(e.message for e in arch.rejected_reasons) or "malformed"
|
|
435
|
+
return f"candidate rejected by linter: {reasons}"
|
|
436
|
+
if arch.status == "plateau":
|
|
437
|
+
return arch.note or "architect plateaued (no fresh proposal)"
|
|
438
|
+
return arch.note or f"architect status={arch.status}"
|
|
439
|
+
|
|
440
|
+
|
|
441
|
+
__all__ = [
|
|
442
|
+
"Engine", "EngineConfig", "CycleResult", "LoopResult", "CycleAborted",
|
|
443
|
+
"CycleCtx", "DeployCtx",
|
|
444
|
+
]
|