archforge-optimizer 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- archforge/__init__.py +76 -0
- archforge/__main__.py +10 -0
- archforge/architect.py +442 -0
- archforge/cli.py +881 -0
- archforge/config.py +140 -0
- archforge/config_init.py +150 -0
- archforge/diff.py +206 -0
- archforge/engine.py +444 -0
- archforge/gatekeeper.py +290 -0
- archforge/host/__init__.py +20 -0
- archforge/host/adapters/__init__.py +41 -0
- archforge/host/adapters/base.py +311 -0
- archforge/host/adapters/helpers.py +163 -0
- archforge/host/adapters/langgraph.py +726 -0
- archforge/host/base.py +105 -0
- archforge/host/fake.py +380 -0
- archforge/judge/__init__.py +20 -0
- archforge/judge/base.py +257 -0
- archforge/judge/scripted.py +145 -0
- archforge/lint.py +180 -0
- archforge/llm/__init__.py +65 -0
- archforge/llm/_common.py +94 -0
- archforge/llm/anthropic.py +90 -0
- archforge/llm/base.py +90 -0
- archforge/llm/gemini.py +112 -0
- archforge/llm/groq.py +63 -0
- archforge/llm/openai.py +63 -0
- archforge/llm/scripted.py +134 -0
- archforge/middleware.py +181 -0
- archforge/models.py +435 -0
- archforge/mutate.py +214 -0
- archforge/otel.py +613 -0
- archforge/runlog.py +103 -0
- archforge/runner.py +153 -0
- archforge/spec_builder.py +126 -0
- archforge/stores/__init__.py +22 -0
- archforge/stores/_jsonl.py +81 -0
- archforge/stores/attempt_store.py +161 -0
- archforge/stores/spec_store.py +188 -0
- archforge/stores/trace_store.py +42 -0
- archforge/suite.py +248 -0
- archforge/userconfig.py +144 -0
- archforge_optimizer-0.1.0.dist-info/METADATA +420 -0
- archforge_optimizer-0.1.0.dist-info/RECORD +47 -0
- archforge_optimizer-0.1.0.dist-info/WHEEL +4 -0
- archforge_optimizer-0.1.0.dist-info/entry_points.txt +2 -0
- archforge_optimizer-0.1.0.dist-info/licenses/LICENSE +21 -0
archforge/cli.py
ADDED
|
@@ -0,0 +1,881 @@
|
|
|
1
|
+
"""ArchForge command-line interface — the Forge.
|
|
2
|
+
|
|
3
|
+
Wires the four organs (Architect, SuiteRunner, Judge, Gatekeeper) plus the
|
|
4
|
+
filesystem stores into the runnable surface the user actually touches:
|
|
5
|
+
|
|
6
|
+
archforge-optimizer lint <spec.json> validate a Spec
|
|
7
|
+
archforge-optimizer evolve [--root R] [--seed S] one Propose-Evaluate-Commit cycle
|
|
8
|
+
archforge-optimizer evolve-loop [...] repeat until budget cap or plateau
|
|
9
|
+
archforge-optimizer approve [<id>...|--all] drain the structural-change queue
|
|
10
|
+
archforge-optimizer reject <id> reject a queued change (active kept)
|
|
11
|
+
archforge-optimizer status incumbent Spec + lineage + counts
|
|
12
|
+
archforge-optimizer report aggregate deltas across attempts
|
|
13
|
+
|
|
14
|
+
Provider seam (the single place "real vs fake" lives at the CLI):
|
|
15
|
+
* `--provider scripted` (default) builds the zero-cost fakes — `FakeHostMAS`,
|
|
16
|
+
`ScriptedJudge`, `ScriptedArchitect` — so `evolve` is runnable end-to-end
|
|
17
|
+
with no LLM. Left unconfigured, the scripted architect simply plateaus
|
|
18
|
+
(it has no proposal to make); a deterministic *promotion* needs the organs
|
|
19
|
+
pre-configured, which is exactly what an embedding test supplies via
|
|
20
|
+
`components=...` (see `tests/integration/test_cli.py`).
|
|
21
|
+
* `--provider anthropic|openai|groq|gemini` wires a real `LLMClient` and the
|
|
22
|
+
real `Architect` + `Judge` over it — one provider swap, same organs.
|
|
23
|
+
|
|
24
|
+
`_ensure_incumbent` bootstraps the root incumbent from `--seed <spec.json>`
|
|
25
|
+
zero-LLM (commit as INCUMBENT + set_active) so the very first `evolve` has an
|
|
26
|
+
active spec to mutate. Approval/rollback keep `active` moving only through the
|
|
27
|
+
Gatekeeper (invariant I1); the CLI never mutates the pointer itself.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
from __future__ import annotations
|
|
31
|
+
|
|
32
|
+
import argparse
|
|
33
|
+
import importlib
|
|
34
|
+
import json
|
|
35
|
+
import os
|
|
36
|
+
import sys
|
|
37
|
+
import warnings
|
|
38
|
+
from dataclasses import dataclass
|
|
39
|
+
from pathlib import Path
|
|
40
|
+
|
|
41
|
+
import archforge.models as m
|
|
42
|
+
from archforge.architect import Architect, ArchitectProtocol, ScriptedArchitect
|
|
43
|
+
from archforge.diff import format_diff, spec_diff
|
|
44
|
+
from archforge.engine import (
|
|
45
|
+
CycleCtx, CycleResult, DeployCtx, Engine, EngineConfig, LoopResult,
|
|
46
|
+
)
|
|
47
|
+
from archforge.gatekeeper import Gatekeeper
|
|
48
|
+
from archforge.host.base import HostMAS, Task
|
|
49
|
+
from archforge.host.fake import FakeHostMAS
|
|
50
|
+
from archforge.judge import ScriptedJudge
|
|
51
|
+
from archforge.judge.base import Judge, JudgeProtocol, default_rubric
|
|
52
|
+
from archforge.lint import lint
|
|
53
|
+
from archforge.runlog import RunLog
|
|
54
|
+
from archforge.stores import AttemptStore, SpecStore, TraceStore
|
|
55
|
+
from archforge.suite import Suite, load_suite_file
|
|
56
|
+
from archforge.config_init import archforge_config_text, env_example_text, _DEFAULT_SUITE_JSON
|
|
57
|
+
|
|
58
|
+
# System config (provider roster, CLI fixtures, PROG, .env loader) — single source
|
|
59
|
+
# in archforge.config. The TUNABLE defaults (tau/delta/repeats/provider/models/…)
|
|
60
|
+
# are NOT imported here; they resolve lazily from archforge.userconfig (the active
|
|
61
|
+
# project config made by `init`) at use time, so importing+running the CLI works
|
|
62
|
+
# before `init` has created .archforge/archforge.py and so an edit takes effect on
|
|
63
|
+
# the next run. See archforge/userconfig.py.
|
|
64
|
+
from archforge import userconfig as ucfg
|
|
65
|
+
from archforge.config import (
|
|
66
|
+
ALL_PROVIDERS as _PROVIDERS,
|
|
67
|
+
DEFAULT_SUITE_ID, DEFAULT_TASK_ID, DEFAULT_TASK_INPUT,
|
|
68
|
+
PROG, load_env,
|
|
69
|
+
)
|
|
70
|
+
from archforge.userconfig import ConfigNotInitialized
|
|
71
|
+
|
|
72
|
+
# Fixed run-state / config-discovery dir (a system path, independent of the
|
|
73
|
+
# tunable DEFAULT_ROOT_DIR which the embedder API reads via archforge.userconfig).
|
|
74
|
+
_DEFAULT_ROOT = ".archforge"
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
# --------------------------------------------------------------------------- #
|
|
78
|
+
# Injectable runtime organs — the test/embedding seam
|
|
79
|
+
# --------------------------------------------------------------------------- #
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
@dataclass
|
|
83
|
+
class Components:
|
|
84
|
+
"""The four organs the Engine runs, injectable so a test/embedding can
|
|
85
|
+
supply pre-configured fakes (a scripted architect with a queued proposal +
|
|
86
|
+
a scripted judge with per-spec aggregates → a deterministic promotion).
|
|
87
|
+
|
|
88
|
+
When `main(..., components=None)` the CLI builds defaults per `--provider`:
|
|
89
|
+
`scripted` builds the inert fakes; a real provider builds the real
|
|
90
|
+
`Architect` + `Judge` over a real `LLMClient`.
|
|
91
|
+
"""
|
|
92
|
+
|
|
93
|
+
host: HostMAS
|
|
94
|
+
judge: JudgeProtocol
|
|
95
|
+
architect: ArchitectProtocol
|
|
96
|
+
suite: Suite
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _import_adapter(dotted: str) -> HostMAS:
|
|
100
|
+
"""Import an external MAS adapter from a ``module:Class`` (or ``module``)
|
|
101
|
+
dotted path and instantiate it. The class implements ``HostMAS`` (the kit's
|
|
102
|
+
``BaseHostAdapter`` does), so it drops straight in as ``components.host`` —
|
|
103
|
+
its own ``__init__`` carries whatever its MAS needs (Lumina loads its base
|
|
104
|
+
prompts; a framework adapter wraps its graph). No PR into core to adapt a
|
|
105
|
+
new MAS: ``--adapter mypkg:MyAdapter`` wires it; ``--provider`` keeps the
|
|
106
|
+
Architect/Judge organs, ``--seed`` the bootstrap Spec."""
|
|
107
|
+
if ":" in dotted:
|
|
108
|
+
modpath, cls = dotted.split(":", 1)
|
|
109
|
+
else:
|
|
110
|
+
modpath, cls = dotted, ""
|
|
111
|
+
module = importlib.import_module(modpath)
|
|
112
|
+
if not cls:
|
|
113
|
+
# Bare module: expect it to expose a ``HostMAS``-protocol attr named
|
|
114
|
+
# ``HostMAS`` or the last path segment; else error loudly.
|
|
115
|
+
cls = "HostMAS"
|
|
116
|
+
try:
|
|
117
|
+
obj = getattr(module, cls)
|
|
118
|
+
except AttributeError as exc:
|
|
119
|
+
raise SystemExit(
|
|
120
|
+
f"--adapter: module {modpath!r} has no attribute {cls!r}. "
|
|
121
|
+
f"Pass it as `module:ClassName`."
|
|
122
|
+
) from exc
|
|
123
|
+
if isinstance(obj, type):
|
|
124
|
+
return obj() # a HostMAS/BaseHostAdapter subclass → instance
|
|
125
|
+
if isinstance(obj, HostMAS):
|
|
126
|
+
return obj # already an instance
|
|
127
|
+
raise SystemExit(f"--adapter: {dotted!r} resolved to a {type(obj).__name__}, "
|
|
128
|
+
"not a HostMAS subclass or instance.")
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
# --------------------------------------------------------------------------- #
|
|
132
|
+
# arg parsing
|
|
133
|
+
# --------------------------------------------------------------------------- #
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _add_store_args(p: argparse.ArgumentParser) -> None:
|
|
137
|
+
p.add_argument("--root", default=_DEFAULT_ROOT,
|
|
138
|
+
help="archforge state directory (default: .archforge)")
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _add_evolve_args(p: argparse.ArgumentParser, *, loop: bool) -> None:
|
|
142
|
+
p.add_argument("--seed", metavar="PATH",
|
|
143
|
+
help="bootstrap the root incumbent from this Spec JSON (no active yet)")
|
|
144
|
+
p.add_argument("--adapter", metavar="DOTTED.PATH[:Class]",
|
|
145
|
+
help="import an external MAS adapter (a HostMAS/BaseHostAdapter "
|
|
146
|
+
"subclass) as the runtime host; pair with --provider for the "
|
|
147
|
+
"Architect/Judge organs. e.g. --adapter archforge_glue:LuminaAdapter")
|
|
148
|
+
p.add_argument("--provider", choices=_PROVIDERS, default=None,
|
|
149
|
+
help="LLM provider (default from archforge.py: anthropic/openai/groq/gemini "
|
|
150
|
+
"are real; scripted is the zero-cost fake)")
|
|
151
|
+
# API keys: a `.env` in the cwd is loaded first (load_env, ∴ real env wins), then
|
|
152
|
+
# the provider SDK reads its key var; --api-key overrides both. Tunable flags
|
|
153
|
+
# default to None here so the active config (.archforge/archforge.py) supplies
|
|
154
|
+
# the real default at resolve time — an explicit flag overrides the file.
|
|
155
|
+
p.add_argument("--env-file", default=None,
|
|
156
|
+
help="load provider API keys from this file before --provider "
|
|
157
|
+
"builds the organs (default from archforge.py; no-op if absent)")
|
|
158
|
+
p.add_argument("--api-key", default=None, help="provider API key (overrides .env/env)")
|
|
159
|
+
p.add_argument("--base-url", default=None, help="provider base URL override")
|
|
160
|
+
p.add_argument("--architect-model", default=None,
|
|
161
|
+
help="model id for the Architect (else the provider default)")
|
|
162
|
+
p.add_argument("--judge-model", default=None,
|
|
163
|
+
help="model id for the Judge (else the provider default)")
|
|
164
|
+
p.add_argument("--suite", metavar="PATH", default=None,
|
|
165
|
+
help="path to a suite.json (overrides the DEFAULT_SUITE_FILE tunable; "
|
|
166
|
+
"the file's tasks define what you optimize against)")
|
|
167
|
+
# thresholds — every default resolves from the active config (.archforge/archforge.py);
|
|
168
|
+
# a flag is None at the parser and filled from ucfg unless the user set it.
|
|
169
|
+
p.add_argument("--tau", type=float, default=None, help="promotion margin τ")
|
|
170
|
+
p.add_argument("--delta", type=float, default=None, help="regression floor δ (>= τ)")
|
|
171
|
+
p.add_argument("--repeats", type=int, default=None, help="R: repeats per eval-suite task")
|
|
172
|
+
if loop:
|
|
173
|
+
p.add_argument("--max-cycles", type=int, default=None,
|
|
174
|
+
help="cap on P-E-C cycles")
|
|
175
|
+
p.add_argument("--plateau-cycles", type=int, default=None,
|
|
176
|
+
help="K consecutive no-promotion cycles → plateau (E8)")
|
|
177
|
+
p.add_argument("--max-tokens-total", type=int, default=None,
|
|
178
|
+
help="total token budget cap; stop at or before reaching it (E3)")
|
|
179
|
+
p.add_argument("--max-tokens-per-cycle", type=int, default=None,
|
|
180
|
+
help="per-cycle token cap; abort mid-cycle if exceeded (E3)")
|
|
181
|
+
p.add_argument("--max-wall-ms-per-cycle", type=float, default=None,
|
|
182
|
+
help="per-cycle wall-clock cap (ms); aborts if exceeded — "
|
|
183
|
+
"bounds non-LLM nodes (retriever/tool/rule) that cost time, not tokens; "
|
|
184
|
+
"default None = no limit")
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def _build_parser() -> argparse.ArgumentParser:
|
|
188
|
+
parser = argparse.ArgumentParser(
|
|
189
|
+
prog=PROG,
|
|
190
|
+
description="ArchForge — a self-improving meta-layer over multi-agent systems.",
|
|
191
|
+
)
|
|
192
|
+
sub = parser.add_subparsers(dest="command", metavar="COMMAND")
|
|
193
|
+
|
|
194
|
+
# --- evolve (one cycle) --------------------------------------------------
|
|
195
|
+
ev = sub.add_parser("evolve",
|
|
196
|
+
help="run one Propose-Evaluate-Commit cycle from the active incumbent")
|
|
197
|
+
_add_store_args(ev)
|
|
198
|
+
_add_evolve_args(ev, loop=False)
|
|
199
|
+
|
|
200
|
+
# --- evolve-loop ---------------------------------------------------------
|
|
201
|
+
evl = sub.add_parser("evolve-loop",
|
|
202
|
+
help="repeat evolve until the budget cap or a plateau")
|
|
203
|
+
_add_store_args(evl)
|
|
204
|
+
_add_evolve_args(evl, loop=True)
|
|
205
|
+
|
|
206
|
+
# --- approve (human gate, I4) --------------------------------------------
|
|
207
|
+
ap = sub.add_parser("approve",
|
|
208
|
+
help="approve queued (PENDING_HUMAN) structural changes → active")
|
|
209
|
+
_add_store_args(ap)
|
|
210
|
+
ap.add_argument("attempts", nargs="*",
|
|
211
|
+
help="attempt ids to approve (default: every pending change, in order)")
|
|
212
|
+
ap.add_argument("--all", action="store_true",
|
|
213
|
+
help="approve every pending change (default when none are named)")
|
|
214
|
+
|
|
215
|
+
# --- reject --------------------------------------------------------------
|
|
216
|
+
rj = sub.add_parser("reject",
|
|
217
|
+
help="reject a queued structural change — active is left alone")
|
|
218
|
+
_add_store_args(rj)
|
|
219
|
+
rj.add_argument("attempt_id", help="attempt id to reject")
|
|
220
|
+
|
|
221
|
+
# --- status / report -----------------------------------------------------
|
|
222
|
+
st = sub.add_parser("status", help="print the incumbent Spec + lineage + counts")
|
|
223
|
+
_add_store_args(st)
|
|
224
|
+
|
|
225
|
+
rp = sub.add_parser("report", help="print aggregate deltas across attempts")
|
|
226
|
+
_add_store_args(rp)
|
|
227
|
+
|
|
228
|
+
# --- lint ----------------------------------------------------------------
|
|
229
|
+
lint_p = sub.add_parser("lint", help="run the Spec Linter on a JSON Spec file")
|
|
230
|
+
_add_store_args(lint_p)
|
|
231
|
+
lint_p.add_argument("path", help="path to a Spec JSON file")
|
|
232
|
+
|
|
233
|
+
# --- init (scaffold user config) -----------------------------------------
|
|
234
|
+
init_p = sub.add_parser("init",
|
|
235
|
+
help="scaffold .archforge/archforge.py + .env.example for this project")
|
|
236
|
+
_add_store_args(init_p) # --root selects where archforge.py is written
|
|
237
|
+
init_p.add_argument("--force", action="store_true",
|
|
238
|
+
help="overwrite an existing .archforge/archforge.py")
|
|
239
|
+
|
|
240
|
+
return parser
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
# --------------------------------------------------------------------------- #
|
|
244
|
+
# helpers
|
|
245
|
+
# --------------------------------------------------------------------------- #
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def _stores(root: str) -> tuple[SpecStore, AttemptStore, TraceStore]:
|
|
249
|
+
return SpecStore(root), AttemptStore(root), TraceStore(root)
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def _arg(args: argparse.Namespace, attr: str, cfg_name: str):
|
|
253
|
+
"""Resolve a CLI tunable: the explicit flag value, else the active-config default.
|
|
254
|
+
|
|
255
|
+
Every tunable flag defaults to ``None`` at the parser (so `--help` works pre-init
|
|
256
|
+
and a user's `.archforge/archforge.py` override isn't frozen into argparse). The
|
|
257
|
+
real default is read here from ``ucfg`` (disk post-init, or the in-memory sane
|
|
258
|
+
template under the pytest gate) only when the flag is omitted.
|
|
259
|
+
"""
|
|
260
|
+
v = getattr(args, attr, None)
|
|
261
|
+
return v if v is not None else ucfg.get(cfg_name)
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def _thresholds(args: argparse.Namespace) -> m.Thresholds:
|
|
265
|
+
return m.Thresholds(
|
|
266
|
+
tau=_arg(args, "tau", "DEFAULT_TAU"),
|
|
267
|
+
delta=_arg(args, "delta", "DEFAULT_DELTA"),
|
|
268
|
+
repeats=_arg(args, "repeats", "DEFAULT_REPEATS"),
|
|
269
|
+
plateau_cycles=_arg(args, "plateau_cycles", "DEFAULT_PLATEAU_CYCLES"),
|
|
270
|
+
)
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
def _config(args: argparse.Namespace) -> EngineConfig:
|
|
274
|
+
return EngineConfig(
|
|
275
|
+
max_cycles=_arg(args, "max_cycles", "DEFAULT_MAX_CYCLES"),
|
|
276
|
+
max_tokens_per_cycle=_arg(args, "max_tokens_per_cycle", "DEFAULT_MAX_TOKENS_PER_CYCLE"),
|
|
277
|
+
max_tokens_total=_arg(args, "max_tokens_total", "DEFAULT_MAX_TOKENS_TOTAL"),
|
|
278
|
+
repeats=_arg(args, "repeats", "DEFAULT_REPEATS"),
|
|
279
|
+
plateau_cycles=_arg(args, "plateau_cycles", "DEFAULT_PLATEAU_CYCLES"),
|
|
280
|
+
max_wall_ms_per_cycle=_arg(args, "max_wall_ms_per_cycle", "DEFAULT_MAX_WALL_MS_PER_CYCLE"),
|
|
281
|
+
)
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def _default_components(args: argparse.Namespace) -> Components:
|
|
285
|
+
"""Build the runtime organs for `--provider` (no injected `components`).
|
|
286
|
+
|
|
287
|
+
`--provider scripted` (default) builds the zero-cost fakes — well-defined but
|
|
288
|
+
inert: the scripted architect has no queued proposal so it plateaus, and the
|
|
289
|
+
scripted judge returns a neutral 0.5 base. A deterministic *promotion* needs
|
|
290
|
+
the organs pre-configured, which is the test path through
|
|
291
|
+
`main(..., components=...)`.
|
|
292
|
+
|
|
293
|
+
A real provider (`anthropic`/`openai`/`groq`/`gemini`)
|
|
294
|
+
builds ONE `LLMClient` via `make_client` and the REAL `Architect` + `Judge`
|
|
295
|
+
over it — the provider abstraction is the single seam, so neither organ
|
|
296
|
+
changes when the provider changes. The host stays `FakeHostMAS` for now
|
|
297
|
+
(a real MAS host is its own integration; the seam already accepts it).
|
|
298
|
+
"""
|
|
299
|
+
|
|
300
|
+
# The eval suite: a JSON sidecar (.archforge/suite.json by default) if present,
|
|
301
|
+
# else the one-task fallback fixture (byte-identical to `init`'s seeded
|
|
302
|
+
# suite.json, so out-of-box == generated-default). --suite overrides the tunable.
|
|
303
|
+
suite_path = getattr(args, "suite", None) or ucfg.get("DEFAULT_SUITE_FILE")
|
|
304
|
+
suite = load_suite_file(suite_path) or Suite(
|
|
305
|
+
suite_id=DEFAULT_SUITE_ID, rubric_id=default_rubric().rubric_id,
|
|
306
|
+
tasks=[Task(task_id=DEFAULT_TASK_ID, input=DEFAULT_TASK_INPUT)])
|
|
307
|
+
provider = args.provider or ucfg.get("PROVIDER")
|
|
308
|
+
if provider == "scripted":
|
|
309
|
+
return Components(host=FakeHostMAS(), judge=ScriptedJudge(),
|
|
310
|
+
architect=ScriptedArchitect(), suite=suite)
|
|
311
|
+
from archforge.llm import make_client, LLMError
|
|
312
|
+
|
|
313
|
+
try:
|
|
314
|
+
llm = make_client(provider, api_key=args.api_key, base_url=args.base_url)
|
|
315
|
+
except LLMError as exc:
|
|
316
|
+
print(f"[provider] {exc}", file=sys.stderr)
|
|
317
|
+
raise
|
|
318
|
+
arch_models = ucfg.get("DEFAULT_ARCHITECT_MODELS")
|
|
319
|
+
judge_models = ucfg.get("DEFAULT_JUDGE_MODELS")
|
|
320
|
+
arch = Architect(llm, model=args.architect_model or arch_models[provider])
|
|
321
|
+
judge = Judge(llm, model=args.judge_model or judge_models[provider],
|
|
322
|
+
rubric=default_rubric())
|
|
323
|
+
return Components(host=FakeHostMAS(), judge=judge, architect=arch, suite=suite)
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def _ensure_incumbent(args: argparse.Namespace, specs: SpecStore) -> str | None:
|
|
327
|
+
"""Make sure an active incumbent exists before `evolve` runs.
|
|
328
|
+
|
|
329
|
+
If one already exists, return its id (no LLM, no overwrite). If none and a
|
|
330
|
+
`--seed` is supplied, lint + commit it as the root INCUMBENT and set active
|
|
331
|
+
(zero-LLM bootstrap). Returns the active id, or None if it could not.
|
|
332
|
+
"""
|
|
333
|
+
|
|
334
|
+
current = specs.active_id()
|
|
335
|
+
if current is not None:
|
|
336
|
+
return current
|
|
337
|
+
seed_path = getattr(args, "seed", None)
|
|
338
|
+
if not seed_path:
|
|
339
|
+
return None
|
|
340
|
+
spec = m.Spec.model_validate(json.loads(Path(seed_path).read_text(encoding="utf-8")))
|
|
341
|
+
faults = lint(spec)
|
|
342
|
+
if faults:
|
|
343
|
+
raise SystemExit(
|
|
344
|
+
"refusing to bootstrap from --seed: spec fails the linter: "
|
|
345
|
+
+ "; ".join(f"{f.code}({f.location or ''}): {f.message}" for f in faults)
|
|
346
|
+
)
|
|
347
|
+
spec_id = specs.commit(spec, parent_spec_id=None, status=m.SpecStatus.INCUMBENT)
|
|
348
|
+
specs.set_active(spec_id)
|
|
349
|
+
return spec_id
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
def _pending(atts: AttemptStore) -> list[m.Attempt]:
|
|
353
|
+
return [a for a in atts.all() if a.verdict is m.Verdict.PENDING_HUMAN]
|
|
354
|
+
|
|
355
|
+
|
|
356
|
+
def _fmt_spec(spec: m.Spec) -> str:
|
|
357
|
+
parent = spec.parent_spec_id or "(root)"
|
|
358
|
+
return (f" spec_id: {spec.spec_id}\n"
|
|
359
|
+
f" parent: {parent}\n"
|
|
360
|
+
f" status: {spec.status.value}\n"
|
|
361
|
+
f" created: {spec.created_at}\n"
|
|
362
|
+
f" nodes: {len(spec.nodes)}\n"
|
|
363
|
+
f" edges: {len(spec.edges)}")
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
# --------------------------------------------------------------------------- #
|
|
367
|
+
# subcommand handlers
|
|
368
|
+
# --------------------------------------------------------------------------- #
|
|
369
|
+
|
|
370
|
+
|
|
371
|
+
def _cmd_status(args: argparse.Namespace) -> int:
|
|
372
|
+
specs, atts, _ = _stores(args.root)
|
|
373
|
+
active_id = specs.active_id()
|
|
374
|
+
if active_id is None:
|
|
375
|
+
print("No incumbent yet. Bootstrap with `archforge-optimizer evolve --seed <spec.json>`.")
|
|
376
|
+
return 0
|
|
377
|
+
spec = specs.get(active_id)
|
|
378
|
+
print("active incumbent:")
|
|
379
|
+
print(_fmt_spec(spec))
|
|
380
|
+
chain = specs.lineage(active_id)
|
|
381
|
+
print("lineage: " + " <- ".join(chain))
|
|
382
|
+
print(f"specs known: {len(specs.known_ids())} "
|
|
383
|
+
f"archived: {len(specs.archived_ids())} "
|
|
384
|
+
f"attempts: {len(atts.all())} "
|
|
385
|
+
f"pending: {len(_pending(atts))}")
|
|
386
|
+
return 0
|
|
387
|
+
|
|
388
|
+
|
|
389
|
+
def _cmd_report(args: argparse.Namespace) -> int:
|
|
390
|
+
specs, atts, _ = _stores(args.root)
|
|
391
|
+
rows = atts.all()
|
|
392
|
+
if not rows:
|
|
393
|
+
print("No attempts recorded yet.")
|
|
394
|
+
return 0
|
|
395
|
+
print(f"{'attempt_id':<18}{'verdict':<14}{'kind':<13}{'target':<8}"
|
|
396
|
+
f"{'mean':>7}{'margin':>9}{'tokens':>8} change")
|
|
397
|
+
print("-" * 90)
|
|
398
|
+
for a in rows:
|
|
399
|
+
r = a.suite_result
|
|
400
|
+
mean = f"{r.mean:.3f}" if r is not None else "-"
|
|
401
|
+
margin = f"{r.margin_vs_incumbent:+.3f}" if r is not None else "-"
|
|
402
|
+
toks = str(r.tokens) if r is not None else "-"
|
|
403
|
+
print(f"{(a.attempt_id or '-'):<18}{a.verdict.value:<14}"
|
|
404
|
+
f"{a.change.kind.value:<13}{a.change.target:<8}"
|
|
405
|
+
f"{mean:>7}{margin:>9}{toks:>8} {a.change.diff}")
|
|
406
|
+
return 0
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
def _cmd_approve(args: argparse.Namespace) -> int:
|
|
410
|
+
specs, atts, _ = _stores(args.root)
|
|
411
|
+
gk = Gatekeeper(specs, atts, thresholds=_thresholds_for_approval())
|
|
412
|
+
if args.all or not args.attempts:
|
|
413
|
+
pending = _pending(atts)
|
|
414
|
+
if not pending:
|
|
415
|
+
print("No queued (PENDING_HUMAN) changes to approve.")
|
|
416
|
+
return 0
|
|
417
|
+
ids = [a.attempt_id for a in pending]
|
|
418
|
+
else:
|
|
419
|
+
ids = list(args.attempts)
|
|
420
|
+
|
|
421
|
+
approved: list[str] = []
|
|
422
|
+
for aid in ids:
|
|
423
|
+
try:
|
|
424
|
+
att = gk.approve(aid) # only path w/ Gatekeeper that moves active
|
|
425
|
+
except ValueError as exc:
|
|
426
|
+
print(f"! {aid}: {exc}")
|
|
427
|
+
continue
|
|
428
|
+
approved.append(aid)
|
|
429
|
+
print(f"approved {aid}: active = {att.candidate_spec_id} (verdict={att.verdict.value})")
|
|
430
|
+
if not approved:
|
|
431
|
+
print("Nothing approved.")
|
|
432
|
+
return 1
|
|
433
|
+
print(f"approved {len(approved)} change(s). active incumbent: {specs.active_id()}")
|
|
434
|
+
return 0
|
|
435
|
+
|
|
436
|
+
|
|
437
|
+
def _cmd_reject(args: argparse.Namespace) -> int:
|
|
438
|
+
specs, atts, _ = _stores(args.root)
|
|
439
|
+
gk = Gatekeeper(specs, atts, thresholds=_thresholds_for_approval())
|
|
440
|
+
try:
|
|
441
|
+
att = gk.reject(args.attempt_id)
|
|
442
|
+
except ValueError as exc:
|
|
443
|
+
print(f"! {args.attempt_id}: {exc}")
|
|
444
|
+
return 1
|
|
445
|
+
print(f"rejected {args.attempt_id}: verdict={att.verdict.value}; "
|
|
446
|
+
f"active unchanged at {specs.active_id()}")
|
|
447
|
+
return 0
|
|
448
|
+
|
|
449
|
+
|
|
450
|
+
def _thresholds_for_approval() -> m.Thresholds:
|
|
451
|
+
# approve/reject never consult τ/δ; defaults are fine (the verdicts already
|
|
452
|
+
# carry the cycle's margin on their suite_result).
|
|
453
|
+
return m.Thresholds()
|
|
454
|
+
|
|
455
|
+
|
|
456
|
+
def _install_resource_warning_quieteners() -> None:
|
|
457
|
+
"""Suppress `ResourceWarning: unclosed <ssl.SSLSocket>` noise emitted by the
|
|
458
|
+
per-call LLM clients' (groq/google-genai) sockets closing in langgraph's
|
|
459
|
+
async executor. Extracted to module scope so the silence contract is pin-able
|
|
460
|
+
by a unit test (the mechanism took three iterations settle — see pitfall below).
|
|
461
|
+
|
|
462
|
+
A real socket's `__del__` emits these via the standard
|
|
463
|
+
`warnings.warn(..., ResourceWarning)` path — controlled by the warnings
|
|
464
|
+
FILTER, which is what makes them fiddly: a library importing after this point
|
|
465
|
+
(httpx/httpcore/chromadb/langgraph) calls `warnings.simplefilter` /
|
|
466
|
+
`filterwarnings` at import, which PREPENDS its entry above ours, and the FIRST
|
|
467
|
+
matching filter wins → a `simplefilter("ignore", ResourceWarning)` we set here
|
|
468
|
+
gets shadowed and the warnings print anyway (observed in the first CLI run).
|
|
469
|
+
The `def __init__(self)` / `threading.py:301` / `langgraph/pregel/_utils.py:235`
|
|
470
|
+
lines are tracemalloc allocation-site fingers ResourceWarning attaches to its
|
|
471
|
+
source resolution, NOT emitters.
|
|
472
|
+
|
|
473
|
+
The robust mechanism (shadow-proof): override `warnings.showwarning` — the
|
|
474
|
+
terminal sink every warning routes to AFTER the filters decide to *show* it (so
|
|
475
|
+
a library's re-armed "default" filter still routes to us, and we drop
|
|
476
|
+
ResourceWarning here). Can't be shadowed by filter re-arming. Drop
|
|
477
|
+
ResourceWarning ONLY — a real archforge DeprecationWarning stays visible for
|
|
478
|
+
debugging. Belt-and-suspenders: also silence `sys.unraisablehook` for the rare
|
|
479
|
+
`__del__`-RAISES path (a genuine bug), limited to ResourceWarning so other
|
|
480
|
+
unraisable exceptions stay loud. Scoped to the evolve command: `init`/`status`/
|
|
481
|
+
`report` + the pytest suite never enter here, retaining full warning visibility.
|
|
482
|
+
archforge opens no raw sockets itself → ResourceWarning is third-party
|
|
483
|
+
async-pool noise.
|
|
484
|
+
"""
|
|
485
|
+
_default_showwarning = warnings.showwarning
|
|
486
|
+
_default_unraisable_hook = sys.unraisablehook
|
|
487
|
+
|
|
488
|
+
def _quiet_showwarning(message, category, filename, lineno, file=None,
|
|
489
|
+
line=None):
|
|
490
|
+
if issubclass(category, ResourceWarning):
|
|
491
|
+
return
|
|
492
|
+
_default_showwarning(message, category, filename, lineno, file, line=line)
|
|
493
|
+
|
|
494
|
+
def _quiet_resource_warning(unr_args, /):
|
|
495
|
+
exc = getattr(unr_args, "exc_value", None)
|
|
496
|
+
if isinstance(exc, ResourceWarning):
|
|
497
|
+
return
|
|
498
|
+
_default_unraisable_hook(unr_args)
|
|
499
|
+
|
|
500
|
+
warnings.showwarning = _quiet_showwarning
|
|
501
|
+
sys.unraisablehook = _quiet_resource_warning
|
|
502
|
+
|
|
503
|
+
|
|
504
|
+
# Activate the quietener at IMPORT time (module scope), not inside `_cmd_evolve`.
|
|
505
|
+
# The unclosed-SSL ResourceWarnings emit in THREE windows: (1) import/init time
|
|
506
|
+
# as libraries (httpx/langgraph/google-genai) spin up + tear down sockets, (2)
|
|
507
|
+
# during the evolve run, (3) at interpreter shutdown GC. Scoping the install to
|
|
508
|
+
# `_cmd_evolve` (the earlier attempt) covered only window 2 — windows 1 and 3
|
|
509
|
+
# still printed (observed: a top batch with `ast.py:46` fingers before cycle 0
|
|
510
|
+
# and a bottom batch with `<sys>:0` fingers after the last cycle, both printing
|
|
511
|
+
# the default `Enable tracemalloc` hint). `python -m archforge.cli` imports this
|
|
512
|
+
# module FIRST (it's `__main__`), so installing here runs before any provider
|
|
513
|
+
# library is imported (those come lazily at evolve time per the import-laziness
|
|
514
|
+
# contract) → the sink is in place for all three windows. The override is
|
|
515
|
+
# shadow-proof against a library re-arming the warnings FILTER (proven in
|
|
516
|
+
# tests/unit/test_cli_warning_quietener.py); a library reassigning
|
|
517
|
+
# `warnings.showwarning` outright would defeat it, but none of the runtime deps
|
|
518
|
+
# do that (only pytest's recorder does, and the CLI isn't under pytest).
|
|
519
|
+
_install_resource_warning_quieteners()
|
|
520
|
+
|
|
521
|
+
|
|
522
|
+
def _cmd_evolve(args: argparse.Namespace, *, components: Components | None,
|
|
523
|
+
loop: bool) -> int:
|
|
524
|
+
# Injected `components` (the test/embedding path) always win — they ARE the
|
|
525
|
+
# organs, by contract. Otherwise build organs per `--provider`. While a real
|
|
526
|
+
# tunable is needed we require `init` to have run (the "pip install → init → CLI
|
|
527
|
+
# works" contract): a project with no .archforge/archforge.py gets the init hint
|
|
528
|
+
# and rc 1 instead of a bogus run. No-op under the pytest gate (tests use the
|
|
529
|
+
# in-memory sane template).
|
|
530
|
+
if components is None:
|
|
531
|
+
try:
|
|
532
|
+
ucfg.ensure_initialized()
|
|
533
|
+
except ConfigNotInitialized as exc:
|
|
534
|
+
print(str(exc), file=sys.stderr)
|
|
535
|
+
return 1
|
|
536
|
+
# Load API keys from the cwd's .env (real env wins; --api-key wins above
|
|
537
|
+
# both). No-op if the file is missing or for --provider scripted.
|
|
538
|
+
load_env(getattr(args, "env_file", None) or ucfg.get("DEFAULT_ENV_FILE"))
|
|
539
|
+
if (args.provider or ucfg.get("PROVIDER")) != "scripted":
|
|
540
|
+
try:
|
|
541
|
+
components = _default_components(args) # builds a real LLMClient
|
|
542
|
+
except Exception:
|
|
543
|
+
return 2 # message already printed
|
|
544
|
+
|
|
545
|
+
specs, atts, traces = _stores(args.root)
|
|
546
|
+
|
|
547
|
+
# zero-LLM bootstrap of the root incumbent from --seed (if none active)
|
|
548
|
+
active = _ensure_incumbent(args, specs)
|
|
549
|
+
if active is None:
|
|
550
|
+
print("No active incumbent and no --seed given; bootstrap with "
|
|
551
|
+
"`archforge-optimizer evolve --seed <spec.json>` first.", file=sys.stderr)
|
|
552
|
+
return 1
|
|
553
|
+
|
|
554
|
+
organs = components or _default_components(args)
|
|
555
|
+
|
|
556
|
+
# --adapter: swap the runtime host for an external MAS adapter (a HostMAS /
|
|
557
|
+
# BaseHostAdapter) while keeping the --provider organs (Architect/Judge).
|
|
558
|
+
# This is the "adapt any MAS" seam: point at an adapter class, no core edit.
|
|
559
|
+
adapter_path = getattr(args, "adapter", None)
|
|
560
|
+
if adapter_path:
|
|
561
|
+
organs = Components(host=_import_adapter(adapter_path), judge=organs.judge,
|
|
562
|
+
architect=organs.architect, suite=organs.suite)
|
|
563
|
+
|
|
564
|
+
# The expressive surfaces (improvements #2/#3/#4/#5) wire through the Engine's
|
|
565
|
+
# opt-in hooks (on_cycle fires each attempted cycle; on_deploy fires on a
|
|
566
|
+
# promote). `on_cycle` renders the legacy line + the compact card to stdout
|
|
567
|
+
# AND appends a JSON form to a fail-soft run-log (``<root>/runs/last.json``);
|
|
568
|
+
# `on_deploy` writes the unified ``optimized.json`` envelope (Tier-2 deploy,
|
|
569
|
+
# auto-synced from the CLI — improvement #4). When the hook fires, the legacy
|
|
570
|
+
# `_print_cycle`/`_print_loop` must NOT re-print the per-cycle line (the hook
|
|
571
|
+
# already did); `hooks_wired` gates that fallback.
|
|
572
|
+
runlog = RunLog(Path(args.root) / "runs" / "last.json")
|
|
573
|
+
if not runlog.enabled:
|
|
574
|
+
print(f" (run log disabled: {runlog.reason})")
|
|
575
|
+
optimized_path = Path(args.root) / "optimized.json"
|
|
576
|
+
|
|
577
|
+
def _on_cycle(result: CycleResult, ctx: CycleCtx) -> None:
|
|
578
|
+
# Per-cycle visibility (hooks fire DURING the run, so evolve-loop now shows
|
|
579
|
+
# each cycle, not just the final summary). The legacy one-liner (kept
|
|
580
|
+
# verbatim — the integration tests assert its substrings) PLUS the card.
|
|
581
|
+
_print_cycle(result)
|
|
582
|
+
if result.attempted:
|
|
583
|
+
print(render_card(result, ctx))
|
|
584
|
+
runlog.append_cycle(card_to_json(result, ctx))
|
|
585
|
+
|
|
586
|
+
def _on_deploy(spec: m.Spec, dctx: DeployCtx) -> None:
|
|
587
|
+
# Tier-2 deploy auto-synced from the CLI (improvement #4): one
|
|
588
|
+
# ``optimized.json`` production loads to apply the winner — knobs + lineage
|
|
589
|
+
# + the decision + scores that justified the promote. Printed notice keeps
|
|
590
|
+
# the user informed; the file is the artifact (rollback = delete it).
|
|
591
|
+
try:
|
|
592
|
+
from archforge.host.adapters import export_optimized
|
|
593
|
+
except ImportError: # pragma: no cover
|
|
594
|
+
return
|
|
595
|
+
export_optimized(spec, optimized_path, parent=dctx.parent,
|
|
596
|
+
promoted_at_cycle=dctx.promoted_at_cycle,
|
|
597
|
+
decision=dctx.decision, cand_run=dctx.cand_run,
|
|
598
|
+
inc_run=dctx.inc_run)
|
|
599
|
+
print(f" deployed: {optimized_path} "
|
|
600
|
+
f"spec_id={spec.spec_id or spec.compute_spec_id()[:8]} "
|
|
601
|
+
f"(knobs + scores + decision)")
|
|
602
|
+
|
|
603
|
+
engine = Engine(
|
|
604
|
+
host=organs.host, judge=organs.judge, architect=organs.architect,
|
|
605
|
+
spec_store=specs, attempt_store=atts, trace_store=traces,
|
|
606
|
+
suite=organs.suite, thresholds=_thresholds(args), config=_config(args),
|
|
607
|
+
on_cycle=_on_cycle, on_deploy=_on_deploy,
|
|
608
|
+
)
|
|
609
|
+
|
|
610
|
+
if not loop:
|
|
611
|
+
engine.evolve_cycle(cycle=0)
|
|
612
|
+
# `on_cycle` already printed the line + card; no post-hoc re-print.
|
|
613
|
+
return 0
|
|
614
|
+
lr = engine.evolve_loop()
|
|
615
|
+
# `on_cycle` printed each cycle's line + card live; the loop summary is the
|
|
616
|
+
# only post-hoc print (legacy one-liner kept verbatim + the summary block).
|
|
617
|
+
print(f"{PROG} evolve-loop: cycles_run={lr.cycles_run} promotions={lr.promotions} "
|
|
618
|
+
f"queued={lr.queued} plateaued={'true' if lr.plateaued else 'false'} "
|
|
619
|
+
f"aborted={'true' if lr.aborted else 'false'} "
|
|
620
|
+
f"final_incumbent={lr.final_incumbent_id} "
|
|
621
|
+
f"final_mean={lr.final_incumbent_mean if lr.final_incumbent_mean is not None else '-'!s}")
|
|
622
|
+
if lr.aborted and lr.abort_reason:
|
|
623
|
+
print(f" abort_reason: {lr.abort_reason}")
|
|
624
|
+
print(summary_block(lr))
|
|
625
|
+
runlog.write_summary(summary_to_json(lr))
|
|
626
|
+
return 0
|
|
627
|
+
|
|
628
|
+
|
|
629
|
+
def _print_cycle(r: CycleResult) -> int:
|
|
630
|
+
if not r.attempted:
|
|
631
|
+
print(f"{PROG} evolve: cycle={r.cycle} attempted=false action=none "
|
|
632
|
+
f"note={r.note}")
|
|
633
|
+
return 0
|
|
634
|
+
action = r.decision.action.value if r.decision else "none"
|
|
635
|
+
inc = f"{r.incumbent_mean:.3f}" if r.incumbent_mean is not None else "-"
|
|
636
|
+
cand = f"{r.candidate_mean:.3f}" if r.candidate_mean is not None else "-"
|
|
637
|
+
margin = f"{r.decision.margin:+.3f}" if r.decision else "-"
|
|
638
|
+
print(f"{PROG} evolve: cycle={r.cycle} attempted=true action={action} "
|
|
639
|
+
f"margin={margin} incumbent_mean={inc} candidate_mean={cand} "
|
|
640
|
+
f"promoted={'true' if r.promoted else 'false'} "
|
|
641
|
+
f"queued={'true' if r.queued else 'false'} "
|
|
642
|
+
f"attempt_id={r.applied_attempt_id} tokens={r.tokens}")
|
|
643
|
+
if r.note:
|
|
644
|
+
print(f" note: {r.note}")
|
|
645
|
+
return 0
|
|
646
|
+
|
|
647
|
+
|
|
648
|
+
# --------------------------------------------------------------------------- #
|
|
649
|
+
# Expressive surfaces — the cycle card (#2/#3/#5) + loop summary (#5)
|
|
650
|
+
# --------------------------------------------------------------------------- #
|
|
651
|
+
#
|
|
652
|
+
# The legacy one-liners above stay verbatim (the integration tests assert their
|
|
653
|
+
# substrings). The card is APPENDED after the legacy line: a 4-line block carrying
|
|
654
|
+
# the mutation diff (#2), the cand-vs-inc rubric scores, the cost delta, and the
|
|
655
|
+
# rejection/accept explanation (#3). It renders BOTH to stdout AND to a JSON run
|
|
656
|
+
# log (improvement #5's "file" sink) when a RunLog is wired. `attempted=false`
|
|
657
|
+
# cycles print only the legacy line (nothing to diff). The envelope (#4) is
|
|
658
|
+
# written by `_on_deploy` on AUTO_PROMOTE; the loop summary card is printed +
|
|
659
|
+
# logged at the end.
|
|
660
|
+
|
|
661
|
+
def _fmt_dims(run) -> str:
|
|
662
|
+
"""Compact ``corr X.XX compl X.XX`` from a SuiteRun's per-repeat rubric
|
|
663
|
+
dims. Tolerant of a None run or a run with no scores (-> empty)."""
|
|
664
|
+
if run is None:
|
|
665
|
+
return ""
|
|
666
|
+
sums: dict[str, float] = {}
|
|
667
|
+
n: dict[str, int] = {}
|
|
668
|
+
for rs in getattr(run, "scores", None) or []:
|
|
669
|
+
for dim, val in (getattr(rs, "rubric_scores", None) or {}).items():
|
|
670
|
+
try:
|
|
671
|
+
v = float(val)
|
|
672
|
+
except (TypeError, ValueError):
|
|
673
|
+
continue
|
|
674
|
+
sums[dim] = sums.get(dim, 0.0) + v
|
|
675
|
+
n[dim] = n.get(dim, 0) + 1
|
|
676
|
+
parts = [f"{d} {sums[d] / n[d]:.2f}" for d in sums]
|
|
677
|
+
return " ".join(parts)
|
|
678
|
+
|
|
679
|
+
|
|
680
|
+
def render_card(result: CycleResult, ctx: CycleCtx) -> str:
|
|
681
|
+
"""The compact 4-line card appended after the legacy cycle line (#2/#3/#5).
|
|
682
|
+
|
|
683
|
+
The lines (intentionally short — full per-step breakdown stays in the
|
|
684
|
+
persisted JSON + ``report``)::
|
|
685
|
+
|
|
686
|
+
change: <Change.kind> on "<target>" — <field: old -> new | roster delta>
|
|
687
|
+
scores: cand <m.mm> [dims] vs inc <m.mm> [dims]
|
|
688
|
+
cost: tokens +<Δ> (inc I → cand C) latency +<Δ>ms
|
|
689
|
+
why: <Decision.reason> (rule=<by_rule>)
|
|
690
|
+
|
|
691
|
+
The ``change`` line uses ``spec_diff`` (real field-level before/after, since
|
|
692
|
+
the mutator's ``payload`` evaporates) summarized via ``format_diff``. Empty
|
|
693
|
+
diff (candidate == parent, e.g. a no-op proposal that passed lint) renders
|
|
694
|
+
``(no change)``. Returns the block WITHOUT a trailing newline so callers can
|
|
695
|
+
join pipes; printing callers add the newline.
|
|
696
|
+
"""
|
|
697
|
+
ch = ctx.change
|
|
698
|
+
entries = spec_diff(ctx.parent_spec, ctx.candidate_spec)
|
|
699
|
+
change_line = format_diff(entries)
|
|
700
|
+
inc = ctx.inc_run
|
|
701
|
+
cand = ctx.cand_run
|
|
702
|
+
inc_mean = f"{inc.mean:.3f}" if getattr(inc, "mean", None) is not None else "-"
|
|
703
|
+
cand_mean = f"{cand.mean:.3f}" if getattr(cand, "mean", None) is not None else "-"
|
|
704
|
+
inc_dims = _fmt_dims(inc)
|
|
705
|
+
cand_dims = _fmt_dims(cand)
|
|
706
|
+
inc_toks = getattr(inc, "tokens", 0) or 0
|
|
707
|
+
cand_toks = getattr(cand, "tokens", 0) or 0
|
|
708
|
+
inc_lat = getattr(inc, "latency_ms", 0.0) or 0.0
|
|
709
|
+
cand_lat = getattr(cand, "latency_ms", 0.0) or 0.0
|
|
710
|
+
rule = result.decision.by_rule if result.decision else ""
|
|
711
|
+
reason = (result.decision.reason if result.decision else result.note) or ""
|
|
712
|
+
action = result.decision.action.value if result.decision else "none"
|
|
713
|
+
lines = [
|
|
714
|
+
f' change: {ch.kind.value} on "{ch.target}" — {change_line}',
|
|
715
|
+
f" scores: cand {cand_mean} [{cand_dims}] vs inc {inc_mean} [{inc_dims}]",
|
|
716
|
+
f" cost: tokens +{cand_toks - inc_toks} (inc {inc_toks} → cand {cand_toks})"
|
|
717
|
+
f" latency +{cand_lat - inc_lat:.1f}ms",
|
|
718
|
+
f" why: {reason} (rule={rule}, action={action})",
|
|
719
|
+
]
|
|
720
|
+
return "\n".join(lines)
|
|
721
|
+
|
|
722
|
+
|
|
723
|
+
def card_to_json(result: CycleResult, ctx: CycleCtx) -> dict:
|
|
724
|
+
"""The machine-readable form appended to the run-log (one entry per cycle)."""
|
|
725
|
+
return {
|
|
726
|
+
"cycle": result.cycle,
|
|
727
|
+
"attempted": result.attempted,
|
|
728
|
+
"action": result.decision.action.value if result.decision else None,
|
|
729
|
+
"change_kind": ctx.change.kind.value,
|
|
730
|
+
"target": ctx.change.target,
|
|
731
|
+
"diff": format_diff(spec_diff(ctx.parent_spec, ctx.candidate_spec)),
|
|
732
|
+
"incumbent_mean": result.incumbent_mean,
|
|
733
|
+
"candidate_mean": result.candidate_mean,
|
|
734
|
+
"margin": result.decision.margin if result.decision else None,
|
|
735
|
+
"by_rule": result.decision.by_rule if result.decision else None,
|
|
736
|
+
"reason": result.decision.reason if result.decision else result.note,
|
|
737
|
+
"tokens": result.tokens,
|
|
738
|
+
"latency_ms": result.latency_ms,
|
|
739
|
+
}
|
|
740
|
+
|
|
741
|
+
|
|
742
|
+
def summary_block(r: LoopResult) -> str:
|
|
743
|
+
"""The end-of-loop summary (#5): what changed, accept/reject, quality/cost
|
|
744
|
+
impact. Printed AFTER the legacy loop one-liner and written to the run-log."""
|
|
745
|
+
final_mean = (f"{r.final_incumbent_mean:.3f}"
|
|
746
|
+
if r.final_incumbent_mean is not None else "-")
|
|
747
|
+
accepted = r.promotions
|
|
748
|
+
rejected = sum(1 for x in r.results if x.attempted and
|
|
749
|
+
x.decision is not None and x.decision.action.value == "discard")
|
|
750
|
+
total_tokens = sum(x.tokens for x in r.results)
|
|
751
|
+
total_lat = sum(x.latency_ms for x in r.results)
|
|
752
|
+
outcome = ("aborted" if r.aborted else
|
|
753
|
+
("plateaued" if r.plateaued else "ran to max_cycles"))
|
|
754
|
+
return ("\narchforge summary: "
|
|
755
|
+
f"cycles={r.cycles_run} {outcome} "
|
|
756
|
+
f"accepted={accepted} rejected={rejected} queued={r.queued} "
|
|
757
|
+
f"final_mean={final_mean} "
|
|
758
|
+
f"total_tokens={total_tokens} latency={total_lat:.1f}ms")
|
|
759
|
+
|
|
760
|
+
|
|
761
|
+
def summary_to_json(r: LoopResult) -> dict:
|
|
762
|
+
return {
|
|
763
|
+
"cycles_run": r.cycles_run,
|
|
764
|
+
"promotions": r.promotions,
|
|
765
|
+
"queued": r.queued,
|
|
766
|
+
"plateaued": r.plateaued,
|
|
767
|
+
"aborted": r.aborted,
|
|
768
|
+
"abort_reason": r.abort_reason,
|
|
769
|
+
"final_incumbent_id": r.final_incumbent_id,
|
|
770
|
+
"final_incumbent_mean": r.final_incumbent_mean,
|
|
771
|
+
"total_tokens": sum(x.tokens for x in r.results),
|
|
772
|
+
"total_latency_ms": sum(x.latency_ms for x in r.results),
|
|
773
|
+
}
|
|
774
|
+
|
|
775
|
+
|
|
776
|
+
def _cmd_lint(path: str) -> int:
|
|
777
|
+
from archforge.models import Spec
|
|
778
|
+
|
|
779
|
+
spec = Spec.model_validate(json.loads(Path(path).read_text(encoding="utf-8")))
|
|
780
|
+
errors = lint(spec)
|
|
781
|
+
if not errors:
|
|
782
|
+
print("OK: spec is structurally valid.")
|
|
783
|
+
return 0
|
|
784
|
+
for e in errors:
|
|
785
|
+
loc = f" [{e.location}]" if e.location else ""
|
|
786
|
+
print(f"{e.code}{loc}: {e.message}")
|
|
787
|
+
return 1
|
|
788
|
+
|
|
789
|
+
|
|
790
|
+
def _cmd_init(args: argparse.Namespace) -> int:
|
|
791
|
+
"""Scaffold `.archforge/archforge.py` (user-editable config) + `.env.example`
|
|
792
|
+
(repo root). Never touches a real `.env`. Refuses to clobber an existing
|
|
793
|
+
`archforge.py` unless `--force`; never overwrites an existing `.env.example`."""
|
|
794
|
+
root = Path(args.root or _DEFAULT_ROOT)
|
|
795
|
+
cfg_path = root / "archforge.py"
|
|
796
|
+
|
|
797
|
+
# 1. archforge.py — refuse-clobber unless --force.
|
|
798
|
+
if cfg_path.exists() and not args.force:
|
|
799
|
+
print(f"! {cfg_path} already exists. Re-run with --force to overwrite "
|
|
800
|
+
"(your edits would be lost).", file=sys.stderr)
|
|
801
|
+
return 2
|
|
802
|
+
root.mkdir(parents=True, exist_ok=True) # .archforge/ (also the run state dir)
|
|
803
|
+
cfg_path.write_text(archforge_config_text(), encoding="utf-8")
|
|
804
|
+
print(f"created: {cfg_path} (edit a value to change a default; the file is ACTIVE as-is)")
|
|
805
|
+
|
|
806
|
+
# 2. .env.example — create once at the repo root (cwd), never overwrite.
|
|
807
|
+
env_example = Path(".env.example")
|
|
808
|
+
if env_example.exists():
|
|
809
|
+
print(f"kept: {env_example} (already present)")
|
|
810
|
+
else:
|
|
811
|
+
env_example.write_text(env_example_text(), encoding="utf-8")
|
|
812
|
+
print(f"created: {env_example}")
|
|
813
|
+
|
|
814
|
+
# 3. suite.json — seed the eval-task sidecar next to archforge.py (so --root
|
|
815
|
+
# relocations also move the seeded suite); never overwrite — the user may have
|
|
816
|
+
# tuned the tasks. Byte-identical to the CLI's one-task fallback fixture.
|
|
817
|
+
suite_path = root / "suite.json"
|
|
818
|
+
if suite_path.exists():
|
|
819
|
+
print(f"kept: {suite_path} (already present)")
|
|
820
|
+
else:
|
|
821
|
+
suite_path.write_text(_DEFAULT_SUITE_JSON, encoding="utf-8")
|
|
822
|
+
print(f"created: {suite_path} (edit the tasks to change what you optimize against)")
|
|
823
|
+
print(f"\nNext: edit {cfg_path}, then run `{PROG} evolve --seed <spec.json>`.")
|
|
824
|
+
return 0
|
|
825
|
+
|
|
826
|
+
|
|
827
|
+
# --------------------------------------------------------------------------- #
|
|
828
|
+
# entry
|
|
829
|
+
# --------------------------------------------------------------------------- #
|
|
830
|
+
|
|
831
|
+
|
|
832
|
+
def main(argv: list[str] | None = None, *,
|
|
833
|
+
components: Components | None = None) -> int:
|
|
834
|
+
"""Run the Forge CLI. `components` injects pre-configured organs (tests/embedding).
|
|
835
|
+
|
|
836
|
+
When `components` is None the CLI builds them per `--provider`: the default
|
|
837
|
+
`scripted` uses inert fakes; `--provider anthropic|openai|groq|gemini` wires
|
|
838
|
+
real LLMs. Only the evolve-family consults `components`; `status`/`report`/
|
|
839
|
+
`approve`/`reject`/`lint` read the stores directly.
|
|
840
|
+
"""
|
|
841
|
+
|
|
842
|
+
# Run in CWD (improvement #1): put the caller's working directory on sys.path
|
|
843
|
+
# so `--adapter module:Class` (and a host adapter that imports the project's
|
|
844
|
+
# own modules — e.g. `aede`) resolves from the dir the user runs in. Under the
|
|
845
|
+
# console script (`archforge-optimizer …`) CWD is NOT on sys.path by default,
|
|
846
|
+
# which forced the earlier `PYTHONPATH=".:.." python -m archforge` friction;
|
|
847
|
+
# under `python -m archforge` CWD is already present, so this is a no-op there.
|
|
848
|
+
# Belt-and-suspenders: insert if missing, never duplicate. All stores/suites/
|
|
849
|
+
# traces/.env are ALREADY CWD-relative via the `--root .archforge` literal, so
|
|
850
|
+
# this 3-line edit is the entirety of #1 (no path redesign needed).
|
|
851
|
+
_cwd = os.getcwd()
|
|
852
|
+
if _cwd not in sys.path:
|
|
853
|
+
sys.path.insert(0, _cwd)
|
|
854
|
+
|
|
855
|
+
parser = _build_parser()
|
|
856
|
+
args = parser.parse_args(argv)
|
|
857
|
+
if args.command is None:
|
|
858
|
+
parser.print_help()
|
|
859
|
+
return 0
|
|
860
|
+
if args.command == "lint":
|
|
861
|
+
return _cmd_lint(args.path)
|
|
862
|
+
if args.command == "status":
|
|
863
|
+
return _cmd_status(args)
|
|
864
|
+
if args.command == "report":
|
|
865
|
+
return _cmd_report(args)
|
|
866
|
+
if args.command == "approve":
|
|
867
|
+
return _cmd_approve(args)
|
|
868
|
+
if args.command == "reject":
|
|
869
|
+
return _cmd_reject(args)
|
|
870
|
+
if args.command == "init":
|
|
871
|
+
return _cmd_init(args)
|
|
872
|
+
if args.command == "evolve":
|
|
873
|
+
return _cmd_evolve(args, components=components, loop=False)
|
|
874
|
+
if args.command == "evolve-loop":
|
|
875
|
+
return _cmd_evolve(args, components=components, loop=True)
|
|
876
|
+
# argparse rejects unknown subcommands before dispatch, so this is unreachable.
|
|
877
|
+
return 0 # pragma: no cover
|
|
878
|
+
|
|
879
|
+
|
|
880
|
+
if __name__ == "__main__": # pragma: no cover
|
|
881
|
+
sys.exit(main())
|