archforge-optimizer 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. archforge/__init__.py +76 -0
  2. archforge/__main__.py +10 -0
  3. archforge/architect.py +442 -0
  4. archforge/cli.py +881 -0
  5. archforge/config.py +140 -0
  6. archforge/config_init.py +150 -0
  7. archforge/diff.py +206 -0
  8. archforge/engine.py +444 -0
  9. archforge/gatekeeper.py +290 -0
  10. archforge/host/__init__.py +20 -0
  11. archforge/host/adapters/__init__.py +41 -0
  12. archforge/host/adapters/base.py +311 -0
  13. archforge/host/adapters/helpers.py +163 -0
  14. archforge/host/adapters/langgraph.py +726 -0
  15. archforge/host/base.py +105 -0
  16. archforge/host/fake.py +380 -0
  17. archforge/judge/__init__.py +20 -0
  18. archforge/judge/base.py +257 -0
  19. archforge/judge/scripted.py +145 -0
  20. archforge/lint.py +180 -0
  21. archforge/llm/__init__.py +65 -0
  22. archforge/llm/_common.py +94 -0
  23. archforge/llm/anthropic.py +90 -0
  24. archforge/llm/base.py +90 -0
  25. archforge/llm/gemini.py +112 -0
  26. archforge/llm/groq.py +63 -0
  27. archforge/llm/openai.py +63 -0
  28. archforge/llm/scripted.py +134 -0
  29. archforge/middleware.py +181 -0
  30. archforge/models.py +435 -0
  31. archforge/mutate.py +214 -0
  32. archforge/otel.py +613 -0
  33. archforge/runlog.py +103 -0
  34. archforge/runner.py +153 -0
  35. archforge/spec_builder.py +126 -0
  36. archforge/stores/__init__.py +22 -0
  37. archforge/stores/_jsonl.py +81 -0
  38. archforge/stores/attempt_store.py +161 -0
  39. archforge/stores/spec_store.py +188 -0
  40. archforge/stores/trace_store.py +42 -0
  41. archforge/suite.py +248 -0
  42. archforge/userconfig.py +144 -0
  43. archforge_optimizer-0.1.0.dist-info/METADATA +420 -0
  44. archforge_optimizer-0.1.0.dist-info/RECORD +47 -0
  45. archforge_optimizer-0.1.0.dist-info/WHEEL +4 -0
  46. archforge_optimizer-0.1.0.dist-info/entry_points.txt +2 -0
  47. archforge_optimizer-0.1.0.dist-info/licenses/LICENSE +21 -0
archforge/cli.py ADDED
@@ -0,0 +1,881 @@
1
+ """ArchForge command-line interface — the Forge.
2
+
3
+ Wires the four organs (Architect, SuiteRunner, Judge, Gatekeeper) plus the
4
+ filesystem stores into the runnable surface the user actually touches:
5
+
6
+ archforge-optimizer lint <spec.json> validate a Spec
7
+ archforge-optimizer evolve [--root R] [--seed S] one Propose-Evaluate-Commit cycle
8
+ archforge-optimizer evolve-loop [...] repeat until budget cap or plateau
9
+ archforge-optimizer approve [<id>...|--all] drain the structural-change queue
10
+ archforge-optimizer reject <id> reject a queued change (active kept)
11
+ archforge-optimizer status incumbent Spec + lineage + counts
12
+ archforge-optimizer report aggregate deltas across attempts
13
+
14
+ Provider seam (the single place "real vs fake" lives at the CLI):
15
+ * `--provider scripted` (default) builds the zero-cost fakes — `FakeHostMAS`,
16
+ `ScriptedJudge`, `ScriptedArchitect` — so `evolve` is runnable end-to-end
17
+ with no LLM. Left unconfigured, the scripted architect simply plateaus
18
+ (it has no proposal to make); a deterministic *promotion* needs the organs
19
+ pre-configured, which is exactly what an embedding test supplies via
20
+ `components=...` (see `tests/integration/test_cli.py`).
21
+ * `--provider anthropic|openai|groq|gemini` wires a real `LLMClient` and the
22
+ real `Architect` + `Judge` over it — one provider swap, same organs.
23
+
24
+ `_ensure_incumbent` bootstraps the root incumbent from `--seed <spec.json>`
25
+ zero-LLM (commit as INCUMBENT + set_active) so the very first `evolve` has an
26
+ active spec to mutate. Approval/rollback keep `active` moving only through the
27
+ Gatekeeper (invariant I1); the CLI never mutates the pointer itself.
28
+ """
29
+
30
+ from __future__ import annotations
31
+
32
+ import argparse
33
+ import importlib
34
+ import json
35
+ import os
36
+ import sys
37
+ import warnings
38
+ from dataclasses import dataclass
39
+ from pathlib import Path
40
+
41
+ import archforge.models as m
42
+ from archforge.architect import Architect, ArchitectProtocol, ScriptedArchitect
43
+ from archforge.diff import format_diff, spec_diff
44
+ from archforge.engine import (
45
+ CycleCtx, CycleResult, DeployCtx, Engine, EngineConfig, LoopResult,
46
+ )
47
+ from archforge.gatekeeper import Gatekeeper
48
+ from archforge.host.base import HostMAS, Task
49
+ from archforge.host.fake import FakeHostMAS
50
+ from archforge.judge import ScriptedJudge
51
+ from archforge.judge.base import Judge, JudgeProtocol, default_rubric
52
+ from archforge.lint import lint
53
+ from archforge.runlog import RunLog
54
+ from archforge.stores import AttemptStore, SpecStore, TraceStore
55
+ from archforge.suite import Suite, load_suite_file
56
+ from archforge.config_init import archforge_config_text, env_example_text, _DEFAULT_SUITE_JSON
57
+
58
+ # System config (provider roster, CLI fixtures, PROG, .env loader) — single source
59
+ # in archforge.config. The TUNABLE defaults (tau/delta/repeats/provider/models/…)
60
+ # are NOT imported here; they resolve lazily from archforge.userconfig (the active
61
+ # project config made by `init`) at use time, so importing+running the CLI works
62
+ # before `init` has created .archforge/archforge.py and so an edit takes effect on
63
+ # the next run. See archforge/userconfig.py.
64
+ from archforge import userconfig as ucfg
65
+ from archforge.config import (
66
+ ALL_PROVIDERS as _PROVIDERS,
67
+ DEFAULT_SUITE_ID, DEFAULT_TASK_ID, DEFAULT_TASK_INPUT,
68
+ PROG, load_env,
69
+ )
70
+ from archforge.userconfig import ConfigNotInitialized
71
+
72
+ # Fixed run-state / config-discovery dir (a system path, independent of the
73
+ # tunable DEFAULT_ROOT_DIR which the embedder API reads via archforge.userconfig).
74
+ _DEFAULT_ROOT = ".archforge"
75
+
76
+
77
+ # --------------------------------------------------------------------------- #
78
+ # Injectable runtime organs — the test/embedding seam
79
+ # --------------------------------------------------------------------------- #
80
+
81
+
82
+ @dataclass
83
+ class Components:
84
+ """The four organs the Engine runs, injectable so a test/embedding can
85
+ supply pre-configured fakes (a scripted architect with a queued proposal +
86
+ a scripted judge with per-spec aggregates → a deterministic promotion).
87
+
88
+ When `main(..., components=None)` the CLI builds defaults per `--provider`:
89
+ `scripted` builds the inert fakes; a real provider builds the real
90
+ `Architect` + `Judge` over a real `LLMClient`.
91
+ """
92
+
93
+ host: HostMAS
94
+ judge: JudgeProtocol
95
+ architect: ArchitectProtocol
96
+ suite: Suite
97
+
98
+
99
+ def _import_adapter(dotted: str) -> HostMAS:
100
+ """Import an external MAS adapter from a ``module:Class`` (or ``module``)
101
+ dotted path and instantiate it. The class implements ``HostMAS`` (the kit's
102
+ ``BaseHostAdapter`` does), so it drops straight in as ``components.host`` —
103
+ its own ``__init__`` carries whatever its MAS needs (Lumina loads its base
104
+ prompts; a framework adapter wraps its graph). No PR into core to adapt a
105
+ new MAS: ``--adapter mypkg:MyAdapter`` wires it; ``--provider`` keeps the
106
+ Architect/Judge organs, ``--seed`` the bootstrap Spec."""
107
+ if ":" in dotted:
108
+ modpath, cls = dotted.split(":", 1)
109
+ else:
110
+ modpath, cls = dotted, ""
111
+ module = importlib.import_module(modpath)
112
+ if not cls:
113
+ # Bare module: expect it to expose a ``HostMAS``-protocol attr named
114
+ # ``HostMAS`` or the last path segment; else error loudly.
115
+ cls = "HostMAS"
116
+ try:
117
+ obj = getattr(module, cls)
118
+ except AttributeError as exc:
119
+ raise SystemExit(
120
+ f"--adapter: module {modpath!r} has no attribute {cls!r}. "
121
+ f"Pass it as `module:ClassName`."
122
+ ) from exc
123
+ if isinstance(obj, type):
124
+ return obj() # a HostMAS/BaseHostAdapter subclass → instance
125
+ if isinstance(obj, HostMAS):
126
+ return obj # already an instance
127
+ raise SystemExit(f"--adapter: {dotted!r} resolved to a {type(obj).__name__}, "
128
+ "not a HostMAS subclass or instance.")
129
+
130
+
131
+ # --------------------------------------------------------------------------- #
132
+ # arg parsing
133
+ # --------------------------------------------------------------------------- #
134
+
135
+
136
+ def _add_store_args(p: argparse.ArgumentParser) -> None:
137
+ p.add_argument("--root", default=_DEFAULT_ROOT,
138
+ help="archforge state directory (default: .archforge)")
139
+
140
+
141
+ def _add_evolve_args(p: argparse.ArgumentParser, *, loop: bool) -> None:
142
+ p.add_argument("--seed", metavar="PATH",
143
+ help="bootstrap the root incumbent from this Spec JSON (no active yet)")
144
+ p.add_argument("--adapter", metavar="DOTTED.PATH[:Class]",
145
+ help="import an external MAS adapter (a HostMAS/BaseHostAdapter "
146
+ "subclass) as the runtime host; pair with --provider for the "
147
+ "Architect/Judge organs. e.g. --adapter archforge_glue:LuminaAdapter")
148
+ p.add_argument("--provider", choices=_PROVIDERS, default=None,
149
+ help="LLM provider (default from archforge.py: anthropic/openai/groq/gemini "
150
+ "are real; scripted is the zero-cost fake)")
151
+ # API keys: a `.env` in the cwd is loaded first (load_env, ∴ real env wins), then
152
+ # the provider SDK reads its key var; --api-key overrides both. Tunable flags
153
+ # default to None here so the active config (.archforge/archforge.py) supplies
154
+ # the real default at resolve time — an explicit flag overrides the file.
155
+ p.add_argument("--env-file", default=None,
156
+ help="load provider API keys from this file before --provider "
157
+ "builds the organs (default from archforge.py; no-op if absent)")
158
+ p.add_argument("--api-key", default=None, help="provider API key (overrides .env/env)")
159
+ p.add_argument("--base-url", default=None, help="provider base URL override")
160
+ p.add_argument("--architect-model", default=None,
161
+ help="model id for the Architect (else the provider default)")
162
+ p.add_argument("--judge-model", default=None,
163
+ help="model id for the Judge (else the provider default)")
164
+ p.add_argument("--suite", metavar="PATH", default=None,
165
+ help="path to a suite.json (overrides the DEFAULT_SUITE_FILE tunable; "
166
+ "the file's tasks define what you optimize against)")
167
+ # thresholds — every default resolves from the active config (.archforge/archforge.py);
168
+ # a flag is None at the parser and filled from ucfg unless the user set it.
169
+ p.add_argument("--tau", type=float, default=None, help="promotion margin τ")
170
+ p.add_argument("--delta", type=float, default=None, help="regression floor δ (>= τ)")
171
+ p.add_argument("--repeats", type=int, default=None, help="R: repeats per eval-suite task")
172
+ if loop:
173
+ p.add_argument("--max-cycles", type=int, default=None,
174
+ help="cap on P-E-C cycles")
175
+ p.add_argument("--plateau-cycles", type=int, default=None,
176
+ help="K consecutive no-promotion cycles → plateau (E8)")
177
+ p.add_argument("--max-tokens-total", type=int, default=None,
178
+ help="total token budget cap; stop at or before reaching it (E3)")
179
+ p.add_argument("--max-tokens-per-cycle", type=int, default=None,
180
+ help="per-cycle token cap; abort mid-cycle if exceeded (E3)")
181
+ p.add_argument("--max-wall-ms-per-cycle", type=float, default=None,
182
+ help="per-cycle wall-clock cap (ms); aborts if exceeded — "
183
+ "bounds non-LLM nodes (retriever/tool/rule) that cost time, not tokens; "
184
+ "default None = no limit")
185
+
186
+
187
+ def _build_parser() -> argparse.ArgumentParser:
188
+ parser = argparse.ArgumentParser(
189
+ prog=PROG,
190
+ description="ArchForge — a self-improving meta-layer over multi-agent systems.",
191
+ )
192
+ sub = parser.add_subparsers(dest="command", metavar="COMMAND")
193
+
194
+ # --- evolve (one cycle) --------------------------------------------------
195
+ ev = sub.add_parser("evolve",
196
+ help="run one Propose-Evaluate-Commit cycle from the active incumbent")
197
+ _add_store_args(ev)
198
+ _add_evolve_args(ev, loop=False)
199
+
200
+ # --- evolve-loop ---------------------------------------------------------
201
+ evl = sub.add_parser("evolve-loop",
202
+ help="repeat evolve until the budget cap or a plateau")
203
+ _add_store_args(evl)
204
+ _add_evolve_args(evl, loop=True)
205
+
206
+ # --- approve (human gate, I4) --------------------------------------------
207
+ ap = sub.add_parser("approve",
208
+ help="approve queued (PENDING_HUMAN) structural changes → active")
209
+ _add_store_args(ap)
210
+ ap.add_argument("attempts", nargs="*",
211
+ help="attempt ids to approve (default: every pending change, in order)")
212
+ ap.add_argument("--all", action="store_true",
213
+ help="approve every pending change (default when none are named)")
214
+
215
+ # --- reject --------------------------------------------------------------
216
+ rj = sub.add_parser("reject",
217
+ help="reject a queued structural change — active is left alone")
218
+ _add_store_args(rj)
219
+ rj.add_argument("attempt_id", help="attempt id to reject")
220
+
221
+ # --- status / report -----------------------------------------------------
222
+ st = sub.add_parser("status", help="print the incumbent Spec + lineage + counts")
223
+ _add_store_args(st)
224
+
225
+ rp = sub.add_parser("report", help="print aggregate deltas across attempts")
226
+ _add_store_args(rp)
227
+
228
+ # --- lint ----------------------------------------------------------------
229
+ lint_p = sub.add_parser("lint", help="run the Spec Linter on a JSON Spec file")
230
+ _add_store_args(lint_p)
231
+ lint_p.add_argument("path", help="path to a Spec JSON file")
232
+
233
+ # --- init (scaffold user config) -----------------------------------------
234
+ init_p = sub.add_parser("init",
235
+ help="scaffold .archforge/archforge.py + .env.example for this project")
236
+ _add_store_args(init_p) # --root selects where archforge.py is written
237
+ init_p.add_argument("--force", action="store_true",
238
+ help="overwrite an existing .archforge/archforge.py")
239
+
240
+ return parser
241
+
242
+
243
+ # --------------------------------------------------------------------------- #
244
+ # helpers
245
+ # --------------------------------------------------------------------------- #
246
+
247
+
248
+ def _stores(root: str) -> tuple[SpecStore, AttemptStore, TraceStore]:
249
+ return SpecStore(root), AttemptStore(root), TraceStore(root)
250
+
251
+
252
+ def _arg(args: argparse.Namespace, attr: str, cfg_name: str):
253
+ """Resolve a CLI tunable: the explicit flag value, else the active-config default.
254
+
255
+ Every tunable flag defaults to ``None`` at the parser (so `--help` works pre-init
256
+ and a user's `.archforge/archforge.py` override isn't frozen into argparse). The
257
+ real default is read here from ``ucfg`` (disk post-init, or the in-memory sane
258
+ template under the pytest gate) only when the flag is omitted.
259
+ """
260
+ v = getattr(args, attr, None)
261
+ return v if v is not None else ucfg.get(cfg_name)
262
+
263
+
264
+ def _thresholds(args: argparse.Namespace) -> m.Thresholds:
265
+ return m.Thresholds(
266
+ tau=_arg(args, "tau", "DEFAULT_TAU"),
267
+ delta=_arg(args, "delta", "DEFAULT_DELTA"),
268
+ repeats=_arg(args, "repeats", "DEFAULT_REPEATS"),
269
+ plateau_cycles=_arg(args, "plateau_cycles", "DEFAULT_PLATEAU_CYCLES"),
270
+ )
271
+
272
+
273
+ def _config(args: argparse.Namespace) -> EngineConfig:
274
+ return EngineConfig(
275
+ max_cycles=_arg(args, "max_cycles", "DEFAULT_MAX_CYCLES"),
276
+ max_tokens_per_cycle=_arg(args, "max_tokens_per_cycle", "DEFAULT_MAX_TOKENS_PER_CYCLE"),
277
+ max_tokens_total=_arg(args, "max_tokens_total", "DEFAULT_MAX_TOKENS_TOTAL"),
278
+ repeats=_arg(args, "repeats", "DEFAULT_REPEATS"),
279
+ plateau_cycles=_arg(args, "plateau_cycles", "DEFAULT_PLATEAU_CYCLES"),
280
+ max_wall_ms_per_cycle=_arg(args, "max_wall_ms_per_cycle", "DEFAULT_MAX_WALL_MS_PER_CYCLE"),
281
+ )
282
+
283
+
284
+ def _default_components(args: argparse.Namespace) -> Components:
285
+ """Build the runtime organs for `--provider` (no injected `components`).
286
+
287
+ `--provider scripted` (default) builds the zero-cost fakes — well-defined but
288
+ inert: the scripted architect has no queued proposal so it plateaus, and the
289
+ scripted judge returns a neutral 0.5 base. A deterministic *promotion* needs
290
+ the organs pre-configured, which is the test path through
291
+ `main(..., components=...)`.
292
+
293
+ A real provider (`anthropic`/`openai`/`groq`/`gemini`)
294
+ builds ONE `LLMClient` via `make_client` and the REAL `Architect` + `Judge`
295
+ over it — the provider abstraction is the single seam, so neither organ
296
+ changes when the provider changes. The host stays `FakeHostMAS` for now
297
+ (a real MAS host is its own integration; the seam already accepts it).
298
+ """
299
+
300
+ # The eval suite: a JSON sidecar (.archforge/suite.json by default) if present,
301
+ # else the one-task fallback fixture (byte-identical to `init`'s seeded
302
+ # suite.json, so out-of-box == generated-default). --suite overrides the tunable.
303
+ suite_path = getattr(args, "suite", None) or ucfg.get("DEFAULT_SUITE_FILE")
304
+ suite = load_suite_file(suite_path) or Suite(
305
+ suite_id=DEFAULT_SUITE_ID, rubric_id=default_rubric().rubric_id,
306
+ tasks=[Task(task_id=DEFAULT_TASK_ID, input=DEFAULT_TASK_INPUT)])
307
+ provider = args.provider or ucfg.get("PROVIDER")
308
+ if provider == "scripted":
309
+ return Components(host=FakeHostMAS(), judge=ScriptedJudge(),
310
+ architect=ScriptedArchitect(), suite=suite)
311
+ from archforge.llm import make_client, LLMError
312
+
313
+ try:
314
+ llm = make_client(provider, api_key=args.api_key, base_url=args.base_url)
315
+ except LLMError as exc:
316
+ print(f"[provider] {exc}", file=sys.stderr)
317
+ raise
318
+ arch_models = ucfg.get("DEFAULT_ARCHITECT_MODELS")
319
+ judge_models = ucfg.get("DEFAULT_JUDGE_MODELS")
320
+ arch = Architect(llm, model=args.architect_model or arch_models[provider])
321
+ judge = Judge(llm, model=args.judge_model or judge_models[provider],
322
+ rubric=default_rubric())
323
+ return Components(host=FakeHostMAS(), judge=judge, architect=arch, suite=suite)
324
+
325
+
326
+ def _ensure_incumbent(args: argparse.Namespace, specs: SpecStore) -> str | None:
327
+ """Make sure an active incumbent exists before `evolve` runs.
328
+
329
+ If one already exists, return its id (no LLM, no overwrite). If none and a
330
+ `--seed` is supplied, lint + commit it as the root INCUMBENT and set active
331
+ (zero-LLM bootstrap). Returns the active id, or None if it could not.
332
+ """
333
+
334
+ current = specs.active_id()
335
+ if current is not None:
336
+ return current
337
+ seed_path = getattr(args, "seed", None)
338
+ if not seed_path:
339
+ return None
340
+ spec = m.Spec.model_validate(json.loads(Path(seed_path).read_text(encoding="utf-8")))
341
+ faults = lint(spec)
342
+ if faults:
343
+ raise SystemExit(
344
+ "refusing to bootstrap from --seed: spec fails the linter: "
345
+ + "; ".join(f"{f.code}({f.location or ''}): {f.message}" for f in faults)
346
+ )
347
+ spec_id = specs.commit(spec, parent_spec_id=None, status=m.SpecStatus.INCUMBENT)
348
+ specs.set_active(spec_id)
349
+ return spec_id
350
+
351
+
352
+ def _pending(atts: AttemptStore) -> list[m.Attempt]:
353
+ return [a for a in atts.all() if a.verdict is m.Verdict.PENDING_HUMAN]
354
+
355
+
356
+ def _fmt_spec(spec: m.Spec) -> str:
357
+ parent = spec.parent_spec_id or "(root)"
358
+ return (f" spec_id: {spec.spec_id}\n"
359
+ f" parent: {parent}\n"
360
+ f" status: {spec.status.value}\n"
361
+ f" created: {spec.created_at}\n"
362
+ f" nodes: {len(spec.nodes)}\n"
363
+ f" edges: {len(spec.edges)}")
364
+
365
+
366
+ # --------------------------------------------------------------------------- #
367
+ # subcommand handlers
368
+ # --------------------------------------------------------------------------- #
369
+
370
+
371
+ def _cmd_status(args: argparse.Namespace) -> int:
372
+ specs, atts, _ = _stores(args.root)
373
+ active_id = specs.active_id()
374
+ if active_id is None:
375
+ print("No incumbent yet. Bootstrap with `archforge-optimizer evolve --seed <spec.json>`.")
376
+ return 0
377
+ spec = specs.get(active_id)
378
+ print("active incumbent:")
379
+ print(_fmt_spec(spec))
380
+ chain = specs.lineage(active_id)
381
+ print("lineage: " + " <- ".join(chain))
382
+ print(f"specs known: {len(specs.known_ids())} "
383
+ f"archived: {len(specs.archived_ids())} "
384
+ f"attempts: {len(atts.all())} "
385
+ f"pending: {len(_pending(atts))}")
386
+ return 0
387
+
388
+
389
+ def _cmd_report(args: argparse.Namespace) -> int:
390
+ specs, atts, _ = _stores(args.root)
391
+ rows = atts.all()
392
+ if not rows:
393
+ print("No attempts recorded yet.")
394
+ return 0
395
+ print(f"{'attempt_id':<18}{'verdict':<14}{'kind':<13}{'target':<8}"
396
+ f"{'mean':>7}{'margin':>9}{'tokens':>8} change")
397
+ print("-" * 90)
398
+ for a in rows:
399
+ r = a.suite_result
400
+ mean = f"{r.mean:.3f}" if r is not None else "-"
401
+ margin = f"{r.margin_vs_incumbent:+.3f}" if r is not None else "-"
402
+ toks = str(r.tokens) if r is not None else "-"
403
+ print(f"{(a.attempt_id or '-'):<18}{a.verdict.value:<14}"
404
+ f"{a.change.kind.value:<13}{a.change.target:<8}"
405
+ f"{mean:>7}{margin:>9}{toks:>8} {a.change.diff}")
406
+ return 0
407
+
408
+
409
+ def _cmd_approve(args: argparse.Namespace) -> int:
410
+ specs, atts, _ = _stores(args.root)
411
+ gk = Gatekeeper(specs, atts, thresholds=_thresholds_for_approval())
412
+ if args.all or not args.attempts:
413
+ pending = _pending(atts)
414
+ if not pending:
415
+ print("No queued (PENDING_HUMAN) changes to approve.")
416
+ return 0
417
+ ids = [a.attempt_id for a in pending]
418
+ else:
419
+ ids = list(args.attempts)
420
+
421
+ approved: list[str] = []
422
+ for aid in ids:
423
+ try:
424
+ att = gk.approve(aid) # only path w/ Gatekeeper that moves active
425
+ except ValueError as exc:
426
+ print(f"! {aid}: {exc}")
427
+ continue
428
+ approved.append(aid)
429
+ print(f"approved {aid}: active = {att.candidate_spec_id} (verdict={att.verdict.value})")
430
+ if not approved:
431
+ print("Nothing approved.")
432
+ return 1
433
+ print(f"approved {len(approved)} change(s). active incumbent: {specs.active_id()}")
434
+ return 0
435
+
436
+
437
+ def _cmd_reject(args: argparse.Namespace) -> int:
438
+ specs, atts, _ = _stores(args.root)
439
+ gk = Gatekeeper(specs, atts, thresholds=_thresholds_for_approval())
440
+ try:
441
+ att = gk.reject(args.attempt_id)
442
+ except ValueError as exc:
443
+ print(f"! {args.attempt_id}: {exc}")
444
+ return 1
445
+ print(f"rejected {args.attempt_id}: verdict={att.verdict.value}; "
446
+ f"active unchanged at {specs.active_id()}")
447
+ return 0
448
+
449
+
450
+ def _thresholds_for_approval() -> m.Thresholds:
451
+ # approve/reject never consult τ/δ; defaults are fine (the verdicts already
452
+ # carry the cycle's margin on their suite_result).
453
+ return m.Thresholds()
454
+
455
+
456
+ def _install_resource_warning_quieteners() -> None:
457
+ """Suppress `ResourceWarning: unclosed <ssl.SSLSocket>` noise emitted by the
458
+ per-call LLM clients' (groq/google-genai) sockets closing in langgraph's
459
+ async executor. Extracted to module scope so the silence contract is pin-able
460
+ by a unit test (the mechanism took three iterations settle — see pitfall below).
461
+
462
+ A real socket's `__del__` emits these via the standard
463
+ `warnings.warn(..., ResourceWarning)` path — controlled by the warnings
464
+ FILTER, which is what makes them fiddly: a library importing after this point
465
+ (httpx/httpcore/chromadb/langgraph) calls `warnings.simplefilter` /
466
+ `filterwarnings` at import, which PREPENDS its entry above ours, and the FIRST
467
+ matching filter wins → a `simplefilter("ignore", ResourceWarning)` we set here
468
+ gets shadowed and the warnings print anyway (observed in the first CLI run).
469
+ The `def __init__(self)` / `threading.py:301` / `langgraph/pregel/_utils.py:235`
470
+ lines are tracemalloc allocation-site fingers ResourceWarning attaches to its
471
+ source resolution, NOT emitters.
472
+
473
+ The robust mechanism (shadow-proof): override `warnings.showwarning` — the
474
+ terminal sink every warning routes to AFTER the filters decide to *show* it (so
475
+ a library's re-armed "default" filter still routes to us, and we drop
476
+ ResourceWarning here). Can't be shadowed by filter re-arming. Drop
477
+ ResourceWarning ONLY — a real archforge DeprecationWarning stays visible for
478
+ debugging. Belt-and-suspenders: also silence `sys.unraisablehook` for the rare
479
+ `__del__`-RAISES path (a genuine bug), limited to ResourceWarning so other
480
+ unraisable exceptions stay loud. Scoped to the evolve command: `init`/`status`/
481
+ `report` + the pytest suite never enter here, retaining full warning visibility.
482
+ archforge opens no raw sockets itself → ResourceWarning is third-party
483
+ async-pool noise.
484
+ """
485
+ _default_showwarning = warnings.showwarning
486
+ _default_unraisable_hook = sys.unraisablehook
487
+
488
+ def _quiet_showwarning(message, category, filename, lineno, file=None,
489
+ line=None):
490
+ if issubclass(category, ResourceWarning):
491
+ return
492
+ _default_showwarning(message, category, filename, lineno, file, line=line)
493
+
494
+ def _quiet_resource_warning(unr_args, /):
495
+ exc = getattr(unr_args, "exc_value", None)
496
+ if isinstance(exc, ResourceWarning):
497
+ return
498
+ _default_unraisable_hook(unr_args)
499
+
500
+ warnings.showwarning = _quiet_showwarning
501
+ sys.unraisablehook = _quiet_resource_warning
502
+
503
+
504
+ # Activate the quietener at IMPORT time (module scope), not inside `_cmd_evolve`.
505
+ # The unclosed-SSL ResourceWarnings emit in THREE windows: (1) import/init time
506
+ # as libraries (httpx/langgraph/google-genai) spin up + tear down sockets, (2)
507
+ # during the evolve run, (3) at interpreter shutdown GC. Scoping the install to
508
+ # `_cmd_evolve` (the earlier attempt) covered only window 2 — windows 1 and 3
509
+ # still printed (observed: a top batch with `ast.py:46` fingers before cycle 0
510
+ # and a bottom batch with `<sys>:0` fingers after the last cycle, both printing
511
+ # the default `Enable tracemalloc` hint). `python -m archforge.cli` imports this
512
+ # module FIRST (it's `__main__`), so installing here runs before any provider
513
+ # library is imported (those come lazily at evolve time per the import-laziness
514
+ # contract) → the sink is in place for all three windows. The override is
515
+ # shadow-proof against a library re-arming the warnings FILTER (proven in
516
+ # tests/unit/test_cli_warning_quietener.py); a library reassigning
517
+ # `warnings.showwarning` outright would defeat it, but none of the runtime deps
518
+ # do that (only pytest's recorder does, and the CLI isn't under pytest).
519
+ _install_resource_warning_quieteners()
520
+
521
+
522
+ def _cmd_evolve(args: argparse.Namespace, *, components: Components | None,
523
+ loop: bool) -> int:
524
+ # Injected `components` (the test/embedding path) always win — they ARE the
525
+ # organs, by contract. Otherwise build organs per `--provider`. While a real
526
+ # tunable is needed we require `init` to have run (the "pip install → init → CLI
527
+ # works" contract): a project with no .archforge/archforge.py gets the init hint
528
+ # and rc 1 instead of a bogus run. No-op under the pytest gate (tests use the
529
+ # in-memory sane template).
530
+ if components is None:
531
+ try:
532
+ ucfg.ensure_initialized()
533
+ except ConfigNotInitialized as exc:
534
+ print(str(exc), file=sys.stderr)
535
+ return 1
536
+ # Load API keys from the cwd's .env (real env wins; --api-key wins above
537
+ # both). No-op if the file is missing or for --provider scripted.
538
+ load_env(getattr(args, "env_file", None) or ucfg.get("DEFAULT_ENV_FILE"))
539
+ if (args.provider or ucfg.get("PROVIDER")) != "scripted":
540
+ try:
541
+ components = _default_components(args) # builds a real LLMClient
542
+ except Exception:
543
+ return 2 # message already printed
544
+
545
+ specs, atts, traces = _stores(args.root)
546
+
547
+ # zero-LLM bootstrap of the root incumbent from --seed (if none active)
548
+ active = _ensure_incumbent(args, specs)
549
+ if active is None:
550
+ print("No active incumbent and no --seed given; bootstrap with "
551
+ "`archforge-optimizer evolve --seed <spec.json>` first.", file=sys.stderr)
552
+ return 1
553
+
554
+ organs = components or _default_components(args)
555
+
556
+ # --adapter: swap the runtime host for an external MAS adapter (a HostMAS /
557
+ # BaseHostAdapter) while keeping the --provider organs (Architect/Judge).
558
+ # This is the "adapt any MAS" seam: point at an adapter class, no core edit.
559
+ adapter_path = getattr(args, "adapter", None)
560
+ if adapter_path:
561
+ organs = Components(host=_import_adapter(adapter_path), judge=organs.judge,
562
+ architect=organs.architect, suite=organs.suite)
563
+
564
+ # The expressive surfaces (improvements #2/#3/#4/#5) wire through the Engine's
565
+ # opt-in hooks (on_cycle fires each attempted cycle; on_deploy fires on a
566
+ # promote). `on_cycle` renders the legacy line + the compact card to stdout
567
+ # AND appends a JSON form to a fail-soft run-log (``<root>/runs/last.json``);
568
+ # `on_deploy` writes the unified ``optimized.json`` envelope (Tier-2 deploy,
569
+ # auto-synced from the CLI — improvement #4). When the hook fires, the legacy
570
+ # `_print_cycle`/`_print_loop` must NOT re-print the per-cycle line (the hook
571
+ # already did); `hooks_wired` gates that fallback.
572
+ runlog = RunLog(Path(args.root) / "runs" / "last.json")
573
+ if not runlog.enabled:
574
+ print(f" (run log disabled: {runlog.reason})")
575
+ optimized_path = Path(args.root) / "optimized.json"
576
+
577
+ def _on_cycle(result: CycleResult, ctx: CycleCtx) -> None:
578
+ # Per-cycle visibility (hooks fire DURING the run, so evolve-loop now shows
579
+ # each cycle, not just the final summary). The legacy one-liner (kept
580
+ # verbatim — the integration tests assert its substrings) PLUS the card.
581
+ _print_cycle(result)
582
+ if result.attempted:
583
+ print(render_card(result, ctx))
584
+ runlog.append_cycle(card_to_json(result, ctx))
585
+
586
+ def _on_deploy(spec: m.Spec, dctx: DeployCtx) -> None:
587
+ # Tier-2 deploy auto-synced from the CLI (improvement #4): one
588
+ # ``optimized.json`` production loads to apply the winner — knobs + lineage
589
+ # + the decision + scores that justified the promote. Printed notice keeps
590
+ # the user informed; the file is the artifact (rollback = delete it).
591
+ try:
592
+ from archforge.host.adapters import export_optimized
593
+ except ImportError: # pragma: no cover
594
+ return
595
+ export_optimized(spec, optimized_path, parent=dctx.parent,
596
+ promoted_at_cycle=dctx.promoted_at_cycle,
597
+ decision=dctx.decision, cand_run=dctx.cand_run,
598
+ inc_run=dctx.inc_run)
599
+ print(f" deployed: {optimized_path} "
600
+ f"spec_id={spec.spec_id or spec.compute_spec_id()[:8]} "
601
+ f"(knobs + scores + decision)")
602
+
603
+ engine = Engine(
604
+ host=organs.host, judge=organs.judge, architect=organs.architect,
605
+ spec_store=specs, attempt_store=atts, trace_store=traces,
606
+ suite=organs.suite, thresholds=_thresholds(args), config=_config(args),
607
+ on_cycle=_on_cycle, on_deploy=_on_deploy,
608
+ )
609
+
610
+ if not loop:
611
+ engine.evolve_cycle(cycle=0)
612
+ # `on_cycle` already printed the line + card; no post-hoc re-print.
613
+ return 0
614
+ lr = engine.evolve_loop()
615
+ # `on_cycle` printed each cycle's line + card live; the loop summary is the
616
+ # only post-hoc print (legacy one-liner kept verbatim + the summary block).
617
+ print(f"{PROG} evolve-loop: cycles_run={lr.cycles_run} promotions={lr.promotions} "
618
+ f"queued={lr.queued} plateaued={'true' if lr.plateaued else 'false'} "
619
+ f"aborted={'true' if lr.aborted else 'false'} "
620
+ f"final_incumbent={lr.final_incumbent_id} "
621
+ f"final_mean={lr.final_incumbent_mean if lr.final_incumbent_mean is not None else '-'!s}")
622
+ if lr.aborted and lr.abort_reason:
623
+ print(f" abort_reason: {lr.abort_reason}")
624
+ print(summary_block(lr))
625
+ runlog.write_summary(summary_to_json(lr))
626
+ return 0
627
+
628
+
629
+ def _print_cycle(r: CycleResult) -> int:
630
+ if not r.attempted:
631
+ print(f"{PROG} evolve: cycle={r.cycle} attempted=false action=none "
632
+ f"note={r.note}")
633
+ return 0
634
+ action = r.decision.action.value if r.decision else "none"
635
+ inc = f"{r.incumbent_mean:.3f}" if r.incumbent_mean is not None else "-"
636
+ cand = f"{r.candidate_mean:.3f}" if r.candidate_mean is not None else "-"
637
+ margin = f"{r.decision.margin:+.3f}" if r.decision else "-"
638
+ print(f"{PROG} evolve: cycle={r.cycle} attempted=true action={action} "
639
+ f"margin={margin} incumbent_mean={inc} candidate_mean={cand} "
640
+ f"promoted={'true' if r.promoted else 'false'} "
641
+ f"queued={'true' if r.queued else 'false'} "
642
+ f"attempt_id={r.applied_attempt_id} tokens={r.tokens}")
643
+ if r.note:
644
+ print(f" note: {r.note}")
645
+ return 0
646
+
647
+
648
+ # --------------------------------------------------------------------------- #
649
+ # Expressive surfaces — the cycle card (#2/#3/#5) + loop summary (#5)
650
+ # --------------------------------------------------------------------------- #
651
+ #
652
+ # The legacy one-liners above stay verbatim (the integration tests assert their
653
+ # substrings). The card is APPENDED after the legacy line: a 4-line block carrying
654
+ # the mutation diff (#2), the cand-vs-inc rubric scores, the cost delta, and the
655
+ # rejection/accept explanation (#3). It renders BOTH to stdout AND to a JSON run
656
+ # log (improvement #5's "file" sink) when a RunLog is wired. `attempted=false`
657
+ # cycles print only the legacy line (nothing to diff). The envelope (#4) is
658
+ # written by `_on_deploy` on AUTO_PROMOTE; the loop summary card is printed +
659
+ # logged at the end.
660
+
661
+ def _fmt_dims(run) -> str:
662
+ """Compact ``corr X.XX compl X.XX`` from a SuiteRun's per-repeat rubric
663
+ dims. Tolerant of a None run or a run with no scores (-> empty)."""
664
+ if run is None:
665
+ return ""
666
+ sums: dict[str, float] = {}
667
+ n: dict[str, int] = {}
668
+ for rs in getattr(run, "scores", None) or []:
669
+ for dim, val in (getattr(rs, "rubric_scores", None) or {}).items():
670
+ try:
671
+ v = float(val)
672
+ except (TypeError, ValueError):
673
+ continue
674
+ sums[dim] = sums.get(dim, 0.0) + v
675
+ n[dim] = n.get(dim, 0) + 1
676
+ parts = [f"{d} {sums[d] / n[d]:.2f}" for d in sums]
677
+ return " ".join(parts)
678
+
679
+
680
+ def render_card(result: CycleResult, ctx: CycleCtx) -> str:
681
+ """The compact 4-line card appended after the legacy cycle line (#2/#3/#5).
682
+
683
+ The lines (intentionally short — full per-step breakdown stays in the
684
+ persisted JSON + ``report``)::
685
+
686
+ change: <Change.kind> on "<target>" — <field: old -> new | roster delta>
687
+ scores: cand <m.mm> [dims] vs inc <m.mm> [dims]
688
+ cost: tokens +<Δ> (inc I → cand C) latency +<Δ>ms
689
+ why: <Decision.reason> (rule=<by_rule>)
690
+
691
+ The ``change`` line uses ``spec_diff`` (real field-level before/after, since
692
+ the mutator's ``payload`` evaporates) summarized via ``format_diff``. Empty
693
+ diff (candidate == parent, e.g. a no-op proposal that passed lint) renders
694
+ ``(no change)``. Returns the block WITHOUT a trailing newline so callers can
695
+ join pipes; printing callers add the newline.
696
+ """
697
+ ch = ctx.change
698
+ entries = spec_diff(ctx.parent_spec, ctx.candidate_spec)
699
+ change_line = format_diff(entries)
700
+ inc = ctx.inc_run
701
+ cand = ctx.cand_run
702
+ inc_mean = f"{inc.mean:.3f}" if getattr(inc, "mean", None) is not None else "-"
703
+ cand_mean = f"{cand.mean:.3f}" if getattr(cand, "mean", None) is not None else "-"
704
+ inc_dims = _fmt_dims(inc)
705
+ cand_dims = _fmt_dims(cand)
706
+ inc_toks = getattr(inc, "tokens", 0) or 0
707
+ cand_toks = getattr(cand, "tokens", 0) or 0
708
+ inc_lat = getattr(inc, "latency_ms", 0.0) or 0.0
709
+ cand_lat = getattr(cand, "latency_ms", 0.0) or 0.0
710
+ rule = result.decision.by_rule if result.decision else ""
711
+ reason = (result.decision.reason if result.decision else result.note) or ""
712
+ action = result.decision.action.value if result.decision else "none"
713
+ lines = [
714
+ f' change: {ch.kind.value} on "{ch.target}" — {change_line}',
715
+ f" scores: cand {cand_mean} [{cand_dims}] vs inc {inc_mean} [{inc_dims}]",
716
+ f" cost: tokens +{cand_toks - inc_toks} (inc {inc_toks} → cand {cand_toks})"
717
+ f" latency +{cand_lat - inc_lat:.1f}ms",
718
+ f" why: {reason} (rule={rule}, action={action})",
719
+ ]
720
+ return "\n".join(lines)
721
+
722
+
723
+ def card_to_json(result: CycleResult, ctx: CycleCtx) -> dict:
724
+ """The machine-readable form appended to the run-log (one entry per cycle)."""
725
+ return {
726
+ "cycle": result.cycle,
727
+ "attempted": result.attempted,
728
+ "action": result.decision.action.value if result.decision else None,
729
+ "change_kind": ctx.change.kind.value,
730
+ "target": ctx.change.target,
731
+ "diff": format_diff(spec_diff(ctx.parent_spec, ctx.candidate_spec)),
732
+ "incumbent_mean": result.incumbent_mean,
733
+ "candidate_mean": result.candidate_mean,
734
+ "margin": result.decision.margin if result.decision else None,
735
+ "by_rule": result.decision.by_rule if result.decision else None,
736
+ "reason": result.decision.reason if result.decision else result.note,
737
+ "tokens": result.tokens,
738
+ "latency_ms": result.latency_ms,
739
+ }
740
+
741
+
742
+ def summary_block(r: LoopResult) -> str:
743
+ """The end-of-loop summary (#5): what changed, accept/reject, quality/cost
744
+ impact. Printed AFTER the legacy loop one-liner and written to the run-log."""
745
+ final_mean = (f"{r.final_incumbent_mean:.3f}"
746
+ if r.final_incumbent_mean is not None else "-")
747
+ accepted = r.promotions
748
+ rejected = sum(1 for x in r.results if x.attempted and
749
+ x.decision is not None and x.decision.action.value == "discard")
750
+ total_tokens = sum(x.tokens for x in r.results)
751
+ total_lat = sum(x.latency_ms for x in r.results)
752
+ outcome = ("aborted" if r.aborted else
753
+ ("plateaued" if r.plateaued else "ran to max_cycles"))
754
+ return ("\narchforge summary: "
755
+ f"cycles={r.cycles_run} {outcome} "
756
+ f"accepted={accepted} rejected={rejected} queued={r.queued} "
757
+ f"final_mean={final_mean} "
758
+ f"total_tokens={total_tokens} latency={total_lat:.1f}ms")
759
+
760
+
761
+ def summary_to_json(r: LoopResult) -> dict:
762
+ return {
763
+ "cycles_run": r.cycles_run,
764
+ "promotions": r.promotions,
765
+ "queued": r.queued,
766
+ "plateaued": r.plateaued,
767
+ "aborted": r.aborted,
768
+ "abort_reason": r.abort_reason,
769
+ "final_incumbent_id": r.final_incumbent_id,
770
+ "final_incumbent_mean": r.final_incumbent_mean,
771
+ "total_tokens": sum(x.tokens for x in r.results),
772
+ "total_latency_ms": sum(x.latency_ms for x in r.results),
773
+ }
774
+
775
+
776
+ def _cmd_lint(path: str) -> int:
777
+ from archforge.models import Spec
778
+
779
+ spec = Spec.model_validate(json.loads(Path(path).read_text(encoding="utf-8")))
780
+ errors = lint(spec)
781
+ if not errors:
782
+ print("OK: spec is structurally valid.")
783
+ return 0
784
+ for e in errors:
785
+ loc = f" [{e.location}]" if e.location else ""
786
+ print(f"{e.code}{loc}: {e.message}")
787
+ return 1
788
+
789
+
790
+ def _cmd_init(args: argparse.Namespace) -> int:
791
+ """Scaffold `.archforge/archforge.py` (user-editable config) + `.env.example`
792
+ (repo root). Never touches a real `.env`. Refuses to clobber an existing
793
+ `archforge.py` unless `--force`; never overwrites an existing `.env.example`."""
794
+ root = Path(args.root or _DEFAULT_ROOT)
795
+ cfg_path = root / "archforge.py"
796
+
797
+ # 1. archforge.py — refuse-clobber unless --force.
798
+ if cfg_path.exists() and not args.force:
799
+ print(f"! {cfg_path} already exists. Re-run with --force to overwrite "
800
+ "(your edits would be lost).", file=sys.stderr)
801
+ return 2
802
+ root.mkdir(parents=True, exist_ok=True) # .archforge/ (also the run state dir)
803
+ cfg_path.write_text(archforge_config_text(), encoding="utf-8")
804
+ print(f"created: {cfg_path} (edit a value to change a default; the file is ACTIVE as-is)")
805
+
806
+ # 2. .env.example — create once at the repo root (cwd), never overwrite.
807
+ env_example = Path(".env.example")
808
+ if env_example.exists():
809
+ print(f"kept: {env_example} (already present)")
810
+ else:
811
+ env_example.write_text(env_example_text(), encoding="utf-8")
812
+ print(f"created: {env_example}")
813
+
814
+ # 3. suite.json — seed the eval-task sidecar next to archforge.py (so --root
815
+ # relocations also move the seeded suite); never overwrite — the user may have
816
+ # tuned the tasks. Byte-identical to the CLI's one-task fallback fixture.
817
+ suite_path = root / "suite.json"
818
+ if suite_path.exists():
819
+ print(f"kept: {suite_path} (already present)")
820
+ else:
821
+ suite_path.write_text(_DEFAULT_SUITE_JSON, encoding="utf-8")
822
+ print(f"created: {suite_path} (edit the tasks to change what you optimize against)")
823
+ print(f"\nNext: edit {cfg_path}, then run `{PROG} evolve --seed <spec.json>`.")
824
+ return 0
825
+
826
+
827
+ # --------------------------------------------------------------------------- #
828
+ # entry
829
+ # --------------------------------------------------------------------------- #
830
+
831
+
832
+ def main(argv: list[str] | None = None, *,
833
+ components: Components | None = None) -> int:
834
+ """Run the Forge CLI. `components` injects pre-configured organs (tests/embedding).
835
+
836
+ When `components` is None the CLI builds them per `--provider`: the default
837
+ `scripted` uses inert fakes; `--provider anthropic|openai|groq|gemini` wires
838
+ real LLMs. Only the evolve-family consults `components`; `status`/`report`/
839
+ `approve`/`reject`/`lint` read the stores directly.
840
+ """
841
+
842
+ # Run in CWD (improvement #1): put the caller's working directory on sys.path
843
+ # so `--adapter module:Class` (and a host adapter that imports the project's
844
+ # own modules — e.g. `aede`) resolves from the dir the user runs in. Under the
845
+ # console script (`archforge-optimizer …`) CWD is NOT on sys.path by default,
846
+ # which forced the earlier `PYTHONPATH=".:.." python -m archforge` friction;
847
+ # under `python -m archforge` CWD is already present, so this is a no-op there.
848
+ # Belt-and-suspenders: insert if missing, never duplicate. All stores/suites/
849
+ # traces/.env are ALREADY CWD-relative via the `--root .archforge` literal, so
850
+ # this 3-line edit is the entirety of #1 (no path redesign needed).
851
+ _cwd = os.getcwd()
852
+ if _cwd not in sys.path:
853
+ sys.path.insert(0, _cwd)
854
+
855
+ parser = _build_parser()
856
+ args = parser.parse_args(argv)
857
+ if args.command is None:
858
+ parser.print_help()
859
+ return 0
860
+ if args.command == "lint":
861
+ return _cmd_lint(args.path)
862
+ if args.command == "status":
863
+ return _cmd_status(args)
864
+ if args.command == "report":
865
+ return _cmd_report(args)
866
+ if args.command == "approve":
867
+ return _cmd_approve(args)
868
+ if args.command == "reject":
869
+ return _cmd_reject(args)
870
+ if args.command == "init":
871
+ return _cmd_init(args)
872
+ if args.command == "evolve":
873
+ return _cmd_evolve(args, components=components, loop=False)
874
+ if args.command == "evolve-loop":
875
+ return _cmd_evolve(args, components=components, loop=True)
876
+ # argparse rejects unknown subcommands before dispatch, so this is unreachable.
877
+ return 0 # pragma: no cover
878
+
879
+
880
+ if __name__ == "__main__": # pragma: no cover
881
+ sys.exit(main())