archforge-optimizer 0.1.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. archforge_optimizer-0.3.0/.gitignore +6 -0
  2. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/PKG-INFO +104 -81
  3. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/README.md +101 -79
  4. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/cli.py +198 -17
  5. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/config.py +1 -1
  6. archforge_optimizer-0.3.0/archforge/config_init.py +592 -0
  7. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/llm/__init__.py +15 -22
  8. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/llm/base.py +30 -2
  9. archforge_optimizer-0.3.0/archforge/llm/litellm.py +141 -0
  10. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/pyproject.toml +5 -4
  11. archforge_optimizer-0.1.0/.gitignore +0 -12
  12. archforge_optimizer-0.1.0/archforge/config_init.py +0 -150
  13. archforge_optimizer-0.1.0/archforge/llm/_common.py +0 -94
  14. archforge_optimizer-0.1.0/archforge/llm/anthropic.py +0 -90
  15. archforge_optimizer-0.1.0/archforge/llm/gemini.py +0 -112
  16. archforge_optimizer-0.1.0/archforge/llm/groq.py +0 -63
  17. archforge_optimizer-0.1.0/archforge/llm/openai.py +0 -63
  18. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/LICENSE +0 -0
  19. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/__init__.py +0 -0
  20. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/__main__.py +0 -0
  21. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/architect.py +0 -0
  22. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/diff.py +0 -0
  23. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/engine.py +0 -0
  24. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/gatekeeper.py +0 -0
  25. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/host/__init__.py +0 -0
  26. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/host/adapters/__init__.py +0 -0
  27. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/host/adapters/base.py +0 -0
  28. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/host/adapters/helpers.py +0 -0
  29. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/host/adapters/langgraph.py +0 -0
  30. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/host/base.py +0 -0
  31. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/host/fake.py +0 -0
  32. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/judge/__init__.py +0 -0
  33. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/judge/base.py +0 -0
  34. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/judge/scripted.py +0 -0
  35. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/lint.py +0 -0
  36. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/llm/scripted.py +0 -0
  37. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/middleware.py +0 -0
  38. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/models.py +0 -0
  39. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/mutate.py +0 -0
  40. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/otel.py +0 -0
  41. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/runlog.py +0 -0
  42. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/runner.py +0 -0
  43. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/spec_builder.py +0 -0
  44. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/stores/__init__.py +0 -0
  45. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/stores/_jsonl.py +0 -0
  46. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/stores/attempt_store.py +0 -0
  47. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/stores/spec_store.py +0 -0
  48. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/stores/trace_store.py +0 -0
  49. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/suite.py +0 -0
  50. {archforge_optimizer-0.1.0 → archforge_optimizer-0.3.0}/archforge/userconfig.py +0 -0
@@ -0,0 +1,6 @@
1
+ __pycache__/
2
+ .pytest_cache/
3
+ data/
4
+ *.db
5
+ .env
6
+
@@ -1,10 +1,11 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: archforge-optimizer
3
- Version: 0.1.0
4
- Summary: ArchForge — a self-improving meta-layer over multi-agent systems
3
+ Version: 0.3.0
4
+ Summary: ArchForge: a self-improving meta-layer over multi-agent systems
5
5
  License-Expression: MIT
6
6
  License-File: LICENSE
7
7
  Requires-Python: >=3.11
8
+ Requires-Dist: litellm
8
9
  Requires-Dist: pydantic>=2.7
9
10
  Requires-Dist: python-dotenv>=1.0
10
11
  Provides-Extra: dev
@@ -30,21 +31,38 @@ Description-Content-Type: text/markdown
30
31
 
31
32
  # ArchForge
32
33
 
34
+ <p align="center">
35
+ <img src="images/logo.png" width="160">
36
+ </p>
37
+
38
+ <p align="center">
39
+ <img src="https://img.shields.io/badge/Python-3.11%2B-3776AB?logo=python&logoColor=white">
40
+ <img src="https://img.shields.io/badge/License-MIT-green.svg">
41
+ </p>
42
+
33
43
  > A self-improving meta-layer over multi-agent systems.
34
- > Point it at your graph, give it a rubric, and it evolves your pipeline — one proven change per cycle.
44
+ > Point it at your graph, give it a rubric, and it evolves your pipeline, one proven change per cycle.
45
+
46
+ **Minimal core dependencies** · **optional provider and tracing integrations** · install from PyPI with `pip install archforge-optimizer` · exercised end-to-end on a real LangGraph MAS (groq + google-genai + chroma, OpenTelemetry-traced).
47
+
48
+ ## What is ArchForge?
35
49
 
36
- **Python ≥ 3.11** · **zero hard deps beyond pydantic** · **MIT-licensed** · install from PyPI with `pip install archforge-optimizer` · exercised end-to-end on a real LangGraph MAS (groq + google-genai + chroma, OpenTelemetry-traced).
50
+ ArchForge sits **on top** of an existing multi-agent system (MAS) and improves it over time. In plain terms:
37
51
 
38
- ArchForge sits **on top** of an existing multi-agent system (MAS) and improves it run-over-run. Each cycle it inspects where the judge docked points, proposes **one** targeted change — rewriting an agent's prompt, tuning a knob, adding a verifier, re-wiring a node, swapping a model — and keeps it only if it measurably beats the incumbent on a held-out suite. The host MAS keeps running tasks as normal; ArchForge observes the runs and feeds back an improved pipeline.
52
+ 1. **Look.** It inspects where the judge docked points on the last run.
53
+ 2. **Propose.** It proposes **one** targeted change: rewriting an agent's prompt, tuning a knob, adding a verifier, rewiring a node, or swapping a model.
54
+ 3. **Keep or drop.** It keeps the change only if it measurably beats the current pipeline on a held-out suite.
55
+
56
+ Your MAS keeps running tasks as normal. ArchForge watches those runs and feeds back an improved pipeline.
39
57
 
40
58
  The design is deliberately minimal and verifiable:
41
59
 
42
- - **One protected incumbent.** A candidate never touches production config; it's promoted only when its mean score beats the incumbent's by at least the margin τ.
43
- - **Immutable, versioned Specs are the single source of truth.** Evolving the pipeline = swapping which Spec the host instantiates, never patching live state.
44
- - **Hybrid autonomy.** Safe small edits (prompt/knob) auto-apply; structural edits (roster/graph/model) queue for human approval.
45
- - **Observation/control asymmetry.** The wrapper records traces and reports the active Spec, but never rewrites prompts mid-run. All mutation happens *between* runs, on the Spec.
60
+ - **One protected incumbent.** A candidate never touches production config. It is promoted only when its mean score beats the incumbent's by at least the margin τ.
61
+ - **Immutable, versioned Specs are the single source of truth.** Evolving the pipeline means swapping which Spec the host instantiates, never patching live state.
62
+ - **Hybrid autonomy.** Safe small edits (prompt/knob) apply automatically. Structural edits (roster/graph/model) queue for human approval.
63
+ - **Observation/control asymmetry.** The wrapper records traces and reports the active Spec, but never rewrites prompts mid-run. All mutation happens between runs, on the Spec.
46
64
 
47
- No ground truth is required — an LLM-as-judge scores each run against a rubric.
65
+ No ground truth is required. An LLM-as-judge scores each run against a rubric.
48
66
 
49
67
  ### At a glance
50
68
 
@@ -56,16 +74,16 @@ No ground truth is required — an LLM-as-judge scores each run against a rubric
56
74
  | [**Judge**](archforge/judge/) | LLM-as-judge: scores each run per a versioned rubric, with a per-step breakdown for credit assignment |
57
75
  | [**Gatekeeper**](archforge/gatekeeper.py) | Decides promote / queue-for-human / discard / rollback by margin `τ` + scope |
58
76
  | [**Stores**](archforge/stores/) | `SpecStore` (versioned, content-addressed) + `TraceStore` + `AttemptStore` (all append-only) |
59
- | [**TracingMiddleware**](archforge/middleware.py) | The host seam — wraps every agent, records each `Step`, reports the active Spec |
77
+ | [**TracingMiddleware**](archforge/middleware.py) | The host seam: wraps every agent, records each `Step`, reports the active Spec |
60
78
 
61
79
  ---
62
80
 
63
81
  ## A Simple Example
64
82
 
65
- One Propose-Evaluate-Commit cycle, **zero cost** — no LLM, no network, no API keys. It wires the scripted organs (a fake host, a scripted Architect that proposes one prompt edit, a scripted Judge that scores it a win) into the real `Engine`, and you watch an auto-promotion end-to-end. To run it for real on your own MAS, replace the scripted organs: pass `--adapter your_pkg.your_host:YourAdapter` and `--provider <llm>` to `archforge-optimizer evolve` (see [Quickstart](#quickstart)).
83
+ One Propose-Evaluate-Commit cycle, **zero cost**: no LLM, no network, no API keys. It wires the scripted organs (a fake host, a scripted Architect that proposes one prompt edit, a scripted Judge that scores it a win) into the real `Engine`, and you watch an auto-promotion end-to-end. To run it for real on your own MAS, replace the scripted organs with `--adapter your_pkg.your_host:YourAdapter` and `--provider <llm>` on `archforge-optimizer evolve` (see [Quickstart](#quickstart)).
66
84
 
67
85
  ```python
68
- # save this as evolve_demo.py — run from an `init`-ed project dir
86
+ # save this as evolve_demo.py, then run it from an init-ed project dir
69
87
  import tempfile
70
88
  from pathlib import Path
71
89
 
@@ -99,12 +117,12 @@ with tempfile.TemporaryDirectory() as d:
99
117
  specs, atts, ts = SpecStore(root_dir), AttemptStore(root_dir), TraceStore(root_dir)
100
118
  rid = specs.commit(seed, parent_spec_id=None, status=m.SpecStatus.INCUMBENT)
101
119
  specs.set_active(rid)
102
- # the candidate's spec_id is content-hashed OVER its parent — mirror the
120
+ # the candidate's spec_id is content-hashed OVER its parent, so mirror the
103
121
  # engine's commit (parent = rid) when scripting the judge's score for it
104
122
  cand.parent_spec_id = rid
105
123
  judge = (ScriptedJudge(rubric=rubric)
106
124
  .set_aggregate(rid, "t1", 0.55) # incumbent baseline
107
- .set_aggregate(cand.compute_spec_id(), "t1", 0.70)) # +0.15 >= τ
125
+ .set_aggregate(cand.compute_spec_id(), "t1", 0.70)) # +0.15 >= tau
108
126
 
109
127
  engine = Engine(
110
128
  host=FakeHostMAS(), judge=judge, architect=ScriptedArchitect().propose(change, {"prompt": "p1"}),
@@ -119,14 +137,14 @@ with tempfile.TemporaryDirectory() as d:
119
137
  ```
120
138
 
121
139
  ```
122
- $ archforge-optimizer init # once — scaffolds the project config the Engine reads
123
- $ python evolve_demo.py
140
+ archforge-optimizer init # once: scaffolds project config + the archforge_optimizer/ adapter package
141
+ python evolve_demo.py
124
142
  action=AUTO_PROMOTE margin=+0.15
125
143
  incumbent_mean=0.55 candidate_mean=0.70
126
144
  active_spec_id=57396a49 promoted=True
127
145
  ```
128
146
 
129
- The candidate's tighter prompt beat the incumbent by `+0.15 ≥ τ`, so the Gatekeeper **auto-promoted** it to the new active Spec — all within immutable, versioned storage. Nothing was patched in place; the host will now instantiate the new Spec on its next run.
147
+ The candidate's tighter prompt beat the incumbent by `+0.15 ≥ τ`, so the Gatekeeper **auto-promoted** it to the new active Spec, all within immutable, versioned storage. Nothing was patched in place; the host will now instantiate the new Spec on its next run.
130
148
 
131
149
  ---
132
150
 
@@ -167,7 +185,7 @@ The candidate's tighter prompt beat the incumbent by `+0.15 ≥ τ`, so the Gate
167
185
  │ small + win → auto-promote │ structural + win → human gate │
168
186
  │ lose → discard │ regress → rollback (lineage) │
169
187
  │ ▼ │
170
- │ SpecStore (versioned, immutable) — "active incumbent" pointer │
188
+ │ SpecStore (versioned, immutable) - "active incumbent" pointer │
171
189
  └─────────────────────────────┬──────────────────────────────────┘
172
190
  └── next host runs use the new incumbent Spec
173
191
  ```
@@ -176,7 +194,7 @@ Four organs, one loop:
176
194
 
177
195
  | Organ | Role |
178
196
  |---|---|
179
- | **Architect** | Reads the last trace + judge scores + history, credit-assigns the rubric loss to a node/route, proposes **one** change. Forgets nothing — skips mutations already tried-and-rejected. |
197
+ | **Architect** | Reads the last trace + judge scores + history, credit-assigns the rubric loss to a node/route, proposes **one** change. Forgets nothing: it skips mutations already tried and rejected. |
180
198
  | **SuiteRunner** | Runs each candidate against the held-out suite `R` times (repeats absorb judge noise). The only component that invokes the host MAS. |
181
199
  | **Judge** | LLM-as-judge: scores each run per a versioned rubric (`grounding`, `correctness`, `completeness`, …), with a per-step breakdown for credit assignment. |
182
200
  | **Gatekeeper** | Decides by margin `τ` + scope: auto-promote small wins, queue structural wins for a human, discard regressions, rollback if a later measurement regresses. |
@@ -185,34 +203,40 @@ Four organs, one loop:
185
203
 
186
204
  ## Quickstart
187
205
 
188
- ArchForge imports with **zero LLM** installed — adapters are import-lazy and self-skip when a provider SDK is absent. A real run needs one provider SDK.
206
+ ArchForge imports with **zero LLM** installed. The one provider client (LiteLLM) is import-lazy, and a real run needs LiteLLM (a core dep) plus the provider SDK(s) you actually run.
189
207
 
190
208
  ```bash
191
- # 1. Install from PyPI
209
+ # 1. Install from PyPI (LiteLLM ships as a core dependency; the provider SDKs it
210
+ # shells out to are optional extras)
192
211
  pip install archforge-optimizer
193
212
 
194
213
  # Optional: install the provider SDK(s) you actually run (none required to import)
195
214
  pip install "archforge-optimizer[providers-groq,providers-gemini]"
196
215
 
197
- # 2. Scaffold per-project config — writes .archforge/archforge.py (tunables),
198
- # .archforge/suite.json (the eval tasks), .env.example (key template)
216
+ # 2. Scaffold per-project config + the adapter package. This writes:
217
+ # .archforge/archforge.py (tunables, with sane defaults active)
218
+ # .archforge/suite.json (the eval tasks you optimize against)
219
+ # archforge_optimizer/ (a generic LangGraph adapter skeleton, 5 files)
220
+ # __init__.py host.py app.py sidecar.py test_smoke_offline.py
199
221
  archforge-optimizer init
200
222
 
201
- # 3. Put your API keys in .env (gitignored) ── e.g. GROQ_API_KEY=..., GEMINI_API_KEY=...
223
+ # 3. Put your provider API key in a root `.env` (gitignored), e.g. GEMINI_API_KEY=...
224
+ # (init never writes or touches .env; it just tells you to put the key there.)
202
225
 
203
- # 4. Lint a Spec before running it — checks DAG validity, node refs, type rules
204
- archforge-optimizer lint path/to/spec.json
226
+ # 4. Edit your MAS details into archforge_optimizer/app.py. Fill every `# EDIT:`
227
+ # marker (the node roster, edges, knobs, summarize/apply_llm_config hooks). Then
228
+ # build the bootstrap Spec from your EDITED adapter: it lints the roster first and
229
+ # writes archforge_optimizer/spec.json only if valid (rc=1 + the faults if not,
230
+ # so fix and rerun).
231
+ archforge-optimizer make-spec # -> archforge_optimizer/spec.json (lint OK)
205
232
 
206
- # 5. Run one Propose-Evaluate-Commit cycle against your MAS, wired by an adapter
207
- archforge-optimizer evolve \
208
- --adapter your_pkg.your_host:YourAdapter \
209
- --seed your_spec.json
233
+ # 5. Run one Propose-Evaluate-Commit cycle against your MAS. evolve auto-defaults
234
+ # --adapter archforge_optimizer.host:AppAdapter and --seed archforge_optimizer/spec.json
235
+ archforge-optimizer evolve
210
236
 
211
- # 6. Run the full loop: repeat evolve until K consecutive non-promotions (plateau)
212
- # or a cycle/budget cap is hit
213
- archforge-optimizer evolve-loop \
214
- --adapter your_pkg.your_host:YourAdapter \
215
- --seed your_spec.json --max-cycles 50
237
+ # 6. Run the full loop: repeat evolve until K consecutive non-promotions (plateau),
238
+ # or set flags in archforge.py
239
+ archforge-optimizer evolve-loop --max-cycles 50
216
240
 
217
241
  # 7. Inspect
218
242
  archforge-optimizer status # print the active incumbent Spec id, lineage, counts
@@ -220,11 +244,7 @@ archforge-optimizer report # print per-attempt score deltas (incumbent vs ca
220
244
  archforge-optimizer approve --all # move PENDING_HUMAN structural wins into active
221
245
  ```
222
246
 
223
- > You can also invoke as `python -m archforge ...` — identical surface.
224
- >
225
- > **From source (development).** Clone the repo and `pip install -e .` for an editable install.
226
-
227
- The `--provider` flag selects the LLM backing the Architect + Judge (`scripted` by default for zero-cost runs; `anthropic` / `openai` / `groq` / `gemini` for real runs). The host MAS is wired via `--adapter my_pkg.my_host:MyAdapter`.
247
+ The `--provider` flag selects the LLM backing the Architect + Judge (`anthropic` / `openai` / `groq` / `gemini` for real runs). Every real provider goes through **one LiteLLM client**: the provider just prefixes the model id (`openai/gpt-4o`, `gemini/gemini-3.6-flash`, …). The host MAS is wired via `--adapter my_pkg.my_host:MyAdapter`. After `init`, `evolve` already defaults it to `archforge_optimizer.host:AppAdapter`, so you only pass the flag for a custom adapter.
228
248
 
229
249
  ---
230
250
 
@@ -244,25 +264,25 @@ One cycle, end-to-end:
244
264
 
245
265
  ### The action space
246
266
 
247
- `ChangeKind` ∈ `prompt_edit | knob | add_node | remove_node | rewire | model_swap`. Scope is mechanical: `small` (prompt/knob) auto-promotes; `structural` (roster/graph/model) requires a human. A Spec Linter validates every candidate *before* it reaches the SuiteRunner — orphans, dangling refs, self-loops, type rules.
267
+ `ChangeKind` ∈ `prompt_edit | knob | add_node | remove_node | rewire | model_swap`. Scope is mechanical: `small` (prompt/knob) auto-promotes; `structural` (roster/graph/model) requires a human. A Spec Linter validates every candidate *before* it reaches the SuiteRunner (orphans, dangling refs, self-loops, type rules).
248
268
 
249
269
  ### Governing invariants
250
270
 
251
- - **I1** — `SpecStore.active()` is the only Spec any host run can instantiate.
252
- - **I2** — no committed Spec ever changes after `commit`.
253
- - **I3** — every non-root Spec has a reachable `parent_spec_id` chain; rollback preserves it.
254
- - **I4** — every structural win goes to `queue_for_human`; auto-promote never bypasses.
255
- - **I5** — no `Attempt.suite_result` ever compares scores across a different `rubric_id` or task set.
271
+ - **I1:** `SpecStore.active()` is the only Spec any host run can instantiate.
272
+ - **I2:** no committed Spec ever changes after `commit`.
273
+ - **I3:** every non-root Spec has a reachable `parent_spec_id` chain; rollback preserves it.
274
+ - **I4:** every structural win goes to `queue_for_human`; auto-promote never bypasses.
275
+ - **I5:** no `Attempt.suite_result` ever compares scores across a different `rubric_id` or task set.
256
276
 
257
277
  ### Error handling, by design
258
278
 
259
- Every failure that touches the lineage **fails closed** — the incumbent is untouched, the candidate discarded or held, traces retained. Noise is absorbed by `R` repeats + margin `τ` + regression floor `δ ≥ τ` (so a noisy measurement never yo-yos the pointer). Host/agent errors mid-run are caught per-task (`Trace.ok=false`, partial trace retained); a candidate that fails > ε of tasks is auto-rejected *before* margin math.
279
+ Every failure that touches the lineage **fails closed**: the incumbent is untouched, the candidate discarded or held, traces retained. Noise is absorbed by `R` repeats + margin `τ` + regression floor `δ ≥ τ` (so a noisy measurement never yo-yos the pointer). Host/agent errors mid-run are caught per-task (`Trace.ok=false`, partial trace retained); a candidate that fails > ε of tasks is auto-rejected *before* margin math.
260
280
 
261
281
  ---
262
282
 
263
283
  ## Adapters: connecting your MAS
264
284
 
265
- ArchForge couples to a host through one protocol — `HostMAS`:
285
+ ArchForge couples to a host through one protocol, `HostMAS`:
266
286
 
267
287
  ```python
268
288
  class HostMAS(Protocol):
@@ -271,12 +291,14 @@ class HostMAS(Protocol):
271
291
 
272
292
  Your adapter builds a runnable pipeline from `spec` (the active incumbent's nodes/edges/prompts/knobs) and threads `TracingMiddleware` through it so every step is recorded. Everything below the seam is your pipeline; everything above it is the Forge.
273
293
 
274
- A **generic LangGraph adapter** ships in `archforge/host/adapters/langgraph.py` and drives a real `graph.stream(...)` — "describe, don't introspect" (it reads node *names*, the stable surface; it never climbs your graph's internals). It is the easiest path for any LangGraph-based MAS. For other frameworks (CrewAI, AutoGen, raw call loops), subclass `BaseHostAdapter` (`archforge/host/adapters/base.py`) — the kit is factored so adapting *any* MAS is cheap, not bespoke-per-framework.
294
+ A **generic LangGraph adapter** ships in `archforge/host/adapters/langgraph.py` and drives a real `graph.stream(...)`: it "describes, doesn't introspect" (it reads node *names*, the stable surface; it never climbs your graph's internals). It is the easiest path for any LangGraph-based MAS. For other frameworks (CrewAI, AutoGen, raw call loops), subclass `BaseHostAdapter` (`archforge/host/adapters/base.py`). The kit is factored so adapting *any* MAS is cheap, not bespoke-per-framework.
275
295
 
276
- Run it via the dotted-path seam:
296
+ **`init` scaffolds the adapter for you.** You don't code the wiring from scratch. `archforge-optimizer init` writes a generic, name-neutral `archforge_optimizer/` package (the LangGraph adapter skeleton above) into your project root. Edit the `# EDIT:` markers in `archforge_optimizer/app.py` to describe your MAS (the node roster `_NODES`, edges `_EDGES`, knob to state map, and the `summarize` / `apply_llm_config` / `reset_llm_config` hooks). Then `archforge-optimizer make-spec` builds and lints `archforge_optimizer/spec.json` from it. Once scaffolded, `evolve` auto-defaults to the scaffold: `--adapter archforge_optimizer.host:AppAdapter` and `--seed archforge_optimizer/spec.json` (pass the flags only for a custom adapter/seed). Per-file clobber guards mean re-running `init` never overwrites your edits unless `--force`, and a missing/half-edited adapter is repaired even when `archforge.py` already exists.
297
+
298
+ Run it via the dotted-path seam (the scaffolded package uses the same `module:Class` form):
277
299
 
278
300
  ```bash
279
- archforge-optimizer evolve-loop --adapter your_pkg.your_host:YourAdapter --seed your_spec.json
301
+ archforge-optimizer evolve-loop # defaults: --adapter archforge_optimizer.host:AppAdapter --seed archforge_optimizer/spec.json
280
302
  ```
281
303
 
282
304
  ---
@@ -286,7 +308,8 @@ archforge-optimizer evolve-loop --adapter your_pkg.your_host:YourAdapter --seed
286
308
  ```
287
309
  archforge-optimizer <command> [flags]
288
310
 
289
- init scaffold .archforge/archforge.py + .env.example + suite.json for this project
311
+ init scaffold .archforge/archforge.py + suite.json + the archforge_optimizer/ adapter package
312
+ make-spec build + lint archforge_optimizer/spec.json from the EDITED adapter (writes only if it passes)
290
313
  lint <path> run the Spec Linter on a JSON Spec file
291
314
  evolve run one Propose-Evaluate-Commit cycle from the active incumbent
292
315
  evolve-loop repeat evolve until the budget cap or a plateau
@@ -301,8 +324,8 @@ archforge-optimizer <command> [flags]
301
324
  | Flag | Purpose |
302
325
  |---|---|
303
326
  | `--root <dir>` | project root holding `.archforge/` (default `.`) |
304
- | `--seed <path>` | bootstrap the root incumbent from a Spec JSON (first run) |
305
- | `--adapter <dotted.path[:Class]>` | your `HostMAS` adapter |
327
+ | `--seed <path>` | bootstrap the root incumbent from a Spec JSON (first run); defaults to `archforge_optimizer/spec.json` when present |
328
+ | `--adapter <dotted.path[:Class]>` | your `HostMAS` adapter; defaults to `archforge_optimizer.host:AppAdapter` when the scaffold is present (not on the `--provider scripted` fake path) |
306
329
  | `--provider {scripted\|anthropic\|openai\|groq\|gemini}` | LLM backing the Architect + Judge |
307
330
  | `--suite <path>` | evaluation suite JSON (default: `.archforge/suite.json`) |
308
331
  | `--tau <float>` | promotion margin τ |
@@ -322,29 +345,29 @@ archforge-optimizer <command> [flags]
322
345
 
323
346
  ## Configuration
324
347
 
325
- Per-project config lives in **`.archforge/archforge.py`** — a plain Python file, **active as-is** (no registration step), so `archforge-optimizer init` produces a working project directory immediately. Edit a value to change a default. `init` scaffolds it with sane defaults: `PROVIDER="gemini"`, `DEFAULT_TAU=0.05`, `DEFAULT_DELTA=0.07`, `DEFAULT_REPEATS=1`, `DEFAULT_MAX_CYCLES=20`, `DEFAULT_PLATEAU_CYCLES=5`, plus the budget caps, the architect model roster, and `DEFAULT_TRACE_TOTAL_BUDGET_TOK=None` (the tracing toggle — see below).
348
+ Per-project config lives in **`.archforge/archforge.py`**, a plain Python file that is **active as-is** (no registration step), so `archforge-optimizer init` produces a working project directory immediately. Edit a value to change a default. `init` scaffolds it with sane defaults: `PROVIDER="gemini"`, `DEFAULT_TAU=0.05`, `DEFAULT_DELTA=0.07`, `DEFAULT_REPEATS=1`, `DEFAULT_MAX_CYCLES=20`, `DEFAULT_PLATEAU_CYCLES=5`, plus the budget caps, the architect model roster, and `DEFAULT_TRACE_TOTAL_BUDGET_TOK=None` (the tracing toggle, see below).
326
349
 
327
- API keys live in **`.env`** (gitignored — your own keys, never logged or committed). `evolve` loads them from `.env` for `--provider != scripted`; the environment always wins, and `--api-key` wins above both.
350
+ API keys live in **`.env`**. `evolve` loads them for `--provider != scripted`; the environment always preferred
328
351
 
329
- The evaluation suite is **`.archforge/suite.json`** — the representative tasks the Judge scores. Optimization targets the *suite*, never a single repeated task (the primary defense against overfitting structural mutations).
352
+ The evaluation suite is **`.archforge/suite.json`**, the representative tasks the Judge scores. Optimization targets the *suite*, never a single repeated task (the primary defense against overfitting structural mutations).
330
353
 
331
354
  ---
332
355
 
333
356
  ## Observability (OpenTelemetry GenAI tracing)
334
357
 
335
- By default, each `Step` the Judge reads carries a host-authored one-liner summary (e.g. `answer_len=1189`) — lossy on the **host-streaming path**. ArchForge can instead auto-instrument your SDK calls as **OpenTelemetry GenAI spans** and project a **bounded slice** of the real prompt/completion into each `Step` — so the Judge compares real content against the task, not length stubs.
358
+ By default, each `Step` the Judge reads carries a host-authored one-liner summary (e.g. `answer_len=1189`), lossy on the **host-streaming path**. ArchForge can instead auto-instrument your SDK calls as **OpenTelemetry GenAI spans** and project a **bounded slice** of the real prompt/completion into each `Step`, so the Judge compares real content against the task, not length stubs.
336
359
 
337
- - **Cooperative attribution.** A forge-owned `wrapped(name, fn)` opens an `archforge.node` parent span; auto-instrumented LLM/retriever spans nest as children by parent-link (not temporal order) — robust to retries, multi-call, and fan-out.
360
+ - **Cooperative attribution.** A forge-owned `wrapped(name, fn)` opens an `archforge.node` parent span; auto-instrumented LLM/retriever spans nest as children by parent-link (not temporal order), robust to retries, multi-call, and fan-out.
338
361
  - **Bounded.** Per-kind caps keep the total judge-prompt token budget bounded; a post-loop shed trims the largest remaining steps while **protecting the final-answer step**.
339
362
  - **Gated, not forked.** `DEFAULT_TRACE_TOTAL_BUDGET_TOK = None` reproduces the lossy `summarize()` path **byte-identically**, so turning rich tracing off yields exactly the same `Step` records the Judge would read without OTel installed. Set a number to turn on rich steps. Toggle, not fork.
340
- - **Zero-dep by default.** `archforge.otel` is import-lazy — `import archforge` and `import archforge.otel` pull **zero** OpenTelemetry. Per-SDK instrumentors (`opentelemetry-instrumentation-<sdk>`) are the MAS owner's install.
341
- - **Secrets stay in-process.** The in-memory span buffer has no exporter — nothing leaves the process. Never wire an OTLP exporter without a redaction processor.
363
+ - **Zero-dep by default.** `archforge.otel` is import-lazy: `import archforge` and `import archforge.otel` pull **zero** OpenTelemetry. Per-SDK instrumentors (`opentelemetry-instrumentation-<sdk>`) are the MAS owner's install.
364
+ - **Secrets stay in-process.** The in-memory span buffer has no exporter, so nothing leaves the process. Never wire an OTLP exporter without a redaction processor.
342
365
 
343
366
  ---
344
367
 
345
368
  ## Deployment: shipping optimizations to production
346
369
 
347
- When a candidate auto-promotes, ArchForge can emit a **deploy envelope** — a self-contained JSON with the promoted Spec, the knobs to overlay, the scores, and the decision (margin + rule). Your MAS reads it at startup and applies the knobs without the Forge on the hot path. Opt-in via the engine's `on_deploy` hook (the CLI wires it to write `.archforge/optimized.json`); `on_cycle` is the richer per-cycle surface (specs, runs, change) for custom rendering/telemetry.
370
+ When a candidate auto-promotes, ArchForge can emit a **deploy envelope**: a self-contained JSON with the promoted Spec, the knobs to overlay, the scores, and the decision (margin + rule). Your MAS reads it at startup and applies the knobs without the Forge on the hot path. Opt in via the engine's `on_deploy` hook (the CLI wires it to write `.archforge/optimized.json`); `on_cycle` is the richer per-cycle surface (specs, runs, change) for custom rendering/telemetry.
348
371
 
349
372
  ---
350
373
 
@@ -352,23 +375,23 @@ When a candidate auto-promotes, ArchForge can emit a **deploy envelope** — a s
352
375
 
353
376
  ```
354
377
  archforge/
355
- cli.py the Forge — argparse entrypoint + per-command wiring
378
+ cli.py the Forge: argparse entrypoint + per-command wiring
356
379
  engine.py the P-E-C orchestrator + loop (E3/E8 budget/plateau)
357
380
  architect.py proposes one change per cycle (credit assignment, dedup)
358
- suite.py SuiteRunner — runs the eval suite R repeats
381
+ suite.py SuiteRunner: runs the eval suite R repeats
359
382
  judge/ LLM-as-judge (base.py + scripted.py)
360
383
  gatekeeper.py decides promote / queue / discard / rollback
361
384
  stores/ SpecStore (versioned) + TraceStore + AttemptStore (append-only)
362
- middleware.py TracingMiddleware — the host seam
385
+ middleware.py TracingMiddleware: the host seam
363
386
  host/ HostMAS protocol + adapter kit (base.py, langgraph.py, ...)
364
- llm/ provider clients (anthropic/openai/groq/gemini, lazy + self-skip)
387
+ llm/ one LiteLLM client for every real provider (import-lazy; provider = model prefix)
365
388
  otel.py OpenTelemetry GenAI tracing (import-lazy, bounded projection)
366
389
  lint.py Spec Linter (validate-DAG, refs, type rules)
367
390
  mutate.py apply a Change to a Spec
368
391
  diff.py spec_diff / format_diff (human-readable mutation deltas)
369
392
  runlog.py per-cycle run log (cards)
370
393
  models.py Spec/Node/Edge/Step/Trace/RunScore/Attempt/Change/...
371
- config.py .py / userconfig.py / config_init.py versioning + tunable resolver + `init`
394
+ config.py / userconfig.py / config_init.py versioning + tunable resolver + `init`
372
395
  ```
373
396
 
374
397
  ---
@@ -376,10 +399,10 @@ archforge/
376
399
  ## Extending ArchForge
377
400
 
378
401
  - **A new MAS.** Subclass `BaseHostAdapter` (or use the LangGraph adapter if you're on LangGraph), implement `instantiate(spec, middleware) -> Runnable`, and pass it via `--adapter`.
379
- - **A new provider.** Add a client under `archforge/llm/` (subclass `LLMClient`); register it in the CLI's `_PROVIDERS`.
402
+ - **A new provider.** Add a provider→prefix entry to `_PROVIDER_PREFIX` in `archforge/llm/litellm.py` and a model default in `.archforge/archforge.py`'s `DEFAULT_ARCHITECT_MODELS`. LiteLLM routes the prefixed model id for you. No new adapter.
380
403
  - **A new mutation kind.** Add it to `ChangeKind` + `scope_for_kind`, implement it in `mutate.apply_change`, and teach the Architect to propose it.
381
- - **A richer rubric.** Write a `suite.json` + rubric; the Judge scores each run against it. Comparisons are only valid within `(rubric_id, suite_id)` — bumping either starts a fresh baseline (I5).
382
- - **Custom cycle/deploy surfaces.** Pass callbacks into the `Engine` constructor: `on_cycle(result, ctx)` fires every cycle (the CLI uses it to print the per-cycle card; `ctx` carries the parent + candidate Specs, both `SuiteRun`s, and the proposed `Change`), and `on_deploy(spec, dctx)` fires only on `AUTO_PROMOTE` (the CLI uses it to write the `optimized.json` deploy envelope; `dctx` carries the parent Spec, the `Decision` with margin + rule, both runs' scores, and the cycle index).
404
+ - **A richer rubric.** Write a `suite.json` + rubric; the Judge scores each run against it. Comparisons are only valid within `(rubric_id, suite_id)`: bumping either starts a fresh baseline (I5).
405
+ - **Custom cycle/deploy surfaces.** Pass callbacks into the `Engine` constructor. `on_cycle(result, ctx)` fires every cycle (the CLI uses it to print the per-cycle card; `ctx` carries the parent + candidate Specs, both `SuiteRun`s, and the proposed `Change`). `on_deploy(spec, dctx)` fires only on `AUTO_PROMOTE` (the CLI uses it to write the `optimized.json` deploy envelope; `dctx` carries the parent Spec, the `Decision` with margin + rule, both runs' scores, and the cycle index).
383
406
 
384
407
  The public model surface (`archforge.models`) is the stable contract: `Spec`, `Node`, `Edge`, `Knobs`, `Step`, `Trace`, `RunScore`, `Attempt`, `Change`, `Thresholds`, and the `ChangeKind`/`Scope`/`Verdict`/`SpecStatus` enums. `archforge.host.base` defines `Task`, `AgentResponse`, `Agent`, `Runnable`, `HostMAS`.
385
408
 
@@ -388,12 +411,10 @@ The public model surface (`archforge.models`) is the stable contract: `Spec`, `N
388
411
  ## Requirements
389
412
 
390
413
  - Python ≥ 3.11 (developed on 3.14)
391
- - `pydantic >= 2.7`, `python-dotenv >= 1.0` (only hard deps — ArchForge imports cleanly with nothing else)
392
- - Provider SDKs (optional, install only what you run): `anthropic`, `openai`, `groq`, `google-genai`
414
+ - `pydantic >= 2.7`, `python-dotenv >= 1.0`, `litellm` (hard deps: LiteLLM is import-lazy, so ArchForge still imports cleanly with nothing else; it is only needed at a real provider call)
415
+ - Provider SDKs (optional, install only what you run; LiteLLM shells out to them): `anthropic`, `openai`, `groq`, `google-genai`
393
416
  - For rich tracing (optional): `opentelemetry-sdk` + the per-SDK instrumentors you call
394
417
 
395
- > **Installing from source (development).** For an editable install, `pip install -e .` from a clone of this repository. A flat `pip install .` makes a *non-editable* copy in site-packages, so any later source edit won't take effect at the CLI — if a repo edit ever seems to no-op, check `python -c "import archforge; print(archforge.__file__)"` resolves to the repo, not site-packages.
396
-
397
418
  ---
398
419
 
399
420
  ## Roadmap
@@ -401,20 +422,22 @@ The public model surface (`archforge.models`) is the stable contract: `Spec`, `N
401
422
  ArchForge is exercised end-to-end on a real LangGraph MAS (groq + google-genai + chroma, OpenTelemetry-traced). Active directions:
402
423
 
403
424
  - **Delegation specs** (replace hand-rolled subsystems with vetted libraries):
404
- - ✅ #1 — Tracing → OpenTelemetry GenAI (lands bounded real prompt/completion slices into the Judge's per-step `Step` records)
405
- - 🚧 #2 — LLM clients → LiteLLM (unify the per-provider clients behind one library)
406
- - 🚧 #3 — Judge → DeepEval / Ragas (rubric scoring via a mature eval framework)
425
+ - ✅ #1: Tracing → OpenTelemetry GenAI (lands bounded real prompt/completion slices into the Judge's per-step `Step` records)
426
+ - ✅ #2: LLM clients → LiteLLM (one client, provider = model prefix; #1)
427
+ - 🚧 #3: Judge → DeepEval / Ragas (rubric scoring via a mature eval framework)
407
428
  - **Adapter kit.** Generalize so adapting *any* MAS is cheap (LangGraph done; CrewAI/AutoGen/raw-loops next).
408
429
  - **Hierarchical search (v2).** A Strategist layer that emits scoped optimization goals, layered over the P-E-C loop once the cheap one-change loop is reliable.
409
430
 
410
- No part of the roadmap requires breaking the model surface — additions are additive and gated behind tunables.
431
+ No part of the roadmap requires breaking the model surface: additions are additive and gated behind tunables.
411
432
 
412
433
  ---
413
434
 
414
435
  ## License
415
436
 
416
- ArchForge is released under the **MIT License** — see [`LICENSE`](LICENSE) for the full text. © 2026 Vedant Pardeshi.
437
+ ArchForge is released under the **MIT License** (see [`LICENSE`](LICENSE) for the full text). © 2026 Vedant Pardeshi.
417
438
 
418
439
  ---
419
440
 
420
- *ArchForge never patches live state — it swaps which versioned pipeline the host uses.*
441
+ <p align="center">
442
+ *ArchForge never patches live state. It swaps which versioned pipeline the host uses.*
443
+ </p>