archforge-optimizer 0.3.0__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. archforge_optimizer-0.3.0/README.md → archforge_optimizer-0.4.0/PKG-INFO +44 -0
  2. archforge_optimizer-0.3.0/PKG-INFO → archforge_optimizer-0.4.0/README.md +11 -31
  3. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/cli.py +145 -5
  4. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/config.py +17 -1
  5. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/config_init.py +7 -0
  6. archforge_optimizer-0.4.0/archforge/judge/__init__.py +56 -0
  7. archforge_optimizer-0.4.0/archforge/judge/deepeval.py +236 -0
  8. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/llm/litellm.py +19 -14
  9. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/userconfig.py +1 -2
  10. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/pyproject.toml +2 -0
  11. archforge_optimizer-0.3.0/archforge/judge/__init__.py +0 -20
  12. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/.gitignore +0 -0
  13. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/LICENSE +0 -0
  14. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/__init__.py +0 -0
  15. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/__main__.py +0 -0
  16. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/architect.py +0 -0
  17. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/diff.py +0 -0
  18. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/engine.py +0 -0
  19. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/gatekeeper.py +0 -0
  20. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/host/__init__.py +0 -0
  21. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/host/adapters/__init__.py +0 -0
  22. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/host/adapters/base.py +0 -0
  23. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/host/adapters/helpers.py +0 -0
  24. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/host/adapters/langgraph.py +0 -0
  25. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/host/base.py +0 -0
  26. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/host/fake.py +0 -0
  27. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/judge/base.py +0 -0
  28. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/judge/scripted.py +0 -0
  29. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/lint.py +0 -0
  30. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/llm/__init__.py +0 -0
  31. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/llm/base.py +0 -0
  32. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/llm/scripted.py +0 -0
  33. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/middleware.py +0 -0
  34. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/models.py +0 -0
  35. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/mutate.py +0 -0
  36. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/otel.py +0 -0
  37. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/runlog.py +0 -0
  38. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/runner.py +0 -0
  39. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/spec_builder.py +0 -0
  40. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/stores/__init__.py +0 -0
  41. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/stores/_jsonl.py +0 -0
  42. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/stores/attempt_store.py +0 -0
  43. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/stores/spec_store.py +0 -0
  44. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/stores/trace_store.py +0 -0
  45. {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/suite.py +0 -0
@@ -1,3 +1,36 @@
1
+ Metadata-Version: 2.5
2
+ Name: archforge-optimizer
3
+ Version: 0.4.0
4
+ Summary: ArchForge: a self-improving meta-layer over multi-agent systems
5
+ License-Expression: MIT
6
+ License-File: LICENSE
7
+ Requires-Python: >=3.11
8
+ Requires-Dist: litellm
9
+ Requires-Dist: pydantic>=2.7
10
+ Requires-Dist: python-dotenv>=1.0
11
+ Provides-Extra: deepeval
12
+ Requires-Dist: deepeval; extra == 'deepeval'
13
+ Provides-Extra: dev
14
+ Requires-Dist: hypothesis>=6; extra == 'dev'
15
+ Requires-Dist: mypy>=1.11; extra == 'dev'
16
+ Requires-Dist: pytest-cov>=5; extra == 'dev'
17
+ Requires-Dist: pytest>=8; extra == 'dev'
18
+ Requires-Dist: ruff>=0.5; extra == 'dev'
19
+ Provides-Extra: providers
20
+ Requires-Dist: anthropic>=0.40; extra == 'providers'
21
+ Requires-Dist: google-genai>=1.0; extra == 'providers'
22
+ Requires-Dist: groq>=0.11; extra == 'providers'
23
+ Requires-Dist: openai>=1.40; extra == 'providers'
24
+ Provides-Extra: providers-anthropic
25
+ Requires-Dist: anthropic>=0.40; extra == 'providers-anthropic'
26
+ Provides-Extra: providers-gemini
27
+ Requires-Dist: google-genai>=1.0; extra == 'providers-gemini'
28
+ Provides-Extra: providers-groq
29
+ Requires-Dist: groq>=0.11; extra == 'providers-groq'
30
+ Provides-Extra: providers-openai
31
+ Requires-Dist: openai>=1.40; extra == 'providers-openai'
32
+ Description-Content-Type: text/markdown
33
+
1
34
  # ArchForge
2
35
 
3
36
  <p align="center">
@@ -215,8 +248,17 @@ archforge-optimizer approve --all # move PENDING_HUMAN structural wins into ac
215
248
 
216
249
  The `--provider` flag selects the LLM backing the Architect + Judge (`anthropic` / `openai` / `groq` / `gemini` for real runs). Every real provider goes through **one LiteLLM client**: the provider just prefixes the model id (`openai/gpt-4o`, `gemini/gemini-3.6-flash`, …). The host MAS is wired via `--adapter my_pkg.my_host:MyAdapter`. After `init`, `evolve` already defaults it to `archforge_optimizer.host:AppAdapter`, so you only pass the flag for a custom adapter.
217
250
 
251
+ ### Evaluation backends
252
+
253
+ The Judge is pluggable behind one `JudgeProtocol` seam (`score` / `score_suite`), so the optimizer never knows which evaluator produced a score. `--evaluator` picks the backend (default `DEFAULT_EVALUATOR="native"` in `.archforge/archforge.py`):
254
+
255
+ - **`native`** (default): the built-in LLM-as-judge. One structured call to your `--provider` model scores every rubric dimension plus a per-step breakdown (used for credit assignment).
256
+ - **`deepeval`**: the external [DeepEval](https://deepeval.com) backend (optional: `pip install "archforge-optimizer[deepeval]"`). Each run is projected into a DeepEval `LLMTestCase` and scored by standalone metrics (`--deepeval-metric answer_relevancy --deepeval-metric faithfulness`, or the `DEFAULT_DEEPEVAL_METRICS` tunable). Metric scores land in `RunScore.rubric_scores`; the aggregate is their mean, comparable to the native judge's [0,1] aggregate. DeepEval is run-level, not per-step, so `step_scores` is empty and credit assignment degrades gracefully (the Architect falls back to conservative, blame-free proposals). The judge model is configurable: it reuses the `--judge-model` / `DEFAULT_JUDGE_MODELS` seam, prefixed with your `--provider` (e.g. `gemini/gemini-3.6-flash`) and routed through LiteLLM, so it scores with the same vendor and env keys as the rest of the run, never DeepEval's OpenAI default.
257
+
218
258
  ---
219
259
 
260
+
261
+
220
262
  ## The optimization loop (P-E-C)
221
263
 
222
264
  One cycle, end-to-end:
@@ -296,6 +338,8 @@ archforge-optimizer <command> [flags]
296
338
  | `--seed <path>` | bootstrap the root incumbent from a Spec JSON (first run); defaults to `archforge_optimizer/spec.json` when present |
297
339
  | `--adapter <dotted.path[:Class]>` | your `HostMAS` adapter; defaults to `archforge_optimizer.host:AppAdapter` when the scaffold is present (not on the `--provider scripted` fake path) |
298
340
  | `--provider {scripted\|anthropic\|openai\|groq\|gemini}` | LLM backing the Architect + Judge |
341
+ | `--evaluator {native\|deepeval}` | evaluation backend (default: `DEFAULT_EVALUATOR`); `deepeval` needs the `[deepeval]` extra |
342
+ | `--deepeval-metric {answer_relevancy\|faithfulness}` | DeepEval metric to score with (repeatable; default: `DEFAULT_DEEPEVAL_METRICS`) |
299
343
  | `--suite <path>` | evaluation suite JSON (default: `.archforge/suite.json`) |
300
344
  | `--tau <float>` | promotion margin τ |
301
345
  | `--delta <float>` | regression floor δ (≥ τ) |
@@ -1,34 +1,3 @@
1
- Metadata-Version: 2.5
2
- Name: archforge-optimizer
3
- Version: 0.3.0
4
- Summary: ArchForge: a self-improving meta-layer over multi-agent systems
5
- License-Expression: MIT
6
- License-File: LICENSE
7
- Requires-Python: >=3.11
8
- Requires-Dist: litellm
9
- Requires-Dist: pydantic>=2.7
10
- Requires-Dist: python-dotenv>=1.0
11
- Provides-Extra: dev
12
- Requires-Dist: hypothesis>=6; extra == 'dev'
13
- Requires-Dist: mypy>=1.11; extra == 'dev'
14
- Requires-Dist: pytest-cov>=5; extra == 'dev'
15
- Requires-Dist: pytest>=8; extra == 'dev'
16
- Requires-Dist: ruff>=0.5; extra == 'dev'
17
- Provides-Extra: providers
18
- Requires-Dist: anthropic>=0.40; extra == 'providers'
19
- Requires-Dist: google-genai>=1.0; extra == 'providers'
20
- Requires-Dist: groq>=0.11; extra == 'providers'
21
- Requires-Dist: openai>=1.40; extra == 'providers'
22
- Provides-Extra: providers-anthropic
23
- Requires-Dist: anthropic>=0.40; extra == 'providers-anthropic'
24
- Provides-Extra: providers-gemini
25
- Requires-Dist: google-genai>=1.0; extra == 'providers-gemini'
26
- Provides-Extra: providers-groq
27
- Requires-Dist: groq>=0.11; extra == 'providers-groq'
28
- Provides-Extra: providers-openai
29
- Requires-Dist: openai>=1.40; extra == 'providers-openai'
30
- Description-Content-Type: text/markdown
31
-
32
1
  # ArchForge
33
2
 
34
3
  <p align="center">
@@ -246,8 +215,17 @@ archforge-optimizer approve --all # move PENDING_HUMAN structural wins into ac
246
215
 
247
216
  The `--provider` flag selects the LLM backing the Architect + Judge (`anthropic` / `openai` / `groq` / `gemini` for real runs). Every real provider goes through **one LiteLLM client**: the provider just prefixes the model id (`openai/gpt-4o`, `gemini/gemini-3.6-flash`, …). The host MAS is wired via `--adapter my_pkg.my_host:MyAdapter`. After `init`, `evolve` already defaults it to `archforge_optimizer.host:AppAdapter`, so you only pass the flag for a custom adapter.
248
217
 
218
+ ### Evaluation backends
219
+
220
+ The Judge is pluggable behind one `JudgeProtocol` seam (`score` / `score_suite`), so the optimizer never knows which evaluator produced a score. `--evaluator` picks the backend (default `DEFAULT_EVALUATOR="native"` in `.archforge/archforge.py`):
221
+
222
+ - **`native`** (default): the built-in LLM-as-judge. One structured call to your `--provider` model scores every rubric dimension plus a per-step breakdown (used for credit assignment).
223
+ - **`deepeval`**: the external [DeepEval](https://deepeval.com) backend (optional: `pip install "archforge-optimizer[deepeval]"`). Each run is projected into a DeepEval `LLMTestCase` and scored by standalone metrics (`--deepeval-metric answer_relevancy --deepeval-metric faithfulness`, or the `DEFAULT_DEEPEVAL_METRICS` tunable). Metric scores land in `RunScore.rubric_scores`; the aggregate is their mean, comparable to the native judge's [0,1] aggregate. DeepEval is run-level, not per-step, so `step_scores` is empty and credit assignment degrades gracefully (the Architect falls back to conservative, blame-free proposals). The judge model is configurable: it reuses the `--judge-model` / `DEFAULT_JUDGE_MODELS` seam, prefixed with your `--provider` (e.g. `gemini/gemini-3.6-flash`) and routed through LiteLLM, so it scores with the same vendor and env keys as the rest of the run, never DeepEval's OpenAI default.
224
+
249
225
  ---
250
226
 
227
+
228
+
251
229
  ## The optimization loop (P-E-C)
252
230
 
253
231
  One cycle, end-to-end:
@@ -327,6 +305,8 @@ archforge-optimizer <command> [flags]
327
305
  | `--seed <path>` | bootstrap the root incumbent from a Spec JSON (first run); defaults to `archforge_optimizer/spec.json` when present |
328
306
  | `--adapter <dotted.path[:Class]>` | your `HostMAS` adapter; defaults to `archforge_optimizer.host:AppAdapter` when the scaffold is present (not on the `--provider scripted` fake path) |
329
307
  | `--provider {scripted\|anthropic\|openai\|groq\|gemini}` | LLM backing the Architect + Judge |
308
+ | `--evaluator {native\|deepeval}` | evaluation backend (default: `DEFAULT_EVALUATOR`); `deepeval` needs the `[deepeval]` extra |
309
+ | `--deepeval-metric {answer_relevancy\|faithfulness}` | DeepEval metric to score with (repeatable; default: `DEFAULT_DEEPEVAL_METRICS`) |
330
310
  | `--suite <path>` | evaluation suite JSON (default: `.archforge/suite.json`) |
331
311
  | `--tau <float>` | promotion margin τ |
332
312
  | `--delta <float>` | regression floor δ (≥ τ) |
@@ -67,10 +67,15 @@ from archforge import userconfig as ucfg
67
67
  from archforge.config import (
68
68
  ALL_PROVIDERS as _PROVIDERS,
69
69
  DEFAULT_SUITE_ID, DEFAULT_TASK_ID, DEFAULT_TASK_INPUT,
70
+ EVALUATORS as _EVALUATORS,
70
71
  PROG, load_env,
71
72
  )
72
73
  from archforge.userconfig import ConfigNotInitialized
73
74
 
75
+ # The DeepEval metrics the `--deepeval-metric` flag accepts (its `choices`). Kept in
76
+ # sync with the factory registry in archforge/judge/deepeval.py.
77
+ _DEEPEVAL_METRICS = ("answer_relevancy", "faithfulness")
78
+
74
79
  # Fixed run-state / config-discovery dir (a system path, independent of the
75
80
  # tunable DEFAULT_ROOT_DIR which the embedder API reads via archforge.userconfig).
76
81
  _DEFAULT_ROOT = ".archforge"
@@ -105,12 +110,42 @@ def _import_adapter(dotted: str) -> HostMAS:
105
110
  its own ``__init__`` carries whatever its MAS needs (Lumina loads its base
106
111
  prompts; a framework adapter wraps its graph). No PR into core to adapt a
107
112
  new MAS: ``--adapter mypkg:MyAdapter`` wires it; ``--provider`` keeps the
108
- Architect/Judge organs, ``--seed`` the bootstrap Spec."""
113
+ Architect/Judge organs, ``--seed`` the bootstrap Spec.
114
+
115
+ Surfaces clear, actionable errors for the common failure modes users hit when
116
+ editing the scaffold's ``app.py``:
117
+
118
+ * ``ImportError``/``ModuleNotFoundError`` — a MAS dependency listed in
119
+ ``app.py``'s imports is not installed (e.g. ``from mymas.graph import
120
+ build_graph`` when ``mymas`` isn't on ``sys.path``).
121
+ * ``KeyError`` — ``node_ids(build_graph())`` can't find a node name that
122
+ ``_GID[...]`` references (a rename in ``build_graph`` that wasn't
123
+ reflected in ``_GID``).
124
+ """
109
125
  if ":" in dotted:
110
126
  modpath, cls = dotted.split(":", 1)
111
127
  else:
112
128
  modpath, cls = dotted, ""
113
- module = importlib.import_module(modpath)
129
+ try:
130
+ module = importlib.import_module(modpath)
131
+ except ModuleNotFoundError as exc:
132
+ # Surface the missing package name explicitly so the user knows what to
133
+ # install, and point at `app.py` as the likely source of the bad import.
134
+ missing = exc.name or str(exc)
135
+ raise SystemExit(
136
+ f"! --adapter: cannot import {modpath!r} — missing module {missing!r}.\n"
137
+ f" Check the imports at the top of archforge_optimizer/app.py: make sure\n"
138
+ f" every package your MAS needs (e.g. `langgraph`, your MAS package) is\n"
139
+ f" installed in the current Python environment.\n"
140
+ f" Original error: {exc}"
141
+ ) from exc
142
+ except ImportError as exc:
143
+ raise SystemExit(
144
+ f"! --adapter: failed to import {modpath!r}.\n"
145
+ f" Check the imports at the top of archforge_optimizer/app.py and make\n"
146
+ f" sure all your MAS dependencies are installed.\n"
147
+ f" Original error: {exc}"
148
+ ) from exc
114
149
  if not cls:
115
150
  # Bare module: expect it to expose a ``HostMAS``-protocol attr named
116
151
  # ``HostMAS`` or the last path segment; else error loudly.
@@ -123,7 +158,33 @@ def _import_adapter(dotted: str) -> HostMAS:
123
158
  f"Pass it as `module:ClassName`."
124
159
  ) from exc
125
160
  if isinstance(obj, type):
126
- return obj() # a HostMAS/BaseHostAdapter subclass → instance
161
+ try:
162
+ return obj() # a HostMAS/BaseHostAdapter subclass → instance
163
+ except KeyError as exc:
164
+ # Usually `_GID = node_ids(build_graph())` in app.py referencing a
165
+ # node name the graph doesn't define — but a KeyError can come from
166
+ # ANY dict/env lookup in the adapter's __init__, so name the likely
167
+ # cause without asserting it.
168
+ app_py = Path.cwd() / "archforge_optimizer" / "app.py"
169
+ raise SystemExit(
170
+ f"! --adapter: {dotted!r} raised KeyError {exc} at startup.\n"
171
+ f" Most common cause: the `_GID` dict in {app_py} references a\n"
172
+ f" node name that `build_graph()` never defines (check the\n"
173
+ f" `add_node(..., name=...)` calls in your graph builder). If\n"
174
+ f" `_GID` matches, look for another dict/env lookup in the\n"
175
+ f" adapter's __init__ (e.g. os.environ[...])."
176
+ ) from exc
177
+ except AssertionError:
178
+ # Re-raise so `make-spec`'s own handler (which formats build_spec()
179
+ # lint failures as clean rc=1 output) can catch it.
180
+ raise
181
+ except Exception as exc: # noqa: BLE001
182
+ app_py = Path.cwd() / "archforge_optimizer" / "app.py"
183
+ raise SystemExit(
184
+ f"! --adapter: failed to instantiate {dotted!r}.\n"
185
+ f" Check {app_py} — the adapter raised an unexpected error at\n"
186
+ f" startup: {type(exc).__name__}: {exc}"
187
+ ) from exc
127
188
  if isinstance(obj, HostMAS):
128
189
  return obj # already an instance
129
190
  raise SystemExit(f"--adapter: {dotted!r} resolved to a {type(obj).__name__}, "
@@ -163,6 +224,14 @@ def _add_evolve_args(p: argparse.ArgumentParser, *, loop: bool) -> None:
163
224
  help="model id for the Architect (else the provider default)")
164
225
  p.add_argument("--judge-model", default=None,
165
226
  help="model id for the Judge (else the provider default)")
227
+ p.add_argument("--evaluator", choices=_EVALUATORS, default=None,
228
+ help="evaluation backend (default from archforge.py): native is the "
229
+ "built-in LLM-as-judge; deepeval is the external DeepEval backend "
230
+ "(needs the [deepeval] extra)")
231
+ p.add_argument("--deepeval-metric", choices=_DEEPEVAL_METRICS, action="append",
232
+ default=None, dest="deepeval_metrics",
233
+ help="DeepEval metric to score with (repeatable; else the "
234
+ "DEFAULT_DEEPEVAL_METRICS tunable)")
166
235
  p.add_argument("--suite", metavar="PATH", default=None,
167
236
  help="path to a suite.json (overrides the DEFAULT_SUITE_FILE tunable; "
168
237
  "the file's tasks define what you optimize against)")
@@ -331,8 +400,34 @@ def _default_components(args: argparse.Namespace) -> Components:
331
400
  arch_models = ucfg.get("DEFAULT_ARCHITECT_MODELS")
332
401
  judge_models = ucfg.get("DEFAULT_JUDGE_MODELS")
333
402
  arch = Architect(llm, model=args.architect_model or arch_models[provider])
334
- judge = Judge(llm, model=args.judge_model or judge_models[provider],
335
- rubric=default_rubric())
403
+
404
+ # Evaluation backend seam (issue #2): `native` keeps the built-in Judge over
405
+ # the provider's LLM; `deepeval` swaps in the external DeepEval backend, which
406
+ # scores independently of the Architect's LLM (it carries its own model/keys).
407
+ # `default=` on both: a config written by an OLDER init (before issue #2)
408
+ # lacks these names and would KeyError otherwise — the defaults keep the
409
+ # pre-existing native behavior for those projects.
410
+ evaluator = getattr(args, "evaluator", None) or ucfg.get(
411
+ "DEFAULT_EVALUATOR", default="native")
412
+ if evaluator == "deepeval":
413
+ from archforge.judge import make_evaluator
414
+
415
+ try:
416
+ metrics = getattr(args, "deepeval_metrics", None) or ucfg.get(
417
+ "DEFAULT_DEEPEVAL_METRICS", default=["answer_relevancy"])
418
+ # DeepEval scores through LiteLLM too: reuse the Judge model seam
419
+ # (--judge-model / DEFAULT_JUDGE_MODELS) and prefix it with the
420
+ # provider so LiteLLM routes to the same vendor (never DeepEval's
421
+ # OpenAI default).
422
+ from archforge.llm.litellm import prefix_model
423
+ de_model = prefix_model(provider, args.judge_model or judge_models[provider])
424
+ judge = make_evaluator("deepeval", model=de_model, metrics=metrics)
425
+ except LLMError as exc:
426
+ print(f"[evaluator] {exc}", file=sys.stderr)
427
+ raise
428
+ else:
429
+ judge = Judge(llm, model=args.judge_model or judge_models[provider],
430
+ rubric=default_rubric())
336
431
  return Components(host=FakeHostMAS(), judge=judge, architect=arch, suite=suite)
337
432
 
338
433
 
@@ -549,6 +644,31 @@ def _scaffolded_spec_available() -> bool:
549
644
  return (Path.cwd() / "archforge_optimizer" / "spec.json").is_file()
550
645
 
551
646
 
647
+ # Unique token written by `init` into the placeholder `build_graph` — present ONLY
648
+ # when the user hasn't replaced it yet (spec E-guard: refuse to evolve template code).
649
+ _TEMPLATE_MARKER = "EDIT app.py: import your MAS"
650
+
651
+
652
+ def _scaffolded_adapter_is_template() -> bool:
653
+ """True when the scaffolded `app.py` still contains the unedited placeholder
654
+ `build_graph` (the ``NotImplementedError`` with the "EDIT app.py" marker).
655
+
656
+ This is the signal that the user ran `init` but hasn't wired their real MAS yet.
657
+ Returning True causes `evolve`/`evolve-loop` to refuse with a clear message
658
+ instead of silently running the Judge against the fake placeholder spec.
659
+ Only the scaffolded adapter path is checked — an explicit ``--adapter`` flag
660
+ skips this (the user owns that adapter).
661
+ """
662
+ app_py = Path.cwd() / "archforge_optimizer" / "app.py"
663
+ if not app_py.is_file():
664
+ return False
665
+ try:
666
+ text = app_py.read_text(encoding="utf-8")
667
+ except OSError:
668
+ return False
669
+ return _TEMPLATE_MARKER in text
670
+
671
+
552
672
  def _cmd_evolve(args: argparse.Namespace, *, components: Components | None,
553
673
  loop: bool) -> int:
554
674
  # Injected `components` (the test/embedding path) always win — they ARE the
@@ -597,6 +717,26 @@ def _cmd_evolve(args: argparse.Namespace, *, components: Components | None,
597
717
  # Skipped on the scripted fake path (FakeHostMAS is the zero-cost host) and
598
718
  # when `components` were injected (test/embedding path owns the host).
599
719
  adapter_path = getattr(args, "adapter", None)
720
+ # Guard: refuse to run against the unedited scaffold template. When the user
721
+ # ran `init` but hasn't replaced `build_graph` in `app.py` yet, the placeholder
722
+ # Spec is valid (it lints) but the MAS is a stub — running `evolve` would silently
723
+ # score fake outputs through the Judge with no real pipeline. Only checked when
724
+ # the scaffold is auto-detected (no explicit --adapter flag) so user-supplied
725
+ # adapters are never blocked. The scripted path is also exempt: it keeps
726
+ # FakeHostMAS and never touches the scaffold, so a template app.py is inert
727
+ # there (and blocking it would break the zero-cost demo in a scaffolded dir).
728
+ if (adapter_path is None and not components_injected
729
+ and (args.provider or ucfg.get("PROVIDER")) != "scripted"
730
+ and _scaffolded_adapter_available()
731
+ and _scaffolded_adapter_is_template()):
732
+ print(
733
+ f"! archforge_optimizer/app.py still contains the placeholder `build_graph` "
734
+ f"(the scaffold written by `{PROG} init` has not been edited yet).\n"
735
+ f" Edit the # EDIT: markers in archforge_optimizer/app.py to wire your real "
736
+ f"MAS, then re-run `{PROG} make-spec` to rebuild the Spec before evolving.",
737
+ file=sys.stderr,
738
+ )
739
+ return 1
600
740
  if (adapter_path is None and not components_injected
601
741
  and _scaffolded_adapter_available()
602
742
  and (args.provider or ucfg.get("PROVIDER")) != "scripted"):
@@ -29,7 +29,7 @@ from pathlib import Path
29
29
  # =========================================================================== #
30
30
  # The package version. Read by hatchling for the built distribution and
31
31
  # re-exported as `archforge.__version__` (archforge/__init__.py). One place.
32
- VERSION: str = "0.3.0"
32
+ VERSION: str = "0.4.0"
33
33
 
34
34
 
35
35
  # =========================================================================== #
@@ -46,6 +46,20 @@ REAL_PROVIDERS: tuple[str, ...] = ("anthropic", "openai", "groq", "gemini")
46
46
  ALL_PROVIDERS: tuple[str, ...] = (SCRIPTED_PROVIDER,) + REAL_PROVIDERS
47
47
 
48
48
 
49
+ # =========================================================================== #
50
+ # Evaluation backends — the ROSTER only (NOT the default choice; that's a tunable)
51
+ # =========================================================================== #
52
+ # The evaluators the CLI's `--evaluator` flag ACCEPTS (its `choices`). `native` is
53
+ # the built-in LLM-as-judge (`Judge` over an `LLMClient`); `deepeval` is the
54
+ # external DeepEval backend (`DeepEvalEvaluator`, an optional extra). WHICH is the
55
+ # default is a tunable → it lives in .archforge/archforge.py (resolved by
56
+ # archforge.userconfig), NOT here. `EVALUATORS` is re-exported from archforge.judge;
57
+ # `make_evaluator(evaluator)` dispatches on these.
58
+ NATIVE_EVALUATOR: str = "native"
59
+ DEEPEVAL_EVALUATOR: str = "deepeval"
60
+ EVALUATORS: tuple[str, ...] = (NATIVE_EVALUATOR, DEEPEVAL_EVALUATOR)
61
+
62
+
49
63
  # =========================================================================== #
50
64
  # Project environment (`.env`) — how secrets reach the provider SDKs
51
65
  # =========================================================================== #
@@ -128,6 +142,8 @@ __all__ = [
128
142
  "VERSION",
129
143
  # llm provider roster
130
144
  "SCRIPTED_PROVIDER", "REAL_PROVIDERS", "ALL_PROVIDERS",
145
+ # evaluation backend roster
146
+ "NATIVE_EVALUATOR", "DEEPEVAL_EVALUATOR", "EVALUATORS",
131
147
  # project environment (.env loader for provider API keys)
132
148
  "load_env",
133
149
  # storage layout
@@ -57,6 +57,13 @@ _FIELDS: tuple[tuple[str, str, str], ...] = (
57
57
  # --- LLM provider
58
58
  ("PROVIDER", '"gemini"',
59
59
  "which LLM to use (scripted|anthropic|openai|groq|gemini); scripted needs no API key"),
60
+ # --- evaluation backend
61
+ ("DEFAULT_EVALUATOR", '"native"',
62
+ "which evaluator scores runs (native|deepeval); native = built-in LLM-as-judge, "
63
+ "deepeval = external DeepEval backend (needs the [deepeval] extra)"),
64
+ ("DEFAULT_DEEPEVAL_METRICS", '["answer_relevancy"]',
65
+ "DeepEval metrics to run per run (answer_relevancy|faithfulness); "
66
+ "each becomes a rubric_scores dimension and the aggregate is their mean"),
60
67
  ("DEFAULT_ARCHITECT_MODELS", _DEFAULT_ARCHITECT_MODELS,
61
68
  "default Architect (proposer) model per provider; a bare LLMClient call falls "
62
69
  "back here too — edit the dict to change it"),
@@ -0,0 +1,56 @@
1
+ """The Judge — LLM-as-judge scoring of runs (spec §3, §4).
2
+
3
+ `Judge.score(trace, task, rubric_id) -> RunScore` turns a run's trace into a
4
+ scored verdict: an aggregate + named rubric dimensions + confidence, plus a
5
+ *per-step breakdown* (StepScore[]) that names which agent lost which points. The
6
+ per-step breakdown is the raw material the Architect uses to credit-assign a
7
+ fault to a node/route.
8
+
9
+ `score_suite(...)` aggregates over R repeats: mean (down-weighted by
10
+ confidence) and a stable aggregate used by the Gatekeeper. `rubric_id` is
11
+ stamped on every score so cross-rubric comparisons never masquerade as
12
+ improvement (invariant I5, spec E2).
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ from typing import Sequence
18
+
19
+ from archforge.config import EVALUATORS
20
+ from archforge.judge.base import (
21
+ Judge, JudgeProtocol, SuiteAggregate, default_rubric,
22
+ )
23
+ from archforge.judge.scripted import ScriptedJudge
24
+ from archforge.llm.base import LLMError
25
+
26
+ # `EVALUATORS` is re-exported (in __all__) straight from archforge.config — the
27
+ # single source of truth the CLI also imports for its `--evaluator` choices.
28
+
29
+
30
+ def make_evaluator(
31
+ evaluator: str, *, model: str | None = None, metrics: Sequence[str] | None = None
32
+ ) -> JudgeProtocol:
33
+ """Build an external evaluation backend by name.
34
+
35
+ Resolves `DeepEvalEvaluator` lazily (so importing DeepEval is deferred to here,
36
+ and the optional extra is only required at construction). Raises `LLMError` for
37
+ an unknown evaluator so the CLI surfaces one clear message across the backend
38
+ seam. The `native` evaluator is NOT built here: it is the built-in `Judge`,
39
+ constructed inline by the CLI from the provider's `LLMClient`.
40
+ """
41
+
42
+ if evaluator == "native":
43
+ raise LLMError(
44
+ "'native' is the built-in Judge; construct it with Judge(llm, model=...)"
45
+ )
46
+ if evaluator == "deepeval":
47
+ from archforge.judge.deepeval import DeepEvalEvaluator
48
+
49
+ return DeepEvalEvaluator(model=model, metrics=metrics) # type: ignore[return-value]
50
+ raise LLMError(f"unknown evaluator {evaluator!r}; expected one of {EVALUATORS}")
51
+
52
+
53
+ __all__ = [
54
+ "Judge", "JudgeProtocol", "ScriptedJudge", "SuiteAggregate", "default_rubric",
55
+ "make_evaluator", "EVALUATORS",
56
+ ]
@@ -0,0 +1,236 @@
1
+ """The DeepEval evaluation backend — an external `JudgeProtocol` (issue #2).
2
+
3
+ `DeepEvalEvaluator` implements the SAME protocol the built-in `Judge` does
4
+ (`JudgeProtocol` in archforge/judge/base.py), so the SuiteRunner and optimizer
5
+ treat it exactly like the native backend: `score(trace, task, rubric_id) ->
6
+ RunScore` and `score_suite(...) -> SuiteAggregate`. Nothing downstream knows or
7
+ cares which backend produced a score — that is the pluggability the issue asks
8
+ for (criterion 5: the optimizer stays independent of the evaluator).
9
+
10
+ Unlike the native `Judge` (one LLM call returning a whole JSON rubric verdict),
11
+ DeepEval scores a run with one or more standalone LLM-judge METRICS
12
+ (AnswerRelevancy, Faithfulness, ...), each yielding a score in [0,1]. We project
13
+ an ArchForge trace into a DeepEval `LLMTestCase` and convert each metric's score
14
+ into a `RunScore`:
15
+
16
+ * `rubric_scores` = {metric slug: metric.score} (the named dimensions)
17
+ * `aggregate` = mean of the metric scores (comparable to the native Judge's
18
+ [0,1] aggregate; invariant I5 still holds: rubric_id is
19
+ stamped on every score)
20
+ * `step_scores` = [] (top-level only — DeepEval is run-level, not per-step;
21
+ credit assignment degrades gracefully to `blame=None`)
22
+
23
+ DeepEval is imported LAZILY inside `__init__` (it is an optional extra and is
24
+ heavy — it pulls an LLM provider SDK + more). A missing install surfaces as a
25
+ clear `LLMError` at construction, never a bare ImportError at import time, so
26
+ `import archforge` stays deepeval-free.
27
+ """
28
+
29
+ from __future__ import annotations
30
+
31
+ from typing import Any, Sequence
32
+
33
+ from archforge.llm.base import LLMError
34
+ from archforge.judge.base import SuiteAggregate, aggregate_scores
35
+ from archforge.host.base import Task
36
+ import archforge.models as m
37
+
38
+
39
+ # A metric name -> factory. Kept as a mapping (not `if` chains) so a configured
40
+ # name resolves to a metric object AND gives us the stable rubric_scores slug.
41
+ _METRIC_FACTORIES: dict[str, Any] = {}
42
+
43
+
44
+ def _metric_slug(name: str) -> str:
45
+ """Canonical rubric_scores key for a DeepEval metric name (lower snake)."""
46
+ return name.strip().lower().replace("-", "_").replace(" ", "_")
47
+
48
+
49
+ class DeepEvalEvaluator:
50
+ """Scores runs via DeepEval metrics, satisfying `JudgeProtocol`.
51
+
52
+ `metrics` are metric NAMES (`"answer_relevancy"`, `"faithfulness"`) or
53
+ already-constructed DeepEval metric objects. Names resolve to metric objects
54
+ lazily at construction (DeepEval stays unimported until then). `model` is a
55
+ LiteLLM-style model string (e.g. `"gemini/gemini-3.6-flash"`; prefix with the
56
+ provider) or an already-built DeepEval model object; a string is wrapped in
57
+ DeepEval's `LiteLLMModel` so scoring routes through LiteLLM with the
58
+ provider's env key, never DeepEval's OpenAI default. When None, DeepEval
59
+ falls back to its own default model.
60
+ """
61
+
62
+ def __init__(
63
+ self,
64
+ *,
65
+ metrics: Sequence[str | Any] | None = None,
66
+ model: str | Any | None = None,
67
+ threshold: float = 0.5,
68
+ ) -> None:
69
+ self._deepeval = _import_deepeval() # LLMError if the extra is missing
70
+ # A string model is wrapped in LiteLLMModel (below) so the provider
71
+ # prefix routes correctly; `_judge_label` is only the provenance stamp
72
+ # on JudgeMeta.
73
+ self._model = self._wrap_model(model)
74
+ if isinstance(model, str):
75
+ self._judge_label = model
76
+ elif model is not None and hasattr(model, "get_model_name"):
77
+ self._judge_label = model.get_model_name() # DeepEvalBaseLLM API
78
+ else:
79
+ self._judge_label = "deepeval"
80
+ self._threshold = threshold
81
+ self._metrics = self._build_metrics(metrics) # list[(slug, metric)]
82
+
83
+ def _wrap_model(self, model: str | Any | None) -> Any:
84
+ """Wrap a string model id in DeepEval's `LiteLLMModel` (lazy import).
85
+
86
+ A bare string handed to a DeepEval metric resolves to DeepEval's OpenAI
87
+ default regardless of any provider prefix, so we wrap it explicitly:
88
+ `LiteLLMModel` is a `DeepEvalBaseLLM`, which `initialize_model` passes
89
+ through untouched, and LiteLLM routes by the model's provider prefix
90
+ using the provider's env key (same as ArchForge's own LiteLLMClient).
91
+ Non-string models (already-built DeepEval models) pass through.
92
+ """
93
+ if model is None or not isinstance(model, str):
94
+ return model
95
+ try:
96
+ from deepeval.models import LiteLLMModel
97
+
98
+ return LiteLLMModel(model=model)
99
+ except LLMError:
100
+ raise
101
+ except Exception as exc: # noqa: BLE001 — deepeval/litellm construction errors
102
+ raise LLMError(
103
+ f"could not build the DeepEval judge model {model!r}: {exc}"
104
+ ) from exc
105
+
106
+ # --------------------------------------------------------------- protocol
107
+ def score(self, trace: m.Trace, task: Task, rubric_id: str) -> m.RunScore:
108
+ test_case = self._llm_test_case(trace, task)
109
+ rubric_scores: dict[str, float] = {}
110
+ for slug, metric in self._metrics:
111
+ # `measure()` is SYNCHRONOUS in deepeval (the async entry point is
112
+ # `a_measure`), matching the sync SuiteRunner path directly — do NOT
113
+ # wrap it in asyncio.run (that raises TypeError on a non-coroutine).
114
+ # A metric LLM outage raises deepeval's own error; convert to
115
+ # LLMError so the SuiteRunner's bounded retry treats it exactly like
116
+ # a native Judge failure (E9), never a fabricated 0.0.
117
+ try:
118
+ metric.measure(test_case)
119
+ except Exception as exc: # noqa: BLE001 — deepeval's error type
120
+ raise LLMError(f"DeepEval metric {slug!r} failed: {exc}") from exc
121
+ rubric_scores[slug] = float(metric.score or 0.0)
122
+
123
+ aggregate = _mean(rubric_scores.values()) if rubric_scores else 0.0
124
+ return m.RunScore(
125
+ run_id=trace.run_id,
126
+ spec_id=trace.spec_id,
127
+ task_id=trace.task_id,
128
+ rubric_scores=rubric_scores,
129
+ aggregate=aggregate,
130
+ confidence=1.0,
131
+ judge_meta=m.JudgeMeta(model=self._judge_label, rubric_id=rubric_id),
132
+ step_scores=[], # top-level only (DeepEval is run-level, not per-step)
133
+ )
134
+
135
+ def score_suite(
136
+ self, scores: Sequence[m.RunScore], *, suite_id: str, rubric_id: str
137
+ ) -> SuiteAggregate:
138
+ # Reuse the pure shared aggregation (same as the native Judge) so the two
139
+ # backends produce structurally identical, comparable SuiteAggregates.
140
+ return aggregate_scores(scores, suite_id=suite_id, rubric_id=rubric_id)
141
+
142
+ # --------------------------------------------------------------- internals
143
+ def _build_metrics(
144
+ self, metrics: Sequence[str | Any] | None
145
+ ) -> list[tuple[str, Any]]:
146
+ built: list[tuple[str, Any]] = []
147
+ for item in (metrics or ["answer_relevancy"]):
148
+ if isinstance(item, str):
149
+ slug = _metric_slug(item)
150
+ factory = _METRIC_FACTORIES.get(slug)
151
+ if factory is None:
152
+ raise LLMError(
153
+ f"unknown DeepEval metric {item!r}; expected one of "
154
+ f"{sorted(_METRIC_FACTORIES)}"
155
+ )
156
+ # A metric validates its judge model at CONSTRUCTION (deepeval
157
+ # raises its own DeepEvalError if no API key is configured). Catch
158
+ # ANY construction failure and re-raise as LLMError so the CLI /
159
+ # SuiteRunner see the same error type as the native Judge (E9).
160
+ try:
161
+ metric = factory(model=self._model, threshold=self._threshold)
162
+ except Exception as exc: # noqa: BLE001 — deepeval's DeepEvalError
163
+ raise LLMError(
164
+ f"could not build DeepEval metric {item!r}: {exc}"
165
+ ) from exc
166
+ built.append((slug, metric))
167
+ else:
168
+ built.append((_metric_slug(type(item).__name__), item))
169
+ return built
170
+
171
+ def _llm_test_case(self, trace: m.Trace, task: Task):
172
+ """Project an ArchForge trace into a DeepEval `LLMTestCase`.
173
+
174
+ `actual_output` is the run's final answer (the traced final_output, else
175
+ the last step's response). Grounding comes from each step's prompt_in
176
+ and is set on BOTH `context` and `retrieval_context`: Faithfulness
177
+ REQUIRES `retrieval_context` (its `_required_params`) and errors without
178
+ it, while RAG-style metrics read `context`. `expected_output` is carried
179
+ through when the task JSON declares one (Task allows extra fields).
180
+ """
181
+ actual_output = trace.final_output or (
182
+ trace.steps[-1].response_out if trace.steps else ""
183
+ )
184
+ expected_output = getattr(task, "expected_output", None)
185
+ grounding = [s.prompt_in for s in trace.steps]
186
+ kwargs: dict[str, Any] = {
187
+ "input": task.input,
188
+ "actual_output": actual_output,
189
+ "context": grounding,
190
+ "retrieval_context": grounding, # Faithfulness requires this param
191
+ }
192
+ if expected_output:
193
+ kwargs["expected_output"] = expected_output
194
+ return self._deepeval.test_case.LLMTestCase(**kwargs)
195
+
196
+
197
+ def _import_deepeval():
198
+ """Lazily import the `deepeval` package, or raise a clear LLMError."""
199
+ try:
200
+ import deepeval
201
+ except ImportError as exc: # pragma: no cover - exercised via monkeypatch
202
+ raise LLMError(
203
+ "DeepEval is not installed; install it with "
204
+ "`pip install archforge-optimizer[deepeval]` to use --evaluator deepeval"
205
+ ) from exc
206
+ return deepeval
207
+
208
+
209
+ def _mean(values: Sequence[float]) -> float:
210
+ return sum(values) / len(values) if values else 0.0
211
+
212
+
213
+ # Register the metrics ArchForge exposes by name. Factories construct the metric
214
+ # objects lazily (deepeval classes are imported inside each factory, not here, so
215
+ # this module stays importable before the extra is installed).
216
+ def _answer_relevancy_factory(*, model, threshold):
217
+ from deepeval.metrics import AnswerRelevancyMetric
218
+
219
+ return AnswerRelevancyMetric(threshold=threshold, model=model)
220
+
221
+
222
+ def _faithfulness_factory(*, model, threshold):
223
+ from deepeval.metrics import FaithfulnessMetric
224
+
225
+ return FaithfulnessMetric(threshold=threshold, model=model)
226
+
227
+
228
+ _METRIC_FACTORIES.update(
229
+ {
230
+ "answer_relevancy": _answer_relevancy_factory,
231
+ "faithfulness": _faithfulness_factory,
232
+ }
233
+ )
234
+
235
+
236
+ __all__ = ["DeepEvalEvaluator"]
@@ -48,6 +48,22 @@ _PROVIDER_PREFIX: dict[str, str] = {
48
48
  }
49
49
 
50
50
 
51
+ def prefix_model(provider: str, model: str) -> str:
52
+ """Prefix a model id with the provider (``openai/gpt-4o``).
53
+
54
+ A model that ALREADY carries THIS provider's prefix (``openai/gpt-4o``
55
+ under provider=openai) is passed through untouched, so fully-qualified
56
+ overrides still work. But a ``/`` alone does NOT mean "already routed" —
57
+ e.g. Groq's model ``openai/gpt-oss-120b`` has a slash in its native id,
58
+ so it must still get the provider prefix (``groq/openai/gpt-oss-120b``)
59
+ or LiteLLM would route it to OpenAI. Prefix unless it already matches.
60
+ """
61
+ prefix = _PROVIDER_PREFIX.get(provider, provider)
62
+ if model.startswith(f"{prefix}/"):
63
+ return model
64
+ return f"{prefix}/{model}"
65
+
66
+
51
67
  class LiteLLMClient:
52
68
  """An `LLMClient` backed by `litellm.completion` for one provider.
53
69
 
@@ -65,19 +81,8 @@ class LiteLLMClient:
65
81
  self._base_url = base_url
66
82
 
67
83
  def _prefixed(self, model: str) -> str:
68
- """Prefix a model id with the provider (``openai/gpt-4o``).
69
-
70
- A model that ALREADY carries THIS provider's prefix (``openai/gpt-4o``
71
- under provider=openai) is passed through untouched, so fully-qualified
72
- overrides still work. But a ``/`` alone does NOT mean "already routed" —
73
- e.g. Groq's model ``openai/gpt-oss-120b`` has a slash in its native id,
74
- so it must still get the provider prefix (``groq/openai/gpt-oss-120b``)
75
- or LiteLLM would route it to OpenAI. Prefix unless it already matches.
76
- """
77
- prefix = _PROVIDER_PREFIX[self._provider]
78
- if model.startswith(f"{prefix}/"):
79
- return model
80
- return f"{prefix}/{model}"
84
+ """Prefix a model id with this client's provider (see `prefix_model`)."""
85
+ return prefix_model(self._provider, model)
81
86
 
82
87
  def _default_model(self) -> str:
83
88
  """The provider's default model id for a bare complete() call, resolved
@@ -138,4 +143,4 @@ class LiteLLMClient:
138
143
  )
139
144
 
140
145
 
141
- __all__ = ["LiteLLMClient"]
146
+ __all__ = ["LiteLLMClient", "prefix_model"]
@@ -39,8 +39,7 @@ _DISCOVERY_DIR: Path = Path(".archforge")
39
39
  _DISCOVERY_FILE: Path = _DISCOVERY_DIR / "archforge.py"
40
40
 
41
41
  _MAIN_MSG = (
42
- "ArchForge is not initialized. Run `archforge-optimizer init` to create "
43
- ".archforge/archforge.py (the project's tunable config), then the CLI will work."
42
+ "ArchForge is not initialized. Run `archforge-optimizer init`."
44
43
  )
45
44
 
46
45
 
@@ -39,6 +39,8 @@ providers = [
39
39
  "google-genai>=1.0",
40
40
  ]
41
41
 
42
+ deepeval = ["deepeval"]
43
+
42
44
  [build-system]
43
45
  requires = ["hatchling"]
44
46
  build-backend = "hatchling.build"
@@ -1,20 +0,0 @@
1
- """The Judge — LLM-as-judge scoring of runs (spec §3, §4).
2
-
3
- `Judge.score(trace, task, rubric_id) -> RunScore` turns a run's trace into a
4
- scored verdict: an aggregate + named rubric dimensions + confidence, plus a
5
- *per-step breakdown* (StepScore[]) that names which agent lost which points. The
6
- per-step breakdown is the raw material the Architect uses to credit-assign a
7
- fault to a node/route.
8
-
9
- `score_suite(...)` aggregates over R repeats: mean (down-weighted by
10
- confidence) and a stable aggregate used by the Gatekeeper. `rubric_id` is
11
- stamped on every score so cross-rubric comparisons never masquerade as
12
- improvement (invariant I5, spec E2).
13
- """
14
-
15
- from __future__ import annotations
16
-
17
- from archforge.judge.base import Judge, SuiteAggregate, default_rubric
18
- from archforge.judge.scripted import ScriptedJudge
19
-
20
- __all__ = ["Judge", "ScriptedJudge", "SuiteAggregate", "default_rubric"]