archforge-optimizer 0.3.0__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- archforge_optimizer-0.3.0/README.md → archforge_optimizer-0.4.0/PKG-INFO +44 -0
- archforge_optimizer-0.3.0/PKG-INFO → archforge_optimizer-0.4.0/README.md +11 -31
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/cli.py +145 -5
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/config.py +17 -1
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/config_init.py +7 -0
- archforge_optimizer-0.4.0/archforge/judge/__init__.py +56 -0
- archforge_optimizer-0.4.0/archforge/judge/deepeval.py +236 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/llm/litellm.py +19 -14
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/userconfig.py +1 -2
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/pyproject.toml +2 -0
- archforge_optimizer-0.3.0/archforge/judge/__init__.py +0 -20
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/.gitignore +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/LICENSE +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/__init__.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/__main__.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/architect.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/diff.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/engine.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/gatekeeper.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/host/__init__.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/host/adapters/__init__.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/host/adapters/base.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/host/adapters/helpers.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/host/adapters/langgraph.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/host/base.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/host/fake.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/judge/base.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/judge/scripted.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/lint.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/llm/__init__.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/llm/base.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/llm/scripted.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/middleware.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/models.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/mutate.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/otel.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/runlog.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/runner.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/spec_builder.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/stores/__init__.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/stores/_jsonl.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/stores/attempt_store.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/stores/spec_store.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/stores/trace_store.py +0 -0
- {archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/suite.py +0 -0
|
@@ -1,3 +1,36 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: archforge-optimizer
|
|
3
|
+
Version: 0.4.0
|
|
4
|
+
Summary: ArchForge: a self-improving meta-layer over multi-agent systems
|
|
5
|
+
License-Expression: MIT
|
|
6
|
+
License-File: LICENSE
|
|
7
|
+
Requires-Python: >=3.11
|
|
8
|
+
Requires-Dist: litellm
|
|
9
|
+
Requires-Dist: pydantic>=2.7
|
|
10
|
+
Requires-Dist: python-dotenv>=1.0
|
|
11
|
+
Provides-Extra: deepeval
|
|
12
|
+
Requires-Dist: deepeval; extra == 'deepeval'
|
|
13
|
+
Provides-Extra: dev
|
|
14
|
+
Requires-Dist: hypothesis>=6; extra == 'dev'
|
|
15
|
+
Requires-Dist: mypy>=1.11; extra == 'dev'
|
|
16
|
+
Requires-Dist: pytest-cov>=5; extra == 'dev'
|
|
17
|
+
Requires-Dist: pytest>=8; extra == 'dev'
|
|
18
|
+
Requires-Dist: ruff>=0.5; extra == 'dev'
|
|
19
|
+
Provides-Extra: providers
|
|
20
|
+
Requires-Dist: anthropic>=0.40; extra == 'providers'
|
|
21
|
+
Requires-Dist: google-genai>=1.0; extra == 'providers'
|
|
22
|
+
Requires-Dist: groq>=0.11; extra == 'providers'
|
|
23
|
+
Requires-Dist: openai>=1.40; extra == 'providers'
|
|
24
|
+
Provides-Extra: providers-anthropic
|
|
25
|
+
Requires-Dist: anthropic>=0.40; extra == 'providers-anthropic'
|
|
26
|
+
Provides-Extra: providers-gemini
|
|
27
|
+
Requires-Dist: google-genai>=1.0; extra == 'providers-gemini'
|
|
28
|
+
Provides-Extra: providers-groq
|
|
29
|
+
Requires-Dist: groq>=0.11; extra == 'providers-groq'
|
|
30
|
+
Provides-Extra: providers-openai
|
|
31
|
+
Requires-Dist: openai>=1.40; extra == 'providers-openai'
|
|
32
|
+
Description-Content-Type: text/markdown
|
|
33
|
+
|
|
1
34
|
# ArchForge
|
|
2
35
|
|
|
3
36
|
<p align="center">
|
|
@@ -215,8 +248,17 @@ archforge-optimizer approve --all # move PENDING_HUMAN structural wins into ac
|
|
|
215
248
|
|
|
216
249
|
The `--provider` flag selects the LLM backing the Architect + Judge (`anthropic` / `openai` / `groq` / `gemini` for real runs). Every real provider goes through **one LiteLLM client**: the provider just prefixes the model id (`openai/gpt-4o`, `gemini/gemini-3.6-flash`, …). The host MAS is wired via `--adapter my_pkg.my_host:MyAdapter`. After `init`, `evolve` already defaults it to `archforge_optimizer.host:AppAdapter`, so you only pass the flag for a custom adapter.
|
|
217
250
|
|
|
251
|
+
### Evaluation backends
|
|
252
|
+
|
|
253
|
+
The Judge is pluggable behind one `JudgeProtocol` seam (`score` / `score_suite`), so the optimizer never knows which evaluator produced a score. `--evaluator` picks the backend (default `DEFAULT_EVALUATOR="native"` in `.archforge/archforge.py`):
|
|
254
|
+
|
|
255
|
+
- **`native`** (default): the built-in LLM-as-judge. One structured call to your `--provider` model scores every rubric dimension plus a per-step breakdown (used for credit assignment).
|
|
256
|
+
- **`deepeval`**: the external [DeepEval](https://deepeval.com) backend (optional: `pip install "archforge-optimizer[deepeval]"`). Each run is projected into a DeepEval `LLMTestCase` and scored by standalone metrics (`--deepeval-metric answer_relevancy --deepeval-metric faithfulness`, or the `DEFAULT_DEEPEVAL_METRICS` tunable). Metric scores land in `RunScore.rubric_scores`; the aggregate is their mean, comparable to the native judge's [0,1] aggregate. DeepEval is run-level, not per-step, so `step_scores` is empty and credit assignment degrades gracefully (the Architect falls back to conservative, blame-free proposals). The judge model is configurable: it reuses the `--judge-model` / `DEFAULT_JUDGE_MODELS` seam, prefixed with your `--provider` (e.g. `gemini/gemini-3.6-flash`) and routed through LiteLLM, so it scores with the same vendor and env keys as the rest of the run, never DeepEval's OpenAI default.
|
|
257
|
+
|
|
218
258
|
---
|
|
219
259
|
|
|
260
|
+
|
|
261
|
+
|
|
220
262
|
## The optimization loop (P-E-C)
|
|
221
263
|
|
|
222
264
|
One cycle, end-to-end:
|
|
@@ -296,6 +338,8 @@ archforge-optimizer <command> [flags]
|
|
|
296
338
|
| `--seed <path>` | bootstrap the root incumbent from a Spec JSON (first run); defaults to `archforge_optimizer/spec.json` when present |
|
|
297
339
|
| `--adapter <dotted.path[:Class]>` | your `HostMAS` adapter; defaults to `archforge_optimizer.host:AppAdapter` when the scaffold is present (not on the `--provider scripted` fake path) |
|
|
298
340
|
| `--provider {scripted\|anthropic\|openai\|groq\|gemini}` | LLM backing the Architect + Judge |
|
|
341
|
+
| `--evaluator {native\|deepeval}` | evaluation backend (default: `DEFAULT_EVALUATOR`); `deepeval` needs the `[deepeval]` extra |
|
|
342
|
+
| `--deepeval-metric {answer_relevancy\|faithfulness}` | DeepEval metric to score with (repeatable; default: `DEFAULT_DEEPEVAL_METRICS`) |
|
|
299
343
|
| `--suite <path>` | evaluation suite JSON (default: `.archforge/suite.json`) |
|
|
300
344
|
| `--tau <float>` | promotion margin τ |
|
|
301
345
|
| `--delta <float>` | regression floor δ (≥ τ) |
|
|
@@ -1,34 +1,3 @@
|
|
|
1
|
-
Metadata-Version: 2.5
|
|
2
|
-
Name: archforge-optimizer
|
|
3
|
-
Version: 0.3.0
|
|
4
|
-
Summary: ArchForge: a self-improving meta-layer over multi-agent systems
|
|
5
|
-
License-Expression: MIT
|
|
6
|
-
License-File: LICENSE
|
|
7
|
-
Requires-Python: >=3.11
|
|
8
|
-
Requires-Dist: litellm
|
|
9
|
-
Requires-Dist: pydantic>=2.7
|
|
10
|
-
Requires-Dist: python-dotenv>=1.0
|
|
11
|
-
Provides-Extra: dev
|
|
12
|
-
Requires-Dist: hypothesis>=6; extra == 'dev'
|
|
13
|
-
Requires-Dist: mypy>=1.11; extra == 'dev'
|
|
14
|
-
Requires-Dist: pytest-cov>=5; extra == 'dev'
|
|
15
|
-
Requires-Dist: pytest>=8; extra == 'dev'
|
|
16
|
-
Requires-Dist: ruff>=0.5; extra == 'dev'
|
|
17
|
-
Provides-Extra: providers
|
|
18
|
-
Requires-Dist: anthropic>=0.40; extra == 'providers'
|
|
19
|
-
Requires-Dist: google-genai>=1.0; extra == 'providers'
|
|
20
|
-
Requires-Dist: groq>=0.11; extra == 'providers'
|
|
21
|
-
Requires-Dist: openai>=1.40; extra == 'providers'
|
|
22
|
-
Provides-Extra: providers-anthropic
|
|
23
|
-
Requires-Dist: anthropic>=0.40; extra == 'providers-anthropic'
|
|
24
|
-
Provides-Extra: providers-gemini
|
|
25
|
-
Requires-Dist: google-genai>=1.0; extra == 'providers-gemini'
|
|
26
|
-
Provides-Extra: providers-groq
|
|
27
|
-
Requires-Dist: groq>=0.11; extra == 'providers-groq'
|
|
28
|
-
Provides-Extra: providers-openai
|
|
29
|
-
Requires-Dist: openai>=1.40; extra == 'providers-openai'
|
|
30
|
-
Description-Content-Type: text/markdown
|
|
31
|
-
|
|
32
1
|
# ArchForge
|
|
33
2
|
|
|
34
3
|
<p align="center">
|
|
@@ -246,8 +215,17 @@ archforge-optimizer approve --all # move PENDING_HUMAN structural wins into ac
|
|
|
246
215
|
|
|
247
216
|
The `--provider` flag selects the LLM backing the Architect + Judge (`anthropic` / `openai` / `groq` / `gemini` for real runs). Every real provider goes through **one LiteLLM client**: the provider just prefixes the model id (`openai/gpt-4o`, `gemini/gemini-3.6-flash`, …). The host MAS is wired via `--adapter my_pkg.my_host:MyAdapter`. After `init`, `evolve` already defaults it to `archforge_optimizer.host:AppAdapter`, so you only pass the flag for a custom adapter.
|
|
248
217
|
|
|
218
|
+
### Evaluation backends
|
|
219
|
+
|
|
220
|
+
The Judge is pluggable behind one `JudgeProtocol` seam (`score` / `score_suite`), so the optimizer never knows which evaluator produced a score. `--evaluator` picks the backend (default `DEFAULT_EVALUATOR="native"` in `.archforge/archforge.py`):
|
|
221
|
+
|
|
222
|
+
- **`native`** (default): the built-in LLM-as-judge. One structured call to your `--provider` model scores every rubric dimension plus a per-step breakdown (used for credit assignment).
|
|
223
|
+
- **`deepeval`**: the external [DeepEval](https://deepeval.com) backend (optional: `pip install "archforge-optimizer[deepeval]"`). Each run is projected into a DeepEval `LLMTestCase` and scored by standalone metrics (`--deepeval-metric answer_relevancy --deepeval-metric faithfulness`, or the `DEFAULT_DEEPEVAL_METRICS` tunable). Metric scores land in `RunScore.rubric_scores`; the aggregate is their mean, comparable to the native judge's [0,1] aggregate. DeepEval is run-level, not per-step, so `step_scores` is empty and credit assignment degrades gracefully (the Architect falls back to conservative, blame-free proposals). The judge model is configurable: it reuses the `--judge-model` / `DEFAULT_JUDGE_MODELS` seam, prefixed with your `--provider` (e.g. `gemini/gemini-3.6-flash`) and routed through LiteLLM, so it scores with the same vendor and env keys as the rest of the run, never DeepEval's OpenAI default.
|
|
224
|
+
|
|
249
225
|
---
|
|
250
226
|
|
|
227
|
+
|
|
228
|
+
|
|
251
229
|
## The optimization loop (P-E-C)
|
|
252
230
|
|
|
253
231
|
One cycle, end-to-end:
|
|
@@ -327,6 +305,8 @@ archforge-optimizer <command> [flags]
|
|
|
327
305
|
| `--seed <path>` | bootstrap the root incumbent from a Spec JSON (first run); defaults to `archforge_optimizer/spec.json` when present |
|
|
328
306
|
| `--adapter <dotted.path[:Class]>` | your `HostMAS` adapter; defaults to `archforge_optimizer.host:AppAdapter` when the scaffold is present (not on the `--provider scripted` fake path) |
|
|
329
307
|
| `--provider {scripted\|anthropic\|openai\|groq\|gemini}` | LLM backing the Architect + Judge |
|
|
308
|
+
| `--evaluator {native\|deepeval}` | evaluation backend (default: `DEFAULT_EVALUATOR`); `deepeval` needs the `[deepeval]` extra |
|
|
309
|
+
| `--deepeval-metric {answer_relevancy\|faithfulness}` | DeepEval metric to score with (repeatable; default: `DEFAULT_DEEPEVAL_METRICS`) |
|
|
330
310
|
| `--suite <path>` | evaluation suite JSON (default: `.archforge/suite.json`) |
|
|
331
311
|
| `--tau <float>` | promotion margin τ |
|
|
332
312
|
| `--delta <float>` | regression floor δ (≥ τ) |
|
|
@@ -67,10 +67,15 @@ from archforge import userconfig as ucfg
|
|
|
67
67
|
from archforge.config import (
|
|
68
68
|
ALL_PROVIDERS as _PROVIDERS,
|
|
69
69
|
DEFAULT_SUITE_ID, DEFAULT_TASK_ID, DEFAULT_TASK_INPUT,
|
|
70
|
+
EVALUATORS as _EVALUATORS,
|
|
70
71
|
PROG, load_env,
|
|
71
72
|
)
|
|
72
73
|
from archforge.userconfig import ConfigNotInitialized
|
|
73
74
|
|
|
75
|
+
# The DeepEval metrics the `--deepeval-metric` flag accepts (its `choices`). Kept in
|
|
76
|
+
# sync with the factory registry in archforge/judge/deepeval.py.
|
|
77
|
+
_DEEPEVAL_METRICS = ("answer_relevancy", "faithfulness")
|
|
78
|
+
|
|
74
79
|
# Fixed run-state / config-discovery dir (a system path, independent of the
|
|
75
80
|
# tunable DEFAULT_ROOT_DIR which the embedder API reads via archforge.userconfig).
|
|
76
81
|
_DEFAULT_ROOT = ".archforge"
|
|
@@ -105,12 +110,42 @@ def _import_adapter(dotted: str) -> HostMAS:
|
|
|
105
110
|
its own ``__init__`` carries whatever its MAS needs (Lumina loads its base
|
|
106
111
|
prompts; a framework adapter wraps its graph). No PR into core to adapt a
|
|
107
112
|
new MAS: ``--adapter mypkg:MyAdapter`` wires it; ``--provider`` keeps the
|
|
108
|
-
Architect/Judge organs, ``--seed`` the bootstrap Spec.
|
|
113
|
+
Architect/Judge organs, ``--seed`` the bootstrap Spec.
|
|
114
|
+
|
|
115
|
+
Surfaces clear, actionable errors for the common failure modes users hit when
|
|
116
|
+
editing the scaffold's ``app.py``:
|
|
117
|
+
|
|
118
|
+
* ``ImportError``/``ModuleNotFoundError`` — a MAS dependency listed in
|
|
119
|
+
``app.py``'s imports is not installed (e.g. ``from mymas.graph import
|
|
120
|
+
build_graph`` when ``mymas`` isn't on ``sys.path``).
|
|
121
|
+
* ``KeyError`` — ``node_ids(build_graph())`` can't find a node name that
|
|
122
|
+
``_GID[...]`` references (a rename in ``build_graph`` that wasn't
|
|
123
|
+
reflected in ``_GID``).
|
|
124
|
+
"""
|
|
109
125
|
if ":" in dotted:
|
|
110
126
|
modpath, cls = dotted.split(":", 1)
|
|
111
127
|
else:
|
|
112
128
|
modpath, cls = dotted, ""
|
|
113
|
-
|
|
129
|
+
try:
|
|
130
|
+
module = importlib.import_module(modpath)
|
|
131
|
+
except ModuleNotFoundError as exc:
|
|
132
|
+
# Surface the missing package name explicitly so the user knows what to
|
|
133
|
+
# install, and point at `app.py` as the likely source of the bad import.
|
|
134
|
+
missing = exc.name or str(exc)
|
|
135
|
+
raise SystemExit(
|
|
136
|
+
f"! --adapter: cannot import {modpath!r} — missing module {missing!r}.\n"
|
|
137
|
+
f" Check the imports at the top of archforge_optimizer/app.py: make sure\n"
|
|
138
|
+
f" every package your MAS needs (e.g. `langgraph`, your MAS package) is\n"
|
|
139
|
+
f" installed in the current Python environment.\n"
|
|
140
|
+
f" Original error: {exc}"
|
|
141
|
+
) from exc
|
|
142
|
+
except ImportError as exc:
|
|
143
|
+
raise SystemExit(
|
|
144
|
+
f"! --adapter: failed to import {modpath!r}.\n"
|
|
145
|
+
f" Check the imports at the top of archforge_optimizer/app.py and make\n"
|
|
146
|
+
f" sure all your MAS dependencies are installed.\n"
|
|
147
|
+
f" Original error: {exc}"
|
|
148
|
+
) from exc
|
|
114
149
|
if not cls:
|
|
115
150
|
# Bare module: expect it to expose a ``HostMAS``-protocol attr named
|
|
116
151
|
# ``HostMAS`` or the last path segment; else error loudly.
|
|
@@ -123,7 +158,33 @@ def _import_adapter(dotted: str) -> HostMAS:
|
|
|
123
158
|
f"Pass it as `module:ClassName`."
|
|
124
159
|
) from exc
|
|
125
160
|
if isinstance(obj, type):
|
|
126
|
-
|
|
161
|
+
try:
|
|
162
|
+
return obj() # a HostMAS/BaseHostAdapter subclass → instance
|
|
163
|
+
except KeyError as exc:
|
|
164
|
+
# Usually `_GID = node_ids(build_graph())` in app.py referencing a
|
|
165
|
+
# node name the graph doesn't define — but a KeyError can come from
|
|
166
|
+
# ANY dict/env lookup in the adapter's __init__, so name the likely
|
|
167
|
+
# cause without asserting it.
|
|
168
|
+
app_py = Path.cwd() / "archforge_optimizer" / "app.py"
|
|
169
|
+
raise SystemExit(
|
|
170
|
+
f"! --adapter: {dotted!r} raised KeyError {exc} at startup.\n"
|
|
171
|
+
f" Most common cause: the `_GID` dict in {app_py} references a\n"
|
|
172
|
+
f" node name that `build_graph()` never defines (check the\n"
|
|
173
|
+
f" `add_node(..., name=...)` calls in your graph builder). If\n"
|
|
174
|
+
f" `_GID` matches, look for another dict/env lookup in the\n"
|
|
175
|
+
f" adapter's __init__ (e.g. os.environ[...])."
|
|
176
|
+
) from exc
|
|
177
|
+
except AssertionError:
|
|
178
|
+
# Re-raise so `make-spec`'s own handler (which formats build_spec()
|
|
179
|
+
# lint failures as clean rc=1 output) can catch it.
|
|
180
|
+
raise
|
|
181
|
+
except Exception as exc: # noqa: BLE001
|
|
182
|
+
app_py = Path.cwd() / "archforge_optimizer" / "app.py"
|
|
183
|
+
raise SystemExit(
|
|
184
|
+
f"! --adapter: failed to instantiate {dotted!r}.\n"
|
|
185
|
+
f" Check {app_py} — the adapter raised an unexpected error at\n"
|
|
186
|
+
f" startup: {type(exc).__name__}: {exc}"
|
|
187
|
+
) from exc
|
|
127
188
|
if isinstance(obj, HostMAS):
|
|
128
189
|
return obj # already an instance
|
|
129
190
|
raise SystemExit(f"--adapter: {dotted!r} resolved to a {type(obj).__name__}, "
|
|
@@ -163,6 +224,14 @@ def _add_evolve_args(p: argparse.ArgumentParser, *, loop: bool) -> None:
|
|
|
163
224
|
help="model id for the Architect (else the provider default)")
|
|
164
225
|
p.add_argument("--judge-model", default=None,
|
|
165
226
|
help="model id for the Judge (else the provider default)")
|
|
227
|
+
p.add_argument("--evaluator", choices=_EVALUATORS, default=None,
|
|
228
|
+
help="evaluation backend (default from archforge.py): native is the "
|
|
229
|
+
"built-in LLM-as-judge; deepeval is the external DeepEval backend "
|
|
230
|
+
"(needs the [deepeval] extra)")
|
|
231
|
+
p.add_argument("--deepeval-metric", choices=_DEEPEVAL_METRICS, action="append",
|
|
232
|
+
default=None, dest="deepeval_metrics",
|
|
233
|
+
help="DeepEval metric to score with (repeatable; else the "
|
|
234
|
+
"DEFAULT_DEEPEVAL_METRICS tunable)")
|
|
166
235
|
p.add_argument("--suite", metavar="PATH", default=None,
|
|
167
236
|
help="path to a suite.json (overrides the DEFAULT_SUITE_FILE tunable; "
|
|
168
237
|
"the file's tasks define what you optimize against)")
|
|
@@ -331,8 +400,34 @@ def _default_components(args: argparse.Namespace) -> Components:
|
|
|
331
400
|
arch_models = ucfg.get("DEFAULT_ARCHITECT_MODELS")
|
|
332
401
|
judge_models = ucfg.get("DEFAULT_JUDGE_MODELS")
|
|
333
402
|
arch = Architect(llm, model=args.architect_model or arch_models[provider])
|
|
334
|
-
|
|
335
|
-
|
|
403
|
+
|
|
404
|
+
# Evaluation backend seam (issue #2): `native` keeps the built-in Judge over
|
|
405
|
+
# the provider's LLM; `deepeval` swaps in the external DeepEval backend, which
|
|
406
|
+
# scores independently of the Architect's LLM (it carries its own model/keys).
|
|
407
|
+
# `default=` on both: a config written by an OLDER init (before issue #2)
|
|
408
|
+
# lacks these names and would KeyError otherwise — the defaults keep the
|
|
409
|
+
# pre-existing native behavior for those projects.
|
|
410
|
+
evaluator = getattr(args, "evaluator", None) or ucfg.get(
|
|
411
|
+
"DEFAULT_EVALUATOR", default="native")
|
|
412
|
+
if evaluator == "deepeval":
|
|
413
|
+
from archforge.judge import make_evaluator
|
|
414
|
+
|
|
415
|
+
try:
|
|
416
|
+
metrics = getattr(args, "deepeval_metrics", None) or ucfg.get(
|
|
417
|
+
"DEFAULT_DEEPEVAL_METRICS", default=["answer_relevancy"])
|
|
418
|
+
# DeepEval scores through LiteLLM too: reuse the Judge model seam
|
|
419
|
+
# (--judge-model / DEFAULT_JUDGE_MODELS) and prefix it with the
|
|
420
|
+
# provider so LiteLLM routes to the same vendor (never DeepEval's
|
|
421
|
+
# OpenAI default).
|
|
422
|
+
from archforge.llm.litellm import prefix_model
|
|
423
|
+
de_model = prefix_model(provider, args.judge_model or judge_models[provider])
|
|
424
|
+
judge = make_evaluator("deepeval", model=de_model, metrics=metrics)
|
|
425
|
+
except LLMError as exc:
|
|
426
|
+
print(f"[evaluator] {exc}", file=sys.stderr)
|
|
427
|
+
raise
|
|
428
|
+
else:
|
|
429
|
+
judge = Judge(llm, model=args.judge_model or judge_models[provider],
|
|
430
|
+
rubric=default_rubric())
|
|
336
431
|
return Components(host=FakeHostMAS(), judge=judge, architect=arch, suite=suite)
|
|
337
432
|
|
|
338
433
|
|
|
@@ -549,6 +644,31 @@ def _scaffolded_spec_available() -> bool:
|
|
|
549
644
|
return (Path.cwd() / "archforge_optimizer" / "spec.json").is_file()
|
|
550
645
|
|
|
551
646
|
|
|
647
|
+
# Unique token written by `init` into the placeholder `build_graph` — present ONLY
|
|
648
|
+
# when the user hasn't replaced it yet (spec E-guard: refuse to evolve template code).
|
|
649
|
+
_TEMPLATE_MARKER = "EDIT app.py: import your MAS"
|
|
650
|
+
|
|
651
|
+
|
|
652
|
+
def _scaffolded_adapter_is_template() -> bool:
|
|
653
|
+
"""True when the scaffolded `app.py` still contains the unedited placeholder
|
|
654
|
+
`build_graph` (the ``NotImplementedError`` with the "EDIT app.py" marker).
|
|
655
|
+
|
|
656
|
+
This is the signal that the user ran `init` but hasn't wired their real MAS yet.
|
|
657
|
+
Returning True causes `evolve`/`evolve-loop` to refuse with a clear message
|
|
658
|
+
instead of silently running the Judge against the fake placeholder spec.
|
|
659
|
+
Only the scaffolded adapter path is checked — an explicit ``--adapter`` flag
|
|
660
|
+
skips this (the user owns that adapter).
|
|
661
|
+
"""
|
|
662
|
+
app_py = Path.cwd() / "archforge_optimizer" / "app.py"
|
|
663
|
+
if not app_py.is_file():
|
|
664
|
+
return False
|
|
665
|
+
try:
|
|
666
|
+
text = app_py.read_text(encoding="utf-8")
|
|
667
|
+
except OSError:
|
|
668
|
+
return False
|
|
669
|
+
return _TEMPLATE_MARKER in text
|
|
670
|
+
|
|
671
|
+
|
|
552
672
|
def _cmd_evolve(args: argparse.Namespace, *, components: Components | None,
|
|
553
673
|
loop: bool) -> int:
|
|
554
674
|
# Injected `components` (the test/embedding path) always win — they ARE the
|
|
@@ -597,6 +717,26 @@ def _cmd_evolve(args: argparse.Namespace, *, components: Components | None,
|
|
|
597
717
|
# Skipped on the scripted fake path (FakeHostMAS is the zero-cost host) and
|
|
598
718
|
# when `components` were injected (test/embedding path owns the host).
|
|
599
719
|
adapter_path = getattr(args, "adapter", None)
|
|
720
|
+
# Guard: refuse to run against the unedited scaffold template. When the user
|
|
721
|
+
# ran `init` but hasn't replaced `build_graph` in `app.py` yet, the placeholder
|
|
722
|
+
# Spec is valid (it lints) but the MAS is a stub — running `evolve` would silently
|
|
723
|
+
# score fake outputs through the Judge with no real pipeline. Only checked when
|
|
724
|
+
# the scaffold is auto-detected (no explicit --adapter flag) so user-supplied
|
|
725
|
+
# adapters are never blocked. The scripted path is also exempt: it keeps
|
|
726
|
+
# FakeHostMAS and never touches the scaffold, so a template app.py is inert
|
|
727
|
+
# there (and blocking it would break the zero-cost demo in a scaffolded dir).
|
|
728
|
+
if (adapter_path is None and not components_injected
|
|
729
|
+
and (args.provider or ucfg.get("PROVIDER")) != "scripted"
|
|
730
|
+
and _scaffolded_adapter_available()
|
|
731
|
+
and _scaffolded_adapter_is_template()):
|
|
732
|
+
print(
|
|
733
|
+
f"! archforge_optimizer/app.py still contains the placeholder `build_graph` "
|
|
734
|
+
f"(the scaffold written by `{PROG} init` has not been edited yet).\n"
|
|
735
|
+
f" Edit the # EDIT: markers in archforge_optimizer/app.py to wire your real "
|
|
736
|
+
f"MAS, then re-run `{PROG} make-spec` to rebuild the Spec before evolving.",
|
|
737
|
+
file=sys.stderr,
|
|
738
|
+
)
|
|
739
|
+
return 1
|
|
600
740
|
if (adapter_path is None and not components_injected
|
|
601
741
|
and _scaffolded_adapter_available()
|
|
602
742
|
and (args.provider or ucfg.get("PROVIDER")) != "scripted"):
|
|
@@ -29,7 +29,7 @@ from pathlib import Path
|
|
|
29
29
|
# =========================================================================== #
|
|
30
30
|
# The package version. Read by hatchling for the built distribution and
|
|
31
31
|
# re-exported as `archforge.__version__` (archforge/__init__.py). One place.
|
|
32
|
-
VERSION: str = "0.
|
|
32
|
+
VERSION: str = "0.4.0"
|
|
33
33
|
|
|
34
34
|
|
|
35
35
|
# =========================================================================== #
|
|
@@ -46,6 +46,20 @@ REAL_PROVIDERS: tuple[str, ...] = ("anthropic", "openai", "groq", "gemini")
|
|
|
46
46
|
ALL_PROVIDERS: tuple[str, ...] = (SCRIPTED_PROVIDER,) + REAL_PROVIDERS
|
|
47
47
|
|
|
48
48
|
|
|
49
|
+
# =========================================================================== #
|
|
50
|
+
# Evaluation backends — the ROSTER only (NOT the default choice; that's a tunable)
|
|
51
|
+
# =========================================================================== #
|
|
52
|
+
# The evaluators the CLI's `--evaluator` flag ACCEPTS (its `choices`). `native` is
|
|
53
|
+
# the built-in LLM-as-judge (`Judge` over an `LLMClient`); `deepeval` is the
|
|
54
|
+
# external DeepEval backend (`DeepEvalEvaluator`, an optional extra). WHICH is the
|
|
55
|
+
# default is a tunable → it lives in .archforge/archforge.py (resolved by
|
|
56
|
+
# archforge.userconfig), NOT here. `EVALUATORS` is re-exported from archforge.judge;
|
|
57
|
+
# `make_evaluator(evaluator)` dispatches on these.
|
|
58
|
+
NATIVE_EVALUATOR: str = "native"
|
|
59
|
+
DEEPEVAL_EVALUATOR: str = "deepeval"
|
|
60
|
+
EVALUATORS: tuple[str, ...] = (NATIVE_EVALUATOR, DEEPEVAL_EVALUATOR)
|
|
61
|
+
|
|
62
|
+
|
|
49
63
|
# =========================================================================== #
|
|
50
64
|
# Project environment (`.env`) — how secrets reach the provider SDKs
|
|
51
65
|
# =========================================================================== #
|
|
@@ -128,6 +142,8 @@ __all__ = [
|
|
|
128
142
|
"VERSION",
|
|
129
143
|
# llm provider roster
|
|
130
144
|
"SCRIPTED_PROVIDER", "REAL_PROVIDERS", "ALL_PROVIDERS",
|
|
145
|
+
# evaluation backend roster
|
|
146
|
+
"NATIVE_EVALUATOR", "DEEPEVAL_EVALUATOR", "EVALUATORS",
|
|
131
147
|
# project environment (.env loader for provider API keys)
|
|
132
148
|
"load_env",
|
|
133
149
|
# storage layout
|
|
@@ -57,6 +57,13 @@ _FIELDS: tuple[tuple[str, str, str], ...] = (
|
|
|
57
57
|
# --- LLM provider
|
|
58
58
|
("PROVIDER", '"gemini"',
|
|
59
59
|
"which LLM to use (scripted|anthropic|openai|groq|gemini); scripted needs no API key"),
|
|
60
|
+
# --- evaluation backend
|
|
61
|
+
("DEFAULT_EVALUATOR", '"native"',
|
|
62
|
+
"which evaluator scores runs (native|deepeval); native = built-in LLM-as-judge, "
|
|
63
|
+
"deepeval = external DeepEval backend (needs the [deepeval] extra)"),
|
|
64
|
+
("DEFAULT_DEEPEVAL_METRICS", '["answer_relevancy"]',
|
|
65
|
+
"DeepEval metrics to run per run (answer_relevancy|faithfulness); "
|
|
66
|
+
"each becomes a rubric_scores dimension and the aggregate is their mean"),
|
|
60
67
|
("DEFAULT_ARCHITECT_MODELS", _DEFAULT_ARCHITECT_MODELS,
|
|
61
68
|
"default Architect (proposer) model per provider; a bare LLMClient call falls "
|
|
62
69
|
"back here too — edit the dict to change it"),
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
"""The Judge — LLM-as-judge scoring of runs (spec §3, §4).
|
|
2
|
+
|
|
3
|
+
`Judge.score(trace, task, rubric_id) -> RunScore` turns a run's trace into a
|
|
4
|
+
scored verdict: an aggregate + named rubric dimensions + confidence, plus a
|
|
5
|
+
*per-step breakdown* (StepScore[]) that names which agent lost which points. The
|
|
6
|
+
per-step breakdown is the raw material the Architect uses to credit-assign a
|
|
7
|
+
fault to a node/route.
|
|
8
|
+
|
|
9
|
+
`score_suite(...)` aggregates over R repeats: mean (down-weighted by
|
|
10
|
+
confidence) and a stable aggregate used by the Gatekeeper. `rubric_id` is
|
|
11
|
+
stamped on every score so cross-rubric comparisons never masquerade as
|
|
12
|
+
improvement (invariant I5, spec E2).
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from typing import Sequence
|
|
18
|
+
|
|
19
|
+
from archforge.config import EVALUATORS
|
|
20
|
+
from archforge.judge.base import (
|
|
21
|
+
Judge, JudgeProtocol, SuiteAggregate, default_rubric,
|
|
22
|
+
)
|
|
23
|
+
from archforge.judge.scripted import ScriptedJudge
|
|
24
|
+
from archforge.llm.base import LLMError
|
|
25
|
+
|
|
26
|
+
# `EVALUATORS` is re-exported (in __all__) straight from archforge.config — the
|
|
27
|
+
# single source of truth the CLI also imports for its `--evaluator` choices.
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def make_evaluator(
|
|
31
|
+
evaluator: str, *, model: str | None = None, metrics: Sequence[str] | None = None
|
|
32
|
+
) -> JudgeProtocol:
|
|
33
|
+
"""Build an external evaluation backend by name.
|
|
34
|
+
|
|
35
|
+
Resolves `DeepEvalEvaluator` lazily (so importing DeepEval is deferred to here,
|
|
36
|
+
and the optional extra is only required at construction). Raises `LLMError` for
|
|
37
|
+
an unknown evaluator so the CLI surfaces one clear message across the backend
|
|
38
|
+
seam. The `native` evaluator is NOT built here: it is the built-in `Judge`,
|
|
39
|
+
constructed inline by the CLI from the provider's `LLMClient`.
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
if evaluator == "native":
|
|
43
|
+
raise LLMError(
|
|
44
|
+
"'native' is the built-in Judge; construct it with Judge(llm, model=...)"
|
|
45
|
+
)
|
|
46
|
+
if evaluator == "deepeval":
|
|
47
|
+
from archforge.judge.deepeval import DeepEvalEvaluator
|
|
48
|
+
|
|
49
|
+
return DeepEvalEvaluator(model=model, metrics=metrics) # type: ignore[return-value]
|
|
50
|
+
raise LLMError(f"unknown evaluator {evaluator!r}; expected one of {EVALUATORS}")
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
__all__ = [
|
|
54
|
+
"Judge", "JudgeProtocol", "ScriptedJudge", "SuiteAggregate", "default_rubric",
|
|
55
|
+
"make_evaluator", "EVALUATORS",
|
|
56
|
+
]
|
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
"""The DeepEval evaluation backend — an external `JudgeProtocol` (issue #2).
|
|
2
|
+
|
|
3
|
+
`DeepEvalEvaluator` implements the SAME protocol the built-in `Judge` does
|
|
4
|
+
(`JudgeProtocol` in archforge/judge/base.py), so the SuiteRunner and optimizer
|
|
5
|
+
treat it exactly like the native backend: `score(trace, task, rubric_id) ->
|
|
6
|
+
RunScore` and `score_suite(...) -> SuiteAggregate`. Nothing downstream knows or
|
|
7
|
+
cares which backend produced a score — that is the pluggability the issue asks
|
|
8
|
+
for (criterion 5: the optimizer stays independent of the evaluator).
|
|
9
|
+
|
|
10
|
+
Unlike the native `Judge` (one LLM call returning a whole JSON rubric verdict),
|
|
11
|
+
DeepEval scores a run with one or more standalone LLM-judge METRICS
|
|
12
|
+
(AnswerRelevancy, Faithfulness, ...), each yielding a score in [0,1]. We project
|
|
13
|
+
an ArchForge trace into a DeepEval `LLMTestCase` and convert each metric's score
|
|
14
|
+
into a `RunScore`:
|
|
15
|
+
|
|
16
|
+
* `rubric_scores` = {metric slug: metric.score} (the named dimensions)
|
|
17
|
+
* `aggregate` = mean of the metric scores (comparable to the native Judge's
|
|
18
|
+
[0,1] aggregate; invariant I5 still holds: rubric_id is
|
|
19
|
+
stamped on every score)
|
|
20
|
+
* `step_scores` = [] (top-level only — DeepEval is run-level, not per-step;
|
|
21
|
+
credit assignment degrades gracefully to `blame=None`)
|
|
22
|
+
|
|
23
|
+
DeepEval is imported LAZILY inside `__init__` (it is an optional extra and is
|
|
24
|
+
heavy — it pulls an LLM provider SDK + more). A missing install surfaces as a
|
|
25
|
+
clear `LLMError` at construction, never a bare ImportError at import time, so
|
|
26
|
+
`import archforge` stays deepeval-free.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
from __future__ import annotations
|
|
30
|
+
|
|
31
|
+
from typing import Any, Sequence
|
|
32
|
+
|
|
33
|
+
from archforge.llm.base import LLMError
|
|
34
|
+
from archforge.judge.base import SuiteAggregate, aggregate_scores
|
|
35
|
+
from archforge.host.base import Task
|
|
36
|
+
import archforge.models as m
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
# A metric name -> factory. Kept as a mapping (not `if` chains) so a configured
|
|
40
|
+
# name resolves to a metric object AND gives us the stable rubric_scores slug.
|
|
41
|
+
_METRIC_FACTORIES: dict[str, Any] = {}
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _metric_slug(name: str) -> str:
|
|
45
|
+
"""Canonical rubric_scores key for a DeepEval metric name (lower snake)."""
|
|
46
|
+
return name.strip().lower().replace("-", "_").replace(" ", "_")
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class DeepEvalEvaluator:
|
|
50
|
+
"""Scores runs via DeepEval metrics, satisfying `JudgeProtocol`.
|
|
51
|
+
|
|
52
|
+
`metrics` are metric NAMES (`"answer_relevancy"`, `"faithfulness"`) or
|
|
53
|
+
already-constructed DeepEval metric objects. Names resolve to metric objects
|
|
54
|
+
lazily at construction (DeepEval stays unimported until then). `model` is a
|
|
55
|
+
LiteLLM-style model string (e.g. `"gemini/gemini-3.6-flash"`; prefix with the
|
|
56
|
+
provider) or an already-built DeepEval model object; a string is wrapped in
|
|
57
|
+
DeepEval's `LiteLLMModel` so scoring routes through LiteLLM with the
|
|
58
|
+
provider's env key, never DeepEval's OpenAI default. When None, DeepEval
|
|
59
|
+
falls back to its own default model.
|
|
60
|
+
"""
|
|
61
|
+
|
|
62
|
+
def __init__(
|
|
63
|
+
self,
|
|
64
|
+
*,
|
|
65
|
+
metrics: Sequence[str | Any] | None = None,
|
|
66
|
+
model: str | Any | None = None,
|
|
67
|
+
threshold: float = 0.5,
|
|
68
|
+
) -> None:
|
|
69
|
+
self._deepeval = _import_deepeval() # LLMError if the extra is missing
|
|
70
|
+
# A string model is wrapped in LiteLLMModel (below) so the provider
|
|
71
|
+
# prefix routes correctly; `_judge_label` is only the provenance stamp
|
|
72
|
+
# on JudgeMeta.
|
|
73
|
+
self._model = self._wrap_model(model)
|
|
74
|
+
if isinstance(model, str):
|
|
75
|
+
self._judge_label = model
|
|
76
|
+
elif model is not None and hasattr(model, "get_model_name"):
|
|
77
|
+
self._judge_label = model.get_model_name() # DeepEvalBaseLLM API
|
|
78
|
+
else:
|
|
79
|
+
self._judge_label = "deepeval"
|
|
80
|
+
self._threshold = threshold
|
|
81
|
+
self._metrics = self._build_metrics(metrics) # list[(slug, metric)]
|
|
82
|
+
|
|
83
|
+
def _wrap_model(self, model: str | Any | None) -> Any:
|
|
84
|
+
"""Wrap a string model id in DeepEval's `LiteLLMModel` (lazy import).
|
|
85
|
+
|
|
86
|
+
A bare string handed to a DeepEval metric resolves to DeepEval's OpenAI
|
|
87
|
+
default regardless of any provider prefix, so we wrap it explicitly:
|
|
88
|
+
`LiteLLMModel` is a `DeepEvalBaseLLM`, which `initialize_model` passes
|
|
89
|
+
through untouched, and LiteLLM routes by the model's provider prefix
|
|
90
|
+
using the provider's env key (same as ArchForge's own LiteLLMClient).
|
|
91
|
+
Non-string models (already-built DeepEval models) pass through.
|
|
92
|
+
"""
|
|
93
|
+
if model is None or not isinstance(model, str):
|
|
94
|
+
return model
|
|
95
|
+
try:
|
|
96
|
+
from deepeval.models import LiteLLMModel
|
|
97
|
+
|
|
98
|
+
return LiteLLMModel(model=model)
|
|
99
|
+
except LLMError:
|
|
100
|
+
raise
|
|
101
|
+
except Exception as exc: # noqa: BLE001 — deepeval/litellm construction errors
|
|
102
|
+
raise LLMError(
|
|
103
|
+
f"could not build the DeepEval judge model {model!r}: {exc}"
|
|
104
|
+
) from exc
|
|
105
|
+
|
|
106
|
+
# --------------------------------------------------------------- protocol
|
|
107
|
+
def score(self, trace: m.Trace, task: Task, rubric_id: str) -> m.RunScore:
|
|
108
|
+
test_case = self._llm_test_case(trace, task)
|
|
109
|
+
rubric_scores: dict[str, float] = {}
|
|
110
|
+
for slug, metric in self._metrics:
|
|
111
|
+
# `measure()` is SYNCHRONOUS in deepeval (the async entry point is
|
|
112
|
+
# `a_measure`), matching the sync SuiteRunner path directly — do NOT
|
|
113
|
+
# wrap it in asyncio.run (that raises TypeError on a non-coroutine).
|
|
114
|
+
# A metric LLM outage raises deepeval's own error; convert to
|
|
115
|
+
# LLMError so the SuiteRunner's bounded retry treats it exactly like
|
|
116
|
+
# a native Judge failure (E9), never a fabricated 0.0.
|
|
117
|
+
try:
|
|
118
|
+
metric.measure(test_case)
|
|
119
|
+
except Exception as exc: # noqa: BLE001 — deepeval's error type
|
|
120
|
+
raise LLMError(f"DeepEval metric {slug!r} failed: {exc}") from exc
|
|
121
|
+
rubric_scores[slug] = float(metric.score or 0.0)
|
|
122
|
+
|
|
123
|
+
aggregate = _mean(rubric_scores.values()) if rubric_scores else 0.0
|
|
124
|
+
return m.RunScore(
|
|
125
|
+
run_id=trace.run_id,
|
|
126
|
+
spec_id=trace.spec_id,
|
|
127
|
+
task_id=trace.task_id,
|
|
128
|
+
rubric_scores=rubric_scores,
|
|
129
|
+
aggregate=aggregate,
|
|
130
|
+
confidence=1.0,
|
|
131
|
+
judge_meta=m.JudgeMeta(model=self._judge_label, rubric_id=rubric_id),
|
|
132
|
+
step_scores=[], # top-level only (DeepEval is run-level, not per-step)
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
def score_suite(
|
|
136
|
+
self, scores: Sequence[m.RunScore], *, suite_id: str, rubric_id: str
|
|
137
|
+
) -> SuiteAggregate:
|
|
138
|
+
# Reuse the pure shared aggregation (same as the native Judge) so the two
|
|
139
|
+
# backends produce structurally identical, comparable SuiteAggregates.
|
|
140
|
+
return aggregate_scores(scores, suite_id=suite_id, rubric_id=rubric_id)
|
|
141
|
+
|
|
142
|
+
# --------------------------------------------------------------- internals
|
|
143
|
+
def _build_metrics(
|
|
144
|
+
self, metrics: Sequence[str | Any] | None
|
|
145
|
+
) -> list[tuple[str, Any]]:
|
|
146
|
+
built: list[tuple[str, Any]] = []
|
|
147
|
+
for item in (metrics or ["answer_relevancy"]):
|
|
148
|
+
if isinstance(item, str):
|
|
149
|
+
slug = _metric_slug(item)
|
|
150
|
+
factory = _METRIC_FACTORIES.get(slug)
|
|
151
|
+
if factory is None:
|
|
152
|
+
raise LLMError(
|
|
153
|
+
f"unknown DeepEval metric {item!r}; expected one of "
|
|
154
|
+
f"{sorted(_METRIC_FACTORIES)}"
|
|
155
|
+
)
|
|
156
|
+
# A metric validates its judge model at CONSTRUCTION (deepeval
|
|
157
|
+
# raises its own DeepEvalError if no API key is configured). Catch
|
|
158
|
+
# ANY construction failure and re-raise as LLMError so the CLI /
|
|
159
|
+
# SuiteRunner see the same error type as the native Judge (E9).
|
|
160
|
+
try:
|
|
161
|
+
metric = factory(model=self._model, threshold=self._threshold)
|
|
162
|
+
except Exception as exc: # noqa: BLE001 — deepeval's DeepEvalError
|
|
163
|
+
raise LLMError(
|
|
164
|
+
f"could not build DeepEval metric {item!r}: {exc}"
|
|
165
|
+
) from exc
|
|
166
|
+
built.append((slug, metric))
|
|
167
|
+
else:
|
|
168
|
+
built.append((_metric_slug(type(item).__name__), item))
|
|
169
|
+
return built
|
|
170
|
+
|
|
171
|
+
def _llm_test_case(self, trace: m.Trace, task: Task):
|
|
172
|
+
"""Project an ArchForge trace into a DeepEval `LLMTestCase`.
|
|
173
|
+
|
|
174
|
+
`actual_output` is the run's final answer (the traced final_output, else
|
|
175
|
+
the last step's response). Grounding comes from each step's prompt_in
|
|
176
|
+
and is set on BOTH `context` and `retrieval_context`: Faithfulness
|
|
177
|
+
REQUIRES `retrieval_context` (its `_required_params`) and errors without
|
|
178
|
+
it, while RAG-style metrics read `context`. `expected_output` is carried
|
|
179
|
+
through when the task JSON declares one (Task allows extra fields).
|
|
180
|
+
"""
|
|
181
|
+
actual_output = trace.final_output or (
|
|
182
|
+
trace.steps[-1].response_out if trace.steps else ""
|
|
183
|
+
)
|
|
184
|
+
expected_output = getattr(task, "expected_output", None)
|
|
185
|
+
grounding = [s.prompt_in for s in trace.steps]
|
|
186
|
+
kwargs: dict[str, Any] = {
|
|
187
|
+
"input": task.input,
|
|
188
|
+
"actual_output": actual_output,
|
|
189
|
+
"context": grounding,
|
|
190
|
+
"retrieval_context": grounding, # Faithfulness requires this param
|
|
191
|
+
}
|
|
192
|
+
if expected_output:
|
|
193
|
+
kwargs["expected_output"] = expected_output
|
|
194
|
+
return self._deepeval.test_case.LLMTestCase(**kwargs)
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def _import_deepeval():
|
|
198
|
+
"""Lazily import the `deepeval` package, or raise a clear LLMError."""
|
|
199
|
+
try:
|
|
200
|
+
import deepeval
|
|
201
|
+
except ImportError as exc: # pragma: no cover - exercised via monkeypatch
|
|
202
|
+
raise LLMError(
|
|
203
|
+
"DeepEval is not installed; install it with "
|
|
204
|
+
"`pip install archforge-optimizer[deepeval]` to use --evaluator deepeval"
|
|
205
|
+
) from exc
|
|
206
|
+
return deepeval
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def _mean(values: Sequence[float]) -> float:
|
|
210
|
+
return sum(values) / len(values) if values else 0.0
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
# Register the metrics ArchForge exposes by name. Factories construct the metric
|
|
214
|
+
# objects lazily (deepeval classes are imported inside each factory, not here, so
|
|
215
|
+
# this module stays importable before the extra is installed).
|
|
216
|
+
def _answer_relevancy_factory(*, model, threshold):
|
|
217
|
+
from deepeval.metrics import AnswerRelevancyMetric
|
|
218
|
+
|
|
219
|
+
return AnswerRelevancyMetric(threshold=threshold, model=model)
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _faithfulness_factory(*, model, threshold):
|
|
223
|
+
from deepeval.metrics import FaithfulnessMetric
|
|
224
|
+
|
|
225
|
+
return FaithfulnessMetric(threshold=threshold, model=model)
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
_METRIC_FACTORIES.update(
|
|
229
|
+
{
|
|
230
|
+
"answer_relevancy": _answer_relevancy_factory,
|
|
231
|
+
"faithfulness": _faithfulness_factory,
|
|
232
|
+
}
|
|
233
|
+
)
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
__all__ = ["DeepEvalEvaluator"]
|
|
@@ -48,6 +48,22 @@ _PROVIDER_PREFIX: dict[str, str] = {
|
|
|
48
48
|
}
|
|
49
49
|
|
|
50
50
|
|
|
51
|
+
def prefix_model(provider: str, model: str) -> str:
|
|
52
|
+
"""Prefix a model id with the provider (``openai/gpt-4o``).
|
|
53
|
+
|
|
54
|
+
A model that ALREADY carries THIS provider's prefix (``openai/gpt-4o``
|
|
55
|
+
under provider=openai) is passed through untouched, so fully-qualified
|
|
56
|
+
overrides still work. But a ``/`` alone does NOT mean "already routed" —
|
|
57
|
+
e.g. Groq's model ``openai/gpt-oss-120b`` has a slash in its native id,
|
|
58
|
+
so it must still get the provider prefix (``groq/openai/gpt-oss-120b``)
|
|
59
|
+
or LiteLLM would route it to OpenAI. Prefix unless it already matches.
|
|
60
|
+
"""
|
|
61
|
+
prefix = _PROVIDER_PREFIX.get(provider, provider)
|
|
62
|
+
if model.startswith(f"{prefix}/"):
|
|
63
|
+
return model
|
|
64
|
+
return f"{prefix}/{model}"
|
|
65
|
+
|
|
66
|
+
|
|
51
67
|
class LiteLLMClient:
|
|
52
68
|
"""An `LLMClient` backed by `litellm.completion` for one provider.
|
|
53
69
|
|
|
@@ -65,19 +81,8 @@ class LiteLLMClient:
|
|
|
65
81
|
self._base_url = base_url
|
|
66
82
|
|
|
67
83
|
def _prefixed(self, model: str) -> str:
|
|
68
|
-
"""Prefix a model id with
|
|
69
|
-
|
|
70
|
-
A model that ALREADY carries THIS provider's prefix (``openai/gpt-4o``
|
|
71
|
-
under provider=openai) is passed through untouched, so fully-qualified
|
|
72
|
-
overrides still work. But a ``/`` alone does NOT mean "already routed" —
|
|
73
|
-
e.g. Groq's model ``openai/gpt-oss-120b`` has a slash in its native id,
|
|
74
|
-
so it must still get the provider prefix (``groq/openai/gpt-oss-120b``)
|
|
75
|
-
or LiteLLM would route it to OpenAI. Prefix unless it already matches.
|
|
76
|
-
"""
|
|
77
|
-
prefix = _PROVIDER_PREFIX[self._provider]
|
|
78
|
-
if model.startswith(f"{prefix}/"):
|
|
79
|
-
return model
|
|
80
|
-
return f"{prefix}/{model}"
|
|
84
|
+
"""Prefix a model id with this client's provider (see `prefix_model`)."""
|
|
85
|
+
return prefix_model(self._provider, model)
|
|
81
86
|
|
|
82
87
|
def _default_model(self) -> str:
|
|
83
88
|
"""The provider's default model id for a bare complete() call, resolved
|
|
@@ -138,4 +143,4 @@ class LiteLLMClient:
|
|
|
138
143
|
)
|
|
139
144
|
|
|
140
145
|
|
|
141
|
-
__all__ = ["LiteLLMClient"]
|
|
146
|
+
__all__ = ["LiteLLMClient", "prefix_model"]
|
|
@@ -39,8 +39,7 @@ _DISCOVERY_DIR: Path = Path(".archforge")
|
|
|
39
39
|
_DISCOVERY_FILE: Path = _DISCOVERY_DIR / "archforge.py"
|
|
40
40
|
|
|
41
41
|
_MAIN_MSG = (
|
|
42
|
-
"ArchForge is not initialized. Run `archforge-optimizer init
|
|
43
|
-
".archforge/archforge.py (the project's tunable config), then the CLI will work."
|
|
42
|
+
"ArchForge is not initialized. Run `archforge-optimizer init`."
|
|
44
43
|
)
|
|
45
44
|
|
|
46
45
|
|
|
@@ -1,20 +0,0 @@
|
|
|
1
|
-
"""The Judge — LLM-as-judge scoring of runs (spec §3, §4).
|
|
2
|
-
|
|
3
|
-
`Judge.score(trace, task, rubric_id) -> RunScore` turns a run's trace into a
|
|
4
|
-
scored verdict: an aggregate + named rubric dimensions + confidence, plus a
|
|
5
|
-
*per-step breakdown* (StepScore[]) that names which agent lost which points. The
|
|
6
|
-
per-step breakdown is the raw material the Architect uses to credit-assign a
|
|
7
|
-
fault to a node/route.
|
|
8
|
-
|
|
9
|
-
`score_suite(...)` aggregates over R repeats: mean (down-weighted by
|
|
10
|
-
confidence) and a stable aggregate used by the Gatekeeper. `rubric_id` is
|
|
11
|
-
stamped on every score so cross-rubric comparisons never masquerade as
|
|
12
|
-
improvement (invariant I5, spec E2).
|
|
13
|
-
"""
|
|
14
|
-
|
|
15
|
-
from __future__ import annotations
|
|
16
|
-
|
|
17
|
-
from archforge.judge.base import Judge, SuiteAggregate, default_rubric
|
|
18
|
-
from archforge.judge.scripted import ScriptedJudge
|
|
19
|
-
|
|
20
|
-
__all__ = ["Judge", "ScriptedJudge", "SuiteAggregate", "default_rubric"]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{archforge_optimizer-0.3.0 → archforge_optimizer-0.4.0}/archforge/host/adapters/langgraph.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|