archforge-optimizer 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- archforge/__init__.py +76 -0
- archforge/__main__.py +10 -0
- archforge/architect.py +442 -0
- archforge/cli.py +881 -0
- archforge/config.py +140 -0
- archforge/config_init.py +150 -0
- archforge/diff.py +206 -0
- archforge/engine.py +444 -0
- archforge/gatekeeper.py +290 -0
- archforge/host/__init__.py +20 -0
- archforge/host/adapters/__init__.py +41 -0
- archforge/host/adapters/base.py +311 -0
- archforge/host/adapters/helpers.py +163 -0
- archforge/host/adapters/langgraph.py +726 -0
- archforge/host/base.py +105 -0
- archforge/host/fake.py +380 -0
- archforge/judge/__init__.py +20 -0
- archforge/judge/base.py +257 -0
- archforge/judge/scripted.py +145 -0
- archforge/lint.py +180 -0
- archforge/llm/__init__.py +65 -0
- archforge/llm/_common.py +94 -0
- archforge/llm/anthropic.py +90 -0
- archforge/llm/base.py +90 -0
- archforge/llm/gemini.py +112 -0
- archforge/llm/groq.py +63 -0
- archforge/llm/openai.py +63 -0
- archforge/llm/scripted.py +134 -0
- archforge/middleware.py +181 -0
- archforge/models.py +435 -0
- archforge/mutate.py +214 -0
- archforge/otel.py +613 -0
- archforge/runlog.py +103 -0
- archforge/runner.py +153 -0
- archforge/spec_builder.py +126 -0
- archforge/stores/__init__.py +22 -0
- archforge/stores/_jsonl.py +81 -0
- archforge/stores/attempt_store.py +161 -0
- archforge/stores/spec_store.py +188 -0
- archforge/stores/trace_store.py +42 -0
- archforge/suite.py +248 -0
- archforge/userconfig.py +144 -0
- archforge_optimizer-0.1.0.dist-info/METADATA +420 -0
- archforge_optimizer-0.1.0.dist-info/RECORD +47 -0
- archforge_optimizer-0.1.0.dist-info/WHEEL +4 -0
- archforge_optimizer-0.1.0.dist-info/entry_points.txt +2 -0
- archforge_optimizer-0.1.0.dist-info/licenses/LICENSE +21 -0
archforge/host/base.py
ADDED
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
"""The host-MAS integration contract (spec §3, §4).
|
|
2
|
+
|
|
3
|
+
This module defines the *seam* between ArchForge and an arbitrary multi-agent
|
|
4
|
+
system. A concrete host (LangGraph/CrewAI/AutoGen/custom) implements `HostMAS`;
|
|
5
|
+
ArchForge talks to it only through this protocol, plus the
|
|
6
|
+
`TracingMiddleware` it injects.
|
|
7
|
+
|
|
8
|
+
Why two collaborating roles:
|
|
9
|
+
* **HostMAS** owns *execution*: it knows the framework's agents and how to
|
|
10
|
+
sequence them per the Spec's graph. It is config-agnostic at the agent level.
|
|
11
|
+
* **TracingMiddleware** owns *observation + config*: it records each Step to
|
|
12
|
+
TraceStore and applies the live Spec's (system_prompt, model, knobs, tools)
|
|
13
|
+
to each agent at invoke time — so reconfiguring the pipeline never requires
|
|
14
|
+
rebuilding the host.
|
|
15
|
+
|
|
16
|
+
`Runnable.run(task)` returns a full `Trace` (ok or not). A mid-run agent crash
|
|
17
|
+
surfaces as a `Trace` with `ok=False`, `error` set, and the Steps that *did*
|
|
18
|
+
complete retained (spec E4) — the trace is never lost.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
from typing import TYPE_CHECKING, Protocol, runtime_checkable
|
|
24
|
+
|
|
25
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
26
|
+
|
|
27
|
+
import archforge.models as m
|
|
28
|
+
|
|
29
|
+
if TYPE_CHECKING: # avoid a runtime import cycle (middleware imports host.base)
|
|
30
|
+
from archforge.middleware import TracingMiddleware
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class Task(BaseModel):
|
|
34
|
+
"""A unit of work for the host MAS to execute (one eval-suite item)."""
|
|
35
|
+
|
|
36
|
+
model_config = ConfigDict(extra="allow")
|
|
37
|
+
|
|
38
|
+
task_id: str
|
|
39
|
+
input: str
|
|
40
|
+
# The rubric to score this task against (spec E2/I5 — comparisons stay
|
|
41
|
+
# within a rubric). None means "use whatever the suite defaults to".
|
|
42
|
+
rubric_id: str | None = None
|
|
43
|
+
suite_id: str | None = None
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class AgentResponse(BaseModel):
|
|
47
|
+
"""What a single agent produced: its text, any tool calls, and perf signals."""
|
|
48
|
+
|
|
49
|
+
model_config = ConfigDict(extra="allow")
|
|
50
|
+
|
|
51
|
+
text: str
|
|
52
|
+
tool_calls: list[m.ToolCall] = Field(default_factory=list)
|
|
53
|
+
perf: m.StepPerf = Field(default_factory=m.StepPerf)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
@runtime_checkable
|
|
57
|
+
class Agent(Protocol):
|
|
58
|
+
"""Host-provided implementation of one node's behaviour.
|
|
59
|
+
|
|
60
|
+
Concrete hosts wrap their framework agent here. `invoke` receives the Spec's
|
|
61
|
+
configuration (`system_prompt`, `model`, `knobs`, `tools`, `kind`) so the same
|
|
62
|
+
agent is reconfigured when the spec evolves — without rebuilding the host. The
|
|
63
|
+
middleware is what actually supplies these arguments at call time.
|
|
64
|
+
|
|
65
|
+
`kind` (the node's `NodeKind`) is OPTIONAL and defaults to `LLM` so every
|
|
66
|
+
pre-existing adapter that knows only LLM agents keeps working unchanged; a
|
|
67
|
+
host that dispatches on kind (e.g. the fake kit's rule/retriever/tool agents)
|
|
68
|
+
reads it to choose its behaviour. It carries NO cost on adapters that ignore it.
|
|
69
|
+
"""
|
|
70
|
+
|
|
71
|
+
node_id: str
|
|
72
|
+
role: str
|
|
73
|
+
|
|
74
|
+
def invoke(
|
|
75
|
+
self,
|
|
76
|
+
prompt: str,
|
|
77
|
+
*,
|
|
78
|
+
system_prompt: str,
|
|
79
|
+
model: str,
|
|
80
|
+
knobs: m.Knobs,
|
|
81
|
+
tools: list[str],
|
|
82
|
+
kind: m.NodeKind = m.NodeKind.LLM,
|
|
83
|
+
) -> AgentResponse: ...
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
@runtime_checkable
|
|
87
|
+
class Runnable(Protocol):
|
|
88
|
+
"""A Spec instantiated into an executable, observable pipeline."""
|
|
89
|
+
|
|
90
|
+
def run(self, task: Task) -> m.Trace: ...
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
@runtime_checkable
|
|
94
|
+
class HostMAS(Protocol):
|
|
95
|
+
"""The multi-agent system being optimized.
|
|
96
|
+
|
|
97
|
+
`instantiate` builds the agents for a Spec and wires each through the
|
|
98
|
+
middleware (which records steps + applies config). The returned `Runnable`
|
|
99
|
+
is what the SuiteRunner calls once per task per repeat.
|
|
100
|
+
"""
|
|
101
|
+
|
|
102
|
+
def instantiate(self, spec: m.Spec, middleware: "TracingMiddleware") -> Runnable: ...
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
__all__ = ["Task", "AgentResponse", "Agent", "Runnable", "HostMAS"]
|
archforge/host/fake.py
ADDED
|
@@ -0,0 +1,380 @@
|
|
|
1
|
+
"""FakeHostMAS — a deterministic, scriptable stand-in for a real MAS.
|
|
2
|
+
|
|
3
|
+
Used by the SuiteRunner and every E2E scenario / smoke test so the optimizer
|
|
4
|
+
runs end-to-end **without any real LLM**. It is deliberately simple but honours
|
|
5
|
+
the real seam's contracts:
|
|
6
|
+
|
|
7
|
+
* reads the Spec graph and runs nodes in a topological order (the order the
|
|
8
|
+
graph implies — sequence edges chain, fanout/join/conditional resolve)
|
|
9
|
+
* each `FakeAgent` produces a deterministic response derived from its config
|
|
10
|
+
and the task; it can be scripted to raise on a chosen call (spec E4
|
|
11
|
+
mid-run crash), so a candidate's behaviour is fully predictable
|
|
12
|
+
* per-run perf is populated (tokens/latency/retries) so cost tracking works
|
|
13
|
+
* the resulting `Trace` is assembled by the `TracingMiddleware` exactly as a
|
|
14
|
+
real host would; a crash yields `ok=False` + a partial trace (E4)
|
|
15
|
+
|
|
16
|
+
This is the *only* piece that knows framework execution details; everything
|
|
17
|
+
above it (SuiteRunner, Judge, Architect, Gatekeeper) treats it as a `HostMAS`.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import hashlib
|
|
23
|
+
from collections import defaultdict
|
|
24
|
+
from typing import Callable
|
|
25
|
+
|
|
26
|
+
import archforge.models as m
|
|
27
|
+
from archforge.config import SHORT_HASH_LEN
|
|
28
|
+
from archforge.host.adapters.helpers import run_id as _run_id, topo_order as _topo_order
|
|
29
|
+
from archforge.host.base import Agent, AgentResponse, HostMAS, Runnable, Task
|
|
30
|
+
from archforge.middleware import TracingMiddleware
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
# --------------------------------------------------------------------------- #
|
|
34
|
+
# Fake agents
|
|
35
|
+
# --------------------------------------------------------------------------- #
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class CrashOnCall(Exception):
|
|
39
|
+
"""Scripted failure of a single agent invocation (spec E4)."""
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _default_responder(node: m.Node, prompt: str, _system: str) -> str:
|
|
43
|
+
"""Deterministic text so identical (node, prompt) -> identical output."""
|
|
44
|
+
|
|
45
|
+
h = hashlib.sha256(f"{node.node_id}|{prompt}".encode()).hexdigest()[:SHORT_HASH_LEN]
|
|
46
|
+
return f"[{node.role}:{node.model}:{h}] {prompt}"
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class _FakeBaseAgent:
|
|
50
|
+
"""Shared scaffolding for every fake agent kind (LLM + non-LLM).
|
|
51
|
+
|
|
52
|
+
Owns the two pieces common to all kinds, lifted out of the old `FakeAgent`:
|
|
53
|
+
* the `crash_on` hook so a test can force ANY node — including a non-LLM
|
|
54
|
+
one — to raise on a chosen invocation index (spec E4 mid-run crash);
|
|
55
|
+
* `invoke_count` so deterministic behaviour + the crash index line up.
|
|
56
|
+
|
|
57
|
+
Each kind subclasses and implements ``_respond`` (what text + tool calls the
|
|
58
|
+
node produces). The base then wraps it with kind-aware perf: an ``llm`` node
|
|
59
|
+
costs deterministic pseudo-tokens (``len//4``, as before); a non-llm node
|
|
60
|
+
costs **ZERO tokens** — its real cost is wall-clock latency, which the
|
|
61
|
+
per-cycle wall cap (engine) measures. That tokens=0 convention is what lets a
|
|
62
|
+
tool/retriever/rule-heavy pipeline stay under the token cap yet be bounded by
|
|
63
|
+
the new wall cap (the design's cost fix). Latency is deterministic + nonzero
|
|
64
|
+
so traces stay reproducible (R-repeat aggregation, E1) and the wall cap is
|
|
65
|
+
exercisable.
|
|
66
|
+
"""
|
|
67
|
+
|
|
68
|
+
def __init__(
|
|
69
|
+
self,
|
|
70
|
+
node: m.Node,
|
|
71
|
+
*,
|
|
72
|
+
crash_on: Callable[[int], bool] | None = None,
|
|
73
|
+
) -> None:
|
|
74
|
+
self._node = node
|
|
75
|
+
self._crash_on = crash_on
|
|
76
|
+
self.invoke_count = 0
|
|
77
|
+
|
|
78
|
+
@property
|
|
79
|
+
def node_id(self) -> str:
|
|
80
|
+
return self._node.node_id
|
|
81
|
+
|
|
82
|
+
@property
|
|
83
|
+
def role(self) -> str:
|
|
84
|
+
return self._node.role
|
|
85
|
+
|
|
86
|
+
def _respond(
|
|
87
|
+
self,
|
|
88
|
+
prompt: str,
|
|
89
|
+
system_prompt: str,
|
|
90
|
+
model: str,
|
|
91
|
+
knobs: m.Knobs,
|
|
92
|
+
tools: list[str],
|
|
93
|
+
kind: m.NodeKind,
|
|
94
|
+
) -> tuple[str, list[m.ToolCall]]:
|
|
95
|
+
raise NotImplementedError
|
|
96
|
+
|
|
97
|
+
def invoke(
|
|
98
|
+
self,
|
|
99
|
+
prompt: str,
|
|
100
|
+
*,
|
|
101
|
+
system_prompt: str,
|
|
102
|
+
model: str,
|
|
103
|
+
knobs: m.Knobs,
|
|
104
|
+
tools: list[str],
|
|
105
|
+
kind: m.NodeKind = m.NodeKind.LLM,
|
|
106
|
+
) -> AgentResponse:
|
|
107
|
+
if self._crash_on is not None and self._crash_on(self.invoke_count):
|
|
108
|
+
self.invoke_count += 1
|
|
109
|
+
raise CrashOnCall(self.node_id)
|
|
110
|
+
self.invoke_count += 1
|
|
111
|
+
|
|
112
|
+
out, tool_calls = self._respond(prompt, system_prompt, model, knobs, tools, kind)
|
|
113
|
+
# LLM-shaped tokens (len//4) for cost+dedup; non-LLM kinds cost time, not
|
|
114
|
+
# tokens — perf.tokens=0 is the cost-cap convention (see class docstring).
|
|
115
|
+
token_count = max(1, len(out) // 4) if kind is m.NodeKind.LLM else 0
|
|
116
|
+
latency = float((token_count % 7) + 1) # deterministic, always nonzero
|
|
117
|
+
return AgentResponse(
|
|
118
|
+
text=out,
|
|
119
|
+
tool_calls=tool_calls,
|
|
120
|
+
perf=m.StepPerf(tokens=token_count, latency_ms=latency,
|
|
121
|
+
retries=0, error=None),
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
class FakeAgent(_FakeBaseAgent):
|
|
126
|
+
"""An LLM node's executor: deterministic, optionally scripted to fail.
|
|
127
|
+
|
|
128
|
+
The historical fake (now the `kind=llm` arm). `invoke` returns a deterministic
|
|
129
|
+
response built from the node config + the incoming prompt, so the same
|
|
130
|
+
(spec, task) always yields the same output (reproducible runs for R-repeat
|
|
131
|
+
aggregation). An optional `crash_on` forces a raise on a chosen invocation
|
|
132
|
+
index. Also used, harmless, for `symbolic` nodes — a `symbolic` node is a
|
|
133
|
+
deterministic transform whose cost is latency not tokens, which the base
|
|
134
|
+
enforces via its kind-aware perf (tokens=0 for non-llm).
|
|
135
|
+
"""
|
|
136
|
+
|
|
137
|
+
def __init__(
|
|
138
|
+
self,
|
|
139
|
+
node: m.Node,
|
|
140
|
+
*,
|
|
141
|
+
crash_on: Callable[[int], bool] | None = None,
|
|
142
|
+
responder: Callable[[m.Node, str, str], str] | None = None,
|
|
143
|
+
) -> None:
|
|
144
|
+
super().__init__(node, crash_on=crash_on)
|
|
145
|
+
self._responder = responder or _default_responder
|
|
146
|
+
|
|
147
|
+
def _respond(
|
|
148
|
+
self,
|
|
149
|
+
prompt: str,
|
|
150
|
+
system_prompt: str,
|
|
151
|
+
model: str,
|
|
152
|
+
knobs: m.Knobs,
|
|
153
|
+
tools: list[str],
|
|
154
|
+
kind: m.NodeKind,
|
|
155
|
+
) -> tuple[str, list[m.ToolCall]]:
|
|
156
|
+
return self._responder(self._node, prompt, system_prompt), []
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
class FakeRuleAgent(_FakeBaseAgent):
|
|
160
|
+
"""A `rule` node (heuristic/scorer/classifier threshold — no prompt, no model).
|
|
161
|
+
|
|
162
|
+
Produces a deterministic label that surfaces its `threshold` knob (the
|
|
163
|
+
`tunable` param the LLM Architect edits on a rule node). Zero tokens.
|
|
164
|
+
"""
|
|
165
|
+
|
|
166
|
+
def _respond(
|
|
167
|
+
self,
|
|
168
|
+
prompt: str,
|
|
169
|
+
system_prompt: str,
|
|
170
|
+
model: str,
|
|
171
|
+
knobs: m.Knobs,
|
|
172
|
+
tools: list[str],
|
|
173
|
+
kind: m.NodeKind,
|
|
174
|
+
) -> tuple[str, list[m.ToolCall]]:
|
|
175
|
+
threshold = getattr(knobs, "threshold", None)
|
|
176
|
+
h = hashlib.sha256(f"{self._node.node_id}|{prompt}".encode()).hexdigest()[:SHORT_HASH_LEN]
|
|
177
|
+
return f"[rule:{self._node.role}:thr={threshold}:{h}] {prompt}", []
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
class FakeRetrieverAgent(_FakeBaseAgent):
|
|
181
|
+
"""A `retriever` node — fetches search/vector context, parameterized by `top_k`.
|
|
182
|
+
|
|
183
|
+
Emits `top_k` deterministic context chunks (default 3 when `knobs.top_k` is
|
|
184
|
+
unset), so an Architect `knob` edit raising `top_k` visibly changes the step's
|
|
185
|
+
output (a different SuiteRun). Zero tokens.
|
|
186
|
+
"""
|
|
187
|
+
|
|
188
|
+
def _respond(
|
|
189
|
+
self,
|
|
190
|
+
prompt: str,
|
|
191
|
+
system_prompt: str,
|
|
192
|
+
model: str,
|
|
193
|
+
knobs: m.Knobs,
|
|
194
|
+
tools: list[str],
|
|
195
|
+
kind: m.NodeKind,
|
|
196
|
+
) -> tuple[str, list[m.ToolCall]]:
|
|
197
|
+
top_k = getattr(knobs, "top_k", None) or 3
|
|
198
|
+
h = hashlib.sha256(f"{self._node.node_id}|{prompt}".encode()).hexdigest()[:SHORT_HASH_LEN]
|
|
199
|
+
chunks = [f"chunk-{i}:{h}" for i in range(int(top_k))]
|
|
200
|
+
return f"[retriever:{self._node.role}:top_k={int(top_k)}] " + "; ".join(chunks), []
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
class FakeToolAgent(_FakeBaseAgent):
|
|
204
|
+
"""A `tool` node — an external call recorded as a `ToolCall`.
|
|
205
|
+
|
|
206
|
+
Produces a deterministic serialized result AND a `ToolCall(tool_id=node_id)`
|
|
207
|
+
so a tool node's run is distinguishable from an LLM's (the trace carries the
|
|
208
|
+
call). Zero tokens.
|
|
209
|
+
"""
|
|
210
|
+
|
|
211
|
+
def _respond(
|
|
212
|
+
self,
|
|
213
|
+
prompt: str,
|
|
214
|
+
system_prompt: str,
|
|
215
|
+
model: str,
|
|
216
|
+
knobs: m.Knobs,
|
|
217
|
+
tools: list[str],
|
|
218
|
+
kind: m.NodeKind,
|
|
219
|
+
) -> tuple[str, list[m.ToolCall]]:
|
|
220
|
+
h = hashlib.sha256(f"{self._node.node_id}|{prompt}".encode()).hexdigest()[:SHORT_HASH_LEN]
|
|
221
|
+
result = f"[tool:{self._node.role}:{h}] {prompt}"
|
|
222
|
+
call = m.ToolCall(tool_id=self._node.node_id, args={"query": prompt},
|
|
223
|
+
result=result, ok=True)
|
|
224
|
+
return result, [call]
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
# --------------------------------------------------------------------------- #
|
|
228
|
+
# Runnable pipeline over the Spec graph
|
|
229
|
+
# --------------------------------------------------------------------------- #
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
class FakePipeline:
|
|
233
|
+
"""A Spec rendered as an executable pipeline.
|
|
234
|
+
|
|
235
|
+
Execution model (a deliberate simplification that covers the spec's edge
|
|
236
|
+
kinds for exercising control flow):
|
|
237
|
+
* nodes run in a topological order derived from the edges
|
|
238
|
+
* the first node consumes the task input as its prompt; every later node
|
|
239
|
+
consumes the previous node's response as its prompt (sequence).
|
|
240
|
+
* a `conditional` edge whose `gate` equals the string form of the prior
|
|
241
|
+
step's response is taken; otherwise the next non-conditional edge is
|
|
242
|
+
followed. (Real hosts will have richer semantics; we only need enough to
|
|
243
|
+
run + trace.)
|
|
244
|
+
* the last node's response is the run's `final_output`.
|
|
245
|
+
"""
|
|
246
|
+
|
|
247
|
+
def __init__(
|
|
248
|
+
self,
|
|
249
|
+
spec: m.Spec,
|
|
250
|
+
middleware: TracingMiddleware,
|
|
251
|
+
agents: dict[str, _FakeBaseAgent],
|
|
252
|
+
) -> None:
|
|
253
|
+
self._spec = spec
|
|
254
|
+
self._mw = middleware
|
|
255
|
+
self._agents = agents
|
|
256
|
+
self._edges_by_src: dict[str, list[m.Edge]] = defaultdict(list)
|
|
257
|
+
for e in spec.edges:
|
|
258
|
+
self._edges_by_src[e.from_].append(e)
|
|
259
|
+
# monotonic counter so each run() mints a UNIQUE run_id, even at R>1
|
|
260
|
+
# (a spec+task no longer collides across repeats). See spec §7 R-repeats.
|
|
261
|
+
self._run_counter = 0
|
|
262
|
+
|
|
263
|
+
def run(self, task: Task) -> m.Trace:
|
|
264
|
+
sid = self._spec.compute_spec_id()
|
|
265
|
+
run_id = _run_id(sid, task.task_id, self._run_counter)
|
|
266
|
+
self._run_counter += 1
|
|
267
|
+
self._mw.begin_run(run_id, self._spec, task.task_id)
|
|
268
|
+
last_text: str | None = None
|
|
269
|
+
final_output: str | None = None
|
|
270
|
+
try:
|
|
271
|
+
order = _topo_order(self._spec)
|
|
272
|
+
if not order:
|
|
273
|
+
# nodeless spec -> nothing to trace; trivial success, empty trace
|
|
274
|
+
return self._mw.end_run(None, ok=True, error=None)
|
|
275
|
+
current_id = order[0]
|
|
276
|
+
prompt = task.input
|
|
277
|
+
visited: set[str] = set()
|
|
278
|
+
while current_id is not None:
|
|
279
|
+
if current_id in visited:
|
|
280
|
+
break # defensive against cycles the linter should already catch
|
|
281
|
+
visited.add(current_id)
|
|
282
|
+
agent = self._agents[current_id]
|
|
283
|
+
wrapped = self._mw.wrap(agent)
|
|
284
|
+
response = wrapped.invoke(prompt)
|
|
285
|
+
last_text = response.text
|
|
286
|
+
final_output = response.text
|
|
287
|
+
current_id, prompt = self._next_node(current_id, last_text)
|
|
288
|
+
trace = self._mw.end_run(final_output, ok=True, error=None)
|
|
289
|
+
return trace
|
|
290
|
+
except CrashOnCall as exc:
|
|
291
|
+
# A scripted mid-run crash — flush partial trace with ok=False (E4)
|
|
292
|
+
trace = self._mw.end_run(last_text, ok=False, error=f"crash in {exc}")
|
|
293
|
+
return trace
|
|
294
|
+
except Exception as exc: # noqa: BLE001 — host errors also flush partial trace
|
|
295
|
+
trace = self._mw.end_run(last_text, ok=False, error=repr(exc))
|
|
296
|
+
return trace
|
|
297
|
+
|
|
298
|
+
def _next_node(self, current_id: str, response: str) -> tuple[str | None, str]:
|
|
299
|
+
"""Pick the next node given the outgoing edges of `current_id`.
|
|
300
|
+
|
|
301
|
+
Returns (next_node_id_or_None, next_prompt). A `conditional` edge is
|
|
302
|
+
taken iff its `gate` equals the current response; otherwise the first
|
|
303
|
+
non-conditional out-edge is followed. If no out-edge, we stop.
|
|
304
|
+
"""
|
|
305
|
+
|
|
306
|
+
outs = self._edges_by_src.get(current_id, [])
|
|
307
|
+
if not outs:
|
|
308
|
+
return None, response
|
|
309
|
+
for edge in outs:
|
|
310
|
+
if edge.type is m.EdgeType.CONDITIONAL:
|
|
311
|
+
if edge.gate is not None and response.strip() == edge.gate:
|
|
312
|
+
return edge.to, response
|
|
313
|
+
else:
|
|
314
|
+
return edge.to, response
|
|
315
|
+
# only conditional edges, none taken -> stop
|
|
316
|
+
return None, response
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
# --------------------------------------------------------------------------- #
|
|
320
|
+
# The host
|
|
321
|
+
# --------------------------------------------------------------------------- #
|
|
322
|
+
|
|
323
|
+
|
|
324
|
+
class FakeHostMAS:
|
|
325
|
+
"""A `HostMAS` that builds a `FakePipeline` from a Spec + node scripts.
|
|
326
|
+
|
|
327
|
+
Dispatches on each node's `kind` (the non-LLM extension) to pick the matching
|
|
328
|
+
fake agent — `llm`/`symbolic` -> `FakeAgent`, `rule` -> `FakeRuleAgent`,
|
|
329
|
+
`retriever` -> `FakeRetrieverAgent`, `tool` -> `FakeToolAgent` — so a mixed
|
|
330
|
+
pipeline (retriever + llm + tool) runs end-to-end **zero-LLM**. An explicit
|
|
331
|
+
`node_scripts[...]` override (e.g. `crash_on`) still applies to whatever kind
|
|
332
|
+
the node is, so E4 mid-run crashes script on non-LLM nodes too.
|
|
333
|
+
|
|
334
|
+
`responder` overrides the LLM-arm responder only (`_default_responder`
|
|
335
|
+
otherwise); it is ignored by the non-LLM fakes (they own their deterministic
|
|
336
|
+
outputs). `node_scripts` lets tests inject per-node behaviour keyed by
|
|
337
|
+
node_id; agents are built per `instantiate` so each candidate Spec gets fresh
|
|
338
|
+
ones.
|
|
339
|
+
"""
|
|
340
|
+
|
|
341
|
+
def __init__(
|
|
342
|
+
self,
|
|
343
|
+
node_scripts: dict[str, dict] | None = None,
|
|
344
|
+
responder: Callable[[m.Node, str, str], str] | None = None,
|
|
345
|
+
) -> None:
|
|
346
|
+
self._node_scripts = node_scripts or {}
|
|
347
|
+
self._responder = responder
|
|
348
|
+
|
|
349
|
+
def _agent_for(
|
|
350
|
+
self, node: m.Node, crash_on: Callable[[int], bool] | None
|
|
351
|
+
) -> _FakeBaseAgent:
|
|
352
|
+
kind = node.kind
|
|
353
|
+
if kind is m.NodeKind.RULE:
|
|
354
|
+
return FakeRuleAgent(node, crash_on=crash_on)
|
|
355
|
+
if kind is m.NodeKind.RETRIEVER:
|
|
356
|
+
return FakeRetrieverAgent(node, crash_on=crash_on)
|
|
357
|
+
if kind is m.NodeKind.TOOL:
|
|
358
|
+
return FakeToolAgent(node, crash_on=crash_on)
|
|
359
|
+
# llm + symbolic share the generic (deterministic) agent; symbolic costs
|
|
360
|
+
# latency not tokens, enforced by the base's kind-aware perf.
|
|
361
|
+
return FakeAgent(node, crash_on=crash_on, responder=self._responder)
|
|
362
|
+
|
|
363
|
+
def instantiate(self, spec: m.Spec, middleware: TracingMiddleware) -> Runnable:
|
|
364
|
+
agents: dict[str, _FakeBaseAgent] = {}
|
|
365
|
+
for node in spec.nodes:
|
|
366
|
+
script = dict(self._node_scripts.get(node.node_id, {}))
|
|
367
|
+
crash_on = script.pop("crash_on", None) if isinstance(script, dict) else None
|
|
368
|
+
agents[node.node_id] = self._agent_for(node, crash_on)
|
|
369
|
+
return FakePipeline(spec, middleware, agents)
|
|
370
|
+
|
|
371
|
+
|
|
372
|
+
# ``topo_order`` and ``run_id`` now live in ``archforge.host.adapters.helpers``
|
|
373
|
+
# (single source for the kit + this module). They are re-imported at the top as
|
|
374
|
+
# ``_topo_order`` / ``_run_id`` so the call sites below are unchanged.
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
__all__ = [
|
|
378
|
+
"FakeHostMAS", "FakeAgent", "FakeRuleAgent", "FakeRetrieverAgent",
|
|
379
|
+
"FakeToolAgent", "FakePipeline", "CrashOnCall",
|
|
380
|
+
]
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
"""The Judge — LLM-as-judge scoring of runs (spec §3, §4).
|
|
2
|
+
|
|
3
|
+
`Judge.score(trace, task, rubric_id) -> RunScore` turns a run's trace into a
|
|
4
|
+
scored verdict: an aggregate + named rubric dimensions + confidence, plus a
|
|
5
|
+
*per-step breakdown* (StepScore[]) that names which agent lost which points. The
|
|
6
|
+
per-step breakdown is the raw material the Architect uses to credit-assign a
|
|
7
|
+
fault to a node/route.
|
|
8
|
+
|
|
9
|
+
`score_suite(...)` aggregates over R repeats: mean (down-weighted by
|
|
10
|
+
confidence) and a stable aggregate used by the Gatekeeper. `rubric_id` is
|
|
11
|
+
stamped on every score so cross-rubric comparisons never masquerade as
|
|
12
|
+
improvement (invariant I5, spec E2).
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from archforge.judge.base import Judge, SuiteAggregate, default_rubric
|
|
18
|
+
from archforge.judge.scripted import ScriptedJudge
|
|
19
|
+
|
|
20
|
+
__all__ = ["Judge", "ScriptedJudge", "SuiteAggregate", "default_rubric"]
|