archforge-optimizer 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. archforge/__init__.py +76 -0
  2. archforge/__main__.py +10 -0
  3. archforge/architect.py +442 -0
  4. archforge/cli.py +881 -0
  5. archforge/config.py +140 -0
  6. archforge/config_init.py +150 -0
  7. archforge/diff.py +206 -0
  8. archforge/engine.py +444 -0
  9. archforge/gatekeeper.py +290 -0
  10. archforge/host/__init__.py +20 -0
  11. archforge/host/adapters/__init__.py +41 -0
  12. archforge/host/adapters/base.py +311 -0
  13. archforge/host/adapters/helpers.py +163 -0
  14. archforge/host/adapters/langgraph.py +726 -0
  15. archforge/host/base.py +105 -0
  16. archforge/host/fake.py +380 -0
  17. archforge/judge/__init__.py +20 -0
  18. archforge/judge/base.py +257 -0
  19. archforge/judge/scripted.py +145 -0
  20. archforge/lint.py +180 -0
  21. archforge/llm/__init__.py +65 -0
  22. archforge/llm/_common.py +94 -0
  23. archforge/llm/anthropic.py +90 -0
  24. archforge/llm/base.py +90 -0
  25. archforge/llm/gemini.py +112 -0
  26. archforge/llm/groq.py +63 -0
  27. archforge/llm/openai.py +63 -0
  28. archforge/llm/scripted.py +134 -0
  29. archforge/middleware.py +181 -0
  30. archforge/models.py +435 -0
  31. archforge/mutate.py +214 -0
  32. archforge/otel.py +613 -0
  33. archforge/runlog.py +103 -0
  34. archforge/runner.py +153 -0
  35. archforge/spec_builder.py +126 -0
  36. archforge/stores/__init__.py +22 -0
  37. archforge/stores/_jsonl.py +81 -0
  38. archforge/stores/attempt_store.py +161 -0
  39. archforge/stores/spec_store.py +188 -0
  40. archforge/stores/trace_store.py +42 -0
  41. archforge/suite.py +248 -0
  42. archforge/userconfig.py +144 -0
  43. archforge_optimizer-0.1.0.dist-info/METADATA +420 -0
  44. archforge_optimizer-0.1.0.dist-info/RECORD +47 -0
  45. archforge_optimizer-0.1.0.dist-info/WHEEL +4 -0
  46. archforge_optimizer-0.1.0.dist-info/entry_points.txt +2 -0
  47. archforge_optimizer-0.1.0.dist-info/licenses/LICENSE +21 -0
archforge/host/base.py ADDED
@@ -0,0 +1,105 @@
1
+ """The host-MAS integration contract (spec §3, §4).
2
+
3
+ This module defines the *seam* between ArchForge and an arbitrary multi-agent
4
+ system. A concrete host (LangGraph/CrewAI/AutoGen/custom) implements `HostMAS`;
5
+ ArchForge talks to it only through this protocol, plus the
6
+ `TracingMiddleware` it injects.
7
+
8
+ Why two collaborating roles:
9
+ * **HostMAS** owns *execution*: it knows the framework's agents and how to
10
+ sequence them per the Spec's graph. It is config-agnostic at the agent level.
11
+ * **TracingMiddleware** owns *observation + config*: it records each Step to
12
+ TraceStore and applies the live Spec's (system_prompt, model, knobs, tools)
13
+ to each agent at invoke time — so reconfiguring the pipeline never requires
14
+ rebuilding the host.
15
+
16
+ `Runnable.run(task)` returns a full `Trace` (ok or not). A mid-run agent crash
17
+ surfaces as a `Trace` with `ok=False`, `error` set, and the Steps that *did*
18
+ complete retained (spec E4) — the trace is never lost.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ from typing import TYPE_CHECKING, Protocol, runtime_checkable
24
+
25
+ from pydantic import BaseModel, ConfigDict, Field
26
+
27
+ import archforge.models as m
28
+
29
+ if TYPE_CHECKING: # avoid a runtime import cycle (middleware imports host.base)
30
+ from archforge.middleware import TracingMiddleware
31
+
32
+
33
+ class Task(BaseModel):
34
+ """A unit of work for the host MAS to execute (one eval-suite item)."""
35
+
36
+ model_config = ConfigDict(extra="allow")
37
+
38
+ task_id: str
39
+ input: str
40
+ # The rubric to score this task against (spec E2/I5 — comparisons stay
41
+ # within a rubric). None means "use whatever the suite defaults to".
42
+ rubric_id: str | None = None
43
+ suite_id: str | None = None
44
+
45
+
46
+ class AgentResponse(BaseModel):
47
+ """What a single agent produced: its text, any tool calls, and perf signals."""
48
+
49
+ model_config = ConfigDict(extra="allow")
50
+
51
+ text: str
52
+ tool_calls: list[m.ToolCall] = Field(default_factory=list)
53
+ perf: m.StepPerf = Field(default_factory=m.StepPerf)
54
+
55
+
56
+ @runtime_checkable
57
+ class Agent(Protocol):
58
+ """Host-provided implementation of one node's behaviour.
59
+
60
+ Concrete hosts wrap their framework agent here. `invoke` receives the Spec's
61
+ configuration (`system_prompt`, `model`, `knobs`, `tools`, `kind`) so the same
62
+ agent is reconfigured when the spec evolves — without rebuilding the host. The
63
+ middleware is what actually supplies these arguments at call time.
64
+
65
+ `kind` (the node's `NodeKind`) is OPTIONAL and defaults to `LLM` so every
66
+ pre-existing adapter that knows only LLM agents keeps working unchanged; a
67
+ host that dispatches on kind (e.g. the fake kit's rule/retriever/tool agents)
68
+ reads it to choose its behaviour. It carries NO cost on adapters that ignore it.
69
+ """
70
+
71
+ node_id: str
72
+ role: str
73
+
74
+ def invoke(
75
+ self,
76
+ prompt: str,
77
+ *,
78
+ system_prompt: str,
79
+ model: str,
80
+ knobs: m.Knobs,
81
+ tools: list[str],
82
+ kind: m.NodeKind = m.NodeKind.LLM,
83
+ ) -> AgentResponse: ...
84
+
85
+
86
+ @runtime_checkable
87
+ class Runnable(Protocol):
88
+ """A Spec instantiated into an executable, observable pipeline."""
89
+
90
+ def run(self, task: Task) -> m.Trace: ...
91
+
92
+
93
+ @runtime_checkable
94
+ class HostMAS(Protocol):
95
+ """The multi-agent system being optimized.
96
+
97
+ `instantiate` builds the agents for a Spec and wires each through the
98
+ middleware (which records steps + applies config). The returned `Runnable`
99
+ is what the SuiteRunner calls once per task per repeat.
100
+ """
101
+
102
+ def instantiate(self, spec: m.Spec, middleware: "TracingMiddleware") -> Runnable: ...
103
+
104
+
105
+ __all__ = ["Task", "AgentResponse", "Agent", "Runnable", "HostMAS"]
archforge/host/fake.py ADDED
@@ -0,0 +1,380 @@
1
+ """FakeHostMAS — a deterministic, scriptable stand-in for a real MAS.
2
+
3
+ Used by the SuiteRunner and every E2E scenario / smoke test so the optimizer
4
+ runs end-to-end **without any real LLM**. It is deliberately simple but honours
5
+ the real seam's contracts:
6
+
7
+ * reads the Spec graph and runs nodes in a topological order (the order the
8
+ graph implies — sequence edges chain, fanout/join/conditional resolve)
9
+ * each `FakeAgent` produces a deterministic response derived from its config
10
+ and the task; it can be scripted to raise on a chosen call (spec E4
11
+ mid-run crash), so a candidate's behaviour is fully predictable
12
+ * per-run perf is populated (tokens/latency/retries) so cost tracking works
13
+ * the resulting `Trace` is assembled by the `TracingMiddleware` exactly as a
14
+ real host would; a crash yields `ok=False` + a partial trace (E4)
15
+
16
+ This is the *only* piece that knows framework execution details; everything
17
+ above it (SuiteRunner, Judge, Architect, Gatekeeper) treats it as a `HostMAS`.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import hashlib
23
+ from collections import defaultdict
24
+ from typing import Callable
25
+
26
+ import archforge.models as m
27
+ from archforge.config import SHORT_HASH_LEN
28
+ from archforge.host.adapters.helpers import run_id as _run_id, topo_order as _topo_order
29
+ from archforge.host.base import Agent, AgentResponse, HostMAS, Runnable, Task
30
+ from archforge.middleware import TracingMiddleware
31
+
32
+
33
+ # --------------------------------------------------------------------------- #
34
+ # Fake agents
35
+ # --------------------------------------------------------------------------- #
36
+
37
+
38
+ class CrashOnCall(Exception):
39
+ """Scripted failure of a single agent invocation (spec E4)."""
40
+
41
+
42
+ def _default_responder(node: m.Node, prompt: str, _system: str) -> str:
43
+ """Deterministic text so identical (node, prompt) -> identical output."""
44
+
45
+ h = hashlib.sha256(f"{node.node_id}|{prompt}".encode()).hexdigest()[:SHORT_HASH_LEN]
46
+ return f"[{node.role}:{node.model}:{h}] {prompt}"
47
+
48
+
49
+ class _FakeBaseAgent:
50
+ """Shared scaffolding for every fake agent kind (LLM + non-LLM).
51
+
52
+ Owns the two pieces common to all kinds, lifted out of the old `FakeAgent`:
53
+ * the `crash_on` hook so a test can force ANY node — including a non-LLM
54
+ one — to raise on a chosen invocation index (spec E4 mid-run crash);
55
+ * `invoke_count` so deterministic behaviour + the crash index line up.
56
+
57
+ Each kind subclasses and implements ``_respond`` (what text + tool calls the
58
+ node produces). The base then wraps it with kind-aware perf: an ``llm`` node
59
+ costs deterministic pseudo-tokens (``len//4``, as before); a non-llm node
60
+ costs **ZERO tokens** — its real cost is wall-clock latency, which the
61
+ per-cycle wall cap (engine) measures. That tokens=0 convention is what lets a
62
+ tool/retriever/rule-heavy pipeline stay under the token cap yet be bounded by
63
+ the new wall cap (the design's cost fix). Latency is deterministic + nonzero
64
+ so traces stay reproducible (R-repeat aggregation, E1) and the wall cap is
65
+ exercisable.
66
+ """
67
+
68
+ def __init__(
69
+ self,
70
+ node: m.Node,
71
+ *,
72
+ crash_on: Callable[[int], bool] | None = None,
73
+ ) -> None:
74
+ self._node = node
75
+ self._crash_on = crash_on
76
+ self.invoke_count = 0
77
+
78
+ @property
79
+ def node_id(self) -> str:
80
+ return self._node.node_id
81
+
82
+ @property
83
+ def role(self) -> str:
84
+ return self._node.role
85
+
86
+ def _respond(
87
+ self,
88
+ prompt: str,
89
+ system_prompt: str,
90
+ model: str,
91
+ knobs: m.Knobs,
92
+ tools: list[str],
93
+ kind: m.NodeKind,
94
+ ) -> tuple[str, list[m.ToolCall]]:
95
+ raise NotImplementedError
96
+
97
+ def invoke(
98
+ self,
99
+ prompt: str,
100
+ *,
101
+ system_prompt: str,
102
+ model: str,
103
+ knobs: m.Knobs,
104
+ tools: list[str],
105
+ kind: m.NodeKind = m.NodeKind.LLM,
106
+ ) -> AgentResponse:
107
+ if self._crash_on is not None and self._crash_on(self.invoke_count):
108
+ self.invoke_count += 1
109
+ raise CrashOnCall(self.node_id)
110
+ self.invoke_count += 1
111
+
112
+ out, tool_calls = self._respond(prompt, system_prompt, model, knobs, tools, kind)
113
+ # LLM-shaped tokens (len//4) for cost+dedup; non-LLM kinds cost time, not
114
+ # tokens — perf.tokens=0 is the cost-cap convention (see class docstring).
115
+ token_count = max(1, len(out) // 4) if kind is m.NodeKind.LLM else 0
116
+ latency = float((token_count % 7) + 1) # deterministic, always nonzero
117
+ return AgentResponse(
118
+ text=out,
119
+ tool_calls=tool_calls,
120
+ perf=m.StepPerf(tokens=token_count, latency_ms=latency,
121
+ retries=0, error=None),
122
+ )
123
+
124
+
125
+ class FakeAgent(_FakeBaseAgent):
126
+ """An LLM node's executor: deterministic, optionally scripted to fail.
127
+
128
+ The historical fake (now the `kind=llm` arm). `invoke` returns a deterministic
129
+ response built from the node config + the incoming prompt, so the same
130
+ (spec, task) always yields the same output (reproducible runs for R-repeat
131
+ aggregation). An optional `crash_on` forces a raise on a chosen invocation
132
+ index. Also used, harmless, for `symbolic` nodes — a `symbolic` node is a
133
+ deterministic transform whose cost is latency not tokens, which the base
134
+ enforces via its kind-aware perf (tokens=0 for non-llm).
135
+ """
136
+
137
+ def __init__(
138
+ self,
139
+ node: m.Node,
140
+ *,
141
+ crash_on: Callable[[int], bool] | None = None,
142
+ responder: Callable[[m.Node, str, str], str] | None = None,
143
+ ) -> None:
144
+ super().__init__(node, crash_on=crash_on)
145
+ self._responder = responder or _default_responder
146
+
147
+ def _respond(
148
+ self,
149
+ prompt: str,
150
+ system_prompt: str,
151
+ model: str,
152
+ knobs: m.Knobs,
153
+ tools: list[str],
154
+ kind: m.NodeKind,
155
+ ) -> tuple[str, list[m.ToolCall]]:
156
+ return self._responder(self._node, prompt, system_prompt), []
157
+
158
+
159
+ class FakeRuleAgent(_FakeBaseAgent):
160
+ """A `rule` node (heuristic/scorer/classifier threshold — no prompt, no model).
161
+
162
+ Produces a deterministic label that surfaces its `threshold` knob (the
163
+ `tunable` param the LLM Architect edits on a rule node). Zero tokens.
164
+ """
165
+
166
+ def _respond(
167
+ self,
168
+ prompt: str,
169
+ system_prompt: str,
170
+ model: str,
171
+ knobs: m.Knobs,
172
+ tools: list[str],
173
+ kind: m.NodeKind,
174
+ ) -> tuple[str, list[m.ToolCall]]:
175
+ threshold = getattr(knobs, "threshold", None)
176
+ h = hashlib.sha256(f"{self._node.node_id}|{prompt}".encode()).hexdigest()[:SHORT_HASH_LEN]
177
+ return f"[rule:{self._node.role}:thr={threshold}:{h}] {prompt}", []
178
+
179
+
180
+ class FakeRetrieverAgent(_FakeBaseAgent):
181
+ """A `retriever` node — fetches search/vector context, parameterized by `top_k`.
182
+
183
+ Emits `top_k` deterministic context chunks (default 3 when `knobs.top_k` is
184
+ unset), so an Architect `knob` edit raising `top_k` visibly changes the step's
185
+ output (a different SuiteRun). Zero tokens.
186
+ """
187
+
188
+ def _respond(
189
+ self,
190
+ prompt: str,
191
+ system_prompt: str,
192
+ model: str,
193
+ knobs: m.Knobs,
194
+ tools: list[str],
195
+ kind: m.NodeKind,
196
+ ) -> tuple[str, list[m.ToolCall]]:
197
+ top_k = getattr(knobs, "top_k", None) or 3
198
+ h = hashlib.sha256(f"{self._node.node_id}|{prompt}".encode()).hexdigest()[:SHORT_HASH_LEN]
199
+ chunks = [f"chunk-{i}:{h}" for i in range(int(top_k))]
200
+ return f"[retriever:{self._node.role}:top_k={int(top_k)}] " + "; ".join(chunks), []
201
+
202
+
203
+ class FakeToolAgent(_FakeBaseAgent):
204
+ """A `tool` node — an external call recorded as a `ToolCall`.
205
+
206
+ Produces a deterministic serialized result AND a `ToolCall(tool_id=node_id)`
207
+ so a tool node's run is distinguishable from an LLM's (the trace carries the
208
+ call). Zero tokens.
209
+ """
210
+
211
+ def _respond(
212
+ self,
213
+ prompt: str,
214
+ system_prompt: str,
215
+ model: str,
216
+ knobs: m.Knobs,
217
+ tools: list[str],
218
+ kind: m.NodeKind,
219
+ ) -> tuple[str, list[m.ToolCall]]:
220
+ h = hashlib.sha256(f"{self._node.node_id}|{prompt}".encode()).hexdigest()[:SHORT_HASH_LEN]
221
+ result = f"[tool:{self._node.role}:{h}] {prompt}"
222
+ call = m.ToolCall(tool_id=self._node.node_id, args={"query": prompt},
223
+ result=result, ok=True)
224
+ return result, [call]
225
+
226
+
227
+ # --------------------------------------------------------------------------- #
228
+ # Runnable pipeline over the Spec graph
229
+ # --------------------------------------------------------------------------- #
230
+
231
+
232
+ class FakePipeline:
233
+ """A Spec rendered as an executable pipeline.
234
+
235
+ Execution model (a deliberate simplification that covers the spec's edge
236
+ kinds for exercising control flow):
237
+ * nodes run in a topological order derived from the edges
238
+ * the first node consumes the task input as its prompt; every later node
239
+ consumes the previous node's response as its prompt (sequence).
240
+ * a `conditional` edge whose `gate` equals the string form of the prior
241
+ step's response is taken; otherwise the next non-conditional edge is
242
+ followed. (Real hosts will have richer semantics; we only need enough to
243
+ run + trace.)
244
+ * the last node's response is the run's `final_output`.
245
+ """
246
+
247
+ def __init__(
248
+ self,
249
+ spec: m.Spec,
250
+ middleware: TracingMiddleware,
251
+ agents: dict[str, _FakeBaseAgent],
252
+ ) -> None:
253
+ self._spec = spec
254
+ self._mw = middleware
255
+ self._agents = agents
256
+ self._edges_by_src: dict[str, list[m.Edge]] = defaultdict(list)
257
+ for e in spec.edges:
258
+ self._edges_by_src[e.from_].append(e)
259
+ # monotonic counter so each run() mints a UNIQUE run_id, even at R>1
260
+ # (a spec+task no longer collides across repeats). See spec §7 R-repeats.
261
+ self._run_counter = 0
262
+
263
+ def run(self, task: Task) -> m.Trace:
264
+ sid = self._spec.compute_spec_id()
265
+ run_id = _run_id(sid, task.task_id, self._run_counter)
266
+ self._run_counter += 1
267
+ self._mw.begin_run(run_id, self._spec, task.task_id)
268
+ last_text: str | None = None
269
+ final_output: str | None = None
270
+ try:
271
+ order = _topo_order(self._spec)
272
+ if not order:
273
+ # nodeless spec -> nothing to trace; trivial success, empty trace
274
+ return self._mw.end_run(None, ok=True, error=None)
275
+ current_id = order[0]
276
+ prompt = task.input
277
+ visited: set[str] = set()
278
+ while current_id is not None:
279
+ if current_id in visited:
280
+ break # defensive against cycles the linter should already catch
281
+ visited.add(current_id)
282
+ agent = self._agents[current_id]
283
+ wrapped = self._mw.wrap(agent)
284
+ response = wrapped.invoke(prompt)
285
+ last_text = response.text
286
+ final_output = response.text
287
+ current_id, prompt = self._next_node(current_id, last_text)
288
+ trace = self._mw.end_run(final_output, ok=True, error=None)
289
+ return trace
290
+ except CrashOnCall as exc:
291
+ # A scripted mid-run crash — flush partial trace with ok=False (E4)
292
+ trace = self._mw.end_run(last_text, ok=False, error=f"crash in {exc}")
293
+ return trace
294
+ except Exception as exc: # noqa: BLE001 — host errors also flush partial trace
295
+ trace = self._mw.end_run(last_text, ok=False, error=repr(exc))
296
+ return trace
297
+
298
+ def _next_node(self, current_id: str, response: str) -> tuple[str | None, str]:
299
+ """Pick the next node given the outgoing edges of `current_id`.
300
+
301
+ Returns (next_node_id_or_None, next_prompt). A `conditional` edge is
302
+ taken iff its `gate` equals the current response; otherwise the first
303
+ non-conditional out-edge is followed. If no out-edge, we stop.
304
+ """
305
+
306
+ outs = self._edges_by_src.get(current_id, [])
307
+ if not outs:
308
+ return None, response
309
+ for edge in outs:
310
+ if edge.type is m.EdgeType.CONDITIONAL:
311
+ if edge.gate is not None and response.strip() == edge.gate:
312
+ return edge.to, response
313
+ else:
314
+ return edge.to, response
315
+ # only conditional edges, none taken -> stop
316
+ return None, response
317
+
318
+
319
+ # --------------------------------------------------------------------------- #
320
+ # The host
321
+ # --------------------------------------------------------------------------- #
322
+
323
+
324
+ class FakeHostMAS:
325
+ """A `HostMAS` that builds a `FakePipeline` from a Spec + node scripts.
326
+
327
+ Dispatches on each node's `kind` (the non-LLM extension) to pick the matching
328
+ fake agent — `llm`/`symbolic` -> `FakeAgent`, `rule` -> `FakeRuleAgent`,
329
+ `retriever` -> `FakeRetrieverAgent`, `tool` -> `FakeToolAgent` — so a mixed
330
+ pipeline (retriever + llm + tool) runs end-to-end **zero-LLM**. An explicit
331
+ `node_scripts[...]` override (e.g. `crash_on`) still applies to whatever kind
332
+ the node is, so E4 mid-run crashes script on non-LLM nodes too.
333
+
334
+ `responder` overrides the LLM-arm responder only (`_default_responder`
335
+ otherwise); it is ignored by the non-LLM fakes (they own their deterministic
336
+ outputs). `node_scripts` lets tests inject per-node behaviour keyed by
337
+ node_id; agents are built per `instantiate` so each candidate Spec gets fresh
338
+ ones.
339
+ """
340
+
341
+ def __init__(
342
+ self,
343
+ node_scripts: dict[str, dict] | None = None,
344
+ responder: Callable[[m.Node, str, str], str] | None = None,
345
+ ) -> None:
346
+ self._node_scripts = node_scripts or {}
347
+ self._responder = responder
348
+
349
+ def _agent_for(
350
+ self, node: m.Node, crash_on: Callable[[int], bool] | None
351
+ ) -> _FakeBaseAgent:
352
+ kind = node.kind
353
+ if kind is m.NodeKind.RULE:
354
+ return FakeRuleAgent(node, crash_on=crash_on)
355
+ if kind is m.NodeKind.RETRIEVER:
356
+ return FakeRetrieverAgent(node, crash_on=crash_on)
357
+ if kind is m.NodeKind.TOOL:
358
+ return FakeToolAgent(node, crash_on=crash_on)
359
+ # llm + symbolic share the generic (deterministic) agent; symbolic costs
360
+ # latency not tokens, enforced by the base's kind-aware perf.
361
+ return FakeAgent(node, crash_on=crash_on, responder=self._responder)
362
+
363
+ def instantiate(self, spec: m.Spec, middleware: TracingMiddleware) -> Runnable:
364
+ agents: dict[str, _FakeBaseAgent] = {}
365
+ for node in spec.nodes:
366
+ script = dict(self._node_scripts.get(node.node_id, {}))
367
+ crash_on = script.pop("crash_on", None) if isinstance(script, dict) else None
368
+ agents[node.node_id] = self._agent_for(node, crash_on)
369
+ return FakePipeline(spec, middleware, agents)
370
+
371
+
372
+ # ``topo_order`` and ``run_id`` now live in ``archforge.host.adapters.helpers``
373
+ # (single source for the kit + this module). They are re-imported at the top as
374
+ # ``_topo_order`` / ``_run_id`` so the call sites below are unchanged.
375
+
376
+
377
+ __all__ = [
378
+ "FakeHostMAS", "FakeAgent", "FakeRuleAgent", "FakeRetrieverAgent",
379
+ "FakeToolAgent", "FakePipeline", "CrashOnCall",
380
+ ]
@@ -0,0 +1,20 @@
1
+ """The Judge — LLM-as-judge scoring of runs (spec §3, §4).
2
+
3
+ `Judge.score(trace, task, rubric_id) -> RunScore` turns a run's trace into a
4
+ scored verdict: an aggregate + named rubric dimensions + confidence, plus a
5
+ *per-step breakdown* (StepScore[]) that names which agent lost which points. The
6
+ per-step breakdown is the raw material the Architect uses to credit-assign a
7
+ fault to a node/route.
8
+
9
+ `score_suite(...)` aggregates over R repeats: mean (down-weighted by
10
+ confidence) and a stable aggregate used by the Gatekeeper. `rubric_id` is
11
+ stamped on every score so cross-rubric comparisons never masquerade as
12
+ improvement (invariant I5, spec E2).
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ from archforge.judge.base import Judge, SuiteAggregate, default_rubric
18
+ from archforge.judge.scripted import ScriptedJudge
19
+
20
+ __all__ = ["Judge", "ScriptedJudge", "SuiteAggregate", "default_rubric"]