archforge-optimizer 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- archforge/__init__.py +76 -0
- archforge/__main__.py +10 -0
- archforge/architect.py +442 -0
- archforge/cli.py +881 -0
- archforge/config.py +140 -0
- archforge/config_init.py +150 -0
- archforge/diff.py +206 -0
- archforge/engine.py +444 -0
- archforge/gatekeeper.py +290 -0
- archforge/host/__init__.py +20 -0
- archforge/host/adapters/__init__.py +41 -0
- archforge/host/adapters/base.py +311 -0
- archforge/host/adapters/helpers.py +163 -0
- archforge/host/adapters/langgraph.py +726 -0
- archforge/host/base.py +105 -0
- archforge/host/fake.py +380 -0
- archforge/judge/__init__.py +20 -0
- archforge/judge/base.py +257 -0
- archforge/judge/scripted.py +145 -0
- archforge/lint.py +180 -0
- archforge/llm/__init__.py +65 -0
- archforge/llm/_common.py +94 -0
- archforge/llm/anthropic.py +90 -0
- archforge/llm/base.py +90 -0
- archforge/llm/gemini.py +112 -0
- archforge/llm/groq.py +63 -0
- archforge/llm/openai.py +63 -0
- archforge/llm/scripted.py +134 -0
- archforge/middleware.py +181 -0
- archforge/models.py +435 -0
- archforge/mutate.py +214 -0
- archforge/otel.py +613 -0
- archforge/runlog.py +103 -0
- archforge/runner.py +153 -0
- archforge/spec_builder.py +126 -0
- archforge/stores/__init__.py +22 -0
- archforge/stores/_jsonl.py +81 -0
- archforge/stores/attempt_store.py +161 -0
- archforge/stores/spec_store.py +188 -0
- archforge/stores/trace_store.py +42 -0
- archforge/suite.py +248 -0
- archforge/userconfig.py +144 -0
- archforge_optimizer-0.1.0.dist-info/METADATA +420 -0
- archforge_optimizer-0.1.0.dist-info/RECORD +47 -0
- archforge_optimizer-0.1.0.dist-info/WHEEL +4 -0
- archforge_optimizer-0.1.0.dist-info/entry_points.txt +2 -0
- archforge_optimizer-0.1.0.dist-info/licenses/LICENSE +21 -0
archforge/gatekeeper.py
ADDED
|
@@ -0,0 +1,290 @@
|
|
|
1
|
+
"""The Gatekeeper — the P-E-C "Commit" step (spec §3, §4, §6, §8).
|
|
2
|
+
|
|
3
|
+
Lone enforcer of "fail closed to the incumbent": it is the only thing that ever
|
|
4
|
+
moves the `active` pointer (via SpecStore.set_active) or flips an Attempt's
|
|
5
|
+
verdict beyond INITIAL. Its `decide(...)` returns a `Decision` naming the action;
|
|
6
|
+
`apply(...)` executes it (or you can decide without applying to inspect).
|
|
7
|
+
|
|
8
|
+
Decision rules (thresholds from `m.Thresholds`):
|
|
9
|
+
* unrunnable (candidate suite >ε crashed) -> DISCARD (before any margin math, E4)
|
|
10
|
+
* cross-rubric/cross-suite -> DISCARD (the only valid comparison is same
|
|
11
|
+
rubric + same suite; I5 — never silently compare across rubrics)
|
|
12
|
+
* candidate_mean - incumbent_mean < τ -> DISCARD (E1 noise inside τ)
|
|
13
|
+
* win + scope SMALL -> AUTO_PROMOTE (active = candidate)
|
|
14
|
+
* win + scope STRUCTURAL -> QUEUE_HUMAN (never auto-promote, I4)
|
|
15
|
+
* promoted-then-regresses (≥ δ on the standard suite) -> ROLLBACK
|
|
16
|
+
(active = parent, candidate archived, verdict ROLLED_BACK; pointer swap, E6)
|
|
17
|
+
|
|
18
|
+
Structural changes are queued even on a clear win — the human gate the user
|
|
19
|
+
locked in. `apply` returns the updated Attempt (verdict set) so the orchestrator
|
|
20
|
+
can record the cycle's outcome. Rollback uses the lineage pointer; no Spec is
|
|
21
|
+
ever deleted (archived, never deleted).
|
|
22
|
+
|
|
23
|
+
Human approval (`approve`) is the second path — besides AUTO_PROMOTE/ROLLBACK —
|
|
24
|
+
that may move `active`: a queued (PENDING_HUMAN) structural change the human
|
|
25
|
+
accepts becomes the incumbent; a rejection (`reject`) leaves the incumbent alone
|
|
26
|
+
and archives the candidate as a recorded dead end.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
from __future__ import annotations
|
|
30
|
+
|
|
31
|
+
from enum import Enum
|
|
32
|
+
from typing import Any
|
|
33
|
+
|
|
34
|
+
from pydantic import BaseModel, ConfigDict
|
|
35
|
+
|
|
36
|
+
import archforge.models as m
|
|
37
|
+
from archforge.stores.attempt_store import AttemptStore
|
|
38
|
+
from archforge.stores.spec_store import SpecStore
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class Action(str, Enum):
|
|
42
|
+
"""What the Gatekeeper decided to do with this candidate."""
|
|
43
|
+
|
|
44
|
+
AUTO_PROMOTE = "auto_promote" # small win -> active = candidate
|
|
45
|
+
QUEUE_HUMAN = "queue_human" # structural win -> Approval Queue
|
|
46
|
+
DISCARD = "discard" # loss / unrunnable / cross-rubric
|
|
47
|
+
ROLLBACK = "rollback" # promoted-then-regressed (E6)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class Decision(BaseModel):
|
|
51
|
+
"""The Gatekeeper's verdict on a candidate. ``Gatekeeper.apply_decision`` executes it."""
|
|
52
|
+
|
|
53
|
+
model_config = ConfigDict(extra="allow")
|
|
54
|
+
|
|
55
|
+
action: Action
|
|
56
|
+
attempt_id: str | None = None # the candidate Attempt this decides
|
|
57
|
+
reason: str # human-readable, surfaced to the report
|
|
58
|
+
margin: float = 0.0 # candidate_mean - incumbent_mean (signed)
|
|
59
|
+
by_rule: str = "" # which rule fired ("auto_promote", "rollback", ...)
|
|
60
|
+
|
|
61
|
+
def __repr__(self) -> str: # pragma: no cover (debug aid)
|
|
62
|
+
return (f"Decision(action={self.action.value}, margin={self.margin:+.3f}, "
|
|
63
|
+
f"rule={self.by_rule}, reason={self.reason!r})")
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class _AttemptNotFoundError(KeyError):
|
|
67
|
+
"""The candidate Attempt given to the Gatekeeper was never persisted."""
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
class Gatekeeper:
|
|
71
|
+
"""The lone enforcer of promotion + rollback.
|
|
72
|
+
|
|
73
|
+
Constructed once per run with the (SpecStore, AttemptStore) it mutates and a
|
|
74
|
+
`Thresholds` (τ, δ, ε). Stateless across calls otherwise — the incumbent + a
|
|
75
|
+
candidate's SuiteRun are passed in `decide`.
|
|
76
|
+
"""
|
|
77
|
+
|
|
78
|
+
def __init__(
|
|
79
|
+
self,
|
|
80
|
+
spec_store: SpecStore,
|
|
81
|
+
attempt_store: AttemptStore,
|
|
82
|
+
*,
|
|
83
|
+
thresholds: m.Thresholds | None = None,
|
|
84
|
+
) -> None:
|
|
85
|
+
self._specs = spec_store
|
|
86
|
+
self._attempts = attempt_store
|
|
87
|
+
self._th = thresholds or m.Thresholds()
|
|
88
|
+
|
|
89
|
+
# ----------------------------------------------------------------- decide
|
|
90
|
+
def decide(
|
|
91
|
+
self,
|
|
92
|
+
attempt_id: str,
|
|
93
|
+
candidate: Any, # SuiteRun (typed loosely to avoid an import cycle)
|
|
94
|
+
incumbent: "m.SuiteRun | None" = None,
|
|
95
|
+
) -> Decision:
|
|
96
|
+
"""Decide the fate of a candidate `SuiteRun` vs the incumbent `SuiteRun`.
|
|
97
|
+
|
|
98
|
+
Both are `SuiteRun` objects (archforge.suite). `incumbent=None` means
|
|
99
|
+
"no incumbent yet" — only AUTO_PROMOTE-like forward progress is valid,
|
|
100
|
+
but a structural candidate still queues for human review.
|
|
101
|
+
"""
|
|
102
|
+
# NOTE: `m.SuiteRun` lives in archforge.suite, not archforge.models, so we
|
|
103
|
+
# accept it as `Any` here and pull attributes defensively (duck-typed).
|
|
104
|
+
|
|
105
|
+
if candidate.unrunnable:
|
|
106
|
+
return _decision(Action.DISCARD, attempt_id,
|
|
107
|
+
"candidate suite was unrunnable (>ε tasks crashed/unscored)",
|
|
108
|
+
rule="unrunnable")
|
|
109
|
+
|
|
110
|
+
if incumbent is not None and not _same_geometry(incumbent, candidate):
|
|
111
|
+
return _decision(Action.DISCARD, attempt_id,
|
|
112
|
+
"candidate vs incumbent differ in rubric or suite "
|
|
113
|
+
"(cross-geometry comparison blocked, I5)",
|
|
114
|
+
rule="cross_geometry")
|
|
115
|
+
|
|
116
|
+
inc_mean = incumbent.mean if incumbent is not None else 0.0
|
|
117
|
+
margin = candidate.mean - inc_mean
|
|
118
|
+
cand_scope = self._candidate_scope(attempt_id)
|
|
119
|
+
|
|
120
|
+
if margin < self._th.tau:
|
|
121
|
+
return _decision(Action.DISCARD, attempt_id,
|
|
122
|
+
f"candidate did not clear margin τ={self._th.tau} "
|
|
123
|
+
f"(margin {margin:+.3f}); noise not promoted (E1)",
|
|
124
|
+
margin=margin, rule="below_margin")
|
|
125
|
+
|
|
126
|
+
# win past τ
|
|
127
|
+
if cand_scope is m.Scope.STRUCTURAL:
|
|
128
|
+
return _decision(Action.QUEUE_HUMAN, attempt_id,
|
|
129
|
+
f"structural win by {margin:+.3f} >= τ but structural "
|
|
130
|
+
f"changes require human approval (I4)",
|
|
131
|
+
margin=margin, rule="structural_wins_queue")
|
|
132
|
+
return _decision(Action.AUTO_PROMOTE, attempt_id,
|
|
133
|
+
f"small win by {margin:+.3f} >= τ={self._th.tau}; promoted",
|
|
134
|
+
margin=margin, rule="auto_promote")
|
|
135
|
+
|
|
136
|
+
# ----------------------------------------------------------------- apply
|
|
137
|
+
def apply_decision(self, decision: Decision) -> m.Attempt:
|
|
138
|
+
"""Execute a Decision against the stores and return the updated Attempt."""
|
|
139
|
+
|
|
140
|
+
if decision.attempt_id is None:
|
|
141
|
+
raise ValueError("cannot apply a Decision with no attempt_id")
|
|
142
|
+
att = self._attempts.require(decision.attempt_id)
|
|
143
|
+
|
|
144
|
+
if decision.action is Action.AUTO_PROMOTE:
|
|
145
|
+
spec_id = att.candidate_spec_id
|
|
146
|
+
self._specs.set_active(spec_id)
|
|
147
|
+
verdict = m.Verdict.PROMOTED
|
|
148
|
+
elif decision.action is Action.QUEUE_HUMAN:
|
|
149
|
+
verdict = m.Verdict.PENDING_HUMAN
|
|
150
|
+
elif decision.action is Action.DISCARD:
|
|
151
|
+
verdict = m.Verdict.REJECTED
|
|
152
|
+
elif decision.action is Action.ROLLBACK:
|
|
153
|
+
self._rollback(att)
|
|
154
|
+
verdict = m.Verdict.ROLLED_BACK
|
|
155
|
+
else: # pragma: no cover (enum exhaustive)
|
|
156
|
+
raise ValueError(f"unknown action {decision.action}")
|
|
157
|
+
|
|
158
|
+
return self._attempts.set_verdict(decision.attempt_id, verdict)
|
|
159
|
+
|
|
160
|
+
# ----------------------------------------------------------------- human gate
|
|
161
|
+
def approve(self, attempt_id: str) -> m.Attempt:
|
|
162
|
+
"""Human approves a queued (PENDING_HUMAN) structural change (spec I4).
|
|
163
|
+
|
|
164
|
+
This is the one path besides AUTO_PROMOTE/ROLLBACK that may move the
|
|
165
|
+
`active` pointer — the human gate the user locked in: a structural win is
|
|
166
|
+
never auto-promoted, only ever promoted through here. `active` -> the
|
|
167
|
+
candidate, verdict -> PROMOTED. Idempotent for an already-promoted attempt
|
|
168
|
+
(a no-op). Raises `ValueError` if the attempt is not queued, so the CLI
|
|
169
|
+
cannot rewrite history (only decide what the gate queued).
|
|
170
|
+
"""
|
|
171
|
+
|
|
172
|
+
att = self._attempts.require(attempt_id)
|
|
173
|
+
if att.verdict is m.Verdict.PROMOTED:
|
|
174
|
+
return att # already approved (idempotent)
|
|
175
|
+
if att.verdict is not m.Verdict.PENDING_HUMAN:
|
|
176
|
+
raise ValueError(
|
|
177
|
+
f"cannot approve attempt {attempt_id}: verdict is "
|
|
178
|
+
f"{att.verdict.value}; only PENDING_HUMAN (queued) changes can be approved"
|
|
179
|
+
)
|
|
180
|
+
self._specs.set_active(att.candidate_spec_id)
|
|
181
|
+
return self._attempts.set_verdict(attempt_id, m.Verdict.PROMOTED)
|
|
182
|
+
|
|
183
|
+
def reject(self, attempt_id: str, *, reason: str = "human-reject") -> m.Attempt:
|
|
184
|
+
"""Human rejects a queued structural change: verdict -> REJECTED, active unchanged.
|
|
185
|
+
|
|
186
|
+
The incumbent is left alone. The candidate Spec is archived (never
|
|
187
|
+
deleted) with `reason="human-reject"` so the dead end is recorded for
|
|
188
|
+
lineage/dedup queries. Idempotent for an already-rejected attempt.
|
|
189
|
+
Raises `ValueError` if the attempt is not queued.
|
|
190
|
+
"""
|
|
191
|
+
|
|
192
|
+
att = self._attempts.require(attempt_id)
|
|
193
|
+
if att.verdict is m.Verdict.REJECTED:
|
|
194
|
+
return att
|
|
195
|
+
if att.verdict is not m.Verdict.PENDING_HUMAN:
|
|
196
|
+
raise ValueError(
|
|
197
|
+
f"cannot reject attempt {attempt_id}: verdict is "
|
|
198
|
+
f"{att.verdict.value}; only PENDING_HUMAN (queued) changes can be rejected"
|
|
199
|
+
)
|
|
200
|
+
self._specs.archive(att.candidate_spec_id, reason=reason)
|
|
201
|
+
return self._attempts.set_verdict(attempt_id, m.Verdict.REJECTED)
|
|
202
|
+
|
|
203
|
+
# ----------------------------------------------------------------- rollback (E6)
|
|
204
|
+
def rollback(
|
|
205
|
+
self,
|
|
206
|
+
attempt_id: str,
|
|
207
|
+
*,
|
|
208
|
+
regressed_mean: float,
|
|
209
|
+
pre_promotion_mean: float,
|
|
210
|
+
) -> Decision:
|
|
211
|
+
"""Decide+apply a rollback after a promoted Spec regressed (E6).
|
|
212
|
+
|
|
213
|
+
Rolls back iff `pre_promotion_mean - regressed_mean >= δ` on the standard
|
|
214
|
+
suite (same rubric+suite, I5). Otherwise the regression is within the
|
|
215
|
+
noise floor and the incumbent is left alone. Returns the Decision (which
|
|
216
|
+
has already been applied if action is ROLLBACK).
|
|
217
|
+
"""
|
|
218
|
+
|
|
219
|
+
drop = pre_promotion_mean - regressed_mean
|
|
220
|
+
# Idempotency: a repeated regression check on an already-rolled-back
|
|
221
|
+
# attempt must not downgrade its verdict or move the active pointer
|
|
222
|
+
# again. Return a benign DISCARD decision without touching the stores.
|
|
223
|
+
current_verdict = self._attempts.require(attempt_id).verdict
|
|
224
|
+
if current_verdict is m.Verdict.ROLLED_BACK:
|
|
225
|
+
return _decision(Action.DISCARD, attempt_id,
|
|
226
|
+
"regression check on an already-rolled-back attempt; "
|
|
227
|
+
"no further action",
|
|
228
|
+
margin=-drop, rule="already_rolled_back")
|
|
229
|
+
|
|
230
|
+
if drop < self._th.delta:
|
|
231
|
+
# Regression is within the δ noise floor: the promotion stands, the
|
|
232
|
+
# incumbent (the promoted Spec) is left in place, and the attempt's
|
|
233
|
+
# verdict is NOT mutated — a real small-won, the dip is just noise.
|
|
234
|
+
return _decision(Action.DISCARD, attempt_id,
|
|
235
|
+
f"regression {drop:+.3f} < δ={self._th.delta}; within the "
|
|
236
|
+
f"noise floor, incumbent left alone",
|
|
237
|
+
margin=-drop, rule="regression_within_floor")
|
|
238
|
+
|
|
239
|
+
d = _decision(Action.ROLLBACK, attempt_id,
|
|
240
|
+
f"regression {drop:+.3f} >= δ={self._th.delta}; rolling back to "
|
|
241
|
+
f"parent (pointer swap, archived, never deleted)",
|
|
242
|
+
margin=-drop, rule="rollback")
|
|
243
|
+
att = self._attempts.require(attempt_id)
|
|
244
|
+
self._rollback(att)
|
|
245
|
+
self._attempts.set_verdict(attempt_id, m.Verdict.ROLLED_BACK)
|
|
246
|
+
return d
|
|
247
|
+
|
|
248
|
+
def _rollback(self, att: m.Attempt) -> None:
|
|
249
|
+
"""Active -> parent; candidate archived. Pointer swap, never deleted (E6/I3)."""
|
|
250
|
+
|
|
251
|
+
spec = self._specs.get(att.candidate_spec_id)
|
|
252
|
+
parent_id = spec.parent_spec_id
|
|
253
|
+
if parent_id is None:
|
|
254
|
+
# root incumbent regressed: nothing to roll back to; leave pointer, mark rejected
|
|
255
|
+
self._specs.archive(att.candidate_spec_id, reason="rollback_rootless")
|
|
256
|
+
return
|
|
257
|
+
if not self._specs.has(parent_id):
|
|
258
|
+
raise _AttemptNotFoundError(
|
|
259
|
+
f"rollback target parent {parent_id} not in store"
|
|
260
|
+
)
|
|
261
|
+
self._specs.set_active(parent_id) # pointer swap — the incumbent reverts
|
|
262
|
+
self._specs.archive(att.candidate_spec_id, reason="rollback")
|
|
263
|
+
|
|
264
|
+
# ----------------------------------------------------------------- helpers
|
|
265
|
+
def _candidate_scope(self, attempt_id: str) -> m.Scope:
|
|
266
|
+
att = self._attempts.require(attempt_id)
|
|
267
|
+
return att.change.scope
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
# --------------------------------------------------------------------------- #
|
|
271
|
+
# helpers (module-private)
|
|
272
|
+
# --------------------------------------------------------------------------- #
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def _decision(
|
|
276
|
+
action: Action, attempt_id: str | None, reason: str,
|
|
277
|
+
*, margin: float = 0.0, rule: str = "",
|
|
278
|
+
) -> Decision:
|
|
279
|
+
return Decision(action=action, attempt_id=attempt_id, reason=reason,
|
|
280
|
+
margin=margin, by_rule=rule)
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def _same_geometry(inc: Any, cand: Any) -> bool:
|
|
284
|
+
"""Invariant I5: comparisons only valid under identical (rubric, suite)."""
|
|
285
|
+
|
|
286
|
+
return (getattr(inc, "rubric_id", None) == getattr(cand, "rubric_id", None)
|
|
287
|
+
and getattr(inc, "suite_id", None) == getattr(cand, "suite_id", None))
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
__all__ = ["Action", "Decision", "Gatekeeper"]
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
"""Host multi-agent system integration (spec §3, §4).
|
|
2
|
+
|
|
3
|
+
ArchForge does not depend on any specific agentic framework. A host MAS
|
|
4
|
+
implements the `HostMAS` contract: it can build an executable pipeline from a
|
|
5
|
+
`Spec` and run a `Task` through it, emitting one `Step` per agent. The
|
|
6
|
+
`TracingMiddleware` attaches as both observer (records steps to TraceStore) and
|
|
7
|
+
config bridge (applies the live Spec's prompt/model/knobs to each agent at
|
|
8
|
+
invoke time — so evolving the pipeline is swapping which Spec the host uses).
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from archforge.host.base import Agent, AgentResponse, HostMAS, Runnable, Task
|
|
14
|
+
from archforge.host.fake import (
|
|
15
|
+
FakeAgent, FakeHostMAS, FakeRuleAgent, FakeRetrieverAgent, FakeToolAgent,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
__all__ = ["Agent", "AgentResponse", "HostMAS", "Runnable", "Task",
|
|
19
|
+
"FakeHostMAS", "FakeAgent", "FakeRuleAgent", "FakeRetrieverAgent",
|
|
20
|
+
"FakeToolAgent"]
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""archforge.host.adapters — the reusable adapter kit.
|
|
2
|
+
|
|
3
|
+
The single place a new MAS author looks: subclass `BaseHostAdapter` + a small
|
|
4
|
+
`BaseAgent` per node, fill `call()` (the node's real work), and override the
|
|
5
|
+
hooks (`execution_order` / `resolve_prompt` / `stage_context`) only when the
|
|
6
|
+
MAS's data-shaping is non-default. The kit owns the run loop, content-decouple,
|
|
7
|
+
config-decay, perf, and partial-trace flush — the recurring scaffolding every
|
|
8
|
+
adapter previously re-derived.
|
|
9
|
+
|
|
10
|
+
`helpers` is also re-exported so an author extending the loop's primitives
|
|
11
|
+
(topo order, run-id, config decay) does so against one source.
|
|
12
|
+
"""
|
|
13
|
+
from archforge.host.adapters.base import (
|
|
14
|
+
BaseAgent, BaseHostAdapter, BasePipeline, CallResult, RunContext,
|
|
15
|
+
)
|
|
16
|
+
from archforge.host.adapters.helpers import (
|
|
17
|
+
KnobVote, cfg_as_kwargs, cfg_decay, estimate_tokens, run_id, topo_order,
|
|
18
|
+
)
|
|
19
|
+
# LangGraph adapter — re-exported eagerly. The module is langgraph-FREE (it
|
|
20
|
+
# imports no langgraph types; the dependency enters only when a concrete app's
|
|
21
|
+
# `graph_factory` builds the real graph), so importing it here keeps
|
|
22
|
+
# `import archforge` framework-free.
|
|
23
|
+
from archforge.host.adapters.langgraph import (
|
|
24
|
+
EdgeSpec, LangGraphApp, LangGraphHostAdapter, LangGraphRunnable, Nd,
|
|
25
|
+
NodeIdMap, build_optimized_envelope, export_optimized, export_spec_sidecar,
|
|
26
|
+
load_optimized, load_spec_sidecar, node_ids,
|
|
27
|
+
)
|
|
28
|
+
# `wrapped` is deliberately NOT re-exported at the package level: a MAS authors
|
|
29
|
+
# the id ONCE in its `build_graph` via `add(name, fn)` and imports `wrapped`
|
|
30
|
+
# from `archforge.host.adapters.langgraph` (the adapter module the MAS already
|
|
31
|
+
# touches) — never from `archforge.otel` directly, and never from this package.
|
|
32
|
+
|
|
33
|
+
__all__ = [
|
|
34
|
+
"BaseHostAdapter", "BaseAgent", "BasePipeline", "CallResult", "RunContext",
|
|
35
|
+
"run_id", "topo_order", "cfg_decay", "cfg_as_kwargs", "KnobVote",
|
|
36
|
+
"estimate_tokens",
|
|
37
|
+
"Nd", "EdgeSpec", "LangGraphApp", "LangGraphHostAdapter",
|
|
38
|
+
"LangGraphRunnable", "NodeIdMap", "node_ids",
|
|
39
|
+
"export_spec_sidecar", "load_spec_sidecar",
|
|
40
|
+
"build_optimized_envelope", "export_optimized", "load_optimized",
|
|
41
|
+
]
|
|
@@ -0,0 +1,311 @@
|
|
|
1
|
+
"""The ArchForge adapter kit — a reusable implementation of the host seam.
|
|
2
|
+
|
|
3
|
+
`HostMAS` / `Agent` / `Runnable` (in ``host/base.py``) are the *contract*; this
|
|
4
|
+
module is a *reusable implementation of it* that owns the scaffolding every MAS
|
|
5
|
+
adapter repeats:
|
|
6
|
+
|
|
7
|
+
* one ``Agent`` per Spec node;
|
|
8
|
+
* the run loop: ``begin_run`` → ordered walk → ``end_run``, with a monotonic
|
|
9
|
+
run-id (unique across R-repeats), partial-trace flush on any crash (E4);
|
|
10
|
+
* the **content-decouple**: score the agent's *content*, thread its
|
|
11
|
+
*plumbing* onward via an ``AgentResponse`` extra (``extra="allow"`` lets a
|
|
12
|
+
``thread`` field ride without a model change);
|
|
13
|
+
* the **config-decay** (phase-2 rule): thread model/temperature/max_tokens
|
|
14
|
+
always (base run == agent default → behaviour-preserving) and system_prompt
|
|
15
|
+
only when mutated.
|
|
16
|
+
|
|
17
|
+
The genuinely-MAS-specific bits are exposed as a few imperative hooks an adapter
|
|
18
|
+
overrides when its MAS needs them — *not* a declarative graph interpreter (so
|
|
19
|
+
hard data-shaping — file-stage handoffs, multi-input joins — is just imperative
|
|
20
|
+
code in a hook, never a drop off a declarative cliff).
|
|
21
|
+
"""
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import time
|
|
25
|
+
from contextlib import nullcontext
|
|
26
|
+
from dataclasses import dataclass, field
|
|
27
|
+
from typing import Any, Callable
|
|
28
|
+
|
|
29
|
+
import archforge.models as m
|
|
30
|
+
from archforge.host.adapters.helpers import cfg_decay, estimate_tokens, run_id, topo_order
|
|
31
|
+
from archforge.host.base import AgentResponse, HostMAS, Runnable, Task
|
|
32
|
+
from archforge.middleware import TracingMiddleware
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
# --------------------------------------------------------------------------- #
|
|
36
|
+
# Run context — what the loop carries between nodes
|
|
37
|
+
# --------------------------------------------------------------------------- #
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass
|
|
41
|
+
class RunContext:
|
|
42
|
+
"""The state threaded through one run, advanced by the loop and read by
|
|
43
|
+
the adapter's ``resolve_prompt``.
|
|
44
|
+
|
|
45
|
+
``last_content`` is what each step's response *recorded* (what the Judge
|
|
46
|
+
scores); ``last_thread`` is what each step *threads onward* (plumbing: a
|
|
47
|
+
file path, a ref… — usually the same as content for text-in/text-out MASes,
|
|
48
|
+
different for file-staged ones like Lumina). ``per_node`` lets a join node
|
|
49
|
+
recall a sibling threaded earlier (the report 3-way join reads sibling
|
|
50
|
+
gap/cross paths here).
|
|
51
|
+
"""
|
|
52
|
+
|
|
53
|
+
last_content: str
|
|
54
|
+
last_thread: str | None = None
|
|
55
|
+
per_node: dict[str, Any] = field(default_factory=dict)
|
|
56
|
+
|
|
57
|
+
def advance(self, node_id: str, resp: AgentResponse) -> None:
|
|
58
|
+
self.last_content = resp.text
|
|
59
|
+
# `thread` rides as an AgentResponse extra (extra="allow"); fall back to
|
|
60
|
+
# the scored content if the adapter didn't set one (text-in/text-out).
|
|
61
|
+
self.last_thread = getattr(resp, "thread", None) or resp.text
|
|
62
|
+
self.per_node[node_id] = self.last_thread
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
# --------------------------------------------------------------------------- #
|
|
66
|
+
# CallResult — the content-decouple made first-class
|
|
67
|
+
# --------------------------------------------------------------------------- #
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
@dataclass
|
|
71
|
+
class CallResult:
|
|
72
|
+
"""What an agent's ``call`` returns: ``thread`` flows to the next node
|
|
73
|
+
(plumbing), ``content`` is what the trace records and the Judge scores.
|
|
74
|
+
|
|
75
|
+
Promoted from the Lumina adapter's ``RunResult`` so every kit-made adapter
|
|
76
|
+
gets content-decouple for free instead of re-inventing it.
|
|
77
|
+
"""
|
|
78
|
+
|
|
79
|
+
thread: str | None
|
|
80
|
+
content: str
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
# --------------------------------------------------------------------------- #
|
|
84
|
+
# BaseAgent — implements `Agent`; owns invoke; author overrides `call`
|
|
85
|
+
# --------------------------------------------------------------------------- #
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
class BaseAgent:
|
|
89
|
+
"""One node's executor. The kit owns ``invoke`` (the content-decouple +
|
|
90
|
+
config-decay + perf measurement); the adapter fills ``call`` — the node's
|
|
91
|
+
real work — and is handed a config map it forwards to its agent call.
|
|
92
|
+
"""
|
|
93
|
+
|
|
94
|
+
def __init__(self, node: m.Node, adapter: "BaseHostAdapter") -> None:
|
|
95
|
+
self._node = node
|
|
96
|
+
self._adapter = adapter
|
|
97
|
+
|
|
98
|
+
@property
|
|
99
|
+
def node_id(self) -> str:
|
|
100
|
+
return self._node.node_id
|
|
101
|
+
|
|
102
|
+
@property
|
|
103
|
+
def role(self) -> str:
|
|
104
|
+
return self._node.role
|
|
105
|
+
|
|
106
|
+
def call(self, prompt: str, vote: "KnobVote") -> CallResult: # noqa: ARG002
|
|
107
|
+
"""The one abstract hook: run the node, return (thread, content).
|
|
108
|
+
|
|
109
|
+
``vote`` is the DECAYED live config (a `KnobVote`: model / temperature /
|
|
110
|
+
max_tokens / retries / system_prompt, each ``None`` meaning "use your
|
|
111
|
+
default"). The adapter binds exactly the keys its agent accepts — the
|
|
112
|
+
kit owns the decay logic, not the agent's signature. Subclasses override.
|
|
113
|
+
"""
|
|
114
|
+
raise NotImplementedError(f"{type(self).__name__}.call() not implemented")
|
|
115
|
+
|
|
116
|
+
def invoke(
|
|
117
|
+
self,
|
|
118
|
+
prompt: str,
|
|
119
|
+
*,
|
|
120
|
+
system_prompt: str | None,
|
|
121
|
+
model: str | None,
|
|
122
|
+
knobs: m.Knobs | None,
|
|
123
|
+
tools: list[str] | None,
|
|
124
|
+
kind: m.NodeKind = m.NodeKind.LLM,
|
|
125
|
+
) -> AgentResponse:
|
|
126
|
+
"""Kit-owned: fulfils the `Agent` protocol. Applies config-decay, calls
|
|
127
|
+
the node's real work, packs an AgentResponse that records *content* and
|
|
128
|
+
carries *plumbing* onward via the `thread` extra.
|
|
129
|
+
|
|
130
|
+
`kind` is accepted for protocol parity with the middleware (which threads
|
|
131
|
+
the live node's `NodeKind` to every wrapped agent) and defaults to `LLM`
|
|
132
|
+
so existing kit adapters keep working unchanged. It is UNUSED here: a kit
|
|
133
|
+
adapter that needs kind-specific behaviour reads `self._node.kind` in its
|
|
134
|
+
`call` override (the node is already held on the agent).
|
|
135
|
+
"""
|
|
136
|
+
vote = self._adapter.cfg_for(self.node_id, system_prompt, model, knobs, tools)
|
|
137
|
+
t0 = time.perf_counter()
|
|
138
|
+
cr = self.call(prompt, vote)
|
|
139
|
+
latency = round((time.perf_counter() - t0) * 1000.0, 3)
|
|
140
|
+
# `text` is what the trace records + the Judge scores (content); the
|
|
141
|
+
# `thread` extra is what flows to the next node (plumbing).
|
|
142
|
+
return AgentResponse(
|
|
143
|
+
text=cr.content,
|
|
144
|
+
tool_calls=[],
|
|
145
|
+
thread=cr.thread if cr.thread is not None else cr.content,
|
|
146
|
+
perf=m.StepPerf(tokens=estimate_tokens(cr.content), latency_ms=latency),
|
|
147
|
+
)
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _estimate(text: str) -> int: # kept as a thin alias for any external callers
|
|
151
|
+
return estimate_tokens(text)
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
# --------------------------------------------------------------------------- #
|
|
155
|
+
# BaseHostAdapter — implements `HostMAS`; owns instantiate; delegates the run
|
|
156
|
+
# --------------------------------------------------------------------------- #
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
class BaseHostAdapter:
|
|
160
|
+
"""A `HostMAS` that builds a `BasePipeline` from a Spec.
|
|
161
|
+
|
|
162
|
+
An adapter subclasses this and overrides the hooks that are non-default for
|
|
163
|
+
its MAS:
|
|
164
|
+
|
|
165
|
+
* ``make_agent`` — wrap one framework agent per Spec node (the per-node
|
|
166
|
+
factory that chooses the right ``BaseAgent`` subclass).
|
|
167
|
+
* ``execution_order`` — default: topological from edges; override for a
|
|
168
|
+
fixed sequence (Lumina's 7-step).
|
|
169
|
+
* ``resolve_prompt`` — default: the last step's content; override for a
|
|
170
|
+
file path when a node needs upstream plumbing (joins, file-stage handoffs).
|
|
171
|
+
* ``stage_context`` — default: none; override for a per-run scratch area
|
|
172
|
+
(Lumina: a temp dir + chdir, released on exit).
|
|
173
|
+
|
|
174
|
+
The run loop, content-decouple, config-decay, partial-trace flush, perf —
|
|
175
|
+
none of those are the adapter's concern; the kit owns them.
|
|
176
|
+
"""
|
|
177
|
+
|
|
178
|
+
#: The seeded/default prompts, so `cfg_decay` can tell a *mutated* prompt
|
|
179
|
+
#: (an Architect prompt_edit) from the default and only thread the former.
|
|
180
|
+
#: Subclasses populate this (often from their base Spec).
|
|
181
|
+
base_prompts: dict[str, str] = {}
|
|
182
|
+
|
|
183
|
+
# ---- hooks (override when the MAS needs non-default behaviour) -------- #
|
|
184
|
+
|
|
185
|
+
def make_agent(self, node: m.Node) -> BaseAgent:
|
|
186
|
+
"""Pick the `BaseAgent` subclass for `node` (one per Spec node)."""
|
|
187
|
+
raise NotImplementedError("make_agent must return a BaseAgent for each node")
|
|
188
|
+
|
|
189
|
+
def execution_order(self, spec: m.Spec) -> list[str]:
|
|
190
|
+
"""Order to run nodes in. Default: topological from the Spec's edges."""
|
|
191
|
+
return topo_order(spec)
|
|
192
|
+
|
|
193
|
+
def resolve_prompt(self, node_id: str, ctx: RunContext) -> str:
|
|
194
|
+
"""What `node_id` receives as its prompt. Default: the last step's
|
|
195
|
+
*content* (text-in/text-out). Override to pass a file path / sibling
|
|
196
|
+
ref when the node needs upstream plumbing (joins, file-stage handoffs)."""
|
|
197
|
+
return ctx.last_content
|
|
198
|
+
|
|
199
|
+
def stage_context(self):
|
|
200
|
+
"""A context manager wrapping one run. Default: none. Override for a
|
|
201
|
+
per-run scratch area — e.g. a temp dir + chdir (Lumina) — that this
|
|
202
|
+
releases on exit. The loop's `finally` cleanup lives here."""
|
|
203
|
+
return nullcontext()
|
|
204
|
+
|
|
205
|
+
# ---- kit-internal seam (rarely overridden) --------------------------- #
|
|
206
|
+
|
|
207
|
+
def cfg_for(
|
|
208
|
+
self, node_id: str, system_prompt: str | None, model: str | None,
|
|
209
|
+
knobs: m.Knobs | None, tools: list[str] | None,
|
|
210
|
+
) -> "KnobVote":
|
|
211
|
+
"""Live-config → the DECAYED values the adapter's ``call`` forwards to
|
|
212
|
+
its agent. Defaults to the phase-2 `cfg_decay` rule; returns a
|
|
213
|
+
``KnobVote`` so each adapter binds EXACTLY the keys its agent accepts
|
|
214
|
+
(the kit owns the decay logic, not the agent's signature). Override only
|
|
215
|
+
if the MAS's config-decay differs."""
|
|
216
|
+
return cfg_decay(node_id, system_prompt, model, knobs, tools,
|
|
217
|
+
base_prompts=self.base_prompts)
|
|
218
|
+
|
|
219
|
+
# ---- kit-owned: the existing HostMAS.instantiate contract, un-touched -- #
|
|
220
|
+
|
|
221
|
+
def instantiate(self, spec: m.Spec, middleware: TracingMiddleware) -> Runnable:
|
|
222
|
+
agents: dict[str, BaseAgent] = {}
|
|
223
|
+
for node in spec.nodes:
|
|
224
|
+
agents[node.node_id] = self._agent_for(node)
|
|
225
|
+
return BasePipeline(spec, middleware, agents, self)
|
|
226
|
+
|
|
227
|
+
def _agent_for(self, node: m.Node) -> BaseAgent:
|
|
228
|
+
"""One agent per Spec node; an unknown node_id (an `add_node` we can't
|
|
229
|
+
run) becomes a _RaisingAgent, surfacing as an errored run that scoring
|
|
230
|
+
rejects — structural changes stay human-gated (I4)."""
|
|
231
|
+
try:
|
|
232
|
+
return self.make_agent(node)
|
|
233
|
+
except NotImplementedError:
|
|
234
|
+
return _RaisingAgent(node, self)
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
# --------------------------------------------------------------------------- #
|
|
238
|
+
# BasePipeline — implements `Runnable`; owns the run loop
|
|
239
|
+
# --------------------------------------------------------------------------- #
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
class BasePipeline:
|
|
243
|
+
"""A Spec rendered as an executable pipeline by the kit's run loop.
|
|
244
|
+
|
|
245
|
+
The loop only calls the adapter's hooks — it never inspects framework
|
|
246
|
+
specifics — so the same loop ranges over a stateless pipeline MAS
|
|
247
|
+
(text-in/text-out, default hooks) and a file-staged one (Lumina, override
|
|
248
|
+
order/prompt/context). ``run`` returns a full `Trace` (ok or not); a mid-run
|
|
249
|
+
crash flushes a partial trace with the Steps that ran (E4) — never lost.
|
|
250
|
+
"""
|
|
251
|
+
|
|
252
|
+
def __init__(
|
|
253
|
+
self,
|
|
254
|
+
spec: m.Spec,
|
|
255
|
+
middleware: TracingMiddleware,
|
|
256
|
+
agents: dict[str, BaseAgent],
|
|
257
|
+
adapter: BaseHostAdapter,
|
|
258
|
+
) -> None:
|
|
259
|
+
self._spec = spec
|
|
260
|
+
self._mw = middleware
|
|
261
|
+
self._agents = agents
|
|
262
|
+
self._adapter = adapter
|
|
263
|
+
self._run_counter = 0
|
|
264
|
+
|
|
265
|
+
def run(self, task: Task) -> m.Trace:
|
|
266
|
+
sid = self._spec.spec_id or self._spec.compute_spec_id()
|
|
267
|
+
rid = run_id(sid, task.task_id, self._run_counter)
|
|
268
|
+
self._run_counter += 1
|
|
269
|
+
self._mw.begin_run(rid, self._spec, task.task_id)
|
|
270
|
+
|
|
271
|
+
present = {n.node_id for n in self._spec.nodes}
|
|
272
|
+
ctx = RunContext(last_content=task.input)
|
|
273
|
+
final_output: str | None = None
|
|
274
|
+
try:
|
|
275
|
+
with self._adapter.stage_context():
|
|
276
|
+
for node_id in self._adapter.execution_order(self._spec):
|
|
277
|
+
if node_id not in present or node_id not in self._agents:
|
|
278
|
+
continue # a remove_node mutation simply skips a step
|
|
279
|
+
prompt = self._adapter.resolve_prompt(node_id, ctx)
|
|
280
|
+
wrapped = self._mw.wrap(self._agents[node_id])
|
|
281
|
+
resp = wrapped.invoke(prompt)
|
|
282
|
+
ctx.advance(node_id, resp)
|
|
283
|
+
final_output = resp.text
|
|
284
|
+
return self._mw.end_run(final_output, ok=True, error=None)
|
|
285
|
+
except Exception as exc: # noqa: BLE001 — flush a partial trace (E4)
|
|
286
|
+
return self._mw.end_run(final_output, ok=False, error=repr(exc))
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
# --------------------------------------------------------------------------- #
|
|
290
|
+
# Unknown-node agent — structural mutations that can't yet run
|
|
291
|
+
# --------------------------------------------------------------------------- #
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
class _RaisingAgent(BaseAgent):
|
|
295
|
+
"""A node the adapter can't execute (an `add_node` it has no entry-point
|
|
296
|
+
for). `invoke` runs (so the trace records the step) but `call` raises,
|
|
297
|
+
surfacing as ok=False — scoring rejects it, structural changes stay
|
|
298
|
+
human-gated (I4)."""
|
|
299
|
+
|
|
300
|
+
def call(self, prompt: str, vote: "KnobVote") -> CallResult: # noqa: ARG002
|
|
301
|
+
raise RuntimeError(
|
|
302
|
+
f"Adapter {type(self._adapter).__name__} can't execute node "
|
|
303
|
+
f"{self.node_id!r}: not in its node map (add_node mutations aren't "
|
|
304
|
+
f"runnable yet; the run errors and is rejected by scoring)."
|
|
305
|
+
)
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
__all__ = [
|
|
309
|
+
"RunContext", "CallResult", "BaseAgent", "BaseHostAdapter",
|
|
310
|
+
"BasePipeline", "run_id", "topo_order", "cfg_decay",
|
|
311
|
+
]
|