archforge-optimizer 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. archforge/__init__.py +76 -0
  2. archforge/__main__.py +10 -0
  3. archforge/architect.py +442 -0
  4. archforge/cli.py +881 -0
  5. archforge/config.py +140 -0
  6. archforge/config_init.py +150 -0
  7. archforge/diff.py +206 -0
  8. archforge/engine.py +444 -0
  9. archforge/gatekeeper.py +290 -0
  10. archforge/host/__init__.py +20 -0
  11. archforge/host/adapters/__init__.py +41 -0
  12. archforge/host/adapters/base.py +311 -0
  13. archforge/host/adapters/helpers.py +163 -0
  14. archforge/host/adapters/langgraph.py +726 -0
  15. archforge/host/base.py +105 -0
  16. archforge/host/fake.py +380 -0
  17. archforge/judge/__init__.py +20 -0
  18. archforge/judge/base.py +257 -0
  19. archforge/judge/scripted.py +145 -0
  20. archforge/lint.py +180 -0
  21. archforge/llm/__init__.py +65 -0
  22. archforge/llm/_common.py +94 -0
  23. archforge/llm/anthropic.py +90 -0
  24. archforge/llm/base.py +90 -0
  25. archforge/llm/gemini.py +112 -0
  26. archforge/llm/groq.py +63 -0
  27. archforge/llm/openai.py +63 -0
  28. archforge/llm/scripted.py +134 -0
  29. archforge/middleware.py +181 -0
  30. archforge/models.py +435 -0
  31. archforge/mutate.py +214 -0
  32. archforge/otel.py +613 -0
  33. archforge/runlog.py +103 -0
  34. archforge/runner.py +153 -0
  35. archforge/spec_builder.py +126 -0
  36. archforge/stores/__init__.py +22 -0
  37. archforge/stores/_jsonl.py +81 -0
  38. archforge/stores/attempt_store.py +161 -0
  39. archforge/stores/spec_store.py +188 -0
  40. archforge/stores/trace_store.py +42 -0
  41. archforge/suite.py +248 -0
  42. archforge/userconfig.py +144 -0
  43. archforge_optimizer-0.1.0.dist-info/METADATA +420 -0
  44. archforge_optimizer-0.1.0.dist-info/RECORD +47 -0
  45. archforge_optimizer-0.1.0.dist-info/WHEEL +4 -0
  46. archforge_optimizer-0.1.0.dist-info/entry_points.txt +2 -0
  47. archforge_optimizer-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,290 @@
1
+ """The Gatekeeper — the P-E-C "Commit" step (spec §3, §4, §6, §8).
2
+
3
+ Lone enforcer of "fail closed to the incumbent": it is the only thing that ever
4
+ moves the `active` pointer (via SpecStore.set_active) or flips an Attempt's
5
+ verdict beyond INITIAL. Its `decide(...)` returns a `Decision` naming the action;
6
+ `apply(...)` executes it (or you can decide without applying to inspect).
7
+
8
+ Decision rules (thresholds from `m.Thresholds`):
9
+ * unrunnable (candidate suite >ε crashed) -> DISCARD (before any margin math, E4)
10
+ * cross-rubric/cross-suite -> DISCARD (the only valid comparison is same
11
+ rubric + same suite; I5 — never silently compare across rubrics)
12
+ * candidate_mean - incumbent_mean < τ -> DISCARD (E1 noise inside τ)
13
+ * win + scope SMALL -> AUTO_PROMOTE (active = candidate)
14
+ * win + scope STRUCTURAL -> QUEUE_HUMAN (never auto-promote, I4)
15
+ * promoted-then-regresses (≥ δ on the standard suite) -> ROLLBACK
16
+ (active = parent, candidate archived, verdict ROLLED_BACK; pointer swap, E6)
17
+
18
+ Structural changes are queued even on a clear win — the human gate the user
19
+ locked in. `apply` returns the updated Attempt (verdict set) so the orchestrator
20
+ can record the cycle's outcome. Rollback uses the lineage pointer; no Spec is
21
+ ever deleted (archived, never deleted).
22
+
23
+ Human approval (`approve`) is the second path — besides AUTO_PROMOTE/ROLLBACK —
24
+ that may move `active`: a queued (PENDING_HUMAN) structural change the human
25
+ accepts becomes the incumbent; a rejection (`reject`) leaves the incumbent alone
26
+ and archives the candidate as a recorded dead end.
27
+ """
28
+
29
+ from __future__ import annotations
30
+
31
+ from enum import Enum
32
+ from typing import Any
33
+
34
+ from pydantic import BaseModel, ConfigDict
35
+
36
+ import archforge.models as m
37
+ from archforge.stores.attempt_store import AttemptStore
38
+ from archforge.stores.spec_store import SpecStore
39
+
40
+
41
+ class Action(str, Enum):
42
+ """What the Gatekeeper decided to do with this candidate."""
43
+
44
+ AUTO_PROMOTE = "auto_promote" # small win -> active = candidate
45
+ QUEUE_HUMAN = "queue_human" # structural win -> Approval Queue
46
+ DISCARD = "discard" # loss / unrunnable / cross-rubric
47
+ ROLLBACK = "rollback" # promoted-then-regressed (E6)
48
+
49
+
50
+ class Decision(BaseModel):
51
+ """The Gatekeeper's verdict on a candidate. ``Gatekeeper.apply_decision`` executes it."""
52
+
53
+ model_config = ConfigDict(extra="allow")
54
+
55
+ action: Action
56
+ attempt_id: str | None = None # the candidate Attempt this decides
57
+ reason: str # human-readable, surfaced to the report
58
+ margin: float = 0.0 # candidate_mean - incumbent_mean (signed)
59
+ by_rule: str = "" # which rule fired ("auto_promote", "rollback", ...)
60
+
61
+ def __repr__(self) -> str: # pragma: no cover (debug aid)
62
+ return (f"Decision(action={self.action.value}, margin={self.margin:+.3f}, "
63
+ f"rule={self.by_rule}, reason={self.reason!r})")
64
+
65
+
66
+ class _AttemptNotFoundError(KeyError):
67
+ """The candidate Attempt given to the Gatekeeper was never persisted."""
68
+
69
+
70
+ class Gatekeeper:
71
+ """The lone enforcer of promotion + rollback.
72
+
73
+ Constructed once per run with the (SpecStore, AttemptStore) it mutates and a
74
+ `Thresholds` (τ, δ, ε). Stateless across calls otherwise — the incumbent + a
75
+ candidate's SuiteRun are passed in `decide`.
76
+ """
77
+
78
+ def __init__(
79
+ self,
80
+ spec_store: SpecStore,
81
+ attempt_store: AttemptStore,
82
+ *,
83
+ thresholds: m.Thresholds | None = None,
84
+ ) -> None:
85
+ self._specs = spec_store
86
+ self._attempts = attempt_store
87
+ self._th = thresholds or m.Thresholds()
88
+
89
+ # ----------------------------------------------------------------- decide
90
+ def decide(
91
+ self,
92
+ attempt_id: str,
93
+ candidate: Any, # SuiteRun (typed loosely to avoid an import cycle)
94
+ incumbent: "m.SuiteRun | None" = None,
95
+ ) -> Decision:
96
+ """Decide the fate of a candidate `SuiteRun` vs the incumbent `SuiteRun`.
97
+
98
+ Both are `SuiteRun` objects (archforge.suite). `incumbent=None` means
99
+ "no incumbent yet" — only AUTO_PROMOTE-like forward progress is valid,
100
+ but a structural candidate still queues for human review.
101
+ """
102
+ # NOTE: `m.SuiteRun` lives in archforge.suite, not archforge.models, so we
103
+ # accept it as `Any` here and pull attributes defensively (duck-typed).
104
+
105
+ if candidate.unrunnable:
106
+ return _decision(Action.DISCARD, attempt_id,
107
+ "candidate suite was unrunnable (>ε tasks crashed/unscored)",
108
+ rule="unrunnable")
109
+
110
+ if incumbent is not None and not _same_geometry(incumbent, candidate):
111
+ return _decision(Action.DISCARD, attempt_id,
112
+ "candidate vs incumbent differ in rubric or suite "
113
+ "(cross-geometry comparison blocked, I5)",
114
+ rule="cross_geometry")
115
+
116
+ inc_mean = incumbent.mean if incumbent is not None else 0.0
117
+ margin = candidate.mean - inc_mean
118
+ cand_scope = self._candidate_scope(attempt_id)
119
+
120
+ if margin < self._th.tau:
121
+ return _decision(Action.DISCARD, attempt_id,
122
+ f"candidate did not clear margin τ={self._th.tau} "
123
+ f"(margin {margin:+.3f}); noise not promoted (E1)",
124
+ margin=margin, rule="below_margin")
125
+
126
+ # win past τ
127
+ if cand_scope is m.Scope.STRUCTURAL:
128
+ return _decision(Action.QUEUE_HUMAN, attempt_id,
129
+ f"structural win by {margin:+.3f} >= τ but structural "
130
+ f"changes require human approval (I4)",
131
+ margin=margin, rule="structural_wins_queue")
132
+ return _decision(Action.AUTO_PROMOTE, attempt_id,
133
+ f"small win by {margin:+.3f} >= τ={self._th.tau}; promoted",
134
+ margin=margin, rule="auto_promote")
135
+
136
+ # ----------------------------------------------------------------- apply
137
+ def apply_decision(self, decision: Decision) -> m.Attempt:
138
+ """Execute a Decision against the stores and return the updated Attempt."""
139
+
140
+ if decision.attempt_id is None:
141
+ raise ValueError("cannot apply a Decision with no attempt_id")
142
+ att = self._attempts.require(decision.attempt_id)
143
+
144
+ if decision.action is Action.AUTO_PROMOTE:
145
+ spec_id = att.candidate_spec_id
146
+ self._specs.set_active(spec_id)
147
+ verdict = m.Verdict.PROMOTED
148
+ elif decision.action is Action.QUEUE_HUMAN:
149
+ verdict = m.Verdict.PENDING_HUMAN
150
+ elif decision.action is Action.DISCARD:
151
+ verdict = m.Verdict.REJECTED
152
+ elif decision.action is Action.ROLLBACK:
153
+ self._rollback(att)
154
+ verdict = m.Verdict.ROLLED_BACK
155
+ else: # pragma: no cover (enum exhaustive)
156
+ raise ValueError(f"unknown action {decision.action}")
157
+
158
+ return self._attempts.set_verdict(decision.attempt_id, verdict)
159
+
160
+ # ----------------------------------------------------------------- human gate
161
+ def approve(self, attempt_id: str) -> m.Attempt:
162
+ """Human approves a queued (PENDING_HUMAN) structural change (spec I4).
163
+
164
+ This is the one path besides AUTO_PROMOTE/ROLLBACK that may move the
165
+ `active` pointer — the human gate the user locked in: a structural win is
166
+ never auto-promoted, only ever promoted through here. `active` -> the
167
+ candidate, verdict -> PROMOTED. Idempotent for an already-promoted attempt
168
+ (a no-op). Raises `ValueError` if the attempt is not queued, so the CLI
169
+ cannot rewrite history (only decide what the gate queued).
170
+ """
171
+
172
+ att = self._attempts.require(attempt_id)
173
+ if att.verdict is m.Verdict.PROMOTED:
174
+ return att # already approved (idempotent)
175
+ if att.verdict is not m.Verdict.PENDING_HUMAN:
176
+ raise ValueError(
177
+ f"cannot approve attempt {attempt_id}: verdict is "
178
+ f"{att.verdict.value}; only PENDING_HUMAN (queued) changes can be approved"
179
+ )
180
+ self._specs.set_active(att.candidate_spec_id)
181
+ return self._attempts.set_verdict(attempt_id, m.Verdict.PROMOTED)
182
+
183
+ def reject(self, attempt_id: str, *, reason: str = "human-reject") -> m.Attempt:
184
+ """Human rejects a queued structural change: verdict -> REJECTED, active unchanged.
185
+
186
+ The incumbent is left alone. The candidate Spec is archived (never
187
+ deleted) with `reason="human-reject"` so the dead end is recorded for
188
+ lineage/dedup queries. Idempotent for an already-rejected attempt.
189
+ Raises `ValueError` if the attempt is not queued.
190
+ """
191
+
192
+ att = self._attempts.require(attempt_id)
193
+ if att.verdict is m.Verdict.REJECTED:
194
+ return att
195
+ if att.verdict is not m.Verdict.PENDING_HUMAN:
196
+ raise ValueError(
197
+ f"cannot reject attempt {attempt_id}: verdict is "
198
+ f"{att.verdict.value}; only PENDING_HUMAN (queued) changes can be rejected"
199
+ )
200
+ self._specs.archive(att.candidate_spec_id, reason=reason)
201
+ return self._attempts.set_verdict(attempt_id, m.Verdict.REJECTED)
202
+
203
+ # ----------------------------------------------------------------- rollback (E6)
204
+ def rollback(
205
+ self,
206
+ attempt_id: str,
207
+ *,
208
+ regressed_mean: float,
209
+ pre_promotion_mean: float,
210
+ ) -> Decision:
211
+ """Decide+apply a rollback after a promoted Spec regressed (E6).
212
+
213
+ Rolls back iff `pre_promotion_mean - regressed_mean >= δ` on the standard
214
+ suite (same rubric+suite, I5). Otherwise the regression is within the
215
+ noise floor and the incumbent is left alone. Returns the Decision (which
216
+ has already been applied if action is ROLLBACK).
217
+ """
218
+
219
+ drop = pre_promotion_mean - regressed_mean
220
+ # Idempotency: a repeated regression check on an already-rolled-back
221
+ # attempt must not downgrade its verdict or move the active pointer
222
+ # again. Return a benign DISCARD decision without touching the stores.
223
+ current_verdict = self._attempts.require(attempt_id).verdict
224
+ if current_verdict is m.Verdict.ROLLED_BACK:
225
+ return _decision(Action.DISCARD, attempt_id,
226
+ "regression check on an already-rolled-back attempt; "
227
+ "no further action",
228
+ margin=-drop, rule="already_rolled_back")
229
+
230
+ if drop < self._th.delta:
231
+ # Regression is within the δ noise floor: the promotion stands, the
232
+ # incumbent (the promoted Spec) is left in place, and the attempt's
233
+ # verdict is NOT mutated — a real small-won, the dip is just noise.
234
+ return _decision(Action.DISCARD, attempt_id,
235
+ f"regression {drop:+.3f} < δ={self._th.delta}; within the "
236
+ f"noise floor, incumbent left alone",
237
+ margin=-drop, rule="regression_within_floor")
238
+
239
+ d = _decision(Action.ROLLBACK, attempt_id,
240
+ f"regression {drop:+.3f} >= δ={self._th.delta}; rolling back to "
241
+ f"parent (pointer swap, archived, never deleted)",
242
+ margin=-drop, rule="rollback")
243
+ att = self._attempts.require(attempt_id)
244
+ self._rollback(att)
245
+ self._attempts.set_verdict(attempt_id, m.Verdict.ROLLED_BACK)
246
+ return d
247
+
248
+ def _rollback(self, att: m.Attempt) -> None:
249
+ """Active -> parent; candidate archived. Pointer swap, never deleted (E6/I3)."""
250
+
251
+ spec = self._specs.get(att.candidate_spec_id)
252
+ parent_id = spec.parent_spec_id
253
+ if parent_id is None:
254
+ # root incumbent regressed: nothing to roll back to; leave pointer, mark rejected
255
+ self._specs.archive(att.candidate_spec_id, reason="rollback_rootless")
256
+ return
257
+ if not self._specs.has(parent_id):
258
+ raise _AttemptNotFoundError(
259
+ f"rollback target parent {parent_id} not in store"
260
+ )
261
+ self._specs.set_active(parent_id) # pointer swap — the incumbent reverts
262
+ self._specs.archive(att.candidate_spec_id, reason="rollback")
263
+
264
+ # ----------------------------------------------------------------- helpers
265
+ def _candidate_scope(self, attempt_id: str) -> m.Scope:
266
+ att = self._attempts.require(attempt_id)
267
+ return att.change.scope
268
+
269
+
270
+ # --------------------------------------------------------------------------- #
271
+ # helpers (module-private)
272
+ # --------------------------------------------------------------------------- #
273
+
274
+
275
+ def _decision(
276
+ action: Action, attempt_id: str | None, reason: str,
277
+ *, margin: float = 0.0, rule: str = "",
278
+ ) -> Decision:
279
+ return Decision(action=action, attempt_id=attempt_id, reason=reason,
280
+ margin=margin, by_rule=rule)
281
+
282
+
283
+ def _same_geometry(inc: Any, cand: Any) -> bool:
284
+ """Invariant I5: comparisons only valid under identical (rubric, suite)."""
285
+
286
+ return (getattr(inc, "rubric_id", None) == getattr(cand, "rubric_id", None)
287
+ and getattr(inc, "suite_id", None) == getattr(cand, "suite_id", None))
288
+
289
+
290
+ __all__ = ["Action", "Decision", "Gatekeeper"]
@@ -0,0 +1,20 @@
1
+ """Host multi-agent system integration (spec §3, §4).
2
+
3
+ ArchForge does not depend on any specific agentic framework. A host MAS
4
+ implements the `HostMAS` contract: it can build an executable pipeline from a
5
+ `Spec` and run a `Task` through it, emitting one `Step` per agent. The
6
+ `TracingMiddleware` attaches as both observer (records steps to TraceStore) and
7
+ config bridge (applies the live Spec's prompt/model/knobs to each agent at
8
+ invoke time — so evolving the pipeline is swapping which Spec the host uses).
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from archforge.host.base import Agent, AgentResponse, HostMAS, Runnable, Task
14
+ from archforge.host.fake import (
15
+ FakeAgent, FakeHostMAS, FakeRuleAgent, FakeRetrieverAgent, FakeToolAgent,
16
+ )
17
+
18
+ __all__ = ["Agent", "AgentResponse", "HostMAS", "Runnable", "Task",
19
+ "FakeHostMAS", "FakeAgent", "FakeRuleAgent", "FakeRetrieverAgent",
20
+ "FakeToolAgent"]
@@ -0,0 +1,41 @@
1
+ """archforge.host.adapters — the reusable adapter kit.
2
+
3
+ The single place a new MAS author looks: subclass `BaseHostAdapter` + a small
4
+ `BaseAgent` per node, fill `call()` (the node's real work), and override the
5
+ hooks (`execution_order` / `resolve_prompt` / `stage_context`) only when the
6
+ MAS's data-shaping is non-default. The kit owns the run loop, content-decouple,
7
+ config-decay, perf, and partial-trace flush — the recurring scaffolding every
8
+ adapter previously re-derived.
9
+
10
+ `helpers` is also re-exported so an author extending the loop's primitives
11
+ (topo order, run-id, config decay) does so against one source.
12
+ """
13
+ from archforge.host.adapters.base import (
14
+ BaseAgent, BaseHostAdapter, BasePipeline, CallResult, RunContext,
15
+ )
16
+ from archforge.host.adapters.helpers import (
17
+ KnobVote, cfg_as_kwargs, cfg_decay, estimate_tokens, run_id, topo_order,
18
+ )
19
+ # LangGraph adapter — re-exported eagerly. The module is langgraph-FREE (it
20
+ # imports no langgraph types; the dependency enters only when a concrete app's
21
+ # `graph_factory` builds the real graph), so importing it here keeps
22
+ # `import archforge` framework-free.
23
+ from archforge.host.adapters.langgraph import (
24
+ EdgeSpec, LangGraphApp, LangGraphHostAdapter, LangGraphRunnable, Nd,
25
+ NodeIdMap, build_optimized_envelope, export_optimized, export_spec_sidecar,
26
+ load_optimized, load_spec_sidecar, node_ids,
27
+ )
28
+ # `wrapped` is deliberately NOT re-exported at the package level: a MAS authors
29
+ # the id ONCE in its `build_graph` via `add(name, fn)` and imports `wrapped`
30
+ # from `archforge.host.adapters.langgraph` (the adapter module the MAS already
31
+ # touches) — never from `archforge.otel` directly, and never from this package.
32
+
33
+ __all__ = [
34
+ "BaseHostAdapter", "BaseAgent", "BasePipeline", "CallResult", "RunContext",
35
+ "run_id", "topo_order", "cfg_decay", "cfg_as_kwargs", "KnobVote",
36
+ "estimate_tokens",
37
+ "Nd", "EdgeSpec", "LangGraphApp", "LangGraphHostAdapter",
38
+ "LangGraphRunnable", "NodeIdMap", "node_ids",
39
+ "export_spec_sidecar", "load_spec_sidecar",
40
+ "build_optimized_envelope", "export_optimized", "load_optimized",
41
+ ]
@@ -0,0 +1,311 @@
1
+ """The ArchForge adapter kit — a reusable implementation of the host seam.
2
+
3
+ `HostMAS` / `Agent` / `Runnable` (in ``host/base.py``) are the *contract*; this
4
+ module is a *reusable implementation of it* that owns the scaffolding every MAS
5
+ adapter repeats:
6
+
7
+ * one ``Agent`` per Spec node;
8
+ * the run loop: ``begin_run`` → ordered walk → ``end_run``, with a monotonic
9
+ run-id (unique across R-repeats), partial-trace flush on any crash (E4);
10
+ * the **content-decouple**: score the agent's *content*, thread its
11
+ *plumbing* onward via an ``AgentResponse`` extra (``extra="allow"`` lets a
12
+ ``thread`` field ride without a model change);
13
+ * the **config-decay** (phase-2 rule): thread model/temperature/max_tokens
14
+ always (base run == agent default → behaviour-preserving) and system_prompt
15
+ only when mutated.
16
+
17
+ The genuinely-MAS-specific bits are exposed as a few imperative hooks an adapter
18
+ overrides when its MAS needs them — *not* a declarative graph interpreter (so
19
+ hard data-shaping — file-stage handoffs, multi-input joins — is just imperative
20
+ code in a hook, never a drop off a declarative cliff).
21
+ """
22
+ from __future__ import annotations
23
+
24
+ import time
25
+ from contextlib import nullcontext
26
+ from dataclasses import dataclass, field
27
+ from typing import Any, Callable
28
+
29
+ import archforge.models as m
30
+ from archforge.host.adapters.helpers import cfg_decay, estimate_tokens, run_id, topo_order
31
+ from archforge.host.base import AgentResponse, HostMAS, Runnable, Task
32
+ from archforge.middleware import TracingMiddleware
33
+
34
+
35
+ # --------------------------------------------------------------------------- #
36
+ # Run context — what the loop carries between nodes
37
+ # --------------------------------------------------------------------------- #
38
+
39
+
40
+ @dataclass
41
+ class RunContext:
42
+ """The state threaded through one run, advanced by the loop and read by
43
+ the adapter's ``resolve_prompt``.
44
+
45
+ ``last_content`` is what each step's response *recorded* (what the Judge
46
+ scores); ``last_thread`` is what each step *threads onward* (plumbing: a
47
+ file path, a ref… — usually the same as content for text-in/text-out MASes,
48
+ different for file-staged ones like Lumina). ``per_node`` lets a join node
49
+ recall a sibling threaded earlier (the report 3-way join reads sibling
50
+ gap/cross paths here).
51
+ """
52
+
53
+ last_content: str
54
+ last_thread: str | None = None
55
+ per_node: dict[str, Any] = field(default_factory=dict)
56
+
57
+ def advance(self, node_id: str, resp: AgentResponse) -> None:
58
+ self.last_content = resp.text
59
+ # `thread` rides as an AgentResponse extra (extra="allow"); fall back to
60
+ # the scored content if the adapter didn't set one (text-in/text-out).
61
+ self.last_thread = getattr(resp, "thread", None) or resp.text
62
+ self.per_node[node_id] = self.last_thread
63
+
64
+
65
+ # --------------------------------------------------------------------------- #
66
+ # CallResult — the content-decouple made first-class
67
+ # --------------------------------------------------------------------------- #
68
+
69
+
70
+ @dataclass
71
+ class CallResult:
72
+ """What an agent's ``call`` returns: ``thread`` flows to the next node
73
+ (plumbing), ``content`` is what the trace records and the Judge scores.
74
+
75
+ Promoted from the Lumina adapter's ``RunResult`` so every kit-made adapter
76
+ gets content-decouple for free instead of re-inventing it.
77
+ """
78
+
79
+ thread: str | None
80
+ content: str
81
+
82
+
83
+ # --------------------------------------------------------------------------- #
84
+ # BaseAgent — implements `Agent`; owns invoke; author overrides `call`
85
+ # --------------------------------------------------------------------------- #
86
+
87
+
88
+ class BaseAgent:
89
+ """One node's executor. The kit owns ``invoke`` (the content-decouple +
90
+ config-decay + perf measurement); the adapter fills ``call`` — the node's
91
+ real work — and is handed a config map it forwards to its agent call.
92
+ """
93
+
94
+ def __init__(self, node: m.Node, adapter: "BaseHostAdapter") -> None:
95
+ self._node = node
96
+ self._adapter = adapter
97
+
98
+ @property
99
+ def node_id(self) -> str:
100
+ return self._node.node_id
101
+
102
+ @property
103
+ def role(self) -> str:
104
+ return self._node.role
105
+
106
+ def call(self, prompt: str, vote: "KnobVote") -> CallResult: # noqa: ARG002
107
+ """The one abstract hook: run the node, return (thread, content).
108
+
109
+ ``vote`` is the DECAYED live config (a `KnobVote`: model / temperature /
110
+ max_tokens / retries / system_prompt, each ``None`` meaning "use your
111
+ default"). The adapter binds exactly the keys its agent accepts — the
112
+ kit owns the decay logic, not the agent's signature. Subclasses override.
113
+ """
114
+ raise NotImplementedError(f"{type(self).__name__}.call() not implemented")
115
+
116
+ def invoke(
117
+ self,
118
+ prompt: str,
119
+ *,
120
+ system_prompt: str | None,
121
+ model: str | None,
122
+ knobs: m.Knobs | None,
123
+ tools: list[str] | None,
124
+ kind: m.NodeKind = m.NodeKind.LLM,
125
+ ) -> AgentResponse:
126
+ """Kit-owned: fulfils the `Agent` protocol. Applies config-decay, calls
127
+ the node's real work, packs an AgentResponse that records *content* and
128
+ carries *plumbing* onward via the `thread` extra.
129
+
130
+ `kind` is accepted for protocol parity with the middleware (which threads
131
+ the live node's `NodeKind` to every wrapped agent) and defaults to `LLM`
132
+ so existing kit adapters keep working unchanged. It is UNUSED here: a kit
133
+ adapter that needs kind-specific behaviour reads `self._node.kind` in its
134
+ `call` override (the node is already held on the agent).
135
+ """
136
+ vote = self._adapter.cfg_for(self.node_id, system_prompt, model, knobs, tools)
137
+ t0 = time.perf_counter()
138
+ cr = self.call(prompt, vote)
139
+ latency = round((time.perf_counter() - t0) * 1000.0, 3)
140
+ # `text` is what the trace records + the Judge scores (content); the
141
+ # `thread` extra is what flows to the next node (plumbing).
142
+ return AgentResponse(
143
+ text=cr.content,
144
+ tool_calls=[],
145
+ thread=cr.thread if cr.thread is not None else cr.content,
146
+ perf=m.StepPerf(tokens=estimate_tokens(cr.content), latency_ms=latency),
147
+ )
148
+
149
+
150
+ def _estimate(text: str) -> int: # kept as a thin alias for any external callers
151
+ return estimate_tokens(text)
152
+
153
+
154
+ # --------------------------------------------------------------------------- #
155
+ # BaseHostAdapter — implements `HostMAS`; owns instantiate; delegates the run
156
+ # --------------------------------------------------------------------------- #
157
+
158
+
159
+ class BaseHostAdapter:
160
+ """A `HostMAS` that builds a `BasePipeline` from a Spec.
161
+
162
+ An adapter subclasses this and overrides the hooks that are non-default for
163
+ its MAS:
164
+
165
+ * ``make_agent`` — wrap one framework agent per Spec node (the per-node
166
+ factory that chooses the right ``BaseAgent`` subclass).
167
+ * ``execution_order`` — default: topological from edges; override for a
168
+ fixed sequence (Lumina's 7-step).
169
+ * ``resolve_prompt`` — default: the last step's content; override for a
170
+ file path when a node needs upstream plumbing (joins, file-stage handoffs).
171
+ * ``stage_context`` — default: none; override for a per-run scratch area
172
+ (Lumina: a temp dir + chdir, released on exit).
173
+
174
+ The run loop, content-decouple, config-decay, partial-trace flush, perf —
175
+ none of those are the adapter's concern; the kit owns them.
176
+ """
177
+
178
+ #: The seeded/default prompts, so `cfg_decay` can tell a *mutated* prompt
179
+ #: (an Architect prompt_edit) from the default and only thread the former.
180
+ #: Subclasses populate this (often from their base Spec).
181
+ base_prompts: dict[str, str] = {}
182
+
183
+ # ---- hooks (override when the MAS needs non-default behaviour) -------- #
184
+
185
+ def make_agent(self, node: m.Node) -> BaseAgent:
186
+ """Pick the `BaseAgent` subclass for `node` (one per Spec node)."""
187
+ raise NotImplementedError("make_agent must return a BaseAgent for each node")
188
+
189
+ def execution_order(self, spec: m.Spec) -> list[str]:
190
+ """Order to run nodes in. Default: topological from the Spec's edges."""
191
+ return topo_order(spec)
192
+
193
+ def resolve_prompt(self, node_id: str, ctx: RunContext) -> str:
194
+ """What `node_id` receives as its prompt. Default: the last step's
195
+ *content* (text-in/text-out). Override to pass a file path / sibling
196
+ ref when the node needs upstream plumbing (joins, file-stage handoffs)."""
197
+ return ctx.last_content
198
+
199
+ def stage_context(self):
200
+ """A context manager wrapping one run. Default: none. Override for a
201
+ per-run scratch area — e.g. a temp dir + chdir (Lumina) — that this
202
+ releases on exit. The loop's `finally` cleanup lives here."""
203
+ return nullcontext()
204
+
205
+ # ---- kit-internal seam (rarely overridden) --------------------------- #
206
+
207
+ def cfg_for(
208
+ self, node_id: str, system_prompt: str | None, model: str | None,
209
+ knobs: m.Knobs | None, tools: list[str] | None,
210
+ ) -> "KnobVote":
211
+ """Live-config → the DECAYED values the adapter's ``call`` forwards to
212
+ its agent. Defaults to the phase-2 `cfg_decay` rule; returns a
213
+ ``KnobVote`` so each adapter binds EXACTLY the keys its agent accepts
214
+ (the kit owns the decay logic, not the agent's signature). Override only
215
+ if the MAS's config-decay differs."""
216
+ return cfg_decay(node_id, system_prompt, model, knobs, tools,
217
+ base_prompts=self.base_prompts)
218
+
219
+ # ---- kit-owned: the existing HostMAS.instantiate contract, un-touched -- #
220
+
221
+ def instantiate(self, spec: m.Spec, middleware: TracingMiddleware) -> Runnable:
222
+ agents: dict[str, BaseAgent] = {}
223
+ for node in spec.nodes:
224
+ agents[node.node_id] = self._agent_for(node)
225
+ return BasePipeline(spec, middleware, agents, self)
226
+
227
+ def _agent_for(self, node: m.Node) -> BaseAgent:
228
+ """One agent per Spec node; an unknown node_id (an `add_node` we can't
229
+ run) becomes a _RaisingAgent, surfacing as an errored run that scoring
230
+ rejects — structural changes stay human-gated (I4)."""
231
+ try:
232
+ return self.make_agent(node)
233
+ except NotImplementedError:
234
+ return _RaisingAgent(node, self)
235
+
236
+
237
+ # --------------------------------------------------------------------------- #
238
+ # BasePipeline — implements `Runnable`; owns the run loop
239
+ # --------------------------------------------------------------------------- #
240
+
241
+
242
+ class BasePipeline:
243
+ """A Spec rendered as an executable pipeline by the kit's run loop.
244
+
245
+ The loop only calls the adapter's hooks — it never inspects framework
246
+ specifics — so the same loop ranges over a stateless pipeline MAS
247
+ (text-in/text-out, default hooks) and a file-staged one (Lumina, override
248
+ order/prompt/context). ``run`` returns a full `Trace` (ok or not); a mid-run
249
+ crash flushes a partial trace with the Steps that ran (E4) — never lost.
250
+ """
251
+
252
+ def __init__(
253
+ self,
254
+ spec: m.Spec,
255
+ middleware: TracingMiddleware,
256
+ agents: dict[str, BaseAgent],
257
+ adapter: BaseHostAdapter,
258
+ ) -> None:
259
+ self._spec = spec
260
+ self._mw = middleware
261
+ self._agents = agents
262
+ self._adapter = adapter
263
+ self._run_counter = 0
264
+
265
+ def run(self, task: Task) -> m.Trace:
266
+ sid = self._spec.spec_id or self._spec.compute_spec_id()
267
+ rid = run_id(sid, task.task_id, self._run_counter)
268
+ self._run_counter += 1
269
+ self._mw.begin_run(rid, self._spec, task.task_id)
270
+
271
+ present = {n.node_id for n in self._spec.nodes}
272
+ ctx = RunContext(last_content=task.input)
273
+ final_output: str | None = None
274
+ try:
275
+ with self._adapter.stage_context():
276
+ for node_id in self._adapter.execution_order(self._spec):
277
+ if node_id not in present or node_id not in self._agents:
278
+ continue # a remove_node mutation simply skips a step
279
+ prompt = self._adapter.resolve_prompt(node_id, ctx)
280
+ wrapped = self._mw.wrap(self._agents[node_id])
281
+ resp = wrapped.invoke(prompt)
282
+ ctx.advance(node_id, resp)
283
+ final_output = resp.text
284
+ return self._mw.end_run(final_output, ok=True, error=None)
285
+ except Exception as exc: # noqa: BLE001 — flush a partial trace (E4)
286
+ return self._mw.end_run(final_output, ok=False, error=repr(exc))
287
+
288
+
289
+ # --------------------------------------------------------------------------- #
290
+ # Unknown-node agent — structural mutations that can't yet run
291
+ # --------------------------------------------------------------------------- #
292
+
293
+
294
+ class _RaisingAgent(BaseAgent):
295
+ """A node the adapter can't execute (an `add_node` it has no entry-point
296
+ for). `invoke` runs (so the trace records the step) but `call` raises,
297
+ surfacing as ok=False — scoring rejects it, structural changes stay
298
+ human-gated (I4)."""
299
+
300
+ def call(self, prompt: str, vote: "KnobVote") -> CallResult: # noqa: ARG002
301
+ raise RuntimeError(
302
+ f"Adapter {type(self._adapter).__name__} can't execute node "
303
+ f"{self.node_id!r}: not in its node map (add_node mutations aren't "
304
+ f"runnable yet; the run errors and is rejected by scoring)."
305
+ )
306
+
307
+
308
+ __all__ = [
309
+ "RunContext", "CallResult", "BaseAgent", "BaseHostAdapter",
310
+ "BasePipeline", "run_id", "topo_order", "cfg_decay",
311
+ ]