archforge-optimizer 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. archforge/__init__.py +76 -0
  2. archforge/__main__.py +10 -0
  3. archforge/architect.py +442 -0
  4. archforge/cli.py +881 -0
  5. archforge/config.py +140 -0
  6. archforge/config_init.py +150 -0
  7. archforge/diff.py +206 -0
  8. archforge/engine.py +444 -0
  9. archforge/gatekeeper.py +290 -0
  10. archforge/host/__init__.py +20 -0
  11. archforge/host/adapters/__init__.py +41 -0
  12. archforge/host/adapters/base.py +311 -0
  13. archforge/host/adapters/helpers.py +163 -0
  14. archforge/host/adapters/langgraph.py +726 -0
  15. archforge/host/base.py +105 -0
  16. archforge/host/fake.py +380 -0
  17. archforge/judge/__init__.py +20 -0
  18. archforge/judge/base.py +257 -0
  19. archforge/judge/scripted.py +145 -0
  20. archforge/lint.py +180 -0
  21. archforge/llm/__init__.py +65 -0
  22. archforge/llm/_common.py +94 -0
  23. archforge/llm/anthropic.py +90 -0
  24. archforge/llm/base.py +90 -0
  25. archforge/llm/gemini.py +112 -0
  26. archforge/llm/groq.py +63 -0
  27. archforge/llm/openai.py +63 -0
  28. archforge/llm/scripted.py +134 -0
  29. archforge/middleware.py +181 -0
  30. archforge/models.py +435 -0
  31. archforge/mutate.py +214 -0
  32. archforge/otel.py +613 -0
  33. archforge/runlog.py +103 -0
  34. archforge/runner.py +153 -0
  35. archforge/spec_builder.py +126 -0
  36. archforge/stores/__init__.py +22 -0
  37. archforge/stores/_jsonl.py +81 -0
  38. archforge/stores/attempt_store.py +161 -0
  39. archforge/stores/spec_store.py +188 -0
  40. archforge/stores/trace_store.py +42 -0
  41. archforge/suite.py +248 -0
  42. archforge/userconfig.py +144 -0
  43. archforge_optimizer-0.1.0.dist-info/METADATA +420 -0
  44. archforge_optimizer-0.1.0.dist-info/RECORD +47 -0
  45. archforge_optimizer-0.1.0.dist-info/WHEEL +4 -0
  46. archforge_optimizer-0.1.0.dist-info/entry_points.txt +2 -0
  47. archforge_optimizer-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,726 @@
1
+ """ArchForge adapter for LangGraph apps — drives the REAL compiled graph.
2
+
3
+ LangGraph (https://github.com/langchain-ai/langgraph) runs a state machine: a
4
+ ``state -> partial_state`` node per superstep, routed by conditional edges and
5
+ runtime loops, driven via ``graph.stream(stream_mode="updates")``. Each stream
6
+ chunk is ``{node_name: partial_state_diff}``.
7
+
8
+ This adapter maps the real graph onto ArchForge's P-E-C loop WITHOUT flattening
9
+ it to the kit's flat topological walk (`BasePipeline`) — that would erase the
10
+ conditional routing and the runtime loop, the very behaviors worth tuning.
11
+ Instead it:
12
+
13
+ * drives ``graph.stream`` and records one ArchForge ``Step`` per emitted node
14
+ (surgery-free via ``mw._record_step``, like Lumina — ``archforge/`` untouched);
15
+ * injects the live Spec's knobs into the graph's ``initial_state`` so a
16
+ candidate's knob edits reach the REAL execution (the only viable shape:
17
+ ``graph.stream`` reveals a node only *after* it runs, so per-upcoming-node
18
+ contextvars are impossible — populate up-front, consult at call time);
19
+ * kind-aware cost: LLM nodes cost tokens (`estimate_tokens`); retriever/rule/
20
+ symbolic nodes cost ~0 tokens (their real cost is wall-clock `latency_ms`),
21
+ matching the mixed-non-LLM convention (bounded via ``--max-wall-ms-per-cycle``).
22
+
23
+ Knob routing (the adapter's one design rule — "describe, don't introspect"):
24
+ * EXTRAS (kind-specific params: a retriever's ``top_k``, a rule node's
25
+ ``threshold``) flow through STATE — the author maps each to a state key via
26
+ ``knob_to_state``, and the node reads ``state.get(key, default)``. ArchForge
27
+ live = inject into ``initial_state``; deploy = edit the node's consts.
28
+ * NAMED LLM knobs (``model``/``temperature``/``max_tokens``/``system_prompt``)
29
+ flow through a CALL-TIME injector the author wires (`apply_llm_config`),
30
+ consulted by the node's LLM call at call time — AEDE's pattern (a module-level
31
+ ``_NODE_CONFIG`` the groq client reads).
32
+
33
+ This module is **langgraph-FREE**: it imports no langgraph types. The langgraph
34
+ dependency enters only when a concrete app's ``graph_factory`` builds the real
35
+ graph. So ``import archforge`` stays framework-free and this module loads even
36
+ without langgraph installed (you just can't *run* a real app until it is).
37
+ """
38
+ from __future__ import annotations
39
+
40
+ import json
41
+ import time
42
+ from dataclasses import dataclass, field
43
+ from pathlib import Path
44
+ from typing import Any, Callable
45
+
46
+ import archforge.models as m
47
+ from archforge import userconfig as ucfg
48
+ from archforge.host.adapters.base import BaseHostAdapter
49
+ from archforge.host.adapters.helpers import KnobVote, estimate_tokens, run_id
50
+ from archforge.lint import lint
51
+ from archforge.middleware import TracingMiddleware
52
+
53
+ # Re-export the OTel cooperative seam so a MAS's ``build_graph`` imports
54
+ # ``wrapped`` from here (the adapter module it already touches) rather than
55
+ # reaching into ``archforge.otel`` directly. ``otel`` is import-lazy (it pulls
56
+ # NO OpenTelemetry at its top level), so merely naming it here keeps
57
+ # ``import archforge`` / this module OTel-free; its OTel imports run lazily only
58
+ # when ``_ensure_tracer`` / ``wrapped`` first execute. ``project`` /
59
+ # ``_shed_to_budget`` are imported lazily INSIDE the methods that use them
60
+ # (below), so they never load unless a budget is actually set.
61
+ from archforge.otel import wrapped as _otel_wrapped # noqa: E402 (lazy module; no eager OTel)
62
+
63
+
64
+ def wrapped(name: str, fn: Callable) -> Callable:
65
+ """Re-exported cooperative OTel seam: wraps ``fn`` to open an
66
+ ``archforge.node`` span so auto-instrumented SDK calls nest as children
67
+ (correlation by parent-link). A MAS's ``build_graph`` routes node fns
68
+ through ``add(name, fn) -> g.add_node(name, wrapped(name, fn))`` so the id
69
+ string is authored ONCE. No-op passthrough when OTel is unavailable.
70
+
71
+ See ``archforge.otel.wrapped`` for the implementation; this thin re-export
72
+ is the seam the MAS imports from (the adapter it already touches)."""
73
+ return _otel_wrapped(name, fn)
74
+
75
+ # The named LLM knobs flow through the call-time injector, NOT state. These are
76
+ # the Knobs model's defined fields to exclude when overlaying state-routed extras.
77
+ # Lifted into ``archforge.models._NAMED_KNOBS`` as the single source of truth
78
+ # (shared with ``archforge.diff``); referenced here as ``m._NAMED_KNOBS``.
79
+ _NAMED_KNOBS = m._NAMED_KNOBS
80
+
81
+
82
+ # --------------------------------------------------------------------------- #
83
+ # Declarative "description" surface — the MAS author fills this in
84
+ # --------------------------------------------------------------------------- #
85
+
86
+
87
+ @dataclass
88
+ class Nd:
89
+ """One node in the MAS's static roster. Fields mirror ``m.Node``.
90
+
91
+ ``knobs`` carries the SEEDED (incumbent) values — the extras that flow through
92
+ state, plus the named LLM knobs that flow through the call-time injector.
93
+ ``tunable`` (a field on ``Knobs``) names which EXTRAS the Architect may edit;
94
+ the named knobs are always editable (back-compat).
95
+ """
96
+ node_id: str
97
+ role: str
98
+ kind: m.NodeKind = m.NodeKind.LLM
99
+ knobs: m.Knobs = field(default_factory=m.Knobs)
100
+ model: str = ""
101
+ system_prompt: str = ""
102
+ tools: list[str] = field(default_factory=list)
103
+
104
+
105
+ @dataclass
106
+ class EdgeSpec:
107
+ """One static edge. CONDITIONAL edges carry a non-empty ``gate`` (a label the
108
+ linter checks for truthiness; the Forge does NOT evaluate it — the real graph
109
+ does). Runtime loops are documented in ``LangGraphApp.runtime_loops`` but
110
+ OMITTED from ``edges``: encoding a back-edge would trip the linter's ``cycle``
111
+ rule, and the real graph drives the loop anyway (the adapter sees each loop
112
+ iteration as an emitted node)."""
113
+ from_: str
114
+ to: str
115
+ kind: m.EdgeType = m.EdgeType.SEQUENCE
116
+ gate: str | None = None
117
+
118
+
119
+ # --------------------------------------------------------------------------- #
120
+ # NodeIdMap — keyed id source (single-authoring, fail-fast on rename)
121
+ # --------------------------------------------------------------------------- #
122
+
123
+
124
+ @dataclass(frozen=True)
125
+ class NodeIdMap:
126
+ """Keyed lookup over a compiled graph's real node names — the authority for
127
+ the node id string, so it is written ONCE (in ``build_graph``'s
128
+ ``add(name, fn)`` line) and sourced everywhere else via ``_GID["name"]``.
129
+
130
+ ``__getitem__`` validates against the compiled graph's node names (drops
131
+ langgraph's built-in ``__start__`` / ``__end__``) and raises ``KeyError``
132
+ listing the declared names on a miss → a rename in ``build_graph`` fails
133
+ LOUDLY at aede_app import (fail-fast per node), never a silently-mislabeled
134
+ run. Introspection is limited to node NAMES — the most stable surface — one
135
+ time at wiring; "describe, don't introspect" still protects the version-
136
+ fragile internals/routing the rest of the adapter avoids.
137
+ """
138
+
139
+ _names: tuple[str, ...]
140
+
141
+ def __getitem__(self, name: str) -> str:
142
+ if name not in self._names:
143
+ raise KeyError(
144
+ f"unknown node {name!r}; declared: {list(self._names)}"
145
+ )
146
+ return name # canonical id == the add_node name
147
+
148
+ def __iter__(self):
149
+ return iter(self._names)
150
+
151
+ def __len__(self) -> int:
152
+ return len(self._names)
153
+
154
+
155
+ def node_ids(graph: Any) -> NodeIdMap:
156
+ """Build a :class:`NodeIdMap` from a compiled langgraph graph's node names
157
+ (drops ``__start__`` / ``__end__``). Instantiating this validates the MAS's
158
+ ``_GID["..."]`` lookups at import time. ``graph`` is the value returned by
159
+ ``graph_factory()`` (i.e. a compiled ``StateGraph``)."""
160
+ names = tuple(n for n in graph.nodes if not str(n).startswith("__"))
161
+ return NodeIdMap(names)
162
+
163
+
164
+ class LangGraphApp:
165
+ """The declarative description of one LangGraph MAS — "describe, don't
166
+ introspect." Subclass it, set the class attributes, and override the two
167
+ behavior hooks. A ``LangGraphHostAdapter`` wraps an instance; the CLI's
168
+ ``--adapter module:Class`` instantiates the adapter, which holds the app.
169
+
170
+ Class attributes the author sets:
171
+ graph_factory : ``() -> compiled graph`` (builds the real state machine;
172
+ may import langgraph). Called once per ``instantiate``.
173
+ nodes : ``list[Nd]`` — the static roster (kinds + seeded knobs).
174
+ edges : ``list[EdgeSpec]`` — the STATIC wiring (no runtime loops).
175
+ knob_to_state : ``dict[str,str]`` — maps an EXTRA knob name to the state
176
+ key the node reads. Named LLM knobs are NOT here (they
177
+ flow through ``apply_llm_config``).
178
+ runtime_loops : ``list[(from, to)]`` — documented runtime back-edges;
179
+ omitted from ``edges`` (informational — NOT enforced
180
+ against the compiled graph, whose internals are
181
+ version-fragile).
182
+ final_output_key: state key holding the run's answer (default ``"answer"``).
183
+ base_prompts : ``{node_id: seeded_prompt}`` for config-decay (default
184
+ ``{}`` — prompts live as node-code literals for most
185
+ LangGraph apps, so the seeded prompt is empty and a
186
+ ``prompt_edit`` rides through).
187
+
188
+ Behavior hooks (override for the MAS; defaults are sensible):
189
+ initialize_state(task_input) -> dict — the graph's initial_state.
190
+ summarize(node_id, partial, merged) -> str — what each Step records as
191
+ ``response_out`` (the Judge scores this; assertions read it).
192
+ apply_llm_config(node_id, vote) -> None — push the live named LLM knobs
193
+ into whatever the node's LLM call consults (a module-level registry).
194
+ Default noop; override iff the MAS has LLM nodes whose model/temp you
195
+ want tunable. Called UP-FRONT per run (graph.stream emits a node only
196
+ after it runs, so this pre-population is the only viable shape).
197
+ reset_llm_config() -> None — clear it before a run (default
198
+ noop).
199
+ """
200
+ # required class attributes (the author MUST set these on the subclass):
201
+ graph_factory: Callable[[], Any]
202
+ nodes: list[Nd]
203
+
204
+ # optional (sensible defaults):
205
+ edges: list[EdgeSpec] = []
206
+ knob_to_state: dict[str, str] = {}
207
+ runtime_loops: list[tuple[str, str]] = []
208
+ final_output_key: str = "answer"
209
+ base_prompts: dict[str, str] = {}
210
+
211
+ # Tracing budget (the OTel trace-projection gate):
212
+ # None -> read the ``DEFAULT_TRACE_TOTAL_BUDGET_TOK`` tunable (the
213
+ # default = None = today's lossy ``summarize()`` path ⇒ the
214
+ # existing suite stays green). Set this None too for parity.
215
+ # int -> cap the per-Judge-prompt total projected tokens; auto-instr
216
+ # SDK calls become OTel GenAI spans, :meth:`_record` projects a
217
+ # BOUNDED slice (LLM nodes only) into each ``Step`` and sheds
218
+ # the largest-evidence-chunk steps first, keeping the final-
219
+ # answer step. A per-app override lets one MAS opt in without
220
+ # touching config.
221
+ trace_total_budget: int | None = None
222
+
223
+ # ---- behavior hooks (override me) -------------------------------------- #
224
+ def initialize_state(self, task_input: str) -> dict[str, Any]:
225
+ f = getattr(type(self), "state_factory", None)
226
+ if f is None:
227
+ raise NotImplementedError(
228
+ "override LangGraphApp.initialize_state(task_input) (or set the "
229
+ "class attribute `state_factory: Callable[[str], dict]`)."
230
+ )
231
+ return f(task_input)
232
+
233
+ def summarize(self, node_id: str, partial: dict, merged: dict) -> str:
234
+ """Default: a deterministic JSON snapshot of the node's state diff
235
+ (bounded). Override to emit what the Judge should score per node."""
236
+ blob = json.dumps(partial, sort_keys=True, default=str)
237
+ return blob[:500]
238
+
239
+ def apply_llm_config(self, node_id: str, vote: KnobVote) -> None:
240
+ """Push the live named LLM knobs (model/temp/max_tokens/system_prompt)
241
+ into the node's LLM-call site. Default noop — override iff the MAS has
242
+ LLM nodes whose model/temp you want tunable."""
243
+ return None
244
+
245
+ def reset_llm_config(self) -> None:
246
+ """Clear the call-time injector before a run (called at run start)."""
247
+ return None
248
+
249
+ # ---- derived (the adapter reads these) --------------------------------- #
250
+ def kind_of(self, node_id: str) -> m.NodeKind:
251
+ for nd in self.nodes:
252
+ if nd.node_id == node_id:
253
+ return nd.kind
254
+ return m.NodeKind.LLM # unknown node -> treat as LLM (costs tokens)
255
+
256
+ def llm_node_ids(self) -> list[str]:
257
+ return [nd.node_id for nd in self.nodes if nd.kind is m.NodeKind.LLM]
258
+
259
+ def build_spec(self) -> m.Spec:
260
+ """Build the bootstrap (incumbent) ``m.Spec`` from this description and
261
+ lint it. A malformed description breaks loudly at adapter construction
262
+ (mirrors ``python -m archforge lint``)."""
263
+ mnodes = [
264
+ m.Node(node_id=nd.node_id, role=nd.role, kind=nd.kind,
265
+ system_prompt=nd.system_prompt, model=nd.model,
266
+ knobs=nd.knobs, tools=list(nd.tools))
267
+ for nd in self.nodes
268
+ ]
269
+ medges = [
270
+ m.Edge(from_=e.from_, to=e.to, type=e.kind, gate=e.gate)
271
+ for e in self.edges
272
+ ]
273
+ spec = m.Spec(nodes=mnodes, edges=medges)
274
+ errs = lint(spec)
275
+ assert not errs, (
276
+ "LangGraph app Spec failed lint: "
277
+ f"{[f'{e.code}@{e.location}: {e.message}' for e in errs]}"
278
+ )
279
+ return spec
280
+
281
+
282
+ # --------------------------------------------------------------------------- #
283
+ # The adapter + runnable
284
+ # --------------------------------------------------------------------------- #
285
+
286
+
287
+ class LangGraphHostAdapter(BaseHostAdapter):
288
+ """A ``HostMAS`` for a LangGraph app. Reuses the kit's ``cfg_for``/``cfg_decay``
289
+ (via ``base_prompts``) + ``run_id``/``estimate_tokens``; OVERRIDES
290
+ ``instantiate`` to return a ``LangGraphRunnable`` that drives ``graph.stream``
291
+ (NOT the kit's flat ``BasePipeline`` walk — that would flatten the
292
+ conditional routing + runtime loop).
293
+
294
+ ``make_agent``/``execution_order``/``resolve_prompt``/``stage_context`` are
295
+ moot under the override (only ``BasePipeline.run`` calls them); leave the kit
296
+ defaults — they're never invoked for a LangGraph host.
297
+ """
298
+
299
+ def __init__(self, app: LangGraphApp) -> None:
300
+ self._app = app
301
+ # build + validate the bootstrap Spec once (catches a malformed app early)
302
+ self._spec = app.build_spec()
303
+ self.base_prompts = app.base_prompts
304
+ # Once-per-process OTel setup: build ArchForge's TracerProvider + the
305
+ # in-memory span buffer and register the repo-declared auto-instrumentors.
306
+ # Idempotent; a no-op when OTel isn't installed (the rich path degrades
307
+ # gracefully to ``summarize()``). Done here — before any
308
+ # ``graph.stream`` — so instrumentors patch the SDKs BEFORE the first call.
309
+ # Lazy import keeps ``import archforge`` OTel-free; touching the module
310
+ # only when a host adapter is constructed.
311
+ from archforge.otel import _ensure_tracer
312
+ _ensure_tracer()
313
+
314
+ def app_spec(self) -> m.Spec:
315
+ """The bootstrap (incumbent) Spec — for ``--seed`` / seeding the store."""
316
+ return self._spec
317
+
318
+ def instantiate(
319
+ self, spec: m.Spec, middleware: TracingMiddleware
320
+ ) -> "LangGraphRunnable":
321
+ graph = self._app.graph_factory()
322
+ return LangGraphRunnable(spec, middleware, self, graph)
323
+
324
+
325
+ class LangGraphRunnable:
326
+ """Drives the real compiled graph and records one ``Step`` per emitted node.
327
+
328
+ Holds the LIVE Spec (candidate or incumbent); its knobs flow into
329
+ ``initial_state`` (extras) and the call-time LLM injector (named LLM knobs),
330
+ both sourced from the live Spec — so a candidate's edits reach the real
331
+ execution. The cost loop is: ``begin_run`` → drive ``graph.stream`` → one
332
+ ``_record_step`` per emitted node → ``end_run`` (assembles + persists the
333
+ ``Trace``; a mid-run crash flushes the Steps that ran, spec E4).
334
+ """
335
+
336
+ def __init__(
337
+ self,
338
+ spec: m.Spec,
339
+ middleware: TracingMiddleware,
340
+ adapter: LangGraphHostAdapter,
341
+ graph: Any,
342
+ ) -> None:
343
+ self._spec = spec
344
+ self._mw = middleware
345
+ self._adapter = adapter
346
+ self._app = adapter._app
347
+ self._graph = graph
348
+ self._run_counter = 0
349
+
350
+ # ---- one run ----------------------------------------------------------- #
351
+ def run(self, task: Any) -> m.Trace:
352
+ sid = self._spec.spec_id or self._spec.compute_spec_id()
353
+ rid = run_id(sid, task.task_id, self._run_counter)
354
+ self._run_counter += 1
355
+ self._mw.begin_run(rid, self._spec, task.task_id)
356
+
357
+ # The trace-projection gate. ``None`` (default) = today's lossy
358
+ # ``summarize()`` path ⇒ identical ``Step`` s ⇒ existing suite green;
359
+ # an int = rich per-step Steps. A per-app ``trace_total_budget`` overrides
360
+ # the ``DEFAULT_TRACE_TOTAL_BUDGET_TOK`` tunable (mirrors the lazy
361
+ # ``ucfg.get`` pattern at engine.py:146). Resolved once per run.
362
+ budget = self._app.trace_total_budget
363
+ if budget is None:
364
+ budget = ucfg.get("DEFAULT_TRACE_TOTAL_BUDGET_TOK")
365
+ self._budget = int(budget) if isinstance(budget, int) and budget > 0 else None
366
+ # rich-path bookkeeping: keep a ref to each Step built during the run so
367
+ # the post-loop shed can mutate them IN PLACE (middleware holds the same
368
+ # refs → ``end_run``'s ``list(self._steps)`` sees trimmed values). Cleared
369
+ # per run; only populated when a budget is set.
370
+ self._run_steps: list[m.Step] = []
371
+ self._final_answer_node: str | None = None
372
+
373
+ # 1. Call-time LLM config: populate the injector UP-FRONT. graph.stream
374
+ # emits a node only *after* it runs, so pre-population is the only
375
+ # viable shape (a per-upcoming-node contextvar is impossible).
376
+ #
377
+ # Read the LIVE Spec's node config (`self._spec` — candidate or
378
+ # incumbent), NOT the app's seeded `Nd` roster: a candidate's knob
379
+ # /model_swap/prompt_edit lives on the live Spec, so the seeded values
380
+ # would silently mask it. Fall back to `nd` for a node the live Spec
381
+ # dropped (a remove_node mutation the static graph still compiles) so
382
+ # the run doesn't crash — structural edits are human-gated regardless.
383
+ live_nodes = {n.node_id: n for n in self._spec.nodes}
384
+ self._app.reset_llm_config()
385
+ for nd in self._app.nodes:
386
+ if nd.kind is not m.NodeKind.LLM:
387
+ continue
388
+ live = live_nodes.get(nd.node_id, nd)
389
+ vote = self._adapter.cfg_for(
390
+ nd.node_id, live.system_prompt, live.model, live.knobs, live.tools
391
+ )
392
+ self._app.apply_llm_config(nd.node_id, vote)
393
+
394
+ # 2. initial_state, then overlay the live Spec's EXTRA knobs (state-routed).
395
+ initial: dict[str, Any] = dict(self._app.initialize_state(task.input))
396
+ for nd in self._app.nodes:
397
+ live = live_nodes.get(nd.node_id, nd)
398
+ extras = live.knobs.model_dump(exclude=_NAMED_KNOBS)
399
+ for kname, kval in extras.items():
400
+ if kval is None:
401
+ continue
402
+ state_key = self._app.knob_to_state.get(kname)
403
+ if state_key is None:
404
+ continue # an extra with no state mapping: host-owned
405
+ initial[state_key] = kval
406
+
407
+ # 3. Drive the real graph: one Step per emitted node. Wall-clock-around-
408
+ # transition (mirrors aede/runner.run_with_timings): when a new node's
409
+ # chunk arrives, the *previous* node is done — record it with the
410
+ # transition time as its latency; the last node is recorded post-loop.
411
+ merged: dict[str, Any] = dict(initial)
412
+ final_output: str | None = None
413
+ current: str | None = None
414
+ current_partial: dict[str, Any] = {}
415
+ node_started_at = 0.0
416
+ try:
417
+ for chunk in self._graph.stream(initial, stream_mode="updates"):
418
+ # AEDE-style graphs have no parallel branch ⇒ one node/superstep.
419
+ # A multi-key chunk means a parallel branch the adapter doesn't
420
+ # yet model — fail loudly with a clear message, not silently.
421
+ assert len(chunk) == 1, (
422
+ "LangGraph adapter saw a multi-key stream chunk — a parallel "
423
+ f"branch it doesn't model yet: {list(chunk)}"
424
+ )
425
+ node_name, partial = next(iter(chunk.items()))
426
+ now = time.perf_counter()
427
+ if current is not None:
428
+ self._record(current, current_partial, now - node_started_at, merged)
429
+ current = node_name
430
+ current_partial = partial or {}
431
+ node_started_at = now
432
+ if partial:
433
+ merged.update(partial)
434
+ # Note the node that write-sets the final-output key: the
435
+ # post-loop shed protects it (the answer-bearing step always
436
+ # reaches the Judge). ``partial`` is this node's state diff,
437
+ # so its carrying ``final_output_key`` marks the producer.
438
+ if (
439
+ self._budget is not None
440
+ and self._app.final_output_key in partial
441
+ ):
442
+ self._final_answer_node = node_name
443
+ if current is not None:
444
+ now = time.perf_counter()
445
+ self._record(current, current_partial, now - node_started_at, merged)
446
+
447
+ # Shed to budget ON THE SUCCESS PATH only: mutate the shared Step
448
+ # objects in place so the Judge's total ingested text ≤ budget, the
449
+ # largest-evidence-chunk steps trims first, the final-answer step is
450
+ # protected. Skipped on the except path → a mid-run crash leaves ran
451
+ # Steps at their per-node caps (E4 holds).
452
+ if self._budget is not None and self._run_steps:
453
+ from archforge.otel import _shed_to_budget
454
+ _shed_to_budget(self._run_steps, self._final_answer_node, self._budget)
455
+
456
+ final_output = merged.get(self._app.final_output_key)
457
+ if final_output is not None and not isinstance(final_output, str):
458
+ final_output = str(final_output)
459
+ return self._mw.end_run(final_output, ok=True, error=None)
460
+ except Exception as exc: # noqa: BLE001 — flush a partial trace (E4)
461
+ return self._mw.end_run(final_output, ok=False, error=repr(exc))
462
+
463
+ # ---- one step ---------------------------------------------------------- #
464
+ def _record(
465
+ self, node_name: str, partial: dict, elapsed_s: float, merged: dict
466
+ ) -> None:
467
+ kind = self._app.kind_of(node_name)
468
+ # The gated OTel trace projection. No budget (the default) → today's lossy
469
+ # ``summarize()`` path byte-identical (parity — the existing suite's
470
+ # substring/shape assertions hold). Budget set → ask ``otel.project`` for
471
+ # the real per-step prompt/completion captured as OTel GenAI spans under
472
+ # the ``archforge.node`` parent; ``None`` (no spans / capture off /
473
+ # non-LLM / OTel absent) → degrade gracefully back to ``summarize()``.
474
+ if self._budget is None:
475
+ resp = self._app.summarize(node_name, partial, merged)
476
+ prompt_in = "" # LangGraph nodes read the query from state,
477
+ # Kind-aware cost: LLM nodes cost tokens; retriever/rule/symbolic ~0
478
+ # (their real cost is wall-clock — `latency_ms`).
479
+ tokens = estimate_tokens(resp) if kind is m.NodeKind.LLM else 0
480
+ else:
481
+ from archforge.otel import project as _project
482
+ proj = _project(node_name, kind, partial, merged)
483
+ if proj is None:
484
+ # graceful degrade: OTel absent / no child spans / non-LLM kind
485
+ resp = self._app.summarize(node_name, partial, merged)
486
+ prompt_in = ""
487
+ tokens = estimate_tokens(resp) if kind is m.NodeKind.LLM else 0
488
+ else:
489
+ resp = proj.response_out
490
+ prompt_in = proj.prompt_in
491
+ tokens = proj.tokens
492
+ step = m.Step(
493
+ node_id=node_name,
494
+ prompt_in=prompt_in,
495
+ response_out=resp,
496
+ perf=m.StepPerf(
497
+ tokens=tokens,
498
+ latency_ms=round(elapsed_s * 1000.0, 3),
499
+ ),
500
+ )
501
+ if self._budget is not None:
502
+ # keep a ref so the post-loop shed can mutate this same object in
503
+ # place (middleware holds the same ref → end_run sees trimmed values).
504
+ self._run_steps.append(step)
505
+ self._mw._record_step(step)
506
+
507
+
508
+ __all__ = [
509
+ "Nd", "EdgeSpec", "LangGraphApp",
510
+ "LangGraphHostAdapter", "LangGraphRunnable",
511
+ "node_ids", "NodeIdMap", "wrapped", # the cooperative OTel seam + keyed id source
512
+ "export_spec_sidecar", "load_spec_sidecar",
513
+ "build_optimized_envelope", "export_optimized", "load_optimized",
514
+ "ENVELOPE_SCHEMA",
515
+ ]
516
+
517
+
518
+ # --------------------------------------------------------------------------- #
519
+ # Tier-2 deploy — the winning Spec's knobs as a reviewable sidecar
520
+ # --------------------------------------------------------------------------- #
521
+ #
522
+ # On an AUTO_PROMOTE the engine fires an opt-in ``on_promote`` callback (see
523
+ # ``engine.Engine.__init__``) with the promoted Spec. A driver wires that to
524
+ # ``export_spec_sidecar`` to write a stable JSON sidecar the MAS overlays onto
525
+ # its config consts at startup — so the optimization reaches *production* (the
526
+ # ordinary `/optimize`-equivalent run, no ArchForge on the hot path). Rollback
527
+ # is deleting the file; the diff is what Git shows between promotes.
528
+ #
529
+ # General: needs only the Spec. The projection is faithful (the Spec's native
530
+ # knob vocabulary), so the per-MAS consumer owns the one-time
531
+ # ``node_id.knob_name → config field`` mapping (e.g. AEDE's
532
+ # ``Settings.from_env`` reads this and overlays onto ``PipelineConfig``). The
533
+ # exporter does NOT guess that mapping — "describe, don't introspect."
534
+
535
+ def _node_knob_projection(node: m.Node) -> dict[str, Any]:
536
+ """One node → its live knobs as a flat ``{knob: value}`` the consumer overlays.
537
+
538
+ Named LLM knobs (model/temperature/max_tokens/retries/system_prompt) are the
539
+ universal vocabulary; extras (top_k/threshold/…) are the kind-specific ones.
540
+ Only *real* values ship (None = unset; blank ``model``/``system_prompt`` =
541
+ "use default"), so the sidecar carries just the knobs the Forge set.
542
+ """
543
+ out: dict[str, Any] = {}
544
+ if node.model:
545
+ out["model"] = node.model
546
+ k = node.knobs
547
+ if k is not None:
548
+ if k.temperature is not None:
549
+ out["temperature"] = k.temperature
550
+ if k.max_tokens is not None:
551
+ out["max_tokens"] = k.max_tokens
552
+ if k.retries is not None:
553
+ out["retries"] = k.retries
554
+ # extras (everything that isn't a named knob / `tunable`)
555
+ for kk, vv in k.model_dump().items():
556
+ if kk in _NAMED_KNOBS:
557
+ continue
558
+ if vv is None:
559
+ continue
560
+ out[kk] = vv
561
+ if node.system_prompt:
562
+ out["system_prompt"] = node.system_prompt
563
+ return out
564
+
565
+
566
+ def export_spec_sidecar(spec: m.Spec, path: str | Path) -> dict[str, dict[str, Any]]:
567
+ """Write the winning Spec's knobs to ``path`` as ``{node_id: {knob: value}}``.
568
+
569
+ A full snapshot (not just deltas) so the consumer can delete-then-replace on
570
+ each promote and Git shows the between-promote diff. Returns the sidecar dict
571
+ (for in-process use / testing). Stable: nodes in Spec order; knobs in
572
+ insertion order; JSON ``indent=2 sort_keys=False`` so the diff is readable.
573
+ """
574
+ sidecar = {n.node_id: _node_knob_projection(n) for n in spec.nodes if n.node_id}
575
+ # drop empty nodes — a node whose knobs all defaulted carries nothing to deploy
576
+ sidecar = {nid: knobs for nid, knobs in sidecar.items() if knobs}
577
+ Path(path).write_text(json.dumps(sidecar, indent=2, default=str), encoding="utf-8")
578
+ return sidecar
579
+
580
+
581
+ def load_spec_sidecar(path: str | Path) -> dict[str, dict[str, Any]]:
582
+ """Read a sidecar written by ``export_spec_sidecar`` → ``{node_id: {knob}}``.
583
+
584
+ The per-MAS consumer wraps this with its ``node.knob → config field`` map."""
585
+ return json.loads(Path(path).read_text(encoding="utf-8"))
586
+
587
+
588
+ # --------------------------------------------------------------------------- #
589
+ # Unified deployment config — the optimized.json envelope (improvement #4)
590
+ # --------------------------------------------------------------------------- #
591
+ #
592
+ # The bare sidecar (``export_spec_sidecar``) ships only ``{node_id:{knob}}`` —
593
+ # enough to overlay the MAS's config consts, but it carries none of the *why*.
594
+ # The envelope is the single file production loads to apply the winner AND audit
595
+ # it: the knobs (today's sidecar body, nested under ``knobs``) plus the lineage,
596
+ # the Gatekeeper decision, and the scores that justified the promote. One artifact
597
+ # → Tier-2 deploy: the MAS reads ``envelope["knobs"]`` (a one-line change to a
598
+ # consumer like AEDE's ``aede_sidecar``), the rest is human-readable provenance.
599
+ # Rollback is deleting the file; the diff is what Git shows between promotes.
600
+ #
601
+ # `knobs` is NESTED UNDER the key named `knobs` (vs the sidecar's flat top level)
602
+ # so the envelope has room for `schema`/`spec_id`/`scores`/… alongside it. The
603
+ # per-MAS consumer changes ONE line: `load_spec_sidecar(path)` →
604
+ # `load_optimized(path)["knobs"]`. That AEDE consumer edit is OUT OF SCOPE here
605
+ # (tracked by the AEDE integration tasks); this module ships the producer.
606
+
607
+ ENVELOPE_SCHEMA = "archforge.optimized/v1"
608
+
609
+
610
+ def _score_dims(run: Any) -> list[dict[str, Any]]:
611
+ """Reduce a SuiteRun's per-RunScore sub-rubrics to a stable dim summary.
612
+
613
+ A ``SuiteRun`` (``archforge.suite``) carries a list of ``m.RunScore`` (one per
614
+ scored repeat), each with a ``rubric_scores: {dim: score}`` map. For the
615
+ envelope we want the mean per rubric dimension across the candidate's scored
616
+ repeats — the same dimensions the Gatekeeper's mean aggregates. Tolerant: any
617
+ duck-typed ``SuiteRun``-like (the engine passes the real one; tests may pass
618
+ a lighter object) with optional ``scores``/``aggregate``.
619
+ """
620
+ scores = getattr(run, "scores", None) or []
621
+ sums: dict[str, float] = {}
622
+ n: dict[str, int] = {}
623
+ for rs in scores:
624
+ dims = getattr(rs, "rubric_scores", None) or {}
625
+ for dim, val in dims.items():
626
+ try:
627
+ v = float(val)
628
+ except (TypeError, ValueError):
629
+ continue
630
+ sums[dim] = sums.get(dim, 0.0) + v
631
+ n[dim] = n.get(dim, 0) + 1
632
+ return [{"dim": d, "mean": round(sums[d] / n[d], 4)} for d in sums]
633
+
634
+
635
+ def build_optimized_envelope(
636
+ spec: m.Spec,
637
+ *,
638
+ parent: m.Spec | None,
639
+ promoted_at_cycle: int,
640
+ decision: "Decision | None" = None,
641
+ cand_run: "Any | None" = None,
642
+ inc_run: "Any | None" = None,
643
+ ) -> dict[str, Any]:
644
+ """The unified deployment config — knobs + provenance (improvement #4).
645
+
646
+ Shape (``archforge.optimized/v1``)::
647
+
648
+ {"schema": "archforge.optimized/v1",
649
+ "spec_id": <candidate id>, "parent_spec_id": <parent id | None>,
650
+ "promoted_at_cycle": <int>,
651
+ "decision": {"action": "auto_promote", "rule": <by_rule>,
652
+ "margin": <±float>, "reason": <str>} | None,
653
+ "scores": {"mean": <cand mean>, "incumbent_mean": <inc mean | None>,
654
+ "dims": [{"dim","mean"}], "rubric_id": <str|None>,
655
+ "suite_id": <str|None>, "tokens": <int>, "latency_ms": <float>} | None,
656
+ "knobs": {node_id: {knob: value}, ...}} # == today's sidecar body
657
+
658
+ ``knobs`` is byte-identical to ``export_spec_sidecar``'s body (same
659
+ ``_node_knob_projection``, same empty-node drop), so a consumer already reading
660
+ the bare sidecar switches by reading ``env["knobs"]`` instead of the top level.
661
+ ``decision``/``scores`` are ``None`` when the caller omits them (an embedder
662
+ building an envelope outside a promote context still gets the knobs + lineage).
663
+ """
664
+ # `knobs` is exactly the sidecar body — the consumer-compat seam.
665
+ knobs = {n.node_id: _node_knob_projection(n) for n in spec.nodes if n.node_id}
666
+ knobs = {nid: kv for nid, kv in knobs.items() if kv}
667
+
668
+ env: dict[str, Any] = {
669
+ "schema": ENVELOPE_SCHEMA,
670
+ "spec_id": spec.spec_id or spec.compute_spec_id(),
671
+ "parent_spec_id": getattr(parent, "spec_id", None) or
672
+ (parent.compute_spec_id() if parent is not None else None),
673
+ "promoted_at_cycle": promoted_at_cycle,
674
+ "decision": None,
675
+ "scores": None,
676
+ "knobs": knobs,
677
+ }
678
+ if decision is not None:
679
+ env["decision"] = {
680
+ "action": getattr(decision.action, "value", str(decision.action)),
681
+ "rule": decision.by_rule,
682
+ "margin": decision.margin,
683
+ "reason": decision.reason,
684
+ }
685
+ if cand_run is not None:
686
+ inc_mean = getattr(inc_run, "mean", None) if inc_run is not None else None
687
+ env["scores"] = {
688
+ "mean": getattr(cand_run, "mean", None),
689
+ "incumbent_mean": inc_mean,
690
+ "dims": _score_dims(cand_run),
691
+ "rubric_id": getattr(cand_run, "rubric_id", None),
692
+ "suite_id": getattr(cand_run, "suite_id", None),
693
+ "tokens": getattr(cand_run, "tokens", 0),
694
+ "latency_ms": getattr(cand_run, "latency_ms", 0.0),
695
+ }
696
+ return env
697
+
698
+
699
+ def export_optimized(
700
+ spec: m.Spec, path: str | Path, *,
701
+ parent: m.Spec | None = None, promoted_at_cycle: int = 0,
702
+ decision: "Decision | None" = None, cand_run: "Any | None" = None,
703
+ inc_run: "Any | None" = None,
704
+ ) -> dict[str, Any]:
705
+ """Write ``build_optimized_envelope(...)`` to ``path`` as ``indent=2`` JSON.
706
+
707
+ Returns the envelope dict (for in-process use / testing). Stable: nodes in
708
+ Spec order, knobs in insertion order; ``indent=2 sort_keys=False`` so the
709
+ between-promote Git diff is readable. The CLI's ``on_deploy`` wires this to
710
+ ``<root>/optimized.json`` on every AUTO_PROMOTE — Tier-2 deploy auto-synced.
711
+ """
712
+ env = build_optimized_envelope(
713
+ spec, parent=parent, promoted_at_cycle=promoted_at_cycle,
714
+ decision=decision, cand_run=cand_run, inc_run=inc_run,
715
+ )
716
+ Path(path).write_text(json.dumps(env, indent=2, default=str), encoding="utf-8")
717
+ return env
718
+
719
+
720
+ def load_optimized(path: str | Path) -> dict[str, Any]:
721
+ """Read an envelope written by ``export_optimized`` (the unified deploy file).
722
+
723
+ The per-MAS consumer reads ``env["knobs"]`` (the sidecar body) and overlays it
724
+ onto its config; the rest is provenance for human review. (Bare sidecar
725
+ consumers keep ``load_spec_sidecar``; this is the envelope's loader.)"""
726
+ return json.loads(Path(path).read_text(encoding="utf-8"))