archforge-optimizer 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. archforge/__init__.py +76 -0
  2. archforge/__main__.py +10 -0
  3. archforge/architect.py +442 -0
  4. archforge/cli.py +881 -0
  5. archforge/config.py +140 -0
  6. archforge/config_init.py +150 -0
  7. archforge/diff.py +206 -0
  8. archforge/engine.py +444 -0
  9. archforge/gatekeeper.py +290 -0
  10. archforge/host/__init__.py +20 -0
  11. archforge/host/adapters/__init__.py +41 -0
  12. archforge/host/adapters/base.py +311 -0
  13. archforge/host/adapters/helpers.py +163 -0
  14. archforge/host/adapters/langgraph.py +726 -0
  15. archforge/host/base.py +105 -0
  16. archforge/host/fake.py +380 -0
  17. archforge/judge/__init__.py +20 -0
  18. archforge/judge/base.py +257 -0
  19. archforge/judge/scripted.py +145 -0
  20. archforge/lint.py +180 -0
  21. archforge/llm/__init__.py +65 -0
  22. archforge/llm/_common.py +94 -0
  23. archforge/llm/anthropic.py +90 -0
  24. archforge/llm/base.py +90 -0
  25. archforge/llm/gemini.py +112 -0
  26. archforge/llm/groq.py +63 -0
  27. archforge/llm/openai.py +63 -0
  28. archforge/llm/scripted.py +134 -0
  29. archforge/middleware.py +181 -0
  30. archforge/models.py +435 -0
  31. archforge/mutate.py +214 -0
  32. archforge/otel.py +613 -0
  33. archforge/runlog.py +103 -0
  34. archforge/runner.py +153 -0
  35. archforge/spec_builder.py +126 -0
  36. archforge/stores/__init__.py +22 -0
  37. archforge/stores/_jsonl.py +81 -0
  38. archforge/stores/attempt_store.py +161 -0
  39. archforge/stores/spec_store.py +188 -0
  40. archforge/stores/trace_store.py +42 -0
  41. archforge/suite.py +248 -0
  42. archforge/userconfig.py +144 -0
  43. archforge_optimizer-0.1.0.dist-info/METADATA +420 -0
  44. archforge_optimizer-0.1.0.dist-info/RECORD +47 -0
  45. archforge_optimizer-0.1.0.dist-info/WHEEL +4 -0
  46. archforge_optimizer-0.1.0.dist-info/entry_points.txt +2 -0
  47. archforge_optimizer-0.1.0.dist-info/licenses/LICENSE +21 -0
archforge/config.py ADDED
@@ -0,0 +1,140 @@
1
+ """ArchForge system config — internals only (NOT user-tunable).
2
+
3
+ This holds the NON-tunable framework plumbing: identity (``VERSION``), the LLM
4
+ provider ROSTER (the tuples the CLI's ``--provider`` choices come from), the
5
+ on-disk storage DIRNAMES + pointer files, the content-addressing hash lengths,
6
+ Anthropic's required ``max_tokens``, the ScriptedJudge noise pattern, the CLI
7
+ program name + its scripted fixtures, and ``load_env`` (the ``.env`` loader for
8
+ provider API keys). **Nothing here is a user preference** — every value a user
9
+ might tune (τ, δ, R, the default provider + models, the rubric, the storage root,
10
+ token budgets, grader resilience …) lives instead in the project's
11
+ ``.archforge/archforge.py`` (made by ``archforge-optimizer init``) and is read
12
+ lazily by ``archforge.userconfig`` at use time. So ``config.py`` ↔ ``archforge.py``
13
+ share NO variable (the two-file split), and this module never depends on the
14
+ user's config — it imports cleanly before ``init`` has run.
15
+
16
+ Pure leaf: stdlib + ``python-dotenv`` (a core dependency) only — imports nothing
17
+ else from ``archforge``, so every other module may ``from archforge.config import …``
18
+ with no risk of a back-edge.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ import importlib.util
24
+ import os
25
+ from pathlib import Path
26
+
27
+ # =========================================================================== #
28
+ # Identity
29
+ # =========================================================================== #
30
+ # The package version. Read by hatchling for the built distribution and
31
+ # re-exported as `archforge.__version__` (archforge/__init__.py). One place.
32
+ VERSION: str = "0.1.0"
33
+
34
+
35
+ # =========================================================================== #
36
+ # LLM providers — the ROSTER only (NOT the default choice; that's a tunable)
37
+ # =========================================================================== #
38
+ # The providers the CLI's `--provider` flag ACCEPTS (its `choices`). `scripted` is
39
+ # the zero-cost option (deterministic fakes, no SDK, no API key); the real
40
+ # providers each need their SDK + an API key. WHICH provider is the default, and
41
+ # each provider's default MODEL id, are tunables → they live in .archforge/archforge.py
42
+ # (resolved by archforge.userconfig), NOT here. `REAL_PROVIDERS`/`ALL_PROVIDERS` are
43
+ # re-exported from archforge.llm; `make_client(provider)` dispatches on these.
44
+ SCRIPTED_PROVIDER: str = "scripted"
45
+ REAL_PROVIDERS: tuple[str, ...] = ("anthropic", "openai", "groq", "gemini")
46
+ ALL_PROVIDERS: tuple[str, ...] = (SCRIPTED_PROVIDER,) + REAL_PROVIDERS
47
+
48
+
49
+ # =========================================================================== #
50
+ # Project environment (`.env`) — how secrets reach the provider SDKs
51
+ # =========================================================================== #
52
+ # A pip-installed `archforge` reads the project folder's `.env` so a checked-out
53
+ # repo "just runs" once API keys are added (the `.env` is gitignored — never commit
54
+ # secrets). `load_env()` populates `os.environ` with `override=False`, so the
55
+ # precedence is: an explicit `--api-key` (passed straight to the SDK by the CLI)
56
+ # > a real process env var > the `.env` file. `python-dotenv` is a CORE dependency
57
+ # here (handles `export ` prefixes, quoting, multiline values robustly); an absent
58
+ # file is a no-op. Provider env-var names are NOT centralized here — the provider
59
+ # SDKs read their own (`ANTHROPIC_API_KEY` … `GEMINI_API_KEY`); this loader only
60
+ # populates env.
61
+ #
62
+ # The DEFAULT `.env` path is a tunable (DEFAULT_ENV_FILE in archforge.py); this
63
+ # loader's own default arg is the literal ".env" so the function stays a pure leaf
64
+ # that never touches the resolver (callable pre-init, e.g. to load keys).
65
+
66
+ def load_env(path: str | os.PathLike[str] | None = ".env") -> None:
67
+ """Populate ``os.environ`` from a ``.env`` file (via python-dotenv, no override).
68
+
69
+ A real process env var therefore always wins (``override=False``); the CLI's
70
+ ``--api-key`` (handed straight to the SDK) wins above both. No-op when the file
71
+ is absent or ``path`` is falsy — so callers (CLI, runner) invoke it
72
+ unconditionally. ``python-dotenv`` (core dep) does the parsing.
73
+ """
74
+ if not path:
75
+ return
76
+ if importlib.util.find_spec("dotenv") is not None:
77
+ from dotenv import load_dotenv # type: ignore[import-not-found]
78
+ load_dotenv(path, override=False) # override=False ⇒ real env wins
79
+
80
+
81
+ # =========================================================================== #
82
+ # Storage layout
83
+ # =========================================================================== #
84
+ # On-disk layout under the CLI's `--root` (a path the user passes, or ".archforge" —
85
+ # a tunable, but the dirnames themselves are plumbing). The stores name their
86
+ # subdirectories + pointer files from these; rare to change.
87
+ SPECS_DIRNAME: str = "specs"
88
+ ATTEMPTS_DIRNAME: str = "attempts"
89
+ TRACES_DIRNAME: str = "traces"
90
+ ACTIVE_POINTER_FILE: str = "active.pointer" # names the active incumbent spec_id
91
+ ARCHIVED_FILE: str = "archived.jsonl" # rolled-back spec_ids (I3 reachability)
92
+
93
+
94
+ # =========================================================================== #
95
+ # Internals — not for tuning
96
+ # =========================================================================== #
97
+ # Bookkeeping values wired into specific consumers. Change only if you know the
98
+ # consumer; these are plumbing, not user preferences.
99
+
100
+ # Content-addressing hash truncation (sha256 hex prefix lengths). SPEC_ID_HASH_LEN
101
+ # is the Spec/Attempt content id (models.py, attempt_store); SHORT_HASH_LEN is the
102
+ # short deterministic suffixes used by the fake host (responder + run_id).
103
+ SPEC_ID_HASH_LEN: int = 16
104
+ SHORT_HASH_LEN: int = 8
105
+
106
+ # Anthropic's messages API requires a max_tokens; used when a caller omits it
107
+ # (only the Anthropic adapter). Not a user preference — a provider requirement.
108
+ ANTHROPIC_DEFAULT_MAX_TOKENS: int = 1024
109
+
110
+ # The ScriptedJudge's deterministic noise band (E1 jitter pattern): the jitter
111
+ # applied to a scripted aggregate is `pattern[run_index % len] * noise_width`,
112
+ # so identical runs yield identical jitter (reproducible E1 margin-boundary tests).
113
+ SCRIPTED_NOISE_PATTERN: tuple[float, ...] = (1.0, 0.0, -1.0, 0.5, -0.5, 0.25, -0.25)
114
+
115
+ # The CLI program name (the installed console command + the prefix on result lines).
116
+ PROG: str = "archforge-optimizer"
117
+
118
+ # The CLI's free-run scripted fixtures: when `--provider scripted` is invoked
119
+ # with no injected `components`, the evolve family runs against this one-task suite
120
+ # so `archforge-optimizer evolve` is runnable end-to-end with zero configuration.
121
+ DEFAULT_SUITE_ID: str = "cli-default"
122
+ DEFAULT_TASK_ID: str = "t1"
123
+ DEFAULT_TASK_INPUT: str = "hello"
124
+
125
+
126
+ __all__ = [
127
+ # identity
128
+ "VERSION",
129
+ # llm provider roster
130
+ "SCRIPTED_PROVIDER", "REAL_PROVIDERS", "ALL_PROVIDERS",
131
+ # project environment (.env loader for provider API keys)
132
+ "load_env",
133
+ # storage layout
134
+ "SPECS_DIRNAME", "ATTEMPTS_DIRNAME", "TRACES_DIRNAME",
135
+ "ACTIVE_POINTER_FILE", "ARCHIVED_FILE",
136
+ # internals
137
+ "SPEC_ID_HASH_LEN", "SHORT_HASH_LEN", "ANTHROPIC_DEFAULT_MAX_TOKENS",
138
+ "SCRIPTED_NOISE_PATTERN", "PROG",
139
+ "DEFAULT_SUITE_ID", "DEFAULT_TASK_ID", "DEFAULT_TASK_INPUT",
140
+ ]
@@ -0,0 +1,150 @@
1
+ """Scaffolding text for `archforge-optimizer init` — the user-tunable config template.
2
+
3
+ Owns the ACTIVE defaults that become `.archforge/archforge.py` (the project's sole
4
+ source of ArchForge tunables, made by `init`) and the `.env.example` key template.
5
+
6
+ Two roles, one source:
7
+ * `init` writes ``archforge_config_text()`` verbatim to ``.archforge/archforge.py`` →
8
+ the user's tunables. It ships with **active sane values** so the CLI works
9
+ immediately after `init`; the user edits a value to change behaviour.
10
+ * Under the test runner, ``archforge.userconfig`` execs the SAME ``TEMPLATE``
11
+ in-memory (no disk file) so the suite sees the sane defaults
12
+ (e.g. ``Thresholds().tau == 0.05``) with zero per-test files.
13
+
14
+ The values here are ArchForge's sane defaults — keep them in sync with what the
15
+ package historically shipped. This module is pure data (string constants); it imports
16
+ nothing from `archforge` and never reads the live `.env`.
17
+ """
18
+ from __future__ import annotations
19
+
20
+ # --------------------------------------------------------------------------- #
21
+ # the sane tunable defaults — the ONE place the values live
22
+ # --------------------------------------------------------------------------- #
23
+ # (name, active-value-as-assignment, one-line purpose). The assignment text is
24
+ # emitted verbatim into the generated file AND exec'd by the resolver, so it must be
25
+ # valid Python and carry the real sane value (note DEFAULT_ARCHITECT_MODELS /
26
+ # DEFAULT_JUDGE_MODELS / DEFAULT_SUB_RUBRICS are the full dicts, not `{}` — the
27
+ # resolver must resolve them without KeyError).
28
+
29
+ _DEFAULT_ARCHITECT_MODELS = (
30
+ '{"anthropic": "claude-sonnet-5", "openai": "gpt-4o", '
31
+ '"groq": "openai/gpt-oss-120b", "gemini": "gemini-3.6-flash"}'
32
+ )
33
+ _DEFAULT_JUDGE_MODELS = (
34
+ '{"anthropic": "claude-sonnet-5", "openai": "gpt-4o", '
35
+ '"groq": "openai/gpt-oss-120b", "gemini": "gemini-3.6-flash"}'
36
+ )
37
+ _DEFAULT_SUB_RUBRICS = (
38
+ '{"correctness": "Is the final answer factually correct and aligned with the task?", '
39
+ '"completeness": "Does the answer address every part of the task?", '
40
+ '"grounding": "Are the claims supported by the inputs/context, not invented?"}'
41
+ )
42
+
43
+ # The starter suite `init` writes to .archforge/suite.json — byte-identical to the
44
+ # one-task CLI fallback fixture (cli-default / t1 / hello), so the generated default
45
+ # round-trips to the same Suite the CLI builds when the file is absent. A per-task
46
+ # rubric_id is optional in the file; omitting it scores against the active rubric.
47
+ _DEFAULT_SUITE_JSON = (
48
+ '{\n'
49
+ ' "suite_id": "cli-default",\n'
50
+ ' "tasks": [\n'
51
+ ' {"task_id": "t1", "input": "hello"}\n'
52
+ ' ]\n'
53
+ '}\n'
54
+ )
55
+
56
+ _FIELDS: tuple[tuple[str, str, str], ...] = (
57
+ # --- LLM provider
58
+ ("PROVIDER", '"gemini"',
59
+ "which LLM to use (scripted|anthropic|openai|groq|gemini); scripted needs no API key"),
60
+ ("DEFAULT_ARCHITECT_MODELS", _DEFAULT_ARCHITECT_MODELS,
61
+ "default Architect (proposer) model per provider; a bare LLMClient call falls "
62
+ "back here too — edit the dict to change it"),
63
+ ("DEFAULT_JUDGE_MODELS", _DEFAULT_JUDGE_MODELS,
64
+ "default Judge (scorer) model per provider — edit the dict to change it"),
65
+ # --- optimization policy
66
+ ("DEFAULT_TAU", "0.05", "how much better a candidate must score to be promoted (τ)"),
67
+ ("DEFAULT_DELTA", "0.07", "how far a promoted run can drop before it's rolled back (δ, >= τ)"),
68
+ ("DEFAULT_REPEATS", "1", "how many times each eval task is run; more = steadier scores, more cost (R)"),
69
+ ("MAX_REPEATS", "3", "upper bound on R"),
70
+ ("DEFAULT_UNRUNNABLE_FRAC", "0.25", "drop a candidate if more than this fraction of its tasks crash (ε)"),
71
+ ("DEFAULT_PLATEAU_CYCLES", "5", "stop after this many cycles in a row with no improvement (K)"),
72
+ ("DEFAULT_MAX_CYCLES", "20", "max optimization cycles per run"),
73
+ # --- grader resilience
74
+ ("DEFAULT_JUDGE_RETRIES", "2", "how many times to retry a failed judge call"),
75
+ ("BACKOFF_CAP_SECONDS", "30.0", "max seconds to wait between judge retries"),
76
+ # --- judge scoring
77
+ ("DEFAULT_RUBRIC_ID", '"default-v1"', "name of the scoring rubric (keep it stable so runs compare)"),
78
+ ("DEFAULT_SUB_RUBRICS", _DEFAULT_SUB_RUBRICS,
79
+ "the rubric's dimensions: what a high score looks like, per dimension"),
80
+ # --- environment / budget / storage
81
+ ("DEFAULT_ROOT_DIR", '".archforge"', "where run state is written (relative to where you run the CLI)"),
82
+ ("DEFAULT_ENV_FILE", '".env"', "the .env file loaded for API keys (a real env var always wins)"),
83
+ ("DEFAULT_SUITE_FILE", '".archforge/suite.json"', "the suite file defining your eval tasks (absent → the one-task default)"),
84
+ ("DEFAULT_MAX_TOKENS_TOTAL", "None", "whole-run token budget cap; None = no limit"),
85
+ ("DEFAULT_MAX_TOKENS_PER_CYCLE", "None", "per-cycle token cap (aborts mid-cycle if exceeded); None = no limit"),
86
+ ("DEFAULT_MAX_WALL_MS_PER_CYCLE", "None", "per-cycle wall-clock cap (ms); aborts if exceeded — "
87
+ "covers non-LLM nodes (retriever/tool/rule) that cost time, not tokens; None = no limit"),
88
+ ("DEFAULT_TRACE_TOTAL_BUDGET_TOK", "None", "total Judge-prompt token budget for OTel trace "
89
+ "projection of per-step LLM prompt/completion; None = lossy summarize() path (parity, "
90
+ "no tracing); an int turns on rich per-step Steps, shedding largest-evidence chunks first"),
91
+ )
92
+
93
+ # section break points in _FIELDS (for grouping the emitted file)
94
+ _BREAKS: dict[int, str] = {
95
+ 3: "# --- optimization policy ------------------------------------------------",
96
+ 10: "# --- grader resilience ---------------------------------------------------",
97
+ 12: "# --- judge scoring -------------------------------------------------------",
98
+ 14: "# --- environment / budget / storage --------------------------------------",
99
+ }
100
+
101
+ _HEADER = """\
102
+ # archforge.py — your ArchForge config (made by `archforge-optimizer init`).
103
+ #
104
+ # The values below already work — the CLI runs as-is after `init`. Edit any value
105
+ # to change that default. Nothing here is required to make ArchForge import.
106
+ #
107
+
108
+ # --- LLM provider ------------------------------------------------------------
109
+ """
110
+
111
+ _FOOTER = """
112
+ # Changes here take effect on the next `archforge-optimizer` run.
113
+ """
114
+
115
+
116
+ def archforge_config_text() -> str:
117
+ """The full body of the generated `.archforge/archforge.py` (active sane defaults)."""
118
+ lines = [_HEADER.rstrip("\n")]
119
+ for i, (name, value, purpose) in enumerate(_FIELDS):
120
+ if i in _BREAKS:
121
+ lines.append("")
122
+ lines.append(_BREAKS[i])
123
+ lines.append(f"# {purpose}")
124
+ lines.append(f"{name} = {value}")
125
+ lines.append(_FOOTER.rstrip("\n"))
126
+ return "\n".join(lines) + "\n"
127
+
128
+
129
+ # The active sane-default assignments, exec'd by archforge.userconfig under the test
130
+ # runner (so the suite sees sane defaults) — same string `init` writes to disk.
131
+ TEMPLATE: str = archforge_config_text()
132
+
133
+
134
+ def env_example_text() -> str:
135
+ """The body of the generated `.env.example` (empty provider key var names)."""
136
+ return (
137
+ "# fill in your API keys for provider you are going to use for archforge-optimizer.\n\n"
138
+ "# ANTHROPIC_API_KEY=\n"
139
+ "# OPENAI_API_KEY=\n"
140
+ "# GROQ_API_KEY=\n"
141
+ "# GEMINI_API_KEY=\n"
142
+ "\n# Other service keys you need for your tasks\n"
143
+ )
144
+
145
+
146
+ # The tunable names — exported so callers/tests enumerate the editable surface.
147
+ EDITABLE_NAMES: tuple[str, ...] = tuple(name for name, _, _ in _FIELDS)
148
+
149
+ __all__ = ["archforge_config_text", "env_example_text", "TEMPLATE", "EDITABLE_NAMES",
150
+ "_DEFAULT_SUITE_JSON"]
archforge/diff.py ADDED
@@ -0,0 +1,206 @@
1
+ """Spec-level structural diff — what changed from the parent Spec to the candidate.
2
+
3
+ ArchForge's ``Change`` record (``m.Change``) carries the *intent* of a mutation
4
+ (kind/target/diff/rationale/scope) but NOT the field-level before/after — the
5
+ ``payload`` dict evaporates after ``apply_change`` (it is metadata, not
6
+ persisted). So an inspectable "what did the Forge actually change" requires
7
+ diffing the two persisted Specs node-by-node. Both specs live in the SpecStore
8
+ (``spec_store.get(parent_id)`` / ``get(candidate_id)``), so the comparison is
9
+ cheap and needs nothing beyond the Specs themselves.
10
+
11
+ ``spec_diff(parent, candidate)`` does exactly that: a pure comparison returning a
12
+ deterministic list of ``DiffEntry`` records. It powers the CLI's per-cycle
13
+ mutation-diff card (improvement #2) and is reuseable by any embedder that wants
14
+ to surface what a candidate changed vs its parent.
15
+
16
+ Only *real* changes surface — identical Specs yield ``[]``. Knob changes are
17
+ classified against ``m._NAMED_KNOBS`` (the single source lifted into
18
+ ``archforge.models``): the named LLM knobs (temperature/retries/max_tokens) and
19
+ the kind-specific extras (top_k/threshold/...) are both reported as
20
+ ``kind="knob"``; ``tunable`` is editability *metadata*, not a knob value, and is
21
+ NOT reported. Deterministic order: parent-spec node order for matched/removed
22
+ nodes, candidate-spec node order for added nodes, then edges.
23
+ """
24
+ from __future__ import annotations
25
+
26
+ from dataclasses import dataclass
27
+ from typing import Any
28
+
29
+ import archforge.models as m
30
+
31
+
32
+ @dataclass(frozen=True)
33
+ class DiffEntry:
34
+ """One field-level change between a parent Spec and its candidate.
35
+
36
+ ``kind`` : category of the change
37
+ (knob|prompt|model|add_node|remove_node|edge|role|kind)
38
+ ``target`` : the node_id (for node fields) or "from->to" (for edges)
39
+ ``field`` : the field name on the target (system_prompt|model|role|kind|
40
+ <knob_name>|edge|gate)
41
+ ``old`` : the parent's value (None for an addition);
42
+ ``new`` : the candidate's value (None for a removal)
43
+ """
44
+
45
+ kind: str
46
+ target: str
47
+ field: str
48
+ old: Any
49
+ new: Any
50
+
51
+ def one_liner(self) -> str:
52
+ """Compact ``field: old -> new`` for a card's change line."""
53
+ return f"{self.field}: {_fmt(self.old)} -> {_fmt(self.new)}"
54
+
55
+ def as_dict(self) -> dict[str, Any]:
56
+ return {"kind": self.kind, "target": self.target, "field": self.field,
57
+ "old": self.old, "new": self.new}
58
+
59
+
60
+ def _fmt(v: Any) -> str:
61
+ if v is None:
62
+ return "(unset)"
63
+ if isinstance(v, bool):
64
+ return str(v).lower()
65
+ return str(v)
66
+
67
+
68
+ # --------------------------------------------------------------------------- #
69
+ # the diff
70
+ # --------------------------------------------------------------------------- #
71
+
72
+
73
+ def spec_diff(parent: m.Spec, candidate: m.Spec) -> list[DiffEntry]:
74
+ """Field-level diff of ``candidate`` vs ``parent`` (deterministic, pure).
75
+
76
+ Node diffs come before edge diffs. Within a matched node the order is
77
+ system_prompt, model, role, kind, then knobs (named knobs in a fixed order,
78
+ then extras). Returns ``[]`` for identical Specs.
79
+ """
80
+ out: list[DiffEntry] = []
81
+ pnodes: dict[str, m.Node] = {n.node_id: n for n in parent.nodes}
82
+ cnodes: dict[str, m.Node] = {n.node_id: n for n in candidate.nodes}
83
+
84
+ # matched + removed nodes (parent order), then added nodes (candidate order).
85
+ for pn in parent.nodes:
86
+ cn = cnodes.get(pn.node_id)
87
+ if cn is None:
88
+ out.append(DiffEntry("remove_node", pn.node_id, "node", pn.node_id, None))
89
+ continue
90
+ out.extend(_node_diff(pn, cn))
91
+ for cn in candidate.nodes:
92
+ if cn.node_id not in pnodes:
93
+ out.append(DiffEntry("add_node", cn.node_id, "node", None, cn.node_id))
94
+
95
+ out.extend(_edge_diff(parent, candidate))
96
+ return out
97
+
98
+
99
+ def _node_diff(parent: m.Node, candidate: m.Node) -> list[DiffEntry]:
100
+ nid = parent.node_id
101
+ out: list[DiffEntry] = []
102
+ if parent.system_prompt != candidate.system_prompt:
103
+ out.append(DiffEntry("prompt", nid, "system_prompt",
104
+ parent.system_prompt, candidate.system_prompt))
105
+ if parent.model != candidate.model:
106
+ out.append(DiffEntry("model", nid, "model", parent.model, candidate.model))
107
+ if parent.role != candidate.role:
108
+ out.append(DiffEntry("role", nid, "role", parent.role, candidate.role))
109
+ if parent.kind is not candidate.kind:
110
+ out.append(DiffEntry("kind", nid, "kind", parent.kind.value, candidate.kind.value))
111
+ out.extend(_knob_diff(nid, parent.knobs, candidate.knobs))
112
+ return out
113
+
114
+
115
+ def _knob_diff(nid: str, pk: m.Knobs, ck: m.Knobs) -> list[DiffEntry]:
116
+ out: list[DiffEntry] = []
117
+ pd = pk.model_dump()
118
+ cd = ck.model_dump()
119
+ # named knobs in a fixed stable order, then extras in candidate insertion
120
+ # order, then parent-only extras (a knob dropped from the candidate).
121
+ named = ["temperature", "retries", "max_tokens"]
122
+ extras = [k for k in cd if k not in m._NAMED_KNOBS]
123
+ parent_extras = [k for k in pd if k not in m._NAMED_KNOBS and k not in extras]
124
+ for key in named + extras + parent_extras:
125
+ pv = pd.get(key)
126
+ cv = cd.get(key)
127
+ if pv != cv:
128
+ out.append(DiffEntry("knob", nid, key, pv, cv))
129
+ return out
130
+
131
+
132
+ def _edge_key(e: m.Edge) -> tuple[str, str, str]:
133
+ return (e.from_, e.to, e.type.value)
134
+
135
+
136
+ def _edge_diff(parent: m.Spec, candidate: m.Spec) -> list[DiffEntry]:
137
+ out: list[DiffEntry] = []
138
+ pedge: dict[tuple[str, str, str], m.Edge] = {_edge_key(e): e for e in parent.edges}
139
+ for pe in parent.edges:
140
+ ce = next((c for c in candidate.edges if _edge_key(c) == _edge_key(pe)), None)
141
+ if ce is None:
142
+ out.append(DiffEntry("edge", _edge_label(pe), "edge",
143
+ _edge_describe(pe), None))
144
+ elif pe.gate != ce.gate:
145
+ out.append(DiffEntry("edge", _edge_label(pe), "gate", pe.gate, ce.gate))
146
+ for ce in candidate.edges:
147
+ if _edge_key(ce) not in pedge:
148
+ out.append(DiffEntry("edge", _edge_label(ce), "edge", None, _edge_describe(ce)))
149
+ return out
150
+
151
+
152
+ def _edge_label(e: m.Edge) -> str:
153
+ return f"{e.from_}->{e.to}"
154
+
155
+
156
+ def _edge_describe(e: m.Edge) -> str:
157
+ s = e.type.value
158
+ if e.gate:
159
+ s += f" gate={e.gate}"
160
+ return s
161
+
162
+
163
+ # --------------------------------------------------------------------------- #
164
+ # rendering helper — a compact one-liner for the CLI's change card
165
+ # --------------------------------------------------------------------------- #
166
+
167
+
168
+ def format_diff(entries: list[DiffEntry]) -> str:
169
+ """Compact one-liner for a card's change line, from a ``spec_diff`` list.
170
+
171
+ Field-level diffs (knob/prompt/model/role/kind on matched nodes) are joined
172
+ with "; " (most changes are a single field). Structural diffs (added/removed
173
+ nodes + edges — ADD_NODE/REMOVE_NODE/REWIRE) are summarized as a roster delta
174
+ (``+node v +edge 1 -edge 1 ~edge 1``) so the card stays one line even when a
175
+ -- proposed change touches several edges/nodes at once. Any field-level diffs
176
+ riding alongside a structural change (e.g. a rewire that also swapped a model)
177
+ are appended so nothing is silently dropped.
178
+ """
179
+ if not entries:
180
+ return "(no change)"
181
+ structural = [e for e in entries if e.kind in ("add_node", "remove_node", "edge")]
182
+ if not structural:
183
+ return "; ".join(e.one_liner() for e in entries)
184
+ parts: list[str] = []
185
+ added = [e.target for e in structural if e.kind == "add_node"]
186
+ removed = [e.target for e in structural if e.kind == "remove_node"]
187
+ added_edges = [e for e in structural if e.kind == "edge" and e.old is None]
188
+ removed_edges = [e for e in structural if e.kind == "edge" and e.new is None]
189
+ changed_edges = [e for e in structural if e.kind == "edge" and e.old is not None and e.new is not None]
190
+ if added:
191
+ parts.append(f"+node {' '.join(added)}")
192
+ if removed:
193
+ parts.append(f"-node {' '.join(removed)}")
194
+ if added_edges:
195
+ parts.append(f"+edge {len(added_edges)}")
196
+ if removed_edges:
197
+ parts.append(f"-edge {len(removed_edges)}")
198
+ if changed_edges:
199
+ parts.append(f"~edge {len(changed_edges)}")
200
+ field_level = [e for e in entries if e.kind not in ("add_node", "remove_node", "edge")]
201
+ if field_level:
202
+ parts.append("; ".join(e.one_liner() for e in field_level))
203
+ return " ".join(parts)
204
+
205
+
206
+ __all__ = ["DiffEntry", "spec_diff", "format_diff"]