archforge-optimizer 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- archforge/__init__.py +76 -0
- archforge/__main__.py +10 -0
- archforge/architect.py +442 -0
- archforge/cli.py +881 -0
- archforge/config.py +140 -0
- archforge/config_init.py +150 -0
- archforge/diff.py +206 -0
- archforge/engine.py +444 -0
- archforge/gatekeeper.py +290 -0
- archforge/host/__init__.py +20 -0
- archforge/host/adapters/__init__.py +41 -0
- archforge/host/adapters/base.py +311 -0
- archforge/host/adapters/helpers.py +163 -0
- archforge/host/adapters/langgraph.py +726 -0
- archforge/host/base.py +105 -0
- archforge/host/fake.py +380 -0
- archforge/judge/__init__.py +20 -0
- archforge/judge/base.py +257 -0
- archforge/judge/scripted.py +145 -0
- archforge/lint.py +180 -0
- archforge/llm/__init__.py +65 -0
- archforge/llm/_common.py +94 -0
- archforge/llm/anthropic.py +90 -0
- archforge/llm/base.py +90 -0
- archforge/llm/gemini.py +112 -0
- archforge/llm/groq.py +63 -0
- archforge/llm/openai.py +63 -0
- archforge/llm/scripted.py +134 -0
- archforge/middleware.py +181 -0
- archforge/models.py +435 -0
- archforge/mutate.py +214 -0
- archforge/otel.py +613 -0
- archforge/runlog.py +103 -0
- archforge/runner.py +153 -0
- archforge/spec_builder.py +126 -0
- archforge/stores/__init__.py +22 -0
- archforge/stores/_jsonl.py +81 -0
- archforge/stores/attempt_store.py +161 -0
- archforge/stores/spec_store.py +188 -0
- archforge/stores/trace_store.py +42 -0
- archforge/suite.py +248 -0
- archforge/userconfig.py +144 -0
- archforge_optimizer-0.1.0.dist-info/METADATA +420 -0
- archforge_optimizer-0.1.0.dist-info/RECORD +47 -0
- archforge_optimizer-0.1.0.dist-info/WHEEL +4 -0
- archforge_optimizer-0.1.0.dist-info/entry_points.txt +2 -0
- archforge_optimizer-0.1.0.dist-info/licenses/LICENSE +21 -0
archforge/config.py
ADDED
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
"""ArchForge system config — internals only (NOT user-tunable).
|
|
2
|
+
|
|
3
|
+
This holds the NON-tunable framework plumbing: identity (``VERSION``), the LLM
|
|
4
|
+
provider ROSTER (the tuples the CLI's ``--provider`` choices come from), the
|
|
5
|
+
on-disk storage DIRNAMES + pointer files, the content-addressing hash lengths,
|
|
6
|
+
Anthropic's required ``max_tokens``, the ScriptedJudge noise pattern, the CLI
|
|
7
|
+
program name + its scripted fixtures, and ``load_env`` (the ``.env`` loader for
|
|
8
|
+
provider API keys). **Nothing here is a user preference** — every value a user
|
|
9
|
+
might tune (τ, δ, R, the default provider + models, the rubric, the storage root,
|
|
10
|
+
token budgets, grader resilience …) lives instead in the project's
|
|
11
|
+
``.archforge/archforge.py`` (made by ``archforge-optimizer init``) and is read
|
|
12
|
+
lazily by ``archforge.userconfig`` at use time. So ``config.py`` ↔ ``archforge.py``
|
|
13
|
+
share NO variable (the two-file split), and this module never depends on the
|
|
14
|
+
user's config — it imports cleanly before ``init`` has run.
|
|
15
|
+
|
|
16
|
+
Pure leaf: stdlib + ``python-dotenv`` (a core dependency) only — imports nothing
|
|
17
|
+
else from ``archforge``, so every other module may ``from archforge.config import …``
|
|
18
|
+
with no risk of a back-edge.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import importlib.util
|
|
24
|
+
import os
|
|
25
|
+
from pathlib import Path
|
|
26
|
+
|
|
27
|
+
# =========================================================================== #
|
|
28
|
+
# Identity
|
|
29
|
+
# =========================================================================== #
|
|
30
|
+
# The package version. Read by hatchling for the built distribution and
|
|
31
|
+
# re-exported as `archforge.__version__` (archforge/__init__.py). One place.
|
|
32
|
+
VERSION: str = "0.1.0"
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
# =========================================================================== #
|
|
36
|
+
# LLM providers — the ROSTER only (NOT the default choice; that's a tunable)
|
|
37
|
+
# =========================================================================== #
|
|
38
|
+
# The providers the CLI's `--provider` flag ACCEPTS (its `choices`). `scripted` is
|
|
39
|
+
# the zero-cost option (deterministic fakes, no SDK, no API key); the real
|
|
40
|
+
# providers each need their SDK + an API key. WHICH provider is the default, and
|
|
41
|
+
# each provider's default MODEL id, are tunables → they live in .archforge/archforge.py
|
|
42
|
+
# (resolved by archforge.userconfig), NOT here. `REAL_PROVIDERS`/`ALL_PROVIDERS` are
|
|
43
|
+
# re-exported from archforge.llm; `make_client(provider)` dispatches on these.
|
|
44
|
+
SCRIPTED_PROVIDER: str = "scripted"
|
|
45
|
+
REAL_PROVIDERS: tuple[str, ...] = ("anthropic", "openai", "groq", "gemini")
|
|
46
|
+
ALL_PROVIDERS: tuple[str, ...] = (SCRIPTED_PROVIDER,) + REAL_PROVIDERS
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
# =========================================================================== #
|
|
50
|
+
# Project environment (`.env`) — how secrets reach the provider SDKs
|
|
51
|
+
# =========================================================================== #
|
|
52
|
+
# A pip-installed `archforge` reads the project folder's `.env` so a checked-out
|
|
53
|
+
# repo "just runs" once API keys are added (the `.env` is gitignored — never commit
|
|
54
|
+
# secrets). `load_env()` populates `os.environ` with `override=False`, so the
|
|
55
|
+
# precedence is: an explicit `--api-key` (passed straight to the SDK by the CLI)
|
|
56
|
+
# > a real process env var > the `.env` file. `python-dotenv` is a CORE dependency
|
|
57
|
+
# here (handles `export ` prefixes, quoting, multiline values robustly); an absent
|
|
58
|
+
# file is a no-op. Provider env-var names are NOT centralized here — the provider
|
|
59
|
+
# SDKs read their own (`ANTHROPIC_API_KEY` … `GEMINI_API_KEY`); this loader only
|
|
60
|
+
# populates env.
|
|
61
|
+
#
|
|
62
|
+
# The DEFAULT `.env` path is a tunable (DEFAULT_ENV_FILE in archforge.py); this
|
|
63
|
+
# loader's own default arg is the literal ".env" so the function stays a pure leaf
|
|
64
|
+
# that never touches the resolver (callable pre-init, e.g. to load keys).
|
|
65
|
+
|
|
66
|
+
def load_env(path: str | os.PathLike[str] | None = ".env") -> None:
|
|
67
|
+
"""Populate ``os.environ`` from a ``.env`` file (via python-dotenv, no override).
|
|
68
|
+
|
|
69
|
+
A real process env var therefore always wins (``override=False``); the CLI's
|
|
70
|
+
``--api-key`` (handed straight to the SDK) wins above both. No-op when the file
|
|
71
|
+
is absent or ``path`` is falsy — so callers (CLI, runner) invoke it
|
|
72
|
+
unconditionally. ``python-dotenv`` (core dep) does the parsing.
|
|
73
|
+
"""
|
|
74
|
+
if not path:
|
|
75
|
+
return
|
|
76
|
+
if importlib.util.find_spec("dotenv") is not None:
|
|
77
|
+
from dotenv import load_dotenv # type: ignore[import-not-found]
|
|
78
|
+
load_dotenv(path, override=False) # override=False ⇒ real env wins
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
# =========================================================================== #
|
|
82
|
+
# Storage layout
|
|
83
|
+
# =========================================================================== #
|
|
84
|
+
# On-disk layout under the CLI's `--root` (a path the user passes, or ".archforge" —
|
|
85
|
+
# a tunable, but the dirnames themselves are plumbing). The stores name their
|
|
86
|
+
# subdirectories + pointer files from these; rare to change.
|
|
87
|
+
SPECS_DIRNAME: str = "specs"
|
|
88
|
+
ATTEMPTS_DIRNAME: str = "attempts"
|
|
89
|
+
TRACES_DIRNAME: str = "traces"
|
|
90
|
+
ACTIVE_POINTER_FILE: str = "active.pointer" # names the active incumbent spec_id
|
|
91
|
+
ARCHIVED_FILE: str = "archived.jsonl" # rolled-back spec_ids (I3 reachability)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
# =========================================================================== #
|
|
95
|
+
# Internals — not for tuning
|
|
96
|
+
# =========================================================================== #
|
|
97
|
+
# Bookkeeping values wired into specific consumers. Change only if you know the
|
|
98
|
+
# consumer; these are plumbing, not user preferences.
|
|
99
|
+
|
|
100
|
+
# Content-addressing hash truncation (sha256 hex prefix lengths). SPEC_ID_HASH_LEN
|
|
101
|
+
# is the Spec/Attempt content id (models.py, attempt_store); SHORT_HASH_LEN is the
|
|
102
|
+
# short deterministic suffixes used by the fake host (responder + run_id).
|
|
103
|
+
SPEC_ID_HASH_LEN: int = 16
|
|
104
|
+
SHORT_HASH_LEN: int = 8
|
|
105
|
+
|
|
106
|
+
# Anthropic's messages API requires a max_tokens; used when a caller omits it
|
|
107
|
+
# (only the Anthropic adapter). Not a user preference — a provider requirement.
|
|
108
|
+
ANTHROPIC_DEFAULT_MAX_TOKENS: int = 1024
|
|
109
|
+
|
|
110
|
+
# The ScriptedJudge's deterministic noise band (E1 jitter pattern): the jitter
|
|
111
|
+
# applied to a scripted aggregate is `pattern[run_index % len] * noise_width`,
|
|
112
|
+
# so identical runs yield identical jitter (reproducible E1 margin-boundary tests).
|
|
113
|
+
SCRIPTED_NOISE_PATTERN: tuple[float, ...] = (1.0, 0.0, -1.0, 0.5, -0.5, 0.25, -0.25)
|
|
114
|
+
|
|
115
|
+
# The CLI program name (the installed console command + the prefix on result lines).
|
|
116
|
+
PROG: str = "archforge-optimizer"
|
|
117
|
+
|
|
118
|
+
# The CLI's free-run scripted fixtures: when `--provider scripted` is invoked
|
|
119
|
+
# with no injected `components`, the evolve family runs against this one-task suite
|
|
120
|
+
# so `archforge-optimizer evolve` is runnable end-to-end with zero configuration.
|
|
121
|
+
DEFAULT_SUITE_ID: str = "cli-default"
|
|
122
|
+
DEFAULT_TASK_ID: str = "t1"
|
|
123
|
+
DEFAULT_TASK_INPUT: str = "hello"
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
__all__ = [
|
|
127
|
+
# identity
|
|
128
|
+
"VERSION",
|
|
129
|
+
# llm provider roster
|
|
130
|
+
"SCRIPTED_PROVIDER", "REAL_PROVIDERS", "ALL_PROVIDERS",
|
|
131
|
+
# project environment (.env loader for provider API keys)
|
|
132
|
+
"load_env",
|
|
133
|
+
# storage layout
|
|
134
|
+
"SPECS_DIRNAME", "ATTEMPTS_DIRNAME", "TRACES_DIRNAME",
|
|
135
|
+
"ACTIVE_POINTER_FILE", "ARCHIVED_FILE",
|
|
136
|
+
# internals
|
|
137
|
+
"SPEC_ID_HASH_LEN", "SHORT_HASH_LEN", "ANTHROPIC_DEFAULT_MAX_TOKENS",
|
|
138
|
+
"SCRIPTED_NOISE_PATTERN", "PROG",
|
|
139
|
+
"DEFAULT_SUITE_ID", "DEFAULT_TASK_ID", "DEFAULT_TASK_INPUT",
|
|
140
|
+
]
|
archforge/config_init.py
ADDED
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
"""Scaffolding text for `archforge-optimizer init` — the user-tunable config template.
|
|
2
|
+
|
|
3
|
+
Owns the ACTIVE defaults that become `.archforge/archforge.py` (the project's sole
|
|
4
|
+
source of ArchForge tunables, made by `init`) and the `.env.example` key template.
|
|
5
|
+
|
|
6
|
+
Two roles, one source:
|
|
7
|
+
* `init` writes ``archforge_config_text()`` verbatim to ``.archforge/archforge.py`` →
|
|
8
|
+
the user's tunables. It ships with **active sane values** so the CLI works
|
|
9
|
+
immediately after `init`; the user edits a value to change behaviour.
|
|
10
|
+
* Under the test runner, ``archforge.userconfig`` execs the SAME ``TEMPLATE``
|
|
11
|
+
in-memory (no disk file) so the suite sees the sane defaults
|
|
12
|
+
(e.g. ``Thresholds().tau == 0.05``) with zero per-test files.
|
|
13
|
+
|
|
14
|
+
The values here are ArchForge's sane defaults — keep them in sync with what the
|
|
15
|
+
package historically shipped. This module is pure data (string constants); it imports
|
|
16
|
+
nothing from `archforge` and never reads the live `.env`.
|
|
17
|
+
"""
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
# --------------------------------------------------------------------------- #
|
|
21
|
+
# the sane tunable defaults — the ONE place the values live
|
|
22
|
+
# --------------------------------------------------------------------------- #
|
|
23
|
+
# (name, active-value-as-assignment, one-line purpose). The assignment text is
|
|
24
|
+
# emitted verbatim into the generated file AND exec'd by the resolver, so it must be
|
|
25
|
+
# valid Python and carry the real sane value (note DEFAULT_ARCHITECT_MODELS /
|
|
26
|
+
# DEFAULT_JUDGE_MODELS / DEFAULT_SUB_RUBRICS are the full dicts, not `{}` — the
|
|
27
|
+
# resolver must resolve them without KeyError).
|
|
28
|
+
|
|
29
|
+
_DEFAULT_ARCHITECT_MODELS = (
|
|
30
|
+
'{"anthropic": "claude-sonnet-5", "openai": "gpt-4o", '
|
|
31
|
+
'"groq": "openai/gpt-oss-120b", "gemini": "gemini-3.6-flash"}'
|
|
32
|
+
)
|
|
33
|
+
_DEFAULT_JUDGE_MODELS = (
|
|
34
|
+
'{"anthropic": "claude-sonnet-5", "openai": "gpt-4o", '
|
|
35
|
+
'"groq": "openai/gpt-oss-120b", "gemini": "gemini-3.6-flash"}'
|
|
36
|
+
)
|
|
37
|
+
_DEFAULT_SUB_RUBRICS = (
|
|
38
|
+
'{"correctness": "Is the final answer factually correct and aligned with the task?", '
|
|
39
|
+
'"completeness": "Does the answer address every part of the task?", '
|
|
40
|
+
'"grounding": "Are the claims supported by the inputs/context, not invented?"}'
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
# The starter suite `init` writes to .archforge/suite.json — byte-identical to the
|
|
44
|
+
# one-task CLI fallback fixture (cli-default / t1 / hello), so the generated default
|
|
45
|
+
# round-trips to the same Suite the CLI builds when the file is absent. A per-task
|
|
46
|
+
# rubric_id is optional in the file; omitting it scores against the active rubric.
|
|
47
|
+
_DEFAULT_SUITE_JSON = (
|
|
48
|
+
'{\n'
|
|
49
|
+
' "suite_id": "cli-default",\n'
|
|
50
|
+
' "tasks": [\n'
|
|
51
|
+
' {"task_id": "t1", "input": "hello"}\n'
|
|
52
|
+
' ]\n'
|
|
53
|
+
'}\n'
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
_FIELDS: tuple[tuple[str, str, str], ...] = (
|
|
57
|
+
# --- LLM provider
|
|
58
|
+
("PROVIDER", '"gemini"',
|
|
59
|
+
"which LLM to use (scripted|anthropic|openai|groq|gemini); scripted needs no API key"),
|
|
60
|
+
("DEFAULT_ARCHITECT_MODELS", _DEFAULT_ARCHITECT_MODELS,
|
|
61
|
+
"default Architect (proposer) model per provider; a bare LLMClient call falls "
|
|
62
|
+
"back here too — edit the dict to change it"),
|
|
63
|
+
("DEFAULT_JUDGE_MODELS", _DEFAULT_JUDGE_MODELS,
|
|
64
|
+
"default Judge (scorer) model per provider — edit the dict to change it"),
|
|
65
|
+
# --- optimization policy
|
|
66
|
+
("DEFAULT_TAU", "0.05", "how much better a candidate must score to be promoted (τ)"),
|
|
67
|
+
("DEFAULT_DELTA", "0.07", "how far a promoted run can drop before it's rolled back (δ, >= τ)"),
|
|
68
|
+
("DEFAULT_REPEATS", "1", "how many times each eval task is run; more = steadier scores, more cost (R)"),
|
|
69
|
+
("MAX_REPEATS", "3", "upper bound on R"),
|
|
70
|
+
("DEFAULT_UNRUNNABLE_FRAC", "0.25", "drop a candidate if more than this fraction of its tasks crash (ε)"),
|
|
71
|
+
("DEFAULT_PLATEAU_CYCLES", "5", "stop after this many cycles in a row with no improvement (K)"),
|
|
72
|
+
("DEFAULT_MAX_CYCLES", "20", "max optimization cycles per run"),
|
|
73
|
+
# --- grader resilience
|
|
74
|
+
("DEFAULT_JUDGE_RETRIES", "2", "how many times to retry a failed judge call"),
|
|
75
|
+
("BACKOFF_CAP_SECONDS", "30.0", "max seconds to wait between judge retries"),
|
|
76
|
+
# --- judge scoring
|
|
77
|
+
("DEFAULT_RUBRIC_ID", '"default-v1"', "name of the scoring rubric (keep it stable so runs compare)"),
|
|
78
|
+
("DEFAULT_SUB_RUBRICS", _DEFAULT_SUB_RUBRICS,
|
|
79
|
+
"the rubric's dimensions: what a high score looks like, per dimension"),
|
|
80
|
+
# --- environment / budget / storage
|
|
81
|
+
("DEFAULT_ROOT_DIR", '".archforge"', "where run state is written (relative to where you run the CLI)"),
|
|
82
|
+
("DEFAULT_ENV_FILE", '".env"', "the .env file loaded for API keys (a real env var always wins)"),
|
|
83
|
+
("DEFAULT_SUITE_FILE", '".archforge/suite.json"', "the suite file defining your eval tasks (absent → the one-task default)"),
|
|
84
|
+
("DEFAULT_MAX_TOKENS_TOTAL", "None", "whole-run token budget cap; None = no limit"),
|
|
85
|
+
("DEFAULT_MAX_TOKENS_PER_CYCLE", "None", "per-cycle token cap (aborts mid-cycle if exceeded); None = no limit"),
|
|
86
|
+
("DEFAULT_MAX_WALL_MS_PER_CYCLE", "None", "per-cycle wall-clock cap (ms); aborts if exceeded — "
|
|
87
|
+
"covers non-LLM nodes (retriever/tool/rule) that cost time, not tokens; None = no limit"),
|
|
88
|
+
("DEFAULT_TRACE_TOTAL_BUDGET_TOK", "None", "total Judge-prompt token budget for OTel trace "
|
|
89
|
+
"projection of per-step LLM prompt/completion; None = lossy summarize() path (parity, "
|
|
90
|
+
"no tracing); an int turns on rich per-step Steps, shedding largest-evidence chunks first"),
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
# section break points in _FIELDS (for grouping the emitted file)
|
|
94
|
+
_BREAKS: dict[int, str] = {
|
|
95
|
+
3: "# --- optimization policy ------------------------------------------------",
|
|
96
|
+
10: "# --- grader resilience ---------------------------------------------------",
|
|
97
|
+
12: "# --- judge scoring -------------------------------------------------------",
|
|
98
|
+
14: "# --- environment / budget / storage --------------------------------------",
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
_HEADER = """\
|
|
102
|
+
# archforge.py — your ArchForge config (made by `archforge-optimizer init`).
|
|
103
|
+
#
|
|
104
|
+
# The values below already work — the CLI runs as-is after `init`. Edit any value
|
|
105
|
+
# to change that default. Nothing here is required to make ArchForge import.
|
|
106
|
+
#
|
|
107
|
+
|
|
108
|
+
# --- LLM provider ------------------------------------------------------------
|
|
109
|
+
"""
|
|
110
|
+
|
|
111
|
+
_FOOTER = """
|
|
112
|
+
# Changes here take effect on the next `archforge-optimizer` run.
|
|
113
|
+
"""
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def archforge_config_text() -> str:
|
|
117
|
+
"""The full body of the generated `.archforge/archforge.py` (active sane defaults)."""
|
|
118
|
+
lines = [_HEADER.rstrip("\n")]
|
|
119
|
+
for i, (name, value, purpose) in enumerate(_FIELDS):
|
|
120
|
+
if i in _BREAKS:
|
|
121
|
+
lines.append("")
|
|
122
|
+
lines.append(_BREAKS[i])
|
|
123
|
+
lines.append(f"# {purpose}")
|
|
124
|
+
lines.append(f"{name} = {value}")
|
|
125
|
+
lines.append(_FOOTER.rstrip("\n"))
|
|
126
|
+
return "\n".join(lines) + "\n"
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
# The active sane-default assignments, exec'd by archforge.userconfig under the test
|
|
130
|
+
# runner (so the suite sees sane defaults) — same string `init` writes to disk.
|
|
131
|
+
TEMPLATE: str = archforge_config_text()
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def env_example_text() -> str:
|
|
135
|
+
"""The body of the generated `.env.example` (empty provider key var names)."""
|
|
136
|
+
return (
|
|
137
|
+
"# fill in your API keys for provider you are going to use for archforge-optimizer.\n\n"
|
|
138
|
+
"# ANTHROPIC_API_KEY=\n"
|
|
139
|
+
"# OPENAI_API_KEY=\n"
|
|
140
|
+
"# GROQ_API_KEY=\n"
|
|
141
|
+
"# GEMINI_API_KEY=\n"
|
|
142
|
+
"\n# Other service keys you need for your tasks\n"
|
|
143
|
+
)
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
# The tunable names — exported so callers/tests enumerate the editable surface.
|
|
147
|
+
EDITABLE_NAMES: tuple[str, ...] = tuple(name for name, _, _ in _FIELDS)
|
|
148
|
+
|
|
149
|
+
__all__ = ["archforge_config_text", "env_example_text", "TEMPLATE", "EDITABLE_NAMES",
|
|
150
|
+
"_DEFAULT_SUITE_JSON"]
|
archforge/diff.py
ADDED
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
"""Spec-level structural diff — what changed from the parent Spec to the candidate.
|
|
2
|
+
|
|
3
|
+
ArchForge's ``Change`` record (``m.Change``) carries the *intent* of a mutation
|
|
4
|
+
(kind/target/diff/rationale/scope) but NOT the field-level before/after — the
|
|
5
|
+
``payload`` dict evaporates after ``apply_change`` (it is metadata, not
|
|
6
|
+
persisted). So an inspectable "what did the Forge actually change" requires
|
|
7
|
+
diffing the two persisted Specs node-by-node. Both specs live in the SpecStore
|
|
8
|
+
(``spec_store.get(parent_id)`` / ``get(candidate_id)``), so the comparison is
|
|
9
|
+
cheap and needs nothing beyond the Specs themselves.
|
|
10
|
+
|
|
11
|
+
``spec_diff(parent, candidate)`` does exactly that: a pure comparison returning a
|
|
12
|
+
deterministic list of ``DiffEntry`` records. It powers the CLI's per-cycle
|
|
13
|
+
mutation-diff card (improvement #2) and is reuseable by any embedder that wants
|
|
14
|
+
to surface what a candidate changed vs its parent.
|
|
15
|
+
|
|
16
|
+
Only *real* changes surface — identical Specs yield ``[]``. Knob changes are
|
|
17
|
+
classified against ``m._NAMED_KNOBS`` (the single source lifted into
|
|
18
|
+
``archforge.models``): the named LLM knobs (temperature/retries/max_tokens) and
|
|
19
|
+
the kind-specific extras (top_k/threshold/...) are both reported as
|
|
20
|
+
``kind="knob"``; ``tunable`` is editability *metadata*, not a knob value, and is
|
|
21
|
+
NOT reported. Deterministic order: parent-spec node order for matched/removed
|
|
22
|
+
nodes, candidate-spec node order for added nodes, then edges.
|
|
23
|
+
"""
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
from dataclasses import dataclass
|
|
27
|
+
from typing import Any
|
|
28
|
+
|
|
29
|
+
import archforge.models as m
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass(frozen=True)
|
|
33
|
+
class DiffEntry:
|
|
34
|
+
"""One field-level change between a parent Spec and its candidate.
|
|
35
|
+
|
|
36
|
+
``kind`` : category of the change
|
|
37
|
+
(knob|prompt|model|add_node|remove_node|edge|role|kind)
|
|
38
|
+
``target`` : the node_id (for node fields) or "from->to" (for edges)
|
|
39
|
+
``field`` : the field name on the target (system_prompt|model|role|kind|
|
|
40
|
+
<knob_name>|edge|gate)
|
|
41
|
+
``old`` : the parent's value (None for an addition);
|
|
42
|
+
``new`` : the candidate's value (None for a removal)
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
kind: str
|
|
46
|
+
target: str
|
|
47
|
+
field: str
|
|
48
|
+
old: Any
|
|
49
|
+
new: Any
|
|
50
|
+
|
|
51
|
+
def one_liner(self) -> str:
|
|
52
|
+
"""Compact ``field: old -> new`` for a card's change line."""
|
|
53
|
+
return f"{self.field}: {_fmt(self.old)} -> {_fmt(self.new)}"
|
|
54
|
+
|
|
55
|
+
def as_dict(self) -> dict[str, Any]:
|
|
56
|
+
return {"kind": self.kind, "target": self.target, "field": self.field,
|
|
57
|
+
"old": self.old, "new": self.new}
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _fmt(v: Any) -> str:
|
|
61
|
+
if v is None:
|
|
62
|
+
return "(unset)"
|
|
63
|
+
if isinstance(v, bool):
|
|
64
|
+
return str(v).lower()
|
|
65
|
+
return str(v)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
# --------------------------------------------------------------------------- #
|
|
69
|
+
# the diff
|
|
70
|
+
# --------------------------------------------------------------------------- #
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def spec_diff(parent: m.Spec, candidate: m.Spec) -> list[DiffEntry]:
|
|
74
|
+
"""Field-level diff of ``candidate`` vs ``parent`` (deterministic, pure).
|
|
75
|
+
|
|
76
|
+
Node diffs come before edge diffs. Within a matched node the order is
|
|
77
|
+
system_prompt, model, role, kind, then knobs (named knobs in a fixed order,
|
|
78
|
+
then extras). Returns ``[]`` for identical Specs.
|
|
79
|
+
"""
|
|
80
|
+
out: list[DiffEntry] = []
|
|
81
|
+
pnodes: dict[str, m.Node] = {n.node_id: n for n in parent.nodes}
|
|
82
|
+
cnodes: dict[str, m.Node] = {n.node_id: n for n in candidate.nodes}
|
|
83
|
+
|
|
84
|
+
# matched + removed nodes (parent order), then added nodes (candidate order).
|
|
85
|
+
for pn in parent.nodes:
|
|
86
|
+
cn = cnodes.get(pn.node_id)
|
|
87
|
+
if cn is None:
|
|
88
|
+
out.append(DiffEntry("remove_node", pn.node_id, "node", pn.node_id, None))
|
|
89
|
+
continue
|
|
90
|
+
out.extend(_node_diff(pn, cn))
|
|
91
|
+
for cn in candidate.nodes:
|
|
92
|
+
if cn.node_id not in pnodes:
|
|
93
|
+
out.append(DiffEntry("add_node", cn.node_id, "node", None, cn.node_id))
|
|
94
|
+
|
|
95
|
+
out.extend(_edge_diff(parent, candidate))
|
|
96
|
+
return out
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _node_diff(parent: m.Node, candidate: m.Node) -> list[DiffEntry]:
|
|
100
|
+
nid = parent.node_id
|
|
101
|
+
out: list[DiffEntry] = []
|
|
102
|
+
if parent.system_prompt != candidate.system_prompt:
|
|
103
|
+
out.append(DiffEntry("prompt", nid, "system_prompt",
|
|
104
|
+
parent.system_prompt, candidate.system_prompt))
|
|
105
|
+
if parent.model != candidate.model:
|
|
106
|
+
out.append(DiffEntry("model", nid, "model", parent.model, candidate.model))
|
|
107
|
+
if parent.role != candidate.role:
|
|
108
|
+
out.append(DiffEntry("role", nid, "role", parent.role, candidate.role))
|
|
109
|
+
if parent.kind is not candidate.kind:
|
|
110
|
+
out.append(DiffEntry("kind", nid, "kind", parent.kind.value, candidate.kind.value))
|
|
111
|
+
out.extend(_knob_diff(nid, parent.knobs, candidate.knobs))
|
|
112
|
+
return out
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _knob_diff(nid: str, pk: m.Knobs, ck: m.Knobs) -> list[DiffEntry]:
|
|
116
|
+
out: list[DiffEntry] = []
|
|
117
|
+
pd = pk.model_dump()
|
|
118
|
+
cd = ck.model_dump()
|
|
119
|
+
# named knobs in a fixed stable order, then extras in candidate insertion
|
|
120
|
+
# order, then parent-only extras (a knob dropped from the candidate).
|
|
121
|
+
named = ["temperature", "retries", "max_tokens"]
|
|
122
|
+
extras = [k for k in cd if k not in m._NAMED_KNOBS]
|
|
123
|
+
parent_extras = [k for k in pd if k not in m._NAMED_KNOBS and k not in extras]
|
|
124
|
+
for key in named + extras + parent_extras:
|
|
125
|
+
pv = pd.get(key)
|
|
126
|
+
cv = cd.get(key)
|
|
127
|
+
if pv != cv:
|
|
128
|
+
out.append(DiffEntry("knob", nid, key, pv, cv))
|
|
129
|
+
return out
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def _edge_key(e: m.Edge) -> tuple[str, str, str]:
|
|
133
|
+
return (e.from_, e.to, e.type.value)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _edge_diff(parent: m.Spec, candidate: m.Spec) -> list[DiffEntry]:
|
|
137
|
+
out: list[DiffEntry] = []
|
|
138
|
+
pedge: dict[tuple[str, str, str], m.Edge] = {_edge_key(e): e for e in parent.edges}
|
|
139
|
+
for pe in parent.edges:
|
|
140
|
+
ce = next((c for c in candidate.edges if _edge_key(c) == _edge_key(pe)), None)
|
|
141
|
+
if ce is None:
|
|
142
|
+
out.append(DiffEntry("edge", _edge_label(pe), "edge",
|
|
143
|
+
_edge_describe(pe), None))
|
|
144
|
+
elif pe.gate != ce.gate:
|
|
145
|
+
out.append(DiffEntry("edge", _edge_label(pe), "gate", pe.gate, ce.gate))
|
|
146
|
+
for ce in candidate.edges:
|
|
147
|
+
if _edge_key(ce) not in pedge:
|
|
148
|
+
out.append(DiffEntry("edge", _edge_label(ce), "edge", None, _edge_describe(ce)))
|
|
149
|
+
return out
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _edge_label(e: m.Edge) -> str:
|
|
153
|
+
return f"{e.from_}->{e.to}"
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def _edge_describe(e: m.Edge) -> str:
|
|
157
|
+
s = e.type.value
|
|
158
|
+
if e.gate:
|
|
159
|
+
s += f" gate={e.gate}"
|
|
160
|
+
return s
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
# --------------------------------------------------------------------------- #
|
|
164
|
+
# rendering helper — a compact one-liner for the CLI's change card
|
|
165
|
+
# --------------------------------------------------------------------------- #
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def format_diff(entries: list[DiffEntry]) -> str:
|
|
169
|
+
"""Compact one-liner for a card's change line, from a ``spec_diff`` list.
|
|
170
|
+
|
|
171
|
+
Field-level diffs (knob/prompt/model/role/kind on matched nodes) are joined
|
|
172
|
+
with "; " (most changes are a single field). Structural diffs (added/removed
|
|
173
|
+
nodes + edges — ADD_NODE/REMOVE_NODE/REWIRE) are summarized as a roster delta
|
|
174
|
+
(``+node v +edge 1 -edge 1 ~edge 1``) so the card stays one line even when a
|
|
175
|
+
-- proposed change touches several edges/nodes at once. Any field-level diffs
|
|
176
|
+
riding alongside a structural change (e.g. a rewire that also swapped a model)
|
|
177
|
+
are appended so nothing is silently dropped.
|
|
178
|
+
"""
|
|
179
|
+
if not entries:
|
|
180
|
+
return "(no change)"
|
|
181
|
+
structural = [e for e in entries if e.kind in ("add_node", "remove_node", "edge")]
|
|
182
|
+
if not structural:
|
|
183
|
+
return "; ".join(e.one_liner() for e in entries)
|
|
184
|
+
parts: list[str] = []
|
|
185
|
+
added = [e.target for e in structural if e.kind == "add_node"]
|
|
186
|
+
removed = [e.target for e in structural if e.kind == "remove_node"]
|
|
187
|
+
added_edges = [e for e in structural if e.kind == "edge" and e.old is None]
|
|
188
|
+
removed_edges = [e for e in structural if e.kind == "edge" and e.new is None]
|
|
189
|
+
changed_edges = [e for e in structural if e.kind == "edge" and e.old is not None and e.new is not None]
|
|
190
|
+
if added:
|
|
191
|
+
parts.append(f"+node {' '.join(added)}")
|
|
192
|
+
if removed:
|
|
193
|
+
parts.append(f"-node {' '.join(removed)}")
|
|
194
|
+
if added_edges:
|
|
195
|
+
parts.append(f"+edge {len(added_edges)}")
|
|
196
|
+
if removed_edges:
|
|
197
|
+
parts.append(f"-edge {len(removed_edges)}")
|
|
198
|
+
if changed_edges:
|
|
199
|
+
parts.append(f"~edge {len(changed_edges)}")
|
|
200
|
+
field_level = [e for e in entries if e.kind not in ("add_node", "remove_node", "edge")]
|
|
201
|
+
if field_level:
|
|
202
|
+
parts.append("; ".join(e.one_liner() for e in field_level))
|
|
203
|
+
return " ".join(parts)
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
__all__ = ["DiffEntry", "spec_diff", "format_diff"]
|