fanbase 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fanbase/__init__.py +3 -0
- fanbase/analyze.py +125 -0
- fanbase/cli.py +264 -0
- fanbase/importers/__init__.py +6 -0
- fanbase/importers/corpus.py +145 -0
- fanbase/importers/ossfuzz_eval.py +112 -0
- fanbase/metrics/__init__.py +12 -0
- fanbase/metrics/intrinsics.py +220 -0
- fanbase/metrics/selfcheck.py +155 -0
- fanbase/registry.py +193 -0
- fanbase/runner.py +107 -0
- fanbase/schemas/__init__.py +0 -0
- fanbase/schemas/format.schema.json +30 -0
- fanbase/schemas/lineage.schema.json +27 -0
- fanbase/schemas/profile.schema.json +20 -0
- fanbase/schemas/profiles.schema.json +12 -0
- fanbase/spec.py +151 -0
- fanbase/validate.py +128 -0
- fanbase-0.1.0.dist-info/METADATA +30 -0
- fanbase-0.1.0.dist-info/RECORD +22 -0
- fanbase-0.1.0.dist-info/WHEEL +4 -0
- fanbase-0.1.0.dist-info/entry_points.txt +2 -0
fanbase/__init__.py
ADDED
fanbase/analyze.py
ADDED
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
"""Summarise and correlate measurement records.
|
|
2
|
+
|
|
3
|
+
The pilot's question is whether the metrics we defined actually separate — whether they
|
|
4
|
+
cluster along the categories of DESIGN.md §6.3, or collapse into one axis. That is a
|
|
5
|
+
correlation question, so this module computes Spearman rank correlation (rank, then
|
|
6
|
+
Pearson) rather than assuming anything about metric distributions.
|
|
7
|
+
|
|
8
|
+
Spearman is hand-rolled to keep the package dependency-light; the corpus is small enough
|
|
9
|
+
that an O(n log n) rank and an O(n) covariance are free.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
import math
|
|
16
|
+
import statistics
|
|
17
|
+
from collections import defaultdict
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
from typing import Any, Iterable, Sequence
|
|
21
|
+
|
|
22
|
+
# Fields that identify or describe a record rather than measure it.
|
|
23
|
+
NON_METRIC_KEYS = {
|
|
24
|
+
"revision", "format", "profile", "lineage", "objective", "sha256", "path", "ok",
|
|
25
|
+
"error", "detail", "model", "effort", "replicate", "genesis", "role_hint", "arm",
|
|
26
|
+
"mode", "spec", "node_types", "fandango_version", "python_version",
|
|
27
|
+
"samples_requested", "kpaths_error",
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def load(*paths: Path) -> list[dict[str, Any]]:
|
|
32
|
+
"""Read JSONL result files, merging records that share a revision."""
|
|
33
|
+
merged: dict[str, dict[str, Any]] = {}
|
|
34
|
+
for path in paths:
|
|
35
|
+
for line in Path(path).read_text().splitlines():
|
|
36
|
+
if not line.strip():
|
|
37
|
+
continue
|
|
38
|
+
record = json.loads(line)
|
|
39
|
+
key = record.get("revision") or record.get("spec") or record.get("path", "")
|
|
40
|
+
merged.setdefault(key, {}).update(record)
|
|
41
|
+
return list(merged.values())
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def metric_names(records: Sequence[dict[str, Any]]) -> list[str]:
|
|
45
|
+
names: set[str] = set()
|
|
46
|
+
for record in records:
|
|
47
|
+
for key, value in record.items():
|
|
48
|
+
if key not in NON_METRIC_KEYS and isinstance(value, (int, float)) and not isinstance(value, bool):
|
|
49
|
+
names.add(key)
|
|
50
|
+
return sorted(names)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _ranks(values: Sequence[float]) -> list[float]:
|
|
54
|
+
"""Fractional ranks, ties averaged — required for Spearman to be well defined."""
|
|
55
|
+
order = sorted(range(len(values)), key=lambda i: values[i])
|
|
56
|
+
ranks = [0.0] * len(values)
|
|
57
|
+
i = 0
|
|
58
|
+
while i < len(order):
|
|
59
|
+
j = i
|
|
60
|
+
while j + 1 < len(order) and values[order[j + 1]] == values[order[i]]:
|
|
61
|
+
j += 1
|
|
62
|
+
shared = (i + j) / 2 + 1
|
|
63
|
+
for k in range(i, j + 1):
|
|
64
|
+
ranks[order[k]] = shared
|
|
65
|
+
i = j + 1
|
|
66
|
+
return ranks
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def spearman(xs: Sequence[float], ys: Sequence[float]) -> float | None:
|
|
70
|
+
if len(xs) < 3:
|
|
71
|
+
return None
|
|
72
|
+
rx, ry = _ranks(xs), _ranks(ys)
|
|
73
|
+
mx, my = statistics.fmean(rx), statistics.fmean(ry)
|
|
74
|
+
num = sum((a - mx) * (b - my) for a, b in zip(rx, ry))
|
|
75
|
+
den = math.sqrt(sum((a - mx) ** 2 for a in rx) * sum((b - my) ** 2 for b in ry))
|
|
76
|
+
return round(num / den, 3) if den else None
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def paired(records: Iterable[dict[str, Any]], a: str, b: str) -> tuple[list[float], list[float]]:
|
|
80
|
+
xs, ys = [], []
|
|
81
|
+
for record in records:
|
|
82
|
+
va, vb = record.get(a), record.get(b)
|
|
83
|
+
if isinstance(va, (int, float)) and isinstance(vb, (int, float)):
|
|
84
|
+
xs.append(float(va))
|
|
85
|
+
ys.append(float(vb))
|
|
86
|
+
return xs, ys
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
@dataclass
|
|
90
|
+
class Summary:
|
|
91
|
+
metric: str
|
|
92
|
+
n: int
|
|
93
|
+
mean: float
|
|
94
|
+
median: float
|
|
95
|
+
minimum: float
|
|
96
|
+
maximum: float
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def summarise(records: Sequence[dict[str, Any]], metrics: Sequence[str] | None = None) -> list[Summary]:
|
|
100
|
+
out = []
|
|
101
|
+
for name in metrics or metric_names(records):
|
|
102
|
+
values = [float(r[name]) for r in records
|
|
103
|
+
if isinstance(r.get(name), (int, float)) and not isinstance(r.get(name), bool)]
|
|
104
|
+
if values:
|
|
105
|
+
out.append(Summary(name, len(values), round(statistics.fmean(values), 3),
|
|
106
|
+
round(statistics.median(values), 3), min(values), max(values)))
|
|
107
|
+
return out
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def group_by(records: Sequence[dict[str, Any]], key: str) -> dict[Any, list[dict[str, Any]]]:
|
|
111
|
+
groups: dict[Any, list[dict[str, Any]]] = defaultdict(list)
|
|
112
|
+
for record in records:
|
|
113
|
+
groups[record.get(key)].append(record)
|
|
114
|
+
return dict(groups)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def correlation_matrix(
|
|
118
|
+
records: Sequence[dict[str, Any]], metrics: Sequence[str], min_pairs: int = 5
|
|
119
|
+
) -> dict[tuple[str, str], float | None]:
|
|
120
|
+
matrix: dict[tuple[str, str], float | None] = {}
|
|
121
|
+
for i, a in enumerate(metrics):
|
|
122
|
+
for b in metrics[i + 1:]:
|
|
123
|
+
xs, ys = paired(records, a, b)
|
|
124
|
+
matrix[(a, b)] = spearman(xs, ys) if len(xs) >= min_pairs else None
|
|
125
|
+
return matrix
|
fanbase/cli.py
ADDED
|
@@ -0,0 +1,264 @@
|
|
|
1
|
+
"""The `fanbase` command.
|
|
2
|
+
|
|
3
|
+
Everything here reads a registry checkout and nothing else — no upstream corpus, no
|
|
4
|
+
network, and no Fandango import unless a subcommand actually measures something.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import argparse
|
|
10
|
+
import os
|
|
11
|
+
import shlex
|
|
12
|
+
import shutil
|
|
13
|
+
import sys
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
|
|
16
|
+
from fanbase.registry import Registry, RegistryError, Revision
|
|
17
|
+
from fanbase.runner import METRICS, RunConfig, run
|
|
18
|
+
from fanbase.analyze import correlation_matrix, group_by, load, metric_names, summarise
|
|
19
|
+
from fanbase.importers.corpus import import_corpus
|
|
20
|
+
from fanbase.validate import validate
|
|
21
|
+
|
|
22
|
+
ENV_REGISTRY = "FANBASE_REGISTRY"
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def find_registry(explicit: str | None) -> Registry:
|
|
26
|
+
"""Locate a registry: --registry, then $FANBASE_REGISTRY, then upwards from cwd."""
|
|
27
|
+
candidates: list[Path] = []
|
|
28
|
+
if explicit:
|
|
29
|
+
candidates.append(Path(explicit))
|
|
30
|
+
if os.environ.get(ENV_REGISTRY):
|
|
31
|
+
candidates.append(Path(os.environ[ENV_REGISTRY]))
|
|
32
|
+
here = Path.cwd().resolve()
|
|
33
|
+
candidates.extend([here, *here.parents])
|
|
34
|
+
|
|
35
|
+
for candidate in candidates:
|
|
36
|
+
if (candidate / "specs").is_dir():
|
|
37
|
+
return Registry(candidate)
|
|
38
|
+
raise RegistryError(
|
|
39
|
+
"no registry found. Pass --registry PATH, set $FANBASE_REGISTRY, "
|
|
40
|
+
"or run inside a checkout containing specs/"
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def install_root() -> Path:
|
|
45
|
+
"""Where Fandango's `include()` will find installed specs.
|
|
46
|
+
|
|
47
|
+
Mirrors Fandango's own search order, so an installed spec is importable with no
|
|
48
|
+
further configuration.
|
|
49
|
+
"""
|
|
50
|
+
if fandango_path := os.environ.get("FANDANGO_PATH"):
|
|
51
|
+
first = fandango_path.split(os.pathsep)[0]
|
|
52
|
+
if first:
|
|
53
|
+
return Path(first)
|
|
54
|
+
if xdg := os.environ.get("XDG_DATA_HOME"):
|
|
55
|
+
return Path(xdg) / "fandango"
|
|
56
|
+
if sys.platform == "darwin":
|
|
57
|
+
return Path.home() / "Library" / "Fandango"
|
|
58
|
+
return Path.home() / ".local" / "share" / "fandango"
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def describe(rev: Revision) -> list[tuple[str, str]]:
|
|
62
|
+
header = rev.spec.header
|
|
63
|
+
rows = [
|
|
64
|
+
("revision", str(rev.coord)),
|
|
65
|
+
("objective", rev.objective),
|
|
66
|
+
("sha256", rev.spec.sha256[:16]),
|
|
67
|
+
("size", f"{rev.spec.size:,} bytes / {rev.spec.lines} lines"),
|
|
68
|
+
]
|
|
69
|
+
for field in ("MODEL", "EFFORT", "REPLICATE", "GENESIS", "PROFILE", "ROLE_HINT", "SOURCE"):
|
|
70
|
+
if field in header:
|
|
71
|
+
rows.append((field.lower(), str(header[field])))
|
|
72
|
+
return rows
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def cmd_list(args, reg: Registry) -> int:
|
|
76
|
+
formats = reg.formats()
|
|
77
|
+
for fmt in formats:
|
|
78
|
+
profiles = reg.profiles(fmt)
|
|
79
|
+
n = sum(len(reg.lineages(fmt, p)) for p in profiles)
|
|
80
|
+
print(f" {fmt:<16} {len(profiles)} profile(s) {n:>4} lineage(s)")
|
|
81
|
+
print(f"\n{len(formats)} formats")
|
|
82
|
+
return 0
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def cmd_search(args, reg: Registry) -> int:
|
|
86
|
+
needle = args.query.lower()
|
|
87
|
+
hits = 0
|
|
88
|
+
for fmt in reg.formats():
|
|
89
|
+
if needle in fmt.lower():
|
|
90
|
+
print(f" {fmt}")
|
|
91
|
+
hits += 1
|
|
92
|
+
continue
|
|
93
|
+
for profile in reg.profiles(fmt):
|
|
94
|
+
for lineage in reg.lineages(fmt, profile):
|
|
95
|
+
if needle in lineage.lower():
|
|
96
|
+
print(f" {fmt}/{profile}/{lineage}")
|
|
97
|
+
hits += 1
|
|
98
|
+
print(f"\n{hits} match(es) for {args.query!r}")
|
|
99
|
+
return 0 if hits else 1
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def cmd_show(args, reg: Registry) -> int:
|
|
103
|
+
rev = reg.resolve(args.ref)
|
|
104
|
+
width = max(len(k) for k, _ in describe(rev))
|
|
105
|
+
for key, value in describe(rev):
|
|
106
|
+
print(f" {key:<{width}} {value}")
|
|
107
|
+
return 0
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def cmd_install(args, reg: Registry) -> int:
|
|
111
|
+
rev = reg.resolve(args.ref)
|
|
112
|
+
root = Path(args.into) if args.into else install_root()
|
|
113
|
+
target = root / rev.coord.format / f"{rev.coord.lineage}.fan"
|
|
114
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
115
|
+
shutil.copy2(rev.path, target)
|
|
116
|
+
print(f"installed {rev.coord} -> {target}")
|
|
117
|
+
print(f' include("{rev.coord.format}/{rev.coord.lineage}.fan")')
|
|
118
|
+
return 0
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def cmd_measure(args, reg: Registry) -> int:
|
|
122
|
+
revisions = list(reg.revisions(fmt=args.format))
|
|
123
|
+
if args.objective:
|
|
124
|
+
revisions = [r for r in revisions if r.objective == args.objective]
|
|
125
|
+
if not revisions:
|
|
126
|
+
print("no revisions matched", file=sys.stderr)
|
|
127
|
+
return 1
|
|
128
|
+
config = RunConfig(
|
|
129
|
+
metric=args.metric,
|
|
130
|
+
python=args.python or sys.executable,
|
|
131
|
+
timeout=args.timeout,
|
|
132
|
+
jobs=args.jobs,
|
|
133
|
+
extra=tuple(shlex.split(args.extra or "")),
|
|
134
|
+
)
|
|
135
|
+
ok, failed = run(reg, config, Path(args.out), revisions=revisions)
|
|
136
|
+
print(f"\n{ok} measured, {failed} failed -> {args.out}")
|
|
137
|
+
return 0
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def cmd_analyze(args, reg: Registry) -> int:
|
|
141
|
+
records = load(*[Path(p) for p in args.results])
|
|
142
|
+
ok = [r for r in records if r.get("ok")]
|
|
143
|
+
print(f"records {len(records)} ok {len(ok)} failed {len(records) - len(ok)}")
|
|
144
|
+
versions = {r.get("fandango_version") for r in ok if r.get("fandango_version")}
|
|
145
|
+
if versions:
|
|
146
|
+
print(f"fandango {', '.join(sorted(versions))}")
|
|
147
|
+
if not ok:
|
|
148
|
+
return 1
|
|
149
|
+
|
|
150
|
+
metrics = [m for m in (args.metrics or metric_names(ok))]
|
|
151
|
+
print("\nmetric summary")
|
|
152
|
+
print(f" {'metric':<26} {'n':>5} {'mean':>10} {'median':>10} {'min':>10} {'max':>10}")
|
|
153
|
+
for s in summarise(ok, metrics):
|
|
154
|
+
print(f" {s.metric:<26} {s.n:>5} {s.mean:>10} {s.median:>10} {s.minimum:>10} {s.maximum:>10}")
|
|
155
|
+
|
|
156
|
+
for key in args.by or []:
|
|
157
|
+
groups = {k: v for k, v in group_by(ok, key).items() if k is not None}
|
|
158
|
+
if len(groups) < 2:
|
|
159
|
+
continue
|
|
160
|
+
print(f"\nby {key}")
|
|
161
|
+
shown = [m for m in metrics if any(m in r for r in ok)][: args.columns]
|
|
162
|
+
print(f" {key:<28} {'n':>4} " + " ".join(f"{m[:14]:>15}" for m in shown))
|
|
163
|
+
for name, rows in sorted(groups.items(), key=lambda kv: str(kv[0])):
|
|
164
|
+
cells = []
|
|
165
|
+
for m in shown:
|
|
166
|
+
s = summarise(rows, [m])
|
|
167
|
+
cells.append(f"{s[0].mean:>15}" if s else f"{'-':>15}")
|
|
168
|
+
print(f" {str(name):<28} {len(rows):>4} " + " ".join(cells))
|
|
169
|
+
|
|
170
|
+
if args.correlate:
|
|
171
|
+
matrix = correlation_matrix(ok, metrics)
|
|
172
|
+
ranked = sorted(
|
|
173
|
+
((v, a, b) for (a, b), v in matrix.items() if v is not None),
|
|
174
|
+
key=lambda t: -abs(t[0]),
|
|
175
|
+
)
|
|
176
|
+
print(f"\nstrongest rank correlations (Spearman, {len(ranked)} pairs)")
|
|
177
|
+
for value, a, b in ranked[: args.top]:
|
|
178
|
+
print(f" {value:>7} {a} ~ {b}")
|
|
179
|
+
return 0
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def cmd_import(args, reg: Registry) -> int:
|
|
183
|
+
written, unchanged, formats = import_corpus(Path(args.corpus), reg.root, args.dry_run)
|
|
184
|
+
verb = "would write" if args.dry_run else "wrote"
|
|
185
|
+
print(f"{verb} {written} revisions, {unchanged} unchanged, across {formats} formats")
|
|
186
|
+
return 0
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def cmd_validate(args, reg: Registry) -> int:
|
|
190
|
+
problems = list(validate(reg))
|
|
191
|
+
errors = [p for p in problems if p.level == "error"]
|
|
192
|
+
shown = problems if args.all else problems[: args.limit]
|
|
193
|
+
for problem in shown:
|
|
194
|
+
print(f" {problem}")
|
|
195
|
+
if len(problems) > len(shown):
|
|
196
|
+
print(f" ... and {len(problems) - len(shown)} more (use --all)")
|
|
197
|
+
print(f"\n{len(errors)} error(s), {len(problems) - len(errors)} warning(s)")
|
|
198
|
+
return 1 if errors else 0
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
202
|
+
ap = argparse.ArgumentParser(prog="fanbase", description=__doc__)
|
|
203
|
+
ap.add_argument("--registry", help="path to a registry checkout")
|
|
204
|
+
sub = ap.add_subparsers(dest="command", required=True)
|
|
205
|
+
|
|
206
|
+
sub.add_parser("list", help="list formats in the registry").set_defaults(fn=cmd_list)
|
|
207
|
+
|
|
208
|
+
p = sub.add_parser("search", help="find formats and lineages")
|
|
209
|
+
p.add_argument("query")
|
|
210
|
+
p.set_defaults(fn=cmd_search)
|
|
211
|
+
|
|
212
|
+
p = sub.add_parser("show", help="show one revision")
|
|
213
|
+
p.add_argument("ref", help="e.g. png/base/normal-claude-sonnet-5-high-r1@v1")
|
|
214
|
+
p.set_defaults(fn=cmd_show)
|
|
215
|
+
|
|
216
|
+
p = sub.add_parser("install", help="copy a revision into Fandango's include path")
|
|
217
|
+
p.add_argument("ref")
|
|
218
|
+
p.add_argument("--into", help="install directory (default: Fandango's data dir)")
|
|
219
|
+
p.set_defaults(fn=cmd_install)
|
|
220
|
+
|
|
221
|
+
p = sub.add_parser("analyze", help="summarise and correlate measurement records")
|
|
222
|
+
p.add_argument("results", nargs="+", help="JSONL files from `fanbase measure`")
|
|
223
|
+
p.add_argument("--metrics", nargs="*", help="restrict to these metrics")
|
|
224
|
+
p.add_argument("--by", nargs="*", default=["objective", "model", "effort"],
|
|
225
|
+
help="group comparisons on these fields")
|
|
226
|
+
p.add_argument("--correlate", action="store_true", help="rank-correlate every metric pair")
|
|
227
|
+
p.add_argument("--top", type=int, default=15)
|
|
228
|
+
p.add_argument("--columns", type=int, default=4)
|
|
229
|
+
p.set_defaults(fn=cmd_analyze)
|
|
230
|
+
|
|
231
|
+
p = sub.add_parser("import", help="import an upstream corpus into the registry")
|
|
232
|
+
p.add_argument("corpus")
|
|
233
|
+
p.add_argument("--dry-run", action="store_true")
|
|
234
|
+
p.set_defaults(fn=cmd_import)
|
|
235
|
+
|
|
236
|
+
p = sub.add_parser("validate", help="check the registry against its schemas")
|
|
237
|
+
p.add_argument("--limit", type=int, default=25)
|
|
238
|
+
p.add_argument("--all", action="store_true")
|
|
239
|
+
p.set_defaults(fn=cmd_validate)
|
|
240
|
+
|
|
241
|
+
p = sub.add_parser("measure", help="run a metric over registry revisions")
|
|
242
|
+
p.add_argument("--metric", choices=sorted(METRICS), default="intrinsics")
|
|
243
|
+
p.add_argument("--format", help="restrict to one format")
|
|
244
|
+
p.add_argument("--objective", help="restrict to one objective")
|
|
245
|
+
p.add_argument("--python", help="interpreter with Fandango (default: this one)")
|
|
246
|
+
p.add_argument("--timeout", type=int, default=120)
|
|
247
|
+
p.add_argument("--jobs", type=int, default=1)
|
|
248
|
+
p.add_argument("--out", default="results/metrics.jsonl")
|
|
249
|
+
p.add_argument("--extra", help='extra worker arguments, one string: --extra "--samples 8"')
|
|
250
|
+
p.set_defaults(fn=cmd_measure)
|
|
251
|
+
return ap
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def main(argv: list[str] | None = None) -> int:
|
|
255
|
+
args = build_parser().parse_args(argv)
|
|
256
|
+
try:
|
|
257
|
+
return args.fn(args, find_registry(args.registry))
|
|
258
|
+
except RegistryError as exc:
|
|
259
|
+
print(f"fanbase: {exc}", file=sys.stderr)
|
|
260
|
+
return 2
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
if __name__ == "__main__":
|
|
264
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
"""Adapters that turn an external, ad-hoc corpus into registry entries.
|
|
2
|
+
|
|
3
|
+
Everything source-specific lives here. The rest of the package knows only the
|
|
4
|
+
registry layout (`fanbase.registry`), so no core code, CLI command, metric, or site
|
|
5
|
+
build ever depends on how some upstream repository happened to name its files.
|
|
6
|
+
"""
|
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Import an upstream corpus into the registry (DESIGN.md §4).
|
|
3
|
+
|
|
4
|
+
Re-runnable and idempotent: every imported revision records the SHA-256 of the file it
|
|
5
|
+
came from, so re-importing an updated corpus rewrites what changed, adds what is new,
|
|
6
|
+
and leaves everything else untouched.
|
|
7
|
+
|
|
8
|
+
Upstream specs carry no provenance header, so one is generated from the source's
|
|
9
|
+
filename and prepended — that is what makes a registry spec self-describing when someone
|
|
10
|
+
copies the single file into their own project.
|
|
11
|
+
|
|
12
|
+
Canonical specs are deliberately *not* created: blessing one is a maintainer decision
|
|
13
|
+
informed by measurements, never an import-time guess.
|
|
14
|
+
|
|
15
|
+
Exposed as `fanbase import CORPUS`.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import hashlib
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
|
|
23
|
+
from fanbase.importers.ossfuzz_eval import describe, load_corpus
|
|
24
|
+
from fanbase.registry import DEFAULT_OBJECTIVE, DEFAULT_PROFILE
|
|
25
|
+
|
|
26
|
+
GENERATED_BANNER = "# --- generated by fanbase import; edit the registry, not this block ---"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _yaml(mapping: dict[str, object]) -> str:
|
|
30
|
+
"""Minimal YAML writer — avoids a dependency for what is only flat key/value data."""
|
|
31
|
+
lines = []
|
|
32
|
+
for key, value in mapping.items():
|
|
33
|
+
if value is None:
|
|
34
|
+
lines.append(f"{key}:")
|
|
35
|
+
elif isinstance(value, bool):
|
|
36
|
+
lines.append(f"{key}: {'true' if value else 'false'}")
|
|
37
|
+
elif isinstance(value, (int, float)):
|
|
38
|
+
lines.append(f"{key}: {value}")
|
|
39
|
+
elif isinstance(value, list):
|
|
40
|
+
# An empty list must stay a list: `key:` alone parses back as null.
|
|
41
|
+
if not value:
|
|
42
|
+
lines.append(f"{key}: []")
|
|
43
|
+
else:
|
|
44
|
+
lines.append(f"{key}:")
|
|
45
|
+
lines.extend(f" - {item}" for item in value)
|
|
46
|
+
else:
|
|
47
|
+
text = str(value).replace('"', '\\"')
|
|
48
|
+
lines.append(f'{key}: "{text}"')
|
|
49
|
+
return "\n".join(lines) + "\n"
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def build_header(name, source_sha: str, source_rel: str) -> str:
|
|
53
|
+
objective = name.objective or DEFAULT_OBJECTIVE
|
|
54
|
+
fields = {
|
|
55
|
+
"INFO": f"Fandango specification for {name.format}",
|
|
56
|
+
"PROFILE": f"{name.format}/{DEFAULT_PROFILE}",
|
|
57
|
+
"OBJECTIVE": objective,
|
|
58
|
+
"GENESIS": "ai",
|
|
59
|
+
"MODEL": name.model or "unknown",
|
|
60
|
+
"EFFORT": name.effort or "unknown",
|
|
61
|
+
"REPLICATE": name.replicate,
|
|
62
|
+
"SOURCE": source_rel,
|
|
63
|
+
"SOURCE_SHA256": source_sha,
|
|
64
|
+
}
|
|
65
|
+
if name.role_hint:
|
|
66
|
+
fields["ROLE_HINT"] = name.role_hint
|
|
67
|
+
lines = [GENERATED_BANNER]
|
|
68
|
+
for key, value in fields.items():
|
|
69
|
+
lines.append(f"{key} = {value!r}" if isinstance(value, int) else f'{key} = "{value}"')
|
|
70
|
+
lines.append(GENERATED_BANNER)
|
|
71
|
+
return "\n".join(lines) + "\n\n"
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def strip_generated_header(text: str) -> str:
|
|
75
|
+
"""Remove a previously generated block so re-import compares like with like."""
|
|
76
|
+
if not text.startswith(GENERATED_BANNER):
|
|
77
|
+
return text
|
|
78
|
+
end = text.find(GENERATED_BANNER, len(GENERATED_BANNER))
|
|
79
|
+
return text[end + len(GENERATED_BANNER):].lstrip("\n") if end != -1 else text
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def write_if_changed(path: Path, content: str, dry_run: bool) -> bool:
|
|
83
|
+
if path.exists() and path.read_text(encoding="utf-8") == content:
|
|
84
|
+
return False
|
|
85
|
+
if not dry_run:
|
|
86
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
87
|
+
path.write_text(content, encoding="utf-8")
|
|
88
|
+
return True
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def import_corpus(corpus: Path, root: Path, dry_run: bool = False) -> tuple[int, int, int]:
|
|
92
|
+
"""Import `corpus` into the registry at `root`. Returns (written, unchanged, formats)."""
|
|
93
|
+
specs = load_corpus(Path(corpus).resolve())
|
|
94
|
+
root = Path(root).resolve()
|
|
95
|
+
args = type("Args", (), {"dry_run": dry_run})()
|
|
96
|
+
written = skipped = 0
|
|
97
|
+
formats: dict[str, set[str]] = {}
|
|
98
|
+
|
|
99
|
+
for spec in specs:
|
|
100
|
+
name = describe(spec)
|
|
101
|
+
lineage = name.lineage if name.recognised else f"legacy-{spec.path.stem}"
|
|
102
|
+
formats.setdefault(name.format, set()).add(lineage)
|
|
103
|
+
|
|
104
|
+
body = strip_generated_header(spec.path.read_text(encoding="utf-8", errors="replace"))
|
|
105
|
+
source_sha = hashlib.sha256(body.encode("utf-8")).hexdigest()
|
|
106
|
+
content = build_header(name, source_sha, spec.rel.as_posix()) + body
|
|
107
|
+
|
|
108
|
+
base = root / "specs" / name.format / "profiles" / DEFAULT_PROFILE / "lineages" / lineage
|
|
109
|
+
changed = write_if_changed(base / "v1.fan", content, args.dry_run)
|
|
110
|
+
write_if_changed(base / "lineage.yaml", _yaml({
|
|
111
|
+
"id": lineage,
|
|
112
|
+
"objective": name.objective or DEFAULT_OBJECTIVE,
|
|
113
|
+
"role_hint": name.role_hint,
|
|
114
|
+
"model": name.model,
|
|
115
|
+
"effort": name.effort,
|
|
116
|
+
"replicate": name.replicate,
|
|
117
|
+
"arm": name.arm,
|
|
118
|
+
"owner": None,
|
|
119
|
+
"concept_doi": None,
|
|
120
|
+
}), args.dry_run)
|
|
121
|
+
written += changed
|
|
122
|
+
skipped += not changed
|
|
123
|
+
|
|
124
|
+
for fmt, lineages in sorted(formats.items()):
|
|
125
|
+
base = root / "specs" / fmt
|
|
126
|
+
write_if_changed(base / "format.yaml", _yaml({
|
|
127
|
+
"id": fmt,
|
|
128
|
+
"name": fmt.upper(),
|
|
129
|
+
"category": None,
|
|
130
|
+
"references": [],
|
|
131
|
+
"oracle_tier": "T0",
|
|
132
|
+
"maintainers": [],
|
|
133
|
+
}), args.dry_run)
|
|
134
|
+
write_if_changed(base / "profiles.yaml", _yaml({
|
|
135
|
+
"default": DEFAULT_PROFILE,
|
|
136
|
+
"profiles": [DEFAULT_PROFILE],
|
|
137
|
+
}), args.dry_run)
|
|
138
|
+
write_if_changed(base / "profiles" / DEFAULT_PROFILE / "profile.yaml", _yaml({
|
|
139
|
+
"id": DEFAULT_PROFILE,
|
|
140
|
+
"note": "placeholder: upstream declares no format version",
|
|
141
|
+
"extends": None,
|
|
142
|
+
"lineages": len(lineages),
|
|
143
|
+
}), args.dry_run)
|
|
144
|
+
|
|
145
|
+
return written, skipped, len(formats)
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
"""Import adapter for the `oss-fuzz-fandango-eval` corpus.
|
|
2
|
+
|
|
3
|
+
That repository is a factorial generation experiment, not a registry: every spec is one
|
|
4
|
+
cell of `format x mode x model x effort x replicate`, encoded in the filename as
|
|
5
|
+
|
|
6
|
+
<format>_<mode>_<model>_<effort>_<replicate>.fan
|
|
7
|
+
png_spicy_claude-opus-4-8_high_2.fan
|
|
8
|
+
|
|
9
|
+
with two arms — `library-broad/` (many formats, shallow) and `library-deep/` (few
|
|
10
|
+
formats, the full grid). This module is the only place in the package that knows any of
|
|
11
|
+
that; everything else consumes `fanbase.registry`.
|
|
12
|
+
|
|
13
|
+
Two mappings are not one-to-one and are deliberate:
|
|
14
|
+
|
|
15
|
+
* `mode` conflates two of our dimensions. `normal` and `spicy` are *objectives* (breadth
|
|
16
|
+
vs boundary), while `modify` describes the *role* the spec is written for (MUTATE) and
|
|
17
|
+
says nothing about what it aims at. Both are emitted separately.
|
|
18
|
+
* A replicate is **not** a revision. Three independent generations under identical
|
|
19
|
+
settings are siblings, not successive improvements, so each becomes its own lineage at
|
|
20
|
+
`v1` rather than `v1..v3` of one lineage. Revisions stay reserved for real patches.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import re
|
|
26
|
+
from dataclasses import dataclass
|
|
27
|
+
from pathlib import Path
|
|
28
|
+
|
|
29
|
+
from fanbase.spec import SpecFile, discover, group_by_content
|
|
30
|
+
|
|
31
|
+
# normal/spicy describe what the spec aims at; modify describes how it is meant to be used.
|
|
32
|
+
OBJECTIVE_BY_MODE = {"normal": "breadth", "spicy": "boundary"}
|
|
33
|
+
ROLE_BY_MODE = {"modify": "mutate"}
|
|
34
|
+
|
|
35
|
+
EFFORTS = ("low", "medium", "high", "minimal", "none")
|
|
36
|
+
|
|
37
|
+
_NAME = re.compile(
|
|
38
|
+
r"^(?P<format>[^_]+)_(?P<mode>[^_]+)_(?P<model>.+?)_"
|
|
39
|
+
r"(?P<effort>" + "|".join(EFFORTS) + r")_(?P<replicate>\d+)$"
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@dataclass
|
|
44
|
+
class ParsedName:
|
|
45
|
+
stem: str
|
|
46
|
+
format: str
|
|
47
|
+
mode: str | None = None
|
|
48
|
+
model: str | None = None
|
|
49
|
+
effort: str | None = None
|
|
50
|
+
replicate: int | None = None
|
|
51
|
+
arm: str | None = None # broad | deep
|
|
52
|
+
recognised: bool = False
|
|
53
|
+
|
|
54
|
+
@property
|
|
55
|
+
def objective(self) -> str | None:
|
|
56
|
+
return OBJECTIVE_BY_MODE.get(self.mode or "")
|
|
57
|
+
|
|
58
|
+
@property
|
|
59
|
+
def role_hint(self) -> str | None:
|
|
60
|
+
return ROLE_BY_MODE.get(self.mode or "")
|
|
61
|
+
|
|
62
|
+
@property
|
|
63
|
+
def lineage(self) -> str:
|
|
64
|
+
"""Experimental cell, replicate included — siblings are separate lineages."""
|
|
65
|
+
if not self.recognised:
|
|
66
|
+
return "legacy"
|
|
67
|
+
return f"{self.mode}-{self.model}-{self.effort}-r{self.replicate}"
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def parse_spec_name(stem: str, arm: str | None = None) -> ParsedName:
|
|
71
|
+
m = _NAME.match(stem)
|
|
72
|
+
if not m:
|
|
73
|
+
return ParsedName(stem=stem, format=stem.split("_")[0].split("-")[0], arm=arm)
|
|
74
|
+
return ParsedName(
|
|
75
|
+
stem=stem,
|
|
76
|
+
format=m.group("format"),
|
|
77
|
+
mode=m.group("mode"),
|
|
78
|
+
model=m.group("model"),
|
|
79
|
+
effort=m.group("effort"),
|
|
80
|
+
replicate=int(m.group("replicate")),
|
|
81
|
+
arm=arm,
|
|
82
|
+
recognised=True,
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def arm_of(spec: SpecFile) -> str | None:
|
|
87
|
+
head = spec.rel.parts[0] if spec.rel.parts else ""
|
|
88
|
+
return head.removeprefix("library-") if head.startswith("library-") else None
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def format_dir(spec: SpecFile) -> str:
|
|
92
|
+
"""The format directory, which sits under the arm: `library-broad/png/library/...`."""
|
|
93
|
+
parts = spec.rel.parts
|
|
94
|
+
return parts[1] if len(parts) > 2 and parts[0].startswith("library-") else parts[0]
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def describe(spec: SpecFile) -> ParsedName:
|
|
98
|
+
return parse_spec_name(spec.path.stem, arm_of(spec))
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def distinct(specs: list[SpecFile]) -> list[SpecFile]:
|
|
102
|
+
"""One representative per content hash.
|
|
103
|
+
|
|
104
|
+
Identical content across two experimental cells is itself a finding — two models
|
|
105
|
+
converging on the same grammar — so callers that care should use `group_by_content`.
|
|
106
|
+
"""
|
|
107
|
+
chosen = [group[0] for group in group_by_content(specs).values()]
|
|
108
|
+
return sorted(chosen, key=lambda s: s.rel.as_posix())
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def load_corpus(root: Path) -> list[SpecFile]:
|
|
112
|
+
return sorted(discover(Path(root)), key=lambda s: s.rel.as_posix())
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""Metric implementations, grouped by the apparatus they require (DESIGN.md §6.2).
|
|
2
|
+
|
|
3
|
+
S the spec alone intrinsics, self-consistency
|
|
4
|
+
+I a seed corpus recall, round-trip against real inputs, invariant battery
|
|
5
|
+
+R a reference layout structural fidelity
|
|
6
|
+
+O a validity oracle validity rate, specificity
|
|
7
|
+
+C instrumented program target coverage, feature reach
|
|
8
|
+
+F a program fleet + time finding yield
|
|
9
|
+
|
|
10
|
+
Anything below `+O` runs in CI on every PR; `+C` and `+F` need the lab machine and
|
|
11
|
+
repeated trials, and a single number from them is not a result.
|
|
12
|
+
"""
|