fanbase 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fanbase-0.1.0/PKG-INFO +30 -0
- fanbase-0.1.0/README.md +18 -0
- fanbase-0.1.0/pyproject.toml +22 -0
- fanbase-0.1.0/src/fanbase/__init__.py +3 -0
- fanbase-0.1.0/src/fanbase/analyze.py +125 -0
- fanbase-0.1.0/src/fanbase/cli.py +264 -0
- fanbase-0.1.0/src/fanbase/importers/__init__.py +6 -0
- fanbase-0.1.0/src/fanbase/importers/corpus.py +145 -0
- fanbase-0.1.0/src/fanbase/importers/ossfuzz_eval.py +112 -0
- fanbase-0.1.0/src/fanbase/metrics/__init__.py +12 -0
- fanbase-0.1.0/src/fanbase/metrics/intrinsics.py +220 -0
- fanbase-0.1.0/src/fanbase/metrics/selfcheck.py +155 -0
- fanbase-0.1.0/src/fanbase/registry.py +193 -0
- fanbase-0.1.0/src/fanbase/runner.py +107 -0
- fanbase-0.1.0/src/fanbase/schemas/__init__.py +0 -0
- fanbase-0.1.0/src/fanbase/schemas/format.schema.json +30 -0
- fanbase-0.1.0/src/fanbase/schemas/lineage.schema.json +27 -0
- fanbase-0.1.0/src/fanbase/schemas/profile.schema.json +20 -0
- fanbase-0.1.0/src/fanbase/schemas/profiles.schema.json +12 -0
- fanbase-0.1.0/src/fanbase/spec.py +151 -0
- fanbase-0.1.0/src/fanbase/validate.py +128 -0
fanbase-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: fanbase
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Registry, evaluation framework and benchmark for Fandango input specifications
|
|
5
|
+
License: Apache-2.0
|
|
6
|
+
Requires-Python: >=3.11
|
|
7
|
+
Requires-Dist: jsonschema>=4.20
|
|
8
|
+
Requires-Dist: pyyaml>=6
|
|
9
|
+
Provides-Extra: metrics
|
|
10
|
+
Requires-Dist: fandango-fuzzer>=1.1; extra == 'metrics'
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
|
|
13
|
+
# fanbase-cli
|
|
14
|
+
|
|
15
|
+
The `fanbase` command and the metric engine behind it.
|
|
16
|
+
|
|
17
|
+
```
|
|
18
|
+
fanbase list # formats in the registry
|
|
19
|
+
fanbase search png # find formats and lineages
|
|
20
|
+
fanbase show png/base/normal-...@v1 # provenance and metrics for one revision
|
|
21
|
+
fanbase install png # materialise into Fandango's include path
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
The registry itself lives in the [`fanbase`](../fanbase) repository; this package reads
|
|
25
|
+
it and never depends on any upstream corpus. Point it at a checkout with `--registry`,
|
|
26
|
+
or set `FANBASE_REGISTRY`.
|
|
27
|
+
|
|
28
|
+
Metrics require the `metrics` extra (`pip install -e '.[metrics]'`), which pulls in
|
|
29
|
+
Fandango. Loading a spec executes its Python, so every measurement runs in an isolated
|
|
30
|
+
subprocess under a timeout.
|
fanbase-0.1.0/README.md
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# fanbase-cli
|
|
2
|
+
|
|
3
|
+
The `fanbase` command and the metric engine behind it.
|
|
4
|
+
|
|
5
|
+
```
|
|
6
|
+
fanbase list # formats in the registry
|
|
7
|
+
fanbase search png # find formats and lineages
|
|
8
|
+
fanbase show png/base/normal-...@v1 # provenance and metrics for one revision
|
|
9
|
+
fanbase install png # materialise into Fandango's include path
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
The registry itself lives in the [`fanbase`](../fanbase) repository; this package reads
|
|
13
|
+
it and never depends on any upstream corpus. Point it at a checkout with `--registry`,
|
|
14
|
+
or set `FANBASE_REGISTRY`.
|
|
15
|
+
|
|
16
|
+
Metrics require the `metrics` extra (`pip install -e '.[metrics]'`), which pulls in
|
|
17
|
+
Fandango. Loading a spec executes its Python, so every measurement runs in an isolated
|
|
18
|
+
subprocess under a timeout.
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "fanbase"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Registry, evaluation framework and benchmark for Fandango input specifications"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.11"
|
|
11
|
+
license = { text = "Apache-2.0" }
|
|
12
|
+
dependencies = ["pyyaml>=6", "jsonschema>=4.20"]
|
|
13
|
+
|
|
14
|
+
[project.optional-dependencies]
|
|
15
|
+
# Metrics that load or run a spec need Fandango; registry, search and install do not.
|
|
16
|
+
metrics = ["fandango-fuzzer>=1.1"]
|
|
17
|
+
|
|
18
|
+
[project.scripts]
|
|
19
|
+
fanbase = "fanbase.cli:main"
|
|
20
|
+
|
|
21
|
+
[tool.hatch.build.targets.wheel]
|
|
22
|
+
packages = ["src/fanbase"]
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
"""Summarise and correlate measurement records.
|
|
2
|
+
|
|
3
|
+
The pilot's question is whether the metrics we defined actually separate — whether they
|
|
4
|
+
cluster along the categories of DESIGN.md §6.3, or collapse into one axis. That is a
|
|
5
|
+
correlation question, so this module computes Spearman rank correlation (rank, then
|
|
6
|
+
Pearson) rather than assuming anything about metric distributions.
|
|
7
|
+
|
|
8
|
+
Spearman is hand-rolled to keep the package dependency-light; the corpus is small enough
|
|
9
|
+
that an O(n log n) rank and an O(n) covariance are free.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
import math
|
|
16
|
+
import statistics
|
|
17
|
+
from collections import defaultdict
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
from typing import Any, Iterable, Sequence
|
|
21
|
+
|
|
22
|
+
# Fields that identify or describe a record rather than measure it.
|
|
23
|
+
NON_METRIC_KEYS = {
|
|
24
|
+
"revision", "format", "profile", "lineage", "objective", "sha256", "path", "ok",
|
|
25
|
+
"error", "detail", "model", "effort", "replicate", "genesis", "role_hint", "arm",
|
|
26
|
+
"mode", "spec", "node_types", "fandango_version", "python_version",
|
|
27
|
+
"samples_requested", "kpaths_error",
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def load(*paths: Path) -> list[dict[str, Any]]:
|
|
32
|
+
"""Read JSONL result files, merging records that share a revision."""
|
|
33
|
+
merged: dict[str, dict[str, Any]] = {}
|
|
34
|
+
for path in paths:
|
|
35
|
+
for line in Path(path).read_text().splitlines():
|
|
36
|
+
if not line.strip():
|
|
37
|
+
continue
|
|
38
|
+
record = json.loads(line)
|
|
39
|
+
key = record.get("revision") or record.get("spec") or record.get("path", "")
|
|
40
|
+
merged.setdefault(key, {}).update(record)
|
|
41
|
+
return list(merged.values())
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def metric_names(records: Sequence[dict[str, Any]]) -> list[str]:
|
|
45
|
+
names: set[str] = set()
|
|
46
|
+
for record in records:
|
|
47
|
+
for key, value in record.items():
|
|
48
|
+
if key not in NON_METRIC_KEYS and isinstance(value, (int, float)) and not isinstance(value, bool):
|
|
49
|
+
names.add(key)
|
|
50
|
+
return sorted(names)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _ranks(values: Sequence[float]) -> list[float]:
|
|
54
|
+
"""Fractional ranks, ties averaged — required for Spearman to be well defined."""
|
|
55
|
+
order = sorted(range(len(values)), key=lambda i: values[i])
|
|
56
|
+
ranks = [0.0] * len(values)
|
|
57
|
+
i = 0
|
|
58
|
+
while i < len(order):
|
|
59
|
+
j = i
|
|
60
|
+
while j + 1 < len(order) and values[order[j + 1]] == values[order[i]]:
|
|
61
|
+
j += 1
|
|
62
|
+
shared = (i + j) / 2 + 1
|
|
63
|
+
for k in range(i, j + 1):
|
|
64
|
+
ranks[order[k]] = shared
|
|
65
|
+
i = j + 1
|
|
66
|
+
return ranks
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def spearman(xs: Sequence[float], ys: Sequence[float]) -> float | None:
|
|
70
|
+
if len(xs) < 3:
|
|
71
|
+
return None
|
|
72
|
+
rx, ry = _ranks(xs), _ranks(ys)
|
|
73
|
+
mx, my = statistics.fmean(rx), statistics.fmean(ry)
|
|
74
|
+
num = sum((a - mx) * (b - my) for a, b in zip(rx, ry))
|
|
75
|
+
den = math.sqrt(sum((a - mx) ** 2 for a in rx) * sum((b - my) ** 2 for b in ry))
|
|
76
|
+
return round(num / den, 3) if den else None
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def paired(records: Iterable[dict[str, Any]], a: str, b: str) -> tuple[list[float], list[float]]:
|
|
80
|
+
xs, ys = [], []
|
|
81
|
+
for record in records:
|
|
82
|
+
va, vb = record.get(a), record.get(b)
|
|
83
|
+
if isinstance(va, (int, float)) and isinstance(vb, (int, float)):
|
|
84
|
+
xs.append(float(va))
|
|
85
|
+
ys.append(float(vb))
|
|
86
|
+
return xs, ys
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
@dataclass
|
|
90
|
+
class Summary:
|
|
91
|
+
metric: str
|
|
92
|
+
n: int
|
|
93
|
+
mean: float
|
|
94
|
+
median: float
|
|
95
|
+
minimum: float
|
|
96
|
+
maximum: float
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def summarise(records: Sequence[dict[str, Any]], metrics: Sequence[str] | None = None) -> list[Summary]:
|
|
100
|
+
out = []
|
|
101
|
+
for name in metrics or metric_names(records):
|
|
102
|
+
values = [float(r[name]) for r in records
|
|
103
|
+
if isinstance(r.get(name), (int, float)) and not isinstance(r.get(name), bool)]
|
|
104
|
+
if values:
|
|
105
|
+
out.append(Summary(name, len(values), round(statistics.fmean(values), 3),
|
|
106
|
+
round(statistics.median(values), 3), min(values), max(values)))
|
|
107
|
+
return out
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def group_by(records: Sequence[dict[str, Any]], key: str) -> dict[Any, list[dict[str, Any]]]:
|
|
111
|
+
groups: dict[Any, list[dict[str, Any]]] = defaultdict(list)
|
|
112
|
+
for record in records:
|
|
113
|
+
groups[record.get(key)].append(record)
|
|
114
|
+
return dict(groups)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def correlation_matrix(
|
|
118
|
+
records: Sequence[dict[str, Any]], metrics: Sequence[str], min_pairs: int = 5
|
|
119
|
+
) -> dict[tuple[str, str], float | None]:
|
|
120
|
+
matrix: dict[tuple[str, str], float | None] = {}
|
|
121
|
+
for i, a in enumerate(metrics):
|
|
122
|
+
for b in metrics[i + 1:]:
|
|
123
|
+
xs, ys = paired(records, a, b)
|
|
124
|
+
matrix[(a, b)] = spearman(xs, ys) if len(xs) >= min_pairs else None
|
|
125
|
+
return matrix
|
|
@@ -0,0 +1,264 @@
|
|
|
1
|
+
"""The `fanbase` command.
|
|
2
|
+
|
|
3
|
+
Everything here reads a registry checkout and nothing else — no upstream corpus, no
|
|
4
|
+
network, and no Fandango import unless a subcommand actually measures something.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import argparse
|
|
10
|
+
import os
|
|
11
|
+
import shlex
|
|
12
|
+
import shutil
|
|
13
|
+
import sys
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
|
|
16
|
+
from fanbase.registry import Registry, RegistryError, Revision
|
|
17
|
+
from fanbase.runner import METRICS, RunConfig, run
|
|
18
|
+
from fanbase.analyze import correlation_matrix, group_by, load, metric_names, summarise
|
|
19
|
+
from fanbase.importers.corpus import import_corpus
|
|
20
|
+
from fanbase.validate import validate
|
|
21
|
+
|
|
22
|
+
ENV_REGISTRY = "FANBASE_REGISTRY"
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def find_registry(explicit: str | None) -> Registry:
|
|
26
|
+
"""Locate a registry: --registry, then $FANBASE_REGISTRY, then upwards from cwd."""
|
|
27
|
+
candidates: list[Path] = []
|
|
28
|
+
if explicit:
|
|
29
|
+
candidates.append(Path(explicit))
|
|
30
|
+
if os.environ.get(ENV_REGISTRY):
|
|
31
|
+
candidates.append(Path(os.environ[ENV_REGISTRY]))
|
|
32
|
+
here = Path.cwd().resolve()
|
|
33
|
+
candidates.extend([here, *here.parents])
|
|
34
|
+
|
|
35
|
+
for candidate in candidates:
|
|
36
|
+
if (candidate / "specs").is_dir():
|
|
37
|
+
return Registry(candidate)
|
|
38
|
+
raise RegistryError(
|
|
39
|
+
"no registry found. Pass --registry PATH, set $FANBASE_REGISTRY, "
|
|
40
|
+
"or run inside a checkout containing specs/"
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def install_root() -> Path:
|
|
45
|
+
"""Where Fandango's `include()` will find installed specs.
|
|
46
|
+
|
|
47
|
+
Mirrors Fandango's own search order, so an installed spec is importable with no
|
|
48
|
+
further configuration.
|
|
49
|
+
"""
|
|
50
|
+
if fandango_path := os.environ.get("FANDANGO_PATH"):
|
|
51
|
+
first = fandango_path.split(os.pathsep)[0]
|
|
52
|
+
if first:
|
|
53
|
+
return Path(first)
|
|
54
|
+
if xdg := os.environ.get("XDG_DATA_HOME"):
|
|
55
|
+
return Path(xdg) / "fandango"
|
|
56
|
+
if sys.platform == "darwin":
|
|
57
|
+
return Path.home() / "Library" / "Fandango"
|
|
58
|
+
return Path.home() / ".local" / "share" / "fandango"
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def describe(rev: Revision) -> list[tuple[str, str]]:
|
|
62
|
+
header = rev.spec.header
|
|
63
|
+
rows = [
|
|
64
|
+
("revision", str(rev.coord)),
|
|
65
|
+
("objective", rev.objective),
|
|
66
|
+
("sha256", rev.spec.sha256[:16]),
|
|
67
|
+
("size", f"{rev.spec.size:,} bytes / {rev.spec.lines} lines"),
|
|
68
|
+
]
|
|
69
|
+
for field in ("MODEL", "EFFORT", "REPLICATE", "GENESIS", "PROFILE", "ROLE_HINT", "SOURCE"):
|
|
70
|
+
if field in header:
|
|
71
|
+
rows.append((field.lower(), str(header[field])))
|
|
72
|
+
return rows
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def cmd_list(args, reg: Registry) -> int:
|
|
76
|
+
formats = reg.formats()
|
|
77
|
+
for fmt in formats:
|
|
78
|
+
profiles = reg.profiles(fmt)
|
|
79
|
+
n = sum(len(reg.lineages(fmt, p)) for p in profiles)
|
|
80
|
+
print(f" {fmt:<16} {len(profiles)} profile(s) {n:>4} lineage(s)")
|
|
81
|
+
print(f"\n{len(formats)} formats")
|
|
82
|
+
return 0
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def cmd_search(args, reg: Registry) -> int:
|
|
86
|
+
needle = args.query.lower()
|
|
87
|
+
hits = 0
|
|
88
|
+
for fmt in reg.formats():
|
|
89
|
+
if needle in fmt.lower():
|
|
90
|
+
print(f" {fmt}")
|
|
91
|
+
hits += 1
|
|
92
|
+
continue
|
|
93
|
+
for profile in reg.profiles(fmt):
|
|
94
|
+
for lineage in reg.lineages(fmt, profile):
|
|
95
|
+
if needle in lineage.lower():
|
|
96
|
+
print(f" {fmt}/{profile}/{lineage}")
|
|
97
|
+
hits += 1
|
|
98
|
+
print(f"\n{hits} match(es) for {args.query!r}")
|
|
99
|
+
return 0 if hits else 1
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def cmd_show(args, reg: Registry) -> int:
|
|
103
|
+
rev = reg.resolve(args.ref)
|
|
104
|
+
width = max(len(k) for k, _ in describe(rev))
|
|
105
|
+
for key, value in describe(rev):
|
|
106
|
+
print(f" {key:<{width}} {value}")
|
|
107
|
+
return 0
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def cmd_install(args, reg: Registry) -> int:
|
|
111
|
+
rev = reg.resolve(args.ref)
|
|
112
|
+
root = Path(args.into) if args.into else install_root()
|
|
113
|
+
target = root / rev.coord.format / f"{rev.coord.lineage}.fan"
|
|
114
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
115
|
+
shutil.copy2(rev.path, target)
|
|
116
|
+
print(f"installed {rev.coord} -> {target}")
|
|
117
|
+
print(f' include("{rev.coord.format}/{rev.coord.lineage}.fan")')
|
|
118
|
+
return 0
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def cmd_measure(args, reg: Registry) -> int:
|
|
122
|
+
revisions = list(reg.revisions(fmt=args.format))
|
|
123
|
+
if args.objective:
|
|
124
|
+
revisions = [r for r in revisions if r.objective == args.objective]
|
|
125
|
+
if not revisions:
|
|
126
|
+
print("no revisions matched", file=sys.stderr)
|
|
127
|
+
return 1
|
|
128
|
+
config = RunConfig(
|
|
129
|
+
metric=args.metric,
|
|
130
|
+
python=args.python or sys.executable,
|
|
131
|
+
timeout=args.timeout,
|
|
132
|
+
jobs=args.jobs,
|
|
133
|
+
extra=tuple(shlex.split(args.extra or "")),
|
|
134
|
+
)
|
|
135
|
+
ok, failed = run(reg, config, Path(args.out), revisions=revisions)
|
|
136
|
+
print(f"\n{ok} measured, {failed} failed -> {args.out}")
|
|
137
|
+
return 0
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def cmd_analyze(args, reg: Registry) -> int:
|
|
141
|
+
records = load(*[Path(p) for p in args.results])
|
|
142
|
+
ok = [r for r in records if r.get("ok")]
|
|
143
|
+
print(f"records {len(records)} ok {len(ok)} failed {len(records) - len(ok)}")
|
|
144
|
+
versions = {r.get("fandango_version") for r in ok if r.get("fandango_version")}
|
|
145
|
+
if versions:
|
|
146
|
+
print(f"fandango {', '.join(sorted(versions))}")
|
|
147
|
+
if not ok:
|
|
148
|
+
return 1
|
|
149
|
+
|
|
150
|
+
metrics = [m for m in (args.metrics or metric_names(ok))]
|
|
151
|
+
print("\nmetric summary")
|
|
152
|
+
print(f" {'metric':<26} {'n':>5} {'mean':>10} {'median':>10} {'min':>10} {'max':>10}")
|
|
153
|
+
for s in summarise(ok, metrics):
|
|
154
|
+
print(f" {s.metric:<26} {s.n:>5} {s.mean:>10} {s.median:>10} {s.minimum:>10} {s.maximum:>10}")
|
|
155
|
+
|
|
156
|
+
for key in args.by or []:
|
|
157
|
+
groups = {k: v for k, v in group_by(ok, key).items() if k is not None}
|
|
158
|
+
if len(groups) < 2:
|
|
159
|
+
continue
|
|
160
|
+
print(f"\nby {key}")
|
|
161
|
+
shown = [m for m in metrics if any(m in r for r in ok)][: args.columns]
|
|
162
|
+
print(f" {key:<28} {'n':>4} " + " ".join(f"{m[:14]:>15}" for m in shown))
|
|
163
|
+
for name, rows in sorted(groups.items(), key=lambda kv: str(kv[0])):
|
|
164
|
+
cells = []
|
|
165
|
+
for m in shown:
|
|
166
|
+
s = summarise(rows, [m])
|
|
167
|
+
cells.append(f"{s[0].mean:>15}" if s else f"{'-':>15}")
|
|
168
|
+
print(f" {str(name):<28} {len(rows):>4} " + " ".join(cells))
|
|
169
|
+
|
|
170
|
+
if args.correlate:
|
|
171
|
+
matrix = correlation_matrix(ok, metrics)
|
|
172
|
+
ranked = sorted(
|
|
173
|
+
((v, a, b) for (a, b), v in matrix.items() if v is not None),
|
|
174
|
+
key=lambda t: -abs(t[0]),
|
|
175
|
+
)
|
|
176
|
+
print(f"\nstrongest rank correlations (Spearman, {len(ranked)} pairs)")
|
|
177
|
+
for value, a, b in ranked[: args.top]:
|
|
178
|
+
print(f" {value:>7} {a} ~ {b}")
|
|
179
|
+
return 0
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def cmd_import(args, reg: Registry) -> int:
|
|
183
|
+
written, unchanged, formats = import_corpus(Path(args.corpus), reg.root, args.dry_run)
|
|
184
|
+
verb = "would write" if args.dry_run else "wrote"
|
|
185
|
+
print(f"{verb} {written} revisions, {unchanged} unchanged, across {formats} formats")
|
|
186
|
+
return 0
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def cmd_validate(args, reg: Registry) -> int:
|
|
190
|
+
problems = list(validate(reg))
|
|
191
|
+
errors = [p for p in problems if p.level == "error"]
|
|
192
|
+
shown = problems if args.all else problems[: args.limit]
|
|
193
|
+
for problem in shown:
|
|
194
|
+
print(f" {problem}")
|
|
195
|
+
if len(problems) > len(shown):
|
|
196
|
+
print(f" ... and {len(problems) - len(shown)} more (use --all)")
|
|
197
|
+
print(f"\n{len(errors)} error(s), {len(problems) - len(errors)} warning(s)")
|
|
198
|
+
return 1 if errors else 0
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
202
|
+
ap = argparse.ArgumentParser(prog="fanbase", description=__doc__)
|
|
203
|
+
ap.add_argument("--registry", help="path to a registry checkout")
|
|
204
|
+
sub = ap.add_subparsers(dest="command", required=True)
|
|
205
|
+
|
|
206
|
+
sub.add_parser("list", help="list formats in the registry").set_defaults(fn=cmd_list)
|
|
207
|
+
|
|
208
|
+
p = sub.add_parser("search", help="find formats and lineages")
|
|
209
|
+
p.add_argument("query")
|
|
210
|
+
p.set_defaults(fn=cmd_search)
|
|
211
|
+
|
|
212
|
+
p = sub.add_parser("show", help="show one revision")
|
|
213
|
+
p.add_argument("ref", help="e.g. png/base/normal-claude-sonnet-5-high-r1@v1")
|
|
214
|
+
p.set_defaults(fn=cmd_show)
|
|
215
|
+
|
|
216
|
+
p = sub.add_parser("install", help="copy a revision into Fandango's include path")
|
|
217
|
+
p.add_argument("ref")
|
|
218
|
+
p.add_argument("--into", help="install directory (default: Fandango's data dir)")
|
|
219
|
+
p.set_defaults(fn=cmd_install)
|
|
220
|
+
|
|
221
|
+
p = sub.add_parser("analyze", help="summarise and correlate measurement records")
|
|
222
|
+
p.add_argument("results", nargs="+", help="JSONL files from `fanbase measure`")
|
|
223
|
+
p.add_argument("--metrics", nargs="*", help="restrict to these metrics")
|
|
224
|
+
p.add_argument("--by", nargs="*", default=["objective", "model", "effort"],
|
|
225
|
+
help="group comparisons on these fields")
|
|
226
|
+
p.add_argument("--correlate", action="store_true", help="rank-correlate every metric pair")
|
|
227
|
+
p.add_argument("--top", type=int, default=15)
|
|
228
|
+
p.add_argument("--columns", type=int, default=4)
|
|
229
|
+
p.set_defaults(fn=cmd_analyze)
|
|
230
|
+
|
|
231
|
+
p = sub.add_parser("import", help="import an upstream corpus into the registry")
|
|
232
|
+
p.add_argument("corpus")
|
|
233
|
+
p.add_argument("--dry-run", action="store_true")
|
|
234
|
+
p.set_defaults(fn=cmd_import)
|
|
235
|
+
|
|
236
|
+
p = sub.add_parser("validate", help="check the registry against its schemas")
|
|
237
|
+
p.add_argument("--limit", type=int, default=25)
|
|
238
|
+
p.add_argument("--all", action="store_true")
|
|
239
|
+
p.set_defaults(fn=cmd_validate)
|
|
240
|
+
|
|
241
|
+
p = sub.add_parser("measure", help="run a metric over registry revisions")
|
|
242
|
+
p.add_argument("--metric", choices=sorted(METRICS), default="intrinsics")
|
|
243
|
+
p.add_argument("--format", help="restrict to one format")
|
|
244
|
+
p.add_argument("--objective", help="restrict to one objective")
|
|
245
|
+
p.add_argument("--python", help="interpreter with Fandango (default: this one)")
|
|
246
|
+
p.add_argument("--timeout", type=int, default=120)
|
|
247
|
+
p.add_argument("--jobs", type=int, default=1)
|
|
248
|
+
p.add_argument("--out", default="results/metrics.jsonl")
|
|
249
|
+
p.add_argument("--extra", help='extra worker arguments, one string: --extra "--samples 8"')
|
|
250
|
+
p.set_defaults(fn=cmd_measure)
|
|
251
|
+
return ap
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def main(argv: list[str] | None = None) -> int:
|
|
255
|
+
args = build_parser().parse_args(argv)
|
|
256
|
+
try:
|
|
257
|
+
return args.fn(args, find_registry(args.registry))
|
|
258
|
+
except RegistryError as exc:
|
|
259
|
+
print(f"fanbase: {exc}", file=sys.stderr)
|
|
260
|
+
return 2
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
if __name__ == "__main__":
|
|
264
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
"""Adapters that turn an external, ad-hoc corpus into registry entries.
|
|
2
|
+
|
|
3
|
+
Everything source-specific lives here. The rest of the package knows only the
|
|
4
|
+
registry layout (`fanbase.registry`), so no core code, CLI command, metric, or site
|
|
5
|
+
build ever depends on how some upstream repository happened to name its files.
|
|
6
|
+
"""
|
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Import an upstream corpus into the registry (DESIGN.md §4).
|
|
3
|
+
|
|
4
|
+
Re-runnable and idempotent: every imported revision records the SHA-256 of the file it
|
|
5
|
+
came from, so re-importing an updated corpus rewrites what changed, adds what is new,
|
|
6
|
+
and leaves everything else untouched.
|
|
7
|
+
|
|
8
|
+
Upstream specs carry no provenance header, so one is generated from the source's
|
|
9
|
+
filename and prepended — that is what makes a registry spec self-describing when someone
|
|
10
|
+
copies the single file into their own project.
|
|
11
|
+
|
|
12
|
+
Canonical specs are deliberately *not* created: blessing one is a maintainer decision
|
|
13
|
+
informed by measurements, never an import-time guess.
|
|
14
|
+
|
|
15
|
+
Exposed as `fanbase import CORPUS`.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import hashlib
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
|
|
23
|
+
from fanbase.importers.ossfuzz_eval import describe, load_corpus
|
|
24
|
+
from fanbase.registry import DEFAULT_OBJECTIVE, DEFAULT_PROFILE
|
|
25
|
+
|
|
26
|
+
GENERATED_BANNER = "# --- generated by fanbase import; edit the registry, not this block ---"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _yaml(mapping: dict[str, object]) -> str:
|
|
30
|
+
"""Minimal YAML writer — avoids a dependency for what is only flat key/value data."""
|
|
31
|
+
lines = []
|
|
32
|
+
for key, value in mapping.items():
|
|
33
|
+
if value is None:
|
|
34
|
+
lines.append(f"{key}:")
|
|
35
|
+
elif isinstance(value, bool):
|
|
36
|
+
lines.append(f"{key}: {'true' if value else 'false'}")
|
|
37
|
+
elif isinstance(value, (int, float)):
|
|
38
|
+
lines.append(f"{key}: {value}")
|
|
39
|
+
elif isinstance(value, list):
|
|
40
|
+
# An empty list must stay a list: `key:` alone parses back as null.
|
|
41
|
+
if not value:
|
|
42
|
+
lines.append(f"{key}: []")
|
|
43
|
+
else:
|
|
44
|
+
lines.append(f"{key}:")
|
|
45
|
+
lines.extend(f" - {item}" for item in value)
|
|
46
|
+
else:
|
|
47
|
+
text = str(value).replace('"', '\\"')
|
|
48
|
+
lines.append(f'{key}: "{text}"')
|
|
49
|
+
return "\n".join(lines) + "\n"
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def build_header(name, source_sha: str, source_rel: str) -> str:
|
|
53
|
+
objective = name.objective or DEFAULT_OBJECTIVE
|
|
54
|
+
fields = {
|
|
55
|
+
"INFO": f"Fandango specification for {name.format}",
|
|
56
|
+
"PROFILE": f"{name.format}/{DEFAULT_PROFILE}",
|
|
57
|
+
"OBJECTIVE": objective,
|
|
58
|
+
"GENESIS": "ai",
|
|
59
|
+
"MODEL": name.model or "unknown",
|
|
60
|
+
"EFFORT": name.effort or "unknown",
|
|
61
|
+
"REPLICATE": name.replicate,
|
|
62
|
+
"SOURCE": source_rel,
|
|
63
|
+
"SOURCE_SHA256": source_sha,
|
|
64
|
+
}
|
|
65
|
+
if name.role_hint:
|
|
66
|
+
fields["ROLE_HINT"] = name.role_hint
|
|
67
|
+
lines = [GENERATED_BANNER]
|
|
68
|
+
for key, value in fields.items():
|
|
69
|
+
lines.append(f"{key} = {value!r}" if isinstance(value, int) else f'{key} = "{value}"')
|
|
70
|
+
lines.append(GENERATED_BANNER)
|
|
71
|
+
return "\n".join(lines) + "\n\n"
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def strip_generated_header(text: str) -> str:
|
|
75
|
+
"""Remove a previously generated block so re-import compares like with like."""
|
|
76
|
+
if not text.startswith(GENERATED_BANNER):
|
|
77
|
+
return text
|
|
78
|
+
end = text.find(GENERATED_BANNER, len(GENERATED_BANNER))
|
|
79
|
+
return text[end + len(GENERATED_BANNER):].lstrip("\n") if end != -1 else text
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def write_if_changed(path: Path, content: str, dry_run: bool) -> bool:
|
|
83
|
+
if path.exists() and path.read_text(encoding="utf-8") == content:
|
|
84
|
+
return False
|
|
85
|
+
if not dry_run:
|
|
86
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
87
|
+
path.write_text(content, encoding="utf-8")
|
|
88
|
+
return True
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def import_corpus(corpus: Path, root: Path, dry_run: bool = False) -> tuple[int, int, int]:
|
|
92
|
+
"""Import `corpus` into the registry at `root`. Returns (written, unchanged, formats)."""
|
|
93
|
+
specs = load_corpus(Path(corpus).resolve())
|
|
94
|
+
root = Path(root).resolve()
|
|
95
|
+
args = type("Args", (), {"dry_run": dry_run})()
|
|
96
|
+
written = skipped = 0
|
|
97
|
+
formats: dict[str, set[str]] = {}
|
|
98
|
+
|
|
99
|
+
for spec in specs:
|
|
100
|
+
name = describe(spec)
|
|
101
|
+
lineage = name.lineage if name.recognised else f"legacy-{spec.path.stem}"
|
|
102
|
+
formats.setdefault(name.format, set()).add(lineage)
|
|
103
|
+
|
|
104
|
+
body = strip_generated_header(spec.path.read_text(encoding="utf-8", errors="replace"))
|
|
105
|
+
source_sha = hashlib.sha256(body.encode("utf-8")).hexdigest()
|
|
106
|
+
content = build_header(name, source_sha, spec.rel.as_posix()) + body
|
|
107
|
+
|
|
108
|
+
base = root / "specs" / name.format / "profiles" / DEFAULT_PROFILE / "lineages" / lineage
|
|
109
|
+
changed = write_if_changed(base / "v1.fan", content, args.dry_run)
|
|
110
|
+
write_if_changed(base / "lineage.yaml", _yaml({
|
|
111
|
+
"id": lineage,
|
|
112
|
+
"objective": name.objective or DEFAULT_OBJECTIVE,
|
|
113
|
+
"role_hint": name.role_hint,
|
|
114
|
+
"model": name.model,
|
|
115
|
+
"effort": name.effort,
|
|
116
|
+
"replicate": name.replicate,
|
|
117
|
+
"arm": name.arm,
|
|
118
|
+
"owner": None,
|
|
119
|
+
"concept_doi": None,
|
|
120
|
+
}), args.dry_run)
|
|
121
|
+
written += changed
|
|
122
|
+
skipped += not changed
|
|
123
|
+
|
|
124
|
+
for fmt, lineages in sorted(formats.items()):
|
|
125
|
+
base = root / "specs" / fmt
|
|
126
|
+
write_if_changed(base / "format.yaml", _yaml({
|
|
127
|
+
"id": fmt,
|
|
128
|
+
"name": fmt.upper(),
|
|
129
|
+
"category": None,
|
|
130
|
+
"references": [],
|
|
131
|
+
"oracle_tier": "T0",
|
|
132
|
+
"maintainers": [],
|
|
133
|
+
}), args.dry_run)
|
|
134
|
+
write_if_changed(base / "profiles.yaml", _yaml({
|
|
135
|
+
"default": DEFAULT_PROFILE,
|
|
136
|
+
"profiles": [DEFAULT_PROFILE],
|
|
137
|
+
}), args.dry_run)
|
|
138
|
+
write_if_changed(base / "profiles" / DEFAULT_PROFILE / "profile.yaml", _yaml({
|
|
139
|
+
"id": DEFAULT_PROFILE,
|
|
140
|
+
"note": "placeholder: upstream declares no format version",
|
|
141
|
+
"extends": None,
|
|
142
|
+
"lineages": len(lineages),
|
|
143
|
+
}), args.dry_run)
|
|
144
|
+
|
|
145
|
+
return written, skipped, len(formats)
|