modelspec-dev 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. api/__init__.py +0 -0
  2. api/class_fit.py +334 -0
  3. api/classes.py +557 -0
  4. api/ranking/__init__.py +12 -0
  5. api/ranking/engine.py +1943 -0
  6. cli/__init__.py +0 -0
  7. cli/modelspec/__init__.py +0 -0
  8. cli/modelspec/cli.py +1819 -0
  9. cli/modelspec/commands/__init__.py +0 -0
  10. cli/modelspec/decide_cmd.py +333 -0
  11. cli/modelspec/offline.py +623 -0
  12. cli/modelspec/snapshot.py +698 -0
  13. cli/modelspec/snapshot_build_cmd.py +49 -0
  14. cli/modelspec/verify_cmd.py +125 -0
  15. cli/modelspec/vocab_cmd.py +204 -0
  16. cli/modelspec/vocabulary_cache.py +54 -0
  17. decision/__init__.py +13 -0
  18. decision/capability.py +872 -0
  19. decision/computed.py +125 -0
  20. decision/contract.py +1575 -0
  21. decision/engine.py +238 -0
  22. decision/excluded.py +34 -0
  23. decision/explain.py +908 -0
  24. decision/filter.py +796 -0
  25. decision/model.py +438 -0
  26. decision/normalise.py +604 -0
  27. decision/optimise.py +320 -0
  28. decision/registry.py +717 -0
  29. decision/relax.py +132 -0
  30. decision/resolve.py +111 -0
  31. decision/schema.py +21 -0
  32. decision/snapshot.py +1483 -0
  33. decision/sources.py +544 -0
  34. decision/templates.py +134 -0
  35. decision/verify.py +1745 -0
  36. decision/vocabulary.py +433 -0
  37. modelspec_dev-0.1.0.dist-info/METADATA +101 -0
  38. modelspec_dev-0.1.0.dist-info/RECORD +63 -0
  39. modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
  40. modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
  41. modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
  42. modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
  43. pipeline/__init__.py +0 -0
  44. pipeline/class_export.py +172 -0
  45. pipeline/hardware.py +434 -0
  46. pipeline/hosts.py +247 -0
  47. pipeline/load.py +224 -0
  48. pipeline/ranking.py +551 -0
  49. registry/domains.yaml +130 -0
  50. registry/facets.yaml +888 -0
  51. registry/harnesses.yaml +79 -0
  52. registry/providers.yaml +354 -0
  53. registry/sources.yaml +3059 -0
  54. registry/templates.yaml +166 -0
  55. schema/__init__.py +0 -0
  56. schema/applicability.py +147 -0
  57. schema/benchmark.py +175 -0
  58. schema/benchmark_eligibility.py +304 -0
  59. schema/card.py +1463 -0
  60. schema/enrichment.py +162 -0
  61. schema/enums.py +327 -0
  62. schema/graph.py +406 -0
  63. schema/suppliers.py +72 -0
decision/vocabulary.py ADDED
@@ -0,0 +1,433 @@
1
+ """The decision vocabulary: what a spec can name, as data (MODEL-153).
2
+
3
+ The site build writes it to ``/api/decision/vocabulary.json`` beside the
4
+ decision snapshot it describes, so a client (the decide page) never carries a
5
+ list of facets or benchmarks of its own:
6
+
7
+ * ``facets``: every registered facet except the parameterised families
8
+ (``evidence.benchmark`` and the like, whose members are the benchmarks
9
+ below), with its label, unit, subject, value type, the condition operators
10
+ that type admits, whether it can be an objective, and how much of the
11
+ snapshot's lineup knows it (``known`` of ``of``), with the values or the
12
+ range it takes there. A facet no lineup candidate knows is listed with
13
+ ``known: 0``; a client should not offer it.
14
+ * ``benchmarks``: every benchmark with verified evidence in the snapshot, with
15
+ its name, unit, domains and directness, how many lineup models have a
16
+ verified row (``models``), how many have one not reported by their own lab
17
+ (``independent_models``), and the range of those values. A benchmark with
18
+ none is not listed.
19
+ * ``domains``: every registered domain with a listed benchmark, its capability
20
+ estimate coverage, and its benchmarks ordered for the explicit drill-down.
21
+ * ``models``: every lineup and archive model, by ID, with the ``display_name``
22
+ and lab (``lab``, ``lab_name``) its card gives. A name the card does not
23
+ give is ``null``; a client shows the ID, never a name made from the slug.
24
+ * ``providers``: every registered provider's display name, by ID.
25
+ * ``coverage``: what the lineup holds, so a client can say what an empty
26
+ answer was measured against without writing it per question. ``models`` is
27
+ the lineup size and ``verified`` how many of those have at least one verified
28
+ evidence row; ``as_of`` is the snapshot date. ``classes`` lists every
29
+ registered model class (zeros included) with the same two counts and, per
30
+ domain, how many of its models have verified evidence there. ``domains``
31
+ lists every registered domain with how many lineup models have verified
32
+ evidence on any benchmark tagged to it (``verified``) and on a direct one
33
+ (``direct``).
34
+
35
+ Operators are the compact condition forms: ``=``, ``!=``, ``<``, ``<=``,
36
+ ``>``, ``>=``, ``between`` (``facet in [low, high]``), ``in`` and ``not in``
37
+ (``facet in {a, b}``), and ``known`` (``known(facet)``).
38
+ """
39
+
40
+ from __future__ import annotations
41
+
42
+ import typing
43
+ from collections import Counter
44
+ from collections.abc import Iterable, Mapping
45
+ from typing import Any
46
+
47
+ from decision.computed import with_computed
48
+ from decision.contract import (
49
+ CONTRACT_VERSION,
50
+ DEFAULT_TASK_TOKENS,
51
+ AllOf,
52
+ AnyOf,
53
+ Known,
54
+ NotOf,
55
+ TaskType,
56
+ parse_spec,
57
+ )
58
+ from decision.engine import decide
59
+ from decision.filter import _INDEPENDENT as INDEPENDENT_MEASURERS
60
+ from decision.templates import load_templates
61
+
62
+ VOCABULARY_VERSION = 1
63
+ MIN_FRONTIER_COVERAGE = 0.50
64
+
65
+ _ORDERED = ("=", "!=", "<", "<=", ">", ">=", "between")
66
+ OPERATORS: Mapping[str, tuple[str, ...]] = {
67
+ "number": (*_ORDERED, "known"),
68
+ "date": (*_ORDERED, "known"),
69
+ "enum": ("=", "!=", "in", "not in", "known"),
70
+ "boolean": ("=", "!=", "known"),
71
+ "set": ("in", "not in", "known"),
72
+ }
73
+ _LITERALS = ("unbounded", "not_offered")
74
+
75
+
76
+ class FrontierCoverageError(ValueError):
77
+ """A domain's default basis would leave most directly measured models unranked."""
78
+
79
+
80
+ def _lineup(snapshot: Any) -> list[str]:
81
+ return [cid for cid in snapshot.candidates() if snapshot.lifecycle(cid) != "retired"]
82
+
83
+
84
+ def frontier_coverage(
85
+ snapshot: Any,
86
+ domain_id: str,
87
+ *,
88
+ benchmark: str | None = None,
89
+ ) -> dict[str, float | int]:
90
+ """Coverage of one ranking basis among lineup models with direct evidence.
91
+
92
+ With no ``benchmark``, the basis is the stored domain capability estimate.
93
+ Passing a benchmark measures the counterfactual single-benchmark basis used
94
+ by the page's explicit "Measured by" drill-down.
95
+ """
96
+ lineup = _lineup(snapshot)
97
+ by_model: dict[str, list[str]] = {}
98
+ for candidate in lineup:
99
+ by_model.setdefault(snapshot.model_of(candidate), []).append(candidate)
100
+ direct_benchmarks = {
101
+ benchmark_id
102
+ for benchmark_id, tags in snapshot.benchmark_domain_tags().items()
103
+ if (domain_id, "direct") in tags
104
+ }
105
+ direct = {
106
+ model_id
107
+ for model_id, candidates in by_model.items()
108
+ if any(
109
+ row.verified and row.benchmark_id in direct_benchmarks
110
+ for candidate in candidates
111
+ for row in snapshot.evidence_for_domain(candidate, domain_id)
112
+ )
113
+ }
114
+ if benchmark is None:
115
+ ranked = {
116
+ model_id
117
+ for model_id in direct
118
+ if any(
119
+ snapshot.capability_estimate(candidate, domain_id) is not None
120
+ for candidate in by_model[model_id]
121
+ )
122
+ }
123
+ else:
124
+ ranked = {
125
+ model_id
126
+ for model_id in direct
127
+ if any(
128
+ row.verified
129
+ for candidate in by_model[model_id]
130
+ for row in snapshot.evidence(candidate, benchmark)
131
+ )
132
+ }
133
+ ratio = len(ranked) / len(direct) if direct else 1.0
134
+ return {"ranked": len(ranked), "direct": len(direct), "ratio": ratio}
135
+
136
+
137
+ def _estimate_models(snapshot: Any, domain_id: str) -> int:
138
+ """How many distinct lineup models have a stored estimate for ``domain_id``."""
139
+ models: set[str] = set()
140
+ for candidate in _lineup(snapshot):
141
+ if snapshot.capability_estimate(candidate, domain_id) is not None:
142
+ models.add(snapshot.model_of(candidate))
143
+ return len(models)
144
+
145
+
146
+ def require_frontier_coverage(
147
+ snapshot: Any,
148
+ domain_id: str,
149
+ *,
150
+ benchmark: str | None = None,
151
+ ) -> dict[str, float | int]:
152
+ """Fail when a proposed default basis ranks less than half the frontier."""
153
+ coverage = frontier_coverage(snapshot, domain_id, benchmark=benchmark)
154
+ if coverage["direct"] and coverage["ratio"] < MIN_FRONTIER_COVERAGE:
155
+ basis = benchmark or f"{domain_id} capability estimate"
156
+ raise FrontierCoverageError(
157
+ f"{domain_id}: default basis {basis} ranks {coverage['ranked']} of "
158
+ f"{coverage['direct']} lineup models with direct evidence "
159
+ f"({coverage['ratio']:.1%}); minimum is {MIN_FRONTIER_COVERAGE:.0%}"
160
+ )
161
+ return coverage
162
+
163
+
164
+ def _number(value: Any) -> float | None:
165
+ if isinstance(value, bool) or not isinstance(value, int | float):
166
+ return None
167
+ return value
168
+
169
+
170
+ def _facet_row(facet: Any, snapshot: Any, subjects: Iterable[str], unit_definition: str | None,
171
+ ) -> dict[str, Any]:
172
+ kind = facet.value_type.kind
173
+ subjects = list(subjects)
174
+ values: list[Any] = []
175
+ for cid in subjects:
176
+ fact = snapshot.fact(cid, facet.id)
177
+ if fact.state == "known" and fact.value is not None:
178
+ values.append(fact.value)
179
+ row: dict[str, Any] = {
180
+ "id": facet.id,
181
+ "label": facet.label or facet.id,
182
+ "definition": facet.definition,
183
+ "subject": facet.subject,
184
+ "value_type": kind,
185
+ "unit": facet.unit,
186
+ "unit_definition": unit_definition,
187
+ "operators": list(OPERATORS[kind]),
188
+ "objective": kind == "number",
189
+ "risk": facet.risk,
190
+ "computed_by": facet.computed_by,
191
+ "known": len(values),
192
+ "of": len(subjects),
193
+ }
194
+ if kind == "number":
195
+ numbers = [n for n in map(_number, values) if n is not None]
196
+ row["range"] = {"min": min(numbers), "max": max(numbers)} if numbers else None
197
+ elif kind == "date":
198
+ dates = sorted(str(v) for v in values)
199
+ row["range"] = {"min": dates[0], "max": dates[-1]} if dates else None
200
+ if kind == "number":
201
+ literals = sorted({v for v in values if v in _LITERALS})
202
+ if literals:
203
+ row["literals"] = literals
204
+ elif kind in ("enum", "boolean", "set"):
205
+ counts: Counter[Any] = Counter()
206
+ for value in values:
207
+ counts.update(set(value) if isinstance(value, list) else {value})
208
+ row["values"] = [
209
+ {"value": v, "count": counts[v],
210
+ **({"label": label} if isinstance(v, str) and (label := facet.value_label(v))
211
+ else {})}
212
+ for v in sorted(counts, key=lambda v: (str(type(v)), str(v)))]
213
+ return row
214
+
215
+
216
+ def _benchmark_rows(snapshot: Any, lineup: list[str], pages: Mapping[str, Mapping[str, Any]],
217
+ ) -> list[dict[str, Any]]:
218
+ tags = snapshot.benchmark_domain_tags()
219
+ rows = []
220
+ for benchmark in snapshot.benchmark_ids():
221
+ models: set[str] = set()
222
+ independent: set[str] = set()
223
+ units: Counter[str] = Counter()
224
+ scores: list[float] = []
225
+ for cid in lineup:
226
+ found = [e for e in snapshot.evidence(cid, benchmark) if e.verified]
227
+ if found:
228
+ models.add(snapshot.model_of(cid))
229
+ units.update(e.unit for e in found if e.unit)
230
+ scores.extend(e.value for e in found)
231
+ if any(e.measured_by in INDEPENDENT_MEASURERS for e in found):
232
+ independent.add(snapshot.model_of(cid))
233
+ if not models:
234
+ continue
235
+ page = pages.get(benchmark) or {}
236
+ metric = page.get("metric") or {}
237
+ unit = sorted(units.items(), key=lambda item: (-item[1], item[0]))[0][0] if units else None
238
+ rows.append({
239
+ "id": benchmark,
240
+ "name": str(page.get("name") or benchmark),
241
+ "unit": unit,
242
+ "higher_is_better": metric.get("direction", "higher_is_better") != "lower_is_better",
243
+ "models": len(models),
244
+ "independent_models": len(independent),
245
+ "range": {"min": min(scores), "max": max(scores)},
246
+ "domains": [{"id": d, "directness": k} for d, k in tags.get(benchmark, ())],
247
+ })
248
+ return rows
249
+
250
+
251
+ def _model_rows(snapshot: Any, cards: Mapping[str, Mapping[str, Any]],
252
+ ) -> dict[str, dict[str, Any]]:
253
+ rows = {}
254
+ for mid in sorted({snapshot.model_of(cid) for cid in snapshot.candidates()}):
255
+ card = cards.get(mid) or {}
256
+ rows[mid] = {
257
+ "display_name": card.get("display_name") or None,
258
+ "lab": card.get("provider") or mid.split("/", 1)[0],
259
+ "lab_name": card.get("provider_display") or None,
260
+ }
261
+ return rows
262
+
263
+
264
+ def _coverage(snapshot: Any, lineup: list[str], registry: Any) -> dict[str, Any]:
265
+ tags = snapshot.benchmark_domain_tags()
266
+ class_of: dict[str, Any] = {}
267
+ for cid in lineup:
268
+ if snapshot.kind(cid) == "model":
269
+ class_of[snapshot.model_of(cid)] = snapshot.fact(cid, "model.class").value
270
+ verified: dict[str, set[str]] = {} # model -> benchmarks with a verified row
271
+ for cid in lineup:
272
+ for benchmark in snapshot.benchmark_ids():
273
+ if any(e.verified for e in snapshot.evidence(cid, benchmark)):
274
+ verified.setdefault(snapshot.model_of(cid), set()).add(benchmark)
275
+
276
+ def in_domain(mid: str, domain: str, *, direct: bool = False) -> bool:
277
+ return any(d == domain and (not direct or k == "direct")
278
+ for b in verified.get(mid, ()) for d, k in tags.get(b, ()))
279
+
280
+ domains = registry.domains()
281
+ classes = []
282
+ for class_id in sorted(registry.allowed_values(registry.facet("model.class")) or ()):
283
+ members = [mid for mid, cls in class_of.items() if cls == class_id]
284
+ per_domain = [{"id": d.id, "verified": n} for d in domains
285
+ if (n := sum(in_domain(mid, d.id) for mid in members))]
286
+ classes.append({
287
+ "id": class_id,
288
+ "models": len(members),
289
+ "verified": sum(mid in verified for mid in members),
290
+ "domains": sorted(per_domain, key=lambda row: (-row["verified"], row["id"])),
291
+ })
292
+ return {
293
+ "as_of": snapshot.as_of.isoformat() if snapshot.as_of else None,
294
+ "models": len(class_of),
295
+ "verified": sum(mid in verified for mid in class_of),
296
+ "classes": classes,
297
+ "domains": [{
298
+ "id": d.id,
299
+ "name": d.name,
300
+ "verified": sum(in_domain(mid, d.id) for mid in class_of),
301
+ "direct": sum(in_domain(mid, d.id, direct=True) for mid in class_of),
302
+ } for d in domains],
303
+ }
304
+
305
+
306
+ def _condition_facet(condition: Any) -> str | None:
307
+ if isinstance(condition, (AnyOf, AllOf)):
308
+ children = condition.any if isinstance(condition, AnyOf) else condition.all
309
+ facets = {_condition_facet(child) for child in children}
310
+ return facets.pop() if len(facets) == 1 else None
311
+ if isinstance(condition, NotOf):
312
+ return _condition_facet(condition.not_)
313
+ return condition.known if isinstance(condition, Known) else condition.facet
314
+
315
+
316
+ def _unavailable_reason(
317
+ decision: Any, funnel: Iterable[Any], template_id: str, registry: Any
318
+ ) -> str:
319
+ """Describe the decision stage that left a template without an answer."""
320
+ for step in funnel:
321
+ if step.after or step.may_qualify:
322
+ continue
323
+ facet_id = _condition_facet(step._condition)
324
+ facet = registry.facet(facet_id) if facet_id else None
325
+ subject = facet.subject if facet else "candidate"
326
+ label = facet.label if facet and facet.label else step.condition
327
+ if template_id == "eu-data" and facet_id == "offering.region":
328
+ label = "Inference region in the EU"
329
+ before = step.offerings_before if subject == "offering" else step.models_before
330
+ noun = subject if before == 1 else f"{subject}s"
331
+ return f"No {subject} passes: {label} — 0 of {before} {noun}"
332
+ if decision.relax_to:
333
+ change = decision.relax_to[0]
334
+ return f"No feasible result; relax {change.condition} to {change.relaxed}."
335
+ if decision.relax:
336
+ return f"No feasible result; relax {decision.relax[0]}."
337
+ return "No feasible result."
338
+
339
+
340
+ def _template_rows(snapshot: Any, registry: Any) -> list[dict[str, Any]]:
341
+ """Templates plus answerability proven by the real decision engine."""
342
+ rows = []
343
+ for template in load_templates(registry=registry):
344
+ spec = parse_spec(template["spec"] | {"explain": "none"}, facets=registry.facet)
345
+ trace = []
346
+ decision = decide(spec, snapshot, facets=registry.facet, _filter_trace=trace.append)
347
+ available = bool(decision.results or decision.may_qualify)
348
+ rows.append(template | {
349
+ "available": available,
350
+ "unavailable_reason": None if available else _unavailable_reason(
351
+ decision, trace[0].funnel, template["id"], registry
352
+ ),
353
+ })
354
+ return rows
355
+
356
+
357
+ def build_vocabulary(snapshot: Any, *, pages: Mapping[str, Mapping[str, Any]] | None = None,
358
+ registry: Any = None, cards: Mapping[str, Mapping[str, Any]] | None = None,
359
+ enforce_frontier_coverage: bool = False,
360
+ ) -> dict[str, Any]:
361
+ """The vocabulary of ``snapshot``. ``pages`` maps benchmark IDs to their page
362
+ front matter (for names and metric direction); ``cards`` maps model IDs to
363
+ their card front matter (for display and lab names); ``registry`` defaults
364
+ to the repository's own. Load ``snapshot`` with its archive to name
365
+ archived models too."""
366
+ if registry is None:
367
+ from decision.registry import default
368
+
369
+ registry = default()
370
+ view = with_computed(snapshot, DEFAULT_TASK_TOKENS)
371
+ lineup = _lineup(view)
372
+ by_kind = {kind: [cid for cid in lineup if view.kind(cid) == kind]
373
+ for kind in ("model", "offering")}
374
+ facets = []
375
+ for facet in registry.facets():
376
+ if facet.parameter is not None:
377
+ continue
378
+ unit = registry.unit(facet.unit).definition if facet.unit else None
379
+ facets.append(_facet_row(facet, view, by_kind[facet.subject], unit))
380
+ benchmarks = _benchmark_rows(view, lineup, pages or {})
381
+ covered = {row["id"]: row for row in benchmarks}
382
+ domains = []
383
+ for domain in registry.domains():
384
+ members = [b for b in covered.values()
385
+ if any(tag["id"] == domain.id for tag in b["domains"])]
386
+ if not members:
387
+ continue
388
+
389
+ estimate_coverage = (
390
+ require_frontier_coverage(snapshot, domain.id)
391
+ if enforce_frontier_coverage
392
+ else frontier_coverage(snapshot, domain.id)
393
+ )
394
+
395
+ # The registry preference leads the explicit drill-down only when it
396
+ # has verified lineup evidence. The domain estimate remains default.
397
+ default = (domain.default_benchmark
398
+ if any(b["id"] == domain.default_benchmark and b["models"] > 0
399
+ for b in members) else None)
400
+
401
+ def order(row: Mapping[str, Any], domain_id: str = domain.id,
402
+ default: str | None = default) -> tuple[Any, ...]:
403
+ direct = any(t == {"id": domain_id, "directness": "direct"} for t in row["domains"])
404
+ return (row["id"] != default, not direct, -row["models"], row["id"])
405
+
406
+ domains.append({
407
+ "id": domain.id,
408
+ "name": domain.name,
409
+ "proxy_only": domain.proxy_only,
410
+ "default_basis": "capability_estimate",
411
+ "estimate_models": _estimate_models(snapshot, domain.id),
412
+ "direct_models": estimate_coverage["direct"],
413
+ # Kept only to preselect the explicit benchmark drill-down. It is
414
+ # never the domain's default ranking basis.
415
+ "default_benchmark": default,
416
+ "benchmarks": [row["id"] for row in sorted(members, key=order)],
417
+ })
418
+ coverage = _coverage(view, lineup, registry)
419
+ return {
420
+ "vocabulary_version": VOCABULARY_VERSION,
421
+ "contract_version": CONTRACT_VERSION,
422
+ "snapshot": snapshot.snapshot_id,
423
+ "default_task_tokens": {"input": DEFAULT_TASK_TOKENS.input,
424
+ "output": DEFAULT_TASK_TOKENS.output},
425
+ "task_types": list(typing.get_args(TaskType)),
426
+ "facets": facets,
427
+ "benchmarks": benchmarks,
428
+ "domains": domains,
429
+ "models": _model_rows(snapshot, cards or {}),
430
+ "providers": {p.id: p.name for p in registry.providers()},
431
+ "coverage": coverage,
432
+ "templates": _template_rows(snapshot, registry),
433
+ }
@@ -0,0 +1,101 @@
1
+ Metadata-Version: 2.5
2
+ Name: modelspec-dev
3
+ Version: 0.1.0
4
+ Summary: A sourced model decision engine with an offline command-line interface
5
+ Project-URL: Homepage, https://modelspec.dev
6
+ Project-URL: Repository, https://github.com/turbobeest/modelspec
7
+ Project-URL: Documentation, https://github.com/turbobeest/modelspec/tree/main/docs
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ License-File: LICENSE-DATA
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Environment :: Console
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.11
17
+ Classifier: Programming Language :: Python :: 3.12
18
+ Classifier: Programming Language :: Python :: 3.13
19
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
20
+ Requires-Python: >=3.11
21
+ Requires-Dist: click>=8.0
22
+ Requires-Dist: falkordb>=1.0
23
+ Requires-Dist: httpx>=0.27
24
+ Requires-Dist: pydantic>=2.0
25
+ Requires-Dist: pyyaml>=6.0
26
+ Requires-Dist: rich>=13.0
27
+ Requires-Dist: typer>=0.12
28
+ Provides-Extra: dev
29
+ Requires-Dist: mypy>=1.11; extra == 'dev'
30
+ Requires-Dist: pytest-asyncio>=0.24; extra == 'dev'
31
+ Requires-Dist: pytest-xdist==3.8.0; extra == 'dev'
32
+ Requires-Dist: pytest>=8.0; extra == 'dev'
33
+ Requires-Dist: ruff>=0.6; extra == 'dev'
34
+ Provides-Extra: researcher
35
+ Requires-Dist: anthropic>=0.39; extra == 'researcher'
36
+ Requires-Dist: beautifulsoup4>=4.12; extra == 'researcher'
37
+ Requires-Dist: celery>=5.4; extra == 'researcher'
38
+ Description-Content-Type: text/markdown
39
+
40
+ <p align="center">
41
+ <img width="2816" height="797" alt="modelspec" src="https://github.com/user-attachments/assets/1e344d67-605d-4577-8ea4-84148d5ae3d5" />
42
+ </p>
43
+
44
+ ModelSpec catalogs AI models as YAML+Markdown cards, exports them to versioned JSON on Cloudflare Pages ([modelspec.dev](https://modelspec.dev)), and ranks from that export — in the browser, or offline from a local snapshot. **No database is on the serving path.**
45
+
46
+ - Site: [modelspec.dev](https://modelspec.dev) · [graph](https://modelspec.dev/graph/) · [downselect](https://modelspec.dev/downselect/)
47
+ - CLI contract: [`docs/cli-contract.md`](docs/cli-contract.md)
48
+ - Current state: [`docs/handoff/current.md`](docs/handoff/current.md)
49
+ - Agent entry: [`AGENTS.md`](AGENTS.md)
50
+
51
+ ## Quick start
52
+
53
+ ```bash
54
+ pipx install modelspec-dev
55
+ modelspec snapshot fetch
56
+ modelspec offline rank coding --json
57
+ ```
58
+
59
+ `snapshot fetch` is the only networked command. After that, rank and fit read the cache.
60
+
61
+ The supported interface is [`docs/cli-contract.md`](docs/cli-contract.md):
62
+
63
+ ```
64
+ modelspec snapshot fetch [--origin URL]
65
+ modelspec snapshot status [--json]
66
+ modelspec offline rank <use-case> [...]
67
+ modelspec offline fit [<hardware-id>]
68
+ ```
69
+
70
+ Graph commands (`stats`, `search`, `info`, `compare`, `rank`, `hardware`, …) need a local FalkorDB and are outside this contract.
71
+
72
+ ## Serving path
73
+
74
+ ```
75
+ models/*.md ──▶ pipeline/build.py ──▶ static JSON on Cloudflare Pages
76
+ (modelspec.dev /api/*.json)
77
+ │
78
+ CLI `snapshot fetch` ───────────────────────┘
79
+ Wizard / 3D graph read the same JSON in the browser.
80
+ ```
81
+
82
+ FalkorDB is optional local graph exploration only. It is not required to rank, fit, or render the sites.
83
+
84
+ ## Contribute
85
+
86
+ Read [`CONTRIBUTING.md`](CONTRIBUTING.md). Sign off commits (`git commit -s`) to certify the [`DCO`](DCO) — not a copyright assignment. Every fact carries a source and the date it was read; unknown means an empty field.
87
+
88
+ Cards live in `models/{provider}/{model-slug}.md`. Edit and open a PR.
89
+
90
+ ## License
91
+
92
+ Two licences, because the code and the corpus want different things.
93
+
94
+ | Part | Licence |
95
+ | --- | --- |
96
+ | Code (everything outside the data directories) | MIT |
97
+ | Data (`models/`, `benchmarks/`, `hardware/`, `hosts/`) | CC BY-SA 4.0 |
98
+
99
+ The corpus is share-alike: build on it, including commercially, but if you redistribute it or a derivative, credit ModelSpec and publish yours under the same terms. The code is permissive so the CLI can go anywhere. Full text in [`LICENSE`](LICENSE) and [`LICENSE-DATA`](LICENSE-DATA).
100
+
101
+ Every card and page records the sources it draws on and the date each was read. Those sources keep their own licences.
@@ -0,0 +1,63 @@
1
+ cli/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
2
+ cli/modelspec/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
3
+ cli/modelspec/cli.py,sha256=oOflhvfSGJqFgTfSw06zA0QNiSje4LuwKMRt4r6D9uE,70636
4
+ cli/modelspec/decide_cmd.py,sha256=1eCDlagaLpylCkzKU11ykvr10Z3R6hqPeLgfIDk1vxE,13732
5
+ cli/modelspec/offline.py,sha256=VD9GA9oCtbOqcnjtmeicVukUl-EsO3y4rxHSPA47uM8,28566
6
+ cli/modelspec/snapshot.py,sha256=GihusgRdcO8V99VoRn0BQ6LFKSH5rnygTMnc_5wr4Ag,29522
7
+ cli/modelspec/snapshot_build_cmd.py,sha256=g6J3q9knDUl_w3yKJeQiCS_Cyqo5uGpN_9ci74d-I_g,2217
8
+ cli/modelspec/verify_cmd.py,sha256=wNs5uLi_pymxUdsvvr9DtbACBNLCElj5QIXe2YymxAI,5038
9
+ cli/modelspec/vocab_cmd.py,sha256=Kpttxbb19S8qqyU6bzp44XK8pOYryfcabNPiTYcP_As,8376
10
+ cli/modelspec/vocabulary_cache.py,sha256=RCvQCgyL2sGX_bgx9e-rFSFTtePa8Cxa6gfcvQI1ufM,1915
11
+ cli/modelspec/commands/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
12
+ decision/__init__.py,sha256=iJ3qiMH-cOBNGmcmICoS6iIy48RqdfzST_AFLk8fG_s,588
13
+ decision/capability.py,sha256=Tm8qNH_ELQIQnJ9uhNAIsSfqVHRn5CEDwqAPscpU8Gg,35652
14
+ decision/computed.py,sha256=uaH2aGuNGyR5dlfa52pYemxvCHYB-JEihBoLpWhOamk,4949
15
+ decision/contract.py,sha256=lxJSRhU7Niq3TGw3wkwPkRgs0iG1esQfAxocarS5cP4,59327
16
+ decision/engine.py,sha256=4DMnEMi4ehjDGMG5LsfQr5nBEmcDaxUVrXs-WAQwd4w,8806
17
+ decision/excluded.py,sha256=xAAqMJqtcQIzeRmIA4R3yEiCEM5Q2gnpqUmIeF_7Z1Q,1150
18
+ decision/explain.py,sha256=iFg4KTlk0hsQsnoyhiXjFdJA2Btgr-BdFS7vGiHlprE,38488
19
+ decision/filter.py,sha256=X4OpQCNyhcg7Ig_qeT-Yt53oXMTDB0-iNTCCX-TnfjM,29489
20
+ decision/model.py,sha256=XpsThjL_RdVW1csb8SH3x2SMN1QdIFi5fdToCg7PWIQ,15402
21
+ decision/normalise.py,sha256=1mVi-Lnlz_dNHK6oDhTuwo2-TFQAhPZ-NqIWE0i7BA4,19718
22
+ decision/optimise.py,sha256=3ZM8SyukjuxV5t6F0OexJYvQhMzD0kGF1sTDlVO2CrA,13951
23
+ decision/registry.py,sha256=h5zw1Fo0aoAYfjLyfUrf3o-xXhidH0GxRcFqadtRze0,29853
24
+ decision/relax.py,sha256=6XAT2NoNLfprOc-azrF3wFa7y1p4v-hejVILJQmEnUQ,5083
25
+ decision/resolve.py,sha256=lP8ntthcc-K8NDiN0zOwSnNLTAlHq7EyuboHjr-h5h4,3517
26
+ decision/schema.py,sha256=YIwpI2XRONuKXgM2a4Nc1C_xr34xh0rKOFoBhAPkbqs,458
27
+ decision/snapshot.py,sha256=UkWTWl0N_3NtaPA-TFUczZIIkRuGN-ftZkTPnK7Fwyk,64458
28
+ decision/sources.py,sha256=06xa6D8fueo2ogN1X5Du1MpRMFoU8B4kpdAHQHZzRj4,19968
29
+ decision/templates.py,sha256=_RkwwcDMuIcxxQMgcfz_AtRMD9tyGyVEFFnoNvpssso,5579
30
+ decision/verify.py,sha256=jgDvPOmjWCYGu2gGKhx4GDDsEn1zDjrpd6hwL2HDW84,75755
31
+ decision/vocabulary.py,sha256=CdWNJsOXrupLsvHwdh0U72F-PilodcpX7ujbzx-DNvI,18345
32
+ schema/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
33
+ schema/applicability.py,sha256=u5qN7J-F7_7_6N6yvrQa_77OJX_aY4VMHoLNamG-e2I,6975
34
+ schema/benchmark.py,sha256=YygwlAWqZEj_0jUdD3tsN27QbHVjzpuyklckXEg4pmU,6325
35
+ schema/benchmark_eligibility.py,sha256=LAlygCfd3ylct9gpIKcBbLDWB2QAVLtZ2cniwGsAgjM,11133
36
+ schema/card.py,sha256=4FcmtgaIySAhWSsCfEM5OGxcGSSvKBI-glGPR9iqmpo,60754
37
+ schema/enrichment.py,sha256=VHs7Y9sNYbsWGFqX7iWc5oXXY7ilppiLSoUe2m7x_DU,7069
38
+ schema/enums.py,sha256=oX-WlDam0I4BwD7ctZH_dR-awmVE3tRzjNJyAx4G3LE,9737
39
+ schema/graph.py,sha256=98lU9EApPWRKVurkUU37pAFO3QH96nEaTZhziiwo6II,17845
40
+ schema/suppliers.py,sha256=bbfOtdKcdFLZwVZUIb5kYlURBMds5cJ6bO58i8yAxpM,2925
41
+ api/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
42
+ api/class_fit.py,sha256=twI-vLxbbkZ697IQPTpAH4_CghlF2E1aj9aLTSHwjh8,13875
43
+ api/classes.py,sha256=efNNRED2ZgtZDbt38kT_jZs5mR6DBkcodzpgOzp1bg0,24218
44
+ api/ranking/__init__.py,sha256=bD3q9uxJWbAxNI6fSEYvBQ6uGndQFt1WsNi5qIWIRAc,419
45
+ api/ranking/engine.py,sha256=vQRa5dHkSi_KvafpCrcvUU6N0nFugcTKM1ku15_goG8,82466
46
+ pipeline/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
47
+ pipeline/class_export.py,sha256=iIHn4D-PtpYSeqI0GPgwfNuE5eFgpgk36y3wf4RiZNo,7325
48
+ pipeline/hardware.py,sha256=O2sksjNnE5ez-S842tqmKLc1q0fw8vdHNeQNwr0sbHs,18588
49
+ pipeline/hosts.py,sha256=J6d59sn7KhnNW0EyIZk26QUgdhbyOOH6O-cCItPI2U4,10031
50
+ pipeline/load.py,sha256=9bEydvLBE8gk-dbDVU3hiQDVrm0uR4R6ZWXOEMbaQ0k,7688
51
+ pipeline/ranking.py,sha256=K2LrEhkbGyDwTnt8SwPWaG8YU7Z2H2QkIzcm13wKoQc,25460
52
+ registry/domains.yaml,sha256=hQR5T7hiqW488h_tOM8yYn9XFuTVvUkM54fy24ZuyPM,5681
53
+ registry/facets.yaml,sha256=qrcWi0kG1ovBZf_cuNih0dnPdGcs2Ess57OKHjlubNc,37111
54
+ registry/harnesses.yaml,sha256=SzytLMcjgnhtegvtKTzAU3qmht4sXkdncVoI4CvzAnc,2948
55
+ registry/providers.yaml,sha256=SDLjIe715tPGJ_4CKny7FiRbqJwyEDlF5xfz7phFFzM,12738
56
+ registry/sources.yaml,sha256=yzIKRCsyGkmlA_rTN_I5lj7OWZP-IgczEWRehzoxk3A,97010
57
+ registry/templates.yaml,sha256=UF3tfh1TjvbN6CCGP2tNLrIuKD25bOpTbExPjJVjP9s,7540
58
+ modelspec_dev-0.1.0.dist-info/METADATA,sha256=SJ9OAOeZv-H-WR4QS_toogSHPpHQSKJRZC16gzP_KC0,4299
59
+ modelspec_dev-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
60
+ modelspec_dev-0.1.0.dist-info/entry_points.txt,sha256=kKMBL87pdMHVJqEsdmL5Karwat8m_NMSc_ExOBofbQo,52
61
+ modelspec_dev-0.1.0.dist-info/licenses/LICENSE,sha256=iCsHytGn2QG7WHx8Z-r3ZIYz5dUL9xatml1XOV4QiMs,2149
62
+ modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA,sha256=KKlSnH0LtNxR9L9cEWo9Fu8kegUvdZFGZ2jd9WP9HPU,20138
63
+ modelspec_dev-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.4
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ modelspec = cli.modelspec.cli:app
@@ -0,0 +1,43 @@
1
+ ModelSpec is licensed in two parts.
2
+
3
+ Code — everything except the data directories below — MIT, as set out in
4
+ full beneath this notice.
5
+
6
+ Data — the curated corpora in `models/`, `benchmarks/`, `hardware/` and
7
+ `hosts/` — Creative Commons Attribution-ShareAlike 4.0
8
+ International (CC BY-SA 4.0), set out in full in LICENSE-DATA.
9
+
10
+ The data is share-alike so that anyone who builds on the corpus and
11
+ redistributes it publishes their improvements under the same terms. The code is
12
+ permissive so the CLI can be embedded anywhere. Individual cards and pages
13
+ record the source and licence of the third-party material they draw on; those
14
+ terms govern that material and are not altered by this file.
15
+
16
+ Decided by the operator on 2026-09-08. The data directories were extended to
17
+ `hardware/` and `hosts/` on 2026-09-16 (MODEL-11); those records are curated data
18
+ on the same terms as the other corpora, and the earlier text covered them only by
19
+ omission.
20
+
21
+ -----------------------------------------------------------------------
22
+
23
+ MIT License
24
+
25
+ Copyright (c) 2026 ModelSpec Contributors
26
+
27
+ Permission is hereby granted, free of charge, to any person obtaining a copy
28
+ of this software and associated documentation files (the "Software"), to deal
29
+ in the Software without restriction, including without limitation the rights
30
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
31
+ copies of the Software, and to permit persons to whom the Software is
32
+ furnished to do so, subject to the following conditions:
33
+
34
+ The above copyright notice and this permission notice shall be included in all
35
+ copies or substantial portions of the Software.
36
+
37
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
38
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
39
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
40
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
41
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
42
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
43
+ SOFTWARE.