modelspec-dev 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. api/__init__.py +0 -0
  2. api/class_fit.py +334 -0
  3. api/classes.py +557 -0
  4. api/ranking/__init__.py +12 -0
  5. api/ranking/engine.py +1943 -0
  6. cli/__init__.py +0 -0
  7. cli/modelspec/__init__.py +0 -0
  8. cli/modelspec/cli.py +1819 -0
  9. cli/modelspec/commands/__init__.py +0 -0
  10. cli/modelspec/decide_cmd.py +333 -0
  11. cli/modelspec/offline.py +623 -0
  12. cli/modelspec/snapshot.py +698 -0
  13. cli/modelspec/snapshot_build_cmd.py +49 -0
  14. cli/modelspec/verify_cmd.py +125 -0
  15. cli/modelspec/vocab_cmd.py +204 -0
  16. cli/modelspec/vocabulary_cache.py +54 -0
  17. decision/__init__.py +13 -0
  18. decision/capability.py +872 -0
  19. decision/computed.py +125 -0
  20. decision/contract.py +1575 -0
  21. decision/engine.py +238 -0
  22. decision/excluded.py +34 -0
  23. decision/explain.py +908 -0
  24. decision/filter.py +796 -0
  25. decision/model.py +438 -0
  26. decision/normalise.py +604 -0
  27. decision/optimise.py +320 -0
  28. decision/registry.py +717 -0
  29. decision/relax.py +132 -0
  30. decision/resolve.py +111 -0
  31. decision/schema.py +21 -0
  32. decision/snapshot.py +1483 -0
  33. decision/sources.py +544 -0
  34. decision/templates.py +134 -0
  35. decision/verify.py +1745 -0
  36. decision/vocabulary.py +433 -0
  37. modelspec_dev-0.1.0.dist-info/METADATA +101 -0
  38. modelspec_dev-0.1.0.dist-info/RECORD +63 -0
  39. modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
  40. modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
  41. modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
  42. modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
  43. pipeline/__init__.py +0 -0
  44. pipeline/class_export.py +172 -0
  45. pipeline/hardware.py +434 -0
  46. pipeline/hosts.py +247 -0
  47. pipeline/load.py +224 -0
  48. pipeline/ranking.py +551 -0
  49. registry/domains.yaml +130 -0
  50. registry/facets.yaml +888 -0
  51. registry/harnesses.yaml +79 -0
  52. registry/providers.yaml +354 -0
  53. registry/sources.yaml +3059 -0
  54. registry/templates.yaml +166 -0
  55. schema/__init__.py +0 -0
  56. schema/applicability.py +147 -0
  57. schema/benchmark.py +175 -0
  58. schema/benchmark_eligibility.py +304 -0
  59. schema/card.py +1463 -0
  60. schema/enrichment.py +162 -0
  61. schema/enums.py +327 -0
  62. schema/graph.py +406 -0
  63. schema/suppliers.py +72 -0
@@ -0,0 +1,49 @@
1
+ """``modelspec snapshot build``: compile the decision snapshot (MODEL-138).
2
+
3
+ Collects cards, offerings, sources and the verification log, keeps only
4
+ verified values from registered, non-excluded sources, runs the premier-set
5
+ completeness gate, and writes a gzipped snapshot. Signed with
6
+ ``MODELSPEC_SNAPSHOT_KEY`` when it is set. Exits 1 on any failure, naming the
7
+ model, facet and source for each gate gap.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from datetime import date
13
+ from pathlib import Path
14
+ from typing import Optional
15
+
16
+ import typer
17
+
18
+ from decision import snapshot as snap
19
+
20
+ EXIT_ERROR = 1
21
+ REPO_ROOT = Path(__file__).resolve().parents[2]
22
+
23
+
24
+ def build(
25
+ out: Path = typer.Option(..., "--out", help="Where to write the snapshot (.json.gz)."),
26
+ root: Path = typer.Option(REPO_ROOT, "--root", help="Repository root."),
27
+ premier: Optional[Path] = typer.Option( # noqa: UP045 - Typer reads the annotation
28
+ None, "--premier", help="Premier list; default premier/slice-1.yaml under --root."),
29
+ as_of: Optional[str] = typer.Option( # noqa: UP045
30
+ None, "--as-of", help="The snapshot's date, YYYY-MM-DD; default today."),
31
+ ) -> None:
32
+ """Build the decision snapshot. Fails when a premier model lacks a guaranteed fact."""
33
+ premier = premier or root / "premier" / "slice-1.yaml"
34
+ try:
35
+ when = date.fromisoformat(as_of) if as_of else date.today()
36
+ built = snap.build_from_repo(root, premier=premier, as_of=when,
37
+ registry=snap.default_registry())
38
+ except ValueError as exc: # SnapshotError, CompletenessError, a bad --as-of
39
+ typer.echo(f"error: {exc}", err=True)
40
+ raise typer.Exit(EXIT_ERROR) from exc
41
+ signed = snap.env_key() is not None
42
+ built.write(out)
43
+ lineup = len(built.content["lineup"]["candidates"])
44
+ typer.echo(f"{built.snapshot_id} {built.content_hash}")
45
+ typer.echo(f"{lineup} candidates in the lineup, "
46
+ f"{len(built.content['archive']['candidates'])} in the archive, "
47
+ f"{built.content['out_of_lineup']} models outside the premier lineup; "
48
+ f"{'signed' if signed else f'unsigned ({snap.KEY_ENV} not set)'}")
49
+ typer.echo(f"written to {out}")
@@ -0,0 +1,125 @@
1
+ """``modelspec verify``: verification and decision-engine accuracy.
2
+
3
+ Reads the queue in ``verification/queue/events.jsonl``, re-reads each claim from
4
+ its retained source copy with the deterministic extractors, appends every
5
+ outcome to ``verification/log.jsonl`` and prints a summary. Pass
6
+ ``--llm-reader claude`` (Claude Sonnet) or ``--llm-reader mistral`` (Mistral
7
+ Large on the local ollama host) to add a prose reader after the deterministic
8
+ readers. A reader is only asked about values collected by another model family.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import json
14
+ from datetime import date
15
+ from pathlib import Path
16
+
17
+ import typer
18
+
19
+ from decision import verify as v
20
+ from decision.sources import CopyStore
21
+
22
+ #: ``--llm-reader`` name -> the ``decision.verify`` factory that builds it.
23
+ READERS = {"claude": "claude_extractor", "mistral": "mistral_extractor"}
24
+
25
+ app = typer.Typer(
26
+ help="Verify queued values or run decision-engine accuracy checks.",
27
+ invoke_without_command=True,
28
+ no_args_is_help=False,
29
+ )
30
+
31
+
32
+ @app.callback()
33
+ def verify(
34
+ ctx: typer.Context,
35
+ changed_only: bool = typer.Option(
36
+ False, "--changed-only", help="Only values re-queued by source change detection."
37
+ ),
38
+ root: Path = typer.Option(v.REPO_ROOT, "--root", help="Repository root."),
39
+ llm_reader: str | None = typer.Option(
40
+ None, "--llm-reader", help="Independent prose reader: claude or mistral."
41
+ ),
42
+ as_json: bool = typer.Option(False, "--json", help="Machine-readable output."),
43
+ ) -> None:
44
+ """Verify queued facts and evidence against their sources (two keys)."""
45
+ if ctx.invoked_subcommand is not None:
46
+ return
47
+ directory = root / "verification"
48
+ extractors = v.deterministic_extractors()
49
+ if llm_reader is not None:
50
+ if llm_reader not in READERS:
51
+ raise typer.BadParameter(
52
+ f"supported readers: {', '.join(READERS)}", param_hint="--llm-reader"
53
+ )
54
+ extractors.append(getattr(v, READERS[llm_reader])())
55
+ report = v.run(
56
+ v.Queue(directory),
57
+ v.VerificationLog(directory),
58
+ v.StoredRegions(CopyStore(), v.load_sources(root / "registry" / "sources.yaml")),
59
+ extractors,
60
+ today=date.today(),
61
+ changed_only=changed_only,
62
+ )
63
+ if as_json:
64
+ typer.echo(json.dumps({"command": "verify", **report.to_dict()}, indent=2))
65
+ return
66
+ scope = "re-queued by change detection" if changed_only else "queued"
67
+ typer.echo(f"Verified what was {scope}: {len(report.results)} value(s).")
68
+ for outcome, n in report.counts.items():
69
+ typer.echo(f" {outcome}: {n}")
70
+ for r in report.results:
71
+ if r.outcome == "verified":
72
+ continue
73
+ detail = (
74
+ "; ".join(f"{d.field}: expected {d.expected}, found {d.found}" for d in r.diffs)
75
+ or r.reason
76
+ or ""
77
+ )
78
+ typer.echo(f" {r.outcome:<11} {v.ref_str(r.target)} {detail}")
79
+ for t in report.unknown:
80
+ typer.echo(f" no claim filed for re-queued {v.ref_str(t)}")
81
+
82
+
83
+ def accuracy(
84
+ profile: str = typer.Option("pr", "--profile", help="pr or nightly."),
85
+ output_dir: Path = typer.Option(Path("accuracy-report"), "--output-dir"),
86
+ root: Path = typer.Option(v.REPO_ROOT, "--root", help="Repository root."),
87
+ config: Path | None = typer.Option(None, "--config", help="Accuracy config YAML."),
88
+ snapshot_file: Path | None = typer.Option(None, "--snapshot-file"),
89
+ report_date: str | None = typer.Option(None, "--date", help="Report date, YYYY-MM-DD."),
90
+ llm_reader: str | None = typer.Option(None, "--llm-reader"),
91
+ sample_size: int | None = typer.Option(None, "--sample-size"),
92
+ seed: str | None = typer.Option(None, "--seed"),
93
+ ) -> None:
94
+ """Run the decision-engine accuracy harness and print its Markdown report."""
95
+ from scripts import accuracy as harness
96
+
97
+ if profile not in harness.PROFILES:
98
+ raise typer.BadParameter("supported profiles: pr, nightly", param_hint="--profile")
99
+ try:
100
+ day = date.fromisoformat(report_date) if report_date else date.today()
101
+ except ValueError as exc:
102
+ raise typer.BadParameter("expected YYYY-MM-DD", param_hint="--date") from exc
103
+ report = harness.run_profile(
104
+ root=root,
105
+ config_path=config or root / "accuracy.yaml",
106
+ profile=profile,
107
+ output_dir=output_dir,
108
+ snapshot_file=snapshot_file,
109
+ as_of=day,
110
+ llm_reader=llm_reader,
111
+ sample_size=sample_size,
112
+ random_seed=seed,
113
+ )
114
+ markdown, payload = harness.write_report(report, output_dir)
115
+ typer.echo(markdown.read_text(encoding="utf-8"), nl=False)
116
+ typer.echo(f"JSON: {payload}")
117
+ if report.status == "fail":
118
+ raise typer.Exit(1)
119
+
120
+
121
+ # The accuracy harness runs repository tests and reads repository-only recall
122
+ # material. Keep it available in a source checkout, but do not advertise a
123
+ # command that an installed wheel cannot execute.
124
+ if (v.REPO_ROOT / "scripts" / "accuracy.py").is_file():
125
+ app.command("accuracy")(accuracy)
@@ -0,0 +1,204 @@
1
+ """Inspect the decision vocabulary cached by ``snapshot fetch``."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ from datetime import UTC, datetime
7
+ from typing import Any, Optional
8
+
9
+ import typer
10
+ from rich.console import Console
11
+ from rich.table import Table
12
+
13
+ from . import offline
14
+ from . import snapshot as snap
15
+ from .snapshot import decision_vocabulary_path
16
+ from .vocabulary_cache import (
17
+ VocabularyInvalidError,
18
+ VocabularyMissingError,
19
+ )
20
+ from .vocabulary_cache import (
21
+ load_cached_vocabulary as _load_cached_vocabulary,
22
+ )
23
+
24
+ SECTIONS = (
25
+ "facets", "benchmarks", "domains", "providers", "task-types", "coverage", "templates"
26
+ )
27
+
28
+
29
+ def load_cached_vocabulary(*, as_json: bool = False) -> dict[str, Any]:
30
+ """Read the current cached vocabulary, or exit like other offline commands."""
31
+ try:
32
+ return _load_cached_vocabulary()
33
+ except VocabularyMissingError as exc:
34
+ offline._emit_error("vocab", str(exc), as_json)
35
+ raise typer.Exit(offline.EXIT_NO_SNAPSHOT) from exc
36
+ except VocabularyInvalidError as exc:
37
+ offline._emit_error("vocab", str(exc), as_json)
38
+ raise typer.Exit(offline.EXIT_ERROR) from exc
39
+
40
+
41
+ def _freshness(vocabulary: dict[str, Any]) -> dict[str, Any]:
42
+ """Use rank-cache metadata when present, with an offline decision fallback."""
43
+ try:
44
+ return snap.load().freshness()
45
+ except (snap.SnapshotMissing, snap.SnapshotInvalid):
46
+ path = decision_vocabulary_path()
47
+ fetched_at = datetime.fromtimestamp(path.stat().st_mtime, tz=UTC)
48
+ age_days = (datetime.now(UTC) - fetched_at).total_seconds() / 86400
49
+ return {
50
+ "fetched_at": fetched_at.isoformat(),
51
+ "age_days": round(age_days, 2),
52
+ "stale": age_days > snap.STALE_AFTER_DAYS,
53
+ "stale_after_days": snap.STALE_AFTER_DAYS,
54
+ "origin": "local decision cache",
55
+ "build_commit": vocabulary.get("snapshot"),
56
+ "built_at": vocabulary.get("coverage", {}).get("as_of"),
57
+ }
58
+
59
+
60
+ def _search(rows: list[dict[str, Any]], text: str | None) -> list[dict[str, Any]]:
61
+ if not text:
62
+ return rows
63
+ needle = text.casefold()
64
+ return [row for row in rows if needle in str(row.get("id", "")).casefold()
65
+ or needle in str(row.get("label") or row.get("name") or "").casefold()]
66
+
67
+
68
+ def _filtered(vocabulary: dict[str, Any], section: str, class_id: str | None,
69
+ domain_id: str | None, search: str | None) -> Any:
70
+ key = "task_types" if section == "task-types" else section
71
+ value = vocabulary.get(key, [] if key != "providers" and key != "coverage" else {})
72
+ if section == "providers":
73
+ rows = [{"id": id_, "name": name} for id_, name in value.items()]
74
+ return _search(rows, search)
75
+ if section == "task-types":
76
+ if not search:
77
+ return value
78
+ needle = search.casefold()
79
+ return [item for item in value if needle in str(item).casefold()]
80
+ if section == "coverage":
81
+ return value
82
+
83
+ rows = list(value)
84
+ if section == "benchmarks" and domain_id:
85
+ rows = [row for row in rows if any(tag.get("id") == domain_id
86
+ for tag in row.get("domains", []))]
87
+ if section == "benchmarks" and class_id:
88
+ classes = vocabulary.get("coverage", {}).get("classes", [])
89
+ selected = next((row for row in classes if row.get("id") == class_id), None)
90
+ covered_domains = ({row["id"] for row in selected.get("domains", [])}
91
+ if selected else set())
92
+ rows = [row for row in rows if any(tag.get("id") in covered_domains
93
+ for tag in row.get("domains", []))]
94
+ return _search(rows, search)
95
+
96
+
97
+ def _table(section: str, value: Any) -> Table:
98
+ table = Table(show_header=True, header_style="bold")
99
+ if section == "facets":
100
+ for column in ("id", "label", "type", "operators", "coverage"):
101
+ table.add_column(column)
102
+ for row in value:
103
+ table.add_row(row["id"], row.get("label", ""), row.get("value_type", ""),
104
+ ", ".join(row.get("operators", [])),
105
+ f"{row.get('known', 0)}/{row.get('of', 0)}")
106
+ elif section == "benchmarks":
107
+ for column in ("id", "name", "domains"):
108
+ table.add_column(column)
109
+ for row in value:
110
+ domains = ", ".join(f"{tag['id']} ({tag['directness']})"
111
+ for tag in row.get("domains", []))
112
+ table.add_row(row["id"], row.get("name", ""), domains)
113
+ elif section == "domains":
114
+ for column in ("id", "name", "proxy_only", "estimate models"):
115
+ table.add_column(column)
116
+ for row in value:
117
+ table.add_row(row["id"], row.get("name", ""), str(row.get("proxy_only", False)),
118
+ str(row.get("estimate_models", 0)))
119
+ elif section == "providers":
120
+ table.add_column("id")
121
+ table.add_column("name")
122
+ for row in value:
123
+ table.add_row(row["id"], row["name"])
124
+ elif section == "task-types":
125
+ table.add_column("task type")
126
+ for item in value:
127
+ table.add_row(str(item))
128
+ elif section == "templates":
129
+ for column in ("id", "name", "available", "reason", "purpose"):
130
+ table.add_column(column)
131
+ for row in value:
132
+ table.add_row(
133
+ row["id"], row.get("name", ""), str(row.get("available", True)),
134
+ row.get("unavailable_reason") or "",
135
+ row.get("purpose", ""),
136
+ )
137
+ else:
138
+ table.add_column("coverage")
139
+ table.add_column("value")
140
+ for key, item in value.items():
141
+ rendered = (
142
+ json.dumps(item, separators=(",", ":"))
143
+ if isinstance(item, list | dict)
144
+ else str(item)
145
+ )
146
+ table.add_row(key, rendered)
147
+ return table
148
+
149
+
150
+ def vocab(
151
+ section: Optional[str] = typer.Argument( # noqa: UP045
152
+ None, help="facets, benchmarks, domains, providers, task-types, coverage or templates"
153
+ ),
154
+ class_id: Optional[str] = typer.Option( # noqa: UP045
155
+ None, "--class", help="Limit entries to a model class where coverage permits."
156
+ ),
157
+ domain: Optional[str] = typer.Option( # noqa: UP045
158
+ None, "--domain", help="Limit benchmarks to a capability domain."
159
+ ),
160
+ search: Optional[str] = typer.Option( # noqa: UP045
161
+ None, "--search", help="Substring match on ID and label or name."
162
+ ),
163
+ as_json: bool = typer.Option(
164
+ False, "--json", help="Print the raw vocabulary or selected section."
165
+ ),
166
+ ) -> None:
167
+ """Show the cached decision vocabulary. Never uses the network."""
168
+ vocabulary = load_cached_vocabulary(as_json=as_json)
169
+ if section is None:
170
+ if as_json:
171
+ typer.echo(json.dumps({
172
+ "schema_version": offline.SCHEMA_VERSION,
173
+ "command": "vocab",
174
+ "freshness": _freshness(vocabulary),
175
+ "result": vocabulary,
176
+ }, indent=2))
177
+ return
178
+ providers = vocabulary.get("providers", {})
179
+ typer.echo(f"snapshot: {vocabulary.get('snapshot', 'unknown')}")
180
+ typer.echo(" " + " ".join(
181
+ f"{name}: {len(vocabulary.get(key, providers if key == 'providers' else []))}"
182
+ for name, key in (("facets", "facets"), ("benchmarks", "benchmarks"),
183
+ ("domains", "domains"), ("providers", "providers"),
184
+ ("task-types", "task_types"), ("templates", "templates"))
185
+ ))
186
+ typer.echo("next: modelspec vocab facets")
187
+ return
188
+ if section not in SECTIONS:
189
+ offline._emit_error(
190
+ "vocab", f"unknown section {section!r}; choose one of {', '.join(SECTIONS)}", as_json
191
+ )
192
+ raise typer.Exit(offline.EXIT_ERROR)
193
+ value = _filtered(vocabulary, section, class_id, domain, search)
194
+ if as_json:
195
+ if section == "providers":
196
+ value = {row["id"]: row["name"] for row in value}
197
+ typer.echo(json.dumps({
198
+ "schema_version": offline.SCHEMA_VERSION,
199
+ "command": "vocab",
200
+ "freshness": _freshness(vocabulary),
201
+ "result": value,
202
+ }, indent=2))
203
+ return
204
+ Console(width=160).print(_table(section, value))
@@ -0,0 +1,54 @@
1
+ """Read the decision vocabulary selected by the local cache."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ from typing import Any
7
+
8
+ from . import snapshot
9
+
10
+
11
+ class VocabularyMissingError(ValueError):
12
+ """No decision vocabulary has been cached."""
13
+
14
+
15
+ class VocabularyInvalidError(ValueError):
16
+ """The selected decision vocabulary cannot be read."""
17
+
18
+
19
+ def load_cached_vocabulary() -> dict[str, Any]:
20
+ """Return the selected vocabulary, distinguishing absence from corruption."""
21
+ try:
22
+ current = (
23
+ snapshot.cache_dir() / snapshot.DECISION_DIRECTORY / "current"
24
+ ).read_text(encoding="utf-8").strip()
25
+ except FileNotFoundError as exc:
26
+ raise VocabularyMissingError(
27
+ "no cached decision vocabulary. Run `modelspec snapshot fetch`."
28
+ ) from exc
29
+ except (OSError, UnicodeError) as exc:
30
+ raise VocabularyInvalidError(
31
+ f"cannot read the cached decision vocabulary: {exc}"
32
+ ) from exc
33
+ if snapshot.DECISION_SNAPSHOT_ID.fullmatch(current) is None:
34
+ raise VocabularyInvalidError(
35
+ "cannot read the cached decision vocabulary: "
36
+ "decision/current contains an invalid snapshot ID"
37
+ )
38
+ path = (
39
+ snapshot.cache_dir() / snapshot.DECISION_DIRECTORY / current
40
+ / snapshot.DECISION_VOCABULARY_FILENAME
41
+ )
42
+ try:
43
+ value = json.loads(path.read_text(encoding="utf-8"))
44
+ except FileNotFoundError as exc:
45
+ raise VocabularyMissingError(
46
+ "no cached decision vocabulary. Run `modelspec snapshot fetch`."
47
+ ) from exc
48
+ except (OSError, UnicodeError, json.JSONDecodeError) as exc:
49
+ raise VocabularyInvalidError(
50
+ f"cannot read the cached decision vocabulary: {exc}"
51
+ ) from exc
52
+ if not isinstance(value, dict):
53
+ raise VocabularyInvalidError("cached decision vocabulary is not a JSON object")
54
+ return value
decision/__init__.py ADDED
@@ -0,0 +1,13 @@
1
+ """The decision engine (slice 1; design: `docs/design/decision-engine.md`). The contract is ``decision.contract``.
2
+
3
+ A new package beside v1. Nothing here is imported by `api/ranking/` or
4
+ `pipeline/ranking.py`, and nothing here changes `/v1/rank`.
5
+ """
6
+
7
+ from decision.contract import CONTRACT_VERSION, Decision, Spec, SpecError, parse_spec, spec_hash
8
+ from decision.engine import decide
9
+ from decision.filter import apply
10
+ from decision.resolve import resolve
11
+
12
+ __all__ = ["CONTRACT_VERSION", "Decision", "Spec", "SpecError", "apply", "decide", "parse_spec",
13
+ "resolve", "spec_hash"]