asm-protocol 0.5.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
openrouter_adapter.py ADDED
@@ -0,0 +1,421 @@
1
+ #!/usr/bin/env python3
2
+ """OpenRouter-to-ASM ephemeral manifest adapter.
3
+
4
+ This adapter is intentionally opportunistic: it maps OpenRouter model metadata
5
+ into temporary ASM manifests so the normal ASM scorer can rank LLM APIs without
6
+ requiring providers to publish manifests first.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import json
12
+ import re
13
+ from datetime import datetime, timezone
14
+ from pathlib import Path
15
+ from typing import Any
16
+ from urllib.request import Request, urlopen
17
+
18
+
19
+ ROOT = Path(__file__).resolve().parent
20
+ OPENROUTER_MODELS_URL = "https://openrouter.ai/api/v1/models"
21
+ OPENROUTER_RANKINGS_URL = "https://openrouter.ai/rankings"
22
+ DEFAULT_RANKINGS_JSON = ROOT / "experiments" / "results" / "external_validation" / "openrouter_rankings.json"
23
+
24
+ # LMArena Elo (benchmark-backed quality signal). Snapshot built from
25
+ # huggingface.co/datasets/lmarena-ai/leaderboard-dataset (text/latest).
26
+ # Canonical location is inside the scorer package so it ships in the wheel;
27
+ # ROOT/scorer resolves in both editable dev and installed layouts. The old
28
+ # data/lmarena path is kept as a fallback.
29
+ def _default_elo_snapshot() -> Path:
30
+ packaged = ROOT / "scorer" / "data" / "elo_snapshot.json"
31
+ return packaged if packaged.exists() else ROOT / "data" / "lmarena" / "elo_snapshot.json"
32
+
33
+
34
+ ARENA_ELO_SNAPSHOT = _default_elo_snapshot()
35
+ ARENA_ELO_ANCHOR_LOW = 1000.0
36
+ ARENA_ELO_ANCHOR_HIGH = 1500.0
37
+
38
+
39
+ def utc_now() -> str:
40
+ return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
41
+
42
+
43
+ def load_openrouter_models(models_json: str | Path | None = None, timeout: int = 20) -> tuple[list[dict], str, str]:
44
+ """Load OpenRouter model records from a JSON cache or the public API."""
45
+ retrieved_at = utc_now()
46
+ if models_json:
47
+ path = Path(models_json)
48
+ data = json.loads(path.read_text(encoding="utf-8"))
49
+ source = f"file:{path}"
50
+ else:
51
+ req = Request(OPENROUTER_MODELS_URL, headers={"User-Agent": "asm-openrouter-adapter/0.1"})
52
+ with urlopen(req, timeout=timeout) as resp:
53
+ data = json.loads(resp.read().decode("utf-8"))
54
+ source = OPENROUTER_MODELS_URL
55
+
56
+ if isinstance(data, dict):
57
+ models = data.get("data", [])
58
+ elif isinstance(data, list):
59
+ models = data
60
+ else:
61
+ raise ValueError("OpenRouter models payload must be an object with data[] or a model list")
62
+
63
+ return list(models), source, retrieved_at
64
+
65
+
66
+ def load_openrouter_rankings(rankings_json: str | Path | None = None) -> tuple[dict[str, dict], int, str | None]:
67
+ """Load cached OpenRouter usage rankings keyed by normalized model id."""
68
+ path = Path(rankings_json) if rankings_json else DEFAULT_RANKINGS_JSON
69
+ if not path.exists():
70
+ return {}, 0, None
71
+
72
+ data = json.loads(path.read_text(encoding="utf-8"))
73
+ models = data.get("models", [])
74
+ by_slug: dict[str, dict] = {}
75
+ for model in models:
76
+ for key in _ranking_keys(model):
77
+ if key and key not in by_slug:
78
+ by_slug[key] = model
79
+ return by_slug, int(data.get("n_models") or len(models)), data.get("generated_at")
80
+
81
+
82
+ def openrouter_models_to_manifests(
83
+ models: list[dict],
84
+ *,
85
+ models_source: str,
86
+ retrieved_at: str,
87
+ ranking_by_slug: dict[str, dict] | None = None,
88
+ ranking_count: int = 0,
89
+ ranking_generated_at: str | None = None,
90
+ elo_index: dict | None = None,
91
+ elo_meta: dict | None = None,
92
+ ) -> list[dict]:
93
+ """Convert OpenRouter model records into ephemeral ASM manifests."""
94
+ ranking_by_slug = ranking_by_slug or {}
95
+ manifests: list[dict] = []
96
+ for model in models:
97
+ manifest = openrouter_model_to_manifest(
98
+ model,
99
+ models_source=models_source,
100
+ retrieved_at=retrieved_at,
101
+ ranking_by_slug=ranking_by_slug,
102
+ ranking_count=ranking_count,
103
+ ranking_generated_at=ranking_generated_at,
104
+ elo_index=elo_index,
105
+ elo_meta=elo_meta,
106
+ )
107
+ if manifest is not None:
108
+ manifests.append(manifest)
109
+ return manifests
110
+
111
+
112
+ def openrouter_model_to_manifest(
113
+ model: dict,
114
+ *,
115
+ models_source: str,
116
+ retrieved_at: str,
117
+ ranking_by_slug: dict[str, dict],
118
+ ranking_count: int,
119
+ ranking_generated_at: str | None,
120
+ elo_index: dict | None = None,
121
+ elo_meta: dict | None = None,
122
+ ) -> dict | None:
123
+ model_id = str(model.get("id") or "").strip()
124
+ if not model_id:
125
+ return None
126
+
127
+ pricing = model.get("pricing") or {}
128
+ prompt_cost = _parse_float(pricing.get("prompt"))
129
+ completion_cost = _parse_float(pricing.get("completion"))
130
+ if prompt_cost is None and completion_cost is None:
131
+ return None
132
+
133
+ dims = []
134
+ if prompt_cost is not None:
135
+ dims.append({
136
+ "dimension": "input_token",
137
+ "unit": "per_1M",
138
+ "cost_per_unit": prompt_cost * 1_000_000,
139
+ "currency": "USD",
140
+ })
141
+ if completion_cost is not None:
142
+ dims.append({
143
+ "dimension": "output_token",
144
+ "unit": "per_1M",
145
+ "cost_per_unit": completion_cost * 1_000_000,
146
+ "currency": "USD",
147
+ })
148
+
149
+ ranking = _find_ranking(model_id, ranking_by_slug)
150
+ usage_metric, usage_board = _usage_quality(model_id, ranking, ranking_count, ranking_generated_at)
151
+ arena_metric, arena_board = _arena_quality(model_id, elo_index, elo_meta)
152
+ if arena_metric is not None:
153
+ # Benchmark-backed quality available -> primary axis.
154
+ quality_metrics = [arena_metric, usage_metric]
155
+ leaderboard = arena_board
156
+ elif elo_index:
157
+ # Elo mode active but this model has no benchmark match. Do NOT let the
158
+ # usage-popularity signal masquerade as quality on the same scale as Elo
159
+ # (mixing incomparable quality scales is a documented failure mode).
160
+ quality_metrics = [_unknown_quality(), usage_metric]
161
+ leaderboard = usage_board
162
+ else:
163
+ quality_metrics = [usage_metric]
164
+ leaderboard = usage_board
165
+ architecture = model.get("architecture") or {}
166
+ top_provider = model.get("top_provider") or {}
167
+ context_length = model.get("context_length") or top_provider.get("context_length")
168
+
169
+ notes = [
170
+ "Ephemeral manifest generated from OpenRouter model metadata.",
171
+ "Pricing is OpenRouter-reported and may change.",
172
+ "OpenRouter does not expose per-model latency or uptime in /api/v1/models.",
173
+ ]
174
+ if arena_metric is not None:
175
+ notes.append("Primary quality = LMArena overall Elo (benchmark-backed); usage signal kept as secondary.")
176
+ else:
177
+ notes.append("Quality metric is a usage signal, not benchmark quality.")
178
+ if ranking_generated_at:
179
+ notes.append(f"Usage ranking snapshot: {ranking_generated_at}.")
180
+ else:
181
+ notes.append("No cached usage ranking snapshot was available; neutral usage score used.")
182
+
183
+ manifest = {
184
+ "asm_version": "0.3",
185
+ "service_id": f"openrouter/{model_id}@current",
186
+ "taxonomy": "ai.llm.chat",
187
+ "display_name": str(model.get("name") or model_id),
188
+ "provider": {
189
+ "name": "OpenRouter",
190
+ "url": "https://openrouter.ai",
191
+ "verified_by": ["openrouter-public-api"],
192
+ },
193
+ "capabilities": {
194
+ "description": f"OpenRouter model endpoint for {model_id}",
195
+ "input_modalities": _schema_modalities(architecture.get("input_modalities")),
196
+ "output_modalities": _schema_modalities(architecture.get("output_modalities")),
197
+ },
198
+ "pricing": {
199
+ "billing_dimensions": dims,
200
+ "estimated": False,
201
+ },
202
+ "quality": {
203
+ "metrics": quality_metrics,
204
+ },
205
+ "provenance": {
206
+ "source_url": models_source if models_source.startswith("http") else OPENROUTER_MODELS_URL,
207
+ "retrieved_at": retrieved_at,
208
+ "last_verified_at": retrieved_at,
209
+ "verification_status": "self_reported",
210
+ "notes": " ".join(notes),
211
+ },
212
+ "updated_at": retrieved_at,
213
+ "ttl": 300,
214
+ }
215
+ if context_length:
216
+ manifest["capabilities"]["context_window"] = int(context_length)
217
+ if leaderboard:
218
+ manifest["quality"]["leaderboard_rank"] = leaderboard
219
+ return manifest
220
+
221
+
222
+ def load_openrouter_manifests(
223
+ *,
224
+ models_json: str | Path | None = None,
225
+ rankings_json: str | Path | None = None,
226
+ arena_elo: bool = True,
227
+ arena_category: str = "overall",
228
+ arena_snapshot_json: str | Path | None = None,
229
+ timeout: int = 20,
230
+ ) -> tuple[list[dict], dict]:
231
+ models, source, retrieved_at = load_openrouter_models(models_json=models_json, timeout=timeout)
232
+ rankings, ranking_count, ranking_generated_at = load_openrouter_rankings(rankings_json=rankings_json)
233
+ elo_index, elo_meta = load_arena_elo(arena_snapshot_json, arena_category) if arena_elo else ({}, {})
234
+ manifests = openrouter_models_to_manifests(
235
+ models,
236
+ models_source=source,
237
+ retrieved_at=retrieved_at,
238
+ ranking_by_slug=rankings,
239
+ ranking_count=ranking_count,
240
+ ranking_generated_at=ranking_generated_at,
241
+ elo_index=elo_index,
242
+ elo_meta=elo_meta,
243
+ )
244
+ elo_matched = sum(
245
+ 1 for m in manifests
246
+ if (m.get("quality") or {}).get("metrics")
247
+ and m["quality"]["metrics"][0].get("name") == "lmarena_elo"
248
+ )
249
+ metadata = {
250
+ "source": source,
251
+ "retrieved_at": retrieved_at,
252
+ "n_models": len(models),
253
+ "n_manifests": len(manifests),
254
+ "ranking_snapshot": ranking_generated_at,
255
+ "ranking_count": ranking_count,
256
+ "arena_elo_snapshot": elo_meta.get("publish_date"),
257
+ "arena_elo_matched": elo_matched,
258
+ }
259
+ return manifests, metadata
260
+
261
+
262
+ def _normalize_model_key(name: str) -> str:
263
+ """Normalize an OpenRouter id or Arena model name to a comparable key."""
264
+ n = (name or "").lower().strip().lstrip("~")
265
+ if "/" in n:
266
+ n = n.split("/", 1)[1] # drop provider prefix on OpenRouter ids
267
+ n = n.split(":", 1)[0] # drop :free / :nitro serving variants
268
+ n = re.sub(r"[ ._]+", "-", n) # unify separators
269
+ n = re.sub(r"-+", "-", n).strip("-")
270
+ for suffix in ("-fast", "-turbo", "-nitro", "-high", "-low", "-online"):
271
+ if n.endswith(suffix): # serving variants share the base model's Elo
272
+ n = n[: -len(suffix)]
273
+ return n
274
+
275
+
276
+ def load_arena_elo(
277
+ snapshot_path: str | Path | None = None,
278
+ category: str = "overall",
279
+ ) -> tuple[dict[str, dict], dict]:
280
+ """Load the LMArena Elo snapshot, indexed by normalized model key."""
281
+ path = Path(snapshot_path) if snapshot_path else ARENA_ELO_SNAPSHOT
282
+ if not path.exists():
283
+ return {}, {}
284
+ snap = json.loads(path.read_text(encoding="utf-8"))
285
+ cat = (snap.get("categories") or {}).get(category) or {}
286
+ index: dict[str, dict] = {}
287
+ for row in cat.get("models", []):
288
+ index.setdefault(_normalize_model_key(row.get("model", "")), row)
289
+ meta = {
290
+ "category": category,
291
+ "source": snap.get("source"),
292
+ "retrieved_at": snap.get("retrieved_at"),
293
+ "publish_date": cat.get("leaderboard_publish_date"),
294
+ "elo_min": cat.get("elo_min"),
295
+ "elo_max": cat.get("elo_max"),
296
+ }
297
+ return index, meta
298
+
299
+
300
+ def _arena_quality(
301
+ model_id: str,
302
+ elo_index: dict[str, dict] | None,
303
+ elo_meta: dict | None,
304
+ ) -> tuple[dict, dict] | tuple[None, None]:
305
+ """Return (quality_metric, leaderboard) from LMArena Elo, or (None, None)."""
306
+ if not elo_index:
307
+ return None, None
308
+ row = elo_index.get(_normalize_model_key(model_id))
309
+ if not row:
310
+ return None, None
311
+ elo = float(row.get("elo", 0.0))
312
+ score = max(0.0, min(1.0, (elo - ARENA_ELO_ANCHOR_LOW) / (ARENA_ELO_ANCHOR_HIGH - ARENA_ELO_ANCHOR_LOW)))
313
+ publish_date = (elo_meta or {}).get("publish_date")
314
+ metric = {
315
+ "name": "lmarena_elo",
316
+ "score": round(score, 4),
317
+ "scale": "0-1",
318
+ "benchmark": f"LMArena overall Elo {int(elo)} (snapshot {publish_date})",
319
+ "benchmark_url": "https://lmarena.ai/leaderboard",
320
+ "self_reported": False,
321
+ }
322
+ leaderboard = {
323
+ "name": f"LMArena overall ({publish_date})",
324
+ "rank": row.get("rank"),
325
+ "elo": elo,
326
+ "votes": row.get("votes"),
327
+ "url": "https://lmarena.ai/leaderboard",
328
+ }
329
+ return metric, leaderboard
330
+
331
+
332
+ def _unknown_quality() -> dict:
333
+ """Neutral quality for models with no benchmark Elo (avoids mixing scales)."""
334
+ return {
335
+ "name": "quality_unknown",
336
+ "score": 0.5,
337
+ "scale": "0-1",
338
+ "benchmark": "No LMArena Elo match; quality scored neutral (not benchmark-backed).",
339
+ "self_reported": True,
340
+ }
341
+
342
+
343
+ def _usage_quality(
344
+ model_id: str,
345
+ ranking: dict | None,
346
+ ranking_count: int,
347
+ ranking_generated_at: str | None,
348
+ ) -> tuple[dict, dict | None]:
349
+ if ranking and ranking_count > 1:
350
+ rank = int(ranking.get("rank_by_prompt_tokens") or ranking_count)
351
+ score = max(0.0, min(1.0, 1 - ((rank - 1) / (ranking_count - 1))))
352
+ metric = {
353
+ "name": "openrouter_usage_signal",
354
+ "score": score,
355
+ "scale": "0-1",
356
+ "benchmark": "OpenRouter 7-day prompt-token usage rank",
357
+ "benchmark_url": OPENROUTER_RANKINGS_URL,
358
+ "self_reported": False,
359
+ }
360
+ leaderboard = {
361
+ "name": "OpenRouter 7-day prompt-token usage",
362
+ "rank": rank,
363
+ "total": ranking_count,
364
+ "url": OPENROUTER_RANKINGS_URL,
365
+ }
366
+ if ranking_generated_at:
367
+ leaderboard["snapshot_date"] = ranking_generated_at[:10]
368
+ return metric, leaderboard
369
+
370
+ return {
371
+ "name": "openrouter_usage_signal",
372
+ "score": 0.5,
373
+ "scale": "0-1",
374
+ "benchmark": "No cached OpenRouter ranking match; neutral usage placeholder",
375
+ "benchmark_url": OPENROUTER_RANKINGS_URL,
376
+ "self_reported": True,
377
+ }, None
378
+
379
+
380
+ def _find_ranking(model_id: str, ranking_by_slug: dict[str, dict]) -> dict | None:
381
+ for key in _model_keys(model_id):
382
+ if key in ranking_by_slug:
383
+ return ranking_by_slug[key]
384
+ return None
385
+
386
+
387
+ def _model_keys(model_id: str) -> list[str]:
388
+ normalized = model_id.lower()
389
+ keys = [normalized]
390
+ if ":" in normalized:
391
+ keys.append(normalized.split(":", 1)[0])
392
+ return list(dict.fromkeys(keys))
393
+
394
+
395
+ def _ranking_keys(model: dict) -> list[str]:
396
+ keys = []
397
+ for field in ("slug", "permaslug"):
398
+ value = str(model.get(field) or "").lower()
399
+ if value:
400
+ keys.append(value)
401
+ return list(dict.fromkeys(keys))
402
+
403
+
404
+ def _parse_float(value: Any) -> float | None:
405
+ if value is None:
406
+ return None
407
+ try:
408
+ parsed = float(value)
409
+ except (TypeError, ValueError):
410
+ return None
411
+ if parsed < 0:
412
+ return None
413
+ return parsed
414
+
415
+
416
+ def _schema_modalities(values: Any) -> list[str]:
417
+ allowed = {"text", "image", "audio", "video", "file"}
418
+ if not isinstance(values, list):
419
+ return ["text"]
420
+ result = [str(v) for v in values if str(v) in allowed]
421
+ return result or ["text"]
scorer/__init__.py ADDED
@@ -0,0 +1,45 @@
1
+ """Public Python imports for ASM scoring."""
2
+
3
+ from .scorer import (
4
+ Constraints,
5
+ Preferences,
6
+ ScoredService,
7
+ ServiceVector,
8
+ ReceiptRecord,
9
+ TrustScore,
10
+ compute_trust_delta,
11
+ compute_trust_score,
12
+ cost_delta_from_receipt,
13
+ exponential_decay_weight,
14
+ filter_services,
15
+ load_manifests,
16
+ parse_manifest,
17
+ score_topsis,
18
+ score_weighted_average,
19
+ select_service,
20
+ _extract_primary_cost,
21
+ _extract_primary_quality,
22
+ _parse_latency,
23
+ )
24
+
25
+ __all__ = [
26
+ "Constraints",
27
+ "Preferences",
28
+ "ScoredService",
29
+ "ServiceVector",
30
+ "ReceiptRecord",
31
+ "TrustScore",
32
+ "compute_trust_delta",
33
+ "compute_trust_score",
34
+ "cost_delta_from_receipt",
35
+ "exponential_decay_weight",
36
+ "filter_services",
37
+ "load_manifests",
38
+ "parse_manifest",
39
+ "score_topsis",
40
+ "score_weighted_average",
41
+ "select_service",
42
+ "_extract_primary_cost",
43
+ "_extract_primary_quality",
44
+ "_parse_latency",
45
+ ]