asm-protocol 0.5.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
scorer/test_scorer.py ADDED
@@ -0,0 +1,447 @@
1
+ #!/usr/bin/env python3
2
+ """ASM Scorer Unit Tests — 3 key test cases.
3
+
4
+ 1. Golden test: verify TOPSIS output on known input
5
+ 2. io_ratio test: effect of io_ratio on ranking (regression test)
6
+ 3. Cross-language parity: Python vs TypeScript output consistency
7
+
8
+ Usage:
9
+ python -m pytest test_scorer.py -v
10
+ python test_scorer.py # Run directly
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import json
16
+ import math
17
+ import subprocess
18
+ import sys
19
+ from pathlib import Path
20
+
21
+ # Add scorer to path
22
+ _SCORER_DIR = str(Path(__file__).resolve().parent)
23
+ if _SCORER_DIR not in sys.path:
24
+ sys.path.insert(0, _SCORER_DIR)
25
+
26
+ from scorer import (
27
+ Preferences,
28
+ ServiceVector,
29
+ load_manifests,
30
+ parse_manifest,
31
+ score_topsis,
32
+ score_weighted_average,
33
+ cost_delta_from_receipt,
34
+ _extract_primary_cost,
35
+ _extract_primary_quality,
36
+ _parse_latency,
37
+ )
38
+
39
+
40
+ # ============================================================
41
+ # Test data: 3 synthetic manifests (known input)
42
+ # ============================================================
43
+
44
+ SYNTHETIC_MANIFESTS = [
45
+ {
46
+ "asm_version": "0.3",
47
+ "service_id": "test/cheap-fast@1.0",
48
+ "taxonomy": "ai.llm.chat",
49
+ "display_name": "Cheap Fast",
50
+ "pricing": {
51
+ "billing_dimensions": [
52
+ {"dimension": "input_token", "unit": "per_1M", "cost_per_unit": 1.0, "currency": "USD"},
53
+ {"dimension": "output_token", "unit": "per_1M", "cost_per_unit": 2.0, "currency": "USD"},
54
+ ]
55
+ },
56
+ "quality": {"metrics": [{"name": "Elo", "score": 1100, "scale": "Elo"}]},
57
+ "sla": {"latency_p50": "300ms", "uptime": 0.99},
58
+ },
59
+ {
60
+ "asm_version": "0.3",
61
+ "service_id": "test/expensive-good@1.0",
62
+ "taxonomy": "ai.llm.chat",
63
+ "display_name": "Expensive Good",
64
+ "pricing": {
65
+ "billing_dimensions": [
66
+ {"dimension": "input_token", "unit": "per_1M", "cost_per_unit": 10.0, "currency": "USD"},
67
+ {"dimension": "output_token", "unit": "per_1M", "cost_per_unit": 30.0, "currency": "USD"},
68
+ ]
69
+ },
70
+ "quality": {"metrics": [{"name": "Elo", "score": 1350, "scale": "Elo"}]},
71
+ "sla": {"latency_p50": "1.5s", "uptime": 0.999},
72
+ },
73
+ {
74
+ "asm_version": "0.3",
75
+ "service_id": "test/balanced-mid@1.0",
76
+ "taxonomy": "ai.llm.chat",
77
+ "display_name": "Balanced Mid",
78
+ "pricing": {
79
+ "billing_dimensions": [
80
+ {"dimension": "input_token", "unit": "per_1M", "cost_per_unit": 3.0, "currency": "USD"},
81
+ {"dimension": "output_token", "unit": "per_1M", "cost_per_unit": 8.0, "currency": "USD"},
82
+ ]
83
+ },
84
+ "quality": {"metrics": [{"name": "Elo", "score": 1250, "scale": "Elo"}]},
85
+ "sla": {"latency_p50": "800ms", "uptime": 0.995},
86
+ },
87
+ ]
88
+
89
+
90
+ def _parse_all(manifests, io_ratio=0.3):
91
+ """Parse manifests into ServiceVector list."""
92
+ return [parse_manifest(m, io_ratio=io_ratio) for m in manifests]
93
+
94
+
95
+ # ============================================================
96
+ # Test 1: Golden Test — TOPSIS output on known input
97
+ # ============================================================
98
+
99
+ def test_topsis_golden():
100
+ """Verify TOPSIS ranking and score range on known synthetic data.
101
+
102
+ Known:
103
+ - Cheap Fast: Low cost, low latency, medium quality
104
+ - Expensive Good: High cost, high latency, high quality
105
+ - Balanced Mid: Medium across all dimensions
106
+
107
+ With equal weights (0.25 each), TOPSIS should favor Balanced Mid or Cheap Fast
108
+ (because they are competitive across most dimensions)。
109
+ """
110
+ services = _parse_all(SYNTHETIC_MANIFESTS)
111
+ prefs = Preferences(cost=0.25, quality=0.25, speed=0.25, reliability=0.25)
112
+ results = score_topsis(services, prefs)
113
+
114
+ # Basic structure validation
115
+ assert len(results) == 3, f"Expected 3 results, got {len(results)}"
116
+ assert results[0].rank == 1
117
+ assert results[1].rank == 2
118
+ assert results[2].rank == 3
119
+
120
+ # Score range: TOPSIS closeness in [0, 1]
121
+ for r in results:
122
+ assert 0.0 <= r.total_score <= 1.0, f"{r.service.display_name} score {r.total_score} out of range [0,1]"
123
+
124
+ # Ranking monotonicity: strictly decreasing scores
125
+ for i in range(len(results) - 1):
126
+ assert results[i].total_score >= results[i + 1].total_score, \
127
+ f"Rank #{i+1} score {results[i].total_score} < Rank #{i+2} score {results[i+1].total_score}"
128
+
129
+ # Breakdown dimension completeness
130
+ for r in results:
131
+ assert set(r.breakdown.keys()) == {"cost", "quality", "speed", "reliability"}, \
132
+ f"Breakdown Dimensions incomplete: {r.breakdown.keys()}"
133
+ for dim, val in r.breakdown.items():
134
+ assert 0.0 <= val <= 1.0, f"{r.service.display_name}.{dim} = {val} out of range [0,1]"
135
+
136
+ # Reasoning non-empty
137
+ for r in results:
138
+ assert len(r.reasoning) > 0, f"{r.service.display_name} reasoning is empty"
139
+
140
+ # Specific ranking: cost-priority, Cheap Fast should rank first
141
+ prefs_cost = Preferences(cost=0.7, quality=0.1, speed=0.1, reliability=0.1)
142
+ results_cost = score_topsis(services, prefs_cost)
143
+ assert results_cost[0].service.service_id == "test/cheap-fast@1.0", \
144
+ f"Cost-priority rank 1 should be Cheap Fast, got {results_cost[0].service.display_name}"
145
+
146
+ # Quality-priority: Expensive Good should rank first
147
+ prefs_quality = Preferences(cost=0.1, quality=0.7, speed=0.1, reliability=0.1)
148
+ results_quality = score_topsis(services, prefs_quality)
149
+ assert results_quality[0].service.service_id == "test/expensive-good@1.0", \
150
+ f"Quality-priority rank 1 should be Expensive Good, got {results_quality[0].service.display_name}"
151
+
152
+ print("✅ Test 1 (Golden Test): PASSED")
153
+
154
+
155
+ # ============================================================
156
+ # Test 2: io_ratio effect on ranking (regression test)
157
+ # ============================================================
158
+
159
+ def test_io_ratio_regression():
160
+ """Verify io_ratio affects cost calculation and final ranking.
161
+
162
+ io_ratio=0.3 (default, chat) → weighted toward output cost
163
+ io_ratio=0.8 (RAG) → weighted toward input cost
164
+
165
+ For Expensive Good (input=10, output=30):
166
+ io_ratio=0.3 → cost = 0.3*10e-6 + 0.7*30e-6 = 24e-6
167
+ io_ratio=0.8 → cost = 0.8*10e-6 + 0.2*30e-6 = 14e-6
168
+
169
+ So in RAG scenario, Expensive Good has lower cost and may rank higher.
170
+ """
171
+ # Verify _extract_primary_cost io_ratio behavior
172
+ pricing = SYNTHETIC_MANIFESTS[1]["pricing"] # Expensive Good
173
+
174
+ cost_chat = _extract_primary_cost(pricing, io_ratio=0.3)
175
+ cost_rag = _extract_primary_cost(pricing, io_ratio=0.8)
176
+
177
+ # Chat scenario weights output (30/M), so cost is higher
178
+ assert cost_chat > cost_rag, \
179
+ f"Chat cost ({cost_chat}) should be greater than RAG cost ({cost_rag}),because output tokens are more expensive"
180
+
181
+ # Exact value verification
182
+ expected_chat = 0.3 * 10 / 1_000_000 + 0.7 * 30 / 1_000_000
183
+ expected_rag = 0.8 * 10 / 1_000_000 + 0.2 * 30 / 1_000_000
184
+ assert math.isclose(cost_chat, expected_chat, rel_tol=1e-9), \
185
+ f"Chat cost {cost_chat} != expected {expected_chat}"
186
+ assert math.isclose(cost_rag, expected_rag, rel_tol=1e-9), \
187
+ f"RAG cost {cost_rag} != expected {expected_rag}"
188
+
189
+ # Verify io_ratio effect on TOPSIS ranking
190
+ services_chat = _parse_all(SYNTHETIC_MANIFESTS, io_ratio=0.3)
191
+ services_rag = _parse_all(SYNTHETIC_MANIFESTS, io_ratio=0.8)
192
+
193
+ prefs = Preferences(cost=0.5, quality=0.2, speed=0.2, reliability=0.1)
194
+ results_chat = score_topsis(services_chat, prefs)
195
+ results_rag = score_topsis(services_rag, prefs)
196
+
197
+ # Rankings should differ (or at least scores)
198
+ chat_order = [r.service.service_id for r in results_chat]
199
+ rag_order = [r.service.service_id for r in results_rag]
200
+
201
+ # Scores must differ
202
+ chat_scores = {r.service.service_id: r.total_score for r in results_chat}
203
+ rag_scores = {r.service.service_id: r.total_score for r in results_rag}
204
+
205
+ any_diff = False
206
+ for sid in chat_scores:
207
+ if not math.isclose(chat_scores[sid], rag_scores[sid], abs_tol=1e-6):
208
+ any_diff = True
209
+ break
210
+
211
+ assert any_diff, "Scores should differ after io_ratio change"
212
+
213
+ # Edge case tests
214
+ cost_0 = _extract_primary_cost(pricing, io_ratio=0.0) # Pure output
215
+ cost_1 = _extract_primary_cost(pricing, io_ratio=1.0) # Pure input
216
+ expected_0 = 30 / 1_000_000 # Pure output token cost
217
+ expected_1 = 10 / 1_000_000 # Pure input token cost
218
+ assert math.isclose(cost_0, expected_0, rel_tol=1e-9)
219
+ assert math.isclose(cost_1, expected_1, rel_tol=1e-9)
220
+
221
+ print("✅ Test 2 (io_ratio Regression): PASSED")
222
+
223
+
224
+ # ============================================================
225
+ # Test 3: Cross-language parity (Python vs TypeScript)
226
+ # ============================================================
227
+
228
+ def test_cross_language_parity():
229
+ """Verify Python and TypeScript scorers produce consistent output on same input.
230
+
231
+ Uses real 14 manifest files, comparing TOPSIS rankings from both implementations.
232
+ Skipped if TypeScript compilation environment unavailable.
233
+ """
234
+ manifest_dir = Path(__file__).resolve().parent.parent / "manifests"
235
+ registry_dir = Path(__file__).resolve().parent.parent / "registry"
236
+ ts_test_script = registry_dir / "src" / "test_topsis.ts"
237
+
238
+ if not manifest_dir.exists():
239
+ print("⚠️ Test 3 (Cross-language): SKIPPED — manifests directory not found")
240
+ return
241
+
242
+ # Python side
243
+ manifests = load_manifests(manifest_dir)
244
+ if len(manifests) < 2:
245
+ print("⚠️ Test 3 (Cross-language): SKIPPED — insufficient manifests")
246
+ return
247
+
248
+ services = _parse_all(manifests, io_ratio=0.3)
249
+ prefs = Preferences(cost=0.3, quality=0.3, speed=0.2, reliability=0.2)
250
+ py_results = score_topsis(services, prefs)
251
+
252
+ py_ranking = [(r.service.service_id, r.total_score) for r in py_results]
253
+
254
+ # TypeScript side — try running
255
+ try:
256
+ result = subprocess.run(
257
+ ["npx", "tsx", str(ts_test_script)],
258
+ cwd=str(registry_dir),
259
+ capture_output=True,
260
+ text=True,
261
+ timeout=30,
262
+ )
263
+ if result.returncode != 0:
264
+ print(f"⚠️ Test 3 (Cross-language): SKIPPED — TypeScript execution failed: {result.stderr[:200]}")
265
+ return
266
+
267
+ # Parse TypeScript output (TOPSIS part only, ignore Weighted Average)
268
+ ts_ranking = []
269
+ in_topsis_section = False
270
+ for line in result.stdout.strip().split("\n"):
271
+ line = line.strip()
272
+ if "TOPSIS" in line and "---" in line:
273
+ in_topsis_section = True
274
+ continue
275
+ if "Weighted Average" in line and "---" in line:
276
+ in_topsis_section = False
277
+ continue
278
+ if not in_topsis_section:
279
+ continue
280
+ if not line.startswith("#"):
281
+ continue
282
+ # Format: #1 service_id: score=0.xxxx ...
283
+ parts = line.split()
284
+ if len(parts) >= 3:
285
+ sid = parts[1].rstrip(":")
286
+ score_str = parts[2].split("=")[1]
287
+ ts_ranking.append((sid, float(score_str)))
288
+
289
+ if not ts_ranking:
290
+ print("⚠️ Test 3 (Cross-language): SKIPPED — Cannot parse TypeScript output")
291
+ return
292
+
293
+ # Compare rankings
294
+ py_order = [sid for sid, _ in py_ranking]
295
+ ts_order = [sid for sid, _ in ts_ranking]
296
+
297
+ assert py_order == ts_order, \
298
+ f"Rankings inconsistent!\nPython: {py_order[:5]}\nTypeScript: {ts_order[:5]}"
299
+
300
+ # Compare scores (0.001 tolerance)
301
+ py_scores = dict(py_ranking)
302
+ ts_scores = dict(ts_ranking)
303
+
304
+ max_diff = 0.0
305
+ for sid in py_scores:
306
+ if sid in ts_scores:
307
+ diff = abs(py_scores[sid] - ts_scores[sid])
308
+ max_diff = max(max_diff, diff)
309
+ assert diff < 0.001, \
310
+ f"{sid}: Python={py_scores[sid]:.4f} vs TS={ts_scores[sid]:.4f}, diff={diff:.6f}"
311
+
312
+ print(f"✅ Test 3 (Cross-language Parity): PASSED — {len(py_ranking)} services ranked consistently, max score diff: {max_diff:.6f}")
313
+
314
+ except FileNotFoundError:
315
+ print("⚠️ Test 3 (Cross-language): SKIPPED — npx/tsx unavailable")
316
+ except subprocess.TimeoutExpired:
317
+ print("⚠️ Test 3 (Cross-language): SKIPPED — TypeScript execution timeout")
318
+
319
+
320
+ # ============================================================
321
+ # Trust Delta receipt extension v0.1 — cost_delta_from_receipt
322
+ # ============================================================
323
+
324
+ def test_cost_delta_agreement():
325
+ """When the receipt's claimed cost matches the manifest-recomputed cost,
326
+ cost_delta should be ~0."""
327
+ manifest = {
328
+ "pricing": {
329
+ "billing_dimensions": [
330
+ {"dimension": "input_token", "unit": "per_1M", "cost_per_unit": 3.00, "currency": "USD"},
331
+ {"dimension": "output_token", "unit": "per_1M", "cost_per_unit": 15.00, "currency": "USD"},
332
+ ]
333
+ }
334
+ }
335
+ receipt = {
336
+ "billing": {
337
+ "currency": "USD",
338
+ "cost": 0.0037,
339
+ "observed_input_tokens": 1000, # 1000 * 3 / 1M = 0.003
340
+ "observed_output_tokens": 50, # 50 * 15 / 1M = 0.00075
341
+ # total = 0.00375
342
+ }
343
+ }
344
+ result = cost_delta_from_receipt(manifest, receipt)
345
+ assert result["claimed_cost"] == 0.0037
346
+ assert abs(result["recomputed_cost"] - 0.00375) < 1e-9, result
347
+ assert abs(result["cost_delta"] - (-0.00005)) < 1e-9, result
348
+ assert "input_token" in result["per_dimension"]
349
+ assert "output_token" in result["per_dimension"]
350
+ print("✅ Test 4 (cost_delta agreement): tiny delta (-5e-5 USD) matches expectation")
351
+
352
+
353
+ def test_cost_delta_divergence():
354
+ """Publisher claims much less than the manifest rate implies — delta is negative
355
+ (publisher under-claims) and substantial."""
356
+ manifest = {
357
+ "pricing": {
358
+ "billing_dimensions": [
359
+ {"dimension": "input_token", "unit": "per_1M", "cost_per_unit": 3.00, "currency": "USD"},
360
+ {"dimension": "output_token", "unit": "per_1M", "cost_per_unit": 15.00, "currency": "USD"},
361
+ ]
362
+ }
363
+ }
364
+ receipt = {
365
+ "billing": {
366
+ "currency": "USD",
367
+ "cost": 0.001, # publisher claims 0.001 USD
368
+ "observed_input_tokens": 1000,
369
+ "observed_output_tokens": 1000,
370
+ # manifest-recomputed = 1000*3/1M + 1000*15/1M = 0.018
371
+ }
372
+ }
373
+ result = cost_delta_from_receipt(manifest, receipt)
374
+ assert abs(result["recomputed_cost"] - 0.018) < 1e-9, result
375
+ assert result["cost_delta"] < -0.01, result # publisher claimed much less than manifest rate
376
+ print(f"✅ Test 5 (cost_delta divergence): delta={result['cost_delta']:.6f} (under-claim of ~$0.017)")
377
+
378
+
379
+ def test_cost_delta_pipeline_run():
380
+ """Pipeline-level service uses a single 'pipeline_run' dimension instead of token-based.
381
+ Receipt carries `quantity` and dimension. cost_delta should fall back to per-run pricing."""
382
+ manifest = {
383
+ "pricing": {
384
+ "billing_dimensions": [
385
+ {"dimension": "pipeline_run", "unit": "per_run", "cost_per_unit": 0.0037, "currency": "USD"},
386
+ ]
387
+ }
388
+ }
389
+ receipt = {
390
+ "billing": {
391
+ "dimension": "pipeline_run",
392
+ "unit": "per_run",
393
+ "quantity": 1,
394
+ "currency": "USD",
395
+ "cost": 0.0037,
396
+ }
397
+ }
398
+ result = cost_delta_from_receipt(manifest, receipt)
399
+ assert abs(result["recomputed_cost"] - 0.0037) < 1e-9, result
400
+ assert abs(result["cost_delta"]) < 1e-9, result
401
+ assert "pipeline_run" in result["per_dimension"]
402
+ print("✅ Test 6 (cost_delta pipeline_run): exact match, delta=0")
403
+
404
+
405
+ # ============================================================
406
+ # Main entry
407
+ # ============================================================
408
+
409
+ def main():
410
+ print("=" * 60)
411
+ print(" ASM Scorer Unit Tests")
412
+ print("=" * 60)
413
+ print()
414
+
415
+ passed = 0
416
+ failed = 0
417
+ skipped = 0
418
+
419
+ tests = [
420
+ ("Test 1: TOPSIS Golden Test", test_topsis_golden),
421
+ ("Test 2: io_ratio Regression", test_io_ratio_regression),
422
+ ("Test 3: Cross-language Parity", test_cross_language_parity),
423
+ ("Test 4: cost_delta_from_receipt — agreement case", test_cost_delta_agreement),
424
+ ("Test 5: cost_delta_from_receipt — divergence case", test_cost_delta_divergence),
425
+ ("Test 6: cost_delta_from_receipt — pipeline_run dimension", test_cost_delta_pipeline_run),
426
+ ]
427
+
428
+ for name, test_fn in tests:
429
+ try:
430
+ test_fn()
431
+ passed += 1
432
+ except AssertionError as e:
433
+ print(f"❌ {name}: FAILED — {e}")
434
+ failed += 1
435
+ except Exception as e:
436
+ print(f"⚠️ {name}: ERROR — {e}")
437
+ failed += 1
438
+
439
+ print(f"\n{'=' * 60}")
440
+ print(f" Results: {passed} passed, {failed} failed")
441
+ print(f"{'=' * 60}")
442
+
443
+ return 0 if failed == 0 else 1
444
+
445
+
446
+ if __name__ == "__main__":
447
+ sys.exit(main())