asm-protocol 0.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- asm_cli.py +626 -0
- asm_protocol-0.5.0.dist-info/METADATA +395 -0
- asm_protocol-0.5.0.dist-info/RECORD +21 -0
- asm_protocol-0.5.0.dist-info/WHEEL +5 -0
- asm_protocol-0.5.0.dist-info/entry_points.txt +4 -0
- asm_protocol-0.5.0.dist-info/licenses/LICENSE +21 -0
- asm_protocol-0.5.0.dist-info/top_level.txt +7 -0
- asm_select_api.py +144 -0
- asm_selector_mcp.py +94 -0
- library_select.py +149 -0
- mcp_server_json_asm.py +177 -0
- openrouter_adapter.py +421 -0
- scorer/__init__.py +45 -0
- scorer/data/elo_snapshot.json +5027 -0
- scorer/scorer.py +883 -0
- scorer/test_langchain_adapter.py +54 -0
- scorer/test_library_select.py +104 -0
- scorer/test_manifests_schema.py +55 -0
- scorer/test_mcp_server_json_asm.py +84 -0
- scorer/test_openrouter_adapter.py +238 -0
- scorer/test_scorer.py +447 -0
scorer/test_scorer.py
ADDED
|
@@ -0,0 +1,447 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""ASM Scorer Unit Tests — 3 key test cases.
|
|
3
|
+
|
|
4
|
+
1. Golden test: verify TOPSIS output on known input
|
|
5
|
+
2. io_ratio test: effect of io_ratio on ranking (regression test)
|
|
6
|
+
3. Cross-language parity: Python vs TypeScript output consistency
|
|
7
|
+
|
|
8
|
+
Usage:
|
|
9
|
+
python -m pytest test_scorer.py -v
|
|
10
|
+
python test_scorer.py # Run directly
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
import math
|
|
17
|
+
import subprocess
|
|
18
|
+
import sys
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
# Add scorer to path
|
|
22
|
+
_SCORER_DIR = str(Path(__file__).resolve().parent)
|
|
23
|
+
if _SCORER_DIR not in sys.path:
|
|
24
|
+
sys.path.insert(0, _SCORER_DIR)
|
|
25
|
+
|
|
26
|
+
from scorer import (
|
|
27
|
+
Preferences,
|
|
28
|
+
ServiceVector,
|
|
29
|
+
load_manifests,
|
|
30
|
+
parse_manifest,
|
|
31
|
+
score_topsis,
|
|
32
|
+
score_weighted_average,
|
|
33
|
+
cost_delta_from_receipt,
|
|
34
|
+
_extract_primary_cost,
|
|
35
|
+
_extract_primary_quality,
|
|
36
|
+
_parse_latency,
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
# ============================================================
|
|
41
|
+
# Test data: 3 synthetic manifests (known input)
|
|
42
|
+
# ============================================================
|
|
43
|
+
|
|
44
|
+
SYNTHETIC_MANIFESTS = [
|
|
45
|
+
{
|
|
46
|
+
"asm_version": "0.3",
|
|
47
|
+
"service_id": "test/cheap-fast@1.0",
|
|
48
|
+
"taxonomy": "ai.llm.chat",
|
|
49
|
+
"display_name": "Cheap Fast",
|
|
50
|
+
"pricing": {
|
|
51
|
+
"billing_dimensions": [
|
|
52
|
+
{"dimension": "input_token", "unit": "per_1M", "cost_per_unit": 1.0, "currency": "USD"},
|
|
53
|
+
{"dimension": "output_token", "unit": "per_1M", "cost_per_unit": 2.0, "currency": "USD"},
|
|
54
|
+
]
|
|
55
|
+
},
|
|
56
|
+
"quality": {"metrics": [{"name": "Elo", "score": 1100, "scale": "Elo"}]},
|
|
57
|
+
"sla": {"latency_p50": "300ms", "uptime": 0.99},
|
|
58
|
+
},
|
|
59
|
+
{
|
|
60
|
+
"asm_version": "0.3",
|
|
61
|
+
"service_id": "test/expensive-good@1.0",
|
|
62
|
+
"taxonomy": "ai.llm.chat",
|
|
63
|
+
"display_name": "Expensive Good",
|
|
64
|
+
"pricing": {
|
|
65
|
+
"billing_dimensions": [
|
|
66
|
+
{"dimension": "input_token", "unit": "per_1M", "cost_per_unit": 10.0, "currency": "USD"},
|
|
67
|
+
{"dimension": "output_token", "unit": "per_1M", "cost_per_unit": 30.0, "currency": "USD"},
|
|
68
|
+
]
|
|
69
|
+
},
|
|
70
|
+
"quality": {"metrics": [{"name": "Elo", "score": 1350, "scale": "Elo"}]},
|
|
71
|
+
"sla": {"latency_p50": "1.5s", "uptime": 0.999},
|
|
72
|
+
},
|
|
73
|
+
{
|
|
74
|
+
"asm_version": "0.3",
|
|
75
|
+
"service_id": "test/balanced-mid@1.0",
|
|
76
|
+
"taxonomy": "ai.llm.chat",
|
|
77
|
+
"display_name": "Balanced Mid",
|
|
78
|
+
"pricing": {
|
|
79
|
+
"billing_dimensions": [
|
|
80
|
+
{"dimension": "input_token", "unit": "per_1M", "cost_per_unit": 3.0, "currency": "USD"},
|
|
81
|
+
{"dimension": "output_token", "unit": "per_1M", "cost_per_unit": 8.0, "currency": "USD"},
|
|
82
|
+
]
|
|
83
|
+
},
|
|
84
|
+
"quality": {"metrics": [{"name": "Elo", "score": 1250, "scale": "Elo"}]},
|
|
85
|
+
"sla": {"latency_p50": "800ms", "uptime": 0.995},
|
|
86
|
+
},
|
|
87
|
+
]
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _parse_all(manifests, io_ratio=0.3):
|
|
91
|
+
"""Parse manifests into ServiceVector list."""
|
|
92
|
+
return [parse_manifest(m, io_ratio=io_ratio) for m in manifests]
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
# ============================================================
|
|
96
|
+
# Test 1: Golden Test — TOPSIS output on known input
|
|
97
|
+
# ============================================================
|
|
98
|
+
|
|
99
|
+
def test_topsis_golden():
|
|
100
|
+
"""Verify TOPSIS ranking and score range on known synthetic data.
|
|
101
|
+
|
|
102
|
+
Known:
|
|
103
|
+
- Cheap Fast: Low cost, low latency, medium quality
|
|
104
|
+
- Expensive Good: High cost, high latency, high quality
|
|
105
|
+
- Balanced Mid: Medium across all dimensions
|
|
106
|
+
|
|
107
|
+
With equal weights (0.25 each), TOPSIS should favor Balanced Mid or Cheap Fast
|
|
108
|
+
(because they are competitive across most dimensions)。
|
|
109
|
+
"""
|
|
110
|
+
services = _parse_all(SYNTHETIC_MANIFESTS)
|
|
111
|
+
prefs = Preferences(cost=0.25, quality=0.25, speed=0.25, reliability=0.25)
|
|
112
|
+
results = score_topsis(services, prefs)
|
|
113
|
+
|
|
114
|
+
# Basic structure validation
|
|
115
|
+
assert len(results) == 3, f"Expected 3 results, got {len(results)}"
|
|
116
|
+
assert results[0].rank == 1
|
|
117
|
+
assert results[1].rank == 2
|
|
118
|
+
assert results[2].rank == 3
|
|
119
|
+
|
|
120
|
+
# Score range: TOPSIS closeness in [0, 1]
|
|
121
|
+
for r in results:
|
|
122
|
+
assert 0.0 <= r.total_score <= 1.0, f"{r.service.display_name} score {r.total_score} out of range [0,1]"
|
|
123
|
+
|
|
124
|
+
# Ranking monotonicity: strictly decreasing scores
|
|
125
|
+
for i in range(len(results) - 1):
|
|
126
|
+
assert results[i].total_score >= results[i + 1].total_score, \
|
|
127
|
+
f"Rank #{i+1} score {results[i].total_score} < Rank #{i+2} score {results[i+1].total_score}"
|
|
128
|
+
|
|
129
|
+
# Breakdown dimension completeness
|
|
130
|
+
for r in results:
|
|
131
|
+
assert set(r.breakdown.keys()) == {"cost", "quality", "speed", "reliability"}, \
|
|
132
|
+
f"Breakdown Dimensions incomplete: {r.breakdown.keys()}"
|
|
133
|
+
for dim, val in r.breakdown.items():
|
|
134
|
+
assert 0.0 <= val <= 1.0, f"{r.service.display_name}.{dim} = {val} out of range [0,1]"
|
|
135
|
+
|
|
136
|
+
# Reasoning non-empty
|
|
137
|
+
for r in results:
|
|
138
|
+
assert len(r.reasoning) > 0, f"{r.service.display_name} reasoning is empty"
|
|
139
|
+
|
|
140
|
+
# Specific ranking: cost-priority, Cheap Fast should rank first
|
|
141
|
+
prefs_cost = Preferences(cost=0.7, quality=0.1, speed=0.1, reliability=0.1)
|
|
142
|
+
results_cost = score_topsis(services, prefs_cost)
|
|
143
|
+
assert results_cost[0].service.service_id == "test/cheap-fast@1.0", \
|
|
144
|
+
f"Cost-priority rank 1 should be Cheap Fast, got {results_cost[0].service.display_name}"
|
|
145
|
+
|
|
146
|
+
# Quality-priority: Expensive Good should rank first
|
|
147
|
+
prefs_quality = Preferences(cost=0.1, quality=0.7, speed=0.1, reliability=0.1)
|
|
148
|
+
results_quality = score_topsis(services, prefs_quality)
|
|
149
|
+
assert results_quality[0].service.service_id == "test/expensive-good@1.0", \
|
|
150
|
+
f"Quality-priority rank 1 should be Expensive Good, got {results_quality[0].service.display_name}"
|
|
151
|
+
|
|
152
|
+
print("✅ Test 1 (Golden Test): PASSED")
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
# ============================================================
|
|
156
|
+
# Test 2: io_ratio effect on ranking (regression test)
|
|
157
|
+
# ============================================================
|
|
158
|
+
|
|
159
|
+
def test_io_ratio_regression():
|
|
160
|
+
"""Verify io_ratio affects cost calculation and final ranking.
|
|
161
|
+
|
|
162
|
+
io_ratio=0.3 (default, chat) → weighted toward output cost
|
|
163
|
+
io_ratio=0.8 (RAG) → weighted toward input cost
|
|
164
|
+
|
|
165
|
+
For Expensive Good (input=10, output=30):
|
|
166
|
+
io_ratio=0.3 → cost = 0.3*10e-6 + 0.7*30e-6 = 24e-6
|
|
167
|
+
io_ratio=0.8 → cost = 0.8*10e-6 + 0.2*30e-6 = 14e-6
|
|
168
|
+
|
|
169
|
+
So in RAG scenario, Expensive Good has lower cost and may rank higher.
|
|
170
|
+
"""
|
|
171
|
+
# Verify _extract_primary_cost io_ratio behavior
|
|
172
|
+
pricing = SYNTHETIC_MANIFESTS[1]["pricing"] # Expensive Good
|
|
173
|
+
|
|
174
|
+
cost_chat = _extract_primary_cost(pricing, io_ratio=0.3)
|
|
175
|
+
cost_rag = _extract_primary_cost(pricing, io_ratio=0.8)
|
|
176
|
+
|
|
177
|
+
# Chat scenario weights output (30/M), so cost is higher
|
|
178
|
+
assert cost_chat > cost_rag, \
|
|
179
|
+
f"Chat cost ({cost_chat}) should be greater than RAG cost ({cost_rag}),because output tokens are more expensive"
|
|
180
|
+
|
|
181
|
+
# Exact value verification
|
|
182
|
+
expected_chat = 0.3 * 10 / 1_000_000 + 0.7 * 30 / 1_000_000
|
|
183
|
+
expected_rag = 0.8 * 10 / 1_000_000 + 0.2 * 30 / 1_000_000
|
|
184
|
+
assert math.isclose(cost_chat, expected_chat, rel_tol=1e-9), \
|
|
185
|
+
f"Chat cost {cost_chat} != expected {expected_chat}"
|
|
186
|
+
assert math.isclose(cost_rag, expected_rag, rel_tol=1e-9), \
|
|
187
|
+
f"RAG cost {cost_rag} != expected {expected_rag}"
|
|
188
|
+
|
|
189
|
+
# Verify io_ratio effect on TOPSIS ranking
|
|
190
|
+
services_chat = _parse_all(SYNTHETIC_MANIFESTS, io_ratio=0.3)
|
|
191
|
+
services_rag = _parse_all(SYNTHETIC_MANIFESTS, io_ratio=0.8)
|
|
192
|
+
|
|
193
|
+
prefs = Preferences(cost=0.5, quality=0.2, speed=0.2, reliability=0.1)
|
|
194
|
+
results_chat = score_topsis(services_chat, prefs)
|
|
195
|
+
results_rag = score_topsis(services_rag, prefs)
|
|
196
|
+
|
|
197
|
+
# Rankings should differ (or at least scores)
|
|
198
|
+
chat_order = [r.service.service_id for r in results_chat]
|
|
199
|
+
rag_order = [r.service.service_id for r in results_rag]
|
|
200
|
+
|
|
201
|
+
# Scores must differ
|
|
202
|
+
chat_scores = {r.service.service_id: r.total_score for r in results_chat}
|
|
203
|
+
rag_scores = {r.service.service_id: r.total_score for r in results_rag}
|
|
204
|
+
|
|
205
|
+
any_diff = False
|
|
206
|
+
for sid in chat_scores:
|
|
207
|
+
if not math.isclose(chat_scores[sid], rag_scores[sid], abs_tol=1e-6):
|
|
208
|
+
any_diff = True
|
|
209
|
+
break
|
|
210
|
+
|
|
211
|
+
assert any_diff, "Scores should differ after io_ratio change"
|
|
212
|
+
|
|
213
|
+
# Edge case tests
|
|
214
|
+
cost_0 = _extract_primary_cost(pricing, io_ratio=0.0) # Pure output
|
|
215
|
+
cost_1 = _extract_primary_cost(pricing, io_ratio=1.0) # Pure input
|
|
216
|
+
expected_0 = 30 / 1_000_000 # Pure output token cost
|
|
217
|
+
expected_1 = 10 / 1_000_000 # Pure input token cost
|
|
218
|
+
assert math.isclose(cost_0, expected_0, rel_tol=1e-9)
|
|
219
|
+
assert math.isclose(cost_1, expected_1, rel_tol=1e-9)
|
|
220
|
+
|
|
221
|
+
print("✅ Test 2 (io_ratio Regression): PASSED")
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
# ============================================================
|
|
225
|
+
# Test 3: Cross-language parity (Python vs TypeScript)
|
|
226
|
+
# ============================================================
|
|
227
|
+
|
|
228
|
+
def test_cross_language_parity():
|
|
229
|
+
"""Verify Python and TypeScript scorers produce consistent output on same input.
|
|
230
|
+
|
|
231
|
+
Uses real 14 manifest files, comparing TOPSIS rankings from both implementations.
|
|
232
|
+
Skipped if TypeScript compilation environment unavailable.
|
|
233
|
+
"""
|
|
234
|
+
manifest_dir = Path(__file__).resolve().parent.parent / "manifests"
|
|
235
|
+
registry_dir = Path(__file__).resolve().parent.parent / "registry"
|
|
236
|
+
ts_test_script = registry_dir / "src" / "test_topsis.ts"
|
|
237
|
+
|
|
238
|
+
if not manifest_dir.exists():
|
|
239
|
+
print("⚠️ Test 3 (Cross-language): SKIPPED — manifests directory not found")
|
|
240
|
+
return
|
|
241
|
+
|
|
242
|
+
# Python side
|
|
243
|
+
manifests = load_manifests(manifest_dir)
|
|
244
|
+
if len(manifests) < 2:
|
|
245
|
+
print("⚠️ Test 3 (Cross-language): SKIPPED — insufficient manifests")
|
|
246
|
+
return
|
|
247
|
+
|
|
248
|
+
services = _parse_all(manifests, io_ratio=0.3)
|
|
249
|
+
prefs = Preferences(cost=0.3, quality=0.3, speed=0.2, reliability=0.2)
|
|
250
|
+
py_results = score_topsis(services, prefs)
|
|
251
|
+
|
|
252
|
+
py_ranking = [(r.service.service_id, r.total_score) for r in py_results]
|
|
253
|
+
|
|
254
|
+
# TypeScript side — try running
|
|
255
|
+
try:
|
|
256
|
+
result = subprocess.run(
|
|
257
|
+
["npx", "tsx", str(ts_test_script)],
|
|
258
|
+
cwd=str(registry_dir),
|
|
259
|
+
capture_output=True,
|
|
260
|
+
text=True,
|
|
261
|
+
timeout=30,
|
|
262
|
+
)
|
|
263
|
+
if result.returncode != 0:
|
|
264
|
+
print(f"⚠️ Test 3 (Cross-language): SKIPPED — TypeScript execution failed: {result.stderr[:200]}")
|
|
265
|
+
return
|
|
266
|
+
|
|
267
|
+
# Parse TypeScript output (TOPSIS part only, ignore Weighted Average)
|
|
268
|
+
ts_ranking = []
|
|
269
|
+
in_topsis_section = False
|
|
270
|
+
for line in result.stdout.strip().split("\n"):
|
|
271
|
+
line = line.strip()
|
|
272
|
+
if "TOPSIS" in line and "---" in line:
|
|
273
|
+
in_topsis_section = True
|
|
274
|
+
continue
|
|
275
|
+
if "Weighted Average" in line and "---" in line:
|
|
276
|
+
in_topsis_section = False
|
|
277
|
+
continue
|
|
278
|
+
if not in_topsis_section:
|
|
279
|
+
continue
|
|
280
|
+
if not line.startswith("#"):
|
|
281
|
+
continue
|
|
282
|
+
# Format: #1 service_id: score=0.xxxx ...
|
|
283
|
+
parts = line.split()
|
|
284
|
+
if len(parts) >= 3:
|
|
285
|
+
sid = parts[1].rstrip(":")
|
|
286
|
+
score_str = parts[2].split("=")[1]
|
|
287
|
+
ts_ranking.append((sid, float(score_str)))
|
|
288
|
+
|
|
289
|
+
if not ts_ranking:
|
|
290
|
+
print("⚠️ Test 3 (Cross-language): SKIPPED — Cannot parse TypeScript output")
|
|
291
|
+
return
|
|
292
|
+
|
|
293
|
+
# Compare rankings
|
|
294
|
+
py_order = [sid for sid, _ in py_ranking]
|
|
295
|
+
ts_order = [sid for sid, _ in ts_ranking]
|
|
296
|
+
|
|
297
|
+
assert py_order == ts_order, \
|
|
298
|
+
f"Rankings inconsistent!\nPython: {py_order[:5]}\nTypeScript: {ts_order[:5]}"
|
|
299
|
+
|
|
300
|
+
# Compare scores (0.001 tolerance)
|
|
301
|
+
py_scores = dict(py_ranking)
|
|
302
|
+
ts_scores = dict(ts_ranking)
|
|
303
|
+
|
|
304
|
+
max_diff = 0.0
|
|
305
|
+
for sid in py_scores:
|
|
306
|
+
if sid in ts_scores:
|
|
307
|
+
diff = abs(py_scores[sid] - ts_scores[sid])
|
|
308
|
+
max_diff = max(max_diff, diff)
|
|
309
|
+
assert diff < 0.001, \
|
|
310
|
+
f"{sid}: Python={py_scores[sid]:.4f} vs TS={ts_scores[sid]:.4f}, diff={diff:.6f}"
|
|
311
|
+
|
|
312
|
+
print(f"✅ Test 3 (Cross-language Parity): PASSED — {len(py_ranking)} services ranked consistently, max score diff: {max_diff:.6f}")
|
|
313
|
+
|
|
314
|
+
except FileNotFoundError:
|
|
315
|
+
print("⚠️ Test 3 (Cross-language): SKIPPED — npx/tsx unavailable")
|
|
316
|
+
except subprocess.TimeoutExpired:
|
|
317
|
+
print("⚠️ Test 3 (Cross-language): SKIPPED — TypeScript execution timeout")
|
|
318
|
+
|
|
319
|
+
|
|
320
|
+
# ============================================================
|
|
321
|
+
# Trust Delta receipt extension v0.1 — cost_delta_from_receipt
|
|
322
|
+
# ============================================================
|
|
323
|
+
|
|
324
|
+
def test_cost_delta_agreement():
|
|
325
|
+
"""When the receipt's claimed cost matches the manifest-recomputed cost,
|
|
326
|
+
cost_delta should be ~0."""
|
|
327
|
+
manifest = {
|
|
328
|
+
"pricing": {
|
|
329
|
+
"billing_dimensions": [
|
|
330
|
+
{"dimension": "input_token", "unit": "per_1M", "cost_per_unit": 3.00, "currency": "USD"},
|
|
331
|
+
{"dimension": "output_token", "unit": "per_1M", "cost_per_unit": 15.00, "currency": "USD"},
|
|
332
|
+
]
|
|
333
|
+
}
|
|
334
|
+
}
|
|
335
|
+
receipt = {
|
|
336
|
+
"billing": {
|
|
337
|
+
"currency": "USD",
|
|
338
|
+
"cost": 0.0037,
|
|
339
|
+
"observed_input_tokens": 1000, # 1000 * 3 / 1M = 0.003
|
|
340
|
+
"observed_output_tokens": 50, # 50 * 15 / 1M = 0.00075
|
|
341
|
+
# total = 0.00375
|
|
342
|
+
}
|
|
343
|
+
}
|
|
344
|
+
result = cost_delta_from_receipt(manifest, receipt)
|
|
345
|
+
assert result["claimed_cost"] == 0.0037
|
|
346
|
+
assert abs(result["recomputed_cost"] - 0.00375) < 1e-9, result
|
|
347
|
+
assert abs(result["cost_delta"] - (-0.00005)) < 1e-9, result
|
|
348
|
+
assert "input_token" in result["per_dimension"]
|
|
349
|
+
assert "output_token" in result["per_dimension"]
|
|
350
|
+
print("✅ Test 4 (cost_delta agreement): tiny delta (-5e-5 USD) matches expectation")
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
def test_cost_delta_divergence():
|
|
354
|
+
"""Publisher claims much less than the manifest rate implies — delta is negative
|
|
355
|
+
(publisher under-claims) and substantial."""
|
|
356
|
+
manifest = {
|
|
357
|
+
"pricing": {
|
|
358
|
+
"billing_dimensions": [
|
|
359
|
+
{"dimension": "input_token", "unit": "per_1M", "cost_per_unit": 3.00, "currency": "USD"},
|
|
360
|
+
{"dimension": "output_token", "unit": "per_1M", "cost_per_unit": 15.00, "currency": "USD"},
|
|
361
|
+
]
|
|
362
|
+
}
|
|
363
|
+
}
|
|
364
|
+
receipt = {
|
|
365
|
+
"billing": {
|
|
366
|
+
"currency": "USD",
|
|
367
|
+
"cost": 0.001, # publisher claims 0.001 USD
|
|
368
|
+
"observed_input_tokens": 1000,
|
|
369
|
+
"observed_output_tokens": 1000,
|
|
370
|
+
# manifest-recomputed = 1000*3/1M + 1000*15/1M = 0.018
|
|
371
|
+
}
|
|
372
|
+
}
|
|
373
|
+
result = cost_delta_from_receipt(manifest, receipt)
|
|
374
|
+
assert abs(result["recomputed_cost"] - 0.018) < 1e-9, result
|
|
375
|
+
assert result["cost_delta"] < -0.01, result # publisher claimed much less than manifest rate
|
|
376
|
+
print(f"✅ Test 5 (cost_delta divergence): delta={result['cost_delta']:.6f} (under-claim of ~$0.017)")
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
def test_cost_delta_pipeline_run():
|
|
380
|
+
"""Pipeline-level service uses a single 'pipeline_run' dimension instead of token-based.
|
|
381
|
+
Receipt carries `quantity` and dimension. cost_delta should fall back to per-run pricing."""
|
|
382
|
+
manifest = {
|
|
383
|
+
"pricing": {
|
|
384
|
+
"billing_dimensions": [
|
|
385
|
+
{"dimension": "pipeline_run", "unit": "per_run", "cost_per_unit": 0.0037, "currency": "USD"},
|
|
386
|
+
]
|
|
387
|
+
}
|
|
388
|
+
}
|
|
389
|
+
receipt = {
|
|
390
|
+
"billing": {
|
|
391
|
+
"dimension": "pipeline_run",
|
|
392
|
+
"unit": "per_run",
|
|
393
|
+
"quantity": 1,
|
|
394
|
+
"currency": "USD",
|
|
395
|
+
"cost": 0.0037,
|
|
396
|
+
}
|
|
397
|
+
}
|
|
398
|
+
result = cost_delta_from_receipt(manifest, receipt)
|
|
399
|
+
assert abs(result["recomputed_cost"] - 0.0037) < 1e-9, result
|
|
400
|
+
assert abs(result["cost_delta"]) < 1e-9, result
|
|
401
|
+
assert "pipeline_run" in result["per_dimension"]
|
|
402
|
+
print("✅ Test 6 (cost_delta pipeline_run): exact match, delta=0")
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
# ============================================================
|
|
406
|
+
# Main entry
|
|
407
|
+
# ============================================================
|
|
408
|
+
|
|
409
|
+
def main():
|
|
410
|
+
print("=" * 60)
|
|
411
|
+
print(" ASM Scorer Unit Tests")
|
|
412
|
+
print("=" * 60)
|
|
413
|
+
print()
|
|
414
|
+
|
|
415
|
+
passed = 0
|
|
416
|
+
failed = 0
|
|
417
|
+
skipped = 0
|
|
418
|
+
|
|
419
|
+
tests = [
|
|
420
|
+
("Test 1: TOPSIS Golden Test", test_topsis_golden),
|
|
421
|
+
("Test 2: io_ratio Regression", test_io_ratio_regression),
|
|
422
|
+
("Test 3: Cross-language Parity", test_cross_language_parity),
|
|
423
|
+
("Test 4: cost_delta_from_receipt — agreement case", test_cost_delta_agreement),
|
|
424
|
+
("Test 5: cost_delta_from_receipt — divergence case", test_cost_delta_divergence),
|
|
425
|
+
("Test 6: cost_delta_from_receipt — pipeline_run dimension", test_cost_delta_pipeline_run),
|
|
426
|
+
]
|
|
427
|
+
|
|
428
|
+
for name, test_fn in tests:
|
|
429
|
+
try:
|
|
430
|
+
test_fn()
|
|
431
|
+
passed += 1
|
|
432
|
+
except AssertionError as e:
|
|
433
|
+
print(f"❌ {name}: FAILED — {e}")
|
|
434
|
+
failed += 1
|
|
435
|
+
except Exception as e:
|
|
436
|
+
print(f"⚠️ {name}: ERROR — {e}")
|
|
437
|
+
failed += 1
|
|
438
|
+
|
|
439
|
+
print(f"\n{'=' * 60}")
|
|
440
|
+
print(f" Results: {passed} passed, {failed} failed")
|
|
441
|
+
print(f"{'=' * 60}")
|
|
442
|
+
|
|
443
|
+
return 0 if failed == 0 else 1
|
|
444
|
+
|
|
445
|
+
|
|
446
|
+
if __name__ == "__main__":
|
|
447
|
+
sys.exit(main())
|