costopt 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- costopt/__init__.py +6 -0
- costopt/anomaly.py +110 -0
- costopt/api/routes.py +354 -0
- costopt/api/server.py +56 -0
- costopt/cache.py +180 -0
- costopt/client.py +307 -0
- costopt/generator.py +226 -0
- costopt/main.py +71 -0
- costopt/pricing.py +118 -0
- costopt/router.py +114 -0
- costopt/telemetry.py +136 -0
- costopt-0.1.0.dist-info/METADATA +200 -0
- costopt-0.1.0.dist-info/RECORD +17 -0
- costopt-0.1.0.dist-info/WHEEL +5 -0
- costopt-0.1.0.dist-info/entry_points.txt +2 -0
- costopt-0.1.0.dist-info/licenses/LICENSE +21 -0
- costopt-0.1.0.dist-info/top_level.txt +1 -0
costopt/__init__.py
ADDED
costopt/anomaly.py
ADDED
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
import sqlite3
|
|
2
|
+
import math
|
|
3
|
+
import logging
|
|
4
|
+
from typing import List, Dict, Any, Tuple
|
|
5
|
+
|
|
6
|
+
logger = logging.getLogger("costopt.anomaly")
|
|
7
|
+
|
|
8
|
+
class AnomalyDetector:
|
|
9
|
+
def __init__(self, telemetry_db_path: str = "costopt_telemetry.db"):
|
|
10
|
+
self.db_path = telemetry_db_path
|
|
11
|
+
|
|
12
|
+
def analyze_daily_cost_anomalies(self, z_threshold: float = 2.0, lookback_days: int = 30) -> List[Dict[str, Any]]:
|
|
13
|
+
"""
|
|
14
|
+
Retrieves daily costs, computes rolling Z-scores, and flags anomalies.
|
|
15
|
+
Returns a list of flagged daily anomaly reports.
|
|
16
|
+
"""
|
|
17
|
+
try:
|
|
18
|
+
with sqlite3.connect(self.db_path) as conn:
|
|
19
|
+
conn.row_factory = sqlite3.Row
|
|
20
|
+
cursor = conn.cursor()
|
|
21
|
+
|
|
22
|
+
# Aggregate total cost by day (YYYY-MM-DD)
|
|
23
|
+
cursor.execute("""
|
|
24
|
+
SELECT
|
|
25
|
+
strftime('%Y-%m-%d', timestamp) as date,
|
|
26
|
+
SUM(cost_actual) as total_cost,
|
|
27
|
+
COUNT(*) as request_count
|
|
28
|
+
FROM telemetry
|
|
29
|
+
GROUP BY date
|
|
30
|
+
ORDER BY date ASC
|
|
31
|
+
LIMIT ?
|
|
32
|
+
""", (lookback_days,))
|
|
33
|
+
|
|
34
|
+
rows = cursor.fetchall()
|
|
35
|
+
if len(rows) < 3:
|
|
36
|
+
# Not enough historical baseline data points to calculate variance/stddev
|
|
37
|
+
logger.warning("Insufficient history points in telemetry database to run anomaly analysis.")
|
|
38
|
+
return []
|
|
39
|
+
|
|
40
|
+
dates = [r["date"] for r in rows]
|
|
41
|
+
costs = [float(r["total_cost"]) for r in rows]
|
|
42
|
+
counts = [int(r["request_count"]) for r in rows]
|
|
43
|
+
|
|
44
|
+
anomalies = []
|
|
45
|
+
n = len(costs)
|
|
46
|
+
|
|
47
|
+
# Compute rolling parameters
|
|
48
|
+
for i in range(2, n):
|
|
49
|
+
current_cost = costs[i]
|
|
50
|
+
current_date = dates[i]
|
|
51
|
+
current_count = counts[i]
|
|
52
|
+
|
|
53
|
+
# History slice up to current index (excluding current point to establish clean baseline)
|
|
54
|
+
history = costs[:i]
|
|
55
|
+
mean = sum(history) / len(history)
|
|
56
|
+
|
|
57
|
+
# Compute standard deviation
|
|
58
|
+
variance = sum((x - mean) ** 2 for x in history) / len(history)
|
|
59
|
+
stddev = math.sqrt(variance)
|
|
60
|
+
|
|
61
|
+
if stddev == 0.0:
|
|
62
|
+
z_score = 0.0
|
|
63
|
+
else:
|
|
64
|
+
z_score = (current_cost - mean) / stddev
|
|
65
|
+
|
|
66
|
+
# Flag anomaly if Z-score exceeds threshold
|
|
67
|
+
if z_score > z_threshold:
|
|
68
|
+
anomalies.append({
|
|
69
|
+
"date": current_date,
|
|
70
|
+
"actual_cost": round(current_cost, 2),
|
|
71
|
+
"expected_mean": round(mean, 2),
|
|
72
|
+
"stddev": round(stddev, 2),
|
|
73
|
+
"z_score": round(z_score, 2),
|
|
74
|
+
"request_count": current_count,
|
|
75
|
+
"severity": "CRITICAL" if z_score > 3.5 else "WARNING"
|
|
76
|
+
})
|
|
77
|
+
|
|
78
|
+
return anomalies
|
|
79
|
+
except Exception as e:
|
|
80
|
+
logger.error(f"Error running cost anomaly detection: {e}")
|
|
81
|
+
return []
|
|
82
|
+
|
|
83
|
+
def get_highest_impact_cost_drivers(self, limit: int = 5) -> List[Dict[str, Any]]:
|
|
84
|
+
"""
|
|
85
|
+
Queries telemetry database to locate applications or models driving the highest spend.
|
|
86
|
+
Useful evidence generation for FinOps recommendations.
|
|
87
|
+
"""
|
|
88
|
+
try:
|
|
89
|
+
with sqlite3.connect(self.db_path) as conn:
|
|
90
|
+
conn.row_factory = sqlite3.Row
|
|
91
|
+
cursor = conn.cursor()
|
|
92
|
+
|
|
93
|
+
# Spend by model
|
|
94
|
+
cursor.execute("""
|
|
95
|
+
SELECT
|
|
96
|
+
model_used,
|
|
97
|
+
provider,
|
|
98
|
+
SUM(cost_actual) as total_spend,
|
|
99
|
+
SUM(savings) as saved_amount,
|
|
100
|
+
COUNT(*) as request_count
|
|
101
|
+
FROM telemetry
|
|
102
|
+
GROUP BY model_used, provider
|
|
103
|
+
ORDER BY total_spend DESC
|
|
104
|
+
LIMIT ?
|
|
105
|
+
""", (limit,))
|
|
106
|
+
|
|
107
|
+
return [dict(row) for row in cursor.fetchall()]
|
|
108
|
+
except Exception as e:
|
|
109
|
+
logger.error(f"Error retrieving cost drivers: {e}")
|
|
110
|
+
return []
|
costopt/api/routes.py
ADDED
|
@@ -0,0 +1,354 @@
|
|
|
1
|
+
import sqlite3
|
|
2
|
+
from fastapi import APIRouter, HTTPException, Query
|
|
3
|
+
from typing import Dict, Any, List, Optional
|
|
4
|
+
from costopt.anomaly import AnomalyDetector
|
|
5
|
+
|
|
6
|
+
router = APIRouter()
|
|
7
|
+
|
|
8
|
+
# Global database paths that can be configured by server init
|
|
9
|
+
_TELEMETRY_DB = "costopt_telemetry.db"
|
|
10
|
+
_CACHE_DB = "costopt_cache.db"
|
|
11
|
+
|
|
12
|
+
def set_db_paths(telemetry_db: str, cache_db: str):
|
|
13
|
+
global _TELEMETRY_DB, _CACHE_DB
|
|
14
|
+
_TELEMETRY_DB = telemetry_db
|
|
15
|
+
_CACHE_DB = cache_db
|
|
16
|
+
|
|
17
|
+
def _get_connection(db_path: str):
|
|
18
|
+
conn = sqlite3.connect(db_path)
|
|
19
|
+
conn.row_factory = sqlite3.Row
|
|
20
|
+
return conn
|
|
21
|
+
|
|
22
|
+
@router.get("/overview")
|
|
23
|
+
def get_overview(env: Optional[str] = None):
|
|
24
|
+
"""Retrieve key aggregate cost and efficiency metrics from logs."""
|
|
25
|
+
try:
|
|
26
|
+
with _get_connection(_TELEMETRY_DB) as conn:
|
|
27
|
+
cursor = conn.cursor()
|
|
28
|
+
|
|
29
|
+
# Filters
|
|
30
|
+
query = "SELECT count(*), sum(cost_original), sum(cost_actual), sum(savings), sum(cache_hit), sum(case when success = 0 then 1 else 0 end) FROM telemetry"
|
|
31
|
+
params = []
|
|
32
|
+
if env:
|
|
33
|
+
query += " WHERE environment = ?"
|
|
34
|
+
params.append(env)
|
|
35
|
+
|
|
36
|
+
cursor.execute(query, params)
|
|
37
|
+
row = cursor.fetchone()
|
|
38
|
+
|
|
39
|
+
count = row[0] or 0
|
|
40
|
+
original = row[1] or 0.0
|
|
41
|
+
actual = row[2] or 0.0
|
|
42
|
+
savings = row[3] or 0.0
|
|
43
|
+
cache_hits = row[4] or 0
|
|
44
|
+
failures = row[5] or 0
|
|
45
|
+
|
|
46
|
+
cache_rate = (cache_hits / count * 100) if count > 0 else 0.0
|
|
47
|
+
error_rate = (failures / count * 100) if count > 0 else 0.0
|
|
48
|
+
|
|
49
|
+
return {
|
|
50
|
+
"total_requests": count,
|
|
51
|
+
"cost_baseline": round(original, 4),
|
|
52
|
+
"cost_actual": round(actual, 4),
|
|
53
|
+
"total_savings": round(savings, 4),
|
|
54
|
+
"cache_hit_rate": round(cache_rate, 2),
|
|
55
|
+
"error_rate": round(error_rate, 2)
|
|
56
|
+
}
|
|
57
|
+
except Exception as e:
|
|
58
|
+
raise HTTPException(status_code=500, detail=f"Database aggregation failed: {e}")
|
|
59
|
+
|
|
60
|
+
@router.get("/charts/savings")
|
|
61
|
+
def get_savings_chart_data(limit: int = 30):
|
|
62
|
+
"""Returns daily cost, baseline cost, and savings metrics for time series plots."""
|
|
63
|
+
try:
|
|
64
|
+
with _get_connection(_TELEMETRY_DB) as conn:
|
|
65
|
+
cursor = conn.cursor()
|
|
66
|
+
cursor.execute("""
|
|
67
|
+
SELECT
|
|
68
|
+
strftime('%Y-%m-%d', timestamp) as date,
|
|
69
|
+
SUM(cost_original) as baseline_cost,
|
|
70
|
+
SUM(cost_actual) as actual_cost,
|
|
71
|
+
SUM(savings) as savings
|
|
72
|
+
FROM telemetry
|
|
73
|
+
GROUP BY date
|
|
74
|
+
ORDER BY date ASC
|
|
75
|
+
LIMIT ?
|
|
76
|
+
""", (limit,))
|
|
77
|
+
rows = cursor.fetchall()
|
|
78
|
+
|
|
79
|
+
return [
|
|
80
|
+
{
|
|
81
|
+
"date": r["date"],
|
|
82
|
+
"baseline_cost": round(r["baseline_cost"], 4),
|
|
83
|
+
"actual_cost": round(r["actual_cost"], 4),
|
|
84
|
+
"savings": round(r["savings"], 4)
|
|
85
|
+
} for r in rows
|
|
86
|
+
]
|
|
87
|
+
except Exception as e:
|
|
88
|
+
raise HTTPException(status_code=500, detail=str(e))
|
|
89
|
+
|
|
90
|
+
@router.get("/charts/models")
|
|
91
|
+
def get_model_distribution():
|
|
92
|
+
"""Returns total request counts and actual cost grouped by provider/model."""
|
|
93
|
+
try:
|
|
94
|
+
with _get_connection(_TELEMETRY_DB) as conn:
|
|
95
|
+
cursor = conn.cursor()
|
|
96
|
+
cursor.execute("""
|
|
97
|
+
SELECT
|
|
98
|
+
provider,
|
|
99
|
+
model_used,
|
|
100
|
+
COUNT(*) as count,
|
|
101
|
+
SUM(cost_actual) as spend
|
|
102
|
+
FROM telemetry
|
|
103
|
+
GROUP BY provider, model_used
|
|
104
|
+
ORDER BY spend DESC
|
|
105
|
+
""")
|
|
106
|
+
rows = cursor.fetchall()
|
|
107
|
+
return [dict(r) for r in rows]
|
|
108
|
+
except Exception as e:
|
|
109
|
+
raise HTTPException(status_code=500, detail=str(e))
|
|
110
|
+
|
|
111
|
+
@router.get("/recommendations")
|
|
112
|
+
def get_recommendations():
|
|
113
|
+
"""Generates concrete, actionable FinOps optimization suggestions backed by database evidence."""
|
|
114
|
+
recommendations = []
|
|
115
|
+
|
|
116
|
+
try:
|
|
117
|
+
with _get_connection(_TELEMETRY_DB) as conn:
|
|
118
|
+
cursor = conn.cursor()
|
|
119
|
+
|
|
120
|
+
# Strategy A: Model Routing Check
|
|
121
|
+
# Look at simple prompts (short length, containing key intent keywords) run on expensive models
|
|
122
|
+
cursor.execute("""
|
|
123
|
+
SELECT
|
|
124
|
+
count(*) as count,
|
|
125
|
+
sum(cost_actual) as cost
|
|
126
|
+
FROM telemetry
|
|
127
|
+
WHERE model_requested IN ('gpt-4', 'claude-3-opus')
|
|
128
|
+
AND length(prompt_hash) > 0 -- just representing prompt presence
|
|
129
|
+
-- simple requests heuristic
|
|
130
|
+
AND (input_tokens < 200)
|
|
131
|
+
""")
|
|
132
|
+
routing_row = cursor.fetchone()
|
|
133
|
+
if routing_row and routing_row["count"] > 0:
|
|
134
|
+
count = routing_row["count"]
|
|
135
|
+
current_cost = routing_row["cost"] or 0.0
|
|
136
|
+
# Let's project saving if routed to mini/haiku (saving ~90% cost)
|
|
137
|
+
estimated_savings = current_cost * 0.90
|
|
138
|
+
recommendations.append({
|
|
139
|
+
"strategy": "Model Routing",
|
|
140
|
+
"title": f"Route short requests from premium models to mini equivalents",
|
|
141
|
+
"description": f"Detected {count} high-cost premium requests with low input token counts (< 200). Routing these to gpt-4o-mini / claude-3-haiku can cut token costs by 90%.",
|
|
142
|
+
"evidence": f"Acreage: {count} requests. Current cost: ${current_cost:.4f}.",
|
|
143
|
+
"estimated_savings": round(estimated_savings, 4),
|
|
144
|
+
"confidence": "HIGH"
|
|
145
|
+
})
|
|
146
|
+
|
|
147
|
+
# Strategy B: Duplicate prompt caching opportunities
|
|
148
|
+
# Look at prompts that repeated but missed cache (cache_hit = 0)
|
|
149
|
+
cursor.execute("""
|
|
150
|
+
SELECT
|
|
151
|
+
prompt_hash,
|
|
152
|
+
count(*) as occurrence_count,
|
|
153
|
+
sum(cost_actual) as wasted_cost
|
|
154
|
+
FROM telemetry
|
|
155
|
+
WHERE cache_hit = 0 AND success = 1
|
|
156
|
+
GROUP BY prompt_hash
|
|
157
|
+
HAVING occurrence_count > 1
|
|
158
|
+
ORDER BY occurrence_count DESC
|
|
159
|
+
LIMIT 5
|
|
160
|
+
""")
|
|
161
|
+
dup_rows = cursor.fetchall()
|
|
162
|
+
if dup_rows:
|
|
163
|
+
total_waste = sum(r["wasted_cost"] for r in dup_rows)
|
|
164
|
+
occurrences = sum(r["occurrence_count"] for r in dup_rows)
|
|
165
|
+
# Savings is the cost of duplicate runs (actual spend - 1st run spend)
|
|
166
|
+
estimated_savings = sum(r["wasted_cost"] * (r["occurrence_count"] - 1) / r["occurrence_count"] for r in dup_rows)
|
|
167
|
+
recommendations.append({
|
|
168
|
+
"strategy": "Duplicate Caching",
|
|
169
|
+
"title": "Enable caching for highly repetitive prompts",
|
|
170
|
+
"description": f"Identified {len(dup_rows)} distinct prompt hashes repeating {occurrences} times without caching. Enabling local SQLite caching will bypass API hits completely.",
|
|
171
|
+
"evidence": f"Total repeated runs waste: ${total_waste:.4f}.",
|
|
172
|
+
"estimated_savings": round(estimated_savings, 4),
|
|
173
|
+
"confidence": "VERY_HIGH"
|
|
174
|
+
})
|
|
175
|
+
|
|
176
|
+
# Strategy C: Retry Waste Analysis
|
|
177
|
+
# Look at costs accumulated due to failed request retries
|
|
178
|
+
cursor.execute("""
|
|
179
|
+
SELECT
|
|
180
|
+
count(*) as failed_count,
|
|
181
|
+
sum(cost_actual) as waste
|
|
182
|
+
FROM telemetry
|
|
183
|
+
WHERE success = 0 AND retry_count > 0
|
|
184
|
+
""")
|
|
185
|
+
retry_row = cursor.fetchone()
|
|
186
|
+
if retry_row and retry_row["failed_count"] > 0:
|
|
187
|
+
count = retry_row["failed_count"]
|
|
188
|
+
waste_cost = retry_row["waste"] or 0.0
|
|
189
|
+
recommendations.append({
|
|
190
|
+
"strategy": "Retry Optimization",
|
|
191
|
+
"title": "Reduce aggressive retry counts on server errors",
|
|
192
|
+
"description": f"Detected {count} request failures that incurred API charge penalties due to repetitive retry attempts.",
|
|
193
|
+
"evidence": f"Wasted retry budget: ${waste_cost:.4f}.",
|
|
194
|
+
"estimated_savings": round(waste_cost, 4),
|
|
195
|
+
"confidence": "MEDIUM"
|
|
196
|
+
})
|
|
197
|
+
|
|
198
|
+
return recommendations
|
|
199
|
+
except Exception as e:
|
|
200
|
+
raise HTTPException(status_code=500, detail=f"Failed building recommendations: {e}")
|
|
201
|
+
|
|
202
|
+
@router.get("/anomalies")
|
|
203
|
+
def get_anomalies(z_score: float = 2.0):
|
|
204
|
+
"""Retrieve cost anomaly analysis reports using Z-score statistics."""
|
|
205
|
+
detector = AnomalyDetector(_TELEMETRY_DB)
|
|
206
|
+
anomalies = detector.analyze_daily_cost_anomalies(z_threshold=z_score)
|
|
207
|
+
return anomalies
|
|
208
|
+
|
|
209
|
+
@router.post("/cache/clear")
|
|
210
|
+
def clear_cache():
|
|
211
|
+
"""Wipes the local cache database."""
|
|
212
|
+
try:
|
|
213
|
+
from costopt.cache import SQLiteCache
|
|
214
|
+
cache = SQLiteCache(_CACHE_DB)
|
|
215
|
+
cache.clear()
|
|
216
|
+
return {"status": "success", "message": "Cache database cleared successfully."}
|
|
217
|
+
except Exception as e:
|
|
218
|
+
raise HTTPException(status_code=500, detail=str(e))
|
|
219
|
+
|
|
220
|
+
@router.get("/telemetry/recent")
|
|
221
|
+
def get_recent_telemetry(limit: int = 10):
|
|
222
|
+
"""Returns the most recent logged telemetry transactions."""
|
|
223
|
+
try:
|
|
224
|
+
with _get_connection(_TELEMETRY_DB) as conn:
|
|
225
|
+
cursor = conn.cursor()
|
|
226
|
+
cursor.execute("""
|
|
227
|
+
SELECT
|
|
228
|
+
timestamp, request_id, model_requested, model_used,
|
|
229
|
+
input_tokens, output_tokens, latency_ms, success,
|
|
230
|
+
cache_hit, cost_original, cost_actual, savings, prompt_hash
|
|
231
|
+
FROM telemetry
|
|
232
|
+
ORDER BY timestamp DESC
|
|
233
|
+
LIMIT ?
|
|
234
|
+
""", (limit,))
|
|
235
|
+
rows = cursor.fetchall()
|
|
236
|
+
return [dict(r) for r in rows]
|
|
237
|
+
except Exception as e:
|
|
238
|
+
raise HTTPException(status_code=500, detail=str(e))
|
|
239
|
+
|
|
240
|
+
@router.post("/telemetry/generate")
|
|
241
|
+
def inject_simulation_data():
|
|
242
|
+
"""Injects 150 mock transactions into the telemetry database."""
|
|
243
|
+
try:
|
|
244
|
+
from costopt.generator import generate_telemetry_dataset
|
|
245
|
+
from costopt.telemetry import SQLiteTelemetryLogger
|
|
246
|
+
events = generate_telemetry_dataset(records_count=150, days=15)
|
|
247
|
+
logger = SQLiteTelemetryLogger(db_path=_TELEMETRY_DB)
|
|
248
|
+
logger._flush_batch(events)
|
|
249
|
+
logger.shutdown()
|
|
250
|
+
return {"status": "success", "message": "Injected 150 mock logs into database."}
|
|
251
|
+
except Exception as e:
|
|
252
|
+
raise HTTPException(status_code=500, detail=str(e))
|
|
253
|
+
|
|
254
|
+
@router.get("/config")
|
|
255
|
+
def get_active_config():
|
|
256
|
+
"""Returns the current costopt.yaml routing rules template."""
|
|
257
|
+
import os
|
|
258
|
+
try:
|
|
259
|
+
config_path = "costopt.yaml"
|
|
260
|
+
if os.path.exists(config_path):
|
|
261
|
+
with open(config_path, "r", encoding="utf-8") as f:
|
|
262
|
+
content = f.read()
|
|
263
|
+
return {"status": "success", "content": content}
|
|
264
|
+
return {"status": "error", "message": "costopt.yaml config file not found."}
|
|
265
|
+
except Exception as e:
|
|
266
|
+
raise HTTPException(status_code=500, detail=str(e))
|
|
267
|
+
|
|
268
|
+
from pydantic import BaseModel
|
|
269
|
+
class SimulationRequest(BaseModel):
|
|
270
|
+
prompt: str
|
|
271
|
+
model: str
|
|
272
|
+
|
|
273
|
+
@router.post("/simulate")
|
|
274
|
+
def simulate_sdk_call(req: SimulationRequest):
|
|
275
|
+
"""Simulates the SDK cache checks and routing rules for a prompt."""
|
|
276
|
+
import time
|
|
277
|
+
logs = []
|
|
278
|
+
logs.append(f"Intercepting ChatCompletion call for requested model='{req.model}'")
|
|
279
|
+
|
|
280
|
+
# Check Cache
|
|
281
|
+
from costopt.cache import SQLiteCache
|
|
282
|
+
cache = SQLiteCache(_CACHE_DB)
|
|
283
|
+
logs.append("Step 1: Computing MD5 hash and checking SQLiteCache...")
|
|
284
|
+
cached_response = cache.get(req.prompt, req.model)
|
|
285
|
+
|
|
286
|
+
from costopt.pricing import calculate_cost
|
|
287
|
+
|
|
288
|
+
if cached_response:
|
|
289
|
+
logs.append("Cache HIT! Serving response from local SQLite cache database.")
|
|
290
|
+
cost_orig = calculate_cost("openai", req.model, 45, 15)
|
|
291
|
+
cost_act = calculate_cost("openai", req.model, 45, 15, cache_hit=True)
|
|
292
|
+
savings = cost_orig - cost_act
|
|
293
|
+
return {
|
|
294
|
+
"status": "success",
|
|
295
|
+
"cache_hit": True,
|
|
296
|
+
"routed": False,
|
|
297
|
+
"original_model": req.model,
|
|
298
|
+
"final_model": req.model,
|
|
299
|
+
"cost_original": round(cost_orig, 6),
|
|
300
|
+
"cost_actual": round(cost_act, 6),
|
|
301
|
+
"savings": round(savings, 6),
|
|
302
|
+
"logs": logs
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
logs.append("Cache MISS. Evaluating routing rules...")
|
|
306
|
+
|
|
307
|
+
# Check Router
|
|
308
|
+
from costopt.router import CostOptRouter
|
|
309
|
+
router_engine = CostOptRouter()
|
|
310
|
+
target_model = router_engine.match_route(req.prompt, req.model)
|
|
311
|
+
|
|
312
|
+
routed = False
|
|
313
|
+
if target_model != req.model:
|
|
314
|
+
logs.append(f"Routing rule match! Redirecting query to cheaper model: '{target_model}'")
|
|
315
|
+
routed = True
|
|
316
|
+
else:
|
|
317
|
+
logs.append(f"No routing rules matched. Executing completion using requested model: '{req.model}'")
|
|
318
|
+
|
|
319
|
+
# Simulated token count cost calculation
|
|
320
|
+
cost_orig = calculate_cost("openai", req.model, 150, 60)
|
|
321
|
+
cost_act = calculate_cost("openai", target_model, 150, 60)
|
|
322
|
+
savings = cost_orig - cost_act
|
|
323
|
+
|
|
324
|
+
return {
|
|
325
|
+
"status": "success",
|
|
326
|
+
"cache_hit": False,
|
|
327
|
+
"routed": routed,
|
|
328
|
+
"original_model": req.model,
|
|
329
|
+
"final_model": target_model,
|
|
330
|
+
"cost_original": round(cost_orig, 6),
|
|
331
|
+
"cost_actual": round(cost_act, 6),
|
|
332
|
+
"savings": round(savings, 6),
|
|
333
|
+
"logs": logs
|
|
334
|
+
}
|
|
335
|
+
|
|
336
|
+
@router.get("/models")
|
|
337
|
+
def get_supported_models():
|
|
338
|
+
"""Returns a list of all model names loaded in the pricing system."""
|
|
339
|
+
try:
|
|
340
|
+
from costopt.pricing import get_all_loaded_models
|
|
341
|
+
raw_data = get_all_loaded_models()
|
|
342
|
+
models = []
|
|
343
|
+
for provider, model_list in raw_data.items():
|
|
344
|
+
for m in model_list:
|
|
345
|
+
models.append({
|
|
346
|
+
"provider": provider,
|
|
347
|
+
"model": m,
|
|
348
|
+
"label": f"{m} ({provider})"
|
|
349
|
+
})
|
|
350
|
+
return models
|
|
351
|
+
except Exception as e:
|
|
352
|
+
raise HTTPException(status_code=500, detail=str(e))
|
|
353
|
+
|
|
354
|
+
|
costopt/api/server.py
ADDED
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import uvicorn
|
|
3
|
+
from fastapi import FastAPI
|
|
4
|
+
from fastapi.staticfiles import StaticFiles
|
|
5
|
+
from fastapi.responses import FileResponse
|
|
6
|
+
from fastapi.middleware.cors import CORSMiddleware
|
|
7
|
+
|
|
8
|
+
from costopt.api.routes import router, set_db_paths
|
|
9
|
+
|
|
10
|
+
app = FastAPI(
|
|
11
|
+
title="LLM CostOpt API",
|
|
12
|
+
description="Cost optimization API serving the developer dashboard",
|
|
13
|
+
version="0.1.0"
|
|
14
|
+
)
|
|
15
|
+
|
|
16
|
+
# Enable CORS for local cross-origin dashboard testing
|
|
17
|
+
app.add_middleware(
|
|
18
|
+
CORSMiddleware,
|
|
19
|
+
allow_origins=["http://localhost:8000", "http://127.0.0.1:8000"],
|
|
20
|
+
allow_credentials=True,
|
|
21
|
+
allow_methods=["*"],
|
|
22
|
+
allow_headers=["*"],
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
# Attach routes
|
|
26
|
+
app.include_router(router, prefix="/api")
|
|
27
|
+
|
|
28
|
+
# Serve dashboard static folder
|
|
29
|
+
DASHBOARD_DIR = os.path.abspath(
|
|
30
|
+
os.path.join(os.path.dirname(__file__), "..", "..", "..", "dashboard")
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
if os.path.exists(DASHBOARD_DIR):
|
|
34
|
+
app.mount("/static", StaticFiles(directory=DASHBOARD_DIR), name="static")
|
|
35
|
+
|
|
36
|
+
@app.get("/")
|
|
37
|
+
def read_index():
|
|
38
|
+
return FileResponse(os.path.join(DASHBOARD_DIR, "index.html"))
|
|
39
|
+
else:
|
|
40
|
+
@app.get("/")
|
|
41
|
+
def read_root():
|
|
42
|
+
return {"status": "running", "message": "Dashboard assets directory not found. API endpoints are operational."}
|
|
43
|
+
|
|
44
|
+
def start_server(
|
|
45
|
+
host: str = "127.0.0.1",
|
|
46
|
+
port: int = 8000,
|
|
47
|
+
telemetry_db: str = "costopt_telemetry.db",
|
|
48
|
+
cache_db: str = "costopt_cache.db"
|
|
49
|
+
) -> None:
|
|
50
|
+
"""Configures database connections and launches the API/Dashboard server."""
|
|
51
|
+
set_db_paths(telemetry_db, cache_db)
|
|
52
|
+
print(f"Starting LLM CostOpt Dashboard on http://{host}:{port}")
|
|
53
|
+
uvicorn.run(app, host=host, port=port)
|
|
54
|
+
|
|
55
|
+
if __name__ == "__main__":
|
|
56
|
+
start_server()
|