costopt 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
costopt/__init__.py ADDED
@@ -0,0 +1,6 @@
1
+ from costopt.client import CostOpt
2
+ from costopt.pricing import calculate_cost, get_pricing
3
+ from costopt.router import CostOptRouter
4
+ from costopt.cache import SQLiteCache
5
+
6
+ __all__ = ["CostOpt", "calculate_cost", "get_pricing", "CostOptRouter", "SQLiteCache"]
costopt/anomaly.py ADDED
@@ -0,0 +1,110 @@
1
+ import sqlite3
2
+ import math
3
+ import logging
4
+ from typing import List, Dict, Any, Tuple
5
+
6
+ logger = logging.getLogger("costopt.anomaly")
7
+
8
+ class AnomalyDetector:
9
+ def __init__(self, telemetry_db_path: str = "costopt_telemetry.db"):
10
+ self.db_path = telemetry_db_path
11
+
12
+ def analyze_daily_cost_anomalies(self, z_threshold: float = 2.0, lookback_days: int = 30) -> List[Dict[str, Any]]:
13
+ """
14
+ Retrieves daily costs, computes rolling Z-scores, and flags anomalies.
15
+ Returns a list of flagged daily anomaly reports.
16
+ """
17
+ try:
18
+ with sqlite3.connect(self.db_path) as conn:
19
+ conn.row_factory = sqlite3.Row
20
+ cursor = conn.cursor()
21
+
22
+ # Aggregate total cost by day (YYYY-MM-DD)
23
+ cursor.execute("""
24
+ SELECT
25
+ strftime('%Y-%m-%d', timestamp) as date,
26
+ SUM(cost_actual) as total_cost,
27
+ COUNT(*) as request_count
28
+ FROM telemetry
29
+ GROUP BY date
30
+ ORDER BY date ASC
31
+ LIMIT ?
32
+ """, (lookback_days,))
33
+
34
+ rows = cursor.fetchall()
35
+ if len(rows) < 3:
36
+ # Not enough historical baseline data points to calculate variance/stddev
37
+ logger.warning("Insufficient history points in telemetry database to run anomaly analysis.")
38
+ return []
39
+
40
+ dates = [r["date"] for r in rows]
41
+ costs = [float(r["total_cost"]) for r in rows]
42
+ counts = [int(r["request_count"]) for r in rows]
43
+
44
+ anomalies = []
45
+ n = len(costs)
46
+
47
+ # Compute rolling parameters
48
+ for i in range(2, n):
49
+ current_cost = costs[i]
50
+ current_date = dates[i]
51
+ current_count = counts[i]
52
+
53
+ # History slice up to current index (excluding current point to establish clean baseline)
54
+ history = costs[:i]
55
+ mean = sum(history) / len(history)
56
+
57
+ # Compute standard deviation
58
+ variance = sum((x - mean) ** 2 for x in history) / len(history)
59
+ stddev = math.sqrt(variance)
60
+
61
+ if stddev == 0.0:
62
+ z_score = 0.0
63
+ else:
64
+ z_score = (current_cost - mean) / stddev
65
+
66
+ # Flag anomaly if Z-score exceeds threshold
67
+ if z_score > z_threshold:
68
+ anomalies.append({
69
+ "date": current_date,
70
+ "actual_cost": round(current_cost, 2),
71
+ "expected_mean": round(mean, 2),
72
+ "stddev": round(stddev, 2),
73
+ "z_score": round(z_score, 2),
74
+ "request_count": current_count,
75
+ "severity": "CRITICAL" if z_score > 3.5 else "WARNING"
76
+ })
77
+
78
+ return anomalies
79
+ except Exception as e:
80
+ logger.error(f"Error running cost anomaly detection: {e}")
81
+ return []
82
+
83
+ def get_highest_impact_cost_drivers(self, limit: int = 5) -> List[Dict[str, Any]]:
84
+ """
85
+ Queries telemetry database to locate applications or models driving the highest spend.
86
+ Useful evidence generation for FinOps recommendations.
87
+ """
88
+ try:
89
+ with sqlite3.connect(self.db_path) as conn:
90
+ conn.row_factory = sqlite3.Row
91
+ cursor = conn.cursor()
92
+
93
+ # Spend by model
94
+ cursor.execute("""
95
+ SELECT
96
+ model_used,
97
+ provider,
98
+ SUM(cost_actual) as total_spend,
99
+ SUM(savings) as saved_amount,
100
+ COUNT(*) as request_count
101
+ FROM telemetry
102
+ GROUP BY model_used, provider
103
+ ORDER BY total_spend DESC
104
+ LIMIT ?
105
+ """, (limit,))
106
+
107
+ return [dict(row) for row in cursor.fetchall()]
108
+ except Exception as e:
109
+ logger.error(f"Error retrieving cost drivers: {e}")
110
+ return []
costopt/api/routes.py ADDED
@@ -0,0 +1,354 @@
1
+ import sqlite3
2
+ from fastapi import APIRouter, HTTPException, Query
3
+ from typing import Dict, Any, List, Optional
4
+ from costopt.anomaly import AnomalyDetector
5
+
6
+ router = APIRouter()
7
+
8
+ # Global database paths that can be configured by server init
9
+ _TELEMETRY_DB = "costopt_telemetry.db"
10
+ _CACHE_DB = "costopt_cache.db"
11
+
12
+ def set_db_paths(telemetry_db: str, cache_db: str):
13
+ global _TELEMETRY_DB, _CACHE_DB
14
+ _TELEMETRY_DB = telemetry_db
15
+ _CACHE_DB = cache_db
16
+
17
+ def _get_connection(db_path: str):
18
+ conn = sqlite3.connect(db_path)
19
+ conn.row_factory = sqlite3.Row
20
+ return conn
21
+
22
+ @router.get("/overview")
23
+ def get_overview(env: Optional[str] = None):
24
+ """Retrieve key aggregate cost and efficiency metrics from logs."""
25
+ try:
26
+ with _get_connection(_TELEMETRY_DB) as conn:
27
+ cursor = conn.cursor()
28
+
29
+ # Filters
30
+ query = "SELECT count(*), sum(cost_original), sum(cost_actual), sum(savings), sum(cache_hit), sum(case when success = 0 then 1 else 0 end) FROM telemetry"
31
+ params = []
32
+ if env:
33
+ query += " WHERE environment = ?"
34
+ params.append(env)
35
+
36
+ cursor.execute(query, params)
37
+ row = cursor.fetchone()
38
+
39
+ count = row[0] or 0
40
+ original = row[1] or 0.0
41
+ actual = row[2] or 0.0
42
+ savings = row[3] or 0.0
43
+ cache_hits = row[4] or 0
44
+ failures = row[5] or 0
45
+
46
+ cache_rate = (cache_hits / count * 100) if count > 0 else 0.0
47
+ error_rate = (failures / count * 100) if count > 0 else 0.0
48
+
49
+ return {
50
+ "total_requests": count,
51
+ "cost_baseline": round(original, 4),
52
+ "cost_actual": round(actual, 4),
53
+ "total_savings": round(savings, 4),
54
+ "cache_hit_rate": round(cache_rate, 2),
55
+ "error_rate": round(error_rate, 2)
56
+ }
57
+ except Exception as e:
58
+ raise HTTPException(status_code=500, detail=f"Database aggregation failed: {e}")
59
+
60
+ @router.get("/charts/savings")
61
+ def get_savings_chart_data(limit: int = 30):
62
+ """Returns daily cost, baseline cost, and savings metrics for time series plots."""
63
+ try:
64
+ with _get_connection(_TELEMETRY_DB) as conn:
65
+ cursor = conn.cursor()
66
+ cursor.execute("""
67
+ SELECT
68
+ strftime('%Y-%m-%d', timestamp) as date,
69
+ SUM(cost_original) as baseline_cost,
70
+ SUM(cost_actual) as actual_cost,
71
+ SUM(savings) as savings
72
+ FROM telemetry
73
+ GROUP BY date
74
+ ORDER BY date ASC
75
+ LIMIT ?
76
+ """, (limit,))
77
+ rows = cursor.fetchall()
78
+
79
+ return [
80
+ {
81
+ "date": r["date"],
82
+ "baseline_cost": round(r["baseline_cost"], 4),
83
+ "actual_cost": round(r["actual_cost"], 4),
84
+ "savings": round(r["savings"], 4)
85
+ } for r in rows
86
+ ]
87
+ except Exception as e:
88
+ raise HTTPException(status_code=500, detail=str(e))
89
+
90
+ @router.get("/charts/models")
91
+ def get_model_distribution():
92
+ """Returns total request counts and actual cost grouped by provider/model."""
93
+ try:
94
+ with _get_connection(_TELEMETRY_DB) as conn:
95
+ cursor = conn.cursor()
96
+ cursor.execute("""
97
+ SELECT
98
+ provider,
99
+ model_used,
100
+ COUNT(*) as count,
101
+ SUM(cost_actual) as spend
102
+ FROM telemetry
103
+ GROUP BY provider, model_used
104
+ ORDER BY spend DESC
105
+ """)
106
+ rows = cursor.fetchall()
107
+ return [dict(r) for r in rows]
108
+ except Exception as e:
109
+ raise HTTPException(status_code=500, detail=str(e))
110
+
111
+ @router.get("/recommendations")
112
+ def get_recommendations():
113
+ """Generates concrete, actionable FinOps optimization suggestions backed by database evidence."""
114
+ recommendations = []
115
+
116
+ try:
117
+ with _get_connection(_TELEMETRY_DB) as conn:
118
+ cursor = conn.cursor()
119
+
120
+ # Strategy A: Model Routing Check
121
+ # Look at simple prompts (short length, containing key intent keywords) run on expensive models
122
+ cursor.execute("""
123
+ SELECT
124
+ count(*) as count,
125
+ sum(cost_actual) as cost
126
+ FROM telemetry
127
+ WHERE model_requested IN ('gpt-4', 'claude-3-opus')
128
+ AND length(prompt_hash) > 0 -- just representing prompt presence
129
+ -- simple requests heuristic
130
+ AND (input_tokens < 200)
131
+ """)
132
+ routing_row = cursor.fetchone()
133
+ if routing_row and routing_row["count"] > 0:
134
+ count = routing_row["count"]
135
+ current_cost = routing_row["cost"] or 0.0
136
+ # Let's project saving if routed to mini/haiku (saving ~90% cost)
137
+ estimated_savings = current_cost * 0.90
138
+ recommendations.append({
139
+ "strategy": "Model Routing",
140
+ "title": f"Route short requests from premium models to mini equivalents",
141
+ "description": f"Detected {count} high-cost premium requests with low input token counts (< 200). Routing these to gpt-4o-mini / claude-3-haiku can cut token costs by 90%.",
142
+ "evidence": f"Acreage: {count} requests. Current cost: ${current_cost:.4f}.",
143
+ "estimated_savings": round(estimated_savings, 4),
144
+ "confidence": "HIGH"
145
+ })
146
+
147
+ # Strategy B: Duplicate prompt caching opportunities
148
+ # Look at prompts that repeated but missed cache (cache_hit = 0)
149
+ cursor.execute("""
150
+ SELECT
151
+ prompt_hash,
152
+ count(*) as occurrence_count,
153
+ sum(cost_actual) as wasted_cost
154
+ FROM telemetry
155
+ WHERE cache_hit = 0 AND success = 1
156
+ GROUP BY prompt_hash
157
+ HAVING occurrence_count > 1
158
+ ORDER BY occurrence_count DESC
159
+ LIMIT 5
160
+ """)
161
+ dup_rows = cursor.fetchall()
162
+ if dup_rows:
163
+ total_waste = sum(r["wasted_cost"] for r in dup_rows)
164
+ occurrences = sum(r["occurrence_count"] for r in dup_rows)
165
+ # Savings is the cost of duplicate runs (actual spend - 1st run spend)
166
+ estimated_savings = sum(r["wasted_cost"] * (r["occurrence_count"] - 1) / r["occurrence_count"] for r in dup_rows)
167
+ recommendations.append({
168
+ "strategy": "Duplicate Caching",
169
+ "title": "Enable caching for highly repetitive prompts",
170
+ "description": f"Identified {len(dup_rows)} distinct prompt hashes repeating {occurrences} times without caching. Enabling local SQLite caching will bypass API hits completely.",
171
+ "evidence": f"Total repeated runs waste: ${total_waste:.4f}.",
172
+ "estimated_savings": round(estimated_savings, 4),
173
+ "confidence": "VERY_HIGH"
174
+ })
175
+
176
+ # Strategy C: Retry Waste Analysis
177
+ # Look at costs accumulated due to failed request retries
178
+ cursor.execute("""
179
+ SELECT
180
+ count(*) as failed_count,
181
+ sum(cost_actual) as waste
182
+ FROM telemetry
183
+ WHERE success = 0 AND retry_count > 0
184
+ """)
185
+ retry_row = cursor.fetchone()
186
+ if retry_row and retry_row["failed_count"] > 0:
187
+ count = retry_row["failed_count"]
188
+ waste_cost = retry_row["waste"] or 0.0
189
+ recommendations.append({
190
+ "strategy": "Retry Optimization",
191
+ "title": "Reduce aggressive retry counts on server errors",
192
+ "description": f"Detected {count} request failures that incurred API charge penalties due to repetitive retry attempts.",
193
+ "evidence": f"Wasted retry budget: ${waste_cost:.4f}.",
194
+ "estimated_savings": round(waste_cost, 4),
195
+ "confidence": "MEDIUM"
196
+ })
197
+
198
+ return recommendations
199
+ except Exception as e:
200
+ raise HTTPException(status_code=500, detail=f"Failed building recommendations: {e}")
201
+
202
+ @router.get("/anomalies")
203
+ def get_anomalies(z_score: float = 2.0):
204
+ """Retrieve cost anomaly analysis reports using Z-score statistics."""
205
+ detector = AnomalyDetector(_TELEMETRY_DB)
206
+ anomalies = detector.analyze_daily_cost_anomalies(z_threshold=z_score)
207
+ return anomalies
208
+
209
+ @router.post("/cache/clear")
210
+ def clear_cache():
211
+ """Wipes the local cache database."""
212
+ try:
213
+ from costopt.cache import SQLiteCache
214
+ cache = SQLiteCache(_CACHE_DB)
215
+ cache.clear()
216
+ return {"status": "success", "message": "Cache database cleared successfully."}
217
+ except Exception as e:
218
+ raise HTTPException(status_code=500, detail=str(e))
219
+
220
+ @router.get("/telemetry/recent")
221
+ def get_recent_telemetry(limit: int = 10):
222
+ """Returns the most recent logged telemetry transactions."""
223
+ try:
224
+ with _get_connection(_TELEMETRY_DB) as conn:
225
+ cursor = conn.cursor()
226
+ cursor.execute("""
227
+ SELECT
228
+ timestamp, request_id, model_requested, model_used,
229
+ input_tokens, output_tokens, latency_ms, success,
230
+ cache_hit, cost_original, cost_actual, savings, prompt_hash
231
+ FROM telemetry
232
+ ORDER BY timestamp DESC
233
+ LIMIT ?
234
+ """, (limit,))
235
+ rows = cursor.fetchall()
236
+ return [dict(r) for r in rows]
237
+ except Exception as e:
238
+ raise HTTPException(status_code=500, detail=str(e))
239
+
240
+ @router.post("/telemetry/generate")
241
+ def inject_simulation_data():
242
+ """Injects 150 mock transactions into the telemetry database."""
243
+ try:
244
+ from costopt.generator import generate_telemetry_dataset
245
+ from costopt.telemetry import SQLiteTelemetryLogger
246
+ events = generate_telemetry_dataset(records_count=150, days=15)
247
+ logger = SQLiteTelemetryLogger(db_path=_TELEMETRY_DB)
248
+ logger._flush_batch(events)
249
+ logger.shutdown()
250
+ return {"status": "success", "message": "Injected 150 mock logs into database."}
251
+ except Exception as e:
252
+ raise HTTPException(status_code=500, detail=str(e))
253
+
254
+ @router.get("/config")
255
+ def get_active_config():
256
+ """Returns the current costopt.yaml routing rules template."""
257
+ import os
258
+ try:
259
+ config_path = "costopt.yaml"
260
+ if os.path.exists(config_path):
261
+ with open(config_path, "r", encoding="utf-8") as f:
262
+ content = f.read()
263
+ return {"status": "success", "content": content}
264
+ return {"status": "error", "message": "costopt.yaml config file not found."}
265
+ except Exception as e:
266
+ raise HTTPException(status_code=500, detail=str(e))
267
+
268
+ from pydantic import BaseModel
269
+ class SimulationRequest(BaseModel):
270
+ prompt: str
271
+ model: str
272
+
273
+ @router.post("/simulate")
274
+ def simulate_sdk_call(req: SimulationRequest):
275
+ """Simulates the SDK cache checks and routing rules for a prompt."""
276
+ import time
277
+ logs = []
278
+ logs.append(f"Intercepting ChatCompletion call for requested model='{req.model}'")
279
+
280
+ # Check Cache
281
+ from costopt.cache import SQLiteCache
282
+ cache = SQLiteCache(_CACHE_DB)
283
+ logs.append("Step 1: Computing MD5 hash and checking SQLiteCache...")
284
+ cached_response = cache.get(req.prompt, req.model)
285
+
286
+ from costopt.pricing import calculate_cost
287
+
288
+ if cached_response:
289
+ logs.append("Cache HIT! Serving response from local SQLite cache database.")
290
+ cost_orig = calculate_cost("openai", req.model, 45, 15)
291
+ cost_act = calculate_cost("openai", req.model, 45, 15, cache_hit=True)
292
+ savings = cost_orig - cost_act
293
+ return {
294
+ "status": "success",
295
+ "cache_hit": True,
296
+ "routed": False,
297
+ "original_model": req.model,
298
+ "final_model": req.model,
299
+ "cost_original": round(cost_orig, 6),
300
+ "cost_actual": round(cost_act, 6),
301
+ "savings": round(savings, 6),
302
+ "logs": logs
303
+ }
304
+
305
+ logs.append("Cache MISS. Evaluating routing rules...")
306
+
307
+ # Check Router
308
+ from costopt.router import CostOptRouter
309
+ router_engine = CostOptRouter()
310
+ target_model = router_engine.match_route(req.prompt, req.model)
311
+
312
+ routed = False
313
+ if target_model != req.model:
314
+ logs.append(f"Routing rule match! Redirecting query to cheaper model: '{target_model}'")
315
+ routed = True
316
+ else:
317
+ logs.append(f"No routing rules matched. Executing completion using requested model: '{req.model}'")
318
+
319
+ # Simulated token count cost calculation
320
+ cost_orig = calculate_cost("openai", req.model, 150, 60)
321
+ cost_act = calculate_cost("openai", target_model, 150, 60)
322
+ savings = cost_orig - cost_act
323
+
324
+ return {
325
+ "status": "success",
326
+ "cache_hit": False,
327
+ "routed": routed,
328
+ "original_model": req.model,
329
+ "final_model": target_model,
330
+ "cost_original": round(cost_orig, 6),
331
+ "cost_actual": round(cost_act, 6),
332
+ "savings": round(savings, 6),
333
+ "logs": logs
334
+ }
335
+
336
+ @router.get("/models")
337
+ def get_supported_models():
338
+ """Returns a list of all model names loaded in the pricing system."""
339
+ try:
340
+ from costopt.pricing import get_all_loaded_models
341
+ raw_data = get_all_loaded_models()
342
+ models = []
343
+ for provider, model_list in raw_data.items():
344
+ for m in model_list:
345
+ models.append({
346
+ "provider": provider,
347
+ "model": m,
348
+ "label": f"{m} ({provider})"
349
+ })
350
+ return models
351
+ except Exception as e:
352
+ raise HTTPException(status_code=500, detail=str(e))
353
+
354
+
costopt/api/server.py ADDED
@@ -0,0 +1,56 @@
1
+ import os
2
+ import uvicorn
3
+ from fastapi import FastAPI
4
+ from fastapi.staticfiles import StaticFiles
5
+ from fastapi.responses import FileResponse
6
+ from fastapi.middleware.cors import CORSMiddleware
7
+
8
+ from costopt.api.routes import router, set_db_paths
9
+
10
+ app = FastAPI(
11
+ title="LLM CostOpt API",
12
+ description="Cost optimization API serving the developer dashboard",
13
+ version="0.1.0"
14
+ )
15
+
16
+ # Enable CORS for local cross-origin dashboard testing
17
+ app.add_middleware(
18
+ CORSMiddleware,
19
+ allow_origins=["http://localhost:8000", "http://127.0.0.1:8000"],
20
+ allow_credentials=True,
21
+ allow_methods=["*"],
22
+ allow_headers=["*"],
23
+ )
24
+
25
+ # Attach routes
26
+ app.include_router(router, prefix="/api")
27
+
28
+ # Serve dashboard static folder
29
+ DASHBOARD_DIR = os.path.abspath(
30
+ os.path.join(os.path.dirname(__file__), "..", "..", "..", "dashboard")
31
+ )
32
+
33
+ if os.path.exists(DASHBOARD_DIR):
34
+ app.mount("/static", StaticFiles(directory=DASHBOARD_DIR), name="static")
35
+
36
+ @app.get("/")
37
+ def read_index():
38
+ return FileResponse(os.path.join(DASHBOARD_DIR, "index.html"))
39
+ else:
40
+ @app.get("/")
41
+ def read_root():
42
+ return {"status": "running", "message": "Dashboard assets directory not found. API endpoints are operational."}
43
+
44
+ def start_server(
45
+ host: str = "127.0.0.1",
46
+ port: int = 8000,
47
+ telemetry_db: str = "costopt_telemetry.db",
48
+ cache_db: str = "costopt_cache.db"
49
+ ) -> None:
50
+ """Configures database connections and launches the API/Dashboard server."""
51
+ set_db_paths(telemetry_db, cache_db)
52
+ print(f"Starting LLM CostOpt Dashboard on http://{host}:{port}")
53
+ uvicorn.run(app, host=host, port=port)
54
+
55
+ if __name__ == "__main__":
56
+ start_server()