api-cost-tracker 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- api_cost_tracker-0.2.0/.gitignore +43 -0
- api_cost_tracker-0.2.0/PKG-INFO +5 -0
- api_cost_tracker-0.2.0/pyproject.toml +16 -0
- api_cost_tracker-0.2.0/src/api_cost_tracker/__init__.py +18 -0
- api_cost_tracker-0.2.0/src/api_cost_tracker/buffer.py +174 -0
- api_cost_tracker-0.2.0/src/api_cost_tracker/tracker.py +345 -0
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
# Dependencies
|
|
2
|
+
node_modules/
|
|
3
|
+
yarn.lock
|
|
4
|
+
package-lock.json
|
|
5
|
+
|
|
6
|
+
# Build output
|
|
7
|
+
website/dist/
|
|
8
|
+
website/.vercel/
|
|
9
|
+
|
|
10
|
+
# Scaffolding
|
|
11
|
+
.agent/
|
|
12
|
+
.agentsync/
|
|
13
|
+
.scaffolding-version
|
|
14
|
+
|
|
15
|
+
# IDE
|
|
16
|
+
.vscode/
|
|
17
|
+
.idea/
|
|
18
|
+
*.swp
|
|
19
|
+
*.swo
|
|
20
|
+
|
|
21
|
+
# Python
|
|
22
|
+
__pycache__/
|
|
23
|
+
*.pyc
|
|
24
|
+
*.db
|
|
25
|
+
*.db-shm
|
|
26
|
+
*.db-wal
|
|
27
|
+
|
|
28
|
+
# Generated indexes
|
|
29
|
+
00_Index_*.md
|
|
30
|
+
|
|
31
|
+
# Auto-generated session stub (recreated by /cleanup as needed)
|
|
32
|
+
PROGRESS.md
|
|
33
|
+
|
|
34
|
+
# uv
|
|
35
|
+
uv.lock
|
|
36
|
+
|
|
37
|
+
# OS
|
|
38
|
+
.DS_Store
|
|
39
|
+
Thumbs.db
|
|
40
|
+
.vercel
|
|
41
|
+
.env*.local
|
|
42
|
+
.scratch/
|
|
43
|
+
/.claude/
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "api-cost-tracker"
|
|
7
|
+
version = "0.2.0"
|
|
8
|
+
description = "Lightweight client for the Synth Insight Labs API cost tracker"
|
|
9
|
+
requires-python = ">=3.11"
|
|
10
|
+
dependencies = []
|
|
11
|
+
|
|
12
|
+
[tool.hatch.build.targets.wheel]
|
|
13
|
+
packages = ["src/api_cost_tracker"]
|
|
14
|
+
|
|
15
|
+
[tool.pytest.ini_options]
|
|
16
|
+
testpaths = ["tests"]
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""
|
|
2
|
+
API Cost Tracker Client — track API spend across all providers and services.
|
|
3
|
+
|
|
4
|
+
Usage:
|
|
5
|
+
from api_cost_tracker import track, log_call
|
|
6
|
+
|
|
7
|
+
# AI API response tracking
|
|
8
|
+
resp = client.messages.create(model="claude-haiku-4-5", ...)
|
|
9
|
+
track(resp, "anthropic", project="ai-memory", caller="classifier")
|
|
10
|
+
|
|
11
|
+
# Non-AI API call logging
|
|
12
|
+
log_call("vercel", service="blob", project="muffinpanrecipes")
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from .tracker import track, track_tokens, log_call
|
|
16
|
+
from .buffer import flush_buffer, pending_count
|
|
17
|
+
|
|
18
|
+
__all__ = ["track", "track_tokens", "log_call", "flush_buffer", "pending_count"]
|
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Local SQLite buffer for offline usage tracking.
|
|
3
|
+
|
|
4
|
+
When the hosted endpoint is unreachable, usage records are buffered here.
|
|
5
|
+
Call flush_buffer() to retry sending buffered records when connectivity returns.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import json
|
|
9
|
+
import sqlite3
|
|
10
|
+
from datetime import datetime
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Optional
|
|
13
|
+
from urllib.error import URLError
|
|
14
|
+
from urllib.request import Request, urlopen
|
|
15
|
+
|
|
16
|
+
# Deliberately still the pre-#7830 name: 0.1.x clients that are not upgraded yet
|
|
17
|
+
# keep writing this file, and sharing it is the only way neither version strands
|
|
18
|
+
# rows the other wrote. The schema is unchanged.
|
|
19
|
+
BUFFER_DB_PATH = Path.home() / ".local" / "share" / "api_trust_tracker" / "buffer.db"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _get_buffer_connection(db_path: Optional[Path] = None) -> sqlite3.Connection:
|
|
23
|
+
"""Create a connection to the buffer database."""
|
|
24
|
+
path = db_path or BUFFER_DB_PATH
|
|
25
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
26
|
+
conn = sqlite3.connect(str(path), timeout=5)
|
|
27
|
+
conn.row_factory = sqlite3.Row
|
|
28
|
+
conn.execute("PRAGMA journal_mode=WAL")
|
|
29
|
+
conn.execute("PRAGMA busy_timeout=5000")
|
|
30
|
+
conn.executescript("""
|
|
31
|
+
CREATE TABLE IF NOT EXISTS pending_records (
|
|
32
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
33
|
+
endpoint TEXT NOT NULL,
|
|
34
|
+
payload TEXT NOT NULL,
|
|
35
|
+
created_at TEXT NOT NULL,
|
|
36
|
+
attempts INTEGER DEFAULT 0,
|
|
37
|
+
last_attempt TEXT
|
|
38
|
+
);
|
|
39
|
+
""")
|
|
40
|
+
return conn
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def buffer_record(
|
|
44
|
+
endpoint: str,
|
|
45
|
+
payload: dict,
|
|
46
|
+
db_path: Optional[Path] = None,
|
|
47
|
+
) -> int:
|
|
48
|
+
"""
|
|
49
|
+
Store a failed POST payload for later retry.
|
|
50
|
+
|
|
51
|
+
Args:
|
|
52
|
+
endpoint: The relative endpoint path (e.g., "/track" or "/log_call").
|
|
53
|
+
payload: The JSON payload that failed to send.
|
|
54
|
+
db_path: Override buffer DB path for testing.
|
|
55
|
+
|
|
56
|
+
Returns:
|
|
57
|
+
The buffered record's ID.
|
|
58
|
+
"""
|
|
59
|
+
conn = _get_buffer_connection(db_path)
|
|
60
|
+
try:
|
|
61
|
+
cursor = conn.execute(
|
|
62
|
+
"""INSERT INTO pending_records (endpoint, payload, created_at)
|
|
63
|
+
VALUES (?, ?, ?)""",
|
|
64
|
+
(endpoint, json.dumps(payload), datetime.now().isoformat()),
|
|
65
|
+
)
|
|
66
|
+
conn.commit()
|
|
67
|
+
return cursor.lastrowid
|
|
68
|
+
finally:
|
|
69
|
+
conn.close()
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def flush_buffer(
|
|
73
|
+
base_url: str,
|
|
74
|
+
api_key: Optional[str] = None,
|
|
75
|
+
db_path: Optional[Path] = None,
|
|
76
|
+
max_batch: int = 100,
|
|
77
|
+
) -> dict:
|
|
78
|
+
"""
|
|
79
|
+
Retry sending buffered records to the hosted endpoint.
|
|
80
|
+
|
|
81
|
+
Args:
|
|
82
|
+
base_url: The base URL (e.g., "https://api.synthinsightlabs.com").
|
|
83
|
+
api_key: API key for authentication.
|
|
84
|
+
db_path: Override buffer DB path for testing.
|
|
85
|
+
max_batch: Maximum records to flush per call.
|
|
86
|
+
|
|
87
|
+
Returns:
|
|
88
|
+
Dict with keys: sent, failed, remaining.
|
|
89
|
+
"""
|
|
90
|
+
conn = _get_buffer_connection(db_path)
|
|
91
|
+
sent = 0
|
|
92
|
+
failed = 0
|
|
93
|
+
|
|
94
|
+
try:
|
|
95
|
+
rows = conn.execute(
|
|
96
|
+
"""SELECT id, endpoint, payload, attempts
|
|
97
|
+
FROM pending_records
|
|
98
|
+
ORDER BY created_at ASC
|
|
99
|
+
LIMIT ?""",
|
|
100
|
+
(max_batch,),
|
|
101
|
+
).fetchall()
|
|
102
|
+
|
|
103
|
+
for row in rows:
|
|
104
|
+
record_id = row["id"]
|
|
105
|
+
endpoint = row["endpoint"]
|
|
106
|
+
payload = row["payload"]
|
|
107
|
+
|
|
108
|
+
url = f"{base_url.rstrip('/')}{endpoint}"
|
|
109
|
+
headers = {"Content-Type": "application/json"}
|
|
110
|
+
if api_key:
|
|
111
|
+
headers["X-API-Key"] = api_key
|
|
112
|
+
|
|
113
|
+
try:
|
|
114
|
+
req = Request(
|
|
115
|
+
url,
|
|
116
|
+
data=payload.encode("utf-8"),
|
|
117
|
+
headers=headers,
|
|
118
|
+
method="POST",
|
|
119
|
+
)
|
|
120
|
+
resp = urlopen(req, timeout=10)
|
|
121
|
+
if resp.status < 300:
|
|
122
|
+
conn.execute(
|
|
123
|
+
"DELETE FROM pending_records WHERE id = ?",
|
|
124
|
+
(record_id,),
|
|
125
|
+
)
|
|
126
|
+
sent += 1
|
|
127
|
+
else:
|
|
128
|
+
conn.execute(
|
|
129
|
+
"""UPDATE pending_records
|
|
130
|
+
SET attempts = attempts + 1, last_attempt = ?
|
|
131
|
+
WHERE id = ?""",
|
|
132
|
+
(datetime.now().isoformat(), record_id),
|
|
133
|
+
)
|
|
134
|
+
failed += 1
|
|
135
|
+
except (URLError, OSError, TimeoutError):
|
|
136
|
+
conn.execute(
|
|
137
|
+
"""UPDATE pending_records
|
|
138
|
+
SET attempts = attempts + 1, last_attempt = ?
|
|
139
|
+
WHERE id = ?""",
|
|
140
|
+
(datetime.now().isoformat(), record_id),
|
|
141
|
+
)
|
|
142
|
+
failed += 1
|
|
143
|
+
|
|
144
|
+
conn.commit()
|
|
145
|
+
|
|
146
|
+
remaining_row = conn.execute(
|
|
147
|
+
"SELECT COUNT(*) AS cnt FROM pending_records"
|
|
148
|
+
).fetchone()
|
|
149
|
+
remaining = remaining_row["cnt"] if remaining_row else 0
|
|
150
|
+
|
|
151
|
+
return {"sent": sent, "failed": failed, "remaining": remaining}
|
|
152
|
+
finally:
|
|
153
|
+
conn.close()
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def pending_count(db_path: Optional[Path] = None) -> int:
|
|
157
|
+
"""Return the number of pending buffered records."""
|
|
158
|
+
conn = _get_buffer_connection(db_path)
|
|
159
|
+
try:
|
|
160
|
+
row = conn.execute("SELECT COUNT(*) AS cnt FROM pending_records").fetchone()
|
|
161
|
+
return row["cnt"] if row else 0
|
|
162
|
+
finally:
|
|
163
|
+
conn.close()
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def clear_buffer(db_path: Optional[Path] = None) -> int:
|
|
167
|
+
"""Clear all buffered records. Returns count deleted."""
|
|
168
|
+
conn = _get_buffer_connection(db_path)
|
|
169
|
+
try:
|
|
170
|
+
cursor = conn.execute("DELETE FROM pending_records")
|
|
171
|
+
conn.commit()
|
|
172
|
+
return cursor.rowcount
|
|
173
|
+
finally:
|
|
174
|
+
conn.close()
|
|
@@ -0,0 +1,345 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Client-side API cost tracker.
|
|
3
|
+
|
|
4
|
+
Two public functions:
|
|
5
|
+
track(response, provider, ...) -- for AI API responses. Extracts tokens, POSTs to server.
|
|
6
|
+
log_call(provider, service, ...) -- for non-AI API calls. Logs that the call happened.
|
|
7
|
+
|
|
8
|
+
Both return immediately. Neither raises exceptions -- a tracking failure (or an
|
|
9
|
+
unknown provider) is logged at WARNING with the call's identity, because that
|
|
10
|
+
call's spend was not recorded.
|
|
11
|
+
The API call must always succeed even if tracking fails.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
import logging
|
|
16
|
+
import os
|
|
17
|
+
import platform
|
|
18
|
+
from datetime import datetime
|
|
19
|
+
from typing import Optional
|
|
20
|
+
from urllib.error import URLError
|
|
21
|
+
from urllib.request import Request, urlopen
|
|
22
|
+
|
|
23
|
+
from .buffer import buffer_record
|
|
24
|
+
|
|
25
|
+
_log = logging.getLogger(__name__)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
# Configurable endpoint
|
|
29
|
+
COST_TRACKER_URL = os.environ.get(
|
|
30
|
+
"COST_TRACKER_URL", "https://api.synthinsightlabs.com"
|
|
31
|
+
)
|
|
32
|
+
COST_TRACKER_API_KEY = os.environ.get("COST_TRACKER_API_KEY")
|
|
33
|
+
|
|
34
|
+
# Timeout for POSTs (seconds) -- keep short so tracking doesn't slow down the caller
|
|
35
|
+
_POST_TIMEOUT = 5
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _post_to_server(endpoint: str, payload: dict) -> bool:
|
|
39
|
+
"""
|
|
40
|
+
POST payload to the hosted endpoint.
|
|
41
|
+
Returns True if successful, False if failed (payload gets buffered).
|
|
42
|
+
"""
|
|
43
|
+
url = f"{COST_TRACKER_URL.rstrip('/')}{endpoint}"
|
|
44
|
+
headers = {"Content-Type": "application/json"}
|
|
45
|
+
if COST_TRACKER_API_KEY:
|
|
46
|
+
headers["X-API-Key"] = COST_TRACKER_API_KEY
|
|
47
|
+
|
|
48
|
+
try:
|
|
49
|
+
req = Request(
|
|
50
|
+
url,
|
|
51
|
+
data=json.dumps(payload).encode("utf-8"),
|
|
52
|
+
headers=headers,
|
|
53
|
+
method="POST",
|
|
54
|
+
)
|
|
55
|
+
resp = urlopen(req, timeout=_POST_TIMEOUT)
|
|
56
|
+
return resp.status < 300
|
|
57
|
+
except (URLError, OSError, TimeoutError): # governance: allow-silent SF002: False is the documented POST-failed result; every caller then writes the payload to the offline buffer via buffer_record for later flush
|
|
58
|
+
return False
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _extract_anthropic(response) -> dict:
|
|
62
|
+
"""Extract token usage from an Anthropic API response."""
|
|
63
|
+
usage = response.usage
|
|
64
|
+
return {
|
|
65
|
+
"model": response.model,
|
|
66
|
+
"prompt_tokens": usage.input_tokens,
|
|
67
|
+
"completion_tokens": usage.output_tokens,
|
|
68
|
+
"cache_read_tokens": getattr(usage, "cache_read_input_tokens", 0) or 0,
|
|
69
|
+
"cache_creation_tokens": getattr(usage, "cache_creation_input_tokens", 0) or 0,
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _extract_openai(response) -> dict:
|
|
74
|
+
"""
|
|
75
|
+
Extract token usage from an OpenAI API response.
|
|
76
|
+
Handles both Chat Completions and Responses API formats.
|
|
77
|
+
"""
|
|
78
|
+
usage = response.usage
|
|
79
|
+
model = response.model
|
|
80
|
+
|
|
81
|
+
# Chat Completions API
|
|
82
|
+
if hasattr(usage, "prompt_tokens"):
|
|
83
|
+
return {
|
|
84
|
+
"model": model,
|
|
85
|
+
"prompt_tokens": usage.prompt_tokens,
|
|
86
|
+
"completion_tokens": usage.completion_tokens,
|
|
87
|
+
"cache_read_tokens": 0,
|
|
88
|
+
"cache_creation_tokens": 0,
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
# Responses API
|
|
92
|
+
return {
|
|
93
|
+
"model": model,
|
|
94
|
+
"prompt_tokens": getattr(usage, "input_tokens", 0) or 0,
|
|
95
|
+
"completion_tokens": getattr(usage, "output_tokens", 0) or 0,
|
|
96
|
+
"cache_read_tokens": 0,
|
|
97
|
+
"cache_creation_tokens": 0,
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _extract_google(response, model: Optional[str] = None) -> dict:
|
|
102
|
+
"""Extract token usage from a Google Gemini API response."""
|
|
103
|
+
meta = response.usage_metadata
|
|
104
|
+
|
|
105
|
+
prompt = getattr(meta, "prompt_token_count", 0) or 0
|
|
106
|
+
completion = getattr(meta, "candidates_token_count", 0) or 0
|
|
107
|
+
|
|
108
|
+
resolved_model = model
|
|
109
|
+
if resolved_model is None:
|
|
110
|
+
resolved_model = getattr(response, "model_version", None)
|
|
111
|
+
if resolved_model is None:
|
|
112
|
+
resolved_model = getattr(response, "model", "unknown")
|
|
113
|
+
|
|
114
|
+
return {
|
|
115
|
+
"model": resolved_model,
|
|
116
|
+
"prompt_tokens": prompt,
|
|
117
|
+
"completion_tokens": completion,
|
|
118
|
+
"cache_read_tokens": 0,
|
|
119
|
+
"cache_creation_tokens": 0,
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _extract_openrouter(response) -> dict:
|
|
124
|
+
"""
|
|
125
|
+
Extract token usage and the billed cost from an OpenRouter response.
|
|
126
|
+
|
|
127
|
+
OpenRouter's usage object is OpenAI-compatible and also carries ``cost``,
|
|
128
|
+
"the total amount charged to your account" in USD-denominated credits,
|
|
129
|
+
returned on every response
|
|
130
|
+
(https://openrouter.ai/docs/use-cases/usage-accounting). That reported
|
|
131
|
+
cost is kept as ``reported_cost_usd`` so no registry price is needed for
|
|
132
|
+
OpenRouter model ids such as ``x-ai/grok-4.3``. A missing or invalid
|
|
133
|
+
cost yields None.
|
|
134
|
+
"""
|
|
135
|
+
data = _extract_openai(response)
|
|
136
|
+
cost = getattr(response.usage, "cost", None)
|
|
137
|
+
valid = isinstance(cost, (int, float)) and not isinstance(cost, bool) and cost >= 0
|
|
138
|
+
data["reported_cost_usd"] = float(cost) if valid else None
|
|
139
|
+
return data
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
_EXTRACTORS = {
|
|
143
|
+
"anthropic": _extract_anthropic,
|
|
144
|
+
"openai": _extract_openai,
|
|
145
|
+
"xai": _extract_openai, # xAI uses OpenAI-compatible format
|
|
146
|
+
"openrouter": _extract_openrouter, # OpenAI-compatible + billed cost
|
|
147
|
+
"google": _extract_google,
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _model_for_warning(response, model: Optional[str]):
|
|
152
|
+
"""Identity for an 'API cost NOT recorded' warning: the explicit model, else response.model."""
|
|
153
|
+
if model:
|
|
154
|
+
return model
|
|
155
|
+
try:
|
|
156
|
+
return getattr(response, "model", None)
|
|
157
|
+
except Exception: # governance: allow-silent SF002: feeds a warning line only; None logs as an unknown model and must never make track() raise
|
|
158
|
+
return None
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def track(
|
|
162
|
+
response,
|
|
163
|
+
provider: str,
|
|
164
|
+
*,
|
|
165
|
+
model: Optional[str] = None,
|
|
166
|
+
project: Optional[str] = None,
|
|
167
|
+
caller: Optional[str] = None,
|
|
168
|
+
service: str = "chat",
|
|
169
|
+
):
|
|
170
|
+
"""
|
|
171
|
+
Track an AI API response. Extracts tokens, sends to server.
|
|
172
|
+
|
|
173
|
+
The response object passes through unchanged -- call this after every API call.
|
|
174
|
+
|
|
175
|
+
Usage:
|
|
176
|
+
resp = client.messages.create(model="claude-haiku-4-5", ...)
|
|
177
|
+
track(resp, "anthropic", project="ai-memory", caller="classifier")
|
|
178
|
+
|
|
179
|
+
resp = openai_client.chat.completions.create(model="gpt-4.1-mini", ...)
|
|
180
|
+
track(resp, "openai", project="flowfi")
|
|
181
|
+
|
|
182
|
+
Args:
|
|
183
|
+
response: The raw SDK response object.
|
|
184
|
+
provider: Provider name ("anthropic", "openai", "xai", "openrouter", "google").
|
|
185
|
+
model: Override model name (extracted from response if not provided).
|
|
186
|
+
project: Project name for attribution.
|
|
187
|
+
caller: Identifies the calling code path.
|
|
188
|
+
service: Service type (default "chat").
|
|
189
|
+
|
|
190
|
+
Returns:
|
|
191
|
+
The original response object, unchanged.
|
|
192
|
+
"""
|
|
193
|
+
try:
|
|
194
|
+
provider = provider.lower()
|
|
195
|
+
extractor = _EXTRACTORS.get(provider)
|
|
196
|
+
if extractor is None:
|
|
197
|
+
# Unknown provider: pass the response through, but the spend is not recorded.
|
|
198
|
+
_log.warning(
|
|
199
|
+
"API cost NOT recorded: unknown provider %r (model=%r, project=%r, caller=%r)",
|
|
200
|
+
provider, _model_for_warning(response, model), project, caller,
|
|
201
|
+
)
|
|
202
|
+
return response
|
|
203
|
+
|
|
204
|
+
# Google extractor needs model hint
|
|
205
|
+
if provider == "google":
|
|
206
|
+
usage_data = extractor(response, model=model)
|
|
207
|
+
else:
|
|
208
|
+
usage_data = extractor(response)
|
|
209
|
+
|
|
210
|
+
# Allow model override
|
|
211
|
+
if model:
|
|
212
|
+
usage_data["model"] = model
|
|
213
|
+
|
|
214
|
+
resolved_model = usage_data.get("model", "unknown")
|
|
215
|
+
prompt_tokens = usage_data.get("prompt_tokens", 0)
|
|
216
|
+
completion_tokens = usage_data.get("completion_tokens", 0)
|
|
217
|
+
cache_read = usage_data.get("cache_read_tokens", 0)
|
|
218
|
+
cache_creation = usage_data.get("cache_creation_tokens", 0)
|
|
219
|
+
total_tokens = prompt_tokens + completion_tokens + cache_read + cache_creation
|
|
220
|
+
|
|
221
|
+
payload = {
|
|
222
|
+
"timestamp": datetime.now().isoformat(),
|
|
223
|
+
"provider": provider,
|
|
224
|
+
"model": resolved_model,
|
|
225
|
+
"service": service,
|
|
226
|
+
"api_type": "ai",
|
|
227
|
+
"project": project,
|
|
228
|
+
"prompt_tokens": prompt_tokens,
|
|
229
|
+
"completion_tokens": completion_tokens,
|
|
230
|
+
"cache_read_tokens": cache_read,
|
|
231
|
+
"cache_creation_tokens": cache_creation,
|
|
232
|
+
"total_tokens": total_tokens,
|
|
233
|
+
# Provider-billed cost when reported (OpenRouter usage.cost);
|
|
234
|
+
# otherwise None and the server calculates it.
|
|
235
|
+
"estimated_cost_usd": usage_data.get("reported_cost_usd"),
|
|
236
|
+
"source_machine": platform.node(),
|
|
237
|
+
"caller": caller,
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
if not _post_to_server("/track", payload):
|
|
241
|
+
buffer_record("/track", payload)
|
|
242
|
+
|
|
243
|
+
except Exception: # governance: allow-silent SF001: documented never-raise contract (the caller's API call already succeeded and must not fail or be retried); the uncounted call is logged at WARNING with its identity
|
|
244
|
+
_log.warning(
|
|
245
|
+
"API cost NOT recorded: track() failed for provider=%r model=%r project=%r caller=%r",
|
|
246
|
+
provider, _model_for_warning(response, model), project, caller,
|
|
247
|
+
exc_info=True,
|
|
248
|
+
)
|
|
249
|
+
|
|
250
|
+
return response
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def track_tokens(
|
|
254
|
+
provider: str,
|
|
255
|
+
model: str,
|
|
256
|
+
prompt_tokens: int,
|
|
257
|
+
completion_tokens: int,
|
|
258
|
+
*,
|
|
259
|
+
cache_read_tokens: int = 0,
|
|
260
|
+
cache_creation_tokens: int = 0,
|
|
261
|
+
project: Optional[str] = None,
|
|
262
|
+
caller: Optional[str] = None,
|
|
263
|
+
service: str = "chat",
|
|
264
|
+
) -> None:
|
|
265
|
+
"""
|
|
266
|
+
Track AI API usage by token counts directly (no SDK response object needed).
|
|
267
|
+
|
|
268
|
+
Use this when calling provider APIs via raw HTTP instead of their SDK.
|
|
269
|
+
|
|
270
|
+
Usage:
|
|
271
|
+
result = json.loads(resp.read())
|
|
272
|
+
usage = result["usage"]
|
|
273
|
+
track_tokens("anthropic", result["model"],
|
|
274
|
+
usage["input_tokens"], usage["output_tokens"],
|
|
275
|
+
project="ai-memory", caller="brain.classify_scope")
|
|
276
|
+
"""
|
|
277
|
+
try:
|
|
278
|
+
total_tokens = prompt_tokens + completion_tokens + cache_read_tokens + cache_creation_tokens
|
|
279
|
+
|
|
280
|
+
payload = {
|
|
281
|
+
"timestamp": datetime.now().isoformat(),
|
|
282
|
+
"provider": provider.lower(),
|
|
283
|
+
"model": model,
|
|
284
|
+
"service": service,
|
|
285
|
+
"api_type": "ai",
|
|
286
|
+
"project": project,
|
|
287
|
+
"prompt_tokens": prompt_tokens,
|
|
288
|
+
"completion_tokens": completion_tokens,
|
|
289
|
+
"cache_read_tokens": cache_read_tokens,
|
|
290
|
+
"cache_creation_tokens": cache_creation_tokens,
|
|
291
|
+
"total_tokens": total_tokens,
|
|
292
|
+
"estimated_cost_usd": None, # Server calculates cost
|
|
293
|
+
"source_machine": platform.node(),
|
|
294
|
+
"caller": caller,
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
if not _post_to_server("/track", payload):
|
|
298
|
+
buffer_record("/track", payload)
|
|
299
|
+
|
|
300
|
+
except Exception: # governance: allow-silent SF001: documented never-raise contract (the caller's API call already succeeded and must not fail or be retried); the uncounted call is logged at WARNING with its identity
|
|
301
|
+
_log.warning(
|
|
302
|
+
"API cost NOT recorded: track_tokens() failed for provider=%r model=%r project=%r caller=%r",
|
|
303
|
+
provider, model, project, caller,
|
|
304
|
+
exc_info=True,
|
|
305
|
+
)
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
def log_call(
|
|
309
|
+
provider: str,
|
|
310
|
+
*,
|
|
311
|
+
service: str = "api",
|
|
312
|
+
project: Optional[str] = None,
|
|
313
|
+
caller: Optional[str] = None,
|
|
314
|
+
) -> None:
|
|
315
|
+
"""
|
|
316
|
+
Log a non-AI API call (Vercel, Doppler, RunPod, etc.).
|
|
317
|
+
|
|
318
|
+
Usage:
|
|
319
|
+
log_call("vercel", service="blob", project="muffinpanrecipes", caller="storage.upload")
|
|
320
|
+
|
|
321
|
+
Args:
|
|
322
|
+
provider: Service provider name.
|
|
323
|
+
service: Specific service/endpoint being called.
|
|
324
|
+
project: Project name for attribution.
|
|
325
|
+
caller: Identifies the calling code path.
|
|
326
|
+
"""
|
|
327
|
+
try:
|
|
328
|
+
payload = {
|
|
329
|
+
"timestamp": datetime.now().isoformat(),
|
|
330
|
+
"provider": provider.lower(),
|
|
331
|
+
"service": service,
|
|
332
|
+
"project": project,
|
|
333
|
+
"caller": caller,
|
|
334
|
+
"source_machine": platform.node(),
|
|
335
|
+
}
|
|
336
|
+
|
|
337
|
+
if not _post_to_server("/log_call", payload):
|
|
338
|
+
buffer_record("/log_call", payload)
|
|
339
|
+
|
|
340
|
+
except Exception: # governance: allow-silent SF001: documented never-raise contract (the caller's API call already succeeded and must not fail or be retried); the uncounted call is logged at WARNING with its identity
|
|
341
|
+
_log.warning(
|
|
342
|
+
"API call NOT recorded: log_call() failed for provider=%r service=%r project=%r caller=%r",
|
|
343
|
+
provider, service, project, caller,
|
|
344
|
+
exc_info=True,
|
|
345
|
+
)
|