tensorcost 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tensorcost/__init__.py +61 -0
- tensorcost/_applied_mode.py +148 -0
- tensorcost/_circuit.py +99 -0
- tensorcost/_config.py +156 -0
- tensorcost/_errors.py +122 -0
- tensorcost/_models.py +40 -0
- tensorcost/_providers/__init__.py +1 -0
- tensorcost/_providers/_anthropic.py +266 -0
- tensorcost/_providers/_detect.py +46 -0
- tensorcost/_providers/_openai.py +295 -0
- tensorcost/_proxy_client.py +408 -0
- tensorcost/_retry.py +99 -0
- tensorcost/_telemetry.py +120 -0
- tensorcost/_transport.py +174 -0
- tensorcost/_wrap.py +162 -0
- tensorcost/py.typed +0 -0
- tensorcost-0.4.0.dist-info/METADATA +281 -0
- tensorcost-0.4.0.dist-info/RECORD +20 -0
- tensorcost-0.4.0.dist-info/WHEEL +4 -0
- tensorcost-0.4.0.dist-info/licenses/LICENSE +201 -0
tensorcost/__init__.py
ADDED
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""TensorCost Python SDK — one-line wrapper for OpenAI and Anthropic clients.
|
|
2
|
+
|
|
3
|
+
Usage:
|
|
4
|
+
|
|
5
|
+
from openai import OpenAI
|
|
6
|
+
from tensorcost import wrap
|
|
7
|
+
|
|
8
|
+
client = wrap(OpenAI(api_key="..."))
|
|
9
|
+
# use `client` exactly like a normal OpenAI client; observations
|
|
10
|
+
# are sent fire-and-forget to TensorCost in the background.
|
|
11
|
+
|
|
12
|
+
The SDK fails open by default: if TensorCost is unreachable, the
|
|
13
|
+
underlying provider call still completes normally.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from ._wrap import wrap
|
|
17
|
+
from ._config import MissingConfigError
|
|
18
|
+
from ._providers._detect import UnsupportedClientError
|
|
19
|
+
from ._errors import (
|
|
20
|
+
TensorCostError,
|
|
21
|
+
TensorCostNetworkError,
|
|
22
|
+
TensorCostTimeoutError,
|
|
23
|
+
TensorCostProxyError,
|
|
24
|
+
TensorCostQuotaError,
|
|
25
|
+
TensorCostProviderError,
|
|
26
|
+
)
|
|
27
|
+
from ._telemetry import (
|
|
28
|
+
LifecycleEvent,
|
|
29
|
+
BeforeRequestEvent,
|
|
30
|
+
AfterResponseEvent,
|
|
31
|
+
OnRetryEvent,
|
|
32
|
+
OnErrorEvent,
|
|
33
|
+
OnFallbackEvent,
|
|
34
|
+
LifecycleEventCallback,
|
|
35
|
+
)
|
|
36
|
+
from ._retry import RetryConfig
|
|
37
|
+
|
|
38
|
+
__version__ = "0.4.0"
|
|
39
|
+
__all__ = [
|
|
40
|
+
"wrap",
|
|
41
|
+
"MissingConfigError",
|
|
42
|
+
"UnsupportedClientError",
|
|
43
|
+
# Error hierarchy
|
|
44
|
+
"TensorCostError",
|
|
45
|
+
"TensorCostNetworkError",
|
|
46
|
+
"TensorCostTimeoutError",
|
|
47
|
+
"TensorCostProxyError",
|
|
48
|
+
"TensorCostQuotaError",
|
|
49
|
+
"TensorCostProviderError",
|
|
50
|
+
# Telemetry types
|
|
51
|
+
"LifecycleEvent",
|
|
52
|
+
"BeforeRequestEvent",
|
|
53
|
+
"AfterResponseEvent",
|
|
54
|
+
"OnRetryEvent",
|
|
55
|
+
"OnErrorEvent",
|
|
56
|
+
"OnFallbackEvent",
|
|
57
|
+
"LifecycleEventCallback",
|
|
58
|
+
# Config types
|
|
59
|
+
"RetryConfig",
|
|
60
|
+
"__version__",
|
|
61
|
+
]
|
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
"""Applied-mode (Layer 2) HTTP routing helpers.
|
|
2
|
+
|
|
3
|
+
When applied_mode=True, chat/completions calls are forwarded through the
|
|
4
|
+
TensorCost inference-proxy rather than going straight to the upstream
|
|
5
|
+
provider. The proxy decides per-request whether to route or pass through;
|
|
6
|
+
the SDK is not involved in that decision.
|
|
7
|
+
|
|
8
|
+
The two required proxy headers:
|
|
9
|
+
x-tc-provider-url The original provider's base URL (e.g. "https://api.openai.com").
|
|
10
|
+
x-tc-provider-auth The auth header value the SDK would have sent to the provider
|
|
11
|
+
(e.g. "Bearer sk-..." for OpenAI, "sk-ant-..." for Anthropic).
|
|
12
|
+
|
|
13
|
+
The proxy is OpenAI-API-compatible — the request body is forwarded verbatim,
|
|
14
|
+
and the response body is returned verbatim to the caller. The customer's
|
|
15
|
+
parsing code sees no difference.
|
|
16
|
+
|
|
17
|
+
This module is an internal helper; nothing here is part of the public API.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
from typing import Any
|
|
23
|
+
|
|
24
|
+
import httpx
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
# Proxy path appended to proxy_url when routing chat/completions.
|
|
28
|
+
PROXY_CHAT_PATH = "/api/inference-proxy/v1/chat/completions"
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def build_proxy_headers(
|
|
32
|
+
provider_url: str,
|
|
33
|
+
provider_auth: str,
|
|
34
|
+
correlation_id: str | None = None,
|
|
35
|
+
environment: str | None = None,
|
|
36
|
+
) -> dict[str, str]:
|
|
37
|
+
"""Return the custom headers the proxy controller requires.
|
|
38
|
+
|
|
39
|
+
Only x-tc-provider-url and x-tc-provider-auth are mandatory; the
|
|
40
|
+
correlation and environment headers are added when present so the
|
|
41
|
+
proxy can tie the request to the observation record and use the
|
|
42
|
+
environment for policy matching.
|
|
43
|
+
"""
|
|
44
|
+
headers: dict[str, str] = {
|
|
45
|
+
"x-tc-provider-url": provider_url,
|
|
46
|
+
"x-tc-provider-auth": provider_auth,
|
|
47
|
+
}
|
|
48
|
+
if correlation_id:
|
|
49
|
+
headers["x-tc-correlation-id"] = correlation_id
|
|
50
|
+
if environment:
|
|
51
|
+
headers["x-tc-environment"] = environment
|
|
52
|
+
return headers
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def openai_provider_auth(client: Any) -> str:
|
|
56
|
+
"""Extract the Authorization header value for an OpenAI-style client.
|
|
57
|
+
|
|
58
|
+
OpenAI uses "Bearer <api_key>" as the Authorization header. We
|
|
59
|
+
read the key from the client object rather than re-capturing a
|
|
60
|
+
closure variable so that key rotation on the client instance is
|
|
61
|
+
respected automatically.
|
|
62
|
+
"""
|
|
63
|
+
# openai.OpenAI stores the key as client.api_key.
|
|
64
|
+
api_key = getattr(client, "api_key", None)
|
|
65
|
+
if api_key:
|
|
66
|
+
return f"Bearer {api_key}"
|
|
67
|
+
# Fallback: return empty string; the proxy will 400 on an absent auth.
|
|
68
|
+
return ""
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def openai_provider_base_url(client: Any) -> str:
|
|
72
|
+
"""Return the upstream base URL for an OpenAI-style client.
|
|
73
|
+
|
|
74
|
+
openai.OpenAI stores it as client.base_url (an httpx.URL). We
|
|
75
|
+
coerce to str and strip trailing slash so the proxy gets a clean URL.
|
|
76
|
+
"""
|
|
77
|
+
base = getattr(client, "base_url", None)
|
|
78
|
+
if base is not None:
|
|
79
|
+
url = str(base).rstrip("/")
|
|
80
|
+
# openai-python sets base_url to "https://api.openai.com/v1" by
|
|
81
|
+
# default. The proxy expects the provider's plain base (no path),
|
|
82
|
+
# because the proxy appends "/v1/chat/completions" itself.
|
|
83
|
+
# Strip any trailing path component that looks like an API version
|
|
84
|
+
# (i.e. ends with /v1, /v2, etc.) so x-tc-provider-url is always
|
|
85
|
+
# the scheme+host portion. The proxy will receive
|
|
86
|
+
# "https://api.openai.com" and know what path to append.
|
|
87
|
+
import re
|
|
88
|
+
url = re.sub(r"/v\d+$", "", url)
|
|
89
|
+
return url
|
|
90
|
+
return "https://api.openai.com"
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def anthropic_provider_auth(client: Any) -> str:
|
|
94
|
+
"""Extract the x-api-key value for an Anthropic client.
|
|
95
|
+
|
|
96
|
+
Anthropic's SDK uses "x-api-key: <key>" rather than Bearer.
|
|
97
|
+
The proxy is responsible for forwarding this in the right header;
|
|
98
|
+
the SDK just surfaces the raw key. We prefix with the literal
|
|
99
|
+
header name so the proxy can reconstruct the Authorization shape.
|
|
100
|
+
"""
|
|
101
|
+
api_key = getattr(client, "api_key", None)
|
|
102
|
+
if api_key:
|
|
103
|
+
# Encode as "x-api-key <key>" so the proxy can distinguish this
|
|
104
|
+
# from OpenAI-style Bearer tokens and set the right header on the
|
|
105
|
+
# upstream call.
|
|
106
|
+
return f"x-api-key {api_key}"
|
|
107
|
+
return ""
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def anthropic_provider_base_url(client: Any) -> str:
|
|
111
|
+
"""Return the upstream base URL for an Anthropic client."""
|
|
112
|
+
base = getattr(client, "base_url", None)
|
|
113
|
+
if base is not None:
|
|
114
|
+
return str(base).rstrip("/")
|
|
115
|
+
return "https://api.anthropic.com"
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def route_via_proxy(
|
|
119
|
+
proxy_url: str,
|
|
120
|
+
bearer_token: str,
|
|
121
|
+
provider_url: str,
|
|
122
|
+
provider_auth: str,
|
|
123
|
+
body: dict[str, Any],
|
|
124
|
+
timeout_s: float = 30.0,
|
|
125
|
+
correlation_id: str | None = None,
|
|
126
|
+
environment: str | None = None,
|
|
127
|
+
) -> httpx.Response:
|
|
128
|
+
"""POST *body* to the inference-proxy and return the raw httpx.Response.
|
|
129
|
+
|
|
130
|
+
This is a synchronous blocking call — it runs on the customer's
|
|
131
|
+
calling thread, same as the original provider call would have.
|
|
132
|
+
|
|
133
|
+
The caller is responsible for forwarding the response status and body
|
|
134
|
+
back to the customer's code.
|
|
135
|
+
"""
|
|
136
|
+
url = proxy_url.rstrip("/") + PROXY_CHAT_PATH
|
|
137
|
+
headers = {
|
|
138
|
+
"Authorization": f"Bearer {bearer_token}",
|
|
139
|
+
"Content-Type": "application/json",
|
|
140
|
+
**build_proxy_headers(
|
|
141
|
+
provider_url=provider_url,
|
|
142
|
+
provider_auth=provider_auth,
|
|
143
|
+
correlation_id=correlation_id,
|
|
144
|
+
environment=environment,
|
|
145
|
+
),
|
|
146
|
+
}
|
|
147
|
+
with httpx.Client(timeout=timeout_s) as client:
|
|
148
|
+
return client.post(url, json=body, headers=headers)
|
tensorcost/_circuit.py
ADDED
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
"""Circuit breaker for the fail-open proxy fallback.
|
|
2
|
+
|
|
3
|
+
Tracks consecutive proxy failures. When the failure count reaches
|
|
4
|
+
``open_threshold``, the circuit opens and subsequent requests go directly to
|
|
5
|
+
the provider instead of through the proxy. The circuit closes again after
|
|
6
|
+
``close_threshold`` consecutive successful probe calls.
|
|
7
|
+
|
|
8
|
+
State machine::
|
|
9
|
+
|
|
10
|
+
CLOSED ──(failures >= open_threshold)──> OPEN
|
|
11
|
+
OPEN ──(next call, probes)───────────> HALF_OPEN
|
|
12
|
+
HALF_OPEN ──(proxy 2xx)────────────────> probe_successes++
|
|
13
|
+
──(proxy 5xx/err)───────────> back to OPEN
|
|
14
|
+
HALF_OPEN ──(probe_successes >= close_threshold)──> CLOSED
|
|
15
|
+
|
|
16
|
+
This is intentionally simple and synchronous — the circuit lives on the SDK
|
|
17
|
+
client instance and is not shared across threads. The transport layer already
|
|
18
|
+
serialises token exchange with a lock; no additional locking is needed here
|
|
19
|
+
because apply-mode proxy calls run on the calling thread.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
from dataclasses import dataclass
|
|
25
|
+
from typing import Literal
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
CircuitState = Literal["CLOSED", "OPEN", "HALF_OPEN"]
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass(frozen=True)
|
|
32
|
+
class CircuitBreakerConfig:
|
|
33
|
+
"""Configuration for the proxy circuit breaker."""
|
|
34
|
+
|
|
35
|
+
#: Consecutive proxy failures before the circuit opens.
|
|
36
|
+
open_threshold: int = 3
|
|
37
|
+
#: Consecutive probe successes before the circuit closes again.
|
|
38
|
+
close_threshold: int = 5
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
DEFAULT_CIRCUIT_CONFIG = CircuitBreakerConfig()
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class CircuitBreaker:
|
|
45
|
+
"""Simple synchronous circuit breaker."""
|
|
46
|
+
|
|
47
|
+
def __init__(self, config: CircuitBreakerConfig = DEFAULT_CIRCUIT_CONFIG) -> None:
|
|
48
|
+
self.config = config
|
|
49
|
+
self._state: CircuitState = "CLOSED"
|
|
50
|
+
self._consecutive_failures: int = 0
|
|
51
|
+
self._probe_successes: int = 0
|
|
52
|
+
|
|
53
|
+
@property
|
|
54
|
+
def state(self) -> CircuitState:
|
|
55
|
+
return self._state
|
|
56
|
+
|
|
57
|
+
@property
|
|
58
|
+
def failures(self) -> int:
|
|
59
|
+
return self._consecutive_failures
|
|
60
|
+
|
|
61
|
+
def should_bypass(self) -> bool:
|
|
62
|
+
"""Return True when calls should skip the proxy entirely."""
|
|
63
|
+
return self._state == "OPEN"
|
|
64
|
+
|
|
65
|
+
def should_probe(self) -> bool:
|
|
66
|
+
"""Return True when the next call should be a proxy probe.
|
|
67
|
+
|
|
68
|
+
Transitions OPEN → HALF_OPEN as a side effect.
|
|
69
|
+
"""
|
|
70
|
+
if self._state == "OPEN":
|
|
71
|
+
self._state = "HALF_OPEN"
|
|
72
|
+
return True
|
|
73
|
+
return False
|
|
74
|
+
|
|
75
|
+
def record_success(self) -> None:
|
|
76
|
+
"""Call after a proxy request succeeds."""
|
|
77
|
+
if self._state == "HALF_OPEN":
|
|
78
|
+
self._probe_successes += 1
|
|
79
|
+
if self._probe_successes >= self.config.close_threshold:
|
|
80
|
+
self._reset()
|
|
81
|
+
elif self._state == "CLOSED":
|
|
82
|
+
self._consecutive_failures = 0
|
|
83
|
+
|
|
84
|
+
def record_failure(self) -> None:
|
|
85
|
+
"""Call after a proxy request fails (5xx or network error)."""
|
|
86
|
+
if self._state == "HALF_OPEN":
|
|
87
|
+
# Probe failed — go back to fully open.
|
|
88
|
+
self._state = "OPEN"
|
|
89
|
+
self._probe_successes = 0
|
|
90
|
+
elif self._state == "CLOSED":
|
|
91
|
+
self._consecutive_failures += 1
|
|
92
|
+
if self._consecutive_failures >= self.config.open_threshold:
|
|
93
|
+
self._state = "OPEN"
|
|
94
|
+
# OPEN + another failure: stay open.
|
|
95
|
+
|
|
96
|
+
def _reset(self) -> None:
|
|
97
|
+
self._state = "CLOSED"
|
|
98
|
+
self._consecutive_failures = 0
|
|
99
|
+
self._probe_successes = 0
|
tensorcost/_config.py
ADDED
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
"""Configuration resolution for the SDK.
|
|
2
|
+
|
|
3
|
+
Resolution order:
|
|
4
|
+
1. Explicit kwargs to ``wrap()``
|
|
5
|
+
2. Environment variables (TENSORCOST_API_KEY, TENSORCOST_BASE_URL,
|
|
6
|
+
TENSORCOST_TENANT_ID, TENSORCOST_PROXY_URL)
|
|
7
|
+
3. Defaults (only for base_url)
|
|
8
|
+
|
|
9
|
+
If ``api_key`` is still missing after resolution, ``MissingConfigError``
|
|
10
|
+
is raised. This is the *one* case the SDK does not fail open: it means
|
|
11
|
+
the SDK was never configured at all.
|
|
12
|
+
|
|
13
|
+
If ``applied_mode=True`` but no proxy URL is available (from kwarg or
|
|
14
|
+
``TENSORCOST_PROXY_URL``), ``MissingConfigError`` is raised at wrap time
|
|
15
|
+
rather than on the first request — misconfigured applied-mode fails
|
|
16
|
+
loudly so the problem surfaces in development, not in production.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import os
|
|
22
|
+
from dataclasses import dataclass, field
|
|
23
|
+
from typing import Callable, Optional
|
|
24
|
+
|
|
25
|
+
from ._retry import RetryConfig, DEFAULT_RETRY_CONFIG
|
|
26
|
+
from ._telemetry import LifecycleEventCallback
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
DEFAULT_BASE_URL = "https://api.tensorcost.com"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class MissingConfigError(RuntimeError):
|
|
33
|
+
"""Raised when required configuration (api_key or proxy_url) is missing."""
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass(frozen=True)
|
|
37
|
+
class ResolvedConfig:
|
|
38
|
+
api_key: str
|
|
39
|
+
base_url: str
|
|
40
|
+
tenant_id: Optional[str]
|
|
41
|
+
# Phase A4 batch 3 — resolved environment tag. ``None`` when the
|
|
42
|
+
# customer did not configure one; the SDK omits the ``environment``
|
|
43
|
+
# field from the envelope in that case so the backend's default
|
|
44
|
+
# ('production') applies.
|
|
45
|
+
environment: Optional[str]
|
|
46
|
+
# Phase A4 batch 4 — resolved provider-connection identifier.
|
|
47
|
+
# ``None`` when the customer did not configure one; the SDK omits
|
|
48
|
+
# the ``connection_id`` field from the envelope in that case.
|
|
49
|
+
connection_id: Optional[str]
|
|
50
|
+
fail_open: bool
|
|
51
|
+
# Applied-mode (Layer 2). When True, wrap functions route chat/completions
|
|
52
|
+
# calls through the TensorCost inference-proxy instead of directly to the
|
|
53
|
+
# upstream provider. Observe-only telemetry still fires on every request.
|
|
54
|
+
applied_mode: bool
|
|
55
|
+
# Resolved proxy base URL. Only meaningful when applied_mode is True.
|
|
56
|
+
# Points to the TensorCost deployment (e.g. "https://api.my-instance.tensorcost.com").
|
|
57
|
+
# The SDK appends /api/inference-proxy/v1/chat/completions to form the target URL.
|
|
58
|
+
proxy_url: Optional[str]
|
|
59
|
+
|
|
60
|
+
# Hardening fields — all have safe defaults in resolve_config().
|
|
61
|
+
retry: RetryConfig = field(default_factory=lambda: DEFAULT_RETRY_CONFIG)
|
|
62
|
+
timeout_s: float = 60.0
|
|
63
|
+
idle_timeout_s: float = 30.0
|
|
64
|
+
on_lifecycle_event: Optional[LifecycleEventCallback] = None
|
|
65
|
+
fail_open_enabled: bool = True
|
|
66
|
+
provider_api_key: Optional[str] = None
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def resolve_config(
|
|
70
|
+
api_key: Optional[str] = None,
|
|
71
|
+
base_url: Optional[str] = None,
|
|
72
|
+
tenant_id: Optional[str] = None,
|
|
73
|
+
environment: Optional[str] = None,
|
|
74
|
+
connection_id: Optional[str] = None,
|
|
75
|
+
fail_open: bool = True,
|
|
76
|
+
applied_mode: bool = False,
|
|
77
|
+
proxy_url: Optional[str] = None,
|
|
78
|
+
# Hardening options.
|
|
79
|
+
retry: Optional[RetryConfig] = None,
|
|
80
|
+
timeout_s: float = 60.0,
|
|
81
|
+
idle_timeout_s: float = 30.0,
|
|
82
|
+
on_lifecycle_event: Optional[LifecycleEventCallback] = None,
|
|
83
|
+
fail_open_enabled: bool = True,
|
|
84
|
+
provider_api_key: Optional[str] = None,
|
|
85
|
+
) -> ResolvedConfig:
|
|
86
|
+
"""Resolve final config values, env vars filling in any blanks."""
|
|
87
|
+
final_api_key = api_key or os.environ.get("TENSORCOST_API_KEY")
|
|
88
|
+
final_base_url = (
|
|
89
|
+
base_url
|
|
90
|
+
or os.environ.get("TENSORCOST_BASE_URL")
|
|
91
|
+
or DEFAULT_BASE_URL
|
|
92
|
+
)
|
|
93
|
+
final_tenant_id = tenant_id or os.environ.get("TENSORCOST_TENANT_ID")
|
|
94
|
+
|
|
95
|
+
# Phase A4 batch 3 — environment tag. Trimmed; capped at 64 chars
|
|
96
|
+
# per project_environment_scoping.md. An empty / whitespace-only
|
|
97
|
+
# value collapses to None so the SDK omits the field from the
|
|
98
|
+
# envelope and the backend's column default ('production') applies.
|
|
99
|
+
raw_env = environment or os.environ.get("TENSORCOST_ENVIRONMENT")
|
|
100
|
+
final_environment: Optional[str] = None
|
|
101
|
+
if raw_env is not None:
|
|
102
|
+
trimmed = raw_env.strip()
|
|
103
|
+
if len(trimmed) > 64:
|
|
104
|
+
raise MissingConfigError(
|
|
105
|
+
"TensorCost environment must be 64 characters or fewer."
|
|
106
|
+
)
|
|
107
|
+
final_environment = trimmed if trimmed else None
|
|
108
|
+
|
|
109
|
+
# Phase A4 batch 4 — provider-connection id. Same trim+cap rules as
|
|
110
|
+
# environment; collapsing blank to ``None`` so the SDK omits the
|
|
111
|
+
# field when not configured.
|
|
112
|
+
raw_conn = connection_id or os.environ.get("TENSORCOST_CONNECTION_ID")
|
|
113
|
+
final_connection_id: Optional[str] = None
|
|
114
|
+
if raw_conn is not None:
|
|
115
|
+
trimmed = raw_conn.strip()
|
|
116
|
+
if len(trimmed) > 64:
|
|
117
|
+
raise MissingConfigError(
|
|
118
|
+
"TensorCost connection_id must be 64 characters or fewer."
|
|
119
|
+
)
|
|
120
|
+
final_connection_id = trimmed if trimmed else None
|
|
121
|
+
|
|
122
|
+
if not final_api_key:
|
|
123
|
+
raise MissingConfigError(
|
|
124
|
+
"TensorCost api_key not found. Pass api_key=... to wrap() or "
|
|
125
|
+
"set the TENSORCOST_API_KEY environment variable."
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
# Resolve proxy_url. Only required when applied_mode is True.
|
|
129
|
+
final_proxy_url = (
|
|
130
|
+
proxy_url or os.environ.get("TENSORCOST_PROXY_URL") or None
|
|
131
|
+
)
|
|
132
|
+
if final_proxy_url is not None:
|
|
133
|
+
final_proxy_url = final_proxy_url.rstrip("/")
|
|
134
|
+
|
|
135
|
+
if applied_mode and not final_proxy_url:
|
|
136
|
+
raise MissingConfigError(
|
|
137
|
+
"applied_mode=True requires a proxy URL. Pass proxy_url=... to wrap() or "
|
|
138
|
+
"set the TENSORCOST_PROXY_URL environment variable."
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
return ResolvedConfig(
|
|
142
|
+
api_key=final_api_key,
|
|
143
|
+
base_url=final_base_url.rstrip("/"),
|
|
144
|
+
tenant_id=final_tenant_id,
|
|
145
|
+
environment=final_environment,
|
|
146
|
+
connection_id=final_connection_id,
|
|
147
|
+
fail_open=fail_open,
|
|
148
|
+
applied_mode=applied_mode,
|
|
149
|
+
proxy_url=final_proxy_url,
|
|
150
|
+
retry=retry if retry is not None else DEFAULT_RETRY_CONFIG,
|
|
151
|
+
timeout_s=timeout_s,
|
|
152
|
+
idle_timeout_s=idle_timeout_s,
|
|
153
|
+
on_lifecycle_event=on_lifecycle_event,
|
|
154
|
+
fail_open_enabled=fail_open_enabled,
|
|
155
|
+
provider_api_key=provider_api_key,
|
|
156
|
+
)
|
tensorcost/_errors.py
ADDED
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
"""Typed error hierarchy for the TensorCost Python SDK.
|
|
2
|
+
|
|
3
|
+
Every error thrown by the SDK's applied-mode request path is a subclass of
|
|
4
|
+
``TensorCostError``. Customer code can ``isinstance``-branch on any of them
|
|
5
|
+
without importing raw exception types from the underlying HTTP layer.
|
|
6
|
+
|
|
7
|
+
Hierarchy::
|
|
8
|
+
|
|
9
|
+
TensorCostError (base)
|
|
10
|
+
├── TensorCostNetworkError couldn't reach the proxy at all
|
|
11
|
+
├── TensorCostTimeoutError request or idle window exceeded limit
|
|
12
|
+
├── TensorCostProxyError proxy responded with 5xx
|
|
13
|
+
├── TensorCostQuotaError proxy responded with 429
|
|
14
|
+
└── TensorCostProviderError proxy forwarded an upstream provider error
|
|
15
|
+
|
|
16
|
+
All subclasses carry:
|
|
17
|
+
|
|
18
|
+
* ``root_cause`` — the underlying Exception, if any (avoids clash with
|
|
19
|
+
``Exception.__cause__`` semantics)
|
|
20
|
+
* ``status`` — HTTP status code (None when the request never completed)
|
|
21
|
+
* ``request_id`` — value of the X-TC-Request-ID header the proxy stamps
|
|
22
|
+
* ``attempt`` — 1-based attempt number at the time the error occurred
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
from typing import Optional
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class TensorCostError(Exception):
|
|
31
|
+
"""Base class for all TensorCost SDK errors."""
|
|
32
|
+
|
|
33
|
+
def __init__(
|
|
34
|
+
self,
|
|
35
|
+
message: str,
|
|
36
|
+
*,
|
|
37
|
+
root_cause: Optional[Exception] = None,
|
|
38
|
+
status: Optional[int] = None,
|
|
39
|
+
request_id: Optional[str] = None,
|
|
40
|
+
attempt: int = 1,
|
|
41
|
+
) -> None:
|
|
42
|
+
super().__init__(message)
|
|
43
|
+
self.root_cause = root_cause
|
|
44
|
+
self.status = status
|
|
45
|
+
self.request_id = request_id
|
|
46
|
+
self.attempt = attempt
|
|
47
|
+
|
|
48
|
+
def __repr__(self) -> str:
|
|
49
|
+
return (
|
|
50
|
+
f"{self.__class__.__name__}("
|
|
51
|
+
f"message={str(self)!r}, "
|
|
52
|
+
f"status={self.status!r}, "
|
|
53
|
+
f"attempt={self.attempt!r})"
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class TensorCostNetworkError(TensorCostError):
|
|
58
|
+
"""The SDK could not establish a connection to the proxy at all."""
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class TensorCostTimeoutError(TensorCostError):
|
|
62
|
+
"""A request or idle-window timeout fired before the operation completed.
|
|
63
|
+
|
|
64
|
+
``timeout_kind`` is ``"total"`` for a wall-clock limit or ``"idle"`` for
|
|
65
|
+
a streaming idle window.
|
|
66
|
+
"""
|
|
67
|
+
|
|
68
|
+
def __init__(
|
|
69
|
+
self,
|
|
70
|
+
message: str,
|
|
71
|
+
timeout_kind: str,
|
|
72
|
+
*,
|
|
73
|
+
root_cause: Optional[Exception] = None,
|
|
74
|
+
status: Optional[int] = None,
|
|
75
|
+
request_id: Optional[str] = None,
|
|
76
|
+
attempt: int = 1,
|
|
77
|
+
) -> None:
|
|
78
|
+
super().__init__(
|
|
79
|
+
message,
|
|
80
|
+
root_cause=root_cause,
|
|
81
|
+
status=status,
|
|
82
|
+
request_id=request_id,
|
|
83
|
+
attempt=attempt,
|
|
84
|
+
)
|
|
85
|
+
self.timeout_kind = timeout_kind # "total" | "idle"
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
class TensorCostProxyError(TensorCostError):
|
|
89
|
+
"""The proxy returned a 5xx response (TensorCost infrastructure error)."""
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
class TensorCostQuotaError(TensorCostError):
|
|
93
|
+
"""The proxy returned 429 Too Many Requests.
|
|
94
|
+
|
|
95
|
+
``retry_after_ms`` is set when the response included a ``Retry-After``
|
|
96
|
+
header (in milliseconds).
|
|
97
|
+
"""
|
|
98
|
+
|
|
99
|
+
def __init__(
|
|
100
|
+
self,
|
|
101
|
+
message: str,
|
|
102
|
+
retry_after_ms: Optional[int] = None,
|
|
103
|
+
*,
|
|
104
|
+
root_cause: Optional[Exception] = None,
|
|
105
|
+
status: int = 429,
|
|
106
|
+
request_id: Optional[str] = None,
|
|
107
|
+
attempt: int = 1,
|
|
108
|
+
) -> None:
|
|
109
|
+
super().__init__(
|
|
110
|
+
message,
|
|
111
|
+
root_cause=root_cause,
|
|
112
|
+
status=status,
|
|
113
|
+
request_id=request_id,
|
|
114
|
+
attempt=attempt,
|
|
115
|
+
)
|
|
116
|
+
self.retry_after_ms = retry_after_ms
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
class TensorCostProviderError(TensorCostError):
|
|
120
|
+
"""The proxy reached the upstream provider, but the provider returned an
|
|
121
|
+
error (4xx / 5xx from OpenAI, Anthropic, etc.). ``status`` reflects the
|
|
122
|
+
provider's HTTP status code."""
|
tensorcost/_models.py
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""Internal type definitions for observation payloads."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Optional, TypedDict
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class Observation(TypedDict, total=False):
|
|
9
|
+
"""The wire shape posted to /api/proxy/observation.
|
|
10
|
+
|
|
11
|
+
Per TIER-2-SDK-PLAN §6. Fields are flat strings/ints to keep the
|
|
12
|
+
backend ingestion fast.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
sdk_version: str
|
|
16
|
+
provider: str # "openai" | "anthropic"
|
|
17
|
+
model: str
|
|
18
|
+
operation: str # "chat.completions" | "completions" | "messages"
|
|
19
|
+
request_at: str # ISO-8601
|
|
20
|
+
response_at: str # ISO-8601
|
|
21
|
+
input_tokens: Optional[int]
|
|
22
|
+
output_tokens: Optional[int]
|
|
23
|
+
cost_usd_cents: Optional[int] # SDK doesn't compute; backend fills
|
|
24
|
+
status: str # "success" | "error"
|
|
25
|
+
error_message: Optional[str]
|
|
26
|
+
correlation_id: str # uuid4
|
|
27
|
+
tenant_id: Optional[str]
|
|
28
|
+
# Phase A4 batch 3 — per-environment data scoping. Optional; the
|
|
29
|
+
# SDK omits the field when ``wrap()`` was not given an
|
|
30
|
+
# ``environment`` kwarg, and the backend then falls back to the
|
|
31
|
+
# column default ('production'). Free-text per
|
|
32
|
+
# project_environment_scoping.md (1..64 chars).
|
|
33
|
+
environment: Optional[str]
|
|
34
|
+
# Phase A4 batch 4 — provider-connection identifier. Optional; the
|
|
35
|
+
# SDK omits the field when ``wrap()`` was not given a
|
|
36
|
+
# ``connection_id`` kwarg. When present, ai-service uses it as the
|
|
37
|
+
# second tier of its env-resolution chain (look up
|
|
38
|
+
# `integration.cloud_account.environment` for the row whose `id`
|
|
39
|
+
# matches, behind a per-process 60s LRU cache).
|
|
40
|
+
connection_id: Optional[str]
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Provider-specific wrappers."""
|