tokenbee-sdk 1.3.0__tar.gz → 2.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tokenbee_sdk-1.3.0 → tokenbee_sdk-2.0.1}/PKG-INFO +10 -10
- {tokenbee_sdk-1.3.0 → tokenbee_sdk-2.0.1}/README.md +9 -9
- {tokenbee_sdk-1.3.0 → tokenbee_sdk-2.0.1}/pyproject.toml +1 -1
- {tokenbee_sdk-1.3.0 → tokenbee_sdk-2.0.1}/setup.py +1 -1
- {tokenbee_sdk-1.3.0 → tokenbee_sdk-2.0.1}/tokenbee/__init__.py +35 -18
- {tokenbee_sdk-1.3.0 → tokenbee_sdk-2.0.1}/tokenbee_sdk.egg-info/PKG-INFO +10 -10
- {tokenbee_sdk-1.3.0 → tokenbee_sdk-2.0.1}/setup.cfg +0 -0
- {tokenbee_sdk-1.3.0 → tokenbee_sdk-2.0.1}/tokenbee_sdk.egg-info/SOURCES.txt +0 -0
- {tokenbee_sdk-1.3.0 → tokenbee_sdk-2.0.1}/tokenbee_sdk.egg-info/dependency_links.txt +0 -0
- {tokenbee_sdk-1.3.0 → tokenbee_sdk-2.0.1}/tokenbee_sdk.egg-info/requires.txt +0 -0
- {tokenbee_sdk-1.3.0 → tokenbee_sdk-2.0.1}/tokenbee_sdk.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: tokenbee-sdk
|
|
3
|
-
Version:
|
|
3
|
+
Version: 2.0.1
|
|
4
4
|
Summary: Official Python SDK for TokenBee LLM inference gateway and observability.
|
|
5
5
|
Author: TokenBee Inc.
|
|
6
6
|
Author-email: "TokenBee Inc." <founders@tokenbee.io>
|
|
@@ -18,13 +18,13 @@ Dynamic: requires-python
|
|
|
18
18
|
|
|
19
19
|
# TokenBee Python SDK
|
|
20
20
|
|
|
21
|
-
Official Python SDK for [TokenBee](https://tokenbee.io)
|
|
21
|
+
Official Python SDK for [TokenBee](https://tokenbee.io) — AI interaction capture, audit, replay, and optimization.
|
|
22
22
|
|
|
23
23
|
## Features
|
|
24
24
|
|
|
25
25
|
- **Unified API**: Access multiple LLM providers (OpenAI, Anthropic, Google, Mistral, etc.) through a single interface.
|
|
26
26
|
- **Intelligent Compression**: Reduce token usage and latency with context-aware compression.
|
|
27
|
-
- **
|
|
27
|
+
- **Configurable Capture**: Retain interaction content when you need it; turn capture off per request or globally.
|
|
28
28
|
- **Built-in Observability**: Automatic tracking of latency, costs, and token usage.
|
|
29
29
|
|
|
30
30
|
## Links
|
|
@@ -54,7 +54,7 @@ client = TokenBee(
|
|
|
54
54
|
|
|
55
55
|
# Send a request
|
|
56
56
|
response = client.send(
|
|
57
|
-
model=TokenBeeModel.
|
|
57
|
+
model=TokenBeeModel.ANTHROPIC_CLAUDE_SONNET_4,
|
|
58
58
|
input={
|
|
59
59
|
"messages": [
|
|
60
60
|
{"role": "user", "content": "Explain quantum entanglement in simple terms."}
|
|
@@ -77,12 +77,12 @@ You can specify the compression rate and method per request. TokenBee uses an in
|
|
|
77
77
|
|
|
78
78
|
```python
|
|
79
79
|
response = client.send(
|
|
80
|
-
model=TokenBeeModel.
|
|
80
|
+
model=TokenBeeModel.ANTHROPIC_CLAUDE_SONNET_4,
|
|
81
81
|
input={
|
|
82
82
|
"messages": [...],
|
|
83
83
|
"compression": "auto", # "auto" (default), "on", or "off"
|
|
84
84
|
"rate": CompressionRate.HIGH, # MEDIUM (0.5), HIGH (0.33), etc.
|
|
85
|
-
"
|
|
85
|
+
"capture": False # per-request: do not retain message/response bodies
|
|
86
86
|
}
|
|
87
87
|
)
|
|
88
88
|
```
|
|
@@ -91,15 +91,15 @@ response = client.send(
|
|
|
91
91
|
- **`rate`**: Controls the aggressiveness of compression. `HIGH` aims for ~67% token reduction.
|
|
92
92
|
- **`sessionId`**: (Optional) String ID to group multiple requests into a single replayable session in the dashboard.
|
|
93
93
|
- **`userId`**: (Optional) String ID to track usage and costs per unique end-user.
|
|
94
|
-
- **`
|
|
94
|
+
- **`capture`**: (Optional) Per-request override. `False` does not retain messages/responses (metadata like tokens/cost/latency can still be recorded). Precedence: per-request → project setting → default on.
|
|
95
95
|
|
|
96
96
|
### Supported Models
|
|
97
97
|
|
|
98
98
|
The SDK provides a `TokenBeeModel` enum with popular models:
|
|
99
99
|
|
|
100
|
-
- `TokenBeeModel.
|
|
101
|
-
- `TokenBeeModel.
|
|
102
|
-
- `TokenBeeModel.
|
|
100
|
+
- `TokenBeeModel.ANTHROPIC_CLAUDE_SONNET_4`
|
|
101
|
+
- `TokenBeeModel.OPENAI_GPT_5_MINI`
|
|
102
|
+
- `TokenBeeModel.GROQ_GPT_OSS_20B`
|
|
103
103
|
- ... and many others.
|
|
104
104
|
|
|
105
105
|
## License
|
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
# TokenBee Python SDK
|
|
2
2
|
|
|
3
|
-
Official Python SDK for [TokenBee](https://tokenbee.io)
|
|
3
|
+
Official Python SDK for [TokenBee](https://tokenbee.io) — AI interaction capture, audit, replay, and optimization.
|
|
4
4
|
|
|
5
5
|
## Features
|
|
6
6
|
|
|
7
7
|
- **Unified API**: Access multiple LLM providers (OpenAI, Anthropic, Google, Mistral, etc.) through a single interface.
|
|
8
8
|
- **Intelligent Compression**: Reduce token usage and latency with context-aware compression.
|
|
9
|
-
- **
|
|
9
|
+
- **Configurable Capture**: Retain interaction content when you need it; turn capture off per request or globally.
|
|
10
10
|
- **Built-in Observability**: Automatic tracking of latency, costs, and token usage.
|
|
11
11
|
|
|
12
12
|
## Links
|
|
@@ -36,7 +36,7 @@ client = TokenBee(
|
|
|
36
36
|
|
|
37
37
|
# Send a request
|
|
38
38
|
response = client.send(
|
|
39
|
-
model=TokenBeeModel.
|
|
39
|
+
model=TokenBeeModel.ANTHROPIC_CLAUDE_SONNET_4,
|
|
40
40
|
input={
|
|
41
41
|
"messages": [
|
|
42
42
|
{"role": "user", "content": "Explain quantum entanglement in simple terms."}
|
|
@@ -59,12 +59,12 @@ You can specify the compression rate and method per request. TokenBee uses an in
|
|
|
59
59
|
|
|
60
60
|
```python
|
|
61
61
|
response = client.send(
|
|
62
|
-
model=TokenBeeModel.
|
|
62
|
+
model=TokenBeeModel.ANTHROPIC_CLAUDE_SONNET_4,
|
|
63
63
|
input={
|
|
64
64
|
"messages": [...],
|
|
65
65
|
"compression": "auto", # "auto" (default), "on", or "off"
|
|
66
66
|
"rate": CompressionRate.HIGH, # MEDIUM (0.5), HIGH (0.33), etc.
|
|
67
|
-
"
|
|
67
|
+
"capture": False # per-request: do not retain message/response bodies
|
|
68
68
|
}
|
|
69
69
|
)
|
|
70
70
|
```
|
|
@@ -73,15 +73,15 @@ response = client.send(
|
|
|
73
73
|
- **`rate`**: Controls the aggressiveness of compression. `HIGH` aims for ~67% token reduction.
|
|
74
74
|
- **`sessionId`**: (Optional) String ID to group multiple requests into a single replayable session in the dashboard.
|
|
75
75
|
- **`userId`**: (Optional) String ID to track usage and costs per unique end-user.
|
|
76
|
-
- **`
|
|
76
|
+
- **`capture`**: (Optional) Per-request override. `False` does not retain messages/responses (metadata like tokens/cost/latency can still be recorded). Precedence: per-request → project setting → default on.
|
|
77
77
|
|
|
78
78
|
### Supported Models
|
|
79
79
|
|
|
80
80
|
The SDK provides a `TokenBeeModel` enum with popular models:
|
|
81
81
|
|
|
82
|
-
- `TokenBeeModel.
|
|
83
|
-
- `TokenBeeModel.
|
|
84
|
-
- `TokenBeeModel.
|
|
82
|
+
- `TokenBeeModel.ANTHROPIC_CLAUDE_SONNET_4`
|
|
83
|
+
- `TokenBeeModel.OPENAI_GPT_5_MINI`
|
|
84
|
+
- `TokenBeeModel.GROQ_GPT_OSS_20B`
|
|
85
85
|
- ... and many others.
|
|
86
86
|
|
|
87
87
|
## License
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import os
|
|
2
2
|
import httpx
|
|
3
3
|
from enum import Enum
|
|
4
|
+
from typing import Optional
|
|
4
5
|
|
|
5
6
|
class CompressionRate(str, Enum):
|
|
6
7
|
LOW = "0.75"
|
|
@@ -20,28 +21,39 @@ class TokenBeeContext(str, Enum):
|
|
|
20
21
|
CODE = "code"
|
|
21
22
|
|
|
22
23
|
class TokenBeeModel(str, Enum):
|
|
23
|
-
# OpenAI
|
|
24
|
-
|
|
24
|
+
# OpenAI — current GPT-5.6 / GPT-5 family
|
|
25
|
+
OPENAI_GPT_5_6_SOL = "openai/gpt-5.6-sol"
|
|
26
|
+
OPENAI_GPT_5_6_TERRA = "openai/gpt-5.6-terra"
|
|
27
|
+
OPENAI_GPT_5_6_LUNA = "openai/gpt-5.6-luna"
|
|
28
|
+
OPENAI_GPT_5 = "openai/gpt-5"
|
|
29
|
+
OPENAI_GPT_5_MINI = "openai/gpt-5-mini"
|
|
30
|
+
OPENAI_GPT_5_NANO = "openai/gpt-5-nano"
|
|
31
|
+
OPENAI_GPT_5_4 = "openai/gpt-5.4"
|
|
32
|
+
OPENAI_GPT_5_4_MINI = "openai/gpt-5.4-mini"
|
|
33
|
+
OPENAI_GPT_5_4_NANO = "openai/gpt-5.4-nano"
|
|
34
|
+
OPENAI_GPT_4_1 = "openai/gpt-4.1"
|
|
35
|
+
OPENAI_GPT_4_1_MINI = "openai/gpt-4.1-mini"
|
|
36
|
+
OPENAI_GPT_4_1_NANO = "openai/gpt-4.1-nano"
|
|
25
37
|
OPENAI_GPT_4O = "openai/gpt-4o"
|
|
26
38
|
OPENAI_GPT_4O_MINI = "openai/gpt-4o-mini"
|
|
39
|
+
OPENAI_O3 = "openai/o3"
|
|
40
|
+
OPENAI_O3_MINI = "openai/o3-mini"
|
|
41
|
+
OPENAI_O4_MINI = "openai/o4-mini"
|
|
27
42
|
OPENAI_O1 = "openai/o1"
|
|
28
43
|
OPENAI_O1_MINI = "openai/o1-mini"
|
|
29
|
-
OPENAI_O3_MINI = "openai/o3-mini"
|
|
30
44
|
|
|
31
45
|
# Anthropic
|
|
46
|
+
ANTHROPIC_CLAUDE_SONNET_4 = "anthropic/claude-sonnet-4-latest"
|
|
47
|
+
ANTHROPIC_CLAUDE_OPUS_4 = "anthropic/claude-opus-4-latest"
|
|
48
|
+
ANTHROPIC_CLAUDE_HAIKU_4 = "anthropic/claude-haiku-4-latest"
|
|
32
49
|
ANTHROPIC_CLAUDE_3_7_SONNET = "anthropic/claude-3-7-sonnet-latest"
|
|
33
50
|
ANTHROPIC_CLAUDE_3_5_SONNET = "anthropic/claude-3-5-sonnet-latest"
|
|
34
51
|
ANTHROPIC_CLAUDE_3_5_HAIKU = "anthropic/claude-3-5-haiku-latest"
|
|
35
|
-
ANTHROPIC_CLAUDE_3_OPUS = "anthropic/claude-3-opus-latest"
|
|
36
52
|
|
|
37
53
|
# Google
|
|
38
|
-
GEMINI_3_1_PRO = "google/gemini-3.1-pro"
|
|
39
|
-
GEMINI_3_1_FLASH = "google/gemini-3.1-flash"
|
|
40
54
|
GEMINI_2_5_PRO = "google/gemini-2.5-pro"
|
|
55
|
+
GEMINI_2_5_FLASH = "google/gemini-2.5-flash"
|
|
41
56
|
GEMINI_2_0_FLASH = "google/gemini-2.0-flash"
|
|
42
|
-
GEMINI_2_0_PRO = "google/gemini-2.0-pro-exp"
|
|
43
|
-
GEMINI_1_5_PRO = "google/gemini-1.5-pro"
|
|
44
|
-
|
|
45
57
|
|
|
46
58
|
# Mistral
|
|
47
59
|
MISTRAL_LARGE = "mistral/mistral-large-latest"
|
|
@@ -54,10 +66,14 @@ class TokenBeeModel(str, Enum):
|
|
|
54
66
|
PERPLEXITY_SONAR_PRO = "perplexity/sonar-pro"
|
|
55
67
|
PERPLEXITY_SONAR_REASONING = "perplexity/sonar-reasoning"
|
|
56
68
|
|
|
57
|
-
# Groq
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
69
|
+
# Groq (current — Aug 2026)
|
|
70
|
+
# Note: model IDs after "groq/" are sent to Groq as-is (may include org prefix).
|
|
71
|
+
GROQ_GPT_OSS_120B = "groq/openai/gpt-oss-120b"
|
|
72
|
+
GROQ_GPT_OSS_20B = "groq/openai/gpt-oss-20b"
|
|
73
|
+
GROQ_GPT_OSS_SAFEGUARD_20B = "groq/openai/gpt-oss-safeguard-20b"
|
|
74
|
+
GROQ_QWEN3_6_27B = "groq/qwen/qwen3.6-27b"
|
|
75
|
+
GROQ_COMPOUND = "groq/groq/compound"
|
|
76
|
+
GROQ_COMPOUND_MINI = "groq/groq/compound-mini"
|
|
61
77
|
|
|
62
78
|
# xAI
|
|
63
79
|
XAI_GROK_3 = "xai/grok-3"
|
|
@@ -75,7 +91,7 @@ class TokenBee:
|
|
|
75
91
|
context: str = TokenBeeContext.AUTO,
|
|
76
92
|
model: str = "",
|
|
77
93
|
provider: str = "",
|
|
78
|
-
|
|
94
|
+
capture: Optional[bool] = None
|
|
79
95
|
):
|
|
80
96
|
self.api_key = api_key
|
|
81
97
|
self.llm_key = llm_key
|
|
@@ -89,8 +105,9 @@ class TokenBee:
|
|
|
89
105
|
"X-TokenBee-Rate": rate,
|
|
90
106
|
"X-TokenBee-Strategy": strategy,
|
|
91
107
|
"X-TokenBee-Context": context,
|
|
92
|
-
"X-TokenBee-Privacy": str(privacy).lower()
|
|
93
108
|
}
|
|
109
|
+
if capture is not None:
|
|
110
|
+
self.headers["X-TokenBee-Capture"] = str(capture).lower()
|
|
94
111
|
if model:
|
|
95
112
|
self.headers["X-TokenBee-Model"] = model
|
|
96
113
|
if provider:
|
|
@@ -115,8 +132,8 @@ class TokenBee:
|
|
|
115
132
|
headers["X-TokenBee-Strategy"] = str(input["strategy"])
|
|
116
133
|
if "context" in input:
|
|
117
134
|
headers["X-TokenBee-Context"] = str(input["context"])
|
|
118
|
-
if "
|
|
119
|
-
headers["X-TokenBee-
|
|
135
|
+
if "capture" in input:
|
|
136
|
+
headers["X-TokenBee-Capture"] = str(input["capture"]).lower()
|
|
120
137
|
if "sessionId" in input:
|
|
121
138
|
headers["X-TB-Session-Id"] = str(input["sessionId"])
|
|
122
139
|
if "userId" in input:
|
|
@@ -128,7 +145,7 @@ class TokenBee:
|
|
|
128
145
|
payload.pop("rate", None)
|
|
129
146
|
payload.pop("strategy", None)
|
|
130
147
|
payload.pop("context", None)
|
|
131
|
-
payload.pop("
|
|
148
|
+
payload.pop("capture", None)
|
|
132
149
|
payload.pop("sessionId", None)
|
|
133
150
|
payload.pop("userId", None)
|
|
134
151
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: tokenbee-sdk
|
|
3
|
-
Version:
|
|
3
|
+
Version: 2.0.1
|
|
4
4
|
Summary: Official Python SDK for TokenBee LLM inference gateway and observability.
|
|
5
5
|
Author: TokenBee Inc.
|
|
6
6
|
Author-email: "TokenBee Inc." <founders@tokenbee.io>
|
|
@@ -18,13 +18,13 @@ Dynamic: requires-python
|
|
|
18
18
|
|
|
19
19
|
# TokenBee Python SDK
|
|
20
20
|
|
|
21
|
-
Official Python SDK for [TokenBee](https://tokenbee.io)
|
|
21
|
+
Official Python SDK for [TokenBee](https://tokenbee.io) — AI interaction capture, audit, replay, and optimization.
|
|
22
22
|
|
|
23
23
|
## Features
|
|
24
24
|
|
|
25
25
|
- **Unified API**: Access multiple LLM providers (OpenAI, Anthropic, Google, Mistral, etc.) through a single interface.
|
|
26
26
|
- **Intelligent Compression**: Reduce token usage and latency with context-aware compression.
|
|
27
|
-
- **
|
|
27
|
+
- **Configurable Capture**: Retain interaction content when you need it; turn capture off per request or globally.
|
|
28
28
|
- **Built-in Observability**: Automatic tracking of latency, costs, and token usage.
|
|
29
29
|
|
|
30
30
|
## Links
|
|
@@ -54,7 +54,7 @@ client = TokenBee(
|
|
|
54
54
|
|
|
55
55
|
# Send a request
|
|
56
56
|
response = client.send(
|
|
57
|
-
model=TokenBeeModel.
|
|
57
|
+
model=TokenBeeModel.ANTHROPIC_CLAUDE_SONNET_4,
|
|
58
58
|
input={
|
|
59
59
|
"messages": [
|
|
60
60
|
{"role": "user", "content": "Explain quantum entanglement in simple terms."}
|
|
@@ -77,12 +77,12 @@ You can specify the compression rate and method per request. TokenBee uses an in
|
|
|
77
77
|
|
|
78
78
|
```python
|
|
79
79
|
response = client.send(
|
|
80
|
-
model=TokenBeeModel.
|
|
80
|
+
model=TokenBeeModel.ANTHROPIC_CLAUDE_SONNET_4,
|
|
81
81
|
input={
|
|
82
82
|
"messages": [...],
|
|
83
83
|
"compression": "auto", # "auto" (default), "on", or "off"
|
|
84
84
|
"rate": CompressionRate.HIGH, # MEDIUM (0.5), HIGH (0.33), etc.
|
|
85
|
-
"
|
|
85
|
+
"capture": False # per-request: do not retain message/response bodies
|
|
86
86
|
}
|
|
87
87
|
)
|
|
88
88
|
```
|
|
@@ -91,15 +91,15 @@ response = client.send(
|
|
|
91
91
|
- **`rate`**: Controls the aggressiveness of compression. `HIGH` aims for ~67% token reduction.
|
|
92
92
|
- **`sessionId`**: (Optional) String ID to group multiple requests into a single replayable session in the dashboard.
|
|
93
93
|
- **`userId`**: (Optional) String ID to track usage and costs per unique end-user.
|
|
94
|
-
- **`
|
|
94
|
+
- **`capture`**: (Optional) Per-request override. `False` does not retain messages/responses (metadata like tokens/cost/latency can still be recorded). Precedence: per-request → project setting → default on.
|
|
95
95
|
|
|
96
96
|
### Supported Models
|
|
97
97
|
|
|
98
98
|
The SDK provides a `TokenBeeModel` enum with popular models:
|
|
99
99
|
|
|
100
|
-
- `TokenBeeModel.
|
|
101
|
-
- `TokenBeeModel.
|
|
102
|
-
- `TokenBeeModel.
|
|
100
|
+
- `TokenBeeModel.ANTHROPIC_CLAUDE_SONNET_4`
|
|
101
|
+
- `TokenBeeModel.OPENAI_GPT_5_MINI`
|
|
102
|
+
- `TokenBeeModel.GROQ_GPT_OSS_20B`
|
|
103
103
|
- ... and many others.
|
|
104
104
|
|
|
105
105
|
## License
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|