tokenbee-sdk 2.0.0__tar.gz → 2.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tokenbee_sdk-2.0.0 → tokenbee_sdk-2.0.1}/PKG-INFO +5 -5
- {tokenbee_sdk-2.0.0 → tokenbee_sdk-2.0.1}/README.md +4 -4
- {tokenbee_sdk-2.0.0 → tokenbee_sdk-2.0.1}/pyproject.toml +1 -1
- {tokenbee_sdk-2.0.0 → tokenbee_sdk-2.0.1}/setup.py +1 -1
- {tokenbee_sdk-2.0.0 → tokenbee_sdk-2.0.1}/tokenbee/__init__.py +7 -5
- {tokenbee_sdk-2.0.0 → tokenbee_sdk-2.0.1}/tokenbee_sdk.egg-info/PKG-INFO +5 -5
- {tokenbee_sdk-2.0.0 → tokenbee_sdk-2.0.1}/setup.cfg +0 -0
- {tokenbee_sdk-2.0.0 → tokenbee_sdk-2.0.1}/tokenbee_sdk.egg-info/SOURCES.txt +0 -0
- {tokenbee_sdk-2.0.0 → tokenbee_sdk-2.0.1}/tokenbee_sdk.egg-info/dependency_links.txt +0 -0
- {tokenbee_sdk-2.0.0 → tokenbee_sdk-2.0.1}/tokenbee_sdk.egg-info/requires.txt +0 -0
- {tokenbee_sdk-2.0.0 → tokenbee_sdk-2.0.1}/tokenbee_sdk.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: tokenbee-sdk
|
|
3
|
-
Version: 2.0.
|
|
3
|
+
Version: 2.0.1
|
|
4
4
|
Summary: Official Python SDK for TokenBee LLM inference gateway and observability.
|
|
5
5
|
Author: TokenBee Inc.
|
|
6
6
|
Author-email: "TokenBee Inc." <founders@tokenbee.io>
|
|
@@ -18,13 +18,13 @@ Dynamic: requires-python
|
|
|
18
18
|
|
|
19
19
|
# TokenBee Python SDK
|
|
20
20
|
|
|
21
|
-
Official Python SDK for [TokenBee](https://tokenbee.io)
|
|
21
|
+
Official Python SDK for [TokenBee](https://tokenbee.io) — AI interaction capture, audit, replay, and optimization.
|
|
22
22
|
|
|
23
23
|
## Features
|
|
24
24
|
|
|
25
25
|
- **Unified API**: Access multiple LLM providers (OpenAI, Anthropic, Google, Mistral, etc.) through a single interface.
|
|
26
26
|
- **Intelligent Compression**: Reduce token usage and latency with context-aware compression.
|
|
27
|
-
- **
|
|
27
|
+
- **Configurable Capture**: Retain interaction content when you need it; turn capture off per request or globally.
|
|
28
28
|
- **Built-in Observability**: Automatic tracking of latency, costs, and token usage.
|
|
29
29
|
|
|
30
30
|
## Links
|
|
@@ -82,7 +82,7 @@ response = client.send(
|
|
|
82
82
|
"messages": [...],
|
|
83
83
|
"compression": "auto", # "auto" (default), "on", or "off"
|
|
84
84
|
"rate": CompressionRate.HIGH, # MEDIUM (0.5), HIGH (0.33), etc.
|
|
85
|
-
"
|
|
85
|
+
"capture": False # per-request: do not retain message/response bodies
|
|
86
86
|
}
|
|
87
87
|
)
|
|
88
88
|
```
|
|
@@ -91,7 +91,7 @@ response = client.send(
|
|
|
91
91
|
- **`rate`**: Controls the aggressiveness of compression. `HIGH` aims for ~67% token reduction.
|
|
92
92
|
- **`sessionId`**: (Optional) String ID to group multiple requests into a single replayable session in the dashboard.
|
|
93
93
|
- **`userId`**: (Optional) String ID to track usage and costs per unique end-user.
|
|
94
|
-
- **`
|
|
94
|
+
- **`capture`**: (Optional) Per-request override. `False` does not retain messages/responses (metadata like tokens/cost/latency can still be recorded). Precedence: per-request → project setting → default on.
|
|
95
95
|
|
|
96
96
|
### Supported Models
|
|
97
97
|
|
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
# TokenBee Python SDK
|
|
2
2
|
|
|
3
|
-
Official Python SDK for [TokenBee](https://tokenbee.io)
|
|
3
|
+
Official Python SDK for [TokenBee](https://tokenbee.io) — AI interaction capture, audit, replay, and optimization.
|
|
4
4
|
|
|
5
5
|
## Features
|
|
6
6
|
|
|
7
7
|
- **Unified API**: Access multiple LLM providers (OpenAI, Anthropic, Google, Mistral, etc.) through a single interface.
|
|
8
8
|
- **Intelligent Compression**: Reduce token usage and latency with context-aware compression.
|
|
9
|
-
- **
|
|
9
|
+
- **Configurable Capture**: Retain interaction content when you need it; turn capture off per request or globally.
|
|
10
10
|
- **Built-in Observability**: Automatic tracking of latency, costs, and token usage.
|
|
11
11
|
|
|
12
12
|
## Links
|
|
@@ -64,7 +64,7 @@ response = client.send(
|
|
|
64
64
|
"messages": [...],
|
|
65
65
|
"compression": "auto", # "auto" (default), "on", or "off"
|
|
66
66
|
"rate": CompressionRate.HIGH, # MEDIUM (0.5), HIGH (0.33), etc.
|
|
67
|
-
"
|
|
67
|
+
"capture": False # per-request: do not retain message/response bodies
|
|
68
68
|
}
|
|
69
69
|
)
|
|
70
70
|
```
|
|
@@ -73,7 +73,7 @@ response = client.send(
|
|
|
73
73
|
- **`rate`**: Controls the aggressiveness of compression. `HIGH` aims for ~67% token reduction.
|
|
74
74
|
- **`sessionId`**: (Optional) String ID to group multiple requests into a single replayable session in the dashboard.
|
|
75
75
|
- **`userId`**: (Optional) String ID to track usage and costs per unique end-user.
|
|
76
|
-
- **`
|
|
76
|
+
- **`capture`**: (Optional) Per-request override. `False` does not retain messages/responses (metadata like tokens/cost/latency can still be recorded). Precedence: per-request → project setting → default on.
|
|
77
77
|
|
|
78
78
|
### Supported Models
|
|
79
79
|
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import os
|
|
2
2
|
import httpx
|
|
3
3
|
from enum import Enum
|
|
4
|
+
from typing import Optional
|
|
4
5
|
|
|
5
6
|
class CompressionRate(str, Enum):
|
|
6
7
|
LOW = "0.75"
|
|
@@ -90,7 +91,7 @@ class TokenBee:
|
|
|
90
91
|
context: str = TokenBeeContext.AUTO,
|
|
91
92
|
model: str = "",
|
|
92
93
|
provider: str = "",
|
|
93
|
-
|
|
94
|
+
capture: Optional[bool] = None
|
|
94
95
|
):
|
|
95
96
|
self.api_key = api_key
|
|
96
97
|
self.llm_key = llm_key
|
|
@@ -104,8 +105,9 @@ class TokenBee:
|
|
|
104
105
|
"X-TokenBee-Rate": rate,
|
|
105
106
|
"X-TokenBee-Strategy": strategy,
|
|
106
107
|
"X-TokenBee-Context": context,
|
|
107
|
-
"X-TokenBee-Privacy": str(privacy).lower()
|
|
108
108
|
}
|
|
109
|
+
if capture is not None:
|
|
110
|
+
self.headers["X-TokenBee-Capture"] = str(capture).lower()
|
|
109
111
|
if model:
|
|
110
112
|
self.headers["X-TokenBee-Model"] = model
|
|
111
113
|
if provider:
|
|
@@ -130,8 +132,8 @@ class TokenBee:
|
|
|
130
132
|
headers["X-TokenBee-Strategy"] = str(input["strategy"])
|
|
131
133
|
if "context" in input:
|
|
132
134
|
headers["X-TokenBee-Context"] = str(input["context"])
|
|
133
|
-
if "
|
|
134
|
-
headers["X-TokenBee-
|
|
135
|
+
if "capture" in input:
|
|
136
|
+
headers["X-TokenBee-Capture"] = str(input["capture"]).lower()
|
|
135
137
|
if "sessionId" in input:
|
|
136
138
|
headers["X-TB-Session-Id"] = str(input["sessionId"])
|
|
137
139
|
if "userId" in input:
|
|
@@ -143,7 +145,7 @@ class TokenBee:
|
|
|
143
145
|
payload.pop("rate", None)
|
|
144
146
|
payload.pop("strategy", None)
|
|
145
147
|
payload.pop("context", None)
|
|
146
|
-
payload.pop("
|
|
148
|
+
payload.pop("capture", None)
|
|
147
149
|
payload.pop("sessionId", None)
|
|
148
150
|
payload.pop("userId", None)
|
|
149
151
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: tokenbee-sdk
|
|
3
|
-
Version: 2.0.
|
|
3
|
+
Version: 2.0.1
|
|
4
4
|
Summary: Official Python SDK for TokenBee LLM inference gateway and observability.
|
|
5
5
|
Author: TokenBee Inc.
|
|
6
6
|
Author-email: "TokenBee Inc." <founders@tokenbee.io>
|
|
@@ -18,13 +18,13 @@ Dynamic: requires-python
|
|
|
18
18
|
|
|
19
19
|
# TokenBee Python SDK
|
|
20
20
|
|
|
21
|
-
Official Python SDK for [TokenBee](https://tokenbee.io)
|
|
21
|
+
Official Python SDK for [TokenBee](https://tokenbee.io) — AI interaction capture, audit, replay, and optimization.
|
|
22
22
|
|
|
23
23
|
## Features
|
|
24
24
|
|
|
25
25
|
- **Unified API**: Access multiple LLM providers (OpenAI, Anthropic, Google, Mistral, etc.) through a single interface.
|
|
26
26
|
- **Intelligent Compression**: Reduce token usage and latency with context-aware compression.
|
|
27
|
-
- **
|
|
27
|
+
- **Configurable Capture**: Retain interaction content when you need it; turn capture off per request or globally.
|
|
28
28
|
- **Built-in Observability**: Automatic tracking of latency, costs, and token usage.
|
|
29
29
|
|
|
30
30
|
## Links
|
|
@@ -82,7 +82,7 @@ response = client.send(
|
|
|
82
82
|
"messages": [...],
|
|
83
83
|
"compression": "auto", # "auto" (default), "on", or "off"
|
|
84
84
|
"rate": CompressionRate.HIGH, # MEDIUM (0.5), HIGH (0.33), etc.
|
|
85
|
-
"
|
|
85
|
+
"capture": False # per-request: do not retain message/response bodies
|
|
86
86
|
}
|
|
87
87
|
)
|
|
88
88
|
```
|
|
@@ -91,7 +91,7 @@ response = client.send(
|
|
|
91
91
|
- **`rate`**: Controls the aggressiveness of compression. `HIGH` aims for ~67% token reduction.
|
|
92
92
|
- **`sessionId`**: (Optional) String ID to group multiple requests into a single replayable session in the dashboard.
|
|
93
93
|
- **`userId`**: (Optional) String ID to track usage and costs per unique end-user.
|
|
94
|
-
- **`
|
|
94
|
+
- **`capture`**: (Optional) Per-request override. `False` does not retain messages/responses (metadata like tokens/cost/latency can still be recorded). Precedence: per-request → project setting → default on.
|
|
95
95
|
|
|
96
96
|
### Supported Models
|
|
97
97
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|