subproto 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fakeup/__init__.py +1 -0
- fakeup/server.py +295 -0
- subproto/__init__.py +7 -0
- subproto/__main__.py +5 -0
- subproto/cli.py +1493 -0
- subproto/compiler.py +431 -0
- subproto/config.py +216 -0
- subproto/dataset.py +330 -0
- subproto/demo.py +260 -0
- subproto/doctor.py +282 -0
- subproto/engine.py +352 -0
- subproto/graph.py +502 -0
- subproto/heuristics.py +177 -0
- subproto/implicit.py +534 -0
- subproto/laya.py +19 -0
- subproto/laya_server.py +389 -0
- subproto/live.py +149 -0
- subproto/pricing.py +43 -0
- subproto/protocol.py +342 -0
- subproto/proxy.py +349 -0
- subproto/report.py +509 -0
- subproto/retrain.py +541 -0
- subproto/style.py +292 -0
- subproto/systemone/__init__.py +28 -0
- subproto/systemone/base.py +123 -0
- subproto/systemone/evidence/ablation.json +404 -0
- subproto/systemone/registry.py +212 -0
- subproto/systemone/router.py +192 -0
- subproto/systemone/versions.py +322 -0
- subproto/telemetry.py +147 -0
- subproto/usage.py +268 -0
- subproto-0.1.0.dist-info/METADATA +708 -0
- subproto-0.1.0.dist-info/RECORD +37 -0
- subproto-0.1.0.dist-info/WHEEL +5 -0
- subproto-0.1.0.dist-info/entry_points.txt +2 -0
- subproto-0.1.0.dist-info/licenses/LICENSE +204 -0
- subproto-0.1.0.dist-info/top_level.txt +2 -0
fakeup/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Mock upstream used for tests and `subproto demo`."""
|
fakeup/server.py
ADDED
|
@@ -0,0 +1,295 @@
|
|
|
1
|
+
"""A mock upstream that speaks just enough of both dialects to test the proxy:
|
|
2
|
+
streamed SSE with realistic usage events, non-streamed JSON, and errors.
|
|
3
|
+
|
|
4
|
+
It used to answer every request with the same usage block. That made the whole billing
|
|
5
|
+
side of the tool unfalsifiable: `subproto report` printed the identical `billed input`
|
|
6
|
+
and `spend` whether the slots had cut 40 % of the request or nothing at all, while
|
|
7
|
+
`subproto live` on the same traffic said the tokens were delivered — two pages of one
|
|
8
|
+
product disagreeing about whether anything happened (S46/B4). The mock now prices the
|
|
9
|
+
body it was actually sent, and it models a prompt cache the way a provider does: the
|
|
10
|
+
part of this request that is the same string as the last one is a cache read, so a
|
|
11
|
+
proxy that rewrites the cached prefix is billed for it (I2 stops being a claim and
|
|
12
|
+
becomes something the meter can see).
|
|
13
|
+
|
|
14
|
+
Output is still a tariff constant: the mock answers "done" and does not measure what it
|
|
15
|
+
wrote, so no page should read an output figure as a result.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
import json
|
|
19
|
+
import os
|
|
20
|
+
import threading
|
|
21
|
+
import time
|
|
22
|
+
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
|
23
|
+
|
|
24
|
+
SLEEP = float(os.environ.get("MOCKUP_SLEEP", "0") or 0)
|
|
25
|
+
|
|
26
|
+
# The same characters-per-token the proxy's own estimator uses, so the two agree on what
|
|
27
|
+
# a request weighs rather than differing by an unexplained factor.
|
|
28
|
+
TOK_CHARS = 3.6
|
|
29
|
+
|
|
30
|
+
# Per dialect: what the mock charges for an answer, and the shape it answers in.
|
|
31
|
+
TARIFFS = {
|
|
32
|
+
"anthropic": {"output": 320, "reasoning": 0},
|
|
33
|
+
"openai": {"output": 320, "reasoning": 0},
|
|
34
|
+
"gemini": {"output": 410, "reasoning": 90},
|
|
35
|
+
"responses": {"output": 210, "reasoning": 64},
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _tok(text):
|
|
40
|
+
return int(len(text or "") / TOK_CHARS + 0.5)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def prompt_text(body):
|
|
44
|
+
"""The request as one string, in the order a provider reads it."""
|
|
45
|
+
parts = []
|
|
46
|
+
if isinstance(body.get("system"), str):
|
|
47
|
+
parts.append(body["system"])
|
|
48
|
+
if isinstance(body.get("instructions"), str):
|
|
49
|
+
parts.append(body["instructions"])
|
|
50
|
+
for tool in body.get("tools") or []:
|
|
51
|
+
parts.append(json.dumps(tool, sort_keys=True))
|
|
52
|
+
for key in ("messages", "input"):
|
|
53
|
+
for item in body.get(key) or []:
|
|
54
|
+
parts.append(item if isinstance(item, str) else json.dumps(item, sort_keys=True))
|
|
55
|
+
if isinstance(body.get("prompt"), str):
|
|
56
|
+
parts.append(body["prompt"])
|
|
57
|
+
if isinstance(body.get("contents"), (list, str)):
|
|
58
|
+
parts.append(json.dumps(body["contents"], sort_keys=True))
|
|
59
|
+
return "\n".join(parts)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _common_prefix(a, b):
|
|
63
|
+
n = 0
|
|
64
|
+
for x, y in zip(a, b):
|
|
65
|
+
if x != y:
|
|
66
|
+
break
|
|
67
|
+
n += 1
|
|
68
|
+
return n
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def price(body, tariff, previous=""):
|
|
72
|
+
"""Usage for this body, given the text the same session sent last time."""
|
|
73
|
+
text = prompt_text(body)
|
|
74
|
+
total = _tok(text)
|
|
75
|
+
cached = _tok(text[:_common_prefix(text, previous)])
|
|
76
|
+
out = dict(TARIFFS[tariff])
|
|
77
|
+
out.update({"model": body.get("model") or "mock-model",
|
|
78
|
+
"input_uncached": max(0, total - cached),
|
|
79
|
+
"cache_write": 0, "cache_read": cached})
|
|
80
|
+
return out
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def anthropic_stream(usage):
|
|
84
|
+
events = [
|
|
85
|
+
("message_start", {"type": "message_start", "message": {
|
|
86
|
+
"id": "mock", "type": "message", "role": "assistant",
|
|
87
|
+
"model": usage["model"], "content": [], "usage": {
|
|
88
|
+
"input_tokens": usage["input_uncached"],
|
|
89
|
+
"cache_creation_input_tokens": usage["cache_write"],
|
|
90
|
+
"cache_read_input_tokens": usage["cache_read"],
|
|
91
|
+
"output_tokens": 4}}}),
|
|
92
|
+
("content_block_start", {"type": "content_block_start", "index": 0,
|
|
93
|
+
"content_block": {"type": "text", "text": ""}}),
|
|
94
|
+
("content_block_delta", {"type": "content_block_delta", "index": 0,
|
|
95
|
+
"delta": {"type": "text_delta", "text": "done"}}),
|
|
96
|
+
("content_block_stop", {"type": "content_block_stop", "index": 0}),
|
|
97
|
+
("message_delta", {"type": "message_delta",
|
|
98
|
+
"delta": {"stop_reason": "end_turn"},
|
|
99
|
+
"usage": {"output_tokens": usage["output"]}}),
|
|
100
|
+
("message_stop", {"type": "message_stop"}),
|
|
101
|
+
]
|
|
102
|
+
for name, payload in events:
|
|
103
|
+
yield "event: %s\ndata: %s\n\n" % (name, json.dumps(payload))
|
|
104
|
+
if SLEEP:
|
|
105
|
+
time.sleep(SLEEP)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def openai_stream(usage):
|
|
109
|
+
for chunk in ({"id": "mock", "object": "chat.completion.chunk", "model": usage["model"],
|
|
110
|
+
"choices": [{"index": 0, "delta": {"content": "done"}, "finish_reason": None}]},
|
|
111
|
+
{"id": "mock", "object": "chat.completion.chunk", "model": usage["model"],
|
|
112
|
+
"choices": [], "usage": {
|
|
113
|
+
"prompt_tokens": usage["input_uncached"] + usage["cache_read"]
|
|
114
|
+
+ usage["cache_write"],
|
|
115
|
+
"completion_tokens": usage["output"],
|
|
116
|
+
"prompt_tokens_details": {"cached_tokens": usage["cache_read"]},
|
|
117
|
+
"completion_tokens_details": {"reasoning_tokens": usage["reasoning"]}}}):
|
|
118
|
+
yield "data: %s\n\n" % json.dumps(chunk)
|
|
119
|
+
if SLEEP:
|
|
120
|
+
time.sleep(SLEEP)
|
|
121
|
+
yield "data: [DONE]\n\n"
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _responses_obj(usage):
|
|
125
|
+
return {"id": "resp_mock", "object": "response", "model": usage["model"],
|
|
126
|
+
"status": "completed",
|
|
127
|
+
"output": [{"type": "message", "role": "assistant",
|
|
128
|
+
"content": [{"type": "output_text", "text": "done"}]}],
|
|
129
|
+
"usage": {
|
|
130
|
+
"input_tokens": usage["input_uncached"] + usage["cache_read"],
|
|
131
|
+
"output_tokens": usage["output"],
|
|
132
|
+
"input_tokens_details": {"cached_tokens": usage["cache_read"]},
|
|
133
|
+
"output_tokens_details": {"reasoning_tokens": usage["reasoning"]}}}
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def responses_stream(usage):
|
|
137
|
+
# OpenAI /v1/responses streams lifecycle events; usage rides on the final
|
|
138
|
+
# `response.completed` event, so the parser must pick it up from there.
|
|
139
|
+
for name, payload in (
|
|
140
|
+
("response.created", {"type": "response.created",
|
|
141
|
+
"response": {"id": "resp_mock", "model": usage["model"],
|
|
142
|
+
"status": "in_progress"}}),
|
|
143
|
+
("response.output_text.delta", {"type": "response.output_text.delta", "delta": "done"}),
|
|
144
|
+
("response.completed", {"type": "response.completed", "response": _responses_obj(usage)}),
|
|
145
|
+
):
|
|
146
|
+
yield "event: %s\ndata: %s\n\n" % (name, json.dumps(payload))
|
|
147
|
+
if SLEEP:
|
|
148
|
+
time.sleep(SLEEP)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _anthropic_obj(usage, echo=None):
|
|
152
|
+
obj = {"id": "mock", "type": "message", "role": "assistant",
|
|
153
|
+
"model": usage["model"], "content": [{"type": "text", "text": "done"}],
|
|
154
|
+
"usage": {"input_tokens": usage["input_uncached"],
|
|
155
|
+
"cache_creation_input_tokens": usage["cache_write"],
|
|
156
|
+
"cache_read_input_tokens": usage["cache_read"],
|
|
157
|
+
"output_tokens": usage["output"]}}
|
|
158
|
+
if echo is not None:
|
|
159
|
+
obj["_mock_echo"] = echo
|
|
160
|
+
return obj
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _gemini_obj(usage):
|
|
164
|
+
return {"candidates": [{"content": {"role": "model",
|
|
165
|
+
"parts": [{"text": "done"}]},
|
|
166
|
+
"finishReason": "STOP"}],
|
|
167
|
+
"modelVersion": usage["model"],
|
|
168
|
+
"usageMetadata": {
|
|
169
|
+
"promptTokenCount": usage["input_uncached"] + usage["cache_read"],
|
|
170
|
+
"candidatesTokenCount": usage["output"],
|
|
171
|
+
"thoughtsTokenCount": usage["reasoning"],
|
|
172
|
+
"cachedContentTokenCount": usage["cache_read"]}}
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def gemini_stream(usage):
|
|
176
|
+
# Gemini streams `:streamGenerateContent` as SSE `data:` frames; each carries
|
|
177
|
+
# cumulative usageMetadata, so the last frame is authoritative.
|
|
178
|
+
yield "data: %s\n\n" % json.dumps(
|
|
179
|
+
{"candidates": [{"content": {"role": "model", "parts": [{"text": "do"}]}}]})
|
|
180
|
+
yield "data: %s\n\n" % json.dumps(_gemini_obj(usage))
|
|
181
|
+
if SLEEP:
|
|
182
|
+
time.sleep(SLEEP)
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _sse(self):
|
|
186
|
+
self.send_response(200)
|
|
187
|
+
self.send_header("content-type", "text/event-stream")
|
|
188
|
+
self.send_header("connection", "close")
|
|
189
|
+
self.close_connection = True
|
|
190
|
+
self.end_headers()
|
|
191
|
+
return self.wfile
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
class Handler(BaseHTTPRequestHandler):
|
|
195
|
+
protocol_version = "HTTP/1.1"
|
|
196
|
+
|
|
197
|
+
def log_message(self, *a):
|
|
198
|
+
pass
|
|
199
|
+
|
|
200
|
+
def _json(self, obj, status=200):
|
|
201
|
+
raw = json.dumps(obj).encode()
|
|
202
|
+
self.send_response(status)
|
|
203
|
+
self.send_header("content-type", "application/json")
|
|
204
|
+
self.send_header("content-length", str(len(raw)))
|
|
205
|
+
self.end_headers()
|
|
206
|
+
self.wfile.write(raw)
|
|
207
|
+
|
|
208
|
+
def do_POST(self):
|
|
209
|
+
raw = self.rfile.read(int(self.headers.get("content-length") or 0))
|
|
210
|
+
try:
|
|
211
|
+
body = json.loads(raw.decode())
|
|
212
|
+
except ValueError:
|
|
213
|
+
return self._json({"error": {"message": "mock: bad json"}}, 400)
|
|
214
|
+
if (self.headers.get("x-force-error") or "") == "1":
|
|
215
|
+
return self._json({"type": "error", "error": {
|
|
216
|
+
"type": "overloaded_error", "message": "mock overloaded"}}, 529)
|
|
217
|
+
model = body.get("model") or "mock-model"
|
|
218
|
+
stream = bool(body.get("stream"))
|
|
219
|
+
is_gemini = "generateContent" in self.path or "streamGenerateContent" in self.path
|
|
220
|
+
is_responses = "/responses" in self.path
|
|
221
|
+
is_anthropic = "/messages" in self.path
|
|
222
|
+
tariff = ("gemini" if is_gemini else "responses" if is_responses
|
|
223
|
+
else "anthropic" if is_anthropic else "openai")
|
|
224
|
+
if is_gemini:
|
|
225
|
+
stream = stream or "streamGenerateContent" in self.path
|
|
226
|
+
text = prompt_text(body)
|
|
227
|
+
key = (tariff, model)
|
|
228
|
+
seen = getattr(self.server, "seen", None)
|
|
229
|
+
lock = getattr(self.server, "seen_lock", None)
|
|
230
|
+
if seen is None: # a bare handler, not behind MockUpstream
|
|
231
|
+
seen, lock = {}, threading.Lock()
|
|
232
|
+
self.server.seen, self.server.seen_lock = seen, lock
|
|
233
|
+
with lock:
|
|
234
|
+
previous = seen.get(key, "")
|
|
235
|
+
usage = price(body, tariff, previous)
|
|
236
|
+
seen[key] = text
|
|
237
|
+
if is_gemini:
|
|
238
|
+
if stream:
|
|
239
|
+
wfile = _sse(self)
|
|
240
|
+
for line in gemini_stream(usage):
|
|
241
|
+
wfile.write(line.encode())
|
|
242
|
+
wfile.flush()
|
|
243
|
+
return
|
|
244
|
+
return self._json(_gemini_obj(usage))
|
|
245
|
+
if is_responses:
|
|
246
|
+
if stream:
|
|
247
|
+
wfile = _sse(self)
|
|
248
|
+
for line in responses_stream(usage):
|
|
249
|
+
wfile.write(line.encode())
|
|
250
|
+
wfile.flush()
|
|
251
|
+
return
|
|
252
|
+
return self._json(_responses_obj(usage))
|
|
253
|
+
if is_anthropic:
|
|
254
|
+
if stream:
|
|
255
|
+
wfile = _sse(self)
|
|
256
|
+
for line in anthropic_stream(usage):
|
|
257
|
+
wfile.write(line.encode())
|
|
258
|
+
wfile.flush()
|
|
259
|
+
return
|
|
260
|
+
return self._json(_anthropic_obj(usage, body if body.get("_mock_echo") else None))
|
|
261
|
+
if stream:
|
|
262
|
+
wfile = _sse(self)
|
|
263
|
+
for line in openai_stream(usage):
|
|
264
|
+
wfile.write(line.encode())
|
|
265
|
+
wfile.flush()
|
|
266
|
+
return
|
|
267
|
+
return self._json({"id": "mock", "object": "chat.completion", "model": model,
|
|
268
|
+
"choices": [{"index": 0, "message": {"role": "assistant",
|
|
269
|
+
"content": "done"}, "finish_reason": "stop"}],
|
|
270
|
+
"usage": {"prompt_tokens": usage["input_uncached"] + usage["cache_read"],
|
|
271
|
+
"completion_tokens": usage["output"],
|
|
272
|
+
"prompt_tokens_details": {"cached_tokens": usage["cache_read"]}}})
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
class MockUpstream(ThreadingHTTPServer):
|
|
276
|
+
daemon_threads = True
|
|
277
|
+
allow_reuse_address = True
|
|
278
|
+
|
|
279
|
+
def __init__(self, port=8799):
|
|
280
|
+
ThreadingHTTPServer.__init__(self, ("127.0.0.1", port), Handler)
|
|
281
|
+
self.port = port
|
|
282
|
+
# What each (dialect, model) session last sent, so the next request can be
|
|
283
|
+
# priced against it. This is the mock's whole memory.
|
|
284
|
+
self.seen = {}
|
|
285
|
+
self.seen_lock = threading.Lock()
|
|
286
|
+
|
|
287
|
+
def start(self):
|
|
288
|
+
import threading
|
|
289
|
+
self.thread = threading.Thread(target=self.serve_forever, daemon=True)
|
|
290
|
+
self.thread.start()
|
|
291
|
+
return self
|
|
292
|
+
|
|
293
|
+
def stop(self):
|
|
294
|
+
self.shutdown()
|
|
295
|
+
self.server_close()
|
subproto/__init__.py
ADDED