loopview 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- loopview/__init__.py +3 -0
- loopview/app.py +252 -0
- loopview/cli.py +128 -0
- loopview/cost/__init__.py +1 -0
- loopview/cost/pricing.json +62 -0
- loopview/cost/pricing.py +74 -0
- loopview/cost/split.py +228 -0
- loopview/demo_data/flagship.otlp.jsonl +13 -0
- loopview/devtools/__init__.py +1 -0
- loopview/devtools/dump_normalized.py +211 -0
- loopview/ingest/__init__.py +1 -0
- loopview/ingest/otlp.py +305 -0
- loopview/ingest/raw.py +47 -0
- loopview/live.py +154 -0
- loopview/normalize/__init__.py +1 -0
- loopview/normalize/adapters/__init__.py +14 -0
- loopview/normalize/adapters/base.py +87 -0
- loopview/normalize/adapters/gen_ai.py +252 -0
- loopview/normalize/adapters/generic.py +22 -0
- loopview/normalize/adapters/openinference.py +367 -0
- loopview/normalize/derived_tools.py +90 -0
- loopview/normalize/loop_nodes.py +193 -0
- loopview/normalize/normalizer.py +326 -0
- loopview/normalize/schema.py +161 -0
- loopview/normalize/transitions.py +106 -0
- loopview/py.typed +0 -0
- loopview/static/assets/elk-worker.min-DfmSo98M.js +22 -0
- loopview/static/assets/index-CEGppBgk.css +1 -0
- loopview/static/assets/index-DzC96LiQ.js +21 -0
- loopview/static/assets/inter-cyrillic-ext-wght-normal-BOeWTOD4.woff2 +0 -0
- loopview/static/assets/inter-cyrillic-wght-normal-DqGufNeO.woff2 +0 -0
- loopview/static/assets/inter-greek-ext-wght-normal-DlzME5K_.woff2 +0 -0
- loopview/static/assets/inter-greek-wght-normal-CkhJZR-_.woff2 +0 -0
- loopview/static/assets/inter-latin-ext-wght-normal-DO1Apj_S.woff2 +0 -0
- loopview/static/assets/inter-latin-wght-normal-Dx4kXJAl.woff2 +0 -0
- loopview/static/assets/inter-vietnamese-wght-normal-CBcvBZtf.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-cyrillic-wght-normal-D73BlboJ.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-greek-wght-normal-Bw9x6K1M.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-latin-ext-wght-normal-DBQx-q_a.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-latin-wght-normal-B9CIFXIH.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-vietnamese-wght-normal-Bt-aOZkq.woff2 +0 -0
- loopview/static/index.html +13 -0
- loopview/store/__init__.py +1 -0
- loopview/store/capture.py +112 -0
- loopview/store/memory.py +165 -0
- loopview/tools/__init__.py +1 -0
- loopview/tools/report.py +347 -0
- loopview-0.1.0.dist-info/METADATA +72 -0
- loopview-0.1.0.dist-info/RECORD +51 -0
- loopview-0.1.0.dist-info/WHEEL +4 -0
- loopview-0.1.0.dist-info/entry_points.txt +3 -0
loopview/cost/split.py
ADDED
|
@@ -0,0 +1,228 @@
|
|
|
1
|
+
"""Where a model call's tokens came from: an estimated split, scaled to the
|
|
2
|
+
reported truth.
|
|
3
|
+
|
|
4
|
+
The totals always come from the provider's reported usage. Only how they divide
|
|
5
|
+
into segments is estimated, from the content the trace recorded:
|
|
6
|
+
|
|
7
|
+
1. Estimate tokens per segment from the recorded content (characters per token).
|
|
8
|
+
2. Scale the estimates so they add up to the reported count. A simple estimate is
|
|
9
|
+
off by up to about 20%, so a gap that size is estimation error and is scaled
|
|
10
|
+
away. A bigger gap means content the trace doesn't have: the known segments are
|
|
11
|
+
stretched by at most 1.25x and the rest is shown as "unattributed", never
|
|
12
|
+
spread over segments it doesn't belong to.
|
|
13
|
+
3. Cache reads and writes are sub-counts of the input. Providers cache the start
|
|
14
|
+
of the prompt (tools, then system, then messages), so those tokens are taken
|
|
15
|
+
from the segments in that order.
|
|
16
|
+
|
|
17
|
+
Every function here is pure: a ModelCall and prices in, numbers out.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
import json
|
|
21
|
+
import math
|
|
22
|
+
from typing import Any, Literal
|
|
23
|
+
|
|
24
|
+
from loopview.cost.pricing import ModelPrice, Pricing
|
|
25
|
+
from loopview.normalize.schema import CallCost, CostSegment, CostSegmentName, MessagePart, ModelCall
|
|
26
|
+
|
|
27
|
+
# A segment counted from content may be stretched by at most this factor to
|
|
28
|
+
# reach the reported total. 1.25 = an estimate covering at least 80% of it.
|
|
29
|
+
MAX_STRETCH = 1.25
|
|
30
|
+
|
|
31
|
+
Segment = CostSegmentName
|
|
32
|
+
# Prompt order: the order providers cache a prefix in. "unattributed" input is
|
|
33
|
+
# placed last: we don't know where it sits, so cache tokens come from it last.
|
|
34
|
+
INPUT_ORDER: list[Segment] = [
|
|
35
|
+
"tool_prompt",
|
|
36
|
+
"tool_definitions",
|
|
37
|
+
"system",
|
|
38
|
+
"history",
|
|
39
|
+
"tool_results",
|
|
40
|
+
"new_input",
|
|
41
|
+
"unattributed",
|
|
42
|
+
]
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
# --- 1. estimating tokens from content ------------------------------------------------
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def estimate_tokens(value: Any) -> float:
|
|
49
|
+
"""Characters per token: about 4 for prose, about 3 for JSON and code, which
|
|
50
|
+
split into more, shorter tokens. Proportions are what matter: the result is
|
|
51
|
+
scaled to the reported count afterwards."""
|
|
52
|
+
if value is None:
|
|
53
|
+
return 0.0
|
|
54
|
+
if isinstance(value, str):
|
|
55
|
+
stripped = value.lstrip()
|
|
56
|
+
return len(value) / (3 if stripped[:1] in ("{", "[") else 4)
|
|
57
|
+
return len(json.dumps(value, ensure_ascii=False)) / 3
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _part_tokens(part: MessagePart) -> float:
|
|
61
|
+
if part.type == "tool_call":
|
|
62
|
+
return estimate_tokens(part.name or "") + estimate_tokens(
|
|
63
|
+
part.arguments if part.arguments is not None else {}
|
|
64
|
+
)
|
|
65
|
+
if part.type == "tool_result":
|
|
66
|
+
return estimate_tokens(part.result)
|
|
67
|
+
return estimate_tokens(part.text or "")
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def estimate_input(call: ModelCall) -> dict[Segment, float]:
|
|
71
|
+
"""Split the input by where it came from. "New input" is what follows the last
|
|
72
|
+
assistant message (the latest user or handoff message); everything before it
|
|
73
|
+
that isn't a tool result or the system prompt is history resent on this call."""
|
|
74
|
+
estimates: dict[Segment, float] = {
|
|
75
|
+
"tool_definitions": sum(estimate_tokens(d) for d in call.tool_definitions),
|
|
76
|
+
"system": 0.0,
|
|
77
|
+
"history": 0.0,
|
|
78
|
+
"tool_results": 0.0,
|
|
79
|
+
"new_input": 0.0,
|
|
80
|
+
}
|
|
81
|
+
last_assistant = max((i for i, m in enumerate(call.input) if m.role == "assistant"), default=-1)
|
|
82
|
+
for index, message in enumerate(call.input):
|
|
83
|
+
for part in message.parts:
|
|
84
|
+
tokens = _part_tokens(part)
|
|
85
|
+
if message.role == "system":
|
|
86
|
+
estimates["system"] += tokens
|
|
87
|
+
elif part.type == "tool_result":
|
|
88
|
+
estimates["tool_results"] += tokens
|
|
89
|
+
elif index > last_assistant:
|
|
90
|
+
estimates["new_input"] += tokens
|
|
91
|
+
else:
|
|
92
|
+
estimates["history"] += tokens
|
|
93
|
+
return estimates
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def estimate_output(call: ModelCall) -> dict[Segment, float]:
|
|
97
|
+
estimates: dict[Segment, float] = {"thinking": 0.0, "reply": 0.0}
|
|
98
|
+
for message in call.output:
|
|
99
|
+
for part in message.parts:
|
|
100
|
+
estimates["thinking" if part.type == "reasoning" else "reply"] += _part_tokens(part)
|
|
101
|
+
return estimates
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def uses_tools(call: ModelCall) -> bool:
|
|
105
|
+
"""Tools were passed if they were recorded, or if the conversation calls them."""
|
|
106
|
+
if call.tool_definitions:
|
|
107
|
+
return True
|
|
108
|
+
return any(
|
|
109
|
+
p.type in ("tool_call", "tool_result") for m in [*call.input, *call.output] for p in m.parts
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
# --- 2. scaling to the reported count ---------------------------------------------------
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def scale_to(estimates: dict[Segment, float], reported: int) -> dict[Segment, int]:
|
|
117
|
+
"""Whole-token segments that add up to exactly `reported`, with any part the
|
|
118
|
+
estimate can't account for under "unattributed" (see the module docstring)."""
|
|
119
|
+
known = sum(estimates.values())
|
|
120
|
+
if reported <= 0:
|
|
121
|
+
return {name: 0 for name in estimates} | {"unattributed": 0}
|
|
122
|
+
if known <= 0:
|
|
123
|
+
return {name: 0 for name in estimates} | {"unattributed": reported}
|
|
124
|
+
factor = min(reported / known, MAX_STRETCH)
|
|
125
|
+
scaled = {name: math.floor(value * factor) for name, value in estimates.items()}
|
|
126
|
+
gap = reported - sum(scaled.values())
|
|
127
|
+
if factor < MAX_STRETCH or gap <= len(scaled):
|
|
128
|
+
# Within the estimate's error: the remainder is rounding, give it to the
|
|
129
|
+
# largest segment so the parts add up exactly.
|
|
130
|
+
largest = max(scaled, key=lambda name: estimates[name])
|
|
131
|
+
scaled[largest] += gap
|
|
132
|
+
gap = 0
|
|
133
|
+
return scaled | {"unattributed": gap}
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def carve_cache(
|
|
137
|
+
segments: dict[Segment, int], cache_read: int, cache_write: int
|
|
138
|
+
) -> dict[Segment, int]:
|
|
139
|
+
"""Move cached tokens out of the input segments, from the start of the prompt:
|
|
140
|
+
cache reads first (the prefix already cached), then cache writes (the part
|
|
141
|
+
being added to the cache). The input total doesn't change."""
|
|
142
|
+
result = dict(segments)
|
|
143
|
+
for name, amount in (("cache_read", cache_read), ("cache_write", cache_write)):
|
|
144
|
+
taken = 0
|
|
145
|
+
for segment in INPUT_ORDER:
|
|
146
|
+
if taken >= amount:
|
|
147
|
+
break
|
|
148
|
+
take = min(result.get(segment, 0), amount - taken)
|
|
149
|
+
result[segment] = result.get(segment, 0) - take
|
|
150
|
+
taken += take
|
|
151
|
+
result[name] = taken # type: ignore[index]
|
|
152
|
+
return result
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
# --- 3. the whole call ---------------------------------------------------------------
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def call_cost(call: ModelCall, pricing: Pricing) -> CallCost:
|
|
159
|
+
usage = call.usage
|
|
160
|
+
match = pricing.for_model(call.model)
|
|
161
|
+
key, price = match if match else (None, None)
|
|
162
|
+
if usage is None or (usage.input_tokens is None and usage.output_tokens is None):
|
|
163
|
+
return CallCost(
|
|
164
|
+
status="no_usage", price_key=key, price_source=price.source if price else None
|
|
165
|
+
)
|
|
166
|
+
|
|
167
|
+
reported_in = usage.input_tokens or 0
|
|
168
|
+
reported_out = usage.output_tokens or 0
|
|
169
|
+
recorded = bool(call.input or call.output or call.tool_definitions)
|
|
170
|
+
|
|
171
|
+
# Input. The provider's documented tool prompt is known exactly, so it is its
|
|
172
|
+
# own segment rather than part of the estimate.
|
|
173
|
+
tool_prompt = 0
|
|
174
|
+
if (
|
|
175
|
+
price
|
|
176
|
+
and price.tool_prompt_tokens
|
|
177
|
+
and uses_tools(call)
|
|
178
|
+
and reported_in > price.tool_prompt_tokens
|
|
179
|
+
):
|
|
180
|
+
tool_prompt = price.tool_prompt_tokens
|
|
181
|
+
if recorded:
|
|
182
|
+
inputs = scale_to(estimate_input(call), reported_in - tool_prompt)
|
|
183
|
+
else:
|
|
184
|
+
inputs = {"unattributed": reported_in - tool_prompt}
|
|
185
|
+
inputs = {"tool_prompt": tool_prompt, **inputs}
|
|
186
|
+
inputs = carve_cache(inputs, usage.cache_read_tokens or 0, usage.cache_write_tokens or 0)
|
|
187
|
+
|
|
188
|
+
# Output. Reported thinking tokens are the truth for thinking; otherwise the
|
|
189
|
+
# split comes from the text, like the input.
|
|
190
|
+
if usage.reasoning_tokens is not None:
|
|
191
|
+
reply_estimate = {"reply": estimate_output(call)["reply"]} if recorded else {}
|
|
192
|
+
outputs = {
|
|
193
|
+
"thinking": usage.reasoning_tokens,
|
|
194
|
+
**scale_to(reply_estimate, reported_out - usage.reasoning_tokens),
|
|
195
|
+
}
|
|
196
|
+
elif recorded:
|
|
197
|
+
outputs = scale_to(estimate_output(call), reported_out)
|
|
198
|
+
else:
|
|
199
|
+
outputs = {"unattributed": reported_out}
|
|
200
|
+
|
|
201
|
+
segments = [
|
|
202
|
+
_segment(name, "input", tokens, price) for name, tokens in inputs.items() if tokens > 0
|
|
203
|
+
]
|
|
204
|
+
segments += [
|
|
205
|
+
_segment(name, "output", tokens, price) for name, tokens in outputs.items() if tokens > 0
|
|
206
|
+
]
|
|
207
|
+
return CallCost(
|
|
208
|
+
status="estimated" if recorded else "content_not_recorded",
|
|
209
|
+
segments=segments,
|
|
210
|
+
tokens=reported_in + reported_out,
|
|
211
|
+
dollars=round(sum(s.dollars or 0 for s in segments), 8) if price else None,
|
|
212
|
+
price_key=key,
|
|
213
|
+
price_source=price.source if price else None,
|
|
214
|
+
)
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def _segment(
|
|
218
|
+
name: Any, side: Literal["input", "output"], tokens: int, price: ModelPrice | None
|
|
219
|
+
) -> CostSegment:
|
|
220
|
+
if price is None:
|
|
221
|
+
dollars = None
|
|
222
|
+
elif name == "cache_read":
|
|
223
|
+
dollars = tokens * price.cache_read / 1e6
|
|
224
|
+
elif name == "cache_write":
|
|
225
|
+
dollars = tokens * price.cache_write / 1e6
|
|
226
|
+
else:
|
|
227
|
+
dollars = tokens * (price.output if side == "output" else price.input) / 1e6
|
|
228
|
+
return CostSegment(name=name, side=side, tokens=tokens, dollars=dollars)
|