tokencur 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tokencur-0.3.0/LICENSE +21 -0
- tokencur-0.3.0/PKG-INFO +386 -0
- tokencur-0.3.0/README.md +346 -0
- tokencur-0.3.0/pyproject.toml +77 -0
- tokencur-0.3.0/setup.cfg +4 -0
- tokencur-0.3.0/src/tokencur/__init__.py +7 -0
- tokencur-0.3.0/src/tokencur/__main__.py +6 -0
- tokencur-0.3.0/src/tokencur/cli.py +349 -0
- tokencur-0.3.0/src/tokencur/dashboard.py +177 -0
- tokencur-0.3.0/src/tokencur/doctor.py +239 -0
- tokencur-0.3.0/src/tokencur/export.py +61 -0
- tokencur-0.3.0/src/tokencur/focus.py +275 -0
- tokencur-0.3.0/src/tokencur/ingest/__init__.py +1 -0
- tokencur-0.3.0/src/tokencur/ingest/claude_code.py +160 -0
- tokencur-0.3.0/src/tokencur/ingest/codex.py +131 -0
- tokencur-0.3.0/src/tokencur/ingest/fields.py +49 -0
- tokencur-0.3.0/src/tokencur/ingest/identity.py +38 -0
- tokencur-0.3.0/src/tokencur/ingest/kimi_code.py +142 -0
- tokencur-0.3.0/src/tokencur/ingest/runpod.py +97 -0
- tokencur-0.3.0/src/tokencur/ingest/stats.py +25 -0
- tokencur-0.3.0/src/tokencur/ledger.py +412 -0
- tokencur-0.3.0/src/tokencur/observatory.py +617 -0
- tokencur-0.3.0/src/tokencur/outcomes.py +384 -0
- tokencur-0.3.0/src/tokencur/prices.py +413 -0
- tokencur-0.3.0/src/tokencur/pricing.py +260 -0
- tokencur-0.3.0/src/tokencur/pricing_data/litellm_snapshot.json +2383 -0
- tokencur-0.3.0/src/tokencur/recommend.py +190 -0
- tokencur-0.3.0/src/tokencur/recommend_cli.py +20 -0
- tokencur-0.3.0/src/tokencur/records.py +118 -0
- tokencur-0.3.0/src/tokencur/report.py +140 -0
- tokencur-0.3.0/src/tokencur/sources.py +50 -0
- tokencur-0.3.0/src/tokencur/terminal.py +33 -0
- tokencur-0.3.0/src/tokencur.egg-info/PKG-INFO +386 -0
- tokencur-0.3.0/src/tokencur.egg-info/SOURCES.txt +58 -0
- tokencur-0.3.0/src/tokencur.egg-info/dependency_links.txt +1 -0
- tokencur-0.3.0/src/tokencur.egg-info/entry_points.txt +2 -0
- tokencur-0.3.0/src/tokencur.egg-info/requires.txt +10 -0
- tokencur-0.3.0/src/tokencur.egg-info/top_level.txt +1 -0
- tokencur-0.3.0/tests/test_claude_code.py +269 -0
- tokencur-0.3.0/tests/test_cli.py +236 -0
- tokencur-0.3.0/tests/test_codex.py +276 -0
- tokencur-0.3.0/tests/test_dashboard.py +53 -0
- tokencur-0.3.0/tests/test_doctor.py +162 -0
- tokencur-0.3.0/tests/test_fixture_privacy.py +89 -0
- tokencur-0.3.0/tests/test_focus.py +134 -0
- tokencur-0.3.0/tests/test_focus_cross.py +108 -0
- tokencur-0.3.0/tests/test_fuzz.py +180 -0
- tokencur-0.3.0/tests/test_golden.py +71 -0
- tokencur-0.3.0/tests/test_kimi_code.py +96 -0
- tokencur-0.3.0/tests/test_ledger.py +292 -0
- tokencur-0.3.0/tests/test_observatory.py +198 -0
- tokencur-0.3.0/tests/test_outcomes.py +349 -0
- tokencur-0.3.0/tests/test_prices.py +216 -0
- tokencur-0.3.0/tests/test_pricing.py +187 -0
- tokencur-0.3.0/tests/test_properties.py +257 -0
- tokencur-0.3.0/tests/test_recommend.py +134 -0
- tokencur-0.3.0/tests/test_report.py +82 -0
- tokencur-0.3.0/tests/test_runpod.py +143 -0
- tokencur-0.3.0/tests/test_sources.py +97 -0
- tokencur-0.3.0/tests/test_update_snapshot.py +100 -0
tokencur-0.3.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Pedro Aguayo Chávez
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
tokencur-0.3.0/PKG-INFO
ADDED
|
@@ -0,0 +1,386 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: tokencur
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: Convert multi-provider AI/LLM usage data into FOCUS-conformant cost datasets
|
|
5
|
+
Author-email: Pedro Aguayo Chávez <aguayochavezpedro@gmail.com>
|
|
6
|
+
License: MIT License
|
|
7
|
+
|
|
8
|
+
Copyright (c) 2026 Pedro Aguayo Chávez
|
|
9
|
+
|
|
10
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
11
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
12
|
+
in the Software without restriction, including without limitation the rights
|
|
13
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
14
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
15
|
+
furnished to do so, subject to the following conditions:
|
|
16
|
+
|
|
17
|
+
The above copyright notice and this permission notice shall be included in all
|
|
18
|
+
copies or substantial portions of the Software.
|
|
19
|
+
|
|
20
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
21
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
22
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
23
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
24
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
25
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
26
|
+
SOFTWARE.
|
|
27
|
+
|
|
28
|
+
Requires-Python: >=3.11
|
|
29
|
+
Description-Content-Type: text/markdown
|
|
30
|
+
License-File: LICENSE
|
|
31
|
+
Provides-Extra: dev
|
|
32
|
+
Requires-Dist: pytest>=8; extra == "dev"
|
|
33
|
+
Requires-Dist: pytest-cov>=7; extra == "dev"
|
|
34
|
+
Requires-Dist: hypothesis>=6.168; extra == "dev"
|
|
35
|
+
Requires-Dist: ruff==0.16.10; extra == "dev"
|
|
36
|
+
Provides-Extra: dashboard
|
|
37
|
+
Requires-Dist: streamlit>=1.35; extra == "dashboard"
|
|
38
|
+
Requires-Dist: duckdb>=1.0; extra == "dashboard"
|
|
39
|
+
Dynamic: license-file
|
|
40
|
+
|
|
41
|
+
# tokencur
|
|
42
|
+
|
|
43
|
+
[](https://github.com/pach-boop/tokencur/actions/workflows/ci.yml)
|
|
44
|
+
[](https://scorecard.dev/viewer/?uri=github.com/pach-boop/tokencur)
|
|
45
|
+
[](pyproject.toml)
|
|
46
|
+
[](LICENSE)
|
|
47
|
+
|
|
48
|
+
**The CUR for your tokens** — an open-source pipeline that turns AI usage into
|
|
49
|
+
[FOCUS](https://focus.finops.org)-conformant cost datasets, validated in CI by the
|
|
50
|
+
FinOps Foundation's own validator. Today it reads the local logs of three coding
|
|
51
|
+
agents — Claude Code, Codex CLI and Kimi Code — prices them against 290+ models, and
|
|
52
|
+
adds unit economics, savings recommendations and a local ledger that keeps the history
|
|
53
|
+
after the agents delete their logs.
|
|
54
|
+
|
|
55
|
+
> A FinOps tool that practices FinOps on itself: the first dataset is my own real AI spend.
|
|
56
|
+
|
|
57
|
+
## Why
|
|
58
|
+
|
|
59
|
+
- **AI cost management is the #1 skill gap in FinOps** (State of FinOps 2026: 98% of
|
|
60
|
+
organizations now manage AI spend).
|
|
61
|
+
- **FOCUS already supports tokens** — the spec has columns for token/credit-based billing,
|
|
62
|
+
with OpenAI usage as an official example. The standard is ready; open implementations
|
|
63
|
+
are not.
|
|
64
|
+
- The FinOps Foundation's [`focus_converters`](https://github.com/finopsfoundation/focus_converters)
|
|
65
|
+
covers AWS, GCP, Azure and OCI — **zero AI providers**. tokencur aims to contribute an
|
|
66
|
+
AI-provider converter upstream.
|
|
67
|
+
|
|
68
|
+
## Related work
|
|
69
|
+
|
|
70
|
+
| Category | Projects | How tokencur differs |
|
|
71
|
+
|---|---|---|
|
|
72
|
+
| Dev observability | Langfuse, Helicone, LiteLLM | Per-request tracing; no FOCUS output, no finance vocabulary |
|
|
73
|
+
| Coding-agent trackers | ccusage, tokscale, TokenTracker, [tokentop](https://github.com/tokentopapp/tokentop), [budi](https://github.com/siropkin/budi) | Dashboards and live monitors (budi also attributes each call to a repo, branch and ticket); none exports FOCUS or keeps a finance vocabulary |
|
|
74
|
+
| Cost per commit | [agent-cost](https://github.com/lucianareynaud/agent-cost) | Links a cost you log by hand after each session to the commits it produced; tokencur attributes every call from the agents' own logs, with no manual step |
|
|
75
|
+
| Enterprise platforms | Finout, Vantage, CloudZero | FOCUS-aligned but closed source and enterprise-priced |
|
|
76
|
+
| Plumbing | OpenCost OpenAI plugin, focus_converters | k8s-bound / cloud-only; tokencur is standalone, multi-provider, analyst-friendly |
|
|
77
|
+
|
|
78
|
+
## Design principles
|
|
79
|
+
|
|
80
|
+
1. **Privacy by construction** — ingestion reads usage metadata only (tokens, models,
|
|
81
|
+
timestamps). Conversation content is never extracted.
|
|
82
|
+
2. **Measure existing spend, don't generate spend to measure** — the first data source is
|
|
83
|
+
local Claude Code session logs, which already exist on disk. Budget: ~$0.
|
|
84
|
+
3. **List-cost showback** — subscription usage isn't billed per token, so costs are
|
|
85
|
+
computed as *API-equivalent list cost*. Pricing has two layers: a curated, dated
|
|
86
|
+
Anthropic rate card ([`pricing.py`](src/tokencur/pricing.py)) that always wins, and a
|
|
87
|
+
vendored snapshot of the community-maintained
|
|
88
|
+
[LiteLLM price database](https://github.com/BerriAI/litellm) as fallback (290+ live
|
|
89
|
+
models across Anthropic, OpenAI, Gemini, DeepSeek, Kimi/Moonshot, GLM/Z.ai and
|
|
90
|
+
Ollama, refreshed daily by the [price-watch action](.github/workflows/price-watch.yml);
|
|
91
|
+
models LiteLLM retires keep their last known rate, so historical usage stays priced).
|
|
92
|
+
Unknown models surface as *unpriced usage* rather than silently costing $0.
|
|
93
|
+
Rates are point-in-time: each call is valued at the list rate in force on its
|
|
94
|
+
day, from a rate history the snapshot keeps. Request options that change the
|
|
95
|
+
price are applied as Claude Code logs them: fast mode (2x), US-only inference
|
|
96
|
+
(1.1x) and the Batch API (0.5x).
|
|
97
|
+
4. **Explainable over clever** — every line that ships is one the maintainer fully
|
|
98
|
+
understands and can defend.
|
|
99
|
+
|
|
100
|
+
## Architecture
|
|
101
|
+
|
|
102
|
+
```mermaid
|
|
103
|
+
flowchart LR
|
|
104
|
+
A["Claude Code · Codex CLI · Kimi Code<br/>local logs"] -->|"ingesters<br/>(metadata only)"| L[("Ledger<br/>SQLite")]
|
|
105
|
+
L --> P["Pricing<br/>curated card + LiteLLM snapshot"]
|
|
106
|
+
W["price-watch action<br/>(daily)"] -.->|refreshes| P
|
|
107
|
+
P --> R["report · recommend"]
|
|
108
|
+
P --> F["FOCUS 1.2 normalizer"]
|
|
109
|
+
F --> E["export: FOCUS CSV<br/>(validated in CI)"]
|
|
110
|
+
F --> O["observatory · dashboard"]
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
Each design decision, with its context and its cost, is recorded as an
|
|
114
|
+
[architecture decision record](docs/adr/README.md).
|
|
115
|
+
|
|
116
|
+
One module per layer — [`ingest/`](src/tokencur/ingest) (one adapter per source),
|
|
117
|
+
[`ledger`](src/tokencur/ledger.py), [`pricing`](src/tokencur/pricing.py),
|
|
118
|
+
[`focus`](src/tokencur/focus.py) — with the commands on top. Every command loads
|
|
119
|
+
records through one function, [`sources.load_records()`](src/tokencur/sources.py).
|
|
120
|
+
|
|
121
|
+
## Quickstart
|
|
122
|
+
|
|
123
|
+
Requires Python 3.11+. No runtime dependencies.
|
|
124
|
+
|
|
125
|
+
```bash
|
|
126
|
+
pip install -e .
|
|
127
|
+
tokencur report # cost summary in your terminal
|
|
128
|
+
tokencur export focus.csv # FOCUS 1.2 conformant dataset
|
|
129
|
+
tokencur recommend # avoided cost + what-if headroom
|
|
130
|
+
tokencur export sept.csv --since 2026-09-01 --until 2026-10-01 # one billing period
|
|
131
|
+
tokencur doctor # read-only health check: log formats, ledger, pricing
|
|
132
|
+
tokencur import runpod FILE # billed cost: a RunPod billing export
|
|
133
|
+
tokencur outcomes [REPO...] # usage value per commit, repository by repository
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
`tokencur --help` lists every command, and `python -m tokencur` works the same.
|
|
137
|
+
Under a negotiated contract, `--discounts discounts.json` (for example
|
|
138
|
+
`{"discounts": {"Anthropic": 0.15}}`) keeps the public price in FOCUS `ListCost`
|
|
139
|
+
and puts the contracted price in `ContractedCost`, `EffectiveCost` and
|
|
140
|
+
`BilledCost`; the report adds a contracted total. The export stays in USD, the
|
|
141
|
+
currency every provider bills in; `tokencur report --currency MXN --fx-rate
|
|
142
|
+
18.37` also shows totals at a rate you give (nothing is fetched).
|
|
143
|
+
Periods are UTC days with `--until` excluded, the way a billing period is cut.
|
|
144
|
+
|
|
145
|
+
With no arguments it scans every known local source on your machine — **Claude Code**
|
|
146
|
+
(`~/.claude/projects`), **Codex CLI** (`~/.codex/sessions`) and **Kimi Code**
|
|
147
|
+
(`~/.kimi-code/sessions`) — and prints per-model, per-source and per-day
|
|
148
|
+
API-equivalent cost, including provider-correct cache economics. Each scan
|
|
149
|
+
is also kept in a local [ledger](#ledger), so history survives when the
|
|
150
|
+
agents delete their old logs.
|
|
151
|
+
|
|
152
|
+
For the visual version — daily trend, cost by model, token-type mix and unit
|
|
153
|
+
economics, each view exposing the DuckDB SQL behind it:
|
|
154
|
+
|
|
155
|
+
```bash
|
|
156
|
+
pip install -e ".[dashboard]"
|
|
157
|
+
streamlit run src/tokencur/dashboard.py
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
Real output over the maintainer's own machine on 2026-10-02 (10.7k model calls
|
|
161
|
+
from ~400 MB of logs, 1.8 s), trimmed:
|
|
162
|
+
|
|
163
|
+
```text
|
|
164
|
+
model calls input output cache_read cache_write cost USD
|
|
165
|
+
claude-opus-5-5 2,545 5,230 4,453,008 1,039,768,719 21,810,960 471.43
|
|
166
|
+
gpt-5.4 2,790 34,683,046 1,807,104 304,527,488 0 189.95
|
|
167
|
+
gpt-5.3-codex 5,010 26,148,134 1,997,552 377,040,128 0 139.71
|
|
168
|
+
claude-fable-5-1 14 418 64,013 1,732,662 185,892 7.36
|
|
169
|
+
gpt-5.2-codex 139 347,399 71,796 4,995,072 0 2.49
|
|
170
|
+
...
|
|
171
|
+
|
|
172
|
+
by source:
|
|
173
|
+
claude-code $482.51
|
|
174
|
+
codex $332.95
|
|
175
|
+
kimi-code $2.44
|
|
176
|
+
|
|
177
|
+
API-EQUIVALENT TOTAL (showback): $817.90
|
|
178
|
+
```
|
|
179
|
+
|
|
180
|
+
## Performance
|
|
181
|
+
|
|
182
|
+
On the maintainer's machine the full pipeline reads 10.7k model calls from
|
|
183
|
+
~400 MB of real logs in 1.8 s. Real transcripts carry the conversations,
|
|
184
|
+
which tokencur skips. For scale,
|
|
185
|
+
[`scripts/benchmark.py`](scripts/benchmark.py) times every stage on synthetic
|
|
186
|
+
metadata-only logs (Python 3.13, Intel i7-1355U, Linux):
|
|
187
|
+
|
|
188
|
+
| Stage | 100k messages | 1M messages |
|
|
189
|
+
|---|---:|---:|
|
|
190
|
+
| Scan logs (38 MB / 376 MB) | 0.6 s | 6.3 s |
|
|
191
|
+
| Ledger write | 0.2 s | 2.5 s |
|
|
192
|
+
| Rescan, adds nothing | 0.2 s | 2.4 s |
|
|
193
|
+
| Ledger read | 0.4 s | 5.1 s |
|
|
194
|
+
| Report | 0.1 s | 1.4 s |
|
|
195
|
+
| FOCUS export (0.4M / 4M rows) | 5.6 s | 55.9 s |
|
|
196
|
+
| Peak memory | 139 MB | 1.1 GB |
|
|
197
|
+
|
|
198
|
+
Time grows linearly. The benchmark holds the scanned and the stored history at
|
|
199
|
+
once, so its peak memory is about twice what one command needs. Reproduce with
|
|
200
|
+
`python scripts/benchmark.py --messages 1000000 --files 2000`.
|
|
201
|
+
|
|
202
|
+
## Ledger
|
|
203
|
+
|
|
204
|
+
Coding agents treat their logs as disposable: Claude Code deletes session
|
|
205
|
+
transcripts after `cleanupPeriodDays` (30 days by default). Computed from the
|
|
206
|
+
logs alone, totals would *shrink* as history disappears. So every command —
|
|
207
|
+
`report`, `export`, `recommend`, the observatory and the dashboard — first
|
|
208
|
+
keeps what it scans in a local SQLite ledger, then reports the ledger's full
|
|
209
|
+
history:
|
|
210
|
+
|
|
211
|
+
- **Deduplicated per usage event** — re-running adds nothing, and a resumed
|
|
212
|
+
session's copied messages count once.
|
|
213
|
+
- **Nothing is forgotten** — events whose logs are gone keep their last known
|
|
214
|
+
values; events still on disk are refreshed from the latest parse.
|
|
215
|
+
- **Metadata only** — token counts, models, timestamps, workspace and session
|
|
216
|
+
ids, the directory each call ran in, in a file readable by its owner only.
|
|
217
|
+
- **Location** — `~/.local/share/tokencur/ledger.sqlite3` (honours
|
|
218
|
+
`$XDG_DATA_HOME`), or wherever `$TOKENCUR_LEDGER` points.
|
|
219
|
+
- **Corrections are audited, not erased** — the schema is versioned and
|
|
220
|
+
migrated in place. Before an upgrade the file is copied to
|
|
221
|
+
`ledger.sqlite3.schema-N.bak`, and rows a later version finds were not
|
|
222
|
+
usage move to a `superseded` table with when and why. Schema 2 retired the
|
|
223
|
+
Codex re-sent reports earlier versions double counted (see the changelog);
|
|
224
|
+
schema 3 records each call's price-changing request options, schema 4 billed
|
|
225
|
+
charges, and schema 5 the directory each call ran in.
|
|
226
|
+
|
|
227
|
+
The ledger keeps only what it has seen: usage deleted before the first run is
|
|
228
|
+
gone. An explicit path (`python -m tokencur report ROOT`) is reported as-is and
|
|
229
|
+
never stored.
|
|
230
|
+
|
|
231
|
+
## Billed cost
|
|
232
|
+
|
|
233
|
+
Showback values usage at list price; a bill is what a provider actually took.
|
|
234
|
+
RunPod is the first billed source: `scripts/fetch_runpod_billing.py` saves the
|
|
235
|
+
billing history from RunPod's REST API (the one step that goes online, when you
|
|
236
|
+
run it), and `tokencur import runpod FILE` keeps it in the ledger. Reports show
|
|
237
|
+
it as a separate total, "BILLED (real money, from provider bills)", the FOCUS
|
|
238
|
+
export gains Compute rows whose `BilledCost` is the billed amount, and the
|
|
239
|
+
observatory shows it as actual money outside subscription leverage
|
|
240
|
+
([ADR 0009](docs/adr/0009-billed-charges-next-to-showback.md)).
|
|
241
|
+
|
|
242
|
+
## Unit economics: usage value per commit
|
|
243
|
+
|
|
244
|
+
Cost alone does not say whether usage paid off. `tokencur outcomes` sets it
|
|
245
|
+
against the commits it went into, per git repository:
|
|
246
|
+
|
|
247
|
+
```
|
|
248
|
+
repository days calls usage value commits agent lines per commit
|
|
249
|
+
~/Proyectos/tokencur 2026-09-24..2026-10-02 424 $70.78 40 40 10,824 $1.77
|
|
250
|
+
```
|
|
251
|
+
|
|
252
|
+
- **Attribution by where each call ran.** Agents log the working directory of
|
|
253
|
+
every call, and the call belongs to the git repository that holds it. Usage
|
|
254
|
+
outside any repository, or in a directory since deleted, is reported as
|
|
255
|
+
unattributed, never spread.
|
|
256
|
+
- **Your commits in the days with usage.** Non-merge commits by your
|
|
257
|
+
`user.email` (or `--all-authors`), within `--since`/`--until` or else the
|
|
258
|
+
repository's days with agent usage. Agent-signed commits and lines changed
|
|
259
|
+
are shown as context.
|
|
260
|
+
- **Read-only and local.** git runs on your repositories; nothing is sent.
|
|
261
|
+
|
|
262
|
+
A commit is a coarse unit: it measures output, not quality. The useful
|
|
263
|
+
comparison is a repository with itself over time. Quality, latency and
|
|
264
|
+
reliability are the next layers
|
|
265
|
+
([ADR 0010](docs/adr/0010-usage-value-per-commit.md)).
|
|
266
|
+
|
|
267
|
+
## When an agent changes its logs
|
|
268
|
+
|
|
269
|
+
Agents change their log formats without notice. `tokencur doctor` scans every
|
|
270
|
+
source read-only and reports files, usage lines, records, unreadable lines and
|
|
271
|
+
the agent versions the logs name. It flags what looks like a format change:
|
|
272
|
+
log files with no usage lines, usage lines that yield no records, or more than
|
|
273
|
+
1% unreadable lines. It also checks the ledger's schema and SQLite integrity
|
|
274
|
+
and prints the pricing snapshot's SHA-256. It exits 1 when anything needs
|
|
275
|
+
attention. A malformed line is skipped, never guessed at, and never stops a
|
|
276
|
+
scan.
|
|
277
|
+
|
|
278
|
+
## Money concepts (read before quoting numbers)
|
|
279
|
+
|
|
280
|
+
tokencur's headline figures are **not** a bill. Three money concepts, kept
|
|
281
|
+
deliberately apart:
|
|
282
|
+
|
|
283
|
+
- **Actual outlay** — the flat subscription fees really paid, declared in
|
|
284
|
+
[`subscriptions.json`](./subscriptions.json), and provider bills imported as
|
|
285
|
+
billed charges (see [Billed cost](#billed-cost)). The only real money here.
|
|
286
|
+
- **Usage value (showback)** — what the same usage would cost at API list
|
|
287
|
+
prices. Subscriptions don't bill per token, so tokencur *values* the usage
|
|
288
|
+
instead of pretending to bill it.
|
|
289
|
+
- **Counterfactuals** — cost avoided by provider caching, and what-if
|
|
290
|
+
right-sizing. Properties of the workload, not actions taken. Under a flat
|
|
291
|
+
subscription, right-sizing buys rate-limit headroom, not dollars.
|
|
292
|
+
|
|
293
|
+
Divide value by outlay and you get the observatory's headline metric:
|
|
294
|
+
**subscription leverage** — how many times over the flat fee pays for itself.
|
|
295
|
+
|
|
296
|
+
## Observatory
|
|
297
|
+
|
|
298
|
+
A public, static snapshot of this repository's own AI spend — the FOCUS dataset
|
|
299
|
+
rendered as a dashboard: **https://pach-boop.github.io/tokencur/observatory/**
|
|
300
|
+
|
|
301
|
+
```bash
|
|
302
|
+
tokencur observatory # regenerates docs/observatory/
|
|
303
|
+
```
|
|
304
|
+
|
|
305
|
+
The snapshot publishes aggregates only (day × service, model and token-bucket
|
|
306
|
+
totals) — workspace names, session ids and message content never enter the
|
|
307
|
+
output, and the page is fully self-contained (no external requests). Like the
|
|
308
|
+
pricing snapshot, it is committed deliberately: the site updates when the
|
|
309
|
+
maintainer decides, not on a schedule.
|
|
310
|
+
|
|
311
|
+
## Price card
|
|
312
|
+
|
|
313
|
+
The public list rates tokencur values usage against, and a dated change log
|
|
314
|
+
built from the git history of the pricing snapshot:
|
|
315
|
+
**https://pach-boop.github.io/tokencur/prices/**
|
|
316
|
+
|
|
317
|
+
```bash
|
|
318
|
+
tokencur prices # regenerates docs/prices/
|
|
319
|
+
```
|
|
320
|
+
|
|
321
|
+
A daily [price-watch action](.github/workflows/price-watch.yml) refreshes the
|
|
322
|
+
vendored snapshot and commits whenever it changes, with a message that names
|
|
323
|
+
what moved (rate moves, models added, models retired upstream and kept at their
|
|
324
|
+
last rate), then regenerates this page from that history.
|
|
325
|
+
|
|
326
|
+
## Roadmap
|
|
327
|
+
|
|
328
|
+
| Phase | Deliverable | Status |
|
|
329
|
+
|---|---|---|
|
|
330
|
+
| 1 | Repo, thesis, related work | ✅ |
|
|
331
|
+
| 2 | Ingest real usage: local agent logs; Anthropic/OpenAI admin-API exports | 🔨 Claude Code, Codex CLI and Kimi Code done; API exports pending |
|
|
332
|
+
| 3 | FOCUS normalizer + CSV export, gated in CI by the [Foundation's own validator](https://github.com/finopsfoundation/focus_validator), cross-checked against [official sample data](https://github.com/FinOps-Open-Cost-and-Usage-Spec/FOCUS-Sample-Data) | ✅ |
|
|
333
|
+
| 4 | DuckDB + Streamlit dashboard: trends, top spend, unit economics | ✅ v1 |
|
|
334
|
+
| 5 | Recommendation engine: caching ROI (measured) + model right-sizing (what-if) | ✅ v1 — batch and local-vs-API break-even need user-supplied inputs, next |
|
|
335
|
+
| 6 | Serverless AWS deployment, documented, with its own measured running cost | ⏳ |
|
|
336
|
+
| 7 | PR to `focus_converters` + bilingual (EN/ES) case study | ⏳ |
|
|
337
|
+
|
|
338
|
+
## Limitations (honest)
|
|
339
|
+
|
|
340
|
+
- Usage comes from three coding agents' local logs (Claude Code, Codex CLI, Kimi
|
|
341
|
+
Code); billed cost from RunPod exports. The
|
|
342
|
+
[Anthropic and OpenAI admin cost APIs](https://github.com/pach-boop/tokencur/issues/7)
|
|
343
|
+
are next and wait on real response samples.
|
|
344
|
+
- The ledger can only keep usage it has seen. Run tokencur more often than
|
|
345
|
+
Claude Code's `cleanupPeriodDays`, or raise that setting in
|
|
346
|
+
`~/.claude/settings.json`.
|
|
347
|
+
- Codex calls are counted when the session's running total moves; per session,
|
|
348
|
+
the counted calls reconcile exactly with Codex's own running total (all 94 of
|
|
349
|
+
the maintainer's rollouts). A forked session that re-copies earlier calls
|
|
350
|
+
counts them once, in a scan and in the ledger, even after the original log is
|
|
351
|
+
gone: a copy is recognized by its timestamp and raw usage, so two distinct
|
|
352
|
+
calls identical to the millisecond would also count once (never observed).
|
|
353
|
+
- Costs are list-price showback, not invoices. Subscription plans bill differently.
|
|
354
|
+
- Usage value per commit counts commits on the checked-out branch, and a commit
|
|
355
|
+
says nothing about size or quality. Usage logged before tokencur 0.3 kept no
|
|
356
|
+
working directory; it is attributed only if its log is still on disk.
|
|
357
|
+
- Rates are point-in-time: each call is valued at the list rate in force on its
|
|
358
|
+
UTC day ([ADR 0008](docs/adr/0008-point-in-time-list-rates.md)). Rate history
|
|
359
|
+
starts with the snapshot on 2026-07-06, so earlier usage is valued at the first
|
|
360
|
+
rate observed, and a move is dated by the day the price-watch bot saw it.
|
|
361
|
+
- Older log formats don't break down cache writes by TTL; totals are attributed to the
|
|
362
|
+
5-minute tier (slight underestimate), documented in the parser.
|
|
363
|
+
- Daily buckets use the UTC dates recorded in the logs; a late-night local session can
|
|
364
|
+
land on the next UTC day.
|
|
365
|
+
- Targets FOCUS 1.2: the newest spec version the Foundation's `focus-validator` can
|
|
366
|
+
check (2.2.1 ships 1.2 rules only), so newer spec versions will follow the validator.
|
|
367
|
+
The export passes it in CI and is cross-checked against the Foundation's official
|
|
368
|
+
sample data (which targets FOCUS 1.0; the tests assert convention compatibility, not
|
|
369
|
+
column equality).
|
|
370
|
+
|
|
371
|
+
## Contributing
|
|
372
|
+
|
|
373
|
+
Small, reviewed changes are welcome. [CONTRIBUTING.md](CONTRIBUTING.md) has the
|
|
374
|
+
ground rules (metadata only, unpriced is never $0, no runtime dependencies) and how
|
|
375
|
+
to add a usage source. Changes land through pull requests with green CI;
|
|
376
|
+
Dependabot, CodeQL and OpenSSF Scorecard run on the repository. Security reports
|
|
377
|
+
go through [SECURITY.md](SECURITY.md), never a public issue.
|
|
378
|
+
|
|
379
|
+
## Transparency
|
|
380
|
+
|
|
381
|
+
Built with AI assistance (Claude). Policy: nothing is merged that the maintainer does
|
|
382
|
+
not fully understand and stand behind.
|
|
383
|
+
|
|
384
|
+
## License
|
|
385
|
+
|
|
386
|
+
[MIT](LICENSE)
|