llm_cost_tracker 0.14.0 → 0.14.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +128 -0
- data/README.md +16 -5
- data/app/assets/llm_cost_tracker/application.css +16 -0
- data/app/controllers/llm_cost_tracker/application_controller.rb +16 -3
- data/app/controllers/llm_cost_tracker/calls_controller.rb +1 -3
- data/app/controllers/llm_cost_tracker/models_controller.rb +2 -4
- data/app/controllers/llm_cost_tracker/pricing_controller.rb +2 -2
- data/app/controllers/llm_cost_tracker/tags_controller.rb +9 -7
- data/app/helpers/llm_cost_tracker/application_helper.rb +5 -26
- data/app/helpers/llm_cost_tracker/chart_helper.rb +6 -10
- data/app/helpers/llm_cost_tracker/dashboard_query_helper.rb +5 -0
- data/app/helpers/llm_cost_tracker/token_usage_helper.rb +8 -28
- data/app/models/llm_cost_tracker/call.rb +10 -14
- data/app/services/llm_cost_tracker/dashboard/data_quality.rb +4 -15
- data/app/services/llm_cost_tracker/dashboard/filter.rb +29 -30
- data/app/services/llm_cost_tracker/dashboard/pagination.rb +2 -5
- data/app/services/llm_cost_tracker/dashboard/params.rb +10 -0
- data/app/services/llm_cost_tracker/dashboard/pricing_overview.rb +2 -1
- data/app/services/llm_cost_tracker/dashboard/spend_anomaly.rb +1 -1
- data/app/services/llm_cost_tracker/dashboard/tag_breakdown.rb +2 -3
- data/app/services/llm_cost_tracker/dashboard/tag_key_explorer.rb +1 -0
- data/app/services/llm_cost_tracker/dashboard/time_series.rb +3 -5
- data/app/services/llm_cost_tracker/dashboard/top_models.rb +1 -2
- data/app/views/llm_cost_tracker/calls/show.html.erb +8 -22
- data/app/views/llm_cost_tracker/data_quality/index.html.erb +1 -1
- data/app/views/llm_cost_tracker/shared/_bar.html.erb +1 -3
- data/app/views/llm_cost_tracker/shared/_filter_pill_date.html.erb +1 -4
- data/app/views/llm_cost_tracker/shared/_filter_pill_model.html.erb +1 -4
- data/app/views/llm_cost_tracker/shared/_filter_pill_provider.html.erb +1 -4
- data/app/views/llm_cost_tracker/shared/_filter_pill_stream.html.erb +1 -4
- data/app/views/llm_cost_tracker/tags/show.html.erb +8 -5
- data/config/routes.rb +6 -1
- data/lib/llm_cost_tracker/budget/per_tag.rb +21 -11
- data/lib/llm_cost_tracker/budget.rb +44 -55
- data/lib/llm_cost_tracker/capture/event_window.rb +100 -0
- data/lib/llm_cost_tracker/capture/sdk_payload.rb +5 -1
- data/lib/llm_cost_tracker/capture/sse.rb +108 -32
- data/lib/llm_cost_tracker/capture/stream_collector.rb +41 -81
- data/lib/llm_cost_tracker/capture/stream_tap.rb +57 -0
- data/lib/llm_cost_tracker/capture/stream_tracker.rb +14 -58
- data/lib/llm_cost_tracker/capture_verifier.rb +1 -7
- data/lib/llm_cost_tracker/charges/line_item.rb +5 -1
- data/lib/llm_cost_tracker/configuration/budgets.rb +1 -1
- data/lib/llm_cost_tracker/configuration/pricing.rb +1 -1
- data/lib/llm_cost_tracker/configuration.rb +5 -9
- data/lib/llm_cost_tracker/doctor/price_check.rb +14 -3
- data/lib/llm_cost_tracker/doctor/schema_check.rb +1 -2
- data/lib/llm_cost_tracker/doctor.rb +19 -2
- data/lib/llm_cost_tracker/engine.rb +0 -1
- data/lib/llm_cost_tracker/errors.rb +13 -0
- data/lib/llm_cost_tracker/event.rb +8 -0
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/async_ingestion_generator.rb +0 -6
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/call_rollups_generator.rb +5 -8
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/install_generator.rb +0 -6
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/create_llm_cost_tracker_async_ingestion.rb.erb +1 -1
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/create_llm_cost_tracker_call_rollups.rb.erb +1 -1
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/create_llm_cost_tracker_calls.rb.erb +1 -1
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/initializer.rb.erb +14 -12
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/upgrade_call_rollups_provider.rb.erb +1 -1
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/upgrade_call_tags_key_value_index.rb.erb +1 -1
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/upgrade_image_tokens.rb.erb +1 -1
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/upgrade_indexes.rb.erb +1 -1
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/upgrade_per_tag_budgets.rb.erb +1 -1
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/upgrade_call_rollups_provider_generator.rb +0 -6
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/upgrade_call_tags_key_value_index_generator.rb +0 -6
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/upgrade_image_tokens_generator.rb +0 -6
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/upgrade_indexes_generator.rb +3 -8
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/upgrade_per_tag_budgets_generator.rb +3 -8
- data/lib/llm_cost_tracker/ingestion/batch.rb +37 -8
- data/lib/llm_cost_tracker/ingestion/inbox.rb +2 -8
- data/lib/llm_cost_tracker/ingestion.rb +3 -1
- data/lib/llm_cost_tracker/integrations/anthropic.rb +70 -11
- data/lib/llm_cost_tracker/integrations/base.rb +56 -19
- data/lib/llm_cost_tracker/integrations/openai/batch_capture.rb +21 -13
- data/lib/llm_cost_tracker/integrations/openai/patches.rb +8 -0
- data/lib/llm_cost_tracker/integrations/openai.rb +42 -59
- data/lib/llm_cost_tracker/integrations/ruby_llm.rb +275 -75
- data/lib/llm_cost_tracker/integrations.rb +1 -20
- data/lib/llm_cost_tracker/ledger/isolation.rb +23 -0
- data/lib/llm_cost_tracker/ledger/period/totals.rb +16 -11
- data/lib/llm_cost_tracker/ledger/rollups.rb +25 -17
- data/lib/llm_cost_tracker/ledger/schema/adapter.rb +14 -3
- data/lib/llm_cost_tracker/ledger/schema/base.rb +5 -4
- data/lib/llm_cost_tracker/ledger/storable.rb +16 -0
- data/lib/llm_cost_tracker/ledger/store.rb +14 -13
- data/lib/llm_cost_tracker/ledger/tags/breakdown.rb +1 -5
- data/lib/llm_cost_tracker/ledger/tags/encoding.rb +3 -12
- data/lib/llm_cost_tracker/ledger/tags/query.rb +0 -2
- data/lib/llm_cost_tracker/ledger.rb +1 -0
- data/lib/llm_cost_tracker/logging.rb +3 -1
- data/lib/llm_cost_tracker/middleware/faraday.rb +67 -61
- data/lib/llm_cost_tracker/parsers.rb +12 -41
- data/lib/llm_cost_tracker/prices.json +3071 -249
- data/lib/llm_cost_tracker/pricing/backfill.rb +46 -13
- data/lib/llm_cost_tracker/pricing/calculation.rb +118 -75
- data/lib/llm_cost_tracker/pricing/effective_prices.rb +11 -10
- data/lib/llm_cost_tracker/pricing/estimator.rb +5 -2
- data/lib/llm_cost_tracker/pricing/matcher.rb +36 -22
- data/lib/llm_cost_tracker/pricing/mode.rb +2 -13
- data/lib/llm_cost_tracker/pricing/price_key.rb +5 -3
- data/lib/llm_cost_tracker/pricing/rate.rb +1 -0
- data/lib/llm_cost_tracker/pricing/registry.rb +17 -17
- data/lib/llm_cost_tracker/pricing/service_rates.rb +1 -8
- data/lib/llm_cost_tracker/pricing/sync/change_printer.rb +6 -1
- data/lib/llm_cost_tracker/pricing/sync/registry_diff.rb +13 -9
- data/lib/llm_cost_tracker/pricing/sync/snapshot_guard.rb +47 -0
- data/lib/llm_cost_tracker/pricing/sync.rb +38 -70
- data/lib/llm_cost_tracker/providers/anthropic/parser.rb +33 -3
- data/lib/llm_cost_tracker/providers/anthropic/response_parser.rb +11 -5
- data/lib/llm_cost_tracker/providers/anthropic/usage_extractor.rb +81 -12
- data/lib/llm_cost_tracker/providers/azure/parser.rb +22 -7
- data/lib/llm_cost_tracker/providers/gemini/parser.rb +175 -47
- data/lib/llm_cost_tracker/providers/gemini/usage_extractor.rb +13 -22
- data/lib/llm_cost_tracker/providers/openai/hosts.rb +1 -1
- data/lib/llm_cost_tracker/providers/openai/parser.rb +11 -13
- data/lib/llm_cost_tracker/providers/openai/response_parser.rb +101 -27
- data/lib/llm_cost_tracker/providers/openai/service_charges.rb +57 -40
- data/lib/llm_cost_tracker/providers/openai/usage_extractor.rb +24 -6
- data/lib/llm_cost_tracker/providers/openai_compatible/parser.rb +7 -1
- data/lib/llm_cost_tracker/redaction.rb +32 -0
- data/lib/llm_cost_tracker/report/data.rb +1 -1
- data/lib/llm_cost_tracker/retention.rb +1 -3
- data/lib/llm_cost_tracker/tags/sanitizer.rb +10 -36
- data/lib/llm_cost_tracker/tracker.rb +17 -10
- data/lib/llm_cost_tracker/usage/catalog.rb +13 -4
- data/lib/llm_cost_tracker/usage/dimension.rb +1 -1
- data/lib/llm_cost_tracker/usage/dimensions.yml +61 -0
- data/lib/llm_cost_tracker/version.rb +1 -1
- data/lib/llm_cost_tracker.rb +4 -2
- data/lib/tasks/llm_cost_tracker.rake +26 -21
- data/llm_cost_tracker.gemspec +63 -0
- metadata +23 -10
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 2c1f13f9f919518298bf4aa5bb9bf33f6fb48851a963d132d48000c2c9ab2477
|
|
4
|
+
data.tar.gz: 1de730cdb5061df2588c74dd0c56ec29aa59ec0fe812188e2ae30df4e206c682
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: ee3ac1c8844c4bc478f85899d9e74734d67040d4b4f124c1751a970924a6d4ccf6fd2d680a2ed2a0d0e51ea79d7e985ad623ff49ed88148129df0276ba6c0925
|
|
7
|
+
data.tar.gz: 2aaa9cad0cd45a6e7df0920f054c258d12f03b2edc99c1c3bcf846d4ba33f4ce62972419cc92c5014711e417cb5f12eee63df2ac2f3c3e691acf47c8c620fdb8
|
data/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,134 @@
|
|
|
2
2
|
|
|
3
3
|
Format: [Keep a Changelog](https://keepachangelog.com/en/1.1.0/). Versioning: [SemVer](https://semver.org/spec/v2.0.0.html).
|
|
4
4
|
|
|
5
|
+
## [0.14.2] - 2026-09-28
|
|
6
|
+
|
|
7
|
+
### Added
|
|
8
|
+
|
|
9
|
+
- `bin/rails llm_cost_tracker:reprice FROM=... [TO=...]` reprices recorded calls, their rollups and per-tag costs at current prices, except provider-billed costs and costs passed to `track`.
|
|
10
|
+
- Bundled xAI and Mistral prices, with batch, priority and regional rates; Faraday captures both once their hosts are in `capture.openai_compatible_providers`.
|
|
11
|
+
- `bin/rails llm_cost_tracker:doctor` warns when `pricing.file` is older than the bundled prices or this month's `:cache` rollups do not match the calls ledger.
|
|
12
|
+
- Gemini Interactions API calls through the Faraday middleware, streamed and `background: true` ones included, are recorded and priced with their grounding.
|
|
13
|
+
- Creating an explicit Gemini context cache through Faraday or `RubyLLM.cache` on RubyLLM 2.x records its estimated storage cost until the cache expires.
|
|
14
|
+
- Price fields with a `_from_YYYY-MM-DD` suffix apply to calls from that date, including in `backfill_unknown_pricing` and `reprice`; bundled Gemini prices use it for announced price changes.
|
|
15
|
+
|
|
16
|
+
### Changed
|
|
17
|
+
|
|
18
|
+
- BREAKING: Rails 8.0+ required; Rails 7.1 and 7.2 no longer receive security fixes upstream.
|
|
19
|
+
- Calls returning a billed `usage.cost`, such as OpenRouter's, are recorded at the billed amount instead of a list-price estimate or `pricing.overrides` rate.
|
|
20
|
+
- The official openai gem pointed at a host in `capture.openai_compatible_providers` records that provider instead of `openai`.
|
|
21
|
+
- A provider-scoped price (`<provider>/<model>`) prices another provider's calls only when that provider has no price for the model.
|
|
22
|
+
- With `budgets.totals_source = :cache`, monthly budgets read past days from the rollups: run `bin/rails llm_cost_tracker:rebuild_rollups` once after deploying and after switching to `:cache`, or they are under-counted.
|
|
23
|
+
- OpenAI Realtime cached audio and image and Gemini cached audio are priced at their own cached rates and stored as `audio_token` / `image_token` line items with `cache_state: read`.
|
|
24
|
+
- `bin/rails llm_cost_tracker:prices:refresh` no longer takes `PREVIEW=1` and stops when it is set; use `bin/rails llm_cost_tracker:prices:check` instead.
|
|
25
|
+
|
|
26
|
+
### Removed
|
|
27
|
+
|
|
28
|
+
- The boot warning for `:ruby_llm` enabled together with `:openai` or `:anthropic`: RubyLLM doesn't call those SDKs, so nothing was recorded twice, and disabling one as it advised lost those calls.
|
|
29
|
+
|
|
30
|
+
### Fixed
|
|
31
|
+
|
|
32
|
+
- Bundled prices cover more OpenAI and Gemini models that recorded unknown cost, and `omni-moderation` calls are `free` instead of `unknown`.
|
|
33
|
+
- `gpt-5.5-pro` prompts over 272K tokens use long-context rates (unknown on Batch and Flex), and `gpt-5.4` Batch and Flex long-context cached input is corrected.
|
|
34
|
+
- OpenAI cached input on Pro models and cache writes on models before GPT-5.6 are priced at the input rate; these calls were `partial`.
|
|
35
|
+
- OpenAI duration-billed transcriptions are priced per second instead of per started minute, and are no longer free in streams or missing through RubyLLM.
|
|
36
|
+
- Transcriptions returned without usage are recorded with unknown cost; they were recorded at $0 or not at all.
|
|
37
|
+
- OpenAI image, transcription and batch calls on regional hosts, and Anthropic batches with `inference_geo: us`, include the data-residency uplift for models with regional rates.
|
|
38
|
+
- The Faraday middleware reads the model from multipart uploads, which were recorded as `unknown`, and records `/v1/audio/speech` by input characters.
|
|
39
|
+
- OpenAI Hosted Shell calls in a hosted container are captured as `container_session` line items, like Code Interpreter.
|
|
40
|
+
- OpenAI Responses created with `background: true` are recorded once, when a poll through the OpenAI SDK or Faraday returns them finished.
|
|
41
|
+
- Responses calls using the `image_generation` tool, RubyLLM 2.x OpenAI chats included, are recorded `partial` instead of `complete` without the image charge.
|
|
42
|
+
- Chat Completions web search fees apply to every call to OpenAI's search models, streams included, and no longer to other models' `url_citation` annotations.
|
|
43
|
+
- A stream on a priced model that ends without usage and has no priced tool charges stores a `nil` total instead of `0.0`, so it shows as unpriced and is in `Call.without_cost`.
|
|
44
|
+
- OpenAI, Azure OpenAI and Anthropic SDK streams price the service tier and speed the provider served, not the requested `priority` or `fast`.
|
|
45
|
+
- OpenAI streams with logprobs record their usage instead of 0 tokens and $0, and an overflowing SDK or `track_stream` capture logs a warning.
|
|
46
|
+
- Groq streams whose usage arrives only in `x_groq.usage` record their tokens instead of 0 tokens and `unknown`.
|
|
47
|
+
- Anthropic streams price cache writes made after `message_start`.
|
|
48
|
+
- Anthropic compaction tokens, which the top-level usage leaves out, are counted; on-demand compactions were recorded as free (not yet through RubyLLM).
|
|
49
|
+
- Anthropic server-side fallback calls are recorded under the serving model, and billed fallback attempts and advisor iterations use their own model's rates; an unpriced one triggers `pricing.unknown_model_behavior` (not yet through RubyLLM).
|
|
50
|
+
- Refusals Anthropic does not bill are recorded at $0 (not yet through RubyLLM), and refusals `Anthropic::BetaRefusalFallbackMiddleware` retried are recorded instead of dropped.
|
|
51
|
+
- Gemini image and audio prompt tokens on single-rate models are priced at the input rate; they were unpriced.
|
|
52
|
+
- Gemini image models are priced per 1M image tokens and price text and thinking output at the text rate; `gemini-2.5-flash-image` images were priced about 770x too low and 3.x image models' images 10-20x too low. Calls already recorded keep their cost. With a local pricing file, run `bin/rails llm_cost_tracker:prices:check`, check that only Gemini image models are flagged, then run `bin/rails llm_cost_tracker:prices:refresh FORCE=1`.
|
|
53
|
+
- Gemini 3 grounding is priced per unique non-empty query, `track_stream` included; image search queries, Maps grounding and the 3.x image models' grounding are now priced.
|
|
54
|
+
- Gemini calls are recorded under the response's `modelVersion`, so `-latest` aliases are priced and `track_stream(provider: :gemini)` works without `model:`.
|
|
55
|
+
- `track_stream(provider: :gemini)` parses native Gemini chunks when `generativelanguage.googleapis.com` is also an OpenAI-compatible provider named `gemini`.
|
|
56
|
+
- Amazon Bedrock Claude ids and inference profiles price as the Anthropic model, with the regional-profile premium from Claude 4.5; GovCloud is not priced at its rate.
|
|
57
|
+
- Groq batch calls are priced at Groq's batch rate, cached tokens included.
|
|
58
|
+
- OpenAI batches that end `expired` or `cancelled` record their completed requests, and image batch results use the batch's model and batch image rates.
|
|
59
|
+
- OpenAI and Anthropic batch results are stored once however often or concurrently the batch is fetched; they could be recorded twice.
|
|
60
|
+
- With `config.enabled = false`, batch retrieval no longer downloads OpenAI output files or queries the database.
|
|
61
|
+
- RubyLLM prices Anthropic US inference, fast mode and OpenAI regional hosts, and its streams outside Bedrock read the service tier, 1-hour cache writes and response id.
|
|
62
|
+
- RubyLLM chats, streamed ones included, record Anthropic and OpenAI web search and Gemini grounding fees; they were stored `complete` without the fee.
|
|
63
|
+
- RubyLLM Anthropic streams count the final cumulative input and every `pause_turn` segment's input, and chats keep earlier segments' cache writes.
|
|
64
|
+
- RubyLLM Gemini chats, Interactions protocol included, are priced from the raw usage, keeping audio and URL-context tool tokens, tiers, grounding and response ids.
|
|
65
|
+
- RubyLLM `paint` prices image output at the image output rate for Gemini image models, `gpt-image-1` and `gpt-image-1-mini`; the image was unpriced.
|
|
66
|
+
- RubyLLM transcriptions price prompt text and audio at their own rates when the response splits them (Gemini without it as text), and streams record `stream: true`.
|
|
67
|
+
- RubyLLM Bedrock Converse chats with prompt caching no longer subtract cache tokens from input twice, and split cache writes into 5-minute and 1-hour writes.
|
|
68
|
+
- RubyLLM Gemini embeddings record their tokens instead of 0, and `gemini-embedding-2` prices image, PDF, audio and video parts by modality, adding a `video_input` rate.
|
|
69
|
+
- The pre-send budget estimate covers RubyLLM calls, whose estimate was $0, and ignores base64 images, PDFs and audio, which inflated it and blocked ordinary vision requests.
|
|
70
|
+
- `llm_cost_tracker:backfill_unknown_pricing` also reprices `partial` calls whose already-priced rates are unchanged, and adds only the difference to rollups.
|
|
71
|
+
- Mode rates derived from a standard cache rate are named by their mode key, such as `batch_cache_read_input`, in `pricing_snapshot` and on line items.
|
|
72
|
+
- With inline ingestion, a per-tag `on_exceeded` fires when several calls for one tag value cross the limit together; each call read the others' spend, so none saw itself as the crossing call and the alert never fired.
|
|
73
|
+
- Under `:raise` and `:block_requests`, every budget a call crosses fires its `on_exceeded` before the first error is raised.
|
|
74
|
+
- `budgets.per_tag` without a `:block_requests` rule no longer checks the database schema before an LLM call is sent, so a process that starts during a database outage no longer fails its LLM calls.
|
|
75
|
+
- A tag key given as both a Symbol and a String is stored once, with the later value, not twice; requeue any async inbox rows quarantined because of it.
|
|
76
|
+
- Tags with a nil or empty value count as untagged on the dashboard, in `cost_by_tag` and for `budgets.per_tag`.
|
|
77
|
+
- An OpenAI SDK response without usage logs a warning instead of being skipped silently.
|
|
78
|
+
- The async worker checks its tables through the Rails schema cache, so an idle poll runs one query instead of six.
|
|
79
|
+
- With `ingestion.mode = :async` and `budgets.totals_source = :cache`, a missing rollups table no longer stops the worker from draining the inbox; it warns once and budget reads use the calls ledger, as inline ingestion does.
|
|
80
|
+
- `track_stream` records events passed as symbol-keyed hashes, such as `to_h` of an OpenAI Realtime `response.done` event; they were ignored and the call was stored as `unknown` with 0 tokens.
|
|
81
|
+
- An OpenAI SDK `chat.completions.stream` without `stream_options: { include_usage: true }` logs the same warning as the Faraday path instead of being stored as `unknown` silently.
|
|
82
|
+
- The overview's monthly budget bar and projection marker and the Data Quality coverage bars are visible again, and the dark theme gets its missing chart colors.
|
|
83
|
+
- In apps whose `Time.zone` is not UTC, daily charts, the previous-period line and the spend-anomaly banner use local days when the database knows the zone name.
|
|
84
|
+
- A tag value page's spend chart covers the selected date range instead of the last 30 days.
|
|
85
|
+
- Sortable dashboard tables no longer return a 500 when the host sets `config.action_controller.include_all_helpers = false`.
|
|
86
|
+
|
|
87
|
+
## [0.14.1] - 2026-09-25
|
|
88
|
+
|
|
89
|
+
### Added
|
|
90
|
+
|
|
91
|
+
- `bin/rails llm_cost_tracker:doctor` and the boot log warn when an instrumented SDK is newer than its tested range (RubyLLM 3.0 and later), instead of the integration reporting installed while it records nothing.
|
|
92
|
+
|
|
93
|
+
### Changed
|
|
94
|
+
|
|
95
|
+
- Ruby 3.3 is supported; the minimum drops from 3.4.
|
|
96
|
+
- The Faraday middleware adds `stream_options: { include_usage: true }` only for OpenAI, OpenRouter, DeepSeek, Groq, and Azure OpenAI on the v1 API or `api-version` 2024-06-01 and later without On Your Data or image input, since other servers can reject it and fail the request. Hosts you add to `config.capture.openai_compatible_providers` are left untouched: set the flag yourself if they support it, or their streams record `usage_source: unknown` with a warning.
|
|
97
|
+
- `bin/rails llm_cost_tracker:prices:refresh` refuses a snapshot that zeroes an existing price or charges for a free one, drops a model's `input` or `output` rate, moves a price 100-fold or more, or switches currency, and leaves your pricing file untouched, so one bad commit to the snapshot can no longer silence budgets or make `:block_requests` block every call. `PREVIEW=1` and `prices:check` list the changes; re-run with `FORCE=1` (or `force: true`) to accept them. New models and smaller changes are not checked, so keep reviewing the refreshed file.
|
|
98
|
+
- `pricing.unknown_model_behavior = :raise` records the call, with `cost_status: unknown`, before it raises `LlmCostTracker::UnknownPricingError`, so a call the provider already billed is no longer missing from the ledger.
|
|
99
|
+
- A malformed `pricing.file` fails `LlmCostTracker.configure` at boot instead of every LLM call. A missing one is logged at boot and reported as an error by `doctor`, and calls use bundled prices until `prices:refresh` creates it; before, every call raised and nothing was recorded.
|
|
100
|
+
|
|
101
|
+
### Removed
|
|
102
|
+
|
|
103
|
+
- An OpenAI or Anthropic SDK stream that your code never iterates is no longer recorded at garbage collection; the finalizer that did it kept every stream alive until process exit.
|
|
104
|
+
|
|
105
|
+
### Fixed
|
|
106
|
+
|
|
107
|
+
- Per-tag `:block_requests` budgets check tags passed to the Faraday middleware (`f.use :llm_cost_tracker, tags: ...`) before the call is sent; they saw only `with_tags` tags, so an over-budget tenant still reached the provider.
|
|
108
|
+
- RubyLLM transcriptions are priced at the model's audio input rate when it has one; they were priced as text, at a third to a half of the real cost.
|
|
109
|
+
- `RubyLLM.transcribe` with a Gemini or Vertex AI model on RubyLLM 1.x is recorded; RubyLLM 1.x sent it through its own Gemini method, bypassing the wrapped `RubyLLM::Provider#transcribe`.
|
|
110
|
+
- Calls made through RubyLLM 2.x are recorded. RubyLLM 2.0 moved token counts behind `response.tokens`, so every call was dropped with a warning while `doctor` reported the integration installed. A 2.x `paint(count:)` returning several images is recorded once, as RubyLLM bills it.
|
|
111
|
+
- An automatically captured LLM call no longer fails, losing the provider's response, when the gem cannot record it, such as when the async inbox pool times out. Only `BudgetExceededError`, `UnknownPricingError` under `:raise`, and `TransactionAbortedError` reach your code; other recording failures are logged. `LlmCostTracker.track` and `track_stream` still raise them.
|
|
112
|
+
- A recording error no longer replaces an exception your code raised while iterating an OpenAI or Anthropic SDK stream or inside `track_stream`, such as `Sidekiq::Shutdown`; the stream is still recorded, tagged `stream_errored`, and the recording error is logged. `TransactionAbortedError` still wins, with your exception as its cause.
|
|
113
|
+
- An OpenAI or Anthropic batch with an unpriced model under `:raise`, or one crossing a budget, records every result before raising once; results after the failing one were lost and `batches.retrieve` raised on every poll.
|
|
114
|
+
- With `ingestion.mode = :async`, one inbox row the database rejects no longer fails its whole batch, which after five attempts quarantined up to 99 other calls and dropped them from the ledger and budget totals. A failed batch is retried one row at a time and only rejected rows are marked failed. Rows already quarantined this way can be requeued as described in `docs/operations.md`.
|
|
115
|
+
- A NUL byte in a stored string, which PostgreSQL rejects, or invalid UTF-8 in a tag value no longer loses the call, and `LlmCostTracker.with_tags` no longer raises on invalid UTF-8. NUL bytes are removed on write, including from inbox rows written by earlier releases, and invalid UTF-8 in tag values becomes U+FFFD.
|
|
116
|
+
- An interrupted Gemini or Azure stream records its model from the request URL instead of `unknown`.
|
|
117
|
+
- The async inbox's `last_error` is no longer empty when the 1,000-byte cut splits a multibyte character.
|
|
118
|
+
- An LLM call inside your own database transaction no longer breaks it when the ledger write fails; on PostgreSQL your app's writes were lost with `PG::InFailedSqlTransaction`. Ledger writes, budget reads, and batch de-duplication run in a savepoint inside an open transaction. On MySQL, where a deadlock rolls back your whole transaction, the gem raises `LlmCostTracker::TransactionAbortedError` instead of swallowing it.
|
|
119
|
+
- With `config.budgets.totals_source = :cache`, the rollup increment inside your transaction waits for its commit on Rails 7.2 and later, so an open transaction no longer holds the rollup row lock and stalls every request recording spend for the same provider. On Rails 7.1 and in non-joinable transactions such as transactional fixtures it still runs immediately, and it is never retried inside a transaction.
|
|
120
|
+
- Long streams record their real usage instead of zero tokens and `$0`. Streams were buffered up to 1 MB, dropping the final usage event (past about 3,000 output tokens through Faraday); they are now decoded as they arrive, keeping only the first and last events plus billable tool-call and grounding events, so memory stays flat at any length. Image streams through Faraday keep their final usage too.
|
|
121
|
+
- Streams captured through the OpenAI and Anthropic SDK integrations are released once your code drops them, instead of staying in memory with their buffered events and request for the life of the process.
|
|
122
|
+
- OpenAI Realtime cached audio is no longer billed twice, at the cache-read rate and again at the audio rate ($0.258 instead of $0.106 for a 10,000-token `gpt-realtime` turn with 5,000 cached audio tokens). Cached audio and image tokens are priced at the model's cache-read rate, below OpenAI's published cached-audio price on the mini models. OpenAI-compatible usage whose audio and image buckets exceed the reported input has the overlap taken out of audio input, then image input.
|
|
123
|
+
- Dashboard links and filter forms carry only the dashboard's own query parameters. They copied every parameter, and Rails reads some as link options, so a crafted link such as `?script_name=//evil.example` pointed Export CSV, pagination, filter, and tag links at another site, others rewrote links or caused a 500, and a 50 KB junk parameter grew a tag page to 10 MB. A query string over 16 KB is a bad request.
|
|
124
|
+
- A dashboard page accepts at most 10 `tag[...]` filters, counting a tag value page's own value, and answers more with a bad request. Each filter adds a subquery, so PostgreSQL planning memory grew with the square of their count, and around 700 got the backend killed for running out of memory, restarting the database. Tag breakdowns no longer offer drill-down links past the limit.
|
|
125
|
+
- Dashboard filters handle bad input without a 500. A list or hash in a tag, provider, model, stream, or usage-source filter is a bad request instead of silently matching nothing, and a NUL byte in a filter value matches nothing on PostgreSQL instead of raising. A CSV export with an invalid filter, a missing call, or a database error renders an HTML error page with the right status instead of a 500 or HTML labelled as CSV. A tag value page's value overrides a filter on the same key, and tag pages for keys containing a dot, such as `/tags/team.name`, load instead of returning 404 or 406.
|
|
126
|
+
- The Pricing page shows the active price source instead of a 500 when `source` is a list or hash, as in `/pricing?source[]=bundled`; a list or hash `provider` filter is a bad request, as on other pages.
|
|
127
|
+
|
|
128
|
+
### Security
|
|
129
|
+
|
|
130
|
+
- From v0.9.0 through v0.14.0, a provider API key sent as a URL parameter could be stored in call tags. When a middleware after `f.use :llm_cost_tracker`, such as `f.response :raise_error`, raised on a streamed Gemini `?key=` or Azure `?api-key=` request, the `stream_interrupted_error` tag stored Faraday's error message with the full URL, shown on the dashboard, in CSV exports, and in notification payloads. The tag now holds only the error class, and the HTTP status goes into a new `stream_interrupted_status` tag. Rotate any key that may have been exposed, and clear stored messages with `UPDATE llm_cost_tracker_call_tags SET value = split_part(value, ': ', 1) WHERE key = 'stream_interrupted_error'` on PostgreSQL or ``UPDATE llm_cost_tracker_call_tags SET value = SUBSTRING_INDEX(value, ': ', 1) WHERE `key` = 'stream_interrupted_error'`` on MySQL. With async ingestion, run it after the inbox drains, and delete quarantined inbox rows whose `payload` contains `stream_interrupted_error`.
|
|
131
|
+
- Provider keys, tokens, `key=`/`token=`/`sig=` URL parameters, URL user info, and `Authorization` headers inside a longer string are replaced with `[REDACTED]` in tag values, log lines, the async inbox's `last_error`, `TransactionAbortedError` messages, and the `source_url` that `llm_cost_tracker:prices:refresh` writes and prints; before, only a tag value that was entirely a key was caught.
|
|
132
|
+
|
|
5
133
|
## [0.14.0] - 2026-08-26
|
|
6
134
|
|
|
7
135
|
### Added
|
data/README.md
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
# LLM Cost Tracker
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
Per-tenant LLM spend attribution and budgets for Rails — in your database, no proxy.
|
|
4
4
|
|
|
5
5
|
[](https://rubygems.org/gems/llm_cost_tracker) [](https://github.com/sergey-homenko/llm_cost_tracker/actions) [](https://codecov.io/gh/sergey-homenko/llm_cost_tracker)
|
|
6
6
|
|
|
7
|
-
Every call
|
|
7
|
+
Every call through RubyLLM, the official OpenAI and Anthropic SDKs, Gemini, or any OpenAI-compatible API is logged with tokens, cost, and your tags. Budgets can block a tenant's next call before it is sent.
|
|
8
8
|
|
|
9
9
|
Not Langfuse, Helicone, or LiteLLM. No prompts, no traces, no replay. Spend attribution only.
|
|
10
10
|
|
|
11
|
-
Requires Ruby 3.
|
|
11
|
+
Requires Ruby 3.3+, Rails 8.0+, PostgreSQL or MySQL.
|
|
12
12
|
|
|
13
13
|
<picture> <source media="(prefers-color-scheme: dark)" srcset="docs/dashboard-overview-dark.png"> <img alt="LLM Cost Tracker dashboard" src="docs/dashboard-overview-light.png"> </picture>
|
|
14
14
|
|
|
@@ -58,7 +58,7 @@ The engine ships without authentication on purpose.
|
|
|
58
58
|
## What lands in the ledger
|
|
59
59
|
|
|
60
60
|
- **Calls.** Provider, model, total tokens, total cost, latency, status.
|
|
61
|
-
- **Line items.** Per-component breakdown — text/audio/cached tokens, tool charges (web search,
|
|
61
|
+
- **Line items.** Per-component breakdown — text/audio/cached tokens, tool charges (web search, grounding, container sessions).
|
|
62
62
|
- **Tags.** Whatever attribution you pass — user, feature, tenant, env.
|
|
63
63
|
- **Provider IDs.** Response, project, API key, workspace — for downstream audits.
|
|
64
64
|
- **Pricing snapshot.** So historical numbers don't drift when prices change.
|
|
@@ -73,11 +73,22 @@ The engine ships without authentication on purpose.
|
|
|
73
73
|
| Azure OpenAI | Faraday or official SDK (auto-detected on `*.openai.azure.com` and Foundry `*.services.ai.azure.com`, both deployments and `/openai/v1/...`) |
|
|
74
74
|
| Google Gemini | Faraday |
|
|
75
75
|
| `ruby-openai` | Faraday |
|
|
76
|
-
| OpenRouter, DeepSeek, Groq,
|
|
76
|
+
| OpenRouter, DeepSeek, Groq; xAI, Mistral, and other gateways (LiteLLM etc.) once their host is added to `config.capture.openai_compatible_providers` | OpenAI-compatible Faraday, or the official OpenAI SDK with `base_url` on that host |
|
|
77
77
|
| Anything else | `LlmCostTracker.track` |
|
|
78
78
|
|
|
79
79
|
Streams capture when the provider emits final usage. OpenAI Faraday streams to `/chat/completions` get `stream_options: { include_usage: true }` auto-injected so the final usage chunk lands in the ledger (opt out via `config.capture.request_stream_usage = false`).
|
|
80
80
|
|
|
81
|
+
Captured does not always mean priced:
|
|
82
|
+
|
|
83
|
+
| Cost comes from | Calls |
|
|
84
|
+
| --- | --- |
|
|
85
|
+
| The amount billed, from `usage.cost` in the response or final stream chunk | OpenRouter through Faraday, the official OpenAI SDK, or `track_stream`, and any OpenAI-compatible gateway that returns `usage.cost`; bundled prices apply when it is missing or the call comes through RubyLLM, other than an OpenRouter chat |
|
|
86
|
+
| Bundled [`prices.json`](lib/llm_cost_tracker/prices.json) | The OpenAI, Anthropic, Gemini, Groq, OpenRouter, xAI, and Mistral models it lists, and the same Claude models on Bedrock through RubyLLM |
|
|
87
|
+
| The OpenAI, Anthropic, or Gemini price for the same model name | Azure OpenAI (by the model in the response, not the deployment name), Vertex AI through RubyLLM, gateways that pass a listed model name through |
|
|
88
|
+
| Nothing: recorded with `cost_status: unknown` | DeepSeek, and through RubyLLM also Perplexity, Ollama, other Bedrock models, and Claude on GovCloud (`us-gov.` profiles) |
|
|
89
|
+
|
|
90
|
+
Add missing prices to `config.pricing.file` or `config.pricing.overrides` ([Pricing](docs/pricing.md)), then run `bin/rails llm_cost_tracker:backfill_unknown_pricing` to price the calls already recorded.
|
|
91
|
+
|
|
81
92
|
## What it isn't
|
|
82
93
|
|
|
83
94
|
- No proxy. Direct calls only.
|
|
@@ -154,6 +154,9 @@
|
|
|
154
154
|
--lct-shadow-md: 0 2px 6px rgba(0, 0, 0, 0.35), 0 1px 2px rgba(0, 0, 0, 0.25);
|
|
155
155
|
--lct-shadow-lg: 0 12px 32px rgba(0, 0, 0, 0.5), 0 2px 6px rgba(0, 0, 0, 0.3);
|
|
156
156
|
--lct-focus-ring: 0 0 0 3px rgba(124, 131, 255, 0.30);
|
|
157
|
+
--lct-chart-marker: rgba(230, 236, 245, 0.45);
|
|
158
|
+
--lct-chart-secondary: rgba(230, 236, 245, 0.32);
|
|
159
|
+
--lct-chip-remove: rgba(230, 236, 245, 0.58);
|
|
157
160
|
}
|
|
158
161
|
|
|
159
162
|
* { box-sizing: border-box; }
|
|
@@ -613,6 +616,19 @@
|
|
|
613
616
|
.lct-stack-swatch { width: 10px; height: 10px; border-radius: 2px; display: inline-block; }
|
|
614
617
|
.lct-stack-meta { color: var(--lct-muted); }
|
|
615
618
|
.lct-stack-empty { color: var(--lct-muted); font-size: var(--fs-sm); margin: 0; }
|
|
619
|
+
.lct-bar-track,
|
|
620
|
+
.lct-budget-track { background: var(--lct-surface-2); border-radius: 999px; overflow: hidden; }
|
|
621
|
+
.lct-bar-track { height: 8px; min-width: 96px; }
|
|
622
|
+
.lct-budget-track { height: 10px; position: relative; }
|
|
623
|
+
.lct-bar-fill,
|
|
624
|
+
.lct-budget-fill { height: 100%; display: block; background: var(--lct-accent); }
|
|
625
|
+
.lct-budget-fill--warn { background: var(--lct-warning); }
|
|
626
|
+
.lct-budget-fill--over { background: var(--lct-danger); }
|
|
627
|
+
.lct-budget-marker { position: absolute; top: 0; bottom: 0; border-left: 2px dashed var(--lct-chart-marker); }
|
|
628
|
+
.lct-budget-projection { display: flex; flex-wrap: wrap; gap: 8px 12px; align-items: baseline; margin: 10px 0 0; color: var(--lct-muted); font-size: var(--fs-sm); }
|
|
629
|
+
.lct-budget-projection strong { color: var(--lct-text); }
|
|
630
|
+
.lct-budget-projection-status { font-weight: 600; }
|
|
631
|
+
.lct-budget-projection-status--over { color: var(--lct-warning-copy); }
|
|
616
632
|
|
|
617
633
|
.lct-legend { display: inline-flex; align-items: center; gap: 6px; font-size: var(--fs-xs); color: var(--lct-muted); }
|
|
618
634
|
.lct-legend-dot { width: 9px; height: 9px; border-radius: 2px; }
|
|
@@ -4,11 +4,14 @@ require "securerandom"
|
|
|
4
4
|
|
|
5
5
|
module LlmCostTracker
|
|
6
6
|
class ApplicationController < ActionController::Base
|
|
7
|
+
MAX_QUERY_BYTES = 16 * 1024
|
|
8
|
+
|
|
7
9
|
layout "llm_cost_tracker/application"
|
|
8
10
|
|
|
9
11
|
protect_from_forgery with: :exception
|
|
10
12
|
|
|
11
13
|
before_action :set_dashboard_security_headers
|
|
14
|
+
before_action :reject_oversized_query
|
|
12
15
|
before_action :ensure_current_schema
|
|
13
16
|
before_action :assign_dashboard_date_range
|
|
14
17
|
|
|
@@ -22,6 +25,12 @@ module LlmCostTracker
|
|
|
22
25
|
|
|
23
26
|
private
|
|
24
27
|
|
|
28
|
+
def reject_oversized_query
|
|
29
|
+
return if request.query_string.bytesize <= MAX_QUERY_BYTES
|
|
30
|
+
|
|
31
|
+
raise LlmCostTracker::InvalidFilterError, "query string exceeds #{MAX_QUERY_BYTES / 1024} KB"
|
|
32
|
+
end
|
|
33
|
+
|
|
25
34
|
def ensure_current_schema
|
|
26
35
|
drift = LlmCostTracker::Dashboard::SetupState.current
|
|
27
36
|
return unless drift
|
|
@@ -40,16 +49,20 @@ module LlmCostTracker
|
|
|
40
49
|
end
|
|
41
50
|
|
|
42
51
|
def render_database_error(_error)
|
|
43
|
-
|
|
52
|
+
render_error_page("database", :internal_server_error)
|
|
44
53
|
end
|
|
45
54
|
|
|
46
55
|
def render_invalid_filter(error)
|
|
47
56
|
@error_message = error.message
|
|
48
|
-
|
|
57
|
+
render_error_page("invalid_filter", :bad_request)
|
|
49
58
|
end
|
|
50
59
|
|
|
51
60
|
def render_not_found
|
|
52
|
-
|
|
61
|
+
render_error_page("not_found", :not_found)
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
def render_error_page(name, status)
|
|
65
|
+
render "llm_cost_tracker/errors/#{name}", status: status, formats: :html, content_type: "text/html"
|
|
53
66
|
end
|
|
54
67
|
|
|
55
68
|
def set_dashboard_security_headers
|
|
@@ -16,11 +16,9 @@ module LlmCostTracker
|
|
|
16
16
|
}.freeze
|
|
17
17
|
|
|
18
18
|
def index
|
|
19
|
-
@sort = params[:sort].to_s
|
|
20
|
-
@dir = params[:dir].to_s
|
|
21
19
|
scope = Dashboard::Filter.call(params: params)
|
|
22
20
|
scope = scope.unknown_pricing if params[:cost_status].to_s == "incomplete"
|
|
23
|
-
ordered_scope = scope.order(*calls_order(
|
|
21
|
+
ordered_scope = scope.order(*calls_order(params[:sort].to_s, params[:dir].to_s))
|
|
24
22
|
|
|
25
23
|
respond_to do |format|
|
|
26
24
|
format.html do
|
|
@@ -5,13 +5,11 @@ module LlmCostTracker
|
|
|
5
5
|
MAX_ROWS = 200
|
|
6
6
|
|
|
7
7
|
def index
|
|
8
|
-
@sort = params[:sort].to_s
|
|
9
|
-
@dir = params[:dir].to_s
|
|
10
8
|
@rows = Dashboard::TopModels.call(
|
|
11
9
|
scope: Dashboard::Filter.call(params: params),
|
|
12
10
|
limit: MAX_ROWS,
|
|
13
|
-
sort:
|
|
14
|
-
direction:
|
|
11
|
+
sort: params[:sort].to_s,
|
|
12
|
+
direction: params[:dir].to_s
|
|
15
13
|
)
|
|
16
14
|
end
|
|
17
15
|
end
|
|
@@ -4,10 +4,10 @@ module LlmCostTracker
|
|
|
4
4
|
class PricingController < ApplicationController
|
|
5
5
|
def index
|
|
6
6
|
@overview = Dashboard::PricingOverview.call
|
|
7
|
-
requested = params[:source]
|
|
7
|
+
requested = params[:source].to_s.to_sym
|
|
8
8
|
@active_source = @overview.fetch(:sources).key?(requested) ? requested : @overview.fetch(:effective_source)
|
|
9
9
|
@source_data = @overview.fetch(:sources).fetch(@active_source)
|
|
10
|
-
@provider_filter = params[:provider].
|
|
10
|
+
@provider_filter = Dashboard::Params.scalar(params[:provider], :provider).presence
|
|
11
11
|
@rows = @source_data.fetch(:rows)
|
|
12
12
|
@rows = @rows.select { |row| row.provider == @provider_filter } if @provider_filter
|
|
13
13
|
@providers = @source_data.fetch(:rows).map(&:provider).compact.uniq.sort
|
|
@@ -7,22 +7,24 @@ module LlmCostTracker
|
|
|
7
7
|
end
|
|
8
8
|
|
|
9
9
|
def show
|
|
10
|
-
|
|
11
|
-
@value = params[:tag_value].to_s
|
|
10
|
+
@value = Dashboard::Params.scalar(params[:tag_value], :tag_value)
|
|
12
11
|
|
|
13
12
|
if @value.empty?
|
|
14
|
-
@
|
|
15
|
-
|
|
16
|
-
|
|
13
|
+
@breakdown = Dashboard::TagBreakdown.call(
|
|
14
|
+
scope: Dashboard::Filter.call(params: params),
|
|
15
|
+
key: params[:key],
|
|
16
|
+
sort: params[:sort].to_s,
|
|
17
|
+
direction: params[:dir].to_s
|
|
18
|
+
)
|
|
17
19
|
else
|
|
18
20
|
@key = LlmCostTracker::Tags::Key.validate!(
|
|
19
21
|
params[:key],
|
|
20
22
|
error_class: LlmCostTracker::InvalidFilterError
|
|
21
23
|
)
|
|
22
|
-
value_scope =
|
|
24
|
+
value_scope = Dashboard::Filter.call(params: params, tags: { @key => @value })
|
|
23
25
|
@value_total_cost = value_scope.sum(:total_cost).to_f
|
|
24
26
|
@value_calls = value_scope.count
|
|
25
|
-
@value_points = Dashboard::TimeSeries.call(scope: value_scope)
|
|
27
|
+
@value_points = Dashboard::TimeSeries.call(scope: value_scope, from: @from_date, to: @to_date)
|
|
26
28
|
end
|
|
27
29
|
end
|
|
28
30
|
end
|
|
@@ -13,6 +13,7 @@ module LlmCostTracker
|
|
|
13
13
|
include PaginationHelper
|
|
14
14
|
include TokenUsageHelper
|
|
15
15
|
include InlineStyleHelper
|
|
16
|
+
include SortableTableHelper
|
|
16
17
|
|
|
17
18
|
def dashboard_section
|
|
18
19
|
path = request.path.to_s
|
|
@@ -120,13 +121,10 @@ module LlmCostTracker
|
|
|
120
121
|
end
|
|
121
122
|
|
|
122
123
|
def tag_chip_entries(tags, limit: 3)
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
visible = normalized.first(limit).map do |key, value|
|
|
127
|
-
{ key: key.to_s, value: tag_value_summary(value) }
|
|
124
|
+
visible = tags.first(limit).map do |key, value|
|
|
125
|
+
{ key: key.to_s, value: truncate_text(value.to_s, TAG_VALUE_SUMMARY_BYTES) }
|
|
128
126
|
end
|
|
129
|
-
visible << { more:
|
|
127
|
+
visible << { more: tags.size - limit } if tags.size > limit
|
|
130
128
|
visible
|
|
131
129
|
end
|
|
132
130
|
|
|
@@ -135,7 +133,7 @@ module LlmCostTracker
|
|
|
135
133
|
end
|
|
136
134
|
|
|
137
135
|
def current_query(overrides = {})
|
|
138
|
-
request.query_parameters.symbolize_keys.merge(overrides)
|
|
136
|
+
request.query_parameters.symbolize_keys.slice(*LlmCostTracker::Dashboard::Params::QUERY_KEYS).merge(overrides)
|
|
139
137
|
end
|
|
140
138
|
|
|
141
139
|
def calls_query_for_model(provider:, model:)
|
|
@@ -144,25 +142,6 @@ module LlmCostTracker
|
|
|
144
142
|
|
|
145
143
|
private
|
|
146
144
|
|
|
147
|
-
def normalized_tags(tags)
|
|
148
|
-
return tags.transform_keys(&:to_s) if tags.is_a?(Hash)
|
|
149
|
-
|
|
150
|
-
JSON.parse(tags || "{}")
|
|
151
|
-
rescue JSON::ParserError, TypeError
|
|
152
|
-
{}
|
|
153
|
-
end
|
|
154
|
-
|
|
155
|
-
def tag_value_summary(value)
|
|
156
|
-
string = case value
|
|
157
|
-
when Hash, Array
|
|
158
|
-
JSON.generate(value)
|
|
159
|
-
else
|
|
160
|
-
value.to_s
|
|
161
|
-
end
|
|
162
|
-
|
|
163
|
-
truncate_text(string, TAG_VALUE_SUMMARY_BYTES)
|
|
164
|
-
end
|
|
165
|
-
|
|
166
145
|
def truncate_text(string, limit)
|
|
167
146
|
return string if string.bytesize <= limit
|
|
168
147
|
|
|
@@ -7,7 +7,7 @@ module LlmCostTracker
|
|
|
7
7
|
|
|
8
8
|
cfg = chart_config(points, comparison_points, height, y_ticks)
|
|
9
9
|
parts = [chart_svg_open(cfg), "<title>Daily spend trend</title>", chart_area_gradient_def]
|
|
10
|
-
parts.concat(
|
|
10
|
+
parts.concat((0..cfg[:y_ticks]).map { |i| chart_tick_line(cfg, i) })
|
|
11
11
|
parts << chart_paths(cfg)
|
|
12
12
|
parts.concat(chart_dots(cfg))
|
|
13
13
|
parts.concat(chart_x_labels(cfg))
|
|
@@ -34,7 +34,7 @@ module LlmCostTracker
|
|
|
34
34
|
peak_index = points.each_with_index.max_by { |point, _| point[:cost].to_f }&.last
|
|
35
35
|
{ width: width, height: height, pad: pad, plot_w: plot_w, plot_h: plot_h,
|
|
36
36
|
max_cost: max_cost, n: points.size, y_ticks: y_ticks, points: points, coords: coords,
|
|
37
|
-
|
|
37
|
+
comparison_coords: comparison_coords,
|
|
38
38
|
peak_index: peak_index }
|
|
39
39
|
end
|
|
40
40
|
|
|
@@ -60,21 +60,17 @@ module LlmCostTracker
|
|
|
60
60
|
"<svg #{attrs}>"
|
|
61
61
|
end
|
|
62
62
|
|
|
63
|
-
def chart_grid_and_axis(cfg)
|
|
64
|
-
(0..cfg[:y_ticks]).map { |i| chart_tick_line(cfg, i) }
|
|
65
|
-
end
|
|
66
|
-
|
|
67
63
|
def chart_tick_line(cfg, idx)
|
|
68
64
|
pad = cfg[:pad]
|
|
69
65
|
right_x = chart_fmt(pad[:left] + cfg[:plot_w])
|
|
70
66
|
left_x = chart_fmt(pad[:left])
|
|
71
67
|
text_x = chart_fmt(pad[:left] - 8)
|
|
72
68
|
value = cfg[:max_cost] * (cfg[:y_ticks] - idx).to_f / cfg[:y_ticks]
|
|
73
|
-
|
|
74
|
-
|
|
69
|
+
tick_y = pad[:top] + (cfg[:plot_h] * idx.to_f / cfg[:y_ticks])
|
|
70
|
+
y = chart_fmt(tick_y)
|
|
71
|
+
label_y = chart_fmt(tick_y + 3)
|
|
75
72
|
grid = %(<line class="lct-chart-grid" x1="#{left_x}" x2="#{right_x}" y1="#{y}" y2="#{y}"/>)
|
|
76
|
-
|
|
77
|
-
text = %(<text class="lct-chart-axis" x="#{text_x}" y="#{label_y}" text-anchor="end">$#{label}</text>)
|
|
73
|
+
text = %(<text class="lct-chart-axis" x="#{text_x}" y="#{label_y}" text-anchor="end">$#{chart_fmt(value)}</text>)
|
|
78
74
|
"#{grid}#{text}"
|
|
79
75
|
end
|
|
80
76
|
|
|
@@ -16,6 +16,11 @@ module LlmCostTracker
|
|
|
16
16
|
query
|
|
17
17
|
end
|
|
18
18
|
|
|
19
|
+
def tag_drilldown_allowed?(key)
|
|
20
|
+
tags = LlmCostTracker::Dashboard::Params.tag_query(current_query[:tag])
|
|
21
|
+
tags.except(key.to_s).size < LlmCostTracker::Dashboard::Filter::MAX_TAG_FILTERS
|
|
22
|
+
end
|
|
23
|
+
|
|
19
24
|
def hidden_query_fields(query, prefix: nil)
|
|
20
25
|
safe_join(query.flat_map do |key, value|
|
|
21
26
|
name = prefix ? "#{prefix}[#{key}]" : key.to_s
|
|
@@ -33,41 +33,21 @@ module LlmCostTracker
|
|
|
33
33
|
}.freeze
|
|
34
34
|
|
|
35
35
|
def token_usage_stack_components
|
|
36
|
-
token_usage_display_components(labels: COMPONENT_LABELS).select do |component|
|
|
37
|
-
component.fetch(:cost_key)
|
|
38
|
-
end
|
|
39
|
-
end
|
|
40
|
-
|
|
41
|
-
def call_line_item_costs_by_component(call)
|
|
42
|
-
call.line_items.each_with_object({}) do |line_item, accumulator|
|
|
43
|
-
component = LlmCostTracker::Usage::Catalog.token_priced_for(
|
|
44
|
-
kind: line_item.kind, direction: line_item.direction, cache_state: line_item.cache_state
|
|
45
|
-
)
|
|
46
|
-
accumulator[component.key] = line_item.cost if component && line_item.cost
|
|
47
|
-
end
|
|
48
|
-
end
|
|
49
|
-
|
|
50
|
-
private
|
|
51
|
-
|
|
52
|
-
def token_usage_display_components(labels:)
|
|
53
36
|
LlmCostTracker::Usage::Catalog.token_priced.map do |component|
|
|
54
37
|
token_key = component.token_key
|
|
55
38
|
{
|
|
56
39
|
token_key: token_key,
|
|
57
|
-
cost_key: component.cost_key,
|
|
58
40
|
price_key: component.key,
|
|
59
|
-
label:
|
|
41
|
+
label: COMPONENT_LABELS.fetch(token_key),
|
|
60
42
|
css_class: STACK_CLASSES[token_key]
|
|
61
43
|
}
|
|
62
|
-
end
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
}
|
|
70
|
-
]
|
|
44
|
+
end
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
def call_line_item_costs_by_component(call)
|
|
48
|
+
LlmCostTracker::Usage::Catalog.costs_by_component(
|
|
49
|
+
call.line_items.map { |item| [item.kind, item.direction, item.cache_state, item.cost] }
|
|
50
|
+
)
|
|
71
51
|
end
|
|
72
52
|
end
|
|
73
53
|
end
|
|
@@ -45,7 +45,10 @@ module LlmCostTracker
|
|
|
45
45
|
def already_recorded?(provider:, provider_response_id:)
|
|
46
46
|
return false if provider_response_id.to_s.empty?
|
|
47
47
|
|
|
48
|
-
|
|
48
|
+
Ledger::Isolation.guard(self) do
|
|
49
|
+
where(provider: provider, provider_response_id: provider_response_id)
|
|
50
|
+
.where.not(usage_source: Usage::Source::UNKNOWN).exists?
|
|
51
|
+
end
|
|
49
52
|
end
|
|
50
53
|
|
|
51
54
|
def by_tag(key, value) = by_tags(key => value)
|
|
@@ -86,8 +89,12 @@ module LlmCostTracker
|
|
|
86
89
|
|
|
87
90
|
def latency_by_provider = group(:provider).average(:latency_ms).transform_values(&:to_f)
|
|
88
91
|
|
|
89
|
-
def group_by_period(period, column: :tracked_at)
|
|
90
|
-
|
|
92
|
+
def group_by_period(period, column: :tracked_at, time_zone: nil)
|
|
93
|
+
column = column.to_s
|
|
94
|
+
raise ArgumentError, "invalid period column: #{column.inspect}" unless column_names.include?(column)
|
|
95
|
+
|
|
96
|
+
bucket = Ledger::Schema::Adapter.period_bucket_sql(connection, period, qualified(column), time_zone: time_zone)
|
|
97
|
+
group(Arel.sql(bucket))
|
|
91
98
|
end
|
|
92
99
|
|
|
93
100
|
def daily_costs(days: 30)
|
|
@@ -109,17 +116,6 @@ module LlmCostTracker
|
|
|
109
116
|
relation = relation.limit(limit) if limit
|
|
110
117
|
relation
|
|
111
118
|
end
|
|
112
|
-
|
|
113
|
-
def period_group_expression(period, column:)
|
|
114
|
-
Ledger::Schema::Adapter.period_bucket_sql(connection, period, period_column_expression(column))
|
|
115
|
-
end
|
|
116
|
-
|
|
117
|
-
def period_column_expression(column)
|
|
118
|
-
column = column.to_s
|
|
119
|
-
return "#{quoted_table_name}.#{connection.quote_column_name(column)}" if column_names.include?(column)
|
|
120
|
-
|
|
121
|
-
raise ArgumentError, "invalid period column: #{column.inspect}"
|
|
122
|
-
end
|
|
123
119
|
end
|
|
124
120
|
|
|
125
121
|
def tag_pairs
|