llm_cost_tracker 0.14.1 → 0.14.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +82 -0
- data/README.md +16 -5
- data/app/assets/llm_cost_tracker/application.css +16 -0
- data/app/controllers/llm_cost_tracker/calls_controller.rb +1 -3
- data/app/controllers/llm_cost_tracker/models_controller.rb +2 -4
- data/app/controllers/llm_cost_tracker/tags_controller.rb +5 -4
- data/app/helpers/llm_cost_tracker/application_helper.rb +4 -25
- data/app/helpers/llm_cost_tracker/chart_helper.rb +6 -10
- data/app/helpers/llm_cost_tracker/token_usage_helper.rb +8 -28
- data/app/models/llm_cost_tracker/call.rb +10 -14
- data/app/services/llm_cost_tracker/dashboard/data_quality.rb +4 -15
- data/app/services/llm_cost_tracker/dashboard/filter.rb +5 -9
- data/app/services/llm_cost_tracker/dashboard/pagination.rb +2 -5
- data/app/services/llm_cost_tracker/dashboard/pricing_overview.rb +2 -1
- data/app/services/llm_cost_tracker/dashboard/spend_anomaly.rb +1 -1
- data/app/services/llm_cost_tracker/dashboard/tag_breakdown.rb +2 -3
- data/app/services/llm_cost_tracker/dashboard/tag_key_explorer.rb +1 -0
- data/app/services/llm_cost_tracker/dashboard/time_series.rb +3 -5
- data/app/services/llm_cost_tracker/dashboard/top_models.rb +1 -2
- data/app/views/llm_cost_tracker/calls/show.html.erb +8 -22
- data/app/views/llm_cost_tracker/data_quality/index.html.erb +1 -1
- data/app/views/llm_cost_tracker/shared/_bar.html.erb +1 -3
- data/app/views/llm_cost_tracker/shared/_filter_pill_date.html.erb +1 -4
- data/app/views/llm_cost_tracker/shared/_filter_pill_model.html.erb +1 -4
- data/app/views/llm_cost_tracker/shared/_filter_pill_provider.html.erb +1 -4
- data/app/views/llm_cost_tracker/shared/_filter_pill_stream.html.erb +1 -4
- data/app/views/llm_cost_tracker/tags/show.html.erb +5 -5
- data/lib/llm_cost_tracker/budget/per_tag.rb +15 -9
- data/lib/llm_cost_tracker/budget.rb +44 -55
- data/lib/llm_cost_tracker/capture/event_window.rb +3 -2
- data/lib/llm_cost_tracker/capture/sdk_payload.rb +5 -1
- data/lib/llm_cost_tracker/capture/stream_collector.rb +25 -26
- data/lib/llm_cost_tracker/capture/stream_tracker.rb +9 -23
- data/lib/llm_cost_tracker/capture_verifier.rb +1 -7
- data/lib/llm_cost_tracker/charges/line_item.rb +5 -1
- data/lib/llm_cost_tracker/configuration/budgets.rb +1 -1
- data/lib/llm_cost_tracker/configuration/pricing.rb +1 -1
- data/lib/llm_cost_tracker/configuration.rb +5 -9
- data/lib/llm_cost_tracker/doctor/price_check.rb +13 -3
- data/lib/llm_cost_tracker/doctor/schema_check.rb +1 -2
- data/lib/llm_cost_tracker/doctor.rb +19 -2
- data/lib/llm_cost_tracker/engine.rb +0 -1
- data/lib/llm_cost_tracker/event.rb +8 -0
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/async_ingestion_generator.rb +0 -6
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/call_rollups_generator.rb +5 -8
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/install_generator.rb +0 -6
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/create_llm_cost_tracker_async_ingestion.rb.erb +1 -1
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/create_llm_cost_tracker_call_rollups.rb.erb +1 -1
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/create_llm_cost_tracker_calls.rb.erb +1 -1
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/initializer.rb.erb +11 -9
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/upgrade_call_rollups_provider.rb.erb +1 -1
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/upgrade_call_tags_key_value_index.rb.erb +1 -1
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/upgrade_image_tokens.rb.erb +1 -1
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/upgrade_indexes.rb.erb +1 -1
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/upgrade_per_tag_budgets.rb.erb +1 -1
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/upgrade_call_rollups_provider_generator.rb +0 -6
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/upgrade_call_tags_key_value_index_generator.rb +0 -6
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/upgrade_image_tokens_generator.rb +0 -6
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/upgrade_indexes_generator.rb +3 -8
- data/lib/llm_cost_tracker/generators/llm_cost_tracker/upgrade_per_tag_budgets_generator.rb +3 -8
- data/lib/llm_cost_tracker/ingestion/inbox.rb +2 -8
- data/lib/llm_cost_tracker/ingestion.rb +3 -1
- data/lib/llm_cost_tracker/integrations/anthropic.rb +63 -10
- data/lib/llm_cost_tracker/integrations/base.rb +39 -18
- data/lib/llm_cost_tracker/integrations/openai/batch_capture.rb +16 -13
- data/lib/llm_cost_tracker/integrations/openai/patches.rb +8 -0
- data/lib/llm_cost_tracker/integrations/openai.rb +42 -59
- data/lib/llm_cost_tracker/integrations/ruby_llm.rb +241 -76
- data/lib/llm_cost_tracker/integrations.rb +1 -20
- data/lib/llm_cost_tracker/ledger/isolation.rb +0 -7
- data/lib/llm_cost_tracker/ledger/period/totals.rb +14 -10
- data/lib/llm_cost_tracker/ledger/rollups.rb +14 -14
- data/lib/llm_cost_tracker/ledger/schema/adapter.rb +14 -3
- data/lib/llm_cost_tracker/ledger/schema/base.rb +5 -4
- data/lib/llm_cost_tracker/ledger/store.rb +2 -2
- data/lib/llm_cost_tracker/ledger/tags/breakdown.rb +1 -5
- data/lib/llm_cost_tracker/ledger/tags/encoding.rb +3 -12
- data/lib/llm_cost_tracker/ledger/tags/query.rb +0 -2
- data/lib/llm_cost_tracker/middleware/faraday.rb +36 -19
- data/lib/llm_cost_tracker/parsers.rb +7 -40
- data/lib/llm_cost_tracker/prices.json +2237 -154
- data/lib/llm_cost_tracker/pricing/backfill.rb +46 -13
- data/lib/llm_cost_tracker/pricing/calculation.rb +118 -75
- data/lib/llm_cost_tracker/pricing/effective_prices.rb +11 -10
- data/lib/llm_cost_tracker/pricing/estimator.rb +5 -2
- data/lib/llm_cost_tracker/pricing/matcher.rb +36 -22
- data/lib/llm_cost_tracker/pricing/mode.rb +2 -13
- data/lib/llm_cost_tracker/pricing/price_key.rb +5 -3
- data/lib/llm_cost_tracker/pricing/rate.rb +1 -0
- data/lib/llm_cost_tracker/pricing/registry.rb +5 -16
- data/lib/llm_cost_tracker/pricing/service_rates.rb +1 -8
- data/lib/llm_cost_tracker/pricing/sync/registry_diff.rb +13 -9
- data/lib/llm_cost_tracker/pricing/sync.rb +22 -71
- data/lib/llm_cost_tracker/providers/anthropic/parser.rb +33 -3
- data/lib/llm_cost_tracker/providers/anthropic/response_parser.rb +11 -5
- data/lib/llm_cost_tracker/providers/anthropic/usage_extractor.rb +81 -12
- data/lib/llm_cost_tracker/providers/azure/parser.rb +3 -7
- data/lib/llm_cost_tracker/providers/gemini/parser.rb +172 -48
- data/lib/llm_cost_tracker/providers/gemini/usage_extractor.rb +13 -22
- data/lib/llm_cost_tracker/providers/openai/hosts.rb +1 -1
- data/lib/llm_cost_tracker/providers/openai/parser.rb +11 -13
- data/lib/llm_cost_tracker/providers/openai/response_parser.rb +96 -26
- data/lib/llm_cost_tracker/providers/openai/service_charges.rb +57 -40
- data/lib/llm_cost_tracker/providers/openai/usage_extractor.rb +14 -1
- data/lib/llm_cost_tracker/providers/openai_compatible/parser.rb +3 -1
- data/lib/llm_cost_tracker/report/data.rb +1 -1
- data/lib/llm_cost_tracker/retention.rb +1 -3
- data/lib/llm_cost_tracker/tracker.rb +17 -9
- data/lib/llm_cost_tracker/usage/catalog.rb +13 -4
- data/lib/llm_cost_tracker/usage/dimension.rb +1 -1
- data/lib/llm_cost_tracker/usage/dimensions.yml +61 -0
- data/lib/llm_cost_tracker/version.rb +1 -1
- data/lib/tasks/llm_cost_tracker.rake +24 -21
- data/llm_cost_tracker.gemspec +63 -0
- metadata +8 -7
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 2c1f13f9f919518298bf4aa5bb9bf33f6fb48851a963d132d48000c2c9ab2477
|
|
4
|
+
data.tar.gz: 1de730cdb5061df2588c74dd0c56ec29aa59ec0fe812188e2ae30df4e206c682
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: ee3ac1c8844c4bc478f85899d9e74734d67040d4b4f124c1751a970924a6d4ccf6fd2d680a2ed2a0d0e51ea79d7e985ad623ff49ed88148129df0276ba6c0925
|
|
7
|
+
data.tar.gz: 2aaa9cad0cd45a6e7df0920f054c258d12f03b2edc99c1c3bcf846d4ba33f4ce62972419cc92c5014711e417cb5f12eee63df2ac2f3c3e691acf47c8c620fdb8
|
data/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,88 @@
|
|
|
2
2
|
|
|
3
3
|
Format: [Keep a Changelog](https://keepachangelog.com/en/1.1.0/). Versioning: [SemVer](https://semver.org/spec/v2.0.0.html).
|
|
4
4
|
|
|
5
|
+
## [0.14.2] - 2026-09-28
|
|
6
|
+
|
|
7
|
+
### Added
|
|
8
|
+
|
|
9
|
+
- `bin/rails llm_cost_tracker:reprice FROM=... [TO=...]` reprices recorded calls, their rollups and per-tag costs at current prices, except provider-billed costs and costs passed to `track`.
|
|
10
|
+
- Bundled xAI and Mistral prices, with batch, priority and regional rates; Faraday captures both once their hosts are in `capture.openai_compatible_providers`.
|
|
11
|
+
- `bin/rails llm_cost_tracker:doctor` warns when `pricing.file` is older than the bundled prices or this month's `:cache` rollups do not match the calls ledger.
|
|
12
|
+
- Gemini Interactions API calls through the Faraday middleware, streamed and `background: true` ones included, are recorded and priced with their grounding.
|
|
13
|
+
- Creating an explicit Gemini context cache through Faraday or `RubyLLM.cache` on RubyLLM 2.x records its estimated storage cost until the cache expires.
|
|
14
|
+
- Price fields with a `_from_YYYY-MM-DD` suffix apply to calls from that date, including in `backfill_unknown_pricing` and `reprice`; bundled Gemini prices use it for announced price changes.
|
|
15
|
+
|
|
16
|
+
### Changed
|
|
17
|
+
|
|
18
|
+
- BREAKING: Rails 8.0+ required; Rails 7.1 and 7.2 no longer receive security fixes upstream.
|
|
19
|
+
- Calls returning a billed `usage.cost`, such as OpenRouter's, are recorded at the billed amount instead of a list-price estimate or `pricing.overrides` rate.
|
|
20
|
+
- The official openai gem pointed at a host in `capture.openai_compatible_providers` records that provider instead of `openai`.
|
|
21
|
+
- A provider-scoped price (`<provider>/<model>`) prices another provider's calls only when that provider has no price for the model.
|
|
22
|
+
- With `budgets.totals_source = :cache`, monthly budgets read past days from the rollups: run `bin/rails llm_cost_tracker:rebuild_rollups` once after deploying and after switching to `:cache`, or they are under-counted.
|
|
23
|
+
- OpenAI Realtime cached audio and image and Gemini cached audio are priced at their own cached rates and stored as `audio_token` / `image_token` line items with `cache_state: read`.
|
|
24
|
+
- `bin/rails llm_cost_tracker:prices:refresh` no longer takes `PREVIEW=1` and stops when it is set; use `bin/rails llm_cost_tracker:prices:check` instead.
|
|
25
|
+
|
|
26
|
+
### Removed
|
|
27
|
+
|
|
28
|
+
- The boot warning for `:ruby_llm` enabled together with `:openai` or `:anthropic`: RubyLLM doesn't call those SDKs, so nothing was recorded twice, and disabling one as it advised lost those calls.
|
|
29
|
+
|
|
30
|
+
### Fixed
|
|
31
|
+
|
|
32
|
+
- Bundled prices cover more OpenAI and Gemini models that recorded unknown cost, and `omni-moderation` calls are `free` instead of `unknown`.
|
|
33
|
+
- `gpt-5.5-pro` prompts over 272K tokens use long-context rates (unknown on Batch and Flex), and `gpt-5.4` Batch and Flex long-context cached input is corrected.
|
|
34
|
+
- OpenAI cached input on Pro models and cache writes on models before GPT-5.6 are priced at the input rate; these calls were `partial`.
|
|
35
|
+
- OpenAI duration-billed transcriptions are priced per second instead of per started minute, and are no longer free in streams or missing through RubyLLM.
|
|
36
|
+
- Transcriptions returned without usage are recorded with unknown cost; they were recorded at $0 or not at all.
|
|
37
|
+
- OpenAI image, transcription and batch calls on regional hosts, and Anthropic batches with `inference_geo: us`, include the data-residency uplift for models with regional rates.
|
|
38
|
+
- The Faraday middleware reads the model from multipart uploads, which were recorded as `unknown`, and records `/v1/audio/speech` by input characters.
|
|
39
|
+
- OpenAI Hosted Shell calls in a hosted container are captured as `container_session` line items, like Code Interpreter.
|
|
40
|
+
- OpenAI Responses created with `background: true` are recorded once, when a poll through the OpenAI SDK or Faraday returns them finished.
|
|
41
|
+
- Responses calls using the `image_generation` tool, RubyLLM 2.x OpenAI chats included, are recorded `partial` instead of `complete` without the image charge.
|
|
42
|
+
- Chat Completions web search fees apply to every call to OpenAI's search models, streams included, and no longer to other models' `url_citation` annotations.
|
|
43
|
+
- A stream on a priced model that ends without usage and has no priced tool charges stores a `nil` total instead of `0.0`, so it shows as unpriced and is in `Call.without_cost`.
|
|
44
|
+
- OpenAI, Azure OpenAI and Anthropic SDK streams price the service tier and speed the provider served, not the requested `priority` or `fast`.
|
|
45
|
+
- OpenAI streams with logprobs record their usage instead of 0 tokens and $0, and an overflowing SDK or `track_stream` capture logs a warning.
|
|
46
|
+
- Groq streams whose usage arrives only in `x_groq.usage` record their tokens instead of 0 tokens and `unknown`.
|
|
47
|
+
- Anthropic streams price cache writes made after `message_start`.
|
|
48
|
+
- Anthropic compaction tokens, which the top-level usage leaves out, are counted; on-demand compactions were recorded as free (not yet through RubyLLM).
|
|
49
|
+
- Anthropic server-side fallback calls are recorded under the serving model, and billed fallback attempts and advisor iterations use their own model's rates; an unpriced one triggers `pricing.unknown_model_behavior` (not yet through RubyLLM).
|
|
50
|
+
- Refusals Anthropic does not bill are recorded at $0 (not yet through RubyLLM), and refusals `Anthropic::BetaRefusalFallbackMiddleware` retried are recorded instead of dropped.
|
|
51
|
+
- Gemini image and audio prompt tokens on single-rate models are priced at the input rate; they were unpriced.
|
|
52
|
+
- Gemini image models are priced per 1M image tokens and price text and thinking output at the text rate; `gemini-2.5-flash-image` images were priced about 770x too low and 3.x image models' images 10-20x too low. Calls already recorded keep their cost. With a local pricing file, run `bin/rails llm_cost_tracker:prices:check`, check that only Gemini image models are flagged, then run `bin/rails llm_cost_tracker:prices:refresh FORCE=1`.
|
|
53
|
+
- Gemini 3 grounding is priced per unique non-empty query, `track_stream` included; image search queries, Maps grounding and the 3.x image models' grounding are now priced.
|
|
54
|
+
- Gemini calls are recorded under the response's `modelVersion`, so `-latest` aliases are priced and `track_stream(provider: :gemini)` works without `model:`.
|
|
55
|
+
- `track_stream(provider: :gemini)` parses native Gemini chunks when `generativelanguage.googleapis.com` is also an OpenAI-compatible provider named `gemini`.
|
|
56
|
+
- Amazon Bedrock Claude ids and inference profiles price as the Anthropic model, with the regional-profile premium from Claude 4.5; GovCloud is not priced at its rate.
|
|
57
|
+
- Groq batch calls are priced at Groq's batch rate, cached tokens included.
|
|
58
|
+
- OpenAI batches that end `expired` or `cancelled` record their completed requests, and image batch results use the batch's model and batch image rates.
|
|
59
|
+
- OpenAI and Anthropic batch results are stored once however often or concurrently the batch is fetched; they could be recorded twice.
|
|
60
|
+
- With `config.enabled = false`, batch retrieval no longer downloads OpenAI output files or queries the database.
|
|
61
|
+
- RubyLLM prices Anthropic US inference, fast mode and OpenAI regional hosts, and its streams outside Bedrock read the service tier, 1-hour cache writes and response id.
|
|
62
|
+
- RubyLLM chats, streamed ones included, record Anthropic and OpenAI web search and Gemini grounding fees; they were stored `complete` without the fee.
|
|
63
|
+
- RubyLLM Anthropic streams count the final cumulative input and every `pause_turn` segment's input, and chats keep earlier segments' cache writes.
|
|
64
|
+
- RubyLLM Gemini chats, Interactions protocol included, are priced from the raw usage, keeping audio and URL-context tool tokens, tiers, grounding and response ids.
|
|
65
|
+
- RubyLLM `paint` prices image output at the image output rate for Gemini image models, `gpt-image-1` and `gpt-image-1-mini`; the image was unpriced.
|
|
66
|
+
- RubyLLM transcriptions price prompt text and audio at their own rates when the response splits them (Gemini without it as text), and streams record `stream: true`.
|
|
67
|
+
- RubyLLM Bedrock Converse chats with prompt caching no longer subtract cache tokens from input twice, and split cache writes into 5-minute and 1-hour writes.
|
|
68
|
+
- RubyLLM Gemini embeddings record their tokens instead of 0, and `gemini-embedding-2` prices image, PDF, audio and video parts by modality, adding a `video_input` rate.
|
|
69
|
+
- The pre-send budget estimate covers RubyLLM calls, whose estimate was $0, and ignores base64 images, PDFs and audio, which inflated it and blocked ordinary vision requests.
|
|
70
|
+
- `llm_cost_tracker:backfill_unknown_pricing` also reprices `partial` calls whose already-priced rates are unchanged, and adds only the difference to rollups.
|
|
71
|
+
- Mode rates derived from a standard cache rate are named by their mode key, such as `batch_cache_read_input`, in `pricing_snapshot` and on line items.
|
|
72
|
+
- With inline ingestion, a per-tag `on_exceeded` fires when several calls for one tag value cross the limit together; each call read the others' spend, so none saw itself as the crossing call and the alert never fired.
|
|
73
|
+
- Under `:raise` and `:block_requests`, every budget a call crosses fires its `on_exceeded` before the first error is raised.
|
|
74
|
+
- `budgets.per_tag` without a `:block_requests` rule no longer checks the database schema before an LLM call is sent, so a process that starts during a database outage no longer fails its LLM calls.
|
|
75
|
+
- A tag key given as both a Symbol and a String is stored once, with the later value, not twice; requeue any async inbox rows quarantined because of it.
|
|
76
|
+
- Tags with a nil or empty value count as untagged on the dashboard, in `cost_by_tag` and for `budgets.per_tag`.
|
|
77
|
+
- An OpenAI SDK response without usage logs a warning instead of being skipped silently.
|
|
78
|
+
- The async worker checks its tables through the Rails schema cache, so an idle poll runs one query instead of six.
|
|
79
|
+
- With `ingestion.mode = :async` and `budgets.totals_source = :cache`, a missing rollups table no longer stops the worker from draining the inbox; it warns once and budget reads use the calls ledger, as inline ingestion does.
|
|
80
|
+
- `track_stream` records events passed as symbol-keyed hashes, such as `to_h` of an OpenAI Realtime `response.done` event; they were ignored and the call was stored as `unknown` with 0 tokens.
|
|
81
|
+
- An OpenAI SDK `chat.completions.stream` without `stream_options: { include_usage: true }` logs the same warning as the Faraday path instead of being stored as `unknown` silently.
|
|
82
|
+
- The overview's monthly budget bar and projection marker and the Data Quality coverage bars are visible again, and the dark theme gets its missing chart colors.
|
|
83
|
+
- In apps whose `Time.zone` is not UTC, daily charts, the previous-period line and the spend-anomaly banner use local days when the database knows the zone name.
|
|
84
|
+
- A tag value page's spend chart covers the selected date range instead of the last 30 days.
|
|
85
|
+
- Sortable dashboard tables no longer return a 500 when the host sets `config.action_controller.include_all_helpers = false`.
|
|
86
|
+
|
|
5
87
|
## [0.14.1] - 2026-09-25
|
|
6
88
|
|
|
7
89
|
### Added
|
data/README.md
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
# LLM Cost Tracker
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
Per-tenant LLM spend attribution and budgets for Rails — in your database, no proxy.
|
|
4
4
|
|
|
5
5
|
[](https://rubygems.org/gems/llm_cost_tracker) [](https://github.com/sergey-homenko/llm_cost_tracker/actions) [](https://codecov.io/gh/sergey-homenko/llm_cost_tracker)
|
|
6
6
|
|
|
7
|
-
Every call
|
|
7
|
+
Every call through RubyLLM, the official OpenAI and Anthropic SDKs, Gemini, or any OpenAI-compatible API is logged with tokens, cost, and your tags. Budgets can block a tenant's next call before it is sent.
|
|
8
8
|
|
|
9
9
|
Not Langfuse, Helicone, or LiteLLM. No prompts, no traces, no replay. Spend attribution only.
|
|
10
10
|
|
|
11
|
-
Requires Ruby 3.3+, Rails
|
|
11
|
+
Requires Ruby 3.3+, Rails 8.0+, PostgreSQL or MySQL.
|
|
12
12
|
|
|
13
13
|
<picture> <source media="(prefers-color-scheme: dark)" srcset="docs/dashboard-overview-dark.png"> <img alt="LLM Cost Tracker dashboard" src="docs/dashboard-overview-light.png"> </picture>
|
|
14
14
|
|
|
@@ -58,7 +58,7 @@ The engine ships without authentication on purpose.
|
|
|
58
58
|
## What lands in the ledger
|
|
59
59
|
|
|
60
60
|
- **Calls.** Provider, model, total tokens, total cost, latency, status.
|
|
61
|
-
- **Line items.** Per-component breakdown — text/audio/cached tokens, tool charges (web search,
|
|
61
|
+
- **Line items.** Per-component breakdown — text/audio/cached tokens, tool charges (web search, grounding, container sessions).
|
|
62
62
|
- **Tags.** Whatever attribution you pass — user, feature, tenant, env.
|
|
63
63
|
- **Provider IDs.** Response, project, API key, workspace — for downstream audits.
|
|
64
64
|
- **Pricing snapshot.** So historical numbers don't drift when prices change.
|
|
@@ -73,11 +73,22 @@ The engine ships without authentication on purpose.
|
|
|
73
73
|
| Azure OpenAI | Faraday or official SDK (auto-detected on `*.openai.azure.com` and Foundry `*.services.ai.azure.com`, both deployments and `/openai/v1/...`) |
|
|
74
74
|
| Google Gemini | Faraday |
|
|
75
75
|
| `ruby-openai` | Faraday |
|
|
76
|
-
| OpenRouter, DeepSeek, Groq,
|
|
76
|
+
| OpenRouter, DeepSeek, Groq; xAI, Mistral, and other gateways (LiteLLM etc.) once their host is added to `config.capture.openai_compatible_providers` | OpenAI-compatible Faraday, or the official OpenAI SDK with `base_url` on that host |
|
|
77
77
|
| Anything else | `LlmCostTracker.track` |
|
|
78
78
|
|
|
79
79
|
Streams capture when the provider emits final usage. OpenAI Faraday streams to `/chat/completions` get `stream_options: { include_usage: true }` auto-injected so the final usage chunk lands in the ledger (opt out via `config.capture.request_stream_usage = false`).
|
|
80
80
|
|
|
81
|
+
Captured does not always mean priced:
|
|
82
|
+
|
|
83
|
+
| Cost comes from | Calls |
|
|
84
|
+
| --- | --- |
|
|
85
|
+
| The amount billed, from `usage.cost` in the response or final stream chunk | OpenRouter through Faraday, the official OpenAI SDK, or `track_stream`, and any OpenAI-compatible gateway that returns `usage.cost`; bundled prices apply when it is missing or the call comes through RubyLLM, other than an OpenRouter chat |
|
|
86
|
+
| Bundled [`prices.json`](lib/llm_cost_tracker/prices.json) | The OpenAI, Anthropic, Gemini, Groq, OpenRouter, xAI, and Mistral models it lists, and the same Claude models on Bedrock through RubyLLM |
|
|
87
|
+
| The OpenAI, Anthropic, or Gemini price for the same model name | Azure OpenAI (by the model in the response, not the deployment name), Vertex AI through RubyLLM, gateways that pass a listed model name through |
|
|
88
|
+
| Nothing: recorded with `cost_status: unknown` | DeepSeek, and through RubyLLM also Perplexity, Ollama, other Bedrock models, and Claude on GovCloud (`us-gov.` profiles) |
|
|
89
|
+
|
|
90
|
+
Add missing prices to `config.pricing.file` or `config.pricing.overrides` ([Pricing](docs/pricing.md)), then run `bin/rails llm_cost_tracker:backfill_unknown_pricing` to price the calls already recorded.
|
|
91
|
+
|
|
81
92
|
## What it isn't
|
|
82
93
|
|
|
83
94
|
- No proxy. Direct calls only.
|
|
@@ -154,6 +154,9 @@
|
|
|
154
154
|
--lct-shadow-md: 0 2px 6px rgba(0, 0, 0, 0.35), 0 1px 2px rgba(0, 0, 0, 0.25);
|
|
155
155
|
--lct-shadow-lg: 0 12px 32px rgba(0, 0, 0, 0.5), 0 2px 6px rgba(0, 0, 0, 0.3);
|
|
156
156
|
--lct-focus-ring: 0 0 0 3px rgba(124, 131, 255, 0.30);
|
|
157
|
+
--lct-chart-marker: rgba(230, 236, 245, 0.45);
|
|
158
|
+
--lct-chart-secondary: rgba(230, 236, 245, 0.32);
|
|
159
|
+
--lct-chip-remove: rgba(230, 236, 245, 0.58);
|
|
157
160
|
}
|
|
158
161
|
|
|
159
162
|
* { box-sizing: border-box; }
|
|
@@ -613,6 +616,19 @@
|
|
|
613
616
|
.lct-stack-swatch { width: 10px; height: 10px; border-radius: 2px; display: inline-block; }
|
|
614
617
|
.lct-stack-meta { color: var(--lct-muted); }
|
|
615
618
|
.lct-stack-empty { color: var(--lct-muted); font-size: var(--fs-sm); margin: 0; }
|
|
619
|
+
.lct-bar-track,
|
|
620
|
+
.lct-budget-track { background: var(--lct-surface-2); border-radius: 999px; overflow: hidden; }
|
|
621
|
+
.lct-bar-track { height: 8px; min-width: 96px; }
|
|
622
|
+
.lct-budget-track { height: 10px; position: relative; }
|
|
623
|
+
.lct-bar-fill,
|
|
624
|
+
.lct-budget-fill { height: 100%; display: block; background: var(--lct-accent); }
|
|
625
|
+
.lct-budget-fill--warn { background: var(--lct-warning); }
|
|
626
|
+
.lct-budget-fill--over { background: var(--lct-danger); }
|
|
627
|
+
.lct-budget-marker { position: absolute; top: 0; bottom: 0; border-left: 2px dashed var(--lct-chart-marker); }
|
|
628
|
+
.lct-budget-projection { display: flex; flex-wrap: wrap; gap: 8px 12px; align-items: baseline; margin: 10px 0 0; color: var(--lct-muted); font-size: var(--fs-sm); }
|
|
629
|
+
.lct-budget-projection strong { color: var(--lct-text); }
|
|
630
|
+
.lct-budget-projection-status { font-weight: 600; }
|
|
631
|
+
.lct-budget-projection-status--over { color: var(--lct-warning-copy); }
|
|
616
632
|
|
|
617
633
|
.lct-legend { display: inline-flex; align-items: center; gap: 6px; font-size: var(--fs-xs); color: var(--lct-muted); }
|
|
618
634
|
.lct-legend-dot { width: 9px; height: 9px; border-radius: 2px; }
|
|
@@ -16,11 +16,9 @@ module LlmCostTracker
|
|
|
16
16
|
}.freeze
|
|
17
17
|
|
|
18
18
|
def index
|
|
19
|
-
@sort = params[:sort].to_s
|
|
20
|
-
@dir = params[:dir].to_s
|
|
21
19
|
scope = Dashboard::Filter.call(params: params)
|
|
22
20
|
scope = scope.unknown_pricing if params[:cost_status].to_s == "incomplete"
|
|
23
|
-
ordered_scope = scope.order(*calls_order(
|
|
21
|
+
ordered_scope = scope.order(*calls_order(params[:sort].to_s, params[:dir].to_s))
|
|
24
22
|
|
|
25
23
|
respond_to do |format|
|
|
26
24
|
format.html do
|
|
@@ -5,13 +5,11 @@ module LlmCostTracker
|
|
|
5
5
|
MAX_ROWS = 200
|
|
6
6
|
|
|
7
7
|
def index
|
|
8
|
-
@sort = params[:sort].to_s
|
|
9
|
-
@dir = params[:dir].to_s
|
|
10
8
|
@rows = Dashboard::TopModels.call(
|
|
11
9
|
scope: Dashboard::Filter.call(params: params),
|
|
12
10
|
limit: MAX_ROWS,
|
|
13
|
-
sort:
|
|
14
|
-
direction:
|
|
11
|
+
sort: params[:sort].to_s,
|
|
12
|
+
direction: params[:dir].to_s
|
|
15
13
|
)
|
|
16
14
|
end
|
|
17
15
|
end
|
|
@@ -10,10 +10,11 @@ module LlmCostTracker
|
|
|
10
10
|
@value = Dashboard::Params.scalar(params[:tag_value], :tag_value)
|
|
11
11
|
|
|
12
12
|
if @value.empty?
|
|
13
|
-
@sort = params[:sort].to_s
|
|
14
|
-
@dir = params[:dir].to_s
|
|
15
13
|
@breakdown = Dashboard::TagBreakdown.call(
|
|
16
|
-
scope: Dashboard::Filter.call(params: params),
|
|
14
|
+
scope: Dashboard::Filter.call(params: params),
|
|
15
|
+
key: params[:key],
|
|
16
|
+
sort: params[:sort].to_s,
|
|
17
|
+
direction: params[:dir].to_s
|
|
17
18
|
)
|
|
18
19
|
else
|
|
19
20
|
@key = LlmCostTracker::Tags::Key.validate!(
|
|
@@ -23,7 +24,7 @@ module LlmCostTracker
|
|
|
23
24
|
value_scope = Dashboard::Filter.call(params: params, tags: { @key => @value })
|
|
24
25
|
@value_total_cost = value_scope.sum(:total_cost).to_f
|
|
25
26
|
@value_calls = value_scope.count
|
|
26
|
-
@value_points = Dashboard::TimeSeries.call(scope: value_scope)
|
|
27
|
+
@value_points = Dashboard::TimeSeries.call(scope: value_scope, from: @from_date, to: @to_date)
|
|
27
28
|
end
|
|
28
29
|
end
|
|
29
30
|
end
|
|
@@ -13,6 +13,7 @@ module LlmCostTracker
|
|
|
13
13
|
include PaginationHelper
|
|
14
14
|
include TokenUsageHelper
|
|
15
15
|
include InlineStyleHelper
|
|
16
|
+
include SortableTableHelper
|
|
16
17
|
|
|
17
18
|
def dashboard_section
|
|
18
19
|
path = request.path.to_s
|
|
@@ -120,13 +121,10 @@ module LlmCostTracker
|
|
|
120
121
|
end
|
|
121
122
|
|
|
122
123
|
def tag_chip_entries(tags, limit: 3)
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
visible = normalized.first(limit).map do |key, value|
|
|
127
|
-
{ key: key.to_s, value: tag_value_summary(value) }
|
|
124
|
+
visible = tags.first(limit).map do |key, value|
|
|
125
|
+
{ key: key.to_s, value: truncate_text(value.to_s, TAG_VALUE_SUMMARY_BYTES) }
|
|
128
126
|
end
|
|
129
|
-
visible << { more:
|
|
127
|
+
visible << { more: tags.size - limit } if tags.size > limit
|
|
130
128
|
visible
|
|
131
129
|
end
|
|
132
130
|
|
|
@@ -144,25 +142,6 @@ module LlmCostTracker
|
|
|
144
142
|
|
|
145
143
|
private
|
|
146
144
|
|
|
147
|
-
def normalized_tags(tags)
|
|
148
|
-
return tags.transform_keys(&:to_s) if tags.is_a?(Hash)
|
|
149
|
-
|
|
150
|
-
JSON.parse(tags || "{}")
|
|
151
|
-
rescue JSON::ParserError, TypeError
|
|
152
|
-
{}
|
|
153
|
-
end
|
|
154
|
-
|
|
155
|
-
def tag_value_summary(value)
|
|
156
|
-
string = case value
|
|
157
|
-
when Hash, Array
|
|
158
|
-
JSON.generate(value)
|
|
159
|
-
else
|
|
160
|
-
value.to_s
|
|
161
|
-
end
|
|
162
|
-
|
|
163
|
-
truncate_text(string, TAG_VALUE_SUMMARY_BYTES)
|
|
164
|
-
end
|
|
165
|
-
|
|
166
145
|
def truncate_text(string, limit)
|
|
167
146
|
return string if string.bytesize <= limit
|
|
168
147
|
|
|
@@ -7,7 +7,7 @@ module LlmCostTracker
|
|
|
7
7
|
|
|
8
8
|
cfg = chart_config(points, comparison_points, height, y_ticks)
|
|
9
9
|
parts = [chart_svg_open(cfg), "<title>Daily spend trend</title>", chart_area_gradient_def]
|
|
10
|
-
parts.concat(
|
|
10
|
+
parts.concat((0..cfg[:y_ticks]).map { |i| chart_tick_line(cfg, i) })
|
|
11
11
|
parts << chart_paths(cfg)
|
|
12
12
|
parts.concat(chart_dots(cfg))
|
|
13
13
|
parts.concat(chart_x_labels(cfg))
|
|
@@ -34,7 +34,7 @@ module LlmCostTracker
|
|
|
34
34
|
peak_index = points.each_with_index.max_by { |point, _| point[:cost].to_f }&.last
|
|
35
35
|
{ width: width, height: height, pad: pad, plot_w: plot_w, plot_h: plot_h,
|
|
36
36
|
max_cost: max_cost, n: points.size, y_ticks: y_ticks, points: points, coords: coords,
|
|
37
|
-
|
|
37
|
+
comparison_coords: comparison_coords,
|
|
38
38
|
peak_index: peak_index }
|
|
39
39
|
end
|
|
40
40
|
|
|
@@ -60,21 +60,17 @@ module LlmCostTracker
|
|
|
60
60
|
"<svg #{attrs}>"
|
|
61
61
|
end
|
|
62
62
|
|
|
63
|
-
def chart_grid_and_axis(cfg)
|
|
64
|
-
(0..cfg[:y_ticks]).map { |i| chart_tick_line(cfg, i) }
|
|
65
|
-
end
|
|
66
|
-
|
|
67
63
|
def chart_tick_line(cfg, idx)
|
|
68
64
|
pad = cfg[:pad]
|
|
69
65
|
right_x = chart_fmt(pad[:left] + cfg[:plot_w])
|
|
70
66
|
left_x = chart_fmt(pad[:left])
|
|
71
67
|
text_x = chart_fmt(pad[:left] - 8)
|
|
72
68
|
value = cfg[:max_cost] * (cfg[:y_ticks] - idx).to_f / cfg[:y_ticks]
|
|
73
|
-
|
|
74
|
-
|
|
69
|
+
tick_y = pad[:top] + (cfg[:plot_h] * idx.to_f / cfg[:y_ticks])
|
|
70
|
+
y = chart_fmt(tick_y)
|
|
71
|
+
label_y = chart_fmt(tick_y + 3)
|
|
75
72
|
grid = %(<line class="lct-chart-grid" x1="#{left_x}" x2="#{right_x}" y1="#{y}" y2="#{y}"/>)
|
|
76
|
-
|
|
77
|
-
text = %(<text class="lct-chart-axis" x="#{text_x}" y="#{label_y}" text-anchor="end">$#{label}</text>)
|
|
73
|
+
text = %(<text class="lct-chart-axis" x="#{text_x}" y="#{label_y}" text-anchor="end">$#{chart_fmt(value)}</text>)
|
|
78
74
|
"#{grid}#{text}"
|
|
79
75
|
end
|
|
80
76
|
|
|
@@ -33,41 +33,21 @@ module LlmCostTracker
|
|
|
33
33
|
}.freeze
|
|
34
34
|
|
|
35
35
|
def token_usage_stack_components
|
|
36
|
-
token_usage_display_components(labels: COMPONENT_LABELS).select do |component|
|
|
37
|
-
component.fetch(:cost_key)
|
|
38
|
-
end
|
|
39
|
-
end
|
|
40
|
-
|
|
41
|
-
def call_line_item_costs_by_component(call)
|
|
42
|
-
call.line_items.each_with_object({}) do |line_item, accumulator|
|
|
43
|
-
component = LlmCostTracker::Usage::Catalog.token_priced_for(
|
|
44
|
-
kind: line_item.kind, direction: line_item.direction, cache_state: line_item.cache_state
|
|
45
|
-
)
|
|
46
|
-
accumulator[component.key] = line_item.cost if component && line_item.cost
|
|
47
|
-
end
|
|
48
|
-
end
|
|
49
|
-
|
|
50
|
-
private
|
|
51
|
-
|
|
52
|
-
def token_usage_display_components(labels:)
|
|
53
36
|
LlmCostTracker::Usage::Catalog.token_priced.map do |component|
|
|
54
37
|
token_key = component.token_key
|
|
55
38
|
{
|
|
56
39
|
token_key: token_key,
|
|
57
|
-
cost_key: component.cost_key,
|
|
58
40
|
price_key: component.key,
|
|
59
|
-
label:
|
|
41
|
+
label: COMPONENT_LABELS.fetch(token_key),
|
|
60
42
|
css_class: STACK_CLASSES[token_key]
|
|
61
43
|
}
|
|
62
|
-
end
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
}
|
|
70
|
-
]
|
|
44
|
+
end
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
def call_line_item_costs_by_component(call)
|
|
48
|
+
LlmCostTracker::Usage::Catalog.costs_by_component(
|
|
49
|
+
call.line_items.map { |item| [item.kind, item.direction, item.cache_state, item.cost] }
|
|
50
|
+
)
|
|
71
51
|
end
|
|
72
52
|
end
|
|
73
53
|
end
|
|
@@ -45,7 +45,10 @@ module LlmCostTracker
|
|
|
45
45
|
def already_recorded?(provider:, provider_response_id:)
|
|
46
46
|
return false if provider_response_id.to_s.empty?
|
|
47
47
|
|
|
48
|
-
Ledger::Isolation.guard(self)
|
|
48
|
+
Ledger::Isolation.guard(self) do
|
|
49
|
+
where(provider: provider, provider_response_id: provider_response_id)
|
|
50
|
+
.where.not(usage_source: Usage::Source::UNKNOWN).exists?
|
|
51
|
+
end
|
|
49
52
|
end
|
|
50
53
|
|
|
51
54
|
def by_tag(key, value) = by_tags(key => value)
|
|
@@ -86,8 +89,12 @@ module LlmCostTracker
|
|
|
86
89
|
|
|
87
90
|
def latency_by_provider = group(:provider).average(:latency_ms).transform_values(&:to_f)
|
|
88
91
|
|
|
89
|
-
def group_by_period(period, column: :tracked_at)
|
|
90
|
-
|
|
92
|
+
def group_by_period(period, column: :tracked_at, time_zone: nil)
|
|
93
|
+
column = column.to_s
|
|
94
|
+
raise ArgumentError, "invalid period column: #{column.inspect}" unless column_names.include?(column)
|
|
95
|
+
|
|
96
|
+
bucket = Ledger::Schema::Adapter.period_bucket_sql(connection, period, qualified(column), time_zone: time_zone)
|
|
97
|
+
group(Arel.sql(bucket))
|
|
91
98
|
end
|
|
92
99
|
|
|
93
100
|
def daily_costs(days: 30)
|
|
@@ -109,17 +116,6 @@ module LlmCostTracker
|
|
|
109
116
|
relation = relation.limit(limit) if limit
|
|
110
117
|
relation
|
|
111
118
|
end
|
|
112
|
-
|
|
113
|
-
def period_group_expression(period, column:)
|
|
114
|
-
Ledger::Schema::Adapter.period_bucket_sql(connection, period, period_column_expression(column))
|
|
115
|
-
end
|
|
116
|
-
|
|
117
|
-
def period_column_expression(column)
|
|
118
|
-
column = column.to_s
|
|
119
|
-
return "#{quoted_table_name}.#{connection.quote_column_name(column)}" if column_names.include?(column)
|
|
120
|
-
|
|
121
|
-
raise ArgumentError, "invalid period column: #{column.inspect}"
|
|
122
|
-
end
|
|
123
119
|
end
|
|
124
120
|
|
|
125
121
|
def tag_pairs
|
|
@@ -13,7 +13,6 @@ module LlmCostTracker
|
|
|
13
13
|
:missing_latency_count,
|
|
14
14
|
:streaming_count,
|
|
15
15
|
:streaming_missing_usage,
|
|
16
|
-
:missing_provider_response_id_count,
|
|
17
16
|
:calls_with_pricing,
|
|
18
17
|
:tagged_calls,
|
|
19
18
|
:calls_with_latency,
|
|
@@ -52,7 +51,7 @@ module LlmCostTracker
|
|
|
52
51
|
return nil unless Budget::PerTag.columns?
|
|
53
52
|
return nil unless LlmCostTracker::CallTag.exists?
|
|
54
53
|
|
|
55
|
-
unseen = budgeted.keys.reject { |key| LlmCostTracker::CallTag.
|
|
54
|
+
unseen = budgeted.keys.reject { |key| LlmCostTracker::CallTag.where(key: key).where.not(value: "").exists? }
|
|
56
55
|
return nil if unseen.empty?
|
|
57
56
|
|
|
58
57
|
UnseenBudgetTags.new(keys: unseen)
|
|
@@ -79,7 +78,6 @@ module LlmCostTracker
|
|
|
79
78
|
missing_latency_count,
|
|
80
79
|
streaming_count,
|
|
81
80
|
streaming_missing_usage,
|
|
82
|
-
missing_provider_response_id_count,
|
|
83
81
|
calls_with_pricing,
|
|
84
82
|
tagged_calls,
|
|
85
83
|
calls_with_latency,
|
|
@@ -142,7 +140,6 @@ module LlmCostTracker
|
|
|
142
140
|
token_value = stats[component.token_key].to_i
|
|
143
141
|
|
|
144
142
|
{
|
|
145
|
-
price_key: component.key,
|
|
146
143
|
token_key: component.token_key,
|
|
147
144
|
cost_key: component.cost_key,
|
|
148
145
|
token_value: token_value,
|
|
@@ -154,7 +151,6 @@ module LlmCostTracker
|
|
|
154
151
|
|
|
155
152
|
rows + [
|
|
156
153
|
{
|
|
157
|
-
price_key: nil,
|
|
158
154
|
token_key: :hidden_output_tokens,
|
|
159
155
|
cost_key: nil,
|
|
160
156
|
token_value: stats.hidden_output_tokens.to_i,
|
|
@@ -178,7 +174,7 @@ module LlmCostTracker
|
|
|
178
174
|
Arel.sql("#{line_item_table}.direction"),
|
|
179
175
|
Arel.sql("#{line_item_table}.cache_state"),
|
|
180
176
|
Arel.sql("COALESCE(SUM(#{line_item_table}.cost), 0)"))
|
|
181
|
-
|
|
177
|
+
Usage::Catalog.costs_by_component(rows)
|
|
182
178
|
end
|
|
183
179
|
|
|
184
180
|
def streaming_health_rows(scope, total_streaming:)
|
|
@@ -221,13 +217,6 @@ module LlmCostTracker
|
|
|
221
217
|
|
|
222
218
|
private
|
|
223
219
|
|
|
224
|
-
def index_costs_by_component(rows)
|
|
225
|
-
rows.each_with_object({}) do |(kind, direction, cache_state, cost), accumulator|
|
|
226
|
-
component = Usage::Catalog.token_priced_for(kind: kind, direction: direction, cache_state: cache_state)
|
|
227
|
-
accumulator[component.key] = cost if component
|
|
228
|
-
end
|
|
229
|
-
end
|
|
230
|
-
|
|
231
220
|
def percentage(numerator, denominator)
|
|
232
221
|
return 0.0 unless denominator.positive?
|
|
233
222
|
|
|
@@ -239,7 +228,6 @@ module LlmCostTracker
|
|
|
239
228
|
selects = [
|
|
240
229
|
"COUNT(*) AS total_calls",
|
|
241
230
|
"#{conditional_count_sql(unknown_pricing)} AS unknown_pricing_count",
|
|
242
|
-
"#{tagged_calls_sql(scope)} AS tagged_calls_count",
|
|
243
231
|
"COUNT(*) - #{tagged_calls_sql(scope)} AS untagged_calls_count",
|
|
244
232
|
"#{conditional_count_sql('latency_ms IS NULL')} AS missing_latency_count",
|
|
245
233
|
"#{conditional_count_sql('stream')} AS streaming_count",
|
|
@@ -301,7 +289,8 @@ module LlmCostTracker
|
|
|
301
289
|
tags_table = LlmCostTracker::CallTag.quoted_table_name
|
|
302
290
|
|
|
303
291
|
"COALESCE(SUM(CASE WHEN EXISTS (SELECT 1 FROM #{tags_table} " \
|
|
304
|
-
"WHERE #{tags_table}.llm_cost_tracker_call_id = #{calls_table}.id
|
|
292
|
+
"WHERE #{tags_table}.llm_cost_tracker_call_id = #{calls_table}.id " \
|
|
293
|
+
"AND #{tags_table}.#{scope.connection.quote_column_name('value')} != '') THEN 1 ELSE 0 END), 0)"
|
|
305
294
|
end
|
|
306
295
|
end
|
|
307
296
|
end
|
|
@@ -39,17 +39,13 @@ module LlmCostTracker
|
|
|
39
39
|
attr_reader :scope, :params, :extra_tags
|
|
40
40
|
|
|
41
41
|
def apply_date_filters(relation)
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
Dashboard::DateRange.
|
|
45
|
-
|
|
46
|
-
default_range = Dashboard::DateRange.call(params: params)
|
|
47
|
-
from_date ||= default_range.from
|
|
48
|
-
to_date ||= default_range.to
|
|
42
|
+
Dashboard::DateRange.validate!(from: Dashboard::DateRange.parse(params, :from),
|
|
43
|
+
to: Dashboard::DateRange.parse(params, :to))
|
|
44
|
+
range = Dashboard::DateRange.call(params: params)
|
|
49
45
|
|
|
50
46
|
relation
|
|
51
|
-
.where(tracked_at:
|
|
52
|
-
.where(tracked_at: ..
|
|
47
|
+
.where(tracked_at: range.from.beginning_of_day..)
|
|
48
|
+
.where(tracked_at: ..range.to.end_of_day)
|
|
53
49
|
end
|
|
54
50
|
|
|
55
51
|
def apply_exact_filter(relation, key)
|
|
@@ -19,11 +19,8 @@ module LlmCostTracker
|
|
|
19
19
|
)
|
|
20
20
|
end
|
|
21
21
|
|
|
22
|
-
def self.integer_param(params, key, default:, min:, max:
|
|
23
|
-
|
|
24
|
-
value = [value, min].max
|
|
25
|
-
value = [value, max].min if max
|
|
26
|
-
value
|
|
22
|
+
def self.integer_param(params, key, default:, min:, max:)
|
|
23
|
+
Integer(params[key], 10).clamp(min, max)
|
|
27
24
|
rescue ArgumentError, TypeError
|
|
28
25
|
default
|
|
29
26
|
end
|
|
@@ -63,9 +63,10 @@ module LlmCostTracker
|
|
|
63
63
|
end
|
|
64
64
|
|
|
65
65
|
def build_rows(prices)
|
|
66
|
+
today = Time.now.utc.to_date.iso8601
|
|
66
67
|
rows = prices.map do |key, rates|
|
|
67
68
|
provider, model = split_key(key.to_s)
|
|
68
|
-
Row.new(provider: provider, model: model, rates: rates)
|
|
69
|
+
Row.new(provider: provider, model: model, rates: Pricing::Matcher.prices_on(rates, today))
|
|
69
70
|
end
|
|
70
71
|
rows.sort_by { |row| [row.provider || "~", row.model] }
|
|
71
72
|
end
|
|
@@ -59,7 +59,7 @@ module LlmCostTracker
|
|
|
59
59
|
.where(tracked_at: window)
|
|
60
60
|
.where.not(total_cost: nil)
|
|
61
61
|
.group(:provider, :model)
|
|
62
|
-
.group_by_period(:day)
|
|
62
|
+
.group_by_period(:day, time_zone: Time.zone)
|
|
63
63
|
.sum(:total_cost)
|
|
64
64
|
.each do |(provider, model, day), total_cost|
|
|
65
65
|
grouped[[provider, model]][Date.iso8601(day.to_s)] += total_cost.to_f
|
|
@@ -4,7 +4,6 @@ module LlmCostTracker
|
|
|
4
4
|
module Dashboard
|
|
5
5
|
class TagBreakdown
|
|
6
6
|
DEFAULT_LIMIT = 100
|
|
7
|
-
SORT_OPTIONS = %w[value calls cost avg_cost].freeze
|
|
8
7
|
DEFAULT_DIRECTIONS = { "value" => "asc", "calls" => "desc", "cost" => "desc", "avg_cost" => "desc" }.freeze
|
|
9
8
|
Row = Data.define(:value, :calls, :total_cost, :average_cost_per_call, :share_percent)
|
|
10
9
|
|
|
@@ -21,7 +20,7 @@ module LlmCostTracker
|
|
|
21
20
|
@key = LlmCostTracker::Tags::Key.validate!(key, error_class: LlmCostTracker::InvalidFilterError)
|
|
22
21
|
limit = limit.to_i
|
|
23
22
|
@limit = limit.positive? ? [limit, DEFAULT_LIMIT].min : DEFAULT_LIMIT
|
|
24
|
-
@sort =
|
|
23
|
+
@sort = DEFAULT_DIRECTIONS.key?(sort.to_s) ? sort.to_s : "cost"
|
|
25
24
|
@direction = Sort::DIRECTIONS.include?(direction.to_s) ? direction.to_s : DEFAULT_DIRECTIONS[@sort]
|
|
26
25
|
end
|
|
27
26
|
|
|
@@ -89,7 +88,7 @@ module LlmCostTracker
|
|
|
89
88
|
def summary_sql
|
|
90
89
|
<<~SQL.squish
|
|
91
90
|
SELECT COUNT(*) AS total_calls,
|
|
92
|
-
COUNT(
|
|
91
|
+
COUNT(CASE WHEN #{tag_present_predicate} THEN 1 END) AS tagged_calls,
|
|
93
92
|
COUNT(DISTINCT CASE WHEN #{tag_present_predicate} THEN #{tag_value_column} END) AS distinct_values
|
|
94
93
|
FROM (#{scope.to_sql}) AS sub
|
|
95
94
|
LEFT OUTER JOIN #{call_tag_table} t ON t.llm_cost_tracker_call_id = sub.id AND t.#{quote_column('key')} = #{quoted_key}
|
|
@@ -38,6 +38,7 @@ module LlmCostTracker
|
|
|
38
38
|
COUNT(DISTINCT t.#{value_column}) AS distinct_values
|
|
39
39
|
FROM (#{scope.to_sql}) AS sub
|
|
40
40
|
INNER JOIN #{tags_table} t ON t.llm_cost_tracker_call_id = sub.id
|
|
41
|
+
WHERE t.#{value_column} != ''
|
|
41
42
|
GROUP BY t.#{key_column}
|
|
42
43
|
ORDER BY calls_count DESC
|
|
43
44
|
LIMIT #{limit}
|
|
@@ -5,18 +5,16 @@ require "date"
|
|
|
5
5
|
module LlmCostTracker
|
|
6
6
|
module Dashboard
|
|
7
7
|
class TimeSeries
|
|
8
|
-
DEFAULT_DAYS = 30
|
|
9
|
-
|
|
10
8
|
class << self
|
|
11
|
-
def call(scope: LlmCostTracker::Call.all
|
|
9
|
+
def call(from:, to:, scope: LlmCostTracker::Call.all)
|
|
12
10
|
new(scope: scope, from: from, to: to).points
|
|
13
11
|
end
|
|
14
12
|
end
|
|
15
13
|
|
|
16
14
|
def initialize(scope:, from:, to:)
|
|
17
15
|
@scope = scope
|
|
16
|
+
@from = from.to_date
|
|
18
17
|
@to = to.to_date
|
|
19
|
-
@from = from ? from.to_date : (@to - (DEFAULT_DAYS - 1))
|
|
20
18
|
end
|
|
21
19
|
|
|
22
20
|
def points
|
|
@@ -35,7 +33,7 @@ module LlmCostTracker
|
|
|
35
33
|
def scoped_costs
|
|
36
34
|
scope
|
|
37
35
|
.where(tracked_at: from.beginning_of_day..to.end_of_day)
|
|
38
|
-
.group_by_period(:day)
|
|
36
|
+
.group_by_period(:day, time_zone: Time.zone)
|
|
39
37
|
.sum(:total_cost)
|
|
40
38
|
.transform_values(&:to_f)
|
|
41
39
|
end
|