llm_cost_tracker 0.14.0 → 0.14.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (133) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +128 -0
  3. data/README.md +16 -5
  4. data/app/assets/llm_cost_tracker/application.css +16 -0
  5. data/app/controllers/llm_cost_tracker/application_controller.rb +16 -3
  6. data/app/controllers/llm_cost_tracker/calls_controller.rb +1 -3
  7. data/app/controllers/llm_cost_tracker/models_controller.rb +2 -4
  8. data/app/controllers/llm_cost_tracker/pricing_controller.rb +2 -2
  9. data/app/controllers/llm_cost_tracker/tags_controller.rb +9 -7
  10. data/app/helpers/llm_cost_tracker/application_helper.rb +5 -26
  11. data/app/helpers/llm_cost_tracker/chart_helper.rb +6 -10
  12. data/app/helpers/llm_cost_tracker/dashboard_query_helper.rb +5 -0
  13. data/app/helpers/llm_cost_tracker/token_usage_helper.rb +8 -28
  14. data/app/models/llm_cost_tracker/call.rb +10 -14
  15. data/app/services/llm_cost_tracker/dashboard/data_quality.rb +4 -15
  16. data/app/services/llm_cost_tracker/dashboard/filter.rb +29 -30
  17. data/app/services/llm_cost_tracker/dashboard/pagination.rb +2 -5
  18. data/app/services/llm_cost_tracker/dashboard/params.rb +10 -0
  19. data/app/services/llm_cost_tracker/dashboard/pricing_overview.rb +2 -1
  20. data/app/services/llm_cost_tracker/dashboard/spend_anomaly.rb +1 -1
  21. data/app/services/llm_cost_tracker/dashboard/tag_breakdown.rb +2 -3
  22. data/app/services/llm_cost_tracker/dashboard/tag_key_explorer.rb +1 -0
  23. data/app/services/llm_cost_tracker/dashboard/time_series.rb +3 -5
  24. data/app/services/llm_cost_tracker/dashboard/top_models.rb +1 -2
  25. data/app/views/llm_cost_tracker/calls/show.html.erb +8 -22
  26. data/app/views/llm_cost_tracker/data_quality/index.html.erb +1 -1
  27. data/app/views/llm_cost_tracker/shared/_bar.html.erb +1 -3
  28. data/app/views/llm_cost_tracker/shared/_filter_pill_date.html.erb +1 -4
  29. data/app/views/llm_cost_tracker/shared/_filter_pill_model.html.erb +1 -4
  30. data/app/views/llm_cost_tracker/shared/_filter_pill_provider.html.erb +1 -4
  31. data/app/views/llm_cost_tracker/shared/_filter_pill_stream.html.erb +1 -4
  32. data/app/views/llm_cost_tracker/tags/show.html.erb +8 -5
  33. data/config/routes.rb +6 -1
  34. data/lib/llm_cost_tracker/budget/per_tag.rb +21 -11
  35. data/lib/llm_cost_tracker/budget.rb +44 -55
  36. data/lib/llm_cost_tracker/capture/event_window.rb +100 -0
  37. data/lib/llm_cost_tracker/capture/sdk_payload.rb +5 -1
  38. data/lib/llm_cost_tracker/capture/sse.rb +108 -32
  39. data/lib/llm_cost_tracker/capture/stream_collector.rb +41 -81
  40. data/lib/llm_cost_tracker/capture/stream_tap.rb +57 -0
  41. data/lib/llm_cost_tracker/capture/stream_tracker.rb +14 -58
  42. data/lib/llm_cost_tracker/capture_verifier.rb +1 -7
  43. data/lib/llm_cost_tracker/charges/line_item.rb +5 -1
  44. data/lib/llm_cost_tracker/configuration/budgets.rb +1 -1
  45. data/lib/llm_cost_tracker/configuration/pricing.rb +1 -1
  46. data/lib/llm_cost_tracker/configuration.rb +5 -9
  47. data/lib/llm_cost_tracker/doctor/price_check.rb +14 -3
  48. data/lib/llm_cost_tracker/doctor/schema_check.rb +1 -2
  49. data/lib/llm_cost_tracker/doctor.rb +19 -2
  50. data/lib/llm_cost_tracker/engine.rb +0 -1
  51. data/lib/llm_cost_tracker/errors.rb +13 -0
  52. data/lib/llm_cost_tracker/event.rb +8 -0
  53. data/lib/llm_cost_tracker/generators/llm_cost_tracker/async_ingestion_generator.rb +0 -6
  54. data/lib/llm_cost_tracker/generators/llm_cost_tracker/call_rollups_generator.rb +5 -8
  55. data/lib/llm_cost_tracker/generators/llm_cost_tracker/install_generator.rb +0 -6
  56. data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/create_llm_cost_tracker_async_ingestion.rb.erb +1 -1
  57. data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/create_llm_cost_tracker_call_rollups.rb.erb +1 -1
  58. data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/create_llm_cost_tracker_calls.rb.erb +1 -1
  59. data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/initializer.rb.erb +14 -12
  60. data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/upgrade_call_rollups_provider.rb.erb +1 -1
  61. data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/upgrade_call_tags_key_value_index.rb.erb +1 -1
  62. data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/upgrade_image_tokens.rb.erb +1 -1
  63. data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/upgrade_indexes.rb.erb +1 -1
  64. data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/upgrade_per_tag_budgets.rb.erb +1 -1
  65. data/lib/llm_cost_tracker/generators/llm_cost_tracker/upgrade_call_rollups_provider_generator.rb +0 -6
  66. data/lib/llm_cost_tracker/generators/llm_cost_tracker/upgrade_call_tags_key_value_index_generator.rb +0 -6
  67. data/lib/llm_cost_tracker/generators/llm_cost_tracker/upgrade_image_tokens_generator.rb +0 -6
  68. data/lib/llm_cost_tracker/generators/llm_cost_tracker/upgrade_indexes_generator.rb +3 -8
  69. data/lib/llm_cost_tracker/generators/llm_cost_tracker/upgrade_per_tag_budgets_generator.rb +3 -8
  70. data/lib/llm_cost_tracker/ingestion/batch.rb +37 -8
  71. data/lib/llm_cost_tracker/ingestion/inbox.rb +2 -8
  72. data/lib/llm_cost_tracker/ingestion.rb +3 -1
  73. data/lib/llm_cost_tracker/integrations/anthropic.rb +70 -11
  74. data/lib/llm_cost_tracker/integrations/base.rb +56 -19
  75. data/lib/llm_cost_tracker/integrations/openai/batch_capture.rb +21 -13
  76. data/lib/llm_cost_tracker/integrations/openai/patches.rb +8 -0
  77. data/lib/llm_cost_tracker/integrations/openai.rb +42 -59
  78. data/lib/llm_cost_tracker/integrations/ruby_llm.rb +275 -75
  79. data/lib/llm_cost_tracker/integrations.rb +1 -20
  80. data/lib/llm_cost_tracker/ledger/isolation.rb +23 -0
  81. data/lib/llm_cost_tracker/ledger/period/totals.rb +16 -11
  82. data/lib/llm_cost_tracker/ledger/rollups.rb +25 -17
  83. data/lib/llm_cost_tracker/ledger/schema/adapter.rb +14 -3
  84. data/lib/llm_cost_tracker/ledger/schema/base.rb +5 -4
  85. data/lib/llm_cost_tracker/ledger/storable.rb +16 -0
  86. data/lib/llm_cost_tracker/ledger/store.rb +14 -13
  87. data/lib/llm_cost_tracker/ledger/tags/breakdown.rb +1 -5
  88. data/lib/llm_cost_tracker/ledger/tags/encoding.rb +3 -12
  89. data/lib/llm_cost_tracker/ledger/tags/query.rb +0 -2
  90. data/lib/llm_cost_tracker/ledger.rb +1 -0
  91. data/lib/llm_cost_tracker/logging.rb +3 -1
  92. data/lib/llm_cost_tracker/middleware/faraday.rb +67 -61
  93. data/lib/llm_cost_tracker/parsers.rb +12 -41
  94. data/lib/llm_cost_tracker/prices.json +3071 -249
  95. data/lib/llm_cost_tracker/pricing/backfill.rb +46 -13
  96. data/lib/llm_cost_tracker/pricing/calculation.rb +118 -75
  97. data/lib/llm_cost_tracker/pricing/effective_prices.rb +11 -10
  98. data/lib/llm_cost_tracker/pricing/estimator.rb +5 -2
  99. data/lib/llm_cost_tracker/pricing/matcher.rb +36 -22
  100. data/lib/llm_cost_tracker/pricing/mode.rb +2 -13
  101. data/lib/llm_cost_tracker/pricing/price_key.rb +5 -3
  102. data/lib/llm_cost_tracker/pricing/rate.rb +1 -0
  103. data/lib/llm_cost_tracker/pricing/registry.rb +17 -17
  104. data/lib/llm_cost_tracker/pricing/service_rates.rb +1 -8
  105. data/lib/llm_cost_tracker/pricing/sync/change_printer.rb +6 -1
  106. data/lib/llm_cost_tracker/pricing/sync/registry_diff.rb +13 -9
  107. data/lib/llm_cost_tracker/pricing/sync/snapshot_guard.rb +47 -0
  108. data/lib/llm_cost_tracker/pricing/sync.rb +38 -70
  109. data/lib/llm_cost_tracker/providers/anthropic/parser.rb +33 -3
  110. data/lib/llm_cost_tracker/providers/anthropic/response_parser.rb +11 -5
  111. data/lib/llm_cost_tracker/providers/anthropic/usage_extractor.rb +81 -12
  112. data/lib/llm_cost_tracker/providers/azure/parser.rb +22 -7
  113. data/lib/llm_cost_tracker/providers/gemini/parser.rb +175 -47
  114. data/lib/llm_cost_tracker/providers/gemini/usage_extractor.rb +13 -22
  115. data/lib/llm_cost_tracker/providers/openai/hosts.rb +1 -1
  116. data/lib/llm_cost_tracker/providers/openai/parser.rb +11 -13
  117. data/lib/llm_cost_tracker/providers/openai/response_parser.rb +101 -27
  118. data/lib/llm_cost_tracker/providers/openai/service_charges.rb +57 -40
  119. data/lib/llm_cost_tracker/providers/openai/usage_extractor.rb +24 -6
  120. data/lib/llm_cost_tracker/providers/openai_compatible/parser.rb +7 -1
  121. data/lib/llm_cost_tracker/redaction.rb +32 -0
  122. data/lib/llm_cost_tracker/report/data.rb +1 -1
  123. data/lib/llm_cost_tracker/retention.rb +1 -3
  124. data/lib/llm_cost_tracker/tags/sanitizer.rb +10 -36
  125. data/lib/llm_cost_tracker/tracker.rb +17 -10
  126. data/lib/llm_cost_tracker/usage/catalog.rb +13 -4
  127. data/lib/llm_cost_tracker/usage/dimension.rb +1 -1
  128. data/lib/llm_cost_tracker/usage/dimensions.yml +61 -0
  129. data/lib/llm_cost_tracker/version.rb +1 -1
  130. data/lib/llm_cost_tracker.rb +4 -2
  131. data/lib/tasks/llm_cost_tracker.rake +26 -21
  132. data/llm_cost_tracker.gemspec +63 -0
  133. metadata +23 -10
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: ffc5fef0c3c18c6926b6c414d56fb0b7c9c4afef6db315139f8bc4626864611a
4
- data.tar.gz: 3b012c1014c7246dd0eebb5413e13c02023cce056312e719a96328e86de749a8
3
+ metadata.gz: 2c1f13f9f919518298bf4aa5bb9bf33f6fb48851a963d132d48000c2c9ab2477
4
+ data.tar.gz: 1de730cdb5061df2588c74dd0c56ec29aa59ec0fe812188e2ae30df4e206c682
5
5
  SHA512:
6
- metadata.gz: b28d7d695a1d6c45fe6c9d77d4f5caebbbde84682b7a261e25123d280181d00361baae80e9ddd65def0b9602c385d48813467ae741a5cde4bc378e4652a45829
7
- data.tar.gz: 617cafb3be05250fd7c9c0b049a37474e1e0a2737d72cfd2744110cd531b68c362e7d15928f54d9a9c2d810da3f451516ff9ae747103bca2d618b77159359a13
6
+ metadata.gz: ee3ac1c8844c4bc478f85899d9e74734d67040d4b4f124c1751a970924a6d4ccf6fd2d680a2ed2a0d0e51ea79d7e985ad623ff49ed88148129df0276ba6c0925
7
+ data.tar.gz: 2aaa9cad0cd45a6e7df0920f054c258d12f03b2edc99c1c3bcf846d4ba33f4ce62972419cc92c5014711e417cb5f12eee63df2ac2f3c3e691acf47c8c620fdb8
data/CHANGELOG.md CHANGED
@@ -2,6 +2,134 @@
2
2
 
3
3
  Format: [Keep a Changelog](https://keepachangelog.com/en/1.1.0/). Versioning: [SemVer](https://semver.org/spec/v2.0.0.html).
4
4
 
5
+ ## [0.14.2] - 2026-09-28
6
+
7
+ ### Added
8
+
9
+ - `bin/rails llm_cost_tracker:reprice FROM=... [TO=...]` reprices recorded calls, their rollups and per-tag costs at current prices, except provider-billed costs and costs passed to `track`.
10
+ - Bundled xAI and Mistral prices, with batch, priority and regional rates; Faraday captures both once their hosts are in `capture.openai_compatible_providers`.
11
+ - `bin/rails llm_cost_tracker:doctor` warns when `pricing.file` is older than the bundled prices or this month's `:cache` rollups do not match the calls ledger.
12
+ - Gemini Interactions API calls through the Faraday middleware, streamed and `background: true` ones included, are recorded and priced with their grounding.
13
+ - Creating an explicit Gemini context cache through Faraday or `RubyLLM.cache` on RubyLLM 2.x records its estimated storage cost until the cache expires.
14
+ - Price fields with a `_from_YYYY-MM-DD` suffix apply to calls from that date, including in `backfill_unknown_pricing` and `reprice`; bundled Gemini prices use it for announced price changes.
15
+
16
+ ### Changed
17
+
18
+ - BREAKING: Rails 8.0+ required; Rails 7.1 and 7.2 no longer receive security fixes upstream.
19
+ - Calls returning a billed `usage.cost`, such as OpenRouter's, are recorded at the billed amount instead of a list-price estimate or `pricing.overrides` rate.
20
+ - The official openai gem pointed at a host in `capture.openai_compatible_providers` records that provider instead of `openai`.
21
+ - A provider-scoped price (`<provider>/<model>`) prices another provider's calls only when that provider has no price for the model.
22
+ - With `budgets.totals_source = :cache`, monthly budgets read past days from the rollups: run `bin/rails llm_cost_tracker:rebuild_rollups` once after deploying and after switching to `:cache`, or they are under-counted.
23
+ - OpenAI Realtime cached audio and image and Gemini cached audio are priced at their own cached rates and stored as `audio_token` / `image_token` line items with `cache_state: read`.
24
+ - `bin/rails llm_cost_tracker:prices:refresh` no longer takes `PREVIEW=1` and stops when it is set; use `bin/rails llm_cost_tracker:prices:check` instead.
25
+
26
+ ### Removed
27
+
28
+ - The boot warning for `:ruby_llm` enabled together with `:openai` or `:anthropic`: RubyLLM doesn't call those SDKs, so nothing was recorded twice, and disabling one as it advised lost those calls.
29
+
30
+ ### Fixed
31
+
32
+ - Bundled prices cover more OpenAI and Gemini models that recorded unknown cost, and `omni-moderation` calls are `free` instead of `unknown`.
33
+ - `gpt-5.5-pro` prompts over 272K tokens use long-context rates (unknown on Batch and Flex), and `gpt-5.4` Batch and Flex long-context cached input is corrected.
34
+ - OpenAI cached input on Pro models and cache writes on models before GPT-5.6 are priced at the input rate; these calls were `partial`.
35
+ - OpenAI duration-billed transcriptions are priced per second instead of per started minute, and are no longer free in streams or missing through RubyLLM.
36
+ - Transcriptions returned without usage are recorded with unknown cost; they were recorded at $0 or not at all.
37
+ - OpenAI image, transcription and batch calls on regional hosts, and Anthropic batches with `inference_geo: us`, include the data-residency uplift for models with regional rates.
38
+ - The Faraday middleware reads the model from multipart uploads, which were recorded as `unknown`, and records `/v1/audio/speech` by input characters.
39
+ - OpenAI Hosted Shell calls in a hosted container are captured as `container_session` line items, like Code Interpreter.
40
+ - OpenAI Responses created with `background: true` are recorded once, when a poll through the OpenAI SDK or Faraday returns them finished.
41
+ - Responses calls using the `image_generation` tool, RubyLLM 2.x OpenAI chats included, are recorded `partial` instead of `complete` without the image charge.
42
+ - Chat Completions web search fees apply to every call to OpenAI's search models, streams included, and no longer to other models' `url_citation` annotations.
43
+ - A stream on a priced model that ends without usage and has no priced tool charges stores a `nil` total instead of `0.0`, so it shows as unpriced and is in `Call.without_cost`.
44
+ - OpenAI, Azure OpenAI and Anthropic SDK streams price the service tier and speed the provider served, not the requested `priority` or `fast`.
45
+ - OpenAI streams with logprobs record their usage instead of 0 tokens and $0, and an overflowing SDK or `track_stream` capture logs a warning.
46
+ - Groq streams whose usage arrives only in `x_groq.usage` record their tokens instead of 0 tokens and `unknown`.
47
+ - Anthropic streams price cache writes made after `message_start`.
48
+ - Anthropic compaction tokens, which the top-level usage leaves out, are counted; on-demand compactions were recorded as free (not yet through RubyLLM).
49
+ - Anthropic server-side fallback calls are recorded under the serving model, and billed fallback attempts and advisor iterations use their own model's rates; an unpriced one triggers `pricing.unknown_model_behavior` (not yet through RubyLLM).
50
+ - Refusals Anthropic does not bill are recorded at $0 (not yet through RubyLLM), and refusals `Anthropic::BetaRefusalFallbackMiddleware` retried are recorded instead of dropped.
51
+ - Gemini image and audio prompt tokens on single-rate models are priced at the input rate; they were unpriced.
52
+ - Gemini image models are priced per 1M image tokens and price text and thinking output at the text rate; `gemini-2.5-flash-image` images were priced about 770x too low and 3.x image models' images 10-20x too low. Calls already recorded keep their cost. With a local pricing file, run `bin/rails llm_cost_tracker:prices:check`, check that only Gemini image models are flagged, then run `bin/rails llm_cost_tracker:prices:refresh FORCE=1`.
53
+ - Gemini 3 grounding is priced per unique non-empty query, `track_stream` included; image search queries, Maps grounding and the 3.x image models' grounding are now priced.
54
+ - Gemini calls are recorded under the response's `modelVersion`, so `-latest` aliases are priced and `track_stream(provider: :gemini)` works without `model:`.
55
+ - `track_stream(provider: :gemini)` parses native Gemini chunks when `generativelanguage.googleapis.com` is also an OpenAI-compatible provider named `gemini`.
56
+ - Amazon Bedrock Claude ids and inference profiles price as the Anthropic model, with the regional-profile premium from Claude 4.5; GovCloud is not priced at its rate.
57
+ - Groq batch calls are priced at Groq's batch rate, cached tokens included.
58
+ - OpenAI batches that end `expired` or `cancelled` record their completed requests, and image batch results use the batch's model and batch image rates.
59
+ - OpenAI and Anthropic batch results are stored once however often or concurrently the batch is fetched; they could be recorded twice.
60
+ - With `config.enabled = false`, batch retrieval no longer downloads OpenAI output files or queries the database.
61
+ - RubyLLM prices Anthropic US inference, fast mode and OpenAI regional hosts, and its streams outside Bedrock read the service tier, 1-hour cache writes and response id.
62
+ - RubyLLM chats, streamed ones included, record Anthropic and OpenAI web search and Gemini grounding fees; they were stored `complete` without the fee.
63
+ - RubyLLM Anthropic streams count the final cumulative input and every `pause_turn` segment's input, and chats keep earlier segments' cache writes.
64
+ - RubyLLM Gemini chats, Interactions protocol included, are priced from the raw usage, keeping audio and URL-context tool tokens, tiers, grounding and response ids.
65
+ - RubyLLM `paint` prices image output at the image output rate for Gemini image models, `gpt-image-1` and `gpt-image-1-mini`; the image was unpriced.
66
+ - RubyLLM transcriptions price prompt text and audio at their own rates when the response splits them (Gemini without it as text), and streams record `stream: true`.
67
+ - RubyLLM Bedrock Converse chats with prompt caching no longer subtract cache tokens from input twice, and split cache writes into 5-minute and 1-hour writes.
68
+ - RubyLLM Gemini embeddings record their tokens instead of 0, and `gemini-embedding-2` prices image, PDF, audio and video parts by modality, adding a `video_input` rate.
69
+ - The pre-send budget estimate covers RubyLLM calls, whose estimate was $0, and ignores base64 images, PDFs and audio, which inflated it and blocked ordinary vision requests.
70
+ - `llm_cost_tracker:backfill_unknown_pricing` also reprices `partial` calls whose already-priced rates are unchanged, and adds only the difference to rollups.
71
+ - Mode rates derived from a standard cache rate are named by their mode key, such as `batch_cache_read_input`, in `pricing_snapshot` and on line items.
72
+ - With inline ingestion, a per-tag `on_exceeded` fires when several calls for one tag value cross the limit together; each call read the others' spend, so none saw itself as the crossing call and the alert never fired.
73
+ - Under `:raise` and `:block_requests`, every budget a call crosses fires its `on_exceeded` before the first error is raised.
74
+ - `budgets.per_tag` without a `:block_requests` rule no longer checks the database schema before an LLM call is sent, so a process that starts during a database outage no longer fails its LLM calls.
75
+ - A tag key given as both a Symbol and a String is stored once, with the later value, not twice; requeue any async inbox rows quarantined because of it.
76
+ - Tags with a nil or empty value count as untagged on the dashboard, in `cost_by_tag` and for `budgets.per_tag`.
77
+ - An OpenAI SDK response without usage logs a warning instead of being skipped silently.
78
+ - The async worker checks its tables through the Rails schema cache, so an idle poll runs one query instead of six.
79
+ - With `ingestion.mode = :async` and `budgets.totals_source = :cache`, a missing rollups table no longer stops the worker from draining the inbox; it warns once and budget reads use the calls ledger, as inline ingestion does.
80
+ - `track_stream` records events passed as symbol-keyed hashes, such as `to_h` of an OpenAI Realtime `response.done` event; they were ignored and the call was stored as `unknown` with 0 tokens.
81
+ - An OpenAI SDK `chat.completions.stream` without `stream_options: { include_usage: true }` logs the same warning as the Faraday path instead of being stored as `unknown` silently.
82
+ - The overview's monthly budget bar and projection marker and the Data Quality coverage bars are visible again, and the dark theme gets its missing chart colors.
83
+ - In apps whose `Time.zone` is not UTC, daily charts, the previous-period line and the spend-anomaly banner use local days when the database knows the zone name.
84
+ - A tag value page's spend chart covers the selected date range instead of the last 30 days.
85
+ - Sortable dashboard tables no longer return a 500 when the host sets `config.action_controller.include_all_helpers = false`.
86
+
87
+ ## [0.14.1] - 2026-09-25
88
+
89
+ ### Added
90
+
91
+ - `bin/rails llm_cost_tracker:doctor` and the boot log warn when an instrumented SDK is newer than its tested range (RubyLLM 3.0 and later), instead of the integration reporting installed while it records nothing.
92
+
93
+ ### Changed
94
+
95
+ - Ruby 3.3 is supported; the minimum drops from 3.4.
96
+ - The Faraday middleware adds `stream_options: { include_usage: true }` only for OpenAI, OpenRouter, DeepSeek, Groq, and Azure OpenAI on the v1 API or `api-version` 2024-06-01 and later without On Your Data or image input, since other servers can reject it and fail the request. Hosts you add to `config.capture.openai_compatible_providers` are left untouched: set the flag yourself if they support it, or their streams record `usage_source: unknown` with a warning.
97
+ - `bin/rails llm_cost_tracker:prices:refresh` refuses a snapshot that zeroes an existing price or charges for a free one, drops a model's `input` or `output` rate, moves a price 100-fold or more, or switches currency, and leaves your pricing file untouched, so one bad commit to the snapshot can no longer silence budgets or make `:block_requests` block every call. `PREVIEW=1` and `prices:check` list the changes; re-run with `FORCE=1` (or `force: true`) to accept them. New models and smaller changes are not checked, so keep reviewing the refreshed file.
98
+ - `pricing.unknown_model_behavior = :raise` records the call, with `cost_status: unknown`, before it raises `LlmCostTracker::UnknownPricingError`, so a call the provider already billed is no longer missing from the ledger.
99
+ - A malformed `pricing.file` fails `LlmCostTracker.configure` at boot instead of every LLM call. A missing one is logged at boot and reported as an error by `doctor`, and calls use bundled prices until `prices:refresh` creates it; before, every call raised and nothing was recorded.
100
+
101
+ ### Removed
102
+
103
+ - An OpenAI or Anthropic SDK stream that your code never iterates is no longer recorded at garbage collection; the finalizer that did it kept every stream alive until process exit.
104
+
105
+ ### Fixed
106
+
107
+ - Per-tag `:block_requests` budgets check tags passed to the Faraday middleware (`f.use :llm_cost_tracker, tags: ...`) before the call is sent; they saw only `with_tags` tags, so an over-budget tenant still reached the provider.
108
+ - RubyLLM transcriptions are priced at the model's audio input rate when it has one; they were priced as text, at a third to a half of the real cost.
109
+ - `RubyLLM.transcribe` with a Gemini or Vertex AI model on RubyLLM 1.x is recorded; RubyLLM 1.x sent it through its own Gemini method, bypassing the wrapped `RubyLLM::Provider#transcribe`.
110
+ - Calls made through RubyLLM 2.x are recorded. RubyLLM 2.0 moved token counts behind `response.tokens`, so every call was dropped with a warning while `doctor` reported the integration installed. A 2.x `paint(count:)` returning several images is recorded once, as RubyLLM bills it.
111
+ - An automatically captured LLM call no longer fails, losing the provider's response, when the gem cannot record it, such as when the async inbox pool times out. Only `BudgetExceededError`, `UnknownPricingError` under `:raise`, and `TransactionAbortedError` reach your code; other recording failures are logged. `LlmCostTracker.track` and `track_stream` still raise them.
112
+ - A recording error no longer replaces an exception your code raised while iterating an OpenAI or Anthropic SDK stream or inside `track_stream`, such as `Sidekiq::Shutdown`; the stream is still recorded, tagged `stream_errored`, and the recording error is logged. `TransactionAbortedError` still wins, with your exception as its cause.
113
+ - An OpenAI or Anthropic batch with an unpriced model under `:raise`, or one crossing a budget, records every result before raising once; results after the failing one were lost and `batches.retrieve` raised on every poll.
114
+ - With `ingestion.mode = :async`, one inbox row the database rejects no longer fails its whole batch, which after five attempts quarantined up to 99 other calls and dropped them from the ledger and budget totals. A failed batch is retried one row at a time and only rejected rows are marked failed. Rows already quarantined this way can be requeued as described in `docs/operations.md`.
115
+ - A NUL byte in a stored string, which PostgreSQL rejects, or invalid UTF-8 in a tag value no longer loses the call, and `LlmCostTracker.with_tags` no longer raises on invalid UTF-8. NUL bytes are removed on write, including from inbox rows written by earlier releases, and invalid UTF-8 in tag values becomes U+FFFD.
116
+ - An interrupted Gemini or Azure stream records its model from the request URL instead of `unknown`.
117
+ - The async inbox's `last_error` is no longer empty when the 1,000-byte cut splits a multibyte character.
118
+ - An LLM call inside your own database transaction no longer breaks it when the ledger write fails; on PostgreSQL your app's writes were lost with `PG::InFailedSqlTransaction`. Ledger writes, budget reads, and batch de-duplication run in a savepoint inside an open transaction. On MySQL, where a deadlock rolls back your whole transaction, the gem raises `LlmCostTracker::TransactionAbortedError` instead of swallowing it.
119
+ - With `config.budgets.totals_source = :cache`, the rollup increment inside your transaction waits for its commit on Rails 7.2 and later, so an open transaction no longer holds the rollup row lock and stalls every request recording spend for the same provider. On Rails 7.1 and in non-joinable transactions such as transactional fixtures it still runs immediately, and it is never retried inside a transaction.
120
+ - Long streams record their real usage instead of zero tokens and `$0`. Streams were buffered up to 1 MB, dropping the final usage event (past about 3,000 output tokens through Faraday); they are now decoded as they arrive, keeping only the first and last events plus billable tool-call and grounding events, so memory stays flat at any length. Image streams through Faraday keep their final usage too.
121
+ - Streams captured through the OpenAI and Anthropic SDK integrations are released once your code drops them, instead of staying in memory with their buffered events and request for the life of the process.
122
+ - OpenAI Realtime cached audio is no longer billed twice, at the cache-read rate and again at the audio rate ($0.258 instead of $0.106 for a 10,000-token `gpt-realtime` turn with 5,000 cached audio tokens). Cached audio and image tokens are priced at the model's cache-read rate, below OpenAI's published cached-audio price on the mini models. OpenAI-compatible usage whose audio and image buckets exceed the reported input has the overlap taken out of audio input, then image input.
123
+ - Dashboard links and filter forms carry only the dashboard's own query parameters. They copied every parameter, and Rails reads some as link options, so a crafted link such as `?script_name=//evil.example` pointed Export CSV, pagination, filter, and tag links at another site, others rewrote links or caused a 500, and a 50 KB junk parameter grew a tag page to 10 MB. A query string over 16 KB is a bad request.
124
+ - A dashboard page accepts at most 10 `tag[...]` filters, counting a tag value page's own value, and answers more with a bad request. Each filter adds a subquery, so PostgreSQL planning memory grew with the square of their count, and around 700 got the backend killed for running out of memory, restarting the database. Tag breakdowns no longer offer drill-down links past the limit.
125
+ - Dashboard filters handle bad input without a 500. A list or hash in a tag, provider, model, stream, or usage-source filter is a bad request instead of silently matching nothing, and a NUL byte in a filter value matches nothing on PostgreSQL instead of raising. A CSV export with an invalid filter, a missing call, or a database error renders an HTML error page with the right status instead of a 500 or HTML labelled as CSV. A tag value page's value overrides a filter on the same key, and tag pages for keys containing a dot, such as `/tags/team.name`, load instead of returning 404 or 406.
126
+ - The Pricing page shows the active price source instead of a 500 when `source` is a list or hash, as in `/pricing?source[]=bundled`; a list or hash `provider` filter is a bad request, as on other pages.
127
+
128
+ ### Security
129
+
130
+ - From v0.9.0 through v0.14.0, a provider API key sent as a URL parameter could be stored in call tags. When a middleware after `f.use :llm_cost_tracker`, such as `f.response :raise_error`, raised on a streamed Gemini `?key=` or Azure `?api-key=` request, the `stream_interrupted_error` tag stored Faraday's error message with the full URL, shown on the dashboard, in CSV exports, and in notification payloads. The tag now holds only the error class, and the HTTP status goes into a new `stream_interrupted_status` tag. Rotate any key that may have been exposed, and clear stored messages with `UPDATE llm_cost_tracker_call_tags SET value = split_part(value, ': ', 1) WHERE key = 'stream_interrupted_error'` on PostgreSQL or ``UPDATE llm_cost_tracker_call_tags SET value = SUBSTRING_INDEX(value, ': ', 1) WHERE `key` = 'stream_interrupted_error'`` on MySQL. With async ingestion, run it after the inbox drains, and delete quarantined inbox rows whose `payload` contains `stream_interrupted_error`.
131
+ - Provider keys, tokens, `key=`/`token=`/`sig=` URL parameters, URL user info, and `Authorization` headers inside a longer string are replaced with `[REDACTED]` in tag values, log lines, the async inbox's `last_error`, `TransactionAbortedError` messages, and the `source_url` that `llm_cost_tracker:prices:refresh` writes and prints; before, only a tag value that was entirely a key was caught.
132
+
5
133
  ## [0.14.0] - 2026-08-26
6
134
 
7
135
  ### Added
data/README.md CHANGED
@@ -1,14 +1,14 @@
1
1
  # LLM Cost Tracker
2
2
 
3
- Self-hosted LLM cost tracking for Rails.
3
+ Per-tenant LLM spend attribution and budgets for Rails — in your database, no proxy.
4
4
 
5
5
  [![Gem Version](https://img.shields.io/gem/v/llm_cost_tracker.svg)](https://rubygems.org/gems/llm_cost_tracker) [![CI](https://github.com/sergey-homenko/llm_cost_tracker/actions/workflows/ruby.yml/badge.svg)](https://github.com/sergey-homenko/llm_cost_tracker/actions) [![codecov](https://codecov.io/gh/sergey-homenko/llm_cost_tracker/branch/main/graph/badge.svg)](https://codecov.io/gh/sergey-homenko/llm_cost_tracker)
6
6
 
7
- Every call your app makes through RubyLLM, the official OpenAI and Anthropic SDKs, Gemini, or any OpenAI-compatible API gets logged: tokens, cost, latency, tags. Calls go app → provider direct. No proxy.
7
+ Every call through RubyLLM, the official OpenAI and Anthropic SDKs, Gemini, or any OpenAI-compatible API is logged with tokens, cost, and your tags. Budgets can block a tenant's next call before it is sent.
8
8
 
9
9
  Not Langfuse, Helicone, or LiteLLM. No prompts, no traces, no replay. Spend attribution only.
10
10
 
11
- Requires Ruby 3.4+, Rails 7.1+, PostgreSQL or MySQL.
11
+ Requires Ruby 3.3+, Rails 8.0+, PostgreSQL or MySQL.
12
12
 
13
13
  <picture> <source media="(prefers-color-scheme: dark)" srcset="docs/dashboard-overview-dark.png"> <img alt="LLM Cost Tracker dashboard" src="docs/dashboard-overview-light.png"> </picture>
14
14
 
@@ -58,7 +58,7 @@ The engine ships without authentication on purpose.
58
58
  ## What lands in the ledger
59
59
 
60
60
  - **Calls.** Provider, model, total tokens, total cost, latency, status.
61
- - **Line items.** Per-component breakdown — text/audio/cached tokens, tool charges (web search, code execution, grounding, container sessions).
61
+ - **Line items.** Per-component breakdown — text/audio/cached tokens, tool charges (web search, grounding, container sessions).
62
62
  - **Tags.** Whatever attribution you pass — user, feature, tenant, env.
63
63
  - **Provider IDs.** Response, project, API key, workspace — for downstream audits.
64
64
  - **Pricing snapshot.** So historical numbers don't drift when prices change.
@@ -73,11 +73,22 @@ The engine ships without authentication on purpose.
73
73
  | Azure OpenAI | Faraday or official SDK (auto-detected on `*.openai.azure.com` and Foundry `*.services.ai.azure.com`, both deployments and `/openai/v1/...`) |
74
74
  | Google Gemini | Faraday |
75
75
  | `ruby-openai` | Faraday |
76
- | OpenRouter, DeepSeek, Groq, LiteLLM-style gateways | OpenAI-compatible Faraday |
76
+ | OpenRouter, DeepSeek, Groq; xAI, Mistral, and other gateways (LiteLLM etc.) once their host is added to `config.capture.openai_compatible_providers` | OpenAI-compatible Faraday, or the official OpenAI SDK with `base_url` on that host |
77
77
  | Anything else | `LlmCostTracker.track` |
78
78
 
79
79
  Streams capture when the provider emits final usage. OpenAI Faraday streams to `/chat/completions` get `stream_options: { include_usage: true }` auto-injected so the final usage chunk lands in the ledger (opt out via `config.capture.request_stream_usage = false`).
80
80
 
81
+ Captured does not always mean priced:
82
+
83
+ | Cost comes from | Calls |
84
+ | --- | --- |
85
+ | The amount billed, from `usage.cost` in the response or final stream chunk | OpenRouter through Faraday, the official OpenAI SDK, or `track_stream`, and any OpenAI-compatible gateway that returns `usage.cost`; bundled prices apply when it is missing or the call comes through RubyLLM, other than an OpenRouter chat |
86
+ | Bundled [`prices.json`](lib/llm_cost_tracker/prices.json) | The OpenAI, Anthropic, Gemini, Groq, OpenRouter, xAI, and Mistral models it lists, and the same Claude models on Bedrock through RubyLLM |
87
+ | The OpenAI, Anthropic, or Gemini price for the same model name | Azure OpenAI (by the model in the response, not the deployment name), Vertex AI through RubyLLM, gateways that pass a listed model name through |
88
+ | Nothing: recorded with `cost_status: unknown` | DeepSeek, and through RubyLLM also Perplexity, Ollama, other Bedrock models, and Claude on GovCloud (`us-gov.` profiles) |
89
+
90
+ Add missing prices to `config.pricing.file` or `config.pricing.overrides` ([Pricing](docs/pricing.md)), then run `bin/rails llm_cost_tracker:backfill_unknown_pricing` to price the calls already recorded.
91
+
81
92
  ## What it isn't
82
93
 
83
94
  - No proxy. Direct calls only.
@@ -154,6 +154,9 @@
154
154
  --lct-shadow-md: 0 2px 6px rgba(0, 0, 0, 0.35), 0 1px 2px rgba(0, 0, 0, 0.25);
155
155
  --lct-shadow-lg: 0 12px 32px rgba(0, 0, 0, 0.5), 0 2px 6px rgba(0, 0, 0, 0.3);
156
156
  --lct-focus-ring: 0 0 0 3px rgba(124, 131, 255, 0.30);
157
+ --lct-chart-marker: rgba(230, 236, 245, 0.45);
158
+ --lct-chart-secondary: rgba(230, 236, 245, 0.32);
159
+ --lct-chip-remove: rgba(230, 236, 245, 0.58);
157
160
  }
158
161
 
159
162
  * { box-sizing: border-box; }
@@ -613,6 +616,19 @@
613
616
  .lct-stack-swatch { width: 10px; height: 10px; border-radius: 2px; display: inline-block; }
614
617
  .lct-stack-meta { color: var(--lct-muted); }
615
618
  .lct-stack-empty { color: var(--lct-muted); font-size: var(--fs-sm); margin: 0; }
619
+ .lct-bar-track,
620
+ .lct-budget-track { background: var(--lct-surface-2); border-radius: 999px; overflow: hidden; }
621
+ .lct-bar-track { height: 8px; min-width: 96px; }
622
+ .lct-budget-track { height: 10px; position: relative; }
623
+ .lct-bar-fill,
624
+ .lct-budget-fill { height: 100%; display: block; background: var(--lct-accent); }
625
+ .lct-budget-fill--warn { background: var(--lct-warning); }
626
+ .lct-budget-fill--over { background: var(--lct-danger); }
627
+ .lct-budget-marker { position: absolute; top: 0; bottom: 0; border-left: 2px dashed var(--lct-chart-marker); }
628
+ .lct-budget-projection { display: flex; flex-wrap: wrap; gap: 8px 12px; align-items: baseline; margin: 10px 0 0; color: var(--lct-muted); font-size: var(--fs-sm); }
629
+ .lct-budget-projection strong { color: var(--lct-text); }
630
+ .lct-budget-projection-status { font-weight: 600; }
631
+ .lct-budget-projection-status--over { color: var(--lct-warning-copy); }
616
632
 
617
633
  .lct-legend { display: inline-flex; align-items: center; gap: 6px; font-size: var(--fs-xs); color: var(--lct-muted); }
618
634
  .lct-legend-dot { width: 9px; height: 9px; border-radius: 2px; }
@@ -4,11 +4,14 @@ require "securerandom"
4
4
 
5
5
  module LlmCostTracker
6
6
  class ApplicationController < ActionController::Base
7
+ MAX_QUERY_BYTES = 16 * 1024
8
+
7
9
  layout "llm_cost_tracker/application"
8
10
 
9
11
  protect_from_forgery with: :exception
10
12
 
11
13
  before_action :set_dashboard_security_headers
14
+ before_action :reject_oversized_query
12
15
  before_action :ensure_current_schema
13
16
  before_action :assign_dashboard_date_range
14
17
 
@@ -22,6 +25,12 @@ module LlmCostTracker
22
25
 
23
26
  private
24
27
 
28
+ def reject_oversized_query
29
+ return if request.query_string.bytesize <= MAX_QUERY_BYTES
30
+
31
+ raise LlmCostTracker::InvalidFilterError, "query string exceeds #{MAX_QUERY_BYTES / 1024} KB"
32
+ end
33
+
25
34
  def ensure_current_schema
26
35
  drift = LlmCostTracker::Dashboard::SetupState.current
27
36
  return unless drift
@@ -40,16 +49,20 @@ module LlmCostTracker
40
49
  end
41
50
 
42
51
  def render_database_error(_error)
43
- render "llm_cost_tracker/errors/database", status: :internal_server_error
52
+ render_error_page("database", :internal_server_error)
44
53
  end
45
54
 
46
55
  def render_invalid_filter(error)
47
56
  @error_message = error.message
48
- render "llm_cost_tracker/errors/invalid_filter", status: :bad_request
57
+ render_error_page("invalid_filter", :bad_request)
49
58
  end
50
59
 
51
60
  def render_not_found
52
- render "llm_cost_tracker/errors/not_found", status: :not_found
61
+ render_error_page("not_found", :not_found)
62
+ end
63
+
64
+ def render_error_page(name, status)
65
+ render "llm_cost_tracker/errors/#{name}", status: status, formats: :html, content_type: "text/html"
53
66
  end
54
67
 
55
68
  def set_dashboard_security_headers
@@ -16,11 +16,9 @@ module LlmCostTracker
16
16
  }.freeze
17
17
 
18
18
  def index
19
- @sort = params[:sort].to_s
20
- @dir = params[:dir].to_s
21
19
  scope = Dashboard::Filter.call(params: params)
22
20
  scope = scope.unknown_pricing if params[:cost_status].to_s == "incomplete"
23
- ordered_scope = scope.order(*calls_order(@sort, @dir))
21
+ ordered_scope = scope.order(*calls_order(params[:sort].to_s, params[:dir].to_s))
24
22
 
25
23
  respond_to do |format|
26
24
  format.html do
@@ -5,13 +5,11 @@ module LlmCostTracker
5
5
  MAX_ROWS = 200
6
6
 
7
7
  def index
8
- @sort = params[:sort].to_s
9
- @dir = params[:dir].to_s
10
8
  @rows = Dashboard::TopModels.call(
11
9
  scope: Dashboard::Filter.call(params: params),
12
10
  limit: MAX_ROWS,
13
- sort: @sort,
14
- direction: @dir
11
+ sort: params[:sort].to_s,
12
+ direction: params[:dir].to_s
15
13
  )
16
14
  end
17
15
  end
@@ -4,10 +4,10 @@ module LlmCostTracker
4
4
  class PricingController < ApplicationController
5
5
  def index
6
6
  @overview = Dashboard::PricingOverview.call
7
- requested = params[:source]&.to_sym
7
+ requested = params[:source].to_s.to_sym
8
8
  @active_source = @overview.fetch(:sources).key?(requested) ? requested : @overview.fetch(:effective_source)
9
9
  @source_data = @overview.fetch(:sources).fetch(@active_source)
10
- @provider_filter = params[:provider].to_s.presence
10
+ @provider_filter = Dashboard::Params.scalar(params[:provider], :provider).presence
11
11
  @rows = @source_data.fetch(:rows)
12
12
  @rows = @rows.select { |row| row.provider == @provider_filter } if @provider_filter
13
13
  @providers = @source_data.fetch(:rows).map(&:provider).compact.uniq.sort
@@ -7,22 +7,24 @@ module LlmCostTracker
7
7
  end
8
8
 
9
9
  def show
10
- scope = Dashboard::Filter.call(params: params)
11
- @value = params[:tag_value].to_s
10
+ @value = Dashboard::Params.scalar(params[:tag_value], :tag_value)
12
11
 
13
12
  if @value.empty?
14
- @sort = params[:sort].to_s
15
- @dir = params[:dir].to_s
16
- @breakdown = Dashboard::TagBreakdown.call(scope: scope, key: params[:key], sort: @sort, direction: @dir)
13
+ @breakdown = Dashboard::TagBreakdown.call(
14
+ scope: Dashboard::Filter.call(params: params),
15
+ key: params[:key],
16
+ sort: params[:sort].to_s,
17
+ direction: params[:dir].to_s
18
+ )
17
19
  else
18
20
  @key = LlmCostTracker::Tags::Key.validate!(
19
21
  params[:key],
20
22
  error_class: LlmCostTracker::InvalidFilterError
21
23
  )
22
- value_scope = scope.by_tag(@key, @value)
24
+ value_scope = Dashboard::Filter.call(params: params, tags: { @key => @value })
23
25
  @value_total_cost = value_scope.sum(:total_cost).to_f
24
26
  @value_calls = value_scope.count
25
- @value_points = Dashboard::TimeSeries.call(scope: value_scope)
27
+ @value_points = Dashboard::TimeSeries.call(scope: value_scope, from: @from_date, to: @to_date)
26
28
  end
27
29
  end
28
30
  end
@@ -13,6 +13,7 @@ module LlmCostTracker
13
13
  include PaginationHelper
14
14
  include TokenUsageHelper
15
15
  include InlineStyleHelper
16
+ include SortableTableHelper
16
17
 
17
18
  def dashboard_section
18
19
  path = request.path.to_s
@@ -120,13 +121,10 @@ module LlmCostTracker
120
121
  end
121
122
 
122
123
  def tag_chip_entries(tags, limit: 3)
123
- normalized = normalized_tags(tags)
124
- return [] if normalized.empty?
125
-
126
- visible = normalized.first(limit).map do |key, value|
127
- { key: key.to_s, value: tag_value_summary(value) }
124
+ visible = tags.first(limit).map do |key, value|
125
+ { key: key.to_s, value: truncate_text(value.to_s, TAG_VALUE_SUMMARY_BYTES) }
128
126
  end
129
- visible << { more: normalized.size - limit } if normalized.size > limit
127
+ visible << { more: tags.size - limit } if tags.size > limit
130
128
  visible
131
129
  end
132
130
 
@@ -135,7 +133,7 @@ module LlmCostTracker
135
133
  end
136
134
 
137
135
  def current_query(overrides = {})
138
- request.query_parameters.symbolize_keys.merge(overrides)
136
+ request.query_parameters.symbolize_keys.slice(*LlmCostTracker::Dashboard::Params::QUERY_KEYS).merge(overrides)
139
137
  end
140
138
 
141
139
  def calls_query_for_model(provider:, model:)
@@ -144,25 +142,6 @@ module LlmCostTracker
144
142
 
145
143
  private
146
144
 
147
- def normalized_tags(tags)
148
- return tags.transform_keys(&:to_s) if tags.is_a?(Hash)
149
-
150
- JSON.parse(tags || "{}")
151
- rescue JSON::ParserError, TypeError
152
- {}
153
- end
154
-
155
- def tag_value_summary(value)
156
- string = case value
157
- when Hash, Array
158
- JSON.generate(value)
159
- else
160
- value.to_s
161
- end
162
-
163
- truncate_text(string, TAG_VALUE_SUMMARY_BYTES)
164
- end
165
-
166
145
  def truncate_text(string, limit)
167
146
  return string if string.bytesize <= limit
168
147
 
@@ -7,7 +7,7 @@ module LlmCostTracker
7
7
 
8
8
  cfg = chart_config(points, comparison_points, height, y_ticks)
9
9
  parts = [chart_svg_open(cfg), "<title>Daily spend trend</title>", chart_area_gradient_def]
10
- parts.concat(chart_grid_and_axis(cfg))
10
+ parts.concat((0..cfg[:y_ticks]).map { |i| chart_tick_line(cfg, i) })
11
11
  parts << chart_paths(cfg)
12
12
  parts.concat(chart_dots(cfg))
13
13
  parts.concat(chart_x_labels(cfg))
@@ -34,7 +34,7 @@ module LlmCostTracker
34
34
  peak_index = points.each_with_index.max_by { |point, _| point[:cost].to_f }&.last
35
35
  { width: width, height: height, pad: pad, plot_w: plot_w, plot_h: plot_h,
36
36
  max_cost: max_cost, n: points.size, y_ticks: y_ticks, points: points, coords: coords,
37
- comparison_points: comparison_points, comparison_coords: comparison_coords,
37
+ comparison_coords: comparison_coords,
38
38
  peak_index: peak_index }
39
39
  end
40
40
 
@@ -60,21 +60,17 @@ module LlmCostTracker
60
60
  "<svg #{attrs}>"
61
61
  end
62
62
 
63
- def chart_grid_and_axis(cfg)
64
- (0..cfg[:y_ticks]).map { |i| chart_tick_line(cfg, i) }
65
- end
66
-
67
63
  def chart_tick_line(cfg, idx)
68
64
  pad = cfg[:pad]
69
65
  right_x = chart_fmt(pad[:left] + cfg[:plot_w])
70
66
  left_x = chart_fmt(pad[:left])
71
67
  text_x = chart_fmt(pad[:left] - 8)
72
68
  value = cfg[:max_cost] * (cfg[:y_ticks] - idx).to_f / cfg[:y_ticks]
73
- y = chart_fmt(pad[:top] + (cfg[:plot_h] * idx.to_f / cfg[:y_ticks]))
74
- label_y = chart_fmt(pad[:top] + (cfg[:plot_h] * idx.to_f / cfg[:y_ticks]) + 3)
69
+ tick_y = pad[:top] + (cfg[:plot_h] * idx.to_f / cfg[:y_ticks])
70
+ y = chart_fmt(tick_y)
71
+ label_y = chart_fmt(tick_y + 3)
75
72
  grid = %(<line class="lct-chart-grid" x1="#{left_x}" x2="#{right_x}" y1="#{y}" y2="#{y}"/>)
76
- label = format("%.2f", value)
77
- text = %(<text class="lct-chart-axis" x="#{text_x}" y="#{label_y}" text-anchor="end">$#{label}</text>)
73
+ text = %(<text class="lct-chart-axis" x="#{text_x}" y="#{label_y}" text-anchor="end">$#{chart_fmt(value)}</text>)
78
74
  "#{grid}#{text}"
79
75
  end
80
76
 
@@ -16,6 +16,11 @@ module LlmCostTracker
16
16
  query
17
17
  end
18
18
 
19
+ def tag_drilldown_allowed?(key)
20
+ tags = LlmCostTracker::Dashboard::Params.tag_query(current_query[:tag])
21
+ tags.except(key.to_s).size < LlmCostTracker::Dashboard::Filter::MAX_TAG_FILTERS
22
+ end
23
+
19
24
  def hidden_query_fields(query, prefix: nil)
20
25
  safe_join(query.flat_map do |key, value|
21
26
  name = prefix ? "#{prefix}[#{key}]" : key.to_s
@@ -33,41 +33,21 @@ module LlmCostTracker
33
33
  }.freeze
34
34
 
35
35
  def token_usage_stack_components
36
- token_usage_display_components(labels: COMPONENT_LABELS).select do |component|
37
- component.fetch(:cost_key)
38
- end
39
- end
40
-
41
- def call_line_item_costs_by_component(call)
42
- call.line_items.each_with_object({}) do |line_item, accumulator|
43
- component = LlmCostTracker::Usage::Catalog.token_priced_for(
44
- kind: line_item.kind, direction: line_item.direction, cache_state: line_item.cache_state
45
- )
46
- accumulator[component.key] = line_item.cost if component && line_item.cost
47
- end
48
- end
49
-
50
- private
51
-
52
- def token_usage_display_components(labels:)
53
36
  LlmCostTracker::Usage::Catalog.token_priced.map do |component|
54
37
  token_key = component.token_key
55
38
  {
56
39
  token_key: token_key,
57
- cost_key: component.cost_key,
58
40
  price_key: component.key,
59
- label: labels.fetch(token_key),
41
+ label: COMPONENT_LABELS.fetch(token_key),
60
42
  css_class: STACK_CLASSES[token_key]
61
43
  }
62
- end + [
63
- {
64
- token_key: :hidden_output_tokens,
65
- cost_key: nil,
66
- price_key: nil,
67
- label: labels.fetch(:hidden_output_tokens),
68
- css_class: nil
69
- }
70
- ]
44
+ end
45
+ end
46
+
47
+ def call_line_item_costs_by_component(call)
48
+ LlmCostTracker::Usage::Catalog.costs_by_component(
49
+ call.line_items.map { |item| [item.kind, item.direction, item.cache_state, item.cost] }
50
+ )
71
51
  end
72
52
  end
73
53
  end
@@ -45,7 +45,10 @@ module LlmCostTracker
45
45
  def already_recorded?(provider:, provider_response_id:)
46
46
  return false if provider_response_id.to_s.empty?
47
47
 
48
- where(provider: provider, provider_response_id: provider_response_id).exists?
48
+ Ledger::Isolation.guard(self) do
49
+ where(provider: provider, provider_response_id: provider_response_id)
50
+ .where.not(usage_source: Usage::Source::UNKNOWN).exists?
51
+ end
49
52
  end
50
53
 
51
54
  def by_tag(key, value) = by_tags(key => value)
@@ -86,8 +89,12 @@ module LlmCostTracker
86
89
 
87
90
  def latency_by_provider = group(:provider).average(:latency_ms).transform_values(&:to_f)
88
91
 
89
- def group_by_period(period, column: :tracked_at)
90
- group(Arel.sql(period_group_expression(period, column: column)))
92
+ def group_by_period(period, column: :tracked_at, time_zone: nil)
93
+ column = column.to_s
94
+ raise ArgumentError, "invalid period column: #{column.inspect}" unless column_names.include?(column)
95
+
96
+ bucket = Ledger::Schema::Adapter.period_bucket_sql(connection, period, qualified(column), time_zone: time_zone)
97
+ group(Arel.sql(bucket))
91
98
  end
92
99
 
93
100
  def daily_costs(days: 30)
@@ -109,17 +116,6 @@ module LlmCostTracker
109
116
  relation = relation.limit(limit) if limit
110
117
  relation
111
118
  end
112
-
113
- def period_group_expression(period, column:)
114
- Ledger::Schema::Adapter.period_bucket_sql(connection, period, period_column_expression(column))
115
- end
116
-
117
- def period_column_expression(column)
118
- column = column.to_s
119
- return "#{quoted_table_name}.#{connection.quote_column_name(column)}" if column_names.include?(column)
120
-
121
- raise ArgumentError, "invalid period column: #{column.inspect}"
122
- end
123
119
  end
124
120
 
125
121
  def tag_pairs