llm_cost_tracker 0.12.0 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +71 -16
  3. data/README.md +16 -28
  4. data/app/controllers/llm_cost_tracker/application_controller.rb +3 -2
  5. data/app/controllers/llm_cost_tracker/data_quality_controller.rb +2 -0
  6. data/app/controllers/llm_cost_tracker/models_controller.rb +3 -1
  7. data/app/helpers/llm_cost_tracker/application_helper.rb +6 -4
  8. data/app/helpers/llm_cost_tracker/dashboard_query_helper.rb +11 -0
  9. data/app/models/llm_cost_tracker/call.rb +9 -3
  10. data/app/models/llm_cost_tracker/call_rollup.rb +19 -3
  11. data/app/models/llm_cost_tracker/ingestion/inbox_entry.rb +1 -0
  12. data/app/services/llm_cost_tracker/dashboard/data_quality.rb +25 -0
  13. data/app/services/llm_cost_tracker/dashboard/monthly_budget.rb +1 -1
  14. data/app/services/llm_cost_tracker/dashboard/pagination.rb +2 -1
  15. data/app/services/llm_cost_tracker/dashboard/pricing_overview.rb +3 -3
  16. data/app/services/llm_cost_tracker/dashboard/setup_state.rb +5 -3
  17. data/app/views/llm_cost_tracker/calls/show.html.erb +8 -10
  18. data/app/views/llm_cost_tracker/data_quality/index.html.erb +22 -0
  19. data/app/views/llm_cost_tracker/pricing/index.html.erb +1 -1
  20. data/app/views/llm_cost_tracker/shared/_filter_pill_date.html.erb +1 -3
  21. data/app/views/llm_cost_tracker/shared/_filter_pill_model.html.erb +1 -3
  22. data/app/views/llm_cost_tracker/shared/_filter_pill_provider.html.erb +1 -3
  23. data/app/views/llm_cost_tracker/shared/_filter_pill_stream.html.erb +1 -3
  24. data/lib/llm_cost_tracker/budget/per_tag.rb +159 -0
  25. data/lib/llm_cost_tracker/budget.rb +120 -32
  26. data/lib/llm_cost_tracker/capture/stream_collector.rb +5 -4
  27. data/lib/llm_cost_tracker/charges/cost_status.rb +4 -3
  28. data/lib/llm_cost_tracker/charges/line_item.rb +3 -2
  29. data/lib/llm_cost_tracker/configuration/budgets.rb +93 -0
  30. data/lib/llm_cost_tracker/configuration/capture.rb +42 -0
  31. data/lib/llm_cost_tracker/configuration/ingestion.rb +20 -0
  32. data/lib/llm_cost_tracker/configuration/mutability.rb +33 -0
  33. data/lib/llm_cost_tracker/configuration/pricing.rb +36 -0
  34. data/lib/llm_cost_tracker/configuration/section.rb +59 -0
  35. data/lib/llm_cost_tracker/configuration/tags.rb +52 -0
  36. data/lib/llm_cost_tracker/configuration.rb +69 -125
  37. data/lib/llm_cost_tracker/deprecator.rb +9 -0
  38. data/lib/llm_cost_tracker/doctor/ingestion_check.rb +17 -8
  39. data/lib/llm_cost_tracker/doctor/price_check.rb +1 -1
  40. data/lib/llm_cost_tracker/doctor.rb +4 -4
  41. data/lib/llm_cost_tracker/engine.rb +4 -0
  42. data/lib/llm_cost_tracker/errors.rb +12 -3
  43. data/lib/llm_cost_tracker/event.rb +3 -1
  44. data/lib/llm_cost_tracker/generators/llm_cost_tracker/async_ingestion_generator.rb +2 -2
  45. data/lib/llm_cost_tracker/generators/llm_cost_tracker/call_rollups_generator.rb +2 -2
  46. data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/create_llm_cost_tracker_async_ingestion.rb.erb +0 -1
  47. data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/create_llm_cost_tracker_calls.rb.erb +8 -4
  48. data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/initializer.rb.erb +50 -30
  49. data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/upgrade_indexes.rb.erb +40 -0
  50. data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/upgrade_per_tag_budgets.rb.erb +41 -0
  51. data/lib/llm_cost_tracker/generators/llm_cost_tracker/upgrade_indexes_generator.rb +30 -0
  52. data/lib/llm_cost_tracker/generators/llm_cost_tracker/upgrade_per_tag_budgets_generator.rb +30 -0
  53. data/lib/llm_cost_tracker/ingestion/batch.rb +27 -7
  54. data/lib/llm_cost_tracker/ingestion/pool.rb +9 -2
  55. data/lib/llm_cost_tracker/ingestion.rb +3 -7
  56. data/lib/llm_cost_tracker/integrations/anthropic.rb +8 -7
  57. data/lib/llm_cost_tracker/integrations/base.rb +1 -1
  58. data/lib/llm_cost_tracker/integrations/openai/batch_capture.rb +10 -12
  59. data/lib/llm_cost_tracker/ledger/period/totals.rb +1 -1
  60. data/lib/llm_cost_tracker/ledger/rollups.rb +34 -4
  61. data/lib/llm_cost_tracker/ledger/store.rb +16 -2
  62. data/lib/llm_cost_tracker/ledger/tags/encoding.rb +15 -5
  63. data/lib/llm_cost_tracker/logging.rb +3 -5
  64. data/lib/llm_cost_tracker/middleware/faraday.rb +1 -1
  65. data/lib/llm_cost_tracker/prices.json +1384 -304
  66. data/lib/llm_cost_tracker/pricing/backfill.rb +11 -1
  67. data/lib/llm_cost_tracker/pricing/calculation.rb +44 -25
  68. data/lib/llm_cost_tracker/pricing/effective_prices.rb +26 -15
  69. data/lib/llm_cost_tracker/pricing/matcher.rb +24 -3
  70. data/lib/llm_cost_tracker/pricing/mode.rb +7 -7
  71. data/lib/llm_cost_tracker/pricing/rate.rb +1 -2
  72. data/lib/llm_cost_tracker/pricing/registry.rb +5 -5
  73. data/lib/llm_cost_tracker/pricing/sync.rb +1 -1
  74. data/lib/llm_cost_tracker/pricing/unknown.rb +11 -8
  75. data/lib/llm_cost_tracker/providers/anthropic/usage_extractor.rb +3 -2
  76. data/lib/llm_cost_tracker/providers/openai/model_families.rb +0 -7
  77. data/lib/llm_cost_tracker/providers/openai/response_parser.rb +5 -2
  78. data/lib/llm_cost_tracker/providers/openai/usage_extractor.rb +5 -4
  79. data/lib/llm_cost_tracker/providers/openai_compatible/parser.rb +2 -2
  80. data/lib/llm_cost_tracker/railtie.rb +3 -7
  81. data/lib/llm_cost_tracker/report/data.rb +2 -2
  82. data/lib/llm_cost_tracker/retention.rb +22 -10
  83. data/lib/llm_cost_tracker/tags/context.rb +3 -3
  84. data/lib/llm_cost_tracker/tags/sanitizer.rb +4 -4
  85. data/lib/llm_cost_tracker/tracker.rb +16 -26
  86. data/lib/llm_cost_tracker/usage/catalog.rb +12 -2
  87. data/lib/llm_cost_tracker/usage/dimensions.yml +0 -24
  88. data/lib/llm_cost_tracker/usage/token_usage.rb +20 -75
  89. data/lib/llm_cost_tracker/version.rb +1 -1
  90. data/lib/llm_cost_tracker.rb +7 -2
  91. data/lib/tasks/llm_cost_tracker.rake +23 -10
  92. metadata +22 -9
@@ -4,6 +4,7 @@ require_relative "../pricing"
4
4
  require_relative "../charges/line_item"
5
5
  require_relative "../ledger/rollups"
6
6
  require_relative "../usage/token_usage"
7
+ require_relative "../budget/per_tag"
7
8
 
8
9
  module LlmCostTracker
9
10
  module Pricing
@@ -30,7 +31,7 @@ module LlmCostTracker
30
31
  rollup_events << rollup_event_for(call, calculation)
31
32
  recomputed += 1
32
33
  end
33
- Ledger::Rollups.increment!(rollup_events) if rollup_events.any?
34
+ Ledger::Rollups.increment!(rollup_events)
34
35
  end
35
36
  end
36
37
 
@@ -61,6 +62,7 @@ module LlmCostTracker
61
62
  pricing_snapshot: calculation.snapshot,
62
63
  cost_status: calculation.cost_status
63
64
  )
65
+ resync_tag_costs(call, calculation.cost.total)
64
66
  token_priced = calculation.priced_line_items.select(&:token?).index_by { |item| dimension_key(item) }
65
67
  service_priced = calculation.priced_line_items.reject(&:token?)
66
68
  token_records, service_records = call.line_items.partition { |record| record.unit == "token" }
@@ -69,6 +71,14 @@ module LlmCostTracker
69
71
  service_records.sort_by(&:position).zip(service_priced).each { |record, priced| apply_rate(record, priced) }
70
72
  end
71
73
 
74
+ def resync_tag_costs(call, total_cost)
75
+ return unless LlmCostTracker::Budget::PerTag.columns?
76
+
77
+ LlmCostTracker::CallTag
78
+ .where(llm_cost_tracker_call_id: call.id)
79
+ .update_all(total_cost: total_cost)
80
+ end
81
+
72
82
  def apply_rate(record, priced)
73
83
  return unless priced
74
84
 
@@ -10,7 +10,8 @@ module LlmCostTracker
10
10
  module Pricing
11
11
  class Calculation
12
12
  RATE_DENOMINATOR_TOKENS = Pricing::RATE_BASIS_QUANTITIES.fetch("per_million_tokens")
13
- private_constant :RATE_DENOMINATOR_TOKENS
13
+ SNAPSHOT_SCHEMA_VERSION = 1
14
+ private_constant :RATE_DENOMINATOR_TOKENS, :SNAPSHOT_SCHEMA_VERSION
14
15
 
15
16
  def self.for(provider:, model:, tokens:, pricing_mode:, line_items: [], usage_source: nil)
16
17
  new(provider: provider,
@@ -61,7 +62,12 @@ module LlmCostTracker
61
62
  def snapshot
62
63
  return @snapshot if defined?(@snapshot)
63
64
 
64
- @snapshot = priceable? ? build_snapshot : nil
65
+ @snapshot =
66
+ if priceable?
67
+ build_snapshot
68
+ elsif kept_service_lines.any?
69
+ build_service_snapshot
70
+ end
65
71
  end
66
72
 
67
73
  def cost
@@ -131,7 +137,7 @@ module LlmCostTracker
131
137
 
132
138
  def build_snapshot
133
139
  {
134
- "schema_version" => 1,
140
+ "schema_version" => SNAPSHOT_SCHEMA_VERSION,
135
141
  "source" => match.source.name,
136
142
  "source_key" => match.key,
137
143
  "source_version" => match.source.version,
@@ -141,6 +147,18 @@ module LlmCostTracker
141
147
  }
142
148
  end
143
149
 
150
+ def build_service_snapshot
151
+ primary = kept_service_lines.first
152
+ {
153
+ "schema_version" => SNAPSHOT_SCHEMA_VERSION,
154
+ "source" => primary.price_source,
155
+ "source_version" => primary.price_source_version,
156
+ "matched_by" => "service_charges",
157
+ "currency" => cost.currency,
158
+ "rates" => service_charge_rates
159
+ }
160
+ end
161
+
144
162
  def token_charge_rates
145
163
  priced_token_line_items.each_with_object({}) do |line_item, rates|
146
164
  next if line_item.price_key.nil? || line_item.rate_amount.nil?
@@ -150,8 +168,8 @@ module LlmCostTracker
150
168
  end
151
169
 
152
170
  def service_charge_rates
153
- priced_line_items.each_with_object({}) do |line_item, rates|
154
- next if line_item.token? || line_item.price_key.nil? || line_item.rate_amount.nil?
171
+ kept_service_lines.each_with_object({}) do |line_item, rates|
172
+ next if line_item.price_key.nil? || line_item.rate_amount.nil?
155
173
 
156
174
  rates[line_item.price_key] ||= rate_entry(line_item.rate_amount, line_item.rate_quantity)
157
175
  end
@@ -162,23 +180,23 @@ module LlmCostTracker
162
180
  end
163
181
 
164
182
  def price_token(line_item)
165
- dimension = dimension_for(line_item)
183
+ dimension = line_item.dimension
166
184
  return line_item unless dimension
167
185
  return line_item.with(cost_status: Charges::CostStatus::UNKNOWN) unless priceable?
168
186
 
169
187
  price = effective[dimension.key]
170
188
  return line_item.with(cost_status: Charges::CostStatus::UNKNOWN) if price.nil?
171
189
 
172
- line_item.with_rate(token_rate(dimension, price))
190
+ line_item.with_rate(token_rate(price))
173
191
  end
174
192
 
175
- def token_rate(dimension, price)
193
+ def token_rate(price)
176
194
  Pricing::Rate.new(
177
- amount: price.to_d,
195
+ amount: price.amount.to_d,
178
196
  quantity: RATE_DENOMINATOR_TOKENS.to_d,
179
197
  currency: match.source.currency,
180
198
  source: match.source.name,
181
- source_key: dimension.key,
199
+ source_key: price.key,
182
200
  source_version: match.source.version
183
201
  )
184
202
  end
@@ -210,30 +228,31 @@ module LlmCostTracker
210
228
  )
211
229
  end
212
230
 
213
- def dimension_for(line_item)
214
- Usage::Catalog.all.find do |dimension|
215
- dimension.kind == line_item.kind &&
216
- dimension.direction == line_item.direction &&
217
- dimension.modality == line_item.modality &&
218
- dimension.cache_state == line_item.cache_state &&
219
- dimension.unit == line_item.unit
231
+ def kept_service_lines
232
+ return @kept_service_lines if defined?(@kept_service_lines)
233
+
234
+ @kept_service_lines = begin
235
+ priced_services = priced_line_items.reject(&:token?).select(&:priced?)
236
+ if priced_services.empty?
237
+ []
238
+ else
239
+ base_currency = base_currency_for(token_cost, priced_services)
240
+ matching, mismatched = priced_services.partition { |line| line.currency.to_s == base_currency.to_s }
241
+ warn_currency_mismatch(mismatched, base_currency) if mismatched.any?
242
+ matching
243
+ end
220
244
  end
221
245
  end
222
246
 
223
247
  def combine_service_lines
224
248
  cost = token_cost
225
- priced_services = priced_line_items.reject(&:token?).select(&:priced?)
226
- return cost if priced_services.empty?
227
-
228
- base_currency = base_currency_for(cost, priced_services)
229
- matching, mismatched = priced_services.partition { |line| line.currency.to_s == base_currency.to_s }
230
- warn_currency_mismatch(mismatched, base_currency) if mismatched.any?
249
+ return cost if kept_service_lines.empty?
231
250
 
232
- service_total = matching.sum(BigDecimal("0")) { |line| line.cost_value.round(8) }
251
+ service_total = kept_service_lines.sum(BigDecimal("0")) { |line| line.cost_value.round(8) }
233
252
  Charges::Cost.new(
234
253
  components: cost ? cost.components : {}.freeze,
235
254
  total: (cost&.total || BigDecimal("0")) + service_total,
236
- currency: (cost&.currency || base_currency).to_s
255
+ currency: (cost&.currency || kept_service_lines.first.currency).to_s
237
256
  )
238
257
  end
239
258
 
@@ -8,38 +8,49 @@ require_relative "price_key"
8
8
  module LlmCostTracker
9
9
  module Pricing
10
10
  module EffectivePrices
11
+ Resolved = Data.define(:amount, :key)
12
+
11
13
  class << self
12
14
  def call(usage:, quantities:, prices:, pricing_mode:)
13
15
  context_tier = context_tier?(usage: usage, prices: prices)
14
16
  orderings = pricing_mode && Mode.permutations_for(pricing_mode)
15
17
 
16
18
  quantities.to_h do |price_key, tokens|
17
- price = if tokens.positive?
18
- price_for(
19
- prices: prices,
20
- key: price_key,
21
- orderings: orderings,
22
- context_tier: context_tier
23
- )
24
- else
25
- BigDecimal("0")
26
- end
27
- [price_key, price]
19
+ resolved = if tokens.positive?
20
+ price_for(
21
+ prices: prices,
22
+ key: price_key,
23
+ orderings: orderings,
24
+ context_tier: context_tier
25
+ )
26
+ else
27
+ Resolved.new(amount: BigDecimal("0"), key: price_key)
28
+ end
29
+ [price_key, resolved]
28
30
  end
29
31
  end
30
32
 
31
33
  private
32
34
 
33
35
  def price_for(prices:, key:, orderings:, context_tier:)
34
- return prices[PriceKey.build(key, above_context: context_tier)] unless orderings
36
+ unless orderings
37
+ standard = PriceKey.build(key, above_context: context_tier)
38
+ return resolve(prices[standard], standard)
39
+ end
35
40
 
36
41
  orderings.each do |mode|
37
- direct = prices[PriceKey.build(key, mode: mode, above_context: context_tier)]
38
- return direct if direct
42
+ table_key = PriceKey.build(key, mode: mode, above_context: context_tier)
43
+ direct = prices[table_key]
44
+ return resolve(direct, table_key) if direct
39
45
  end
40
46
  return nil if %w[input output].include?(key)
41
47
 
42
- derived_mode_price(prices: prices, key: key, modes: orderings, context_tier: context_tier)
48
+ derived = derived_mode_price(prices: prices, key: key, modes: orderings, context_tier: context_tier)
49
+ resolve(derived, nil)
50
+ end
51
+
52
+ def resolve(amount, key)
53
+ amount && Resolved.new(amount: amount, key: key)
43
54
  end
44
55
 
45
56
  def derived_mode_price(prices:, key:, modes:, context_tier:)
@@ -9,22 +9,43 @@ module LlmCostTracker
9
9
  module Matcher
10
10
  Match = Data.define(:source, :key, :prices, :matched_by)
11
11
 
12
+ CACHE_LIMIT = 2048
13
+ private_constant :CACHE_LIMIT
14
+
12
15
  class << self
13
16
  def lookup(provider:, model:)
14
17
  provider_name = provider.to_s.presence
15
18
  model_name = model.to_s
16
19
  return nil if model_name.empty?
17
20
 
18
- lookup_match(provider_name: provider_name, model_name: model_name)
21
+ sources = Registry.sources
22
+ reset_cache(sources) unless @cache_sources.equal?(sources)
23
+ key = [provider_name, model_name].freeze
24
+ return @cache[key] if @cache.key?(key)
25
+
26
+ @cache.clear if @cache.size >= CACHE_LIMIT
27
+ @cache[key] = lookup_match(sources, provider_name, model_name)
28
+ end
29
+
30
+ def modifier_priced?(provider:, model:, modifier:)
31
+ prices = lookup(provider: provider, model: model)&.prices
32
+ return false unless prices
33
+
34
+ prices.any? { |key, _| key.to_s.include?(modifier) }
19
35
  end
20
36
 
21
37
  private
22
38
 
23
- def lookup_match(provider_name:, model_name:)
39
+ def reset_cache(sources)
40
+ @cache_sources = sources
41
+ @cache = {}
42
+ end
43
+
44
+ def lookup_match(sources, provider_name, model_name)
24
45
  provider_model = provider_name ? "#{provider_name}/#{model_name}" : model_name
25
46
  normalized = normalize_model_name(model_name)
26
47
 
27
- Registry.sources.each do |source|
48
+ sources.each do |source|
28
49
  match = match_in_source(source, provider_model, model_name, normalized)
29
50
  return match if match
30
51
  end
@@ -4,8 +4,8 @@ module LlmCostTracker
4
4
  module Pricing
5
5
  module Mode
6
6
  STANDARD_MODE_VALUES = %w[auto default standard standard_only unspecified].freeze
7
- COMPOUND_MODIFIERS = %w[data_residency].freeze
8
7
  KNOWN_MODIFIERS = %w[batch flex priority scale fast on_demand data_residency].freeze
8
+ HOST_DERIVED_MODIFIERS = %w[data_residency].freeze
9
9
  MAX_PERMUTED_MODIFIERS = 6
10
10
 
11
11
  def self.normalize(value)
@@ -23,7 +23,7 @@ module LlmCostTracker
23
23
  return normalize(request_mode) if provider_mode.to_s.strip.empty?
24
24
 
25
25
  provider_tokens = tokenize(provider_mode) - STANDARD_MODE_VALUES
26
- request_host_tokens = tokenize(request_mode || "") & COMPOUND_MODIFIERS
26
+ request_host_tokens = tokenize(request_mode || "") & HOST_DERIVED_MODIFIERS
27
27
  combined = provider_tokens | request_host_tokens
28
28
  return nil if combined.empty?
29
29
 
@@ -41,12 +41,12 @@ module LlmCostTracker
41
41
  loop do
42
42
  break if remaining.empty?
43
43
 
44
- compound = COMPOUND_MODIFIERS.find do |token|
45
- remaining == token || remaining.start_with?("#{token}_")
44
+ known = KNOWN_MODIFIERS.find do |modifier|
45
+ remaining == modifier || remaining.start_with?("#{modifier}_")
46
46
  end
47
- if compound
48
- tokens << compound
49
- remaining = remaining.delete_prefix(compound).delete_prefix("_")
47
+ if known
48
+ tokens << known
49
+ remaining = remaining.delete_prefix(known).delete_prefix("_")
50
50
  else
51
51
  first, _, rest = remaining.partition("_")
52
52
  tokens << first unless first.empty?
@@ -9,8 +9,7 @@ module LlmCostTracker
9
9
  "per_1k_requests" => 1_000,
10
10
  "per_session" => 1,
11
11
  "per_hour" => 1,
12
- "per_minute" => 1,
13
- "per_image" => 1
12
+ "per_minute" => 1
14
13
  }.freeze
15
14
 
16
15
  Rate = Data.define(:amount, :quantity, :currency, :source, :source_key, :source_version)
@@ -101,7 +101,7 @@ module LlmCostTracker
101
101
  end
102
102
 
103
103
  def prices_file_mtime_iso
104
- path = LlmCostTracker.configuration.prices_file
104
+ path = LlmCostTracker.configuration.pricing.file
105
105
  return nil unless path && File.exist?(path)
106
106
 
107
107
  @prices_file_mtime_iso ||= File.mtime(path).utc.iso8601
@@ -113,16 +113,16 @@ module LlmCostTracker
113
113
  [
114
114
  Source.new(
115
115
  name: "pricing_overrides",
116
- prices: config.pricing_overrides,
116
+ prices: config.pricing.overrides,
117
117
  rates: {},
118
118
  currency: upcased_currency(nil),
119
119
  version: "configuration"
120
120
  ),
121
121
  Source.new(
122
122
  name: "prices_file",
123
- prices: file_prices(config.prices_file),
124
- rates: file_rates(config.prices_file),
125
- currency: upcased_currency(file_metadata(config.prices_file)["currency"]),
123
+ prices: file_prices(config.pricing.file),
124
+ rates: file_rates(config.pricing.file),
125
+ currency: upcased_currency(file_metadata(config.pricing.file)["currency"]),
126
126
  version: prices_file_mtime_iso
127
127
  ),
128
128
  Source.new(
@@ -26,7 +26,7 @@ module LlmCostTracker
26
26
  output = env["OUTPUT"].to_s.strip.presence
27
27
  return output if output
28
28
 
29
- prices_file = config.prices_file
29
+ prices_file = config.pricing.file
30
30
  return prices_file.to_s if prices_file
31
31
 
32
32
  Rails.root.join(DEFAULT_OUTPUT_PATH).to_s
@@ -9,14 +9,14 @@ module LlmCostTracker
9
9
  WARN_CACHE_LIMIT = 1024
10
10
 
11
11
  class << self
12
- def process(model)
12
+ def process(model, pricing_mode: nil)
13
13
  model = model.to_s.presence || Event::UNKNOWN_MODEL
14
14
 
15
- case LlmCostTracker.configuration.unknown_pricing_behavior
15
+ case LlmCostTracker.configuration.pricing.unknown_model_behavior
16
16
  when :ignore
17
17
  nil
18
18
  when :warn
19
- warn_missing(model)
19
+ warn_missing(model, pricing_mode.to_s.presence)
20
20
  when :raise
21
21
  raise UnknownPricingError.new(model: model)
22
22
  end
@@ -24,19 +24,22 @@ module LlmCostTracker
24
24
 
25
25
  private
26
26
 
27
- def warn_missing(model)
27
+ def warn_missing(model, pricing_mode)
28
+ key = [model, pricing_mode].freeze
28
29
  should_warn = MUTEX.synchronize do
29
30
  @warned_models ||= Set.new
30
- next false if @warned_models.size >= WARN_CACHE_LIMIT && !@warned_models.include?(model)
31
+ next false if @warned_models.size >= WARN_CACHE_LIMIT && !@warned_models.include?(key)
31
32
 
32
- @warned_models.add?(model)
33
+ @warned_models.add?(key)
33
34
  end
34
35
  return unless should_warn
35
36
 
37
+ subject = "model #{model.inspect}"
38
+ subject += " at pricing_mode #{pricing_mode.inspect}" if pricing_mode
36
39
  Logging.warn(
37
- "No pricing configured for model #{model.inspect}. " \
40
+ "No pricing configured for #{subject}. " \
38
41
  "Cost and budget guardrails will be skipped for this event. " \
39
- "Add a pricing_overrides entry or set unknown_pricing_behavior."
42
+ "Add a pricing.overrides entry or set pricing.unknown_model_behavior."
40
43
  )
41
44
  end
42
45
  end
@@ -22,12 +22,13 @@ module LlmCostTracker
22
22
  output_tokens: output,
23
23
  cache_read_input_tokens: cache_read,
24
24
  cache_write_input_tokens: cache_write,
25
- cache_write_extended_input_tokens: cache_write_extended
25
+ cache_write_extended_input_tokens: cache_write_extended,
26
+ hidden_output_tokens: usage.dig(:output_tokens_details, :thinking_tokens).to_i
26
27
  )
27
28
  end
28
29
 
29
30
  def self.pricing_mode(request:, usage:)
30
- speed = request&.dig(:speed)
31
+ speed = usage&.dig(:speed) || request&.dig(:speed)
31
32
  service_tier = usage&.dig(:service_tier) || request&.dig(:service_tier)
32
33
  geo = (usage&.dig(:inference_geo) || request&.dig(:inference_geo)).to_s.downcase
33
34
 
@@ -4,9 +4,6 @@ module LlmCostTracker
4
4
  module Providers
5
5
  module Openai
6
6
  module ModelFamilies
7
- DATA_RESIDENCY_MODEL_PATTERN =
8
- /\Agpt-5\.(?:4|5)(?:-(?:mini|nano|pro|codex(?:-mini|-max)?))?(?:-\d{4}-\d{2}-\d{2})?\z/
9
-
10
7
  IMAGE_OUTPUT_MODEL_PATTERN = /\Agpt-image-/i
11
8
 
12
9
  CHARACTER_BILLED_TTS_MODEL_PATTERN = /\Atts-1(-hd)?\z/
@@ -19,10 +16,6 @@ module LlmCostTracker
19
16
  NON_REASONING_GPT5_PATTERN = /\Agpt-5(?:\.\d+)?-chat\b/i
20
17
 
21
18
  CHAT_COMPLETIONS_SEARCH_MODEL_PATTERN = /-search-(?:preview|api)\b/i
22
- def self.data_residency?(model)
23
- model.to_s.match?(DATA_RESIDENCY_MODEL_PATTERN)
24
- end
25
-
26
19
  def self.image_output?(model)
27
20
  model.to_s.match?(IMAGE_OUTPUT_MODEL_PATTERN)
28
21
  end
@@ -16,7 +16,10 @@ module LlmCostTracker
16
16
  class << self
17
17
  def combined_pricing_mode(host:, model:, service_tier:)
18
18
  modes = [Pricing::Mode.normalize(service_tier)]
19
- modes << "data_residency" if Hosts.data_residency?(host) && ModelFamilies.data_residency?(model)
19
+ if Hosts.data_residency?(host) &&
20
+ Pricing::Matcher.modifier_priced?(provider: "openai", model: model, modifier: "data_residency")
21
+ modes << "data_residency"
22
+ end
20
23
  Pricing::Mode.compose(modes)
21
24
  end
22
25
 
@@ -26,7 +29,7 @@ module LlmCostTracker
26
29
 
27
30
  model = response["model"] || request["model"]
28
31
  service_line_items =
29
- ServiceCharges.service_line_items_for(response, request: request, model: response["model"]) +
32
+ ServiceCharges.service_line_items_for(response, request: request, model: model) +
30
33
  ServiceCharges.transcription_line_items(usage)
31
34
  Event.build(
32
35
  provider: provider,
@@ -12,6 +12,7 @@ module LlmCostTracker
12
12
  input_tokens = (usage[:input_tokens] || usage[:prompt_tokens]).to_i
13
13
  output_tokens = (usage[:output_tokens] || usage[:completion_tokens]).to_i
14
14
  cache_read = cache_read_input_tokens(usage)
15
+ cache_write = cache_write_input_tokens(usage)
15
16
  audio_input = audio_input_tokens(usage)
16
17
  audio_output = audio_output_tokens(usage)
17
18
  image_input = image_input_tokens(usage)
@@ -24,10 +25,11 @@ module LlmCostTracker
24
25
  )
25
26
 
26
27
  Usage::TokenUsage.build(
27
- input_tokens: [input_tokens - cache_read - audio_input - image_input, 0].max,
28
+ input_tokens: [input_tokens - cache_read - cache_write - audio_input - image_input, 0].max,
28
29
  output_tokens: regular_output,
29
30
  total_tokens: usage[:total_tokens],
30
31
  cache_read_input_tokens: cache_read,
32
+ cache_write_input_tokens: cache_write,
31
33
  audio_input_tokens: audio_input,
32
34
  audio_output_tokens: audio_output,
33
35
  image_input_tokens: image_input,
@@ -46,12 +48,11 @@ module LlmCostTracker
46
48
  return default_to_image ? [remainder, 0] : [0, remainder]
47
49
  end
48
50
 
49
- text_output = text_output_details
50
- text_output = [output_tokens - image_output_details - audio_output, 0].max if text_output.zero?
51
- [image_output_details, text_output]
51
+ [image_output_details, [output_tokens - image_output_details - audio_output, 0].max]
52
52
  end
53
53
 
54
54
  def self.cache_read_input_tokens(usage) = detail(usage, INPUT_DETAIL_KEYS, :cached_tokens)
55
+ def self.cache_write_input_tokens(usage) = detail(usage, INPUT_DETAIL_KEYS, :cache_write_tokens)
55
56
  def self.hidden_output_tokens(usage) = detail(usage, OUTPUT_DETAIL_KEYS, :reasoning_tokens)
56
57
  def self.audio_input_tokens(usage) = detail(usage, INPUT_DETAIL_KEYS, :audio_tokens)
57
58
  def self.audio_output_tokens(usage) = detail(usage, OUTPUT_DETAIL_KEYS, :audio_tokens)
@@ -14,7 +14,7 @@ module LlmCostTracker
14
14
  end
15
15
 
16
16
  def provider_names
17
- custom = LlmCostTracker.configuration.openai_compatible_providers.each_value.map do |provider|
17
+ custom = LlmCostTracker.configuration.capture.openai_compatible_providers.each_value.map do |provider|
18
18
  provider.to_s.downcase
19
19
  end
20
20
  ["openai_compatible", *custom].uniq
@@ -23,7 +23,7 @@ module LlmCostTracker
23
23
  def provider_for_uri(uri)
24
24
  return nil unless uri
25
25
 
26
- LlmCostTracker.configuration.openai_compatible_providers[uri.host.to_s.downcase]&.to_s
26
+ LlmCostTracker.configuration.capture.openai_compatible_providers[uri.host.to_s.downcase]&.to_s
27
27
  end
28
28
  end
29
29
 
@@ -8,14 +8,10 @@ module LlmCostTracker
8
8
  app.config.eager_load_paths << models_path unless app.config.eager_load_paths.include?(models_path)
9
9
  end
10
10
 
11
+ GENERATOR_FILES = File.expand_path("generators/llm_cost_tracker/*_generator.rb", __dir__)
12
+
11
13
  generators do
12
- require_relative "generators/llm_cost_tracker/install_generator"
13
- require_relative "generators/llm_cost_tracker/prices_generator"
14
- require_relative "generators/llm_cost_tracker/call_rollups_generator"
15
- require_relative "generators/llm_cost_tracker/async_ingestion_generator"
16
- require_relative "generators/llm_cost_tracker/upgrade_call_rollups_provider_generator"
17
- require_relative "generators/llm_cost_tracker/upgrade_image_tokens_generator"
18
- require_relative "generators/llm_cost_tracker/upgrade_call_tags_key_value_index_generator"
14
+ Dir[GENERATOR_FILES].each { |path| require path }
19
15
  end
20
16
  end
21
17
  end
@@ -34,7 +34,7 @@ module LlmCostTracker
34
34
  end
35
35
  from = now - days.days
36
36
  scope = LlmCostTracker::Call.where(tracked_at: from..now)
37
- tag_breakdowns ||= LlmCostTracker.configuration.report_tag_breakdowns || []
37
+ tag_breakdowns ||= LlmCostTracker.configuration.tags.report_breakdown_keys || []
38
38
  aggregate = totals(scope)
39
39
 
40
40
  new(
@@ -55,7 +55,7 @@ module LlmCostTracker
55
55
  def self.totals(scope)
56
56
  scope
57
57
  .select(
58
- "COALESCE(SUM(total_cost), 0) AS total_cost, " \
58
+ "COALESCE(SUM(#{LlmCostTracker::Call.qualified(:total_cost)}), 0) AS total_cost, " \
59
59
  "COUNT(*) AS requests_count, " \
60
60
  "AVG(latency_ms) AS average_latency_ms, " \
61
61
  "COALESCE(SUM(CASE WHEN #{Charges::CostStatus.unknown_pricing_sql} " \
@@ -3,6 +3,8 @@
3
3
  module LlmCostTracker
4
4
  module Retention
5
5
  DEFAULT_BATCH_SIZE = 5_000
6
+ ROLLUP_COLUMNS = %i[tracked_at total_cost pricing_snapshot provider].freeze
7
+ private_constant :ROLLUP_COLUMNS
6
8
 
7
9
  class << self
8
10
  def prune(older_than:, batch_size: DEFAULT_BATCH_SIZE, now: Time.now.utc)
@@ -26,11 +28,23 @@ module LlmCostTracker
26
28
  require_relative "ingestion"
27
29
  return 0 unless LlmCostTracker::Ingestion::InboxEntry.table_exists?
28
30
 
29
- LlmCostTracker::Ingestion::InboxEntry.where(tracked_at: ...cutoff).delete_all
31
+ scope = LlmCostTracker::Ingestion::InboxEntry.where(tracked_at: ...cutoff)
32
+ warn_undrained(scope.pending)
33
+ scope.delete_all
30
34
  end
31
35
 
32
36
  private
33
37
 
38
+ def warn_undrained(scope)
39
+ count, cost = scope.pick(Arel.sql("COUNT(*)"), Arel.sql("COALESCE(SUM(total_cost), 0)"))
40
+ return if count.to_i.zero?
41
+
42
+ Logging.warn(
43
+ "Retention.prune_inbox is deleting #{count} inbox row(s) worth #{cost} that were never drained " \
44
+ "into the ledger. Their spend is lost. Drain the inbox before pruning, or raise the retention window."
45
+ )
46
+ end
47
+
34
48
  def resolve_cutoff(older_than, now)
35
49
  cutoff = case older_than
36
50
  when Time, DateTime then older_than.utc
@@ -58,22 +72,20 @@ module LlmCostTracker
58
72
 
59
73
  def prune_batch(cutoff, batch_size)
60
74
  LlmCostTracker::Call.transaction do
61
- cache_rollups = LlmCostTracker.configuration.cache_rollups
62
- rows = prunable_rows(cutoff, batch_size, with_rollup_columns: cache_rollups)
75
+ rows = prunable_rows(cutoff, batch_size)
63
76
  next 0 if rows.empty?
64
77
 
65
- ids = cache_rollups ? rows.map(&:id) : rows
66
- deleted = LlmCostTracker::Call.where(id: ids).delete_all
67
- LlmCostTracker::Ledger::Rollups.decrement!(rows) if cache_rollups && deleted.positive?
78
+ deleted = LlmCostTracker::Call.where(id: rows.map(&:id)).delete_all
79
+ LlmCostTracker::Ledger::Rollups.decrement!(rows) if deleted.positive?
68
80
  deleted
69
81
  end
70
82
  end
71
83
 
72
- def prunable_rows(cutoff, batch_size, with_rollup_columns:)
84
+ def prunable_rows(cutoff, batch_size)
73
85
  relation = LlmCostTracker::Call.where(tracked_at: ...cutoff).order(:id).limit(batch_size).lock
74
- return relation.pluck(:id) unless with_rollup_columns
75
-
76
- relation.select(:id, :tracked_at, :total_cost, :pricing_snapshot, :provider).to_a
86
+ columns = [:id]
87
+ columns += ROLLUP_COLUMNS if LlmCostTracker::Ledger::Rollups.cache_active?
88
+ relation.select(*columns).to_a
77
89
  end
78
90
  end
79
91
  end
@@ -18,15 +18,15 @@ module LlmCostTracker
18
18
 
19
19
  def tags
20
20
  config = LlmCostTracker.configuration
21
- base = config.static_sanitized_default_tags ||
22
- Sanitizer.call(call_default_tags(config.default_tags).to_h)
21
+ base = config.tags.static_sanitized_default ||
22
+ Sanitizer.call(call_default_tags(config.tags.default).to_h)
23
23
  base.merge(*Array(ActiveSupport::IsolatedExecutionState[KEY]))
24
24
  end
25
25
 
26
26
  def call_default_tags(proc_or_lambda)
27
27
  proc_or_lambda.call
28
28
  rescue StandardError => e
29
- Logging.warn("LlmCostTracker default_tags proc raised: #{e.class}: #{e.message}; using empty default tags")
29
+ Logging.warn("LlmCostTracker tags.default proc raised: #{e.class}: #{e.message}; using empty default tags")
30
30
  {}
31
31
  end
32
32
  end