llm_cost_tracker 0.13.0 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (81) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +53 -16
  3. data/README.md +16 -28
  4. data/app/controllers/llm_cost_tracker/application_controller.rb +3 -2
  5. data/app/controllers/llm_cost_tracker/data_quality_controller.rb +1 -0
  6. data/app/controllers/llm_cost_tracker/models_controller.rb +3 -1
  7. data/app/helpers/llm_cost_tracker/dashboard_query_helper.rb +11 -0
  8. data/app/models/llm_cost_tracker/call.rb +9 -3
  9. data/app/models/llm_cost_tracker/call_rollup.rb +19 -3
  10. data/app/services/llm_cost_tracker/dashboard/data_quality.rb +13 -0
  11. data/app/services/llm_cost_tracker/dashboard/monthly_budget.rb +1 -1
  12. data/app/services/llm_cost_tracker/dashboard/pagination.rb +2 -1
  13. data/app/services/llm_cost_tracker/dashboard/pricing_overview.rb +3 -3
  14. data/app/services/llm_cost_tracker/dashboard/setup_state.rb +5 -3
  15. data/app/views/llm_cost_tracker/calls/show.html.erb +6 -8
  16. data/app/views/llm_cost_tracker/data_quality/index.html.erb +11 -0
  17. data/app/views/llm_cost_tracker/shared/_filter_pill_date.html.erb +1 -3
  18. data/app/views/llm_cost_tracker/shared/_filter_pill_model.html.erb +1 -3
  19. data/app/views/llm_cost_tracker/shared/_filter_pill_provider.html.erb +1 -3
  20. data/app/views/llm_cost_tracker/shared/_filter_pill_stream.html.erb +1 -3
  21. data/lib/llm_cost_tracker/budget/per_tag.rb +159 -0
  22. data/lib/llm_cost_tracker/budget.rb +120 -32
  23. data/lib/llm_cost_tracker/charges/cost_status.rb +4 -3
  24. data/lib/llm_cost_tracker/configuration/budgets.rb +93 -0
  25. data/lib/llm_cost_tracker/configuration/capture.rb +42 -0
  26. data/lib/llm_cost_tracker/configuration/ingestion.rb +20 -0
  27. data/lib/llm_cost_tracker/configuration/mutability.rb +33 -0
  28. data/lib/llm_cost_tracker/configuration/pricing.rb +36 -0
  29. data/lib/llm_cost_tracker/configuration/section.rb +59 -0
  30. data/lib/llm_cost_tracker/configuration/tags.rb +52 -0
  31. data/lib/llm_cost_tracker/configuration.rb +69 -125
  32. data/lib/llm_cost_tracker/deprecator.rb +9 -0
  33. data/lib/llm_cost_tracker/doctor/ingestion_check.rb +17 -8
  34. data/lib/llm_cost_tracker/doctor/price_check.rb +1 -1
  35. data/lib/llm_cost_tracker/doctor.rb +4 -4
  36. data/lib/llm_cost_tracker/engine.rb +4 -0
  37. data/lib/llm_cost_tracker/errors.rb +12 -3
  38. data/lib/llm_cost_tracker/generators/llm_cost_tracker/async_ingestion_generator.rb +2 -2
  39. data/lib/llm_cost_tracker/generators/llm_cost_tracker/call_rollups_generator.rb +2 -2
  40. data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/create_llm_cost_tracker_async_ingestion.rb.erb +0 -1
  41. data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/create_llm_cost_tracker_calls.rb.erb +8 -4
  42. data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/initializer.rb.erb +50 -30
  43. data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/upgrade_indexes.rb.erb +40 -0
  44. data/lib/llm_cost_tracker/generators/llm_cost_tracker/templates/upgrade_per_tag_budgets.rb.erb +41 -0
  45. data/lib/llm_cost_tracker/generators/llm_cost_tracker/upgrade_indexes_generator.rb +30 -0
  46. data/lib/llm_cost_tracker/generators/llm_cost_tracker/upgrade_per_tag_budgets_generator.rb +30 -0
  47. data/lib/llm_cost_tracker/ingestion/batch.rb +3 -1
  48. data/lib/llm_cost_tracker/ingestion/pool.rb +9 -2
  49. data/lib/llm_cost_tracker/ingestion.rb +3 -7
  50. data/lib/llm_cost_tracker/integrations/openai/batch_capture.rb +10 -12
  51. data/lib/llm_cost_tracker/ledger/period/totals.rb +1 -1
  52. data/lib/llm_cost_tracker/ledger/rollups.rb +34 -4
  53. data/lib/llm_cost_tracker/ledger/store.rb +9 -2
  54. data/lib/llm_cost_tracker/ledger/tags/encoding.rb +15 -5
  55. data/lib/llm_cost_tracker/logging.rb +3 -5
  56. data/lib/llm_cost_tracker/middleware/faraday.rb +1 -1
  57. data/lib/llm_cost_tracker/prices.json +1291 -313
  58. data/lib/llm_cost_tracker/pricing/backfill.rb +11 -1
  59. data/lib/llm_cost_tracker/pricing/calculation.rb +4 -4
  60. data/lib/llm_cost_tracker/pricing/effective_prices.rb +26 -15
  61. data/lib/llm_cost_tracker/pricing/matcher.rb +7 -0
  62. data/lib/llm_cost_tracker/pricing/rate.rb +1 -2
  63. data/lib/llm_cost_tracker/pricing/registry.rb +5 -5
  64. data/lib/llm_cost_tracker/pricing/sync.rb +1 -1
  65. data/lib/llm_cost_tracker/pricing/unknown.rb +11 -8
  66. data/lib/llm_cost_tracker/providers/anthropic/usage_extractor.rb +3 -2
  67. data/lib/llm_cost_tracker/providers/openai/model_families.rb +0 -7
  68. data/lib/llm_cost_tracker/providers/openai/response_parser.rb +5 -2
  69. data/lib/llm_cost_tracker/providers/openai/usage_extractor.rb +5 -4
  70. data/lib/llm_cost_tracker/providers/openai_compatible/parser.rb +2 -2
  71. data/lib/llm_cost_tracker/railtie.rb +3 -7
  72. data/lib/llm_cost_tracker/report/data.rb +2 -2
  73. data/lib/llm_cost_tracker/retention.rb +22 -10
  74. data/lib/llm_cost_tracker/tags/context.rb +3 -3
  75. data/lib/llm_cost_tracker/tags/sanitizer.rb +4 -4
  76. data/lib/llm_cost_tracker/tracker.rb +14 -23
  77. data/lib/llm_cost_tracker/usage/catalog.rb +1 -2
  78. data/lib/llm_cost_tracker/version.rb +1 -1
  79. data/lib/llm_cost_tracker.rb +6 -1
  80. data/lib/tasks/llm_cost_tracker.rake +23 -10
  81. metadata +22 -9
@@ -4,6 +4,7 @@ require_relative "../pricing"
4
4
  require_relative "../charges/line_item"
5
5
  require_relative "../ledger/rollups"
6
6
  require_relative "../usage/token_usage"
7
+ require_relative "../budget/per_tag"
7
8
 
8
9
  module LlmCostTracker
9
10
  module Pricing
@@ -30,7 +31,7 @@ module LlmCostTracker
30
31
  rollup_events << rollup_event_for(call, calculation)
31
32
  recomputed += 1
32
33
  end
33
- Ledger::Rollups.increment!(rollup_events) if rollup_events.any?
34
+ Ledger::Rollups.increment!(rollup_events)
34
35
  end
35
36
  end
36
37
 
@@ -61,6 +62,7 @@ module LlmCostTracker
61
62
  pricing_snapshot: calculation.snapshot,
62
63
  cost_status: calculation.cost_status
63
64
  )
65
+ resync_tag_costs(call, calculation.cost.total)
64
66
  token_priced = calculation.priced_line_items.select(&:token?).index_by { |item| dimension_key(item) }
65
67
  service_priced = calculation.priced_line_items.reject(&:token?)
66
68
  token_records, service_records = call.line_items.partition { |record| record.unit == "token" }
@@ -69,6 +71,14 @@ module LlmCostTracker
69
71
  service_records.sort_by(&:position).zip(service_priced).each { |record, priced| apply_rate(record, priced) }
70
72
  end
71
73
 
74
+ def resync_tag_costs(call, total_cost)
75
+ return unless LlmCostTracker::Budget::PerTag.columns?
76
+
77
+ LlmCostTracker::CallTag
78
+ .where(llm_cost_tracker_call_id: call.id)
79
+ .update_all(total_cost: total_cost)
80
+ end
81
+
72
82
  def apply_rate(record, priced)
73
83
  return unless priced
74
84
 
@@ -187,16 +187,16 @@ module LlmCostTracker
187
187
  price = effective[dimension.key]
188
188
  return line_item.with(cost_status: Charges::CostStatus::UNKNOWN) if price.nil?
189
189
 
190
- line_item.with_rate(token_rate(dimension, price))
190
+ line_item.with_rate(token_rate(price))
191
191
  end
192
192
 
193
- def token_rate(dimension, price)
193
+ def token_rate(price)
194
194
  Pricing::Rate.new(
195
- amount: price.to_d,
195
+ amount: price.amount.to_d,
196
196
  quantity: RATE_DENOMINATOR_TOKENS.to_d,
197
197
  currency: match.source.currency,
198
198
  source: match.source.name,
199
- source_key: dimension.key,
199
+ source_key: price.key,
200
200
  source_version: match.source.version
201
201
  )
202
202
  end
@@ -8,38 +8,49 @@ require_relative "price_key"
8
8
  module LlmCostTracker
9
9
  module Pricing
10
10
  module EffectivePrices
11
+ Resolved = Data.define(:amount, :key)
12
+
11
13
  class << self
12
14
  def call(usage:, quantities:, prices:, pricing_mode:)
13
15
  context_tier = context_tier?(usage: usage, prices: prices)
14
16
  orderings = pricing_mode && Mode.permutations_for(pricing_mode)
15
17
 
16
18
  quantities.to_h do |price_key, tokens|
17
- price = if tokens.positive?
18
- price_for(
19
- prices: prices,
20
- key: price_key,
21
- orderings: orderings,
22
- context_tier: context_tier
23
- )
24
- else
25
- BigDecimal("0")
26
- end
27
- [price_key, price]
19
+ resolved = if tokens.positive?
20
+ price_for(
21
+ prices: prices,
22
+ key: price_key,
23
+ orderings: orderings,
24
+ context_tier: context_tier
25
+ )
26
+ else
27
+ Resolved.new(amount: BigDecimal("0"), key: price_key)
28
+ end
29
+ [price_key, resolved]
28
30
  end
29
31
  end
30
32
 
31
33
  private
32
34
 
33
35
  def price_for(prices:, key:, orderings:, context_tier:)
34
- return prices[PriceKey.build(key, above_context: context_tier)] unless orderings
36
+ unless orderings
37
+ standard = PriceKey.build(key, above_context: context_tier)
38
+ return resolve(prices[standard], standard)
39
+ end
35
40
 
36
41
  orderings.each do |mode|
37
- direct = prices[PriceKey.build(key, mode: mode, above_context: context_tier)]
38
- return direct if direct
42
+ table_key = PriceKey.build(key, mode: mode, above_context: context_tier)
43
+ direct = prices[table_key]
44
+ return resolve(direct, table_key) if direct
39
45
  end
40
46
  return nil if %w[input output].include?(key)
41
47
 
42
- derived_mode_price(prices: prices, key: key, modes: orderings, context_tier: context_tier)
48
+ derived = derived_mode_price(prices: prices, key: key, modes: orderings, context_tier: context_tier)
49
+ resolve(derived, nil)
50
+ end
51
+
52
+ def resolve(amount, key)
53
+ amount && Resolved.new(amount: amount, key: key)
43
54
  end
44
55
 
45
56
  def derived_mode_price(prices:, key:, modes:, context_tier:)
@@ -27,6 +27,13 @@ module LlmCostTracker
27
27
  @cache[key] = lookup_match(sources, provider_name, model_name)
28
28
  end
29
29
 
30
+ def modifier_priced?(provider:, model:, modifier:)
31
+ prices = lookup(provider: provider, model: model)&.prices
32
+ return false unless prices
33
+
34
+ prices.any? { |key, _| key.to_s.include?(modifier) }
35
+ end
36
+
30
37
  private
31
38
 
32
39
  def reset_cache(sources)
@@ -9,8 +9,7 @@ module LlmCostTracker
9
9
  "per_1k_requests" => 1_000,
10
10
  "per_session" => 1,
11
11
  "per_hour" => 1,
12
- "per_minute" => 1,
13
- "per_image" => 1
12
+ "per_minute" => 1
14
13
  }.freeze
15
14
 
16
15
  Rate = Data.define(:amount, :quantity, :currency, :source, :source_key, :source_version)
@@ -101,7 +101,7 @@ module LlmCostTracker
101
101
  end
102
102
 
103
103
  def prices_file_mtime_iso
104
- path = LlmCostTracker.configuration.prices_file
104
+ path = LlmCostTracker.configuration.pricing.file
105
105
  return nil unless path && File.exist?(path)
106
106
 
107
107
  @prices_file_mtime_iso ||= File.mtime(path).utc.iso8601
@@ -113,16 +113,16 @@ module LlmCostTracker
113
113
  [
114
114
  Source.new(
115
115
  name: "pricing_overrides",
116
- prices: config.pricing_overrides,
116
+ prices: config.pricing.overrides,
117
117
  rates: {},
118
118
  currency: upcased_currency(nil),
119
119
  version: "configuration"
120
120
  ),
121
121
  Source.new(
122
122
  name: "prices_file",
123
- prices: file_prices(config.prices_file),
124
- rates: file_rates(config.prices_file),
125
- currency: upcased_currency(file_metadata(config.prices_file)["currency"]),
123
+ prices: file_prices(config.pricing.file),
124
+ rates: file_rates(config.pricing.file),
125
+ currency: upcased_currency(file_metadata(config.pricing.file)["currency"]),
126
126
  version: prices_file_mtime_iso
127
127
  ),
128
128
  Source.new(
@@ -26,7 +26,7 @@ module LlmCostTracker
26
26
  output = env["OUTPUT"].to_s.strip.presence
27
27
  return output if output
28
28
 
29
- prices_file = config.prices_file
29
+ prices_file = config.pricing.file
30
30
  return prices_file.to_s if prices_file
31
31
 
32
32
  Rails.root.join(DEFAULT_OUTPUT_PATH).to_s
@@ -9,14 +9,14 @@ module LlmCostTracker
9
9
  WARN_CACHE_LIMIT = 1024
10
10
 
11
11
  class << self
12
- def process(model)
12
+ def process(model, pricing_mode: nil)
13
13
  model = model.to_s.presence || Event::UNKNOWN_MODEL
14
14
 
15
- case LlmCostTracker.configuration.unknown_pricing_behavior
15
+ case LlmCostTracker.configuration.pricing.unknown_model_behavior
16
16
  when :ignore
17
17
  nil
18
18
  when :warn
19
- warn_missing(model)
19
+ warn_missing(model, pricing_mode.to_s.presence)
20
20
  when :raise
21
21
  raise UnknownPricingError.new(model: model)
22
22
  end
@@ -24,19 +24,22 @@ module LlmCostTracker
24
24
 
25
25
  private
26
26
 
27
- def warn_missing(model)
27
+ def warn_missing(model, pricing_mode)
28
+ key = [model, pricing_mode].freeze
28
29
  should_warn = MUTEX.synchronize do
29
30
  @warned_models ||= Set.new
30
- next false if @warned_models.size >= WARN_CACHE_LIMIT && !@warned_models.include?(model)
31
+ next false if @warned_models.size >= WARN_CACHE_LIMIT && !@warned_models.include?(key)
31
32
 
32
- @warned_models.add?(model)
33
+ @warned_models.add?(key)
33
34
  end
34
35
  return unless should_warn
35
36
 
37
+ subject = "model #{model.inspect}"
38
+ subject += " at pricing_mode #{pricing_mode.inspect}" if pricing_mode
36
39
  Logging.warn(
37
- "No pricing configured for model #{model.inspect}. " \
40
+ "No pricing configured for #{subject}. " \
38
41
  "Cost and budget guardrails will be skipped for this event. " \
39
- "Add a pricing_overrides entry or set unknown_pricing_behavior."
42
+ "Add a pricing.overrides entry or set pricing.unknown_model_behavior."
40
43
  )
41
44
  end
42
45
  end
@@ -22,12 +22,13 @@ module LlmCostTracker
22
22
  output_tokens: output,
23
23
  cache_read_input_tokens: cache_read,
24
24
  cache_write_input_tokens: cache_write,
25
- cache_write_extended_input_tokens: cache_write_extended
25
+ cache_write_extended_input_tokens: cache_write_extended,
26
+ hidden_output_tokens: usage.dig(:output_tokens_details, :thinking_tokens).to_i
26
27
  )
27
28
  end
28
29
 
29
30
  def self.pricing_mode(request:, usage:)
30
- speed = request&.dig(:speed)
31
+ speed = usage&.dig(:speed) || request&.dig(:speed)
31
32
  service_tier = usage&.dig(:service_tier) || request&.dig(:service_tier)
32
33
  geo = (usage&.dig(:inference_geo) || request&.dig(:inference_geo)).to_s.downcase
33
34
 
@@ -4,9 +4,6 @@ module LlmCostTracker
4
4
  module Providers
5
5
  module Openai
6
6
  module ModelFamilies
7
- DATA_RESIDENCY_MODEL_PATTERN =
8
- /\Agpt-5\.(?:4|5)(?:-(?:mini|nano|pro|codex(?:-mini|-max)?))?(?:-\d{4}-\d{2}-\d{2})?\z/
9
-
10
7
  IMAGE_OUTPUT_MODEL_PATTERN = /\Agpt-image-/i
11
8
 
12
9
  CHARACTER_BILLED_TTS_MODEL_PATTERN = /\Atts-1(-hd)?\z/
@@ -19,10 +16,6 @@ module LlmCostTracker
19
16
  NON_REASONING_GPT5_PATTERN = /\Agpt-5(?:\.\d+)?-chat\b/i
20
17
 
21
18
  CHAT_COMPLETIONS_SEARCH_MODEL_PATTERN = /-search-(?:preview|api)\b/i
22
- def self.data_residency?(model)
23
- model.to_s.match?(DATA_RESIDENCY_MODEL_PATTERN)
24
- end
25
-
26
19
  def self.image_output?(model)
27
20
  model.to_s.match?(IMAGE_OUTPUT_MODEL_PATTERN)
28
21
  end
@@ -16,7 +16,10 @@ module LlmCostTracker
16
16
  class << self
17
17
  def combined_pricing_mode(host:, model:, service_tier:)
18
18
  modes = [Pricing::Mode.normalize(service_tier)]
19
- modes << "data_residency" if Hosts.data_residency?(host) && ModelFamilies.data_residency?(model)
19
+ if Hosts.data_residency?(host) &&
20
+ Pricing::Matcher.modifier_priced?(provider: "openai", model: model, modifier: "data_residency")
21
+ modes << "data_residency"
22
+ end
20
23
  Pricing::Mode.compose(modes)
21
24
  end
22
25
 
@@ -26,7 +29,7 @@ module LlmCostTracker
26
29
 
27
30
  model = response["model"] || request["model"]
28
31
  service_line_items =
29
- ServiceCharges.service_line_items_for(response, request: request, model: response["model"]) +
32
+ ServiceCharges.service_line_items_for(response, request: request, model: model) +
30
33
  ServiceCharges.transcription_line_items(usage)
31
34
  Event.build(
32
35
  provider: provider,
@@ -12,6 +12,7 @@ module LlmCostTracker
12
12
  input_tokens = (usage[:input_tokens] || usage[:prompt_tokens]).to_i
13
13
  output_tokens = (usage[:output_tokens] || usage[:completion_tokens]).to_i
14
14
  cache_read = cache_read_input_tokens(usage)
15
+ cache_write = cache_write_input_tokens(usage)
15
16
  audio_input = audio_input_tokens(usage)
16
17
  audio_output = audio_output_tokens(usage)
17
18
  image_input = image_input_tokens(usage)
@@ -24,10 +25,11 @@ module LlmCostTracker
24
25
  )
25
26
 
26
27
  Usage::TokenUsage.build(
27
- input_tokens: [input_tokens - cache_read - audio_input - image_input, 0].max,
28
+ input_tokens: [input_tokens - cache_read - cache_write - audio_input - image_input, 0].max,
28
29
  output_tokens: regular_output,
29
30
  total_tokens: usage[:total_tokens],
30
31
  cache_read_input_tokens: cache_read,
32
+ cache_write_input_tokens: cache_write,
31
33
  audio_input_tokens: audio_input,
32
34
  audio_output_tokens: audio_output,
33
35
  image_input_tokens: image_input,
@@ -46,12 +48,11 @@ module LlmCostTracker
46
48
  return default_to_image ? [remainder, 0] : [0, remainder]
47
49
  end
48
50
 
49
- text_output = text_output_details
50
- text_output = [output_tokens - image_output_details - audio_output, 0].max if text_output.zero?
51
- [image_output_details, text_output]
51
+ [image_output_details, [output_tokens - image_output_details - audio_output, 0].max]
52
52
  end
53
53
 
54
54
  def self.cache_read_input_tokens(usage) = detail(usage, INPUT_DETAIL_KEYS, :cached_tokens)
55
+ def self.cache_write_input_tokens(usage) = detail(usage, INPUT_DETAIL_KEYS, :cache_write_tokens)
55
56
  def self.hidden_output_tokens(usage) = detail(usage, OUTPUT_DETAIL_KEYS, :reasoning_tokens)
56
57
  def self.audio_input_tokens(usage) = detail(usage, INPUT_DETAIL_KEYS, :audio_tokens)
57
58
  def self.audio_output_tokens(usage) = detail(usage, OUTPUT_DETAIL_KEYS, :audio_tokens)
@@ -14,7 +14,7 @@ module LlmCostTracker
14
14
  end
15
15
 
16
16
  def provider_names
17
- custom = LlmCostTracker.configuration.openai_compatible_providers.each_value.map do |provider|
17
+ custom = LlmCostTracker.configuration.capture.openai_compatible_providers.each_value.map do |provider|
18
18
  provider.to_s.downcase
19
19
  end
20
20
  ["openai_compatible", *custom].uniq
@@ -23,7 +23,7 @@ module LlmCostTracker
23
23
  def provider_for_uri(uri)
24
24
  return nil unless uri
25
25
 
26
- LlmCostTracker.configuration.openai_compatible_providers[uri.host.to_s.downcase]&.to_s
26
+ LlmCostTracker.configuration.capture.openai_compatible_providers[uri.host.to_s.downcase]&.to_s
27
27
  end
28
28
  end
29
29
 
@@ -8,14 +8,10 @@ module LlmCostTracker
8
8
  app.config.eager_load_paths << models_path unless app.config.eager_load_paths.include?(models_path)
9
9
  end
10
10
 
11
+ GENERATOR_FILES = File.expand_path("generators/llm_cost_tracker/*_generator.rb", __dir__)
12
+
11
13
  generators do
12
- require_relative "generators/llm_cost_tracker/install_generator"
13
- require_relative "generators/llm_cost_tracker/prices_generator"
14
- require_relative "generators/llm_cost_tracker/call_rollups_generator"
15
- require_relative "generators/llm_cost_tracker/async_ingestion_generator"
16
- require_relative "generators/llm_cost_tracker/upgrade_call_rollups_provider_generator"
17
- require_relative "generators/llm_cost_tracker/upgrade_image_tokens_generator"
18
- require_relative "generators/llm_cost_tracker/upgrade_call_tags_key_value_index_generator"
14
+ Dir[GENERATOR_FILES].each { |path| require path }
19
15
  end
20
16
  end
21
17
  end
@@ -34,7 +34,7 @@ module LlmCostTracker
34
34
  end
35
35
  from = now - days.days
36
36
  scope = LlmCostTracker::Call.where(tracked_at: from..now)
37
- tag_breakdowns ||= LlmCostTracker.configuration.report_tag_breakdowns || []
37
+ tag_breakdowns ||= LlmCostTracker.configuration.tags.report_breakdown_keys || []
38
38
  aggregate = totals(scope)
39
39
 
40
40
  new(
@@ -55,7 +55,7 @@ module LlmCostTracker
55
55
  def self.totals(scope)
56
56
  scope
57
57
  .select(
58
- "COALESCE(SUM(total_cost), 0) AS total_cost, " \
58
+ "COALESCE(SUM(#{LlmCostTracker::Call.qualified(:total_cost)}), 0) AS total_cost, " \
59
59
  "COUNT(*) AS requests_count, " \
60
60
  "AVG(latency_ms) AS average_latency_ms, " \
61
61
  "COALESCE(SUM(CASE WHEN #{Charges::CostStatus.unknown_pricing_sql} " \
@@ -3,6 +3,8 @@
3
3
  module LlmCostTracker
4
4
  module Retention
5
5
  DEFAULT_BATCH_SIZE = 5_000
6
+ ROLLUP_COLUMNS = %i[tracked_at total_cost pricing_snapshot provider].freeze
7
+ private_constant :ROLLUP_COLUMNS
6
8
 
7
9
  class << self
8
10
  def prune(older_than:, batch_size: DEFAULT_BATCH_SIZE, now: Time.now.utc)
@@ -26,11 +28,23 @@ module LlmCostTracker
26
28
  require_relative "ingestion"
27
29
  return 0 unless LlmCostTracker::Ingestion::InboxEntry.table_exists?
28
30
 
29
- LlmCostTracker::Ingestion::InboxEntry.where(tracked_at: ...cutoff).delete_all
31
+ scope = LlmCostTracker::Ingestion::InboxEntry.where(tracked_at: ...cutoff)
32
+ warn_undrained(scope.pending)
33
+ scope.delete_all
30
34
  end
31
35
 
32
36
  private
33
37
 
38
+ def warn_undrained(scope)
39
+ count, cost = scope.pick(Arel.sql("COUNT(*)"), Arel.sql("COALESCE(SUM(total_cost), 0)"))
40
+ return if count.to_i.zero?
41
+
42
+ Logging.warn(
43
+ "Retention.prune_inbox is deleting #{count} inbox row(s) worth #{cost} that were never drained " \
44
+ "into the ledger. Their spend is lost. Drain the inbox before pruning, or raise the retention window."
45
+ )
46
+ end
47
+
34
48
  def resolve_cutoff(older_than, now)
35
49
  cutoff = case older_than
36
50
  when Time, DateTime then older_than.utc
@@ -58,22 +72,20 @@ module LlmCostTracker
58
72
 
59
73
  def prune_batch(cutoff, batch_size)
60
74
  LlmCostTracker::Call.transaction do
61
- cache_rollups = LlmCostTracker.configuration.cache_rollups
62
- rows = prunable_rows(cutoff, batch_size, with_rollup_columns: cache_rollups)
75
+ rows = prunable_rows(cutoff, batch_size)
63
76
  next 0 if rows.empty?
64
77
 
65
- ids = cache_rollups ? rows.map(&:id) : rows
66
- deleted = LlmCostTracker::Call.where(id: ids).delete_all
67
- LlmCostTracker::Ledger::Rollups.decrement!(rows) if cache_rollups && deleted.positive?
78
+ deleted = LlmCostTracker::Call.where(id: rows.map(&:id)).delete_all
79
+ LlmCostTracker::Ledger::Rollups.decrement!(rows) if deleted.positive?
68
80
  deleted
69
81
  end
70
82
  end
71
83
 
72
- def prunable_rows(cutoff, batch_size, with_rollup_columns:)
84
+ def prunable_rows(cutoff, batch_size)
73
85
  relation = LlmCostTracker::Call.where(tracked_at: ...cutoff).order(:id).limit(batch_size).lock
74
- return relation.pluck(:id) unless with_rollup_columns
75
-
76
- relation.select(:id, :tracked_at, :total_cost, :pricing_snapshot, :provider).to_a
86
+ columns = [:id]
87
+ columns += ROLLUP_COLUMNS if LlmCostTracker::Ledger::Rollups.cache_active?
88
+ relation.select(*columns).to_a
77
89
  end
78
90
  end
79
91
  end
@@ -18,15 +18,15 @@ module LlmCostTracker
18
18
 
19
19
  def tags
20
20
  config = LlmCostTracker.configuration
21
- base = config.static_sanitized_default_tags ||
22
- Sanitizer.call(call_default_tags(config.default_tags).to_h)
21
+ base = config.tags.static_sanitized_default ||
22
+ Sanitizer.call(call_default_tags(config.tags.default).to_h)
23
23
  base.merge(*Array(ActiveSupport::IsolatedExecutionState[KEY]))
24
24
  end
25
25
 
26
26
  def call_default_tags(proc_or_lambda)
27
27
  proc_or_lambda.call
28
28
  rescue StandardError => e
29
- Logging.warn("LlmCostTracker default_tags proc raised: #{e.class}: #{e.message}; using empty default tags")
29
+ Logging.warn("LlmCostTracker tags.default proc raised: #{e.class}: #{e.message}; using empty default tags")
30
30
  {}
31
31
  end
32
32
  end
@@ -23,9 +23,9 @@ module LlmCostTracker
23
23
  class << self
24
24
  def call(tags, config: LlmCostTracker.configuration)
25
25
  tags = (tags || {}).to_h
26
- redacted = config.normalized_redacted_tag_keys
27
- limit = [config.max_tag_value_bytesize.to_i, 0].max
28
- max_count = [config.max_tag_count.to_i, 0].max
26
+ redacted = config.tags.normalized_redacted_keys
27
+ limit = [config.tags.max_value_bytesize.to_i, 0].max
28
+ max_count = [config.tags.max_count.to_i, 0].max
29
29
  tags.to_a.last(max_count).each_with_object({}) do |(key, value), sanitized|
30
30
  next unless valid_key?(key)
31
31
 
@@ -43,7 +43,7 @@ module LlmCostTracker
43
43
 
44
44
  def cap(tags, config: LlmCostTracker.configuration)
45
45
  tags = (tags || {}).to_h
46
- max_count = [config.max_tag_count.to_i, 0].max
46
+ max_count = [config.tags.max_count.to_i, 0].max
47
47
  return tags if tags.size <= max_count
48
48
 
49
49
  tags.to_a.last(max_count).to_h
@@ -25,25 +25,14 @@ module LlmCostTracker
25
25
  usage_source: event.usage_source
26
26
  )
27
27
 
28
- if enforce_budget
29
- Budget.enforce!(provider: event.provider,
30
- model: event.model,
31
- estimate: calculation.cost&.total,
32
- force: true)
33
- end
28
+ tags = build_tags(context_tags: context_tags, metadata: metadata)
34
29
 
35
30
  if calculation.token_cost.nil? && event.token_usage.total_tokens.positive? &&
36
31
  calculation.priced_line_items.none?(&:priced?)
37
- Pricing::Unknown.process(event.model)
32
+ Pricing::Unknown.process(event.model, pricing_mode: calculation.mode)
38
33
  end
39
34
 
40
- event = build_event(
41
- event: event,
42
- calculation: calculation,
43
- metadata: metadata,
44
- latency_ms: latency_ms,
45
- context_tags: context_tags
46
- )
35
+ event = build_event(event: event, calculation: calculation, tags: tags, latency_ms: latency_ms)
47
36
 
48
37
  if Ingestion.async?
49
38
  Ingestion::Inbox.save(event)
@@ -54,11 +43,19 @@ module LlmCostTracker
54
43
 
55
44
  yield if block_given?
56
45
  notify_subscribers(event)
57
- Budget.check!(event)
46
+ behavior_override = :raise if enforce_budget
47
+ Budget.check!(event, behavior_override: behavior_override)
48
+ Budget.check_persisted!([event], behavior_override: behavior_override) unless Ingestion.async?
58
49
 
59
50
  event
60
51
  end
61
52
 
53
+ def build_tags(context_tags:, metadata:)
54
+ resolved = (context_tags || LlmCostTracker::Tags::Context.tags).to_h
55
+ sanitized_metadata = LlmCostTracker::Tags::Sanitizer.call(metadata.to_h)
56
+ LlmCostTracker::Tags::Sanitizer.cap(resolved.merge(sanitized_metadata)).freeze
57
+ end
58
+
62
59
  private
63
60
 
64
61
  def notify_subscribers(event)
@@ -69,13 +66,12 @@ module LlmCostTracker
69
66
  Logging.warn("Subscriber raised on #{EVENT_NAME}: #{e.class}: #{e.message}")
70
67
  end
71
68
 
72
- def build_event(event:, calculation:, metadata:, latency_ms:, context_tags:)
73
- context_tags = (context_tags || LlmCostTracker::Tags::Context.tags).to_h
69
+ def build_event(event:, calculation:, tags:, latency_ms:)
74
70
  event.with(
75
71
  event_id: SecureRandom.uuid,
76
72
  pricing_mode: calculation.mode,
77
73
  cost: calculation.cost,
78
- tags: build_tags(context_tags: context_tags, metadata: metadata),
74
+ tags: tags,
79
75
  latency_ms: finite_latency_ms(latency_ms),
80
76
  tracked_at: Time.now.utc,
81
77
  cost_status: calculation.cost_status,
@@ -84,11 +80,6 @@ module LlmCostTracker
84
80
  )
85
81
  end
86
82
 
87
- def build_tags(context_tags:, metadata:)
88
- sanitized_metadata = LlmCostTracker::Tags::Sanitizer.call(metadata.to_h)
89
- LlmCostTracker::Tags::Sanitizer.cap(context_tags.merge(sanitized_metadata)).freeze
90
- end
91
-
92
83
  def finite_latency_ms(latency_ms)
93
84
  return nil if latency_ms.nil?
94
85
 
@@ -16,8 +16,7 @@ module LlmCostTracker
16
16
  "request" => "per_request",
17
17
  "session" => "per_session",
18
18
  "hour" => "per_hour",
19
- "minute" => "per_minute",
20
- "image" => "per_image"
19
+ "minute" => "per_minute"
21
20
  }.freeze
22
21
 
23
22
  class << self
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module LlmCostTracker
4
- VERSION = "0.13.0"
4
+ VERSION = "0.14.0"
5
5
  end
@@ -110,7 +110,12 @@ module LlmCostTracker
110
110
  provider_workspace_id: nil,
111
111
  pricing_mode: nil)
112
112
  require_relative "llm_cost_tracker/capture/stream_collector"
113
- Budget.enforce!(provider: provider, model: model, force: true) if enforce_budget
113
+ if enforce_budget
114
+ Budget.enforce!(provider: provider,
115
+ model: model,
116
+ force: true,
117
+ tags: Tracker.build_tags(context_tags: nil, metadata: tags))
118
+ end
114
119
  collector = Capture::StreamCollector.new(
115
120
  provider: provider.to_s,
116
121
  model: model,