ruby_llm-contract 1.1.0 → 1.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +86 -0
  3. data/README.md +12 -3
  4. data/docs/guide/getting_started.md +17 -1
  5. data/docs/guide/relation_to_agent.md +8 -2
  6. data/lib/ruby_llm/contract/adapters/response.rb +28 -2
  7. data/lib/ruby_llm/contract/adapters/ruby_llm.rb +72 -18
  8. data/lib/ruby_llm/contract/concerns/eval_host.rb +2 -2
  9. data/lib/ruby_llm/contract/concerns/usage_aggregator.rb +8 -5
  10. data/lib/ruby_llm/contract/cost_calculator.rb +69 -41
  11. data/lib/ruby_llm/contract/eval/case_executor.rb +1 -1
  12. data/lib/ruby_llm/contract/eval/case_result.rb +3 -0
  13. data/lib/ruby_llm/contract/eval/eval_history.rb +2 -1
  14. data/lib/ruby_llm/contract/eval/evaluator/proc_evaluator.rb +6 -2
  15. data/lib/ruby_llm/contract/eval/model_comparison.rb +11 -1
  16. data/lib/ruby_llm/contract/eval/prompt_diff_serializer.rb +1 -1
  17. data/lib/ruby_llm/contract/eval/recommender.rb +12 -5
  18. data/lib/ruby_llm/contract/eval/report.rb +4 -1
  19. data/lib/ruby_llm/contract/eval/report_stats.rb +10 -5
  20. data/lib/ruby_llm/contract/eval/report_storage.rb +14 -8
  21. data/lib/ruby_llm/contract/eval/unknown_cost_gate.rb +0 -8
  22. data/lib/ruby_llm/contract/minitest.rb +2 -2
  23. data/lib/ruby_llm/contract/pipeline/runner.rb +8 -0
  24. data/lib/ruby_llm/contract/pipeline/trace.rb +10 -0
  25. data/lib/ruby_llm/contract/railtie.rb +1 -1
  26. data/lib/ruby_llm/contract/rake_task/suite_gate.rb +4 -4
  27. data/lib/ruby_llm/contract/rake_task.rb +4 -4
  28. data/lib/ruby_llm/contract/rspec/pass_eval.rb +2 -2
  29. data/lib/ruby_llm/contract/step/base.rb +21 -12
  30. data/lib/ruby_llm/contract/step/dsl.rb +5 -14
  31. data/lib/ruby_llm/contract/step/limit_checker.rb +2 -1
  32. data/lib/ruby_llm/contract/step/result_builder.rb +32 -3
  33. data/lib/ruby_llm/contract/step/retry_executor.rb +35 -4
  34. data/lib/ruby_llm/contract/step/runner.rb +7 -1
  35. data/lib/ruby_llm/contract/step/runner_config.rb +2 -2
  36. data/lib/ruby_llm/contract/step/trace.rb +54 -20
  37. data/lib/ruby_llm/contract/token_estimator.rb +5 -3
  38. data/lib/ruby_llm/contract/unknown_policy.rb +21 -0
  39. data/lib/ruby_llm/contract/version.rb +1 -1
  40. data/lib/ruby_llm/contract.rb +17 -4
  41. metadata +2 -1
@@ -69,7 +69,8 @@ module RubyLLM
69
69
 
70
70
  lines = ["#{runs.length} runs"]
71
71
  runs.last(5).each do |r|
72
- lines << " #{r[:date]} score=#{r[:score].round(2)} cost=$#{format("%.6f", r[:total_cost] || r[:cost] || 0)}"
72
+ cost = format("%.6f", r[:total_cost] || r[:cost] || 0)
73
+ lines << " #{r[:date]} score=#{r[:score].round(2)} cost=$#{cost}#{" + unknown" if r[:cost_unknown]}"
73
74
  end
74
75
  lines.join("\n")
75
76
  end
@@ -6,6 +6,10 @@ module RubyLLM
6
6
  module Evaluator
7
7
  # Adapts custom Ruby callables to the EvaluationResult contract.
8
8
  class ProcEvaluator
9
+ # Report hides these under a failure: they repeat the PASS/FAIL label.
10
+ PASSED_DETAILS = "passed"
11
+ FAILED_DETAILS = "not passed"
12
+
9
13
  def initialize(callable)
10
14
  @callable = callable
11
15
  end
@@ -36,9 +40,9 @@ module RubyLLM
36
40
  def build_evaluation_result(result)
37
41
  case result
38
42
  when true
39
- EvaluationResult.new(score: 1.0, passed: true, details: "passed")
43
+ EvaluationResult.new(score: 1.0, passed: true, details: PASSED_DETAILS)
40
44
  when false
41
- EvaluationResult.new(score: 0.0, passed: false, details: "not passed")
45
+ EvaluationResult.new(score: 0.0, passed: false, details: FAILED_DETAILS)
42
46
  when Numeric
43
47
  EvaluationResult.new(score: result, passed: result >= 0.5, details: "custom score: #{result}")
44
48
  else
@@ -36,8 +36,12 @@ module RubyLLM
36
36
  @reports[resolve_key(candidate)]&.total_cost
37
37
  end
38
38
 
39
+ # A report with unpriced cases has only a subtotal, so it cannot be
40
+ # called the cheapest.
39
41
  def best_for(min_score: 0.0)
40
- eligible = @reports.select { |_, report| report.score > 0.0 && report.score >= min_score }
42
+ eligible = @reports.select do |_, report|
43
+ report.score > 0.0 && report.score >= min_score && !cost_unknown?(report)
44
+ end
41
45
  return nil if eligible.empty?
42
46
 
43
47
  eligible.min_by { |_, report| report.total_cost }&.first
@@ -45,6 +49,8 @@ module RubyLLM
45
49
 
46
50
  def cost_per_point
47
51
  @reports.transform_values do |report|
52
+ next Float::INFINITY if cost_unknown?(report)
53
+
48
54
  report.score.positive? ? report.total_cost / report.score : Float::INFINITY
49
55
  end
50
56
  end
@@ -163,6 +169,10 @@ module RubyLLM
163
169
  )
164
170
  end
165
171
 
172
+ def cost_unknown?(report)
173
+ report.respond_to?(:unknown_cost_results) && report.unknown_cost_results.any?
174
+ end
175
+
166
176
  def resolve_key(candidate)
167
177
  case candidate
168
178
  when String then candidate
@@ -5,7 +5,7 @@ module RubyLLM
5
5
  module Eval
6
6
  class PromptDiffSerializer
7
7
  def call(report)
8
- report.results.reject { |result| result.step_status == :skipped }.map do |result|
8
+ report.results.reject { |result| result.step_status == CaseResult::SKIPPED_STATUS }.map do |result|
9
9
  {
10
10
  name: result.name,
11
11
  input: result.input,
@@ -40,7 +40,7 @@ module RubyLLM
40
40
  report = @comparison.reports[label]
41
41
  next nil unless report
42
42
 
43
- evaluated_count = report.results.count { |r| r.step_status != :skipped }
43
+ evaluated_count = report.results.count { |r| r.step_status != CaseResult::SKIPPED_STATUS }
44
44
  cases_count = [evaluated_count, 1].max
45
45
  cost_per_call = report.total_cost.to_f / cases_count
46
46
 
@@ -51,7 +51,8 @@ module RubyLLM
51
51
  cost_per_call: cost_per_call,
52
52
  latency: report.avg_latency_ms || Float::INFINITY,
53
53
  pass_rate_ratio: report.pass_rate_ratio,
54
- total_cost: report.total_cost
54
+ total_cost: report.total_cost,
55
+ cost_unknown: unknown_cost?(report)
55
56
  }
56
57
  end
57
58
  end
@@ -114,9 +115,9 @@ module RubyLLM
114
115
 
115
116
  current_label = ModelComparison.candidate_label(@current_config)
116
117
  current_report = @comparison.reports[current_label]
117
- return {} unless current_report
118
+ return {} if current_report.nil? || unknown_cost?(current_report)
118
119
 
119
- current_evaluated = current_report.results.count { |r| r.step_status != :skipped }
120
+ current_evaluated = current_report.results.count { |r| r.step_status != CaseResult::SKIPPED_STATUS }
120
121
  current_cases = [current_evaluated, 1].max
121
122
  current_cost = current_report.total_cost.to_f / current_cases
122
123
  diff = current_cost - best[:cost_per_call]
@@ -125,8 +126,14 @@ module RubyLLM
125
126
  { per_call: diff.round(6), monthly_at: { 10_000 => (diff * 10_000).round(2) } }
126
127
  end
127
128
 
129
+ # A subtotal over unpriced cases is not a cost to rank by. Zero stays
130
+ # "unknown pricing" too: offline runs report $0 without any call.
128
131
  def cost_known?(scored_candidate)
129
- scored_candidate[:cost_per_call]&.positive?
132
+ !scored_candidate[:cost_unknown] && scored_candidate[:cost_per_call]&.positive?
133
+ end
134
+
135
+ def unknown_cost?(report)
136
+ report.respond_to?(:unknown_cost_results) && report.unknown_cost_results.any?
130
137
  end
131
138
  end
132
139
  end
@@ -1,6 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require "forwardable"
4
+ require_relative "evaluator/proc_evaluator"
4
5
 
5
6
  module RubyLLM
6
7
  module Contract
@@ -10,9 +11,11 @@ module RubyLLM
10
11
 
11
12
  attr_reader :dataset_name, :results, :step_name
12
13
 
13
- GENERIC_DETAILS = ["passed", "not passed"].freeze
14
+ GENERIC_DETAILS = [Evaluator::ProcEvaluator::PASSED_DETAILS, Evaluator::ProcEvaluator::FAILED_DETAILS].freeze
14
15
  HISTORY_DIR = ".eval_history"
16
+ HISTORY_EXT = "jsonl"
15
17
  BASELINE_DIR = ".eval_baselines"
18
+ BASELINE_EXT = "json"
16
19
 
17
20
  def_delegators :@stats, :score, :passed, :failed, :skipped, :failures, :pass_rate, :pass_rate_ratio,
18
21
  :total_cost, :unknown_cost_results, :avg_latency_ms, :passed?,
@@ -23,7 +23,7 @@ module RubyLLM
23
23
  end
24
24
 
25
25
  def skipped
26
- @results.count { |result| result.step_status == :skipped }
26
+ @results.count { |result| result.step_status == CaseResult::SKIPPED_STATUS }
27
27
  end
28
28
 
29
29
  def failures
@@ -63,7 +63,7 @@ module RubyLLM
63
63
  end
64
64
 
65
65
  def evaluated_results
66
- @evaluated_results ||= @results.reject { |result| result.step_status == :skipped }
66
+ @evaluated_results ||= @results.reject { |result| result.step_status == CaseResult::SKIPPED_STATUS }
67
67
  end
68
68
 
69
69
  def evaluated_results_count
@@ -85,7 +85,7 @@ module RubyLLM
85
85
  def single_shot_cost
86
86
  return nil unless production_mode?
87
87
 
88
- evaluated_results.sum { |r| first_attempt_cost(r) || r.cost || 0.0 }
88
+ evaluated_results.sum { |r| first_attempt_cost(r) || 0.0 }
89
89
  end
90
90
 
91
91
  def effective_cost
@@ -116,9 +116,14 @@ module RubyLLM
116
116
 
117
117
  private
118
118
 
119
+ # With attempts, only the first one's cost: falling back to the case
120
+ # cost when it is unpriced would report a later attempt's spend as the
121
+ # single-shot cost.
119
122
  def first_attempt_cost(result)
120
- first = (result.attempts || []).first
121
- first && first[:cost]
123
+ attempts = result.attempts || []
124
+ return result.cost if attempts.empty?
125
+
126
+ attempts.first[:cost]
122
127
  end
123
128
 
124
129
  def first_attempt_latency(result)
@@ -13,7 +13,7 @@ module RubyLLM
13
13
  end
14
14
 
15
15
  def save_history!(path: nil, model: nil, reasoning_effort: nil)
16
- file = path || storage_path(Report::HISTORY_DIR, "jsonl", model: model, reasoning_effort: reasoning_effort)
16
+ file = path || storage_path(Report::HISTORY_DIR, Report::HISTORY_EXT, model: model, reasoning_effort: reasoning_effort)
17
17
  entry = history_entry
18
18
  entry[:model] = model if model
19
19
  entry[:reasoning_effort] = reasoning_effort if reasoning_effort
@@ -22,19 +22,19 @@ module RubyLLM
22
22
  end
23
23
 
24
24
  def eval_history(path: nil, model: nil, reasoning_effort: nil)
25
- EvalHistory.load(path || storage_path(Report::HISTORY_DIR, "jsonl", model: model,
26
- reasoning_effort: reasoning_effort))
25
+ EvalHistory.load(path || storage_path(Report::HISTORY_DIR, Report::HISTORY_EXT,
26
+ model: model, reasoning_effort: reasoning_effort))
27
27
  end
28
28
 
29
29
  def save_baseline!(path: nil, model: nil, reasoning_effort: nil)
30
- file = path || storage_path(Report::BASELINE_DIR, "json", model: model, reasoning_effort: reasoning_effort)
30
+ file = path || storage_path(Report::BASELINE_DIR, Report::BASELINE_EXT, model: model, reasoning_effort: reasoning_effort)
31
31
  FileUtils.mkdir_p(File.dirname(file))
32
32
  File.write(file, JSON.pretty_generate(serialize_for_baseline))
33
33
  file
34
34
  end
35
35
 
36
36
  def compare_with_baseline(path: nil, model: nil, reasoning_effort: nil)
37
- file = path || storage_path(Report::BASELINE_DIR, "json", model: model, reasoning_effort: reasoning_effort)
37
+ file = path || storage_path(Report::BASELINE_DIR, Report::BASELINE_EXT, model: model, reasoning_effort: reasoning_effort)
38
38
  raise ArgumentError, "No baseline found at #{file}" unless File.exist?(file)
39
39
 
40
40
  baseline_data = JSON.parse(File.read(file), symbolize_names: true)
@@ -47,13 +47,15 @@ module RubyLLM
47
47
  end
48
48
 
49
49
  def baseline_exists?(path: nil, model: nil, reasoning_effort: nil)
50
- File.exist?(path || storage_path(Report::BASELINE_DIR, "json", model: model, reasoning_effort: reasoning_effort))
50
+ File.exist?(path || storage_path(Report::BASELINE_DIR, Report::BASELINE_EXT, model: model, reasoning_effort: reasoning_effort))
51
51
  end
52
52
 
53
53
  private
54
54
 
55
+ # cost_unknown is written only when true, so older files and runs that
56
+ # priced every case keep the same shape.
55
57
  def history_entry
56
- {
58
+ entry = {
57
59
  date: Time.now.strftime("%Y-%m-%d"),
58
60
  score: @stats.score,
59
61
  total_cost: @stats.total_cost,
@@ -61,6 +63,8 @@ module RubyLLM
61
63
  pass_rate_ratio: @stats.pass_rate_ratio,
62
64
  cases_count: @stats.evaluated_results_count
63
65
  }
66
+ entry[:cost_unknown] = true if @stats.unknown_cost_results.any?
67
+ entry
64
68
  end
65
69
 
66
70
  def serialize_for_baseline
@@ -74,13 +78,15 @@ module RubyLLM
74
78
  end
75
79
 
76
80
  def serialize_case(result)
77
- {
81
+ serialized = {
78
82
  name: result.name,
79
83
  passed: result.passed?,
80
84
  score: result.score,
81
85
  details: result.details,
82
86
  cost: result.cost
83
87
  }
88
+ serialized[:cost_unknown] = true if result.cost_unknown?
89
+ serialized
84
90
  end
85
91
 
86
92
  def storage_path(root_dir, extension, model:, reasoning_effort: nil)
@@ -8,14 +8,6 @@ module RubyLLM
8
8
  # gate that only compares totals passes whatever those cases really cost.
9
9
  # Shared by the rake task, the pass_eval matcher and assert_eval_passes.
10
10
  module UnknownCostGate
11
- MODES = %i[refuse warn].freeze
12
-
13
- def self.validate!(mode)
14
- return mode if MODES.include?(mode)
15
-
16
- raise ArgumentError, "on_unknown_pricing must be :refuse or :warn, got #{mode.inspect}"
17
- end
18
-
19
11
  # Returns the failure message under :refuse, nil when the gate holds.
20
12
  def self.check(reports, mode:)
21
13
  names = reports.flat_map(&:unknown_cost_results).map(&:name).uniq
@@ -31,8 +31,8 @@ module RubyLLM
31
31
  end
32
32
 
33
33
  def assert_eval_passes(step, eval_name, minimum_score: nil, maximum_cost: nil, context: {}, msg: nil,
34
- on_unknown_pricing: :refuse)
35
- Eval::UnknownCostGate.validate!(on_unknown_pricing)
34
+ on_unknown_pricing: UnknownPolicy::DEFAULT)
35
+ UnknownPolicy.validate!("on_unknown_pricing", on_unknown_pricing)
36
36
  report = step.run_eval(eval_name, context: context)
37
37
 
38
38
  if minimum_score
@@ -83,6 +83,7 @@ module RubyLLM
83
83
  total_usage: aggregate_usage(traces),
84
84
  step_traces: traces
85
85
  )
86
+ warn_incomplete_budget_usage(trace)
86
87
 
87
88
  Result.new(
88
89
  status: execution.status, step_results: execution.step_results,
@@ -91,6 +92,13 @@ module RubyLLM
91
92
  )
92
93
  end
93
94
 
95
+ def warn_incomplete_budget_usage(trace)
96
+ return unless @token_budget && !trace.usage_complete?
97
+
98
+ warn "[ruby_llm-contract] token_budget checked against incomplete token counts: " \
99
+ "a provider left out usage for at least one step, so the total undercounts"
100
+ end
101
+
94
102
  def elapsed_ms(start_time)
95
103
  ((Process.clock_gettime(Process::CLOCK_MONOTONIC) - start_time) * 1000).round
96
104
  end
@@ -40,6 +40,16 @@ module RubyLLM
40
40
  @step_traces.any? { |step_trace| step_trace.respond_to?(:cost_unknown?) && step_trace.cost_unknown? }
41
41
  end
42
42
 
43
+ # False when any step's provider left out token counts, so total_usage
44
+ # (and a token_budget checked against it) undercounts.
45
+ def usage_complete?
46
+ return true unless @step_traces.is_a?(Array)
47
+
48
+ @step_traces.none? do |step_trace|
49
+ step_trace.respond_to?(:usage_complete) && step_trace.usage_complete == false
50
+ end
51
+ end
52
+
43
53
  def to_h
44
54
  { trace_id: @trace_id, total_latency_ms: @total_latency_ms,
45
55
  total_usage: @total_usage, step_traces: @step_traces,
@@ -6,7 +6,7 @@ module RubyLLM
6
6
  # Ignore eval/ subdirs BEFORE Zeitwerk setup — eval files don't define
7
7
  # constants, they call define_eval on existing Step classes.
8
8
  initializer "ruby_llm_contract.ignore_eval_dirs", before: :set_autoload_paths do |app|
9
- %w[app/contracts/eval app/steps/eval].each do |path|
9
+ RubyLLM::Contract::RAILS_EVAL_DIRS.each do |path|
10
10
  full = app.root.join(path)
11
11
  next unless full.exist?
12
12
 
@@ -18,7 +18,7 @@ module RubyLLM
18
18
  end
19
19
 
20
20
  def self.evaluate(host_reports:, minimum_score:, maximum_cost:, fail_on_regression:,
21
- on_unknown_pricing: :refuse)
21
+ on_unknown_pricing: UnknownPolicy::DEFAULT)
22
22
  new(host_reports: host_reports,
23
23
  minimum_score: minimum_score,
24
24
  maximum_cost: maximum_cost,
@@ -29,12 +29,12 @@ module RubyLLM
29
29
  attr_reader :verdict
30
30
 
31
31
  def initialize(host_reports:, minimum_score:, maximum_cost:, fail_on_regression:,
32
- on_unknown_pricing: :refuse)
32
+ on_unknown_pricing: UnknownPolicy::DEFAULT)
33
33
  @host_reports = host_reports
34
34
  @minimum_score = minimum_score
35
35
  @maximum_cost = maximum_cost
36
36
  @fail_on_regression = fail_on_regression
37
- @on_unknown_pricing = Eval::UnknownCostGate.validate!(on_unknown_pricing)
37
+ @on_unknown_pricing = UnknownPolicy.validate!("on_unknown_pricing", on_unknown_pricing)
38
38
  @verdict = build_verdict
39
39
  end
40
40
 
@@ -56,7 +56,7 @@ module RubyLLM
56
56
  passed_reports, all_passed = score_each_report
57
57
  Verdict.new(
58
58
  passed: all_passed,
59
- abort_reason: all_passed ? nil : "Eval suite FAILED",
59
+ abort_reason: all_passed ? nil : "one or more evals did not pass",
60
60
  passed_reports: passed_reports,
61
61
  suite_cost: suite_cost
62
62
  )
@@ -2,7 +2,7 @@
2
2
 
3
3
  require "rake"
4
4
  require "rake/tasklib"
5
- require_relative "eval/unknown_cost_gate"
5
+ require_relative "unknown_policy"
6
6
  require_relative "rake_task/suite_gate"
7
7
 
8
8
  module RubyLLM
@@ -21,12 +21,12 @@ module RubyLLM
21
21
  @maximum_cost = nil # nil = no cost limit; float = budget cap (suite-level)
22
22
  @eval_dirs = [] # directories to load eval files from (non-Rails)
23
23
  # With maximum_cost: :refuse fails on cases the budget cannot price; :warn gates priced cases only
24
- @on_unknown_pricing = :refuse
24
+ @on_unknown_pricing = UnknownPolicy::DEFAULT
25
25
  @save_baseline = false
26
26
  @fail_on_regression = false
27
27
  @track_history = false
28
28
  block&.call(self)
29
- Eval::UnknownCostGate.validate!(@on_unknown_pricing)
29
+ UnknownPolicy.validate!("on_unknown_pricing", @on_unknown_pricing)
30
30
  define_task
31
31
  end
32
32
 
@@ -134,7 +134,7 @@ module RubyLLM
134
134
  abort("STEP is required, e.g. STEP=MatchProblemsToPages") if step_name.empty?
135
135
  raw_candidates = ENV["CANDIDATES"].to_s.strip
136
136
  abort("CANDIDATES is required, e.g. CANDIDATES=gpt-5-nano,gpt-5-mini@low,gpt-5-mini") if raw_candidates.empty?
137
- min_score = ENV.fetch("MIN_SCORE", "0.95").to_f
137
+ min_score = ENV.fetch("MIN_SCORE") { Eval::DEFAULT_MIN_SCORE }.to_f
138
138
  runs = parse_runs(ENV.fetch("RUNS", "1"))
139
139
 
140
140
  host = RubyLLM::Contract.eval_hosts.find { |h| h.name == step_name }
@@ -65,7 +65,7 @@ RSpec::Matchers.define :pass_eval do |eval_name|
65
65
  end
66
66
 
67
67
  chain :on_unknown_pricing do |mode|
68
- @on_unknown_pricing = RubyLLM::Contract::Eval::UnknownCostGate.validate!(mode)
68
+ @on_unknown_pricing = RubyLLM::Contract::UnknownPolicy.validate!("on_unknown_pricing", mode)
69
69
  end
70
70
 
71
71
  chain :without_regressions do
@@ -82,7 +82,7 @@ RSpec::Matchers.define :pass_eval do |eval_name|
82
82
  @context ||= {}
83
83
  @minimum_score ||= nil
84
84
  @maximum_cost ||= nil
85
- @on_unknown_pricing ||= :refuse
85
+ @on_unknown_pricing ||= RubyLLM::Contract::UnknownPolicy::DEFAULT
86
86
  @unknown_cost_failure = nil
87
87
  @unknown_cost_only = false
88
88
  @check_regressions ||= false
@@ -26,9 +26,9 @@ module RubyLLM
26
26
  context: context).results.first
27
27
  end
28
28
 
29
- def estimate_cost(input:, model: nil, attachment: nil)
29
+ def estimate_cost(input:, model: nil, attachment: nil, provider: nil)
30
30
  model_name = estimated_model_name(model)
31
- model_info = CostCalculator.find_model(model_name)
31
+ model_info = CostCalculator.find_model(model_name, provider: provider)
32
32
  return nil unless model_info
33
33
 
34
34
  text_tokens = TokenEstimator.estimate(build_messages(input))
@@ -48,12 +48,13 @@ module RubyLLM
48
48
  output_tokens_estimate: output_tokens,
49
49
  estimated_cost: CostCalculator.calculate(
50
50
  model_name: model_name,
51
- usage: { input_tokens: input_tokens, output_tokens: output_tokens }
51
+ usage: { input_tokens: input_tokens, output_tokens: output_tokens },
52
+ provider: provider
52
53
  )
53
54
  }
54
55
  end
55
56
 
56
- def estimate_eval_cost(eval_name, models: nil)
57
+ def estimate_eval_cost(eval_name, models: nil, provider: nil)
57
58
  defn = send(:all_eval_definitions)[eval_name.to_s]
58
59
  raise ArgumentError, "No eval '#{eval_name}' defined" unless defn
59
60
 
@@ -61,7 +62,7 @@ module RubyLLM
61
62
  cases = defn.build_dataset.cases
62
63
 
63
64
  model_list.each_with_object({}) do |model_name, result|
64
- result[model_name] = estimate_eval_cost_for_model(cases, model_name)
65
+ result[model_name] = estimate_eval_cost_for_model(cases, model_name, provider)
65
66
  end
66
67
  end
67
68
 
@@ -90,8 +91,10 @@ module RubyLLM
90
91
  ).call
91
92
  end
92
93
 
93
- KNOWN_CONTEXT_KEYS = %i[adapter model temperature max_tokens provider assume_model_exists
94
- reasoning_effort retry_policy_override attachment].freeze
94
+ # Forwarded to the adapter as-is. A key known but not forwarded would be
95
+ # silently dropped, so the known list is built from this one.
96
+ ADAPTER_CONTEXT_KEYS = %i[provider assume_model_exists max_tokens reasoning_effort attachment].freeze
97
+ KNOWN_CONTEXT_KEYS = (%i[adapter model temperature retry_policy_override] + ADAPTER_CONTEXT_KEYS).freeze
95
98
 
96
99
  include Concerns::ContextHelpers
97
100
 
@@ -133,7 +136,7 @@ module RubyLLM
133
136
  estimate = attachment_token_estimate if respond_to?(:attachment_token_estimate)
134
137
  return [estimate, false] unless estimate.nil?
135
138
 
136
- mode = respond_to?(:on_unknown_attachment_size) ? on_unknown_attachment_size : :refuse
139
+ mode = respond_to?(:on_unknown_attachment_size) ? on_unknown_attachment_size : UnknownPolicy::DEFAULT
137
140
  if mode == :warn
138
141
  warn "[ruby_llm-contract] attachment present but attachment_token_estimate not " \
139
142
  "declared on #{name || self} — estimate_cost proceeds without attachment cost"
@@ -143,9 +146,9 @@ module RubyLLM
143
146
  [0, true]
144
147
  end
145
148
 
146
- def estimate_eval_cost_for_model(cases, model_name)
149
+ def estimate_eval_cost_for_model(cases, model_name, provider)
147
150
  cases.sum do |test_case|
148
- estimate = estimate_cost(input: test_case.input, model: model_name)
151
+ estimate = estimate_cost(input: test_case.input, model: model_name, provider: provider)
149
152
  # Two misses floor to 0.0, not one: model absent from the registry
150
153
  # (nil estimate), or present with unreadable pricing (hash whose
151
154
  # estimated_cost is nil). Documented as a floor, not a fail-closed.
@@ -193,7 +196,7 @@ module RubyLLM
193
196
 
194
197
  def runtime_settings(context)
195
198
  policy = context.key?(:retry_policy_override) ? context[:retry_policy_override] : retry_policy
196
- extra = context.slice(:provider, :assume_model_exists, :max_tokens, :reasoning_effort, :attachment)
199
+ extra = context.slice(*ADAPTER_CONTEXT_KEYS)
197
200
 
198
201
  # Always pass the class-level `thinking` config to the adapter when
199
202
  # set, so fields like `budget` survive a per-call `reasoning_effort`
@@ -285,12 +288,18 @@ module RubyLLM
285
288
  "model=#{trace.model} status=#{result.status} " \
286
289
  "latency=#{trace.latency_ms}ms " \
287
290
  "tokens=#{trace.usage&.dig(:input_tokens) || 0}+#{trace.usage&.dig(:output_tokens) || 0} " \
288
- "cost=$#{format("%.6f", trace.cost || 0)}"
291
+ "cost=#{log_cost(trace)}"
289
292
  logger.info(msg)
290
293
 
291
294
  log_failed_observations(result, logger)
292
295
  end
293
296
 
297
+ def log_cost(trace)
298
+ return "unknown" if trace.respond_to?(:cost_unknown?) && trace.cost_unknown?
299
+
300
+ "$#{format("%.6f", trace.cost || 0)}"
301
+ end
302
+
294
303
  def log_failed_observations(result, logger)
295
304
  failed = result.observations.select { |o| !o[:passed] }
296
305
  return if failed.empty?
@@ -151,12 +151,10 @@ module RubyLLM
151
151
  if amount
152
152
  validate_positive!("max_cost", amount)
153
153
 
154
- if on_unknown_pricing && !%i[refuse warn].include?(on_unknown_pricing)
155
- raise ArgumentError, "on_unknown_pricing must be :refuse or :warn, got #{on_unknown_pricing.inspect}"
156
- end
154
+ UnknownPolicy.validate!("on_unknown_pricing", on_unknown_pricing) if on_unknown_pricing
157
155
 
158
156
  @max_cost = amount
159
- @on_unknown_pricing = on_unknown_pricing || :refuse
157
+ @on_unknown_pricing = on_unknown_pricing || UnknownPolicy::DEFAULT
160
158
  return @max_cost
161
159
  end
162
160
 
@@ -164,7 +162,7 @@ module RubyLLM
164
162
  end
165
163
 
166
164
  def on_unknown_pricing
167
- inherited_value(:on_unknown_pricing) || :refuse
165
+ inherited_value(:on_unknown_pricing) || UnknownPolicy::DEFAULT
168
166
  end
169
167
 
170
168
  def attachment_token_estimate(n = nil)
@@ -182,16 +180,9 @@ module RubyLLM
182
180
  end
183
181
 
184
182
  def on_unknown_attachment_size(mode = nil)
185
- if mode
186
- unless %i[refuse warn].include?(mode)
187
- raise ArgumentError,
188
- "on_unknown_attachment_size must be :refuse or :warn, got #{mode.inspect}"
189
- end
190
-
191
- return @on_unknown_attachment_size = mode
192
- end
183
+ return @on_unknown_attachment_size = UnknownPolicy.validate!("on_unknown_attachment_size", mode) if mode
193
184
 
194
- inherited_value(:on_unknown_attachment_size) || :refuse
185
+ inherited_value(:on_unknown_attachment_size) || UnknownPolicy::DEFAULT
195
186
  end
196
187
 
197
188
  def model(name = nil)
@@ -70,7 +70,8 @@ module RubyLLM
70
70
  estimated_output = effective_max_output || (estimated * DEFAULT_OUTPUT_RATIO)
71
71
  estimated_cost = CostCalculator.calculate(
72
72
  model_name: model_name,
73
- usage: { input_tokens: estimated, output_tokens: estimated_output }
73
+ usage: { input_tokens: estimated, output_tokens: estimated_output },
74
+ provider: pricing_provider
74
75
  )
75
76
 
76
77
  if estimated_cost.nil?
@@ -4,12 +4,21 @@ module RubyLLM
4
4
  module Contract
5
5
  module Step
6
6
  class ResultBuilder
7
- def initialize(contract_definition:, output_type:, output_schema:, model:, observers:)
7
+ # Why the provider stopped, worded for a failed result. Anthropic also
8
+ # reports an exhausted context window as :max_tokens, so the message
9
+ # does not claim max_output was the limit hit.
10
+ FINISH_REASON_ERRORS = {
11
+ max_tokens: "provider reported token-limit termination",
12
+ content_filter: "provider reported content filtering or refusal"
13
+ }.freeze
14
+
15
+ def initialize(contract_definition:, output_type:, output_schema:, model:, observers:, max_output: nil)
8
16
  @contract_definition = contract_definition
9
17
  @output_type = output_type
10
18
  @output_schema = output_schema
11
19
  @model = model
12
20
  @observers = observers
21
+ @max_output = max_output
13
22
  end
14
23
 
15
24
  def error_result(error_result:, messages:)
@@ -25,13 +34,14 @@ module RubyLLM
25
34
  def success_result(response:, messages:, latency_ms:, input:)
26
35
  raw_output = response.content
27
36
  validation_result = validate_output(raw_output, input)
28
- trace = Trace.new(messages: messages, model: @model, latency_ms: latency_ms, usage: response.usage)
37
+ trace = Trace.new(messages: messages, model: @model, latency_ms: latency_ms, usage: response.usage,
38
+ **response_accounting(response))
29
39
 
30
40
  Result.new(
31
41
  status: validation_result[:status],
32
42
  raw_output: raw_output,
33
43
  parsed_output: validation_result[:parsed_output],
34
- validation_errors: validation_result[:errors],
44
+ validation_errors: validation_errors(validation_result, response),
35
45
  trace: trace,
36
46
  observations: observations_for(validation_result, input)
37
47
  )
@@ -39,6 +49,25 @@ module RubyLLM
39
49
 
40
50
  private
41
51
 
52
+ # Only a failed result says why the provider stopped; a result that
53
+ # passed validation keeps finish_reason in its trace alone.
54
+ def validation_errors(validation_result, response)
55
+ errors = validation_result[:errors]
56
+ return errors if validation_result[:status] == :ok
57
+
58
+ reason = response.respond_to?(:finish_reason) ? response.finish_reason : nil
59
+ message = FINISH_REASON_ERRORS[reason]
60
+ return errors unless message
61
+
62
+ message += "; configured max_output: #{@max_output}" if reason == :max_tokens && @max_output
63
+ [*errors, message]
64
+ end
65
+
66
+ # Custom adapters may return any object with `content` and `usage`.
67
+ def response_accounting(response)
68
+ response.respond_to?(:trace_accounting) ? response.trace_accounting : {}
69
+ end
70
+
42
71
  def observations_for(validation_result, input)
43
72
  return [] unless validation_result[:status] == :ok && @observers.any?
44
73