ruby_llm-contract 0.10.6 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. checksums.yaml +4 -4
  2. data/.ruby-version +1 -0
  3. data/CHANGELOG.md +354 -211
  4. data/README.md +12 -2
  5. data/docs/guide/getting_started.md +3 -1
  6. data/docs/guide/llm_judge.md +1 -1
  7. data/docs/guide/multimodal_input.md +4 -3
  8. data/docs/guide/output_schema.md +2 -2
  9. data/docs/guide/testing.md +1 -1
  10. data/examples/README.md +1 -1
  11. data/lib/ruby_llm/contract/adapters/ruby_llm.rb +13 -9
  12. data/lib/ruby_llm/contract/concerns/context_helpers.rb +0 -2
  13. data/lib/ruby_llm/contract/concerns/deep_symbolize.rb +0 -1
  14. data/lib/ruby_llm/contract/contract/parser.rb +0 -3
  15. data/lib/ruby_llm/contract/contract/schema_validator/bound_rule.rb +0 -1
  16. data/lib/ruby_llm/contract/contract/schema_validator/enum_rule.rb +0 -1
  17. data/lib/ruby_llm/contract/contract/schema_validator/node.rb +0 -4
  18. data/lib/ruby_llm/contract/contract/schema_validator/scalar_rules.rb +0 -1
  19. data/lib/ruby_llm/contract/contract/schema_validator/type_rule.rb +0 -1
  20. data/lib/ruby_llm/contract/cost_calculator.rb +28 -2
  21. data/lib/ruby_llm/contract/eval/aggregated_report.rb +4 -0
  22. data/lib/ruby_llm/contract/eval/baseline_diff.rb +22 -3
  23. data/lib/ruby_llm/contract/eval/candidate_label.rb +31 -0
  24. data/lib/ruby_llm/contract/eval/case_executor.rb +2 -2
  25. data/lib/ruby_llm/contract/eval/case_result.rb +17 -3
  26. data/lib/ruby_llm/contract/eval/case_result_builder.rb +2 -1
  27. data/lib/ruby_llm/contract/eval/contract_detail_builder.rb +0 -2
  28. data/lib/ruby_llm/contract/eval/model_comparison.rb +3 -2
  29. data/lib/ruby_llm/contract/eval/pipeline_result_adapter.rb +1 -2
  30. data/lib/ruby_llm/contract/eval/prompt_diff_comparator.rb +0 -1
  31. data/lib/ruby_llm/contract/eval/prompt_diff_presenter.rb +0 -1
  32. data/lib/ruby_llm/contract/eval/prompt_diff_serializer.rb +0 -1
  33. data/lib/ruby_llm/contract/eval/report.rb +1 -1
  34. data/lib/ruby_llm/contract/eval/report_presenter.rb +0 -1
  35. data/lib/ruby_llm/contract/eval/report_stats.rb +5 -1
  36. data/lib/ruby_llm/contract/eval/report_storage.rb +0 -1
  37. data/lib/ruby_llm/contract/eval/retry_optimizer.rb +10 -20
  38. data/lib/ruby_llm/contract/eval/step_expectation_applier.rb +2 -1
  39. data/lib/ruby_llm/contract/eval/trait_evaluator.rb +0 -2
  40. data/lib/ruby_llm/contract/eval/unknown_cost_gate.rb +40 -0
  41. data/lib/ruby_llm/contract/eval.rb +2 -0
  42. data/lib/ruby_llm/contract/minitest.rb +6 -1
  43. data/lib/ruby_llm/contract/pipeline/runner.rb +0 -1
  44. data/lib/ruby_llm/contract/pipeline/trace.rb +7 -0
  45. data/lib/ruby_llm/contract/rake_task/suite_gate.rb +18 -21
  46. data/lib/ruby_llm/contract/rake_task.rb +9 -4
  47. data/lib/ruby_llm/contract/rspec/pass_eval.rb +80 -52
  48. data/lib/ruby_llm/contract/step/base.rb +12 -3
  49. data/lib/ruby_llm/contract/step/dsl.rb +2 -4
  50. data/lib/ruby_llm/contract/step/limit_checker.rb +0 -2
  51. data/lib/ruby_llm/contract/step/retry_executor.rb +0 -2
  52. data/lib/ruby_llm/contract/step/trace.rb +19 -0
  53. data/lib/ruby_llm/contract/token_estimator.rb +0 -2
  54. data/lib/ruby_llm/contract/version.rb +1 -1
  55. data/lib/ruby_llm/contract.rb +0 -4
  56. data/ruby_llm-contract.gemspec +6 -5
  57. metadata +9 -11
  58. data/.rubocop.yml +0 -58
  59. data/Gemfile +0 -13
  60. data/Gemfile.lock +0 -278
  61. data/Rakefile +0 -8
@@ -13,6 +13,10 @@ module RubyLLM
13
13
  # result.print_summary
14
14
  # result.to_dsl # => copy-paste retry_policy
15
15
  class RetryOptimizer
16
+ # Documented CANDIDATES= shorthand, parsed by
17
+ # OptimizeRakeTask#parse_candidates and rendered in this table's headers.
18
+ EFFORT_SEPARATOR = "@"
19
+
16
20
  Result = Struct.new(:step_name, :eval_names, :candidate_labels, :score_matrix,
17
21
  :constraining_eval, :chain, :chain_details, keyword_init: true) do
18
22
  # Terminology alias — `hardest_eval` is the narrative name used in docs;
@@ -80,11 +84,10 @@ module RubyLLM
80
84
  end
81
85
 
82
86
  def short_candidate_label(label)
83
- label
84
- .sub("gpt-5-", "")
85
- .sub("gpt-4.1", "4.1")
86
- .sub(" (effort: ", "@")
87
- .sub(")", "")
87
+ config = CandidateLabel.parse(label)
88
+ model = config[:model].sub("gpt-5-", "").sub("gpt-4.1", "4.1")
89
+ effort = config[:reasoning_effort]
90
+ effort ? "#{model}#{EFFORT_SEPARATOR}#{effort}" : model
88
91
  end
89
92
 
90
93
  def print_dsl(io)
@@ -176,11 +179,9 @@ module RubyLLM
176
179
  def build_chain(matrix, labels, evals)
177
180
  total = evals.size
178
181
 
179
- # Find cheapest model that passes every eval — the safe fallback.
180
182
  safe_fallback = labels.find { |l| evals.all? { |e| (matrix.dig(e, l) || 0) >= @min_score } }
181
183
  return [[], []] unless safe_fallback
182
184
 
183
- # Prepend cheaper models that pass a strict subset.
184
185
  chain = []
185
186
  details = []
186
187
  covered_evals = Set.new
@@ -193,27 +194,16 @@ module RubyLLM
193
194
  next if new_additions.empty?
194
195
 
195
196
  covered_evals.merge(new_additions)
196
- chain << parse_label_to_config(label)
197
+ chain << CandidateLabel.parse(label)
197
198
  details << { label: label, passes: new_additions.size, cost: label }
198
199
  end
199
200
 
200
- # Always end with the safe fallback.
201
- chain << parse_label_to_config(safe_fallback)
201
+ chain << CandidateLabel.parse(safe_fallback)
202
202
  details << { label: safe_fallback, passes: total, cost: safe_fallback }
203
203
 
204
204
  [chain, details]
205
205
  end
206
206
 
207
- def parse_label_to_config(label)
208
- if label.match?(/\(effort: (\w+)\)/)
209
- model = label.sub(/\s*\(effort:.*/, "").strip
210
- effort = label.match(/\(effort: (\w+)\)/)[1]
211
- { model: model, reasoning_effort: effort }
212
- else
213
- { model: label }
214
- end
215
- end
216
-
217
207
  def empty_result(evals)
218
208
  Result.new(
219
209
  step_name: @step.name || @step.to_s,
@@ -58,7 +58,8 @@ module RubyLLM
58
58
  label: "FAIL",
59
59
  details: "step expectations failed: #{failure_details}",
60
60
  duration_ms: result.duration_ms,
61
- cost: result.cost
61
+ cost: result.cost,
62
+ cost_unknown: result.cost_unknown?
62
63
  )
63
64
  end
64
65
  end
@@ -3,8 +3,6 @@
3
3
  module RubyLLM
4
4
  module Contract
5
5
  module Eval
6
- # Extracted from Runner to reduce class length.
7
- # Evaluates expected_traits against parsed output.
8
6
  module TraitEvaluator
9
7
  private
10
8
 
@@ -0,0 +1,40 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Contract
5
+ module Eval
6
+ # Suite-level counterpart of the step-level `max_cost ... on_unknown_pricing:`.
7
+ # A report's total_cost counts an unpriced case as 0.0, so a maximum_cost
8
+ # gate that only compares totals passes whatever those cases really cost.
9
+ # Shared by the rake task, the pass_eval matcher and assert_eval_passes.
10
+ module UnknownCostGate
11
+ MODES = %i[refuse warn].freeze
12
+
13
+ def self.validate!(mode)
14
+ return mode if MODES.include?(mode)
15
+
16
+ raise ArgumentError, "on_unknown_pricing must be :refuse or :warn, got #{mode.inspect}"
17
+ end
18
+
19
+ # Returns the failure message under :refuse, nil when the gate holds.
20
+ def self.check(reports, mode:)
21
+ names = reports.flat_map(&:unknown_cost_results).map(&:name).uniq
22
+ return nil if names.empty?
23
+
24
+ if mode == :warn
25
+ warn "[ruby_llm-contract] #{describe(names)} - maximum_cost checked against priced cases only"
26
+ return nil
27
+ end
28
+
29
+ "#{describe(names)}. Register pricing via CostCalculator.register_model " \
30
+ "or set on_unknown_pricing to :warn to check maximum_cost against priced cases only."
31
+ end
32
+
33
+ def self.describe(names)
34
+ "#{names.length} case(s) have unknown cost (model has no pricing data): #{names.join(", ")}"
35
+ end
36
+ private_class_method :describe
37
+ end
38
+ end
39
+ end
40
+ end
@@ -22,7 +22,9 @@ require_relative "eval/report_presenter"
22
22
  require_relative "eval/report_storage"
23
23
  require_relative "eval/report"
24
24
  require_relative "eval/aggregated_report"
25
+ require_relative "eval/unknown_cost_gate"
25
26
  require_relative "eval/eval_definition"
27
+ require_relative "eval/candidate_label"
26
28
  require_relative "eval/model_comparison"
27
29
  require_relative "eval/baseline_diff"
28
30
  require_relative "eval/prompt_diff_serializer"
@@ -30,7 +30,9 @@ module RubyLLM
30
30
  refute result.ok?, msg || "Expected step result NOT to satisfy contract, but it passed"
31
31
  end
32
32
 
33
- def assert_eval_passes(step, eval_name, minimum_score: nil, maximum_cost: nil, context: {}, msg: nil)
33
+ def assert_eval_passes(step, eval_name, minimum_score: nil, maximum_cost: nil, context: {}, msg: nil,
34
+ on_unknown_pricing: :refuse)
35
+ Eval::UnknownCostGate.validate!(on_unknown_pricing)
34
36
  report = step.run_eval(eval_name, context: context)
35
37
 
36
38
  if minimum_score
@@ -42,6 +44,9 @@ module RubyLLM
42
44
  end
43
45
 
44
46
  if maximum_cost
47
+ unknown_cost_failure = Eval::UnknownCostGate.check([report], mode: on_unknown_pricing)
48
+ assert unknown_cost_failure.nil?,
49
+ msg || "Expected #{eval_name} eval to stay within maximum cost, but #{unknown_cost_failure}"
45
50
  assert report.total_cost <= maximum_cost,
46
51
  msg || "Expected #{eval_name} eval cost <= $#{format("%.4f", maximum_cost)}, got $#{format("%.4f", report.total_cost)}"
47
52
  end
@@ -95,7 +95,6 @@ module RubyLLM
95
95
  ((Process.clock_gettime(Process::CLOCK_MONOTONIC) - start_time) * 1000).round
96
96
  end
97
97
 
98
- # Encapsulates mutable state during pipeline execution
99
98
  class ExecutionState
100
99
  attr_reader :trace_id, :step_results, :step_traces, :outputs_by_step,
101
100
  :current_input, :status, :failed_step
@@ -33,6 +33,13 @@ module RubyLLM
33
33
  value.dig(*rest)
34
34
  end
35
35
 
36
+ # total_cost sums only the steps that could be priced.
37
+ def cost_unknown?
38
+ return false unless @step_traces.is_a?(Array)
39
+
40
+ @step_traces.any? { |step_trace| step_trace.respond_to?(:cost_unknown?) && step_trace.cost_unknown? }
41
+ end
42
+
36
43
  def to_h
37
44
  { trace_id: @trace_id, total_latency_ms: @total_latency_ms,
38
45
  total_usage: @total_usage, step_traces: @step_traces,
@@ -3,21 +3,10 @@
3
3
  module RubyLLM
4
4
  module Contract
5
5
  class RakeTask < ::Rake::TaskLib
6
- # Encapsulates the pass/fail gate that runs after `RakeTask#define_task`
7
- # has collected eval reports. Extracted from the prior `define_task`
8
- # god-method so each gating dimension (cost, score, regression) is
9
- # testable in isolation.
10
- #
11
- # Returns a `Verdict` value object with:
12
- # - `passed?` — overall gate verdict
13
- # - `abort_reason` — String for `abort` when `passed? == false`, nil otherwise
14
- # - `passed_reports` — [[host, report], ...] of reports that individually passed
15
- # (used to decide which baselines to save)
16
- # - `suite_cost` — total cost across all reports
17
- #
18
6
  # Gate ordering (preserved from pre-refactor behaviour):
19
- # 1. cost gate runs FIRST — if `maximum_cost` set and exceeded, the
20
- # suite aborts before any score check; passed_reports is empty.
7
+ # 1. cost gate runs FIRST — if `maximum_cost` set and exceeded, or any
8
+ # case has unknown cost under on_unknown_pricing :refuse, the suite
9
+ # aborts before any score check; passed_reports is empty.
21
10
  # 2. score gate runs per-report; a report passes if
22
11
  # `report_meets_score?` AND `!check_regression`.
23
12
  # 3. overall passed = ALL reports passed AND cost gate not tripped.
@@ -28,20 +17,24 @@ module RubyLLM
28
17
  end
29
18
  end
30
19
 
31
- def self.evaluate(host_reports:, minimum_score:, maximum_cost:, fail_on_regression:)
20
+ def self.evaluate(host_reports:, minimum_score:, maximum_cost:, fail_on_regression:,
21
+ on_unknown_pricing: :refuse)
32
22
  new(host_reports: host_reports,
33
23
  minimum_score: minimum_score,
34
24
  maximum_cost: maximum_cost,
35
- fail_on_regression: fail_on_regression).verdict
25
+ fail_on_regression: fail_on_regression,
26
+ on_unknown_pricing: on_unknown_pricing).verdict
36
27
  end
37
28
 
38
29
  attr_reader :verdict
39
30
 
40
- def initialize(host_reports:, minimum_score:, maximum_cost:, fail_on_regression:)
31
+ def initialize(host_reports:, minimum_score:, maximum_cost:, fail_on_regression:,
32
+ on_unknown_pricing: :refuse)
41
33
  @host_reports = host_reports
42
34
  @minimum_score = minimum_score
43
35
  @maximum_cost = maximum_cost
44
36
  @fail_on_regression = fail_on_regression
37
+ @on_unknown_pricing = Eval::UnknownCostGate.validate!(on_unknown_pricing)
45
38
  @verdict = build_verdict
46
39
  end
47
40
 
@@ -49,11 +42,12 @@ module RubyLLM
49
42
 
50
43
  def build_verdict
51
44
  suite_cost = compute_suite_cost
45
+ cost_failure = cost_failure_message(suite_cost)
52
46
 
53
- if cost_exceeded?(suite_cost)
47
+ if cost_failure
54
48
  return Verdict.new(
55
49
  passed: false,
56
- abort_reason: cost_abort_message(suite_cost),
50
+ abort_reason: cost_failure,
57
51
  passed_reports: [],
58
52
  suite_cost: suite_cost
59
53
  )
@@ -72,8 +66,11 @@ module RubyLLM
72
66
  @host_reports.sum { |_host, report| report.total_cost }
73
67
  end
74
68
 
75
- def cost_exceeded?(suite_cost)
76
- @maximum_cost && suite_cost > @maximum_cost
69
+ def cost_failure_message(suite_cost)
70
+ return nil unless @maximum_cost
71
+ return cost_abort_message(suite_cost) if suite_cost > @maximum_cost
72
+
73
+ Eval::UnknownCostGate.check(@host_reports.map(&:last), mode: @on_unknown_pricing)
77
74
  end
78
75
 
79
76
  def cost_abort_message(suite_cost)
@@ -2,13 +2,15 @@
2
2
 
3
3
  require "rake"
4
4
  require "rake/tasklib"
5
+ require_relative "eval/unknown_cost_gate"
5
6
  require_relative "rake_task/suite_gate"
6
7
 
7
8
  module RubyLLM
8
9
  module Contract
9
10
  class RakeTask < ::Rake::TaskLib
10
11
  attr_accessor :name, :context, :fail_on_empty, :minimum_score, :maximum_cost,
11
- :eval_dirs, :save_baseline, :fail_on_regression, :track_history
12
+ :eval_dirs, :save_baseline, :fail_on_regression, :track_history,
13
+ :on_unknown_pricing
12
14
 
13
15
  def initialize(name = :"ruby_llm_contract:eval", &block)
14
16
  super()
@@ -18,10 +20,13 @@ module RubyLLM
18
20
  @minimum_score = nil # nil = require 100%; float = threshold
19
21
  @maximum_cost = nil # nil = no cost limit; float = budget cap (suite-level)
20
22
  @eval_dirs = [] # directories to load eval files from (non-Rails)
23
+ # With maximum_cost: :refuse fails on cases the budget cannot price; :warn gates priced cases only
24
+ @on_unknown_pricing = :refuse
21
25
  @save_baseline = false
22
26
  @fail_on_regression = false
23
27
  @track_history = false
24
28
  block&.call(self)
29
+ Eval::UnknownCostGate.validate!(@on_unknown_pricing)
25
30
  define_task
26
31
  end
27
32
 
@@ -44,7 +49,8 @@ module RubyLLM
44
49
  host_reports: host_reports,
45
50
  minimum_score: @minimum_score,
46
51
  maximum_cost: @maximum_cost,
47
- fail_on_regression: @fail_on_regression
52
+ fail_on_regression: @fail_on_regression,
53
+ on_unknown_pricing: @on_unknown_pricing
48
54
  )
49
55
 
50
56
  abort "\nEval suite FAILED: #{verdict.abort_reason}" unless verdict.passed?
@@ -164,7 +170,7 @@ module RubyLLM
164
170
  Array(JSON.parse(raw))
165
171
  else
166
172
  raw.split(",").map(&:strip).reject(&:empty?).map do |entry|
167
- model, effort = entry.split("@", 2)
173
+ model, effort = entry.split(Eval::RetryOptimizer::EFFORT_SEPARATOR, 2)
168
174
  config = { model: model.strip }
169
175
  config[:reasoning_effort] = effort.strip if effort && !effort.empty?
170
176
  config
@@ -191,7 +197,6 @@ module RubyLLM
191
197
  end
192
198
  end
193
199
 
194
- # Auto-register the optimize task when this file is loaded
195
200
  OptimizeRakeTask.new
196
201
  end
197
202
  end
@@ -64,6 +64,10 @@ RSpec::Matchers.define :pass_eval do |eval_name|
64
64
  @maximum_cost = cost
65
65
  end
66
66
 
67
+ chain :on_unknown_pricing do |mode|
68
+ @on_unknown_pricing = RubyLLM::Contract::Eval::UnknownCostGate.validate!(mode)
69
+ end
70
+
67
71
  chain :without_regressions do
68
72
  @check_regressions = true
69
73
  end
@@ -78,6 +82,9 @@ RSpec::Matchers.define :pass_eval do |eval_name|
78
82
  @context ||= {}
79
83
  @minimum_score ||= nil
80
84
  @maximum_cost ||= nil
85
+ @on_unknown_pricing ||= :refuse
86
+ @unknown_cost_failure = nil
87
+ @unknown_cost_only = false
81
88
  @check_regressions ||= false
82
89
  @comparison_step ||= nil
83
90
  @error = nil
@@ -97,7 +104,11 @@ RSpec::Matchers.define :pass_eval do |eval_name|
97
104
  @report.passed?
98
105
  end
99
106
 
100
- cost_ok = @maximum_cost ? @report.total_cost <= @maximum_cost : true
107
+ cost_ok = true
108
+ if @maximum_cost
109
+ @unknown_cost_failure = RubyLLM::Contract::Eval::UnknownCostGate.check([@report], mode: @on_unknown_pricing)
110
+ cost_ok = @report.total_cost <= @maximum_cost && @unknown_cost_failure.nil?
111
+ end
101
112
 
102
113
  regression_ok = if @prompt_diff
103
114
  @prompt_diff.safe_to_switch?
@@ -108,6 +119,7 @@ RSpec::Matchers.define :pass_eval do |eval_name|
108
119
  true
109
120
  end
110
121
 
122
+ @unknown_cost_only = score_ok && regression_ok && @report.total_cost <= @maximum_cost if @unknown_cost_failure
111
123
  score_ok && cost_ok && regression_ok
112
124
  rescue StandardError => e
113
125
  @error = e
@@ -115,71 +127,87 @@ RSpec::Matchers.define :pass_eval do |eval_name|
115
127
  end
116
128
 
117
129
  failure_message do
118
- if @prompt_diff && !@prompt_diff.safe_to_switch?
119
- msg = "expected #{@eval_name} eval to be safe to switch from baseline prompt\n"
120
-
121
- # Check empty sides first — most fundamental problem
122
- bl_empty = @prompt_diff.baseline_empty?
123
- cd_empty = @prompt_diff.candidate_empty?
124
- if bl_empty || cd_empty
125
- msg += " One side has no evaluated cases (all skipped or no adapter?)\n"
126
- if sample_response_only_compare?
127
- msg += " compare_with ignores sample_response; pass model: or with_context(adapter: ...)\n"
130
+ other_failure_message = lambda do
131
+ if @prompt_diff && !@prompt_diff.safe_to_switch?
132
+ msg = "expected #{@eval_name} eval to be safe to switch from baseline prompt\n"
133
+
134
+ # Check empty sides first — most fundamental problem
135
+ bl_empty = @prompt_diff.baseline_empty?
136
+ cd_empty = @prompt_diff.candidate_empty?
137
+ if bl_empty || cd_empty
138
+ msg += " One side has no evaluated cases (all skipped or no adapter?)\n"
139
+ if sample_response_only_compare?
140
+ msg += " compare_with ignores sample_response; pass model: or with_context(adapter: ...)\n"
141
+ end
142
+ msg += " Candidate score: #{@prompt_diff.candidate_score}, Baseline score: #{@prompt_diff.baseline_score}"
143
+ next msg
128
144
  end
129
- msg += " Candidate score: #{@prompt_diff.candidate_score}, Baseline score: #{@prompt_diff.baseline_score}"
130
- next msg
131
- end
132
145
 
133
- # Check dataset comparability — names, inputs, AND expected must match
134
- unless @prompt_diff.cases_comparable?
135
- unless @prompt_diff.case_names_match?
136
- mm = @prompt_diff.mismatched_cases
137
- msg += " Case set mismatch — candidate and baseline must have identical cases:\n"
138
- mm[:only_in_baseline].each { |n| msg += " only in baseline: #{n}\n" }
139
- mm[:only_in_candidate].each { |n| msg += " only in candidate: #{n}\n" }
140
- end
141
- @prompt_diff.input_mismatches.each do |m|
142
- msg += " Input mismatch for '#{m[:case]}' — same name but different inputs\n"
146
+ # Check dataset comparability — names, inputs, AND expected must match
147
+ unless @prompt_diff.cases_comparable?
148
+ unless @prompt_diff.case_names_match?
149
+ mm = @prompt_diff.mismatched_cases
150
+ msg += " Case set mismatch — candidate and baseline must have identical cases:\n"
151
+ mm[:only_in_baseline].each { |n| msg += " only in baseline: #{n}\n" }
152
+ mm[:only_in_candidate].each { |n| msg += " only in candidate: #{n}\n" }
153
+ end
154
+ @prompt_diff.input_mismatches.each do |m|
155
+ msg += " Input mismatch for '#{m[:case]}' — same name but different inputs\n"
156
+ end
157
+ @prompt_diff.expected_mismatches.each do |m|
158
+ msg += " Expected mismatch for '#{m[:case]}' — same name/input but different expected values\n"
159
+ end
160
+ next msg
143
161
  end
144
- @prompt_diff.expected_mismatches.each do |m|
145
- msg += " Expected mismatch for '#{m[:case]}' — same name/input but different expected values\n"
162
+
163
+ # Check per-case score regressions (even if global average is flat)
164
+ if @prompt_diff.score_regressions.any?
165
+ msg += " Per-case score regressions (#{@prompt_diff.score_regressions.length}):\n"
166
+ @prompt_diff.score_regressions.each do |r|
167
+ msg += " #{r[:case]}: #{r[:baseline_score]} -> #{r[:candidate_score]} (#{r[:delta]})\n"
168
+ end
169
+ msg += " Score delta: #{@prompt_diff.score_delta}"
170
+ next msg
146
171
  end
147
- next msg
148
- end
149
172
 
150
- # Check per-case score regressions (even if global average is flat)
151
- if @prompt_diff.score_regressions.any?
152
- msg += " Per-case score regressions (#{@prompt_diff.score_regressions.length}):\n"
153
- @prompt_diff.score_regressions.each do |r|
154
- msg += " #{r[:case]}: #{r[:baseline_score]} -> #{r[:candidate_score]} (#{r[:delta]})\n"
173
+ # Check pass/fail regressions and removed cases
174
+ removed = @prompt_diff.removed_passing_cases
175
+ reg_count = @prompt_diff.regressions.length + removed.length
176
+ msg += " Found #{reg_count} regression(s):\n"
177
+ @prompt_diff.regressions.each do |r|
178
+ msg += " #{r[:case]}: was PASS, now FAIL -- #{r[:detail]}\n"
179
+ end
180
+ removed.each do |name|
181
+ msg += " #{name}: REMOVED (was passing in baseline)\n"
155
182
  end
156
183
  msg += " Score delta: #{@prompt_diff.score_delta}"
157
184
  next msg
158
185
  end
159
186
 
160
- # Check pass/fail regressions and removed cases
161
- removed = @prompt_diff.removed_passing_cases
162
- reg_count = @prompt_diff.regressions.length + removed.length
163
- msg += " Found #{reg_count} regression(s):\n"
164
- @prompt_diff.regressions.each do |r|
165
- msg += " #{r[:case]}: was PASS, now FAIL -- #{r[:detail]}\n"
166
- end
167
- removed.each do |name|
168
- msg += " #{name}: REMOVED (was passing in baseline)\n"
187
+ msg = format_failure_message(@eval_name, @error, @report, @minimum_score, @maximum_cost)
188
+ if @diff&.regressed?
189
+ msg += "\n\nRegressions from baseline:\n"
190
+ @diff.regressions.each do |r|
191
+ msg += " #{r[:case]}: was PASS, now FAIL -- #{r[:detail]}\n"
192
+ end
193
+ @diff.skipped_passing_cases.each do |s|
194
+ msg += " #{s[:case]}: was PASS, now SKIPPED -- #{s[:detail]}\n"
195
+ end
196
+ msg += " Score delta: #{@diff.score_delta}"
169
197
  end
170
- msg += " Score delta: #{@prompt_diff.score_delta}"
171
- next msg
198
+ msg
172
199
  end
173
200
 
174
- msg = format_failure_message(@eval_name, @error, @report, @minimum_score, @maximum_cost)
175
- if @diff&.regressed?
176
- msg += "\n\nRegressions from baseline:\n"
177
- @diff.regressions.each do |r|
178
- msg += " #{r[:case]}: was PASS, now FAIL -- #{r[:detail]}\n"
179
- end
180
- msg += " Score delta: #{@diff.score_delta}"
201
+ # Cost first, as in the rake suite gate: an unpriced total cannot be trusted.
202
+ # An exception outranks it, and any other failure is still listed.
203
+ if @unknown_cost_failure && @error.nil?
204
+ unknown_cost = "expected #{@eval_name} eval to stay within maximum cost, but #{@unknown_cost_failure}"
205
+ next unknown_cost if @unknown_cost_only
206
+
207
+ next "#{unknown_cost}\n\n#{other_failure_message.call}"
181
208
  end
182
- msg
209
+
210
+ other_failure_message.call
183
211
  end
184
212
 
185
213
  failure_message_when_negated do
@@ -6,6 +6,11 @@ module RubyLLM
6
6
  class Base
7
7
  DEFAULT_OUTPUT_TOKENS = 256
8
8
 
9
+ # Eval::CaseExecutor substring-matches this to choose skip over raise.
10
+ # Reworded without the consumer, every skipped eval case becomes a hard
11
+ # failure of the whole run, so both ends read it from here.
12
+ NO_ADAPTER_MESSAGE = "No adapter configured"
13
+
9
14
  def self.inherited(subclass)
10
15
  super
11
16
  Contract.register_eval_host(subclass) if respond_to?(:eval_defined?) && eval_defined?
@@ -141,7 +146,10 @@ module RubyLLM
141
146
  def estimate_eval_cost_for_model(cases, model_name)
142
147
  cases.sum do |test_case|
143
148
  estimate = estimate_cost(input: test_case.input, model: model_name)
144
- estimate ? estimate[:estimated_cost] : 0.0
149
+ # Two misses floor to 0.0, not one: model absent from the registry
150
+ # (nil estimate), or present with unreadable pricing (hash whose
151
+ # estimated_cost is nil). Documented as a floor, not a fail-closed.
152
+ (estimate && estimate[:estimated_cost]) || 0.0
145
153
  end.round(6)
146
154
  end
147
155
 
@@ -225,8 +233,9 @@ module RubyLLM
225
233
  adapter = context[:adapter] || RubyLLM::Contract.configuration.default_adapter
226
234
  return adapter if adapter
227
235
 
228
- raise RubyLLM::Contract::Error, "No adapter configured. Set one with RubyLLM::Contract.configure " \
229
- "{ |c| c.default_adapter = ... } or pass context: { adapter: ... }"
236
+ raise RubyLLM::Contract::Error,
237
+ "#{NO_ADAPTER_MESSAGE}. Set one with RubyLLM::Contract.configure " \
238
+ "{ |c| c.default_adapter = ... } or pass context: { adapter: ... }"
230
239
  end
231
240
 
232
241
  # ADR-0021 deliverable 2: narrow ArgumentError rescue to DSL-setup phase only.
@@ -3,8 +3,6 @@
3
3
  module RubyLLM
4
4
  module Contract
5
5
  module Step
6
- # Extracted from Base to reduce class length.
7
- # DSL accessor methods for step definition (input_type, output_type, prompt, etc.).
8
6
  module Dsl # rubocop:disable Metrics/ModuleLength
9
7
  # Sentinel signalling "explicitly reset" (`some_attr(:default)`).
10
8
  # Distinguishes reset (lookup stops at this class, returns nil) from
@@ -65,8 +63,8 @@ module RubyLLM
65
63
 
66
64
  def output_schema(&block)
67
65
  if block
68
- require "ruby_llm/schema"
69
- @output_schema = ::RubyLLM::Schema.create(&block)
66
+ require "schematist"
67
+ @output_schema = ::Schematist::Schema.create(&block)
70
68
  elsif defined?(@output_schema)
71
69
  @output_schema
72
70
  elsif superclass.respond_to?(:output_schema)
@@ -3,8 +3,6 @@
3
3
  module RubyLLM
4
4
  module Contract
5
5
  module Step
6
- # Extracted from Runner to reduce class length.
7
- # Handles input token limit and cost limit checks.
8
6
  module LimitChecker
9
7
  private
10
8
 
@@ -3,8 +3,6 @@
3
3
  module RubyLLM
4
4
  module Contract
5
5
  module Step
6
- # Extracted from Base to reduce class length.
7
- # Handles retry logic: run_with_retry, build_retry_result, aggregate usage, build attempt entries.
8
6
  module RetryExecutor
9
7
  include Concerns::UsageAggregator
10
8
 
@@ -50,6 +50,25 @@ module RubyLLM
50
50
  )
51
51
  end
52
52
 
53
+ # True when a call ran on a named model, used tokens, and still got no
54
+ # cost, so any total built from this trace undercounts. A zero-token call
55
+ # (Test adapter, sample_response) costs 0 whatever the pricing. A retried
56
+ # step is judged per attempt: a priced subtotal must not hide an unpriced
57
+ # attempt, and the merged trace may have re-priced on the last model.
58
+ def cost_unknown?
59
+ if @attempts.is_a?(Array) && !@attempts.empty?
60
+ @attempts.any? { |attempt| self.class.unpriced?(attempt[:model], attempt[:usage], attempt[:cost]) }
61
+ else
62
+ self.class.unpriced?(@model, @usage, @cost)
63
+ end
64
+ end
65
+
66
+ def self.unpriced?(model, usage, cost)
67
+ return false unless model && cost.nil? && usage.is_a?(Hash)
68
+
69
+ ((usage[:input_tokens] || 0) + (usage[:output_tokens] || 0)).positive?
70
+ end
71
+
53
72
  def to_h
54
73
  { messages: @messages, model: @model, latency_ms: @latency_ms,
55
74
  usage: @usage, attempts: @attempts, cost: @cost }.compact
@@ -23,8 +23,6 @@ module RubyLLM
23
23
  module TokenEstimator
24
24
  CHARS_PER_TOKEN = 4
25
25
 
26
- # Heuristic estimate. Returns an integer token count.
27
- # See module docstring for accuracy caveats.
28
26
  def self.estimate(messages)
29
27
  return 0 unless messages.is_a?(Array)
30
28
 
@@ -2,6 +2,6 @@
2
2
 
3
3
  module RubyLLM
4
4
  module Contract
5
- VERSION = "0.10.6"
5
+ VERSION = "1.1.0"
6
6
  end
7
7
  end
@@ -21,8 +21,6 @@ module RubyLLM
21
21
  step_adapter_overrides.clear
22
22
  end
23
23
 
24
- # --- Eval host registry ---
25
-
26
24
  def register_eval_host(klass)
27
25
  eval_hosts << klass unless eval_hosts.include?(klass)
28
26
  end
@@ -101,9 +99,7 @@ module RubyLLM
101
99
 
102
100
  private
103
101
 
104
- # Filter stale hosts, deduplicate by name (last wins), prune registry in-place
105
102
  def live_eval_hosts
106
- # Remove hosts without evals
107
103
  @eval_hosts&.reject! { |h| !h.respond_to?(:eval_defined?) || !h.eval_defined? }
108
104
 
109
105
  # Deduplicate: if two classes share a name (reload), keep the latest