ruby_llm-contract 1.0.0 → 1.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. checksums.yaml +4 -4
  2. data/.ruby-version +1 -0
  3. data/CHANGELOG.md +287 -224
  4. data/README.md +1 -1
  5. data/docs/guide/getting_started.md +3 -1
  6. data/docs/guide/testing.md +1 -1
  7. data/lib/ruby_llm/contract/concerns/eval_host.rb +2 -2
  8. data/lib/ruby_llm/contract/cost_calculator.rb +5 -4
  9. data/lib/ruby_llm/contract/eval/aggregated_report.rb +4 -0
  10. data/lib/ruby_llm/contract/eval/baseline_diff.rb +19 -2
  11. data/lib/ruby_llm/contract/eval/case_executor.rb +1 -1
  12. data/lib/ruby_llm/contract/eval/case_result.rb +12 -3
  13. data/lib/ruby_llm/contract/eval/case_result_builder.rb +2 -1
  14. data/lib/ruby_llm/contract/eval/evaluator/proc_evaluator.rb +6 -2
  15. data/lib/ruby_llm/contract/eval/prompt_diff_serializer.rb +1 -1
  16. data/lib/ruby_llm/contract/eval/recommender.rb +2 -2
  17. data/lib/ruby_llm/contract/eval/report.rb +5 -2
  18. data/lib/ruby_llm/contract/eval/report_stats.rb +7 -2
  19. data/lib/ruby_llm/contract/eval/report_storage.rb +6 -6
  20. data/lib/ruby_llm/contract/eval/step_expectation_applier.rb +2 -1
  21. data/lib/ruby_llm/contract/eval/unknown_cost_gate.rb +32 -0
  22. data/lib/ruby_llm/contract/eval.rb +1 -0
  23. data/lib/ruby_llm/contract/minitest.rb +6 -1
  24. data/lib/ruby_llm/contract/pipeline/trace.rb +7 -0
  25. data/lib/ruby_llm/contract/railtie.rb +1 -1
  26. data/lib/ruby_llm/contract/rake_task/suite_gate.rb +19 -10
  27. data/lib/ruby_llm/contract/rake_task.rb +9 -3
  28. data/lib/ruby_llm/contract/rspec/pass_eval.rb +80 -52
  29. data/lib/ruby_llm/contract/step/base.rb +6 -4
  30. data/lib/ruby_llm/contract/step/dsl.rb +5 -14
  31. data/lib/ruby_llm/contract/step/runner_config.rb +2 -2
  32. data/lib/ruby_llm/contract/step/trace.rb +19 -0
  33. data/lib/ruby_llm/contract/unknown_policy.rb +21 -0
  34. data/lib/ruby_llm/contract/version.rb +1 -1
  35. data/lib/ruby_llm/contract.rb +17 -4
  36. metadata +4 -1
@@ -64,6 +64,10 @@ RSpec::Matchers.define :pass_eval do |eval_name|
64
64
  @maximum_cost = cost
65
65
  end
66
66
 
67
+ chain :on_unknown_pricing do |mode|
68
+ @on_unknown_pricing = RubyLLM::Contract::UnknownPolicy.validate!("on_unknown_pricing", mode)
69
+ end
70
+
67
71
  chain :without_regressions do
68
72
  @check_regressions = true
69
73
  end
@@ -78,6 +82,9 @@ RSpec::Matchers.define :pass_eval do |eval_name|
78
82
  @context ||= {}
79
83
  @minimum_score ||= nil
80
84
  @maximum_cost ||= nil
85
+ @on_unknown_pricing ||= RubyLLM::Contract::UnknownPolicy::DEFAULT
86
+ @unknown_cost_failure = nil
87
+ @unknown_cost_only = false
81
88
  @check_regressions ||= false
82
89
  @comparison_step ||= nil
83
90
  @error = nil
@@ -97,7 +104,11 @@ RSpec::Matchers.define :pass_eval do |eval_name|
97
104
  @report.passed?
98
105
  end
99
106
 
100
- cost_ok = @maximum_cost ? @report.total_cost <= @maximum_cost : true
107
+ cost_ok = true
108
+ if @maximum_cost
109
+ @unknown_cost_failure = RubyLLM::Contract::Eval::UnknownCostGate.check([@report], mode: @on_unknown_pricing)
110
+ cost_ok = @report.total_cost <= @maximum_cost && @unknown_cost_failure.nil?
111
+ end
101
112
 
102
113
  regression_ok = if @prompt_diff
103
114
  @prompt_diff.safe_to_switch?
@@ -108,6 +119,7 @@ RSpec::Matchers.define :pass_eval do |eval_name|
108
119
  true
109
120
  end
110
121
 
122
+ @unknown_cost_only = score_ok && regression_ok && @report.total_cost <= @maximum_cost if @unknown_cost_failure
111
123
  score_ok && cost_ok && regression_ok
112
124
  rescue StandardError => e
113
125
  @error = e
@@ -115,71 +127,87 @@ RSpec::Matchers.define :pass_eval do |eval_name|
115
127
  end
116
128
 
117
129
  failure_message do
118
- if @prompt_diff && !@prompt_diff.safe_to_switch?
119
- msg = "expected #{@eval_name} eval to be safe to switch from baseline prompt\n"
120
-
121
- # Check empty sides first — most fundamental problem
122
- bl_empty = @prompt_diff.baseline_empty?
123
- cd_empty = @prompt_diff.candidate_empty?
124
- if bl_empty || cd_empty
125
- msg += " One side has no evaluated cases (all skipped or no adapter?)\n"
126
- if sample_response_only_compare?
127
- msg += " compare_with ignores sample_response; pass model: or with_context(adapter: ...)\n"
130
+ other_failure_message = lambda do
131
+ if @prompt_diff && !@prompt_diff.safe_to_switch?
132
+ msg = "expected #{@eval_name} eval to be safe to switch from baseline prompt\n"
133
+
134
+ # Check empty sides first — most fundamental problem
135
+ bl_empty = @prompt_diff.baseline_empty?
136
+ cd_empty = @prompt_diff.candidate_empty?
137
+ if bl_empty || cd_empty
138
+ msg += " One side has no evaluated cases (all skipped or no adapter?)\n"
139
+ if sample_response_only_compare?
140
+ msg += " compare_with ignores sample_response; pass model: or with_context(adapter: ...)\n"
141
+ end
142
+ msg += " Candidate score: #{@prompt_diff.candidate_score}, Baseline score: #{@prompt_diff.baseline_score}"
143
+ next msg
128
144
  end
129
- msg += " Candidate score: #{@prompt_diff.candidate_score}, Baseline score: #{@prompt_diff.baseline_score}"
130
- next msg
131
- end
132
145
 
133
- # Check dataset comparability — names, inputs, AND expected must match
134
- unless @prompt_diff.cases_comparable?
135
- unless @prompt_diff.case_names_match?
136
- mm = @prompt_diff.mismatched_cases
137
- msg += " Case set mismatch — candidate and baseline must have identical cases:\n"
138
- mm[:only_in_baseline].each { |n| msg += " only in baseline: #{n}\n" }
139
- mm[:only_in_candidate].each { |n| msg += " only in candidate: #{n}\n" }
140
- end
141
- @prompt_diff.input_mismatches.each do |m|
142
- msg += " Input mismatch for '#{m[:case]}' — same name but different inputs\n"
146
+ # Check dataset comparability — names, inputs, AND expected must match
147
+ unless @prompt_diff.cases_comparable?
148
+ unless @prompt_diff.case_names_match?
149
+ mm = @prompt_diff.mismatched_cases
150
+ msg += " Case set mismatch — candidate and baseline must have identical cases:\n"
151
+ mm[:only_in_baseline].each { |n| msg += " only in baseline: #{n}\n" }
152
+ mm[:only_in_candidate].each { |n| msg += " only in candidate: #{n}\n" }
153
+ end
154
+ @prompt_diff.input_mismatches.each do |m|
155
+ msg += " Input mismatch for '#{m[:case]}' — same name but different inputs\n"
156
+ end
157
+ @prompt_diff.expected_mismatches.each do |m|
158
+ msg += " Expected mismatch for '#{m[:case]}' — same name/input but different expected values\n"
159
+ end
160
+ next msg
143
161
  end
144
- @prompt_diff.expected_mismatches.each do |m|
145
- msg += " Expected mismatch for '#{m[:case]}' — same name/input but different expected values\n"
162
+
163
+ # Check per-case score regressions (even if global average is flat)
164
+ if @prompt_diff.score_regressions.any?
165
+ msg += " Per-case score regressions (#{@prompt_diff.score_regressions.length}):\n"
166
+ @prompt_diff.score_regressions.each do |r|
167
+ msg += " #{r[:case]}: #{r[:baseline_score]} -> #{r[:candidate_score]} (#{r[:delta]})\n"
168
+ end
169
+ msg += " Score delta: #{@prompt_diff.score_delta}"
170
+ next msg
146
171
  end
147
- next msg
148
- end
149
172
 
150
- # Check per-case score regressions (even if global average is flat)
151
- if @prompt_diff.score_regressions.any?
152
- msg += " Per-case score regressions (#{@prompt_diff.score_regressions.length}):\n"
153
- @prompt_diff.score_regressions.each do |r|
154
- msg += " #{r[:case]}: #{r[:baseline_score]} -> #{r[:candidate_score]} (#{r[:delta]})\n"
173
+ # Check pass/fail regressions and removed cases
174
+ removed = @prompt_diff.removed_passing_cases
175
+ reg_count = @prompt_diff.regressions.length + removed.length
176
+ msg += " Found #{reg_count} regression(s):\n"
177
+ @prompt_diff.regressions.each do |r|
178
+ msg += " #{r[:case]}: was PASS, now FAIL -- #{r[:detail]}\n"
179
+ end
180
+ removed.each do |name|
181
+ msg += " #{name}: REMOVED (was passing in baseline)\n"
155
182
  end
156
183
  msg += " Score delta: #{@prompt_diff.score_delta}"
157
184
  next msg
158
185
  end
159
186
 
160
- # Check pass/fail regressions and removed cases
161
- removed = @prompt_diff.removed_passing_cases
162
- reg_count = @prompt_diff.regressions.length + removed.length
163
- msg += " Found #{reg_count} regression(s):\n"
164
- @prompt_diff.regressions.each do |r|
165
- msg += " #{r[:case]}: was PASS, now FAIL -- #{r[:detail]}\n"
166
- end
167
- removed.each do |name|
168
- msg += " #{name}: REMOVED (was passing in baseline)\n"
187
+ msg = format_failure_message(@eval_name, @error, @report, @minimum_score, @maximum_cost)
188
+ if @diff&.regressed?
189
+ msg += "\n\nRegressions from baseline:\n"
190
+ @diff.regressions.each do |r|
191
+ msg += " #{r[:case]}: was PASS, now FAIL -- #{r[:detail]}\n"
192
+ end
193
+ @diff.skipped_passing_cases.each do |s|
194
+ msg += " #{s[:case]}: was PASS, now SKIPPED -- #{s[:detail]}\n"
195
+ end
196
+ msg += " Score delta: #{@diff.score_delta}"
169
197
  end
170
- msg += " Score delta: #{@prompt_diff.score_delta}"
171
- next msg
198
+ msg
172
199
  end
173
200
 
174
- msg = format_failure_message(@eval_name, @error, @report, @minimum_score, @maximum_cost)
175
- if @diff&.regressed?
176
- msg += "\n\nRegressions from baseline:\n"
177
- @diff.regressions.each do |r|
178
- msg += " #{r[:case]}: was PASS, now FAIL -- #{r[:detail]}\n"
179
- end
180
- msg += " Score delta: #{@diff.score_delta}"
201
+ # Cost first, as in the rake suite gate: an unpriced total cannot be trusted.
202
+ # An exception outranks it, and any other failure is still listed.
203
+ if @unknown_cost_failure && @error.nil?
204
+ unknown_cost = "expected #{@eval_name} eval to stay within maximum cost, but #{@unknown_cost_failure}"
205
+ next unknown_cost if @unknown_cost_only
206
+
207
+ next "#{unknown_cost}\n\n#{other_failure_message.call}"
181
208
  end
182
- msg
209
+
210
+ other_failure_message.call
183
211
  end
184
212
 
185
213
  failure_message_when_negated do
@@ -90,8 +90,10 @@ module RubyLLM
90
90
  ).call
91
91
  end
92
92
 
93
- KNOWN_CONTEXT_KEYS = %i[adapter model temperature max_tokens provider assume_model_exists
94
- reasoning_effort retry_policy_override attachment].freeze
93
+ # Forwarded to the adapter as-is. A key known but not forwarded would be
94
+ # silently dropped, so the known list is built from this one.
95
+ ADAPTER_CONTEXT_KEYS = %i[provider assume_model_exists max_tokens reasoning_effort attachment].freeze
96
+ KNOWN_CONTEXT_KEYS = (%i[adapter model temperature retry_policy_override] + ADAPTER_CONTEXT_KEYS).freeze
95
97
 
96
98
  include Concerns::ContextHelpers
97
99
 
@@ -133,7 +135,7 @@ module RubyLLM
133
135
  estimate = attachment_token_estimate if respond_to?(:attachment_token_estimate)
134
136
  return [estimate, false] unless estimate.nil?
135
137
 
136
- mode = respond_to?(:on_unknown_attachment_size) ? on_unknown_attachment_size : :refuse
138
+ mode = respond_to?(:on_unknown_attachment_size) ? on_unknown_attachment_size : UnknownPolicy::DEFAULT
137
139
  if mode == :warn
138
140
  warn "[ruby_llm-contract] attachment present but attachment_token_estimate not " \
139
141
  "declared on #{name || self} — estimate_cost proceeds without attachment cost"
@@ -193,7 +195,7 @@ module RubyLLM
193
195
 
194
196
  def runtime_settings(context)
195
197
  policy = context.key?(:retry_policy_override) ? context[:retry_policy_override] : retry_policy
196
- extra = context.slice(:provider, :assume_model_exists, :max_tokens, :reasoning_effort, :attachment)
198
+ extra = context.slice(*ADAPTER_CONTEXT_KEYS)
197
199
 
198
200
  # Always pass the class-level `thinking` config to the adapter when
199
201
  # set, so fields like `budget` survive a per-call `reasoning_effort`
@@ -151,12 +151,10 @@ module RubyLLM
151
151
  if amount
152
152
  validate_positive!("max_cost", amount)
153
153
 
154
- if on_unknown_pricing && !%i[refuse warn].include?(on_unknown_pricing)
155
- raise ArgumentError, "on_unknown_pricing must be :refuse or :warn, got #{on_unknown_pricing.inspect}"
156
- end
154
+ UnknownPolicy.validate!("on_unknown_pricing", on_unknown_pricing) if on_unknown_pricing
157
155
 
158
156
  @max_cost = amount
159
- @on_unknown_pricing = on_unknown_pricing || :refuse
157
+ @on_unknown_pricing = on_unknown_pricing || UnknownPolicy::DEFAULT
160
158
  return @max_cost
161
159
  end
162
160
 
@@ -164,7 +162,7 @@ module RubyLLM
164
162
  end
165
163
 
166
164
  def on_unknown_pricing
167
- inherited_value(:on_unknown_pricing) || :refuse
165
+ inherited_value(:on_unknown_pricing) || UnknownPolicy::DEFAULT
168
166
  end
169
167
 
170
168
  def attachment_token_estimate(n = nil)
@@ -182,16 +180,9 @@ module RubyLLM
182
180
  end
183
181
 
184
182
  def on_unknown_attachment_size(mode = nil)
185
- if mode
186
- unless %i[refuse warn].include?(mode)
187
- raise ArgumentError,
188
- "on_unknown_attachment_size must be :refuse or :warn, got #{mode.inspect}"
189
- end
190
-
191
- return @on_unknown_attachment_size = mode
192
- end
183
+ return @on_unknown_attachment_size = UnknownPolicy.validate!("on_unknown_attachment_size", mode) if mode
193
184
 
194
- inherited_value(:on_unknown_attachment_size) || :refuse
185
+ inherited_value(:on_unknown_attachment_size) || UnknownPolicy::DEFAULT
195
186
  end
196
187
 
197
188
  def model(name = nil)
@@ -28,8 +28,8 @@ module RubyLLM
28
28
  def self.build(input_type:, output_type:, prompt_block:, contract_definition:,
29
29
  adapter:, model:,
30
30
  output_schema: nil, max_output: nil,
31
- max_input: nil, max_cost: nil, on_unknown_pricing: :refuse,
32
- attachment_token_estimate: nil, on_unknown_attachment_size: :refuse,
31
+ max_input: nil, max_cost: nil, on_unknown_pricing: UnknownPolicy::DEFAULT,
32
+ attachment_token_estimate: nil, on_unknown_attachment_size: UnknownPolicy::DEFAULT,
33
33
  temperature: nil, extra_options: {}, observers: [])
34
34
  new(
35
35
  input_type: input_type, output_type: output_type,
@@ -50,6 +50,25 @@ module RubyLLM
50
50
  )
51
51
  end
52
52
 
53
+ # True when a call ran on a named model, used tokens, and still got no
54
+ # cost, so any total built from this trace undercounts. A zero-token call
55
+ # (Test adapter, sample_response) costs 0 whatever the pricing. A retried
56
+ # step is judged per attempt: a priced subtotal must not hide an unpriced
57
+ # attempt, and the merged trace may have re-priced on the last model.
58
+ def cost_unknown?
59
+ if @attempts.is_a?(Array) && !@attempts.empty?
60
+ @attempts.any? { |attempt| self.class.unpriced?(attempt[:model], attempt[:usage], attempt[:cost]) }
61
+ else
62
+ self.class.unpriced?(@model, @usage, @cost)
63
+ end
64
+ end
65
+
66
+ def self.unpriced?(model, usage, cost)
67
+ return false unless model && cost.nil? && usage.is_a?(Hash)
68
+
69
+ ((usage[:input_tokens] || 0) + (usage[:output_tokens] || 0)).positive?
70
+ end
71
+
53
72
  def to_h
54
73
  { messages: @messages, model: @model, latency_ms: @latency_ms,
55
74
  usage: @usage, attempts: @attempts, cost: @cost }.compact
@@ -0,0 +1,21 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Contract
5
+ # What a limit does when the number it needs is unknown: model pricing for
6
+ # max_cost/maximum_cost, attachment size for max_input. One vocabulary for
7
+ # every such option, step-level and eval-level, so a new mode or a changed
8
+ # default reaches all of them. Standalone: the rake task loads it before the
9
+ # rest of the gem to validate its settings at definition time.
10
+ module UnknownPolicy
11
+ MODES = %i[refuse warn].freeze
12
+ DEFAULT = :refuse
13
+
14
+ def self.validate!(option, mode)
15
+ return mode if MODES.include?(mode)
16
+
17
+ raise ArgumentError, "#{option} must be #{MODES.map(&:inspect).join(" or ")}, got #{mode.inspect}"
18
+ end
19
+ end
20
+ end
21
+ end
@@ -2,6 +2,6 @@
2
2
 
3
3
  module RubyLLM
4
4
  module Contract
5
- VERSION = "1.0.0"
5
+ VERSION = "1.1.1"
6
6
  end
7
7
  end
@@ -2,10 +2,19 @@
2
2
 
3
3
  require_relative "contract/version"
4
4
  require_relative "contract/errors"
5
+ require_relative "contract/unknown_policy"
5
6
  require_relative "contract/types"
6
7
 
7
8
  module RubyLLM
8
9
  module Contract
10
+ # Rails dirs holding Step classes. Each one's eval/ subdir holds define_eval
11
+ # files, which define no constant, so Zeitwerk must ignore exactly those.
12
+ RAILS_CONTRACT_DIRS = %w[app/contracts app/steps].freeze
13
+ RAILS_EVAL_DIRS = RAILS_CONTRACT_DIRS.map { |dir| "#{dir}/eval" }.freeze
14
+
15
+ # Set while load_evals! runs: redefining an eval is then a reload, not a mistake.
16
+ RELOADING_KEY = :ruby_llm_contract_reloading
17
+
9
18
  class << self
10
19
  def configuration
11
20
  @configuration ||= Configuration.new
@@ -51,7 +60,7 @@ module RubyLLM
51
60
  def load_evals!(*dirs)
52
61
  dirs = dirs.flatten.compact
53
62
  if dirs.empty? && defined?(::Rails)
54
- dirs = %w[app/steps/eval app/contracts/eval].filter_map do |path|
63
+ dirs = RAILS_EVAL_DIRS.filter_map do |path|
55
64
  full = ::Rails.root.join(path)
56
65
  full.to_s if full.exist?
57
66
  end
@@ -64,7 +73,7 @@ module RubyLLM
64
73
  eager_load_contract_dirs! if defined?(::Rails)
65
74
 
66
75
  # Clear file-sourced evals ONCE, then load ALL dirs.
67
- Thread.current[:ruby_llm_contract_reloading] = true
76
+ Thread.current[RELOADING_KEY] = true
68
77
  eval_hosts.each do |host|
69
78
  host.clear_file_sourced_evals! if host.respond_to?(:clear_file_sourced_evals!)
70
79
  end
@@ -73,7 +82,11 @@ module RubyLLM
73
82
  Dir[File.join(d, "**", "*_eval.rb")].each { |f| load f }
74
83
  end
75
84
  ensure
76
- Thread.current[:ruby_llm_contract_reloading] = false
85
+ Thread.current[RELOADING_KEY] = false
86
+ end
87
+
88
+ def reloading?
89
+ Thread.current[RELOADING_KEY] ? true : false
77
90
  end
78
91
 
79
92
  def normalize_candidate_config(entry)
@@ -111,7 +124,7 @@ module RubyLLM
111
124
  end
112
125
 
113
126
  def eager_load_contract_dirs!
114
- %w[app/contracts app/steps].each do |path|
127
+ RAILS_CONTRACT_DIRS.each do |path|
115
128
  full = ::Rails.root.join(path)
116
129
  next unless full.exist?
117
130
 
metadata CHANGED
@@ -1,7 +1,7 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: ruby_llm-contract
3
3
  version: !ruby/object:Gem::Version
4
- version: 1.0.0
4
+ version: 1.1.1
5
5
  platform: ruby
6
6
  authors:
7
7
  - Justyna
@@ -59,6 +59,7 @@ executables: []
59
59
  extensions: []
60
60
  extra_rdoc_files: []
61
61
  files:
62
+ - ".ruby-version"
62
63
  - CHANGELOG.md
63
64
  - LICENSE
64
65
  - README.md
@@ -152,6 +153,7 @@ files:
152
153
  - lib/ruby_llm/contract/eval/step_expectation_applier.rb
153
154
  - lib/ruby_llm/contract/eval/step_result_normalizer.rb
154
155
  - lib/ruby_llm/contract/eval/trait_evaluator.rb
156
+ - lib/ruby_llm/contract/eval/unknown_cost_gate.rb
155
157
  - lib/ruby_llm/contract/minitest.rb
156
158
  - lib/ruby_llm/contract/pipeline.rb
157
159
  - lib/ruby_llm/contract/pipeline/base.rb
@@ -191,6 +193,7 @@ files:
191
193
  - lib/ruby_llm/contract/step/trace.rb
192
194
  - lib/ruby_llm/contract/token_estimator.rb
193
195
  - lib/ruby_llm/contract/types.rb
196
+ - lib/ruby_llm/contract/unknown_policy.rb
194
197
  - lib/ruby_llm/contract/version.rb
195
198
  - ruby_llm-contract.gemspec
196
199
  homepage: https://github.com/justi/ruby_llm-contract