ruby_llm-contract 0.10.6 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.ruby-version +1 -0
- data/CHANGELOG.md +354 -211
- data/README.md +12 -2
- data/docs/guide/getting_started.md +3 -1
- data/docs/guide/llm_judge.md +1 -1
- data/docs/guide/multimodal_input.md +4 -3
- data/docs/guide/output_schema.md +2 -2
- data/docs/guide/testing.md +1 -1
- data/examples/README.md +1 -1
- data/lib/ruby_llm/contract/adapters/ruby_llm.rb +13 -9
- data/lib/ruby_llm/contract/concerns/context_helpers.rb +0 -2
- data/lib/ruby_llm/contract/concerns/deep_symbolize.rb +0 -1
- data/lib/ruby_llm/contract/contract/parser.rb +0 -3
- data/lib/ruby_llm/contract/contract/schema_validator/bound_rule.rb +0 -1
- data/lib/ruby_llm/contract/contract/schema_validator/enum_rule.rb +0 -1
- data/lib/ruby_llm/contract/contract/schema_validator/node.rb +0 -4
- data/lib/ruby_llm/contract/contract/schema_validator/scalar_rules.rb +0 -1
- data/lib/ruby_llm/contract/contract/schema_validator/type_rule.rb +0 -1
- data/lib/ruby_llm/contract/cost_calculator.rb +28 -2
- data/lib/ruby_llm/contract/eval/aggregated_report.rb +4 -0
- data/lib/ruby_llm/contract/eval/baseline_diff.rb +22 -3
- data/lib/ruby_llm/contract/eval/candidate_label.rb +31 -0
- data/lib/ruby_llm/contract/eval/case_executor.rb +2 -2
- data/lib/ruby_llm/contract/eval/case_result.rb +17 -3
- data/lib/ruby_llm/contract/eval/case_result_builder.rb +2 -1
- data/lib/ruby_llm/contract/eval/contract_detail_builder.rb +0 -2
- data/lib/ruby_llm/contract/eval/model_comparison.rb +3 -2
- data/lib/ruby_llm/contract/eval/pipeline_result_adapter.rb +1 -2
- data/lib/ruby_llm/contract/eval/prompt_diff_comparator.rb +0 -1
- data/lib/ruby_llm/contract/eval/prompt_diff_presenter.rb +0 -1
- data/lib/ruby_llm/contract/eval/prompt_diff_serializer.rb +0 -1
- data/lib/ruby_llm/contract/eval/report.rb +1 -1
- data/lib/ruby_llm/contract/eval/report_presenter.rb +0 -1
- data/lib/ruby_llm/contract/eval/report_stats.rb +5 -1
- data/lib/ruby_llm/contract/eval/report_storage.rb +0 -1
- data/lib/ruby_llm/contract/eval/retry_optimizer.rb +10 -20
- data/lib/ruby_llm/contract/eval/step_expectation_applier.rb +2 -1
- data/lib/ruby_llm/contract/eval/trait_evaluator.rb +0 -2
- data/lib/ruby_llm/contract/eval/unknown_cost_gate.rb +40 -0
- data/lib/ruby_llm/contract/eval.rb +2 -0
- data/lib/ruby_llm/contract/minitest.rb +6 -1
- data/lib/ruby_llm/contract/pipeline/runner.rb +0 -1
- data/lib/ruby_llm/contract/pipeline/trace.rb +7 -0
- data/lib/ruby_llm/contract/rake_task/suite_gate.rb +18 -21
- data/lib/ruby_llm/contract/rake_task.rb +9 -4
- data/lib/ruby_llm/contract/rspec/pass_eval.rb +80 -52
- data/lib/ruby_llm/contract/step/base.rb +12 -3
- data/lib/ruby_llm/contract/step/dsl.rb +2 -4
- data/lib/ruby_llm/contract/step/limit_checker.rb +0 -2
- data/lib/ruby_llm/contract/step/retry_executor.rb +0 -2
- data/lib/ruby_llm/contract/step/trace.rb +19 -0
- data/lib/ruby_llm/contract/token_estimator.rb +0 -2
- data/lib/ruby_llm/contract/version.rb +1 -1
- data/lib/ruby_llm/contract.rb +0 -4
- data/ruby_llm-contract.gemspec +6 -5
- metadata +9 -11
- data/.rubocop.yml +0 -58
- data/Gemfile +0 -13
- data/Gemfile.lock +0 -278
- data/Rakefile +0 -8
|
@@ -13,6 +13,10 @@ module RubyLLM
|
|
|
13
13
|
# result.print_summary
|
|
14
14
|
# result.to_dsl # => copy-paste retry_policy
|
|
15
15
|
class RetryOptimizer
|
|
16
|
+
# Documented CANDIDATES= shorthand, parsed by
|
|
17
|
+
# OptimizeRakeTask#parse_candidates and rendered in this table's headers.
|
|
18
|
+
EFFORT_SEPARATOR = "@"
|
|
19
|
+
|
|
16
20
|
Result = Struct.new(:step_name, :eval_names, :candidate_labels, :score_matrix,
|
|
17
21
|
:constraining_eval, :chain, :chain_details, keyword_init: true) do
|
|
18
22
|
# Terminology alias — `hardest_eval` is the narrative name used in docs;
|
|
@@ -80,11 +84,10 @@ module RubyLLM
|
|
|
80
84
|
end
|
|
81
85
|
|
|
82
86
|
def short_candidate_label(label)
|
|
83
|
-
label
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
.sub(")", "")
|
|
87
|
+
config = CandidateLabel.parse(label)
|
|
88
|
+
model = config[:model].sub("gpt-5-", "").sub("gpt-4.1", "4.1")
|
|
89
|
+
effort = config[:reasoning_effort]
|
|
90
|
+
effort ? "#{model}#{EFFORT_SEPARATOR}#{effort}" : model
|
|
88
91
|
end
|
|
89
92
|
|
|
90
93
|
def print_dsl(io)
|
|
@@ -176,11 +179,9 @@ module RubyLLM
|
|
|
176
179
|
def build_chain(matrix, labels, evals)
|
|
177
180
|
total = evals.size
|
|
178
181
|
|
|
179
|
-
# Find cheapest model that passes every eval — the safe fallback.
|
|
180
182
|
safe_fallback = labels.find { |l| evals.all? { |e| (matrix.dig(e, l) || 0) >= @min_score } }
|
|
181
183
|
return [[], []] unless safe_fallback
|
|
182
184
|
|
|
183
|
-
# Prepend cheaper models that pass a strict subset.
|
|
184
185
|
chain = []
|
|
185
186
|
details = []
|
|
186
187
|
covered_evals = Set.new
|
|
@@ -193,27 +194,16 @@ module RubyLLM
|
|
|
193
194
|
next if new_additions.empty?
|
|
194
195
|
|
|
195
196
|
covered_evals.merge(new_additions)
|
|
196
|
-
chain <<
|
|
197
|
+
chain << CandidateLabel.parse(label)
|
|
197
198
|
details << { label: label, passes: new_additions.size, cost: label }
|
|
198
199
|
end
|
|
199
200
|
|
|
200
|
-
|
|
201
|
-
chain << parse_label_to_config(safe_fallback)
|
|
201
|
+
chain << CandidateLabel.parse(safe_fallback)
|
|
202
202
|
details << { label: safe_fallback, passes: total, cost: safe_fallback }
|
|
203
203
|
|
|
204
204
|
[chain, details]
|
|
205
205
|
end
|
|
206
206
|
|
|
207
|
-
def parse_label_to_config(label)
|
|
208
|
-
if label.match?(/\(effort: (\w+)\)/)
|
|
209
|
-
model = label.sub(/\s*\(effort:.*/, "").strip
|
|
210
|
-
effort = label.match(/\(effort: (\w+)\)/)[1]
|
|
211
|
-
{ model: model, reasoning_effort: effort }
|
|
212
|
-
else
|
|
213
|
-
{ model: label }
|
|
214
|
-
end
|
|
215
|
-
end
|
|
216
|
-
|
|
217
207
|
def empty_result(evals)
|
|
218
208
|
Result.new(
|
|
219
209
|
step_name: @step.name || @step.to_s,
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
module Contract
|
|
5
|
+
module Eval
|
|
6
|
+
# Suite-level counterpart of the step-level `max_cost ... on_unknown_pricing:`.
|
|
7
|
+
# A report's total_cost counts an unpriced case as 0.0, so a maximum_cost
|
|
8
|
+
# gate that only compares totals passes whatever those cases really cost.
|
|
9
|
+
# Shared by the rake task, the pass_eval matcher and assert_eval_passes.
|
|
10
|
+
module UnknownCostGate
|
|
11
|
+
MODES = %i[refuse warn].freeze
|
|
12
|
+
|
|
13
|
+
def self.validate!(mode)
|
|
14
|
+
return mode if MODES.include?(mode)
|
|
15
|
+
|
|
16
|
+
raise ArgumentError, "on_unknown_pricing must be :refuse or :warn, got #{mode.inspect}"
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
# Returns the failure message under :refuse, nil when the gate holds.
|
|
20
|
+
def self.check(reports, mode:)
|
|
21
|
+
names = reports.flat_map(&:unknown_cost_results).map(&:name).uniq
|
|
22
|
+
return nil if names.empty?
|
|
23
|
+
|
|
24
|
+
if mode == :warn
|
|
25
|
+
warn "[ruby_llm-contract] #{describe(names)} - maximum_cost checked against priced cases only"
|
|
26
|
+
return nil
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
"#{describe(names)}. Register pricing via CostCalculator.register_model " \
|
|
30
|
+
"or set on_unknown_pricing to :warn to check maximum_cost against priced cases only."
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
def self.describe(names)
|
|
34
|
+
"#{names.length} case(s) have unknown cost (model has no pricing data): #{names.join(", ")}"
|
|
35
|
+
end
|
|
36
|
+
private_class_method :describe
|
|
37
|
+
end
|
|
38
|
+
end
|
|
39
|
+
end
|
|
40
|
+
end
|
|
@@ -22,7 +22,9 @@ require_relative "eval/report_presenter"
|
|
|
22
22
|
require_relative "eval/report_storage"
|
|
23
23
|
require_relative "eval/report"
|
|
24
24
|
require_relative "eval/aggregated_report"
|
|
25
|
+
require_relative "eval/unknown_cost_gate"
|
|
25
26
|
require_relative "eval/eval_definition"
|
|
27
|
+
require_relative "eval/candidate_label"
|
|
26
28
|
require_relative "eval/model_comparison"
|
|
27
29
|
require_relative "eval/baseline_diff"
|
|
28
30
|
require_relative "eval/prompt_diff_serializer"
|
|
@@ -30,7 +30,9 @@ module RubyLLM
|
|
|
30
30
|
refute result.ok?, msg || "Expected step result NOT to satisfy contract, but it passed"
|
|
31
31
|
end
|
|
32
32
|
|
|
33
|
-
def assert_eval_passes(step, eval_name, minimum_score: nil, maximum_cost: nil, context: {}, msg: nil
|
|
33
|
+
def assert_eval_passes(step, eval_name, minimum_score: nil, maximum_cost: nil, context: {}, msg: nil,
|
|
34
|
+
on_unknown_pricing: :refuse)
|
|
35
|
+
Eval::UnknownCostGate.validate!(on_unknown_pricing)
|
|
34
36
|
report = step.run_eval(eval_name, context: context)
|
|
35
37
|
|
|
36
38
|
if minimum_score
|
|
@@ -42,6 +44,9 @@ module RubyLLM
|
|
|
42
44
|
end
|
|
43
45
|
|
|
44
46
|
if maximum_cost
|
|
47
|
+
unknown_cost_failure = Eval::UnknownCostGate.check([report], mode: on_unknown_pricing)
|
|
48
|
+
assert unknown_cost_failure.nil?,
|
|
49
|
+
msg || "Expected #{eval_name} eval to stay within maximum cost, but #{unknown_cost_failure}"
|
|
45
50
|
assert report.total_cost <= maximum_cost,
|
|
46
51
|
msg || "Expected #{eval_name} eval cost <= $#{format("%.4f", maximum_cost)}, got $#{format("%.4f", report.total_cost)}"
|
|
47
52
|
end
|
|
@@ -95,7 +95,6 @@ module RubyLLM
|
|
|
95
95
|
((Process.clock_gettime(Process::CLOCK_MONOTONIC) - start_time) * 1000).round
|
|
96
96
|
end
|
|
97
97
|
|
|
98
|
-
# Encapsulates mutable state during pipeline execution
|
|
99
98
|
class ExecutionState
|
|
100
99
|
attr_reader :trace_id, :step_results, :step_traces, :outputs_by_step,
|
|
101
100
|
:current_input, :status, :failed_step
|
|
@@ -33,6 +33,13 @@ module RubyLLM
|
|
|
33
33
|
value.dig(*rest)
|
|
34
34
|
end
|
|
35
35
|
|
|
36
|
+
# total_cost sums only the steps that could be priced.
|
|
37
|
+
def cost_unknown?
|
|
38
|
+
return false unless @step_traces.is_a?(Array)
|
|
39
|
+
|
|
40
|
+
@step_traces.any? { |step_trace| step_trace.respond_to?(:cost_unknown?) && step_trace.cost_unknown? }
|
|
41
|
+
end
|
|
42
|
+
|
|
36
43
|
def to_h
|
|
37
44
|
{ trace_id: @trace_id, total_latency_ms: @total_latency_ms,
|
|
38
45
|
total_usage: @total_usage, step_traces: @step_traces,
|
|
@@ -3,21 +3,10 @@
|
|
|
3
3
|
module RubyLLM
|
|
4
4
|
module Contract
|
|
5
5
|
class RakeTask < ::Rake::TaskLib
|
|
6
|
-
# Encapsulates the pass/fail gate that runs after `RakeTask#define_task`
|
|
7
|
-
# has collected eval reports. Extracted from the prior `define_task`
|
|
8
|
-
# god-method so each gating dimension (cost, score, regression) is
|
|
9
|
-
# testable in isolation.
|
|
10
|
-
#
|
|
11
|
-
# Returns a `Verdict` value object with:
|
|
12
|
-
# - `passed?` — overall gate verdict
|
|
13
|
-
# - `abort_reason` — String for `abort` when `passed? == false`, nil otherwise
|
|
14
|
-
# - `passed_reports` — [[host, report], ...] of reports that individually passed
|
|
15
|
-
# (used to decide which baselines to save)
|
|
16
|
-
# - `suite_cost` — total cost across all reports
|
|
17
|
-
#
|
|
18
6
|
# Gate ordering (preserved from pre-refactor behaviour):
|
|
19
|
-
# 1. cost gate runs FIRST — if `maximum_cost` set and exceeded,
|
|
20
|
-
#
|
|
7
|
+
# 1. cost gate runs FIRST — if `maximum_cost` set and exceeded, or any
|
|
8
|
+
# case has unknown cost under on_unknown_pricing :refuse, the suite
|
|
9
|
+
# aborts before any score check; passed_reports is empty.
|
|
21
10
|
# 2. score gate runs per-report; a report passes if
|
|
22
11
|
# `report_meets_score?` AND `!check_regression`.
|
|
23
12
|
# 3. overall passed = ALL reports passed AND cost gate not tripped.
|
|
@@ -28,20 +17,24 @@ module RubyLLM
|
|
|
28
17
|
end
|
|
29
18
|
end
|
|
30
19
|
|
|
31
|
-
def self.evaluate(host_reports:, minimum_score:, maximum_cost:, fail_on_regression
|
|
20
|
+
def self.evaluate(host_reports:, minimum_score:, maximum_cost:, fail_on_regression:,
|
|
21
|
+
on_unknown_pricing: :refuse)
|
|
32
22
|
new(host_reports: host_reports,
|
|
33
23
|
minimum_score: minimum_score,
|
|
34
24
|
maximum_cost: maximum_cost,
|
|
35
|
-
fail_on_regression: fail_on_regression
|
|
25
|
+
fail_on_regression: fail_on_regression,
|
|
26
|
+
on_unknown_pricing: on_unknown_pricing).verdict
|
|
36
27
|
end
|
|
37
28
|
|
|
38
29
|
attr_reader :verdict
|
|
39
30
|
|
|
40
|
-
def initialize(host_reports:, minimum_score:, maximum_cost:, fail_on_regression
|
|
31
|
+
def initialize(host_reports:, minimum_score:, maximum_cost:, fail_on_regression:,
|
|
32
|
+
on_unknown_pricing: :refuse)
|
|
41
33
|
@host_reports = host_reports
|
|
42
34
|
@minimum_score = minimum_score
|
|
43
35
|
@maximum_cost = maximum_cost
|
|
44
36
|
@fail_on_regression = fail_on_regression
|
|
37
|
+
@on_unknown_pricing = Eval::UnknownCostGate.validate!(on_unknown_pricing)
|
|
45
38
|
@verdict = build_verdict
|
|
46
39
|
end
|
|
47
40
|
|
|
@@ -49,11 +42,12 @@ module RubyLLM
|
|
|
49
42
|
|
|
50
43
|
def build_verdict
|
|
51
44
|
suite_cost = compute_suite_cost
|
|
45
|
+
cost_failure = cost_failure_message(suite_cost)
|
|
52
46
|
|
|
53
|
-
if
|
|
47
|
+
if cost_failure
|
|
54
48
|
return Verdict.new(
|
|
55
49
|
passed: false,
|
|
56
|
-
abort_reason:
|
|
50
|
+
abort_reason: cost_failure,
|
|
57
51
|
passed_reports: [],
|
|
58
52
|
suite_cost: suite_cost
|
|
59
53
|
)
|
|
@@ -72,8 +66,11 @@ module RubyLLM
|
|
|
72
66
|
@host_reports.sum { |_host, report| report.total_cost }
|
|
73
67
|
end
|
|
74
68
|
|
|
75
|
-
def
|
|
76
|
-
|
|
69
|
+
def cost_failure_message(suite_cost)
|
|
70
|
+
return nil unless @maximum_cost
|
|
71
|
+
return cost_abort_message(suite_cost) if suite_cost > @maximum_cost
|
|
72
|
+
|
|
73
|
+
Eval::UnknownCostGate.check(@host_reports.map(&:last), mode: @on_unknown_pricing)
|
|
77
74
|
end
|
|
78
75
|
|
|
79
76
|
def cost_abort_message(suite_cost)
|
|
@@ -2,13 +2,15 @@
|
|
|
2
2
|
|
|
3
3
|
require "rake"
|
|
4
4
|
require "rake/tasklib"
|
|
5
|
+
require_relative "eval/unknown_cost_gate"
|
|
5
6
|
require_relative "rake_task/suite_gate"
|
|
6
7
|
|
|
7
8
|
module RubyLLM
|
|
8
9
|
module Contract
|
|
9
10
|
class RakeTask < ::Rake::TaskLib
|
|
10
11
|
attr_accessor :name, :context, :fail_on_empty, :minimum_score, :maximum_cost,
|
|
11
|
-
:eval_dirs, :save_baseline, :fail_on_regression, :track_history
|
|
12
|
+
:eval_dirs, :save_baseline, :fail_on_regression, :track_history,
|
|
13
|
+
:on_unknown_pricing
|
|
12
14
|
|
|
13
15
|
def initialize(name = :"ruby_llm_contract:eval", &block)
|
|
14
16
|
super()
|
|
@@ -18,10 +20,13 @@ module RubyLLM
|
|
|
18
20
|
@minimum_score = nil # nil = require 100%; float = threshold
|
|
19
21
|
@maximum_cost = nil # nil = no cost limit; float = budget cap (suite-level)
|
|
20
22
|
@eval_dirs = [] # directories to load eval files from (non-Rails)
|
|
23
|
+
# With maximum_cost: :refuse fails on cases the budget cannot price; :warn gates priced cases only
|
|
24
|
+
@on_unknown_pricing = :refuse
|
|
21
25
|
@save_baseline = false
|
|
22
26
|
@fail_on_regression = false
|
|
23
27
|
@track_history = false
|
|
24
28
|
block&.call(self)
|
|
29
|
+
Eval::UnknownCostGate.validate!(@on_unknown_pricing)
|
|
25
30
|
define_task
|
|
26
31
|
end
|
|
27
32
|
|
|
@@ -44,7 +49,8 @@ module RubyLLM
|
|
|
44
49
|
host_reports: host_reports,
|
|
45
50
|
minimum_score: @minimum_score,
|
|
46
51
|
maximum_cost: @maximum_cost,
|
|
47
|
-
fail_on_regression: @fail_on_regression
|
|
52
|
+
fail_on_regression: @fail_on_regression,
|
|
53
|
+
on_unknown_pricing: @on_unknown_pricing
|
|
48
54
|
)
|
|
49
55
|
|
|
50
56
|
abort "\nEval suite FAILED: #{verdict.abort_reason}" unless verdict.passed?
|
|
@@ -164,7 +170,7 @@ module RubyLLM
|
|
|
164
170
|
Array(JSON.parse(raw))
|
|
165
171
|
else
|
|
166
172
|
raw.split(",").map(&:strip).reject(&:empty?).map do |entry|
|
|
167
|
-
model, effort = entry.split(
|
|
173
|
+
model, effort = entry.split(Eval::RetryOptimizer::EFFORT_SEPARATOR, 2)
|
|
168
174
|
config = { model: model.strip }
|
|
169
175
|
config[:reasoning_effort] = effort.strip if effort && !effort.empty?
|
|
170
176
|
config
|
|
@@ -191,7 +197,6 @@ module RubyLLM
|
|
|
191
197
|
end
|
|
192
198
|
end
|
|
193
199
|
|
|
194
|
-
# Auto-register the optimize task when this file is loaded
|
|
195
200
|
OptimizeRakeTask.new
|
|
196
201
|
end
|
|
197
202
|
end
|
|
@@ -64,6 +64,10 @@ RSpec::Matchers.define :pass_eval do |eval_name|
|
|
|
64
64
|
@maximum_cost = cost
|
|
65
65
|
end
|
|
66
66
|
|
|
67
|
+
chain :on_unknown_pricing do |mode|
|
|
68
|
+
@on_unknown_pricing = RubyLLM::Contract::Eval::UnknownCostGate.validate!(mode)
|
|
69
|
+
end
|
|
70
|
+
|
|
67
71
|
chain :without_regressions do
|
|
68
72
|
@check_regressions = true
|
|
69
73
|
end
|
|
@@ -78,6 +82,9 @@ RSpec::Matchers.define :pass_eval do |eval_name|
|
|
|
78
82
|
@context ||= {}
|
|
79
83
|
@minimum_score ||= nil
|
|
80
84
|
@maximum_cost ||= nil
|
|
85
|
+
@on_unknown_pricing ||= :refuse
|
|
86
|
+
@unknown_cost_failure = nil
|
|
87
|
+
@unknown_cost_only = false
|
|
81
88
|
@check_regressions ||= false
|
|
82
89
|
@comparison_step ||= nil
|
|
83
90
|
@error = nil
|
|
@@ -97,7 +104,11 @@ RSpec::Matchers.define :pass_eval do |eval_name|
|
|
|
97
104
|
@report.passed?
|
|
98
105
|
end
|
|
99
106
|
|
|
100
|
-
cost_ok =
|
|
107
|
+
cost_ok = true
|
|
108
|
+
if @maximum_cost
|
|
109
|
+
@unknown_cost_failure = RubyLLM::Contract::Eval::UnknownCostGate.check([@report], mode: @on_unknown_pricing)
|
|
110
|
+
cost_ok = @report.total_cost <= @maximum_cost && @unknown_cost_failure.nil?
|
|
111
|
+
end
|
|
101
112
|
|
|
102
113
|
regression_ok = if @prompt_diff
|
|
103
114
|
@prompt_diff.safe_to_switch?
|
|
@@ -108,6 +119,7 @@ RSpec::Matchers.define :pass_eval do |eval_name|
|
|
|
108
119
|
true
|
|
109
120
|
end
|
|
110
121
|
|
|
122
|
+
@unknown_cost_only = score_ok && regression_ok && @report.total_cost <= @maximum_cost if @unknown_cost_failure
|
|
111
123
|
score_ok && cost_ok && regression_ok
|
|
112
124
|
rescue StandardError => e
|
|
113
125
|
@error = e
|
|
@@ -115,71 +127,87 @@ RSpec::Matchers.define :pass_eval do |eval_name|
|
|
|
115
127
|
end
|
|
116
128
|
|
|
117
129
|
failure_message do
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
130
|
+
other_failure_message = lambda do
|
|
131
|
+
if @prompt_diff && !@prompt_diff.safe_to_switch?
|
|
132
|
+
msg = "expected #{@eval_name} eval to be safe to switch from baseline prompt\n"
|
|
133
|
+
|
|
134
|
+
# Check empty sides first — most fundamental problem
|
|
135
|
+
bl_empty = @prompt_diff.baseline_empty?
|
|
136
|
+
cd_empty = @prompt_diff.candidate_empty?
|
|
137
|
+
if bl_empty || cd_empty
|
|
138
|
+
msg += " One side has no evaluated cases (all skipped or no adapter?)\n"
|
|
139
|
+
if sample_response_only_compare?
|
|
140
|
+
msg += " compare_with ignores sample_response; pass model: or with_context(adapter: ...)\n"
|
|
141
|
+
end
|
|
142
|
+
msg += " Candidate score: #{@prompt_diff.candidate_score}, Baseline score: #{@prompt_diff.baseline_score}"
|
|
143
|
+
next msg
|
|
128
144
|
end
|
|
129
|
-
msg += " Candidate score: #{@prompt_diff.candidate_score}, Baseline score: #{@prompt_diff.baseline_score}"
|
|
130
|
-
next msg
|
|
131
|
-
end
|
|
132
145
|
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
146
|
+
# Check dataset comparability — names, inputs, AND expected must match
|
|
147
|
+
unless @prompt_diff.cases_comparable?
|
|
148
|
+
unless @prompt_diff.case_names_match?
|
|
149
|
+
mm = @prompt_diff.mismatched_cases
|
|
150
|
+
msg += " Case set mismatch — candidate and baseline must have identical cases:\n"
|
|
151
|
+
mm[:only_in_baseline].each { |n| msg += " only in baseline: #{n}\n" }
|
|
152
|
+
mm[:only_in_candidate].each { |n| msg += " only in candidate: #{n}\n" }
|
|
153
|
+
end
|
|
154
|
+
@prompt_diff.input_mismatches.each do |m|
|
|
155
|
+
msg += " Input mismatch for '#{m[:case]}' — same name but different inputs\n"
|
|
156
|
+
end
|
|
157
|
+
@prompt_diff.expected_mismatches.each do |m|
|
|
158
|
+
msg += " Expected mismatch for '#{m[:case]}' — same name/input but different expected values\n"
|
|
159
|
+
end
|
|
160
|
+
next msg
|
|
143
161
|
end
|
|
144
|
-
|
|
145
|
-
|
|
162
|
+
|
|
163
|
+
# Check per-case score regressions (even if global average is flat)
|
|
164
|
+
if @prompt_diff.score_regressions.any?
|
|
165
|
+
msg += " Per-case score regressions (#{@prompt_diff.score_regressions.length}):\n"
|
|
166
|
+
@prompt_diff.score_regressions.each do |r|
|
|
167
|
+
msg += " #{r[:case]}: #{r[:baseline_score]} -> #{r[:candidate_score]} (#{r[:delta]})\n"
|
|
168
|
+
end
|
|
169
|
+
msg += " Score delta: #{@prompt_diff.score_delta}"
|
|
170
|
+
next msg
|
|
146
171
|
end
|
|
147
|
-
next msg
|
|
148
|
-
end
|
|
149
172
|
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
173
|
+
# Check pass/fail regressions and removed cases
|
|
174
|
+
removed = @prompt_diff.removed_passing_cases
|
|
175
|
+
reg_count = @prompt_diff.regressions.length + removed.length
|
|
176
|
+
msg += " Found #{reg_count} regression(s):\n"
|
|
177
|
+
@prompt_diff.regressions.each do |r|
|
|
178
|
+
msg += " #{r[:case]}: was PASS, now FAIL -- #{r[:detail]}\n"
|
|
179
|
+
end
|
|
180
|
+
removed.each do |name|
|
|
181
|
+
msg += " #{name}: REMOVED (was passing in baseline)\n"
|
|
155
182
|
end
|
|
156
183
|
msg += " Score delta: #{@prompt_diff.score_delta}"
|
|
157
184
|
next msg
|
|
158
185
|
end
|
|
159
186
|
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
187
|
+
msg = format_failure_message(@eval_name, @error, @report, @minimum_score, @maximum_cost)
|
|
188
|
+
if @diff&.regressed?
|
|
189
|
+
msg += "\n\nRegressions from baseline:\n"
|
|
190
|
+
@diff.regressions.each do |r|
|
|
191
|
+
msg += " #{r[:case]}: was PASS, now FAIL -- #{r[:detail]}\n"
|
|
192
|
+
end
|
|
193
|
+
@diff.skipped_passing_cases.each do |s|
|
|
194
|
+
msg += " #{s[:case]}: was PASS, now SKIPPED -- #{s[:detail]}\n"
|
|
195
|
+
end
|
|
196
|
+
msg += " Score delta: #{@diff.score_delta}"
|
|
169
197
|
end
|
|
170
|
-
msg
|
|
171
|
-
next msg
|
|
198
|
+
msg
|
|
172
199
|
end
|
|
173
200
|
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
@
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
201
|
+
# Cost first, as in the rake suite gate: an unpriced total cannot be trusted.
|
|
202
|
+
# An exception outranks it, and any other failure is still listed.
|
|
203
|
+
if @unknown_cost_failure && @error.nil?
|
|
204
|
+
unknown_cost = "expected #{@eval_name} eval to stay within maximum cost, but #{@unknown_cost_failure}"
|
|
205
|
+
next unknown_cost if @unknown_cost_only
|
|
206
|
+
|
|
207
|
+
next "#{unknown_cost}\n\n#{other_failure_message.call}"
|
|
181
208
|
end
|
|
182
|
-
|
|
209
|
+
|
|
210
|
+
other_failure_message.call
|
|
183
211
|
end
|
|
184
212
|
|
|
185
213
|
failure_message_when_negated do
|
|
@@ -6,6 +6,11 @@ module RubyLLM
|
|
|
6
6
|
class Base
|
|
7
7
|
DEFAULT_OUTPUT_TOKENS = 256
|
|
8
8
|
|
|
9
|
+
# Eval::CaseExecutor substring-matches this to choose skip over raise.
|
|
10
|
+
# Reworded without the consumer, every skipped eval case becomes a hard
|
|
11
|
+
# failure of the whole run, so both ends read it from here.
|
|
12
|
+
NO_ADAPTER_MESSAGE = "No adapter configured"
|
|
13
|
+
|
|
9
14
|
def self.inherited(subclass)
|
|
10
15
|
super
|
|
11
16
|
Contract.register_eval_host(subclass) if respond_to?(:eval_defined?) && eval_defined?
|
|
@@ -141,7 +146,10 @@ module RubyLLM
|
|
|
141
146
|
def estimate_eval_cost_for_model(cases, model_name)
|
|
142
147
|
cases.sum do |test_case|
|
|
143
148
|
estimate = estimate_cost(input: test_case.input, model: model_name)
|
|
144
|
-
|
|
149
|
+
# Two misses floor to 0.0, not one: model absent from the registry
|
|
150
|
+
# (nil estimate), or present with unreadable pricing (hash whose
|
|
151
|
+
# estimated_cost is nil). Documented as a floor, not a fail-closed.
|
|
152
|
+
(estimate && estimate[:estimated_cost]) || 0.0
|
|
145
153
|
end.round(6)
|
|
146
154
|
end
|
|
147
155
|
|
|
@@ -225,8 +233,9 @@ module RubyLLM
|
|
|
225
233
|
adapter = context[:adapter] || RubyLLM::Contract.configuration.default_adapter
|
|
226
234
|
return adapter if adapter
|
|
227
235
|
|
|
228
|
-
raise RubyLLM::Contract::Error,
|
|
229
|
-
|
|
236
|
+
raise RubyLLM::Contract::Error,
|
|
237
|
+
"#{NO_ADAPTER_MESSAGE}. Set one with RubyLLM::Contract.configure " \
|
|
238
|
+
"{ |c| c.default_adapter = ... } or pass context: { adapter: ... }"
|
|
230
239
|
end
|
|
231
240
|
|
|
232
241
|
# ADR-0021 deliverable 2: narrow ArgumentError rescue to DSL-setup phase only.
|
|
@@ -3,8 +3,6 @@
|
|
|
3
3
|
module RubyLLM
|
|
4
4
|
module Contract
|
|
5
5
|
module Step
|
|
6
|
-
# Extracted from Base to reduce class length.
|
|
7
|
-
# DSL accessor methods for step definition (input_type, output_type, prompt, etc.).
|
|
8
6
|
module Dsl # rubocop:disable Metrics/ModuleLength
|
|
9
7
|
# Sentinel signalling "explicitly reset" (`some_attr(:default)`).
|
|
10
8
|
# Distinguishes reset (lookup stops at this class, returns nil) from
|
|
@@ -65,8 +63,8 @@ module RubyLLM
|
|
|
65
63
|
|
|
66
64
|
def output_schema(&block)
|
|
67
65
|
if block
|
|
68
|
-
require "
|
|
69
|
-
@output_schema = ::
|
|
66
|
+
require "schematist"
|
|
67
|
+
@output_schema = ::Schematist::Schema.create(&block)
|
|
70
68
|
elsif defined?(@output_schema)
|
|
71
69
|
@output_schema
|
|
72
70
|
elsif superclass.respond_to?(:output_schema)
|
|
@@ -50,6 +50,25 @@ module RubyLLM
|
|
|
50
50
|
)
|
|
51
51
|
end
|
|
52
52
|
|
|
53
|
+
# True when a call ran on a named model, used tokens, and still got no
|
|
54
|
+
# cost, so any total built from this trace undercounts. A zero-token call
|
|
55
|
+
# (Test adapter, sample_response) costs 0 whatever the pricing. A retried
|
|
56
|
+
# step is judged per attempt: a priced subtotal must not hide an unpriced
|
|
57
|
+
# attempt, and the merged trace may have re-priced on the last model.
|
|
58
|
+
def cost_unknown?
|
|
59
|
+
if @attempts.is_a?(Array) && !@attempts.empty?
|
|
60
|
+
@attempts.any? { |attempt| self.class.unpriced?(attempt[:model], attempt[:usage], attempt[:cost]) }
|
|
61
|
+
else
|
|
62
|
+
self.class.unpriced?(@model, @usage, @cost)
|
|
63
|
+
end
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
def self.unpriced?(model, usage, cost)
|
|
67
|
+
return false unless model && cost.nil? && usage.is_a?(Hash)
|
|
68
|
+
|
|
69
|
+
((usage[:input_tokens] || 0) + (usage[:output_tokens] || 0)).positive?
|
|
70
|
+
end
|
|
71
|
+
|
|
53
72
|
def to_h
|
|
54
73
|
{ messages: @messages, model: @model, latency_ms: @latency_ms,
|
|
55
74
|
usage: @usage, attempts: @attempts, cost: @cost }.compact
|
data/lib/ruby_llm/contract.rb
CHANGED
|
@@ -21,8 +21,6 @@ module RubyLLM
|
|
|
21
21
|
step_adapter_overrides.clear
|
|
22
22
|
end
|
|
23
23
|
|
|
24
|
-
# --- Eval host registry ---
|
|
25
|
-
|
|
26
24
|
def register_eval_host(klass)
|
|
27
25
|
eval_hosts << klass unless eval_hosts.include?(klass)
|
|
28
26
|
end
|
|
@@ -101,9 +99,7 @@ module RubyLLM
|
|
|
101
99
|
|
|
102
100
|
private
|
|
103
101
|
|
|
104
|
-
# Filter stale hosts, deduplicate by name (last wins), prune registry in-place
|
|
105
102
|
def live_eval_hosts
|
|
106
|
-
# Remove hosts without evals
|
|
107
103
|
@eval_hosts&.reject! { |h| !h.respond_to?(:eval_defined?) || !h.eval_defined? }
|
|
108
104
|
|
|
109
105
|
# Deduplicate: if two classes share a name (reload), keep the latest
|