ruby_llm-contract 1.0.0 → 1.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.ruby-version +1 -0
- data/CHANGELOG.md +287 -224
- data/README.md +1 -1
- data/docs/guide/getting_started.md +3 -1
- data/docs/guide/testing.md +1 -1
- data/lib/ruby_llm/contract/concerns/eval_host.rb +2 -2
- data/lib/ruby_llm/contract/cost_calculator.rb +5 -4
- data/lib/ruby_llm/contract/eval/aggregated_report.rb +4 -0
- data/lib/ruby_llm/contract/eval/baseline_diff.rb +19 -2
- data/lib/ruby_llm/contract/eval/case_executor.rb +1 -1
- data/lib/ruby_llm/contract/eval/case_result.rb +12 -3
- data/lib/ruby_llm/contract/eval/case_result_builder.rb +2 -1
- data/lib/ruby_llm/contract/eval/evaluator/proc_evaluator.rb +6 -2
- data/lib/ruby_llm/contract/eval/prompt_diff_serializer.rb +1 -1
- data/lib/ruby_llm/contract/eval/recommender.rb +2 -2
- data/lib/ruby_llm/contract/eval/report.rb +5 -2
- data/lib/ruby_llm/contract/eval/report_stats.rb +7 -2
- data/lib/ruby_llm/contract/eval/report_storage.rb +6 -6
- data/lib/ruby_llm/contract/eval/step_expectation_applier.rb +2 -1
- data/lib/ruby_llm/contract/eval/unknown_cost_gate.rb +32 -0
- data/lib/ruby_llm/contract/eval.rb +1 -0
- data/lib/ruby_llm/contract/minitest.rb +6 -1
- data/lib/ruby_llm/contract/pipeline/trace.rb +7 -0
- data/lib/ruby_llm/contract/railtie.rb +1 -1
- data/lib/ruby_llm/contract/rake_task/suite_gate.rb +19 -10
- data/lib/ruby_llm/contract/rake_task.rb +9 -3
- data/lib/ruby_llm/contract/rspec/pass_eval.rb +80 -52
- data/lib/ruby_llm/contract/step/base.rb +6 -4
- data/lib/ruby_llm/contract/step/dsl.rb +5 -14
- data/lib/ruby_llm/contract/step/runner_config.rb +2 -2
- data/lib/ruby_llm/contract/step/trace.rb +19 -0
- data/lib/ruby_llm/contract/unknown_policy.rb +21 -0
- data/lib/ruby_llm/contract/version.rb +1 -1
- data/lib/ruby_llm/contract.rb +17 -4
- metadata +4 -1
data/README.md
CHANGED
|
@@ -168,7 +168,7 @@ Different layers, complementary. [`ruby_llm-tribunal`](https://github.com/Alqemi
|
|
|
168
168
|
|
|
169
169
|
## Status & versioning
|
|
170
170
|
|
|
171
|
-
Stable
|
|
171
|
+
Stable since **1.0.0**. Semver tracked; breaking changes flagged in [CHANGELOG](CHANGELOG.md). Pin `~> 1.0`.
|
|
172
172
|
|
|
173
173
|
What 1.0 promises not to break inside the 1.x line: the documented DSL
|
|
174
174
|
(`prompt`, `output_schema`, `validate`, `retry_policy`, `max_cost`, `define_eval`,
|
|
@@ -145,7 +145,7 @@ report.save_baseline!
|
|
|
145
145
|
expect(SummarizeArticle).to pass_eval("regression").without_regressions
|
|
146
146
|
```
|
|
147
147
|
|
|
148
|
-
`without_regressions` fails the build only if a previously-passing case now fails — a new model version, a prompt tweak, or an upstream change that silently lowered quality.
|
|
148
|
+
`without_regressions` fails the build only if a previously-passing case now fails — a new model version, a prompt tweak, or an upstream change that silently lowered quality. A previously-passing case that was skipped this run (no adapter configured) also fails the gate, listed as `SKIPPED` rather than as a regression.
|
|
149
149
|
|
|
150
150
|
## Budget caps
|
|
151
151
|
|
|
@@ -173,6 +173,8 @@ max_cost 0.01, on_unknown_pricing: :warn
|
|
|
173
173
|
|
|
174
174
|
Default is `:refuse`. Use `:warn` only when you accept running without a cost ceiling (fine-tuned models you trust, private endpoints).
|
|
175
175
|
|
|
176
|
+
The eval-level gates - `with_maximum_cost` on `pass_eval`, `maximum_cost` on the rake task, and `maximum_cost:` on `assert_eval_passes` - follow the same rule since 1.1.0. A report's total counts a case it cannot price as $0, so when any case ran on a model without pricing data the gate fails and names the cases instead of comparing an undercounted total. Opt out with `.on_unknown_pricing(:warn)`, `t.on_unknown_pricing = :warn` or `on_unknown_pricing: :warn`, which checks the budget against the priced cases and prints a warning. Offline runs (`sample_response`, a `Test` adapter without `usage:`) report zero tokens, cost $0 with or without pricing, and are never flagged. The check sees only reported tokens: an adapter that returns no token counts looks like a zero-token run.
|
|
177
|
+
|
|
176
178
|
### Preflight cost estimates
|
|
177
179
|
|
|
178
180
|
Check what a call is likely to cost before invoking it:
|
data/docs/guide/testing.md
CHANGED
|
@@ -160,7 +160,7 @@ end
|
|
|
160
160
|
|
|
161
161
|
- `.with_context(model: "gpt-4.1-mini")` — pick model / pass adapter
|
|
162
162
|
- `.with_minimum_score(0.8)` — gate on average score
|
|
163
|
-
- `.with_maximum_cost(0.01)` — gate on total cost
|
|
163
|
+
- `.with_maximum_cost(0.01)` — gate on total cost; fails when a case's model has no pricing data unless you add `.on_unknown_pricing(:warn)`
|
|
164
164
|
- `.without_regressions` — block any previously-passing case that now fails (reads the baseline)
|
|
165
165
|
- `.compared_with(SummarizeArticleV1)` — A/B against another step; implies regression check
|
|
166
166
|
|
|
@@ -16,13 +16,13 @@ module RubyLLM
|
|
|
16
16
|
@file_sourced_evals ||= Set.new
|
|
17
17
|
key = name.to_s
|
|
18
18
|
|
|
19
|
-
if @eval_definitions.key?(key) && !
|
|
19
|
+
if @eval_definitions.key?(key) && !Contract.reloading?
|
|
20
20
|
warn "[ruby_llm-contract] Redefining eval '#{key}' on #{self}. " \
|
|
21
21
|
"This replaces the previous definition."
|
|
22
22
|
end
|
|
23
23
|
|
|
24
24
|
@eval_definitions[key] = Eval::EvalDefinition.new(key, step_class: self, &)
|
|
25
|
-
@file_sourced_evals.add(key) if
|
|
25
|
+
@file_sourced_evals.add(key) if Contract.reloading?
|
|
26
26
|
Contract.register_eval_host(self)
|
|
27
27
|
register_subclasses(self)
|
|
28
28
|
end
|
|
@@ -97,10 +97,11 @@ module RubyLLM
|
|
|
97
97
|
# moved provider pricing into a nested value object and dropped those flat
|
|
98
98
|
# readers, so it has to be walked: pricing -> text_tokens -> standard.
|
|
99
99
|
#
|
|
100
|
-
# nil means "unknown pricing"
|
|
101
|
-
#
|
|
102
|
-
#
|
|
103
|
-
#
|
|
100
|
+
# nil means "unknown pricing". The guards keep the two known shapes
|
|
101
|
+
# explicit, but `calculate` rescues StandardError, so a future shape this
|
|
102
|
+
# walk cannot read also degrades to nil: `max_cost` then refuses the call
|
|
103
|
+
# citing "no pricing data" rather than the real cause, and paths that floor
|
|
104
|
+
# unknown cost to 0.0 (eval cost estimates, report totals) undercount.
|
|
104
105
|
def self.prices_for(model_info)
|
|
105
106
|
if model_info.respond_to?(:input_price_per_million)
|
|
106
107
|
return { input: model_info.input_price_per_million,
|
|
@@ -44,6 +44,10 @@ module RubyLLM
|
|
|
44
44
|
@runs.sum(&:total_cost) / @runs.length.to_f
|
|
45
45
|
end
|
|
46
46
|
|
|
47
|
+
def unknown_cost_results
|
|
48
|
+
@runs.flat_map(&:unknown_cost_results)
|
|
49
|
+
end
|
|
50
|
+
|
|
47
51
|
def avg_latency_ms
|
|
48
52
|
latencies = @runs.filter_map(&:avg_latency_ms)
|
|
49
53
|
return nil if latencies.empty?
|
|
@@ -19,6 +19,7 @@ module RubyLLM
|
|
|
19
19
|
current = @current[name]
|
|
20
20
|
next unless current
|
|
21
21
|
next unless baseline[:passed] && !current[:passed]
|
|
22
|
+
next if skipped?(current)
|
|
22
23
|
|
|
23
24
|
{
|
|
24
25
|
case: name,
|
|
@@ -47,8 +48,19 @@ module RubyLLM
|
|
|
47
48
|
(current_score - baseline_score).round(4)
|
|
48
49
|
end
|
|
49
50
|
|
|
51
|
+
# Passed in the baseline, not run now (no adapter): not evidence of a
|
|
52
|
+
# regression, but not evidence it still passes either, so the gate fails.
|
|
53
|
+
def skipped_passing_cases
|
|
54
|
+
@baseline.filter_map do |name, baseline|
|
|
55
|
+
current = @current[name]
|
|
56
|
+
next unless current && baseline[:passed] && skipped?(current)
|
|
57
|
+
|
|
58
|
+
{ case: name, detail: current[:details] }
|
|
59
|
+
end
|
|
60
|
+
end
|
|
61
|
+
|
|
50
62
|
def regressed?
|
|
51
|
-
regressions.any? || removed_passing_cases.any?
|
|
63
|
+
regressions.any? || removed_passing_cases.any? || skipped_passing_cases.any?
|
|
52
64
|
end
|
|
53
65
|
|
|
54
66
|
def removed_passing_cases
|
|
@@ -70,6 +82,7 @@ module RubyLLM
|
|
|
70
82
|
def to_s
|
|
71
83
|
lines = ["Score: #{baseline_score.round(2)} → #{current_score.round(2)} (#{format_delta})"]
|
|
72
84
|
regressions.each { |r| lines << " REGRESSED #{r[:case]}: #{r[:detail]}" }
|
|
85
|
+
skipped_passing_cases.each { |s| lines << " SKIPPED #{s[:case]}: #{s[:detail]}" }
|
|
73
86
|
improvements.each { |r| lines << " IMPROVED #{r[:case]}" }
|
|
74
87
|
new_cases.each { |c| lines << " NEW #{c}" }
|
|
75
88
|
removed_cases.each { |c| lines << " REMOVED #{c}" }
|
|
@@ -82,12 +95,16 @@ module RubyLLM
|
|
|
82
95
|
# Asymmetric on purpose: the baseline side arrives already filtered by
|
|
83
96
|
# ReportStats#evaluated_results, the current side does not, so this
|
|
84
97
|
# string check is the only skip filter on one of the two sides.
|
|
85
|
-
evaluated = cases.reject { |c|
|
|
98
|
+
evaluated = cases.reject { |c| skipped?(c) }
|
|
86
99
|
return 0.0 if evaluated.empty?
|
|
87
100
|
|
|
88
101
|
evaluated.sum { |c| c[:score] } / evaluated.length
|
|
89
102
|
end
|
|
90
103
|
|
|
104
|
+
def skipped?(serialized_case)
|
|
105
|
+
serialized_case[:details]&.start_with?(CaseResult::SKIPPED_DETAILS_PREFIX)
|
|
106
|
+
end
|
|
107
|
+
|
|
91
108
|
def index_by_name(cases)
|
|
92
109
|
cases.each_with_object({}) { |c, h| h[c[:name]] = c }
|
|
93
110
|
end
|
|
@@ -7,6 +7,9 @@ module RubyLLM
|
|
|
7
7
|
# BaselineDiff#compute_score keys off this prefix to keep skipped cases
|
|
8
8
|
# out of the score denominator. Both ends must read it from here.
|
|
9
9
|
SKIPPED_DETAILS_PREFIX = "skipped:"
|
|
10
|
+
# Not run (no adapter). Readers compare step_status, so any result object
|
|
11
|
+
# exposing it works; kept out of score, pass rate and cost-per-call.
|
|
12
|
+
SKIPPED_STATUS = :skipped
|
|
10
13
|
|
|
11
14
|
def self.skipped_details(reason)
|
|
12
15
|
"#{SKIPPED_DETAILS_PREFIX} #{reason}"
|
|
@@ -15,8 +18,9 @@ module RubyLLM
|
|
|
15
18
|
attr_reader :name, :input, :output, :expected, :step_status,
|
|
16
19
|
:score, :details, :duration_ms, :cost, :attempts
|
|
17
20
|
|
|
18
|
-
def initialize(name:, input:, output:, expected:, step_status:,
|
|
19
|
-
score:, passed:, label: nil, details: nil, duration_ms: nil, cost: nil, attempts: nil
|
|
21
|
+
def initialize(name:, input:, output:, expected:, step_status:, # rubocop:disable Metrics/ParameterLists
|
|
22
|
+
score:, passed:, label: nil, details: nil, duration_ms: nil, cost: nil, attempts: nil,
|
|
23
|
+
cost_unknown: false)
|
|
20
24
|
@name = name
|
|
21
25
|
@input = input
|
|
22
26
|
@output = output
|
|
@@ -29,6 +33,7 @@ module RubyLLM
|
|
|
29
33
|
@duration_ms = duration_ms
|
|
30
34
|
@cost = cost
|
|
31
35
|
@attempts = attempts
|
|
36
|
+
@cost_unknown = cost_unknown
|
|
32
37
|
freeze
|
|
33
38
|
end
|
|
34
39
|
|
|
@@ -40,6 +45,10 @@ module RubyLLM
|
|
|
40
45
|
!@passed
|
|
41
46
|
end
|
|
42
47
|
|
|
48
|
+
def cost_unknown?
|
|
49
|
+
@cost_unknown
|
|
50
|
+
end
|
|
51
|
+
|
|
43
52
|
def label
|
|
44
53
|
@label || (@passed ? "PASS" : "FAIL")
|
|
45
54
|
end
|
|
@@ -69,7 +78,7 @@ module RubyLLM
|
|
|
69
78
|
duration_ms: @duration_ms,
|
|
70
79
|
cost: @cost,
|
|
71
80
|
attempts: @attempts
|
|
72
|
-
}
|
|
81
|
+
}.tap { |hash| hash[:cost_unknown] = true if @cost_unknown }
|
|
73
82
|
end
|
|
74
83
|
|
|
75
84
|
private
|
|
@@ -19,7 +19,8 @@ module RubyLLM
|
|
|
19
19
|
details: evaluation.details,
|
|
20
20
|
duration_ms: trace_metric(trace, :total_latency_ms, :latency_ms),
|
|
21
21
|
cost: trace_metric(trace, :total_cost, :cost),
|
|
22
|
-
attempts: trace_attempts(trace)
|
|
22
|
+
attempts: trace_attempts(trace),
|
|
23
|
+
cost_unknown: trace.respond_to?(:cost_unknown?) && trace.cost_unknown?
|
|
23
24
|
)
|
|
24
25
|
end
|
|
25
26
|
|
|
@@ -6,6 +6,10 @@ module RubyLLM
|
|
|
6
6
|
module Evaluator
|
|
7
7
|
# Adapts custom Ruby callables to the EvaluationResult contract.
|
|
8
8
|
class ProcEvaluator
|
|
9
|
+
# Report hides these under a failure: they repeat the PASS/FAIL label.
|
|
10
|
+
PASSED_DETAILS = "passed"
|
|
11
|
+
FAILED_DETAILS = "not passed"
|
|
12
|
+
|
|
9
13
|
def initialize(callable)
|
|
10
14
|
@callable = callable
|
|
11
15
|
end
|
|
@@ -36,9 +40,9 @@ module RubyLLM
|
|
|
36
40
|
def build_evaluation_result(result)
|
|
37
41
|
case result
|
|
38
42
|
when true
|
|
39
|
-
EvaluationResult.new(score: 1.0, passed: true, details:
|
|
43
|
+
EvaluationResult.new(score: 1.0, passed: true, details: PASSED_DETAILS)
|
|
40
44
|
when false
|
|
41
|
-
EvaluationResult.new(score: 0.0, passed: false, details:
|
|
45
|
+
EvaluationResult.new(score: 0.0, passed: false, details: FAILED_DETAILS)
|
|
42
46
|
when Numeric
|
|
43
47
|
EvaluationResult.new(score: result, passed: result >= 0.5, details: "custom score: #{result}")
|
|
44
48
|
else
|
|
@@ -5,7 +5,7 @@ module RubyLLM
|
|
|
5
5
|
module Eval
|
|
6
6
|
class PromptDiffSerializer
|
|
7
7
|
def call(report)
|
|
8
|
-
report.results.reject { |result| result.step_status ==
|
|
8
|
+
report.results.reject { |result| result.step_status == CaseResult::SKIPPED_STATUS }.map do |result|
|
|
9
9
|
{
|
|
10
10
|
name: result.name,
|
|
11
11
|
input: result.input,
|
|
@@ -40,7 +40,7 @@ module RubyLLM
|
|
|
40
40
|
report = @comparison.reports[label]
|
|
41
41
|
next nil unless report
|
|
42
42
|
|
|
43
|
-
evaluated_count = report.results.count { |r| r.step_status !=
|
|
43
|
+
evaluated_count = report.results.count { |r| r.step_status != CaseResult::SKIPPED_STATUS }
|
|
44
44
|
cases_count = [evaluated_count, 1].max
|
|
45
45
|
cost_per_call = report.total_cost.to_f / cases_count
|
|
46
46
|
|
|
@@ -116,7 +116,7 @@ module RubyLLM
|
|
|
116
116
|
current_report = @comparison.reports[current_label]
|
|
117
117
|
return {} unless current_report
|
|
118
118
|
|
|
119
|
-
current_evaluated = current_report.results.count { |r| r.step_status !=
|
|
119
|
+
current_evaluated = current_report.results.count { |r| r.step_status != CaseResult::SKIPPED_STATUS }
|
|
120
120
|
current_cases = [current_evaluated, 1].max
|
|
121
121
|
current_cost = current_report.total_cost.to_f / current_cases
|
|
122
122
|
diff = current_cost - best[:cost_per_call]
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
require "forwardable"
|
|
4
|
+
require_relative "evaluator/proc_evaluator"
|
|
4
5
|
|
|
5
6
|
module RubyLLM
|
|
6
7
|
module Contract
|
|
@@ -10,12 +11,14 @@ module RubyLLM
|
|
|
10
11
|
|
|
11
12
|
attr_reader :dataset_name, :results, :step_name
|
|
12
13
|
|
|
13
|
-
GENERIC_DETAILS = [
|
|
14
|
+
GENERIC_DETAILS = [Evaluator::ProcEvaluator::PASSED_DETAILS, Evaluator::ProcEvaluator::FAILED_DETAILS].freeze
|
|
14
15
|
HISTORY_DIR = ".eval_history"
|
|
16
|
+
HISTORY_EXT = "jsonl"
|
|
15
17
|
BASELINE_DIR = ".eval_baselines"
|
|
18
|
+
BASELINE_EXT = "json"
|
|
16
19
|
|
|
17
20
|
def_delegators :@stats, :score, :passed, :failed, :skipped, :failures, :pass_rate, :pass_rate_ratio,
|
|
18
|
-
:total_cost, :avg_latency_ms, :passed?,
|
|
21
|
+
:total_cost, :unknown_cost_results, :avg_latency_ms, :passed?,
|
|
19
22
|
:production_mode?, :escalation_rate, :single_shot_cost, :single_shot_latency_ms,
|
|
20
23
|
:effective_cost, :effective_latency_ms, :latency_percentiles
|
|
21
24
|
def_delegators :@presenter, :summary, :to_s, :print_summary
|
|
@@ -23,7 +23,7 @@ module RubyLLM
|
|
|
23
23
|
end
|
|
24
24
|
|
|
25
25
|
def skipped
|
|
26
|
-
@results.count { |result| result.step_status ==
|
|
26
|
+
@results.count { |result| result.step_status == CaseResult::SKIPPED_STATUS }
|
|
27
27
|
end
|
|
28
28
|
|
|
29
29
|
def failures
|
|
@@ -44,6 +44,11 @@ module RubyLLM
|
|
|
44
44
|
@results.sum { |result| result.cost || 0.0 }
|
|
45
45
|
end
|
|
46
46
|
|
|
47
|
+
# Cases total_cost could not price; it counts them as 0.0.
|
|
48
|
+
def unknown_cost_results
|
|
49
|
+
evaluated_results.select(&:cost_unknown?)
|
|
50
|
+
end
|
|
51
|
+
|
|
47
52
|
def avg_latency_ms
|
|
48
53
|
latencies = @results.filter_map(&:duration_ms)
|
|
49
54
|
return nil if latencies.empty?
|
|
@@ -58,7 +63,7 @@ module RubyLLM
|
|
|
58
63
|
end
|
|
59
64
|
|
|
60
65
|
def evaluated_results
|
|
61
|
-
@evaluated_results ||= @results.reject { |result| result.step_status ==
|
|
66
|
+
@evaluated_results ||= @results.reject { |result| result.step_status == CaseResult::SKIPPED_STATUS }
|
|
62
67
|
end
|
|
63
68
|
|
|
64
69
|
def evaluated_results_count
|
|
@@ -13,7 +13,7 @@ module RubyLLM
|
|
|
13
13
|
end
|
|
14
14
|
|
|
15
15
|
def save_history!(path: nil, model: nil, reasoning_effort: nil)
|
|
16
|
-
file = path || storage_path(Report::HISTORY_DIR,
|
|
16
|
+
file = path || storage_path(Report::HISTORY_DIR, Report::HISTORY_EXT, model: model, reasoning_effort: reasoning_effort)
|
|
17
17
|
entry = history_entry
|
|
18
18
|
entry[:model] = model if model
|
|
19
19
|
entry[:reasoning_effort] = reasoning_effort if reasoning_effort
|
|
@@ -22,19 +22,19 @@ module RubyLLM
|
|
|
22
22
|
end
|
|
23
23
|
|
|
24
24
|
def eval_history(path: nil, model: nil, reasoning_effort: nil)
|
|
25
|
-
EvalHistory.load(path || storage_path(Report::HISTORY_DIR,
|
|
26
|
-
|
|
25
|
+
EvalHistory.load(path || storage_path(Report::HISTORY_DIR, Report::HISTORY_EXT,
|
|
26
|
+
model: model, reasoning_effort: reasoning_effort))
|
|
27
27
|
end
|
|
28
28
|
|
|
29
29
|
def save_baseline!(path: nil, model: nil, reasoning_effort: nil)
|
|
30
|
-
file = path || storage_path(Report::BASELINE_DIR,
|
|
30
|
+
file = path || storage_path(Report::BASELINE_DIR, Report::BASELINE_EXT, model: model, reasoning_effort: reasoning_effort)
|
|
31
31
|
FileUtils.mkdir_p(File.dirname(file))
|
|
32
32
|
File.write(file, JSON.pretty_generate(serialize_for_baseline))
|
|
33
33
|
file
|
|
34
34
|
end
|
|
35
35
|
|
|
36
36
|
def compare_with_baseline(path: nil, model: nil, reasoning_effort: nil)
|
|
37
|
-
file = path || storage_path(Report::BASELINE_DIR,
|
|
37
|
+
file = path || storage_path(Report::BASELINE_DIR, Report::BASELINE_EXT, model: model, reasoning_effort: reasoning_effort)
|
|
38
38
|
raise ArgumentError, "No baseline found at #{file}" unless File.exist?(file)
|
|
39
39
|
|
|
40
40
|
baseline_data = JSON.parse(File.read(file), symbolize_names: true)
|
|
@@ -47,7 +47,7 @@ module RubyLLM
|
|
|
47
47
|
end
|
|
48
48
|
|
|
49
49
|
def baseline_exists?(path: nil, model: nil, reasoning_effort: nil)
|
|
50
|
-
File.exist?(path || storage_path(Report::BASELINE_DIR,
|
|
50
|
+
File.exist?(path || storage_path(Report::BASELINE_DIR, Report::BASELINE_EXT, model: model, reasoning_effort: reasoning_effort))
|
|
51
51
|
end
|
|
52
52
|
|
|
53
53
|
private
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
module Contract
|
|
5
|
+
module Eval
|
|
6
|
+
# Suite-level counterpart of the step-level `max_cost ... on_unknown_pricing:`.
|
|
7
|
+
# A report's total_cost counts an unpriced case as 0.0, so a maximum_cost
|
|
8
|
+
# gate that only compares totals passes whatever those cases really cost.
|
|
9
|
+
# Shared by the rake task, the pass_eval matcher and assert_eval_passes.
|
|
10
|
+
module UnknownCostGate
|
|
11
|
+
# Returns the failure message under :refuse, nil when the gate holds.
|
|
12
|
+
def self.check(reports, mode:)
|
|
13
|
+
names = reports.flat_map(&:unknown_cost_results).map(&:name).uniq
|
|
14
|
+
return nil if names.empty?
|
|
15
|
+
|
|
16
|
+
if mode == :warn
|
|
17
|
+
warn "[ruby_llm-contract] #{describe(names)} - maximum_cost checked against priced cases only"
|
|
18
|
+
return nil
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
"#{describe(names)}. Register pricing via CostCalculator.register_model " \
|
|
22
|
+
"or set on_unknown_pricing to :warn to check maximum_cost against priced cases only."
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
def self.describe(names)
|
|
26
|
+
"#{names.length} case(s) have unknown cost (model has no pricing data): #{names.join(", ")}"
|
|
27
|
+
end
|
|
28
|
+
private_class_method :describe
|
|
29
|
+
end
|
|
30
|
+
end
|
|
31
|
+
end
|
|
32
|
+
end
|
|
@@ -22,6 +22,7 @@ require_relative "eval/report_presenter"
|
|
|
22
22
|
require_relative "eval/report_storage"
|
|
23
23
|
require_relative "eval/report"
|
|
24
24
|
require_relative "eval/aggregated_report"
|
|
25
|
+
require_relative "eval/unknown_cost_gate"
|
|
25
26
|
require_relative "eval/eval_definition"
|
|
26
27
|
require_relative "eval/candidate_label"
|
|
27
28
|
require_relative "eval/model_comparison"
|
|
@@ -30,7 +30,9 @@ module RubyLLM
|
|
|
30
30
|
refute result.ok?, msg || "Expected step result NOT to satisfy contract, but it passed"
|
|
31
31
|
end
|
|
32
32
|
|
|
33
|
-
def assert_eval_passes(step, eval_name, minimum_score: nil, maximum_cost: nil, context: {}, msg: nil
|
|
33
|
+
def assert_eval_passes(step, eval_name, minimum_score: nil, maximum_cost: nil, context: {}, msg: nil,
|
|
34
|
+
on_unknown_pricing: UnknownPolicy::DEFAULT)
|
|
35
|
+
UnknownPolicy.validate!("on_unknown_pricing", on_unknown_pricing)
|
|
34
36
|
report = step.run_eval(eval_name, context: context)
|
|
35
37
|
|
|
36
38
|
if minimum_score
|
|
@@ -42,6 +44,9 @@ module RubyLLM
|
|
|
42
44
|
end
|
|
43
45
|
|
|
44
46
|
if maximum_cost
|
|
47
|
+
unknown_cost_failure = Eval::UnknownCostGate.check([report], mode: on_unknown_pricing)
|
|
48
|
+
assert unknown_cost_failure.nil?,
|
|
49
|
+
msg || "Expected #{eval_name} eval to stay within maximum cost, but #{unknown_cost_failure}"
|
|
45
50
|
assert report.total_cost <= maximum_cost,
|
|
46
51
|
msg || "Expected #{eval_name} eval cost <= $#{format("%.4f", maximum_cost)}, got $#{format("%.4f", report.total_cost)}"
|
|
47
52
|
end
|
|
@@ -33,6 +33,13 @@ module RubyLLM
|
|
|
33
33
|
value.dig(*rest)
|
|
34
34
|
end
|
|
35
35
|
|
|
36
|
+
# total_cost sums only the steps that could be priced.
|
|
37
|
+
def cost_unknown?
|
|
38
|
+
return false unless @step_traces.is_a?(Array)
|
|
39
|
+
|
|
40
|
+
@step_traces.any? { |step_trace| step_trace.respond_to?(:cost_unknown?) && step_trace.cost_unknown? }
|
|
41
|
+
end
|
|
42
|
+
|
|
36
43
|
def to_h
|
|
37
44
|
{ trace_id: @trace_id, total_latency_ms: @total_latency_ms,
|
|
38
45
|
total_usage: @total_usage, step_traces: @step_traces,
|
|
@@ -6,7 +6,7 @@ module RubyLLM
|
|
|
6
6
|
# Ignore eval/ subdirs BEFORE Zeitwerk setup — eval files don't define
|
|
7
7
|
# constants, they call define_eval on existing Step classes.
|
|
8
8
|
initializer "ruby_llm_contract.ignore_eval_dirs", before: :set_autoload_paths do |app|
|
|
9
|
-
|
|
9
|
+
RubyLLM::Contract::RAILS_EVAL_DIRS.each do |path|
|
|
10
10
|
full = app.root.join(path)
|
|
11
11
|
next unless full.exist?
|
|
12
12
|
|
|
@@ -4,8 +4,9 @@ module RubyLLM
|
|
|
4
4
|
module Contract
|
|
5
5
|
class RakeTask < ::Rake::TaskLib
|
|
6
6
|
# Gate ordering (preserved from pre-refactor behaviour):
|
|
7
|
-
# 1. cost gate runs FIRST — if `maximum_cost` set and exceeded,
|
|
8
|
-
#
|
|
7
|
+
# 1. cost gate runs FIRST — if `maximum_cost` set and exceeded, or any
|
|
8
|
+
# case has unknown cost under on_unknown_pricing :refuse, the suite
|
|
9
|
+
# aborts before any score check; passed_reports is empty.
|
|
9
10
|
# 2. score gate runs per-report; a report passes if
|
|
10
11
|
# `report_meets_score?` AND `!check_regression`.
|
|
11
12
|
# 3. overall passed = ALL reports passed AND cost gate not tripped.
|
|
@@ -16,20 +17,24 @@ module RubyLLM
|
|
|
16
17
|
end
|
|
17
18
|
end
|
|
18
19
|
|
|
19
|
-
def self.evaluate(host_reports:, minimum_score:, maximum_cost:, fail_on_regression
|
|
20
|
+
def self.evaluate(host_reports:, minimum_score:, maximum_cost:, fail_on_regression:,
|
|
21
|
+
on_unknown_pricing: UnknownPolicy::DEFAULT)
|
|
20
22
|
new(host_reports: host_reports,
|
|
21
23
|
minimum_score: minimum_score,
|
|
22
24
|
maximum_cost: maximum_cost,
|
|
23
|
-
fail_on_regression: fail_on_regression
|
|
25
|
+
fail_on_regression: fail_on_regression,
|
|
26
|
+
on_unknown_pricing: on_unknown_pricing).verdict
|
|
24
27
|
end
|
|
25
28
|
|
|
26
29
|
attr_reader :verdict
|
|
27
30
|
|
|
28
|
-
def initialize(host_reports:, minimum_score:, maximum_cost:, fail_on_regression
|
|
31
|
+
def initialize(host_reports:, minimum_score:, maximum_cost:, fail_on_regression:,
|
|
32
|
+
on_unknown_pricing: UnknownPolicy::DEFAULT)
|
|
29
33
|
@host_reports = host_reports
|
|
30
34
|
@minimum_score = minimum_score
|
|
31
35
|
@maximum_cost = maximum_cost
|
|
32
36
|
@fail_on_regression = fail_on_regression
|
|
37
|
+
@on_unknown_pricing = UnknownPolicy.validate!("on_unknown_pricing", on_unknown_pricing)
|
|
33
38
|
@verdict = build_verdict
|
|
34
39
|
end
|
|
35
40
|
|
|
@@ -37,11 +42,12 @@ module RubyLLM
|
|
|
37
42
|
|
|
38
43
|
def build_verdict
|
|
39
44
|
suite_cost = compute_suite_cost
|
|
45
|
+
cost_failure = cost_failure_message(suite_cost)
|
|
40
46
|
|
|
41
|
-
if
|
|
47
|
+
if cost_failure
|
|
42
48
|
return Verdict.new(
|
|
43
49
|
passed: false,
|
|
44
|
-
abort_reason:
|
|
50
|
+
abort_reason: cost_failure,
|
|
45
51
|
passed_reports: [],
|
|
46
52
|
suite_cost: suite_cost
|
|
47
53
|
)
|
|
@@ -50,7 +56,7 @@ module RubyLLM
|
|
|
50
56
|
passed_reports, all_passed = score_each_report
|
|
51
57
|
Verdict.new(
|
|
52
58
|
passed: all_passed,
|
|
53
|
-
abort_reason: all_passed ? nil : "
|
|
59
|
+
abort_reason: all_passed ? nil : "one or more evals did not pass",
|
|
54
60
|
passed_reports: passed_reports,
|
|
55
61
|
suite_cost: suite_cost
|
|
56
62
|
)
|
|
@@ -60,8 +66,11 @@ module RubyLLM
|
|
|
60
66
|
@host_reports.sum { |_host, report| report.total_cost }
|
|
61
67
|
end
|
|
62
68
|
|
|
63
|
-
def
|
|
64
|
-
|
|
69
|
+
def cost_failure_message(suite_cost)
|
|
70
|
+
return nil unless @maximum_cost
|
|
71
|
+
return cost_abort_message(suite_cost) if suite_cost > @maximum_cost
|
|
72
|
+
|
|
73
|
+
Eval::UnknownCostGate.check(@host_reports.map(&:last), mode: @on_unknown_pricing)
|
|
65
74
|
end
|
|
66
75
|
|
|
67
76
|
def cost_abort_message(suite_cost)
|
|
@@ -2,13 +2,15 @@
|
|
|
2
2
|
|
|
3
3
|
require "rake"
|
|
4
4
|
require "rake/tasklib"
|
|
5
|
+
require_relative "unknown_policy"
|
|
5
6
|
require_relative "rake_task/suite_gate"
|
|
6
7
|
|
|
7
8
|
module RubyLLM
|
|
8
9
|
module Contract
|
|
9
10
|
class RakeTask < ::Rake::TaskLib
|
|
10
11
|
attr_accessor :name, :context, :fail_on_empty, :minimum_score, :maximum_cost,
|
|
11
|
-
:eval_dirs, :save_baseline, :fail_on_regression, :track_history
|
|
12
|
+
:eval_dirs, :save_baseline, :fail_on_regression, :track_history,
|
|
13
|
+
:on_unknown_pricing
|
|
12
14
|
|
|
13
15
|
def initialize(name = :"ruby_llm_contract:eval", &block)
|
|
14
16
|
super()
|
|
@@ -18,10 +20,13 @@ module RubyLLM
|
|
|
18
20
|
@minimum_score = nil # nil = require 100%; float = threshold
|
|
19
21
|
@maximum_cost = nil # nil = no cost limit; float = budget cap (suite-level)
|
|
20
22
|
@eval_dirs = [] # directories to load eval files from (non-Rails)
|
|
23
|
+
# With maximum_cost: :refuse fails on cases the budget cannot price; :warn gates priced cases only
|
|
24
|
+
@on_unknown_pricing = UnknownPolicy::DEFAULT
|
|
21
25
|
@save_baseline = false
|
|
22
26
|
@fail_on_regression = false
|
|
23
27
|
@track_history = false
|
|
24
28
|
block&.call(self)
|
|
29
|
+
UnknownPolicy.validate!("on_unknown_pricing", @on_unknown_pricing)
|
|
25
30
|
define_task
|
|
26
31
|
end
|
|
27
32
|
|
|
@@ -44,7 +49,8 @@ module RubyLLM
|
|
|
44
49
|
host_reports: host_reports,
|
|
45
50
|
minimum_score: @minimum_score,
|
|
46
51
|
maximum_cost: @maximum_cost,
|
|
47
|
-
fail_on_regression: @fail_on_regression
|
|
52
|
+
fail_on_regression: @fail_on_regression,
|
|
53
|
+
on_unknown_pricing: @on_unknown_pricing
|
|
48
54
|
)
|
|
49
55
|
|
|
50
56
|
abort "\nEval suite FAILED: #{verdict.abort_reason}" unless verdict.passed?
|
|
@@ -128,7 +134,7 @@ module RubyLLM
|
|
|
128
134
|
abort("STEP is required, e.g. STEP=MatchProblemsToPages") if step_name.empty?
|
|
129
135
|
raw_candidates = ENV["CANDIDATES"].to_s.strip
|
|
130
136
|
abort("CANDIDATES is required, e.g. CANDIDATES=gpt-5-nano,gpt-5-mini@low,gpt-5-mini") if raw_candidates.empty?
|
|
131
|
-
min_score = ENV.fetch("MIN_SCORE"
|
|
137
|
+
min_score = ENV.fetch("MIN_SCORE") { Eval::DEFAULT_MIN_SCORE }.to_f
|
|
132
138
|
runs = parse_runs(ENV.fetch("RUNS", "1"))
|
|
133
139
|
|
|
134
140
|
host = RubyLLM::Contract.eval_hosts.find { |h| h.name == step_name }
|