ruby_llm-contract 1.1.0 → 1.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +86 -0
- data/README.md +12 -3
- data/docs/guide/getting_started.md +17 -1
- data/docs/guide/relation_to_agent.md +8 -2
- data/lib/ruby_llm/contract/adapters/response.rb +28 -2
- data/lib/ruby_llm/contract/adapters/ruby_llm.rb +72 -18
- data/lib/ruby_llm/contract/concerns/eval_host.rb +2 -2
- data/lib/ruby_llm/contract/concerns/usage_aggregator.rb +8 -5
- data/lib/ruby_llm/contract/cost_calculator.rb +69 -41
- data/lib/ruby_llm/contract/eval/case_executor.rb +1 -1
- data/lib/ruby_llm/contract/eval/case_result.rb +3 -0
- data/lib/ruby_llm/contract/eval/eval_history.rb +2 -1
- data/lib/ruby_llm/contract/eval/evaluator/proc_evaluator.rb +6 -2
- data/lib/ruby_llm/contract/eval/model_comparison.rb +11 -1
- data/lib/ruby_llm/contract/eval/prompt_diff_serializer.rb +1 -1
- data/lib/ruby_llm/contract/eval/recommender.rb +12 -5
- data/lib/ruby_llm/contract/eval/report.rb +4 -1
- data/lib/ruby_llm/contract/eval/report_stats.rb +10 -5
- data/lib/ruby_llm/contract/eval/report_storage.rb +14 -8
- data/lib/ruby_llm/contract/eval/unknown_cost_gate.rb +0 -8
- data/lib/ruby_llm/contract/minitest.rb +2 -2
- data/lib/ruby_llm/contract/pipeline/runner.rb +8 -0
- data/lib/ruby_llm/contract/pipeline/trace.rb +10 -0
- data/lib/ruby_llm/contract/railtie.rb +1 -1
- data/lib/ruby_llm/contract/rake_task/suite_gate.rb +4 -4
- data/lib/ruby_llm/contract/rake_task.rb +4 -4
- data/lib/ruby_llm/contract/rspec/pass_eval.rb +2 -2
- data/lib/ruby_llm/contract/step/base.rb +21 -12
- data/lib/ruby_llm/contract/step/dsl.rb +5 -14
- data/lib/ruby_llm/contract/step/limit_checker.rb +2 -1
- data/lib/ruby_llm/contract/step/result_builder.rb +32 -3
- data/lib/ruby_llm/contract/step/retry_executor.rb +35 -4
- data/lib/ruby_llm/contract/step/runner.rb +7 -1
- data/lib/ruby_llm/contract/step/runner_config.rb +2 -2
- data/lib/ruby_llm/contract/step/trace.rb +54 -20
- data/lib/ruby_llm/contract/token_estimator.rb +5 -3
- data/lib/ruby_llm/contract/unknown_policy.rb +21 -0
- data/lib/ruby_llm/contract/version.rb +1 -1
- data/lib/ruby_llm/contract.rb +17 -4
- metadata +2 -1
|
@@ -69,7 +69,8 @@ module RubyLLM
|
|
|
69
69
|
|
|
70
70
|
lines = ["#{runs.length} runs"]
|
|
71
71
|
runs.last(5).each do |r|
|
|
72
|
-
|
|
72
|
+
cost = format("%.6f", r[:total_cost] || r[:cost] || 0)
|
|
73
|
+
lines << " #{r[:date]} score=#{r[:score].round(2)} cost=$#{cost}#{" + unknown" if r[:cost_unknown]}"
|
|
73
74
|
end
|
|
74
75
|
lines.join("\n")
|
|
75
76
|
end
|
|
@@ -6,6 +6,10 @@ module RubyLLM
|
|
|
6
6
|
module Evaluator
|
|
7
7
|
# Adapts custom Ruby callables to the EvaluationResult contract.
|
|
8
8
|
class ProcEvaluator
|
|
9
|
+
# Report hides these under a failure: they repeat the PASS/FAIL label.
|
|
10
|
+
PASSED_DETAILS = "passed"
|
|
11
|
+
FAILED_DETAILS = "not passed"
|
|
12
|
+
|
|
9
13
|
def initialize(callable)
|
|
10
14
|
@callable = callable
|
|
11
15
|
end
|
|
@@ -36,9 +40,9 @@ module RubyLLM
|
|
|
36
40
|
def build_evaluation_result(result)
|
|
37
41
|
case result
|
|
38
42
|
when true
|
|
39
|
-
EvaluationResult.new(score: 1.0, passed: true, details:
|
|
43
|
+
EvaluationResult.new(score: 1.0, passed: true, details: PASSED_DETAILS)
|
|
40
44
|
when false
|
|
41
|
-
EvaluationResult.new(score: 0.0, passed: false, details:
|
|
45
|
+
EvaluationResult.new(score: 0.0, passed: false, details: FAILED_DETAILS)
|
|
42
46
|
when Numeric
|
|
43
47
|
EvaluationResult.new(score: result, passed: result >= 0.5, details: "custom score: #{result}")
|
|
44
48
|
else
|
|
@@ -36,8 +36,12 @@ module RubyLLM
|
|
|
36
36
|
@reports[resolve_key(candidate)]&.total_cost
|
|
37
37
|
end
|
|
38
38
|
|
|
39
|
+
# A report with unpriced cases has only a subtotal, so it cannot be
|
|
40
|
+
# called the cheapest.
|
|
39
41
|
def best_for(min_score: 0.0)
|
|
40
|
-
eligible = @reports.select
|
|
42
|
+
eligible = @reports.select do |_, report|
|
|
43
|
+
report.score > 0.0 && report.score >= min_score && !cost_unknown?(report)
|
|
44
|
+
end
|
|
41
45
|
return nil if eligible.empty?
|
|
42
46
|
|
|
43
47
|
eligible.min_by { |_, report| report.total_cost }&.first
|
|
@@ -45,6 +49,8 @@ module RubyLLM
|
|
|
45
49
|
|
|
46
50
|
def cost_per_point
|
|
47
51
|
@reports.transform_values do |report|
|
|
52
|
+
next Float::INFINITY if cost_unknown?(report)
|
|
53
|
+
|
|
48
54
|
report.score.positive? ? report.total_cost / report.score : Float::INFINITY
|
|
49
55
|
end
|
|
50
56
|
end
|
|
@@ -163,6 +169,10 @@ module RubyLLM
|
|
|
163
169
|
)
|
|
164
170
|
end
|
|
165
171
|
|
|
172
|
+
def cost_unknown?(report)
|
|
173
|
+
report.respond_to?(:unknown_cost_results) && report.unknown_cost_results.any?
|
|
174
|
+
end
|
|
175
|
+
|
|
166
176
|
def resolve_key(candidate)
|
|
167
177
|
case candidate
|
|
168
178
|
when String then candidate
|
|
@@ -5,7 +5,7 @@ module RubyLLM
|
|
|
5
5
|
module Eval
|
|
6
6
|
class PromptDiffSerializer
|
|
7
7
|
def call(report)
|
|
8
|
-
report.results.reject { |result| result.step_status ==
|
|
8
|
+
report.results.reject { |result| result.step_status == CaseResult::SKIPPED_STATUS }.map do |result|
|
|
9
9
|
{
|
|
10
10
|
name: result.name,
|
|
11
11
|
input: result.input,
|
|
@@ -40,7 +40,7 @@ module RubyLLM
|
|
|
40
40
|
report = @comparison.reports[label]
|
|
41
41
|
next nil unless report
|
|
42
42
|
|
|
43
|
-
evaluated_count = report.results.count { |r| r.step_status !=
|
|
43
|
+
evaluated_count = report.results.count { |r| r.step_status != CaseResult::SKIPPED_STATUS }
|
|
44
44
|
cases_count = [evaluated_count, 1].max
|
|
45
45
|
cost_per_call = report.total_cost.to_f / cases_count
|
|
46
46
|
|
|
@@ -51,7 +51,8 @@ module RubyLLM
|
|
|
51
51
|
cost_per_call: cost_per_call,
|
|
52
52
|
latency: report.avg_latency_ms || Float::INFINITY,
|
|
53
53
|
pass_rate_ratio: report.pass_rate_ratio,
|
|
54
|
-
total_cost: report.total_cost
|
|
54
|
+
total_cost: report.total_cost,
|
|
55
|
+
cost_unknown: unknown_cost?(report)
|
|
55
56
|
}
|
|
56
57
|
end
|
|
57
58
|
end
|
|
@@ -114,9 +115,9 @@ module RubyLLM
|
|
|
114
115
|
|
|
115
116
|
current_label = ModelComparison.candidate_label(@current_config)
|
|
116
117
|
current_report = @comparison.reports[current_label]
|
|
117
|
-
return {}
|
|
118
|
+
return {} if current_report.nil? || unknown_cost?(current_report)
|
|
118
119
|
|
|
119
|
-
current_evaluated = current_report.results.count { |r| r.step_status !=
|
|
120
|
+
current_evaluated = current_report.results.count { |r| r.step_status != CaseResult::SKIPPED_STATUS }
|
|
120
121
|
current_cases = [current_evaluated, 1].max
|
|
121
122
|
current_cost = current_report.total_cost.to_f / current_cases
|
|
122
123
|
diff = current_cost - best[:cost_per_call]
|
|
@@ -125,8 +126,14 @@ module RubyLLM
|
|
|
125
126
|
{ per_call: diff.round(6), monthly_at: { 10_000 => (diff * 10_000).round(2) } }
|
|
126
127
|
end
|
|
127
128
|
|
|
129
|
+
# A subtotal over unpriced cases is not a cost to rank by. Zero stays
|
|
130
|
+
# "unknown pricing" too: offline runs report $0 without any call.
|
|
128
131
|
def cost_known?(scored_candidate)
|
|
129
|
-
scored_candidate[:cost_per_call]&.positive?
|
|
132
|
+
!scored_candidate[:cost_unknown] && scored_candidate[:cost_per_call]&.positive?
|
|
133
|
+
end
|
|
134
|
+
|
|
135
|
+
def unknown_cost?(report)
|
|
136
|
+
report.respond_to?(:unknown_cost_results) && report.unknown_cost_results.any?
|
|
130
137
|
end
|
|
131
138
|
end
|
|
132
139
|
end
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
require "forwardable"
|
|
4
|
+
require_relative "evaluator/proc_evaluator"
|
|
4
5
|
|
|
5
6
|
module RubyLLM
|
|
6
7
|
module Contract
|
|
@@ -10,9 +11,11 @@ module RubyLLM
|
|
|
10
11
|
|
|
11
12
|
attr_reader :dataset_name, :results, :step_name
|
|
12
13
|
|
|
13
|
-
GENERIC_DETAILS = [
|
|
14
|
+
GENERIC_DETAILS = [Evaluator::ProcEvaluator::PASSED_DETAILS, Evaluator::ProcEvaluator::FAILED_DETAILS].freeze
|
|
14
15
|
HISTORY_DIR = ".eval_history"
|
|
16
|
+
HISTORY_EXT = "jsonl"
|
|
15
17
|
BASELINE_DIR = ".eval_baselines"
|
|
18
|
+
BASELINE_EXT = "json"
|
|
16
19
|
|
|
17
20
|
def_delegators :@stats, :score, :passed, :failed, :skipped, :failures, :pass_rate, :pass_rate_ratio,
|
|
18
21
|
:total_cost, :unknown_cost_results, :avg_latency_ms, :passed?,
|
|
@@ -23,7 +23,7 @@ module RubyLLM
|
|
|
23
23
|
end
|
|
24
24
|
|
|
25
25
|
def skipped
|
|
26
|
-
@results.count { |result| result.step_status ==
|
|
26
|
+
@results.count { |result| result.step_status == CaseResult::SKIPPED_STATUS }
|
|
27
27
|
end
|
|
28
28
|
|
|
29
29
|
def failures
|
|
@@ -63,7 +63,7 @@ module RubyLLM
|
|
|
63
63
|
end
|
|
64
64
|
|
|
65
65
|
def evaluated_results
|
|
66
|
-
@evaluated_results ||= @results.reject { |result| result.step_status ==
|
|
66
|
+
@evaluated_results ||= @results.reject { |result| result.step_status == CaseResult::SKIPPED_STATUS }
|
|
67
67
|
end
|
|
68
68
|
|
|
69
69
|
def evaluated_results_count
|
|
@@ -85,7 +85,7 @@ module RubyLLM
|
|
|
85
85
|
def single_shot_cost
|
|
86
86
|
return nil unless production_mode?
|
|
87
87
|
|
|
88
|
-
evaluated_results.sum { |r| first_attempt_cost(r) ||
|
|
88
|
+
evaluated_results.sum { |r| first_attempt_cost(r) || 0.0 }
|
|
89
89
|
end
|
|
90
90
|
|
|
91
91
|
def effective_cost
|
|
@@ -116,9 +116,14 @@ module RubyLLM
|
|
|
116
116
|
|
|
117
117
|
private
|
|
118
118
|
|
|
119
|
+
# With attempts, only the first one's cost: falling back to the case
|
|
120
|
+
# cost when it is unpriced would report a later attempt's spend as the
|
|
121
|
+
# single-shot cost.
|
|
119
122
|
def first_attempt_cost(result)
|
|
120
|
-
|
|
121
|
-
|
|
123
|
+
attempts = result.attempts || []
|
|
124
|
+
return result.cost if attempts.empty?
|
|
125
|
+
|
|
126
|
+
attempts.first[:cost]
|
|
122
127
|
end
|
|
123
128
|
|
|
124
129
|
def first_attempt_latency(result)
|
|
@@ -13,7 +13,7 @@ module RubyLLM
|
|
|
13
13
|
end
|
|
14
14
|
|
|
15
15
|
def save_history!(path: nil, model: nil, reasoning_effort: nil)
|
|
16
|
-
file = path || storage_path(Report::HISTORY_DIR,
|
|
16
|
+
file = path || storage_path(Report::HISTORY_DIR, Report::HISTORY_EXT, model: model, reasoning_effort: reasoning_effort)
|
|
17
17
|
entry = history_entry
|
|
18
18
|
entry[:model] = model if model
|
|
19
19
|
entry[:reasoning_effort] = reasoning_effort if reasoning_effort
|
|
@@ -22,19 +22,19 @@ module RubyLLM
|
|
|
22
22
|
end
|
|
23
23
|
|
|
24
24
|
def eval_history(path: nil, model: nil, reasoning_effort: nil)
|
|
25
|
-
EvalHistory.load(path || storage_path(Report::HISTORY_DIR,
|
|
26
|
-
|
|
25
|
+
EvalHistory.load(path || storage_path(Report::HISTORY_DIR, Report::HISTORY_EXT,
|
|
26
|
+
model: model, reasoning_effort: reasoning_effort))
|
|
27
27
|
end
|
|
28
28
|
|
|
29
29
|
def save_baseline!(path: nil, model: nil, reasoning_effort: nil)
|
|
30
|
-
file = path || storage_path(Report::BASELINE_DIR,
|
|
30
|
+
file = path || storage_path(Report::BASELINE_DIR, Report::BASELINE_EXT, model: model, reasoning_effort: reasoning_effort)
|
|
31
31
|
FileUtils.mkdir_p(File.dirname(file))
|
|
32
32
|
File.write(file, JSON.pretty_generate(serialize_for_baseline))
|
|
33
33
|
file
|
|
34
34
|
end
|
|
35
35
|
|
|
36
36
|
def compare_with_baseline(path: nil, model: nil, reasoning_effort: nil)
|
|
37
|
-
file = path || storage_path(Report::BASELINE_DIR,
|
|
37
|
+
file = path || storage_path(Report::BASELINE_DIR, Report::BASELINE_EXT, model: model, reasoning_effort: reasoning_effort)
|
|
38
38
|
raise ArgumentError, "No baseline found at #{file}" unless File.exist?(file)
|
|
39
39
|
|
|
40
40
|
baseline_data = JSON.parse(File.read(file), symbolize_names: true)
|
|
@@ -47,13 +47,15 @@ module RubyLLM
|
|
|
47
47
|
end
|
|
48
48
|
|
|
49
49
|
def baseline_exists?(path: nil, model: nil, reasoning_effort: nil)
|
|
50
|
-
File.exist?(path || storage_path(Report::BASELINE_DIR,
|
|
50
|
+
File.exist?(path || storage_path(Report::BASELINE_DIR, Report::BASELINE_EXT, model: model, reasoning_effort: reasoning_effort))
|
|
51
51
|
end
|
|
52
52
|
|
|
53
53
|
private
|
|
54
54
|
|
|
55
|
+
# cost_unknown is written only when true, so older files and runs that
|
|
56
|
+
# priced every case keep the same shape.
|
|
55
57
|
def history_entry
|
|
56
|
-
{
|
|
58
|
+
entry = {
|
|
57
59
|
date: Time.now.strftime("%Y-%m-%d"),
|
|
58
60
|
score: @stats.score,
|
|
59
61
|
total_cost: @stats.total_cost,
|
|
@@ -61,6 +63,8 @@ module RubyLLM
|
|
|
61
63
|
pass_rate_ratio: @stats.pass_rate_ratio,
|
|
62
64
|
cases_count: @stats.evaluated_results_count
|
|
63
65
|
}
|
|
66
|
+
entry[:cost_unknown] = true if @stats.unknown_cost_results.any?
|
|
67
|
+
entry
|
|
64
68
|
end
|
|
65
69
|
|
|
66
70
|
def serialize_for_baseline
|
|
@@ -74,13 +78,15 @@ module RubyLLM
|
|
|
74
78
|
end
|
|
75
79
|
|
|
76
80
|
def serialize_case(result)
|
|
77
|
-
{
|
|
81
|
+
serialized = {
|
|
78
82
|
name: result.name,
|
|
79
83
|
passed: result.passed?,
|
|
80
84
|
score: result.score,
|
|
81
85
|
details: result.details,
|
|
82
86
|
cost: result.cost
|
|
83
87
|
}
|
|
88
|
+
serialized[:cost_unknown] = true if result.cost_unknown?
|
|
89
|
+
serialized
|
|
84
90
|
end
|
|
85
91
|
|
|
86
92
|
def storage_path(root_dir, extension, model:, reasoning_effort: nil)
|
|
@@ -8,14 +8,6 @@ module RubyLLM
|
|
|
8
8
|
# gate that only compares totals passes whatever those cases really cost.
|
|
9
9
|
# Shared by the rake task, the pass_eval matcher and assert_eval_passes.
|
|
10
10
|
module UnknownCostGate
|
|
11
|
-
MODES = %i[refuse warn].freeze
|
|
12
|
-
|
|
13
|
-
def self.validate!(mode)
|
|
14
|
-
return mode if MODES.include?(mode)
|
|
15
|
-
|
|
16
|
-
raise ArgumentError, "on_unknown_pricing must be :refuse or :warn, got #{mode.inspect}"
|
|
17
|
-
end
|
|
18
|
-
|
|
19
11
|
# Returns the failure message under :refuse, nil when the gate holds.
|
|
20
12
|
def self.check(reports, mode:)
|
|
21
13
|
names = reports.flat_map(&:unknown_cost_results).map(&:name).uniq
|
|
@@ -31,8 +31,8 @@ module RubyLLM
|
|
|
31
31
|
end
|
|
32
32
|
|
|
33
33
|
def assert_eval_passes(step, eval_name, minimum_score: nil, maximum_cost: nil, context: {}, msg: nil,
|
|
34
|
-
on_unknown_pricing:
|
|
35
|
-
|
|
34
|
+
on_unknown_pricing: UnknownPolicy::DEFAULT)
|
|
35
|
+
UnknownPolicy.validate!("on_unknown_pricing", on_unknown_pricing)
|
|
36
36
|
report = step.run_eval(eval_name, context: context)
|
|
37
37
|
|
|
38
38
|
if minimum_score
|
|
@@ -83,6 +83,7 @@ module RubyLLM
|
|
|
83
83
|
total_usage: aggregate_usage(traces),
|
|
84
84
|
step_traces: traces
|
|
85
85
|
)
|
|
86
|
+
warn_incomplete_budget_usage(trace)
|
|
86
87
|
|
|
87
88
|
Result.new(
|
|
88
89
|
status: execution.status, step_results: execution.step_results,
|
|
@@ -91,6 +92,13 @@ module RubyLLM
|
|
|
91
92
|
)
|
|
92
93
|
end
|
|
93
94
|
|
|
95
|
+
def warn_incomplete_budget_usage(trace)
|
|
96
|
+
return unless @token_budget && !trace.usage_complete?
|
|
97
|
+
|
|
98
|
+
warn "[ruby_llm-contract] token_budget checked against incomplete token counts: " \
|
|
99
|
+
"a provider left out usage for at least one step, so the total undercounts"
|
|
100
|
+
end
|
|
101
|
+
|
|
94
102
|
def elapsed_ms(start_time)
|
|
95
103
|
((Process.clock_gettime(Process::CLOCK_MONOTONIC) - start_time) * 1000).round
|
|
96
104
|
end
|
|
@@ -40,6 +40,16 @@ module RubyLLM
|
|
|
40
40
|
@step_traces.any? { |step_trace| step_trace.respond_to?(:cost_unknown?) && step_trace.cost_unknown? }
|
|
41
41
|
end
|
|
42
42
|
|
|
43
|
+
# False when any step's provider left out token counts, so total_usage
|
|
44
|
+
# (and a token_budget checked against it) undercounts.
|
|
45
|
+
def usage_complete?
|
|
46
|
+
return true unless @step_traces.is_a?(Array)
|
|
47
|
+
|
|
48
|
+
@step_traces.none? do |step_trace|
|
|
49
|
+
step_trace.respond_to?(:usage_complete) && step_trace.usage_complete == false
|
|
50
|
+
end
|
|
51
|
+
end
|
|
52
|
+
|
|
43
53
|
def to_h
|
|
44
54
|
{ trace_id: @trace_id, total_latency_ms: @total_latency_ms,
|
|
45
55
|
total_usage: @total_usage, step_traces: @step_traces,
|
|
@@ -6,7 +6,7 @@ module RubyLLM
|
|
|
6
6
|
# Ignore eval/ subdirs BEFORE Zeitwerk setup — eval files don't define
|
|
7
7
|
# constants, they call define_eval on existing Step classes.
|
|
8
8
|
initializer "ruby_llm_contract.ignore_eval_dirs", before: :set_autoload_paths do |app|
|
|
9
|
-
|
|
9
|
+
RubyLLM::Contract::RAILS_EVAL_DIRS.each do |path|
|
|
10
10
|
full = app.root.join(path)
|
|
11
11
|
next unless full.exist?
|
|
12
12
|
|
|
@@ -18,7 +18,7 @@ module RubyLLM
|
|
|
18
18
|
end
|
|
19
19
|
|
|
20
20
|
def self.evaluate(host_reports:, minimum_score:, maximum_cost:, fail_on_regression:,
|
|
21
|
-
on_unknown_pricing:
|
|
21
|
+
on_unknown_pricing: UnknownPolicy::DEFAULT)
|
|
22
22
|
new(host_reports: host_reports,
|
|
23
23
|
minimum_score: minimum_score,
|
|
24
24
|
maximum_cost: maximum_cost,
|
|
@@ -29,12 +29,12 @@ module RubyLLM
|
|
|
29
29
|
attr_reader :verdict
|
|
30
30
|
|
|
31
31
|
def initialize(host_reports:, minimum_score:, maximum_cost:, fail_on_regression:,
|
|
32
|
-
on_unknown_pricing:
|
|
32
|
+
on_unknown_pricing: UnknownPolicy::DEFAULT)
|
|
33
33
|
@host_reports = host_reports
|
|
34
34
|
@minimum_score = minimum_score
|
|
35
35
|
@maximum_cost = maximum_cost
|
|
36
36
|
@fail_on_regression = fail_on_regression
|
|
37
|
-
@on_unknown_pricing =
|
|
37
|
+
@on_unknown_pricing = UnknownPolicy.validate!("on_unknown_pricing", on_unknown_pricing)
|
|
38
38
|
@verdict = build_verdict
|
|
39
39
|
end
|
|
40
40
|
|
|
@@ -56,7 +56,7 @@ module RubyLLM
|
|
|
56
56
|
passed_reports, all_passed = score_each_report
|
|
57
57
|
Verdict.new(
|
|
58
58
|
passed: all_passed,
|
|
59
|
-
abort_reason: all_passed ? nil : "
|
|
59
|
+
abort_reason: all_passed ? nil : "one or more evals did not pass",
|
|
60
60
|
passed_reports: passed_reports,
|
|
61
61
|
suite_cost: suite_cost
|
|
62
62
|
)
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
require "rake"
|
|
4
4
|
require "rake/tasklib"
|
|
5
|
-
require_relative "
|
|
5
|
+
require_relative "unknown_policy"
|
|
6
6
|
require_relative "rake_task/suite_gate"
|
|
7
7
|
|
|
8
8
|
module RubyLLM
|
|
@@ -21,12 +21,12 @@ module RubyLLM
|
|
|
21
21
|
@maximum_cost = nil # nil = no cost limit; float = budget cap (suite-level)
|
|
22
22
|
@eval_dirs = [] # directories to load eval files from (non-Rails)
|
|
23
23
|
# With maximum_cost: :refuse fails on cases the budget cannot price; :warn gates priced cases only
|
|
24
|
-
@on_unknown_pricing =
|
|
24
|
+
@on_unknown_pricing = UnknownPolicy::DEFAULT
|
|
25
25
|
@save_baseline = false
|
|
26
26
|
@fail_on_regression = false
|
|
27
27
|
@track_history = false
|
|
28
28
|
block&.call(self)
|
|
29
|
-
|
|
29
|
+
UnknownPolicy.validate!("on_unknown_pricing", @on_unknown_pricing)
|
|
30
30
|
define_task
|
|
31
31
|
end
|
|
32
32
|
|
|
@@ -134,7 +134,7 @@ module RubyLLM
|
|
|
134
134
|
abort("STEP is required, e.g. STEP=MatchProblemsToPages") if step_name.empty?
|
|
135
135
|
raw_candidates = ENV["CANDIDATES"].to_s.strip
|
|
136
136
|
abort("CANDIDATES is required, e.g. CANDIDATES=gpt-5-nano,gpt-5-mini@low,gpt-5-mini") if raw_candidates.empty?
|
|
137
|
-
min_score = ENV.fetch("MIN_SCORE"
|
|
137
|
+
min_score = ENV.fetch("MIN_SCORE") { Eval::DEFAULT_MIN_SCORE }.to_f
|
|
138
138
|
runs = parse_runs(ENV.fetch("RUNS", "1"))
|
|
139
139
|
|
|
140
140
|
host = RubyLLM::Contract.eval_hosts.find { |h| h.name == step_name }
|
|
@@ -65,7 +65,7 @@ RSpec::Matchers.define :pass_eval do |eval_name|
|
|
|
65
65
|
end
|
|
66
66
|
|
|
67
67
|
chain :on_unknown_pricing do |mode|
|
|
68
|
-
@on_unknown_pricing = RubyLLM::Contract::
|
|
68
|
+
@on_unknown_pricing = RubyLLM::Contract::UnknownPolicy.validate!("on_unknown_pricing", mode)
|
|
69
69
|
end
|
|
70
70
|
|
|
71
71
|
chain :without_regressions do
|
|
@@ -82,7 +82,7 @@ RSpec::Matchers.define :pass_eval do |eval_name|
|
|
|
82
82
|
@context ||= {}
|
|
83
83
|
@minimum_score ||= nil
|
|
84
84
|
@maximum_cost ||= nil
|
|
85
|
-
@on_unknown_pricing ||=
|
|
85
|
+
@on_unknown_pricing ||= RubyLLM::Contract::UnknownPolicy::DEFAULT
|
|
86
86
|
@unknown_cost_failure = nil
|
|
87
87
|
@unknown_cost_only = false
|
|
88
88
|
@check_regressions ||= false
|
|
@@ -26,9 +26,9 @@ module RubyLLM
|
|
|
26
26
|
context: context).results.first
|
|
27
27
|
end
|
|
28
28
|
|
|
29
|
-
def estimate_cost(input:, model: nil, attachment: nil)
|
|
29
|
+
def estimate_cost(input:, model: nil, attachment: nil, provider: nil)
|
|
30
30
|
model_name = estimated_model_name(model)
|
|
31
|
-
model_info = CostCalculator.find_model(model_name)
|
|
31
|
+
model_info = CostCalculator.find_model(model_name, provider: provider)
|
|
32
32
|
return nil unless model_info
|
|
33
33
|
|
|
34
34
|
text_tokens = TokenEstimator.estimate(build_messages(input))
|
|
@@ -48,12 +48,13 @@ module RubyLLM
|
|
|
48
48
|
output_tokens_estimate: output_tokens,
|
|
49
49
|
estimated_cost: CostCalculator.calculate(
|
|
50
50
|
model_name: model_name,
|
|
51
|
-
usage: { input_tokens: input_tokens, output_tokens: output_tokens }
|
|
51
|
+
usage: { input_tokens: input_tokens, output_tokens: output_tokens },
|
|
52
|
+
provider: provider
|
|
52
53
|
)
|
|
53
54
|
}
|
|
54
55
|
end
|
|
55
56
|
|
|
56
|
-
def estimate_eval_cost(eval_name, models: nil)
|
|
57
|
+
def estimate_eval_cost(eval_name, models: nil, provider: nil)
|
|
57
58
|
defn = send(:all_eval_definitions)[eval_name.to_s]
|
|
58
59
|
raise ArgumentError, "No eval '#{eval_name}' defined" unless defn
|
|
59
60
|
|
|
@@ -61,7 +62,7 @@ module RubyLLM
|
|
|
61
62
|
cases = defn.build_dataset.cases
|
|
62
63
|
|
|
63
64
|
model_list.each_with_object({}) do |model_name, result|
|
|
64
|
-
result[model_name] = estimate_eval_cost_for_model(cases, model_name)
|
|
65
|
+
result[model_name] = estimate_eval_cost_for_model(cases, model_name, provider)
|
|
65
66
|
end
|
|
66
67
|
end
|
|
67
68
|
|
|
@@ -90,8 +91,10 @@ module RubyLLM
|
|
|
90
91
|
).call
|
|
91
92
|
end
|
|
92
93
|
|
|
93
|
-
|
|
94
|
-
|
|
94
|
+
# Forwarded to the adapter as-is. A key known but not forwarded would be
|
|
95
|
+
# silently dropped, so the known list is built from this one.
|
|
96
|
+
ADAPTER_CONTEXT_KEYS = %i[provider assume_model_exists max_tokens reasoning_effort attachment].freeze
|
|
97
|
+
KNOWN_CONTEXT_KEYS = (%i[adapter model temperature retry_policy_override] + ADAPTER_CONTEXT_KEYS).freeze
|
|
95
98
|
|
|
96
99
|
include Concerns::ContextHelpers
|
|
97
100
|
|
|
@@ -133,7 +136,7 @@ module RubyLLM
|
|
|
133
136
|
estimate = attachment_token_estimate if respond_to?(:attachment_token_estimate)
|
|
134
137
|
return [estimate, false] unless estimate.nil?
|
|
135
138
|
|
|
136
|
-
mode = respond_to?(:on_unknown_attachment_size) ? on_unknown_attachment_size :
|
|
139
|
+
mode = respond_to?(:on_unknown_attachment_size) ? on_unknown_attachment_size : UnknownPolicy::DEFAULT
|
|
137
140
|
if mode == :warn
|
|
138
141
|
warn "[ruby_llm-contract] attachment present but attachment_token_estimate not " \
|
|
139
142
|
"declared on #{name || self} — estimate_cost proceeds without attachment cost"
|
|
@@ -143,9 +146,9 @@ module RubyLLM
|
|
|
143
146
|
[0, true]
|
|
144
147
|
end
|
|
145
148
|
|
|
146
|
-
def estimate_eval_cost_for_model(cases, model_name)
|
|
149
|
+
def estimate_eval_cost_for_model(cases, model_name, provider)
|
|
147
150
|
cases.sum do |test_case|
|
|
148
|
-
estimate = estimate_cost(input: test_case.input, model: model_name)
|
|
151
|
+
estimate = estimate_cost(input: test_case.input, model: model_name, provider: provider)
|
|
149
152
|
# Two misses floor to 0.0, not one: model absent from the registry
|
|
150
153
|
# (nil estimate), or present with unreadable pricing (hash whose
|
|
151
154
|
# estimated_cost is nil). Documented as a floor, not a fail-closed.
|
|
@@ -193,7 +196,7 @@ module RubyLLM
|
|
|
193
196
|
|
|
194
197
|
def runtime_settings(context)
|
|
195
198
|
policy = context.key?(:retry_policy_override) ? context[:retry_policy_override] : retry_policy
|
|
196
|
-
extra = context.slice(
|
|
199
|
+
extra = context.slice(*ADAPTER_CONTEXT_KEYS)
|
|
197
200
|
|
|
198
201
|
# Always pass the class-level `thinking` config to the adapter when
|
|
199
202
|
# set, so fields like `budget` survive a per-call `reasoning_effort`
|
|
@@ -285,12 +288,18 @@ module RubyLLM
|
|
|
285
288
|
"model=#{trace.model} status=#{result.status} " \
|
|
286
289
|
"latency=#{trace.latency_ms}ms " \
|
|
287
290
|
"tokens=#{trace.usage&.dig(:input_tokens) || 0}+#{trace.usage&.dig(:output_tokens) || 0} " \
|
|
288
|
-
"cost
|
|
291
|
+
"cost=#{log_cost(trace)}"
|
|
289
292
|
logger.info(msg)
|
|
290
293
|
|
|
291
294
|
log_failed_observations(result, logger)
|
|
292
295
|
end
|
|
293
296
|
|
|
297
|
+
def log_cost(trace)
|
|
298
|
+
return "unknown" if trace.respond_to?(:cost_unknown?) && trace.cost_unknown?
|
|
299
|
+
|
|
300
|
+
"$#{format("%.6f", trace.cost || 0)}"
|
|
301
|
+
end
|
|
302
|
+
|
|
294
303
|
def log_failed_observations(result, logger)
|
|
295
304
|
failed = result.observations.select { |o| !o[:passed] }
|
|
296
305
|
return if failed.empty?
|
|
@@ -151,12 +151,10 @@ module RubyLLM
|
|
|
151
151
|
if amount
|
|
152
152
|
validate_positive!("max_cost", amount)
|
|
153
153
|
|
|
154
|
-
|
|
155
|
-
raise ArgumentError, "on_unknown_pricing must be :refuse or :warn, got #{on_unknown_pricing.inspect}"
|
|
156
|
-
end
|
|
154
|
+
UnknownPolicy.validate!("on_unknown_pricing", on_unknown_pricing) if on_unknown_pricing
|
|
157
155
|
|
|
158
156
|
@max_cost = amount
|
|
159
|
-
@on_unknown_pricing = on_unknown_pricing ||
|
|
157
|
+
@on_unknown_pricing = on_unknown_pricing || UnknownPolicy::DEFAULT
|
|
160
158
|
return @max_cost
|
|
161
159
|
end
|
|
162
160
|
|
|
@@ -164,7 +162,7 @@ module RubyLLM
|
|
|
164
162
|
end
|
|
165
163
|
|
|
166
164
|
def on_unknown_pricing
|
|
167
|
-
inherited_value(:on_unknown_pricing) ||
|
|
165
|
+
inherited_value(:on_unknown_pricing) || UnknownPolicy::DEFAULT
|
|
168
166
|
end
|
|
169
167
|
|
|
170
168
|
def attachment_token_estimate(n = nil)
|
|
@@ -182,16 +180,9 @@ module RubyLLM
|
|
|
182
180
|
end
|
|
183
181
|
|
|
184
182
|
def on_unknown_attachment_size(mode = nil)
|
|
185
|
-
if mode
|
|
186
|
-
unless %i[refuse warn].include?(mode)
|
|
187
|
-
raise ArgumentError,
|
|
188
|
-
"on_unknown_attachment_size must be :refuse or :warn, got #{mode.inspect}"
|
|
189
|
-
end
|
|
190
|
-
|
|
191
|
-
return @on_unknown_attachment_size = mode
|
|
192
|
-
end
|
|
183
|
+
return @on_unknown_attachment_size = UnknownPolicy.validate!("on_unknown_attachment_size", mode) if mode
|
|
193
184
|
|
|
194
|
-
inherited_value(:on_unknown_attachment_size) ||
|
|
185
|
+
inherited_value(:on_unknown_attachment_size) || UnknownPolicy::DEFAULT
|
|
195
186
|
end
|
|
196
187
|
|
|
197
188
|
def model(name = nil)
|
|
@@ -70,7 +70,8 @@ module RubyLLM
|
|
|
70
70
|
estimated_output = effective_max_output || (estimated * DEFAULT_OUTPUT_RATIO)
|
|
71
71
|
estimated_cost = CostCalculator.calculate(
|
|
72
72
|
model_name: model_name,
|
|
73
|
-
usage: { input_tokens: estimated, output_tokens: estimated_output }
|
|
73
|
+
usage: { input_tokens: estimated, output_tokens: estimated_output },
|
|
74
|
+
provider: pricing_provider
|
|
74
75
|
)
|
|
75
76
|
|
|
76
77
|
if estimated_cost.nil?
|
|
@@ -4,12 +4,21 @@ module RubyLLM
|
|
|
4
4
|
module Contract
|
|
5
5
|
module Step
|
|
6
6
|
class ResultBuilder
|
|
7
|
-
|
|
7
|
+
# Why the provider stopped, worded for a failed result. Anthropic also
|
|
8
|
+
# reports an exhausted context window as :max_tokens, so the message
|
|
9
|
+
# does not claim max_output was the limit hit.
|
|
10
|
+
FINISH_REASON_ERRORS = {
|
|
11
|
+
max_tokens: "provider reported token-limit termination",
|
|
12
|
+
content_filter: "provider reported content filtering or refusal"
|
|
13
|
+
}.freeze
|
|
14
|
+
|
|
15
|
+
def initialize(contract_definition:, output_type:, output_schema:, model:, observers:, max_output: nil)
|
|
8
16
|
@contract_definition = contract_definition
|
|
9
17
|
@output_type = output_type
|
|
10
18
|
@output_schema = output_schema
|
|
11
19
|
@model = model
|
|
12
20
|
@observers = observers
|
|
21
|
+
@max_output = max_output
|
|
13
22
|
end
|
|
14
23
|
|
|
15
24
|
def error_result(error_result:, messages:)
|
|
@@ -25,13 +34,14 @@ module RubyLLM
|
|
|
25
34
|
def success_result(response:, messages:, latency_ms:, input:)
|
|
26
35
|
raw_output = response.content
|
|
27
36
|
validation_result = validate_output(raw_output, input)
|
|
28
|
-
trace = Trace.new(messages: messages, model: @model, latency_ms: latency_ms, usage: response.usage
|
|
37
|
+
trace = Trace.new(messages: messages, model: @model, latency_ms: latency_ms, usage: response.usage,
|
|
38
|
+
**response_accounting(response))
|
|
29
39
|
|
|
30
40
|
Result.new(
|
|
31
41
|
status: validation_result[:status],
|
|
32
42
|
raw_output: raw_output,
|
|
33
43
|
parsed_output: validation_result[:parsed_output],
|
|
34
|
-
validation_errors: validation_result
|
|
44
|
+
validation_errors: validation_errors(validation_result, response),
|
|
35
45
|
trace: trace,
|
|
36
46
|
observations: observations_for(validation_result, input)
|
|
37
47
|
)
|
|
@@ -39,6 +49,25 @@ module RubyLLM
|
|
|
39
49
|
|
|
40
50
|
private
|
|
41
51
|
|
|
52
|
+
# Only a failed result says why the provider stopped; a result that
|
|
53
|
+
# passed validation keeps finish_reason in its trace alone.
|
|
54
|
+
def validation_errors(validation_result, response)
|
|
55
|
+
errors = validation_result[:errors]
|
|
56
|
+
return errors if validation_result[:status] == :ok
|
|
57
|
+
|
|
58
|
+
reason = response.respond_to?(:finish_reason) ? response.finish_reason : nil
|
|
59
|
+
message = FINISH_REASON_ERRORS[reason]
|
|
60
|
+
return errors unless message
|
|
61
|
+
|
|
62
|
+
message += "; configured max_output: #{@max_output}" if reason == :max_tokens && @max_output
|
|
63
|
+
[*errors, message]
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
# Custom adapters may return any object with `content` and `usage`.
|
|
67
|
+
def response_accounting(response)
|
|
68
|
+
response.respond_to?(:trace_accounting) ? response.trace_accounting : {}
|
|
69
|
+
end
|
|
70
|
+
|
|
42
71
|
def observations_for(validation_result, input)
|
|
43
72
|
return [] unless validation_result[:status] == :ok && @observers.any?
|
|
44
73
|
|