ruby_llm-contract 1.1.0 → 1.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +30 -0
- data/lib/ruby_llm/contract/concerns/eval_host.rb +2 -2
- data/lib/ruby_llm/contract/eval/case_executor.rb +1 -1
- data/lib/ruby_llm/contract/eval/case_result.rb +3 -0
- data/lib/ruby_llm/contract/eval/evaluator/proc_evaluator.rb +6 -2
- data/lib/ruby_llm/contract/eval/prompt_diff_serializer.rb +1 -1
- data/lib/ruby_llm/contract/eval/recommender.rb +2 -2
- data/lib/ruby_llm/contract/eval/report.rb +4 -1
- data/lib/ruby_llm/contract/eval/report_stats.rb +2 -2
- data/lib/ruby_llm/contract/eval/report_storage.rb +6 -6
- data/lib/ruby_llm/contract/eval/unknown_cost_gate.rb +0 -8
- data/lib/ruby_llm/contract/minitest.rb +2 -2
- data/lib/ruby_llm/contract/railtie.rb +1 -1
- data/lib/ruby_llm/contract/rake_task/suite_gate.rb +4 -4
- data/lib/ruby_llm/contract/rake_task.rb +4 -4
- data/lib/ruby_llm/contract/rspec/pass_eval.rb +2 -2
- data/lib/ruby_llm/contract/step/base.rb +6 -4
- data/lib/ruby_llm/contract/step/dsl.rb +5 -14
- data/lib/ruby_llm/contract/step/runner_config.rb +2 -2
- data/lib/ruby_llm/contract/unknown_policy.rb +21 -0
- data/lib/ruby_llm/contract/version.rb +1 -1
- data/lib/ruby_llm/contract.rb +17 -4
- metadata +2 -1
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 390af91f8200f1098440872d22515fa6232888d13d21a149c0b11f0bf3c15d48
|
|
4
|
+
data.tar.gz: d39d8a0c2e8914d08e4d996e6a3eb4d2c53613304cabcc6e077af2d9963ad4f5
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 81e03a20720536700acfe5bc840720a84675ca738e248fbdb490dac1dc63479c93924923d16cb28433329b53d24b796a42332736d29ec6f4d42a41240dfd4367
|
|
7
|
+
data.tar.gz: a433acb982c831c7ee1e2db5675135c4a42762d75c78e53708673b8ca50f5f165fe2cba9a088828b95738f24f29afb01e2033f694e92c69182662a6307492aa4
|
data/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,35 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 1.1.1 (2026-10-10)
|
|
4
|
+
|
|
5
|
+
### Fixed
|
|
6
|
+
|
|
7
|
+
- **A failed rake eval gate printed "Eval suite FAILED: Eval suite FAILED".** The
|
|
8
|
+
message now reads "Eval suite FAILED: one or more evals did not pass".
|
|
9
|
+
|
|
10
|
+
### Changed (internal, no API change)
|
|
11
|
+
|
|
12
|
+
- Values that two or more places must agree on now have one owner, so one side can no
|
|
13
|
+
longer drift alone: the `:refuse`/`:warn` modes and their default for every
|
|
14
|
+
`on_unknown_*` option (`UnknownPolicy`), the Rails contract and eval dirs that the
|
|
15
|
+
Railtie ignores and `load_evals!` loads, the reload flag, the `:skipped` step status
|
|
16
|
+
(`CaseResult::SKIPPED_STATUS`), baseline and history file extensions, the context keys
|
|
17
|
+
forwarded to the adapter, the generic "passed"/"not passed" details the report hides,
|
|
18
|
+
and the optimize task's `MIN_SCORE` default, which now reads `Eval::DEFAULT_MIN_SCORE`.
|
|
19
|
+
The unknown-context-key warning lists the known keys in a different order, and in a
|
|
20
|
+
Rails app `load_evals!` now loads `app/contracts/eval` before `app/steps/eval` (the order
|
|
21
|
+
the Railtie and eager loading already used); this matters only if both dirs define an
|
|
22
|
+
eval of the same name on the same class, where the later file wins.
|
|
23
|
+
|
|
24
|
+
### Tests
|
|
25
|
+
|
|
26
|
+
- A real attachment through the RubyLLM 2.x adapter with HTTP intercepted (webmock):
|
|
27
|
+
pins the `input_image` part on the wire and the usage read back.
|
|
28
|
+
- A minimal Rails app booted in a child process: pins that Zeitwerk ignores the
|
|
29
|
+
`eval/` dirs, that evals register after boot, and that the rake tasks appear.
|
|
30
|
+
- `rubocop` is pinned to `~> 1.92.0`: CI resolves without a lockfile, so a floating
|
|
31
|
+
version let new cops land on CI before they ran locally.
|
|
32
|
+
|
|
3
33
|
## 1.1.0 (2026-10-10)
|
|
4
34
|
|
|
5
35
|
### Changed - may turn a green CI red
|
|
@@ -16,13 +16,13 @@ module RubyLLM
|
|
|
16
16
|
@file_sourced_evals ||= Set.new
|
|
17
17
|
key = name.to_s
|
|
18
18
|
|
|
19
|
-
if @eval_definitions.key?(key) && !
|
|
19
|
+
if @eval_definitions.key?(key) && !Contract.reloading?
|
|
20
20
|
warn "[ruby_llm-contract] Redefining eval '#{key}' on #{self}. " \
|
|
21
21
|
"This replaces the previous definition."
|
|
22
22
|
end
|
|
23
23
|
|
|
24
24
|
@eval_definitions[key] = Eval::EvalDefinition.new(key, step_class: self, &)
|
|
25
|
-
@file_sourced_evals.add(key) if
|
|
25
|
+
@file_sourced_evals.add(key) if Contract.reloading?
|
|
26
26
|
Contract.register_eval_host(self)
|
|
27
27
|
register_subclasses(self)
|
|
28
28
|
end
|
|
@@ -7,6 +7,9 @@ module RubyLLM
|
|
|
7
7
|
# BaselineDiff#compute_score keys off this prefix to keep skipped cases
|
|
8
8
|
# out of the score denominator. Both ends must read it from here.
|
|
9
9
|
SKIPPED_DETAILS_PREFIX = "skipped:"
|
|
10
|
+
# Not run (no adapter). Readers compare step_status, so any result object
|
|
11
|
+
# exposing it works; kept out of score, pass rate and cost-per-call.
|
|
12
|
+
SKIPPED_STATUS = :skipped
|
|
10
13
|
|
|
11
14
|
def self.skipped_details(reason)
|
|
12
15
|
"#{SKIPPED_DETAILS_PREFIX} #{reason}"
|
|
@@ -6,6 +6,10 @@ module RubyLLM
|
|
|
6
6
|
module Evaluator
|
|
7
7
|
# Adapts custom Ruby callables to the EvaluationResult contract.
|
|
8
8
|
class ProcEvaluator
|
|
9
|
+
# Report hides these under a failure: they repeat the PASS/FAIL label.
|
|
10
|
+
PASSED_DETAILS = "passed"
|
|
11
|
+
FAILED_DETAILS = "not passed"
|
|
12
|
+
|
|
9
13
|
def initialize(callable)
|
|
10
14
|
@callable = callable
|
|
11
15
|
end
|
|
@@ -36,9 +40,9 @@ module RubyLLM
|
|
|
36
40
|
def build_evaluation_result(result)
|
|
37
41
|
case result
|
|
38
42
|
when true
|
|
39
|
-
EvaluationResult.new(score: 1.0, passed: true, details:
|
|
43
|
+
EvaluationResult.new(score: 1.0, passed: true, details: PASSED_DETAILS)
|
|
40
44
|
when false
|
|
41
|
-
EvaluationResult.new(score: 0.0, passed: false, details:
|
|
45
|
+
EvaluationResult.new(score: 0.0, passed: false, details: FAILED_DETAILS)
|
|
42
46
|
when Numeric
|
|
43
47
|
EvaluationResult.new(score: result, passed: result >= 0.5, details: "custom score: #{result}")
|
|
44
48
|
else
|
|
@@ -5,7 +5,7 @@ module RubyLLM
|
|
|
5
5
|
module Eval
|
|
6
6
|
class PromptDiffSerializer
|
|
7
7
|
def call(report)
|
|
8
|
-
report.results.reject { |result| result.step_status ==
|
|
8
|
+
report.results.reject { |result| result.step_status == CaseResult::SKIPPED_STATUS }.map do |result|
|
|
9
9
|
{
|
|
10
10
|
name: result.name,
|
|
11
11
|
input: result.input,
|
|
@@ -40,7 +40,7 @@ module RubyLLM
|
|
|
40
40
|
report = @comparison.reports[label]
|
|
41
41
|
next nil unless report
|
|
42
42
|
|
|
43
|
-
evaluated_count = report.results.count { |r| r.step_status !=
|
|
43
|
+
evaluated_count = report.results.count { |r| r.step_status != CaseResult::SKIPPED_STATUS }
|
|
44
44
|
cases_count = [evaluated_count, 1].max
|
|
45
45
|
cost_per_call = report.total_cost.to_f / cases_count
|
|
46
46
|
|
|
@@ -116,7 +116,7 @@ module RubyLLM
|
|
|
116
116
|
current_report = @comparison.reports[current_label]
|
|
117
117
|
return {} unless current_report
|
|
118
118
|
|
|
119
|
-
current_evaluated = current_report.results.count { |r| r.step_status !=
|
|
119
|
+
current_evaluated = current_report.results.count { |r| r.step_status != CaseResult::SKIPPED_STATUS }
|
|
120
120
|
current_cases = [current_evaluated, 1].max
|
|
121
121
|
current_cost = current_report.total_cost.to_f / current_cases
|
|
122
122
|
diff = current_cost - best[:cost_per_call]
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
require "forwardable"
|
|
4
|
+
require_relative "evaluator/proc_evaluator"
|
|
4
5
|
|
|
5
6
|
module RubyLLM
|
|
6
7
|
module Contract
|
|
@@ -10,9 +11,11 @@ module RubyLLM
|
|
|
10
11
|
|
|
11
12
|
attr_reader :dataset_name, :results, :step_name
|
|
12
13
|
|
|
13
|
-
GENERIC_DETAILS = [
|
|
14
|
+
GENERIC_DETAILS = [Evaluator::ProcEvaluator::PASSED_DETAILS, Evaluator::ProcEvaluator::FAILED_DETAILS].freeze
|
|
14
15
|
HISTORY_DIR = ".eval_history"
|
|
16
|
+
HISTORY_EXT = "jsonl"
|
|
15
17
|
BASELINE_DIR = ".eval_baselines"
|
|
18
|
+
BASELINE_EXT = "json"
|
|
16
19
|
|
|
17
20
|
def_delegators :@stats, :score, :passed, :failed, :skipped, :failures, :pass_rate, :pass_rate_ratio,
|
|
18
21
|
:total_cost, :unknown_cost_results, :avg_latency_ms, :passed?,
|
|
@@ -23,7 +23,7 @@ module RubyLLM
|
|
|
23
23
|
end
|
|
24
24
|
|
|
25
25
|
def skipped
|
|
26
|
-
@results.count { |result| result.step_status ==
|
|
26
|
+
@results.count { |result| result.step_status == CaseResult::SKIPPED_STATUS }
|
|
27
27
|
end
|
|
28
28
|
|
|
29
29
|
def failures
|
|
@@ -63,7 +63,7 @@ module RubyLLM
|
|
|
63
63
|
end
|
|
64
64
|
|
|
65
65
|
def evaluated_results
|
|
66
|
-
@evaluated_results ||= @results.reject { |result| result.step_status ==
|
|
66
|
+
@evaluated_results ||= @results.reject { |result| result.step_status == CaseResult::SKIPPED_STATUS }
|
|
67
67
|
end
|
|
68
68
|
|
|
69
69
|
def evaluated_results_count
|
|
@@ -13,7 +13,7 @@ module RubyLLM
|
|
|
13
13
|
end
|
|
14
14
|
|
|
15
15
|
def save_history!(path: nil, model: nil, reasoning_effort: nil)
|
|
16
|
-
file = path || storage_path(Report::HISTORY_DIR,
|
|
16
|
+
file = path || storage_path(Report::HISTORY_DIR, Report::HISTORY_EXT, model: model, reasoning_effort: reasoning_effort)
|
|
17
17
|
entry = history_entry
|
|
18
18
|
entry[:model] = model if model
|
|
19
19
|
entry[:reasoning_effort] = reasoning_effort if reasoning_effort
|
|
@@ -22,19 +22,19 @@ module RubyLLM
|
|
|
22
22
|
end
|
|
23
23
|
|
|
24
24
|
def eval_history(path: nil, model: nil, reasoning_effort: nil)
|
|
25
|
-
EvalHistory.load(path || storage_path(Report::HISTORY_DIR,
|
|
26
|
-
|
|
25
|
+
EvalHistory.load(path || storage_path(Report::HISTORY_DIR, Report::HISTORY_EXT,
|
|
26
|
+
model: model, reasoning_effort: reasoning_effort))
|
|
27
27
|
end
|
|
28
28
|
|
|
29
29
|
def save_baseline!(path: nil, model: nil, reasoning_effort: nil)
|
|
30
|
-
file = path || storage_path(Report::BASELINE_DIR,
|
|
30
|
+
file = path || storage_path(Report::BASELINE_DIR, Report::BASELINE_EXT, model: model, reasoning_effort: reasoning_effort)
|
|
31
31
|
FileUtils.mkdir_p(File.dirname(file))
|
|
32
32
|
File.write(file, JSON.pretty_generate(serialize_for_baseline))
|
|
33
33
|
file
|
|
34
34
|
end
|
|
35
35
|
|
|
36
36
|
def compare_with_baseline(path: nil, model: nil, reasoning_effort: nil)
|
|
37
|
-
file = path || storage_path(Report::BASELINE_DIR,
|
|
37
|
+
file = path || storage_path(Report::BASELINE_DIR, Report::BASELINE_EXT, model: model, reasoning_effort: reasoning_effort)
|
|
38
38
|
raise ArgumentError, "No baseline found at #{file}" unless File.exist?(file)
|
|
39
39
|
|
|
40
40
|
baseline_data = JSON.parse(File.read(file), symbolize_names: true)
|
|
@@ -47,7 +47,7 @@ module RubyLLM
|
|
|
47
47
|
end
|
|
48
48
|
|
|
49
49
|
def baseline_exists?(path: nil, model: nil, reasoning_effort: nil)
|
|
50
|
-
File.exist?(path || storage_path(Report::BASELINE_DIR,
|
|
50
|
+
File.exist?(path || storage_path(Report::BASELINE_DIR, Report::BASELINE_EXT, model: model, reasoning_effort: reasoning_effort))
|
|
51
51
|
end
|
|
52
52
|
|
|
53
53
|
private
|
|
@@ -8,14 +8,6 @@ module RubyLLM
|
|
|
8
8
|
# gate that only compares totals passes whatever those cases really cost.
|
|
9
9
|
# Shared by the rake task, the pass_eval matcher and assert_eval_passes.
|
|
10
10
|
module UnknownCostGate
|
|
11
|
-
MODES = %i[refuse warn].freeze
|
|
12
|
-
|
|
13
|
-
def self.validate!(mode)
|
|
14
|
-
return mode if MODES.include?(mode)
|
|
15
|
-
|
|
16
|
-
raise ArgumentError, "on_unknown_pricing must be :refuse or :warn, got #{mode.inspect}"
|
|
17
|
-
end
|
|
18
|
-
|
|
19
11
|
# Returns the failure message under :refuse, nil when the gate holds.
|
|
20
12
|
def self.check(reports, mode:)
|
|
21
13
|
names = reports.flat_map(&:unknown_cost_results).map(&:name).uniq
|
|
@@ -31,8 +31,8 @@ module RubyLLM
|
|
|
31
31
|
end
|
|
32
32
|
|
|
33
33
|
def assert_eval_passes(step, eval_name, minimum_score: nil, maximum_cost: nil, context: {}, msg: nil,
|
|
34
|
-
on_unknown_pricing:
|
|
35
|
-
|
|
34
|
+
on_unknown_pricing: UnknownPolicy::DEFAULT)
|
|
35
|
+
UnknownPolicy.validate!("on_unknown_pricing", on_unknown_pricing)
|
|
36
36
|
report = step.run_eval(eval_name, context: context)
|
|
37
37
|
|
|
38
38
|
if minimum_score
|
|
@@ -6,7 +6,7 @@ module RubyLLM
|
|
|
6
6
|
# Ignore eval/ subdirs BEFORE Zeitwerk setup — eval files don't define
|
|
7
7
|
# constants, they call define_eval on existing Step classes.
|
|
8
8
|
initializer "ruby_llm_contract.ignore_eval_dirs", before: :set_autoload_paths do |app|
|
|
9
|
-
|
|
9
|
+
RubyLLM::Contract::RAILS_EVAL_DIRS.each do |path|
|
|
10
10
|
full = app.root.join(path)
|
|
11
11
|
next unless full.exist?
|
|
12
12
|
|
|
@@ -18,7 +18,7 @@ module RubyLLM
|
|
|
18
18
|
end
|
|
19
19
|
|
|
20
20
|
def self.evaluate(host_reports:, minimum_score:, maximum_cost:, fail_on_regression:,
|
|
21
|
-
on_unknown_pricing:
|
|
21
|
+
on_unknown_pricing: UnknownPolicy::DEFAULT)
|
|
22
22
|
new(host_reports: host_reports,
|
|
23
23
|
minimum_score: minimum_score,
|
|
24
24
|
maximum_cost: maximum_cost,
|
|
@@ -29,12 +29,12 @@ module RubyLLM
|
|
|
29
29
|
attr_reader :verdict
|
|
30
30
|
|
|
31
31
|
def initialize(host_reports:, minimum_score:, maximum_cost:, fail_on_regression:,
|
|
32
|
-
on_unknown_pricing:
|
|
32
|
+
on_unknown_pricing: UnknownPolicy::DEFAULT)
|
|
33
33
|
@host_reports = host_reports
|
|
34
34
|
@minimum_score = minimum_score
|
|
35
35
|
@maximum_cost = maximum_cost
|
|
36
36
|
@fail_on_regression = fail_on_regression
|
|
37
|
-
@on_unknown_pricing =
|
|
37
|
+
@on_unknown_pricing = UnknownPolicy.validate!("on_unknown_pricing", on_unknown_pricing)
|
|
38
38
|
@verdict = build_verdict
|
|
39
39
|
end
|
|
40
40
|
|
|
@@ -56,7 +56,7 @@ module RubyLLM
|
|
|
56
56
|
passed_reports, all_passed = score_each_report
|
|
57
57
|
Verdict.new(
|
|
58
58
|
passed: all_passed,
|
|
59
|
-
abort_reason: all_passed ? nil : "
|
|
59
|
+
abort_reason: all_passed ? nil : "one or more evals did not pass",
|
|
60
60
|
passed_reports: passed_reports,
|
|
61
61
|
suite_cost: suite_cost
|
|
62
62
|
)
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
require "rake"
|
|
4
4
|
require "rake/tasklib"
|
|
5
|
-
require_relative "
|
|
5
|
+
require_relative "unknown_policy"
|
|
6
6
|
require_relative "rake_task/suite_gate"
|
|
7
7
|
|
|
8
8
|
module RubyLLM
|
|
@@ -21,12 +21,12 @@ module RubyLLM
|
|
|
21
21
|
@maximum_cost = nil # nil = no cost limit; float = budget cap (suite-level)
|
|
22
22
|
@eval_dirs = [] # directories to load eval files from (non-Rails)
|
|
23
23
|
# With maximum_cost: :refuse fails on cases the budget cannot price; :warn gates priced cases only
|
|
24
|
-
@on_unknown_pricing =
|
|
24
|
+
@on_unknown_pricing = UnknownPolicy::DEFAULT
|
|
25
25
|
@save_baseline = false
|
|
26
26
|
@fail_on_regression = false
|
|
27
27
|
@track_history = false
|
|
28
28
|
block&.call(self)
|
|
29
|
-
|
|
29
|
+
UnknownPolicy.validate!("on_unknown_pricing", @on_unknown_pricing)
|
|
30
30
|
define_task
|
|
31
31
|
end
|
|
32
32
|
|
|
@@ -134,7 +134,7 @@ module RubyLLM
|
|
|
134
134
|
abort("STEP is required, e.g. STEP=MatchProblemsToPages") if step_name.empty?
|
|
135
135
|
raw_candidates = ENV["CANDIDATES"].to_s.strip
|
|
136
136
|
abort("CANDIDATES is required, e.g. CANDIDATES=gpt-5-nano,gpt-5-mini@low,gpt-5-mini") if raw_candidates.empty?
|
|
137
|
-
min_score = ENV.fetch("MIN_SCORE"
|
|
137
|
+
min_score = ENV.fetch("MIN_SCORE") { Eval::DEFAULT_MIN_SCORE }.to_f
|
|
138
138
|
runs = parse_runs(ENV.fetch("RUNS", "1"))
|
|
139
139
|
|
|
140
140
|
host = RubyLLM::Contract.eval_hosts.find { |h| h.name == step_name }
|
|
@@ -65,7 +65,7 @@ RSpec::Matchers.define :pass_eval do |eval_name|
|
|
|
65
65
|
end
|
|
66
66
|
|
|
67
67
|
chain :on_unknown_pricing do |mode|
|
|
68
|
-
@on_unknown_pricing = RubyLLM::Contract::
|
|
68
|
+
@on_unknown_pricing = RubyLLM::Contract::UnknownPolicy.validate!("on_unknown_pricing", mode)
|
|
69
69
|
end
|
|
70
70
|
|
|
71
71
|
chain :without_regressions do
|
|
@@ -82,7 +82,7 @@ RSpec::Matchers.define :pass_eval do |eval_name|
|
|
|
82
82
|
@context ||= {}
|
|
83
83
|
@minimum_score ||= nil
|
|
84
84
|
@maximum_cost ||= nil
|
|
85
|
-
@on_unknown_pricing ||=
|
|
85
|
+
@on_unknown_pricing ||= RubyLLM::Contract::UnknownPolicy::DEFAULT
|
|
86
86
|
@unknown_cost_failure = nil
|
|
87
87
|
@unknown_cost_only = false
|
|
88
88
|
@check_regressions ||= false
|
|
@@ -90,8 +90,10 @@ module RubyLLM
|
|
|
90
90
|
).call
|
|
91
91
|
end
|
|
92
92
|
|
|
93
|
-
|
|
94
|
-
|
|
93
|
+
# Forwarded to the adapter as-is. A key known but not forwarded would be
|
|
94
|
+
# silently dropped, so the known list is built from this one.
|
|
95
|
+
ADAPTER_CONTEXT_KEYS = %i[provider assume_model_exists max_tokens reasoning_effort attachment].freeze
|
|
96
|
+
KNOWN_CONTEXT_KEYS = (%i[adapter model temperature retry_policy_override] + ADAPTER_CONTEXT_KEYS).freeze
|
|
95
97
|
|
|
96
98
|
include Concerns::ContextHelpers
|
|
97
99
|
|
|
@@ -133,7 +135,7 @@ module RubyLLM
|
|
|
133
135
|
estimate = attachment_token_estimate if respond_to?(:attachment_token_estimate)
|
|
134
136
|
return [estimate, false] unless estimate.nil?
|
|
135
137
|
|
|
136
|
-
mode = respond_to?(:on_unknown_attachment_size) ? on_unknown_attachment_size :
|
|
138
|
+
mode = respond_to?(:on_unknown_attachment_size) ? on_unknown_attachment_size : UnknownPolicy::DEFAULT
|
|
137
139
|
if mode == :warn
|
|
138
140
|
warn "[ruby_llm-contract] attachment present but attachment_token_estimate not " \
|
|
139
141
|
"declared on #{name || self} — estimate_cost proceeds without attachment cost"
|
|
@@ -193,7 +195,7 @@ module RubyLLM
|
|
|
193
195
|
|
|
194
196
|
def runtime_settings(context)
|
|
195
197
|
policy = context.key?(:retry_policy_override) ? context[:retry_policy_override] : retry_policy
|
|
196
|
-
extra = context.slice(
|
|
198
|
+
extra = context.slice(*ADAPTER_CONTEXT_KEYS)
|
|
197
199
|
|
|
198
200
|
# Always pass the class-level `thinking` config to the adapter when
|
|
199
201
|
# set, so fields like `budget` survive a per-call `reasoning_effort`
|
|
@@ -151,12 +151,10 @@ module RubyLLM
|
|
|
151
151
|
if amount
|
|
152
152
|
validate_positive!("max_cost", amount)
|
|
153
153
|
|
|
154
|
-
|
|
155
|
-
raise ArgumentError, "on_unknown_pricing must be :refuse or :warn, got #{on_unknown_pricing.inspect}"
|
|
156
|
-
end
|
|
154
|
+
UnknownPolicy.validate!("on_unknown_pricing", on_unknown_pricing) if on_unknown_pricing
|
|
157
155
|
|
|
158
156
|
@max_cost = amount
|
|
159
|
-
@on_unknown_pricing = on_unknown_pricing ||
|
|
157
|
+
@on_unknown_pricing = on_unknown_pricing || UnknownPolicy::DEFAULT
|
|
160
158
|
return @max_cost
|
|
161
159
|
end
|
|
162
160
|
|
|
@@ -164,7 +162,7 @@ module RubyLLM
|
|
|
164
162
|
end
|
|
165
163
|
|
|
166
164
|
def on_unknown_pricing
|
|
167
|
-
inherited_value(:on_unknown_pricing) ||
|
|
165
|
+
inherited_value(:on_unknown_pricing) || UnknownPolicy::DEFAULT
|
|
168
166
|
end
|
|
169
167
|
|
|
170
168
|
def attachment_token_estimate(n = nil)
|
|
@@ -182,16 +180,9 @@ module RubyLLM
|
|
|
182
180
|
end
|
|
183
181
|
|
|
184
182
|
def on_unknown_attachment_size(mode = nil)
|
|
185
|
-
if mode
|
|
186
|
-
unless %i[refuse warn].include?(mode)
|
|
187
|
-
raise ArgumentError,
|
|
188
|
-
"on_unknown_attachment_size must be :refuse or :warn, got #{mode.inspect}"
|
|
189
|
-
end
|
|
190
|
-
|
|
191
|
-
return @on_unknown_attachment_size = mode
|
|
192
|
-
end
|
|
183
|
+
return @on_unknown_attachment_size = UnknownPolicy.validate!("on_unknown_attachment_size", mode) if mode
|
|
193
184
|
|
|
194
|
-
inherited_value(:on_unknown_attachment_size) ||
|
|
185
|
+
inherited_value(:on_unknown_attachment_size) || UnknownPolicy::DEFAULT
|
|
195
186
|
end
|
|
196
187
|
|
|
197
188
|
def model(name = nil)
|
|
@@ -28,8 +28,8 @@ module RubyLLM
|
|
|
28
28
|
def self.build(input_type:, output_type:, prompt_block:, contract_definition:,
|
|
29
29
|
adapter:, model:,
|
|
30
30
|
output_schema: nil, max_output: nil,
|
|
31
|
-
max_input: nil, max_cost: nil, on_unknown_pricing:
|
|
32
|
-
attachment_token_estimate: nil, on_unknown_attachment_size:
|
|
31
|
+
max_input: nil, max_cost: nil, on_unknown_pricing: UnknownPolicy::DEFAULT,
|
|
32
|
+
attachment_token_estimate: nil, on_unknown_attachment_size: UnknownPolicy::DEFAULT,
|
|
33
33
|
temperature: nil, extra_options: {}, observers: [])
|
|
34
34
|
new(
|
|
35
35
|
input_type: input_type, output_type: output_type,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
module Contract
|
|
5
|
+
# What a limit does when the number it needs is unknown: model pricing for
|
|
6
|
+
# max_cost/maximum_cost, attachment size for max_input. One vocabulary for
|
|
7
|
+
# every such option, step-level and eval-level, so a new mode or a changed
|
|
8
|
+
# default reaches all of them. Standalone: the rake task loads it before the
|
|
9
|
+
# rest of the gem to validate its settings at definition time.
|
|
10
|
+
module UnknownPolicy
|
|
11
|
+
MODES = %i[refuse warn].freeze
|
|
12
|
+
DEFAULT = :refuse
|
|
13
|
+
|
|
14
|
+
def self.validate!(option, mode)
|
|
15
|
+
return mode if MODES.include?(mode)
|
|
16
|
+
|
|
17
|
+
raise ArgumentError, "#{option} must be #{MODES.map(&:inspect).join(" or ")}, got #{mode.inspect}"
|
|
18
|
+
end
|
|
19
|
+
end
|
|
20
|
+
end
|
|
21
|
+
end
|
data/lib/ruby_llm/contract.rb
CHANGED
|
@@ -2,10 +2,19 @@
|
|
|
2
2
|
|
|
3
3
|
require_relative "contract/version"
|
|
4
4
|
require_relative "contract/errors"
|
|
5
|
+
require_relative "contract/unknown_policy"
|
|
5
6
|
require_relative "contract/types"
|
|
6
7
|
|
|
7
8
|
module RubyLLM
|
|
8
9
|
module Contract
|
|
10
|
+
# Rails dirs holding Step classes. Each one's eval/ subdir holds define_eval
|
|
11
|
+
# files, which define no constant, so Zeitwerk must ignore exactly those.
|
|
12
|
+
RAILS_CONTRACT_DIRS = %w[app/contracts app/steps].freeze
|
|
13
|
+
RAILS_EVAL_DIRS = RAILS_CONTRACT_DIRS.map { |dir| "#{dir}/eval" }.freeze
|
|
14
|
+
|
|
15
|
+
# Set while load_evals! runs: redefining an eval is then a reload, not a mistake.
|
|
16
|
+
RELOADING_KEY = :ruby_llm_contract_reloading
|
|
17
|
+
|
|
9
18
|
class << self
|
|
10
19
|
def configuration
|
|
11
20
|
@configuration ||= Configuration.new
|
|
@@ -51,7 +60,7 @@ module RubyLLM
|
|
|
51
60
|
def load_evals!(*dirs)
|
|
52
61
|
dirs = dirs.flatten.compact
|
|
53
62
|
if dirs.empty? && defined?(::Rails)
|
|
54
|
-
dirs =
|
|
63
|
+
dirs = RAILS_EVAL_DIRS.filter_map do |path|
|
|
55
64
|
full = ::Rails.root.join(path)
|
|
56
65
|
full.to_s if full.exist?
|
|
57
66
|
end
|
|
@@ -64,7 +73,7 @@ module RubyLLM
|
|
|
64
73
|
eager_load_contract_dirs! if defined?(::Rails)
|
|
65
74
|
|
|
66
75
|
# Clear file-sourced evals ONCE, then load ALL dirs.
|
|
67
|
-
Thread.current[
|
|
76
|
+
Thread.current[RELOADING_KEY] = true
|
|
68
77
|
eval_hosts.each do |host|
|
|
69
78
|
host.clear_file_sourced_evals! if host.respond_to?(:clear_file_sourced_evals!)
|
|
70
79
|
end
|
|
@@ -73,7 +82,11 @@ module RubyLLM
|
|
|
73
82
|
Dir[File.join(d, "**", "*_eval.rb")].each { |f| load f }
|
|
74
83
|
end
|
|
75
84
|
ensure
|
|
76
|
-
Thread.current[
|
|
85
|
+
Thread.current[RELOADING_KEY] = false
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
def reloading?
|
|
89
|
+
Thread.current[RELOADING_KEY] ? true : false
|
|
77
90
|
end
|
|
78
91
|
|
|
79
92
|
def normalize_candidate_config(entry)
|
|
@@ -111,7 +124,7 @@ module RubyLLM
|
|
|
111
124
|
end
|
|
112
125
|
|
|
113
126
|
def eager_load_contract_dirs!
|
|
114
|
-
|
|
127
|
+
RAILS_CONTRACT_DIRS.each do |path|
|
|
115
128
|
full = ::Rails.root.join(path)
|
|
116
129
|
next unless full.exist?
|
|
117
130
|
|
metadata
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: ruby_llm-contract
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 1.1.
|
|
4
|
+
version: 1.1.1
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Justyna
|
|
@@ -193,6 +193,7 @@ files:
|
|
|
193
193
|
- lib/ruby_llm/contract/step/trace.rb
|
|
194
194
|
- lib/ruby_llm/contract/token_estimator.rb
|
|
195
195
|
- lib/ruby_llm/contract/types.rb
|
|
196
|
+
- lib/ruby_llm/contract/unknown_policy.rb
|
|
196
197
|
- lib/ruby_llm/contract/version.rb
|
|
197
198
|
- ruby_llm-contract.gemspec
|
|
198
199
|
homepage: https://github.com/justi/ruby_llm-contract
|