ruby_llm-contract 1.1.1 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +103 -0
- data/README.md +18 -3
- data/docs/guide/getting_started.md +86 -1
- data/docs/guide/llm_judge.md +14 -0
- data/docs/guide/rails_integration.md +8 -0
- data/docs/guide/relation_to_agent.md +8 -2
- data/lib/ruby_llm/contract/adapters/response.rb +28 -2
- data/lib/ruby_llm/contract/adapters/ruby_llm.rb +136 -28
- data/lib/ruby_llm/contract/adapters/test.rb +6 -0
- data/lib/ruby_llm/contract/concerns/usage_aggregator.rb +8 -5
- data/lib/ruby_llm/contract/configuration.rb +5 -1
- data/lib/ruby_llm/contract/cost_calculator.rb +98 -44
- data/lib/ruby_llm/contract/eval/eval_history.rb +2 -1
- data/lib/ruby_llm/contract/eval/model_comparison.rb +11 -1
- data/lib/ruby_llm/contract/eval/recommender.rb +10 -3
- data/lib/ruby_llm/contract/eval/report_stats.rb +8 -3
- data/lib/ruby_llm/contract/eval/report_storage.rb +8 -2
- data/lib/ruby_llm/contract/pipeline/base.rb +3 -1
- data/lib/ruby_llm/contract/pipeline/runner.rb +11 -1
- data/lib/ruby_llm/contract/pipeline/trace.rb +10 -0
- data/lib/ruby_llm/contract/provider_options.rb +45 -0
- data/lib/ruby_llm/contract/step/adapter_caller.rb +2 -1
- data/lib/ruby_llm/contract/step/base.rb +29 -12
- data/lib/ruby_llm/contract/step/dsl.rb +54 -0
- data/lib/ruby_llm/contract/step/limit_checker.rb +48 -13
- data/lib/ruby_llm/contract/step/result_builder.rb +66 -4
- data/lib/ruby_llm/contract/step/retry_executor.rb +35 -4
- data/lib/ruby_llm/contract/step/retry_policy.rb +31 -5
- data/lib/ruby_llm/contract/step/runner.rb +20 -1
- data/lib/ruby_llm/contract/step/runner_config.rb +6 -3
- data/lib/ruby_llm/contract/step/trace.rb +61 -20
- data/lib/ruby_llm/contract/token_estimator.rb +5 -3
- data/lib/ruby_llm/contract/version.rb +1 -1
- data/lib/ruby_llm/contract/workflow_scope.rb +48 -0
- data/lib/ruby_llm/contract.rb +2 -0
- metadata +3 -1
|
@@ -4,6 +4,10 @@ module RubyLLM
|
|
|
4
4
|
module Contract
|
|
5
5
|
module Concerns
|
|
6
6
|
module UsageAggregator
|
|
7
|
+
# Breakdown keys an adapter may add next to input/output. They are parts
|
|
8
|
+
# of input_tokens/output_tokens, not extra tokens, so sum_tokens skips them.
|
|
9
|
+
USAGE_DETAIL_KEYS = %i[cache_read_tokens cache_write_tokens thinking_tokens].freeze
|
|
10
|
+
|
|
7
11
|
private
|
|
8
12
|
|
|
9
13
|
def extract_usage(trace_entry)
|
|
@@ -24,18 +28,17 @@ module RubyLLM
|
|
|
24
28
|
end
|
|
25
29
|
|
|
26
30
|
def aggregate_usage(traces)
|
|
27
|
-
|
|
28
|
-
output_total = 0
|
|
31
|
+
totals = Hash.new(0)
|
|
29
32
|
|
|
30
33
|
traces.each do |trace_entry|
|
|
31
34
|
usage = extract_usage(trace_entry)
|
|
32
35
|
next unless usage.is_a?(Hash)
|
|
33
36
|
|
|
34
|
-
|
|
35
|
-
output_total += usage[:output_tokens] || 0
|
|
37
|
+
[:input_tokens, :output_tokens, *USAGE_DETAIL_KEYS].each { |key| totals[key] += usage[key] || 0 }
|
|
36
38
|
end
|
|
37
39
|
|
|
38
|
-
{ input_tokens:
|
|
40
|
+
{ input_tokens: totals[:input_tokens], output_tokens: totals[:output_tokens] }
|
|
41
|
+
.merge(totals.slice(*USAGE_DETAIL_KEYS).select { |_, count| count.positive? })
|
|
39
42
|
end
|
|
40
43
|
end
|
|
41
44
|
end
|
|
@@ -9,13 +9,17 @@ module RubyLLM
|
|
|
9
9
|
#
|
|
10
10
|
# Then configure contract-specific options:
|
|
11
11
|
# RubyLLM::Contract.configure { |c| c.default_model = "gpt-4.1-mini" }
|
|
12
|
+
#
|
|
13
|
+
# `workflow_instrumentation = true` wraps steps and pipelines in
|
|
14
|
+
# `RubyLLM.workflow` (see WorkflowScope).
|
|
12
15
|
class Configuration
|
|
13
|
-
attr_accessor :default_adapter, :default_model, :logger
|
|
16
|
+
attr_accessor :default_adapter, :default_model, :logger, :workflow_instrumentation
|
|
14
17
|
|
|
15
18
|
def initialize
|
|
16
19
|
@default_adapter = nil
|
|
17
20
|
@default_model = nil
|
|
18
21
|
@logger = nil
|
|
22
|
+
@workflow_instrumentation = false
|
|
19
23
|
end
|
|
20
24
|
end
|
|
21
25
|
end
|
|
@@ -7,27 +7,31 @@ module RubyLLM
|
|
|
7
7
|
# **What this module does (public surface):**
|
|
8
8
|
#
|
|
9
9
|
# 1. **Fine-tune / custom-model pricing registry** — `register_model`
|
|
10
|
-
# fills
|
|
11
|
-
# upstream
|
|
12
|
-
# (e.g. `ft:gpt-4o-custom`) need their pricing supplied
|
|
10
|
+
# fills a gap RubyLLM still has in 2.x: there is no simple
|
|
11
|
+
# upstream API to price a model its registry lacks, so fine-tuned
|
|
12
|
+
# models (e.g. `ft:gpt-4o-custom`) need their pricing supplied
|
|
13
|
+
# locally. A registered price also overrides RubyLLM's for that id.
|
|
13
14
|
# 2. **Lookup with fallback chain** — `calculate(model_name:, usage:)`
|
|
14
15
|
# checks the custom registry first, falls back to
|
|
15
16
|
# `RubyLLM.models.find(model_name)`, returns `nil` on miss.
|
|
16
17
|
#
|
|
17
18
|
# **What this module is NOT:**
|
|
18
19
|
#
|
|
19
|
-
# - Not a
|
|
20
|
-
#
|
|
21
|
-
#
|
|
22
|
-
#
|
|
23
|
-
#
|
|
20
|
+
# - Not a substitute for RubyLLM's pricing — for any model in
|
|
21
|
+
# `RubyLLM.models`, RubyLLM's own `Model#cost_for` does the math.
|
|
22
|
+
# - Not the source of a finished call's cost: the RubyLLM adapter takes
|
|
23
|
+
# that from the response (`response.cost`), which also covers what
|
|
24
|
+
# the provider reported and RubyLLM's own retries. This module prices
|
|
25
|
+
# pre-flight estimates and calls from adapters that report no cost.
|
|
24
26
|
#
|
|
25
27
|
# The reason this module exists at all is the registry + retry usage
|
|
26
28
|
# aggregation across attempts (the latter sits in `Step::RetryExecutor`,
|
|
27
29
|
# which calls `calculate` per attempt and sums; not in this module).
|
|
28
30
|
module CostCalculator
|
|
29
31
|
# Simple struct for custom-registered model pricing
|
|
30
|
-
RegisteredModel = Struct.new(:input_price_per_million, :output_price_per_million,
|
|
32
|
+
RegisteredModel = Struct.new(:input_price_per_million, :output_price_per_million,
|
|
33
|
+
:cache_read_price_per_million, :cache_write_price_per_million,
|
|
34
|
+
keyword_init: true)
|
|
31
35
|
|
|
32
36
|
@custom_models = {}
|
|
33
37
|
|
|
@@ -38,13 +42,22 @@ module RubyLLM
|
|
|
38
42
|
# CostCalculator.register_model("ft:gpt-4o-custom",
|
|
39
43
|
# input_per_1m: 3.0, output_per_1m: 6.0)
|
|
40
44
|
#
|
|
41
|
-
|
|
45
|
+
# Prompt-cache prices are optional: without `cache_read_per_1m` cache
|
|
46
|
+
# reads are charged at the input price; without `cache_write_per_1m` a
|
|
47
|
+
# call that wrote to the cache has no price (writes can cost more than
|
|
48
|
+
# input).
|
|
49
|
+
def self.register_model(model_name, input_per_1m:, output_per_1m:, cache_read_per_1m: nil,
|
|
50
|
+
cache_write_per_1m: nil)
|
|
42
51
|
validate_price!(:input_per_1m, input_per_1m)
|
|
43
52
|
validate_price!(:output_per_1m, output_per_1m)
|
|
53
|
+
validate_price!(:cache_read_per_1m, cache_read_per_1m) unless cache_read_per_1m.nil?
|
|
54
|
+
validate_price!(:cache_write_per_1m, cache_write_per_1m) unless cache_write_per_1m.nil?
|
|
44
55
|
|
|
45
56
|
@custom_models[model_name] = RegisteredModel.new(
|
|
46
57
|
input_price_per_million: input_per_1m,
|
|
47
|
-
output_price_per_million: output_per_1m
|
|
58
|
+
output_price_per_million: output_per_1m,
|
|
59
|
+
cache_read_price_per_million: cache_read_per_1m,
|
|
60
|
+
cache_write_price_per_million: cache_write_per_1m
|
|
48
61
|
)
|
|
49
62
|
end
|
|
50
63
|
|
|
@@ -60,8 +73,9 @@ module RubyLLM
|
|
|
60
73
|
|
|
61
74
|
# Look up cost for a single model + usage hash.
|
|
62
75
|
# Returns nil if model is unknown (custom registry miss + RubyLLM miss),
|
|
63
|
-
#
|
|
64
|
-
#
|
|
76
|
+
# or if any used token category has no price, so callers can decide
|
|
77
|
+
# whether to refuse the call or proceed (see `on_unknown_pricing:` step
|
|
78
|
+
# option for the budget-gating policy).
|
|
65
79
|
#
|
|
66
80
|
# CostCalculator.calculate(
|
|
67
81
|
# model_name: "gpt-4o-mini",
|
|
@@ -69,13 +83,17 @@ module RubyLLM
|
|
|
69
83
|
# )
|
|
70
84
|
# # => 0.00069 (or nil if model not registered)
|
|
71
85
|
#
|
|
72
|
-
#
|
|
73
|
-
#
|
|
74
|
-
#
|
|
75
|
-
|
|
86
|
+
# `usage[:input_tokens]` is the whole prompt; `:cache_read_tokens` and
|
|
87
|
+
# `:cache_write_tokens`, when present, are the parts of it served from or
|
|
88
|
+
# written to the provider's prompt cache. `provider:` picks the provider's
|
|
89
|
+
# price for a model id several providers list at different prices.
|
|
90
|
+
#
|
|
91
|
+
# Aggregating across retry attempts is done in `Step::RetryExecutor`,
|
|
92
|
+
# not here.
|
|
93
|
+
def self.calculate(model_name:, usage:, provider: nil)
|
|
76
94
|
return nil unless model_name && usage.is_a?(Hash)
|
|
77
95
|
|
|
78
|
-
model_info = find_model(model_name)
|
|
96
|
+
model_info = find_model(model_name, provider: provider)
|
|
79
97
|
return nil unless model_info
|
|
80
98
|
|
|
81
99
|
compute_cost(model_info, usage)
|
|
@@ -83,55 +101,90 @@ module RubyLLM
|
|
|
83
101
|
nil
|
|
84
102
|
end
|
|
85
103
|
|
|
86
|
-
def self.
|
|
87
|
-
|
|
88
|
-
return nil unless prices
|
|
89
|
-
|
|
90
|
-
input_cost = token_cost(usage[:input_tokens], prices[:input])
|
|
91
|
-
output_cost = token_cost(usage[:output_tokens], prices[:output])
|
|
92
|
-
(input_cost + output_cost).round(6)
|
|
104
|
+
def self.registered?(model_name)
|
|
105
|
+
@custom_models.key?(model_name)
|
|
93
106
|
end
|
|
94
107
|
|
|
95
108
|
# Two shapes reach here. `RegisteredModel` (our own struct, from
|
|
96
|
-
# `register_model`) exposes flat `*_price_per_million` readers
|
|
97
|
-
#
|
|
98
|
-
#
|
|
109
|
+
# `register_model`) exposes flat `*_price_per_million` readers and is
|
|
110
|
+
# priced here. A RubyLLM model is priced by RubyLLM itself
|
|
111
|
+
# (`Model#cost_for`), which knows cache, long-context and per-provider
|
|
112
|
+
# prices and returns nil when a used category has no price.
|
|
99
113
|
#
|
|
100
|
-
# nil means "unknown pricing".
|
|
101
|
-
#
|
|
102
|
-
#
|
|
103
|
-
#
|
|
104
|
-
#
|
|
105
|
-
def self.
|
|
114
|
+
# nil means "unknown pricing". `calculate` rescues StandardError, so a
|
|
115
|
+
# future shape neither branch can read also degrades to nil: `max_cost`
|
|
116
|
+
# then refuses the call citing "no pricing data" rather than the real
|
|
117
|
+
# cause, and paths that floor unknown cost to 0.0 (eval cost estimates,
|
|
118
|
+
# report totals) undercount.
|
|
119
|
+
def self.compute_cost(model_info, usage)
|
|
106
120
|
if model_info.respond_to?(:input_price_per_million)
|
|
107
|
-
|
|
108
|
-
|
|
121
|
+
registered_cost(model_info, usage)
|
|
122
|
+
elsif model_info.respond_to?(:cost_for)
|
|
123
|
+
model_info.cost_for(tokens_for(usage)).total&.round(6)
|
|
109
124
|
end
|
|
125
|
+
end
|
|
110
126
|
|
|
111
|
-
|
|
112
|
-
|
|
127
|
+
# input_tokens includes the cache reads and writes, so each part is
|
|
128
|
+
# charged once, at its own price. Without a cache-read price, reads are
|
|
129
|
+
# charged at the input price, which assumes they cost no more than
|
|
130
|
+
# uncached input (true of every provider in RubyLLM's registry, not
|
|
131
|
+
# guaranteed for a custom one). Cache writes can cost more than input
|
|
132
|
+
# (Anthropic charges 1.25x-2x), so without a cache-write price a call
|
|
133
|
+
# that wrote to the cache has no price rather than an undercount.
|
|
134
|
+
def self.registered_cost(model_info, usage)
|
|
135
|
+
cache_read = usage[:cache_read_tokens] || 0
|
|
136
|
+
cache_write = usage[:cache_write_tokens] || 0
|
|
137
|
+
read_price, write_price = registered_cache_prices(model_info)
|
|
138
|
+
return nil if cache_write.positive? && write_price.nil?
|
|
139
|
+
|
|
140
|
+
uncached = [(usage[:input_tokens] || 0) - cache_read - cache_write, 0].max
|
|
141
|
+
(token_cost(uncached, model_info.input_price_per_million) +
|
|
142
|
+
token_cost(cache_read, read_price) +
|
|
143
|
+
token_cost(cache_write, write_price.to_f) +
|
|
144
|
+
token_cost(usage[:output_tokens], model_info.output_price_per_million)).round(6)
|
|
145
|
+
end
|
|
146
|
+
|
|
147
|
+
# [read price, write price]; a model without a read price reads at its
|
|
148
|
+
# input price. Duck-typed registry entries may lack both readers.
|
|
149
|
+
def self.registered_cache_prices(model_info)
|
|
150
|
+
read = model_info.cache_read_price_per_million if model_info.respond_to?(:cache_read_price_per_million)
|
|
151
|
+
write = model_info.cache_write_price_per_million if model_info.respond_to?(:cache_write_price_per_million)
|
|
152
|
+
[read || model_info.input_price_per_million, write]
|
|
153
|
+
end
|
|
113
154
|
|
|
114
|
-
|
|
155
|
+
# RubyLLM counts uncached input apart from cache reads and writes; our
|
|
156
|
+
# usage counts them all in input_tokens.
|
|
157
|
+
def self.tokens_for(usage)
|
|
158
|
+
cache_read = usage[:cache_read_tokens] || 0
|
|
159
|
+
cache_write = usage[:cache_write_tokens] || 0
|
|
160
|
+
::RubyLLM::Tokens.new(
|
|
161
|
+
input: [(usage[:input_tokens] || 0) - cache_read - cache_write, 0].max,
|
|
162
|
+
output: usage[:output_tokens] || 0,
|
|
163
|
+
cache_read: cache_read.positive? ? cache_read : nil,
|
|
164
|
+
cache_write: cache_write.positive? ? cache_write : nil,
|
|
165
|
+
# Priced by RubyLLM only for models with a separate reasoning price.
|
|
166
|
+
thinking: usage[:thinking_tokens]
|
|
167
|
+
)
|
|
115
168
|
end
|
|
116
|
-
private_class_method :prices_for
|
|
117
169
|
|
|
118
170
|
# Provider pricing is denominated per 1M tokens; divide here to get
|
|
119
171
|
# the dollar cost for the actual usage count. Named constant for
|
|
120
172
|
# consistency with how RubyLLM and provider docs express prices.
|
|
121
173
|
TOKENS_PER_MILLION = 1_000_000.0
|
|
122
174
|
|
|
175
|
+
# `register_model` validates both prices, so neither is nil here.
|
|
123
176
|
def self.token_cost(tokens, price_per_million)
|
|
124
|
-
(tokens || 0) *
|
|
177
|
+
(tokens || 0) * price_per_million / TOKENS_PER_MILLION
|
|
125
178
|
end
|
|
126
179
|
|
|
127
|
-
def self.find_model(model_name)
|
|
180
|
+
def self.find_model(model_name, provider: nil)
|
|
128
181
|
# Check custom registry first
|
|
129
182
|
custom = @custom_models[model_name]
|
|
130
183
|
return custom if custom
|
|
131
184
|
|
|
132
185
|
return nil unless defined?(RubyLLM)
|
|
133
186
|
|
|
134
|
-
RubyLLM.models.find(model_name)
|
|
187
|
+
provider ? RubyLLM.models.find(model_name, provider: provider) : RubyLLM.models.find(model_name)
|
|
135
188
|
rescue StandardError
|
|
136
189
|
nil
|
|
137
190
|
end
|
|
@@ -146,7 +199,8 @@ module RubyLLM
|
|
|
146
199
|
# to inspect model pricing before invoking `calculate` (e.g., to short-
|
|
147
200
|
# circuit estimate when the model is unknown). Exposing it removes
|
|
148
201
|
# `CostCalculator.send(:find_model)` workarounds at call sites.
|
|
149
|
-
private_class_method :compute_cost, :
|
|
202
|
+
private_class_method :compute_cost, :registered_cost, :registered_cache_prices, :tokens_for, :token_cost,
|
|
203
|
+
:validate_price!
|
|
150
204
|
end
|
|
151
205
|
end
|
|
152
206
|
end
|
|
@@ -69,7 +69,8 @@ module RubyLLM
|
|
|
69
69
|
|
|
70
70
|
lines = ["#{runs.length} runs"]
|
|
71
71
|
runs.last(5).each do |r|
|
|
72
|
-
|
|
72
|
+
cost = format("%.6f", r[:total_cost] || r[:cost] || 0)
|
|
73
|
+
lines << " #{r[:date]} score=#{r[:score].round(2)} cost=$#{cost}#{" + unknown" if r[:cost_unknown]}"
|
|
73
74
|
end
|
|
74
75
|
lines.join("\n")
|
|
75
76
|
end
|
|
@@ -36,8 +36,12 @@ module RubyLLM
|
|
|
36
36
|
@reports[resolve_key(candidate)]&.total_cost
|
|
37
37
|
end
|
|
38
38
|
|
|
39
|
+
# A report with unpriced cases has only a subtotal, so it cannot be
|
|
40
|
+
# called the cheapest.
|
|
39
41
|
def best_for(min_score: 0.0)
|
|
40
|
-
eligible = @reports.select
|
|
42
|
+
eligible = @reports.select do |_, report|
|
|
43
|
+
report.score > 0.0 && report.score >= min_score && !cost_unknown?(report)
|
|
44
|
+
end
|
|
41
45
|
return nil if eligible.empty?
|
|
42
46
|
|
|
43
47
|
eligible.min_by { |_, report| report.total_cost }&.first
|
|
@@ -45,6 +49,8 @@ module RubyLLM
|
|
|
45
49
|
|
|
46
50
|
def cost_per_point
|
|
47
51
|
@reports.transform_values do |report|
|
|
52
|
+
next Float::INFINITY if cost_unknown?(report)
|
|
53
|
+
|
|
48
54
|
report.score.positive? ? report.total_cost / report.score : Float::INFINITY
|
|
49
55
|
end
|
|
50
56
|
end
|
|
@@ -163,6 +169,10 @@ module RubyLLM
|
|
|
163
169
|
)
|
|
164
170
|
end
|
|
165
171
|
|
|
172
|
+
def cost_unknown?(report)
|
|
173
|
+
report.respond_to?(:unknown_cost_results) && report.unknown_cost_results.any?
|
|
174
|
+
end
|
|
175
|
+
|
|
166
176
|
def resolve_key(candidate)
|
|
167
177
|
case candidate
|
|
168
178
|
when String then candidate
|
|
@@ -51,7 +51,8 @@ module RubyLLM
|
|
|
51
51
|
cost_per_call: cost_per_call,
|
|
52
52
|
latency: report.avg_latency_ms || Float::INFINITY,
|
|
53
53
|
pass_rate_ratio: report.pass_rate_ratio,
|
|
54
|
-
total_cost: report.total_cost
|
|
54
|
+
total_cost: report.total_cost,
|
|
55
|
+
cost_unknown: unknown_cost?(report)
|
|
55
56
|
}
|
|
56
57
|
end
|
|
57
58
|
end
|
|
@@ -114,7 +115,7 @@ module RubyLLM
|
|
|
114
115
|
|
|
115
116
|
current_label = ModelComparison.candidate_label(@current_config)
|
|
116
117
|
current_report = @comparison.reports[current_label]
|
|
117
|
-
return {}
|
|
118
|
+
return {} if current_report.nil? || unknown_cost?(current_report)
|
|
118
119
|
|
|
119
120
|
current_evaluated = current_report.results.count { |r| r.step_status != CaseResult::SKIPPED_STATUS }
|
|
120
121
|
current_cases = [current_evaluated, 1].max
|
|
@@ -125,8 +126,14 @@ module RubyLLM
|
|
|
125
126
|
{ per_call: diff.round(6), monthly_at: { 10_000 => (diff * 10_000).round(2) } }
|
|
126
127
|
end
|
|
127
128
|
|
|
129
|
+
# A subtotal over unpriced cases is not a cost to rank by. Zero stays
|
|
130
|
+
# "unknown pricing" too: offline runs report $0 without any call.
|
|
128
131
|
def cost_known?(scored_candidate)
|
|
129
|
-
scored_candidate[:cost_per_call]&.positive?
|
|
132
|
+
!scored_candidate[:cost_unknown] && scored_candidate[:cost_per_call]&.positive?
|
|
133
|
+
end
|
|
134
|
+
|
|
135
|
+
def unknown_cost?(report)
|
|
136
|
+
report.respond_to?(:unknown_cost_results) && report.unknown_cost_results.any?
|
|
130
137
|
end
|
|
131
138
|
end
|
|
132
139
|
end
|
|
@@ -85,7 +85,7 @@ module RubyLLM
|
|
|
85
85
|
def single_shot_cost
|
|
86
86
|
return nil unless production_mode?
|
|
87
87
|
|
|
88
|
-
evaluated_results.sum { |r| first_attempt_cost(r) ||
|
|
88
|
+
evaluated_results.sum { |r| first_attempt_cost(r) || 0.0 }
|
|
89
89
|
end
|
|
90
90
|
|
|
91
91
|
def effective_cost
|
|
@@ -116,9 +116,14 @@ module RubyLLM
|
|
|
116
116
|
|
|
117
117
|
private
|
|
118
118
|
|
|
119
|
+
# With attempts, only the first one's cost: falling back to the case
|
|
120
|
+
# cost when it is unpriced would report a later attempt's spend as the
|
|
121
|
+
# single-shot cost.
|
|
119
122
|
def first_attempt_cost(result)
|
|
120
|
-
|
|
121
|
-
|
|
123
|
+
attempts = result.attempts || []
|
|
124
|
+
return result.cost if attempts.empty?
|
|
125
|
+
|
|
126
|
+
attempts.first[:cost]
|
|
122
127
|
end
|
|
123
128
|
|
|
124
129
|
def first_attempt_latency(result)
|
|
@@ -52,8 +52,10 @@ module RubyLLM
|
|
|
52
52
|
|
|
53
53
|
private
|
|
54
54
|
|
|
55
|
+
# cost_unknown is written only when true, so older files and runs that
|
|
56
|
+
# priced every case keep the same shape.
|
|
55
57
|
def history_entry
|
|
56
|
-
{
|
|
58
|
+
entry = {
|
|
57
59
|
date: Time.now.strftime("%Y-%m-%d"),
|
|
58
60
|
score: @stats.score,
|
|
59
61
|
total_cost: @stats.total_cost,
|
|
@@ -61,6 +63,8 @@ module RubyLLM
|
|
|
61
63
|
pass_rate_ratio: @stats.pass_rate_ratio,
|
|
62
64
|
cases_count: @stats.evaluated_results_count
|
|
63
65
|
}
|
|
66
|
+
entry[:cost_unknown] = true if @stats.unknown_cost_results.any?
|
|
67
|
+
entry
|
|
64
68
|
end
|
|
65
69
|
|
|
66
70
|
def serialize_for_baseline
|
|
@@ -74,13 +78,15 @@ module RubyLLM
|
|
|
74
78
|
end
|
|
75
79
|
|
|
76
80
|
def serialize_case(result)
|
|
77
|
-
{
|
|
81
|
+
serialized = {
|
|
78
82
|
name: result.name,
|
|
79
83
|
passed: result.passed?,
|
|
80
84
|
score: result.score,
|
|
81
85
|
details: result.details,
|
|
82
86
|
cost: result.cost
|
|
83
87
|
}
|
|
88
|
+
serialized[:cost_unknown] = true if result.cost_unknown?
|
|
89
|
+
serialized
|
|
84
90
|
end
|
|
85
91
|
|
|
86
92
|
def storage_path(root_dir, extension, model:, reasoning_effort: nil)
|
|
@@ -48,7 +48,9 @@ module RubyLLM
|
|
|
48
48
|
end
|
|
49
49
|
|
|
50
50
|
def run(input, context: {}, timeout_ms: nil)
|
|
51
|
-
|
|
51
|
+
WorkflowScope.workflow(name || "anonymous pipeline") do
|
|
52
|
+
Runner.new(steps: steps, context: context, timeout_ms: timeout_ms, token_budget: token_budget).call(input)
|
|
53
|
+
end
|
|
52
54
|
end
|
|
53
55
|
|
|
54
56
|
def test(input, responses: {}, timeout_ms: nil)
|
|
@@ -37,7 +37,9 @@ module RubyLLM
|
|
|
37
37
|
|
|
38
38
|
def execute_step(step_def, execution)
|
|
39
39
|
step_context = build_step_context(step_def)
|
|
40
|
-
result = step_def[:
|
|
40
|
+
result = WorkflowScope.step(step_def[:alias]) do
|
|
41
|
+
step_def[:step_class].run(execution.current_input, context: step_context)
|
|
42
|
+
end
|
|
41
43
|
|
|
42
44
|
execution.record_step(step_def[:alias], result)
|
|
43
45
|
end
|
|
@@ -83,6 +85,7 @@ module RubyLLM
|
|
|
83
85
|
total_usage: aggregate_usage(traces),
|
|
84
86
|
step_traces: traces
|
|
85
87
|
)
|
|
88
|
+
warn_incomplete_budget_usage(trace)
|
|
86
89
|
|
|
87
90
|
Result.new(
|
|
88
91
|
status: execution.status, step_results: execution.step_results,
|
|
@@ -91,6 +94,13 @@ module RubyLLM
|
|
|
91
94
|
)
|
|
92
95
|
end
|
|
93
96
|
|
|
97
|
+
def warn_incomplete_budget_usage(trace)
|
|
98
|
+
return unless @token_budget && !trace.usage_complete?
|
|
99
|
+
|
|
100
|
+
warn "[ruby_llm-contract] token_budget checked against incomplete token counts: " \
|
|
101
|
+
"a provider left out usage for at least one step, so the total undercounts"
|
|
102
|
+
end
|
|
103
|
+
|
|
94
104
|
def elapsed_ms(start_time)
|
|
95
105
|
((Process.clock_gettime(Process::CLOCK_MONOTONIC) - start_time) * 1000).round
|
|
96
106
|
end
|
|
@@ -40,6 +40,16 @@ module RubyLLM
|
|
|
40
40
|
@step_traces.any? { |step_trace| step_trace.respond_to?(:cost_unknown?) && step_trace.cost_unknown? }
|
|
41
41
|
end
|
|
42
42
|
|
|
43
|
+
# False when any step's provider left out token counts, so total_usage
|
|
44
|
+
# (and a token_budget checked against it) undercounts.
|
|
45
|
+
def usage_complete?
|
|
46
|
+
return true unless @step_traces.is_a?(Array)
|
|
47
|
+
|
|
48
|
+
@step_traces.none? do |step_trace|
|
|
49
|
+
step_trace.respond_to?(:usage_complete) && step_trace.usage_complete == false
|
|
50
|
+
end
|
|
51
|
+
end
|
|
52
|
+
|
|
43
53
|
def to_h
|
|
44
54
|
{ trace_id: @trace_id, total_latency_ms: @total_latency_ms,
|
|
45
55
|
total_usage: @total_usage, step_traces: @step_traces,
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
module Contract
|
|
5
|
+
# Options in the provider's own request vocabulary, passed to
|
|
6
|
+
# `chat.with_provider_options` and merged into the request as-is. Keys the
|
|
7
|
+
# contract itself sets are refused: an override would make the call differ
|
|
8
|
+
# from what max_cost, max_output and the schema were checked against.
|
|
9
|
+
# Only top-level keys are checked; a provider that nests these settings
|
|
10
|
+
# (Gemini's generationConfig, for one) is not.
|
|
11
|
+
module ProviderOptions
|
|
12
|
+
RESERVED_KEYS = %w[model messages input instructions system temperature stream
|
|
13
|
+
max_tokens max_output_tokens max_completion_tokens].freeze
|
|
14
|
+
# Where OpenAI's protocols put the structured-output schema.
|
|
15
|
+
SCHEMA_KEYS = %w[response_format text].freeze
|
|
16
|
+
|
|
17
|
+
def self.validate!(options, schema: false)
|
|
18
|
+
raise ArgumentError, "provider_options must be a Hash, got #{options.class}" unless options.is_a?(Hash)
|
|
19
|
+
|
|
20
|
+
reserved = schema ? RESERVED_KEYS + SCHEMA_KEYS : RESERVED_KEYS
|
|
21
|
+
clashes = options.keys.map(&:to_s) & reserved
|
|
22
|
+
return options if clashes.empty?
|
|
23
|
+
|
|
24
|
+
raise ArgumentError, "provider_options cannot set #{clashes.join(", ")}: the step sets " \
|
|
25
|
+
"#{clashes.one? ? "it" : "them"} (use the step DSL or context instead)"
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
# A frozen deep copy, so a hash the caller keeps mutating cannot change
|
|
29
|
+
# a step definition shared across threads.
|
|
30
|
+
def self.frozen_copy(object)
|
|
31
|
+
case object
|
|
32
|
+
when Hash then object.to_h { |key, value| [key, frozen_copy(value)] }.freeze
|
|
33
|
+
when Array then object.map { |value| frozen_copy(value) }.freeze
|
|
34
|
+
else object.frozen? ? object : object.dup.freeze
|
|
35
|
+
end
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
# Context keys win over the class-level ones; nil when nothing is set.
|
|
39
|
+
def self.merge(class_level, context_level)
|
|
40
|
+
merged = (class_level || {}).merge(context_level || {})
|
|
41
|
+
merged.empty? ? nil : merged
|
|
42
|
+
end
|
|
43
|
+
end
|
|
44
|
+
end
|
|
45
|
+
end
|