ruby_llm-contract 1.1.0 → 1.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +86 -0
  3. data/README.md +12 -3
  4. data/docs/guide/getting_started.md +17 -1
  5. data/docs/guide/relation_to_agent.md +8 -2
  6. data/lib/ruby_llm/contract/adapters/response.rb +28 -2
  7. data/lib/ruby_llm/contract/adapters/ruby_llm.rb +72 -18
  8. data/lib/ruby_llm/contract/concerns/eval_host.rb +2 -2
  9. data/lib/ruby_llm/contract/concerns/usage_aggregator.rb +8 -5
  10. data/lib/ruby_llm/contract/cost_calculator.rb +69 -41
  11. data/lib/ruby_llm/contract/eval/case_executor.rb +1 -1
  12. data/lib/ruby_llm/contract/eval/case_result.rb +3 -0
  13. data/lib/ruby_llm/contract/eval/eval_history.rb +2 -1
  14. data/lib/ruby_llm/contract/eval/evaluator/proc_evaluator.rb +6 -2
  15. data/lib/ruby_llm/contract/eval/model_comparison.rb +11 -1
  16. data/lib/ruby_llm/contract/eval/prompt_diff_serializer.rb +1 -1
  17. data/lib/ruby_llm/contract/eval/recommender.rb +12 -5
  18. data/lib/ruby_llm/contract/eval/report.rb +4 -1
  19. data/lib/ruby_llm/contract/eval/report_stats.rb +10 -5
  20. data/lib/ruby_llm/contract/eval/report_storage.rb +14 -8
  21. data/lib/ruby_llm/contract/eval/unknown_cost_gate.rb +0 -8
  22. data/lib/ruby_llm/contract/minitest.rb +2 -2
  23. data/lib/ruby_llm/contract/pipeline/runner.rb +8 -0
  24. data/lib/ruby_llm/contract/pipeline/trace.rb +10 -0
  25. data/lib/ruby_llm/contract/railtie.rb +1 -1
  26. data/lib/ruby_llm/contract/rake_task/suite_gate.rb +4 -4
  27. data/lib/ruby_llm/contract/rake_task.rb +4 -4
  28. data/lib/ruby_llm/contract/rspec/pass_eval.rb +2 -2
  29. data/lib/ruby_llm/contract/step/base.rb +21 -12
  30. data/lib/ruby_llm/contract/step/dsl.rb +5 -14
  31. data/lib/ruby_llm/contract/step/limit_checker.rb +2 -1
  32. data/lib/ruby_llm/contract/step/result_builder.rb +32 -3
  33. data/lib/ruby_llm/contract/step/retry_executor.rb +35 -4
  34. data/lib/ruby_llm/contract/step/runner.rb +7 -1
  35. data/lib/ruby_llm/contract/step/runner_config.rb +2 -2
  36. data/lib/ruby_llm/contract/step/trace.rb +54 -20
  37. data/lib/ruby_llm/contract/token_estimator.rb +5 -3
  38. data/lib/ruby_llm/contract/unknown_policy.rb +21 -0
  39. data/lib/ruby_llm/contract/version.rb +1 -1
  40. data/lib/ruby_llm/contract.rb +17 -4
  41. metadata +2 -1
@@ -6,6 +6,9 @@ module RubyLLM
6
6
  module RetryExecutor
7
7
  include Concerns::UsageAggregator
8
8
 
9
+ # Copied from each attempt's trace into its attempts entry when present.
10
+ ATTEMPT_TRACE_FIELDS = %i[usage latency_ms cost finish_reason].freeze
11
+
9
12
  private
10
13
 
11
14
  def run_with_retry(input, adapter:, default_model:, policy:, context_temperature: nil, extra_options: {})
@@ -39,7 +42,9 @@ module RubyLLM
39
42
  observations: last.observations,
40
43
  trace: last.trace.merge(
41
44
  attempts: attempt_log, usage: aggregated_usage,
42
- cost: total_cost, latency_ms: total_latency
45
+ cost: total_cost, latency_ms: total_latency,
46
+ usage_complete: aggregate_flag(all_attempts, :usage_complete),
47
+ cost_complete: aggregate_cost_complete(all_attempts)
43
48
  )
44
49
  )
45
50
  end
@@ -53,12 +58,38 @@ module RubyLLM
53
58
  end
54
59
 
55
60
  def append_trace_fields(entry, trace)
56
- entry[:usage] = trace.usage if trace.respond_to?(:usage) && trace.usage
57
- entry[:latency_ms] = trace.latency_ms if trace.respond_to?(:latency_ms) && trace.latency_ms
58
- entry[:cost] = trace.cost if trace.respond_to?(:cost) && trace.cost
61
+ ATTEMPT_TRACE_FIELDS.each do |field|
62
+ value = trace_value(trace, field)
63
+ entry[field] = value if value
64
+ end
65
+ entry[:cost_unknown] = true if trace_value(trace, :cost_unknown?)
66
+ entry[:usage_complete] = false if trace_value(trace, :usage_complete) == false
59
67
  entry
60
68
  end
61
69
 
70
+ def trace_value(trace, method)
71
+ trace.public_send(method) if trace.respond_to?(method)
72
+ end
73
+
74
+ # The subtotal is complete only if every attempt was; the last attempt's
75
+ # flag must not speak for it. nil when no attempt reported accounting
76
+ # state, so a retried Test-adapter step serializes exactly as before.
77
+ def aggregate_cost_complete(all_attempts)
78
+ traces = all_attempts.map { |a| a[:result].trace }
79
+ return false if traces.any? { |trace| trace.respond_to?(:cost_unknown?) && trace.cost_unknown? }
80
+
81
+ aggregate_flag(all_attempts, :cost_complete)
82
+ end
83
+
84
+ def aggregate_flag(all_attempts, flag)
85
+ flags = all_attempts.map { |a| a[:result].trace }
86
+ .select { |trace| trace.respond_to?(flag) }
87
+ .map { |trace| trace.public_send(flag) }.compact
88
+ return nil if flags.empty?
89
+
90
+ flags.all?
91
+ end
92
+
62
93
  def sum_attempt_costs(all_attempts)
63
94
  costs = extract_trace_values(all_attempts, :cost)
64
95
  return nil if costs.empty?
@@ -59,7 +59,8 @@ module RubyLLM
59
59
  output_type: @config.output_type,
60
60
  output_schema: @config.output_schema,
61
61
  model: @config.model,
62
- observers: @config.observers
62
+ observers: @config.observers,
63
+ max_output: @config.effective_max_output
63
64
  )
64
65
  end
65
66
 
@@ -87,6 +88,11 @@ module RubyLLM
87
88
  @config.on_unknown_attachment_size
88
89
  end
89
90
 
91
+ # The provider the adapter will call, so the estimate uses its price.
92
+ def pricing_provider
93
+ @config.extra_options&.dig(:provider)
94
+ end
95
+
90
96
  def attachment_present?
91
97
  opts = @config.extra_options
92
98
  !opts.nil? && !opts[:attachment].nil?
@@ -28,8 +28,8 @@ module RubyLLM
28
28
  def self.build(input_type:, output_type:, prompt_block:, contract_definition:,
29
29
  adapter:, model:,
30
30
  output_schema: nil, max_output: nil,
31
- max_input: nil, max_cost: nil, on_unknown_pricing: :refuse,
32
- attachment_token_estimate: nil, on_unknown_attachment_size: :refuse,
31
+ max_input: nil, max_cost: nil, on_unknown_pricing: UnknownPolicy::DEFAULT,
32
+ attachment_token_estimate: nil, on_unknown_attachment_size: UnknownPolicy::DEFAULT,
33
33
  temperature: nil, extra_options: {}, observers: [])
34
34
  new(
35
35
  input_type: input_type, output_type: output_type,
@@ -7,19 +7,34 @@ module RubyLLM
7
7
  include Concerns::TraceEquality
8
8
  include Concerns::DeepFreeze
9
9
 
10
- attr_reader :messages, :model, :latency_ms, :usage, :attempts, :cost
11
-
12
- def initialize(messages: nil, model: nil, latency_ms: nil, usage: nil, attempts: nil, cost: nil)
10
+ attr_reader :messages, :model, :latency_ms, :usage, :attempts, :cost,
11
+ :usage_complete, :cost_complete, :finish_reason
12
+
13
+ # Marks `cost:` as not given, as distinct from an explicit nil. Only an
14
+ # omitted cost is priced from the registry; nil from an adapter that
15
+ # reports its own cost means "unknown" and must stay unknown.
16
+ COST_UNSET = Object.new.freeze
17
+
18
+ # usage_complete / cost_complete are nil for traces that carry no
19
+ # accounting state (Test adapter, older adapters, hand-built traces);
20
+ # those keep the token-count rule in `unpriced?`.
21
+ def initialize(messages: nil, model: nil, latency_ms: nil, usage: nil, attempts: nil, cost: COST_UNSET,
22
+ usage_complete: nil, cost_complete: nil, finish_reason: nil)
13
23
  @messages = deep_dup_freeze(messages)
14
24
  @model = model.frozen? ? model : model&.dup&.freeze
15
25
  @latency_ms = latency_ms
16
26
  @usage = deep_dup_freeze(usage)
17
27
  @attempts = deep_dup_freeze(attempts)
18
- @cost = cost || CostCalculator.calculate(model_name: model, usage: usage)
28
+ @usage_complete = usage_complete
29
+ @cost_complete = cost_complete
30
+ @finish_reason = finish_reason
31
+ @cost = resolve_cost(cost)
32
+ @keep_nil_cost = keep_nil_cost?(cost)
19
33
  freeze
20
34
  end
21
35
 
22
- KNOWN_KEYS = %i[messages model latency_ms usage attempts cost].freeze
36
+ KNOWN_KEYS = %i[messages model latency_ms usage attempts cost
37
+ usage_complete cost_complete finish_reason].freeze
23
38
 
24
39
  def [](key)
25
40
  return nil unless KNOWN_KEYS.include?(key.to_sym)
@@ -39,27 +54,27 @@ module RubyLLM
39
54
  end
40
55
  alias has_key? key?
41
56
 
57
+ # Never re-prices: a retry merges a subtotal and an aggregate usage that
58
+ # no single model's pricing describes.
42
59
  def merge(**overrides)
43
- self.class.new(
44
- messages: overrides.fetch(:messages, @messages),
45
- model: overrides.fetch(:model, @model),
46
- latency_ms: overrides.fetch(:latency_ms, @latency_ms),
47
- usage: overrides.fetch(:usage, @usage),
48
- attempts: overrides.fetch(:attempts, @attempts),
49
- cost: overrides.fetch(:cost, @cost)
50
- )
60
+ self.class.new(**KNOWN_KEYS.to_h { |key| [key, overrides.fetch(key) { public_send(key) }] })
51
61
  end
52
62
 
53
- # True when a call ran on a named model, used tokens, and still got no
54
- # cost, so any total built from this trace undercounts. A zero-token call
63
+ # True when a total built from this trace undercounts: the adapter said
64
+ # its cost does not cover the call, or (no accounting state) a call ran
65
+ # on a named model, used tokens, and still got no cost. A zero-token call
55
66
  # (Test adapter, sample_response) costs 0 whatever the pricing. A retried
56
67
  # step is judged per attempt: a priced subtotal must not hide an unpriced
57
- # attempt, and the merged trace may have re-priced on the last model.
68
+ # attempt.
58
69
  def cost_unknown?
70
+ return true if @cost_complete == false
71
+
59
72
  if @attempts.is_a?(Array) && !@attempts.empty?
60
- @attempts.any? { |attempt| self.class.unpriced?(attempt[:model], attempt[:usage], attempt[:cost]) }
73
+ @attempts.any? do |attempt|
74
+ attempt[:cost_unknown] || self.class.unpriced?(attempt[:model], attempt[:usage], attempt[:cost])
75
+ end
61
76
  else
62
- self.class.unpriced?(@model, @usage, @cost)
77
+ @cost_complete.nil? && self.class.unpriced?(@model, @usage, @cost)
63
78
  end
64
79
  end
65
80
 
@@ -70,8 +85,11 @@ module RubyLLM
70
85
  end
71
86
 
72
87
  def to_h
73
- { messages: @messages, model: @model, latency_ms: @latency_ms,
74
- usage: @usage, attempts: @attempts, cost: @cost }.compact
88
+ hash = { messages: @messages, model: @model, latency_ms: @latency_ms,
89
+ usage: @usage, attempts: @attempts, cost: @cost, usage_complete: @usage_complete,
90
+ cost_complete: @cost_complete, finish_reason: @finish_reason }.compact
91
+ hash[:cost] = nil if @keep_nil_cost
92
+ hash
75
93
  end
76
94
 
77
95
  def to_s
@@ -80,6 +98,22 @@ module RubyLLM
80
98
 
81
99
  private
82
100
 
101
+ # A given nil cost that the registry would price must stay `cost: nil` in
102
+ # to_h, or `Trace.new(**to_h)` would come back priced. Otherwise it is
103
+ # left out, so a merged trace compares equal to the one it came from.
104
+ def keep_nil_cost?(cost)
105
+ return false unless cost.nil? && @cost_complete.nil?
106
+
107
+ !CostCalculator.calculate(model_name: @model, usage: @usage).nil?
108
+ end
109
+
110
+ def resolve_cost(cost)
111
+ return cost unless cost.equal?(COST_UNSET)
112
+ return nil unless @cost_complete.nil?
113
+
114
+ CostCalculator.calculate(model_name: @model, usage: @usage)
115
+ end
116
+
83
117
  def build_summary_parts
84
118
  parts = [@model || "no-model"]
85
119
  parts << "#{@latency_ms}ms" if @latency_ms
@@ -11,9 +11,11 @@ module RubyLLM
11
11
  # - Worse for non-English text, code, structured data, and unusual scripts
12
12
  # - Useless for models with very different tokenizers (e.g. some open-source models)
13
13
  #
14
- # RubyLLM 1.14 ships no pre-flight tokenizer either; once the API call
15
- # returns, `RubyLLM::Tokens` provides accurate counts from provider usage
16
- # data. This estimator is for the *pre-flight refusal* path only — its job
14
+ # RubyLLM 2.x can count tokens exactly before a call (`chat.count_tokens`),
15
+ # but only on some providers and at the price of an extra request, so this
16
+ # local estimate stays the default; once the API call returns,
17
+ # `RubyLLM::Tokens` provides accurate counts from provider usage data.
18
+ # This estimator is for the *pre-flight refusal* path only - its job
17
19
  # is to answer "is this call almost certainly within budget?" with enough
18
20
  # accuracy that runaway prompts get caught, while accepting that the
19
21
  # boundary cases will be wrong.
@@ -0,0 +1,21 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Contract
5
+ # What a limit does when the number it needs is unknown: model pricing for
6
+ # max_cost/maximum_cost, attachment size for max_input. One vocabulary for
7
+ # every such option, step-level and eval-level, so a new mode or a changed
8
+ # default reaches all of them. Standalone: the rake task loads it before the
9
+ # rest of the gem to validate its settings at definition time.
10
+ module UnknownPolicy
11
+ MODES = %i[refuse warn].freeze
12
+ DEFAULT = :refuse
13
+
14
+ def self.validate!(option, mode)
15
+ return mode if MODES.include?(mode)
16
+
17
+ raise ArgumentError, "#{option} must be #{MODES.map(&:inspect).join(" or ")}, got #{mode.inspect}"
18
+ end
19
+ end
20
+ end
21
+ end
@@ -2,6 +2,6 @@
2
2
 
3
3
  module RubyLLM
4
4
  module Contract
5
- VERSION = "1.1.0"
5
+ VERSION = "1.1.2"
6
6
  end
7
7
  end
@@ -2,10 +2,19 @@
2
2
 
3
3
  require_relative "contract/version"
4
4
  require_relative "contract/errors"
5
+ require_relative "contract/unknown_policy"
5
6
  require_relative "contract/types"
6
7
 
7
8
  module RubyLLM
8
9
  module Contract
10
+ # Rails dirs holding Step classes. Each one's eval/ subdir holds define_eval
11
+ # files, which define no constant, so Zeitwerk must ignore exactly those.
12
+ RAILS_CONTRACT_DIRS = %w[app/contracts app/steps].freeze
13
+ RAILS_EVAL_DIRS = RAILS_CONTRACT_DIRS.map { |dir| "#{dir}/eval" }.freeze
14
+
15
+ # Set while load_evals! runs: redefining an eval is then a reload, not a mistake.
16
+ RELOADING_KEY = :ruby_llm_contract_reloading
17
+
9
18
  class << self
10
19
  def configuration
11
20
  @configuration ||= Configuration.new
@@ -51,7 +60,7 @@ module RubyLLM
51
60
  def load_evals!(*dirs)
52
61
  dirs = dirs.flatten.compact
53
62
  if dirs.empty? && defined?(::Rails)
54
- dirs = %w[app/steps/eval app/contracts/eval].filter_map do |path|
63
+ dirs = RAILS_EVAL_DIRS.filter_map do |path|
55
64
  full = ::Rails.root.join(path)
56
65
  full.to_s if full.exist?
57
66
  end
@@ -64,7 +73,7 @@ module RubyLLM
64
73
  eager_load_contract_dirs! if defined?(::Rails)
65
74
 
66
75
  # Clear file-sourced evals ONCE, then load ALL dirs.
67
- Thread.current[:ruby_llm_contract_reloading] = true
76
+ Thread.current[RELOADING_KEY] = true
68
77
  eval_hosts.each do |host|
69
78
  host.clear_file_sourced_evals! if host.respond_to?(:clear_file_sourced_evals!)
70
79
  end
@@ -73,7 +82,11 @@ module RubyLLM
73
82
  Dir[File.join(d, "**", "*_eval.rb")].each { |f| load f }
74
83
  end
75
84
  ensure
76
- Thread.current[:ruby_llm_contract_reloading] = false
85
+ Thread.current[RELOADING_KEY] = false
86
+ end
87
+
88
+ def reloading?
89
+ Thread.current[RELOADING_KEY] ? true : false
77
90
  end
78
91
 
79
92
  def normalize_candidate_config(entry)
@@ -111,7 +124,7 @@ module RubyLLM
111
124
  end
112
125
 
113
126
  def eager_load_contract_dirs!
114
- %w[app/contracts app/steps].each do |path|
127
+ RAILS_CONTRACT_DIRS.each do |path|
115
128
  full = ::Rails.root.join(path)
116
129
  next unless full.exist?
117
130
 
metadata CHANGED
@@ -1,7 +1,7 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: ruby_llm-contract
3
3
  version: !ruby/object:Gem::Version
4
- version: 1.1.0
4
+ version: 1.1.2
5
5
  platform: ruby
6
6
  authors:
7
7
  - Justyna
@@ -193,6 +193,7 @@ files:
193
193
  - lib/ruby_llm/contract/step/trace.rb
194
194
  - lib/ruby_llm/contract/token_estimator.rb
195
195
  - lib/ruby_llm/contract/types.rb
196
+ - lib/ruby_llm/contract/unknown_policy.rb
196
197
  - lib/ruby_llm/contract/version.rb
197
198
  - ruby_llm-contract.gemspec
198
199
  homepage: https://github.com/justi/ruby_llm-contract