squishling 0.2.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -8,23 +8,69 @@ module Squishling
8
8
  present, holds additional state about the caller. Respond only with JSON matching the required schema.
9
9
  NOTE
10
10
 
11
+ # The most of a rejected output that is forwarded to the next escalation step.
12
+ MAX_FORWARDED_CHARS = 4_000
13
+
14
+ # One escalation's request: the system prompt, the output Schema, and the first user message. The judge's
15
+ # role (:judge) skips squish_validate and names the judge in errors and logs.
16
+ Prompt = Data.define(:instructions, :schema, :input, :role)
17
+
18
+ # One of a sampling harness's two samples, between rounds: the steps it may attempt, in order.
19
+ Sample = Struct.new(:name, :steps, :chat, :rejected, :result)
20
+
11
21
  def initialize(definition, receiver, inputs)
12
22
  @definition = definition
13
23
  @receiver = receiver
14
24
  @inputs = inputs
25
+ @client = LLMClient.new(definition.label)
26
+ @output_check = OutputCheck.new(definition, receiver, inputs)
15
27
  end
16
28
 
17
- # Makes each attempt of the escalation (Definition#escalation_path) until an output passes the schema and
18
- # the squish_validate check. Consecutive attempts on the same step continue the same conversation, so the
19
- # model sees what it got wrong; a new step starts a fresh chat told about the last rejected output.
29
+ # Runs the harness (Definition#harness) over the escalation (Definition#escalation_path).
20
30
  def call
21
- @instructions = @definition.instructions(@receiver)
22
- @schema = @definition.schema
23
- raise ConfigurationError, "#{@definition.label} has no instructions" if @instructions.nil? || @instructions.empty?
24
- raise ConfigurationError, "#{@definition.label} has no output_schema" unless @schema
31
+ purpose = @definition.purpose(@receiver)
32
+ schema = @definition.schema
33
+ raise ConfigurationError, "#{@definition.label} has no purpose" if purpose.nil? || purpose.empty?
34
+ raise ConfigurationError, "#{@definition.label} has no output_schema" unless schema
25
35
 
36
+ harness = @definition.harness
26
37
  path = @definition.escalation_path
27
- input = JSON.generate(payload)
38
+ body = payload
39
+ prompt = Prompt.new(instructions: "#{purpose}\n\n#{INPUT_NOTE}", schema:,
40
+ input: JSON.generate(body), role: nil)
41
+ return escalate(path, prompt) unless harness.squishsum?
42
+
43
+ # The sample steps and the judge are resolved before any sample is requested, so a missing one fails
44
+ # before costing anything.
45
+ sample_steps, rest = sampling_plan(path, harness)
46
+ judge = harness.judged? && Judge.new(definition: @definition, receiver: @receiver, client: @client, harness:,
47
+ rest:, purpose:, payload: body, escalate: method(:escalate),
48
+ log_failure: method(:log_failure))
49
+ sample(sample_steps, prompt, harness, judge)
50
+ end
51
+
52
+ private
53
+
54
+ # Splits the escalation into runs of consecutive identical steps (a step's attempts). A squishsum samples
55
+ # the first run twice; an ensemble samples the first run, then the second. Returns each sample's steps and
56
+ # the rest of the escalation, which is where a judge looks for its default step.
57
+ def sampling_plan(path, harness)
58
+ runs = path.chunk_while { |step, following| step == following }.to_a
59
+ if harness.ensemble? && runs.size < 2
60
+ raise ConfigurationError, "#{@definition.label}: the #{harness.type} harness needs a second escalation " \
61
+ "step to sample (adjacent identical steps count as one step's attempts)"
62
+ end
63
+
64
+ return [runs.first(2), runs.drop(2).flatten(1)] if harness.ensemble?
65
+
66
+ [[runs.first, runs.first], runs.drop(1).flatten(1)]
67
+ end
68
+
69
+ # Makes each attempt of the path until an output passes the schema and the squish_validate check.
70
+ # Consecutive attempts on the same step continue the same conversation, so the model sees what it got wrong;
71
+ # a new step starts a fresh chat told about the last rejected output (unless the step sets
72
+ # forward_rejected: false).
73
+ def escalate(path, prompt)
28
74
  chat = nil
29
75
  rejected = nil # [raw content, errors] of the last invalid output
30
76
  models = []
@@ -35,92 +81,126 @@ module Squishling
35
81
  if chat && step == path[attempt - 2]
36
82
  message = retry_message(rejected.last)
37
83
  else
38
- chat = start_chat(step)
39
- message = rejected ? escalation_message(input, *rejected) : input
84
+ chat = start_chat(step, prompt)
85
+ message = rejected && step.forward_rejected ? escalation_message(prompt.input, *rejected) : prompt.input
40
86
  end
41
87
 
42
88
  begin
43
- response = ask(chat, message, step)
89
+ response = @client.ask(chat, message, step)
44
90
  rescue LLMError => e
91
+ squawk(prompt, path, attempt, nil, e)
45
92
  raise if last
46
93
 
47
94
  # The failed request may be half-recorded in this chat, so the next attempt starts a fresh one.
48
95
  chat = nil
49
- log_failure(attempt, path, e.message)
96
+ log_failure(prompt.role, attempt, path, e.message)
50
97
  next
51
98
  end
52
99
 
53
- result, errors = check(response.content)
54
- return result if errors.empty?
100
+ result, errors = @output_check.call(response.content, prompt)
101
+ if errors.empty?
102
+ squawk(prompt, path, attempt, response, nil)
103
+ return result
104
+ end
55
105
 
56
- raise InvalidOutputError.new(errors:, raw: response.content, attempts: attempt, models:) if last
106
+ error = InvalidOutputError.new(errors:, raw: response.content, attempts: attempt, models: models.dup,
107
+ source: source(prompt.role))
108
+ squawk(prompt, path, attempt, response, error)
109
+ raise error if last
57
110
 
58
111
  rejected = [response.content, errors]
59
- log_failure(attempt, path, errors.join("; "))
112
+ log_failure(prompt.role, attempt, path, errors.join("; "))
60
113
  end
61
114
  end
62
115
 
63
- private
116
+ # Two samples, each on its own steps and retried per their attempts the same way escalate retries.
117
+ # Each round sends the pending samples' requests concurrently; everything else runs on this thread.
118
+ def sample(sample_steps, prompt, harness, judge)
119
+ samples = %w[a b].zip(sample_steps).map { |name, steps| Sample.new(name, steps) }
64
120
 
65
- def start_chat(step)
66
- chat = build_chat(step)
67
- chat.with_instructions("#{@instructions}\n\n#{INPUT_NOTE}")
68
- chat.with_schema(@schema.llm_schema)
69
- apply_params(chat, step.params)
70
- chat
71
- end
121
+ 1.upto(sample_steps.map(&:size).max) do |attempt|
122
+ pending = samples.reject(&:result)
123
+ break if pending.empty?
72
124
 
73
- def build_chat(step)
74
- RubyLLM.chat(**chat_options(step))
75
- rescue RubyLLM::ModelNotFoundError, RubyLLM::ConfigurationError => e
76
- raise ConfigurationError, "#{@definition.label}: #{e.message}"
77
- end
125
+ jobs = pending.map do |sample|
126
+ step = sample.steps[attempt - 1]
127
+ message = sample_message(sample, step, prompt)
128
+ -> { @client.ask(sample.chat, message, step) }
129
+ end
130
+ outcomes = LLMClient.concurrently(jobs)
131
+ # A setup mistake (or anything that isn't a failed request) ends the call, whichever sample hit it.
132
+ fatal = outcomes.map(&:last).find { |error| error && !error.is_a?(LLMError) }
133
+ raise fatal if fatal
78
134
 
79
- # RubyLLM validates some settings locally (e.g. an impossible thinking budget for the model)
80
- # and raises ArgumentError before any request is sent.
81
- def apply_params(chat, params)
82
- Params.apply(chat, params)
83
- rescue ArgumentError => e
84
- raise ConfigurationError, "#{@definition.label}: invalid params #{params.inspect} (#{e.message})"
135
+ pending.zip(outcomes).each { |sample, outcome| record(sample, outcome, prompt, attempt) }
136
+ end
137
+
138
+ settle(samples, harness, judge)
85
139
  end
86
140
 
87
- # Transient HTTP failures are already retried by RubyLLM (RubyLLM.config.max_retries); anything
88
- # that still fails is surfaced as an LLMError, which moves on to the next attempt of the
89
- # escalation (or propagates from the last one). A 400 means the request we built is invalid (an
90
- # unsupported param, a schema the provider rejects), so it's a setup mistake: escalating or a
91
- # fallback would otherwise hide it on every call.
92
- def ask(chat, message, step)
93
- chat.ask(message)
94
- rescue RubyLLM::ConfigurationError, RubyLLM::UnauthorizedError, RubyLLM::ForbiddenError => e
95
- raise ConfigurationError, "#{@definition.label}: #{e.class}: #{e.message}"
96
- rescue RubyLLM::BadRequestError => e
97
- raise ConfigurationError,
98
- "#{@definition.label}: the provider rejected the request (#{e.message})#{params_hint(step.params)}"
99
- rescue RubyLLM::Error, Faraday::Error => e
100
- raise LLMError, "#{@definition.label}: #{e.class}: #{e.message}"
141
+ # The next message for a sample: a retry in its chat after invalid output, otherwise a fresh chat.
142
+ def sample_message(sample, step, prompt)
143
+ return retry_message(sample.rejected.last) if sample.chat
144
+
145
+ sample.chat = start_chat(step, prompt)
146
+ sample.rejected && step.forward_rejected ? escalation_message(prompt.input, *sample.rejected) : prompt.input
101
147
  end
102
148
 
103
- # Models missing from RubyLLM's registry (e.g. newly released ones) are only usable when a
104
- # provider is named, so RubyLLM is told to assume they exist.
105
- def chat_options(step)
106
- model = step.model
107
- provider = step.provider
108
- options = { model:, provider: }.compact
109
- options[:assume_model_exists] = true if model && provider && !known_model?(model, provider)
110
- options
149
+ def record(sample, (response, error), prompt, attempt)
150
+ steps = sample.steps
151
+ last = attempt == steps.size
152
+ role = "sample #{sample.name}"
153
+ if error
154
+ squawk(prompt, steps, attempt, nil, error)
155
+ raise error if last
156
+
157
+ sample.chat = nil
158
+ return log_failure(role, attempt, steps, error.message)
159
+ end
160
+
161
+ result, errors = @output_check.call(response.content, prompt)
162
+ if errors.empty?
163
+ squawk(prompt, steps, attempt, response, nil)
164
+ return sample.result = result
165
+ end
166
+
167
+ models = steps.first(attempt).map(&:model)
168
+ invalid = InvalidOutputError.new(errors:, raw: response.content, attempts: attempt, models:)
169
+ squawk(prompt, steps, attempt, response, invalid)
170
+ raise invalid if last
171
+
172
+ sample.rejected = [response.content, errors]
173
+ log_failure(role, attempt, steps, errors.join("; "))
111
174
  end
112
175
 
113
- def known_model?(model, provider)
114
- RubyLLM.models.find(model, provider:)
115
- true
116
- rescue RubyLLM::ModelNotFoundError
117
- false
176
+ # Agreeing samples are accepted; otherwise the judge, if any, picks one or the call fails.
177
+ def settle(samples, harness, judge)
178
+ first, second = samples.map(&:result)
179
+ return first if agree?(first, second, harness.compare)
180
+
181
+ models = samples.map { |sample| sample.steps.first.model }
182
+ unless judge
183
+ log_warning("samples disagreed")
184
+ raise DisagreementError.new(candidates: [first, second], models:)
185
+ end
186
+
187
+ log_warning("samples disagreed, asking the judge (#{judge.step.display_name})")
188
+ choice, reason, detail = judge.verdict(first, second)
189
+ return { a: first, b: second }.fetch(choice) unless choice == :neither
190
+
191
+ raise DisagreementError.new(candidates: [first, second], verdict: :neither, reason:, detail:,
192
+ models: models + [judge.step.model])
118
193
  end
119
194
 
120
- def params_hint(params)
121
- return "" if params.empty?
195
+ # compare: is the developer's own code, so its exceptions propagate unwrapped.
196
+ def agree?(first, second, compare)
197
+ return Schema.jsonify(first.to_h) == Schema.jsonify(second.to_h) unless compare
198
+
199
+ @receiver.instance_exec(first, second, **@inputs, &compare) ? true : false
200
+ end
122
201
 
123
- ". Check params #{params.inspect}; reasoning models often reject sampling params such as temperature and top_p"
202
+ def start_chat(step, prompt)
203
+ @client.chat(step, instructions: prompt.instructions, schema: prompt.schema.llm_schema)
124
204
  end
125
205
 
126
206
  # Only the arguments, the declared squish_context names, and a squish! call's context leave the process.
@@ -154,92 +234,67 @@ module Squishling
154
234
  @receiver.instance_variable_get(:"@#{name}")
155
235
  end
156
236
 
157
- # RubyLLM parses structured output itself and leaves the raw string when that fails, so
158
- # content is a Hash on success, or a String/nil when the model refused, was cut off, or
159
- # ignored the schema.
160
- def parse(content)
161
- return [nil, ["response was empty"]] if content.nil? || (content.is_a?(String) && content.strip.empty?)
162
- return [content, []] unless content.is_a?(String)
163
-
164
- [JSON.parse(strip_code_fence(content)), []]
165
- rescue JSON::ParserError => e
166
- [nil, ["response was not valid JSON (#{e.message.lines.first&.strip})"]]
167
- end
168
-
169
- # Models without native structured output sometimes wrap JSON in a markdown code fence.
170
- def strip_code_fence(text)
171
- text[/\A\s*```(?:json)?\s*\n(.*?)\n\s*```\s*\z/m, 1] || text
172
- end
173
-
174
- # Parses and validates a response: [typed result, []] when it passes, [nil or result, errors] otherwise.
175
- def check(content)
176
- data, errors = parse(content)
177
- errors = @schema.validate(data) if errors.empty?
178
- return [nil, errors] if errors.any?
179
-
180
- result = @schema.build(data, squished: true)
181
- [result, validator_errors(result)]
237
+ def retry_message(errors)
238
+ "Your previous response was rejected:\n- #{errors.join("\n- ")}\nRespond again with corrected JSON only."
182
239
  end
183
240
 
184
- # Runs the squish_validate check, if any. Exceptions it raises are the user's own and propagate as-is.
185
- def validator_errors(result)
186
- validator = @definition.validator
187
- return [] unless validator
188
-
189
- value = @receiver.instance_exec(result, **@inputs, &validator)
190
- case value
191
- when nil, true then []
192
- when false then ["the output was rejected by squish_validate"]
193
- when String, Array then Array(value).flatten.compact.map(&:to_s).reject { |message| message.strip.empty? }
194
- else
195
- return contract_errors(value) if value.respond_to?(:success?) && value.respond_to?(:errors)
196
-
197
- raise ConfigurationError, "#{@definition.label}: squish_validate must return nil, true, false, a String, " \
198
- "an Array of Strings, or a validation result, got #{value.class}"
241
+ # The first message to a fresh chat after an earlier model's output was rejected. The rejected output is
242
+ # model-generated and may cross providers, so it is capped at MAX_FORWARDED_CHARS.
243
+ def escalation_message(input, raw, errors)
244
+ previous = rejected_text(raw)
245
+ previous = "(an empty response)" if previous.strip.empty? || raw.nil?
246
+ if previous.length > MAX_FORWARDED_CHARS
247
+ omitted = previous.length - MAX_FORWARDED_CHARS
248
+ previous = "#{previous[0, MAX_FORWARDED_CHARS]}... [truncated, #{omitted} more characters]"
199
249
  end
250
+ "#{input}\n\nA previous attempt at this request returned:\n#{previous}\n" \
251
+ "It was rejected:\n- #{errors.join("\n- ")}\nRespond with corrected JSON only."
200
252
  end
201
253
 
202
- # A dry-validation style result: errors.to_h is { key => ["message", ...] }, nested for nested keys.
203
- def contract_errors(value)
204
- return [] if value.success?
254
+ # The rejected output as text. Output that was rejected for being unrepresentable in JSON (Infinity, invalid
255
+ # UTF-8) can't be serialized or measured as is, so it is scrubbed or replaced rather than raising.
256
+ def rejected_text(raw)
257
+ return utf8(raw).scrub("?") if raw.is_a?(String)
205
258
 
206
- errors = value.errors
207
- errors = errors.to_h if !errors.is_a?(Array) && errors.respond_to?(:to_h)
208
- messages = errors.is_a?(Hash) ? flatten_messages(errors) : Array(errors).map(&:to_s)
209
- messages.empty? ? ["the output was rejected by squish_validate"] : messages
259
+ JSON.generate(raw)
260
+ rescue JSON::JSONError
261
+ "(a response JSON can't represent)"
210
262
  end
211
263
 
212
- def flatten_messages(errors, prefix = nil)
213
- errors.flat_map do |key, value|
214
- path = [prefix, key].compact.join(".")
215
- next flatten_messages(value, path) if value.is_a?(Hash)
216
-
217
- Array(value).map { |message| path.empty? ? message.to_s : "#{path} #{message}" }
218
- end
264
+ # A binary-tagged string counts every byte as valid, so it is read as UTF-8 before scrubbing.
265
+ def utf8(text)
266
+ text.encoding == Encoding::BINARY ? text.dup.force_encoding(Encoding::UTF_8) : text
219
267
  end
220
268
 
221
- def retry_message(errors)
222
- "Your previous response was rejected:\n- #{errors.join("\n- ")}\nRespond again with corrected JSON only."
269
+ # Runs the observability hook, if any, with this attempt's raw output (nil when the call itself failed),
270
+ # the error that ended it (nil when it was accepted), and what was asked.
271
+ def squawk(prompt, path, attempt, response, error)
272
+ hook = @definition.squawk
273
+ return unless hook
274
+
275
+ step = path[attempt - 1]
276
+ metadata = {
277
+ label: @definition.label, attempt:, attempts: path.size, final: attempt == path.size,
278
+ model: response&.model || step.model, provider: step.provider, params: step.params, input: prompt.input,
279
+ usage: response&.tokens&.to_h
280
+ }
281
+ Squawk.call(hook, output: response&.content, metadata:, error:)
223
282
  end
224
283
 
225
- # The first message to a fresh chat after an earlier model's output was rejected.
226
- def escalation_message(input, raw, errors)
227
- previous = raw.is_a?(String) ? raw : JSON.generate(raw)
228
- previous = "(an empty response)" if previous.strip.empty? || raw.nil?
229
- "#{input}\n\nA previous attempt at this request returned:\n#{previous}\n" \
230
- "It was rejected:\n- #{errors.join("\n- ")}\nRespond with corrected JSON only."
284
+ def source(role)
285
+ role == :judge ? "Judge" : "LLM"
231
286
  end
232
287
 
233
- def log_failure(attempt, path, reason)
288
+ def log_failure(role, attempt, path, reason)
234
289
  step = path[attempt - 1]
235
290
  next_step = path[attempt]
236
- action = next_step == step ? "retrying" : "escalating to #{model_name(next_step)}"
237
- Squishling.config.logger&.warn("[Squishling] #{@definition.label} attempt #{attempt} of #{path.size} " \
238
- "(#{model_name(step)}) failed, #{action}: #{reason}")
291
+ action = next_step == step ? "retrying" : "escalating to #{next_step.display_name}"
292
+ who = role ? "#{role} attempt" : "attempt"
293
+ log_warning("#{who} #{attempt} of #{path.size} (#{step.display_name}) failed, #{action}: #{reason}")
239
294
  end
240
295
 
241
- def model_name(step)
242
- step.model || "RubyLLM default model"
296
+ def log_warning(message)
297
+ Squishling.config.logger&.warn("[Squishling] #{@definition.label} #{message}")
243
298
  end
244
299
  end
245
300
  end
@@ -0,0 +1,142 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Squishling
4
+ # Picks between a judged harness's two disagreeing samples (judged_squishsum or judged_ensemble), or rejects
5
+ # both. The judge is the harness's judge: step, or else the escalation's next step after the ones the samples
6
+ # ran on. A chat judge answers with a strict verdict schema; a :judgment judge (a System One decision model such
7
+ # as Jev) answers one choice question through RubyLLM.judge, and its pick counts only at or above
8
+ # min_confidence.
9
+ class Judge
10
+ VERDICT_SCHEMA = {
11
+ "type" => "object",
12
+ "properties" => {
13
+ "verdict" => { "type" => "string", "enum" => %w[a b neither] },
14
+ "reason" => { "type" => "string" }
15
+ },
16
+ "required" => %w[verdict reason],
17
+ "additionalProperties" => false
18
+ }.freeze
19
+
20
+ INPUT_NOTE = <<~NOTE
21
+ The input is a JSON object. "input" is what the operation received ("arguments" and, when present,
22
+ "context"); "output_schema" is the format both candidates follow; "candidates" holds "a" and "b".
23
+ Respond only with JSON matching the required schema: your verdict ("a", "b", or "neither") and a short reason.
24
+ NOTE
25
+
26
+ CHOICES = {
27
+ a: "Candidate a correctly carries out the purpose for this input (also when both do)",
28
+ b: "Candidate b correctly carries out the purpose for this input",
29
+ neither: "Neither candidate is clearly correct"
30
+ }.freeze
31
+
32
+ def initialize(definition:, receiver:, client:, harness:, rest:, purpose:, payload:, escalate:, log_failure:)
33
+ @definition = definition
34
+ @receiver = receiver
35
+ @client = client
36
+ @harness = harness
37
+ @purpose = purpose
38
+ @payload = payload
39
+ @escalate = escalate
40
+ @log_failure = log_failure
41
+ @steps = resolve_steps(rest)
42
+ end
43
+
44
+ # The judge's first attempt, for logs and DisagreementError#models.
45
+ def step
46
+ @steps.first
47
+ end
48
+
49
+ # [:a | :b, reason, detail] for the chosen candidate, or [:neither, reason, detail]. The detail is text
50
+ # Squishling wrote (nil for a chat judge, whose reason is model-written) and is safe for error messages.
51
+ def verdict(first, second)
52
+ candidates = { a: Schema.jsonify(first.to_h), b: Schema.jsonify(second.to_h) }
53
+ judgment? ? judgment(candidates) : chat(candidates)
54
+ end
55
+
56
+ private
57
+
58
+ def judgment?
59
+ @harness.judge&.fetch(:type) == :judgment
60
+ end
61
+
62
+ # A declared judge brings its own model and provider. A chat judge gets the method's generation params
63
+ # under its own, like any escalation step; a judgment judge sends only its own params, as provider options.
64
+ def resolve_steps(rest)
65
+ if (judge = @harness.judge)
66
+ params = judge[:type] == :judgment ? {} : @definition.params
67
+ return ModelPath.steps([judge], provider: nil, params:)
68
+ end
69
+
70
+ if rest.empty?
71
+ ordinal = @harness.ensemble? ? "third" : "second"
72
+ raise ConfigurationError, "#{@definition.label}: the #{@harness.type} harness needs a judge: or a " \
73
+ "#{ordinal} escalation step to judge with"
74
+ end
75
+
76
+ rest.take_while { |step| step == rest.first }
77
+ end
78
+
79
+ def chat(candidates)
80
+ prompt = Invoker::Prompt.new(
81
+ instructions: "#{judge_instructions}\n\nThe operation's purpose:\n<purpose>\n#{@purpose}\n" \
82
+ "</purpose>\n\n#{INPUT_NOTE}",
83
+ schema: Schema.for(VERDICT_SCHEMA), input: JSON.generate(state(candidates)), role: :judge
84
+ )
85
+ result = @escalate.call(@steps, prompt)
86
+ [result[:verdict].to_sym, result[:reason], nil]
87
+ end
88
+
89
+ def judgment(candidates)
90
+ input = { purpose: @purpose, **state(candidates) }
91
+ questions = { winner: { type: :choice, instructions: judge_instructions, options: CHOICES } }
92
+ answer = judge_with_retries(input, questions)[:winner]
93
+ unless answer.respond_to?(:choice) && answer.respond_to?(:probabilities)
94
+ raise InvalidOutputError.new(errors: ["the judgment didn't answer the winner question"], source: "Judge")
95
+ end
96
+
97
+ interpret(answer.choice.to_s.to_sym, answer.probabilities)
98
+ end
99
+
100
+ # The threshold is checked on the exact probability; it's rounded only for the reason text.
101
+ def interpret(choice, probabilities)
102
+ probability = probabilities.to_h.find { |key, _| key.to_s == choice.to_s }&.last.to_f
103
+ minimum = @harness.judge[:min_confidence]
104
+ shown = probability.round(3)
105
+ picked = %i[a b].include?(choice)
106
+ if picked && probability >= minimum
107
+ text = "chosen with probability #{shown}"
108
+ return [choice, text, text]
109
+ end
110
+
111
+ text =
112
+ if choice == :neither then "the judge chose neither (probability #{shown})"
113
+ elsif picked then "the judge chose #{choice} with probability #{shown}, below min_confidence #{minimum}"
114
+ else "the judge returned an unrecognized choice (probability #{shown})"
115
+ end
116
+ [:neither, text, text]
117
+ end
118
+
119
+ # A judgment always matches its questions, so only a failed request (LLMError) moves on to the next attempt.
120
+ def judge_with_retries(input, questions)
121
+ @steps.each.with_index(1) do |step, attempt|
122
+ return @client.judge(input, questions:, step:)
123
+ rescue LLMError => e
124
+ raise if attempt == @steps.size
125
+
126
+ @log_failure.call(:judge, attempt, @steps, e.message)
127
+ end
128
+ end
129
+
130
+ def state(candidates)
131
+ { input: @payload, output_schema: @definition.schema.llm_schema["schema"], candidates: }
132
+ end
133
+
134
+ def judge_instructions
135
+ value = @harness.judge_instructions || Harness::DEFAULT_JUDGE_INSTRUCTIONS
136
+ value = @receiver.instance_exec(&value) if value.is_a?(Proc)
137
+ return value if value.is_a?(String) && !value.strip.empty?
138
+
139
+ raise ConfigurationError, "#{@definition.label}: the judge_instructions proc must return a non-blank String"
140
+ end
141
+ end
142
+ end