squishling 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +83 -0
- data/README.md +85 -30
- data/docs/configuration.md +40 -16
- data/docs/failures.md +68 -8
- data/docs/harnesses.md +198 -0
- data/docs/measuring-tokens.md +148 -0
- data/docs/naming.md +43 -0
- data/docs/routing.md +36 -27
- data/docs/schemas.md +4 -3
- data/lib/squishling/appendices.rb +3 -3
- data/lib/squishling/class_methods.rb +60 -29
- data/lib/squishling/collisions.rb +54 -0
- data/lib/squishling/configuration.rb +19 -0
- data/lib/squishling/definition.rb +48 -23
- data/lib/squishling/errors.rb +22 -1
- data/lib/squishling/harness.rb +153 -0
- data/lib/squishling/invoker.rb +188 -133
- data/lib/squishling/judge.rb +142 -0
- data/lib/squishling/llm_client.rb +150 -0
- data/lib/squishling/model_path.rb +21 -7
- data/lib/squishling/output_check.rb +109 -0
- data/lib/squishling/params.rb +37 -1
- data/lib/squishling/router.rb +8 -2
- data/lib/squishling/source.rb +5 -5
- data/lib/squishling/squawk.rb +52 -0
- data/lib/squishling/version.rb +1 -1
- data/lib/squishling.rb +24 -7
- metadata +12 -3
data/lib/squishling/invoker.rb
CHANGED
|
@@ -8,23 +8,69 @@ module Squishling
|
|
|
8
8
|
present, holds additional state about the caller. Respond only with JSON matching the required schema.
|
|
9
9
|
NOTE
|
|
10
10
|
|
|
11
|
+
# The most of a rejected output that is forwarded to the next escalation step.
|
|
12
|
+
MAX_FORWARDED_CHARS = 4_000
|
|
13
|
+
|
|
14
|
+
# One escalation's request: the system prompt, the output Schema, and the first user message. The judge's
|
|
15
|
+
# role (:judge) skips squish_validate and names the judge in errors and logs.
|
|
16
|
+
Prompt = Data.define(:instructions, :schema, :input, :role)
|
|
17
|
+
|
|
18
|
+
# One of a sampling harness's two samples, between rounds: the steps it may attempt, in order.
|
|
19
|
+
Sample = Struct.new(:name, :steps, :chat, :rejected, :result)
|
|
20
|
+
|
|
11
21
|
def initialize(definition, receiver, inputs)
|
|
12
22
|
@definition = definition
|
|
13
23
|
@receiver = receiver
|
|
14
24
|
@inputs = inputs
|
|
25
|
+
@client = LLMClient.new(definition.label)
|
|
26
|
+
@output_check = OutputCheck.new(definition, receiver, inputs)
|
|
15
27
|
end
|
|
16
28
|
|
|
17
|
-
#
|
|
18
|
-
# the squish_validate check. Consecutive attempts on the same step continue the same conversation, so the
|
|
19
|
-
# model sees what it got wrong; a new step starts a fresh chat told about the last rejected output.
|
|
29
|
+
# Runs the harness (Definition#harness) over the escalation (Definition#escalation_path).
|
|
20
30
|
def call
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
raise ConfigurationError, "#{@definition.label} has no
|
|
24
|
-
raise ConfigurationError, "#{@definition.label} has no output_schema" unless
|
|
31
|
+
purpose = @definition.purpose(@receiver)
|
|
32
|
+
schema = @definition.schema
|
|
33
|
+
raise ConfigurationError, "#{@definition.label} has no purpose" if purpose.nil? || purpose.empty?
|
|
34
|
+
raise ConfigurationError, "#{@definition.label} has no output_schema" unless schema
|
|
25
35
|
|
|
36
|
+
harness = @definition.harness
|
|
26
37
|
path = @definition.escalation_path
|
|
27
|
-
|
|
38
|
+
body = payload
|
|
39
|
+
prompt = Prompt.new(instructions: "#{purpose}\n\n#{INPUT_NOTE}", schema:,
|
|
40
|
+
input: JSON.generate(body), role: nil)
|
|
41
|
+
return escalate(path, prompt) unless harness.squishsum?
|
|
42
|
+
|
|
43
|
+
# The sample steps and the judge are resolved before any sample is requested, so a missing one fails
|
|
44
|
+
# before costing anything.
|
|
45
|
+
sample_steps, rest = sampling_plan(path, harness)
|
|
46
|
+
judge = harness.judged? && Judge.new(definition: @definition, receiver: @receiver, client: @client, harness:,
|
|
47
|
+
rest:, purpose:, payload: body, escalate: method(:escalate),
|
|
48
|
+
log_failure: method(:log_failure))
|
|
49
|
+
sample(sample_steps, prompt, harness, judge)
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
private
|
|
53
|
+
|
|
54
|
+
# Splits the escalation into runs of consecutive identical steps (a step's attempts). A squishsum samples
|
|
55
|
+
# the first run twice; an ensemble samples the first run, then the second. Returns each sample's steps and
|
|
56
|
+
# the rest of the escalation, which is where a judge looks for its default step.
|
|
57
|
+
def sampling_plan(path, harness)
|
|
58
|
+
runs = path.chunk_while { |step, following| step == following }.to_a
|
|
59
|
+
if harness.ensemble? && runs.size < 2
|
|
60
|
+
raise ConfigurationError, "#{@definition.label}: the #{harness.type} harness needs a second escalation " \
|
|
61
|
+
"step to sample (adjacent identical steps count as one step's attempts)"
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
return [runs.first(2), runs.drop(2).flatten(1)] if harness.ensemble?
|
|
65
|
+
|
|
66
|
+
[[runs.first, runs.first], runs.drop(1).flatten(1)]
|
|
67
|
+
end
|
|
68
|
+
|
|
69
|
+
# Makes each attempt of the path until an output passes the schema and the squish_validate check.
|
|
70
|
+
# Consecutive attempts on the same step continue the same conversation, so the model sees what it got wrong;
|
|
71
|
+
# a new step starts a fresh chat told about the last rejected output (unless the step sets
|
|
72
|
+
# forward_rejected: false).
|
|
73
|
+
def escalate(path, prompt)
|
|
28
74
|
chat = nil
|
|
29
75
|
rejected = nil # [raw content, errors] of the last invalid output
|
|
30
76
|
models = []
|
|
@@ -35,92 +81,126 @@ module Squishling
|
|
|
35
81
|
if chat && step == path[attempt - 2]
|
|
36
82
|
message = retry_message(rejected.last)
|
|
37
83
|
else
|
|
38
|
-
chat = start_chat(step)
|
|
39
|
-
message = rejected ? escalation_message(input, *rejected) : input
|
|
84
|
+
chat = start_chat(step, prompt)
|
|
85
|
+
message = rejected && step.forward_rejected ? escalation_message(prompt.input, *rejected) : prompt.input
|
|
40
86
|
end
|
|
41
87
|
|
|
42
88
|
begin
|
|
43
|
-
response = ask(chat, message, step)
|
|
89
|
+
response = @client.ask(chat, message, step)
|
|
44
90
|
rescue LLMError => e
|
|
91
|
+
squawk(prompt, path, attempt, nil, e)
|
|
45
92
|
raise if last
|
|
46
93
|
|
|
47
94
|
# The failed request may be half-recorded in this chat, so the next attempt starts a fresh one.
|
|
48
95
|
chat = nil
|
|
49
|
-
log_failure(attempt, path, e.message)
|
|
96
|
+
log_failure(prompt.role, attempt, path, e.message)
|
|
50
97
|
next
|
|
51
98
|
end
|
|
52
99
|
|
|
53
|
-
result, errors =
|
|
54
|
-
|
|
100
|
+
result, errors = @output_check.call(response.content, prompt)
|
|
101
|
+
if errors.empty?
|
|
102
|
+
squawk(prompt, path, attempt, response, nil)
|
|
103
|
+
return result
|
|
104
|
+
end
|
|
55
105
|
|
|
56
|
-
|
|
106
|
+
error = InvalidOutputError.new(errors:, raw: response.content, attempts: attempt, models: models.dup,
|
|
107
|
+
source: source(prompt.role))
|
|
108
|
+
squawk(prompt, path, attempt, response, error)
|
|
109
|
+
raise error if last
|
|
57
110
|
|
|
58
111
|
rejected = [response.content, errors]
|
|
59
|
-
log_failure(attempt, path, errors.join("; "))
|
|
112
|
+
log_failure(prompt.role, attempt, path, errors.join("; "))
|
|
60
113
|
end
|
|
61
114
|
end
|
|
62
115
|
|
|
63
|
-
|
|
116
|
+
# Two samples, each on its own steps and retried per their attempts the same way escalate retries.
|
|
117
|
+
# Each round sends the pending samples' requests concurrently; everything else runs on this thread.
|
|
118
|
+
def sample(sample_steps, prompt, harness, judge)
|
|
119
|
+
samples = %w[a b].zip(sample_steps).map { |name, steps| Sample.new(name, steps) }
|
|
64
120
|
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
chat.with_schema(@schema.llm_schema)
|
|
69
|
-
apply_params(chat, step.params)
|
|
70
|
-
chat
|
|
71
|
-
end
|
|
121
|
+
1.upto(sample_steps.map(&:size).max) do |attempt|
|
|
122
|
+
pending = samples.reject(&:result)
|
|
123
|
+
break if pending.empty?
|
|
72
124
|
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
125
|
+
jobs = pending.map do |sample|
|
|
126
|
+
step = sample.steps[attempt - 1]
|
|
127
|
+
message = sample_message(sample, step, prompt)
|
|
128
|
+
-> { @client.ask(sample.chat, message, step) }
|
|
129
|
+
end
|
|
130
|
+
outcomes = LLMClient.concurrently(jobs)
|
|
131
|
+
# A setup mistake (or anything that isn't a failed request) ends the call, whichever sample hit it.
|
|
132
|
+
fatal = outcomes.map(&:last).find { |error| error && !error.is_a?(LLMError) }
|
|
133
|
+
raise fatal if fatal
|
|
78
134
|
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
rescue ArgumentError => e
|
|
84
|
-
raise ConfigurationError, "#{@definition.label}: invalid params #{params.inspect} (#{e.message})"
|
|
135
|
+
pending.zip(outcomes).each { |sample, outcome| record(sample, outcome, prompt, attempt) }
|
|
136
|
+
end
|
|
137
|
+
|
|
138
|
+
settle(samples, harness, judge)
|
|
85
139
|
end
|
|
86
140
|
|
|
87
|
-
#
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
chat.ask(message)
|
|
94
|
-
rescue RubyLLM::ConfigurationError, RubyLLM::UnauthorizedError, RubyLLM::ForbiddenError => e
|
|
95
|
-
raise ConfigurationError, "#{@definition.label}: #{e.class}: #{e.message}"
|
|
96
|
-
rescue RubyLLM::BadRequestError => e
|
|
97
|
-
raise ConfigurationError,
|
|
98
|
-
"#{@definition.label}: the provider rejected the request (#{e.message})#{params_hint(step.params)}"
|
|
99
|
-
rescue RubyLLM::Error, Faraday::Error => e
|
|
100
|
-
raise LLMError, "#{@definition.label}: #{e.class}: #{e.message}"
|
|
141
|
+
# The next message for a sample: a retry in its chat after invalid output, otherwise a fresh chat.
|
|
142
|
+
def sample_message(sample, step, prompt)
|
|
143
|
+
return retry_message(sample.rejected.last) if sample.chat
|
|
144
|
+
|
|
145
|
+
sample.chat = start_chat(step, prompt)
|
|
146
|
+
sample.rejected && step.forward_rejected ? escalation_message(prompt.input, *sample.rejected) : prompt.input
|
|
101
147
|
end
|
|
102
148
|
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
149
|
+
def record(sample, (response, error), prompt, attempt)
|
|
150
|
+
steps = sample.steps
|
|
151
|
+
last = attempt == steps.size
|
|
152
|
+
role = "sample #{sample.name}"
|
|
153
|
+
if error
|
|
154
|
+
squawk(prompt, steps, attempt, nil, error)
|
|
155
|
+
raise error if last
|
|
156
|
+
|
|
157
|
+
sample.chat = nil
|
|
158
|
+
return log_failure(role, attempt, steps, error.message)
|
|
159
|
+
end
|
|
160
|
+
|
|
161
|
+
result, errors = @output_check.call(response.content, prompt)
|
|
162
|
+
if errors.empty?
|
|
163
|
+
squawk(prompt, steps, attempt, response, nil)
|
|
164
|
+
return sample.result = result
|
|
165
|
+
end
|
|
166
|
+
|
|
167
|
+
models = steps.first(attempt).map(&:model)
|
|
168
|
+
invalid = InvalidOutputError.new(errors:, raw: response.content, attempts: attempt, models:)
|
|
169
|
+
squawk(prompt, steps, attempt, response, invalid)
|
|
170
|
+
raise invalid if last
|
|
171
|
+
|
|
172
|
+
sample.rejected = [response.content, errors]
|
|
173
|
+
log_failure(role, attempt, steps, errors.join("; "))
|
|
111
174
|
end
|
|
112
175
|
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
176
|
+
# Agreeing samples are accepted; otherwise the judge, if any, picks one or the call fails.
|
|
177
|
+
def settle(samples, harness, judge)
|
|
178
|
+
first, second = samples.map(&:result)
|
|
179
|
+
return first if agree?(first, second, harness.compare)
|
|
180
|
+
|
|
181
|
+
models = samples.map { |sample| sample.steps.first.model }
|
|
182
|
+
unless judge
|
|
183
|
+
log_warning("samples disagreed")
|
|
184
|
+
raise DisagreementError.new(candidates: [first, second], models:)
|
|
185
|
+
end
|
|
186
|
+
|
|
187
|
+
log_warning("samples disagreed, asking the judge (#{judge.step.display_name})")
|
|
188
|
+
choice, reason, detail = judge.verdict(first, second)
|
|
189
|
+
return { a: first, b: second }.fetch(choice) unless choice == :neither
|
|
190
|
+
|
|
191
|
+
raise DisagreementError.new(candidates: [first, second], verdict: :neither, reason:, detail:,
|
|
192
|
+
models: models + [judge.step.model])
|
|
118
193
|
end
|
|
119
194
|
|
|
120
|
-
|
|
121
|
-
|
|
195
|
+
# compare: is the developer's own code, so its exceptions propagate unwrapped.
|
|
196
|
+
def agree?(first, second, compare)
|
|
197
|
+
return Schema.jsonify(first.to_h) == Schema.jsonify(second.to_h) unless compare
|
|
198
|
+
|
|
199
|
+
@receiver.instance_exec(first, second, **@inputs, &compare) ? true : false
|
|
200
|
+
end
|
|
122
201
|
|
|
123
|
-
|
|
202
|
+
def start_chat(step, prompt)
|
|
203
|
+
@client.chat(step, instructions: prompt.instructions, schema: prompt.schema.llm_schema)
|
|
124
204
|
end
|
|
125
205
|
|
|
126
206
|
# Only the arguments, the declared squish_context names, and a squish! call's context leave the process.
|
|
@@ -154,92 +234,67 @@ module Squishling
|
|
|
154
234
|
@receiver.instance_variable_get(:"@#{name}")
|
|
155
235
|
end
|
|
156
236
|
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
# ignored the schema.
|
|
160
|
-
def parse(content)
|
|
161
|
-
return [nil, ["response was empty"]] if content.nil? || (content.is_a?(String) && content.strip.empty?)
|
|
162
|
-
return [content, []] unless content.is_a?(String)
|
|
163
|
-
|
|
164
|
-
[JSON.parse(strip_code_fence(content)), []]
|
|
165
|
-
rescue JSON::ParserError => e
|
|
166
|
-
[nil, ["response was not valid JSON (#{e.message.lines.first&.strip})"]]
|
|
167
|
-
end
|
|
168
|
-
|
|
169
|
-
# Models without native structured output sometimes wrap JSON in a markdown code fence.
|
|
170
|
-
def strip_code_fence(text)
|
|
171
|
-
text[/\A\s*```(?:json)?\s*\n(.*?)\n\s*```\s*\z/m, 1] || text
|
|
172
|
-
end
|
|
173
|
-
|
|
174
|
-
# Parses and validates a response: [typed result, []] when it passes, [nil or result, errors] otherwise.
|
|
175
|
-
def check(content)
|
|
176
|
-
data, errors = parse(content)
|
|
177
|
-
errors = @schema.validate(data) if errors.empty?
|
|
178
|
-
return [nil, errors] if errors.any?
|
|
179
|
-
|
|
180
|
-
result = @schema.build(data, squished: true)
|
|
181
|
-
[result, validator_errors(result)]
|
|
237
|
+
def retry_message(errors)
|
|
238
|
+
"Your previous response was rejected:\n- #{errors.join("\n- ")}\nRespond again with corrected JSON only."
|
|
182
239
|
end
|
|
183
240
|
|
|
184
|
-
#
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
when false then ["the output was rejected by squish_validate"]
|
|
193
|
-
when String, Array then Array(value).flatten.compact.map(&:to_s).reject { |message| message.strip.empty? }
|
|
194
|
-
else
|
|
195
|
-
return contract_errors(value) if value.respond_to?(:success?) && value.respond_to?(:errors)
|
|
196
|
-
|
|
197
|
-
raise ConfigurationError, "#{@definition.label}: squish_validate must return nil, true, false, a String, " \
|
|
198
|
-
"an Array of Strings, or a validation result, got #{value.class}"
|
|
241
|
+
# The first message to a fresh chat after an earlier model's output was rejected. The rejected output is
|
|
242
|
+
# model-generated and may cross providers, so it is capped at MAX_FORWARDED_CHARS.
|
|
243
|
+
def escalation_message(input, raw, errors)
|
|
244
|
+
previous = rejected_text(raw)
|
|
245
|
+
previous = "(an empty response)" if previous.strip.empty? || raw.nil?
|
|
246
|
+
if previous.length > MAX_FORWARDED_CHARS
|
|
247
|
+
omitted = previous.length - MAX_FORWARDED_CHARS
|
|
248
|
+
previous = "#{previous[0, MAX_FORWARDED_CHARS]}... [truncated, #{omitted} more characters]"
|
|
199
249
|
end
|
|
250
|
+
"#{input}\n\nA previous attempt at this request returned:\n#{previous}\n" \
|
|
251
|
+
"It was rejected:\n- #{errors.join("\n- ")}\nRespond with corrected JSON only."
|
|
200
252
|
end
|
|
201
253
|
|
|
202
|
-
#
|
|
203
|
-
|
|
204
|
-
|
|
254
|
+
# The rejected output as text. Output that was rejected for being unrepresentable in JSON (Infinity, invalid
|
|
255
|
+
# UTF-8) can't be serialized or measured as is, so it is scrubbed or replaced rather than raising.
|
|
256
|
+
def rejected_text(raw)
|
|
257
|
+
return utf8(raw).scrub("?") if raw.is_a?(String)
|
|
205
258
|
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
messages.empty? ? ["the output was rejected by squish_validate"] : messages
|
|
259
|
+
JSON.generate(raw)
|
|
260
|
+
rescue JSON::JSONError
|
|
261
|
+
"(a response JSON can't represent)"
|
|
210
262
|
end
|
|
211
263
|
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
next flatten_messages(value, path) if value.is_a?(Hash)
|
|
216
|
-
|
|
217
|
-
Array(value).map { |message| path.empty? ? message.to_s : "#{path} #{message}" }
|
|
218
|
-
end
|
|
264
|
+
# A binary-tagged string counts every byte as valid, so it is read as UTF-8 before scrubbing.
|
|
265
|
+
def utf8(text)
|
|
266
|
+
text.encoding == Encoding::BINARY ? text.dup.force_encoding(Encoding::UTF_8) : text
|
|
219
267
|
end
|
|
220
268
|
|
|
221
|
-
|
|
222
|
-
|
|
269
|
+
# Runs the observability hook, if any, with this attempt's raw output (nil when the call itself failed),
|
|
270
|
+
# the error that ended it (nil when it was accepted), and what was asked.
|
|
271
|
+
def squawk(prompt, path, attempt, response, error)
|
|
272
|
+
hook = @definition.squawk
|
|
273
|
+
return unless hook
|
|
274
|
+
|
|
275
|
+
step = path[attempt - 1]
|
|
276
|
+
metadata = {
|
|
277
|
+
label: @definition.label, attempt:, attempts: path.size, final: attempt == path.size,
|
|
278
|
+
model: response&.model || step.model, provider: step.provider, params: step.params, input: prompt.input,
|
|
279
|
+
usage: response&.tokens&.to_h
|
|
280
|
+
}
|
|
281
|
+
Squawk.call(hook, output: response&.content, metadata:, error:)
|
|
223
282
|
end
|
|
224
283
|
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
previous = raw.is_a?(String) ? raw : JSON.generate(raw)
|
|
228
|
-
previous = "(an empty response)" if previous.strip.empty? || raw.nil?
|
|
229
|
-
"#{input}\n\nA previous attempt at this request returned:\n#{previous}\n" \
|
|
230
|
-
"It was rejected:\n- #{errors.join("\n- ")}\nRespond with corrected JSON only."
|
|
284
|
+
def source(role)
|
|
285
|
+
role == :judge ? "Judge" : "LLM"
|
|
231
286
|
end
|
|
232
287
|
|
|
233
|
-
def log_failure(attempt, path, reason)
|
|
288
|
+
def log_failure(role, attempt, path, reason)
|
|
234
289
|
step = path[attempt - 1]
|
|
235
290
|
next_step = path[attempt]
|
|
236
|
-
action = next_step == step ? "retrying" : "escalating to #{
|
|
237
|
-
|
|
238
|
-
|
|
291
|
+
action = next_step == step ? "retrying" : "escalating to #{next_step.display_name}"
|
|
292
|
+
who = role ? "#{role} attempt" : "attempt"
|
|
293
|
+
log_warning("#{who} #{attempt} of #{path.size} (#{step.display_name}) failed, #{action}: #{reason}")
|
|
239
294
|
end
|
|
240
295
|
|
|
241
|
-
def
|
|
242
|
-
|
|
296
|
+
def log_warning(message)
|
|
297
|
+
Squishling.config.logger&.warn("[Squishling] #{@definition.label} #{message}")
|
|
243
298
|
end
|
|
244
299
|
end
|
|
245
300
|
end
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Squishling
|
|
4
|
+
# Picks between a judged harness's two disagreeing samples (judged_squishsum or judged_ensemble), or rejects
|
|
5
|
+
# both. The judge is the harness's judge: step, or else the escalation's next step after the ones the samples
|
|
6
|
+
# ran on. A chat judge answers with a strict verdict schema; a :judgment judge (a System One decision model such
|
|
7
|
+
# as Jev) answers one choice question through RubyLLM.judge, and its pick counts only at or above
|
|
8
|
+
# min_confidence.
|
|
9
|
+
class Judge
|
|
10
|
+
VERDICT_SCHEMA = {
|
|
11
|
+
"type" => "object",
|
|
12
|
+
"properties" => {
|
|
13
|
+
"verdict" => { "type" => "string", "enum" => %w[a b neither] },
|
|
14
|
+
"reason" => { "type" => "string" }
|
|
15
|
+
},
|
|
16
|
+
"required" => %w[verdict reason],
|
|
17
|
+
"additionalProperties" => false
|
|
18
|
+
}.freeze
|
|
19
|
+
|
|
20
|
+
INPUT_NOTE = <<~NOTE
|
|
21
|
+
The input is a JSON object. "input" is what the operation received ("arguments" and, when present,
|
|
22
|
+
"context"); "output_schema" is the format both candidates follow; "candidates" holds "a" and "b".
|
|
23
|
+
Respond only with JSON matching the required schema: your verdict ("a", "b", or "neither") and a short reason.
|
|
24
|
+
NOTE
|
|
25
|
+
|
|
26
|
+
CHOICES = {
|
|
27
|
+
a: "Candidate a correctly carries out the purpose for this input (also when both do)",
|
|
28
|
+
b: "Candidate b correctly carries out the purpose for this input",
|
|
29
|
+
neither: "Neither candidate is clearly correct"
|
|
30
|
+
}.freeze
|
|
31
|
+
|
|
32
|
+
def initialize(definition:, receiver:, client:, harness:, rest:, purpose:, payload:, escalate:, log_failure:)
|
|
33
|
+
@definition = definition
|
|
34
|
+
@receiver = receiver
|
|
35
|
+
@client = client
|
|
36
|
+
@harness = harness
|
|
37
|
+
@purpose = purpose
|
|
38
|
+
@payload = payload
|
|
39
|
+
@escalate = escalate
|
|
40
|
+
@log_failure = log_failure
|
|
41
|
+
@steps = resolve_steps(rest)
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
# The judge's first attempt, for logs and DisagreementError#models.
|
|
45
|
+
def step
|
|
46
|
+
@steps.first
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
# [:a | :b, reason, detail] for the chosen candidate, or [:neither, reason, detail]. The detail is text
|
|
50
|
+
# Squishling wrote (nil for a chat judge, whose reason is model-written) and is safe for error messages.
|
|
51
|
+
def verdict(first, second)
|
|
52
|
+
candidates = { a: Schema.jsonify(first.to_h), b: Schema.jsonify(second.to_h) }
|
|
53
|
+
judgment? ? judgment(candidates) : chat(candidates)
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
private
|
|
57
|
+
|
|
58
|
+
def judgment?
|
|
59
|
+
@harness.judge&.fetch(:type) == :judgment
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
# A declared judge brings its own model and provider. A chat judge gets the method's generation params
|
|
63
|
+
# under its own, like any escalation step; a judgment judge sends only its own params, as provider options.
|
|
64
|
+
def resolve_steps(rest)
|
|
65
|
+
if (judge = @harness.judge)
|
|
66
|
+
params = judge[:type] == :judgment ? {} : @definition.params
|
|
67
|
+
return ModelPath.steps([judge], provider: nil, params:)
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
if rest.empty?
|
|
71
|
+
ordinal = @harness.ensemble? ? "third" : "second"
|
|
72
|
+
raise ConfigurationError, "#{@definition.label}: the #{@harness.type} harness needs a judge: or a " \
|
|
73
|
+
"#{ordinal} escalation step to judge with"
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
rest.take_while { |step| step == rest.first }
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
def chat(candidates)
|
|
80
|
+
prompt = Invoker::Prompt.new(
|
|
81
|
+
instructions: "#{judge_instructions}\n\nThe operation's purpose:\n<purpose>\n#{@purpose}\n" \
|
|
82
|
+
"</purpose>\n\n#{INPUT_NOTE}",
|
|
83
|
+
schema: Schema.for(VERDICT_SCHEMA), input: JSON.generate(state(candidates)), role: :judge
|
|
84
|
+
)
|
|
85
|
+
result = @escalate.call(@steps, prompt)
|
|
86
|
+
[result[:verdict].to_sym, result[:reason], nil]
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
def judgment(candidates)
|
|
90
|
+
input = { purpose: @purpose, **state(candidates) }
|
|
91
|
+
questions = { winner: { type: :choice, instructions: judge_instructions, options: CHOICES } }
|
|
92
|
+
answer = judge_with_retries(input, questions)[:winner]
|
|
93
|
+
unless answer.respond_to?(:choice) && answer.respond_to?(:probabilities)
|
|
94
|
+
raise InvalidOutputError.new(errors: ["the judgment didn't answer the winner question"], source: "Judge")
|
|
95
|
+
end
|
|
96
|
+
|
|
97
|
+
interpret(answer.choice.to_s.to_sym, answer.probabilities)
|
|
98
|
+
end
|
|
99
|
+
|
|
100
|
+
# The threshold is checked on the exact probability; it's rounded only for the reason text.
|
|
101
|
+
def interpret(choice, probabilities)
|
|
102
|
+
probability = probabilities.to_h.find { |key, _| key.to_s == choice.to_s }&.last.to_f
|
|
103
|
+
minimum = @harness.judge[:min_confidence]
|
|
104
|
+
shown = probability.round(3)
|
|
105
|
+
picked = %i[a b].include?(choice)
|
|
106
|
+
if picked && probability >= minimum
|
|
107
|
+
text = "chosen with probability #{shown}"
|
|
108
|
+
return [choice, text, text]
|
|
109
|
+
end
|
|
110
|
+
|
|
111
|
+
text =
|
|
112
|
+
if choice == :neither then "the judge chose neither (probability #{shown})"
|
|
113
|
+
elsif picked then "the judge chose #{choice} with probability #{shown}, below min_confidence #{minimum}"
|
|
114
|
+
else "the judge returned an unrecognized choice (probability #{shown})"
|
|
115
|
+
end
|
|
116
|
+
[:neither, text, text]
|
|
117
|
+
end
|
|
118
|
+
|
|
119
|
+
# A judgment always matches its questions, so only a failed request (LLMError) moves on to the next attempt.
|
|
120
|
+
def judge_with_retries(input, questions)
|
|
121
|
+
@steps.each.with_index(1) do |step, attempt|
|
|
122
|
+
return @client.judge(input, questions:, step:)
|
|
123
|
+
rescue LLMError => e
|
|
124
|
+
raise if attempt == @steps.size
|
|
125
|
+
|
|
126
|
+
@log_failure.call(:judge, attempt, @steps, e.message)
|
|
127
|
+
end
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
def state(candidates)
|
|
131
|
+
{ input: @payload, output_schema: @definition.schema.llm_schema["schema"], candidates: }
|
|
132
|
+
end
|
|
133
|
+
|
|
134
|
+
def judge_instructions
|
|
135
|
+
value = @harness.judge_instructions || Harness::DEFAULT_JUDGE_INSTRUCTIONS
|
|
136
|
+
value = @receiver.instance_exec(&value) if value.is_a?(Proc)
|
|
137
|
+
return value if value.is_a?(String) && !value.strip.empty?
|
|
138
|
+
|
|
139
|
+
raise ConfigurationError, "#{@definition.label}: the judge_instructions proc must return a non-blank String"
|
|
140
|
+
end
|
|
141
|
+
end
|
|
142
|
+
end
|