ruby_llm-llm_judge 0.1.0 → 0.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +4 -0
- data/lib/ruby_llm/llm_judge/engine.rb +8 -8
- data/lib/ruby_llm/llm_judge/version.rb +1 -1
- data/lib/ruby_llm/llm_judge.rb +12 -0
- metadata +1 -1
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 451771cadd6d75e7ea6361b43814d186bc0b9e3aeb0dea5f128fdc3c4a68e982
|
|
4
|
+
data.tar.gz: ca97b903a28238275417a96551abbf440198eb2dea95d52196cfb2e5d1b0d173
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: ff30bd52e7b0dcb79f8804458e622d725ad3570351e00d180efb9ed040d4bb07c8551c37024f1730ca876125f52513e0e28a557fb9ed8f9d22211b286266aa6d
|
|
7
|
+
data.tar.gz: 25259a68bba8ffc259d343f076b7962788f236143ca8dabe43c03e626181f9bac24bfb50a98b50951130478be94951c56944775072614732c1341aa3218626eb
|
data/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,9 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.1.1 (2026-09-28)
|
|
4
|
+
|
|
5
|
+
- Fix invalid scoring responses raising `NoMethodError` on RubyLLM 1.13, which also skipped the corrective retry. They now raise `RubyLLM::LLMJudge::Error`, a `RubyLLM::Error`, on every supported RubyLLM version.
|
|
6
|
+
|
|
3
7
|
## 0.1.0 (2026-09-28)
|
|
4
8
|
|
|
5
9
|
Initial release.
|
|
@@ -74,7 +74,7 @@ module RubyLLM
|
|
|
74
74
|
chosen = judgment.answers.fetch(question.name)
|
|
75
75
|
probabilities = chosen.probabilities.values
|
|
76
76
|
if probabilities.count(probabilities.max) > 1
|
|
77
|
-
raise
|
|
77
|
+
raise Error, "Scoring model could not resolve the tie for #{question.name}"
|
|
78
78
|
end
|
|
79
79
|
tie_breaks << { question: question.name, ratings:, judgment: }
|
|
80
80
|
end
|
|
@@ -128,13 +128,13 @@ module RubyLLM
|
|
|
128
128
|
|
|
129
129
|
def parse_single_request(content, specs)
|
|
130
130
|
parsed = JSON.parse(content)
|
|
131
|
-
raise
|
|
131
|
+
raise Error, 'Scoring model returned a non-object response' unless parsed.is_a?(Hash)
|
|
132
132
|
|
|
133
133
|
reported = parsed.fetch('answers')
|
|
134
134
|
expected_questions = specs.map { |question, _| question.name.to_s }
|
|
135
135
|
unless reported.is_a?(Hash) && reported.keys.sort == expected_questions.sort
|
|
136
136
|
actual = reported.is_a?(Hash) ? reported.keys : reported.class.name
|
|
137
|
-
raise
|
|
137
|
+
raise Error, "Question IDs must be #{expected_questions.inspect}; got #{actual.inspect}"
|
|
138
138
|
end
|
|
139
139
|
|
|
140
140
|
normalization = {}
|
|
@@ -143,23 +143,23 @@ module RubyLLM
|
|
|
143
143
|
keys = options.map { |name, _| name.to_s }
|
|
144
144
|
unless distribution.is_a?(Hash) && distribution.keys.sort == keys.sort
|
|
145
145
|
actual = distribution.is_a?(Hash) ? distribution.keys : distribution.class.name
|
|
146
|
-
raise
|
|
146
|
+
raise Error, "Option IDs for #{question.name} must be #{keys.inspect}; got #{actual.inspect}"
|
|
147
147
|
end
|
|
148
148
|
|
|
149
149
|
values = keys.map { |key| distribution.fetch(key) }
|
|
150
150
|
unless values.all? { |value| value.is_a?(Numeric) && value.finite? && (0..1).cover?(value) }
|
|
151
|
-
raise
|
|
151
|
+
raise Error, "Scoring model returned invalid probabilities for #{question.name}"
|
|
152
152
|
end
|
|
153
153
|
|
|
154
154
|
total = values.sum.to_f
|
|
155
|
-
raise
|
|
155
|
+
raise Error, "Scoring model returned zero probability for #{question.name}" unless total.positive?
|
|
156
156
|
|
|
157
157
|
normalization[question.name.to_s] = total
|
|
158
158
|
[question.name, answer(question, values.map { |value| value / total })]
|
|
159
159
|
end
|
|
160
160
|
[answers, reported, normalization]
|
|
161
161
|
rescue JSON::ParserError, KeyError, TypeError => error
|
|
162
|
-
raise
|
|
162
|
+
raise Error, "Scoring model returned invalid JSON probabilities: #{error.message}"
|
|
163
163
|
end
|
|
164
164
|
|
|
165
165
|
def single_request_prompt(input, specs)
|
|
@@ -268,7 +268,7 @@ module RubyLLM
|
|
|
268
268
|
if /\A[0-9]\z/.match?(digit)
|
|
269
269
|
return { digit: digit.to_i, tokens: aggregate_tokens(attempts.map(&:tokens)) }
|
|
270
270
|
end
|
|
271
|
-
raise
|
|
271
|
+
raise Error, 'Scoring model returned no single digit' if attempts.size > @malformed_retries
|
|
272
272
|
end
|
|
273
273
|
end
|
|
274
274
|
|
data/lib/ruby_llm/llm_judge.rb
CHANGED
|
@@ -10,6 +10,18 @@ module RubyLLM
|
|
|
10
10
|
# RubyLLM 2.1 added the Judge API. Earlier versions run the engine directly
|
|
11
11
|
# and return Legacy stand-ins with the same readers.
|
|
12
12
|
NATIVE = defined?(RubyLLM::Judge) ? true : false
|
|
13
|
+
|
|
14
|
+
# Raised for invalid scoring responses. RubyLLM::Error takes (response, message)
|
|
15
|
+
# before 1.16 and (message, response:) on main, so build it for either.
|
|
16
|
+
class Error < RubyLLM::Error
|
|
17
|
+
def initialize(message = nil)
|
|
18
|
+
if RubyLLM::Error.instance_method(:initialize).parameters.include?(%i[key response])
|
|
19
|
+
super(message)
|
|
20
|
+
else
|
|
21
|
+
super(nil, message)
|
|
22
|
+
end
|
|
23
|
+
end
|
|
24
|
+
end
|
|
13
25
|
end
|
|
14
26
|
end
|
|
15
27
|
|