aireview 2.0.0 → 2.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,387 @@
1
+ # frozen_string_literal: true
2
+ require 'json'
3
+ require 'logger'
4
+ require 'yaml'
5
+ require_relative 'utils'
6
+ require_relative 'candidate_checker'
7
+ require_relative 'jev_client'
8
+
9
+ module Aireview
10
+ # Jev as a critic. Every candidate becomes a few questions
11
+ # (prompts/jev_questions.yml) over one state: the MR and Jira sections, the
12
+ # candidates with the hunks they point at, the diff. The answers turn into
13
+ # keep, reject or unverifiable by thresholds, then duplicates among the
14
+ # kept ones are dropped. Jev cannot rewrite a finding, so there is no
15
+ # refinement. Used as the critique engine (JevStage) and in shadow mode.
16
+ class JevCritic
17
+ QUESTIONS = YAML.safe_load_file(File.expand_path('prompts/jev_questions.yml', __dir__)).freeze
18
+ # The questions whose answers decide; severity only goes to the log. Their
19
+ # templates go into the review key whole, not as asked about a stub
20
+ # candidate: a single stub gets no duplicate_of question.
21
+ DECISION_QUESTIONS = %w[real_issue enough_context version_claim duplicate_of].freeze
22
+ # Jev limits (docs.typesafe.ai/models): the state plus the longest
23
+ # question, and the state plus all questions together. Questions count
24
+ # in both, so the state budget is what they leave.
25
+ STATE_WITH_LONGEST_QUESTION_TOKENS = 32_000
26
+ REQUEST_TOKENS = 64_000
27
+ # There is no Jev tokenizer here: characters are converted pessimistically
28
+ # (code and Cyrillic take more tokens than English prose) with a margin.
29
+ # Every request logs the real input_tokens to compare against.
30
+ CHARS_PER_TOKEN = 2.5
31
+ SAFETY = 0.9
32
+ CANDIDATE_FIELDS = %w[id file line quoted_code category severity problem why suggestion note].freeze
33
+ TRUNCATED = '[truncated to fit the Jev request]'
34
+ NOT_SHOWN = '(the file is not shown in the diff)'
35
+ TASK = 'Second pass of a merge request review: check the candidate findings of the first pass ' \
36
+ 'against the diff and the requirements.'
37
+ SEVERITY_RANK = {'critical' => 0, 'major' => 1, 'minor' => 2}.freeze
38
+ # The docs only say that a 422 body names the offending field; a size
39
+ # rejection is recognized by an explicit phrase about going over a limit.
40
+ # Bare words like "context" are not enough: they occur in question names
41
+ # (enough_context_C1). Any other 422 is a malformed request, and asking
42
+ # again would not help.
43
+ SIZE_ERROR = /
44
+ \bexceed(?:s|ed|ing)?\b |
45
+ \btoo\s+(?:long|large|many\s+tokens)\b |
46
+ \bmaximum\s+(?:context|length|tokens?)\b |
47
+ \b(?:token|context|length)\s+limit\b
48
+ /ix
49
+
50
+ # decision — :keep, :reject or :unverifiable; answers — the numbers
51
+ # behind it, for the log.
52
+ Assessment = Struct.new(:id, :decision, :reason, :answers, keyword_init: true) do
53
+ # Rounded for reading; the exact numbers stay in answers.
54
+ def numbers
55
+ answers.map do |name, value|
56
+ value.is_a?(Hash) ? "#{name}=#{value[:choice]}/#{round(value[:confidence])}" : "#{name}=#{round(value)}"
57
+ end.join(' ')
58
+ end
59
+
60
+ private
61
+
62
+ def round(value)
63
+ value.is_a?(Numeric) ? value.round(2) : value
64
+ end
65
+ end
66
+ # answers — every answer by question key, for dropping duplicates once
67
+ # the verdicts of Jev and of the LLM are merged.
68
+ Result = Struct.new(:assessments, :answers, :requests, :model, keyword_init: true)
69
+ Request = Struct.new(:candidates, :state, :questions, keyword_init: true)
70
+ # What the requests of one assessment gathered.
71
+ Asked = Struct.new(:answers, :unfit, :requests, :model, keyword_init: true)
72
+
73
+ # scrub — the secret scrubber of the context: the candidates are LLM text
74
+ # and are scrubbed like the candidates JSON of the Critique prompt.
75
+ def initialize(client:, thresholds:, review_instructions: nil, scrub: ->(text) { text },
76
+ logger: Logger.new($stderr))
77
+ @client = client
78
+ @thresholds = thresholds
79
+ @scrub = scrub
80
+ instructions = Aireview::Utils.presence(review_instructions.to_s.strip)
81
+ @review_instructions = instructions && scrub.call(instructions)
82
+ @logger = logger
83
+ end
84
+
85
+ # Kept candidates in generate order: an edge comes from a
86
+ # duplicate_of_<id> answer that names another kept id with enough
87
+ # confidence. In a group of duplicates the most severe one stays, on a
88
+ # tie the one generate listed first. Returns {dropped id => kept id}.
89
+ # The client and the critic as the config sets them up; scrub — the
90
+ # secret scrubber of the context.
91
+ def self.build(config:, scrub:, logger:)
92
+ new(client: JevClient.new(config: config, logger: logger), thresholds: config.jev_thresholds,
93
+ review_instructions: config.review_instructions, scrub: scrub, logger: logger)
94
+ end
95
+
96
+ def self.decision_templates
97
+ QUESTIONS.slice(*DECISION_QUESTIONS)
98
+ end
99
+
100
+ def self.duplicates(kept, answers, threshold:)
101
+ ids = kept.map { |candidate| field(candidate, 'id') }
102
+ rank = kept.to_h do |candidate|
103
+ [field(candidate, 'id'), SEVERITY_RANK.fetch(field(candidate, 'severity'), SEVERITY_RANK.size)]
104
+ end
105
+ duplicate_groups(ids, answers, threshold).each_with_object({}) do |group, dropped|
106
+ winner = group.min_by { |id| [rank[id], ids.index(id)] }
107
+ (group - [winner]).each { |id| dropped[id] = winner }
108
+ end
109
+ end
110
+
111
+ def self.duplicate_groups(ids, answers, threshold)
112
+ group_of = ids.to_h { |id| [id, [id]] }
113
+ ids.each do |id|
114
+ other = duplicate_answer(id, ids, answers, threshold)
115
+ join_groups(group_of, id, other) if other
116
+ end
117
+ group_of.values.uniq(&:object_id).select { |group| group.size > 1 }
118
+ end
119
+
120
+ def self.join_groups(group_of, id, other)
121
+ return if group_of[id].equal?(group_of[other])
122
+
123
+ merged = group_of[id] + group_of[other]
124
+ merged.each { |member| group_of[member] = merged }
125
+ end
126
+
127
+ # The other kept id a duplicate_of answer names with enough confidence.
128
+ def self.duplicate_answer(id, ids, answers, threshold)
129
+ answer = answers["duplicate_of_#{id}"]
130
+ return nil unless answer.is_a?(Hash) && answer['confidence'].to_f >= threshold
131
+
132
+ other = answer['choice']
133
+ other if other != id && ids.include?(other)
134
+ end
135
+
136
+ def self.field(candidate, key)
137
+ (candidate[key] || candidate[key.to_sym]).to_s.strip
138
+ end
139
+
140
+ # dedupe — false when the caller merges the verdicts with the LLM ones
141
+ # first and drops duplicates over the whole set (JevStage).
142
+ def assess(context:, candidates:, dedupe: true)
143
+ candidates = candidates.map { |candidate| normalize(candidate) }
144
+ sections = diff_sections(context.diff_text)
145
+ asked = Asked.new(answers: {}, unfit: [], requests: 0, model: nil)
146
+ plan_requests(context, candidates, sections).each { |request| ask(request, asked, context, sections) }
147
+ assessments = candidates.map { |candidate| decide(candidate, asked.answers, asked.unfit) }
148
+ assessments = drop_duplicates(assessments, candidates, asked.answers) if dedupe
149
+ Result.new(assessments: assessments, answers: asked.answers, requests: asked.requests, model: asked.model)
150
+ end
151
+
152
+ # The first request as it would be sent, for --dry-run.
153
+ def preview(context:, candidates:)
154
+ candidates = candidates.map { |candidate| normalize(candidate) }
155
+ plan_requests(context, candidates, diff_sections(context.diff_text)).first
156
+ end
157
+
158
+ private
159
+
160
+ def normalize(candidate)
161
+ CANDIDATE_FIELDS.to_h do |field|
162
+ value = candidate[field] || candidate[field.to_sym]
163
+ [field, value.is_a?(String) ? @scrub.call(value) : value]
164
+ end.compact.merge('id' => self.class.field(candidate, 'id'))
165
+ end
166
+
167
+ # One request for all candidates when it fits, otherwise one per
168
+ # candidate (then without duplicate_of: a request sees one candidate).
169
+ # A candidate that does not fit even alone is unverifiable.
170
+ def plan_requests(context, candidates, sections)
171
+ shared = build_request(context, candidates, sections)
172
+ return [shared] if shared.state || candidates.size == 1
173
+
174
+ @logger.info('Jev: the candidates do not fit one request, asking about each separately')
175
+ candidates.map { |candidate| build_request(context, [candidate], sections) }
176
+ end
177
+
178
+ def build_request(context, candidates, sections)
179
+ ids = candidates.map { |candidate| candidate['id'] }
180
+ questions = candidates.map { |candidate| questions_for(candidate['id'], ids) }.reduce({}, :merge)
181
+ state = {
182
+ 'task' => TASK,
183
+ 'requirements' => context.sections,
184
+ 'project_instructions' => @review_instructions,
185
+ 'context_truncated' => coverage_notes(context.coverage),
186
+ 'candidates' => candidates.to_h { |candidate| [candidate['id'], with_evidence(candidate, sections)] },
187
+ 'diff' => context.diff_text
188
+ }.compact
189
+ Request.new(candidates: candidates, state: fit(state, state_budget(questions)), questions: questions)
190
+ end
191
+
192
+ def questions_for(id, ids)
193
+ questions = %w[real_issue enough_context version_claim severity].to_h do |name|
194
+ ["#{name}_#{id}", question(QUESTIONS.fetch(name), id)]
195
+ end
196
+ others = ids - [id]
197
+ questions["duplicate_of_#{id}"] = duplicate_question(id, others) unless others.empty?
198
+ questions
199
+ end
200
+
201
+ def question(template, id)
202
+ {
203
+ 'type' => template.fetch('type'),
204
+ 'instructions' => format(template.fetch('instructions'), id: id),
205
+ 'criteria' => template.fetch('criteria').transform_values { |text| format(text, id: id) }
206
+ }
207
+ end
208
+
209
+ def duplicate_question(id, others)
210
+ template = QUESTIONS.fetch('duplicate_of')
211
+ criteria = others.to_h { |other| [other, format(template.fetch('other'), other: other)] }
212
+ {
213
+ 'type' => template.fetch('type'),
214
+ 'instructions' => format(template.fetch('instructions'), id: id),
215
+ 'criteria' => criteria.merge('none' => template.fetch('none'))
216
+ }
217
+ end
218
+
219
+ def state_budget(questions)
220
+ sizes = questions.values.map { |question| tokens(JSON.generate(question).length) }
221
+ room = [STATE_WITH_LONGEST_QUESTION_TOKENS - sizes.max, REQUEST_TOKENS - sizes.sum].min
222
+ (room * SAFETY * CHARS_PER_TOKEN).floor
223
+ end
224
+
225
+ def tokens(chars)
226
+ (chars / CHARS_PER_TOKEN).ceil
227
+ end
228
+
229
+ # The diff goes first, then the MR and Jira sections; the candidates and
230
+ # their evidence are never cut. nil — does not fit even then.
231
+ def fit(state, budget)
232
+ %w[diff requirements].each do |key|
233
+ break if size(state) <= budget
234
+
235
+ state = shrink(state, key, budget) if state[key]
236
+ end
237
+ size(state) <= budget ? state : nil
238
+ end
239
+
240
+ # Every character of the text takes at least one character of JSON, so
241
+ # cutting the overflow in characters is enough; the marker costs its
242
+ # length plus the escaped newline.
243
+ def shrink(state, key, budget)
244
+ text = state[key].to_s
245
+ keep = text.length - (size(state) - budget) - TRUNCATED.length - 2
246
+ state.merge(key => keep.positive? ? "#{text[0, keep]}\n#{TRUNCATED}" : TRUNCATED)
247
+ end
248
+
249
+ def size(state)
250
+ JSON.generate(state).length
251
+ end
252
+
253
+ def coverage_notes(coverage)
254
+ return nil if coverage.nil? || coverage.complete?
255
+
256
+ notes = coverage.truncated_sections.map { |label| "#{label} truncated" }
257
+ notes << "files not shown: #{coverage.files_not_shown.join(', ')}" unless coverage.files_not_shown.empty?
258
+ notes += coverage.files_partial.map { |file| "#{file[:path]}: #{file[:shown]} of #{file[:total]} hunks shown" }
259
+ notes << "diff not available: #{coverage.files_unavailable.join(', ')}" unless coverage.files_unavailable.empty?
260
+ notes
261
+ end
262
+
263
+ # The hunk the line falls into; the whole shown diff of the file when
264
+ # the line is unknown or the anchoring check left a note.
265
+ def with_evidence(candidate, sections)
266
+ section = sections[candidate['file'].to_s] || sections[candidate['file'].to_s.sub(%r{\A(?:\./|[ab]/)}, '')]
267
+ evidence = if section.nil?
268
+ NOT_SHOWN
269
+ elsif candidate['line'].is_a?(Integer) && !candidate['note']
270
+ hunk = section[:hunks].find { |item| item[:range].cover?(candidate['line']) }
271
+ hunk ? hunk[:text] : section[:text]
272
+ else
273
+ section[:text]
274
+ end
275
+ candidate.merge('evidence' => evidence)
276
+ end
277
+
278
+ # The shown diff by file: its whole text and every hunk with the range of
279
+ # new-file lines it covers (the headers as CandidateChecker reads them).
280
+ def diff_sections(diff_text)
281
+ sections = {}
282
+ section = nil
283
+ diff_text.to_s.each_line do |line|
284
+ if (header = line.match(CandidateChecker::FILE_HEADER))
285
+ section = {text: +'', hunks: []}
286
+ header.captures.map(&:strip).each { |path| sections[path] = section }
287
+ elsif section && (hunk_header = line.match(CandidateChecker::HUNK_HEADER))
288
+ section[:hunks] << new_hunk(hunk_header)
289
+ end
290
+ add_line(section, line) if section
291
+ end
292
+ sections
293
+ end
294
+
295
+ def new_hunk(header)
296
+ start = header[1].to_i
297
+ length = header[2] ? header[2].to_i : 1
298
+ {range: start..(start + length - 1), text: +''}
299
+ end
300
+
301
+ def add_line(section, line)
302
+ section[:text] << line
303
+ section[:hunks].last[:text] << line unless section[:hunks].empty?
304
+ end
305
+
306
+ # The estimate goes to the log next to the input_tokens Jev reports, to
307
+ # check CHARS_PER_TOKEN against real requests. When Jev rejects a request
308
+ # as too large after all (the estimate missed), its candidates are asked
309
+ # about one by one; one that is too large alone is unverifiable.
310
+ def ask(request, asked, context, sections)
311
+ ids = request.candidates.map { |candidate| candidate['id'] }
312
+ return asked.unfit.concat(ids) unless request.state
313
+
314
+ chars = size(request.state) + request.questions.values.sum { |question| JSON.generate(question).length }
315
+ @logger.info("Jev request estimated at ~#{tokens(chars)} tokens (#{chars} chars of state and questions)")
316
+ asked.requests += 1
317
+ result = @client.evaluate(state: request.state, questions: request.questions)
318
+ asked.answers.merge!(result.answers)
319
+ asked.model = result.model
320
+ rescue JevError => e
321
+ raise unless e.status == 422 && e.message.match?(SIZE_ERROR)
322
+
323
+ ask_separately(request, asked, context, sections, e)
324
+ end
325
+
326
+ def ask_separately(request, asked, context, sections, error)
327
+ ids = request.candidates.map { |candidate| candidate['id'] }
328
+ if ids.size == 1
329
+ @logger.warn("Jev rejected the request about #{ids.first} as too large, " \
330
+ "it stays unverifiable: #{error.message}")
331
+ return asked.unfit.concat(ids)
332
+ end
333
+
334
+ @logger.warn("Jev rejected the request as too large, asking about each candidate separately: #{error.message}")
335
+ request.candidates.each do |candidate|
336
+ ask(build_request(context, [candidate], sections), asked, context, sections)
337
+ end
338
+ end
339
+
340
+ def decide(candidate, answers, unfit)
341
+ id = candidate['id']
342
+ numbers = numbers(id, answers)
343
+ assessment = ->(decision, reason) { Assessment.new(id: id, decision: decision, reason: reason, answers: numbers) }
344
+ if unfit.include?(id)
345
+ return assessment.call(:unverifiable, 'the candidate and its evidence do not fit a Jev request')
346
+ end
347
+ if numbers[:version_claim] >= @thresholds[:version_claim]
348
+ return assessment.call(:reject, 'claims that a version does not exist')
349
+ end
350
+ if numbers[:enough_context] < @thresholds[:enough_context]
351
+ return assessment.call(:unverifiable, 'not enough context in the request')
352
+ end
353
+
354
+ if numbers[:real_issue] >= @thresholds[:keep_above]
355
+ assessment.call(:keep, 'real issue')
356
+ else
357
+ assessment.call(:reject, 'not a confirmed issue')
358
+ end
359
+ end
360
+
361
+ def numbers(id, answers)
362
+ noul = ->(name) { answers.dig("#{name}_#{id}", 'noul') }
363
+ choice = lambda do |name|
364
+ answer = answers["#{name}_#{id}"]
365
+ answer && {choice: answer['choice'], confidence: answer['confidence']}
366
+ end
367
+ {
368
+ real_issue: noul.call('real_issue'), enough_context: noul.call('enough_context'),
369
+ version_claim: noul.call('version_claim'), duplicate_of: choice.call('duplicate_of'),
370
+ severity: choice.call('severity')
371
+ }.compact
372
+ end
373
+
374
+ def drop_duplicates(assessments, candidates, answers)
375
+ kept_ids = assessments.select { |assessment| assessment.decision == :keep }.map(&:id)
376
+ kept = candidates.select { |candidate| kept_ids.include?(candidate['id']) }
377
+ dropped = self.class.duplicates(kept, answers, threshold: @thresholds[:duplicate])
378
+ assessments.map do |assessment|
379
+ winner = dropped[assessment.id]
380
+ next assessment unless winner
381
+
382
+ Assessment.new(id: assessment.id, decision: :reject, reason: "duplicate of #{winner}",
383
+ answers: assessment.answers)
384
+ end
385
+ end
386
+ end
387
+ end
@@ -0,0 +1,82 @@
1
+ # frozen_string_literal: true
2
+ require 'json'
3
+ require 'logger'
4
+ require_relative 'errors'
5
+ require_relative 'utils'
6
+ require_relative 'jev_critic'
7
+
8
+ module Aireview
9
+ # llm.jev.shadow: after the LLM Critique the same candidates go to Jev, and
10
+ # its decisions are logged next to the Critique verdicts. The review result
11
+ # and the review key do not depend on it, and a Jev failure is a warning:
12
+ # this is data for choosing thresholds, not a check.
13
+ class JevShadow
14
+ def initialize(config:, scrub:, logger: Logger.new($stderr), critic: nil)
15
+ @config = config
16
+ @scrub = scrub
17
+ @logger = logger
18
+ @critic = critic
19
+ end
20
+
21
+ # accepted — what the LLM Critique kept; only the ids are compared.
22
+ def run(context:, candidates:, accepted:)
23
+ critic = self.critic
24
+ return @logger.warn('Jev shadow skipped: JEV_API_KEY is not set') unless critic
25
+
26
+ started_at = Process.clock_gettime(Process::CLOCK_MONOTONIC)
27
+ result = critic.assess(context: context, candidates: candidates)
28
+ seconds = Process.clock_gettime(Process::CLOCK_MONOTONIC) - started_at
29
+ log(result, accepted.map { |candidate| (candidate['id'] || candidate[:id]).to_s }, seconds)
30
+ rescue JevError => e
31
+ @logger.warn("Jev shadow failed, the review is not affected: #{e.message}")
32
+ # The shadow is an experiment running in real review jobs: a bug in it
33
+ # must not cost a review either, but it must stay visible as a bug.
34
+ rescue StandardError => e
35
+ @logger.warn("Jev shadow crashed, the review is not affected: #{e.class}: #{e.message} " \
36
+ "(#{e.backtrace&.first})")
37
+ end
38
+
39
+ private
40
+
41
+ def critic
42
+ return @critic if @critic
43
+ return nil if Aireview::Utils.blank?(@config.jev_api_key)
44
+
45
+ @critic = JevCritic.build(config: @config, scrub: @scrub, logger: @logger)
46
+ end
47
+
48
+ def log(result, critique_kept, seconds)
49
+ result.assessments.each do |assessment|
50
+ critique = critique_kept.include?(assessment.id) ? 'keep' : 'reject'
51
+ @logger.info("Jev shadow #{assessment.id}: #{assessment.decision} (#{assessment.reason}; " \
52
+ "#{assessment.numbers}), critique: #{critique}")
53
+ end
54
+ @logger.info("Jev shadow: #{summary(result.assessments, critique_kept)} " \
55
+ "(model=#{result.model}, requests=#{result.requests}, #{format('%.1fs', seconds)})")
56
+ @logger.info("Jev shadow data: #{JSON.generate(data(result, critique_kept))}")
57
+ end
58
+
59
+ # The lines above round for reading; thresholds are chosen from this one,
60
+ # which keeps the numbers exactly as Jev returned them.
61
+ def data(result, critique_kept)
62
+ {
63
+ model: result.model,
64
+ candidates: result.assessments.map do |assessment|
65
+ {id: assessment.id, decision: assessment.decision, reason: assessment.reason,
66
+ critique: critique_kept.include?(assessment.id) ? 'keep' : 'reject', **assessment.answers}
67
+ end
68
+ }
69
+ end
70
+
71
+ def summary(assessments, critique_kept)
72
+ decided = assessments.reject { |assessment| assessment.decision == :unverifiable }
73
+ agreed = decided.count do |assessment|
74
+ (assessment.decision == :keep) == critique_kept.include?(assessment.id)
75
+ end
76
+ counts = %i[keep reject unverifiable].map do |decision|
77
+ "#{decision} #{assessments.count { |assessment| assessment.decision == decision }}"
78
+ end
79
+ "agrees with critique on #{agreed} of #{decided.size} decided candidate(s); #{counts.join(', ')}"
80
+ end
81
+ end
82
+ end
@@ -0,0 +1,97 @@
1
+ # frozen_string_literal: true
2
+ require 'logger'
3
+ require_relative 'errors'
4
+ require_relative 'jev_critic'
5
+
6
+ module Aireview
7
+ # The Critique stage with llm.critique.engine: jev. Jev decides keep or
8
+ # reject; the candidates it cannot judge (not enough context, too large)
9
+ # and, when Jev fails, all of them go to the LLM Critique with
10
+ # llm.jev.fallback: model, or are rejected / fail the run with fail.
11
+ # Duplicates are dropped once, over the merged verdicts of Jev and the LLM:
12
+ # neither side sees what the other kept.
13
+ class JevStage
14
+ # The report line: :jev (Jev decided everything), :jev_partial (some
15
+ # went to the LLM), :jev_failed (the LLM decided everything).
16
+ Outcome = Struct.new(:accepted, :note, keyword_init: true)
17
+
18
+ def initialize(config:, critic:, logger: Logger.new($stderr))
19
+ @config = config
20
+ @critic = critic
21
+ @logger = logger
22
+ end
23
+
24
+ # llm_critique — the LLM Critique of the pipeline for a subset of the
25
+ # candidates; returns the kept ones, refined.
26
+ def run(context:, candidates:, llm_critique:)
27
+ @logger.info("Pipeline critique pass started (engine=jev, model=#{@config.jev_model})")
28
+ begin
29
+ result = @critic.assess(context: context, candidates: candidates, dedupe: false)
30
+ rescue JevError => e
31
+ return fall_back(e, candidates, llm_critique)
32
+ end
33
+ result.assessments.each { |assessment| log(assessment) }
34
+
35
+ unverifiable = pick(candidates, result, :unverifiable)
36
+ accepted = pick(candidates, result, :keep) + judge_unverifiable(unverifiable, llm_critique)
37
+ accepted = drop_duplicates(in_generate_order(accepted, candidates), result.answers)
38
+ @logger.info("Pipeline critique pass completed with #{accepted.size} kept candidate(s) (engine=jev, " \
39
+ "model=#{result.model}, requests=#{result.requests})")
40
+ Outcome.new(accepted: accepted, note: unverifiable.empty? || fail? ? :jev : :jev_partial)
41
+ end
42
+
43
+ private
44
+
45
+ def fail?
46
+ @config.jev_fallback == 'fail'
47
+ end
48
+
49
+ def fall_back(error, candidates, llm_critique)
50
+ if fail?
51
+ raise JevError.new("Jev critique failed and llm.jev.fallback is fail: #{error.message}", status: error.status)
52
+ end
53
+
54
+ @logger.warn("Jev critique failed, the LLM critique takes over (llm.jev.fallback: model): #{error.message}")
55
+ Outcome.new(accepted: llm_critique.call(candidates), note: :jev_failed)
56
+ end
57
+
58
+ def judge_unverifiable(unverifiable, llm_critique)
59
+ return [] if unverifiable.empty?
60
+
61
+ ids = unverifiable.map { |candidate| id_of(candidate) }.join(', ')
62
+ if fail?
63
+ @logger.info("Critique reject #{ids}: Jev could not judge them and llm.jev.fallback is fail")
64
+ return []
65
+ end
66
+
67
+ @logger.info("Pipeline critique: #{ids} go to the LLM critique, Jev could not judge them " \
68
+ "(model=#{@config.critique_model})")
69
+ llm_critique.call(unverifiable)
70
+ end
71
+
72
+ def pick(candidates, result, decision)
73
+ ids = result.assessments.select { |assessment| assessment.decision == decision }.map(&:id)
74
+ candidates.select { |candidate| ids.include?(id_of(candidate)) }
75
+ end
76
+
77
+ def in_generate_order(accepted, candidates)
78
+ order = candidates.map { |candidate| id_of(candidate) }
79
+ accepted.sort_by { |candidate| order.index(id_of(candidate)) || order.size }
80
+ end
81
+
82
+ def drop_duplicates(accepted, answers)
83
+ dropped = JevCritic.duplicates(accepted, answers, threshold: @config.jev_thresholds[:duplicate])
84
+ dropped.each { |id, winner| @logger.info("Critique reject #{id}: duplicate of #{winner} (Jev)") }
85
+ accepted.reject { |candidate| dropped.key?(id_of(candidate)) }
86
+ end
87
+
88
+ def log(assessment)
89
+ message = "Critique (jev) #{assessment.decision} #{assessment.id}: #{assessment.reason} (#{assessment.numbers})"
90
+ assessment.decision == :keep ? @logger.debug(message) : @logger.info(message)
91
+ end
92
+
93
+ def id_of(candidate)
94
+ (candidate['id'] || candidate[:id]).to_s.strip
95
+ end
96
+ end
97
+ end