aireview 2.1.0 → 2.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,387 @@
1
+ # frozen_string_literal: true
2
+ require 'json'
3
+ require 'logger'
4
+ require 'yaml'
5
+ require_relative 'utils'
6
+ require_relative 'candidate_checker'
7
+ require_relative 'jev_client'
8
+
9
+ module Aireview
10
+ # Jev as a critic. Every candidate becomes a few questions
11
+ # (prompts/jev_questions.yml) over one state: the MR and Jira sections, the
12
+ # candidates with the hunks they point at, the diff. The answers turn into
13
+ # keep, reject or unverifiable by thresholds, then duplicates among the
14
+ # kept ones are dropped. Jev cannot rewrite a finding, so there is no
15
+ # refinement. Used as the critique engine (JevStage) and in shadow mode.
16
+ class JevCritic
17
+ QUESTIONS = YAML.safe_load_file(File.expand_path('prompts/jev_questions.yml', __dir__)).freeze
18
+ # The questions whose answers decide; severity only goes to the log. Their
19
+ # templates go into the review key whole, not as asked about a stub
20
+ # candidate: a single stub gets no duplicate_of question.
21
+ DECISION_QUESTIONS = %w[real_issue enough_context version_claim duplicate_of].freeze
22
+ # Jev limits (docs.typesafe.ai/models): the state plus the longest
23
+ # question, and the state plus all questions together. Questions count
24
+ # in both, so the state budget is what they leave.
25
+ STATE_WITH_LONGEST_QUESTION_TOKENS = 32_000
26
+ REQUEST_TOKENS = 64_000
27
+ # There is no Jev tokenizer here: characters are converted pessimistically
28
+ # (code and Cyrillic take more tokens than English prose) with a margin.
29
+ # Every request logs the real input_tokens to compare against.
30
+ CHARS_PER_TOKEN = 2.5
31
+ SAFETY = 0.9
32
+ CANDIDATE_FIELDS = %w[id file line quoted_code category severity problem why suggestion note].freeze
33
+ TRUNCATED = '[truncated to fit the Jev request]'
34
+ NOT_SHOWN = '(the file is not shown in the diff)'
35
+ TASK = 'Second pass of a merge request review: check the candidate findings of the first pass ' \
36
+ 'against the diff and the requirements.'
37
+ SEVERITY_RANK = {'critical' => 0, 'major' => 1, 'minor' => 2}.freeze
38
+ # The docs only say that a 422 body names the offending field; a size
39
+ # rejection is recognized by an explicit phrase about going over a limit.
40
+ # Bare words like "context" are not enough: they occur in question names
41
+ # (enough_context_C1). Any other 422 is a malformed request, and asking
42
+ # again would not help.
43
+ SIZE_ERROR = /
44
+ \bexceed(?:s|ed|ing)?\b |
45
+ \btoo\s+(?:long|large|many\s+tokens)\b |
46
+ \bmaximum\s+(?:context|length|tokens?)\b |
47
+ \b(?:token|context|length)\s+limit\b
48
+ /ix
49
+
50
+ # decision — :keep, :reject or :unverifiable; answers — the numbers
51
+ # behind it, for the log.
52
+ Assessment = Struct.new(:id, :decision, :reason, :answers, keyword_init: true) do
53
+ # Rounded for reading; the exact numbers stay in answers.
54
+ def numbers
55
+ answers.map do |name, value|
56
+ value.is_a?(Hash) ? "#{name}=#{value[:choice]}/#{round(value[:confidence])}" : "#{name}=#{round(value)}"
57
+ end.join(' ')
58
+ end
59
+
60
+ private
61
+
62
+ def round(value)
63
+ value.is_a?(Numeric) ? value.round(2) : value
64
+ end
65
+ end
66
+ # answers — every answer by question key, for dropping duplicates once
67
+ # the verdicts of Jev and of the LLM are merged.
68
+ Result = Struct.new(:assessments, :answers, :requests, :model, keyword_init: true)
69
+ Request = Struct.new(:candidates, :state, :questions, keyword_init: true)
70
+ # What the requests of one assessment gathered.
71
+ Asked = Struct.new(:answers, :unfit, :requests, :model, keyword_init: true)
72
+
73
+ # scrub — the secret scrubber of the context: the candidates are LLM text
74
+ # and are scrubbed like the candidates JSON of the Critique prompt.
75
+ def initialize(client:, thresholds:, review_instructions: nil, scrub: ->(text) { text },
76
+ logger: Logger.new($stderr))
77
+ @client = client
78
+ @thresholds = thresholds
79
+ @scrub = scrub
80
+ instructions = Aireview::Utils.presence(review_instructions.to_s.strip)
81
+ @review_instructions = instructions && scrub.call(instructions)
82
+ @logger = logger
83
+ end
84
+
85
+ # Kept candidates in generate order: an edge comes from a
86
+ # duplicate_of_<id> answer that names another kept id with enough
87
+ # confidence. In a group of duplicates the most severe one stays, on a
88
+ # tie the one generate listed first. Returns {dropped id => kept id}.
89
+ # The client and the critic as the config sets them up; scrub — the
90
+ # secret scrubber of the context.
91
+ def self.build(config:, scrub:, logger:)
92
+ new(client: JevClient.new(config: config, logger: logger), thresholds: config.jev_thresholds,
93
+ review_instructions: config.review_instructions, scrub: scrub, logger: logger)
94
+ end
95
+
96
+ def self.decision_templates
97
+ QUESTIONS.slice(*DECISION_QUESTIONS)
98
+ end
99
+
100
+ def self.duplicates(kept, answers, threshold:)
101
+ ids = kept.map { |candidate| field(candidate, 'id') }
102
+ rank = kept.to_h do |candidate|
103
+ [field(candidate, 'id'), SEVERITY_RANK.fetch(field(candidate, 'severity'), SEVERITY_RANK.size)]
104
+ end
105
+ duplicate_groups(ids, answers, threshold).each_with_object({}) do |group, dropped|
106
+ winner = group.min_by { |id| [rank[id], ids.index(id)] }
107
+ (group - [winner]).each { |id| dropped[id] = winner }
108
+ end
109
+ end
110
+
111
+ def self.duplicate_groups(ids, answers, threshold)
112
+ group_of = ids.to_h { |id| [id, [id]] }
113
+ ids.each do |id|
114
+ other = duplicate_answer(id, ids, answers, threshold)
115
+ join_groups(group_of, id, other) if other
116
+ end
117
+ group_of.values.uniq(&:object_id).select { |group| group.size > 1 }
118
+ end
119
+
120
+ def self.join_groups(group_of, id, other)
121
+ return if group_of[id].equal?(group_of[other])
122
+
123
+ merged = group_of[id] + group_of[other]
124
+ merged.each { |member| group_of[member] = merged }
125
+ end
126
+
127
+ # The other kept id a duplicate_of answer names with enough confidence.
128
+ def self.duplicate_answer(id, ids, answers, threshold)
129
+ answer = answers["duplicate_of_#{id}"]
130
+ return nil unless answer.is_a?(Hash) && answer['confidence'].to_f >= threshold
131
+
132
+ other = answer['choice']
133
+ other if other != id && ids.include?(other)
134
+ end
135
+
136
+ def self.field(candidate, key)
137
+ (candidate[key] || candidate[key.to_sym]).to_s.strip
138
+ end
139
+
140
+ # dedupe — false when the caller merges the verdicts with the LLM ones
141
+ # first and drops duplicates over the whole set (JevStage).
142
+ def assess(context:, candidates:, dedupe: true)
143
+ candidates = candidates.map { |candidate| normalize(candidate) }
144
+ sections = diff_sections(context.diff_text)
145
+ asked = Asked.new(answers: {}, unfit: [], requests: 0, model: nil)
146
+ plan_requests(context, candidates, sections).each { |request| ask(request, asked, context, sections) }
147
+ assessments = candidates.map { |candidate| decide(candidate, asked.answers, asked.unfit) }
148
+ assessments = drop_duplicates(assessments, candidates, asked.answers) if dedupe
149
+ Result.new(assessments: assessments, answers: asked.answers, requests: asked.requests, model: asked.model)
150
+ end
151
+
152
+ # The first request as it would be sent, for --dry-run.
153
+ def preview(context:, candidates:)
154
+ candidates = candidates.map { |candidate| normalize(candidate) }
155
+ plan_requests(context, candidates, diff_sections(context.diff_text)).first
156
+ end
157
+
158
+ private
159
+
160
+ def normalize(candidate)
161
+ CANDIDATE_FIELDS.to_h do |field|
162
+ value = candidate[field] || candidate[field.to_sym]
163
+ [field, value.is_a?(String) ? @scrub.call(value) : value]
164
+ end.compact.merge('id' => self.class.field(candidate, 'id'))
165
+ end
166
+
167
+ # One request for all candidates when it fits, otherwise one per
168
+ # candidate (then without duplicate_of: a request sees one candidate).
169
+ # A candidate that does not fit even alone is unverifiable.
170
+ def plan_requests(context, candidates, sections)
171
+ shared = build_request(context, candidates, sections)
172
+ return [shared] if shared.state || candidates.size == 1
173
+
174
+ @logger.info('Jev: the candidates do not fit one request, asking about each separately')
175
+ candidates.map { |candidate| build_request(context, [candidate], sections) }
176
+ end
177
+
178
+ def build_request(context, candidates, sections)
179
+ ids = candidates.map { |candidate| candidate['id'] }
180
+ questions = candidates.map { |candidate| questions_for(candidate['id'], ids) }.reduce({}, :merge)
181
+ state = {
182
+ 'task' => TASK,
183
+ 'requirements' => context.sections,
184
+ 'project_instructions' => @review_instructions,
185
+ 'context_truncated' => coverage_notes(context.coverage),
186
+ 'candidates' => candidates.to_h { |candidate| [candidate['id'], with_evidence(candidate, sections)] },
187
+ 'diff' => context.diff_text
188
+ }.compact
189
+ Request.new(candidates: candidates, state: fit(state, state_budget(questions)), questions: questions)
190
+ end
191
+
192
+ def questions_for(id, ids)
193
+ questions = %w[real_issue enough_context version_claim severity].to_h do |name|
194
+ ["#{name}_#{id}", question(QUESTIONS.fetch(name), id)]
195
+ end
196
+ others = ids - [id]
197
+ questions["duplicate_of_#{id}"] = duplicate_question(id, others) unless others.empty?
198
+ questions
199
+ end
200
+
201
+ def question(template, id)
202
+ {
203
+ 'type' => template.fetch('type'),
204
+ 'instructions' => format(template.fetch('instructions'), id: id),
205
+ 'criteria' => template.fetch('criteria').transform_values { |text| format(text, id: id) }
206
+ }
207
+ end
208
+
209
+ def duplicate_question(id, others)
210
+ template = QUESTIONS.fetch('duplicate_of')
211
+ criteria = others.to_h { |other| [other, format(template.fetch('other'), other: other)] }
212
+ {
213
+ 'type' => template.fetch('type'),
214
+ 'instructions' => format(template.fetch('instructions'), id: id),
215
+ 'criteria' => criteria.merge('none' => template.fetch('none'))
216
+ }
217
+ end
218
+
219
+ def state_budget(questions)
220
+ sizes = questions.values.map { |question| tokens(JSON.generate(question).length) }
221
+ room = [STATE_WITH_LONGEST_QUESTION_TOKENS - sizes.max, REQUEST_TOKENS - sizes.sum].min
222
+ (room * SAFETY * CHARS_PER_TOKEN).floor
223
+ end
224
+
225
+ def tokens(chars)
226
+ (chars / CHARS_PER_TOKEN).ceil
227
+ end
228
+
229
+ # The diff goes first, then the MR and Jira sections; the candidates and
230
+ # their evidence are never cut. nil — does not fit even then.
231
+ def fit(state, budget)
232
+ %w[diff requirements].each do |key|
233
+ break if size(state) <= budget
234
+
235
+ state = shrink(state, key, budget) if state[key]
236
+ end
237
+ size(state) <= budget ? state : nil
238
+ end
239
+
240
+ # Every character of the text takes at least one character of JSON, so
241
+ # cutting the overflow in characters is enough; the marker costs its
242
+ # length plus the escaped newline.
243
+ def shrink(state, key, budget)
244
+ text = state[key].to_s
245
+ keep = text.length - (size(state) - budget) - TRUNCATED.length - 2
246
+ state.merge(key => keep.positive? ? "#{text[0, keep]}\n#{TRUNCATED}" : TRUNCATED)
247
+ end
248
+
249
+ def size(state)
250
+ JSON.generate(state).length
251
+ end
252
+
253
+ def coverage_notes(coverage)
254
+ return nil if coverage.nil? || coverage.complete?
255
+
256
+ notes = coverage.truncated_sections.map { |label| "#{label} truncated" }
257
+ notes << "files not shown: #{coverage.files_not_shown.join(', ')}" unless coverage.files_not_shown.empty?
258
+ notes += coverage.files_partial.map { |file| "#{file[:path]}: #{file[:shown]} of #{file[:total]} hunks shown" }
259
+ notes << "diff not available: #{coverage.files_unavailable.join(', ')}" unless coverage.files_unavailable.empty?
260
+ notes
261
+ end
262
+
263
+ # The hunk the line falls into; the whole shown diff of the file when
264
+ # the line is unknown or the anchoring check left a note.
265
+ def with_evidence(candidate, sections)
266
+ section = sections[candidate['file'].to_s] || sections[candidate['file'].to_s.sub(%r{\A(?:\./|[ab]/)}, '')]
267
+ evidence = if section.nil?
268
+ NOT_SHOWN
269
+ elsif candidate['line'].is_a?(Integer) && !candidate['note']
270
+ hunk = section[:hunks].find { |item| item[:range].cover?(candidate['line']) }
271
+ hunk ? hunk[:text] : section[:text]
272
+ else
273
+ section[:text]
274
+ end
275
+ candidate.merge('evidence' => evidence)
276
+ end
277
+
278
+ # The shown diff by file: its whole text and every hunk with the range of
279
+ # new-file lines it covers (the headers as CandidateChecker reads them).
280
+ def diff_sections(diff_text)
281
+ sections = {}
282
+ section = nil
283
+ diff_text.to_s.each_line do |line|
284
+ if (header = line.match(CandidateChecker::FILE_HEADER))
285
+ section = {text: +'', hunks: []}
286
+ header.captures.map(&:strip).each { |path| sections[path] = section }
287
+ elsif section && (hunk_header = line.match(CandidateChecker::HUNK_HEADER))
288
+ section[:hunks] << new_hunk(hunk_header)
289
+ end
290
+ add_line(section, line) if section
291
+ end
292
+ sections
293
+ end
294
+
295
+ def new_hunk(header)
296
+ start = header[1].to_i
297
+ length = header[2] ? header[2].to_i : 1
298
+ {range: start..(start + length - 1), text: +''}
299
+ end
300
+
301
+ def add_line(section, line)
302
+ section[:text] << line
303
+ section[:hunks].last[:text] << line unless section[:hunks].empty?
304
+ end
305
+
306
+ # The estimate goes to the log next to the input_tokens Jev reports, to
307
+ # check CHARS_PER_TOKEN against real requests. When Jev rejects a request
308
+ # as too large after all (the estimate missed), its candidates are asked
309
+ # about one by one; one that is too large alone is unverifiable.
310
+ def ask(request, asked, context, sections)
311
+ ids = request.candidates.map { |candidate| candidate['id'] }
312
+ return asked.unfit.concat(ids) unless request.state
313
+
314
+ chars = size(request.state) + request.questions.values.sum { |question| JSON.generate(question).length }
315
+ @logger.info("Jev request estimated at ~#{tokens(chars)} tokens (#{chars} chars of state and questions)")
316
+ asked.requests += 1
317
+ result = @client.evaluate(state: request.state, questions: request.questions)
318
+ asked.answers.merge!(result.answers)
319
+ asked.model = result.model
320
+ rescue JevError => e
321
+ raise unless e.status == 422 && e.message.match?(SIZE_ERROR)
322
+
323
+ ask_separately(request, asked, context, sections, e)
324
+ end
325
+
326
+ def ask_separately(request, asked, context, sections, error)
327
+ ids = request.candidates.map { |candidate| candidate['id'] }
328
+ if ids.size == 1
329
+ @logger.warn("Jev rejected the request about #{ids.first} as too large, " \
330
+ "it stays unverifiable: #{error.message}")
331
+ return asked.unfit.concat(ids)
332
+ end
333
+
334
+ @logger.warn("Jev rejected the request as too large, asking about each candidate separately: #{error.message}")
335
+ request.candidates.each do |candidate|
336
+ ask(build_request(context, [candidate], sections), asked, context, sections)
337
+ end
338
+ end
339
+
340
+ def decide(candidate, answers, unfit)
341
+ id = candidate['id']
342
+ numbers = numbers(id, answers)
343
+ assessment = ->(decision, reason) { Assessment.new(id: id, decision: decision, reason: reason, answers: numbers) }
344
+ if unfit.include?(id)
345
+ return assessment.call(:unverifiable, 'the candidate and its evidence do not fit a Jev request')
346
+ end
347
+ if numbers[:version_claim] >= @thresholds[:version_claim]
348
+ return assessment.call(:reject, 'claims that a version does not exist')
349
+ end
350
+ if numbers[:enough_context] < @thresholds[:enough_context]
351
+ return assessment.call(:unverifiable, 'not enough context in the request')
352
+ end
353
+
354
+ if numbers[:real_issue] >= @thresholds[:keep_above]
355
+ assessment.call(:keep, 'real issue')
356
+ else
357
+ assessment.call(:reject, 'not a confirmed issue')
358
+ end
359
+ end
360
+
361
+ def numbers(id, answers)
362
+ noul = ->(name) { answers.dig("#{name}_#{id}", 'noul') }
363
+ choice = lambda do |name|
364
+ answer = answers["#{name}_#{id}"]
365
+ answer && {choice: answer['choice'], confidence: answer['confidence']}
366
+ end
367
+ {
368
+ real_issue: noul.call('real_issue'), enough_context: noul.call('enough_context'),
369
+ version_claim: noul.call('version_claim'), duplicate_of: choice.call('duplicate_of'),
370
+ severity: choice.call('severity')
371
+ }.compact
372
+ end
373
+
374
+ def drop_duplicates(assessments, candidates, answers)
375
+ kept_ids = assessments.select { |assessment| assessment.decision == :keep }.map(&:id)
376
+ kept = candidates.select { |candidate| kept_ids.include?(candidate['id']) }
377
+ dropped = self.class.duplicates(kept, answers, threshold: @thresholds[:duplicate])
378
+ assessments.map do |assessment|
379
+ winner = dropped[assessment.id]
380
+ next assessment unless winner
381
+
382
+ Assessment.new(id: assessment.id, decision: :reject, reason: "duplicate of #{winner}",
383
+ answers: assessment.answers)
384
+ end
385
+ end
386
+ end
387
+ end
@@ -0,0 +1,82 @@
1
+ # frozen_string_literal: true
2
+ require 'json'
3
+ require 'logger'
4
+ require_relative 'errors'
5
+ require_relative 'utils'
6
+ require_relative 'jev_critic'
7
+
8
+ module Aireview
9
+ # llm.jev.shadow: after the LLM Critique the same candidates go to Jev, and
10
+ # its decisions are logged next to the Critique verdicts. The review result
11
+ # and the review key do not depend on it, and a Jev failure is a warning:
12
+ # this is data for choosing thresholds, not a check.
13
+ class JevShadow
14
+ def initialize(config:, scrub:, logger: Logger.new($stderr), critic: nil)
15
+ @config = config
16
+ @scrub = scrub
17
+ @logger = logger
18
+ @critic = critic
19
+ end
20
+
21
+ # accepted — what the LLM Critique kept; only the ids are compared.
22
+ def run(context:, candidates:, accepted:)
23
+ critic = self.critic
24
+ return @logger.warn('Jev shadow skipped: JEV_API_KEY is not set') unless critic
25
+
26
+ started_at = Process.clock_gettime(Process::CLOCK_MONOTONIC)
27
+ result = critic.assess(context: context, candidates: candidates)
28
+ seconds = Process.clock_gettime(Process::CLOCK_MONOTONIC) - started_at
29
+ log(result, accepted.map { |candidate| (candidate['id'] || candidate[:id]).to_s }, seconds)
30
+ rescue JevError => e
31
+ @logger.warn("Jev shadow failed, the review is not affected: #{e.message}")
32
+ # The shadow is an experiment running in real review jobs: a bug in it
33
+ # must not cost a review either, but it must stay visible as a bug.
34
+ rescue StandardError => e
35
+ @logger.warn("Jev shadow crashed, the review is not affected: #{e.class}: #{e.message} " \
36
+ "(#{e.backtrace&.first})")
37
+ end
38
+
39
+ private
40
+
41
+ def critic
42
+ return @critic if @critic
43
+ return nil if Aireview::Utils.blank?(@config.jev_api_key)
44
+
45
+ @critic = JevCritic.build(config: @config, scrub: @scrub, logger: @logger)
46
+ end
47
+
48
+ def log(result, critique_kept, seconds)
49
+ result.assessments.each do |assessment|
50
+ critique = critique_kept.include?(assessment.id) ? 'keep' : 'reject'
51
+ @logger.info("Jev shadow #{assessment.id}: #{assessment.decision} (#{assessment.reason}; " \
52
+ "#{assessment.numbers}), critique: #{critique}")
53
+ end
54
+ @logger.info("Jev shadow: #{summary(result.assessments, critique_kept)} " \
55
+ "(model=#{result.model}, requests=#{result.requests}, #{format('%.1fs', seconds)})")
56
+ @logger.info("Jev shadow data: #{JSON.generate(data(result, critique_kept))}")
57
+ end
58
+
59
+ # The lines above round for reading; thresholds are chosen from this one,
60
+ # which keeps the numbers exactly as Jev returned them.
61
+ def data(result, critique_kept)
62
+ {
63
+ model: result.model,
64
+ candidates: result.assessments.map do |assessment|
65
+ {id: assessment.id, decision: assessment.decision, reason: assessment.reason,
66
+ critique: critique_kept.include?(assessment.id) ? 'keep' : 'reject', **assessment.answers}
67
+ end
68
+ }
69
+ end
70
+
71
+ def summary(assessments, critique_kept)
72
+ decided = assessments.reject { |assessment| assessment.decision == :unverifiable }
73
+ agreed = decided.count do |assessment|
74
+ (assessment.decision == :keep) == critique_kept.include?(assessment.id)
75
+ end
76
+ counts = %i[keep reject unverifiable].map do |decision|
77
+ "#{decision} #{assessments.count { |assessment| assessment.decision == decision }}"
78
+ end
79
+ "agrees with critique on #{agreed} of #{decided.size} decided candidate(s); #{counts.join(', ')}"
80
+ end
81
+ end
82
+ end
@@ -0,0 +1,97 @@
1
+ # frozen_string_literal: true
2
+ require 'logger'
3
+ require_relative 'errors'
4
+ require_relative 'jev_critic'
5
+
6
+ module Aireview
7
+ # The Critique stage with llm.critique.engine: jev. Jev decides keep or
8
+ # reject; the candidates it cannot judge (not enough context, too large)
9
+ # and, when Jev fails, all of them go to the LLM Critique with
10
+ # llm.jev.fallback: model, or are rejected / fail the run with fail.
11
+ # Duplicates are dropped once, over the merged verdicts of Jev and the LLM:
12
+ # neither side sees what the other kept.
13
+ class JevStage
14
+ # The report line: :jev (Jev decided everything), :jev_partial (some
15
+ # went to the LLM), :jev_failed (the LLM decided everything).
16
+ Outcome = Struct.new(:accepted, :note, keyword_init: true)
17
+
18
+ def initialize(config:, critic:, logger: Logger.new($stderr))
19
+ @config = config
20
+ @critic = critic
21
+ @logger = logger
22
+ end
23
+
24
+ # llm_critique — the LLM Critique of the pipeline for a subset of the
25
+ # candidates; returns the kept ones, refined.
26
+ def run(context:, candidates:, llm_critique:)
27
+ @logger.info("Pipeline critique pass started (engine=jev, model=#{@config.jev_model})")
28
+ begin
29
+ result = @critic.assess(context: context, candidates: candidates, dedupe: false)
30
+ rescue JevError => e
31
+ return fall_back(e, candidates, llm_critique)
32
+ end
33
+ result.assessments.each { |assessment| log(assessment) }
34
+
35
+ unverifiable = pick(candidates, result, :unverifiable)
36
+ accepted = pick(candidates, result, :keep) + judge_unverifiable(unverifiable, llm_critique)
37
+ accepted = drop_duplicates(in_generate_order(accepted, candidates), result.answers)
38
+ @logger.info("Pipeline critique pass completed with #{accepted.size} kept candidate(s) (engine=jev, " \
39
+ "model=#{result.model}, requests=#{result.requests})")
40
+ Outcome.new(accepted: accepted, note: unverifiable.empty? || fail? ? :jev : :jev_partial)
41
+ end
42
+
43
+ private
44
+
45
+ def fail?
46
+ @config.jev_fallback == 'fail'
47
+ end
48
+
49
+ def fall_back(error, candidates, llm_critique)
50
+ if fail?
51
+ raise JevError.new("Jev critique failed and llm.jev.fallback is fail: #{error.message}", status: error.status)
52
+ end
53
+
54
+ @logger.warn("Jev critique failed, the LLM critique takes over (llm.jev.fallback: model): #{error.message}")
55
+ Outcome.new(accepted: llm_critique.call(candidates), note: :jev_failed)
56
+ end
57
+
58
+ def judge_unverifiable(unverifiable, llm_critique)
59
+ return [] if unverifiable.empty?
60
+
61
+ ids = unverifiable.map { |candidate| id_of(candidate) }.join(', ')
62
+ if fail?
63
+ @logger.info("Critique reject #{ids}: Jev could not judge them and llm.jev.fallback is fail")
64
+ return []
65
+ end
66
+
67
+ @logger.info("Pipeline critique: #{ids} go to the LLM critique, Jev could not judge them " \
68
+ "(model=#{@config.critique_model})")
69
+ llm_critique.call(unverifiable)
70
+ end
71
+
72
+ def pick(candidates, result, decision)
73
+ ids = result.assessments.select { |assessment| assessment.decision == decision }.map(&:id)
74
+ candidates.select { |candidate| ids.include?(id_of(candidate)) }
75
+ end
76
+
77
+ def in_generate_order(accepted, candidates)
78
+ order = candidates.map { |candidate| id_of(candidate) }
79
+ accepted.sort_by { |candidate| order.index(id_of(candidate)) || order.size }
80
+ end
81
+
82
+ def drop_duplicates(accepted, answers)
83
+ dropped = JevCritic.duplicates(accepted, answers, threshold: @config.jev_thresholds[:duplicate])
84
+ dropped.each { |id, winner| @logger.info("Critique reject #{id}: duplicate of #{winner} (Jev)") }
85
+ accepted.reject { |candidate| dropped.key?(id_of(candidate)) }
86
+ end
87
+
88
+ def log(assessment)
89
+ message = "Critique (jev) #{assessment.decision} #{assessment.id}: #{assessment.reason} (#{assessment.numbers})"
90
+ assessment.decision == :keep ? @logger.debug(message) : @logger.info(message)
91
+ end
92
+
93
+ def id_of(candidate)
94
+ (candidate['id'] || candidate[:id]).to_s.strip
95
+ end
96
+ end
97
+ end
@@ -30,7 +30,7 @@ module Aireview
30
30
  stage = prompt.stage.to_s
31
31
  model = candidate.model
32
32
  @logger.info("LLM #{stage} request started (model=#{model}, temperature=#{prompt.temperature})")
33
- chat = build_chat(context: context(stage, candidate.provider, key, key_index), stage: stage,
33
+ chat = build_chat(context: context(stage, candidate, key, key_index), stage: stage,
34
34
  model: model, provider: candidate.provider)
35
35
  chat = configure_reasoning(chat: chat, model: model, provider: candidate.provider)
36
36
  .with_temperature(prompt.temperature.to_f)
@@ -99,18 +99,19 @@ module Aireview
99
99
  context.chat(model: model, provider: provider.to_sym, assume_model_exists: true)
100
100
  end
101
101
 
102
- # A RubyLLM context per stage, provider and key index: switching the key
103
- # is another context, not an edit of the global config.
104
- def context(stage, provider, key, key_index)
105
- @contexts[[stage, provider, key_index]] ||= build_context(provider.to_s, key)
102
+ # A RubyLLM context per stage, key source (a provider's API or a server
103
+ # of your own) and key index: switching the key is another context, not
104
+ # an edit of the global config.
105
+ def context(stage, candidate, key, key_index)
106
+ @contexts[[stage, candidate.key_source, key_index]] ||= build_context(candidate, key)
106
107
  end
107
108
 
108
- def build_context(provider, api_key)
109
+ def build_context(candidate, api_key)
109
110
  RubyLLM.context do |ruby_config|
110
111
  configure_http_proxy(ruby_config)
111
112
  ruby_config.request_timeout = @config.llm_timeout.to_f
112
113
  ruby_config.max_retries = 0
113
- configure_provider(ruby_config, provider, api_key)
114
+ configure_provider(ruby_config, candidate.provider.to_s, api_key, candidate.api_base)
114
115
  end
115
116
  end
116
117
 
@@ -120,27 +121,30 @@ module Aireview
120
121
  ruby_config.http_proxy = @config.llm_http_proxy
121
122
  end
122
123
 
123
- def configure_provider(ruby_config, provider, api_key)
124
+ # api_base — the model's own server; without it the provider's API,
125
+ # or LLM_API_BASE for the providers that always took it.
126
+ def configure_provider(ruby_config, provider, api_key, api_base)
124
127
  case provider
125
128
  when 'gemini', 'openai', 'openrouter'
126
- configure_remote_provider(ruby_config, provider, api_key)
129
+ configure_remote_provider(ruby_config, provider, api_key, api_base || @config.llm_api_base)
127
130
  when 'anthropic'
128
131
  ruby_config.anthropic_api_key = api_key
132
+ ruby_config.anthropic_api_base = api_base if api_base
129
133
  when 'ollama'
130
- ruby_config.ollama_api_base = @config.ollama_api_base
134
+ ruby_config.ollama_api_base = api_base || @config.ollama_api_base
131
135
  else
132
136
  raise ConfigError, "Unsupported LLM provider: #{provider.inspect}"
133
137
  end
134
138
  end
135
139
 
136
140
  # RubyLLM 2 sends OpenAI requests to the Responses API; a compatible
137
- # server behind LLM_API_BASE usually has Chat Completions only.
138
- def configure_remote_provider(ruby_config, provider, api_key)
141
+ # server usually has Chat Completions only.
142
+ def configure_remote_provider(ruby_config, provider, api_key, api_base)
139
143
  ruby_config.public_send("#{provider}_api_key=", api_key)
140
144
  ruby_config.openai_protocol = :chat_completions if provider == 'openai'
141
- return unless Aireview::Utils.present?(@config.llm_api_base)
145
+ return unless Aireview::Utils.present?(api_base)
142
146
 
143
- ruby_config.public_send("#{provider}_api_base=", @config.llm_api_base)
147
+ ruby_config.public_send("#{provider}_api_base=", api_base)
144
148
  end
145
149
  end
146
150
  end
@@ -34,7 +34,7 @@ module Aireview
34
34
 
35
35
  Slot = Struct.new(:candidate, :index)
36
36
  # The key the next model of the same provider continues from.
37
- Carry = Struct.new(:provider, :key_index)
37
+ Carry = Struct.new(:key_source, :key_index)
38
38
  Delay = Struct.new(:seconds, :source)
39
39
 
40
40
  SWITCH_HINT = 'Try again later or switch model via --generate-model/--critique-model.'
@@ -213,7 +213,7 @@ module Aireview
213
213
  log_switch(stage, route, visits)
214
214
  status, response = try_route(stage, route, visits, &request)
215
215
  return [:ok, remember(stage, route, response)] if status == :ok
216
- return [:next_model, Carry.new(route.candidate.provider, route.key_index)] if status == :next_model
216
+ return [:next_model, Carry.new(route.candidate.key_source, route.key_index)] if status == :next_model
217
217
  end
218
218
  state(slot.candidate).exclude('daily quota exhausted on every key') unless tried
219
219
  [:next_model, carry]
@@ -224,7 +224,7 @@ module Aireview
224
224
  # is bound to "key + model", a failure on one model does not write the
225
225
  # key off for another.
226
226
  def candidate_routes(stage, slot, carry)
227
- keys = @config.provider_api_keys(slot.candidate.provider)
227
+ keys = @config.candidate_api_keys(slot.candidate)
228
228
  routes = keys.each_with_index.map do |key, key_index|
229
229
  Route.new(candidate: slot.candidate, candidate_index: slot.index, key: key,
230
230
  key_index: key_index, key_count: keys.size)
@@ -233,7 +233,7 @@ module Aireview
233
233
  end
234
234
 
235
235
  def start_key(stage, slot, carry)
236
- return carry.key_index if carry && carry.provider == slot.candidate.provider
236
+ return carry.key_index if carry && carry.key_source == slot.candidate.key_source
237
237
 
238
238
  cursor_candidate, cursor_key = @cursor.fetch(stage, [0, 0])
239
239
  cursor_candidate == slot.index ? cursor_key : 0