aireview 2.1.0 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +37 -0
- data/README.md +176 -5
- data/config/.aireview.yml.example +6 -0
- data/lib/aireview/cli.rb +11 -2
- data/lib/aireview/config.rb +26 -8
- data/lib/aireview/config_fallbacks.rb +43 -15
- data/lib/aireview/config_jev.rb +137 -0
- data/lib/aireview/config_layers.rb +1 -1
- data/lib/aireview/config_loader.rb +32 -6
- data/lib/aireview/context_builder.rb +13 -10
- data/lib/aireview/dry_run_prompts.rb +104 -0
- data/lib/aireview/dry_run_report.rb +40 -8
- data/lib/aireview/errors.rb +11 -0
- data/lib/aireview/jev_client.rb +129 -0
- data/lib/aireview/jev_critic.rb +387 -0
- data/lib/aireview/jev_shadow.rb +82 -0
- data/lib/aireview/jev_stage.rb +97 -0
- data/lib/aireview/llm_client.rb +18 -14
- data/lib/aireview/llm_router.rb +4 -4
- data/lib/aireview/model_candidate.rb +34 -5
- data/lib/aireview/model_checker.rb +46 -5
- data/lib/aireview/model_pool.rb +57 -28
- data/lib/aireview/prompts/jev_questions.yml +79 -0
- data/lib/aireview/review_marker.rb +7 -2
- data/lib/aireview/review_pipeline.rb +40 -55
- data/lib/aireview/review_renderer.rb +18 -2
- data/lib/aireview/stage_chains.rb +6 -3
- data/lib/aireview/version.rb +1 -1
- metadata +12 -4
|
@@ -0,0 +1,387 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
require 'json'
|
|
3
|
+
require 'logger'
|
|
4
|
+
require 'yaml'
|
|
5
|
+
require_relative 'utils'
|
|
6
|
+
require_relative 'candidate_checker'
|
|
7
|
+
require_relative 'jev_client'
|
|
8
|
+
|
|
9
|
+
module Aireview
|
|
10
|
+
# Jev as a critic. Every candidate becomes a few questions
|
|
11
|
+
# (prompts/jev_questions.yml) over one state: the MR and Jira sections, the
|
|
12
|
+
# candidates with the hunks they point at, the diff. The answers turn into
|
|
13
|
+
# keep, reject or unverifiable by thresholds, then duplicates among the
|
|
14
|
+
# kept ones are dropped. Jev cannot rewrite a finding, so there is no
|
|
15
|
+
# refinement. Used as the critique engine (JevStage) and in shadow mode.
|
|
16
|
+
class JevCritic
|
|
17
|
+
QUESTIONS = YAML.safe_load_file(File.expand_path('prompts/jev_questions.yml', __dir__)).freeze
|
|
18
|
+
# The questions whose answers decide; severity only goes to the log. Their
|
|
19
|
+
# templates go into the review key whole, not as asked about a stub
|
|
20
|
+
# candidate: a single stub gets no duplicate_of question.
|
|
21
|
+
DECISION_QUESTIONS = %w[real_issue enough_context version_claim duplicate_of].freeze
|
|
22
|
+
# Jev limits (docs.typesafe.ai/models): the state plus the longest
|
|
23
|
+
# question, and the state plus all questions together. Questions count
|
|
24
|
+
# in both, so the state budget is what they leave.
|
|
25
|
+
STATE_WITH_LONGEST_QUESTION_TOKENS = 32_000
|
|
26
|
+
REQUEST_TOKENS = 64_000
|
|
27
|
+
# There is no Jev tokenizer here: characters are converted pessimistically
|
|
28
|
+
# (code and Cyrillic take more tokens than English prose) with a margin.
|
|
29
|
+
# Every request logs the real input_tokens to compare against.
|
|
30
|
+
CHARS_PER_TOKEN = 2.5
|
|
31
|
+
SAFETY = 0.9
|
|
32
|
+
CANDIDATE_FIELDS = %w[id file line quoted_code category severity problem why suggestion note].freeze
|
|
33
|
+
TRUNCATED = '[truncated to fit the Jev request]'
|
|
34
|
+
NOT_SHOWN = '(the file is not shown in the diff)'
|
|
35
|
+
TASK = 'Second pass of a merge request review: check the candidate findings of the first pass ' \
|
|
36
|
+
'against the diff and the requirements.'
|
|
37
|
+
SEVERITY_RANK = {'critical' => 0, 'major' => 1, 'minor' => 2}.freeze
|
|
38
|
+
# The docs only say that a 422 body names the offending field; a size
|
|
39
|
+
# rejection is recognized by an explicit phrase about going over a limit.
|
|
40
|
+
# Bare words like "context" are not enough: they occur in question names
|
|
41
|
+
# (enough_context_C1). Any other 422 is a malformed request, and asking
|
|
42
|
+
# again would not help.
|
|
43
|
+
SIZE_ERROR = /
|
|
44
|
+
\bexceed(?:s|ed|ing)?\b |
|
|
45
|
+
\btoo\s+(?:long|large|many\s+tokens)\b |
|
|
46
|
+
\bmaximum\s+(?:context|length|tokens?)\b |
|
|
47
|
+
\b(?:token|context|length)\s+limit\b
|
|
48
|
+
/ix
|
|
49
|
+
|
|
50
|
+
# decision — :keep, :reject or :unverifiable; answers — the numbers
|
|
51
|
+
# behind it, for the log.
|
|
52
|
+
Assessment = Struct.new(:id, :decision, :reason, :answers, keyword_init: true) do
|
|
53
|
+
# Rounded for reading; the exact numbers stay in answers.
|
|
54
|
+
def numbers
|
|
55
|
+
answers.map do |name, value|
|
|
56
|
+
value.is_a?(Hash) ? "#{name}=#{value[:choice]}/#{round(value[:confidence])}" : "#{name}=#{round(value)}"
|
|
57
|
+
end.join(' ')
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
private
|
|
61
|
+
|
|
62
|
+
def round(value)
|
|
63
|
+
value.is_a?(Numeric) ? value.round(2) : value
|
|
64
|
+
end
|
|
65
|
+
end
|
|
66
|
+
# answers — every answer by question key, for dropping duplicates once
|
|
67
|
+
# the verdicts of Jev and of the LLM are merged.
|
|
68
|
+
Result = Struct.new(:assessments, :answers, :requests, :model, keyword_init: true)
|
|
69
|
+
Request = Struct.new(:candidates, :state, :questions, keyword_init: true)
|
|
70
|
+
# What the requests of one assessment gathered.
|
|
71
|
+
Asked = Struct.new(:answers, :unfit, :requests, :model, keyword_init: true)
|
|
72
|
+
|
|
73
|
+
# scrub — the secret scrubber of the context: the candidates are LLM text
|
|
74
|
+
# and are scrubbed like the candidates JSON of the Critique prompt.
|
|
75
|
+
def initialize(client:, thresholds:, review_instructions: nil, scrub: ->(text) { text },
|
|
76
|
+
logger: Logger.new($stderr))
|
|
77
|
+
@client = client
|
|
78
|
+
@thresholds = thresholds
|
|
79
|
+
@scrub = scrub
|
|
80
|
+
instructions = Aireview::Utils.presence(review_instructions.to_s.strip)
|
|
81
|
+
@review_instructions = instructions && scrub.call(instructions)
|
|
82
|
+
@logger = logger
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
# Kept candidates in generate order: an edge comes from a
|
|
86
|
+
# duplicate_of_<id> answer that names another kept id with enough
|
|
87
|
+
# confidence. In a group of duplicates the most severe one stays, on a
|
|
88
|
+
# tie the one generate listed first. Returns {dropped id => kept id}.
|
|
89
|
+
# The client and the critic as the config sets them up; scrub — the
|
|
90
|
+
# secret scrubber of the context.
|
|
91
|
+
def self.build(config:, scrub:, logger:)
|
|
92
|
+
new(client: JevClient.new(config: config, logger: logger), thresholds: config.jev_thresholds,
|
|
93
|
+
review_instructions: config.review_instructions, scrub: scrub, logger: logger)
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
def self.decision_templates
|
|
97
|
+
QUESTIONS.slice(*DECISION_QUESTIONS)
|
|
98
|
+
end
|
|
99
|
+
|
|
100
|
+
def self.duplicates(kept, answers, threshold:)
|
|
101
|
+
ids = kept.map { |candidate| field(candidate, 'id') }
|
|
102
|
+
rank = kept.to_h do |candidate|
|
|
103
|
+
[field(candidate, 'id'), SEVERITY_RANK.fetch(field(candidate, 'severity'), SEVERITY_RANK.size)]
|
|
104
|
+
end
|
|
105
|
+
duplicate_groups(ids, answers, threshold).each_with_object({}) do |group, dropped|
|
|
106
|
+
winner = group.min_by { |id| [rank[id], ids.index(id)] }
|
|
107
|
+
(group - [winner]).each { |id| dropped[id] = winner }
|
|
108
|
+
end
|
|
109
|
+
end
|
|
110
|
+
|
|
111
|
+
def self.duplicate_groups(ids, answers, threshold)
|
|
112
|
+
group_of = ids.to_h { |id| [id, [id]] }
|
|
113
|
+
ids.each do |id|
|
|
114
|
+
other = duplicate_answer(id, ids, answers, threshold)
|
|
115
|
+
join_groups(group_of, id, other) if other
|
|
116
|
+
end
|
|
117
|
+
group_of.values.uniq(&:object_id).select { |group| group.size > 1 }
|
|
118
|
+
end
|
|
119
|
+
|
|
120
|
+
def self.join_groups(group_of, id, other)
|
|
121
|
+
return if group_of[id].equal?(group_of[other])
|
|
122
|
+
|
|
123
|
+
merged = group_of[id] + group_of[other]
|
|
124
|
+
merged.each { |member| group_of[member] = merged }
|
|
125
|
+
end
|
|
126
|
+
|
|
127
|
+
# The other kept id a duplicate_of answer names with enough confidence.
|
|
128
|
+
def self.duplicate_answer(id, ids, answers, threshold)
|
|
129
|
+
answer = answers["duplicate_of_#{id}"]
|
|
130
|
+
return nil unless answer.is_a?(Hash) && answer['confidence'].to_f >= threshold
|
|
131
|
+
|
|
132
|
+
other = answer['choice']
|
|
133
|
+
other if other != id && ids.include?(other)
|
|
134
|
+
end
|
|
135
|
+
|
|
136
|
+
def self.field(candidate, key)
|
|
137
|
+
(candidate[key] || candidate[key.to_sym]).to_s.strip
|
|
138
|
+
end
|
|
139
|
+
|
|
140
|
+
# dedupe — false when the caller merges the verdicts with the LLM ones
|
|
141
|
+
# first and drops duplicates over the whole set (JevStage).
|
|
142
|
+
def assess(context:, candidates:, dedupe: true)
|
|
143
|
+
candidates = candidates.map { |candidate| normalize(candidate) }
|
|
144
|
+
sections = diff_sections(context.diff_text)
|
|
145
|
+
asked = Asked.new(answers: {}, unfit: [], requests: 0, model: nil)
|
|
146
|
+
plan_requests(context, candidates, sections).each { |request| ask(request, asked, context, sections) }
|
|
147
|
+
assessments = candidates.map { |candidate| decide(candidate, asked.answers, asked.unfit) }
|
|
148
|
+
assessments = drop_duplicates(assessments, candidates, asked.answers) if dedupe
|
|
149
|
+
Result.new(assessments: assessments, answers: asked.answers, requests: asked.requests, model: asked.model)
|
|
150
|
+
end
|
|
151
|
+
|
|
152
|
+
# The first request as it would be sent, for --dry-run.
|
|
153
|
+
def preview(context:, candidates:)
|
|
154
|
+
candidates = candidates.map { |candidate| normalize(candidate) }
|
|
155
|
+
plan_requests(context, candidates, diff_sections(context.diff_text)).first
|
|
156
|
+
end
|
|
157
|
+
|
|
158
|
+
private
|
|
159
|
+
|
|
160
|
+
def normalize(candidate)
|
|
161
|
+
CANDIDATE_FIELDS.to_h do |field|
|
|
162
|
+
value = candidate[field] || candidate[field.to_sym]
|
|
163
|
+
[field, value.is_a?(String) ? @scrub.call(value) : value]
|
|
164
|
+
end.compact.merge('id' => self.class.field(candidate, 'id'))
|
|
165
|
+
end
|
|
166
|
+
|
|
167
|
+
# One request for all candidates when it fits, otherwise one per
|
|
168
|
+
# candidate (then without duplicate_of: a request sees one candidate).
|
|
169
|
+
# A candidate that does not fit even alone is unverifiable.
|
|
170
|
+
def plan_requests(context, candidates, sections)
|
|
171
|
+
shared = build_request(context, candidates, sections)
|
|
172
|
+
return [shared] if shared.state || candidates.size == 1
|
|
173
|
+
|
|
174
|
+
@logger.info('Jev: the candidates do not fit one request, asking about each separately')
|
|
175
|
+
candidates.map { |candidate| build_request(context, [candidate], sections) }
|
|
176
|
+
end
|
|
177
|
+
|
|
178
|
+
def build_request(context, candidates, sections)
|
|
179
|
+
ids = candidates.map { |candidate| candidate['id'] }
|
|
180
|
+
questions = candidates.map { |candidate| questions_for(candidate['id'], ids) }.reduce({}, :merge)
|
|
181
|
+
state = {
|
|
182
|
+
'task' => TASK,
|
|
183
|
+
'requirements' => context.sections,
|
|
184
|
+
'project_instructions' => @review_instructions,
|
|
185
|
+
'context_truncated' => coverage_notes(context.coverage),
|
|
186
|
+
'candidates' => candidates.to_h { |candidate| [candidate['id'], with_evidence(candidate, sections)] },
|
|
187
|
+
'diff' => context.diff_text
|
|
188
|
+
}.compact
|
|
189
|
+
Request.new(candidates: candidates, state: fit(state, state_budget(questions)), questions: questions)
|
|
190
|
+
end
|
|
191
|
+
|
|
192
|
+
def questions_for(id, ids)
|
|
193
|
+
questions = %w[real_issue enough_context version_claim severity].to_h do |name|
|
|
194
|
+
["#{name}_#{id}", question(QUESTIONS.fetch(name), id)]
|
|
195
|
+
end
|
|
196
|
+
others = ids - [id]
|
|
197
|
+
questions["duplicate_of_#{id}"] = duplicate_question(id, others) unless others.empty?
|
|
198
|
+
questions
|
|
199
|
+
end
|
|
200
|
+
|
|
201
|
+
def question(template, id)
|
|
202
|
+
{
|
|
203
|
+
'type' => template.fetch('type'),
|
|
204
|
+
'instructions' => format(template.fetch('instructions'), id: id),
|
|
205
|
+
'criteria' => template.fetch('criteria').transform_values { |text| format(text, id: id) }
|
|
206
|
+
}
|
|
207
|
+
end
|
|
208
|
+
|
|
209
|
+
def duplicate_question(id, others)
|
|
210
|
+
template = QUESTIONS.fetch('duplicate_of')
|
|
211
|
+
criteria = others.to_h { |other| [other, format(template.fetch('other'), other: other)] }
|
|
212
|
+
{
|
|
213
|
+
'type' => template.fetch('type'),
|
|
214
|
+
'instructions' => format(template.fetch('instructions'), id: id),
|
|
215
|
+
'criteria' => criteria.merge('none' => template.fetch('none'))
|
|
216
|
+
}
|
|
217
|
+
end
|
|
218
|
+
|
|
219
|
+
def state_budget(questions)
|
|
220
|
+
sizes = questions.values.map { |question| tokens(JSON.generate(question).length) }
|
|
221
|
+
room = [STATE_WITH_LONGEST_QUESTION_TOKENS - sizes.max, REQUEST_TOKENS - sizes.sum].min
|
|
222
|
+
(room * SAFETY * CHARS_PER_TOKEN).floor
|
|
223
|
+
end
|
|
224
|
+
|
|
225
|
+
def tokens(chars)
|
|
226
|
+
(chars / CHARS_PER_TOKEN).ceil
|
|
227
|
+
end
|
|
228
|
+
|
|
229
|
+
# The diff goes first, then the MR and Jira sections; the candidates and
|
|
230
|
+
# their evidence are never cut. nil — does not fit even then.
|
|
231
|
+
def fit(state, budget)
|
|
232
|
+
%w[diff requirements].each do |key|
|
|
233
|
+
break if size(state) <= budget
|
|
234
|
+
|
|
235
|
+
state = shrink(state, key, budget) if state[key]
|
|
236
|
+
end
|
|
237
|
+
size(state) <= budget ? state : nil
|
|
238
|
+
end
|
|
239
|
+
|
|
240
|
+
# Every character of the text takes at least one character of JSON, so
|
|
241
|
+
# cutting the overflow in characters is enough; the marker costs its
|
|
242
|
+
# length plus the escaped newline.
|
|
243
|
+
def shrink(state, key, budget)
|
|
244
|
+
text = state[key].to_s
|
|
245
|
+
keep = text.length - (size(state) - budget) - TRUNCATED.length - 2
|
|
246
|
+
state.merge(key => keep.positive? ? "#{text[0, keep]}\n#{TRUNCATED}" : TRUNCATED)
|
|
247
|
+
end
|
|
248
|
+
|
|
249
|
+
def size(state)
|
|
250
|
+
JSON.generate(state).length
|
|
251
|
+
end
|
|
252
|
+
|
|
253
|
+
def coverage_notes(coverage)
|
|
254
|
+
return nil if coverage.nil? || coverage.complete?
|
|
255
|
+
|
|
256
|
+
notes = coverage.truncated_sections.map { |label| "#{label} truncated" }
|
|
257
|
+
notes << "files not shown: #{coverage.files_not_shown.join(', ')}" unless coverage.files_not_shown.empty?
|
|
258
|
+
notes += coverage.files_partial.map { |file| "#{file[:path]}: #{file[:shown]} of #{file[:total]} hunks shown" }
|
|
259
|
+
notes << "diff not available: #{coverage.files_unavailable.join(', ')}" unless coverage.files_unavailable.empty?
|
|
260
|
+
notes
|
|
261
|
+
end
|
|
262
|
+
|
|
263
|
+
# The hunk the line falls into; the whole shown diff of the file when
|
|
264
|
+
# the line is unknown or the anchoring check left a note.
|
|
265
|
+
def with_evidence(candidate, sections)
|
|
266
|
+
section = sections[candidate['file'].to_s] || sections[candidate['file'].to_s.sub(%r{\A(?:\./|[ab]/)}, '')]
|
|
267
|
+
evidence = if section.nil?
|
|
268
|
+
NOT_SHOWN
|
|
269
|
+
elsif candidate['line'].is_a?(Integer) && !candidate['note']
|
|
270
|
+
hunk = section[:hunks].find { |item| item[:range].cover?(candidate['line']) }
|
|
271
|
+
hunk ? hunk[:text] : section[:text]
|
|
272
|
+
else
|
|
273
|
+
section[:text]
|
|
274
|
+
end
|
|
275
|
+
candidate.merge('evidence' => evidence)
|
|
276
|
+
end
|
|
277
|
+
|
|
278
|
+
# The shown diff by file: its whole text and every hunk with the range of
|
|
279
|
+
# new-file lines it covers (the headers as CandidateChecker reads them).
|
|
280
|
+
def diff_sections(diff_text)
|
|
281
|
+
sections = {}
|
|
282
|
+
section = nil
|
|
283
|
+
diff_text.to_s.each_line do |line|
|
|
284
|
+
if (header = line.match(CandidateChecker::FILE_HEADER))
|
|
285
|
+
section = {text: +'', hunks: []}
|
|
286
|
+
header.captures.map(&:strip).each { |path| sections[path] = section }
|
|
287
|
+
elsif section && (hunk_header = line.match(CandidateChecker::HUNK_HEADER))
|
|
288
|
+
section[:hunks] << new_hunk(hunk_header)
|
|
289
|
+
end
|
|
290
|
+
add_line(section, line) if section
|
|
291
|
+
end
|
|
292
|
+
sections
|
|
293
|
+
end
|
|
294
|
+
|
|
295
|
+
def new_hunk(header)
|
|
296
|
+
start = header[1].to_i
|
|
297
|
+
length = header[2] ? header[2].to_i : 1
|
|
298
|
+
{range: start..(start + length - 1), text: +''}
|
|
299
|
+
end
|
|
300
|
+
|
|
301
|
+
def add_line(section, line)
|
|
302
|
+
section[:text] << line
|
|
303
|
+
section[:hunks].last[:text] << line unless section[:hunks].empty?
|
|
304
|
+
end
|
|
305
|
+
|
|
306
|
+
# The estimate goes to the log next to the input_tokens Jev reports, to
|
|
307
|
+
# check CHARS_PER_TOKEN against real requests. When Jev rejects a request
|
|
308
|
+
# as too large after all (the estimate missed), its candidates are asked
|
|
309
|
+
# about one by one; one that is too large alone is unverifiable.
|
|
310
|
+
def ask(request, asked, context, sections)
|
|
311
|
+
ids = request.candidates.map { |candidate| candidate['id'] }
|
|
312
|
+
return asked.unfit.concat(ids) unless request.state
|
|
313
|
+
|
|
314
|
+
chars = size(request.state) + request.questions.values.sum { |question| JSON.generate(question).length }
|
|
315
|
+
@logger.info("Jev request estimated at ~#{tokens(chars)} tokens (#{chars} chars of state and questions)")
|
|
316
|
+
asked.requests += 1
|
|
317
|
+
result = @client.evaluate(state: request.state, questions: request.questions)
|
|
318
|
+
asked.answers.merge!(result.answers)
|
|
319
|
+
asked.model = result.model
|
|
320
|
+
rescue JevError => e
|
|
321
|
+
raise unless e.status == 422 && e.message.match?(SIZE_ERROR)
|
|
322
|
+
|
|
323
|
+
ask_separately(request, asked, context, sections, e)
|
|
324
|
+
end
|
|
325
|
+
|
|
326
|
+
def ask_separately(request, asked, context, sections, error)
|
|
327
|
+
ids = request.candidates.map { |candidate| candidate['id'] }
|
|
328
|
+
if ids.size == 1
|
|
329
|
+
@logger.warn("Jev rejected the request about #{ids.first} as too large, " \
|
|
330
|
+
"it stays unverifiable: #{error.message}")
|
|
331
|
+
return asked.unfit.concat(ids)
|
|
332
|
+
end
|
|
333
|
+
|
|
334
|
+
@logger.warn("Jev rejected the request as too large, asking about each candidate separately: #{error.message}")
|
|
335
|
+
request.candidates.each do |candidate|
|
|
336
|
+
ask(build_request(context, [candidate], sections), asked, context, sections)
|
|
337
|
+
end
|
|
338
|
+
end
|
|
339
|
+
|
|
340
|
+
def decide(candidate, answers, unfit)
|
|
341
|
+
id = candidate['id']
|
|
342
|
+
numbers = numbers(id, answers)
|
|
343
|
+
assessment = ->(decision, reason) { Assessment.new(id: id, decision: decision, reason: reason, answers: numbers) }
|
|
344
|
+
if unfit.include?(id)
|
|
345
|
+
return assessment.call(:unverifiable, 'the candidate and its evidence do not fit a Jev request')
|
|
346
|
+
end
|
|
347
|
+
if numbers[:version_claim] >= @thresholds[:version_claim]
|
|
348
|
+
return assessment.call(:reject, 'claims that a version does not exist')
|
|
349
|
+
end
|
|
350
|
+
if numbers[:enough_context] < @thresholds[:enough_context]
|
|
351
|
+
return assessment.call(:unverifiable, 'not enough context in the request')
|
|
352
|
+
end
|
|
353
|
+
|
|
354
|
+
if numbers[:real_issue] >= @thresholds[:keep_above]
|
|
355
|
+
assessment.call(:keep, 'real issue')
|
|
356
|
+
else
|
|
357
|
+
assessment.call(:reject, 'not a confirmed issue')
|
|
358
|
+
end
|
|
359
|
+
end
|
|
360
|
+
|
|
361
|
+
def numbers(id, answers)
|
|
362
|
+
noul = ->(name) { answers.dig("#{name}_#{id}", 'noul') }
|
|
363
|
+
choice = lambda do |name|
|
|
364
|
+
answer = answers["#{name}_#{id}"]
|
|
365
|
+
answer && {choice: answer['choice'], confidence: answer['confidence']}
|
|
366
|
+
end
|
|
367
|
+
{
|
|
368
|
+
real_issue: noul.call('real_issue'), enough_context: noul.call('enough_context'),
|
|
369
|
+
version_claim: noul.call('version_claim'), duplicate_of: choice.call('duplicate_of'),
|
|
370
|
+
severity: choice.call('severity')
|
|
371
|
+
}.compact
|
|
372
|
+
end
|
|
373
|
+
|
|
374
|
+
def drop_duplicates(assessments, candidates, answers)
|
|
375
|
+
kept_ids = assessments.select { |assessment| assessment.decision == :keep }.map(&:id)
|
|
376
|
+
kept = candidates.select { |candidate| kept_ids.include?(candidate['id']) }
|
|
377
|
+
dropped = self.class.duplicates(kept, answers, threshold: @thresholds[:duplicate])
|
|
378
|
+
assessments.map do |assessment|
|
|
379
|
+
winner = dropped[assessment.id]
|
|
380
|
+
next assessment unless winner
|
|
381
|
+
|
|
382
|
+
Assessment.new(id: assessment.id, decision: :reject, reason: "duplicate of #{winner}",
|
|
383
|
+
answers: assessment.answers)
|
|
384
|
+
end
|
|
385
|
+
end
|
|
386
|
+
end
|
|
387
|
+
end
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
require 'json'
|
|
3
|
+
require 'logger'
|
|
4
|
+
require_relative 'errors'
|
|
5
|
+
require_relative 'utils'
|
|
6
|
+
require_relative 'jev_critic'
|
|
7
|
+
|
|
8
|
+
module Aireview
|
|
9
|
+
# llm.jev.shadow: after the LLM Critique the same candidates go to Jev, and
|
|
10
|
+
# its decisions are logged next to the Critique verdicts. The review result
|
|
11
|
+
# and the review key do not depend on it, and a Jev failure is a warning:
|
|
12
|
+
# this is data for choosing thresholds, not a check.
|
|
13
|
+
class JevShadow
|
|
14
|
+
def initialize(config:, scrub:, logger: Logger.new($stderr), critic: nil)
|
|
15
|
+
@config = config
|
|
16
|
+
@scrub = scrub
|
|
17
|
+
@logger = logger
|
|
18
|
+
@critic = critic
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
# accepted — what the LLM Critique kept; only the ids are compared.
|
|
22
|
+
def run(context:, candidates:, accepted:)
|
|
23
|
+
critic = self.critic
|
|
24
|
+
return @logger.warn('Jev shadow skipped: JEV_API_KEY is not set') unless critic
|
|
25
|
+
|
|
26
|
+
started_at = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
27
|
+
result = critic.assess(context: context, candidates: candidates)
|
|
28
|
+
seconds = Process.clock_gettime(Process::CLOCK_MONOTONIC) - started_at
|
|
29
|
+
log(result, accepted.map { |candidate| (candidate['id'] || candidate[:id]).to_s }, seconds)
|
|
30
|
+
rescue JevError => e
|
|
31
|
+
@logger.warn("Jev shadow failed, the review is not affected: #{e.message}")
|
|
32
|
+
# The shadow is an experiment running in real review jobs: a bug in it
|
|
33
|
+
# must not cost a review either, but it must stay visible as a bug.
|
|
34
|
+
rescue StandardError => e
|
|
35
|
+
@logger.warn("Jev shadow crashed, the review is not affected: #{e.class}: #{e.message} " \
|
|
36
|
+
"(#{e.backtrace&.first})")
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
private
|
|
40
|
+
|
|
41
|
+
def critic
|
|
42
|
+
return @critic if @critic
|
|
43
|
+
return nil if Aireview::Utils.blank?(@config.jev_api_key)
|
|
44
|
+
|
|
45
|
+
@critic = JevCritic.build(config: @config, scrub: @scrub, logger: @logger)
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
def log(result, critique_kept, seconds)
|
|
49
|
+
result.assessments.each do |assessment|
|
|
50
|
+
critique = critique_kept.include?(assessment.id) ? 'keep' : 'reject'
|
|
51
|
+
@logger.info("Jev shadow #{assessment.id}: #{assessment.decision} (#{assessment.reason}; " \
|
|
52
|
+
"#{assessment.numbers}), critique: #{critique}")
|
|
53
|
+
end
|
|
54
|
+
@logger.info("Jev shadow: #{summary(result.assessments, critique_kept)} " \
|
|
55
|
+
"(model=#{result.model}, requests=#{result.requests}, #{format('%.1fs', seconds)})")
|
|
56
|
+
@logger.info("Jev shadow data: #{JSON.generate(data(result, critique_kept))}")
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
# The lines above round for reading; thresholds are chosen from this one,
|
|
60
|
+
# which keeps the numbers exactly as Jev returned them.
|
|
61
|
+
def data(result, critique_kept)
|
|
62
|
+
{
|
|
63
|
+
model: result.model,
|
|
64
|
+
candidates: result.assessments.map do |assessment|
|
|
65
|
+
{id: assessment.id, decision: assessment.decision, reason: assessment.reason,
|
|
66
|
+
critique: critique_kept.include?(assessment.id) ? 'keep' : 'reject', **assessment.answers}
|
|
67
|
+
end
|
|
68
|
+
}
|
|
69
|
+
end
|
|
70
|
+
|
|
71
|
+
def summary(assessments, critique_kept)
|
|
72
|
+
decided = assessments.reject { |assessment| assessment.decision == :unverifiable }
|
|
73
|
+
agreed = decided.count do |assessment|
|
|
74
|
+
(assessment.decision == :keep) == critique_kept.include?(assessment.id)
|
|
75
|
+
end
|
|
76
|
+
counts = %i[keep reject unverifiable].map do |decision|
|
|
77
|
+
"#{decision} #{assessments.count { |assessment| assessment.decision == decision }}"
|
|
78
|
+
end
|
|
79
|
+
"agrees with critique on #{agreed} of #{decided.size} decided candidate(s); #{counts.join(', ')}"
|
|
80
|
+
end
|
|
81
|
+
end
|
|
82
|
+
end
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
require 'logger'
|
|
3
|
+
require_relative 'errors'
|
|
4
|
+
require_relative 'jev_critic'
|
|
5
|
+
|
|
6
|
+
module Aireview
|
|
7
|
+
# The Critique stage with llm.critique.engine: jev. Jev decides keep or
|
|
8
|
+
# reject; the candidates it cannot judge (not enough context, too large)
|
|
9
|
+
# and, when Jev fails, all of them go to the LLM Critique with
|
|
10
|
+
# llm.jev.fallback: model, or are rejected / fail the run with fail.
|
|
11
|
+
# Duplicates are dropped once, over the merged verdicts of Jev and the LLM:
|
|
12
|
+
# neither side sees what the other kept.
|
|
13
|
+
class JevStage
|
|
14
|
+
# The report line: :jev (Jev decided everything), :jev_partial (some
|
|
15
|
+
# went to the LLM), :jev_failed (the LLM decided everything).
|
|
16
|
+
Outcome = Struct.new(:accepted, :note, keyword_init: true)
|
|
17
|
+
|
|
18
|
+
def initialize(config:, critic:, logger: Logger.new($stderr))
|
|
19
|
+
@config = config
|
|
20
|
+
@critic = critic
|
|
21
|
+
@logger = logger
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
# llm_critique — the LLM Critique of the pipeline for a subset of the
|
|
25
|
+
# candidates; returns the kept ones, refined.
|
|
26
|
+
def run(context:, candidates:, llm_critique:)
|
|
27
|
+
@logger.info("Pipeline critique pass started (engine=jev, model=#{@config.jev_model})")
|
|
28
|
+
begin
|
|
29
|
+
result = @critic.assess(context: context, candidates: candidates, dedupe: false)
|
|
30
|
+
rescue JevError => e
|
|
31
|
+
return fall_back(e, candidates, llm_critique)
|
|
32
|
+
end
|
|
33
|
+
result.assessments.each { |assessment| log(assessment) }
|
|
34
|
+
|
|
35
|
+
unverifiable = pick(candidates, result, :unverifiable)
|
|
36
|
+
accepted = pick(candidates, result, :keep) + judge_unverifiable(unverifiable, llm_critique)
|
|
37
|
+
accepted = drop_duplicates(in_generate_order(accepted, candidates), result.answers)
|
|
38
|
+
@logger.info("Pipeline critique pass completed with #{accepted.size} kept candidate(s) (engine=jev, " \
|
|
39
|
+
"model=#{result.model}, requests=#{result.requests})")
|
|
40
|
+
Outcome.new(accepted: accepted, note: unverifiable.empty? || fail? ? :jev : :jev_partial)
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
private
|
|
44
|
+
|
|
45
|
+
def fail?
|
|
46
|
+
@config.jev_fallback == 'fail'
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
def fall_back(error, candidates, llm_critique)
|
|
50
|
+
if fail?
|
|
51
|
+
raise JevError.new("Jev critique failed and llm.jev.fallback is fail: #{error.message}", status: error.status)
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
@logger.warn("Jev critique failed, the LLM critique takes over (llm.jev.fallback: model): #{error.message}")
|
|
55
|
+
Outcome.new(accepted: llm_critique.call(candidates), note: :jev_failed)
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
def judge_unverifiable(unverifiable, llm_critique)
|
|
59
|
+
return [] if unverifiable.empty?
|
|
60
|
+
|
|
61
|
+
ids = unverifiable.map { |candidate| id_of(candidate) }.join(', ')
|
|
62
|
+
if fail?
|
|
63
|
+
@logger.info("Critique reject #{ids}: Jev could not judge them and llm.jev.fallback is fail")
|
|
64
|
+
return []
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
@logger.info("Pipeline critique: #{ids} go to the LLM critique, Jev could not judge them " \
|
|
68
|
+
"(model=#{@config.critique_model})")
|
|
69
|
+
llm_critique.call(unverifiable)
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
def pick(candidates, result, decision)
|
|
73
|
+
ids = result.assessments.select { |assessment| assessment.decision == decision }.map(&:id)
|
|
74
|
+
candidates.select { |candidate| ids.include?(id_of(candidate)) }
|
|
75
|
+
end
|
|
76
|
+
|
|
77
|
+
def in_generate_order(accepted, candidates)
|
|
78
|
+
order = candidates.map { |candidate| id_of(candidate) }
|
|
79
|
+
accepted.sort_by { |candidate| order.index(id_of(candidate)) || order.size }
|
|
80
|
+
end
|
|
81
|
+
|
|
82
|
+
def drop_duplicates(accepted, answers)
|
|
83
|
+
dropped = JevCritic.duplicates(accepted, answers, threshold: @config.jev_thresholds[:duplicate])
|
|
84
|
+
dropped.each { |id, winner| @logger.info("Critique reject #{id}: duplicate of #{winner} (Jev)") }
|
|
85
|
+
accepted.reject { |candidate| dropped.key?(id_of(candidate)) }
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
def log(assessment)
|
|
89
|
+
message = "Critique (jev) #{assessment.decision} #{assessment.id}: #{assessment.reason} (#{assessment.numbers})"
|
|
90
|
+
assessment.decision == :keep ? @logger.debug(message) : @logger.info(message)
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
def id_of(candidate)
|
|
94
|
+
(candidate['id'] || candidate[:id]).to_s.strip
|
|
95
|
+
end
|
|
96
|
+
end
|
|
97
|
+
end
|
data/lib/aireview/llm_client.rb
CHANGED
|
@@ -30,7 +30,7 @@ module Aireview
|
|
|
30
30
|
stage = prompt.stage.to_s
|
|
31
31
|
model = candidate.model
|
|
32
32
|
@logger.info("LLM #{stage} request started (model=#{model}, temperature=#{prompt.temperature})")
|
|
33
|
-
chat = build_chat(context: context(stage, candidate
|
|
33
|
+
chat = build_chat(context: context(stage, candidate, key, key_index), stage: stage,
|
|
34
34
|
model: model, provider: candidate.provider)
|
|
35
35
|
chat = configure_reasoning(chat: chat, model: model, provider: candidate.provider)
|
|
36
36
|
.with_temperature(prompt.temperature.to_f)
|
|
@@ -99,18 +99,19 @@ module Aireview
|
|
|
99
99
|
context.chat(model: model, provider: provider.to_sym, assume_model_exists: true)
|
|
100
100
|
end
|
|
101
101
|
|
|
102
|
-
# A RubyLLM context per stage,
|
|
103
|
-
#
|
|
104
|
-
|
|
105
|
-
|
|
102
|
+
# A RubyLLM context per stage, key source (a provider's API or a server
|
|
103
|
+
# of your own) and key index: switching the key is another context, not
|
|
104
|
+
# an edit of the global config.
|
|
105
|
+
def context(stage, candidate, key, key_index)
|
|
106
|
+
@contexts[[stage, candidate.key_source, key_index]] ||= build_context(candidate, key)
|
|
106
107
|
end
|
|
107
108
|
|
|
108
|
-
def build_context(
|
|
109
|
+
def build_context(candidate, api_key)
|
|
109
110
|
RubyLLM.context do |ruby_config|
|
|
110
111
|
configure_http_proxy(ruby_config)
|
|
111
112
|
ruby_config.request_timeout = @config.llm_timeout.to_f
|
|
112
113
|
ruby_config.max_retries = 0
|
|
113
|
-
configure_provider(ruby_config, provider, api_key)
|
|
114
|
+
configure_provider(ruby_config, candidate.provider.to_s, api_key, candidate.api_base)
|
|
114
115
|
end
|
|
115
116
|
end
|
|
116
117
|
|
|
@@ -120,27 +121,30 @@ module Aireview
|
|
|
120
121
|
ruby_config.http_proxy = @config.llm_http_proxy
|
|
121
122
|
end
|
|
122
123
|
|
|
123
|
-
|
|
124
|
+
# api_base — the model's own server; without it the provider's API,
|
|
125
|
+
# or LLM_API_BASE for the providers that always took it.
|
|
126
|
+
def configure_provider(ruby_config, provider, api_key, api_base)
|
|
124
127
|
case provider
|
|
125
128
|
when 'gemini', 'openai', 'openrouter'
|
|
126
|
-
configure_remote_provider(ruby_config, provider, api_key)
|
|
129
|
+
configure_remote_provider(ruby_config, provider, api_key, api_base || @config.llm_api_base)
|
|
127
130
|
when 'anthropic'
|
|
128
131
|
ruby_config.anthropic_api_key = api_key
|
|
132
|
+
ruby_config.anthropic_api_base = api_base if api_base
|
|
129
133
|
when 'ollama'
|
|
130
|
-
ruby_config.ollama_api_base = @config.ollama_api_base
|
|
134
|
+
ruby_config.ollama_api_base = api_base || @config.ollama_api_base
|
|
131
135
|
else
|
|
132
136
|
raise ConfigError, "Unsupported LLM provider: #{provider.inspect}"
|
|
133
137
|
end
|
|
134
138
|
end
|
|
135
139
|
|
|
136
140
|
# RubyLLM 2 sends OpenAI requests to the Responses API; a compatible
|
|
137
|
-
# server
|
|
138
|
-
def configure_remote_provider(ruby_config, provider, api_key)
|
|
141
|
+
# server usually has Chat Completions only.
|
|
142
|
+
def configure_remote_provider(ruby_config, provider, api_key, api_base)
|
|
139
143
|
ruby_config.public_send("#{provider}_api_key=", api_key)
|
|
140
144
|
ruby_config.openai_protocol = :chat_completions if provider == 'openai'
|
|
141
|
-
return unless Aireview::Utils.present?(
|
|
145
|
+
return unless Aireview::Utils.present?(api_base)
|
|
142
146
|
|
|
143
|
-
ruby_config.public_send("#{provider}_api_base=",
|
|
147
|
+
ruby_config.public_send("#{provider}_api_base=", api_base)
|
|
144
148
|
end
|
|
145
149
|
end
|
|
146
150
|
end
|
data/lib/aireview/llm_router.rb
CHANGED
|
@@ -34,7 +34,7 @@ module Aireview
|
|
|
34
34
|
|
|
35
35
|
Slot = Struct.new(:candidate, :index)
|
|
36
36
|
# The key the next model of the same provider continues from.
|
|
37
|
-
Carry = Struct.new(:
|
|
37
|
+
Carry = Struct.new(:key_source, :key_index)
|
|
38
38
|
Delay = Struct.new(:seconds, :source)
|
|
39
39
|
|
|
40
40
|
SWITCH_HINT = 'Try again later or switch model via --generate-model/--critique-model.'
|
|
@@ -213,7 +213,7 @@ module Aireview
|
|
|
213
213
|
log_switch(stage, route, visits)
|
|
214
214
|
status, response = try_route(stage, route, visits, &request)
|
|
215
215
|
return [:ok, remember(stage, route, response)] if status == :ok
|
|
216
|
-
return [:next_model, Carry.new(route.candidate.
|
|
216
|
+
return [:next_model, Carry.new(route.candidate.key_source, route.key_index)] if status == :next_model
|
|
217
217
|
end
|
|
218
218
|
state(slot.candidate).exclude('daily quota exhausted on every key') unless tried
|
|
219
219
|
[:next_model, carry]
|
|
@@ -224,7 +224,7 @@ module Aireview
|
|
|
224
224
|
# is bound to "key + model", a failure on one model does not write the
|
|
225
225
|
# key off for another.
|
|
226
226
|
def candidate_routes(stage, slot, carry)
|
|
227
|
-
keys = @config.
|
|
227
|
+
keys = @config.candidate_api_keys(slot.candidate)
|
|
228
228
|
routes = keys.each_with_index.map do |key, key_index|
|
|
229
229
|
Route.new(candidate: slot.candidate, candidate_index: slot.index, key: key,
|
|
230
230
|
key_index: key_index, key_count: keys.size)
|
|
@@ -233,7 +233,7 @@ module Aireview
|
|
|
233
233
|
end
|
|
234
234
|
|
|
235
235
|
def start_key(stage, slot, carry)
|
|
236
|
-
return carry.key_index if carry && carry.
|
|
236
|
+
return carry.key_index if carry && carry.key_source == slot.candidate.key_source
|
|
237
237
|
|
|
238
238
|
cursor_candidate, cursor_key = @cursor.fetch(stage, [0, 0])
|
|
239
239
|
cursor_candidate == slot.index ? cursor_key : 0
|