aireview 0.2.1 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,15 +1,21 @@
1
1
  # frozen_string_literal: true
2
2
  require 'json'
3
3
  require_relative 'errors'
4
+ require_relative 'stages'
4
5
  require_relative 'context_builder'
6
+ require_relative 'candidate_checker'
7
+ require_relative 'result_parser'
5
8
  require_relative 'review_renderer'
6
9
  require_relative 'review_schemas'
7
10
  require_relative 'reviewer'
8
11
 
9
12
  module Aireview
13
+ # A review run: context → Generate → anchoring check against the diff →
14
+ # Critique → report. Invalid JSON is repaired once by the same model; when
15
+ # the repair is invalid too, the stage restarts on another model with the
16
+ # original request.
10
17
  class ReviewPipeline
11
- class SchemaError < StandardError
12
- end
18
+ SchemaError = ResultParser::SchemaError
13
19
 
14
20
  REPAIR_SYSTEM_PROMPT = <<~PROMPT.strip.freeze
15
21
  You fix invalid JSON produced by another LLM call.
@@ -23,6 +29,7 @@ module Aireview
23
29
 
24
30
  def initialize(config:, reviewer: nil, context_builder: nil, logger: Logger.new($stderr))
25
31
  @config = config
32
+ @parser = ResultParser.new
26
33
  @reviewer = reviewer || Reviewer.new(config: config, logger: logger)
27
34
  @context_builder = context_builder || ContextBuilder.new(config: config, logger: logger)
28
35
  @logger = logger
@@ -37,29 +44,31 @@ module Aireview
37
44
  )
38
45
  generate_prompt = @context_builder.build_generate_prompt(context)
39
46
  @logger.info("Pipeline generate pass started (model=#{@config.generate_model})")
40
- candidates_raw = @reviewer.generate(**generate_prompt)
41
- generate_result = parse_with_repair(
42
- raw: candidates_raw,
43
- kind: 'generate result',
44
- expected: :generate,
45
- repair_stage: :generate
46
- )
47
+ generate_result = run_stage('generate') do
48
+ parse_with_repair(
49
+ raw: @reviewer.generate(**generate_prompt),
50
+ kind: 'generate result',
51
+ expected: 'generate',
52
+ repair_stage: 'generate'
53
+ )
54
+ end
47
55
  summary = generate_result['summary']
48
56
  candidates = Array(generate_result['candidates'])
49
- @logger.info("Pipeline generate pass completed with #{candidates.size} candidate(s)")
57
+ @logger.info("Pipeline generate pass completed with #{candidates.size} candidate(s) " \
58
+ "(model=#{@reviewer.answered_model('generate')})")
59
+ candidates = check_candidates(context: context, changes: changes, candidates: candidates)
50
60
 
51
- accepted = if critique
52
- @logger.info("Pipeline critique pass started (model=#{@config.critique_model})")
53
- critique_candidates(context: context, candidates: candidates)
54
- else
55
- @logger.info('Pipeline critique pass skipped')
56
- candidates
57
- end
61
+ accepted = critique ? maybe_critique(context: context, candidates: candidates) : skip_critique(candidates)
58
62
 
59
63
  @logger.info("Pipeline finished with #{accepted.size} accepted finding(s)")
60
64
 
61
- renderer = ReviewRenderer.new(language: @config.review_language)
62
- renderer.render(accepted, summary: summary, coverage: context.coverage)
65
+ ReviewRenderer.new(language: @config.review_language).render(
66
+ accepted,
67
+ summary: summary,
68
+ coverage: context.coverage,
69
+ fallback_models: @reviewer.fallback_models,
70
+ critique_weaker: critique && @reviewer.critique_weaker?
71
+ )
63
72
  end
64
73
 
65
74
  def dry_run_prompts(merge_request:, changes:, jira_issue: nil, critique: true)
@@ -83,6 +92,15 @@ module Aireview
83
92
  generate_temperature: @config.generate_temperature,
84
93
  critique_model: @config.critique_model,
85
94
  critique_temperature: @config.critique_temperature,
95
+ generate_fallbacks: @config.fallback_names('generate'),
96
+ critique_fallbacks: critique ? @config.fallback_names('critique') : [],
97
+ sources: setting_sources(critique),
98
+ config_paths: @config.layer_paths,
99
+ warnings: @config.warnings,
100
+ critique_rule: critique ? @config.routing.rule : nil,
101
+ api_keys: @config.api_key_counts(critique ? STAGES : ['generate']),
102
+ time_budget: @config.llm_time_budget,
103
+ overloaded_quarantine: @config.overloaded_quarantine,
86
104
  coverage: context.coverage,
87
105
  sizes: context.sizes
88
106
  }
@@ -90,20 +108,76 @@ module Aireview
90
108
 
91
109
  private
92
110
 
111
+ # Where the model, provider and reserves of a stage came from, for --dry-run.
112
+ def setting_sources(critique)
113
+ (critique ? %w[generate critique] : %w[generate]).to_h do |stage|
114
+ [stage.to_sym, {
115
+ model: @config.stage_model_source(stage),
116
+ provider: @config.stage_provider_source(stage),
117
+ fallbacks: @config.stage_fallbacks_source(stage)
118
+ }]
119
+ end
120
+ end
121
+
122
+ # A stage is a request, parsing and one repair by the same model. An
123
+ # invalid result after the repair, like a repair with no requests left,
124
+ # excludes the model for the stage, and the stage starts over on the next
125
+ # one — with the original request. API errors pass through: the router
126
+ # handles them with reserves, and exhausted routes are exhausted for a
127
+ # restart too.
128
+ def run_stage(stage)
129
+ loop do
130
+ return yield
131
+ rescue ParseError => e
132
+ reason = "#{e.is_a?(RepairImpossibleError) ? 'repair impossible' : 'invalid result'}: #{e.message}"
133
+ excluded = @reviewer.exclude_answered_model(stage: stage, reason: reason)
134
+ raise unless excluded
135
+
136
+ @logger.warn("Pipeline #{stage}: restarting on the next model after #{excluded} (#{e.message})")
137
+ end
138
+ end
139
+
140
+ # Anchoring is checked against the diff the model saw, before Critique:
141
+ # it gets the notes, the report gets the reset line and the missing-quote mark.
142
+ def check_candidates(context:, changes:, candidates:)
143
+ CandidateChecker.new(
144
+ changes: changes,
145
+ diff_text: context.diff_text,
146
+ coverage: context.coverage,
147
+ logger: @logger
148
+ ).check(candidates)
149
+ end
150
+
151
+ def skip_critique(candidates, reason = nil)
152
+ @logger.info(['Pipeline critique pass skipped', reason].compact.join(': '))
153
+ candidates
154
+ end
155
+
156
+ # Critique has nothing to filter without candidates: the LLM request
157
+ # would waste quota and time.
158
+ def maybe_critique(context:, candidates:)
159
+ return skip_critique(candidates, 'no candidates') if candidates.empty?
160
+
161
+ @logger.info("Pipeline critique pass started (model=#{@config.critique_model})")
162
+ critique_candidates(context: context, candidates: candidates)
163
+ end
164
+
93
165
  def critique_candidates(context:, candidates:)
94
166
  candidates_json = JSON.pretty_generate(candidates)
95
167
  candidates_by_id = index_candidates_by_id(candidates)
96
168
  critique_prompt = @context_builder.build_critique_prompt(context, candidates_json: candidates_json)
97
- critique_raw = @reviewer.critique(**critique_prompt)
98
- critique_result = parse_with_repair(
99
- raw: critique_raw,
100
- kind: 'critique result',
101
- expected: :critique,
102
- repair_stage: :critique,
103
- critique_candidate_ids: candidates_by_id.keys
104
- )
169
+ critique_result = run_stage('critique') do
170
+ parse_with_repair(
171
+ raw: @reviewer.critique(**critique_prompt),
172
+ kind: 'critique result',
173
+ expected: 'critique',
174
+ repair_stage: 'critique',
175
+ critique_candidate_ids: candidates_by_id.keys
176
+ )
177
+ end
105
178
  verdicts = Array(critique_result['verdicts'])
106
- @logger.info("Pipeline critique pass completed with #{verdicts.size} verdict(s)")
179
+ @logger.info("Pipeline critique pass completed with #{verdicts.size} verdict(s) " \
180
+ "(model=#{@reviewer.answered_model('critique')})")
107
181
  apply_critique_verdicts(
108
182
  verdicts: verdicts,
109
183
  candidates_by_id: candidates_by_id
@@ -156,108 +230,11 @@ module Aireview
156
230
  end
157
231
 
158
232
  def parse_expected_result(raw, expected, critique_candidate_ids: nil)
159
- parsed = raw.is_a?(Hash) ? raw : JSON.parse(strip_code_fences(raw.to_s))
160
-
161
- case expected
162
- when :generate
163
- parsed = normalize_generate_result(parsed)
164
- when :critique
165
- parsed = normalize_critique_result(parsed, critique_candidate_ids: critique_candidate_ids)
166
- else
167
- raise ArgumentError, "Unknown expected JSON schema: #{expected.inspect}"
168
- end
169
-
170
- parsed
171
- end
172
-
173
- def strip_code_fences(text)
174
- stripped = text.to_s.strip
175
- return stripped unless stripped.start_with?('```')
176
-
177
- stripped
178
- .sub(/\A```[[:alnum:]_-]*[ \t]*\r?\n?/, '')
179
- .sub(/\r?\n?```[ \t]*\z/, '')
180
- .strip
181
- end
182
-
183
- def normalize_generate_result(parsed)
184
- parsed = {'summary' => nil, 'candidates' => parsed} if parsed.is_a?(Array)
185
- validate_generate_result_shape!(parsed)
186
- parsed['summary'] = nil unless parsed.key?('summary')
187
- candidate_ids = parsed['candidates'].map { |candidate| normalize_id(value(candidate, 'id')) }
188
- validate_identifiers!(
189
- candidate_ids,
190
- missing_message: 'each generate candidate must include a non-empty id',
191
- duplicate_prefix: 'duplicate generate candidate ids'
192
- )
193
-
194
- parsed
195
- end
196
-
197
- def normalize_critique_result(parsed, critique_candidate_ids:)
198
- validate_critique_result_shape!(parsed)
199
- verdicts = parsed['verdicts']
200
- verdict_ids = verdicts.map { |verdict| normalize_id(value(verdict, 'id')) }
201
- validate_identifiers!(
202
- verdict_ids,
203
- missing_message: 'each verdict must include a non-empty id',
204
- duplicate_prefix: 'duplicate verdict ids'
205
- )
206
- validate_expected_verdict_ids!(verdict_ids, critique_candidate_ids)
207
- verdicts.each { |verdict| validate_verdict!(verdict) }
208
-
209
- parsed
210
- end
211
-
212
- def validate_generate_result_shape!(parsed)
213
- valid_shape = parsed.is_a?(Hash) && parsed['candidates'].is_a?(Array)
214
- raise SchemaError, 'expected an object with summary and candidates array' unless valid_shape
215
- raise SchemaError, 'each generate candidate must be an object' unless parsed['candidates'].all?(Hash)
216
- end
217
-
218
- def validate_critique_result_shape!(parsed)
219
- valid_shape = parsed.is_a?(Hash) && parsed['verdicts'].is_a?(Array)
220
- raise SchemaError, 'expected an object with verdicts array' unless valid_shape
221
- raise SchemaError, 'each critique verdict must be an object' unless parsed['verdicts'].all?(Hash)
222
- end
223
-
224
- def validate_identifiers!(identifiers, missing_message:, duplicate_prefix:)
225
- raise SchemaError, missing_message unless identifiers.all?
226
-
227
- duplicate_ids = identifiers.group_by(&:itself).select { |_, ids| ids.size > 1 }.keys
228
- return if duplicate_ids.empty?
229
-
230
- raise SchemaError, "#{duplicate_prefix}: #{duplicate_ids.join(', ')}"
231
- end
232
-
233
- def validate_expected_verdict_ids!(verdict_ids, expected_ids)
234
- return unless expected_ids
235
-
236
- unknown_ids = verdict_ids - expected_ids
237
- missing_ids = expected_ids - verdict_ids
238
- raise SchemaError, "unknown verdict ids: #{unknown_ids.join(', ')}" unless unknown_ids.empty?
239
- raise SchemaError, "missing verdict ids: #{missing_ids.join(', ')}" unless missing_ids.empty?
240
- end
241
-
242
- def validate_verdict!(verdict)
243
- id = normalize_id(value(verdict, 'id'))
244
- decision = normalize_decision(value(verdict, 'decision'))
245
- raise SchemaError, "invalid verdict decision for #{id}" unless %w[keep reject].include?(decision)
246
-
247
- refinement = value(verdict, 'refinement')
248
- raise SchemaError, "reject verdict cannot include refinement for #{id}" if invalid_refinement?(verdict, decision)
249
- return if refinement.nil?
250
- return if refinement.is_a?(Hash)
251
-
252
- raise SchemaError, "refinement must be an object for #{id}"
253
- end
254
-
255
- def invalid_refinement?(verdict, decision)
256
- refinement_key?(verdict) && decision != 'keep'
233
+ @parser.parse(raw, expected: expected, critique_candidate_ids: critique_candidate_ids)
257
234
  end
258
235
 
259
236
  def repair_json(raw:, kind:, expected:, stage:, critique_candidate_ids: nil)
260
- schema = expected == :critique ? ReviewSchemas.critique : ReviewSchemas.generate
237
+ schema = expected == 'critique' ? ReviewSchemas.critique : ReviewSchemas.generate
261
238
  user_prompt = <<~PROMPT
262
239
  The previous #{kind} response was invalid.
263
240
 
@@ -270,19 +247,28 @@ module Aireview
270
247
  Invalid response:
271
248
  #{raw}
272
249
  PROMPT
273
- if stage == :critique && critique_candidate_ids
250
+ if stage == 'critique' && critique_candidate_ids
274
251
  user_prompt << "\nExpected candidate ids: #{critique_candidate_ids.join(', ')}\n"
275
252
  end
276
253
 
277
254
  @logger.info("Pipeline #{stage} repair started for #{kind}")
278
- prompt = @context_builder.check_stage_size!(stage, REPAIR_SYSTEM_PROMPT, user_prompt)
279
- if stage == :critique
280
- @reviewer.critique(**prompt)
255
+ prompt = repair_prompt(stage, user_prompt)
256
+ if stage == 'critique'
257
+ @reviewer.critique(**prompt, pinned: true)
281
258
  else
282
- @reviewer.generate(**prompt)
259
+ @reviewer.generate(**prompt, pinned: true)
283
260
  end
284
261
  end
285
262
 
263
+ # A repair that does not fit the stage limit is an invalid result of this
264
+ # model, not a size error of the original request: the stage moves to
265
+ # the next model with the original prompt.
266
+ def repair_prompt(stage, user_prompt)
267
+ @context_builder.check_stage_size!(stage, REPAIR_SYSTEM_PROMPT, user_prompt)
268
+ rescue ContextBudgetError => e
269
+ raise ParseError, "repair request does not fit the stage limit: #{e.message}"
270
+ end
271
+
286
272
  def index_candidates_by_id(candidates)
287
273
  candidates.each_with_object({}) do |candidate, result|
288
274
  next unless candidate.is_a?(Hash)
@@ -345,21 +331,14 @@ module Aireview
345
331
  hash[key] || hash[key.to_sym]
346
332
  end
347
333
 
348
- def normalize_id(value)
349
- presence(value)
350
- end
351
-
352
334
  def normalize_decision(value)
353
335
  value.to_s.strip.downcase
354
336
  end
355
337
 
356
- def refinement_key?(verdict)
357
- verdict.key?('refinement') || verdict.key?(:refinement)
358
- end
359
-
360
338
  def presence(value)
361
339
  string = value.to_s.strip
362
340
  string.empty? ? nil : string
363
341
  end
342
+ alias normalize_id presence
364
343
  end
365
344
  end
@@ -44,7 +44,10 @@ module Aireview
44
44
  sections_truncated: 'sections truncated',
45
45
  hunks_of: 'hunks shown',
46
46
  diff_unavailable: 'diff not available',
47
- section_list: 'Truncated sections'
47
+ section_list: 'Truncated sections',
48
+ fallback_used: 'Fallback model used',
49
+ critique_weaker: 'Critique ran on a model weaker than Generate: the findings were checked less strictly.',
50
+ quote_missing: 'quote not found in the diff'
48
51
  },
49
52
  'ru' => {
50
53
  summary: 'Сводка',
@@ -68,7 +71,10 @@ module Aireview
68
71
  sections_truncated: 'секций усечено',
69
72
  hunks_of: 'хунков показано',
70
73
  diff_unavailable: 'дифф недоступен',
71
- section_list: 'Усечённые секции'
74
+ section_list: 'Усечённые секции',
75
+ fallback_used: 'Использована резервная модель',
76
+ critique_weaker: 'Критика выполнена моделью слабее generate: замечания проверены менее строго.',
77
+ quote_missing: 'цитата не найдена в диффе'
72
78
  }
73
79
  }.freeze
74
80
 
@@ -76,14 +82,12 @@ module Aireview
76
82
  @labels = LABELS.fetch(language.to_s) { LABELS.fetch(DEFAULT_LANGUAGE) }
77
83
  end
78
84
 
79
- # coverage: факты усечения контекста от пайплайна, не текст модели.
80
- # result по-прежнему про найденные проблемы; неполнота покрытия
81
- # дописывается рядом с ним, чтобы строка результата не читалась как
82
- # «проверено всё».
83
- def render(accepted, summary:, coverage: nil)
84
- findings = sorted_findings(Array(accepted)).first(TOTAL_FINDINGS_LIMIT)
85
- mismatches = findings.select { |finding| category(finding) == 'task_mismatch' }.first(MISMATCH_LIMIT)
86
- important = findings.select { |finding| important_finding?(finding) }.first(IMPORTANT_LIMIT)
85
+ # coverage: the truncation facts from the pipeline, not the model's text.
86
+ # result is still about the findings; incomplete coverage is written
87
+ # next to it so that the result line does not read as "everything was
88
+ # checked".
89
+ def render(accepted, summary:, coverage: nil, fallback_models: {}, critique_weaker: false)
90
+ mismatches, important = select_findings(Array(accepted))
87
91
  result = mismatches.empty? && important.empty? ? 'ok' : 'needs attention'
88
92
 
89
93
  <<~MARKDOWN.rstrip
@@ -102,13 +106,25 @@ module Aireview
102
106
  ## #{label(:result)}
103
107
 
104
108
  #{result}#{partial_note(coverage)}
105
- #{coverage_block(coverage)}
109
+ #{coverage_block(coverage)}#{fallback_note(fallback_models)}#{weaker_note(critique_weaker)}
106
110
  #{label(:disclaimer)}
107
111
  MARKDOWN
108
112
  end
109
113
 
110
114
  private
111
115
 
116
+ # First select what is shown at all, then the section limits and only
117
+ # then the total limit: a finding that cannot be shown because of its
118
+ # category or a section limit must not take a slot in the total.
119
+ def select_findings(accepted)
120
+ sorted = sorted_findings(accepted)
121
+ mismatches = sorted.select { |finding| category(finding) == 'task_mismatch' }.first(MISMATCH_LIMIT)
122
+ important = sorted.select { |finding| important_finding?(finding) }.first(IMPORTANT_LIMIT)
123
+ shown = sorted.select { |finding| mismatches.include?(finding) || important.include?(finding) }
124
+ .first(TOTAL_FINDINGS_LIMIT)
125
+ [mismatches & shown, important & shown]
126
+ end
127
+
112
128
  def sorted_findings(findings)
113
129
  findings
114
130
  .grep(Hash)
@@ -167,13 +183,28 @@ module Aireview
167
183
  "\n## #{label(:not_reviewed)}\n\n#{lines.join("\n")}\n"
168
184
  end
169
185
 
186
+ # A key switch stays in the logs; a model switch is visible to the
187
+ # reader, because a fallback model may review less well than the primary.
188
+ def weaker_note(critique_weaker)
189
+ return '' unless critique_weaker
190
+
191
+ "\n#{label(:critique_weaker)}\n"
192
+ end
193
+
194
+ def fallback_note(fallback_models)
195
+ return '' if fallback_models.nil? || fallback_models.empty?
196
+
197
+ used = fallback_models.map { |stage, model| "#{stage} — #{model}" }.join(', ')
198
+ "\n#{label(:fallback_used)}: #{used}.\n"
199
+ end
200
+
170
201
  def location(finding)
171
202
  file = presence(value(finding, 'file'))
172
203
  line = value(finding, 'line')
173
204
  return label(:not_specified) unless file
174
- return file if line.nil? || line.to_s.empty?
175
205
 
176
- "#{file}:#{line}"
206
+ location = line.to_s.empty? ? file : "#{file}:#{line}"
207
+ value(finding, 'quote_missing') ? "#{location} (#{label(:quote_missing)})" : location
177
208
  end
178
209
 
179
210
  def label(key)