aireview 2.0.0 → 2.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
  require 'json'
3
3
  require_relative 'errors'
4
+ require_relative 'utils'
4
5
  require_relative 'stages'
5
6
  require_relative 'context_builder'
6
7
  require_relative 'candidate_checker'
@@ -8,12 +9,15 @@ require_relative 'result_parser'
8
9
  require_relative 'review_renderer'
9
10
  require_relative 'review_schemas'
10
11
  require_relative 'reviewer'
12
+ require_relative 'jev_shadow'
13
+ require_relative 'jev_stage'
14
+ require_relative 'dry_run_prompts'
11
15
 
12
16
  module Aireview
13
17
  # A review run: context → Generate → anchoring check against the diff →
14
- # Critique → report. Invalid JSON is repaired once by the same model; when
15
- # the repair is invalid too, the stage restarts on another model with the
16
- # original request.
18
+ # Critique (an LLM, or Jev with llm.critique.engine: jev) → report. Invalid
19
+ # JSON is repaired once by the same model; when the repair is invalid too,
20
+ # the stage restarts on another model with the original request.
17
21
  class ReviewPipeline
18
22
  SchemaError = ResultParser::SchemaError
19
23
 
@@ -23,15 +27,22 @@ module Aireview
23
27
  Do not use markdown, code fences, comments, or text outside JSON.
24
28
  Do not add new review findings.
25
29
  PROMPT
26
- DRY_RUN_CANDIDATES_JSON = '[{"id":"C1","file":"path/from/diff.rb","line":1,' \
27
- '"quoted_code":"...","problem":"...","why":"...","suggestion":"...",' \
28
- '"category":"bug","severity":"major"}]'
29
-
30
- def initialize(config:, reviewer: nil, context_builder: nil, logger: Logger.new($stderr))
30
+ # Finish reasons after which an unparsable answer gets no repair.
31
+ CUT_OFF_REASONS = {
32
+ max_tokens: 'cut off at the output limit (max_tokens)',
33
+ content_filter: 'blocked by the provider (content_filter)'
34
+ }.freeze
35
+
36
+ # jev — jev_shadow: and jev_stage: for tests; by default they are built
37
+ # from the config, the stage only when Jev is the engine.
38
+ def initialize(config:, reviewer: nil, context_builder: nil, logger: Logger.new($stderr), **jev)
31
39
  @config = config
32
40
  @parser = ResultParser.new
33
41
  @reviewer = reviewer || Reviewer.new(config: config, logger: logger)
34
42
  @context_builder = context_builder || ContextBuilder.new(config: config, logger: logger)
43
+ @jev_shadow = jev[:jev_shadow] ||
44
+ JevShadow.new(config: config, scrub: @context_builder.method(:scrub_text), logger: logger)
45
+ @jev_stage = jev[:jev_stage]
35
46
  @logger = logger
36
47
  end
37
48
 
@@ -58,7 +69,11 @@ module Aireview
58
69
  "(model=#{@reviewer.answered_model('generate')})")
59
70
  candidates = check_candidates(context: context, changes: changes, candidates: candidates)
60
71
 
61
- accepted = critique ? maybe_critique(context: context, candidates: candidates) : skip_critique(candidates)
72
+ accepted, jev_note = if critique
73
+ maybe_critique(context: context, candidates: candidates)
74
+ else
75
+ skip_critique(candidates)
76
+ end
62
77
 
63
78
  @logger.info("Pipeline finished with #{accepted.size} accepted finding(s)")
64
79
 
@@ -67,58 +82,19 @@ module Aireview
67
82
  summary: summary,
68
83
  coverage: context.coverage,
69
84
  fallback_models: @reviewer.fallback_models,
70
- critique_weaker: critique && @reviewer.critique_weaker?
85
+ critique_weaker: critique && @reviewer.critique_weaker?,
86
+ jev_note: jev_note
71
87
  )
72
88
  end
73
89
 
90
+ # The prompts of the run and everything --dry-run shows; see DryRunPrompts.
74
91
  def dry_run_prompts(merge_request:, changes:, jira_issue: nil, critique: true)
75
- @config.require_models!
76
-
77
- context = @context_builder.prepare(
78
- merge_request: merge_request,
79
- changes: changes,
80
- jira_issue: jira_issue,
81
- critique: critique
82
- )
83
- generate_prompt = @context_builder.build_generate_prompt(context)
84
- critique_prompt = if critique
85
- @context_builder.build_critique_prompt(context, candidates_json: DRY_RUN_CANDIDATES_JSON)
86
- end
87
-
88
- {
89
- generate_prompt: generate_prompt,
90
- critique_prompt: critique_prompt,
91
- generate_model: @config.generate_model,
92
- generate_temperature: @config.generate_temperature,
93
- critique_model: @config.critique_model,
94
- critique_temperature: @config.critique_temperature,
95
- generate_fallbacks: @config.fallback_names('generate'),
96
- critique_fallbacks: critique ? @config.fallback_names('critique') : [],
97
- sources: setting_sources(critique),
98
- config_paths: @config.layer_paths,
99
- warnings: @config.warnings,
100
- critique_rule: critique ? @config.routing.rule : nil,
101
- api_keys: @config.api_key_counts(critique ? STAGES : ['generate']),
102
- time_budget: @config.llm_time_budget,
103
- overloaded_quarantine: @config.overloaded_quarantine,
104
- coverage: context.coverage,
105
- sizes: context.sizes
106
- }
92
+ DryRunPrompts.new(config: @config, context_builder: @context_builder, logger: @logger)
93
+ .build(merge_request: merge_request, changes: changes, jira_issue: jira_issue, critique: critique)
107
94
  end
108
95
 
109
96
  private
110
97
 
111
- # Where the model, provider and reserves of a stage came from, for --dry-run.
112
- def setting_sources(critique)
113
- (critique ? %w[generate critique] : %w[generate]).to_h do |stage|
114
- [stage.to_sym, {
115
- model: @config.stage_model_source(stage),
116
- provider: @config.stage_provider_source(stage),
117
- fallbacks: @config.stage_fallbacks_source(stage)
118
- }]
119
- end
120
- end
121
-
122
98
  # A stage is a request, parsing and one repair by the same model. An
123
99
  # invalid result after the repair, like a repair with no requests left,
124
100
  # excludes the model for the stage, and the stage starts over on the next
@@ -148,18 +124,32 @@ module Aireview
148
124
  ).check(candidates)
149
125
  end
150
126
 
127
+ # Returns the candidates and no report note, like every critique path.
151
128
  def skip_critique(candidates, reason = nil)
152
129
  @logger.info(['Pipeline critique pass skipped', reason].compact.join(': '))
153
- candidates
130
+ [candidates, nil]
154
131
  end
155
132
 
156
133
  # Critique has nothing to filter without candidates: the LLM request
157
- # would waste quota and time.
134
+ # would waste quota and time. Returns [accepted, note for the report].
158
135
  def maybe_critique(context:, candidates:)
159
136
  return skip_critique(candidates, 'no candidates') if candidates.empty?
137
+ return jev_critique(context: context, candidates: candidates) if @config.jev_critique?
160
138
 
161
139
  @logger.info("Pipeline critique pass started (model=#{@config.critique_model})")
162
- critique_candidates(context: context, candidates: candidates)
140
+ accepted = critique_candidates(context: context, candidates: candidates)
141
+ @jev_shadow.run(context: context, candidates: candidates, accepted: accepted) if @config.jev_shadow?
142
+ [accepted, nil]
143
+ end
144
+
145
+ def jev_critique(context:, candidates:)
146
+ @jev_stage ||= JevStage.new(
147
+ config: @config, logger: @logger,
148
+ critic: JevCritic.build(config: @config, scrub: @context_builder.method(:scrub_text), logger: @logger)
149
+ )
150
+ llm_critique = ->(subset) { critique_candidates(context: context, candidates: subset) }
151
+ outcome = @jev_stage.run(context: context, candidates: candidates, llm_critique: llm_critique)
152
+ [outcome.accepted, outcome.note]
163
153
  end
164
154
 
165
155
  def critique_candidates(context:, candidates:)
@@ -214,6 +204,7 @@ module Aireview
214
204
  def parse_string_with_repair(raw:, kind:, expected:, repair_stage:, critique_candidate_ids: nil)
215
205
  parse_expected_result(raw, expected, critique_candidate_ids: critique_candidate_ids)
216
206
  rescue JSON::ParserError, SchemaError => e
207
+ raise_if_cut_off(stage: repair_stage, kind: kind, error: e)
217
208
  @logger.warn("Invalid #{kind} JSON, requesting one repair: #{e.message}")
218
209
  repaired = repair_json(
219
210
  raw: raw,
@@ -225,10 +216,19 @@ module Aireview
225
216
  begin
226
217
  parse_expected_result(repaired, expected, critique_candidate_ids: critique_candidate_ids)
227
218
  rescue JSON::ParserError, SchemaError => second_error
219
+ raise_if_cut_off(stage: repair_stage, kind: "#{kind} repair", error: second_error)
228
220
  raise ParseError, "LLM returned invalid #{kind} JSON after repair: #{second_error.message}"
229
221
  end
230
222
  end
231
223
 
224
+ # An answer the provider cut off or blocked is not a JSON mistake: the
225
+ # same model would cut the repair off too, so the stage goes to the next
226
+ # model at once. A valid answer is taken whatever the reason.
227
+ def raise_if_cut_off(stage:, kind:, error:)
228
+ reason = CUT_OFF_REASONS[@reviewer.finish_reason(stage)]
229
+ raise ParseError, "LLM #{kind} was #{reason}: #{error.message}" if reason
230
+ end
231
+
232
232
  def parse_expected_result(raw, expected, critique_candidate_ids: nil)
233
233
  @parser.parse(raw, expected: expected, critique_candidate_ids: critique_candidate_ids)
234
234
  end
@@ -47,6 +47,10 @@ module Aireview
47
47
  section_list: 'Truncated sections',
48
48
  fallback_used: 'Fallback model used',
49
49
  critique_weaker: 'Critique ran on a model weaker than Generate: the findings were checked less strictly.',
50
+ jev: 'The findings were checked by Jev, a fast classifier, without refining their wording.',
51
+ jev_partial: 'The findings were checked by Jev, a fast classifier, without refining their wording; ' \
52
+ 'those Jev could not judge were checked by the LLM critique.',
53
+ jev_failed: 'Jev was unavailable: the findings were checked by the LLM critique.',
50
54
  quote_missing: 'quote not found in the diff'
51
55
  },
52
56
  'ru' => {
@@ -74,6 +78,10 @@ module Aireview
74
78
  section_list: 'Усечённые секции',
75
79
  fallback_used: 'Использована резервная модель',
76
80
  critique_weaker: 'Критика выполнена моделью слабее generate: замечания проверены менее строго.',
81
+ jev: 'Замечания проверены быстрым классификатором Jev, без уточнения формулировок.',
82
+ jev_partial: 'Замечания проверены быстрым классификатором Jev, без уточнения формулировок; ' \
83
+ 'те, что Jev не смог оценить, проверила LLM-критика.',
84
+ jev_failed: 'Jev был недоступен: замечания проверила LLM-критика.',
77
85
  quote_missing: 'цитата не найдена в диффе'
78
86
  }
79
87
  }.freeze
@@ -86,7 +94,9 @@ module Aireview
86
94
  # result is still about the findings; incomplete coverage is written
87
95
  # next to it so that the result line does not read as "everything was
88
96
  # checked".
89
- def render(accepted, summary:, coverage: nil, fallback_models: {}, critique_weaker: false)
97
+ # jev_note — how Jev took part in the critique (see JevStage::Outcome).
98
+ # The facts about the run are named one by one on purpose.
99
+ def render(accepted, summary:, coverage: nil, fallback_models: {}, critique_weaker: false, jev_note: nil) # rubocop:disable Metrics/ParameterLists
90
100
  mismatches, important = select_findings(Array(accepted))
91
101
  result = mismatches.empty? && important.empty? ? 'ok' : 'needs attention'
92
102
 
@@ -106,7 +116,7 @@ module Aireview
106
116
  ## #{label(:result)}
107
117
 
108
118
  #{result}#{partial_note(coverage)}
109
- #{coverage_block(coverage)}#{fallback_note(fallback_models)}#{weaker_note(critique_weaker)}
119
+ #{coverage_block(coverage)}#{fallback_note(fallback_models)}#{weaker_note(critique_weaker)}#{jev_note(jev_note)}
110
120
  #{label(:disclaimer)}
111
121
  MARKDOWN
112
122
  end
@@ -191,6 +201,12 @@ module Aireview
191
201
  "\n#{label(:critique_weaker)}\n"
192
202
  end
193
203
 
204
+ # Jev decides keep/reject but cannot refine: the reader should know the
205
+ # wording is the first pass's own.
206
+ def jev_note(note)
207
+ note ? "\n#{label(note)}\n" : ''
208
+ end
209
+
194
210
  def fallback_note(fallback_models)
195
211
  return '' if fallback_models.nil? || fallback_models.empty?
196
212
 
@@ -16,6 +16,7 @@ module Aireview
16
16
  @logger = logger
17
17
  @router = router || LlmRouter.new(config: config, logger: logger)
18
18
  @client = client || LlmClient.new(config: config, logger: logger)
19
+ @finish_reasons = {}
19
20
  end
20
21
 
21
22
  # pinned — a request only to the model that answered last in the stage
@@ -47,6 +48,12 @@ module Aireview
47
48
  @router.critique_weaker?
48
49
  end
49
50
 
51
+ # Why the last answer of the stage stopped (:stop, :max_tokens,
52
+ # :content_filter…), nil when the provider did not say or no answer came.
53
+ def finish_reason(stage)
54
+ @finish_reasons[stage.to_s]
55
+ end
56
+
50
57
  # An invalid result: the model is excluded for the stage, the next
51
58
  # request of the stage goes to another. Returns the excluded model or nil.
52
59
  def exclude_answered_model(stage:, reason:)
@@ -58,11 +65,13 @@ module Aireview
58
65
  # A pinned route giving up is not an API error for the pipeline but
59
66
  # "repair impossible": the same fate as an invalid result.
60
67
  def call_llm(prompt, pinned:)
68
+ @finish_reasons.delete(prompt.stage)
61
69
  response = @router.call(stage: prompt.stage, request_chars: prompt.chars, pinned: pinned) do |route, timeout|
62
70
  @client.request(prompt, candidate: route.candidate, key: route.key, key_index: route.key_index,
63
71
  timeout: timeout)
64
72
  end
65
- response.content
73
+ @finish_reasons[prompt.stage] = response.finish_reason
74
+ LlmClient.content(response)
66
75
  rescue RouteExhaustedError => e
67
76
  raise RepairImpossibleError, e.message
68
77
  end
@@ -26,7 +26,8 @@ module Aireview
26
26
  primary = ModelCandidate.new(
27
27
  provider: settings.fetch(:provider).to_s,
28
28
  model: settings[:model],
29
- max_prompt_chars: settings.fetch(:max_prompt_chars)
29
+ max_prompt_chars: settings.fetch(:max_prompt_chars),
30
+ api_base: ModelCandidate.api_base(settings[:api_base], "llm.#{stage}.api_base")
30
31
  )
31
32
  return [primary] if only_primary
32
33
 
@@ -34,7 +35,8 @@ module Aireview
34
35
  end
35
36
 
36
37
  # A reserve without a provider inherits the stage provider, without a
37
- # limit its limit.
38
+ # limit its limit. The address is not inherited: a reserve on another
39
+ # server names its own.
38
40
  def self.fallbacks(stage, items, primary)
39
41
  items.each_with_index.map do |item, index|
40
42
  item = ModelCandidate.parse_item(item)
@@ -46,7 +48,8 @@ module Aireview
46
48
  ModelCandidate.new(
47
49
  provider: (item['provider'] || primary.provider).to_s,
48
50
  model: item['model'].to_s,
49
- max_prompt_chars: limit.nil? ? primary.max_prompt_chars : positive_limit(limit, name)
51
+ max_prompt_chars: limit.nil? ? primary.max_prompt_chars : positive_limit(limit, name),
52
+ api_base: ModelCandidate.api_base(item['api_base'], "#{name}.api_base")
50
53
  )
51
54
  end
52
55
  end
@@ -1,4 +1,4 @@
1
1
  # frozen_string_literal: true
2
2
  module Aireview
3
- VERSION = '2.0.0'
3
+ VERSION = '2.2.0'
4
4
  end
metadata CHANGED
@@ -1,14 +1,14 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: aireview
3
3
  version: !ruby/object:Gem::Version
4
- version: 2.0.0
4
+ version: 2.2.0
5
5
  platform: ruby
6
6
  authors:
7
7
  - Denis Levenko
8
8
  autorequire:
9
9
  bindir: bin
10
10
  cert_chain: []
11
- date: 2026-09-19 00:00:00.000000000 Z
11
+ date: 2026-09-27 00:00:00.000000000 Z
12
12
  dependencies:
13
13
  - !ruby/object:Gem::Dependency
14
14
  name: dotenv
@@ -50,18 +50,19 @@ dependencies:
50
50
  requirements:
51
51
  - - '='
52
52
  - !ruby/object:Gem::Version
53
- version: 1.16.0
53
+ version: 2.0.0
54
54
  type: :runtime
55
55
  prerelease: false
56
56
  version_requirements: !ruby/object:Gem::Requirement
57
57
  requirements:
58
58
  - - '='
59
59
  - !ruby/object:Gem::Version
60
- version: 1.16.0
60
+ version: 2.0.0
61
61
  description: 'Reviews self-hosted GitLab merge requests with a two-pass LLM pipeline:
62
62
  the first pass finds candidate findings, the second one critiques them and drops
63
- the weak ones. Supports Gemini and local Ollama, optional Jira context and posting
64
- a single updatable review note back to the merge request.'
63
+ the weak ones. Supports Gemini, Ollama, OpenAI, Anthropic, OpenRouter and OpenAI-compatible
64
+ servers, Jev as the critic, optional Jira context and posting a single updatable
65
+ review note back to the merge request.'
65
66
  email:
66
67
  executables:
67
68
  - aireview
@@ -80,15 +81,21 @@ files:
80
81
  - lib/aireview/cli.rb
81
82
  - lib/aireview/config.rb
82
83
  - lib/aireview/config_fallbacks.rb
84
+ - lib/aireview/config_jev.rb
83
85
  - lib/aireview/config_layers.rb
84
86
  - lib/aireview/config_limits.rb
85
87
  - lib/aireview/config_loader.rb
86
88
  - lib/aireview/context_budget.rb
87
89
  - lib/aireview/context_builder.rb
88
90
  - lib/aireview/diff_fetcher.rb
91
+ - lib/aireview/dry_run_prompts.rb
89
92
  - lib/aireview/dry_run_report.rb
90
93
  - lib/aireview/errors.rb
91
94
  - lib/aireview/gitlab_client.rb
95
+ - lib/aireview/jev_client.rb
96
+ - lib/aireview/jev_critic.rb
97
+ - lib/aireview/jev_shadow.rb
98
+ - lib/aireview/jev_stage.rb
92
99
  - lib/aireview/jira_client.rb
93
100
  - lib/aireview/llm_client.rb
94
101
  - lib/aireview/llm_failure.rb
@@ -101,6 +108,7 @@ files:
101
108
  - lib/aireview/output_schemas.rb
102
109
  - lib/aireview/prompts/critique.txt
103
110
  - lib/aireview/prompts/generate.txt
111
+ - lib/aireview/prompts/jev_questions.yml
104
112
  - lib/aireview/publisher.rb
105
113
  - lib/aireview/result_parser.rb
106
114
  - lib/aireview/review_marker.rb