aireview 2.1.0 → 2.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,19 +1,48 @@
1
1
  # frozen_string_literal: true
2
+ require_relative 'errors'
2
3
 
3
4
  module Aireview
4
- KNOWN_PROVIDERS = %w[gemini ollama].freeze
5
+ # Providers a "provider/name" string may start with. An OpenRouter name
6
+ # carries a slash of its own, so in a string it always goes with the
7
+ # prefix: openrouter/anthropic/claude-sonnet-4.5.
8
+ KNOWN_PROVIDERS = %w[gemini ollama openai anthropic openrouter].freeze
5
9
 
6
- # A model in a stage chain: provider, name and request size limit.
7
- # Compared by provider and name — the limit depends on the stage.
8
- ModelCandidate = Struct.new(:provider, :model, :max_prompt_chars, keyword_init: true) do
10
+ # A model in a stage chain: provider, name, request size limit and, for a
11
+ # server of your own, its address. Compared by provider, name and address
12
+ # — the limit depends on the stage.
13
+ ModelCandidate = Struct.new(:provider, :model, :max_prompt_chars, :api_base, keyword_init: true) do
14
+ # The address is part of the name, scheme included: two servers with the
15
+ # same model (http and https ones too) are two models for quarantine,
16
+ # reserves and the review key.
9
17
  def to_s
10
- "#{provider}/#{model}"
18
+ name = "#{provider}/#{model}"
19
+ api_base ? "#{name}@#{api_base.chomp('/')}" : name
20
+ end
21
+
22
+ # Every way a start or a CLI override may name this model: bare,
23
+ # "provider/name", or the full name with the address. One list for the
24
+ # config check and for picking the model, so they cannot disagree.
25
+ def names
26
+ [model.to_s, "#{provider}/#{model}", to_s].uniq
11
27
  end
12
28
 
13
29
  def same_model?(other)
14
30
  to_s == other.to_s
15
31
  end
16
32
 
33
+ # Models sharing keys: a provider's own API, or one server of your own.
34
+ def key_source
35
+ api_base ? "#{provider}@#{api_base}" : provider.to_s
36
+ end
37
+
38
+ # api_base of a model or a stage: an http(s) address or nothing.
39
+ def self.api_base(value, name)
40
+ return nil if value.nil? || value.to_s.strip.empty?
41
+ return value.to_s.strip if value.to_s.strip.match?(%r{\Ahttps?://\S+\z})
42
+
43
+ raise ConfigError, "#{name} must be an http(s) URL, got #{value.inspect}"
44
+ end
45
+
17
46
  # A "provider/name" string or a bare name (the provider is separated by
18
47
  # a slash because Ollama tags contain a colon), or a hash with model.
19
48
  def self.parse_item(item)
@@ -8,6 +8,7 @@ require_relative 'review_pipeline'
8
8
  require_relative 'result_parser'
9
9
  require_relative 'llm_client'
10
10
  require_relative 'output_schemas'
11
+ require_relative 'jev_client'
11
12
 
12
13
  module Aireview
13
14
  # `aireview models check`: every model of both stage chains gets one
@@ -16,6 +17,10 @@ module Aireview
16
17
  # validation as in a run: the provider's catalog is not consulted, "the
17
18
  # model is listed" does not mean "our request with the schema passes on
18
19
  # it". No reserves, quarantine or walking — this is a check, not a review.
20
+ # Only the stages that need an LLM are checked (Jev with fallback: fail
21
+ # needs no LLM Critique); Jev gets a probe of its own when it is the
22
+ # critique engine, and in shadow mode too, but then its result is shown
23
+ # without counting: the shadow never blocks a release.
19
24
  class ModelChecker
20
25
  PROBE_MERGE_REQUEST = {
21
26
  'title' => 'Fix order total',
@@ -32,6 +37,10 @@ module Aireview
32
37
  }
33
38
  ].freeze
34
39
  PROBE_CANDIDATE_IDS = ['C1'].freeze
40
+ JEV_PROBE_STATE = 'The order total must include the quantity.'
41
+ JEV_PROBE_QUESTIONS = {
42
+ 'mentions_quantity' => {'type' => 'noul', 'instructions' => 'Does the text mention a quantity?'}
43
+ }.freeze
35
44
  # ok and skipped do not block a release, everything else does. In strict
36
45
  # mode (a runner where Ollama must be running) skipped fails too.
37
46
  PASSING = %i[ok skipped].freeze
@@ -52,6 +61,7 @@ module Aireview
52
61
  @strict = strict
53
62
  client = dependencies[:client]
54
63
  sleeper = dependencies[:sleeper]
64
+ @jev_client = dependencies[:jev_client]
55
65
  @logger = logger
56
66
  @client = client || LlmClient.new(config: config, logger: logger)
57
67
  @sleeper = sleeper || ->(seconds) { sleep(seconds) }
@@ -63,13 +73,14 @@ module Aireview
63
73
  # 1 — at least one did not.
64
74
  def run
65
75
  @config.require_llm_configuration!
66
- candidates = STAGES.flat_map { |stage| @config.stage_chain(stage) }.uniq(&:to_s)
76
+ stages = @config.llm_stages
77
+ candidates = stages.flat_map { |stage| @config.stage_chain(stage) }.uniq(&:to_s)
67
78
  prompts = probe_prompts
68
- @out.puts("Checking #{candidates.size} model(s) with the generate and critique schemas")
79
+ @out.puts("Checking #{candidates.size} model(s) with the #{stages.join(' and ')} schemas")
69
80
  results = candidates.flat_map do |candidate|
70
- STAGES.map { |stage| check(candidate, stage, prompts).tap { |result| @out.puts(result) } }
81
+ stages.map { |stage| check(candidate, stage, prompts).tap { |result| @out.puts(result) } }
71
82
  end
72
- summary(results)
83
+ summary(results + jev_results)
73
84
  end
74
85
 
75
86
  private
@@ -110,10 +121,40 @@ module Aireview
110
121
  stage: stage, system: prompt[:system_prompt], user: prompt[:user_prompt], temperature: 0,
111
122
  schema: stage == 'critique' ? CritiqueOutputSchema : GenerateOutputSchema
112
123
  )
113
- key = @config.provider_api_keys(candidate.provider).first
124
+ key = @config.candidate_api_keys(candidate).first
114
125
  LlmClient.content(@client.request(request, candidate: candidate, key: key, timeout: @config.llm_timeout.to_f))
115
126
  end
116
127
 
128
+ # The Jev probe counts only when Jev is the critique engine.
129
+ def jev_results
130
+ counted = @config.jev_critique?
131
+ return [] unless counted || @config.jev_shadow?
132
+
133
+ result = check_jev
134
+ result.detail = [result.detail, 'shadow only, not counted'].compact.join('; ') unless counted
135
+ @out.puts(result)
136
+ counted ? [result] : []
137
+ end
138
+
139
+ def check_jev
140
+ name = "jev/#{@config.jev_model}"
141
+ if Aireview::Utils.blank?(@config.jev_api_key)
142
+ return Result.new(candidate: name, stage: 'jev', status: :skipped, detail: 'JEV_API_KEY is not set')
143
+ end
144
+
145
+ started_at = Process.clock_gettime(Process::CLOCK_MONOTONIC)
146
+ jev_client.evaluate(state: JEV_PROBE_STATE, questions: JEV_PROBE_QUESTIONS)
147
+ Result.new(candidate: name, stage: 'jev', status: :ok,
148
+ seconds: Process.clock_gettime(Process::CLOCK_MONOTONIC) - started_at)
149
+ rescue JevError => e
150
+ status = e.status.nil? || [429, 529].include?(e.status) ? :unverified : :failed
151
+ Result.new(candidate: name, stage: 'jev', status: status, detail: e.message)
152
+ end
153
+
154
+ def jev_client
155
+ @jev_client ||= JevClient.new(config: @config, logger: @logger)
156
+ end
157
+
117
158
  # missing — the provider has no such model; unverified — the provider
118
159
  # could not answer right now (overload, quota, timeout); skipped — Ollama
119
160
  # is not running where the check runs; failed — everything else.
@@ -1,5 +1,6 @@
1
1
  # frozen_string_literal: true
2
2
  require_relative 'errors'
3
+ require_relative 'stages'
3
4
  require_relative 'utils'
4
5
  require_relative 'model_candidate'
5
6
  require_relative 'stage_chains'
@@ -26,11 +27,24 @@ module Aireview
26
27
  return false if Aireview::Utils.blank?(model)
27
28
 
28
29
  Array(items).each_with_index.any? do |item, index|
29
- parsed = parse_item(item, index, provider)
30
- parsed[:model] == model.to_s || "#{parsed[:provider]}/#{parsed[:model]}" == model.to_s
30
+ names(parse_item(item, index, provider)).include?(model.to_s)
31
31
  end
32
32
  end
33
33
 
34
+ def self.item_candidate(item)
35
+ ModelCandidate.new(provider: item[:provider], model: item[:model], api_base: item[:api_base])
36
+ end
37
+
38
+ # The full name of a pool item, with the address of a server of your own.
39
+ def self.item_name(item)
40
+ item_candidate(item).to_s
41
+ end
42
+
43
+ # How a model may be named in a start or a CLI override (ModelCandidate#names).
44
+ def self.names(item)
45
+ item_candidate(item).names
46
+ end
47
+
34
48
  def self.parse_item(item, index, provider)
35
49
  item = ModelCandidate.parse_item(item)
36
50
  name = "llm.models[#{index}]"
@@ -41,7 +55,8 @@ module Aireview
41
55
  {
42
56
  provider: (item['provider'] || provider).to_s,
43
57
  model: item['model'].to_s,
44
- max_prompt_chars: limit.nil? ? nil : StageChains.positive_limit(limit, name)
58
+ max_prompt_chars: limit.nil? ? nil : StageChains.positive_limit(limit, name),
59
+ api_base: ModelCandidate.api_base(item['api_base'], "#{name}.api_base")
45
60
  }
46
61
  end
47
62
 
@@ -50,30 +65,38 @@ module Aireview
50
65
  # the pool (image defaults versus the project's LLM_MODELS): such a start
51
66
  # missing from the pool is replaced by the first model with a warning,
52
67
  # an explicit start outside the pool is a configuration error;
53
- # own_chains — stages with a chain of their own.
54
- # Nine named settings read better than a struct for its own sake.
68
+ # own_chains — stages with a chain of their own; stages — the stages
69
+ # that go to an LLM in this run: without Critique (Jev decides with no
70
+ # fallback, --no-critique) the plan has no critique chain and no critique
71
+ # policy to validate or to sign.
72
+ # Ten named settings read better than a struct for its own sake.
55
73
  def initialize(items:, provider:, limits:, starts: {}, inherited_starts: [], rank: nil, allow_weaker: false, # rubocop:disable Metrics/ParameterLists
56
- own_chains: nil, only_primary: false)
74
+ own_chains: nil, only_primary: false, stages: STAGES)
75
+ @stages = stages.map(&:to_s)
57
76
  @items = parse_items(items, provider)
58
77
  @limits = limits.transform_keys(&:to_s)
59
- @rank = validate_rank(rank)
60
- @allow_weaker = allow_weaker == true
78
+ @rank, @allow_weaker = critique_policy(rank, allow_weaker)
61
79
  @own_chains = own_chains || StageChains.new({})
62
80
  @only_primary = only_primary
63
81
  @warnings = []
64
82
  @starts = resolve_starts(starts.transform_keys(&:to_s), inherited_starts.map(&:to_s))
65
83
  end
66
84
 
85
+ def stage?(stage)
86
+ @stages.include?(stage.to_s)
87
+ end
88
+
67
89
  def pool(stage = 'generate')
68
90
  limit = @limits.fetch(stage.to_s)
69
91
  @items.map do |item|
70
92
  ModelCandidate.new(provider: item[:provider], model: item[:model],
71
- max_prompt_chars: item[:max_prompt_chars] || limit)
93
+ max_prompt_chars: item[:max_prompt_chars] || limit, api_base: item[:api_base])
72
94
  end
73
95
  end
74
96
 
75
97
  def chain(stage)
76
98
  stage = stage.to_s
99
+ raise ArgumentError, "unknown LLM stage #{stage.inspect}" unless stage?(stage)
77
100
  return @own_chains.chain(stage) unless pool_stage?(stage)
78
101
 
79
102
  trim(pool(stage).rotate(index_of(@starts[stage] || @items.first[:model])))
@@ -107,21 +130,20 @@ module Aireview
107
130
 
108
131
  # The order and policy of the pool go into the review key: they decide
109
132
  # which model checks the findings. nil when a stage is outside the pool.
133
+ # Without Critique only the generate part: a critique setting nobody
134
+ # uses must not change the key, and the pool must stay in it.
110
135
  def signature
111
- return nil unless both_stages_in_pool?
136
+ return nil unless all_stages_in_pool?
112
137
 
113
- {
114
- 'models' => pool.map(&:to_s),
115
- 'generate_start' => @starts['generate'],
116
- 'critique_start' => @starts['critique'],
117
- 'rank' => @rank,
118
- 'allow_weaker' => @allow_weaker
119
- }
138
+ signature = {'models' => pool.map(&:to_s), 'generate_start' => @starts['generate']}
139
+ return signature unless stage?('critique')
140
+
141
+ signature.merge('critique_start' => @starts['critique'], 'rank' => @rank, 'allow_weaker' => @allow_weaker)
120
142
  end
121
143
 
122
144
  # The critique selection rule in words, for --dry-run.
123
145
  def rule
124
- return nil unless both_stages_in_pool?
146
+ return nil unless stage?('critique') && all_stages_in_pool?
125
147
  return @rank if @rank == 'any' || !@allow_weaker
126
148
 
127
149
  "#{@rank}, weaker allowed"
@@ -132,7 +154,7 @@ module Aireview
132
154
  end
133
155
 
134
156
  def pool_stage?(stage)
135
- !@own_chains.stage?(stage)
157
+ stage?(stage) && !@own_chains.stage?(stage)
136
158
  end
137
159
 
138
160
  def pool_member?(model)
@@ -147,12 +169,12 @@ module Aireview
147
169
 
148
170
  private
149
171
 
150
- def both_stages_in_pool?
151
- STAGES.all? { |stage| pool_stage?(stage) }
172
+ def all_stages_in_pool?
173
+ @stages.all? { |stage| pool_stage?(stage) }
152
174
  end
153
175
 
154
176
  def rank_applies?(after)
155
- both_stages_in_pool? && @rank != 'any' && pool_member?(after)
177
+ all_stages_in_pool? && @rank != 'any' && pool_member?(after)
156
178
  end
157
179
 
158
180
  def trim(chain)
@@ -174,7 +196,7 @@ module Aireview
174
196
 
175
197
  # The start of a stage with its own chain is not checked: it is outside the pool.
176
198
  def resolve_starts(starts, inherited)
177
- STAGES.to_h do |stage|
199
+ @stages.to_h do |stage|
178
200
  start = starts[stage]
179
201
  next [stage, nil] if Aireview::Utils.blank?(start) || !pool_stage?(stage)
180
202
  next [stage, start] if pool_member?(start)
@@ -192,15 +214,22 @@ module Aireview
192
214
  raise ConfigError, "#{model} is not in llm.models: #{pool.join(', ')}"
193
215
  end
194
216
 
195
- # A model is given by name or as "provider/name"; a candidate matches by provider and name.
217
+ # A model is given by name, as "provider/name" or by its full name; a
218
+ # candidate matches by provider, name and server address.
196
219
  def match?(item, model)
197
- return "#{item[:provider]}/#{item[:model]}" == model.to_s if model.is_a?(ModelCandidate)
220
+ return self.class.item_name(item) == model.to_s if model.is_a?(ModelCandidate)
198
221
 
199
- item[:model] == model.to_s || "#{item[:provider]}/#{item[:model]}" == model.to_s
222
+ self.class.names(item).include?(model.to_s)
200
223
  end
201
224
 
202
225
  def match_candidate?(candidate, model)
203
- candidate.model == model.to_s || candidate.to_s == model.to_s
226
+ candidate.names.include?(model.to_s)
227
+ end
228
+
229
+ def critique_policy(rank, allow_weaker)
230
+ return [nil, false] unless stage?('critique')
231
+
232
+ [validate_rank(rank), allow_weaker == true]
204
233
  end
205
234
 
206
235
  def validate_rank(rank)
@@ -214,7 +243,7 @@ module Aireview
214
243
  parsed = Array(items).each_with_index.map { |item, index| self.class.parse_item(item, index, provider) }
215
244
  raise ConfigError, 'llm.models must not be empty' if parsed.empty?
216
245
 
217
- names = parsed.map { |item| "#{item[:provider]}/#{item[:model]}" }
246
+ names = parsed.map { |item| self.class.item_name(item) }
218
247
  duplicates = names.tally.select { |_, count| count > 1 }.keys
219
248
  raise ConfigError, "llm.models has duplicates: #{duplicates.join(', ')}" unless duplicates.empty?
220
249
 
@@ -0,0 +1,79 @@
1
+ # Jev questions about one candidate, the counterpart of critique.txt: the
2
+ # rules are the same, only asked as separate yes/no and choice questions.
3
+ # When critique.txt changes, change these too (spec/prompts_spec.rb checks
4
+ # the key rules in both). %{id} is the candidate id; the state holds the
5
+ # candidate at `candidates.<id>` and the diff at `diff`. Keys "true" and
6
+ # "false" are quoted: YAML would read them as booleans.
7
+ real_issue:
8
+ type: noul
9
+ instructions: >-
10
+ Is candidate `candidates.%{id}` a real, well-supported problem in the code
11
+ changed by this merge request?
12
+ criteria:
13
+ "true": >-
14
+ The problem is directly confirmed by the diff, the MR description, the
15
+ Jira task or the changed control/data flow, and it can lead to a real
16
+ defect, regression, security/performance issue, data loss or a mismatch
17
+ with the task. For category task_mismatch: the diff contains a concrete
18
+ artifact (debug code such as puts, p, binding.pry, byebug, console.log,
19
+ debugger; commented-out blocks; unused debug/test/helper methods;
20
+ temporary TODO/HACK/FIXME/DEBUG markers from this diff; disabled tests
21
+ such as skip, xit, pending without an explanation) or a change that
22
+ directly contradicts the MR or Jira.
23
+ "false": >-
24
+ The finding is not confirmed by the diff or Jira; it is based on an
25
+ assumption about code outside the diff; it contradicts the diff; it is
26
+ not actionable; it boils down to "may be redundant", "may conflict",
27
+ "may be unnecessary" or "worth checking" without a clear sign of
28
+ breakage, in any category. For category task_mismatch: implementation
29
+ details (config flags, deploy settings, new fields, helper methods),
30
+ accompanying refactoring and any change whose relation to the task is
31
+ plausible. A candidate with a `note` could not be anchored to the diff
32
+ mechanically and needs to be checked against the diff more carefully.
33
+ Missing text in a truncated part of the context does not mean a
34
+ missing requirement or missing code.
35
+ enough_context:
36
+ type: noul
37
+ instructions: >-
38
+ Does the state contain enough to confirm or refute candidate
39
+ `candidates.%{id}`?
40
+ criteria:
41
+ "true": >-
42
+ The code the finding is about is present in `candidates.%{id}.evidence`
43
+ or in `diff`, and for a finding about the task the relevant MR or Jira
44
+ requirements are present and not cut off where they matter.
45
+ "false": >-
46
+ The decisive code or requirement is missing: the file is not shown, the
47
+ relevant part is truncated, or the finding depends on code or
48
+ requirements that are not in the state.
49
+ version_claim:
50
+ type: noul
51
+ instructions: >-
52
+ Does candidate `candidates.%{id}` claim that a version of a package,
53
+ library, tool or image tag does not exist or has not been released yet?
54
+ criteria:
55
+ "true": >-
56
+ The finding says that a specified version does not exist, is not
57
+ released yet or is invalid because of its number.
58
+ "false": >-
59
+ The finding is about anything else, including syntax errors and explicit
60
+ contradictions with the MR or Jira requirements.
61
+ # Only when the request holds more than one candidate; the options are the
62
+ # other ids plus none.
63
+ duplicate_of:
64
+ type: choice
65
+ instructions: >-
66
+ Does candidate `candidates.%{id}` describe the same problem as another
67
+ candidate?
68
+ other: The same problem as candidate `candidates.%{other}`.
69
+ none: A problem that no other candidate describes.
70
+ # Goes to the log only; its answer does not decide anything yet.
71
+ severity:
72
+ type: choice
73
+ instructions: >-
74
+ How severe is the problem described by candidate `candidates.%{id}`, if it
75
+ is real?
76
+ criteria:
77
+ critical: Breaks production behaviour, loses or corrupts data, or opens a security hole.
78
+ major: A real defect or regression on a common path, or a clear mismatch with the task.
79
+ minor: A limited edge case, a test gap or a maintainability issue.
@@ -23,14 +23,19 @@ module Aireview
23
23
  # that way the diff, the MR description, the Jira context, the review
24
24
  # instructions and ignore_paths enter it by themselves. What else affects
25
25
  # the result — provider, model and temperature of the stages, the shared
26
- # pool with its critique policy — is known by Config#result_signature.
27
- # Without a pool the key is the same as before.
26
+ # pool with its critique policy, Jev as the critique engine — is known by
27
+ # Config#result_signature. Without a pool the key is the same as before.
28
+ # The LLM Critique counts only when it can run (its prompt is there),
29
+ # Jev only when it decides (its question templates are there): with the
30
+ # default engine the key is what it was before Jev.
28
31
  def key(prompts:, config:)
29
32
  signature = config.result_signature
30
33
  source = {
31
34
  'generate' => [*signature['generate'], prompts[:generate_prompt]],
32
35
  'critique' => prompts[:critique_prompt] ? [*signature['critique'], prompts[:critique_prompt]] : nil
33
36
  }
37
+ jev = prompts[:jev_questions]
38
+ source['critique_engine'] = [*signature['critique_engine'], jev] if jev
34
39
  source['pool'] = signature['pool'] if signature['pool']
35
40
 
36
41
  Digest::SHA256.hexdigest(JSON.generate(source))[0, 16]
@@ -1,6 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
  require 'json'
3
3
  require_relative 'errors'
4
+ require_relative 'utils'
4
5
  require_relative 'stages'
5
6
  require_relative 'context_builder'
6
7
  require_relative 'candidate_checker'
@@ -8,12 +9,15 @@ require_relative 'result_parser'
8
9
  require_relative 'review_renderer'
9
10
  require_relative 'review_schemas'
10
11
  require_relative 'reviewer'
12
+ require_relative 'jev_shadow'
13
+ require_relative 'jev_stage'
14
+ require_relative 'dry_run_prompts'
11
15
 
12
16
  module Aireview
13
17
  # A review run: context → Generate → anchoring check against the diff →
14
- # Critique → report. Invalid JSON is repaired once by the same model; when
15
- # the repair is invalid too, the stage restarts on another model with the
16
- # original request.
18
+ # Critique (an LLM, or Jev with llm.critique.engine: jev) → report. Invalid
19
+ # JSON is repaired once by the same model; when the repair is invalid too,
20
+ # the stage restarts on another model with the original request.
17
21
  class ReviewPipeline
18
22
  SchemaError = ResultParser::SchemaError
19
23
 
@@ -23,20 +27,22 @@ module Aireview
23
27
  Do not use markdown, code fences, comments, or text outside JSON.
24
28
  Do not add new review findings.
25
29
  PROMPT
26
- DRY_RUN_CANDIDATES_JSON = '[{"id":"C1","file":"path/from/diff.rb","line":1,' \
27
- '"quoted_code":"...","problem":"...","why":"...","suggestion":"...",' \
28
- '"category":"bug","severity":"major"}]'
29
30
  # Finish reasons after which an unparsable answer gets no repair.
30
31
  CUT_OFF_REASONS = {
31
32
  max_tokens: 'cut off at the output limit (max_tokens)',
32
33
  content_filter: 'blocked by the provider (content_filter)'
33
34
  }.freeze
34
35
 
35
- def initialize(config:, reviewer: nil, context_builder: nil, logger: Logger.new($stderr))
36
+ # jev — jev_shadow: and jev_stage: for tests; by default they are built
37
+ # from the config, the stage only when Jev is the engine.
38
+ def initialize(config:, reviewer: nil, context_builder: nil, logger: Logger.new($stderr), **jev)
36
39
  @config = config
37
40
  @parser = ResultParser.new
38
41
  @reviewer = reviewer || Reviewer.new(config: config, logger: logger)
39
42
  @context_builder = context_builder || ContextBuilder.new(config: config, logger: logger)
43
+ @jev_shadow = jev[:jev_shadow] ||
44
+ JevShadow.new(config: config, scrub: @context_builder.method(:scrub_text), logger: logger)
45
+ @jev_stage = jev[:jev_stage]
40
46
  @logger = logger
41
47
  end
42
48
 
@@ -63,7 +69,11 @@ module Aireview
63
69
  "(model=#{@reviewer.answered_model('generate')})")
64
70
  candidates = check_candidates(context: context, changes: changes, candidates: candidates)
65
71
 
66
- accepted = critique ? maybe_critique(context: context, candidates: candidates) : skip_critique(candidates)
72
+ accepted, jev_note = if critique
73
+ maybe_critique(context: context, candidates: candidates)
74
+ else
75
+ skip_critique(candidates)
76
+ end
67
77
 
68
78
  @logger.info("Pipeline finished with #{accepted.size} accepted finding(s)")
69
79
 
@@ -72,58 +82,19 @@ module Aireview
72
82
  summary: summary,
73
83
  coverage: context.coverage,
74
84
  fallback_models: @reviewer.fallback_models,
75
- critique_weaker: critique && @reviewer.critique_weaker?
85
+ critique_weaker: critique && @reviewer.critique_weaker?,
86
+ jev_note: jev_note
76
87
  )
77
88
  end
78
89
 
90
+ # The prompts of the run and everything --dry-run shows; see DryRunPrompts.
79
91
  def dry_run_prompts(merge_request:, changes:, jira_issue: nil, critique: true)
80
- @config.require_models!
81
-
82
- context = @context_builder.prepare(
83
- merge_request: merge_request,
84
- changes: changes,
85
- jira_issue: jira_issue,
86
- critique: critique
87
- )
88
- generate_prompt = @context_builder.build_generate_prompt(context)
89
- critique_prompt = if critique
90
- @context_builder.build_critique_prompt(context, candidates_json: DRY_RUN_CANDIDATES_JSON)
91
- end
92
-
93
- {
94
- generate_prompt: generate_prompt,
95
- critique_prompt: critique_prompt,
96
- generate_model: @config.generate_model,
97
- generate_temperature: @config.generate_temperature,
98
- critique_model: @config.critique_model,
99
- critique_temperature: @config.critique_temperature,
100
- generate_fallbacks: @config.fallback_names('generate'),
101
- critique_fallbacks: critique ? @config.fallback_names('critique') : [],
102
- sources: setting_sources(critique),
103
- config_paths: @config.layer_paths,
104
- warnings: @config.warnings,
105
- critique_rule: critique ? @config.routing.rule : nil,
106
- api_keys: @config.api_key_counts(critique ? STAGES : ['generate']),
107
- time_budget: @config.llm_time_budget,
108
- overloaded_quarantine: @config.overloaded_quarantine,
109
- coverage: context.coverage,
110
- sizes: context.sizes
111
- }
92
+ DryRunPrompts.new(config: @config, context_builder: @context_builder, logger: @logger)
93
+ .build(merge_request: merge_request, changes: changes, jira_issue: jira_issue, critique: critique)
112
94
  end
113
95
 
114
96
  private
115
97
 
116
- # Where the model, provider and reserves of a stage came from, for --dry-run.
117
- def setting_sources(critique)
118
- (critique ? %w[generate critique] : %w[generate]).to_h do |stage|
119
- [stage.to_sym, {
120
- model: @config.stage_model_source(stage),
121
- provider: @config.stage_provider_source(stage),
122
- fallbacks: @config.stage_fallbacks_source(stage)
123
- }]
124
- end
125
- end
126
-
127
98
  # A stage is a request, parsing and one repair by the same model. An
128
99
  # invalid result after the repair, like a repair with no requests left,
129
100
  # excludes the model for the stage, and the stage starts over on the next
@@ -153,18 +124,32 @@ module Aireview
153
124
  ).check(candidates)
154
125
  end
155
126
 
127
+ # Returns the candidates and no report note, like every critique path.
156
128
  def skip_critique(candidates, reason = nil)
157
129
  @logger.info(['Pipeline critique pass skipped', reason].compact.join(': '))
158
- candidates
130
+ [candidates, nil]
159
131
  end
160
132
 
161
133
  # Critique has nothing to filter without candidates: the LLM request
162
- # would waste quota and time.
134
+ # would waste quota and time. Returns [accepted, note for the report].
163
135
  def maybe_critique(context:, candidates:)
164
136
  return skip_critique(candidates, 'no candidates') if candidates.empty?
137
+ return jev_critique(context: context, candidates: candidates) if @config.jev_critique?
165
138
 
166
139
  @logger.info("Pipeline critique pass started (model=#{@config.critique_model})")
167
- critique_candidates(context: context, candidates: candidates)
140
+ accepted = critique_candidates(context: context, candidates: candidates)
141
+ @jev_shadow.run(context: context, candidates: candidates, accepted: accepted) if @config.jev_shadow?
142
+ [accepted, nil]
143
+ end
144
+
145
+ def jev_critique(context:, candidates:)
146
+ @jev_stage ||= JevStage.new(
147
+ config: @config, logger: @logger,
148
+ critic: JevCritic.build(config: @config, scrub: @context_builder.method(:scrub_text), logger: @logger)
149
+ )
150
+ llm_critique = ->(subset) { critique_candidates(context: context, candidates: subset) }
151
+ outcome = @jev_stage.run(context: context, candidates: candidates, llm_critique: llm_critique)
152
+ [outcome.accepted, outcome.note]
168
153
  end
169
154
 
170
155
  def critique_candidates(context:, candidates:)
@@ -47,6 +47,10 @@ module Aireview
47
47
  section_list: 'Truncated sections',
48
48
  fallback_used: 'Fallback model used',
49
49
  critique_weaker: 'Critique ran on a model weaker than Generate: the findings were checked less strictly.',
50
+ jev: 'The findings were checked by Jev, a fast classifier, without refining their wording.',
51
+ jev_partial: 'The findings were checked by Jev, a fast classifier, without refining their wording; ' \
52
+ 'those Jev could not judge were checked by the LLM critique.',
53
+ jev_failed: 'Jev was unavailable: the findings were checked by the LLM critique.',
50
54
  quote_missing: 'quote not found in the diff'
51
55
  },
52
56
  'ru' => {
@@ -74,6 +78,10 @@ module Aireview
74
78
  section_list: 'Усечённые секции',
75
79
  fallback_used: 'Использована резервная модель',
76
80
  critique_weaker: 'Критика выполнена моделью слабее generate: замечания проверены менее строго.',
81
+ jev: 'Замечания проверены быстрым классификатором Jev, без уточнения формулировок.',
82
+ jev_partial: 'Замечания проверены быстрым классификатором Jev, без уточнения формулировок; ' \
83
+ 'те, что Jev не смог оценить, проверила LLM-критика.',
84
+ jev_failed: 'Jev был недоступен: замечания проверила LLM-критика.',
77
85
  quote_missing: 'цитата не найдена в диффе'
78
86
  }
79
87
  }.freeze
@@ -86,7 +94,9 @@ module Aireview
86
94
  # result is still about the findings; incomplete coverage is written
87
95
  # next to it so that the result line does not read as "everything was
88
96
  # checked".
89
- def render(accepted, summary:, coverage: nil, fallback_models: {}, critique_weaker: false)
97
+ # jev_note — how Jev took part in the critique (see JevStage::Outcome).
98
+ # The facts about the run are named one by one on purpose.
99
+ def render(accepted, summary:, coverage: nil, fallback_models: {}, critique_weaker: false, jev_note: nil) # rubocop:disable Metrics/ParameterLists
90
100
  mismatches, important = select_findings(Array(accepted))
91
101
  result = mismatches.empty? && important.empty? ? 'ok' : 'needs attention'
92
102
 
@@ -106,7 +116,7 @@ module Aireview
106
116
  ## #{label(:result)}
107
117
 
108
118
  #{result}#{partial_note(coverage)}
109
- #{coverage_block(coverage)}#{fallback_note(fallback_models)}#{weaker_note(critique_weaker)}
119
+ #{coverage_block(coverage)}#{fallback_note(fallback_models)}#{weaker_note(critique_weaker)}#{jev_note(jev_note)}
110
120
  #{label(:disclaimer)}
111
121
  MARKDOWN
112
122
  end
@@ -191,6 +201,12 @@ module Aireview
191
201
  "\n#{label(:critique_weaker)}\n"
192
202
  end
193
203
 
204
+ # Jev decides keep/reject but cannot refine: the reader should know the
205
+ # wording is the first pass's own.
206
+ def jev_note(note)
207
+ note ? "\n#{label(note)}\n" : ''
208
+ end
209
+
194
210
  def fallback_note(fallback_models)
195
211
  return '' if fallback_models.nil? || fallback_models.empty?
196
212