aireview 0.2.1 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,224 @@
1
+ # frozen_string_literal: true
2
+ require_relative 'errors'
3
+ require_relative 'utils'
4
+ require_relative 'model_candidate'
5
+ require_relative 'stage_chains'
6
+
7
+ module Aireview
8
+ # A routing plan from a shared pool: models in order of priority (the
9
+ # first is the preferred one for Critique). Generate walks the pool from
10
+ # its start model downwards and round again; Critique takes the first
11
+ # live model not below the one that answered in Generate
12
+ # (rank: not_below_generate), self-critique by the same model is the last
13
+ # permitted option, lower only with allow_weaker. A stage with a model of
14
+ # its own does not use the pool: its chain is independent and the rank
15
+ # rule does not apply, as with rank: any.
16
+ class ModelPool
17
+ CRITIQUE_RANKS = %w[not_below_generate any].freeze
18
+ DEFAULT_CRITIQUE_RANK = 'not_below_generate'
19
+
20
+ attr_reader :warnings
21
+
22
+ # Whether a model is in the raw llm.models list — without building the
23
+ # plan, so that a CLI override does not trip over an invalid start of
24
+ # the old plan.
25
+ def self.member?(items, provider, model)
26
+ return false if Aireview::Utils.blank?(model)
27
+
28
+ Array(items).each_with_index.any? do |item, index|
29
+ parsed = parse_item(item, index, provider)
30
+ parsed[:model] == model.to_s || "#{parsed[:provider]}/#{parsed[:model]}" == model.to_s
31
+ end
32
+ end
33
+
34
+ def self.parse_item(item, index, provider)
35
+ item = ModelCandidate.parse_item(item)
36
+ name = "llm.models[#{index}]"
37
+ raise ConfigError, "#{name} must be a model name or a hash with model" unless item.is_a?(Hash)
38
+ raise ConfigError, "#{name}.model is required" if Aireview::Utils.blank?(item['model'])
39
+
40
+ limit = item['max_prompt_chars']
41
+ {
42
+ provider: (item['provider'] || provider).to_s,
43
+ model: item['model'].to_s,
44
+ max_prompt_chars: limit.nil? ? nil : StageChains.positive_limit(limit, name)
45
+ }
46
+ end
47
+
48
+ # items — the raw llm.models; starts — the start per stage (a model name
49
+ # or nil); inherited_starts — stages whose start came from a layer below
50
+ # the pool (image defaults versus the project's LLM_MODELS): such a start
51
+ # missing from the pool is replaced by the first model with a warning,
52
+ # an explicit start outside the pool is a configuration error;
53
+ # own_chains — stages with a chain of their own.
54
+ # Nine named settings read better than a struct for its own sake.
55
+ def initialize(items:, provider:, limits:, starts: {}, inherited_starts: [], rank: nil, allow_weaker: false, # rubocop:disable Metrics/ParameterLists
56
+ own_chains: nil, only_primary: false)
57
+ @items = parse_items(items, provider)
58
+ @limits = limits.transform_keys(&:to_s)
59
+ @rank = validate_rank(rank)
60
+ @allow_weaker = allow_weaker == true
61
+ @own_chains = own_chains || StageChains.new({})
62
+ @only_primary = only_primary
63
+ @warnings = []
64
+ @starts = resolve_starts(starts.transform_keys(&:to_s), inherited_starts.map(&:to_s))
65
+ end
66
+
67
+ def pool(stage = 'generate')
68
+ limit = @limits.fetch(stage.to_s)
69
+ @items.map do |item|
70
+ ModelCandidate.new(provider: item[:provider], model: item[:model],
71
+ max_prompt_chars: item[:max_prompt_chars] || limit)
72
+ end
73
+ end
74
+
75
+ def chain(stage)
76
+ stage = stage.to_s
77
+ return @own_chains.chain(stage) unless pool_stage?(stage)
78
+
79
+ trim(pool(stage).rotate(index_of(@starts[stage] || @items.first[:model])))
80
+ end
81
+
82
+ # The models not below the one that answered in Generate, in pool order
83
+ # (including that one), then — only with allow_weaker — the rest. An
84
+ # explicit critique start goes first only when it is permitted itself:
85
+ # Generate walks round the pool and may answer with a model above its
86
+ # own start, so this cannot be checked statically; a start that is not
87
+ # permitted does not bypass the ban on a weaker critique, it is skipped.
88
+ def critique_chain(after:)
89
+ return chain('critique') unless rank_applies?(after)
90
+
91
+ models = pool('critique')
92
+ limit = index_of(after)
93
+ allowed = models.first(limit + 1)
94
+ allowed += models.drop(limit + 1) if @allow_weaker
95
+ trim(with_start_first(allowed, @starts['critique']))
96
+ end
97
+
98
+ def primary(stage)
99
+ chain(stage).first
100
+ end
101
+
102
+ def weaker?(critique_candidate, generate_candidate)
103
+ return false unless pool_member?(critique_candidate) && pool_member?(generate_candidate)
104
+
105
+ index_of(critique_candidate) > index_of(generate_candidate)
106
+ end
107
+
108
+ # The order and policy of the pool go into the review key: they decide
109
+ # which model checks the findings. nil when a stage is outside the pool.
110
+ def signature
111
+ return nil unless both_stages_in_pool?
112
+
113
+ {
114
+ 'models' => pool.map(&:to_s),
115
+ 'generate_start' => @starts['generate'],
116
+ 'critique_start' => @starts['critique'],
117
+ 'rank' => @rank,
118
+ 'allow_weaker' => @allow_weaker
119
+ }
120
+ end
121
+
122
+ # The critique selection rule in words, for --dry-run.
123
+ def rule
124
+ return nil unless both_stages_in_pool?
125
+ return @rank if @rank == 'any' || !@allow_weaker
126
+
127
+ "#{@rank}, weaker allowed"
128
+ end
129
+
130
+ def pool?
131
+ true
132
+ end
133
+
134
+ def pool_stage?(stage)
135
+ !@own_chains.stage?(stage)
136
+ end
137
+
138
+ def pool_member?(model)
139
+ return false if model.nil?
140
+
141
+ @items.any? { |item| match?(item, model) }
142
+ end
143
+
144
+ def start_used?(stage)
145
+ !@starts[stage.to_s].nil?
146
+ end
147
+
148
+ private
149
+
150
+ def both_stages_in_pool?
151
+ STAGES.all? { |stage| pool_stage?(stage) }
152
+ end
153
+
154
+ def rank_applies?(after)
155
+ both_stages_in_pool? && @rank != 'any' && pool_member?(after)
156
+ end
157
+
158
+ def trim(chain)
159
+ @only_primary ? chain.first(1) : chain
160
+ end
161
+
162
+ def with_start_first(chain, start)
163
+ return chain unless start
164
+
165
+ head = chain.find { |candidate| match_candidate?(candidate, start) }
166
+ unless head
167
+ @warnings << "llm.critique.start #{start} is below the model that answered in generate and allow_weaker " \
168
+ 'is off, ignoring it'
169
+ return chain
170
+ end
171
+
172
+ [head, *chain.reject { |candidate| candidate.equal?(head) }]
173
+ end
174
+
175
+ # The start of a stage with its own chain is not checked: it is outside the pool.
176
+ def resolve_starts(starts, inherited)
177
+ STAGES.to_h do |stage|
178
+ start = starts[stage]
179
+ next [stage, nil] if Aireview::Utils.blank?(start) || !pool_stage?(stage)
180
+ next [stage, start] if pool_member?(start)
181
+ raise ConfigError, "#{start} is not in llm.models: #{pool.join(', ')}" unless inherited.include?(stage)
182
+
183
+ @warnings << "llm.#{stage}.start #{start} is not in the overriding llm.models, starting from #{pool.first}"
184
+ [stage, nil]
185
+ end
186
+ end
187
+
188
+ def index_of(model)
189
+ index = @items.index { |item| match?(item, model) }
190
+ return index if index
191
+
192
+ raise ConfigError, "#{model} is not in llm.models: #{pool.join(', ')}"
193
+ end
194
+
195
+ # A model is given by name or as "provider/name"; a candidate matches by provider and name.
196
+ def match?(item, model)
197
+ return "#{item[:provider]}/#{item[:model]}" == model.to_s if model.is_a?(ModelCandidate)
198
+
199
+ item[:model] == model.to_s || "#{item[:provider]}/#{item[:model]}" == model.to_s
200
+ end
201
+
202
+ def match_candidate?(candidate, model)
203
+ candidate.model == model.to_s || candidate.to_s == model.to_s
204
+ end
205
+
206
+ def validate_rank(rank)
207
+ rank = (rank || DEFAULT_CRITIQUE_RANK).to_s
208
+ return rank if CRITIQUE_RANKS.include?(rank)
209
+
210
+ raise ConfigError, "llm.critique.rank must be one of #{CRITIQUE_RANKS.join(', ')}, got #{rank.inspect}"
211
+ end
212
+
213
+ def parse_items(items, provider)
214
+ parsed = Array(items).each_with_index.map { |item, index| self.class.parse_item(item, index, provider) }
215
+ raise ConfigError, 'llm.models must not be empty' if parsed.empty?
216
+
217
+ names = parsed.map { |item| "#{item[:provider]}/#{item[:model]}" }
218
+ duplicates = names.tally.select { |_, count| count > 1 }.keys
219
+ raise ConfigError, "llm.models has duplicates: #{duplicates.join(', ')}" unless duplicates.empty?
220
+
221
+ parsed
222
+ end
223
+ end
224
+ end
@@ -0,0 +1,82 @@
1
+ # frozen_string_literal: true
2
+ require 'set'
3
+
4
+ module Aireview
5
+ # What the router learned about a model during the run. Three things with
6
+ # different lifetimes: sent requests are counted per stage, a quarantine
7
+ # lasts until a moment in time, an exclusion lasts until the end of the
8
+ # run (the provider has no such model, every key is out of quota) or the
9
+ # end of the stage (an invalid result). Keys with an exhausted daily quota
10
+ # are remembered separately: a quota is a property of "key + model", a
11
+ # quarantine a property of the model.
12
+ class ModelState
13
+ # Every request sent, the short retry and the JSON repair included,
14
+ # spends an attempt regardless of its outcome; the keys of a model share
15
+ # one counter, a quarantine does not reset it.
16
+ MAX_REQUESTS_PER_STAGE = 3
17
+
18
+ attr_reader :excluded_reason
19
+
20
+ def initialize(limit: MAX_REQUESTS_PER_STAGE)
21
+ @limit = limit
22
+ @sent = Hash.new(0)
23
+ @quarantined_until = nil
24
+ @excluded_reason = nil
25
+ @stage_exclusions = {}
26
+ @exhausted_keys = Set.new
27
+ end
28
+
29
+ def sent(stage)
30
+ @sent[stage.to_s]
31
+ end
32
+
33
+ def record_request(stage)
34
+ @sent[stage.to_s] += 1
35
+ end
36
+
37
+ def requests_left?(stage)
38
+ sent(stage) < @limit
39
+ end
40
+
41
+ def quarantine(until_time)
42
+ @quarantined_until = until_time
43
+ end
44
+
45
+ # A model that answered is not overloaded, whichever key answered.
46
+ def lift_quarantine
47
+ @quarantined_until = nil
48
+ end
49
+
50
+ def quarantine_left(now)
51
+ return 0 unless @quarantined_until
52
+
53
+ [@quarantined_until - now, 0].max
54
+ end
55
+
56
+ def exclude(reason)
57
+ @excluded_reason = reason
58
+ end
59
+
60
+ def exclude_for_stage(stage, reason)
61
+ @stage_exclusions[stage.to_s] = reason
62
+ end
63
+
64
+ def exhaust_key(key_index)
65
+ @exhausted_keys << key_index
66
+ end
67
+
68
+ def key_exhausted?(key_index)
69
+ @exhausted_keys.include?(key_index)
70
+ end
71
+
72
+ # Why the model cannot be tried in the stage; nil — it can (the
73
+ # quarantine is checked separately: it is a wait, not a ban).
74
+ def skip_reason(stage)
75
+ return @excluded_reason if @excluded_reason
76
+ return "excluded for this stage: #{@stage_exclusions[stage.to_s]}" if @stage_exclusions.key?(stage.to_s)
77
+ return "attempt limit of #{@limit} reached" unless requests_left?(stage)
78
+
79
+ nil
80
+ end
81
+ end
82
+ end
@@ -48,6 +48,13 @@ and reject the rest as duplicates.
48
48
 
49
49
  Do not add new findings. Do not change id.
50
50
  Do not change file, line, quoted_code. They do not need to be reinvented.
51
+
52
+ A candidate may carry a note field: the result of a mechanical check of its
53
+ link to the diff (the quote was not found in the shown diff, the line was
54
+ reset to null, the file was shown partially). It means the link could not be
55
+ confirmed, not that the finding is made up; take the note into account when
56
+ deciding keep/reject and check such a candidate against the diff more
57
+ carefully.
51
58
  Use refinement only when it makes a keep finding more precise.
52
59
 
53
60
  The answer must be a valid JSON object only, without markdown and without any
@@ -10,9 +10,9 @@ module Aireview
10
10
  @logger = logger
11
11
  end
12
12
 
13
- # Собственная заметка с ревью: {id:, key:} или nil. Метки недостаточно —
14
- # её может процитировать кто угодно, поэтому сверяем ещё и автора. Старый
15
- # формат без метки подхватываем только если заметки с меткой нет.
13
+ # Our own review note: {id:, key:} or nil. The marker alone is not
14
+ # enough — anyone can quote it, so the author is checked too. The old
15
+ # format without a marker is picked up only when no marked note exists.
16
16
  def existing_review(project_id:, iid:)
17
17
  author_id = current_user_id
18
18
  legacy = nil
@@ -49,9 +49,9 @@ module Aireview
49
49
  "#{ReviewMarker.build(key)}\n#{PREFIX}\n\n#{review_body}"
50
50
  end
51
51
 
52
- # Без надёжного автора матчинг по одной метке небезопасен: процитировать её
53
- # может кто угодно, и тогда чужая заметка либо отменит ревью, либо будет
54
- # перезаписана. Поэтому ошибку не глушим.
52
+ # Without a reliable author, matching by the marker alone is unsafe:
53
+ # anyone can quote it, and then someone else's note would either cancel
54
+ # the review or be overwritten. So the error is not swallowed.
55
55
  def current_user_id
56
56
  return @current_user_id if defined?(@current_user_id)
57
57
 
@@ -0,0 +1,98 @@
1
+ # frozen_string_literal: true
2
+ require 'json'
3
+ require_relative 'utils'
4
+
5
+ module Aireview
6
+ # Parses and validates the shape of an LLM answer: JSON (code fences
7
+ # included) or a ready structure by the schema → a hash with string keys
8
+ # and checked ids. A shape error is SchemaError; whether to repair the
9
+ # answer with another request is the pipeline's decision. The same parser
10
+ # lets `aireview models check` judge whether a model holds the schema.
11
+ class ResultParser
12
+ class SchemaError < StandardError
13
+ end
14
+
15
+ DECISIONS = %w[keep reject].freeze
16
+
17
+ # expected — :generate or :critique; critique_candidate_ids — the
18
+ # candidate ids Critique must give a verdict on, each of them.
19
+ def parse(raw, expected:, critique_candidate_ids: nil)
20
+ parsed = Utils.normalize_hash(raw.is_a?(Hash) ? raw : JSON.parse(strip_code_fences(raw.to_s)))
21
+
22
+ case expected.to_sym
23
+ when :generate then generate_result(parsed)
24
+ when :critique then critique_result(parsed, critique_candidate_ids)
25
+ else raise ArgumentError, "Unknown expected JSON schema: #{expected.inspect}"
26
+ end
27
+ end
28
+
29
+ private
30
+
31
+ def strip_code_fences(text)
32
+ stripped = text.strip
33
+ return stripped unless stripped.start_with?('```')
34
+
35
+ stripped
36
+ .sub(/\A```[[:alnum:]_-]*[ \t]*\r?\n?/, '')
37
+ .sub(/\r?\n?```[ \t]*\z/, '')
38
+ .strip
39
+ end
40
+
41
+ def generate_result(parsed)
42
+ parsed = {'summary' => nil, 'candidates' => parsed} if parsed.is_a?(Array)
43
+ valid_shape = parsed.is_a?(Hash) && parsed['candidates'].is_a?(Array)
44
+ raise SchemaError, 'expected an object with summary and candidates array' unless valid_shape
45
+ raise SchemaError, 'each generate candidate must be an object' unless parsed['candidates'].all?(Hash)
46
+
47
+ parsed['summary'] = nil unless parsed.key?('summary')
48
+ identifiers!(parsed['candidates'].map { |candidate| Utils.presence(candidate['id']) },
49
+ missing: 'each generate candidate must include a non-empty id',
50
+ duplicates: 'duplicate generate candidate ids')
51
+ parsed
52
+ end
53
+
54
+ def critique_result(parsed, expected_ids)
55
+ valid_shape = parsed.is_a?(Hash) && parsed['verdicts'].is_a?(Array)
56
+ raise SchemaError, 'expected an object with verdicts array' unless valid_shape
57
+ raise SchemaError, 'each critique verdict must be an object' unless parsed['verdicts'].all?(Hash)
58
+
59
+ verdict_ids = parsed['verdicts'].map { |verdict| Utils.presence(verdict['id']) }
60
+ identifiers!(verdict_ids, missing: 'each verdict must include a non-empty id',
61
+ duplicates: 'duplicate verdict ids')
62
+ expected_verdict_ids!(verdict_ids, expected_ids)
63
+ parsed['verdicts'].each { |verdict| verdict!(verdict) }
64
+ parsed
65
+ end
66
+
67
+ def identifiers!(identifiers, missing:, duplicates:)
68
+ raise SchemaError, missing unless identifiers.all?
69
+
70
+ duplicate_ids = identifiers.tally.select { |_, count| count > 1 }.keys
71
+ raise SchemaError, "#{duplicates}: #{duplicate_ids.join(', ')}" unless duplicate_ids.empty?
72
+ end
73
+
74
+ def expected_verdict_ids!(verdict_ids, expected_ids)
75
+ return unless expected_ids
76
+
77
+ unknown_ids = verdict_ids - expected_ids
78
+ missing_ids = expected_ids - verdict_ids
79
+ raise SchemaError, "unknown verdict ids: #{unknown_ids.join(', ')}" unless unknown_ids.empty?
80
+ raise SchemaError, "missing verdict ids: #{missing_ids.join(', ')}" unless missing_ids.empty?
81
+ end
82
+
83
+ def verdict!(verdict)
84
+ id = Utils.presence(verdict['id'])
85
+ decision = verdict['decision'].to_s.strip.downcase
86
+ raise SchemaError, "invalid verdict decision for #{id}" unless DECISIONS.include?(decision)
87
+ if verdict.key?('refinement') && decision != 'keep'
88
+ raise SchemaError,
89
+ "reject verdict cannot include refinement for #{id}"
90
+ end
91
+
92
+ refinement = verdict['refinement']
93
+ return if refinement.nil? || refinement.is_a?(Hash)
94
+
95
+ raise SchemaError, "refinement must be an object for #{id}"
96
+ end
97
+ end
98
+ end
@@ -3,8 +3,8 @@ require 'digest'
3
3
  require 'json'
4
4
 
5
5
  module Aireview
6
- # Скрытая метка в теле заметки: по ней ревью находит собственный комментарий
7
- # и понимает, менялось ли с прошлого раза то, что влияет на результат.
6
+ # A hidden marker in the note body: it lets the review find its own
7
+ # comment and tell whether anything that affects the result changed.
8
8
  module ReviewMarker
9
9
  PATTERN = /<!--\s*aireview:key=([0-9a-f]+)\s*-->/
10
10
 
@@ -19,28 +19,27 @@ module Aireview
19
19
  match && match[1]
20
20
  end
21
21
 
22
- # Ключ считается от готовых промптов, а не от одного SHA: так в него сами
23
- # собой попадают дифф, описание MR, контекст Jira, инструкции ревью и
24
- # ignore_paths. Модели и провайдеры добавляются рядом — на промпт они не
25
- # влияют, но на результат влияют.
22
+ # The key is computed from the assembled prompts, not from a single SHA:
23
+ # that way the diff, the MR description, the Jira context, the review
24
+ # instructions and ignore_paths enter it by themselves. What else affects
25
+ # the result — provider, model and temperature of the stages, the shared
26
+ # pool with its critique policy — is known by Config#result_signature.
27
+ # Without a pool the key is the same as before.
26
28
  def key(prompts:, config:)
29
+ signature = config.result_signature
27
30
  source = {
28
- 'generate' => [
29
- config.generate_provider,
30
- prompts[:generate_model],
31
- prompts[:generate_temperature],
32
- prompts[:generate_prompt]
33
- ],
34
- 'critique' => critique_source(prompts, config)
31
+ 'generate' => [*signature['generate'], prompts[:generate_prompt]],
32
+ 'critique' => prompts[:critique_prompt] ? [*signature['critique'], prompts[:critique_prompt]] : nil
35
33
  }
34
+ source['pool'] = signature['pool'] if signature['pool']
36
35
 
37
36
  Digest::SHA256.hexdigest(JSON.generate(source))[0, 16]
38
37
  end
39
38
 
40
- # То, что делает результат ревью устаревшим: новый коммит, смена целевой
41
- # ветки, перебазирование со сдвигом базы сравнения, а также правка
42
- # заголовка или описания — из них берутся требования, с которыми ревью
43
- # сверяет код.
39
+ # What makes a review result stale: a new commit, a target branch change,
40
+ # a rebase that moves the comparison base, and an edit of the title or
41
+ # the description — the requirements the review checks the code against
42
+ # come from those.
44
43
  def state(merge_request)
45
44
  {
46
45
  'sha' => merge_request['sha'],
@@ -50,16 +49,5 @@ module Aireview
50
49
  'description' => merge_request['description']
51
50
  }
52
51
  end
53
-
54
- def critique_source(prompts, config)
55
- return nil unless prompts[:critique_prompt]
56
-
57
- [
58
- config.critique_provider,
59
- prompts[:critique_model],
60
- prompts[:critique_temperature],
61
- prompts[:critique_prompt]
62
- ]
63
- end
64
52
  end
65
53
  end