aireview 0.3.0 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,12 +2,13 @@
2
2
  require_relative 'errors'
3
3
 
4
4
  module Aireview
5
- # Укладывает контекст ревью в бюджет символов и запоминает, что при этом не
6
- # вошло. Секции MR и Jira режутся до своих лимитов с сохранением начала,
7
- # дифф по целым файлам, затем по целым хункам; внутри хунка не режем.
5
+ # Fits the review context into a character budget and remembers what was
6
+ # left out. The MR and Jira sections are cut to their limits keeping the
7
+ # beginning, the diff by whole files, then by whole hunks; a hunk is never
8
+ # cut inside.
8
9
  module ContextBudget
9
- # Пути, которые не вошли, перечисляются в конце диффа; список ограничен,
10
- # чтобы сам не съел бюджет.
10
+ # The paths that did not fit are listed at the end of the diff; the list
11
+ # is capped so that it does not eat the budget itself.
11
12
  NOT_SHOWN_LIST_LIMIT = 20
12
13
  TRAILER_RESERVE_CHARS = 400
13
14
 
@@ -26,7 +27,8 @@ module Aireview
26
27
 
27
28
  Packed = Struct.new(:text, :shown_hunks, :total_hunks, keyword_init: true)
28
29
 
29
- # Начало важнее конца: требования и критерии приёмки обычно там.
30
+ # The beginning matters more than the end: requirements and acceptance
31
+ # criteria usually live there.
30
32
  def self.truncate_section(text, limit:, label:, coverage:)
31
33
  text = text.to_s
32
34
  return text if text.length <= limit
@@ -39,11 +41,11 @@ module Aireview
39
41
  Packer.new(entries, budget: budget, coverage: coverage).pack
40
42
  end
41
43
 
42
- # Файлы без хунков идут первыми: они дёшевы и всегда полезны для картины
43
- # MR. Текстовые файлы идут в порядке GitLab, пока влезают; первый файл,
44
- # который не влезает, показывается частично, всё после него не показывается.
45
- # Хунк, который не влез бы даже в пустой бюджет, пропускается с пометкой,
46
- # а не останавливает раскладку.
44
+ # Files without hunks go first: they are cheap and always useful for the
45
+ # picture of the MR. Text files go in GitLab order while they fit; the
46
+ # first file that does not fit is shown partially, everything after it is
47
+ # not shown. A hunk that would not fit even into an empty budget is
48
+ # skipped with a mark instead of stopping the layout.
47
49
  class Packer
48
50
  def initialize(entries, budget:, coverage:)
49
51
  @non_text, @text = entries.partition { |entry| !entry.text? }
@@ -62,9 +64,9 @@ module Aireview
62
64
 
63
65
  private
64
66
 
65
- # Что-то придётся опустить, значит нужен хвост со списком пропущенного.
66
- # @used считает весь собранный текст, включая разделители между
67
- # файлами: результат не должен выйти за бюджет ни на символ.
67
+ # Something has to be left out, so a trailer listing the skipped paths
68
+ # is needed. @used counts the whole assembled text, separators between
69
+ # files included: the result must not exceed the budget by a character.
68
70
  def pack_within_limit
69
71
  @parts = @non_text.map(&:render)
70
72
  @used = joined_length(@parts)
@@ -95,12 +97,12 @@ module Aireview
95
97
  shown_hunks
96
98
  end
97
99
 
98
- # Место под следующий кусок с учётом разделителя перед ним.
100
+ # Room for the next piece, the separator before it included.
99
101
  def remaining
100
102
  @limit - @used - (@parts.empty? ? 0 : 1)
101
103
  end
102
104
 
103
- # Возвращает [текст, число показанных хунков, остановлена ли раскладка].
105
+ # Returns [text, number of shown hunks, whether the layout stopped].
104
106
  def pack_entry(entry)
105
107
  full = entry.render
106
108
  return [full, entry.hunks.size, false] if full.length <= remaining
@@ -113,10 +115,10 @@ module Aireview
113
115
  [entry.header + body + partial_marker(entry, shown), shown, stopped]
114
116
  end
115
117
 
116
- # Место сначала отдаётся хункам, которые можно показать, и только на
117
- # остаток добавляются пометки о слишком больших: иначе пометки могли бы
118
- # вытеснить единственный подходящий хунк. Факт пропуска в покрытие
119
- # попадает независимо от того, есть ли для пометки место.
118
+ # Room goes first to the hunks that can be shown, and only the remainder
119
+ # to the marks about oversized ones: otherwise the marks could push out
120
+ # the only fitting hunk. The skip is recorded in the coverage whether
121
+ # or not there is room for the mark.
120
122
  def pack_hunks(entry)
121
123
  base = entry.header.length + partial_marker(entry, 0).length
122
124
  shown = []
@@ -1,6 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
  require_relative 'utils'
3
3
  require_relative 'errors'
4
+ require_relative 'stages'
4
5
  require_relative 'secret_scrubber'
5
6
  require_relative 'diff_fetcher'
6
7
  require_relative 'context_budget'
@@ -15,14 +16,13 @@ module Aireview
15
16
  }.freeze
16
17
  CHANGES_HEADER = "Changes:\n"
17
18
  CANDIDATES_HEADER = "\n\nCandidates JSON from Generate:\n"
18
- # Резерв под кандидатов в промпте критика: три кандидата по ~1 500
19
- # символов. Оценка, не гарантия; фактический размер проверяется перед
20
- # отправкой.
19
+ # Room for the candidates in the Critique prompt: three candidates of
20
+ # ~1,500 characters. An estimate, not a guarantee; the actual size is
21
+ # checked before sending.
21
22
  CANDIDATES_RESERVE_CHARS = 4_500
22
- STAGES = %i[generate critique].freeze
23
23
 
24
- # Контекст одного прогона: обе стадии получают одинаковые MR, Jira и дифф,
25
- # усечённые один раз под самую тесную из стадий.
24
+ # The context of one run: both stages get the same MR, Jira and diff,
25
+ # truncated once for the tightest of the stages.
26
26
  Context = Struct.new(:user_prompt, :diff_text, :coverage, :sizes, keyword_init: true)
27
27
 
28
28
  def initialize(config:, logger: Logger.new($stderr))
@@ -53,16 +53,16 @@ module Aireview
53
53
  end
54
54
 
55
55
  def build_generate_prompt(context)
56
- check_stage_size!(:generate, system_prompt(:generate), context.user_prompt)
56
+ check_stage_size!('generate', system_prompt('generate'), context.user_prompt)
57
57
  end
58
58
 
59
59
  def build_critique_prompt(context, candidates_json:)
60
60
  user = "#{context.user_prompt}#{CANDIDATES_HEADER}#{scrub_text(candidates_json)}"
61
- check_stage_size!(:critique, system_prompt(:critique), user)
61
+ check_stage_size!('critique', system_prompt('critique'), user)
62
62
  end
63
63
 
64
64
  def system_prompt(stage)
65
- template = stage.to_sym == :critique ? CRITIQUE_PROMPT_TEMPLATE : GENERATE_PROMPT_TEMPLATE
65
+ template = stage.to_s == 'critique' ? CRITIQUE_PROMPT_TEMPLATE : GENERATE_PROMPT_TEMPLATE
66
66
  extras = []
67
67
  if Aireview::Utils.present?(@config.review_instructions)
68
68
  extras << "Additional project instructions:\n#{scrub_text(@config.review_instructions.strip)}"
@@ -72,10 +72,11 @@ module Aireview
72
72
  [template, *extras].join("\n\n")
73
73
  end
74
74
 
75
- # Проверка перед отправкой: если кандидаты вышли за резерв и запрос не
76
- # помещается, это ошибка, а не повод молча резать контекст, который
77
- # генератор уже видел.
75
+ # A check before sending: when the candidates exceed the reserve and the
76
+ # request does not fit, that is an error, not a reason to silently cut
77
+ # the context Generate has already seen.
78
78
  def check_stage_size!(stage, system, user)
79
+ stage = stage.to_s
79
80
  limit = @config.max_prompt_chars(stage)
80
81
  total = system.length + user.length
81
82
  if total > limit
@@ -89,10 +90,10 @@ module Aireview
89
90
 
90
91
  private
91
92
 
92
- # Минимум по стадиям: контекст один на прогон, поэтому он должен
93
- # помещаться в каждую из них вместе с её системным промптом и резервом.
93
+ # The minimum over the stages: the context is one per run, so it must fit
94
+ # into each of them together with its system prompt and reserve.
94
95
  def context_budget(critique:)
95
- stages = critique ? STAGES : [:generate]
96
+ stages = critique ? STAGES : ['generate']
96
97
  budgets = stages.to_h { |stage| [stage, stage_budget(stage)] }
97
98
  stage, budget = budgets.min_by { |_, value| value }
98
99
  return budget if budget.positive?
@@ -104,7 +105,7 @@ module Aireview
104
105
  end
105
106
 
106
107
  def stage_budget(stage)
107
- reserve = stage == :critique ? CANDIDATES_RESERVE_CHARS + CANDIDATES_HEADER.length : 0
108
+ reserve = stage == 'critique' ? CANDIDATES_RESERVE_CHARS + CANDIDATES_HEADER.length : 0
108
109
  @config.max_prompt_chars(stage) - system_prompt(stage).length - reserve
109
110
  end
110
111
 
@@ -152,7 +153,7 @@ module Aireview
152
153
  end
153
154
 
154
155
  def context_sizes(fixed:, packed:, budget:, diff_budget:, critique:)
155
- stages = critique ? STAGES : [:generate]
156
+ stages = critique ? STAGES : ['generate']
156
157
  {
157
158
  context_budget: budget,
158
159
  diff_budget: diff_budget,
@@ -7,13 +7,14 @@ module Aireview
7
7
  DIFF_UNAVAILABLE = '[diff not available]'
8
8
  BINARY_DIFF = /\ABinary files .* differ/
9
9
 
10
- # Один файл из ответа GitLab: заголовок, хунки и что с ним можно делать.
10
+ # One file from the GitLab answer: the header, the hunks and what can be done with it.
11
11
  # kind:
12
- # :text есть хунки, код можно проверить;
13
- # :no_text_changes переименование, смена режима, пустой файл: проверять
14
- # нечего;
15
- # :unavailable GitLab не отдал дифф (too_large, бинарник, пустой
16
- # дифф без причины): код есть, но проверить его не удалось.
12
+ # :text there are hunks, the code can be checked;
13
+ # :no_text_changes a rename, a mode change, an empty file: nothing to
14
+ # check;
15
+ # :unavailable GitLab did not return the diff (too_large, a binary,
16
+ # an empty diff without a reason): the code exists,
17
+ # but could not be checked.
17
18
  class Entry
18
19
  attr_reader :path, :kind, :header, :hunks
19
20
 
@@ -49,9 +50,9 @@ module Aireview
49
50
 
50
51
  private
51
52
 
52
- # Пустой дифф без объяснимой причины (переименование, смена режима,
53
- # пустой новый или удалённый файл) считаем недоступным: код есть, но
54
- # GitLab его не отдал.
53
+ # An empty diff without an explainable reason (a rename, a mode change,
54
+ # an empty new or deleted file) counts as unavailable: the code exists,
55
+ # but GitLab did not return it.
55
56
  def classify(change, diff)
56
57
  return :unavailable if change['too_large']
57
58
  return :unavailable if diff.match?(BINARY_DIFF)
@@ -67,8 +68,8 @@ module Aireview
67
68
  modes.none?(&:nil?) && modes.uniq.size == 2
68
69
  end
69
70
 
70
- # Хунки режем по заголовкам @@; текст до первого @@ (или дифф без них,
71
- # например заглушка про секретный файл) считается одним хунком.
71
+ # Hunks are split at the @@ headers; the text before the first @@ (or a
72
+ # diff without any, such as the secret-file placeholder) counts as one hunk.
72
73
  def split_hunks(diff)
73
74
  diff = "#{diff}\n" unless diff.end_with?("\n")
74
75
  pieces = diff.split(/^(?=@@ )/)
@@ -1,7 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Aireview
4
- # Вывод --dry-run: настройки, сводка контекста и промпты обеих стадий.
4
+ # The --dry-run output: settings, the context summary and the prompts of both stages.
5
5
  class DryRunReport
6
6
  def initialize(out)
7
7
  @out = out
@@ -33,15 +33,37 @@ module Aireview
33
33
 
34
34
  def render_settings(dry_run)
35
35
  @out.puts('=== LLM SETTINGS ===')
36
- @out.puts("Generate: #{dry_run[:generate_model]} temperature=#{dry_run[:generate_temperature]}")
37
- list('fallbacks', dry_run[:generate_fallbacks], separator: ' -> ')
36
+ render_config_paths(dry_run[:config_paths])
37
+ render_stage(dry_run, :generate)
38
38
  if dry_run[:critique_prompt]
39
- @out.puts("Critique: #{dry_run[:critique_model]} temperature=#{dry_run[:critique_temperature]}")
40
- list('fallbacks', dry_run[:critique_fallbacks], separator: ' -> ')
39
+ render_stage(dry_run, :critique)
41
40
  else
42
41
  @out.puts('Critique: disabled')
43
42
  end
43
+ @out.puts("Critique rule: #{dry_run[:critique_rule]}") if dry_run[:critique_rule]
44
44
  render_reserves(dry_run)
45
+ list('warnings', dry_run[:warnings], separator: "\n ")
46
+ end
47
+
48
+ def render_config_paths(paths)
49
+ return if paths.nil? || paths.empty?
50
+
51
+ @out.puts("Config: #{paths.map { |name, path| "#{name} #{path}" }.join(', ')}")
52
+ end
53
+
54
+ # The source of every setting is the layer it came from: built-in, image
55
+ # defaults, .aireview.yml, env or cli.
56
+ def render_stage(dry_run, stage)
57
+ sources = dry_run.dig(:sources, stage) || {}
58
+ @out.puts("#{stage.capitalize}: #{dry_run[:"#{stage}_model"]} " \
59
+ "temperature=#{dry_run[:"#{stage}_temperature"]}#{origin(sources, :model, :provider)}")
60
+ fallbacks = dry_run[:"#{stage}_fallbacks"]
61
+ list('fallbacks', fallbacks, separator: ' -> ', suffix: origin(sources, :fallbacks))
62
+ end
63
+
64
+ def origin(sources, *keys)
65
+ parts = keys.filter_map { |key| "#{key} from #{sources[key]}" if sources[key] }
66
+ parts.empty? ? '' : " (#{parts.join(', ')})"
45
67
  end
46
68
 
47
69
  def render_context_sizes(sizes)
@@ -67,12 +89,15 @@ module Aireview
67
89
  def render_reserves(dry_run)
68
90
  keys = Array(dry_run[:api_keys]).map { |provider, count| "#{provider} #{count}" }
69
91
  @out.puts("API keys: #{keys.join(', ')}") unless keys.empty?
70
- @out.puts("Time budget: #{dry_run[:time_budget]}s") if dry_run[:time_budget]
92
+ return unless dry_run[:time_budget]
93
+
94
+ quarantine = dry_run[:overloaded_quarantine]
95
+ @out.puts("Time budget: #{dry_run[:time_budget]}s#{", overloaded quarantine: #{quarantine}s" if quarantine}")
71
96
  end
72
97
 
73
- def list(title, items, separator: ', ')
98
+ def list(title, items, separator: ', ', suffix: '')
74
99
  items = Array(items)
75
- @out.puts(" #{title}: #{items.join(separator)}") unless items.empty?
100
+ @out.puts(" #{title}: #{items.join(separator)}#{suffix}") unless items.empty?
76
101
  end
77
102
  end
78
103
  end
@@ -3,7 +3,15 @@ module Aireview
3
3
  class Error < StandardError; end
4
4
  class ConfigError < Error; end
5
5
  class ParseError < Error; end
6
+ # Repairing invalid JSON is impossible: the model that answered is out of
7
+ # attempts or quarantined with no budget to wait. For the pipeline it is
8
+ # the same as an invalid result — the stage restarts on another model.
9
+ class RepairImpossibleError < ParseError; end
6
10
  class ApiError < Error; end
11
+ # A pinned route (the JSON repair by the same model) could not answer: out
12
+ # of attempts, excluded, or quarantined with no budget to wait. Not fatal
13
+ # for the run — the stage restarts on another model.
14
+ class RouteExhaustedError < ApiError; end
7
15
  class ContextBudgetError < Error; end
8
16
  class HelpRequested < Error; end
9
17
  end
@@ -35,8 +35,9 @@ module Aireview
35
35
  get_json('user')
36
36
  end
37
37
 
38
- # Retry создаёт новый job в том же pipeline. Старые попытки видны только
39
- # с include_retried; один лишь новый CI_JOB_ID бывает и после обычного push.
38
+ # A Retry creates a new job in the same pipeline. Earlier attempts are
39
+ # visible only with include_retried; a new CI_JOB_ID alone happens after an
40
+ # ordinary push too.
40
41
  def retried_job?(project_id, job_id)
41
42
  current = get_json("projects/#{project_id}/jobs/#{job_id}")
42
43
  validate_retry_job!(current, pipeline: true)
@@ -54,13 +55,13 @@ module Aireview
54
55
  raise ApiError, 'Too many pipeline jobs to determine whether this job is a retry'
55
56
  end
56
57
 
57
- # Заметки отдаются страницами и сортируются по времени создания, а
58
- # обновление комментария в этом порядке его не поднимает: на длинном MR
59
- # своё ревью оказывается далеко не на первой странице. Обрывать обход молча
60
- # нельзя — по неполному списку ревью решит, что заметки нет, и создаст
61
- # вторую, поэтому упираемся в предел с ошибкой. Порядок задаётся явно: от
62
- # него зависит, какую из старых заметок без метки подхватит ревью, и
63
- # полагаться тут на дефолт API не стоит.
58
+ # Notes come in pages sorted by creation time, and updating a comment
59
+ # does not move it up in that order: on a long MR our review is far from
60
+ # the first page. Stopping the walk silently is not an option — with an
61
+ # incomplete list the review would decide the note is missing and create
62
+ # a second one, so the limit fails with an error. The order is explicit:
63
+ # it decides which old unmarked note the review picks up, and the API
64
+ # default is not to be relied on here.
64
65
  def fetch_merge_request_notes(project_id, iid)
65
66
  notes = []
66
67
  page = 1
@@ -0,0 +1,146 @@
1
+ # frozen_string_literal: true
2
+ require 'json'
3
+ require 'timeout'
4
+ require_relative 'errors'
5
+ require_relative 'utils'
6
+
7
+ module Aireview
8
+ # One request to one model with one key through RubyLLM. Retries, keys and
9
+ # reserves belong to LlmRouter: the built-in RubyLLM/Faraday retries (3 by
10
+ # default) are off, otherwise every router attempt would turn into four
11
+ # HTTP requests and burn quota before the error reaches the classifier.
12
+ class LlmClient
13
+ # What a stage sends to the model; the same for every route of the stage.
14
+ Prompt = Struct.new(:stage, :system, :user, :temperature, :schema, keyword_init: true) do
15
+ def chars
16
+ system.length + user.length
17
+ end
18
+ end
19
+
20
+ def initialize(config:, logger: Logger.new($stderr))
21
+ @config = config
22
+ @logger = logger
23
+ @contexts = {}
24
+ end
25
+
26
+ # Returns the RubyLLM answer; read it with LlmClient.content. A request
27
+ # error is re-raised as is — LlmFailure classifies it.
28
+ def request(prompt, candidate:, key:, timeout:, key_index: 0)
29
+ load_ruby_llm
30
+ stage = prompt.stage.to_s
31
+ model = candidate.model
32
+ @logger.info("LLM #{stage} request started (model=#{model}, temperature=#{prompt.temperature})")
33
+ chat = build_chat(context: context(stage, candidate.provider, key, key_index), stage: stage,
34
+ model: model, provider: candidate.provider)
35
+ chat = configure_reasoning(chat: chat, model: model, provider: candidate.provider)
36
+ .with_temperature(prompt.temperature.to_f)
37
+ .with_schema(prompt.schema)
38
+ chat.with_instructions(prompt.system)
39
+ response = Timeout.timeout(timeout) { chat.ask(prompt.user) }
40
+ @logger.info("LLM #{stage} request completed (model=#{model}#{token_counts(response)})")
41
+ response
42
+ rescue Timeout::Error
43
+ @logger.warn("LLM #{stage} request timed out after #{timeout.round} seconds (model=#{model})")
44
+ raise
45
+ end
46
+
47
+ # The answer as RubyLLM 1.x gave it under a schema: the parsed JSON when
48
+ # the text is JSON, the text itself otherwise (the pipeline repairs it).
49
+ # RubyLLM 2 always returns the text, and a Hash that breaks the schema
50
+ # would go to a repair request instead of the next model. An empty
51
+ # answer stays an empty String (Message#parsed would turn it into nil);
52
+ # JSON null becomes nil, as in 1.x.
53
+ def self.content(response)
54
+ content = response.content
55
+ return content unless content.is_a?(String) && !content.empty?
56
+
57
+ response.parsed
58
+ rescue JSON::ParserError
59
+ content
60
+ end
61
+
62
+ private
63
+
64
+ # Token counts as the provider reported them, one line per attempt. They
65
+ # are not summed: Gemini already counts thinking into output. Input is
66
+ # what was not read from or written to a cache; the prompt size is
67
+ # input + cache_read + cache_write, and a retry of the same prompt on
68
+ # Gemini is often served from its implicit cache.
69
+ def token_counts(response)
70
+ tokens = response.tokens
71
+ cached = {cache_read: tokens.cache_read, cache_write: tokens.cache_write}.reject { |_, count| count.to_i.zero? }
72
+ counts = {input: tokens.input, output: tokens.output, thinking: tokens.thinking}.compact.merge(cached)
73
+ return '' if counts.empty?
74
+
75
+ ", tokens: #{counts.map { |name, count| "#{name}=#{count}" }.join(' ')}"
76
+ end
77
+
78
+ def load_ruby_llm
79
+ require 'ruby_llm'
80
+ rescue LoadError => e
81
+ @logger.error("LLM setup failed: #{e.message}")
82
+ raise ConfigError, "Missing dependency: #{e.message}"
83
+ end
84
+
85
+ def configure_reasoning(chat:, model:, provider:)
86
+ return chat unless provider == 'ollama' && model.start_with?('gpt-oss:')
87
+
88
+ chat.with_thinking(effort: :low)
89
+ end
90
+
91
+ def build_chat(context:, stage:, model:, provider:)
92
+ context.chat(model: model, provider: provider.to_sym)
93
+ rescue RubyLLM::ModelNotFoundError
94
+ @logger.warn(
95
+ "LLM #{stage}: model not found in RubyLLM registry; " \
96
+ "using fallback with incomplete model metadata " \
97
+ "(model=#{model}, provider=#{provider})"
98
+ )
99
+ context.chat(model: model, provider: provider.to_sym, assume_model_exists: true)
100
+ end
101
+
102
+ # A RubyLLM context per stage, provider and key index: switching the key
103
+ # is another context, not an edit of the global config.
104
+ def context(stage, provider, key, key_index)
105
+ @contexts[[stage, provider, key_index]] ||= build_context(provider.to_s, key)
106
+ end
107
+
108
+ def build_context(provider, api_key)
109
+ RubyLLM.context do |ruby_config|
110
+ configure_http_proxy(ruby_config)
111
+ ruby_config.request_timeout = @config.llm_timeout.to_f
112
+ ruby_config.max_retries = 0
113
+ configure_provider(ruby_config, provider, api_key)
114
+ end
115
+ end
116
+
117
+ def configure_http_proxy(ruby_config)
118
+ return unless Aireview::Utils.present?(@config.llm_http_proxy)
119
+
120
+ ruby_config.http_proxy = @config.llm_http_proxy
121
+ end
122
+
123
+ def configure_provider(ruby_config, provider, api_key)
124
+ case provider
125
+ when 'gemini', 'openai', 'openrouter'
126
+ configure_remote_provider(ruby_config, provider, api_key)
127
+ when 'anthropic'
128
+ ruby_config.anthropic_api_key = api_key
129
+ when 'ollama'
130
+ ruby_config.ollama_api_base = @config.ollama_api_base
131
+ else
132
+ raise ConfigError, "Unsupported LLM provider: #{provider.inspect}"
133
+ end
134
+ end
135
+
136
+ # RubyLLM 2 sends OpenAI requests to the Responses API; a compatible
137
+ # server behind LLM_API_BASE usually has Chat Completions only.
138
+ def configure_remote_provider(ruby_config, provider, api_key)
139
+ ruby_config.public_send("#{provider}_api_key=", api_key)
140
+ ruby_config.openai_protocol = :chat_completions if provider == 'openai'
141
+ return unless Aireview::Utils.present?(@config.llm_api_base)
142
+
143
+ ruby_config.public_send("#{provider}_api_base=", @config.llm_api_base)
144
+ end
145
+ end
146
+ end
@@ -3,17 +3,27 @@ require 'json'
3
3
  require 'timeout'
4
4
 
5
5
  module Aireview
6
- # Классификация ошибки LLM-запроса. Отвечает только на вопрос «что это»,
7
- # решение «повторить, сменить ключ или модель» принимает LlmRouter.
6
+ # Classifies an LLM request error. Answers only "what is it"; the decision
7
+ # to retry, switch the key or the model belongs to LlmRouter.
8
8
  #
9
- # :daily_quota — суточная квота проекта на модель, повторы бесполезны;
10
- # :rate_limit — минутный лимит, пройдёт через подсказанное время;
11
- # :overloaded — 503/«high demand» у модели;
12
- # :timeout — ответа нет дольше LLM_TIMEOUT;
13
- # :fatal — ошибка API, которую резервы не лечат;
14
- # :unhandled — не ошибка провайдера, пробрасывается как есть.
9
+ # :daily_quota — the project's daily quota for the model, retries are useless;
10
+ # :rate_limit — a per-minute limit, passes after the hinted time;
11
+ # :overloaded — 503 / "high demand" on the model;
12
+ # :timeout — no answer for longer than LLM_TIMEOUT;
13
+ # :unavailable — the provider has no such model: retired, a typo, not pulled into Ollama;
14
+ # :fatal — an API error that reserves do not cure;
15
+ # :unhandled — not a provider error, re-raised as is.
15
16
  module LlmFailure
16
- KINDS = %i[daily_quota rate_limit overloaded timeout fatal unhandled].freeze
17
+ KINDS = %i[daily_quota rate_limit overloaded timeout unavailable fatal unhandled].freeze
18
+ # Only by the provider's text about the model: a bare 404 is also what a
19
+ # wrong LLM_API_BASE or proxy answers, and the next model will not help.
20
+ # Ollama: `model 'x' not found` (through /v1) and `model "x" not found,
21
+ # try pulling it first` (older versions and /api).
22
+ UNAVAILABLE_MODEL_TEXT = Regexp.union(
23
+ /\bmodels?\/[\w.:-]+ is not found\b/i,
24
+ /\bis not supported for generateContent\b/i,
25
+ /\bmodel ['"][^'"]+['"] not found\b/i
26
+ )
17
27
  QUOTA_FAILURE_TYPE = 'type.googleapis.com/google.rpc.QuotaFailure'
18
28
  DAILY_QUOTA_ID = /PerDay/i
19
29
  DAILY_QUOTA_TEXT = /\bper\s+day\b|\bdaily\b/i
@@ -21,18 +31,22 @@ module Aireview
21
31
 
22
32
  module_function
23
33
 
24
- # Сведения о квоте смотрим раньше класса исключения: RubyLLM превращает
25
- # 429 со словом input_token в ContextLengthExceededError, хотя это
26
- # исчерпанная токенная квота, а не слишком длинный запрос.
34
+ # Quota details are checked before the exception class: RubyLLM turns a
35
+ # 429 mentioning input_token into ContextLengthExceededError, although
36
+ # it is an exhausted token quota, not an oversized request.
27
37
  def classify(error)
28
38
  return :timeout if transport_timeout?(error)
29
39
  return :unhandled unless ruby_llm_error?(error)
30
40
 
31
- quota_kind(error) || (overloaded?(error) ? :overloaded : :fatal)
41
+ quota_kind(error) || model_kind(error) || (overloaded?(error) ? :overloaded : :fatal)
32
42
  end
33
43
 
34
- # Внешний Timeout.timeout и таймауты транспорта Faraday: последние не
35
- # наследуют ни Timeout::Error, ни RubyLLM::Error.
44
+ def model_kind(error)
45
+ :unavailable if error.message.to_s.match?(UNAVAILABLE_MODEL_TEXT)
46
+ end
47
+
48
+ # The outer Timeout.timeout and Faraday's transport timeouts: the latter
49
+ # inherit neither Timeout::Error nor RubyLLM::Error.
36
50
  def transport_timeout?(error)
37
51
  return true if error.is_a?(Timeout::Error) || error.is_a?(Errno::ETIMEDOUT)
38
52
 
@@ -47,10 +61,10 @@ module Aireview
47
61
  defined?(RubyLLM::Error) && error.is_a?(RubyLLM::Error)
48
62
  end
49
63
 
50
- # Google кладёт вид квоты в QuotaFailure.violations[].quotaId
51
- # (GenerateRequestsPerDay… / …PerMinute…). Метрика free_tier_requests
52
- # одна и та же у обоих, по ней не различить. Текст сообщения — запасной
53
- # признак, когда тела ответа нет.
64
+ # Google puts the kind of quota into QuotaFailure.violations[].quotaId
65
+ # (GenerateRequestsPerDay… / …PerMinute…). The free_tier_requests metric
66
+ # is the same for both, it cannot tell them apart. The message text is
67
+ # the fallback signal when there is no response body.
54
68
  def quota_kind(error)
55
69
  ids = quota_ids(error)
56
70
  return daily_quota_id?(ids) ? :daily_quota : :rate_limit unless ids.empty?
@@ -86,5 +100,8 @@ module Aireview
86
100
  match = message.to_s.match(RETRY_AFTER)
87
101
  match[1].to_f if match
88
102
  end
103
+
104
+ private_class_method :model_kind, :transport_timeout?, :overloaded?, :ruby_llm_error?, :quota_kind,
105
+ :daily_quota_id?, :quota_ids, :response_body
89
106
  end
90
107
  end