aireview 0.3.0 → 2.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +82 -0
- data/README.md +301 -29
- data/config/defaults.yml +47 -0
- data/lib/aireview/candidate_checker.rb +24 -22
- data/lib/aireview/cli.rb +79 -19
- data/lib/aireview/config.rb +65 -180
- data/lib/aireview/config_fallbacks.rb +79 -68
- data/lib/aireview/config_layers.rb +110 -0
- data/lib/aireview/config_limits.rb +10 -35
- data/lib/aireview/config_loader.rb +240 -0
- data/lib/aireview/context_budget.rb +22 -20
- data/lib/aireview/context_builder.rb +18 -17
- data/lib/aireview/diff_fetcher.rb +12 -11
- data/lib/aireview/dry_run_report.rb +33 -8
- data/lib/aireview/errors.rb +8 -0
- data/lib/aireview/gitlab_client.rb +10 -9
- data/lib/aireview/llm_client.rb +146 -0
- data/lib/aireview/llm_failure.rb +36 -19
- data/lib/aireview/llm_router.rb +315 -158
- data/lib/aireview/model_candidate.rb +28 -0
- data/lib/aireview/model_checker.rb +148 -0
- data/lib/aireview/model_pool.rb +224 -0
- data/lib/aireview/model_state.rb +82 -0
- data/lib/aireview/output_schemas.rb +3 -3
- data/lib/aireview/publisher.rb +6 -6
- data/lib/aireview/result_parser.rb +98 -0
- data/lib/aireview/review_marker.rb +16 -28
- data/lib/aireview/review_pipeline.rb +102 -85
- data/lib/aireview/review_renderer.rb +19 -11
- data/lib/aireview/reviewer.rb +44 -107
- data/lib/aireview/stage_chains.rb +113 -0
- data/lib/aireview/stages.rb +7 -0
- data/lib/aireview/utils.rb +29 -0
- data/lib/aireview/version.rb +1 -1
- data/lib/aireview.rb +1 -0
- metadata +21 -5
- data/lib/aireview/result_validation.rb +0 -65
|
@@ -2,12 +2,13 @@
|
|
|
2
2
|
require_relative 'errors'
|
|
3
3
|
|
|
4
4
|
module Aireview
|
|
5
|
-
#
|
|
6
|
-
#
|
|
7
|
-
#
|
|
5
|
+
# Fits the review context into a character budget and remembers what was
|
|
6
|
+
# left out. The MR and Jira sections are cut to their limits keeping the
|
|
7
|
+
# beginning, the diff by whole files, then by whole hunks; a hunk is never
|
|
8
|
+
# cut inside.
|
|
8
9
|
module ContextBudget
|
|
9
|
-
#
|
|
10
|
-
#
|
|
10
|
+
# The paths that did not fit are listed at the end of the diff; the list
|
|
11
|
+
# is capped so that it does not eat the budget itself.
|
|
11
12
|
NOT_SHOWN_LIST_LIMIT = 20
|
|
12
13
|
TRAILER_RESERVE_CHARS = 400
|
|
13
14
|
|
|
@@ -26,7 +27,8 @@ module Aireview
|
|
|
26
27
|
|
|
27
28
|
Packed = Struct.new(:text, :shown_hunks, :total_hunks, keyword_init: true)
|
|
28
29
|
|
|
29
|
-
#
|
|
30
|
+
# The beginning matters more than the end: requirements and acceptance
|
|
31
|
+
# criteria usually live there.
|
|
30
32
|
def self.truncate_section(text, limit:, label:, coverage:)
|
|
31
33
|
text = text.to_s
|
|
32
34
|
return text if text.length <= limit
|
|
@@ -39,11 +41,11 @@ module Aireview
|
|
|
39
41
|
Packer.new(entries, budget: budget, coverage: coverage).pack
|
|
40
42
|
end
|
|
41
43
|
|
|
42
|
-
#
|
|
43
|
-
# MR.
|
|
44
|
-
#
|
|
45
|
-
#
|
|
46
|
-
#
|
|
44
|
+
# Files without hunks go first: they are cheap and always useful for the
|
|
45
|
+
# picture of the MR. Text files go in GitLab order while they fit; the
|
|
46
|
+
# first file that does not fit is shown partially, everything after it is
|
|
47
|
+
# not shown. A hunk that would not fit even into an empty budget is
|
|
48
|
+
# skipped with a mark instead of stopping the layout.
|
|
47
49
|
class Packer
|
|
48
50
|
def initialize(entries, budget:, coverage:)
|
|
49
51
|
@non_text, @text = entries.partition { |entry| !entry.text? }
|
|
@@ -62,9 +64,9 @@ module Aireview
|
|
|
62
64
|
|
|
63
65
|
private
|
|
64
66
|
|
|
65
|
-
#
|
|
66
|
-
# @used
|
|
67
|
-
#
|
|
67
|
+
# Something has to be left out, so a trailer listing the skipped paths
|
|
68
|
+
# is needed. @used counts the whole assembled text, separators between
|
|
69
|
+
# files included: the result must not exceed the budget by a character.
|
|
68
70
|
def pack_within_limit
|
|
69
71
|
@parts = @non_text.map(&:render)
|
|
70
72
|
@used = joined_length(@parts)
|
|
@@ -95,12 +97,12 @@ module Aireview
|
|
|
95
97
|
shown_hunks
|
|
96
98
|
end
|
|
97
99
|
|
|
98
|
-
#
|
|
100
|
+
# Room for the next piece, the separator before it included.
|
|
99
101
|
def remaining
|
|
100
102
|
@limit - @used - (@parts.empty? ? 0 : 1)
|
|
101
103
|
end
|
|
102
104
|
|
|
103
|
-
#
|
|
105
|
+
# Returns [text, number of shown hunks, whether the layout stopped].
|
|
104
106
|
def pack_entry(entry)
|
|
105
107
|
full = entry.render
|
|
106
108
|
return [full, entry.hunks.size, false] if full.length <= remaining
|
|
@@ -113,10 +115,10 @@ module Aireview
|
|
|
113
115
|
[entry.header + body + partial_marker(entry, shown), shown, stopped]
|
|
114
116
|
end
|
|
115
117
|
|
|
116
|
-
#
|
|
117
|
-
#
|
|
118
|
-
#
|
|
119
|
-
#
|
|
118
|
+
# Room goes first to the hunks that can be shown, and only the remainder
|
|
119
|
+
# to the marks about oversized ones: otherwise the marks could push out
|
|
120
|
+
# the only fitting hunk. The skip is recorded in the coverage whether
|
|
121
|
+
# or not there is room for the mark.
|
|
120
122
|
def pack_hunks(entry)
|
|
121
123
|
base = entry.header.length + partial_marker(entry, 0).length
|
|
122
124
|
shown = []
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
require_relative 'utils'
|
|
3
3
|
require_relative 'errors'
|
|
4
|
+
require_relative 'stages'
|
|
4
5
|
require_relative 'secret_scrubber'
|
|
5
6
|
require_relative 'diff_fetcher'
|
|
6
7
|
require_relative 'context_budget'
|
|
@@ -15,14 +16,13 @@ module Aireview
|
|
|
15
16
|
}.freeze
|
|
16
17
|
CHANGES_HEADER = "Changes:\n"
|
|
17
18
|
CANDIDATES_HEADER = "\n\nCandidates JSON from Generate:\n"
|
|
18
|
-
#
|
|
19
|
-
#
|
|
20
|
-
#
|
|
19
|
+
# Room for the candidates in the Critique prompt: three candidates of
|
|
20
|
+
# ~1,500 characters. An estimate, not a guarantee; the actual size is
|
|
21
|
+
# checked before sending.
|
|
21
22
|
CANDIDATES_RESERVE_CHARS = 4_500
|
|
22
|
-
STAGES = %i[generate critique].freeze
|
|
23
23
|
|
|
24
|
-
#
|
|
25
|
-
#
|
|
24
|
+
# The context of one run: both stages get the same MR, Jira and diff,
|
|
25
|
+
# truncated once for the tightest of the stages.
|
|
26
26
|
Context = Struct.new(:user_prompt, :diff_text, :coverage, :sizes, keyword_init: true)
|
|
27
27
|
|
|
28
28
|
def initialize(config:, logger: Logger.new($stderr))
|
|
@@ -53,16 +53,16 @@ module Aireview
|
|
|
53
53
|
end
|
|
54
54
|
|
|
55
55
|
def build_generate_prompt(context)
|
|
56
|
-
check_stage_size!(
|
|
56
|
+
check_stage_size!('generate', system_prompt('generate'), context.user_prompt)
|
|
57
57
|
end
|
|
58
58
|
|
|
59
59
|
def build_critique_prompt(context, candidates_json:)
|
|
60
60
|
user = "#{context.user_prompt}#{CANDIDATES_HEADER}#{scrub_text(candidates_json)}"
|
|
61
|
-
check_stage_size!(
|
|
61
|
+
check_stage_size!('critique', system_prompt('critique'), user)
|
|
62
62
|
end
|
|
63
63
|
|
|
64
64
|
def system_prompt(stage)
|
|
65
|
-
template = stage.
|
|
65
|
+
template = stage.to_s == 'critique' ? CRITIQUE_PROMPT_TEMPLATE : GENERATE_PROMPT_TEMPLATE
|
|
66
66
|
extras = []
|
|
67
67
|
if Aireview::Utils.present?(@config.review_instructions)
|
|
68
68
|
extras << "Additional project instructions:\n#{scrub_text(@config.review_instructions.strip)}"
|
|
@@ -72,10 +72,11 @@ module Aireview
|
|
|
72
72
|
[template, *extras].join("\n\n")
|
|
73
73
|
end
|
|
74
74
|
|
|
75
|
-
#
|
|
76
|
-
#
|
|
77
|
-
#
|
|
75
|
+
# A check before sending: when the candidates exceed the reserve and the
|
|
76
|
+
# request does not fit, that is an error, not a reason to silently cut
|
|
77
|
+
# the context Generate has already seen.
|
|
78
78
|
def check_stage_size!(stage, system, user)
|
|
79
|
+
stage = stage.to_s
|
|
79
80
|
limit = @config.max_prompt_chars(stage)
|
|
80
81
|
total = system.length + user.length
|
|
81
82
|
if total > limit
|
|
@@ -89,10 +90,10 @@ module Aireview
|
|
|
89
90
|
|
|
90
91
|
private
|
|
91
92
|
|
|
92
|
-
#
|
|
93
|
-
#
|
|
93
|
+
# The minimum over the stages: the context is one per run, so it must fit
|
|
94
|
+
# into each of them together with its system prompt and reserve.
|
|
94
95
|
def context_budget(critique:)
|
|
95
|
-
stages = critique ? STAGES : [
|
|
96
|
+
stages = critique ? STAGES : ['generate']
|
|
96
97
|
budgets = stages.to_h { |stage| [stage, stage_budget(stage)] }
|
|
97
98
|
stage, budget = budgets.min_by { |_, value| value }
|
|
98
99
|
return budget if budget.positive?
|
|
@@ -104,7 +105,7 @@ module Aireview
|
|
|
104
105
|
end
|
|
105
106
|
|
|
106
107
|
def stage_budget(stage)
|
|
107
|
-
reserve = stage ==
|
|
108
|
+
reserve = stage == 'critique' ? CANDIDATES_RESERVE_CHARS + CANDIDATES_HEADER.length : 0
|
|
108
109
|
@config.max_prompt_chars(stage) - system_prompt(stage).length - reserve
|
|
109
110
|
end
|
|
110
111
|
|
|
@@ -152,7 +153,7 @@ module Aireview
|
|
|
152
153
|
end
|
|
153
154
|
|
|
154
155
|
def context_sizes(fixed:, packed:, budget:, diff_budget:, critique:)
|
|
155
|
-
stages = critique ? STAGES : [
|
|
156
|
+
stages = critique ? STAGES : ['generate']
|
|
156
157
|
{
|
|
157
158
|
context_budget: budget,
|
|
158
159
|
diff_budget: diff_budget,
|
|
@@ -7,13 +7,14 @@ module Aireview
|
|
|
7
7
|
DIFF_UNAVAILABLE = '[diff not available]'
|
|
8
8
|
BINARY_DIFF = /\ABinary files .* differ/
|
|
9
9
|
|
|
10
|
-
#
|
|
10
|
+
# One file from the GitLab answer: the header, the hunks and what can be done with it.
|
|
11
11
|
# kind:
|
|
12
|
-
# :text
|
|
13
|
-
# :no_text_changes
|
|
14
|
-
#
|
|
15
|
-
# :unavailable GitLab
|
|
16
|
-
#
|
|
12
|
+
# :text there are hunks, the code can be checked;
|
|
13
|
+
# :no_text_changes a rename, a mode change, an empty file: nothing to
|
|
14
|
+
# check;
|
|
15
|
+
# :unavailable GitLab did not return the diff (too_large, a binary,
|
|
16
|
+
# an empty diff without a reason): the code exists,
|
|
17
|
+
# but could not be checked.
|
|
17
18
|
class Entry
|
|
18
19
|
attr_reader :path, :kind, :header, :hunks
|
|
19
20
|
|
|
@@ -49,9 +50,9 @@ module Aireview
|
|
|
49
50
|
|
|
50
51
|
private
|
|
51
52
|
|
|
52
|
-
#
|
|
53
|
-
#
|
|
54
|
-
# GitLab
|
|
53
|
+
# An empty diff without an explainable reason (a rename, a mode change,
|
|
54
|
+
# an empty new or deleted file) counts as unavailable: the code exists,
|
|
55
|
+
# but GitLab did not return it.
|
|
55
56
|
def classify(change, diff)
|
|
56
57
|
return :unavailable if change['too_large']
|
|
57
58
|
return :unavailable if diff.match?(BINARY_DIFF)
|
|
@@ -67,8 +68,8 @@ module Aireview
|
|
|
67
68
|
modes.none?(&:nil?) && modes.uniq.size == 2
|
|
68
69
|
end
|
|
69
70
|
|
|
70
|
-
#
|
|
71
|
-
#
|
|
71
|
+
# Hunks are split at the @@ headers; the text before the first @@ (or a
|
|
72
|
+
# diff without any, such as the secret-file placeholder) counts as one hunk.
|
|
72
73
|
def split_hunks(diff)
|
|
73
74
|
diff = "#{diff}\n" unless diff.end_with?("\n")
|
|
74
75
|
pieces = diff.split(/^(?=@@ )/)
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
module Aireview
|
|
4
|
-
#
|
|
4
|
+
# The --dry-run output: settings, the context summary and the prompts of both stages.
|
|
5
5
|
class DryRunReport
|
|
6
6
|
def initialize(out)
|
|
7
7
|
@out = out
|
|
@@ -33,15 +33,37 @@ module Aireview
|
|
|
33
33
|
|
|
34
34
|
def render_settings(dry_run)
|
|
35
35
|
@out.puts('=== LLM SETTINGS ===')
|
|
36
|
-
|
|
37
|
-
|
|
36
|
+
render_config_paths(dry_run[:config_paths])
|
|
37
|
+
render_stage(dry_run, :generate)
|
|
38
38
|
if dry_run[:critique_prompt]
|
|
39
|
-
|
|
40
|
-
list('fallbacks', dry_run[:critique_fallbacks], separator: ' -> ')
|
|
39
|
+
render_stage(dry_run, :critique)
|
|
41
40
|
else
|
|
42
41
|
@out.puts('Critique: disabled')
|
|
43
42
|
end
|
|
43
|
+
@out.puts("Critique rule: #{dry_run[:critique_rule]}") if dry_run[:critique_rule]
|
|
44
44
|
render_reserves(dry_run)
|
|
45
|
+
list('warnings', dry_run[:warnings], separator: "\n ")
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
def render_config_paths(paths)
|
|
49
|
+
return if paths.nil? || paths.empty?
|
|
50
|
+
|
|
51
|
+
@out.puts("Config: #{paths.map { |name, path| "#{name} #{path}" }.join(', ')}")
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
# The source of every setting is the layer it came from: built-in, image
|
|
55
|
+
# defaults, .aireview.yml, env or cli.
|
|
56
|
+
def render_stage(dry_run, stage)
|
|
57
|
+
sources = dry_run.dig(:sources, stage) || {}
|
|
58
|
+
@out.puts("#{stage.capitalize}: #{dry_run[:"#{stage}_model"]} " \
|
|
59
|
+
"temperature=#{dry_run[:"#{stage}_temperature"]}#{origin(sources, :model, :provider)}")
|
|
60
|
+
fallbacks = dry_run[:"#{stage}_fallbacks"]
|
|
61
|
+
list('fallbacks', fallbacks, separator: ' -> ', suffix: origin(sources, :fallbacks))
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
def origin(sources, *keys)
|
|
65
|
+
parts = keys.filter_map { |key| "#{key} from #{sources[key]}" if sources[key] }
|
|
66
|
+
parts.empty? ? '' : " (#{parts.join(', ')})"
|
|
45
67
|
end
|
|
46
68
|
|
|
47
69
|
def render_context_sizes(sizes)
|
|
@@ -67,12 +89,15 @@ module Aireview
|
|
|
67
89
|
def render_reserves(dry_run)
|
|
68
90
|
keys = Array(dry_run[:api_keys]).map { |provider, count| "#{provider} #{count}" }
|
|
69
91
|
@out.puts("API keys: #{keys.join(', ')}") unless keys.empty?
|
|
70
|
-
|
|
92
|
+
return unless dry_run[:time_budget]
|
|
93
|
+
|
|
94
|
+
quarantine = dry_run[:overloaded_quarantine]
|
|
95
|
+
@out.puts("Time budget: #{dry_run[:time_budget]}s#{", overloaded quarantine: #{quarantine}s" if quarantine}")
|
|
71
96
|
end
|
|
72
97
|
|
|
73
|
-
def list(title, items, separator: ', ')
|
|
98
|
+
def list(title, items, separator: ', ', suffix: '')
|
|
74
99
|
items = Array(items)
|
|
75
|
-
@out.puts(" #{title}: #{items.join(separator)}") unless items.empty?
|
|
100
|
+
@out.puts(" #{title}: #{items.join(separator)}#{suffix}") unless items.empty?
|
|
76
101
|
end
|
|
77
102
|
end
|
|
78
103
|
end
|
data/lib/aireview/errors.rb
CHANGED
|
@@ -3,7 +3,15 @@ module Aireview
|
|
|
3
3
|
class Error < StandardError; end
|
|
4
4
|
class ConfigError < Error; end
|
|
5
5
|
class ParseError < Error; end
|
|
6
|
+
# Repairing invalid JSON is impossible: the model that answered is out of
|
|
7
|
+
# attempts or quarantined with no budget to wait. For the pipeline it is
|
|
8
|
+
# the same as an invalid result — the stage restarts on another model.
|
|
9
|
+
class RepairImpossibleError < ParseError; end
|
|
6
10
|
class ApiError < Error; end
|
|
11
|
+
# A pinned route (the JSON repair by the same model) could not answer: out
|
|
12
|
+
# of attempts, excluded, or quarantined with no budget to wait. Not fatal
|
|
13
|
+
# for the run — the stage restarts on another model.
|
|
14
|
+
class RouteExhaustedError < ApiError; end
|
|
7
15
|
class ContextBudgetError < Error; end
|
|
8
16
|
class HelpRequested < Error; end
|
|
9
17
|
end
|
|
@@ -35,8 +35,9 @@ module Aireview
|
|
|
35
35
|
get_json('user')
|
|
36
36
|
end
|
|
37
37
|
|
|
38
|
-
# Retry
|
|
39
|
-
#
|
|
38
|
+
# A Retry creates a new job in the same pipeline. Earlier attempts are
|
|
39
|
+
# visible only with include_retried; a new CI_JOB_ID alone happens after an
|
|
40
|
+
# ordinary push too.
|
|
40
41
|
def retried_job?(project_id, job_id)
|
|
41
42
|
current = get_json("projects/#{project_id}/jobs/#{job_id}")
|
|
42
43
|
validate_retry_job!(current, pipeline: true)
|
|
@@ -54,13 +55,13 @@ module Aireview
|
|
|
54
55
|
raise ApiError, 'Too many pipeline jobs to determine whether this job is a retry'
|
|
55
56
|
end
|
|
56
57
|
|
|
57
|
-
#
|
|
58
|
-
#
|
|
59
|
-
#
|
|
60
|
-
#
|
|
61
|
-
#
|
|
62
|
-
#
|
|
63
|
-
#
|
|
58
|
+
# Notes come in pages sorted by creation time, and updating a comment
|
|
59
|
+
# does not move it up in that order: on a long MR our review is far from
|
|
60
|
+
# the first page. Stopping the walk silently is not an option — with an
|
|
61
|
+
# incomplete list the review would decide the note is missing and create
|
|
62
|
+
# a second one, so the limit fails with an error. The order is explicit:
|
|
63
|
+
# it decides which old unmarked note the review picks up, and the API
|
|
64
|
+
# default is not to be relied on here.
|
|
64
65
|
def fetch_merge_request_notes(project_id, iid)
|
|
65
66
|
notes = []
|
|
66
67
|
page = 1
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
require 'json'
|
|
3
|
+
require 'timeout'
|
|
4
|
+
require_relative 'errors'
|
|
5
|
+
require_relative 'utils'
|
|
6
|
+
|
|
7
|
+
module Aireview
|
|
8
|
+
# One request to one model with one key through RubyLLM. Retries, keys and
|
|
9
|
+
# reserves belong to LlmRouter: the built-in RubyLLM/Faraday retries (3 by
|
|
10
|
+
# default) are off, otherwise every router attempt would turn into four
|
|
11
|
+
# HTTP requests and burn quota before the error reaches the classifier.
|
|
12
|
+
class LlmClient
|
|
13
|
+
# What a stage sends to the model; the same for every route of the stage.
|
|
14
|
+
Prompt = Struct.new(:stage, :system, :user, :temperature, :schema, keyword_init: true) do
|
|
15
|
+
def chars
|
|
16
|
+
system.length + user.length
|
|
17
|
+
end
|
|
18
|
+
end
|
|
19
|
+
|
|
20
|
+
def initialize(config:, logger: Logger.new($stderr))
|
|
21
|
+
@config = config
|
|
22
|
+
@logger = logger
|
|
23
|
+
@contexts = {}
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
# Returns the RubyLLM answer; read it with LlmClient.content. A request
|
|
27
|
+
# error is re-raised as is — LlmFailure classifies it.
|
|
28
|
+
def request(prompt, candidate:, key:, timeout:, key_index: 0)
|
|
29
|
+
load_ruby_llm
|
|
30
|
+
stage = prompt.stage.to_s
|
|
31
|
+
model = candidate.model
|
|
32
|
+
@logger.info("LLM #{stage} request started (model=#{model}, temperature=#{prompt.temperature})")
|
|
33
|
+
chat = build_chat(context: context(stage, candidate.provider, key, key_index), stage: stage,
|
|
34
|
+
model: model, provider: candidate.provider)
|
|
35
|
+
chat = configure_reasoning(chat: chat, model: model, provider: candidate.provider)
|
|
36
|
+
.with_temperature(prompt.temperature.to_f)
|
|
37
|
+
.with_schema(prompt.schema)
|
|
38
|
+
chat.with_instructions(prompt.system)
|
|
39
|
+
response = Timeout.timeout(timeout) { chat.ask(prompt.user) }
|
|
40
|
+
@logger.info("LLM #{stage} request completed (model=#{model}#{token_counts(response)})")
|
|
41
|
+
response
|
|
42
|
+
rescue Timeout::Error
|
|
43
|
+
@logger.warn("LLM #{stage} request timed out after #{timeout.round} seconds (model=#{model})")
|
|
44
|
+
raise
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
# The answer as RubyLLM 1.x gave it under a schema: the parsed JSON when
|
|
48
|
+
# the text is JSON, the text itself otherwise (the pipeline repairs it).
|
|
49
|
+
# RubyLLM 2 always returns the text, and a Hash that breaks the schema
|
|
50
|
+
# would go to a repair request instead of the next model. An empty
|
|
51
|
+
# answer stays an empty String (Message#parsed would turn it into nil);
|
|
52
|
+
# JSON null becomes nil, as in 1.x.
|
|
53
|
+
def self.content(response)
|
|
54
|
+
content = response.content
|
|
55
|
+
return content unless content.is_a?(String) && !content.empty?
|
|
56
|
+
|
|
57
|
+
response.parsed
|
|
58
|
+
rescue JSON::ParserError
|
|
59
|
+
content
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
private
|
|
63
|
+
|
|
64
|
+
# Token counts as the provider reported them, one line per attempt. They
|
|
65
|
+
# are not summed: Gemini already counts thinking into output. Input is
|
|
66
|
+
# what was not read from or written to a cache; the prompt size is
|
|
67
|
+
# input + cache_read + cache_write, and a retry of the same prompt on
|
|
68
|
+
# Gemini is often served from its implicit cache.
|
|
69
|
+
def token_counts(response)
|
|
70
|
+
tokens = response.tokens
|
|
71
|
+
cached = {cache_read: tokens.cache_read, cache_write: tokens.cache_write}.reject { |_, count| count.to_i.zero? }
|
|
72
|
+
counts = {input: tokens.input, output: tokens.output, thinking: tokens.thinking}.compact.merge(cached)
|
|
73
|
+
return '' if counts.empty?
|
|
74
|
+
|
|
75
|
+
", tokens: #{counts.map { |name, count| "#{name}=#{count}" }.join(' ')}"
|
|
76
|
+
end
|
|
77
|
+
|
|
78
|
+
def load_ruby_llm
|
|
79
|
+
require 'ruby_llm'
|
|
80
|
+
rescue LoadError => e
|
|
81
|
+
@logger.error("LLM setup failed: #{e.message}")
|
|
82
|
+
raise ConfigError, "Missing dependency: #{e.message}"
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
def configure_reasoning(chat:, model:, provider:)
|
|
86
|
+
return chat unless provider == 'ollama' && model.start_with?('gpt-oss:')
|
|
87
|
+
|
|
88
|
+
chat.with_thinking(effort: :low)
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
def build_chat(context:, stage:, model:, provider:)
|
|
92
|
+
context.chat(model: model, provider: provider.to_sym)
|
|
93
|
+
rescue RubyLLM::ModelNotFoundError
|
|
94
|
+
@logger.warn(
|
|
95
|
+
"LLM #{stage}: model not found in RubyLLM registry; " \
|
|
96
|
+
"using fallback with incomplete model metadata " \
|
|
97
|
+
"(model=#{model}, provider=#{provider})"
|
|
98
|
+
)
|
|
99
|
+
context.chat(model: model, provider: provider.to_sym, assume_model_exists: true)
|
|
100
|
+
end
|
|
101
|
+
|
|
102
|
+
# A RubyLLM context per stage, provider and key index: switching the key
|
|
103
|
+
# is another context, not an edit of the global config.
|
|
104
|
+
def context(stage, provider, key, key_index)
|
|
105
|
+
@contexts[[stage, provider, key_index]] ||= build_context(provider.to_s, key)
|
|
106
|
+
end
|
|
107
|
+
|
|
108
|
+
def build_context(provider, api_key)
|
|
109
|
+
RubyLLM.context do |ruby_config|
|
|
110
|
+
configure_http_proxy(ruby_config)
|
|
111
|
+
ruby_config.request_timeout = @config.llm_timeout.to_f
|
|
112
|
+
ruby_config.max_retries = 0
|
|
113
|
+
configure_provider(ruby_config, provider, api_key)
|
|
114
|
+
end
|
|
115
|
+
end
|
|
116
|
+
|
|
117
|
+
def configure_http_proxy(ruby_config)
|
|
118
|
+
return unless Aireview::Utils.present?(@config.llm_http_proxy)
|
|
119
|
+
|
|
120
|
+
ruby_config.http_proxy = @config.llm_http_proxy
|
|
121
|
+
end
|
|
122
|
+
|
|
123
|
+
def configure_provider(ruby_config, provider, api_key)
|
|
124
|
+
case provider
|
|
125
|
+
when 'gemini', 'openai', 'openrouter'
|
|
126
|
+
configure_remote_provider(ruby_config, provider, api_key)
|
|
127
|
+
when 'anthropic'
|
|
128
|
+
ruby_config.anthropic_api_key = api_key
|
|
129
|
+
when 'ollama'
|
|
130
|
+
ruby_config.ollama_api_base = @config.ollama_api_base
|
|
131
|
+
else
|
|
132
|
+
raise ConfigError, "Unsupported LLM provider: #{provider.inspect}"
|
|
133
|
+
end
|
|
134
|
+
end
|
|
135
|
+
|
|
136
|
+
# RubyLLM 2 sends OpenAI requests to the Responses API; a compatible
|
|
137
|
+
# server behind LLM_API_BASE usually has Chat Completions only.
|
|
138
|
+
def configure_remote_provider(ruby_config, provider, api_key)
|
|
139
|
+
ruby_config.public_send("#{provider}_api_key=", api_key)
|
|
140
|
+
ruby_config.openai_protocol = :chat_completions if provider == 'openai'
|
|
141
|
+
return unless Aireview::Utils.present?(@config.llm_api_base)
|
|
142
|
+
|
|
143
|
+
ruby_config.public_send("#{provider}_api_base=", @config.llm_api_base)
|
|
144
|
+
end
|
|
145
|
+
end
|
|
146
|
+
end
|
data/lib/aireview/llm_failure.rb
CHANGED
|
@@ -3,17 +3,27 @@ require 'json'
|
|
|
3
3
|
require 'timeout'
|
|
4
4
|
|
|
5
5
|
module Aireview
|
|
6
|
-
#
|
|
7
|
-
#
|
|
6
|
+
# Classifies an LLM request error. Answers only "what is it"; the decision
|
|
7
|
+
# to retry, switch the key or the model belongs to LlmRouter.
|
|
8
8
|
#
|
|
9
|
-
# :daily_quota —
|
|
10
|
-
# :rate_limit —
|
|
11
|
-
# :overloaded — 503
|
|
12
|
-
# :timeout —
|
|
13
|
-
# :
|
|
14
|
-
# :
|
|
9
|
+
# :daily_quota — the project's daily quota for the model, retries are useless;
|
|
10
|
+
# :rate_limit — a per-minute limit, passes after the hinted time;
|
|
11
|
+
# :overloaded — 503 / "high demand" on the model;
|
|
12
|
+
# :timeout — no answer for longer than LLM_TIMEOUT;
|
|
13
|
+
# :unavailable — the provider has no such model: retired, a typo, not pulled into Ollama;
|
|
14
|
+
# :fatal — an API error that reserves do not cure;
|
|
15
|
+
# :unhandled — not a provider error, re-raised as is.
|
|
15
16
|
module LlmFailure
|
|
16
|
-
KINDS = %i[daily_quota rate_limit overloaded timeout fatal unhandled].freeze
|
|
17
|
+
KINDS = %i[daily_quota rate_limit overloaded timeout unavailable fatal unhandled].freeze
|
|
18
|
+
# Only by the provider's text about the model: a bare 404 is also what a
|
|
19
|
+
# wrong LLM_API_BASE or proxy answers, and the next model will not help.
|
|
20
|
+
# Ollama: `model 'x' not found` (through /v1) and `model "x" not found,
|
|
21
|
+
# try pulling it first` (older versions and /api).
|
|
22
|
+
UNAVAILABLE_MODEL_TEXT = Regexp.union(
|
|
23
|
+
/\bmodels?\/[\w.:-]+ is not found\b/i,
|
|
24
|
+
/\bis not supported for generateContent\b/i,
|
|
25
|
+
/\bmodel ['"][^'"]+['"] not found\b/i
|
|
26
|
+
)
|
|
17
27
|
QUOTA_FAILURE_TYPE = 'type.googleapis.com/google.rpc.QuotaFailure'
|
|
18
28
|
DAILY_QUOTA_ID = /PerDay/i
|
|
19
29
|
DAILY_QUOTA_TEXT = /\bper\s+day\b|\bdaily\b/i
|
|
@@ -21,18 +31,22 @@ module Aireview
|
|
|
21
31
|
|
|
22
32
|
module_function
|
|
23
33
|
|
|
24
|
-
#
|
|
25
|
-
# 429
|
|
26
|
-
#
|
|
34
|
+
# Quota details are checked before the exception class: RubyLLM turns a
|
|
35
|
+
# 429 mentioning input_token into ContextLengthExceededError, although
|
|
36
|
+
# it is an exhausted token quota, not an oversized request.
|
|
27
37
|
def classify(error)
|
|
28
38
|
return :timeout if transport_timeout?(error)
|
|
29
39
|
return :unhandled unless ruby_llm_error?(error)
|
|
30
40
|
|
|
31
|
-
quota_kind(error) || (overloaded?(error) ? :overloaded : :fatal)
|
|
41
|
+
quota_kind(error) || model_kind(error) || (overloaded?(error) ? :overloaded : :fatal)
|
|
32
42
|
end
|
|
33
43
|
|
|
34
|
-
|
|
35
|
-
|
|
44
|
+
def model_kind(error)
|
|
45
|
+
:unavailable if error.message.to_s.match?(UNAVAILABLE_MODEL_TEXT)
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
# The outer Timeout.timeout and Faraday's transport timeouts: the latter
|
|
49
|
+
# inherit neither Timeout::Error nor RubyLLM::Error.
|
|
36
50
|
def transport_timeout?(error)
|
|
37
51
|
return true if error.is_a?(Timeout::Error) || error.is_a?(Errno::ETIMEDOUT)
|
|
38
52
|
|
|
@@ -47,10 +61,10 @@ module Aireview
|
|
|
47
61
|
defined?(RubyLLM::Error) && error.is_a?(RubyLLM::Error)
|
|
48
62
|
end
|
|
49
63
|
|
|
50
|
-
# Google
|
|
51
|
-
# (GenerateRequestsPerDay… / …PerMinute…).
|
|
52
|
-
#
|
|
53
|
-
#
|
|
64
|
+
# Google puts the kind of quota into QuotaFailure.violations[].quotaId
|
|
65
|
+
# (GenerateRequestsPerDay… / …PerMinute…). The free_tier_requests metric
|
|
66
|
+
# is the same for both, it cannot tell them apart. The message text is
|
|
67
|
+
# the fallback signal when there is no response body.
|
|
54
68
|
def quota_kind(error)
|
|
55
69
|
ids = quota_ids(error)
|
|
56
70
|
return daily_quota_id?(ids) ? :daily_quota : :rate_limit unless ids.empty?
|
|
@@ -86,5 +100,8 @@ module Aireview
|
|
|
86
100
|
match = message.to_s.match(RETRY_AFTER)
|
|
87
101
|
match[1].to_f if match
|
|
88
102
|
end
|
|
103
|
+
|
|
104
|
+
private_class_method :model_kind, :transport_timeout?, :overloaded?, :ruby_llm_error?, :quota_kind,
|
|
105
|
+
:daily_quota_id?, :quota_ids, :response_body
|
|
89
106
|
end
|
|
90
107
|
end
|