aireview 2.1.0 → 2.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +47 -0
- data/README.md +182 -5
- data/config/.aireview.yml.example +6 -0
- data/lib/aireview/cli.rb +11 -2
- data/lib/aireview/config.rb +26 -8
- data/lib/aireview/config_fallbacks.rb +43 -15
- data/lib/aireview/config_jev.rb +137 -0
- data/lib/aireview/config_layers.rb +1 -1
- data/lib/aireview/config_loader.rb +32 -6
- data/lib/aireview/context_builder.rb +13 -10
- data/lib/aireview/dry_run_prompts.rb +104 -0
- data/lib/aireview/dry_run_report.rb +40 -8
- data/lib/aireview/errors.rb +11 -0
- data/lib/aireview/jev_client.rb +129 -0
- data/lib/aireview/jev_critic.rb +387 -0
- data/lib/aireview/jev_shadow.rb +82 -0
- data/lib/aireview/jev_stage.rb +97 -0
- data/lib/aireview/llm_client.rb +52 -21
- data/lib/aireview/llm_router.rb +4 -4
- data/lib/aireview/model_candidate.rb +34 -5
- data/lib/aireview/model_checker.rb +46 -5
- data/lib/aireview/model_pool.rb +57 -28
- data/lib/aireview/prompts/jev_questions.yml +79 -0
- data/lib/aireview/review_marker.rb +7 -2
- data/lib/aireview/review_pipeline.rb +40 -55
- data/lib/aireview/review_renderer.rb +18 -2
- data/lib/aireview/stage_chains.rb +6 -3
- data/lib/aireview/version.rb +1 -1
- metadata +12 -4
data/lib/aireview/llm_client.rb
CHANGED
|
@@ -17,6 +17,16 @@ module Aireview
|
|
|
17
17
|
end
|
|
18
18
|
end
|
|
19
19
|
|
|
20
|
+
# RubyLLM 2 sends the temperature as given, while 1.x replaced it with
|
|
21
|
+
# 1.0 for OpenAI reasoning models (o1, o3, gpt-5…) and dropped it for the
|
|
22
|
+
# search ones (gpt-4o-search-preview…), which take no other: such a
|
|
23
|
+
# request is a BadRequest, fatal for the router, and the stage would not
|
|
24
|
+
# even reach a fallback model. The RubyLLM registry knows which models
|
|
25
|
+
# take a temperature; where it does not (a server of your own,
|
|
26
|
+
# assume_model_exists, search models with an empty flag) the name
|
|
27
|
+
# decides, as in 1.x.
|
|
28
|
+
NO_TEMPERATURE_MODELS = %r{\A(?:openai/)?(?:o\d|gpt-5)|-search}
|
|
29
|
+
|
|
20
30
|
def initialize(config:, logger: Logger.new($stderr))
|
|
21
31
|
@config = config
|
|
22
32
|
@logger = logger
|
|
@@ -26,16 +36,10 @@ module Aireview
|
|
|
26
36
|
# Returns the RubyLLM answer; read it with LlmClient.content. A request
|
|
27
37
|
# error is re-raised as is — LlmFailure classifies it.
|
|
28
38
|
def request(prompt, candidate:, key:, timeout:, key_index: 0)
|
|
29
|
-
load_ruby_llm
|
|
30
39
|
stage = prompt.stage.to_s
|
|
31
40
|
model = candidate.model
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
model: model, provider: candidate.provider)
|
|
35
|
-
chat = configure_reasoning(chat: chat, model: model, provider: candidate.provider)
|
|
36
|
-
.with_temperature(prompt.temperature.to_f)
|
|
37
|
-
.with_schema(prompt.schema)
|
|
38
|
-
chat.with_instructions(prompt.system)
|
|
41
|
+
chat = prepare(prompt, candidate: candidate, key: key, key_index: key_index)
|
|
42
|
+
@logger.info("LLM #{stage} request started (model=#{model}, temperature=#{chat.temperature || 'model default'})")
|
|
39
43
|
response = Timeout.timeout(timeout) { chat.ask(prompt.user) }
|
|
40
44
|
@logger.info("LLM #{stage} request completed (model=#{model}#{token_counts(response)})")
|
|
41
45
|
response
|
|
@@ -44,6 +48,21 @@ module Aireview
|
|
|
44
48
|
raise
|
|
45
49
|
end
|
|
46
50
|
|
|
51
|
+
# The request as it will go, without sending it: the chat with the model,
|
|
52
|
+
# key, schema, instructions and temperature set. chat.render builds it —
|
|
53
|
+
# that is how the specs check the real request without the network.
|
|
54
|
+
def prepare(prompt, candidate:, key:, key_index: 0)
|
|
55
|
+
load_ruby_llm
|
|
56
|
+
stage = prompt.stage.to_s
|
|
57
|
+
chat = build_chat(context: context(stage, candidate, key, key_index), stage: stage,
|
|
58
|
+
model: candidate.model, provider: candidate.provider)
|
|
59
|
+
chat = configure_reasoning(chat: chat, model: candidate.model, provider: candidate.provider)
|
|
60
|
+
.with_temperature(temperature_for(chat, prompt, candidate))
|
|
61
|
+
.with_schema(prompt.schema)
|
|
62
|
+
chat.with_instructions(prompt.system)
|
|
63
|
+
chat
|
|
64
|
+
end
|
|
65
|
+
|
|
47
66
|
# The answer as RubyLLM 1.x gave it under a schema: the parsed JSON when
|
|
48
67
|
# the text is JSON, the text itself otherwise (the pipeline repairs it).
|
|
49
68
|
# RubyLLM 2 always returns the text, and a Hash that breaks the schema
|
|
@@ -82,6 +101,14 @@ module Aireview
|
|
|
82
101
|
raise ConfigError, "Missing dependency: #{e.message}"
|
|
83
102
|
end
|
|
84
103
|
|
|
104
|
+
# The temperature for a model that takes one (see NO_TEMPERATURE_MODELS);
|
|
105
|
+
# nil — its own default, left out of the request.
|
|
106
|
+
def temperature_for(chat, prompt, candidate)
|
|
107
|
+
accepts = chat.model.metadata[:temperature] if chat.model.respond_to?(:metadata)
|
|
108
|
+
accepts = !candidate.model.to_s.match?(NO_TEMPERATURE_MODELS) if accepts.nil?
|
|
109
|
+
accepts ? prompt.temperature.to_f : nil
|
|
110
|
+
end
|
|
111
|
+
|
|
85
112
|
def configure_reasoning(chat:, model:, provider:)
|
|
86
113
|
return chat unless provider == 'ollama' && model.start_with?('gpt-oss:')
|
|
87
114
|
|
|
@@ -99,18 +126,19 @@ module Aireview
|
|
|
99
126
|
context.chat(model: model, provider: provider.to_sym, assume_model_exists: true)
|
|
100
127
|
end
|
|
101
128
|
|
|
102
|
-
# A RubyLLM context per stage,
|
|
103
|
-
#
|
|
104
|
-
|
|
105
|
-
|
|
129
|
+
# A RubyLLM context per stage, key source (a provider's API or a server
|
|
130
|
+
# of your own) and key index: switching the key is another context, not
|
|
131
|
+
# an edit of the global config.
|
|
132
|
+
def context(stage, candidate, key, key_index)
|
|
133
|
+
@contexts[[stage, candidate.key_source, key_index]] ||= build_context(candidate, key)
|
|
106
134
|
end
|
|
107
135
|
|
|
108
|
-
def build_context(
|
|
136
|
+
def build_context(candidate, api_key)
|
|
109
137
|
RubyLLM.context do |ruby_config|
|
|
110
138
|
configure_http_proxy(ruby_config)
|
|
111
139
|
ruby_config.request_timeout = @config.llm_timeout.to_f
|
|
112
140
|
ruby_config.max_retries = 0
|
|
113
|
-
configure_provider(ruby_config, provider, api_key)
|
|
141
|
+
configure_provider(ruby_config, candidate.provider.to_s, api_key, candidate.api_base)
|
|
114
142
|
end
|
|
115
143
|
end
|
|
116
144
|
|
|
@@ -120,27 +148,30 @@ module Aireview
|
|
|
120
148
|
ruby_config.http_proxy = @config.llm_http_proxy
|
|
121
149
|
end
|
|
122
150
|
|
|
123
|
-
|
|
151
|
+
# api_base — the model's own server; without it the provider's API,
|
|
152
|
+
# or LLM_API_BASE for the providers that always took it.
|
|
153
|
+
def configure_provider(ruby_config, provider, api_key, api_base)
|
|
124
154
|
case provider
|
|
125
155
|
when 'gemini', 'openai', 'openrouter'
|
|
126
|
-
configure_remote_provider(ruby_config, provider, api_key)
|
|
156
|
+
configure_remote_provider(ruby_config, provider, api_key, api_base || @config.llm_api_base)
|
|
127
157
|
when 'anthropic'
|
|
128
158
|
ruby_config.anthropic_api_key = api_key
|
|
159
|
+
ruby_config.anthropic_api_base = api_base if api_base
|
|
129
160
|
when 'ollama'
|
|
130
|
-
ruby_config.ollama_api_base = @config.ollama_api_base
|
|
161
|
+
ruby_config.ollama_api_base = api_base || @config.ollama_api_base
|
|
131
162
|
else
|
|
132
163
|
raise ConfigError, "Unsupported LLM provider: #{provider.inspect}"
|
|
133
164
|
end
|
|
134
165
|
end
|
|
135
166
|
|
|
136
167
|
# RubyLLM 2 sends OpenAI requests to the Responses API; a compatible
|
|
137
|
-
# server
|
|
138
|
-
def configure_remote_provider(ruby_config, provider, api_key)
|
|
168
|
+
# server usually has Chat Completions only.
|
|
169
|
+
def configure_remote_provider(ruby_config, provider, api_key, api_base)
|
|
139
170
|
ruby_config.public_send("#{provider}_api_key=", api_key)
|
|
140
171
|
ruby_config.openai_protocol = :chat_completions if provider == 'openai'
|
|
141
|
-
return unless Aireview::Utils.present?(
|
|
172
|
+
return unless Aireview::Utils.present?(api_base)
|
|
142
173
|
|
|
143
|
-
ruby_config.public_send("#{provider}_api_base=",
|
|
174
|
+
ruby_config.public_send("#{provider}_api_base=", api_base)
|
|
144
175
|
end
|
|
145
176
|
end
|
|
146
177
|
end
|
data/lib/aireview/llm_router.rb
CHANGED
|
@@ -34,7 +34,7 @@ module Aireview
|
|
|
34
34
|
|
|
35
35
|
Slot = Struct.new(:candidate, :index)
|
|
36
36
|
# The key the next model of the same provider continues from.
|
|
37
|
-
Carry = Struct.new(:
|
|
37
|
+
Carry = Struct.new(:key_source, :key_index)
|
|
38
38
|
Delay = Struct.new(:seconds, :source)
|
|
39
39
|
|
|
40
40
|
SWITCH_HINT = 'Try again later or switch model via --generate-model/--critique-model.'
|
|
@@ -213,7 +213,7 @@ module Aireview
|
|
|
213
213
|
log_switch(stage, route, visits)
|
|
214
214
|
status, response = try_route(stage, route, visits, &request)
|
|
215
215
|
return [:ok, remember(stage, route, response)] if status == :ok
|
|
216
|
-
return [:next_model, Carry.new(route.candidate.
|
|
216
|
+
return [:next_model, Carry.new(route.candidate.key_source, route.key_index)] if status == :next_model
|
|
217
217
|
end
|
|
218
218
|
state(slot.candidate).exclude('daily quota exhausted on every key') unless tried
|
|
219
219
|
[:next_model, carry]
|
|
@@ -224,7 +224,7 @@ module Aireview
|
|
|
224
224
|
# is bound to "key + model", a failure on one model does not write the
|
|
225
225
|
# key off for another.
|
|
226
226
|
def candidate_routes(stage, slot, carry)
|
|
227
|
-
keys = @config.
|
|
227
|
+
keys = @config.candidate_api_keys(slot.candidate)
|
|
228
228
|
routes = keys.each_with_index.map do |key, key_index|
|
|
229
229
|
Route.new(candidate: slot.candidate, candidate_index: slot.index, key: key,
|
|
230
230
|
key_index: key_index, key_count: keys.size)
|
|
@@ -233,7 +233,7 @@ module Aireview
|
|
|
233
233
|
end
|
|
234
234
|
|
|
235
235
|
def start_key(stage, slot, carry)
|
|
236
|
-
return carry.key_index if carry && carry.
|
|
236
|
+
return carry.key_index if carry && carry.key_source == slot.candidate.key_source
|
|
237
237
|
|
|
238
238
|
cursor_candidate, cursor_key = @cursor.fetch(stage, [0, 0])
|
|
239
239
|
cursor_candidate == slot.index ? cursor_key : 0
|
|
@@ -1,19 +1,48 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
|
+
require_relative 'errors'
|
|
2
3
|
|
|
3
4
|
module Aireview
|
|
4
|
-
|
|
5
|
+
# Providers a "provider/name" string may start with. An OpenRouter name
|
|
6
|
+
# carries a slash of its own, so in a string it always goes with the
|
|
7
|
+
# prefix: openrouter/anthropic/claude-sonnet-4.5.
|
|
8
|
+
KNOWN_PROVIDERS = %w[gemini ollama openai anthropic openrouter].freeze
|
|
5
9
|
|
|
6
|
-
# A model in a stage chain: provider, name
|
|
7
|
-
#
|
|
8
|
-
|
|
10
|
+
# A model in a stage chain: provider, name, request size limit and, for a
|
|
11
|
+
# server of your own, its address. Compared by provider, name and address
|
|
12
|
+
# — the limit depends on the stage.
|
|
13
|
+
ModelCandidate = Struct.new(:provider, :model, :max_prompt_chars, :api_base, keyword_init: true) do
|
|
14
|
+
# The address is part of the name, scheme included: two servers with the
|
|
15
|
+
# same model (http and https ones too) are two models for quarantine,
|
|
16
|
+
# reserves and the review key.
|
|
9
17
|
def to_s
|
|
10
|
-
"#{provider}/#{model}"
|
|
18
|
+
name = "#{provider}/#{model}"
|
|
19
|
+
api_base ? "#{name}@#{api_base.chomp('/')}" : name
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
# Every way a start or a CLI override may name this model: bare,
|
|
23
|
+
# "provider/name", or the full name with the address. One list for the
|
|
24
|
+
# config check and for picking the model, so they cannot disagree.
|
|
25
|
+
def names
|
|
26
|
+
[model.to_s, "#{provider}/#{model}", to_s].uniq
|
|
11
27
|
end
|
|
12
28
|
|
|
13
29
|
def same_model?(other)
|
|
14
30
|
to_s == other.to_s
|
|
15
31
|
end
|
|
16
32
|
|
|
33
|
+
# Models sharing keys: a provider's own API, or one server of your own.
|
|
34
|
+
def key_source
|
|
35
|
+
api_base ? "#{provider}@#{api_base}" : provider.to_s
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
# api_base of a model or a stage: an http(s) address or nothing.
|
|
39
|
+
def self.api_base(value, name)
|
|
40
|
+
return nil if value.nil? || value.to_s.strip.empty?
|
|
41
|
+
return value.to_s.strip if value.to_s.strip.match?(%r{\Ahttps?://\S+\z})
|
|
42
|
+
|
|
43
|
+
raise ConfigError, "#{name} must be an http(s) URL, got #{value.inspect}"
|
|
44
|
+
end
|
|
45
|
+
|
|
17
46
|
# A "provider/name" string or a bare name (the provider is separated by
|
|
18
47
|
# a slash because Ollama tags contain a colon), or a hash with model.
|
|
19
48
|
def self.parse_item(item)
|
|
@@ -8,6 +8,7 @@ require_relative 'review_pipeline'
|
|
|
8
8
|
require_relative 'result_parser'
|
|
9
9
|
require_relative 'llm_client'
|
|
10
10
|
require_relative 'output_schemas'
|
|
11
|
+
require_relative 'jev_client'
|
|
11
12
|
|
|
12
13
|
module Aireview
|
|
13
14
|
# `aireview models check`: every model of both stage chains gets one
|
|
@@ -16,6 +17,10 @@ module Aireview
|
|
|
16
17
|
# validation as in a run: the provider's catalog is not consulted, "the
|
|
17
18
|
# model is listed" does not mean "our request with the schema passes on
|
|
18
19
|
# it". No reserves, quarantine or walking — this is a check, not a review.
|
|
20
|
+
# Only the stages that need an LLM are checked (Jev with fallback: fail
|
|
21
|
+
# needs no LLM Critique); Jev gets a probe of its own when it is the
|
|
22
|
+
# critique engine, and in shadow mode too, but then its result is shown
|
|
23
|
+
# without counting: the shadow never blocks a release.
|
|
19
24
|
class ModelChecker
|
|
20
25
|
PROBE_MERGE_REQUEST = {
|
|
21
26
|
'title' => 'Fix order total',
|
|
@@ -32,6 +37,10 @@ module Aireview
|
|
|
32
37
|
}
|
|
33
38
|
].freeze
|
|
34
39
|
PROBE_CANDIDATE_IDS = ['C1'].freeze
|
|
40
|
+
JEV_PROBE_STATE = 'The order total must include the quantity.'
|
|
41
|
+
JEV_PROBE_QUESTIONS = {
|
|
42
|
+
'mentions_quantity' => {'type' => 'noul', 'instructions' => 'Does the text mention a quantity?'}
|
|
43
|
+
}.freeze
|
|
35
44
|
# ok and skipped do not block a release, everything else does. In strict
|
|
36
45
|
# mode (a runner where Ollama must be running) skipped fails too.
|
|
37
46
|
PASSING = %i[ok skipped].freeze
|
|
@@ -52,6 +61,7 @@ module Aireview
|
|
|
52
61
|
@strict = strict
|
|
53
62
|
client = dependencies[:client]
|
|
54
63
|
sleeper = dependencies[:sleeper]
|
|
64
|
+
@jev_client = dependencies[:jev_client]
|
|
55
65
|
@logger = logger
|
|
56
66
|
@client = client || LlmClient.new(config: config, logger: logger)
|
|
57
67
|
@sleeper = sleeper || ->(seconds) { sleep(seconds) }
|
|
@@ -63,13 +73,14 @@ module Aireview
|
|
|
63
73
|
# 1 — at least one did not.
|
|
64
74
|
def run
|
|
65
75
|
@config.require_llm_configuration!
|
|
66
|
-
|
|
76
|
+
stages = @config.llm_stages
|
|
77
|
+
candidates = stages.flat_map { |stage| @config.stage_chain(stage) }.uniq(&:to_s)
|
|
67
78
|
prompts = probe_prompts
|
|
68
|
-
@out.puts("Checking #{candidates.size} model(s) with the
|
|
79
|
+
@out.puts("Checking #{candidates.size} model(s) with the #{stages.join(' and ')} schemas")
|
|
69
80
|
results = candidates.flat_map do |candidate|
|
|
70
|
-
|
|
81
|
+
stages.map { |stage| check(candidate, stage, prompts).tap { |result| @out.puts(result) } }
|
|
71
82
|
end
|
|
72
|
-
summary(results)
|
|
83
|
+
summary(results + jev_results)
|
|
73
84
|
end
|
|
74
85
|
|
|
75
86
|
private
|
|
@@ -110,10 +121,40 @@ module Aireview
|
|
|
110
121
|
stage: stage, system: prompt[:system_prompt], user: prompt[:user_prompt], temperature: 0,
|
|
111
122
|
schema: stage == 'critique' ? CritiqueOutputSchema : GenerateOutputSchema
|
|
112
123
|
)
|
|
113
|
-
key = @config.
|
|
124
|
+
key = @config.candidate_api_keys(candidate).first
|
|
114
125
|
LlmClient.content(@client.request(request, candidate: candidate, key: key, timeout: @config.llm_timeout.to_f))
|
|
115
126
|
end
|
|
116
127
|
|
|
128
|
+
# The Jev probe counts only when Jev is the critique engine.
|
|
129
|
+
def jev_results
|
|
130
|
+
counted = @config.jev_critique?
|
|
131
|
+
return [] unless counted || @config.jev_shadow?
|
|
132
|
+
|
|
133
|
+
result = check_jev
|
|
134
|
+
result.detail = [result.detail, 'shadow only, not counted'].compact.join('; ') unless counted
|
|
135
|
+
@out.puts(result)
|
|
136
|
+
counted ? [result] : []
|
|
137
|
+
end
|
|
138
|
+
|
|
139
|
+
def check_jev
|
|
140
|
+
name = "jev/#{@config.jev_model}"
|
|
141
|
+
if Aireview::Utils.blank?(@config.jev_api_key)
|
|
142
|
+
return Result.new(candidate: name, stage: 'jev', status: :skipped, detail: 'JEV_API_KEY is not set')
|
|
143
|
+
end
|
|
144
|
+
|
|
145
|
+
started_at = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
146
|
+
jev_client.evaluate(state: JEV_PROBE_STATE, questions: JEV_PROBE_QUESTIONS)
|
|
147
|
+
Result.new(candidate: name, stage: 'jev', status: :ok,
|
|
148
|
+
seconds: Process.clock_gettime(Process::CLOCK_MONOTONIC) - started_at)
|
|
149
|
+
rescue JevError => e
|
|
150
|
+
status = e.status.nil? || [429, 529].include?(e.status) ? :unverified : :failed
|
|
151
|
+
Result.new(candidate: name, stage: 'jev', status: status, detail: e.message)
|
|
152
|
+
end
|
|
153
|
+
|
|
154
|
+
def jev_client
|
|
155
|
+
@jev_client ||= JevClient.new(config: @config, logger: @logger)
|
|
156
|
+
end
|
|
157
|
+
|
|
117
158
|
# missing — the provider has no such model; unverified — the provider
|
|
118
159
|
# could not answer right now (overload, quota, timeout); skipped — Ollama
|
|
119
160
|
# is not running where the check runs; failed — everything else.
|
data/lib/aireview/model_pool.rb
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
require_relative 'errors'
|
|
3
|
+
require_relative 'stages'
|
|
3
4
|
require_relative 'utils'
|
|
4
5
|
require_relative 'model_candidate'
|
|
5
6
|
require_relative 'stage_chains'
|
|
@@ -26,11 +27,24 @@ module Aireview
|
|
|
26
27
|
return false if Aireview::Utils.blank?(model)
|
|
27
28
|
|
|
28
29
|
Array(items).each_with_index.any? do |item, index|
|
|
29
|
-
|
|
30
|
-
parsed[:model] == model.to_s || "#{parsed[:provider]}/#{parsed[:model]}" == model.to_s
|
|
30
|
+
names(parse_item(item, index, provider)).include?(model.to_s)
|
|
31
31
|
end
|
|
32
32
|
end
|
|
33
33
|
|
|
34
|
+
def self.item_candidate(item)
|
|
35
|
+
ModelCandidate.new(provider: item[:provider], model: item[:model], api_base: item[:api_base])
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
# The full name of a pool item, with the address of a server of your own.
|
|
39
|
+
def self.item_name(item)
|
|
40
|
+
item_candidate(item).to_s
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
# How a model may be named in a start or a CLI override (ModelCandidate#names).
|
|
44
|
+
def self.names(item)
|
|
45
|
+
item_candidate(item).names
|
|
46
|
+
end
|
|
47
|
+
|
|
34
48
|
def self.parse_item(item, index, provider)
|
|
35
49
|
item = ModelCandidate.parse_item(item)
|
|
36
50
|
name = "llm.models[#{index}]"
|
|
@@ -41,7 +55,8 @@ module Aireview
|
|
|
41
55
|
{
|
|
42
56
|
provider: (item['provider'] || provider).to_s,
|
|
43
57
|
model: item['model'].to_s,
|
|
44
|
-
max_prompt_chars: limit.nil? ? nil : StageChains.positive_limit(limit, name)
|
|
58
|
+
max_prompt_chars: limit.nil? ? nil : StageChains.positive_limit(limit, name),
|
|
59
|
+
api_base: ModelCandidate.api_base(item['api_base'], "#{name}.api_base")
|
|
45
60
|
}
|
|
46
61
|
end
|
|
47
62
|
|
|
@@ -50,30 +65,38 @@ module Aireview
|
|
|
50
65
|
# the pool (image defaults versus the project's LLM_MODELS): such a start
|
|
51
66
|
# missing from the pool is replaced by the first model with a warning,
|
|
52
67
|
# an explicit start outside the pool is a configuration error;
|
|
53
|
-
# own_chains — stages with a chain of their own
|
|
54
|
-
#
|
|
68
|
+
# own_chains — stages with a chain of their own; stages — the stages
|
|
69
|
+
# that go to an LLM in this run: without Critique (Jev decides with no
|
|
70
|
+
# fallback, --no-critique) the plan has no critique chain and no critique
|
|
71
|
+
# policy to validate or to sign.
|
|
72
|
+
# Ten named settings read better than a struct for its own sake.
|
|
55
73
|
def initialize(items:, provider:, limits:, starts: {}, inherited_starts: [], rank: nil, allow_weaker: false, # rubocop:disable Metrics/ParameterLists
|
|
56
|
-
own_chains: nil, only_primary: false)
|
|
74
|
+
own_chains: nil, only_primary: false, stages: STAGES)
|
|
75
|
+
@stages = stages.map(&:to_s)
|
|
57
76
|
@items = parse_items(items, provider)
|
|
58
77
|
@limits = limits.transform_keys(&:to_s)
|
|
59
|
-
@rank =
|
|
60
|
-
@allow_weaker = allow_weaker == true
|
|
78
|
+
@rank, @allow_weaker = critique_policy(rank, allow_weaker)
|
|
61
79
|
@own_chains = own_chains || StageChains.new({})
|
|
62
80
|
@only_primary = only_primary
|
|
63
81
|
@warnings = []
|
|
64
82
|
@starts = resolve_starts(starts.transform_keys(&:to_s), inherited_starts.map(&:to_s))
|
|
65
83
|
end
|
|
66
84
|
|
|
85
|
+
def stage?(stage)
|
|
86
|
+
@stages.include?(stage.to_s)
|
|
87
|
+
end
|
|
88
|
+
|
|
67
89
|
def pool(stage = 'generate')
|
|
68
90
|
limit = @limits.fetch(stage.to_s)
|
|
69
91
|
@items.map do |item|
|
|
70
92
|
ModelCandidate.new(provider: item[:provider], model: item[:model],
|
|
71
|
-
max_prompt_chars: item[:max_prompt_chars] || limit)
|
|
93
|
+
max_prompt_chars: item[:max_prompt_chars] || limit, api_base: item[:api_base])
|
|
72
94
|
end
|
|
73
95
|
end
|
|
74
96
|
|
|
75
97
|
def chain(stage)
|
|
76
98
|
stage = stage.to_s
|
|
99
|
+
raise ArgumentError, "unknown LLM stage #{stage.inspect}" unless stage?(stage)
|
|
77
100
|
return @own_chains.chain(stage) unless pool_stage?(stage)
|
|
78
101
|
|
|
79
102
|
trim(pool(stage).rotate(index_of(@starts[stage] || @items.first[:model])))
|
|
@@ -107,21 +130,20 @@ module Aireview
|
|
|
107
130
|
|
|
108
131
|
# The order and policy of the pool go into the review key: they decide
|
|
109
132
|
# which model checks the findings. nil when a stage is outside the pool.
|
|
133
|
+
# Without Critique only the generate part: a critique setting nobody
|
|
134
|
+
# uses must not change the key, and the pool must stay in it.
|
|
110
135
|
def signature
|
|
111
|
-
return nil unless
|
|
136
|
+
return nil unless all_stages_in_pool?
|
|
112
137
|
|
|
113
|
-
{
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
'rank' => @rank,
|
|
118
|
-
'allow_weaker' => @allow_weaker
|
|
119
|
-
}
|
|
138
|
+
signature = {'models' => pool.map(&:to_s), 'generate_start' => @starts['generate']}
|
|
139
|
+
return signature unless stage?('critique')
|
|
140
|
+
|
|
141
|
+
signature.merge('critique_start' => @starts['critique'], 'rank' => @rank, 'allow_weaker' => @allow_weaker)
|
|
120
142
|
end
|
|
121
143
|
|
|
122
144
|
# The critique selection rule in words, for --dry-run.
|
|
123
145
|
def rule
|
|
124
|
-
return nil unless
|
|
146
|
+
return nil unless stage?('critique') && all_stages_in_pool?
|
|
125
147
|
return @rank if @rank == 'any' || !@allow_weaker
|
|
126
148
|
|
|
127
149
|
"#{@rank}, weaker allowed"
|
|
@@ -132,7 +154,7 @@ module Aireview
|
|
|
132
154
|
end
|
|
133
155
|
|
|
134
156
|
def pool_stage?(stage)
|
|
135
|
-
!@own_chains.stage?(stage)
|
|
157
|
+
stage?(stage) && !@own_chains.stage?(stage)
|
|
136
158
|
end
|
|
137
159
|
|
|
138
160
|
def pool_member?(model)
|
|
@@ -147,12 +169,12 @@ module Aireview
|
|
|
147
169
|
|
|
148
170
|
private
|
|
149
171
|
|
|
150
|
-
def
|
|
151
|
-
|
|
172
|
+
def all_stages_in_pool?
|
|
173
|
+
@stages.all? { |stage| pool_stage?(stage) }
|
|
152
174
|
end
|
|
153
175
|
|
|
154
176
|
def rank_applies?(after)
|
|
155
|
-
|
|
177
|
+
all_stages_in_pool? && @rank != 'any' && pool_member?(after)
|
|
156
178
|
end
|
|
157
179
|
|
|
158
180
|
def trim(chain)
|
|
@@ -174,7 +196,7 @@ module Aireview
|
|
|
174
196
|
|
|
175
197
|
# The start of a stage with its own chain is not checked: it is outside the pool.
|
|
176
198
|
def resolve_starts(starts, inherited)
|
|
177
|
-
|
|
199
|
+
@stages.to_h do |stage|
|
|
178
200
|
start = starts[stage]
|
|
179
201
|
next [stage, nil] if Aireview::Utils.blank?(start) || !pool_stage?(stage)
|
|
180
202
|
next [stage, start] if pool_member?(start)
|
|
@@ -192,15 +214,22 @@ module Aireview
|
|
|
192
214
|
raise ConfigError, "#{model} is not in llm.models: #{pool.join(', ')}"
|
|
193
215
|
end
|
|
194
216
|
|
|
195
|
-
# A model is given by name
|
|
217
|
+
# A model is given by name, as "provider/name" or by its full name; a
|
|
218
|
+
# candidate matches by provider, name and server address.
|
|
196
219
|
def match?(item, model)
|
|
197
|
-
return
|
|
220
|
+
return self.class.item_name(item) == model.to_s if model.is_a?(ModelCandidate)
|
|
198
221
|
|
|
199
|
-
item
|
|
222
|
+
self.class.names(item).include?(model.to_s)
|
|
200
223
|
end
|
|
201
224
|
|
|
202
225
|
def match_candidate?(candidate, model)
|
|
203
|
-
candidate.
|
|
226
|
+
candidate.names.include?(model.to_s)
|
|
227
|
+
end
|
|
228
|
+
|
|
229
|
+
def critique_policy(rank, allow_weaker)
|
|
230
|
+
return [nil, false] unless stage?('critique')
|
|
231
|
+
|
|
232
|
+
[validate_rank(rank), allow_weaker == true]
|
|
204
233
|
end
|
|
205
234
|
|
|
206
235
|
def validate_rank(rank)
|
|
@@ -214,7 +243,7 @@ module Aireview
|
|
|
214
243
|
parsed = Array(items).each_with_index.map { |item, index| self.class.parse_item(item, index, provider) }
|
|
215
244
|
raise ConfigError, 'llm.models must not be empty' if parsed.empty?
|
|
216
245
|
|
|
217
|
-
names = parsed.map { |item|
|
|
246
|
+
names = parsed.map { |item| self.class.item_name(item) }
|
|
218
247
|
duplicates = names.tally.select { |_, count| count > 1 }.keys
|
|
219
248
|
raise ConfigError, "llm.models has duplicates: #{duplicates.join(', ')}" unless duplicates.empty?
|
|
220
249
|
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
# Jev questions about one candidate, the counterpart of critique.txt: the
|
|
2
|
+
# rules are the same, only asked as separate yes/no and choice questions.
|
|
3
|
+
# When critique.txt changes, change these too (spec/prompts_spec.rb checks
|
|
4
|
+
# the key rules in both). %{id} is the candidate id; the state holds the
|
|
5
|
+
# candidate at `candidates.<id>` and the diff at `diff`. Keys "true" and
|
|
6
|
+
# "false" are quoted: YAML would read them as booleans.
|
|
7
|
+
real_issue:
|
|
8
|
+
type: noul
|
|
9
|
+
instructions: >-
|
|
10
|
+
Is candidate `candidates.%{id}` a real, well-supported problem in the code
|
|
11
|
+
changed by this merge request?
|
|
12
|
+
criteria:
|
|
13
|
+
"true": >-
|
|
14
|
+
The problem is directly confirmed by the diff, the MR description, the
|
|
15
|
+
Jira task or the changed control/data flow, and it can lead to a real
|
|
16
|
+
defect, regression, security/performance issue, data loss or a mismatch
|
|
17
|
+
with the task. For category task_mismatch: the diff contains a concrete
|
|
18
|
+
artifact (debug code such as puts, p, binding.pry, byebug, console.log,
|
|
19
|
+
debugger; commented-out blocks; unused debug/test/helper methods;
|
|
20
|
+
temporary TODO/HACK/FIXME/DEBUG markers from this diff; disabled tests
|
|
21
|
+
such as skip, xit, pending without an explanation) or a change that
|
|
22
|
+
directly contradicts the MR or Jira.
|
|
23
|
+
"false": >-
|
|
24
|
+
The finding is not confirmed by the diff or Jira; it is based on an
|
|
25
|
+
assumption about code outside the diff; it contradicts the diff; it is
|
|
26
|
+
not actionable; it boils down to "may be redundant", "may conflict",
|
|
27
|
+
"may be unnecessary" or "worth checking" without a clear sign of
|
|
28
|
+
breakage, in any category. For category task_mismatch: implementation
|
|
29
|
+
details (config flags, deploy settings, new fields, helper methods),
|
|
30
|
+
accompanying refactoring and any change whose relation to the task is
|
|
31
|
+
plausible. A candidate with a `note` could not be anchored to the diff
|
|
32
|
+
mechanically and needs to be checked against the diff more carefully.
|
|
33
|
+
Missing text in a truncated part of the context does not mean a
|
|
34
|
+
missing requirement or missing code.
|
|
35
|
+
enough_context:
|
|
36
|
+
type: noul
|
|
37
|
+
instructions: >-
|
|
38
|
+
Does the state contain enough to confirm or refute candidate
|
|
39
|
+
`candidates.%{id}`?
|
|
40
|
+
criteria:
|
|
41
|
+
"true": >-
|
|
42
|
+
The code the finding is about is present in `candidates.%{id}.evidence`
|
|
43
|
+
or in `diff`, and for a finding about the task the relevant MR or Jira
|
|
44
|
+
requirements are present and not cut off where they matter.
|
|
45
|
+
"false": >-
|
|
46
|
+
The decisive code or requirement is missing: the file is not shown, the
|
|
47
|
+
relevant part is truncated, or the finding depends on code or
|
|
48
|
+
requirements that are not in the state.
|
|
49
|
+
version_claim:
|
|
50
|
+
type: noul
|
|
51
|
+
instructions: >-
|
|
52
|
+
Does candidate `candidates.%{id}` claim that a version of a package,
|
|
53
|
+
library, tool or image tag does not exist or has not been released yet?
|
|
54
|
+
criteria:
|
|
55
|
+
"true": >-
|
|
56
|
+
The finding says that a specified version does not exist, is not
|
|
57
|
+
released yet or is invalid because of its number.
|
|
58
|
+
"false": >-
|
|
59
|
+
The finding is about anything else, including syntax errors and explicit
|
|
60
|
+
contradictions with the MR or Jira requirements.
|
|
61
|
+
# Only when the request holds more than one candidate; the options are the
|
|
62
|
+
# other ids plus none.
|
|
63
|
+
duplicate_of:
|
|
64
|
+
type: choice
|
|
65
|
+
instructions: >-
|
|
66
|
+
Does candidate `candidates.%{id}` describe the same problem as another
|
|
67
|
+
candidate?
|
|
68
|
+
other: The same problem as candidate `candidates.%{other}`.
|
|
69
|
+
none: A problem that no other candidate describes.
|
|
70
|
+
# Goes to the log only; its answer does not decide anything yet.
|
|
71
|
+
severity:
|
|
72
|
+
type: choice
|
|
73
|
+
instructions: >-
|
|
74
|
+
How severe is the problem described by candidate `candidates.%{id}`, if it
|
|
75
|
+
is real?
|
|
76
|
+
criteria:
|
|
77
|
+
critical: Breaks production behaviour, loses or corrupts data, or opens a security hole.
|
|
78
|
+
major: A real defect or regression on a common path, or a clear mismatch with the task.
|
|
79
|
+
minor: A limited edge case, a test gap or a maintainability issue.
|
|
@@ -23,14 +23,19 @@ module Aireview
|
|
|
23
23
|
# that way the diff, the MR description, the Jira context, the review
|
|
24
24
|
# instructions and ignore_paths enter it by themselves. What else affects
|
|
25
25
|
# the result — provider, model and temperature of the stages, the shared
|
|
26
|
-
# pool with its critique policy — is known by
|
|
27
|
-
# Without a pool the key is the same as before.
|
|
26
|
+
# pool with its critique policy, Jev as the critique engine — is known by
|
|
27
|
+
# Config#result_signature. Without a pool the key is the same as before.
|
|
28
|
+
# The LLM Critique counts only when it can run (its prompt is there),
|
|
29
|
+
# Jev only when it decides (its question templates are there): with the
|
|
30
|
+
# default engine the key is what it was before Jev.
|
|
28
31
|
def key(prompts:, config:)
|
|
29
32
|
signature = config.result_signature
|
|
30
33
|
source = {
|
|
31
34
|
'generate' => [*signature['generate'], prompts[:generate_prompt]],
|
|
32
35
|
'critique' => prompts[:critique_prompt] ? [*signature['critique'], prompts[:critique_prompt]] : nil
|
|
33
36
|
}
|
|
37
|
+
jev = prompts[:jev_questions]
|
|
38
|
+
source['critique_engine'] = [*signature['critique_engine'], jev] if jev
|
|
34
39
|
source['pool'] = signature['pool'] if signature['pool']
|
|
35
40
|
|
|
36
41
|
Digest::SHA256.hexdigest(JSON.generate(source))[0, 16]
|