aireview 0.2.1 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,477 @@
1
+ # frozen_string_literal: true
2
+ require 'set'
3
+ require_relative 'errors'
4
+ require_relative 'llm_failure'
5
+ require_relative 'model_state'
6
+
7
+ module Aireview
8
+ # Walks the models and keys of a stage. An overload is a property of the
9
+ # model, a quota is a property of "key + model", so a 503 switches the
10
+ # model and a quota switches the key. An overloaded model goes into
11
+ # quarantine and is skipped until it expires; once the chain has been
12
+ # walked, the walk goes round again over the models whose quarantine has
13
+ # expired. Two limits: the time budget of the run and the number of
14
+ # requests sent per model per stage (ModelState).
15
+ class LlmRouter # rubocop:disable Metrics/ClassLength
16
+ Route = Struct.new(:candidate, :candidate_index, :key, :key_index, :key_count, keyword_init: true) do
17
+ def fallback?
18
+ candidate_index.positive?
19
+ end
20
+
21
+ def to_s
22
+ key_count > 1 ? "#{candidate} (key #{key_index + 1}/#{key_count})" : candidate.to_s
23
+ end
24
+ end
25
+
26
+ # A journal entry about a visit to a route, for the error message: either
27
+ # the kind of failure with the number of requests, or a note on why the
28
+ # route was skipped.
29
+ Visit = Struct.new(:route, :kind, :tries, :error, :note, keyword_init: true) do
30
+ def to_s
31
+ "#{route}: #{note || "#{LlmRouter::KIND_LABELS.fetch(kind)} after #{tries} attempt(s)"}"
32
+ end
33
+ end
34
+
35
+ Slot = Struct.new(:candidate, :index)
36
+ # The key the next model of the same provider continues from.
37
+ Carry = Struct.new(:provider, :key_index)
38
+ Delay = Struct.new(:seconds, :source)
39
+
40
+ SWITCH_HINT = 'Try again later or switch model via --generate-model/--critique-model.'
41
+
42
+ KIND_LABELS = {
43
+ daily_quota: 'daily quota exhausted',
44
+ rate_limit: 'rate limited',
45
+ overloaded: 'overloaded',
46
+ timeout: 'timed out',
47
+ unavailable: 'model unavailable'
48
+ }.freeze
49
+
50
+ MAX_ATTEMPTS_PER_MODEL = ModelState::MAX_REQUESTS_PER_STAGE
51
+ # The first failure of a visit to a model gets one short retry, the
52
+ # second one a quarantine and the next model.
53
+ SHORT_RETRY_DELAY = 30.0
54
+ SHORT_RETRY_JITTER_RANGE = 0.85..1.15
55
+ RATE_LIMIT_BASE_DELAY = 2.0
56
+ RATE_LIMIT_JITTER_RANGE = 2.0..5.0
57
+ PROVIDER_RETRY_DELAY_MULTIPLIER_RANGE = 2.0..2.4
58
+ RETRY_WAIT_LOG_FORMAT = 'LLM %<stage>s request will sleep %<delay>.1fs before retry%<source>s ' \
59
+ '(request %<next_request>d/%<max_requests>d of the model in this stage, model=%<model>s)'
60
+
61
+ # clock and sleeper are injected in tests: the schedule is checked without
62
+ # real waiting.
63
+ def initialize(config:, logger:, routing: nil, clock: nil, sleeper: nil)
64
+ @config = config
65
+ @routing = routing || config.routing
66
+ @logger = logger
67
+ @clock = clock || -> { Process.clock_gettime(Process::CLOCK_MONOTONIC) }
68
+ @sleeper = sleeper || ->(seconds) { sleep(seconds) }
69
+ @models = Hash.new { |states, name| states[name] = ModelState.new }
70
+ @cursor = {}
71
+ @used = {}
72
+ @deadline = nil
73
+ end
74
+
75
+ # The block receives a route and a request timeout, makes the request and
76
+ # returns the answer. An error raised by the block is classified, then
77
+ # comes a retry, another key, another model, or ApiError once the routes
78
+ # are exhausted. pinned — only the model that answered last in this
79
+ # stage (the JSON repair): its failure is RouteExhaustedError, and the
80
+ # stage restarts on another model.
81
+ def call(stage:, request_chars:, pinned: false, &request)
82
+ @deadline ||= now + @config.llm_time_budget
83
+ visits = []
84
+ noted = Set.new
85
+ carry = nil
86
+ loop do
87
+ slots = available_slots(stage, request_chars, pinned, visits, noted)
88
+ slot = slots.empty? ? nil : ready_slot(stage, slots, visits)
89
+ raise exhausted_error(stage, visits, pinned) unless slot
90
+
91
+ status, value = try_candidate(stage, slot, visits, carry, &request)
92
+ return value if status == :ok
93
+
94
+ carry = value
95
+ end
96
+ end
97
+
98
+ # Stages answered by a model other than the primary one, for the report.
99
+ def fallback_models
100
+ @used.select { |_, route| route.fallback? }.transform_values { |route| route.candidate.to_s }
101
+ end
102
+
103
+ # The model that answered last in the stage and its place in the chain.
104
+ def answered(stage)
105
+ route = @used[stage.to_s]
106
+ return nil unless route
107
+
108
+ "#{route.candidate} (#{route.candidate_index + 1}/#{chain_for(stage.to_s).size})"
109
+ end
110
+
111
+ # Critique answered with a model below Generate in the pool, for the report.
112
+ def critique_weaker?
113
+ generate = @used['generate']
114
+ critique = @used['critique']
115
+ return false unless generate && critique
116
+
117
+ @routing.weaker?(critique.candidate, generate.candidate)
118
+ end
119
+
120
+ # Excludes the model that answered until the end of the stage: its result
121
+ # is invalid (not JSON, not the schema) even after the repair. The next
122
+ # request of the stage goes to another model. Returns the excluded model;
123
+ # nil — nothing to exclude.
124
+ def exclude_answered(stage:, reason:)
125
+ route = @used[stage.to_s]
126
+ return nil unless route
127
+
128
+ state(route.candidate).exclude_for_stage(stage, reason)
129
+ @logger.warn("LLM #{stage}: #{route.candidate} excluded for this stage: #{reason}")
130
+ route.candidate
131
+ end
132
+
133
+ def remaining_time
134
+ @deadline ? @deadline - now : @config.llm_time_budget.to_f
135
+ end
136
+
137
+ private
138
+
139
+ def state(candidate)
140
+ @models[candidate.to_s]
141
+ end
142
+
143
+ # The models that can still be tried, in walking order. A skipped model
144
+ # enters the journal once per call, and only if it is not already there
145
+ # with an error of this same call.
146
+ def available_slots(stage, request_chars, pinned, visits, noted)
147
+ chain = pinned ? pinned_chain(stage) : ordered_chain(stage)
148
+ chain.filter_map do |candidate, index|
149
+ note = skip_reason(stage, candidate, request_chars)
150
+ next Slot.new(candidate, index) unless note
151
+
152
+ note_skip(visits, noted, candidate, index, note)
153
+ nil
154
+ end
155
+ end
156
+
157
+ def note_skip(visits, noted, candidate, index, note)
158
+ return unless noted.add?([index, note])
159
+ return if visits.any? { |visit| visit.error && visit.route.candidate == candidate }
160
+
161
+ visits << Visit.new(route: bare_route(candidate, index), note: note)
162
+ end
163
+
164
+ def bare_route(candidate, index)
165
+ Route.new(candidate: candidate, candidate_index: index, key_count: 1)
166
+ end
167
+
168
+ def skip_reason(stage, candidate, request_chars)
169
+ if request_chars > candidate.max_prompt_chars
170
+ "skipped, request #{request_chars} chars over max_prompt_chars=#{candidate.max_prompt_chars}"
171
+ else
172
+ state(candidate).skip_reason(stage)
173
+ end
174
+ end
175
+
176
+ # The first model not in quarantine; when all are, wait for the nearest
177
+ # release if it fits into the budget. nil — it does not.
178
+ def ready_slot(stage, slots, visits)
179
+ ensure_time_left!(stage, bare_route(slots.first.candidate, slots.first.index), visits)
180
+ moment = now
181
+ ready = slots.find { |slot| state(slot.candidate).quarantine_left(moment) <= 0 }
182
+ return ready if ready
183
+
184
+ slot = slots.min_by { |item| state(item.candidate).quarantine_left(moment) }
185
+ wait = state(slot.candidate).quarantine_left(moment)
186
+ if wait > remaining_time
187
+ visits << Visit.new(route: bare_route(slot.candidate, slot.index),
188
+ note: format('quarantined for another %.0fs, over the time budget', wait))
189
+ return nil
190
+ end
191
+
192
+ @logger.warn(format('LLM %<stage>s: every model is quarantined, waiting %<wait>.0fs for %<model>s',
193
+ stage: stage, wait: wait, model: slot.candidate))
194
+ pause(wait)
195
+ slot
196
+ end
197
+
198
+ # On :next_model hands over the current key: an overload is a property of
199
+ # the model, and the next model of the same provider continues from the
200
+ # same key instead of going back to the first one, whose quota may
201
+ # already be gone.
202
+ #
203
+ # Every visit either sends a request or excludes the model: otherwise the
204
+ # walk round the pool would never stop. A model whose keys are all out of
205
+ # daily quota is excluded until the end of the run.
206
+ def try_candidate(stage, slot, visits, carry, &request)
207
+ routes = candidate_routes(stage, slot, carry)
208
+ tried = false
209
+ routes.each do |route|
210
+ next if quota_exhausted?(stage, route, visits)
211
+
212
+ tried = true
213
+ log_switch(stage, route, visits)
214
+ status, response = try_route(stage, route, visits, &request)
215
+ return [:ok, remember(stage, route, response)] if status == :ok
216
+ return [:next_model, Carry.new(route.candidate.provider, route.key_index)] if status == :next_model
217
+ end
218
+ state(slot.candidate).exclude('daily quota exhausted on every key') unless tried
219
+ [:next_model, carry]
220
+ end
221
+
222
+ # The keys start from the carried one (after an overload — the current
223
+ # key, after an answer — the one that answered), the rest follow: a quota
224
+ # is bound to "key + model", a failure on one model does not write the
225
+ # key off for another.
226
+ def candidate_routes(stage, slot, carry)
227
+ keys = @config.provider_api_keys(slot.candidate.provider)
228
+ routes = keys.each_with_index.map do |key, key_index|
229
+ Route.new(candidate: slot.candidate, candidate_index: slot.index, key: key,
230
+ key_index: key_index, key_count: keys.size)
231
+ end
232
+ routes.rotate(start_key(stage, slot, carry))
233
+ end
234
+
235
+ def start_key(stage, slot, carry)
236
+ return carry.key_index if carry && carry.provider == slot.candidate.provider
237
+
238
+ cursor_candidate, cursor_key = @cursor.fetch(stage, [0, 0])
239
+ cursor_candidate == slot.index ? cursor_key : 0
240
+ end
241
+
242
+ # A visit to a route: requests until an answer, a retry or a refusal. try
243
+ # is the request number within this visit; the per-stage counter of the
244
+ # model is kept by ModelState.
245
+ def try_route(stage, route, visits)
246
+ model = state(route.candidate)
247
+ try = 0
248
+ loop do
249
+ return [:next_model, nil] unless requests_left?(stage, route)
250
+
251
+ try += 1
252
+ ensure_time_left!(stage, route, visits)
253
+ model.record_request(stage)
254
+ begin
255
+ return [:ok, yield(route, request_timeout)]
256
+ rescue StandardError => e
257
+ decision = handle_failure(stage, route, try, e, visits)
258
+ return [decision, nil] unless decision == :retry
259
+ end
260
+ end
261
+ end
262
+
263
+ # :retry — the pause has been waited out, retry; otherwise the decision for the route.
264
+ def handle_failure(stage, route, try, error, visits)
265
+ kind = LlmFailure.classify(error)
266
+ raise error if kind == :unhandled
267
+
268
+ @logger.warn("LLM #{stage} request failed (#{route}): #{error.class}: #{error.message}")
269
+ decision, delay = decide(kind, try, error)
270
+ raise ApiError, single_route_message(error) if decision == :fail
271
+
272
+ if decision == :retry
273
+ return :retry if requests_left?(stage, route) && waited_before_retry?(stage, route, delay)
274
+
275
+ decision = give_up(kind)
276
+ end
277
+ record_failure(stage, route, kind, decision, error)
278
+ visits << Visit.new(route: route, kind: kind, tries: try, error: error)
279
+ decision
280
+ end
281
+
282
+ def requests_left?(stage, route)
283
+ return true if state(route.candidate).requests_left?(stage)
284
+
285
+ @logger.warn("LLM #{stage}: #{route.candidate} has used its #{MAX_ATTEMPTS_PER_MODEL} attempts")
286
+ false
287
+ end
288
+
289
+ # :retry with a delay, :next_key, :next_model or :fail.
290
+ def decide(kind, try, error)
291
+ case kind
292
+ when :fatal then [:fail]
293
+ when :unavailable then [:next_model]
294
+ when :daily_quota then [:next_key]
295
+ when :rate_limit then try == 1 ? [:retry, rate_limit_delay(error)] : [:next_key]
296
+ else try == 1 ? [:retry, short_delay] : [:next_model]
297
+ end
298
+ end
299
+
300
+ def give_up(kind)
301
+ kind == :rate_limit ? :next_key : :next_model
302
+ end
303
+
304
+ # A daily quota marks the key of the model, a missing model marks the
305
+ # model for the whole run, an overload or a timeout quarantines it, a
306
+ # per-minute limit quarantines it for the provider's hint (the next key
307
+ # is tried at once; the quarantine only affects the next visit).
308
+ def record_failure(stage, route, kind, decision, error)
309
+ model = state(route.candidate)
310
+ case kind
311
+ when :daily_quota then model.exhaust_key(route.key_index)
312
+ when :unavailable then model.exclude('model unavailable earlier in this run')
313
+ when :rate_limit then quarantine(stage, route, rate_limit_delay(error).seconds)
314
+ when :overloaded, :timeout then quarantine(stage, route, @config.overloaded_quarantine) if decision == :next_model
315
+ end
316
+ end
317
+
318
+ def quarantine(stage, route, seconds)
319
+ state(route.candidate).quarantine(now + seconds)
320
+ @logger.warn(format('LLM %<stage>s: %<model>s quarantined for %<seconds>.0fs',
321
+ stage: stage, model: route.candidate, seconds: seconds))
322
+ end
323
+
324
+ def short_delay
325
+ Delay.new(SHORT_RETRY_DELAY * rand(SHORT_RETRY_JITTER_RANGE), '')
326
+ end
327
+
328
+ def rate_limit_delay(error)
329
+ hint = LlmFailure.retry_after_seconds(error.message)
330
+ return Delay.new(RATE_LIMIT_BASE_DELAY + rand(RATE_LIMIT_JITTER_RANGE), '') unless hint
331
+
332
+ multiplier = rand(PROVIDER_RETRY_DELAY_MULTIPLIER_RANGE)
333
+ Delay.new(hint * multiplier, format(' (provider retry hint %<hint>.1fs, multiplier %<multiplier>.2fx)',
334
+ hint: hint, multiplier: multiplier))
335
+ end
336
+
337
+ # false — the pause does not fit into the time budget, no retry.
338
+ def waited_before_retry?(stage, route, delay)
339
+ if delay.seconds > remaining_time
340
+ @logger.warn(
341
+ format('LLM %<stage>s: no time budget left for a %<delay>.0fs pause (%<left>.0fs remaining, %<route>s)',
342
+ stage: stage, delay: delay.seconds, left: [remaining_time, 0].max, route: route)
343
+ )
344
+ return false
345
+ end
346
+
347
+ @logger.warn(
348
+ format(RETRY_WAIT_LOG_FORMAT, stage: stage, delay: delay.seconds, source: delay.source,
349
+ next_request: state(route.candidate).sent(stage) + 1,
350
+ max_requests: MAX_ATTEMPTS_PER_MODEL, model: route.candidate.model)
351
+ )
352
+ started_at = now
353
+ pause(delay.seconds)
354
+ @logger.info(format('LLM %<stage>s retry wait completed after %<waited>.1fs (model=%<model>s)',
355
+ stage: stage, waited: now - started_at, model: route.candidate.model))
356
+ true
357
+ end
358
+
359
+ def ensure_time_left!(stage, route, visits)
360
+ return if remaining_time >= 1
361
+
362
+ visits << Visit.new(route: route, note: 'no time budget left')
363
+ raise ApiError, "LLM time budget of #{@config.llm_time_budget}s is exhausted. " \
364
+ "#{exhausted_message(stage, visits)}"
365
+ end
366
+
367
+ def request_timeout
368
+ [@config.llm_timeout.to_f, remaining_time].min
369
+ end
370
+
371
+ def quota_exhausted?(stage, route, visits)
372
+ return false unless state(route.candidate).key_exhausted?(route.key_index)
373
+
374
+ @logger.info("LLM #{stage}: skipping #{route}, daily quota exhausted earlier in this run")
375
+ visits << Visit.new(route: route, note: 'daily quota exhausted earlier in this run')
376
+ true
377
+ end
378
+
379
+ # A model that answered is not overloaded, whichever key answered: a
380
+ # quarantine set through another key of the same model is lifted.
381
+ def remember(stage, route, response)
382
+ @cursor[stage] = [route.candidate_index, route.key_index]
383
+ @used[stage] = route
384
+ state(route.candidate).lift_quarantine
385
+ response
386
+ end
387
+
388
+ def log_switch(stage, route, visits)
389
+ return if visits.empty?
390
+
391
+ @logger.warn("LLM #{stage}: switching to #{route} after #{visits.last}")
392
+ end
393
+
394
+ # --- walking order ---
395
+
396
+ # The next request of the stage (the JSON repair, for instance) starts
397
+ # from the model that answered; the rest stay in reserve after it.
398
+ def ordered_chain(stage)
399
+ start_candidate, = @cursor.fetch(stage, [0, 0])
400
+ chain_for(stage).each_with_index.to_a.rotate(start_candidate)
401
+ end
402
+
403
+ # The critique chain depends on the model that answered in Generate:
404
+ # with a shared pool Critique does not go below it. Computed once per
405
+ # stage so that the repair and a restart walk the same chain.
406
+ def chain_for(stage)
407
+ @chains ||= {}
408
+ @chains[stage] ||= build_chain(stage)
409
+ end
410
+
411
+ # Warnings the plan raises while building a chain (an ignored
412
+ # critique.start, for instance) were not there when the CLI started —
413
+ # the router logs them.
414
+ def build_chain(stage)
415
+ known = @routing.warnings.size
416
+ chain = if stage == 'critique' && @used['generate']
417
+ @routing.critique_chain(after: @used['generate'].candidate)
418
+ else
419
+ @routing.chain(stage)
420
+ end
421
+ @routing.warnings.drop(known).each { |warning| @logger.warn(warning) }
422
+ chain
423
+ end
424
+
425
+ def pinned_chain(stage)
426
+ route = @used[stage]
427
+ raise ApiError, "LLM #{stage}: no model has answered yet, nothing to pin" unless route
428
+
429
+ [[route.candidate, route.candidate_index]]
430
+ end
431
+
432
+ # --- error messages ---
433
+
434
+ def exhausted_error(stage, visits, pinned)
435
+ (pinned ? RouteExhaustedError : ApiError).new(exhausted_message(stage, visits))
436
+ end
437
+
438
+ # One error on one route — a short message about it; otherwise the list
439
+ # of routes with reasons.
440
+ def exhausted_message(stage, visits)
441
+ return single_route_message(visits.first.error) if visits.size == 1 && visits.first.error
442
+
443
+ "LLM #{stage} request failed on every configured route: #{visits.join('; ')}. #{SWITCH_HINT}"
444
+ end
445
+
446
+ def single_route_message(error)
447
+ case LlmFailure.classify(error)
448
+ when :timeout
449
+ "LLM request timed out after #{@config.llm_timeout} seconds. #{SWITCH_HINT}"
450
+ when :overloaded
451
+ "LLM service is temporarily unavailable or overloaded: #{error.message}. #{SWITCH_HINT}"
452
+ when :rate_limit, :daily_quota
453
+ "LLM rate limit exceeded: #{error.message}. #{SWITCH_HINT}"
454
+ when :unavailable
455
+ "LLM model is unavailable: #{error.message}. #{SWITCH_HINT}"
456
+ else
457
+ fatal_message(error)
458
+ end
459
+ end
460
+
461
+ def fatal_message(error)
462
+ if defined?(RubyLLM::ContextLengthExceededError) && error.is_a?(RubyLLM::ContextLengthExceededError)
463
+ "LLM context limit exceeded: #{error.message}. Try reducing the MR diff or ignore more paths."
464
+ else
465
+ "LLM API request failed: #{error.message}"
466
+ end
467
+ end
468
+
469
+ def now
470
+ @clock.call
471
+ end
472
+
473
+ def pause(seconds)
474
+ @sleeper.call(seconds)
475
+ end
476
+ end
477
+ end
@@ -0,0 +1,28 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Aireview
4
+ KNOWN_PROVIDERS = %w[gemini ollama].freeze
5
+
6
+ # A model in a stage chain: provider, name and request size limit.
7
+ # Compared by provider and name — the limit depends on the stage.
8
+ ModelCandidate = Struct.new(:provider, :model, :max_prompt_chars, keyword_init: true) do
9
+ def to_s
10
+ "#{provider}/#{model}"
11
+ end
12
+
13
+ def same_model?(other)
14
+ to_s == other.to_s
15
+ end
16
+
17
+ # A "provider/name" string or a bare name (the provider is separated by
18
+ # a slash because Ollama tags contain a colon), or a hash with model.
19
+ def self.parse_item(item)
20
+ return item unless item.is_a?(String)
21
+
22
+ provider, model = item.split('/', 2)
23
+ return {'provider' => provider, 'model' => model} if model && KNOWN_PROVIDERS.include?(provider)
24
+
25
+ {'model' => item}
26
+ end
27
+ end
28
+ end
@@ -0,0 +1,148 @@
1
+ # frozen_string_literal: true
2
+ require 'logger'
3
+ require 'socket'
4
+ require_relative 'errors'
5
+ require_relative 'stages'
6
+ require_relative 'llm_failure'
7
+ require_relative 'review_pipeline'
8
+ require_relative 'result_parser'
9
+ require_relative 'llm_client'
10
+ require_relative 'output_schemas'
11
+
12
+ module Aireview
13
+ # `aireview models check`: every model of both stage chains gets one
14
+ # request with the production Generate schema and one with the Critique
15
+ # schema on a tiny synthetic MR. The answer goes through the same
16
+ # validation as in a run: the provider's catalog is not consulted, "the
17
+ # model is listed" does not mean "our request with the schema passes on
18
+ # it". No reserves, quarantine or walking — this is a check, not a review.
19
+ class ModelChecker
20
+ PROBE_MERGE_REQUEST = {
21
+ 'title' => 'Fix order total',
22
+ 'description' => 'The total must include the quantity.',
23
+ 'source_branch' => 'fix/order-total',
24
+ 'target_branch' => 'master',
25
+ 'author' => {'name' => 'aireview'}
26
+ }.freeze
27
+ PROBE_CHANGES = [
28
+ {
29
+ 'old_path' => 'app/models/order.rb',
30
+ 'new_path' => 'app/models/order.rb',
31
+ 'diff' => "@@ -1,5 +1,5 @@\n class Order\n def total\n- price * quantity\n+ price\n end\n"
32
+ }
33
+ ].freeze
34
+ PROBE_CANDIDATE_IDS = ['C1'].freeze
35
+ # ok and skipped do not block a release, everything else does. In strict
36
+ # mode (a runner where Ollama must be running) skipped fails too.
37
+ PASSING = %i[ok skipped].freeze
38
+ STRICT_PASSING = %i[ok].freeze
39
+
40
+ Result = Struct.new(:candidate, :stage, :status, :detail, :seconds, keyword_init: true) do
41
+ def to_s
42
+ text = seconds ? "#{status} (#{format('%.1fs', seconds)})" : status.to_s
43
+ text = "#{text}: #{detail}" if detail
44
+ "#{candidate.to_s.ljust(32)} #{stage.ljust(8)} #{text}"
45
+ end
46
+ end
47
+
48
+ # strict — skipped counts as a failure (a runner where Ollama must be running).
49
+ def initialize(config:, out:, strict: false, logger: Logger.new($stderr), **dependencies)
50
+ @config = config
51
+ @out = out
52
+ @strict = strict
53
+ client = dependencies[:client]
54
+ sleeper = dependencies[:sleeper]
55
+ @logger = logger
56
+ @client = client || LlmClient.new(config: config, logger: logger)
57
+ @sleeper = sleeper || ->(seconds) { sleep(seconds) }
58
+ @pipeline = ReviewPipeline.new(config: config, logger: logger)
59
+ @parser = ResultParser.new
60
+ end
61
+
62
+ # Exit code: 0 — every model answered by the schema (or was skipped),
63
+ # 1 — at least one did not.
64
+ def run
65
+ @config.require_llm_configuration!
66
+ candidates = STAGES.flat_map { |stage| @config.stage_chain(stage) }.uniq(&:to_s)
67
+ prompts = probe_prompts
68
+ @out.puts("Checking #{candidates.size} model(s) with the generate and critique schemas")
69
+ results = candidates.flat_map do |candidate|
70
+ STAGES.map { |stage| check(candidate, stage, prompts).tap { |result| @out.puts(result) } }
71
+ end
72
+ summary(results)
73
+ end
74
+
75
+ private
76
+
77
+ def probe_prompts
78
+ dry_run = @pipeline.dry_run_prompts(merge_request: PROBE_MERGE_REQUEST, changes: PROBE_CHANGES, critique: true)
79
+ {'generate' => dry_run[:generate_prompt], 'critique' => dry_run[:critique_prompt]}
80
+ end
81
+
82
+ def check(candidate, stage, prompts)
83
+ started_at = Process.clock_gettime(Process::CLOCK_MONOTONIC)
84
+ raw = probe_with_one_retry(candidate, stage, prompts[stage])
85
+ @parser.parse(raw, expected: stage, critique_candidate_ids: PROBE_CANDIDATE_IDS)
86
+ Result.new(candidate: candidate, stage: stage, status: :ok,
87
+ seconds: Process.clock_gettime(Process::CLOCK_MONOTONIC) - started_at)
88
+ rescue JSON::ParserError, ResultParser::SchemaError => e
89
+ Result.new(candidate: candidate, stage: stage, status: :invalid,
90
+ detail: "LLM returned invalid #{stage} result: #{e.message}")
91
+ rescue StandardError => e
92
+ Result.new(candidate: candidate, stage: stage, status: failure_status(candidate, e), detail: e.message)
93
+ end
94
+
95
+ # A per-minute limit is no reason to call the model broken: one retry
96
+ # after the provider's hint, then "could not verify".
97
+ def probe_with_one_retry(candidate, stage, prompt)
98
+ probe(candidate, stage, prompt)
99
+ rescue StandardError => e
100
+ raise unless LlmFailure.classify(e) == :rate_limit
101
+
102
+ delay = LlmFailure.retry_after_seconds(e.message) || LlmRouter::RATE_LIMIT_BASE_DELAY
103
+ @logger.warn("Model check: #{candidate} rate limited, retrying once in #{delay.round}s")
104
+ @sleeper.call(delay)
105
+ probe(candidate, stage, prompt)
106
+ end
107
+
108
+ def probe(candidate, stage, prompt)
109
+ request = LlmClient::Prompt.new(
110
+ stage: stage, system: prompt[:system_prompt], user: prompt[:user_prompt], temperature: 0,
111
+ schema: stage == 'critique' ? CritiqueOutputSchema : GenerateOutputSchema
112
+ )
113
+ key = @config.provider_api_keys(candidate.provider).first
114
+ @client.request(request, candidate: candidate, key: key, timeout: @config.llm_timeout.to_f).content
115
+ end
116
+
117
+ # missing — the provider has no such model; unverified — the provider
118
+ # could not answer right now (overload, quota, timeout); skipped — Ollama
119
+ # is not running where the check runs; failed — everything else.
120
+ def failure_status(candidate, error)
121
+ return :skipped if candidate.provider == 'ollama' && connection_failed?(error)
122
+
123
+ case LlmFailure.classify(error)
124
+ when :unavailable then :missing
125
+ when :rate_limit, :daily_quota, :overloaded, :timeout then :unverified
126
+ else :failed
127
+ end
128
+ end
129
+
130
+ def connection_failed?(error)
131
+ return true if error.is_a?(Errno::ECONNREFUSED) || error.is_a?(SocketError)
132
+
133
+ defined?(Faraday::ConnectionFailed) && error.is_a?(Faraday::ConnectionFailed)
134
+ end
135
+
136
+ def summary(results)
137
+ counts = results.group_by(&:status).transform_values(&:size)
138
+ passed = results.all? { |result| passing_statuses.include?(result.status) }
139
+ totals = counts.map { |status, count| "#{count} #{status}" }.join(', ')
140
+ @out.puts("Result: #{totals} -> #{passed ? 'PASSED' : 'FAILED'}")
141
+ passed ? 0 : 1
142
+ end
143
+
144
+ def passing_statuses
145
+ @strict ? STRICT_PASSING : PASSING
146
+ end
147
+ end
148
+ end