aireview 2.1.0 → 2.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -26,10 +26,21 @@ module Aireview
26
26
  'review_mode' => 'REVIEW_MODE',
27
27
  'llm_api_base' => 'LLM_API_BASE',
28
28
  'ollama_api_base' => 'OLLAMA_API_BASE',
29
- 'llm_http_proxy' => 'LLM_HTTP_PROXY'
29
+ 'llm_http_proxy' => 'LLM_HTTP_PROXY',
30
+ 'jev_api_key' => 'JEV_API_KEY'
31
+ }.freeze
32
+ PROVIDER_KEY_MAPPING = {
33
+ 'gemini' => 'GEMINI_API_KEY',
34
+ 'openai' => 'OPENAI_API_KEY',
35
+ 'anthropic' => 'ANTHROPIC_API_KEY',
36
+ 'openrouter' => 'OPENROUTER_API_KEY'
37
+ }.freeze
38
+ PROVIDER_KEYS_MAPPING = {
39
+ 'gemini' => 'GEMINI_API_KEYS',
40
+ 'openai' => 'OPENAI_API_KEYS',
41
+ 'anthropic' => 'ANTHROPIC_API_KEYS',
42
+ 'openrouter' => 'OPENROUTER_API_KEYS'
30
43
  }.freeze
31
- PROVIDER_KEY_MAPPING = {'gemini' => 'GEMINI_API_KEY'}.freeze
32
- PROVIDER_KEYS_MAPPING = {'gemini' => 'GEMINI_API_KEYS'}.freeze
33
44
  CONTEXT_ENV = {
34
45
  'max_diff_chars' => 'MAX_DIFF_CHARS',
35
46
  'max_mr_description_chars' => 'MAX_MR_DESCRIPTION_CHARS',
@@ -38,7 +49,8 @@ module Aireview
38
49
  }.freeze
39
50
  LLM_ENV = %w[
40
51
  LLM_PROVIDER LLM_TEMPERATURE LLM_TIMEOUT LLM_MAX_PROMPT_CHARS LLM_TIME_BUDGET LLM_OVERLOADED_QUARANTINE
41
- LLM_MODELS LLM_CRITIQUE_RANK LLM_CRITIQUE_ALLOW_WEAKER
52
+ LLM_MODELS LLM_CRITIQUE_RANK LLM_CRITIQUE_ALLOW_WEAKER LLM_CRITIQUE_ENGINE
53
+ LLM_JEV_SHADOW LLM_JEV_MODEL LLM_JEV_FALLBACK LLM_JEV_KEEP_ABOVE
42
54
  ].freeze
43
55
  LLM_STAGE_ENV_SUFFIXES = %w[PROVIDER MODEL TEMPERATURE MAX_PROMPT_CHARS FALLBACK_MODEL START].freeze
44
56
  IMAGE_DEFAULTS_ENV = 'AIREVIEW_DEFAULTS'
@@ -141,8 +153,21 @@ module Aireview
141
153
  'time_budget' => parse_integer(env['LLM_TIME_BUDGET'], 'LLM_TIME_BUDGET'),
142
154
  'overloaded_quarantine' => parse_integer(env['LLM_OVERLOADED_QUARANTINE'], 'LLM_OVERLOADED_QUARANTINE'),
143
155
  'generate' => llm_stage_env_config(env, 'GENERATE'),
144
- 'critique' => llm_stage_env_config(env, 'CRITIQUE')
145
- }.compact.reject { |key, value| %w[generate critique].include?(key) && value.empty? }
156
+ 'critique' => llm_stage_env_config(env, 'CRITIQUE'),
157
+ 'jev' => jev_env_config(env)
158
+ }.compact.reject { |key, value| %w[generate critique jev].include?(key) && value.empty? }
159
+ end
160
+
161
+ # LLM_JEV_SHADOW=true|false, LLM_JEV_MODEL — a pinned Jev version,
162
+ # LLM_JEV_FALLBACK=model|fail, LLM_JEV_KEEP_ABOVE — the keep threshold.
163
+ def jev_env_config(env)
164
+ {
165
+ 'shadow' => parse_boolean(env['LLM_JEV_SHADOW'], 'LLM_JEV_SHADOW'),
166
+ 'model' => Aireview::Utils.presence(env['LLM_JEV_MODEL']),
167
+ 'fallback' => Aireview::Utils.presence(env['LLM_JEV_FALLBACK']),
168
+ # Validated by ConfigJev: a typo must fail, not fall back to the default.
169
+ 'keep_above' => Aireview::Utils.presence(env['LLM_JEV_KEEP_ABOVE'])
170
+ }.compact
146
171
  end
147
172
 
148
173
  def llm_stage_env_config(env, stage)
@@ -176,6 +201,7 @@ module Aireview
176
201
  'generate' => {'start' => env['LLM_GENERATE_START']}.compact,
177
202
  'critique' => {
178
203
  'start' => env['LLM_CRITIQUE_START'],
204
+ 'engine' => Aireview::Utils.presence(env['LLM_CRITIQUE_ENGINE']),
179
205
  'rank' => env['LLM_CRITIQUE_RANK'],
180
206
  'allow_weaker' => parse_boolean(env['LLM_CRITIQUE_ALLOW_WEAKER'], 'LLM_CRITIQUE_ALLOW_WEAKER')
181
207
  }.compact
@@ -22,8 +22,9 @@ module Aireview
22
22
  CANDIDATES_RESERVE_CHARS = 4_500
23
23
 
24
24
  # The context of one run: both stages get the same MR, Jira and diff,
25
- # truncated once for the tightest of the stages.
26
- Context = Struct.new(:user_prompt, :diff_text, :coverage, :sizes, keyword_init: true)
25
+ # truncated once for the tightest of the stages. sections — the MR and
26
+ # Jira part without the diff, for Jev.
27
+ Context = Struct.new(:user_prompt, :sections, :diff_text, :coverage, :sizes, keyword_init: true)
27
28
 
28
29
  def initialize(config:, logger: Logger.new($stderr))
29
30
  @config = config
@@ -49,7 +50,8 @@ module Aireview
49
50
 
50
51
  sizes = context_sizes(fixed: fixed, packed: packed, budget: budget, diff_budget: diff_budget, critique: critique)
51
52
  log_sizes(sizes)
52
- Context.new(user_prompt: fixed + packed.text, diff_text: packed.text, coverage: coverage, sizes: sizes)
53
+ Context.new(user_prompt: fixed + packed.text, sections: sections.join("\n\n"), diff_text: packed.text,
54
+ coverage: coverage, sizes: sizes)
53
55
  end
54
56
 
55
57
  def build_generate_prompt(context)
@@ -88,12 +90,17 @@ module Aireview
88
90
  {system_prompt: system, user_prompt: user}
89
91
  end
90
92
 
93
+ def scrub_text(text)
94
+ @secret_scrubber.scrub_text(text.to_s)
95
+ end
96
+
91
97
  private
92
98
 
93
99
  # The minimum over the stages: the context is one per run, so it must fit
94
- # into each of them together with its system prompt and reserve.
100
+ # into each of them together with its system prompt and reserve. Only the
101
+ # stages that go to an LLM count: Jev as the critic has limits of its own.
95
102
  def context_budget(critique:)
96
- stages = critique ? STAGES : ['generate']
103
+ stages = @config.llm_stages(critique: critique)
97
104
  budgets = stages.to_h { |stage| [stage, stage_budget(stage)] }
98
105
  stage, budget = budgets.min_by { |_, value| value }
99
106
  return budget if budget.positive?
@@ -153,7 +160,7 @@ module Aireview
153
160
  end
154
161
 
155
162
  def context_sizes(fixed:, packed:, budget:, diff_budget:, critique:)
156
- stages = critique ? STAGES : ['generate']
163
+ stages = @config.llm_stages(critique: critique)
157
164
  {
158
165
  context_budget: budget,
159
166
  diff_budget: diff_budget,
@@ -187,10 +194,6 @@ module Aireview
187
194
  Aireview::Utils.presence(scrubbed) || '(empty)'
188
195
  end
189
196
 
190
- def scrub_text(text)
191
- @secret_scrubber.scrub_text(text.to_s)
192
- end
193
-
194
197
  def language_name(code)
195
198
  LANGUAGE_NAMES.fetch(code.to_s, code.to_s)
196
199
  end
@@ -0,0 +1,104 @@
1
+ # frozen_string_literal: true
2
+ require 'json'
3
+ require 'logger'
4
+ require_relative 'utils'
5
+ require_relative 'jev_critic'
6
+
7
+ module Aireview
8
+ # The prompts of a run without calling a model, and everything --dry-run
9
+ # shows. The review key is computed from them too (ReviewMarker): the LLM
10
+ # Critique prompt is built only when an LLM Critique can run, the Jev
11
+ # question templates only when Jev decides.
12
+ class DryRunPrompts
13
+ CANDIDATES_JSON = '[{"id":"C1","file":"path/from/diff.rb","line":1,' \
14
+ '"quoted_code":"...","problem":"...","why":"...","suggestion":"...",' \
15
+ '"category":"bug","severity":"major"}]'
16
+
17
+ def initialize(config:, context_builder:, logger: Logger.new($stderr))
18
+ @config = config
19
+ @context_builder = context_builder
20
+ @logger = logger
21
+ end
22
+
23
+ def build(merge_request:, changes:, jira_issue: nil, critique: true)
24
+ @config.require_models!(critique: critique)
25
+ stages = @config.llm_stages(critique: critique)
26
+ context = @context_builder.prepare(
27
+ merge_request: merge_request,
28
+ changes: changes,
29
+ jira_issue: jira_issue,
30
+ critique: critique
31
+ )
32
+
33
+ prompts(context, stages, critique).merge(settings(context, stages, critique))
34
+ end
35
+
36
+ private
37
+
38
+ def prompts(context, stages, critique)
39
+ jev = @config.jev_critique?(critique: critique)
40
+ critique_prompt = if stages.include?('critique')
41
+ @context_builder.build_critique_prompt(context, candidates_json: CANDIDATES_JSON)
42
+ end
43
+ {
44
+ generate_prompt: @context_builder.build_generate_prompt(context),
45
+ critique_prompt: critique_prompt,
46
+ jev_questions: jev ? JevCritic.decision_templates : nil,
47
+ jev_critique: jev ? jev_critique_settings(context) : nil,
48
+ jev_shadow: critique && !jev ? jev_shadow_settings : nil
49
+ }
50
+ end
51
+
52
+ def settings(context, stages, critique)
53
+ llm_critique = stages.include?('critique')
54
+ {
55
+ generate_model: @config.generate_model,
56
+ generate_temperature: @config.generate_temperature,
57
+ critique_model: @config.critique_model,
58
+ critique_temperature: @config.critique_temperature,
59
+ generate_fallbacks: @config.fallback_names('generate'),
60
+ critique_fallbacks: llm_critique ? @config.fallback_names('critique') : [],
61
+ sources: setting_sources(stages),
62
+ config_paths: @config.layer_paths,
63
+ warnings: @config.warnings,
64
+ critique_rule: llm_critique ? @config.routing.rule : nil,
65
+ api_keys: @config.api_key_counts(stages),
66
+ time_budget: @config.llm_time_budget,
67
+ overloaded_quarantine: @config.overloaded_quarantine,
68
+ coverage: context.coverage,
69
+ sizes: context.sizes
70
+ }
71
+ end
72
+
73
+ # The Jev request as it would go for the stub candidate; nothing is sent,
74
+ # so the critic needs no client. Only whether the key is set: its value
75
+ # never leaves.
76
+ def jev_critique_settings(context)
77
+ critic = JevCritic.new(client: nil, thresholds: @config.jev_thresholds,
78
+ review_instructions: @config.review_instructions,
79
+ scrub: @context_builder.method(:scrub_text), logger: @logger)
80
+ request = critic.preview(context: context, candidates: JSON.parse(CANDIDATES_JSON))
81
+ {model: @config.jev_model, key: Aireview::Utils.present?(@config.jev_api_key),
82
+ thresholds: @config.jev_thresholds, fallback: @config.jev_fallback,
83
+ state: request.state, questions: request.questions}
84
+ end
85
+
86
+ # Only whether the key is set: its value never leaves.
87
+ def jev_shadow_settings
88
+ return nil unless @config.jev_shadow?
89
+
90
+ {model: @config.jev_model, key: Aireview::Utils.present?(@config.jev_api_key), thresholds: @config.jev_thresholds}
91
+ end
92
+
93
+ # Where the model, provider and reserves of a stage came from, for --dry-run.
94
+ def setting_sources(stages)
95
+ stages.to_h do |stage|
96
+ [stage.to_sym, {
97
+ model: @config.stage_model_source(stage),
98
+ provider: @config.stage_provider_source(stage),
99
+ fallbacks: @config.stage_fallbacks_source(stage)
100
+ }]
101
+ end
102
+ end
103
+ end
104
+ end
@@ -1,4 +1,5 @@
1
1
  # frozen_string_literal: true
2
+ require 'json'
2
3
 
3
4
  module Aireview
4
5
  # The --dry-run output: settings, the context summary and the prompts of both stages.
@@ -19,32 +20,63 @@ module Aireview
19
20
  @out.puts
20
21
  @out.puts('=== GENERATE USER PROMPT ===')
21
22
  @out.puts(dry_run.dig(:generate_prompt, :user_prompt))
22
- return unless dry_run[:critique_prompt]
23
+ render_critique_prompt(dry_run[:critique_prompt]) if dry_run[:critique_prompt]
24
+ render_jev_request(dry_run[:jev_critique]) if dry_run[:jev_critique]
25
+ end
26
+
27
+ private
23
28
 
29
+ def render_critique_prompt(prompt)
24
30
  @out.puts
25
31
  @out.puts('=== CRITIQUE SYSTEM PROMPT ===')
26
- @out.puts(dry_run.dig(:critique_prompt, :system_prompt))
32
+ @out.puts(prompt[:system_prompt])
27
33
  @out.puts
28
34
  @out.puts('=== CRITIQUE USER PROMPT ===')
29
- @out.puts(dry_run.dig(:critique_prompt, :user_prompt))
35
+ @out.puts(prompt[:user_prompt])
30
36
  end
31
37
 
32
- private
38
+ # The request Jev would get for the stub candidate.
39
+ def render_jev_request(jev)
40
+ @out.puts
41
+ @out.puts('=== JEV STATE ===')
42
+ @out.puts(JSON.pretty_generate(jev[:state]))
43
+ @out.puts
44
+ @out.puts('=== JEV QUESTIONS ===')
45
+ @out.puts(JSON.pretty_generate(jev[:questions]))
46
+ end
33
47
 
48
+ # With Jev as the engine the LLM Critique line is its fallback.
34
49
  def render_settings(dry_run)
35
50
  @out.puts('=== LLM SETTINGS ===')
36
51
  render_config_paths(dry_run[:config_paths])
37
52
  render_stage(dry_run, :generate)
53
+ render_jev_critique(dry_run[:jev_critique])
38
54
  if dry_run[:critique_prompt]
39
- render_stage(dry_run, :critique)
40
- else
55
+ render_stage(dry_run, :critique, title: dry_run[:jev_critique] ? 'Critique fallback' : 'Critique')
56
+ elsif !dry_run[:jev_critique]
41
57
  @out.puts('Critique: disabled')
42
58
  end
43
59
  @out.puts("Critique rule: #{dry_run[:critique_rule]}") if dry_run[:critique_rule]
60
+ render_jev_shadow(dry_run[:jev_shadow])
44
61
  render_reserves(dry_run)
45
62
  list('warnings', dry_run[:warnings], separator: "\n ")
46
63
  end
47
64
 
65
+ def render_jev_critique(jev)
66
+ return unless jev
67
+
68
+ thresholds = jev[:thresholds].map { |name, value| "#{name}=#{value}" }.join(' ')
69
+ @out.puts("Critique: jev #{jev[:model]} (key #{jev[:key] ? 'set' : 'missing'}; " \
70
+ "fallback: #{jev[:fallback]}; #{thresholds})")
71
+ end
72
+
73
+ def render_jev_shadow(jev)
74
+ return unless jev
75
+
76
+ thresholds = jev[:thresholds].map { |name, value| "#{name}=#{value}" }.join(' ')
77
+ @out.puts("Jev shadow: #{jev[:model]} (key #{jev[:key] ? 'set' : 'missing'}; log only, #{thresholds})")
78
+ end
79
+
48
80
  def render_config_paths(paths)
49
81
  return if paths.nil? || paths.empty?
50
82
 
@@ -53,9 +85,9 @@ module Aireview
53
85
 
54
86
  # The source of every setting is the layer it came from: built-in, image
55
87
  # defaults, .aireview.yml, env or cli.
56
- def render_stage(dry_run, stage)
88
+ def render_stage(dry_run, stage, title: stage.capitalize)
57
89
  sources = dry_run.dig(:sources, stage) || {}
58
- @out.puts("#{stage.capitalize}: #{dry_run[:"#{stage}_model"]} " \
90
+ @out.puts("#{title}: #{dry_run[:"#{stage}_model"]} " \
59
91
  "temperature=#{dry_run[:"#{stage}_temperature"]}#{origin(sources, :model, :provider)}")
60
92
  fallbacks = dry_run[:"#{stage}_fallbacks"]
61
93
  list('fallbacks', fallbacks, separator: ' -> ', suffix: origin(sources, :fallbacks))
@@ -14,4 +14,15 @@ module Aireview
14
14
  class RouteExhaustedError < ApiError; end
15
15
  class ContextBudgetError < Error; end
16
16
  class HelpRequested < Error; end
17
+
18
+ # A Jev request failed: network, HTTP status, an answer of the wrong shape.
19
+ # status is the HTTP status when the server answered.
20
+ class JevError < Error
21
+ attr_reader :status
22
+
23
+ def initialize(message, status: nil)
24
+ super(message)
25
+ @status = status
26
+ end
27
+ end
17
28
  end
@@ -0,0 +1,129 @@
1
+ # frozen_string_literal: true
2
+ require 'json'
3
+ require 'logger'
4
+ require_relative 'errors'
5
+ require_relative 'utils'
6
+
7
+ module Aireview
8
+ # One evaluation request to Jev (TypeSafe, POST /v1/systemone): a state and
9
+ # named questions in, one answer per question out. RubyLLM does not know
10
+ # Jev and the router is built for text models, so Jev has its own small
11
+ # client: one retry on 429/529, no key rotation, no reserves. A failure is
12
+ # a JevError; the caller decides whether it matters.
13
+ class JevClient
14
+ API_URL = 'https://api.typesafe.ai/v1/'
15
+ OPEN_TIMEOUT = 5
16
+ RETRY_STATUSES = [429, 529].freeze
17
+ RETRY_DELAY = 2
18
+ # retry-after beyond this is not worth waiting for in a review job.
19
+ MAX_RETRY_DELAY = 10
20
+
21
+ Result = Struct.new(:model, :answers, :usage, keyword_init: true)
22
+
23
+ # Jev goes out the same way as the LLM providers, through
24
+ # LLM_HTTP_PROXY when it is set. dependencies — connection: and sleeper:
25
+ # for tests.
26
+ def initialize(config:, logger: Logger.new($stderr), **dependencies)
27
+ require 'faraday'
28
+
29
+ raise ConfigError, 'JEV_API_KEY is required' if Aireview::Utils.blank?(config.jev_api_key)
30
+
31
+ @api_key = config.jev_api_key
32
+ @model = config.jev_model
33
+ @logger = logger
34
+ @connection = dependencies[:connection] || build_connection(config.jev_timeout, config.llm_http_proxy)
35
+ @sleeper = dependencies[:sleeper] || ->(seconds) { sleep(seconds) }
36
+ end
37
+
38
+ # questions — {key => question}; returns Result with answers under the
39
+ # same keys, each checked against the type of its question.
40
+ def evaluate(state:, questions:)
41
+ body = JSON.generate(model: @model, state: state, questions: questions)
42
+ @logger.info("Jev request started (model=#{@model}, questions=#{questions.size})")
43
+ response = post_with_one_retry(body)
44
+ result = parse(response, questions)
45
+ log_completed(result)
46
+ result
47
+ end
48
+
49
+ private
50
+
51
+ def build_connection(timeout, proxy)
52
+ Faraday.new(url: API_URL, proxy: Aireview::Utils.presence(proxy)) do |builder|
53
+ builder.options.open_timeout = OPEN_TIMEOUT
54
+ builder.options.timeout = timeout
55
+ builder.adapter Faraday.default_adapter
56
+ end
57
+ end
58
+
59
+ def post_with_one_retry(body)
60
+ response = post(body)
61
+ return response unless RETRY_STATUSES.include?(response.status.to_i)
62
+
63
+ delay = retry_delay(response)
64
+ @logger.warn("Jev answered #{response.status}, retrying once in #{delay}s")
65
+ @sleeper.call(delay)
66
+ post(body)
67
+ end
68
+
69
+ def post(body)
70
+ @connection.post('systemone') do |request|
71
+ request.headers['Authorization'] = "Bearer #{@api_key}"
72
+ request.headers['Content-Type'] = 'application/json'
73
+ request.body = body
74
+ end
75
+ rescue Faraday::Error => e
76
+ raise JevError, "Jev request failed: #{e.class}: #{e.message}"
77
+ end
78
+
79
+ def retry_delay(response)
80
+ seconds = Integer(response.headers['retry-after'].to_s, exception: false)
81
+ seconds&.positive? ? [seconds, MAX_RETRY_DELAY].min : RETRY_DELAY
82
+ end
83
+
84
+ def parse(response, questions)
85
+ status = response.status.to_i
86
+ unless status.between?(200, 299)
87
+ raise JevError.new("Jev API error #{status}: #{response.body.to_s[0, 500]}", status: status)
88
+ end
89
+
90
+ payload = JSON.parse(response.body.to_s)
91
+ answers = payload['answers'] if payload.is_a?(Hash)
92
+ raise JevError, 'Jev returned no answers' unless answers.is_a?(Hash)
93
+
94
+ questions.each { |key, question| check_answer!(key, question, answers[key.to_s]) }
95
+ Result.new(model: payload['model'], answers: answers, usage: payload['usage'])
96
+ rescue JSON::ParserError
97
+ raise JevError, "Jev returned invalid JSON: #{response.body.to_s[0, 500]}"
98
+ end
99
+
100
+ def check_answer!(key, question, answer)
101
+ type = question[:type] || question['type']
102
+ return if answer.is_a?(Hash) && answer['type'] == type && valid_value?(type, answer)
103
+
104
+ raise JevError, "Jev returned an invalid #{type} answer for #{key}: #{answer.inspect}"
105
+ end
106
+
107
+ def valid_value?(type, answer)
108
+ case type
109
+ when 'noul' then probability?(answer['noul'])
110
+ when 'choice' then answer['choice'].is_a?(String) && probability?(answer['confidence'])
111
+ else true
112
+ end
113
+ end
114
+
115
+ def probability?(value)
116
+ value.is_a?(Numeric) && value.between?(0, 1)
117
+ end
118
+
119
+ # The answering version is logged: an alias resolves on the server, and
120
+ # a pinned version that answers as another one breaks the thresholds.
121
+ def log_completed(result)
122
+ tokens = result.usage.is_a?(Hash) ? ", tokens: input=#{result.usage['input_tokens']}" : ''
123
+ @logger.info("Jev request completed (model=#{result.model}#{tokens})")
124
+ return if result.model == @model || !@model.match?(/\Ajev-\d/)
125
+
126
+ @logger.warn("Jev answered as #{result.model.inspect}, not the requested #{@model}")
127
+ end
128
+ end
129
+ end