aireview 2.0.0 → 2.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,137 @@
1
+ # frozen_string_literal: true
2
+ require_relative 'errors'
3
+ require_relative 'stages'
4
+ require_relative 'utils'
5
+
6
+ module Aireview
7
+ # The critique engine and the Jev (TypeSafe) settings. llm.critique.engine
8
+ # picks who checks the candidates: an LLM (model, the default) or Jev, a
9
+ # fast classifier that decides keep/reject but cannot refine a finding.
10
+ # llm.jev.fallback says what happens when Jev cannot decide: the LLM
11
+ # Critique takes over (model) or the run fails (fail). llm.jev.shadow runs
12
+ # Jev next to the LLM Critique for the log only.
13
+ module ConfigJev
14
+ CRITIQUE_ENGINES = %w[model jev].freeze
15
+ JEV_FALLBACKS = %w[model fail].freeze
16
+ # A pinned version, not an alias: thresholds are tuned against one
17
+ # version, and an alias moves when a release ships.
18
+ DEFAULT_JEV_MODEL = 'jev-1.13.0'
19
+ JEV_ALIASES = %w[jev-latest jev-preview].freeze
20
+ DEFAULT_JEV_TIMEOUT = 10
21
+ # Provisional values until real thresholds are chosen from the shadow
22
+ # logs; the logs carry the raw probabilities for that.
23
+ JEV_THRESHOLD_DEFAULTS = {
24
+ 'keep_above' => 0.5,
25
+ 'enough_context' => 0.5,
26
+ 'version_claim' => 0.5,
27
+ 'duplicate' => 0.5
28
+ }.freeze
29
+
30
+ def critique_engine
31
+ one_of!(dig('llm', 'critique', 'engine') || 'model', CRITIQUE_ENGINES, 'llm.critique.engine')
32
+ end
33
+
34
+ def jev_fallback
35
+ one_of!(dig('llm', 'jev', 'fallback') || 'model', JEV_FALLBACKS, 'llm.jev.fallback')
36
+ end
37
+
38
+ # Jev decides in this run: the engine is jev and the run has a critique
39
+ # at all (--no-critique switches off every engine, as the critique:
40
+ # argument or as a CLI override of the config).
41
+ def jev_critique?(critique: true)
42
+ critique && !critique_disabled? && critique_engine == 'jev'
43
+ end
44
+
45
+ # The stages that need an LLM in a run. Jev with fallback: model still
46
+ # needs a ready LLM Critique; with fallback: fail it needs none.
47
+ def llm_stages(critique: true)
48
+ return ['generate'] unless critique && !critique_disabled?
49
+ return ['generate'] if jev_critique? && jev_fallback == 'fail'
50
+
51
+ STAGES
52
+ end
53
+
54
+ def critique_disabled?
55
+ @data['critique_disabled'] == true
56
+ end
57
+
58
+ def jev_shadow?
59
+ value = dig('llm', 'jev', 'shadow')
60
+ return false if value.nil?
61
+ return value if [true, false].include?(value)
62
+
63
+ raise ConfigError, "llm.jev.shadow must be true or false, got #{value.inspect}"
64
+ end
65
+
66
+ def jev_model
67
+ Aireview::Utils.presence(dig('llm', 'jev', 'model')) || DEFAULT_JEV_MODEL
68
+ end
69
+
70
+ def jev_api_key
71
+ @data['jev_api_key']
72
+ end
73
+
74
+ def jev_timeout
75
+ positive_integer!(dig('llm', 'jev', 'timeout') || DEFAULT_JEV_TIMEOUT, 'llm.jev.timeout')
76
+ end
77
+
78
+ def jev_thresholds
79
+ JEV_THRESHOLD_DEFAULTS.to_h do |name, default|
80
+ value = dig('llm', 'jev', name)
81
+ [name.to_sym, value.nil? ? default : probability!(value, "llm.jev.#{name}")]
82
+ end
83
+ end
84
+
85
+ # Everything besides the questions that changes a Jev decision; nil when
86
+ # Jev does not decide. The fallback is part of it: it decides what
87
+ # happens to the candidates Jev could not judge.
88
+ def jev_signature
89
+ return nil unless critique_engine == 'jev'
90
+
91
+ ['jev', jev_model, *jev_thresholds.values, jev_fallback]
92
+ end
93
+
94
+ # Jev as the critic needs a key and a pinned version: the thresholds are
95
+ # tuned against one version, and falling back to the LLM silently would
96
+ # hide a broken setup.
97
+ def require_jev!(critique: true)
98
+ return unless jev_critique?(critique: critique)
99
+ raise ConfigError, 'llm.critique.engine is jev, but JEV_API_KEY is not set' if Aireview::Utils.blank?(jev_api_key)
100
+ return unless JEV_ALIASES.include?(jev_model)
101
+
102
+ raise ConfigError, "llm.jev.model #{jev_model.inspect} is an alias; with llm.critique.engine: jev " \
103
+ 'pin a version such as jev-1.13.0'
104
+ end
105
+
106
+ def jev_warnings
107
+ return [] unless jev_shadow?
108
+ return ['llm.jev.shadow is ignored: Jev already decides as the critique engine'] if critique_engine == 'jev'
109
+
110
+ warnings = []
111
+ if Aireview::Utils.blank?(jev_api_key)
112
+ warnings << 'llm.jev.shadow is on, but JEV_API_KEY is not set: Jev is skipped'
113
+ end
114
+ if JEV_ALIASES.include?(jev_model)
115
+ warnings << "llm.jev.model #{jev_model.inspect} is an alias: its answers change when TypeSafe ships " \
116
+ 'a release; pin a version such as jev-1.13.0'
117
+ end
118
+ warnings
119
+ end
120
+
121
+ private
122
+
123
+ def one_of!(value, allowed, name)
124
+ value = value.to_s
125
+ return value if allowed.include?(value)
126
+
127
+ raise ConfigError, "#{name} must be one of #{allowed.join(', ')}, got #{value.inspect}"
128
+ end
129
+
130
+ def probability!(value, name)
131
+ number = Float(value, exception: false) if value.is_a?(Numeric) || value.is_a?(String)
132
+ return number if number&.between?(0, 1)
133
+
134
+ raise ConfigError, "#{name} must be a number from 0 to 1, got #{value.inspect}"
135
+ end
136
+ end
137
+ end
@@ -65,7 +65,7 @@ module Aireview
65
65
 
66
66
  # Configuration and plan warnings; the CLI and --dry-run print them.
67
67
  def warnings
68
- STAGES.flat_map { |stage| stage_provider_warnings(stage) } + routing.warnings
68
+ llm_stages.flat_map { |stage| stage_provider_warnings(stage) } + routing.warnings + jev_warnings
69
69
  end
70
70
 
71
71
  # The paths of the file layers, for --dry-run.
@@ -26,10 +26,21 @@ module Aireview
26
26
  'review_mode' => 'REVIEW_MODE',
27
27
  'llm_api_base' => 'LLM_API_BASE',
28
28
  'ollama_api_base' => 'OLLAMA_API_BASE',
29
- 'llm_http_proxy' => 'LLM_HTTP_PROXY'
29
+ 'llm_http_proxy' => 'LLM_HTTP_PROXY',
30
+ 'jev_api_key' => 'JEV_API_KEY'
31
+ }.freeze
32
+ PROVIDER_KEY_MAPPING = {
33
+ 'gemini' => 'GEMINI_API_KEY',
34
+ 'openai' => 'OPENAI_API_KEY',
35
+ 'anthropic' => 'ANTHROPIC_API_KEY',
36
+ 'openrouter' => 'OPENROUTER_API_KEY'
37
+ }.freeze
38
+ PROVIDER_KEYS_MAPPING = {
39
+ 'gemini' => 'GEMINI_API_KEYS',
40
+ 'openai' => 'OPENAI_API_KEYS',
41
+ 'anthropic' => 'ANTHROPIC_API_KEYS',
42
+ 'openrouter' => 'OPENROUTER_API_KEYS'
30
43
  }.freeze
31
- PROVIDER_KEY_MAPPING = {'gemini' => 'GEMINI_API_KEY'}.freeze
32
- PROVIDER_KEYS_MAPPING = {'gemini' => 'GEMINI_API_KEYS'}.freeze
33
44
  CONTEXT_ENV = {
34
45
  'max_diff_chars' => 'MAX_DIFF_CHARS',
35
46
  'max_mr_description_chars' => 'MAX_MR_DESCRIPTION_CHARS',
@@ -38,7 +49,8 @@ module Aireview
38
49
  }.freeze
39
50
  LLM_ENV = %w[
40
51
  LLM_PROVIDER LLM_TEMPERATURE LLM_TIMEOUT LLM_MAX_PROMPT_CHARS LLM_TIME_BUDGET LLM_OVERLOADED_QUARANTINE
41
- LLM_MODELS LLM_CRITIQUE_RANK LLM_CRITIQUE_ALLOW_WEAKER
52
+ LLM_MODELS LLM_CRITIQUE_RANK LLM_CRITIQUE_ALLOW_WEAKER LLM_CRITIQUE_ENGINE
53
+ LLM_JEV_SHADOW LLM_JEV_MODEL LLM_JEV_FALLBACK LLM_JEV_KEEP_ABOVE
42
54
  ].freeze
43
55
  LLM_STAGE_ENV_SUFFIXES = %w[PROVIDER MODEL TEMPERATURE MAX_PROMPT_CHARS FALLBACK_MODEL START].freeze
44
56
  IMAGE_DEFAULTS_ENV = 'AIREVIEW_DEFAULTS'
@@ -141,8 +153,21 @@ module Aireview
141
153
  'time_budget' => parse_integer(env['LLM_TIME_BUDGET'], 'LLM_TIME_BUDGET'),
142
154
  'overloaded_quarantine' => parse_integer(env['LLM_OVERLOADED_QUARANTINE'], 'LLM_OVERLOADED_QUARANTINE'),
143
155
  'generate' => llm_stage_env_config(env, 'GENERATE'),
144
- 'critique' => llm_stage_env_config(env, 'CRITIQUE')
145
- }.compact.reject { |key, value| %w[generate critique].include?(key) && value.empty? }
156
+ 'critique' => llm_stage_env_config(env, 'CRITIQUE'),
157
+ 'jev' => jev_env_config(env)
158
+ }.compact.reject { |key, value| %w[generate critique jev].include?(key) && value.empty? }
159
+ end
160
+
161
+ # LLM_JEV_SHADOW=true|false, LLM_JEV_MODEL — a pinned Jev version,
162
+ # LLM_JEV_FALLBACK=model|fail, LLM_JEV_KEEP_ABOVE — the keep threshold.
163
+ def jev_env_config(env)
164
+ {
165
+ 'shadow' => parse_boolean(env['LLM_JEV_SHADOW'], 'LLM_JEV_SHADOW'),
166
+ 'model' => Aireview::Utils.presence(env['LLM_JEV_MODEL']),
167
+ 'fallback' => Aireview::Utils.presence(env['LLM_JEV_FALLBACK']),
168
+ # Validated by ConfigJev: a typo must fail, not fall back to the default.
169
+ 'keep_above' => Aireview::Utils.presence(env['LLM_JEV_KEEP_ABOVE'])
170
+ }.compact
146
171
  end
147
172
 
148
173
  def llm_stage_env_config(env, stage)
@@ -176,6 +201,7 @@ module Aireview
176
201
  'generate' => {'start' => env['LLM_GENERATE_START']}.compact,
177
202
  'critique' => {
178
203
  'start' => env['LLM_CRITIQUE_START'],
204
+ 'engine' => Aireview::Utils.presence(env['LLM_CRITIQUE_ENGINE']),
179
205
  'rank' => env['LLM_CRITIQUE_RANK'],
180
206
  'allow_weaker' => parse_boolean(env['LLM_CRITIQUE_ALLOW_WEAKER'], 'LLM_CRITIQUE_ALLOW_WEAKER')
181
207
  }.compact
@@ -22,8 +22,9 @@ module Aireview
22
22
  CANDIDATES_RESERVE_CHARS = 4_500
23
23
 
24
24
  # The context of one run: both stages get the same MR, Jira and diff,
25
- # truncated once for the tightest of the stages.
26
- Context = Struct.new(:user_prompt, :diff_text, :coverage, :sizes, keyword_init: true)
25
+ # truncated once for the tightest of the stages. sections — the MR and
26
+ # Jira part without the diff, for Jev.
27
+ Context = Struct.new(:user_prompt, :sections, :diff_text, :coverage, :sizes, keyword_init: true)
27
28
 
28
29
  def initialize(config:, logger: Logger.new($stderr))
29
30
  @config = config
@@ -49,7 +50,8 @@ module Aireview
49
50
 
50
51
  sizes = context_sizes(fixed: fixed, packed: packed, budget: budget, diff_budget: diff_budget, critique: critique)
51
52
  log_sizes(sizes)
52
- Context.new(user_prompt: fixed + packed.text, diff_text: packed.text, coverage: coverage, sizes: sizes)
53
+ Context.new(user_prompt: fixed + packed.text, sections: sections.join("\n\n"), diff_text: packed.text,
54
+ coverage: coverage, sizes: sizes)
53
55
  end
54
56
 
55
57
  def build_generate_prompt(context)
@@ -88,12 +90,17 @@ module Aireview
88
90
  {system_prompt: system, user_prompt: user}
89
91
  end
90
92
 
93
+ def scrub_text(text)
94
+ @secret_scrubber.scrub_text(text.to_s)
95
+ end
96
+
91
97
  private
92
98
 
93
99
  # The minimum over the stages: the context is one per run, so it must fit
94
- # into each of them together with its system prompt and reserve.
100
+ # into each of them together with its system prompt and reserve. Only the
101
+ # stages that go to an LLM count: Jev as the critic has limits of its own.
95
102
  def context_budget(critique:)
96
- stages = critique ? STAGES : ['generate']
103
+ stages = @config.llm_stages(critique: critique)
97
104
  budgets = stages.to_h { |stage| [stage, stage_budget(stage)] }
98
105
  stage, budget = budgets.min_by { |_, value| value }
99
106
  return budget if budget.positive?
@@ -153,7 +160,7 @@ module Aireview
153
160
  end
154
161
 
155
162
  def context_sizes(fixed:, packed:, budget:, diff_budget:, critique:)
156
- stages = critique ? STAGES : ['generate']
163
+ stages = @config.llm_stages(critique: critique)
157
164
  {
158
165
  context_budget: budget,
159
166
  diff_budget: diff_budget,
@@ -187,10 +194,6 @@ module Aireview
187
194
  Aireview::Utils.presence(scrubbed) || '(empty)'
188
195
  end
189
196
 
190
- def scrub_text(text)
191
- @secret_scrubber.scrub_text(text.to_s)
192
- end
193
-
194
197
  def language_name(code)
195
198
  LANGUAGE_NAMES.fetch(code.to_s, code.to_s)
196
199
  end
@@ -0,0 +1,104 @@
1
+ # frozen_string_literal: true
2
+ require 'json'
3
+ require 'logger'
4
+ require_relative 'utils'
5
+ require_relative 'jev_critic'
6
+
7
+ module Aireview
8
+ # The prompts of a run without calling a model, and everything --dry-run
9
+ # shows. The review key is computed from them too (ReviewMarker): the LLM
10
+ # Critique prompt is built only when an LLM Critique can run, the Jev
11
+ # question templates only when Jev decides.
12
+ class DryRunPrompts
13
+ CANDIDATES_JSON = '[{"id":"C1","file":"path/from/diff.rb","line":1,' \
14
+ '"quoted_code":"...","problem":"...","why":"...","suggestion":"...",' \
15
+ '"category":"bug","severity":"major"}]'
16
+
17
+ def initialize(config:, context_builder:, logger: Logger.new($stderr))
18
+ @config = config
19
+ @context_builder = context_builder
20
+ @logger = logger
21
+ end
22
+
23
+ def build(merge_request:, changes:, jira_issue: nil, critique: true)
24
+ @config.require_models!(critique: critique)
25
+ stages = @config.llm_stages(critique: critique)
26
+ context = @context_builder.prepare(
27
+ merge_request: merge_request,
28
+ changes: changes,
29
+ jira_issue: jira_issue,
30
+ critique: critique
31
+ )
32
+
33
+ prompts(context, stages, critique).merge(settings(context, stages, critique))
34
+ end
35
+
36
+ private
37
+
38
+ def prompts(context, stages, critique)
39
+ jev = @config.jev_critique?(critique: critique)
40
+ critique_prompt = if stages.include?('critique')
41
+ @context_builder.build_critique_prompt(context, candidates_json: CANDIDATES_JSON)
42
+ end
43
+ {
44
+ generate_prompt: @context_builder.build_generate_prompt(context),
45
+ critique_prompt: critique_prompt,
46
+ jev_questions: jev ? JevCritic.decision_templates : nil,
47
+ jev_critique: jev ? jev_critique_settings(context) : nil,
48
+ jev_shadow: critique && !jev ? jev_shadow_settings : nil
49
+ }
50
+ end
51
+
52
+ def settings(context, stages, critique)
53
+ llm_critique = stages.include?('critique')
54
+ {
55
+ generate_model: @config.generate_model,
56
+ generate_temperature: @config.generate_temperature,
57
+ critique_model: @config.critique_model,
58
+ critique_temperature: @config.critique_temperature,
59
+ generate_fallbacks: @config.fallback_names('generate'),
60
+ critique_fallbacks: llm_critique ? @config.fallback_names('critique') : [],
61
+ sources: setting_sources(stages),
62
+ config_paths: @config.layer_paths,
63
+ warnings: @config.warnings,
64
+ critique_rule: llm_critique ? @config.routing.rule : nil,
65
+ api_keys: @config.api_key_counts(stages),
66
+ time_budget: @config.llm_time_budget,
67
+ overloaded_quarantine: @config.overloaded_quarantine,
68
+ coverage: context.coverage,
69
+ sizes: context.sizes
70
+ }
71
+ end
72
+
73
+ # The Jev request as it would go for the stub candidate; nothing is sent,
74
+ # so the critic needs no client. Only whether the key is set: its value
75
+ # never leaves.
76
+ def jev_critique_settings(context)
77
+ critic = JevCritic.new(client: nil, thresholds: @config.jev_thresholds,
78
+ review_instructions: @config.review_instructions,
79
+ scrub: @context_builder.method(:scrub_text), logger: @logger)
80
+ request = critic.preview(context: context, candidates: JSON.parse(CANDIDATES_JSON))
81
+ {model: @config.jev_model, key: Aireview::Utils.present?(@config.jev_api_key),
82
+ thresholds: @config.jev_thresholds, fallback: @config.jev_fallback,
83
+ state: request.state, questions: request.questions}
84
+ end
85
+
86
+ # Only whether the key is set: its value never leaves.
87
+ def jev_shadow_settings
88
+ return nil unless @config.jev_shadow?
89
+
90
+ {model: @config.jev_model, key: Aireview::Utils.present?(@config.jev_api_key), thresholds: @config.jev_thresholds}
91
+ end
92
+
93
+ # Where the model, provider and reserves of a stage came from, for --dry-run.
94
+ def setting_sources(stages)
95
+ stages.to_h do |stage|
96
+ [stage.to_sym, {
97
+ model: @config.stage_model_source(stage),
98
+ provider: @config.stage_provider_source(stage),
99
+ fallbacks: @config.stage_fallbacks_source(stage)
100
+ }]
101
+ end
102
+ end
103
+ end
104
+ end
@@ -1,4 +1,5 @@
1
1
  # frozen_string_literal: true
2
+ require 'json'
2
3
 
3
4
  module Aireview
4
5
  # The --dry-run output: settings, the context summary and the prompts of both stages.
@@ -19,32 +20,63 @@ module Aireview
19
20
  @out.puts
20
21
  @out.puts('=== GENERATE USER PROMPT ===')
21
22
  @out.puts(dry_run.dig(:generate_prompt, :user_prompt))
22
- return unless dry_run[:critique_prompt]
23
+ render_critique_prompt(dry_run[:critique_prompt]) if dry_run[:critique_prompt]
24
+ render_jev_request(dry_run[:jev_critique]) if dry_run[:jev_critique]
25
+ end
26
+
27
+ private
23
28
 
29
+ def render_critique_prompt(prompt)
24
30
  @out.puts
25
31
  @out.puts('=== CRITIQUE SYSTEM PROMPT ===')
26
- @out.puts(dry_run.dig(:critique_prompt, :system_prompt))
32
+ @out.puts(prompt[:system_prompt])
27
33
  @out.puts
28
34
  @out.puts('=== CRITIQUE USER PROMPT ===')
29
- @out.puts(dry_run.dig(:critique_prompt, :user_prompt))
35
+ @out.puts(prompt[:user_prompt])
30
36
  end
31
37
 
32
- private
38
+ # The request Jev would get for the stub candidate.
39
+ def render_jev_request(jev)
40
+ @out.puts
41
+ @out.puts('=== JEV STATE ===')
42
+ @out.puts(JSON.pretty_generate(jev[:state]))
43
+ @out.puts
44
+ @out.puts('=== JEV QUESTIONS ===')
45
+ @out.puts(JSON.pretty_generate(jev[:questions]))
46
+ end
33
47
 
48
+ # With Jev as the engine the LLM Critique line is its fallback.
34
49
  def render_settings(dry_run)
35
50
  @out.puts('=== LLM SETTINGS ===')
36
51
  render_config_paths(dry_run[:config_paths])
37
52
  render_stage(dry_run, :generate)
53
+ render_jev_critique(dry_run[:jev_critique])
38
54
  if dry_run[:critique_prompt]
39
- render_stage(dry_run, :critique)
40
- else
55
+ render_stage(dry_run, :critique, title: dry_run[:jev_critique] ? 'Critique fallback' : 'Critique')
56
+ elsif !dry_run[:jev_critique]
41
57
  @out.puts('Critique: disabled')
42
58
  end
43
59
  @out.puts("Critique rule: #{dry_run[:critique_rule]}") if dry_run[:critique_rule]
60
+ render_jev_shadow(dry_run[:jev_shadow])
44
61
  render_reserves(dry_run)
45
62
  list('warnings', dry_run[:warnings], separator: "\n ")
46
63
  end
47
64
 
65
+ def render_jev_critique(jev)
66
+ return unless jev
67
+
68
+ thresholds = jev[:thresholds].map { |name, value| "#{name}=#{value}" }.join(' ')
69
+ @out.puts("Critique: jev #{jev[:model]} (key #{jev[:key] ? 'set' : 'missing'}; " \
70
+ "fallback: #{jev[:fallback]}; #{thresholds})")
71
+ end
72
+
73
+ def render_jev_shadow(jev)
74
+ return unless jev
75
+
76
+ thresholds = jev[:thresholds].map { |name, value| "#{name}=#{value}" }.join(' ')
77
+ @out.puts("Jev shadow: #{jev[:model]} (key #{jev[:key] ? 'set' : 'missing'}; log only, #{thresholds})")
78
+ end
79
+
48
80
  def render_config_paths(paths)
49
81
  return if paths.nil? || paths.empty?
50
82
 
@@ -53,9 +85,9 @@ module Aireview
53
85
 
54
86
  # The source of every setting is the layer it came from: built-in, image
55
87
  # defaults, .aireview.yml, env or cli.
56
- def render_stage(dry_run, stage)
88
+ def render_stage(dry_run, stage, title: stage.capitalize)
57
89
  sources = dry_run.dig(:sources, stage) || {}
58
- @out.puts("#{stage.capitalize}: #{dry_run[:"#{stage}_model"]} " \
90
+ @out.puts("#{title}: #{dry_run[:"#{stage}_model"]} " \
59
91
  "temperature=#{dry_run[:"#{stage}_temperature"]}#{origin(sources, :model, :provider)}")
60
92
  fallbacks = dry_run[:"#{stage}_fallbacks"]
61
93
  list('fallbacks', fallbacks, separator: ' -> ', suffix: origin(sources, :fallbacks))
@@ -14,4 +14,15 @@ module Aireview
14
14
  class RouteExhaustedError < ApiError; end
15
15
  class ContextBudgetError < Error; end
16
16
  class HelpRequested < Error; end
17
+
18
+ # A Jev request failed: network, HTTP status, an answer of the wrong shape.
19
+ # status is the HTTP status when the server answered.
20
+ class JevError < Error
21
+ attr_reader :status
22
+
23
+ def initialize(message, status: nil)
24
+ super(message)
25
+ @status = status
26
+ end
27
+ end
17
28
  end
@@ -0,0 +1,129 @@
1
+ # frozen_string_literal: true
2
+ require 'json'
3
+ require 'logger'
4
+ require_relative 'errors'
5
+ require_relative 'utils'
6
+
7
+ module Aireview
8
+ # One evaluation request to Jev (TypeSafe, POST /v1/systemone): a state and
9
+ # named questions in, one answer per question out. RubyLLM does not know
10
+ # Jev and the router is built for text models, so Jev has its own small
11
+ # client: one retry on 429/529, no key rotation, no reserves. A failure is
12
+ # a JevError; the caller decides whether it matters.
13
+ class JevClient
14
+ API_URL = 'https://api.typesafe.ai/v1/'
15
+ OPEN_TIMEOUT = 5
16
+ RETRY_STATUSES = [429, 529].freeze
17
+ RETRY_DELAY = 2
18
+ # retry-after beyond this is not worth waiting for in a review job.
19
+ MAX_RETRY_DELAY = 10
20
+
21
+ Result = Struct.new(:model, :answers, :usage, keyword_init: true)
22
+
23
+ # Jev goes out the same way as the LLM providers, through
24
+ # LLM_HTTP_PROXY when it is set. dependencies — connection: and sleeper:
25
+ # for tests.
26
+ def initialize(config:, logger: Logger.new($stderr), **dependencies)
27
+ require 'faraday'
28
+
29
+ raise ConfigError, 'JEV_API_KEY is required' if Aireview::Utils.blank?(config.jev_api_key)
30
+
31
+ @api_key = config.jev_api_key
32
+ @model = config.jev_model
33
+ @logger = logger
34
+ @connection = dependencies[:connection] || build_connection(config.jev_timeout, config.llm_http_proxy)
35
+ @sleeper = dependencies[:sleeper] || ->(seconds) { sleep(seconds) }
36
+ end
37
+
38
+ # questions — {key => question}; returns Result with answers under the
39
+ # same keys, each checked against the type of its question.
40
+ def evaluate(state:, questions:)
41
+ body = JSON.generate(model: @model, state: state, questions: questions)
42
+ @logger.info("Jev request started (model=#{@model}, questions=#{questions.size})")
43
+ response = post_with_one_retry(body)
44
+ result = parse(response, questions)
45
+ log_completed(result)
46
+ result
47
+ end
48
+
49
+ private
50
+
51
+ def build_connection(timeout, proxy)
52
+ Faraday.new(url: API_URL, proxy: Aireview::Utils.presence(proxy)) do |builder|
53
+ builder.options.open_timeout = OPEN_TIMEOUT
54
+ builder.options.timeout = timeout
55
+ builder.adapter Faraday.default_adapter
56
+ end
57
+ end
58
+
59
+ def post_with_one_retry(body)
60
+ response = post(body)
61
+ return response unless RETRY_STATUSES.include?(response.status.to_i)
62
+
63
+ delay = retry_delay(response)
64
+ @logger.warn("Jev answered #{response.status}, retrying once in #{delay}s")
65
+ @sleeper.call(delay)
66
+ post(body)
67
+ end
68
+
69
+ def post(body)
70
+ @connection.post('systemone') do |request|
71
+ request.headers['Authorization'] = "Bearer #{@api_key}"
72
+ request.headers['Content-Type'] = 'application/json'
73
+ request.body = body
74
+ end
75
+ rescue Faraday::Error => e
76
+ raise JevError, "Jev request failed: #{e.class}: #{e.message}"
77
+ end
78
+
79
+ def retry_delay(response)
80
+ seconds = Integer(response.headers['retry-after'].to_s, exception: false)
81
+ seconds&.positive? ? [seconds, MAX_RETRY_DELAY].min : RETRY_DELAY
82
+ end
83
+
84
+ def parse(response, questions)
85
+ status = response.status.to_i
86
+ unless status.between?(200, 299)
87
+ raise JevError.new("Jev API error #{status}: #{response.body.to_s[0, 500]}", status: status)
88
+ end
89
+
90
+ payload = JSON.parse(response.body.to_s)
91
+ answers = payload['answers'] if payload.is_a?(Hash)
92
+ raise JevError, 'Jev returned no answers' unless answers.is_a?(Hash)
93
+
94
+ questions.each { |key, question| check_answer!(key, question, answers[key.to_s]) }
95
+ Result.new(model: payload['model'], answers: answers, usage: payload['usage'])
96
+ rescue JSON::ParserError
97
+ raise JevError, "Jev returned invalid JSON: #{response.body.to_s[0, 500]}"
98
+ end
99
+
100
+ def check_answer!(key, question, answer)
101
+ type = question[:type] || question['type']
102
+ return if answer.is_a?(Hash) && answer['type'] == type && valid_value?(type, answer)
103
+
104
+ raise JevError, "Jev returned an invalid #{type} answer for #{key}: #{answer.inspect}"
105
+ end
106
+
107
+ def valid_value?(type, answer)
108
+ case type
109
+ when 'noul' then probability?(answer['noul'])
110
+ when 'choice' then answer['choice'].is_a?(String) && probability?(answer['confidence'])
111
+ else true
112
+ end
113
+ end
114
+
115
+ def probability?(value)
116
+ value.is_a?(Numeric) && value.between?(0, 1)
117
+ end
118
+
119
+ # The answering version is logged: an alias resolves on the server, and
120
+ # a pinned version that answers as another one breaks the thresholds.
121
+ def log_completed(result)
122
+ tokens = result.usage.is_a?(Hash) ? ", tokens: input=#{result.usage['input_tokens']}" : ''
123
+ @logger.info("Jev request completed (model=#{result.model}#{tokens})")
124
+ return if result.model == @model || !@model.match?(/\Ajev-\d/)
125
+
126
+ @logger.warn("Jev answered as #{result.model.inspect}, not the requested #{@model}")
127
+ end
128
+ end
129
+ end