aireview 2.0.0 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +73 -0
- data/README.md +206 -12
- data/config/.aireview.yml.example +6 -0
- data/lib/aireview/cli.rb +44 -9
- data/lib/aireview/config.rb +26 -8
- data/lib/aireview/config_fallbacks.rb +43 -15
- data/lib/aireview/config_jev.rb +137 -0
- data/lib/aireview/config_layers.rb +1 -1
- data/lib/aireview/config_loader.rb +32 -6
- data/lib/aireview/context_builder.rb +13 -10
- data/lib/aireview/dry_run_prompts.rb +104 -0
- data/lib/aireview/dry_run_report.rb +40 -8
- data/lib/aireview/errors.rb +11 -0
- data/lib/aireview/jev_client.rb +129 -0
- data/lib/aireview/jev_critic.rb +387 -0
- data/lib/aireview/jev_shadow.rb +82 -0
- data/lib/aireview/jev_stage.rb +97 -0
- data/lib/aireview/llm_client.rb +53 -16
- data/lib/aireview/llm_router.rb +4 -4
- data/lib/aireview/model_candidate.rb +34 -5
- data/lib/aireview/model_checker.rb +47 -6
- data/lib/aireview/model_pool.rb +57 -28
- data/lib/aireview/output_schemas.rb +3 -3
- data/lib/aireview/prompts/jev_questions.yml +79 -0
- data/lib/aireview/review_marker.rb +7 -2
- data/lib/aireview/review_pipeline.rb +56 -56
- data/lib/aireview/review_renderer.rb +18 -2
- data/lib/aireview/reviewer.rb +10 -1
- data/lib/aireview/stage_chains.rb +6 -3
- data/lib/aireview/version.rb +1 -1
- metadata +14 -6
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
require_relative 'errors'
|
|
3
|
+
require_relative 'stages'
|
|
4
|
+
require_relative 'utils'
|
|
5
|
+
|
|
6
|
+
module Aireview
|
|
7
|
+
# The critique engine and the Jev (TypeSafe) settings. llm.critique.engine
|
|
8
|
+
# picks who checks the candidates: an LLM (model, the default) or Jev, a
|
|
9
|
+
# fast classifier that decides keep/reject but cannot refine a finding.
|
|
10
|
+
# llm.jev.fallback says what happens when Jev cannot decide: the LLM
|
|
11
|
+
# Critique takes over (model) or the run fails (fail). llm.jev.shadow runs
|
|
12
|
+
# Jev next to the LLM Critique for the log only.
|
|
13
|
+
module ConfigJev
|
|
14
|
+
CRITIQUE_ENGINES = %w[model jev].freeze
|
|
15
|
+
JEV_FALLBACKS = %w[model fail].freeze
|
|
16
|
+
# A pinned version, not an alias: thresholds are tuned against one
|
|
17
|
+
# version, and an alias moves when a release ships.
|
|
18
|
+
DEFAULT_JEV_MODEL = 'jev-1.13.0'
|
|
19
|
+
JEV_ALIASES = %w[jev-latest jev-preview].freeze
|
|
20
|
+
DEFAULT_JEV_TIMEOUT = 10
|
|
21
|
+
# Provisional values until real thresholds are chosen from the shadow
|
|
22
|
+
# logs; the logs carry the raw probabilities for that.
|
|
23
|
+
JEV_THRESHOLD_DEFAULTS = {
|
|
24
|
+
'keep_above' => 0.5,
|
|
25
|
+
'enough_context' => 0.5,
|
|
26
|
+
'version_claim' => 0.5,
|
|
27
|
+
'duplicate' => 0.5
|
|
28
|
+
}.freeze
|
|
29
|
+
|
|
30
|
+
def critique_engine
|
|
31
|
+
one_of!(dig('llm', 'critique', 'engine') || 'model', CRITIQUE_ENGINES, 'llm.critique.engine')
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
def jev_fallback
|
|
35
|
+
one_of!(dig('llm', 'jev', 'fallback') || 'model', JEV_FALLBACKS, 'llm.jev.fallback')
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
# Jev decides in this run: the engine is jev and the run has a critique
|
|
39
|
+
# at all (--no-critique switches off every engine, as the critique:
|
|
40
|
+
# argument or as a CLI override of the config).
|
|
41
|
+
def jev_critique?(critique: true)
|
|
42
|
+
critique && !critique_disabled? && critique_engine == 'jev'
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
# The stages that need an LLM in a run. Jev with fallback: model still
|
|
46
|
+
# needs a ready LLM Critique; with fallback: fail it needs none.
|
|
47
|
+
def llm_stages(critique: true)
|
|
48
|
+
return ['generate'] unless critique && !critique_disabled?
|
|
49
|
+
return ['generate'] if jev_critique? && jev_fallback == 'fail'
|
|
50
|
+
|
|
51
|
+
STAGES
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
def critique_disabled?
|
|
55
|
+
@data['critique_disabled'] == true
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
def jev_shadow?
|
|
59
|
+
value = dig('llm', 'jev', 'shadow')
|
|
60
|
+
return false if value.nil?
|
|
61
|
+
return value if [true, false].include?(value)
|
|
62
|
+
|
|
63
|
+
raise ConfigError, "llm.jev.shadow must be true or false, got #{value.inspect}"
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
def jev_model
|
|
67
|
+
Aireview::Utils.presence(dig('llm', 'jev', 'model')) || DEFAULT_JEV_MODEL
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
def jev_api_key
|
|
71
|
+
@data['jev_api_key']
|
|
72
|
+
end
|
|
73
|
+
|
|
74
|
+
def jev_timeout
|
|
75
|
+
positive_integer!(dig('llm', 'jev', 'timeout') || DEFAULT_JEV_TIMEOUT, 'llm.jev.timeout')
|
|
76
|
+
end
|
|
77
|
+
|
|
78
|
+
def jev_thresholds
|
|
79
|
+
JEV_THRESHOLD_DEFAULTS.to_h do |name, default|
|
|
80
|
+
value = dig('llm', 'jev', name)
|
|
81
|
+
[name.to_sym, value.nil? ? default : probability!(value, "llm.jev.#{name}")]
|
|
82
|
+
end
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
# Everything besides the questions that changes a Jev decision; nil when
|
|
86
|
+
# Jev does not decide. The fallback is part of it: it decides what
|
|
87
|
+
# happens to the candidates Jev could not judge.
|
|
88
|
+
def jev_signature
|
|
89
|
+
return nil unless critique_engine == 'jev'
|
|
90
|
+
|
|
91
|
+
['jev', jev_model, *jev_thresholds.values, jev_fallback]
|
|
92
|
+
end
|
|
93
|
+
|
|
94
|
+
# Jev as the critic needs a key and a pinned version: the thresholds are
|
|
95
|
+
# tuned against one version, and falling back to the LLM silently would
|
|
96
|
+
# hide a broken setup.
|
|
97
|
+
def require_jev!(critique: true)
|
|
98
|
+
return unless jev_critique?(critique: critique)
|
|
99
|
+
raise ConfigError, 'llm.critique.engine is jev, but JEV_API_KEY is not set' if Aireview::Utils.blank?(jev_api_key)
|
|
100
|
+
return unless JEV_ALIASES.include?(jev_model)
|
|
101
|
+
|
|
102
|
+
raise ConfigError, "llm.jev.model #{jev_model.inspect} is an alias; with llm.critique.engine: jev " \
|
|
103
|
+
'pin a version such as jev-1.13.0'
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
def jev_warnings
|
|
107
|
+
return [] unless jev_shadow?
|
|
108
|
+
return ['llm.jev.shadow is ignored: Jev already decides as the critique engine'] if critique_engine == 'jev'
|
|
109
|
+
|
|
110
|
+
warnings = []
|
|
111
|
+
if Aireview::Utils.blank?(jev_api_key)
|
|
112
|
+
warnings << 'llm.jev.shadow is on, but JEV_API_KEY is not set: Jev is skipped'
|
|
113
|
+
end
|
|
114
|
+
if JEV_ALIASES.include?(jev_model)
|
|
115
|
+
warnings << "llm.jev.model #{jev_model.inspect} is an alias: its answers change when TypeSafe ships " \
|
|
116
|
+
'a release; pin a version such as jev-1.13.0'
|
|
117
|
+
end
|
|
118
|
+
warnings
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
private
|
|
122
|
+
|
|
123
|
+
def one_of!(value, allowed, name)
|
|
124
|
+
value = value.to_s
|
|
125
|
+
return value if allowed.include?(value)
|
|
126
|
+
|
|
127
|
+
raise ConfigError, "#{name} must be one of #{allowed.join(', ')}, got #{value.inspect}"
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
def probability!(value, name)
|
|
131
|
+
number = Float(value, exception: false) if value.is_a?(Numeric) || value.is_a?(String)
|
|
132
|
+
return number if number&.between?(0, 1)
|
|
133
|
+
|
|
134
|
+
raise ConfigError, "#{name} must be a number from 0 to 1, got #{value.inspect}"
|
|
135
|
+
end
|
|
136
|
+
end
|
|
137
|
+
end
|
|
@@ -65,7 +65,7 @@ module Aireview
|
|
|
65
65
|
|
|
66
66
|
# Configuration and plan warnings; the CLI and --dry-run print them.
|
|
67
67
|
def warnings
|
|
68
|
-
|
|
68
|
+
llm_stages.flat_map { |stage| stage_provider_warnings(stage) } + routing.warnings + jev_warnings
|
|
69
69
|
end
|
|
70
70
|
|
|
71
71
|
# The paths of the file layers, for --dry-run.
|
|
@@ -26,10 +26,21 @@ module Aireview
|
|
|
26
26
|
'review_mode' => 'REVIEW_MODE',
|
|
27
27
|
'llm_api_base' => 'LLM_API_BASE',
|
|
28
28
|
'ollama_api_base' => 'OLLAMA_API_BASE',
|
|
29
|
-
'llm_http_proxy' => 'LLM_HTTP_PROXY'
|
|
29
|
+
'llm_http_proxy' => 'LLM_HTTP_PROXY',
|
|
30
|
+
'jev_api_key' => 'JEV_API_KEY'
|
|
31
|
+
}.freeze
|
|
32
|
+
PROVIDER_KEY_MAPPING = {
|
|
33
|
+
'gemini' => 'GEMINI_API_KEY',
|
|
34
|
+
'openai' => 'OPENAI_API_KEY',
|
|
35
|
+
'anthropic' => 'ANTHROPIC_API_KEY',
|
|
36
|
+
'openrouter' => 'OPENROUTER_API_KEY'
|
|
37
|
+
}.freeze
|
|
38
|
+
PROVIDER_KEYS_MAPPING = {
|
|
39
|
+
'gemini' => 'GEMINI_API_KEYS',
|
|
40
|
+
'openai' => 'OPENAI_API_KEYS',
|
|
41
|
+
'anthropic' => 'ANTHROPIC_API_KEYS',
|
|
42
|
+
'openrouter' => 'OPENROUTER_API_KEYS'
|
|
30
43
|
}.freeze
|
|
31
|
-
PROVIDER_KEY_MAPPING = {'gemini' => 'GEMINI_API_KEY'}.freeze
|
|
32
|
-
PROVIDER_KEYS_MAPPING = {'gemini' => 'GEMINI_API_KEYS'}.freeze
|
|
33
44
|
CONTEXT_ENV = {
|
|
34
45
|
'max_diff_chars' => 'MAX_DIFF_CHARS',
|
|
35
46
|
'max_mr_description_chars' => 'MAX_MR_DESCRIPTION_CHARS',
|
|
@@ -38,7 +49,8 @@ module Aireview
|
|
|
38
49
|
}.freeze
|
|
39
50
|
LLM_ENV = %w[
|
|
40
51
|
LLM_PROVIDER LLM_TEMPERATURE LLM_TIMEOUT LLM_MAX_PROMPT_CHARS LLM_TIME_BUDGET LLM_OVERLOADED_QUARANTINE
|
|
41
|
-
LLM_MODELS LLM_CRITIQUE_RANK LLM_CRITIQUE_ALLOW_WEAKER
|
|
52
|
+
LLM_MODELS LLM_CRITIQUE_RANK LLM_CRITIQUE_ALLOW_WEAKER LLM_CRITIQUE_ENGINE
|
|
53
|
+
LLM_JEV_SHADOW LLM_JEV_MODEL LLM_JEV_FALLBACK LLM_JEV_KEEP_ABOVE
|
|
42
54
|
].freeze
|
|
43
55
|
LLM_STAGE_ENV_SUFFIXES = %w[PROVIDER MODEL TEMPERATURE MAX_PROMPT_CHARS FALLBACK_MODEL START].freeze
|
|
44
56
|
IMAGE_DEFAULTS_ENV = 'AIREVIEW_DEFAULTS'
|
|
@@ -141,8 +153,21 @@ module Aireview
|
|
|
141
153
|
'time_budget' => parse_integer(env['LLM_TIME_BUDGET'], 'LLM_TIME_BUDGET'),
|
|
142
154
|
'overloaded_quarantine' => parse_integer(env['LLM_OVERLOADED_QUARANTINE'], 'LLM_OVERLOADED_QUARANTINE'),
|
|
143
155
|
'generate' => llm_stage_env_config(env, 'GENERATE'),
|
|
144
|
-
'critique' => llm_stage_env_config(env, 'CRITIQUE')
|
|
145
|
-
|
|
156
|
+
'critique' => llm_stage_env_config(env, 'CRITIQUE'),
|
|
157
|
+
'jev' => jev_env_config(env)
|
|
158
|
+
}.compact.reject { |key, value| %w[generate critique jev].include?(key) && value.empty? }
|
|
159
|
+
end
|
|
160
|
+
|
|
161
|
+
# LLM_JEV_SHADOW=true|false, LLM_JEV_MODEL — a pinned Jev version,
|
|
162
|
+
# LLM_JEV_FALLBACK=model|fail, LLM_JEV_KEEP_ABOVE — the keep threshold.
|
|
163
|
+
def jev_env_config(env)
|
|
164
|
+
{
|
|
165
|
+
'shadow' => parse_boolean(env['LLM_JEV_SHADOW'], 'LLM_JEV_SHADOW'),
|
|
166
|
+
'model' => Aireview::Utils.presence(env['LLM_JEV_MODEL']),
|
|
167
|
+
'fallback' => Aireview::Utils.presence(env['LLM_JEV_FALLBACK']),
|
|
168
|
+
# Validated by ConfigJev: a typo must fail, not fall back to the default.
|
|
169
|
+
'keep_above' => Aireview::Utils.presence(env['LLM_JEV_KEEP_ABOVE'])
|
|
170
|
+
}.compact
|
|
146
171
|
end
|
|
147
172
|
|
|
148
173
|
def llm_stage_env_config(env, stage)
|
|
@@ -176,6 +201,7 @@ module Aireview
|
|
|
176
201
|
'generate' => {'start' => env['LLM_GENERATE_START']}.compact,
|
|
177
202
|
'critique' => {
|
|
178
203
|
'start' => env['LLM_CRITIQUE_START'],
|
|
204
|
+
'engine' => Aireview::Utils.presence(env['LLM_CRITIQUE_ENGINE']),
|
|
179
205
|
'rank' => env['LLM_CRITIQUE_RANK'],
|
|
180
206
|
'allow_weaker' => parse_boolean(env['LLM_CRITIQUE_ALLOW_WEAKER'], 'LLM_CRITIQUE_ALLOW_WEAKER')
|
|
181
207
|
}.compact
|
|
@@ -22,8 +22,9 @@ module Aireview
|
|
|
22
22
|
CANDIDATES_RESERVE_CHARS = 4_500
|
|
23
23
|
|
|
24
24
|
# The context of one run: both stages get the same MR, Jira and diff,
|
|
25
|
-
# truncated once for the tightest of the stages.
|
|
26
|
-
|
|
25
|
+
# truncated once for the tightest of the stages. sections — the MR and
|
|
26
|
+
# Jira part without the diff, for Jev.
|
|
27
|
+
Context = Struct.new(:user_prompt, :sections, :diff_text, :coverage, :sizes, keyword_init: true)
|
|
27
28
|
|
|
28
29
|
def initialize(config:, logger: Logger.new($stderr))
|
|
29
30
|
@config = config
|
|
@@ -49,7 +50,8 @@ module Aireview
|
|
|
49
50
|
|
|
50
51
|
sizes = context_sizes(fixed: fixed, packed: packed, budget: budget, diff_budget: diff_budget, critique: critique)
|
|
51
52
|
log_sizes(sizes)
|
|
52
|
-
Context.new(user_prompt: fixed + packed.text,
|
|
53
|
+
Context.new(user_prompt: fixed + packed.text, sections: sections.join("\n\n"), diff_text: packed.text,
|
|
54
|
+
coverage: coverage, sizes: sizes)
|
|
53
55
|
end
|
|
54
56
|
|
|
55
57
|
def build_generate_prompt(context)
|
|
@@ -88,12 +90,17 @@ module Aireview
|
|
|
88
90
|
{system_prompt: system, user_prompt: user}
|
|
89
91
|
end
|
|
90
92
|
|
|
93
|
+
def scrub_text(text)
|
|
94
|
+
@secret_scrubber.scrub_text(text.to_s)
|
|
95
|
+
end
|
|
96
|
+
|
|
91
97
|
private
|
|
92
98
|
|
|
93
99
|
# The minimum over the stages: the context is one per run, so it must fit
|
|
94
|
-
# into each of them together with its system prompt and reserve.
|
|
100
|
+
# into each of them together with its system prompt and reserve. Only the
|
|
101
|
+
# stages that go to an LLM count: Jev as the critic has limits of its own.
|
|
95
102
|
def context_budget(critique:)
|
|
96
|
-
stages = critique
|
|
103
|
+
stages = @config.llm_stages(critique: critique)
|
|
97
104
|
budgets = stages.to_h { |stage| [stage, stage_budget(stage)] }
|
|
98
105
|
stage, budget = budgets.min_by { |_, value| value }
|
|
99
106
|
return budget if budget.positive?
|
|
@@ -153,7 +160,7 @@ module Aireview
|
|
|
153
160
|
end
|
|
154
161
|
|
|
155
162
|
def context_sizes(fixed:, packed:, budget:, diff_budget:, critique:)
|
|
156
|
-
stages = critique
|
|
163
|
+
stages = @config.llm_stages(critique: critique)
|
|
157
164
|
{
|
|
158
165
|
context_budget: budget,
|
|
159
166
|
diff_budget: diff_budget,
|
|
@@ -187,10 +194,6 @@ module Aireview
|
|
|
187
194
|
Aireview::Utils.presence(scrubbed) || '(empty)'
|
|
188
195
|
end
|
|
189
196
|
|
|
190
|
-
def scrub_text(text)
|
|
191
|
-
@secret_scrubber.scrub_text(text.to_s)
|
|
192
|
-
end
|
|
193
|
-
|
|
194
197
|
def language_name(code)
|
|
195
198
|
LANGUAGE_NAMES.fetch(code.to_s, code.to_s)
|
|
196
199
|
end
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
require 'json'
|
|
3
|
+
require 'logger'
|
|
4
|
+
require_relative 'utils'
|
|
5
|
+
require_relative 'jev_critic'
|
|
6
|
+
|
|
7
|
+
module Aireview
|
|
8
|
+
# The prompts of a run without calling a model, and everything --dry-run
|
|
9
|
+
# shows. The review key is computed from them too (ReviewMarker): the LLM
|
|
10
|
+
# Critique prompt is built only when an LLM Critique can run, the Jev
|
|
11
|
+
# question templates only when Jev decides.
|
|
12
|
+
class DryRunPrompts
|
|
13
|
+
CANDIDATES_JSON = '[{"id":"C1","file":"path/from/diff.rb","line":1,' \
|
|
14
|
+
'"quoted_code":"...","problem":"...","why":"...","suggestion":"...",' \
|
|
15
|
+
'"category":"bug","severity":"major"}]'
|
|
16
|
+
|
|
17
|
+
def initialize(config:, context_builder:, logger: Logger.new($stderr))
|
|
18
|
+
@config = config
|
|
19
|
+
@context_builder = context_builder
|
|
20
|
+
@logger = logger
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
def build(merge_request:, changes:, jira_issue: nil, critique: true)
|
|
24
|
+
@config.require_models!(critique: critique)
|
|
25
|
+
stages = @config.llm_stages(critique: critique)
|
|
26
|
+
context = @context_builder.prepare(
|
|
27
|
+
merge_request: merge_request,
|
|
28
|
+
changes: changes,
|
|
29
|
+
jira_issue: jira_issue,
|
|
30
|
+
critique: critique
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
prompts(context, stages, critique).merge(settings(context, stages, critique))
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
private
|
|
37
|
+
|
|
38
|
+
def prompts(context, stages, critique)
|
|
39
|
+
jev = @config.jev_critique?(critique: critique)
|
|
40
|
+
critique_prompt = if stages.include?('critique')
|
|
41
|
+
@context_builder.build_critique_prompt(context, candidates_json: CANDIDATES_JSON)
|
|
42
|
+
end
|
|
43
|
+
{
|
|
44
|
+
generate_prompt: @context_builder.build_generate_prompt(context),
|
|
45
|
+
critique_prompt: critique_prompt,
|
|
46
|
+
jev_questions: jev ? JevCritic.decision_templates : nil,
|
|
47
|
+
jev_critique: jev ? jev_critique_settings(context) : nil,
|
|
48
|
+
jev_shadow: critique && !jev ? jev_shadow_settings : nil
|
|
49
|
+
}
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
def settings(context, stages, critique)
|
|
53
|
+
llm_critique = stages.include?('critique')
|
|
54
|
+
{
|
|
55
|
+
generate_model: @config.generate_model,
|
|
56
|
+
generate_temperature: @config.generate_temperature,
|
|
57
|
+
critique_model: @config.critique_model,
|
|
58
|
+
critique_temperature: @config.critique_temperature,
|
|
59
|
+
generate_fallbacks: @config.fallback_names('generate'),
|
|
60
|
+
critique_fallbacks: llm_critique ? @config.fallback_names('critique') : [],
|
|
61
|
+
sources: setting_sources(stages),
|
|
62
|
+
config_paths: @config.layer_paths,
|
|
63
|
+
warnings: @config.warnings,
|
|
64
|
+
critique_rule: llm_critique ? @config.routing.rule : nil,
|
|
65
|
+
api_keys: @config.api_key_counts(stages),
|
|
66
|
+
time_budget: @config.llm_time_budget,
|
|
67
|
+
overloaded_quarantine: @config.overloaded_quarantine,
|
|
68
|
+
coverage: context.coverage,
|
|
69
|
+
sizes: context.sizes
|
|
70
|
+
}
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
# The Jev request as it would go for the stub candidate; nothing is sent,
|
|
74
|
+
# so the critic needs no client. Only whether the key is set: its value
|
|
75
|
+
# never leaves.
|
|
76
|
+
def jev_critique_settings(context)
|
|
77
|
+
critic = JevCritic.new(client: nil, thresholds: @config.jev_thresholds,
|
|
78
|
+
review_instructions: @config.review_instructions,
|
|
79
|
+
scrub: @context_builder.method(:scrub_text), logger: @logger)
|
|
80
|
+
request = critic.preview(context: context, candidates: JSON.parse(CANDIDATES_JSON))
|
|
81
|
+
{model: @config.jev_model, key: Aireview::Utils.present?(@config.jev_api_key),
|
|
82
|
+
thresholds: @config.jev_thresholds, fallback: @config.jev_fallback,
|
|
83
|
+
state: request.state, questions: request.questions}
|
|
84
|
+
end
|
|
85
|
+
|
|
86
|
+
# Only whether the key is set: its value never leaves.
|
|
87
|
+
def jev_shadow_settings
|
|
88
|
+
return nil unless @config.jev_shadow?
|
|
89
|
+
|
|
90
|
+
{model: @config.jev_model, key: Aireview::Utils.present?(@config.jev_api_key), thresholds: @config.jev_thresholds}
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
# Where the model, provider and reserves of a stage came from, for --dry-run.
|
|
94
|
+
def setting_sources(stages)
|
|
95
|
+
stages.to_h do |stage|
|
|
96
|
+
[stage.to_sym, {
|
|
97
|
+
model: @config.stage_model_source(stage),
|
|
98
|
+
provider: @config.stage_provider_source(stage),
|
|
99
|
+
fallbacks: @config.stage_fallbacks_source(stage)
|
|
100
|
+
}]
|
|
101
|
+
end
|
|
102
|
+
end
|
|
103
|
+
end
|
|
104
|
+
end
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
|
+
require 'json'
|
|
2
3
|
|
|
3
4
|
module Aireview
|
|
4
5
|
# The --dry-run output: settings, the context summary and the prompts of both stages.
|
|
@@ -19,32 +20,63 @@ module Aireview
|
|
|
19
20
|
@out.puts
|
|
20
21
|
@out.puts('=== GENERATE USER PROMPT ===')
|
|
21
22
|
@out.puts(dry_run.dig(:generate_prompt, :user_prompt))
|
|
22
|
-
|
|
23
|
+
render_critique_prompt(dry_run[:critique_prompt]) if dry_run[:critique_prompt]
|
|
24
|
+
render_jev_request(dry_run[:jev_critique]) if dry_run[:jev_critique]
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
private
|
|
23
28
|
|
|
29
|
+
def render_critique_prompt(prompt)
|
|
24
30
|
@out.puts
|
|
25
31
|
@out.puts('=== CRITIQUE SYSTEM PROMPT ===')
|
|
26
|
-
@out.puts(
|
|
32
|
+
@out.puts(prompt[:system_prompt])
|
|
27
33
|
@out.puts
|
|
28
34
|
@out.puts('=== CRITIQUE USER PROMPT ===')
|
|
29
|
-
@out.puts(
|
|
35
|
+
@out.puts(prompt[:user_prompt])
|
|
30
36
|
end
|
|
31
37
|
|
|
32
|
-
|
|
38
|
+
# The request Jev would get for the stub candidate.
|
|
39
|
+
def render_jev_request(jev)
|
|
40
|
+
@out.puts
|
|
41
|
+
@out.puts('=== JEV STATE ===')
|
|
42
|
+
@out.puts(JSON.pretty_generate(jev[:state]))
|
|
43
|
+
@out.puts
|
|
44
|
+
@out.puts('=== JEV QUESTIONS ===')
|
|
45
|
+
@out.puts(JSON.pretty_generate(jev[:questions]))
|
|
46
|
+
end
|
|
33
47
|
|
|
48
|
+
# With Jev as the engine the LLM Critique line is its fallback.
|
|
34
49
|
def render_settings(dry_run)
|
|
35
50
|
@out.puts('=== LLM SETTINGS ===')
|
|
36
51
|
render_config_paths(dry_run[:config_paths])
|
|
37
52
|
render_stage(dry_run, :generate)
|
|
53
|
+
render_jev_critique(dry_run[:jev_critique])
|
|
38
54
|
if dry_run[:critique_prompt]
|
|
39
|
-
render_stage(dry_run, :critique)
|
|
40
|
-
|
|
55
|
+
render_stage(dry_run, :critique, title: dry_run[:jev_critique] ? 'Critique fallback' : 'Critique')
|
|
56
|
+
elsif !dry_run[:jev_critique]
|
|
41
57
|
@out.puts('Critique: disabled')
|
|
42
58
|
end
|
|
43
59
|
@out.puts("Critique rule: #{dry_run[:critique_rule]}") if dry_run[:critique_rule]
|
|
60
|
+
render_jev_shadow(dry_run[:jev_shadow])
|
|
44
61
|
render_reserves(dry_run)
|
|
45
62
|
list('warnings', dry_run[:warnings], separator: "\n ")
|
|
46
63
|
end
|
|
47
64
|
|
|
65
|
+
def render_jev_critique(jev)
|
|
66
|
+
return unless jev
|
|
67
|
+
|
|
68
|
+
thresholds = jev[:thresholds].map { |name, value| "#{name}=#{value}" }.join(' ')
|
|
69
|
+
@out.puts("Critique: jev #{jev[:model]} (key #{jev[:key] ? 'set' : 'missing'}; " \
|
|
70
|
+
"fallback: #{jev[:fallback]}; #{thresholds})")
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
def render_jev_shadow(jev)
|
|
74
|
+
return unless jev
|
|
75
|
+
|
|
76
|
+
thresholds = jev[:thresholds].map { |name, value| "#{name}=#{value}" }.join(' ')
|
|
77
|
+
@out.puts("Jev shadow: #{jev[:model]} (key #{jev[:key] ? 'set' : 'missing'}; log only, #{thresholds})")
|
|
78
|
+
end
|
|
79
|
+
|
|
48
80
|
def render_config_paths(paths)
|
|
49
81
|
return if paths.nil? || paths.empty?
|
|
50
82
|
|
|
@@ -53,9 +85,9 @@ module Aireview
|
|
|
53
85
|
|
|
54
86
|
# The source of every setting is the layer it came from: built-in, image
|
|
55
87
|
# defaults, .aireview.yml, env or cli.
|
|
56
|
-
def render_stage(dry_run, stage)
|
|
88
|
+
def render_stage(dry_run, stage, title: stage.capitalize)
|
|
57
89
|
sources = dry_run.dig(:sources, stage) || {}
|
|
58
|
-
@out.puts("#{
|
|
90
|
+
@out.puts("#{title}: #{dry_run[:"#{stage}_model"]} " \
|
|
59
91
|
"temperature=#{dry_run[:"#{stage}_temperature"]}#{origin(sources, :model, :provider)}")
|
|
60
92
|
fallbacks = dry_run[:"#{stage}_fallbacks"]
|
|
61
93
|
list('fallbacks', fallbacks, separator: ' -> ', suffix: origin(sources, :fallbacks))
|
data/lib/aireview/errors.rb
CHANGED
|
@@ -14,4 +14,15 @@ module Aireview
|
|
|
14
14
|
class RouteExhaustedError < ApiError; end
|
|
15
15
|
class ContextBudgetError < Error; end
|
|
16
16
|
class HelpRequested < Error; end
|
|
17
|
+
|
|
18
|
+
# A Jev request failed: network, HTTP status, an answer of the wrong shape.
|
|
19
|
+
# status is the HTTP status when the server answered.
|
|
20
|
+
class JevError < Error
|
|
21
|
+
attr_reader :status
|
|
22
|
+
|
|
23
|
+
def initialize(message, status: nil)
|
|
24
|
+
super(message)
|
|
25
|
+
@status = status
|
|
26
|
+
end
|
|
27
|
+
end
|
|
17
28
|
end
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
require 'json'
|
|
3
|
+
require 'logger'
|
|
4
|
+
require_relative 'errors'
|
|
5
|
+
require_relative 'utils'
|
|
6
|
+
|
|
7
|
+
module Aireview
|
|
8
|
+
# One evaluation request to Jev (TypeSafe, POST /v1/systemone): a state and
|
|
9
|
+
# named questions in, one answer per question out. RubyLLM does not know
|
|
10
|
+
# Jev and the router is built for text models, so Jev has its own small
|
|
11
|
+
# client: one retry on 429/529, no key rotation, no reserves. A failure is
|
|
12
|
+
# a JevError; the caller decides whether it matters.
|
|
13
|
+
class JevClient
|
|
14
|
+
API_URL = 'https://api.typesafe.ai/v1/'
|
|
15
|
+
OPEN_TIMEOUT = 5
|
|
16
|
+
RETRY_STATUSES = [429, 529].freeze
|
|
17
|
+
RETRY_DELAY = 2
|
|
18
|
+
# retry-after beyond this is not worth waiting for in a review job.
|
|
19
|
+
MAX_RETRY_DELAY = 10
|
|
20
|
+
|
|
21
|
+
Result = Struct.new(:model, :answers, :usage, keyword_init: true)
|
|
22
|
+
|
|
23
|
+
# Jev goes out the same way as the LLM providers, through
|
|
24
|
+
# LLM_HTTP_PROXY when it is set. dependencies — connection: and sleeper:
|
|
25
|
+
# for tests.
|
|
26
|
+
def initialize(config:, logger: Logger.new($stderr), **dependencies)
|
|
27
|
+
require 'faraday'
|
|
28
|
+
|
|
29
|
+
raise ConfigError, 'JEV_API_KEY is required' if Aireview::Utils.blank?(config.jev_api_key)
|
|
30
|
+
|
|
31
|
+
@api_key = config.jev_api_key
|
|
32
|
+
@model = config.jev_model
|
|
33
|
+
@logger = logger
|
|
34
|
+
@connection = dependencies[:connection] || build_connection(config.jev_timeout, config.llm_http_proxy)
|
|
35
|
+
@sleeper = dependencies[:sleeper] || ->(seconds) { sleep(seconds) }
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
# questions — {key => question}; returns Result with answers under the
|
|
39
|
+
# same keys, each checked against the type of its question.
|
|
40
|
+
def evaluate(state:, questions:)
|
|
41
|
+
body = JSON.generate(model: @model, state: state, questions: questions)
|
|
42
|
+
@logger.info("Jev request started (model=#{@model}, questions=#{questions.size})")
|
|
43
|
+
response = post_with_one_retry(body)
|
|
44
|
+
result = parse(response, questions)
|
|
45
|
+
log_completed(result)
|
|
46
|
+
result
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
private
|
|
50
|
+
|
|
51
|
+
def build_connection(timeout, proxy)
|
|
52
|
+
Faraday.new(url: API_URL, proxy: Aireview::Utils.presence(proxy)) do |builder|
|
|
53
|
+
builder.options.open_timeout = OPEN_TIMEOUT
|
|
54
|
+
builder.options.timeout = timeout
|
|
55
|
+
builder.adapter Faraday.default_adapter
|
|
56
|
+
end
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
def post_with_one_retry(body)
|
|
60
|
+
response = post(body)
|
|
61
|
+
return response unless RETRY_STATUSES.include?(response.status.to_i)
|
|
62
|
+
|
|
63
|
+
delay = retry_delay(response)
|
|
64
|
+
@logger.warn("Jev answered #{response.status}, retrying once in #{delay}s")
|
|
65
|
+
@sleeper.call(delay)
|
|
66
|
+
post(body)
|
|
67
|
+
end
|
|
68
|
+
|
|
69
|
+
def post(body)
|
|
70
|
+
@connection.post('systemone') do |request|
|
|
71
|
+
request.headers['Authorization'] = "Bearer #{@api_key}"
|
|
72
|
+
request.headers['Content-Type'] = 'application/json'
|
|
73
|
+
request.body = body
|
|
74
|
+
end
|
|
75
|
+
rescue Faraday::Error => e
|
|
76
|
+
raise JevError, "Jev request failed: #{e.class}: #{e.message}"
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
def retry_delay(response)
|
|
80
|
+
seconds = Integer(response.headers['retry-after'].to_s, exception: false)
|
|
81
|
+
seconds&.positive? ? [seconds, MAX_RETRY_DELAY].min : RETRY_DELAY
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
def parse(response, questions)
|
|
85
|
+
status = response.status.to_i
|
|
86
|
+
unless status.between?(200, 299)
|
|
87
|
+
raise JevError.new("Jev API error #{status}: #{response.body.to_s[0, 500]}", status: status)
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
payload = JSON.parse(response.body.to_s)
|
|
91
|
+
answers = payload['answers'] if payload.is_a?(Hash)
|
|
92
|
+
raise JevError, 'Jev returned no answers' unless answers.is_a?(Hash)
|
|
93
|
+
|
|
94
|
+
questions.each { |key, question| check_answer!(key, question, answers[key.to_s]) }
|
|
95
|
+
Result.new(model: payload['model'], answers: answers, usage: payload['usage'])
|
|
96
|
+
rescue JSON::ParserError
|
|
97
|
+
raise JevError, "Jev returned invalid JSON: #{response.body.to_s[0, 500]}"
|
|
98
|
+
end
|
|
99
|
+
|
|
100
|
+
def check_answer!(key, question, answer)
|
|
101
|
+
type = question[:type] || question['type']
|
|
102
|
+
return if answer.is_a?(Hash) && answer['type'] == type && valid_value?(type, answer)
|
|
103
|
+
|
|
104
|
+
raise JevError, "Jev returned an invalid #{type} answer for #{key}: #{answer.inspect}"
|
|
105
|
+
end
|
|
106
|
+
|
|
107
|
+
def valid_value?(type, answer)
|
|
108
|
+
case type
|
|
109
|
+
when 'noul' then probability?(answer['noul'])
|
|
110
|
+
when 'choice' then answer['choice'].is_a?(String) && probability?(answer['confidence'])
|
|
111
|
+
else true
|
|
112
|
+
end
|
|
113
|
+
end
|
|
114
|
+
|
|
115
|
+
def probability?(value)
|
|
116
|
+
value.is_a?(Numeric) && value.between?(0, 1)
|
|
117
|
+
end
|
|
118
|
+
|
|
119
|
+
# The answering version is logged: an alias resolves on the server, and
|
|
120
|
+
# a pinned version that answers as another one breaks the thresholds.
|
|
121
|
+
def log_completed(result)
|
|
122
|
+
tokens = result.usage.is_a?(Hash) ? ", tokens: input=#{result.usage['input_tokens']}" : ''
|
|
123
|
+
@logger.info("Jev request completed (model=#{result.model}#{tokens})")
|
|
124
|
+
return if result.model == @model || !@model.match?(/\Ajev-\d/)
|
|
125
|
+
|
|
126
|
+
@logger.warn("Jev answered as #{result.model.inspect}, not the requested #{@model}")
|
|
127
|
+
end
|
|
128
|
+
end
|
|
129
|
+
end
|