aireview 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +26 -0
- data/CONTRIBUTORS.md +8 -0
- data/README.md +82 -2
- data/lib/aireview/cli.rb +9 -29
- data/lib/aireview/config.rb +12 -4
- data/lib/aireview/config_limits.rb +85 -0
- data/lib/aireview/context_budget.rb +197 -0
- data/lib/aireview/context_builder.rb +125 -39
- data/lib/aireview/diff_fetcher.rb +79 -16
- data/lib/aireview/dry_run_report.rb +64 -0
- data/lib/aireview/errors.rb +1 -0
- data/lib/aireview/prompts/critique.txt +10 -0
- data/lib/aireview/prompts/generate.txt +9 -0
- data/lib/aireview/review_pipeline.rb +24 -32
- data/lib/aireview/review_renderer.rb +55 -5
- data/lib/aireview/reviewer.rb +15 -6
- data/lib/aireview/version.rb +1 -1
- data/lib/aireview.rb +1 -0
- metadata +7 -2
|
@@ -1,16 +1,29 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
require_relative 'utils'
|
|
3
|
+
require_relative 'errors'
|
|
3
4
|
require_relative 'secret_scrubber'
|
|
5
|
+
require_relative 'diff_fetcher'
|
|
6
|
+
require_relative 'context_budget'
|
|
4
7
|
|
|
5
8
|
module Aireview
|
|
6
9
|
class ContextBuilder
|
|
7
|
-
MAX_DIFF_CHARS = 120_000
|
|
8
10
|
GENERATE_PROMPT_TEMPLATE = File.read(File.expand_path('prompts/generate.txt', __dir__)).strip.freeze
|
|
9
11
|
CRITIQUE_PROMPT_TEMPLATE = File.read(File.expand_path('prompts/critique.txt', __dir__)).strip.freeze
|
|
10
12
|
LANGUAGE_NAMES = {
|
|
11
13
|
'ru' => 'Russian',
|
|
12
14
|
'en' => 'English'
|
|
13
15
|
}.freeze
|
|
16
|
+
CHANGES_HEADER = "Changes:\n"
|
|
17
|
+
CANDIDATES_HEADER = "\n\nCandidates JSON from Generate:\n"
|
|
18
|
+
# Резерв под кандидатов в промпте критика: три кандидата по ~1 500
|
|
19
|
+
# символов. Оценка, не гарантия; фактический размер проверяется перед
|
|
20
|
+
# отправкой.
|
|
21
|
+
CANDIDATES_RESERVE_CHARS = 4_500
|
|
22
|
+
STAGES = %i[generate critique].freeze
|
|
23
|
+
|
|
24
|
+
# Контекст одного прогона: обе стадии получают одинаковые MR, Jira и дифф,
|
|
25
|
+
# усечённые один раз под самую тесную из стадий.
|
|
26
|
+
Context = Struct.new(:user_prompt, :coverage, :sizes, keyword_init: true)
|
|
14
27
|
|
|
15
28
|
def initialize(config:, logger: Logger.new($stderr))
|
|
16
29
|
@config = config
|
|
@@ -20,32 +33,36 @@ module Aireview
|
|
|
20
33
|
secret_files: config.secret_files,
|
|
21
34
|
logger: logger
|
|
22
35
|
)
|
|
36
|
+
@diff_fetcher = DiffFetcher.new(ignore_paths: [], logger: logger)
|
|
23
37
|
end
|
|
24
38
|
|
|
25
|
-
def
|
|
26
|
-
|
|
27
|
-
|
|
39
|
+
def prepare(merge_request:, changes:, jira_issue: nil, critique: true)
|
|
40
|
+
coverage = ContextBudget::Coverage.empty
|
|
41
|
+
budget = context_budget(critique: critique)
|
|
42
|
+
sections = merge_request_sections(merge_request, coverage: coverage)
|
|
43
|
+
sections << jira_section(jira_issue, coverage: coverage) if jira_issue
|
|
44
|
+
fixed = "#{sections.join("\n\n")}\n\n#{CHANGES_HEADER}"
|
|
28
45
|
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
user_prompt: user_prompt(merge_request: merge_request, changes_text: changes_text, jira_issue: jira_issue)
|
|
33
|
-
}
|
|
34
|
-
end
|
|
46
|
+
diff_budget = [budget - fixed.length, @config.max_diff_chars].min
|
|
47
|
+
entries = @diff_fetcher.entries(@secret_scrubber.scrub_changes(changes))
|
|
48
|
+
packed = ContextBudget.pack_entries(entries, budget: diff_budget, coverage: coverage)
|
|
35
49
|
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
50
|
+
sizes = context_sizes(fixed: fixed, packed: packed, budget: budget, diff_budget: diff_budget, critique: critique)
|
|
51
|
+
log_sizes(sizes)
|
|
52
|
+
Context.new(user_prompt: fixed + packed.text, coverage: coverage, sizes: sizes)
|
|
53
|
+
end
|
|
39
54
|
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
user_prompt: user
|
|
43
|
-
}
|
|
55
|
+
def build_generate_prompt(context)
|
|
56
|
+
check_stage_size!(:generate, system_prompt(:generate), context.user_prompt)
|
|
44
57
|
end
|
|
45
58
|
|
|
46
|
-
|
|
59
|
+
def build_critique_prompt(context, candidates_json:)
|
|
60
|
+
user = "#{context.user_prompt}#{CANDIDATES_HEADER}#{scrub_text(candidates_json)}"
|
|
61
|
+
check_stage_size!(:critique, system_prompt(:critique), user)
|
|
62
|
+
end
|
|
47
63
|
|
|
48
|
-
def system_prompt(
|
|
64
|
+
def system_prompt(stage)
|
|
65
|
+
template = stage.to_sym == :critique ? CRITIQUE_PROMPT_TEMPLATE : GENERATE_PROMPT_TEMPLATE
|
|
49
66
|
extras = []
|
|
50
67
|
if Aireview::Utils.present?(@config.review_instructions)
|
|
51
68
|
extras << "Additional project instructions:\n#{scrub_text(@config.review_instructions.strip)}"
|
|
@@ -55,44 +72,113 @@ module Aireview
|
|
|
55
72
|
[template, *extras].join("\n\n")
|
|
56
73
|
end
|
|
57
74
|
|
|
58
|
-
|
|
59
|
-
|
|
75
|
+
# Проверка перед отправкой: если кандидаты вышли за резерв и запрос не
|
|
76
|
+
# помещается, это ошибка, а не повод молча резать контекст, который
|
|
77
|
+
# генератор уже видел.
|
|
78
|
+
def check_stage_size!(stage, system, user)
|
|
79
|
+
limit = @config.max_prompt_chars(stage)
|
|
80
|
+
total = system.length + user.length
|
|
81
|
+
if total > limit
|
|
82
|
+
raise ContextBudgetError,
|
|
83
|
+
"#{stage.capitalize} request is #{total} chars, over llm.#{stage}.max_prompt_chars=#{limit} " \
|
|
84
|
+
"(system prompt #{system.length}, context #{user.length})"
|
|
85
|
+
end
|
|
86
|
+
|
|
87
|
+
{system_prompt: system, user_prompt: user}
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
private
|
|
91
|
+
|
|
92
|
+
# Минимум по стадиям: контекст один на прогон, поэтому он должен
|
|
93
|
+
# помещаться в каждую из них вместе с её системным промптом и резервом.
|
|
94
|
+
def context_budget(critique:)
|
|
95
|
+
stages = critique ? STAGES : [:generate]
|
|
96
|
+
budgets = stages.to_h { |stage| [stage, stage_budget(stage)] }
|
|
97
|
+
stage, budget = budgets.min_by { |_, value| value }
|
|
98
|
+
return budget if budget.positive?
|
|
99
|
+
|
|
100
|
+
raise ContextBudgetError,
|
|
101
|
+
"System prompt of the #{stage} stage (#{system_prompt(stage).length} chars, including " \
|
|
102
|
+
"review_instructions) leaves no room for the merge request within llm.#{stage}.max_prompt_chars=" \
|
|
103
|
+
"#{@config.max_prompt_chars(stage)}"
|
|
104
|
+
end
|
|
60
105
|
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
sections.join("\n\n")
|
|
106
|
+
def stage_budget(stage)
|
|
107
|
+
reserve = stage == :critique ? CANDIDATES_RESERVE_CHARS + CANDIDATES_HEADER.length : 0
|
|
108
|
+
@config.max_prompt_chars(stage) - system_prompt(stage).length - reserve
|
|
65
109
|
end
|
|
66
110
|
|
|
67
|
-
def merge_request_sections(merge_request)
|
|
111
|
+
def merge_request_sections(merge_request, coverage:)
|
|
68
112
|
source_branch = scrub_text(merge_request['source_branch'])
|
|
69
113
|
target_branch = scrub_text(merge_request['target_branch'])
|
|
114
|
+
description = ContextBudget.truncate_section(
|
|
115
|
+
scrub_optional_text(merge_request['description']),
|
|
116
|
+
limit: @config.max_mr_description_chars,
|
|
117
|
+
label: 'MR description',
|
|
118
|
+
coverage: coverage
|
|
119
|
+
)
|
|
70
120
|
|
|
71
121
|
[
|
|
72
122
|
"MR: #{scrub_text(merge_request['title'])}",
|
|
73
123
|
"Author: #{scrub_text(merge_request.dig('author', 'name'))}",
|
|
74
124
|
"Branch: #{source_branch} -> #{target_branch}",
|
|
75
|
-
"Description:\n#{
|
|
125
|
+
"Description:\n#{description}"
|
|
76
126
|
]
|
|
77
127
|
end
|
|
78
128
|
|
|
79
|
-
def jira_section(jira_issue)
|
|
129
|
+
def jira_section(jira_issue, coverage:)
|
|
130
|
+
description = ContextBudget.truncate_section(
|
|
131
|
+
scrub_optional_text(jira_issue['description']),
|
|
132
|
+
limit: @config.max_jira_description_chars,
|
|
133
|
+
label: 'Jira description',
|
|
134
|
+
coverage: coverage
|
|
135
|
+
)
|
|
80
136
|
section = "Jira task (#{scrub_text(jira_issue['key'])}):\n"
|
|
81
137
|
section << "Summary: #{scrub_text(jira_issue['summary'])}\n"
|
|
82
|
-
section << "Description:\n#{
|
|
83
|
-
|
|
84
|
-
comments = jira_issue['comments']
|
|
85
|
-
return section
|
|
86
|
-
|
|
87
|
-
|
|
138
|
+
section << "Description:\n#{description}"
|
|
139
|
+
|
|
140
|
+
comments = Array(jira_issue['comments'])
|
|
141
|
+
return section if comments.empty?
|
|
142
|
+
|
|
143
|
+
rendered = comments.each_with_index.map do |comment, index|
|
|
144
|
+
ContextBudget.truncate_section(
|
|
145
|
+
scrub_text(comment),
|
|
146
|
+
limit: @config.max_jira_comment_chars,
|
|
147
|
+
label: "Jira comment #{index + 1}",
|
|
148
|
+
coverage: coverage
|
|
149
|
+
)
|
|
150
|
+
end
|
|
151
|
+
section << "\nRecent comments:\n#{rendered.join("\n")}"
|
|
88
152
|
end
|
|
89
153
|
|
|
90
|
-
def
|
|
91
|
-
|
|
92
|
-
|
|
154
|
+
def context_sizes(fixed:, packed:, budget:, diff_budget:, critique:)
|
|
155
|
+
stages = critique ? STAGES : [:generate]
|
|
156
|
+
{
|
|
157
|
+
context_budget: budget,
|
|
158
|
+
diff_budget: diff_budget,
|
|
159
|
+
sections: fixed.length,
|
|
160
|
+
diff: packed.text.length,
|
|
161
|
+
hunks_shown: packed.shown_hunks,
|
|
162
|
+
hunks_total: packed.total_hunks,
|
|
163
|
+
stages: stages.to_h do |stage|
|
|
164
|
+
system = system_prompt(stage).length
|
|
165
|
+
[stage, {system_prompt: system, request: system + fixed.length + packed.text.length,
|
|
166
|
+
max_prompt_chars: @config.max_prompt_chars(stage)}]
|
|
167
|
+
end
|
|
168
|
+
}
|
|
169
|
+
end
|
|
93
170
|
|
|
94
|
-
|
|
95
|
-
|
|
171
|
+
def log_sizes(sizes)
|
|
172
|
+
@logger.debug(
|
|
173
|
+
"Context: sections #{sizes[:sections]} chars, diff #{sizes[:diff]} chars " \
|
|
174
|
+
"(budget #{sizes[:diff_budget]}, hunks #{sizes[:hunks_shown]}/#{sizes[:hunks_total]})"
|
|
175
|
+
)
|
|
176
|
+
sizes[:stages].each do |stage, stage_sizes|
|
|
177
|
+
@logger.debug(
|
|
178
|
+
"Context #{stage}: request #{stage_sizes[:request]} chars (~#{stage_sizes[:request] / 4} tokens) " \
|
|
179
|
+
"of max #{stage_sizes[:max_prompt_chars]}, system prompt #{stage_sizes[:system_prompt]}"
|
|
180
|
+
)
|
|
181
|
+
end
|
|
96
182
|
end
|
|
97
183
|
|
|
98
184
|
def scrub_optional_text(text)
|
|
@@ -3,7 +3,80 @@ require_relative 'utils'
|
|
|
3
3
|
|
|
4
4
|
module Aireview
|
|
5
5
|
class DiffFetcher
|
|
6
|
-
|
|
6
|
+
NO_TEXT_CHANGES = '[no text changes]'
|
|
7
|
+
DIFF_UNAVAILABLE = '[diff not available]'
|
|
8
|
+
BINARY_DIFF = /\ABinary files .* differ/
|
|
9
|
+
|
|
10
|
+
# Один файл из ответа GitLab: заголовок, хунки и что с ним можно делать.
|
|
11
|
+
# kind:
|
|
12
|
+
# :text есть хунки, код можно проверить;
|
|
13
|
+
# :no_text_changes переименование, смена режима, пустой файл: проверять
|
|
14
|
+
# нечего;
|
|
15
|
+
# :unavailable GitLab не отдал дифф (too_large, бинарник, пустой
|
|
16
|
+
# дифф без причины): код есть, но проверить его не удалось.
|
|
17
|
+
class Entry
|
|
18
|
+
attr_reader :path, :kind, :header, :hunks
|
|
19
|
+
|
|
20
|
+
def initialize(change)
|
|
21
|
+
old_path = change['old_path'] || change['new_path']
|
|
22
|
+
new_path = change['new_path'] || change['old_path']
|
|
23
|
+
@path = new_path
|
|
24
|
+
@header = "diff --git a/#{old_path} b/#{new_path}\n--- a/#{old_path}\n+++ b/#{new_path}\n"
|
|
25
|
+
diff = change['diff'].to_s
|
|
26
|
+
@kind = classify(change, diff)
|
|
27
|
+
@hunks = @kind == :text ? split_hunks(diff) : []
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
def text?
|
|
31
|
+
kind == :text
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
def unavailable?
|
|
35
|
+
kind == :unavailable
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
def body
|
|
39
|
+
case kind
|
|
40
|
+
when :text then hunks.join
|
|
41
|
+
when :no_text_changes then "#{NO_TEXT_CHANGES}\n"
|
|
42
|
+
else "#{DIFF_UNAVAILABLE}\n"
|
|
43
|
+
end
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
def render
|
|
47
|
+
header + body
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
private
|
|
51
|
+
|
|
52
|
+
# Пустой дифф без объяснимой причины (переименование, смена режима,
|
|
53
|
+
# пустой новый или удалённый файл) считаем недоступным: код есть, но
|
|
54
|
+
# GitLab его не отдал.
|
|
55
|
+
def classify(change, diff)
|
|
56
|
+
return :unavailable if change['too_large']
|
|
57
|
+
return :unavailable if diff.match?(BINARY_DIFF)
|
|
58
|
+
return :text unless diff.strip.empty?
|
|
59
|
+
|
|
60
|
+
empty_diff_explained?(change) ? :no_text_changes : :unavailable
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
def empty_diff_explained?(change)
|
|
64
|
+
return true if %w[renamed_file new_file deleted_file].any? { |flag| change[flag] }
|
|
65
|
+
|
|
66
|
+
modes = change.values_at('a_mode', 'b_mode')
|
|
67
|
+
modes.none?(&:nil?) && modes.uniq.size == 2
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
# Хунки режем по заголовкам @@; текст до первого @@ (или дифф без них,
|
|
71
|
+
# например заглушка про секретный файл) считается одним хунком.
|
|
72
|
+
def split_hunks(diff)
|
|
73
|
+
diff = "#{diff}\n" unless diff.end_with?("\n")
|
|
74
|
+
pieces = diff.split(/^(?=@@ )/)
|
|
75
|
+
return pieces if pieces.size <= 1 || pieces.first.start_with?('@@ ')
|
|
76
|
+
|
|
77
|
+
[pieces[0] + pieces[1], *pieces[2..]]
|
|
78
|
+
end
|
|
79
|
+
end
|
|
7
80
|
|
|
8
81
|
def initialize(ignore_paths:, logger: Logger.new($stderr))
|
|
9
82
|
@ignore_paths = Array(ignore_paths).compact
|
|
@@ -16,8 +89,12 @@ module Aireview
|
|
|
16
89
|
end
|
|
17
90
|
end
|
|
18
91
|
|
|
92
|
+
def entries(changes)
|
|
93
|
+
Array(changes).map { |change| Entry.new(change) }
|
|
94
|
+
end
|
|
95
|
+
|
|
19
96
|
def render(changes)
|
|
20
|
-
|
|
97
|
+
entries(changes).map(&:render).join("\n")
|
|
21
98
|
end
|
|
22
99
|
|
|
23
100
|
private
|
|
@@ -29,19 +106,5 @@ module Aireview
|
|
|
29
106
|
File.fnmatch?(pattern, path, File::FNM_DOTMATCH | File::FNM_EXTGLOB)
|
|
30
107
|
end
|
|
31
108
|
end
|
|
32
|
-
|
|
33
|
-
def render_change(change)
|
|
34
|
-
old_path = change['old_path'] || change['new_path']
|
|
35
|
-
new_path = change['new_path'] || change['old_path']
|
|
36
|
-
diff = change['diff'].to_s
|
|
37
|
-
diff = DIFF_UNAVAILABLE if diff.strip.empty?
|
|
38
|
-
|
|
39
|
-
<<~DIFF
|
|
40
|
-
diff --git a/#{old_path} b/#{new_path}
|
|
41
|
-
--- a/#{old_path}
|
|
42
|
-
+++ b/#{new_path}
|
|
43
|
-
#{diff}
|
|
44
|
-
DIFF
|
|
45
|
-
end
|
|
46
109
|
end
|
|
47
110
|
end
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Aireview
|
|
4
|
+
# Вывод --dry-run: настройки, сводка контекста и промпты обеих стадий.
|
|
5
|
+
class DryRunReport
|
|
6
|
+
def initialize(out)
|
|
7
|
+
@out = out
|
|
8
|
+
end
|
|
9
|
+
|
|
10
|
+
def render(dry_run)
|
|
11
|
+
@out.puts('=== LLM SETTINGS ===')
|
|
12
|
+
@out.puts("Generate: #{dry_run[:generate_model]} temperature=#{dry_run[:generate_temperature]}")
|
|
13
|
+
if dry_run[:critique_prompt]
|
|
14
|
+
@out.puts("Critique: #{dry_run[:critique_model]} temperature=#{dry_run[:critique_temperature]}")
|
|
15
|
+
else
|
|
16
|
+
@out.puts('Critique: disabled')
|
|
17
|
+
end
|
|
18
|
+
@out.puts
|
|
19
|
+
@out.puts('=== CONTEXT ===')
|
|
20
|
+
render_context_sizes(dry_run[:sizes])
|
|
21
|
+
render_coverage(dry_run[:coverage])
|
|
22
|
+
@out.puts
|
|
23
|
+
@out.puts('=== GENERATE SYSTEM PROMPT ===')
|
|
24
|
+
@out.puts(dry_run.dig(:generate_prompt, :system_prompt))
|
|
25
|
+
@out.puts
|
|
26
|
+
@out.puts('=== GENERATE USER PROMPT ===')
|
|
27
|
+
@out.puts(dry_run.dig(:generate_prompt, :user_prompt))
|
|
28
|
+
return unless dry_run[:critique_prompt]
|
|
29
|
+
|
|
30
|
+
@out.puts
|
|
31
|
+
@out.puts('=== CRITIQUE SYSTEM PROMPT ===')
|
|
32
|
+
@out.puts(dry_run.dig(:critique_prompt, :system_prompt))
|
|
33
|
+
@out.puts
|
|
34
|
+
@out.puts('=== CRITIQUE USER PROMPT ===')
|
|
35
|
+
@out.puts(dry_run.dig(:critique_prompt, :user_prompt))
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
private
|
|
39
|
+
|
|
40
|
+
def render_context_sizes(sizes)
|
|
41
|
+
@out.puts("Sections: #{sizes[:sections]} chars, diff: #{sizes[:diff]} chars " \
|
|
42
|
+
"(budget #{sizes[:diff_budget]}, hunks #{sizes[:hunks_shown]}/#{sizes[:hunks_total]})")
|
|
43
|
+
sizes[:stages].each do |stage, stage_sizes|
|
|
44
|
+
@out.puts("#{stage.capitalize} request: #{stage_sizes[:request]} chars " \
|
|
45
|
+
"(~#{stage_sizes[:request] / 4} tokens) of max #{stage_sizes[:max_prompt_chars]}, " \
|
|
46
|
+
"system prompt #{stage_sizes[:system_prompt]}")
|
|
47
|
+
end
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
def render_coverage(coverage)
|
|
51
|
+
return @out.puts('Coverage: complete') if coverage.complete?
|
|
52
|
+
|
|
53
|
+
@out.puts('Coverage: partial')
|
|
54
|
+
list('truncated sections', coverage.truncated_sections)
|
|
55
|
+
list('files not shown', coverage.files_not_shown)
|
|
56
|
+
coverage.files_partial.each { |file| @out.puts(" #{file[:path]}: #{file[:shown]} of #{file[:total]} hunks") }
|
|
57
|
+
list('diff not available', coverage.files_unavailable)
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
def list(title, items)
|
|
61
|
+
@out.puts(" #{title}: #{items.join(', ')}") unless items.empty?
|
|
62
|
+
end
|
|
63
|
+
end
|
|
64
|
+
end
|
data/lib/aireview/errors.rb
CHANGED
|
@@ -4,6 +4,10 @@ You are given the diff, the MR description, the Jira context and the candidates
|
|
|
4
4
|
from the first pass. For every candidate return a verdict with decision=keep
|
|
5
5
|
or reject.
|
|
6
6
|
|
|
7
|
+
Parts of the context may be truncated to fit the budget; markers in square
|
|
8
|
+
brackets show where. Missing text in a truncated part does not mean a missing
|
|
9
|
+
requirement or missing code.
|
|
10
|
+
|
|
7
11
|
Be a strict filter:
|
|
8
12
|
- when in doubt, choose reject;
|
|
9
13
|
- keep only the most important and well-supported findings;
|
|
@@ -23,6 +27,12 @@ Implementation details (config flags, deploy settings, new fields, helper
|
|
|
23
27
|
methods), accompanying refactoring and any changes whose relation to the task
|
|
24
28
|
is plausible are NOT task_mismatch. Reject such candidates.
|
|
25
29
|
|
|
30
|
+
Do not verify that the specified versions of packages, libraries, tools and
|
|
31
|
+
image tags exist. Do not report that a version does not exist or has not been
|
|
32
|
+
released yet. Check syntax errors and explicit contradictions with the MR/Jira
|
|
33
|
+
requirements as usual.
|
|
34
|
+
Reject candidates claiming that a version does not exist.
|
|
35
|
+
|
|
26
36
|
Reject a candidate if it:
|
|
27
37
|
- is not confirmed by the diff or Jira;
|
|
28
38
|
- is based on an assumption about code outside the diff;
|
|
@@ -3,6 +3,15 @@ You are doing the first pass of a merge request review.
|
|
|
3
3
|
Look only at the diff, the MR description and the Jira context, if present.
|
|
4
4
|
Do not draw conclusions about code outside the diff.
|
|
5
5
|
|
|
6
|
+
Parts of the context may be truncated to fit the budget; markers in square
|
|
7
|
+
brackets show where. Missing text in a truncated part does not mean a missing
|
|
8
|
+
requirement or missing code.
|
|
9
|
+
|
|
10
|
+
Do not verify that the specified versions of packages, libraries, tools and
|
|
11
|
+
image tags exist. Do not report that a version does not exist or has not been
|
|
12
|
+
released yet. Check syntax errors and explicit contradictions with the MR/Jira
|
|
13
|
+
requirements as usual.
|
|
14
|
+
|
|
6
15
|
Look only for the most important and well-supported findings:
|
|
7
16
|
- bugs, regressions, data loss;
|
|
8
17
|
- security and performance risks;
|
|
@@ -28,12 +28,14 @@ module Aireview
|
|
|
28
28
|
@logger = logger
|
|
29
29
|
end
|
|
30
30
|
|
|
31
|
-
def run(merge_request:,
|
|
32
|
-
|
|
31
|
+
def run(merge_request:, changes:, jira_issue: nil, critique: true)
|
|
32
|
+
context = @context_builder.prepare(
|
|
33
33
|
merge_request: merge_request,
|
|
34
|
-
|
|
35
|
-
jira_issue: jira_issue
|
|
34
|
+
changes: changes,
|
|
35
|
+
jira_issue: jira_issue,
|
|
36
|
+
critique: critique
|
|
36
37
|
)
|
|
38
|
+
generate_prompt = @context_builder.build_generate_prompt(context)
|
|
37
39
|
@logger.info("Pipeline generate pass started (model=#{@config.generate_model})")
|
|
38
40
|
candidates_raw = @reviewer.generate(**generate_prompt)
|
|
39
41
|
generate_result = parse_with_repair(
|
|
@@ -48,12 +50,7 @@ module Aireview
|
|
|
48
50
|
|
|
49
51
|
accepted = if critique
|
|
50
52
|
@logger.info("Pipeline critique pass started (model=#{@config.critique_model})")
|
|
51
|
-
critique_candidates(
|
|
52
|
-
merge_request: merge_request,
|
|
53
|
-
changes_text: changes_text,
|
|
54
|
-
jira_issue: jira_issue,
|
|
55
|
-
candidates: candidates
|
|
56
|
-
)
|
|
53
|
+
critique_candidates(context: context, candidates: candidates)
|
|
57
54
|
else
|
|
58
55
|
@logger.info('Pipeline critique pass skipped')
|
|
59
56
|
candidates
|
|
@@ -61,25 +58,22 @@ module Aireview
|
|
|
61
58
|
|
|
62
59
|
@logger.info("Pipeline finished with #{accepted.size} accepted finding(s)")
|
|
63
60
|
|
|
64
|
-
ReviewRenderer.new(language: @config.review_language)
|
|
61
|
+
renderer = ReviewRenderer.new(language: @config.review_language)
|
|
62
|
+
renderer.render(accepted, summary: summary, coverage: context.coverage)
|
|
65
63
|
end
|
|
66
64
|
|
|
67
|
-
def dry_run_prompts(merge_request:,
|
|
65
|
+
def dry_run_prompts(merge_request:, changes:, jira_issue: nil, critique: true)
|
|
68
66
|
@config.require_models!
|
|
69
67
|
|
|
70
|
-
|
|
68
|
+
context = @context_builder.prepare(
|
|
71
69
|
merge_request: merge_request,
|
|
72
|
-
|
|
73
|
-
jira_issue: jira_issue
|
|
70
|
+
changes: changes,
|
|
71
|
+
jira_issue: jira_issue,
|
|
72
|
+
critique: critique
|
|
74
73
|
)
|
|
75
|
-
|
|
74
|
+
generate_prompt = @context_builder.build_generate_prompt(context)
|
|
76
75
|
critique_prompt = if critique
|
|
77
|
-
@context_builder.build_critique_prompt(
|
|
78
|
-
merge_request: merge_request,
|
|
79
|
-
changes_text: changes_text,
|
|
80
|
-
jira_issue: jira_issue,
|
|
81
|
-
candidates_json: DRY_RUN_CANDIDATES_JSON
|
|
82
|
-
)
|
|
76
|
+
@context_builder.build_critique_prompt(context, candidates_json: DRY_RUN_CANDIDATES_JSON)
|
|
83
77
|
end
|
|
84
78
|
|
|
85
79
|
{
|
|
@@ -88,21 +82,18 @@ module Aireview
|
|
|
88
82
|
generate_model: @config.generate_model,
|
|
89
83
|
generate_temperature: @config.generate_temperature,
|
|
90
84
|
critique_model: @config.critique_model,
|
|
91
|
-
critique_temperature: @config.critique_temperature
|
|
85
|
+
critique_temperature: @config.critique_temperature,
|
|
86
|
+
coverage: context.coverage,
|
|
87
|
+
sizes: context.sizes
|
|
92
88
|
}
|
|
93
89
|
end
|
|
94
90
|
|
|
95
91
|
private
|
|
96
92
|
|
|
97
|
-
def critique_candidates(
|
|
93
|
+
def critique_candidates(context:, candidates:)
|
|
98
94
|
candidates_json = JSON.pretty_generate(candidates)
|
|
99
95
|
candidates_by_id = index_candidates_by_id(candidates)
|
|
100
|
-
critique_prompt = @context_builder.build_critique_prompt(
|
|
101
|
-
merge_request: merge_request,
|
|
102
|
-
changes_text: changes_text,
|
|
103
|
-
jira_issue: jira_issue,
|
|
104
|
-
candidates_json: candidates_json
|
|
105
|
-
)
|
|
96
|
+
critique_prompt = @context_builder.build_critique_prompt(context, candidates_json: candidates_json)
|
|
106
97
|
critique_raw = @reviewer.critique(**critique_prompt)
|
|
107
98
|
critique_result = parse_with_repair(
|
|
108
99
|
raw: critique_raw,
|
|
@@ -284,10 +275,11 @@ module Aireview
|
|
|
284
275
|
end
|
|
285
276
|
|
|
286
277
|
@logger.info("Pipeline #{stage} repair started for #{kind}")
|
|
278
|
+
prompt = @context_builder.check_stage_size!(stage, REPAIR_SYSTEM_PROMPT, user_prompt)
|
|
287
279
|
if stage == :critique
|
|
288
|
-
@reviewer.critique(
|
|
280
|
+
@reviewer.critique(**prompt)
|
|
289
281
|
else
|
|
290
|
-
@reviewer.generate(
|
|
282
|
+
@reviewer.generate(**prompt)
|
|
291
283
|
end
|
|
292
284
|
end
|
|
293
285
|
|
|
@@ -35,7 +35,16 @@ module Aireview
|
|
|
35
35
|
where: 'Where',
|
|
36
36
|
problem: 'Problem',
|
|
37
37
|
why: 'Why it matters',
|
|
38
|
-
suggestion: 'Suggestion'
|
|
38
|
+
suggestion: 'Suggestion',
|
|
39
|
+
partial: 'Partial review',
|
|
40
|
+
not_reviewed: 'Not reviewed',
|
|
41
|
+
files_not_shown: 'files not reviewed',
|
|
42
|
+
files_partial: 'files reviewed partially',
|
|
43
|
+
files_unavailable: 'files without an available diff',
|
|
44
|
+
sections_truncated: 'sections truncated',
|
|
45
|
+
hunks_of: 'hunks shown',
|
|
46
|
+
diff_unavailable: 'diff not available',
|
|
47
|
+
section_list: 'Truncated sections'
|
|
39
48
|
},
|
|
40
49
|
'ru' => {
|
|
41
50
|
summary: 'Сводка',
|
|
@@ -50,7 +59,16 @@ module Aireview
|
|
|
50
59
|
where: 'Где',
|
|
51
60
|
problem: 'Проблема',
|
|
52
61
|
why: 'Почему важно',
|
|
53
|
-
suggestion: 'Предложение'
|
|
62
|
+
suggestion: 'Предложение',
|
|
63
|
+
partial: 'Ревью частичное',
|
|
64
|
+
not_reviewed: 'Не вошло в ревью',
|
|
65
|
+
files_not_shown: 'файлов не проверено',
|
|
66
|
+
files_partial: 'файлов проверено частично',
|
|
67
|
+
files_unavailable: 'файлов без доступного диффа',
|
|
68
|
+
sections_truncated: 'секций усечено',
|
|
69
|
+
hunks_of: 'хунков показано',
|
|
70
|
+
diff_unavailable: 'дифф недоступен',
|
|
71
|
+
section_list: 'Усечённые секции'
|
|
54
72
|
}
|
|
55
73
|
}.freeze
|
|
56
74
|
|
|
@@ -58,7 +76,11 @@ module Aireview
|
|
|
58
76
|
@labels = LABELS.fetch(language.to_s) { LABELS.fetch(DEFAULT_LANGUAGE) }
|
|
59
77
|
end
|
|
60
78
|
|
|
61
|
-
|
|
79
|
+
# coverage: факты усечения контекста от пайплайна, не текст модели.
|
|
80
|
+
# result по-прежнему про найденные проблемы; неполнота покрытия
|
|
81
|
+
# дописывается рядом с ним, чтобы строка результата не читалась как
|
|
82
|
+
# «проверено всё».
|
|
83
|
+
def render(accepted, summary:, coverage: nil)
|
|
62
84
|
findings = sorted_findings(Array(accepted)).first(TOTAL_FINDINGS_LIMIT)
|
|
63
85
|
mismatches = findings.select { |finding| category(finding) == 'task_mismatch' }.first(MISMATCH_LIMIT)
|
|
64
86
|
important = findings.select { |finding| important_finding?(finding) }.first(IMPORTANT_LIMIT)
|
|
@@ -79,8 +101,8 @@ module Aireview
|
|
|
79
101
|
|
|
80
102
|
## #{label(:result)}
|
|
81
103
|
|
|
82
|
-
#{result}
|
|
83
|
-
|
|
104
|
+
#{result}#{partial_note(coverage)}
|
|
105
|
+
#{coverage_block(coverage)}
|
|
84
106
|
#{label(:disclaimer)}
|
|
85
107
|
MARKDOWN
|
|
86
108
|
end
|
|
@@ -117,6 +139,34 @@ module Aireview
|
|
|
117
139
|
end.join("\n\n")
|
|
118
140
|
end
|
|
119
141
|
|
|
142
|
+
def partial_note(coverage)
|
|
143
|
+
return '' if coverage.nil? || coverage.complete?
|
|
144
|
+
|
|
145
|
+
counts = {
|
|
146
|
+
files_not_shown: coverage.files_not_shown.size,
|
|
147
|
+
files_partial: coverage.files_partial.size,
|
|
148
|
+
files_unavailable: coverage.files_unavailable.size,
|
|
149
|
+
sections_truncated: coverage.truncated_sections.size
|
|
150
|
+
}.reject { |_, count| count.zero? }.map { |key, count| "#{count} #{label(key)}" }
|
|
151
|
+
|
|
152
|
+
". #{label(:partial)}: #{counts.join(', ')}."
|
|
153
|
+
end
|
|
154
|
+
|
|
155
|
+
def coverage_block(coverage)
|
|
156
|
+
return '' if coverage.nil? || coverage.complete?
|
|
157
|
+
|
|
158
|
+
lines = coverage.files_not_shown.map { |path| "- #{path}" }
|
|
159
|
+
lines += coverage.files_partial.map do |file|
|
|
160
|
+
"- #{file[:path]}: #{file[:shown]}/#{file[:total]} #{label(:hunks_of)}"
|
|
161
|
+
end
|
|
162
|
+
lines += coverage.files_unavailable.map { |path| "- #{path}: #{label(:diff_unavailable)}" }
|
|
163
|
+
unless coverage.truncated_sections.empty?
|
|
164
|
+
lines << "- #{label(:section_list)}: #{coverage.truncated_sections.join(', ')}"
|
|
165
|
+
end
|
|
166
|
+
|
|
167
|
+
"\n## #{label(:not_reviewed)}\n\n#{lines.join("\n")}\n"
|
|
168
|
+
end
|
|
169
|
+
|
|
120
170
|
def location(finding)
|
|
121
171
|
file = presence(value(finding, 'file'))
|
|
122
172
|
line = value(finding, 'line')
|