ast-merge-git 7.1.1 → 7.1.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- checksums.yaml.gz.sig +0 -0
- data/README.md +24 -14
- data/exe/ast-merge-git +149 -0
- data/lib/ast/merge/git/corpus.rb +580 -0
- data/lib/ast/merge/git/local_benchmark.rb +1199 -0
- data/lib/ast/merge/git/version.rb +1 -1
- data/lib/ast/merge/git.rb +196 -343
- data/sig/ast/merge/git.rbs +106 -0
- data.tar.gz.sig +0 -0
- metadata +26 -37
- metadata.gz.sig +3 -2
|
@@ -0,0 +1,1199 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require 'digest'
|
|
4
|
+
require 'fileutils'
|
|
5
|
+
require 'json'
|
|
6
|
+
require 'open3'
|
|
7
|
+
require 'rbconfig'
|
|
8
|
+
|
|
9
|
+
module Ast
|
|
10
|
+
module Merge
|
|
11
|
+
module Git
|
|
12
|
+
# Offline Slice 1022-compatible corpus validation and deterministic selection.
|
|
13
|
+
# rubocop:disable Metrics/AbcSize, Metrics/ClassLength, Metrics/CyclomaticComplexity, Metrics/MethodLength, Metrics/PerceivedComplexity -- benchmark evidence is intentionally explicit
|
|
14
|
+
class LocalBenchmark
|
|
15
|
+
Error = Class.new(StandardError)
|
|
16
|
+
SCHEMA = 'structuredmerge.benchmark.corpus/v1'
|
|
17
|
+
CASE_SCHEMA = 'structuredmerge.benchmark/v1'
|
|
18
|
+
OPERATIONS = %w[merge3 metamorphic diff].freeze
|
|
19
|
+
EXECUTABLE_OPERATIONS = %w[merge3 metamorphic].freeze
|
|
20
|
+
PARTITIONS = %w[sentinel gold metamorphic].freeze
|
|
21
|
+
EXPECTATIONS = %w[clean conflict error excluded_ambiguous].freeze
|
|
22
|
+
SEVERITIES = %w[none low high critical].freeze
|
|
23
|
+
PRESERVATION = %w[required allowed_to_change not_applicable].freeze
|
|
24
|
+
TRANSFORMATIONS = %w[
|
|
25
|
+
rename move reorder formatting comment independent_edit delete_modify duplicate_key_identity
|
|
26
|
+
schema_aware_mutation
|
|
27
|
+
].freeze
|
|
28
|
+
METAMORPHIC_INVARIANTS = %w[
|
|
29
|
+
comment-retained no-semantic-edit same-json-value same-jsonc-value
|
|
30
|
+
].freeze
|
|
31
|
+
ID_PATTERN = /\A[a-z0-9]+(?:[.-][a-z0-9]+)*\z/
|
|
32
|
+
PROVENANCE_FIELDS = %w[
|
|
33
|
+
origin_uri revision spdx_license license_evidence_uri authorship author_review reviewer derivation
|
|
34
|
+
].freeze
|
|
35
|
+
SELECTOR_FIELDS = %w[provider_id family dialect backend profile require].freeze
|
|
36
|
+
INPUT_ROLES = {
|
|
37
|
+
'merge3' => %w[base ours theirs],
|
|
38
|
+
'metamorphic' => %w[source transformed],
|
|
39
|
+
'diff' => %w[before after]
|
|
40
|
+
}.freeze
|
|
41
|
+
|
|
42
|
+
attr_reader :document, :corpus_digest, :path
|
|
43
|
+
|
|
44
|
+
def self.load(path)
|
|
45
|
+
source = File.binread(path)
|
|
46
|
+
new(JSON.parse(source), path: path, corpus_digest: Digest::SHA256.hexdigest(source)).tap(&:validate!)
|
|
47
|
+
rescue JSON::ParserError => e
|
|
48
|
+
raise Error, "invalid corpus JSON: #{e.message}"
|
|
49
|
+
rescue SystemCallError => e
|
|
50
|
+
raise Error, "cannot read corpus: #{e.message}"
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
def initialize(document, path: nil, corpus_digest: nil)
|
|
54
|
+
@document = document
|
|
55
|
+
@path = path && Pathname(path).expand_path
|
|
56
|
+
@corpus_digest = corpus_digest || digest(canonical_json(document))
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
def validate!
|
|
60
|
+
require_keys(document, %w[schema_version kind id version extends provenance profiles capability_map
|
|
61
|
+
selection competitors cases expected_summary], 'corpus')
|
|
62
|
+
error!('unsupported corpus schema') unless document['schema_version'] == SCHEMA
|
|
63
|
+
error!('corpus must extend Slice 1022 v1') unless document['extends'] == CASE_SCHEMA
|
|
64
|
+
error!('network must be denied') unless document['network_policy'] == 'denied'
|
|
65
|
+
error!('services must be empty') unless document['services'] == []
|
|
66
|
+
validate_provenance!(document['provenance'], 'corpus')
|
|
67
|
+
validate_profiles!
|
|
68
|
+
validate_capability_map!
|
|
69
|
+
validate_competitors!
|
|
70
|
+
validate_cases!
|
|
71
|
+
validate_summary!
|
|
72
|
+
true
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
def cases
|
|
76
|
+
validate!
|
|
77
|
+
document.fetch('cases')
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
def select(profile:, changed_paths: [])
|
|
81
|
+
validate!
|
|
82
|
+
profile = profile.to_s
|
|
83
|
+
definition = document.fetch('profiles')[profile]
|
|
84
|
+
error!("unknown profile: #{profile}") unless definition
|
|
85
|
+
paths = changed_paths.map(&:to_s).uniq.sort
|
|
86
|
+
inferred = paths.to_h { |changed| [changed, capabilities_for(changed)] }
|
|
87
|
+
capabilities = inferred.values.flatten.uniq.sort
|
|
88
|
+
direct = cases.select { |item| direct_case?(item, capabilities) }.map { |item| item.fetch('id') }
|
|
89
|
+
sentinels = definition.fetch('mandatory_sentinels')
|
|
90
|
+
selected = ordered(sentinels + direct)
|
|
91
|
+
population = cases.map { |item| item.fetch('id') } - selected
|
|
92
|
+
neighbors = neighbor_order(population).first(definition.fetch('neighbor_count'))
|
|
93
|
+
selected = ordered(selected + neighbors)
|
|
94
|
+
selected = sentinels if profile == 'micro'
|
|
95
|
+
|
|
96
|
+
{
|
|
97
|
+
'profile' => profile,
|
|
98
|
+
'seed' => document.dig('selection', 'seed'),
|
|
99
|
+
'selected_case_ids' => selected,
|
|
100
|
+
'excluded_case_ids' => cases.map { |item| item.fetch('id') } - selected,
|
|
101
|
+
'changed_paths' => inferred.map { |changed, caps| { 'path' => changed, 'capabilities' => caps } },
|
|
102
|
+
'inferred_capabilities' => capabilities,
|
|
103
|
+
'direct_cases' => direct,
|
|
104
|
+
'direct_case_reasons' => capabilities.to_h do |capability|
|
|
105
|
+
matching = cases.filter_map do |item|
|
|
106
|
+
item['id'] if direct_case?(item, [capability])
|
|
107
|
+
end
|
|
108
|
+
[capability, matching]
|
|
109
|
+
end,
|
|
110
|
+
'sentinels' => sentinels,
|
|
111
|
+
'neighbor_sample' => {
|
|
112
|
+
'population' => population,
|
|
113
|
+
'ordering_algorithm' => document.dig('selection', 'neighbor_order'),
|
|
114
|
+
'seed' => document.dig('selection', 'seed'),
|
|
115
|
+
'selected_case_ids' => neighbors
|
|
116
|
+
},
|
|
117
|
+
'unsupported_selected_cases' => selected.reject do |id|
|
|
118
|
+
EXECUTABLE_OPERATIONS.include?(case_by_id(id)['operation'])
|
|
119
|
+
end,
|
|
120
|
+
'budgets' => definition.fetch('budgets'),
|
|
121
|
+
'explanation' => explanation(
|
|
122
|
+
profile,
|
|
123
|
+
inferred: inferred,
|
|
124
|
+
direct: direct,
|
|
125
|
+
sentinels: sentinels,
|
|
126
|
+
population: population,
|
|
127
|
+
neighbors: neighbors
|
|
128
|
+
)
|
|
129
|
+
}
|
|
130
|
+
end
|
|
131
|
+
|
|
132
|
+
def case_by_id(id)
|
|
133
|
+
document.fetch('cases').find { |item| item['id'] == id } || error!("unknown case: #{id}")
|
|
134
|
+
end
|
|
135
|
+
|
|
136
|
+
private
|
|
137
|
+
|
|
138
|
+
def validate_profiles!
|
|
139
|
+
error!('profiles must be exactly micro and dev') unless document['profiles'].keys.sort == %w[dev micro]
|
|
140
|
+
document['profiles'].each do |name, profile|
|
|
141
|
+
require_keys(profile, %w[mandatory_sentinels neighbor_count budgets], "profile #{name}")
|
|
142
|
+
require_keys(profile['budgets'], %w[wall_seconds case_count output_bytes], "profile #{name} budgets")
|
|
143
|
+
end
|
|
144
|
+
end
|
|
145
|
+
|
|
146
|
+
def validate_capability_map!
|
|
147
|
+
error!('capability_map must not be empty') unless document['capability_map'].is_a?(Array) &&
|
|
148
|
+
document['capability_map'].any?
|
|
149
|
+
document['capability_map'].each do |entry|
|
|
150
|
+
require_keys(entry, %w[path_prefix capabilities], 'capability map entry')
|
|
151
|
+
error!('capability map path must be relative') if Pathname(entry['path_prefix']).absolute?
|
|
152
|
+
unless entry['capabilities'] == entry['capabilities'].sort
|
|
153
|
+
error!('capability map capabilities must be sorted')
|
|
154
|
+
end
|
|
155
|
+
end
|
|
156
|
+
end
|
|
157
|
+
|
|
158
|
+
def validate_competitors!
|
|
159
|
+
document.fetch('competitors').each do |id, competitor|
|
|
160
|
+
require_keys(
|
|
161
|
+
competitor,
|
|
162
|
+
%w[adapter_id source_url source_revision version spdx_license reuse_posture toolchain
|
|
163
|
+
build_command operations dialects],
|
|
164
|
+
"competitor #{id}"
|
|
165
|
+
)
|
|
166
|
+
error!("competitor #{id}: adapter ID differs") unless competitor['adapter_id'] == id
|
|
167
|
+
unless /\A[0-9a-f]{40}\z/.match?(competitor['source_revision'])
|
|
168
|
+
error!("competitor #{id}: source revision must be a full SHA")
|
|
169
|
+
end
|
|
170
|
+
error!("competitor #{id}: only merge3 is admitted") unless competitor['operations'] == ['merge3']
|
|
171
|
+
next if competitor['dialects'] == competitor['dialects'].sort
|
|
172
|
+
|
|
173
|
+
error!("competitor #{id}: dialects must be sorted")
|
|
174
|
+
end
|
|
175
|
+
end
|
|
176
|
+
|
|
177
|
+
def validate_cases!
|
|
178
|
+
records = document['cases']
|
|
179
|
+
error!('cases must not be empty') unless records.is_a?(Array) && records.any?
|
|
180
|
+
ids = records.map { |item| validate_case!(item) }
|
|
181
|
+
error!('duplicate case ID') unless ids.uniq.length == ids.length
|
|
182
|
+
id_set = ids.to_h { |id| [id, true] }
|
|
183
|
+
records.select { |item| item['operation'] == 'metamorphic' }.each do |item|
|
|
184
|
+
error!("#{item['id']}: parent case is dangling") unless id_set[item['parent_case_id']]
|
|
185
|
+
end
|
|
186
|
+
sentinels = records.select { |item| item['partition'] == 'sentinel' }.map { |item| item['id'] }
|
|
187
|
+
document['profiles'].each_value do |profile|
|
|
188
|
+
error!('profile sentinels differ from corpus sentinels') unless profile['mandatory_sentinels'] == sentinels
|
|
189
|
+
end
|
|
190
|
+
end
|
|
191
|
+
|
|
192
|
+
def validate_case!(item)
|
|
193
|
+
require_keys(item, %w[schema_version kind id operation family provider dialect capabilities partition
|
|
194
|
+
provenance oracle acceptable_equivalence preservation_policy
|
|
195
|
+
false_auto_merge_severity selector inputs independent_edits independent_edit_ids
|
|
196
|
+
expected_conflict_regions], 'case')
|
|
197
|
+
id = item['id']
|
|
198
|
+
error!("#{id}: invalid stable case ID") unless ID_PATTERN.match?(id.to_s)
|
|
199
|
+
error!("#{id}: incompatible case schema") unless item['schema_version'] == CASE_SCHEMA
|
|
200
|
+
error!("#{id}: invalid kind") unless item['kind'] == 'benchmark_case'
|
|
201
|
+
error!("#{id}: unsupported operation") unless OPERATIONS.include?(item['operation'])
|
|
202
|
+
error!("#{id}: unsupported partition") unless PARTITIONS.include?(item['partition'])
|
|
203
|
+
unless item['capabilities'] == item['capabilities'].uniq.sort
|
|
204
|
+
error!("#{id}: capabilities must be unique and sorted")
|
|
205
|
+
end
|
|
206
|
+
error!("#{id}: unsupported severity") unless SEVERITIES.include?(item['false_auto_merge_severity'])
|
|
207
|
+
validate_provenance!(item['provenance'], id)
|
|
208
|
+
validate_oracle!(item, id)
|
|
209
|
+
validate_selector!(item['selector'], id)
|
|
210
|
+
validate_inputs!(item, id)
|
|
211
|
+
validate_edits!(item, id)
|
|
212
|
+
validate_preservation!(item, id)
|
|
213
|
+
validate_operation!(item, id)
|
|
214
|
+
id
|
|
215
|
+
end
|
|
216
|
+
|
|
217
|
+
def validate_provenance!(provenance, label)
|
|
218
|
+
require_keys(provenance, PROVENANCE_FIELDS, "#{label} provenance")
|
|
219
|
+
error!("#{label}: authorship must be reviewed") unless provenance['author_review'] == 'reviewed'
|
|
220
|
+
error!("#{label}: SPDX license is required") if provenance['spdx_license'].to_s.empty?
|
|
221
|
+
end
|
|
222
|
+
|
|
223
|
+
def validate_oracle!(item, id)
|
|
224
|
+
oracle = item['oracle']
|
|
225
|
+
require_keys(oracle, %w[class artifact admission score_eligible procedure], "#{id} oracle")
|
|
226
|
+
validate_inline!(oracle['artifact'], "#{id} oracle artifact")
|
|
227
|
+
error!("#{id}: oracle procedure cannot accept parse validity alone") if oracle['procedure'].to_s.empty?
|
|
228
|
+
error!("#{id}: acceptable equivalence must be explicit") unless item['acceptable_equivalence'].is_a?(Array) &&
|
|
229
|
+
item['acceptable_equivalence'].any?
|
|
230
|
+
return unless oracle['class'] == 'exact' && item.dig('expected', 'outcome') == 'clean'
|
|
231
|
+
return if oracle.dig('artifact', 'bytes') == item.dig('expected', 'output', 'bytes')
|
|
232
|
+
|
|
233
|
+
error!("#{id}: exact oracle artifact must match expected output bytes")
|
|
234
|
+
end
|
|
235
|
+
|
|
236
|
+
def validate_selector!(selector, id)
|
|
237
|
+
require_keys(selector, SELECTOR_FIELDS, "#{id} selector")
|
|
238
|
+
end
|
|
239
|
+
|
|
240
|
+
def validate_inputs!(item, id)
|
|
241
|
+
roles = INPUT_ROLES.fetch(item['operation'])
|
|
242
|
+
require_keys(item['inputs'], roles, "#{id} inputs")
|
|
243
|
+
item['inputs'].each { |role, record| validate_inline!(record, "#{id} #{role}") }
|
|
244
|
+
end
|
|
245
|
+
|
|
246
|
+
def validate_inline!(record, label)
|
|
247
|
+
require_keys(record, %w[mode bytes sha256], label)
|
|
248
|
+
error!("#{label}: only inline authored evidence is admitted") unless record['mode'] == 'inline'
|
|
249
|
+
error!("#{label}: input exceeds Slice 1022 inline limit") if record['bytes'].bytesize > 4096
|
|
250
|
+
error!("#{label}: SHA-256 does not match exact bytes") unless record['sha256'] == digest(record['bytes'])
|
|
251
|
+
end
|
|
252
|
+
|
|
253
|
+
def validate_edits!(item, id)
|
|
254
|
+
edits = item['independent_edits']
|
|
255
|
+
error!("#{id}: independent edits must be an array") unless edits.is_a?(Array)
|
|
256
|
+
edit_ids = edits.map { |edit| edit.fetch('id') }
|
|
257
|
+
error!("#{id}: duplicate independent edit ID") unless edit_ids.uniq.length == edit_ids.length
|
|
258
|
+
error!("#{id}: independent edit IDs differ") unless item['independent_edit_ids'] == edit_ids
|
|
259
|
+
end
|
|
260
|
+
|
|
261
|
+
def validate_preservation!(item, id)
|
|
262
|
+
required = %w[comments formatting order encoding line_endings unknown_fields source_regions]
|
|
263
|
+
require_keys(item['preservation_policy'], required, "#{id} preservation")
|
|
264
|
+
return if item['preservation_policy'].values.all? { |value| PRESERVATION.include?(value) }
|
|
265
|
+
|
|
266
|
+
error!("#{id}: invalid preservation requirement")
|
|
267
|
+
end
|
|
268
|
+
|
|
269
|
+
def validate_operation!(item, id)
|
|
270
|
+
if item['operation'] == 'merge3'
|
|
271
|
+
require_keys(item, %w[expected expected_conflicts], id)
|
|
272
|
+
expectation = item.dig('expected', 'outcome')
|
|
273
|
+
error!("#{id}: unsupported expected outcome") unless EXPECTATIONS.include?(expectation)
|
|
274
|
+
output = item.dig('expected', 'output')
|
|
275
|
+
validate_inline!(output, "#{id} expected output") if output
|
|
276
|
+
expected_conflict = expectation == 'conflict'
|
|
277
|
+
error!("#{id}: conflict expectation mismatch") unless item['expected_conflicts'] == expected_conflict
|
|
278
|
+
elsif item['operation'] == 'metamorphic'
|
|
279
|
+
validate_metamorphic!(item, id)
|
|
280
|
+
end
|
|
281
|
+
end
|
|
282
|
+
|
|
283
|
+
def validate_metamorphic!(item, id)
|
|
284
|
+
require_keys(item, %w[generator parent_case_id transformations expected_invariants], id)
|
|
285
|
+
require_keys(item['generator'], %w[id version sha256 seed], "#{id} generator")
|
|
286
|
+
digest = item.dig('generator', 'sha256')
|
|
287
|
+
error!("#{id}: generator SHA-256 is malformed") unless /\A[0-9a-f]{64}\z/.match?(digest)
|
|
288
|
+
error!("#{id}: expected invariants must not be empty") if item['expected_invariants'].empty?
|
|
289
|
+
unknown_invariants = item['expected_invariants'] - METAMORPHIC_INVARIANTS
|
|
290
|
+
error!("#{id}: unsupported expected invariants: #{unknown_invariants.join(', ')}") if unknown_invariants.any?
|
|
291
|
+
item['transformations'].each do |transformation|
|
|
292
|
+
require_keys(transformation, %w[id type parameters], "#{id} transformation")
|
|
293
|
+
error!("#{id}: unknown transformation") unless TRANSFORMATIONS.include?(transformation['type'])
|
|
294
|
+
next if transformation.dig('parameters', 'deterministic')
|
|
295
|
+
|
|
296
|
+
error!("#{id}: transformation must be deterministic")
|
|
297
|
+
end
|
|
298
|
+
end
|
|
299
|
+
|
|
300
|
+
def validate_summary!
|
|
301
|
+
summary = document['expected_summary']
|
|
302
|
+
error!('expected case count differs') unless summary['case_count'] == document['cases'].length
|
|
303
|
+
actual = document['cases'].group_by { |item| item['partition'] }.transform_values(&:length)
|
|
304
|
+
error!('expected partition counts differ') unless summary['partition_counts'] == actual
|
|
305
|
+
operations = document['cases'].group_by { |item| item['operation'] }.transform_values(&:length)
|
|
306
|
+
error!('expected operation counts differ') unless summary['operation_counts'] == operations
|
|
307
|
+
families = document['cases'].map { |item| item['family'] }.uniq.sort
|
|
308
|
+
error!('expected families differ') unless summary['families'] == families
|
|
309
|
+
expected_micro = document.dig('profiles', 'micro', 'mandatory_sentinels')
|
|
310
|
+
error!('expected micro case IDs differ') unless summary['micro_case_ids'] == expected_micro
|
|
311
|
+
end
|
|
312
|
+
|
|
313
|
+
def capabilities_for(path)
|
|
314
|
+
document['capability_map'].filter_map do |entry|
|
|
315
|
+
entry['capabilities'] if path.start_with?(entry['path_prefix'])
|
|
316
|
+
end.flatten.uniq.sort
|
|
317
|
+
end
|
|
318
|
+
|
|
319
|
+
def direct_case?(item, capabilities)
|
|
320
|
+
case_capabilities = item['capabilities'] + [item['family'], item['dialect']]
|
|
321
|
+
(case_capabilities & capabilities).any?
|
|
322
|
+
end
|
|
323
|
+
|
|
324
|
+
def ordered(ids)
|
|
325
|
+
order = document['cases'].map { |item| item['id'] }
|
|
326
|
+
ids.uniq.sort_by { |id| order.index(id) }
|
|
327
|
+
end
|
|
328
|
+
|
|
329
|
+
def neighbor_order(ids)
|
|
330
|
+
seed = document.dig('selection', 'seed')
|
|
331
|
+
ids.sort_by { |id| [digest("#{seed}\0#{id}"), id] }
|
|
332
|
+
end
|
|
333
|
+
|
|
334
|
+
def explanation(profile, details)
|
|
335
|
+
{
|
|
336
|
+
'profile_rule' => profile == 'micro' ? 'mandatory sentinels only' : 'sentinels + direct + neighbors',
|
|
337
|
+
'changed_paths' => details[:inferred].map { |path, caps| "#{path} => #{caps.join(',')}" },
|
|
338
|
+
'direct_cases' => details[:direct],
|
|
339
|
+
'sentinels' => details[:sentinels],
|
|
340
|
+
'neighbors' => {
|
|
341
|
+
'population' => details[:population],
|
|
342
|
+
'algorithm' => document.dig('selection', 'neighbor_order'),
|
|
343
|
+
'selected' => details[:neighbors]
|
|
344
|
+
},
|
|
345
|
+
'budget_rule' => 'selection fits declared case budget; no silent extension or dropping'
|
|
346
|
+
}
|
|
347
|
+
end
|
|
348
|
+
|
|
349
|
+
def require_keys(hash, keys, label)
|
|
350
|
+
error!("#{label} must be an object") unless hash.is_a?(Hash)
|
|
351
|
+
missing = keys.reject { |key| hash.key?(key) }
|
|
352
|
+
error!("#{label} missing: #{missing.join(', ')}") if missing.any?
|
|
353
|
+
end
|
|
354
|
+
|
|
355
|
+
def digest(content)
|
|
356
|
+
Digest::SHA256.hexdigest(content)
|
|
357
|
+
end
|
|
358
|
+
|
|
359
|
+
def canonical_json(value)
|
|
360
|
+
JSON.generate(deep_sort(value))
|
|
361
|
+
end
|
|
362
|
+
|
|
363
|
+
def deep_sort(value)
|
|
364
|
+
case value
|
|
365
|
+
when Hash then value.keys.sort.to_h { |key| [key, deep_sort(value.fetch(key))] }
|
|
366
|
+
when Array then value.map { |item| deep_sort(item) }
|
|
367
|
+
else value
|
|
368
|
+
end
|
|
369
|
+
end
|
|
370
|
+
|
|
371
|
+
def error!(message)
|
|
372
|
+
raise Error, message
|
|
373
|
+
end
|
|
374
|
+
end
|
|
375
|
+
|
|
376
|
+
# Executes the same authored merge bytes through Git and the installed driver.
|
|
377
|
+
class LocalBenchmarkRunner
|
|
378
|
+
DEFAULT_TIMEOUT = 30
|
|
379
|
+
MERGIRAF_EXTENSIONS = {
|
|
380
|
+
'bash' => 'sh',
|
|
381
|
+
'html' => 'html',
|
|
382
|
+
'json' => 'json',
|
|
383
|
+
'markdown' => 'md',
|
|
384
|
+
'ruby' => 'rb',
|
|
385
|
+
'toml' => 'toml',
|
|
386
|
+
'typescript' => 'ts',
|
|
387
|
+
'yaml' => 'yaml'
|
|
388
|
+
}.freeze
|
|
389
|
+
|
|
390
|
+
def initialize(benchmark:, driver_path:, tmp_root:, timeout: DEFAULT_TIMEOUT, competitor_paths: {})
|
|
391
|
+
@benchmark = benchmark
|
|
392
|
+
@driver_path = Pathname(driver_path).expand_path
|
|
393
|
+
@tmp_root = Pathname(tmp_root).expand_path
|
|
394
|
+
@timeout = Integer(timeout)
|
|
395
|
+
@competitor_paths = competitor_paths.transform_values { |path| Pathname(path).expand_path }
|
|
396
|
+
end
|
|
397
|
+
|
|
398
|
+
def run(profile:, changed_paths: [])
|
|
399
|
+
verify_environment!
|
|
400
|
+
selection = @benchmark.select(profile: profile, changed_paths: changed_paths)
|
|
401
|
+
original_cwd = Dir.pwd
|
|
402
|
+
results = selection['selected_case_ids'].flat_map do |id|
|
|
403
|
+
item = @benchmark.case_by_id(id)
|
|
404
|
+
execute_case(item) + execute_competitors(item)
|
|
405
|
+
end
|
|
406
|
+
raise LocalBenchmark::Error, 'runner changed its working directory' unless Dir.pwd == original_cwd
|
|
407
|
+
|
|
408
|
+
{
|
|
409
|
+
'schema_version' => 'structuredmerge.benchmark.run/v1',
|
|
410
|
+
'kind' => 'paired_local_run',
|
|
411
|
+
'corpus_id' => @benchmark.document['id'],
|
|
412
|
+
'corpus_digest' => @benchmark.corpus_digest,
|
|
413
|
+
'selection' => selection,
|
|
414
|
+
'cache_identity' => cache_identity(selection),
|
|
415
|
+
'competitors' => competitor_provenance,
|
|
416
|
+
'results' => results
|
|
417
|
+
}
|
|
418
|
+
ensure
|
|
419
|
+
Dir.chdir(original_cwd) if original_cwd && Dir.pwd != original_cwd
|
|
420
|
+
end
|
|
421
|
+
|
|
422
|
+
private
|
|
423
|
+
|
|
424
|
+
def verify_environment!
|
|
425
|
+
error!("missing installed driver: #{@driver_path}") unless @driver_path.file? && @driver_path.executable?
|
|
426
|
+
@competitor_paths.each do |id, path|
|
|
427
|
+
error!("unknown competitor: #{id}") unless @benchmark.document.fetch('competitors').key?(id)
|
|
428
|
+
error!("missing competitor executable: #{path}") unless path.file? && path.executable?
|
|
429
|
+
end
|
|
430
|
+
root = Pathname(__dir__).join('..', '..', '..', '..').realpath
|
|
431
|
+
resolved = @tmp_root.exist? ? @tmp_root.realpath : @tmp_root.dirname.realpath.join(@tmp_root.basename)
|
|
432
|
+
error!('tmp_root must be inside the ast-merge-git repository') unless resolved.to_s.start_with?("#{root}/")
|
|
433
|
+
end
|
|
434
|
+
|
|
435
|
+
def execute_case(item)
|
|
436
|
+
return execute_merge3_case(item) if item['operation'] == 'merge3'
|
|
437
|
+
return execute_metamorphic_case(item) if item['operation'] == 'metamorphic'
|
|
438
|
+
|
|
439
|
+
unsupported_pair(item)
|
|
440
|
+
end
|
|
441
|
+
|
|
442
|
+
def execute_competitors(item)
|
|
443
|
+
@competitor_paths.map do |id, path|
|
|
444
|
+
metadata = @benchmark.document.dig('competitors', id)
|
|
445
|
+
unless metadata['operations'].include?(item['operation']) && metadata['dialects'].include?(item['dialect'])
|
|
446
|
+
next raw_result(
|
|
447
|
+
item,
|
|
448
|
+
id,
|
|
449
|
+
nil,
|
|
450
|
+
outcome: 'unsupported',
|
|
451
|
+
unsupported_reason: "#{id} does not support #{item['operation']}/#{item['dialect']}"
|
|
452
|
+
)
|
|
453
|
+
end
|
|
454
|
+
|
|
455
|
+
execute_mergiraf(item, path)
|
|
456
|
+
end
|
|
457
|
+
end
|
|
458
|
+
|
|
459
|
+
def execute_mergiraf(item, path)
|
|
460
|
+
workspace = @tmp_root.join("#{safe_id(item['id'])}-mergiraf-#{Process.pid}")
|
|
461
|
+
FileUtils.rm_rf(workspace)
|
|
462
|
+
FileUtils.mkdir_p(workspace)
|
|
463
|
+
write_roles(item, workspace)
|
|
464
|
+
output = workspace.join('output')
|
|
465
|
+
capture = timed_capture(
|
|
466
|
+
oracle_free_env,
|
|
467
|
+
path.to_s,
|
|
468
|
+
'merge',
|
|
469
|
+
'base',
|
|
470
|
+
'ours',
|
|
471
|
+
'theirs',
|
|
472
|
+
'--output',
|
|
473
|
+
output.to_s,
|
|
474
|
+
'--path-name',
|
|
475
|
+
"#{item['id']}.#{MERGIRAF_EXTENSIONS.fetch(item['dialect'])}",
|
|
476
|
+
'--conflict-marker-size',
|
|
477
|
+
'7',
|
|
478
|
+
chdir: workspace
|
|
479
|
+
)
|
|
480
|
+
capture[:output] = output.file? ? output.binread : ''
|
|
481
|
+
raw_result(item, 'mergiraf', capture)
|
|
482
|
+
ensure
|
|
483
|
+
FileUtils.rm_rf(workspace) if workspace
|
|
484
|
+
end
|
|
485
|
+
|
|
486
|
+
def execute_merge3_case(item)
|
|
487
|
+
workspace = @tmp_root.join("#{safe_id(item['id'])}-#{Process.pid}")
|
|
488
|
+
FileUtils.rm_rf(workspace)
|
|
489
|
+
FileUtils.mkdir_p(workspace)
|
|
490
|
+
baseline = execute_baseline(item, workspace)
|
|
491
|
+
candidate = execute_candidate(item, workspace)
|
|
492
|
+
rerun = execute_candidate(item, workspace)
|
|
493
|
+
candidate['deterministic_correctness_rerun'] = correctness_record(candidate) == correctness_record(rerun)
|
|
494
|
+
[baseline, candidate]
|
|
495
|
+
ensure
|
|
496
|
+
FileUtils.rm_rf(workspace) if workspace
|
|
497
|
+
end
|
|
498
|
+
|
|
499
|
+
def execute_metamorphic_case(item)
|
|
500
|
+
workspace = @tmp_root.join("#{safe_id(item['id'])}-#{Process.pid}")
|
|
501
|
+
FileUtils.rm_rf(workspace)
|
|
502
|
+
FileUtils.mkdir_p(workspace)
|
|
503
|
+
write_metamorphic_roles(item, workspace)
|
|
504
|
+
baseline = execute_metamorphic_baseline(item, workspace)
|
|
505
|
+
candidate = execute_metamorphic_candidate(item, workspace)
|
|
506
|
+
rerun = execute_metamorphic_candidate(item, workspace)
|
|
507
|
+
candidate['deterministic_correctness_rerun'] = correctness_record(candidate) == correctness_record(rerun)
|
|
508
|
+
[baseline, candidate]
|
|
509
|
+
ensure
|
|
510
|
+
FileUtils.rm_rf(workspace) if workspace
|
|
511
|
+
end
|
|
512
|
+
|
|
513
|
+
def unsupported_pair(item)
|
|
514
|
+
%w[git.unsupported structuredmerge.unsupported].map do |adapter|
|
|
515
|
+
raw_result(item, adapter, nil, outcome: 'unsupported',
|
|
516
|
+
unsupported_reason: "no installed #{item['operation']} adapter")
|
|
517
|
+
end
|
|
518
|
+
end
|
|
519
|
+
|
|
520
|
+
def execute_baseline(item, workspace)
|
|
521
|
+
write_roles(item, workspace)
|
|
522
|
+
capture = timed_capture({}, 'git', 'merge-file', '-p', 'ours', 'base', 'theirs', chdir: workspace)
|
|
523
|
+
capture[:output] = capture[:stdout]
|
|
524
|
+
raw_result(item, 'git.merge-file', capture)
|
|
525
|
+
end
|
|
526
|
+
|
|
527
|
+
def execute_candidate(item, workspace)
|
|
528
|
+
write_roles(item, workspace)
|
|
529
|
+
selector = item.fetch('selector')
|
|
530
|
+
capture = timed_capture(
|
|
531
|
+
candidate_env(selector),
|
|
532
|
+
@driver_path.to_s, 'base', 'ours', 'theirs', "#{item['id']}.#{item['dialect']}", '7',
|
|
533
|
+
chdir: workspace
|
|
534
|
+
)
|
|
535
|
+
capture[:output] = workspace.join('ours').binread
|
|
536
|
+
raw_result(item, 'ast-merge-git', capture)
|
|
537
|
+
end
|
|
538
|
+
|
|
539
|
+
def write_roles(item, workspace)
|
|
540
|
+
%w[base ours theirs].each do |role|
|
|
541
|
+
workspace.join(role).binwrite(item.dig('inputs', role, 'bytes'))
|
|
542
|
+
end
|
|
543
|
+
end
|
|
544
|
+
|
|
545
|
+
def write_metamorphic_roles(item, workspace)
|
|
546
|
+
%w[source transformed].each do |role|
|
|
547
|
+
workspace.join(role).binwrite(item.dig('inputs', role, 'bytes'))
|
|
548
|
+
end
|
|
549
|
+
end
|
|
550
|
+
|
|
551
|
+
def execute_metamorphic_baseline(item, workspace)
|
|
552
|
+
capture = timed_capture(
|
|
553
|
+
{},
|
|
554
|
+
'git', 'diff', '--no-index', '--no-color', '--no-ext-diff', '--', 'source', 'transformed',
|
|
555
|
+
chdir: workspace
|
|
556
|
+
)
|
|
557
|
+
capture[:output] = capture[:stdout]
|
|
558
|
+
metamorphic_result(
|
|
559
|
+
item,
|
|
560
|
+
'git.diff',
|
|
561
|
+
'baseline',
|
|
562
|
+
capture,
|
|
563
|
+
edit_signal: capture[:status] == 1
|
|
564
|
+
)
|
|
565
|
+
end
|
|
566
|
+
|
|
567
|
+
def execute_metamorphic_candidate(item, workspace)
|
|
568
|
+
capture, changes = capture_provider_diff(item, workspace)
|
|
569
|
+
metamorphic_result(
|
|
570
|
+
item,
|
|
571
|
+
'ast-merge-provider.diff2',
|
|
572
|
+
'candidate',
|
|
573
|
+
capture,
|
|
574
|
+
edit_signal: changes.any?
|
|
575
|
+
)
|
|
576
|
+
end
|
|
577
|
+
|
|
578
|
+
def capture_provider_diff(item, workspace)
|
|
579
|
+
selector = item.fetch('selector')
|
|
580
|
+
capture = timed_capture(
|
|
581
|
+
candidate_env(selector),
|
|
582
|
+
@driver_path.to_s,
|
|
583
|
+
'benchmark-provider-diff',
|
|
584
|
+
'source',
|
|
585
|
+
'transformed',
|
|
586
|
+
"#{item['id']}.#{item['dialect']}",
|
|
587
|
+
chdir: workspace
|
|
588
|
+
)
|
|
589
|
+
result = capture[:stdout].empty? ? {} : JSON.parse(capture[:stdout])
|
|
590
|
+
changes = capture[:status].zero? ? result.fetch('changes') : []
|
|
591
|
+
capture[:output] = JSON.generate(changes)
|
|
592
|
+
[capture, changes]
|
|
593
|
+
rescue JSON::ParserError, KeyError => e
|
|
594
|
+
capture ||= { stdout: '', stderr: '', status: 2, duration_ns: 0 }
|
|
595
|
+
capture[:stderr] = [capture[:stderr], e.message].reject(&:empty?).join("\n")
|
|
596
|
+
capture[:status] = 2
|
|
597
|
+
capture[:output] = ''
|
|
598
|
+
[capture, []]
|
|
599
|
+
end
|
|
600
|
+
|
|
601
|
+
def timed_capture(env, *command, chdir:)
|
|
602
|
+
started = Process.clock_gettime(Process::CLOCK_MONOTONIC, :nanosecond)
|
|
603
|
+
stdin, stdout_io, stderr_io, process = Open3.popen3(env, *command, chdir: chdir.to_s)
|
|
604
|
+
[stdin, stdout_io, stderr_io].each(&:binmode)
|
|
605
|
+
stdin.close
|
|
606
|
+
stdout_reader = Thread.new { stdout_io.read }
|
|
607
|
+
stderr_reader = Thread.new { stderr_io.read }
|
|
608
|
+
timed_out = process.join(@timeout).nil?
|
|
609
|
+
terminate_process(process) if timed_out
|
|
610
|
+
stdout = stdout_reader.value
|
|
611
|
+
stderr = stderr_reader.value
|
|
612
|
+
stderr = [stderr, "timeout after #{@timeout}s"].reject(&:empty?).join("\n") if timed_out
|
|
613
|
+
{ stdout: stdout, stderr: stderr, status: timed_out ? 2 : process.value.exitstatus,
|
|
614
|
+
duration_ns: Process.clock_gettime(Process::CLOCK_MONOTONIC, :nanosecond) - started }
|
|
615
|
+
rescue Errno::ENOENT => e
|
|
616
|
+
{ stdout: '', stderr: e.message, status: 2, output: '', duration_ns: 0 }
|
|
617
|
+
ensure
|
|
618
|
+
[stdin, stdout_io, stderr_io].compact.each { |io| io.close unless io.closed? }
|
|
619
|
+
end
|
|
620
|
+
|
|
621
|
+
def terminate_process(process)
|
|
622
|
+
Process.kill('TERM', process.pid)
|
|
623
|
+
return if process.join(1)
|
|
624
|
+
|
|
625
|
+
Process.kill('KILL', process.pid)
|
|
626
|
+
process.join
|
|
627
|
+
rescue Errno::ESRCH, Errno::ECHILD
|
|
628
|
+
process.join
|
|
629
|
+
end
|
|
630
|
+
|
|
631
|
+
def raw_result(item, adapter, capture, outcome: nil, unsupported_reason: nil)
|
|
632
|
+
output = capture&.fetch(:output, '') || ''
|
|
633
|
+
checks = equivalence_checks(item, output)
|
|
634
|
+
markers = conflict_regions(output)
|
|
635
|
+
classified = outcome || classify(item, capture.fetch(:status), checks)
|
|
636
|
+
eligible = item.dig('oracle', 'score_eligible') && !%w[unsupported excluded_ambiguous].include?(classified)
|
|
637
|
+
{
|
|
638
|
+
'schema_version' => 'structuredmerge.benchmark.result/v1',
|
|
639
|
+
'id' => "result.#{safe_id(item['id'])}.#{safe_id(adapter)}",
|
|
640
|
+
'case_id' => item['id'],
|
|
641
|
+
'adapter_id' => adapter,
|
|
642
|
+
'adapter_role' => adapter_role(adapter),
|
|
643
|
+
'case_tags' => item.slice(
|
|
644
|
+
'operation', 'partition', 'family', 'provider', 'dialect',
|
|
645
|
+
'capabilities', 'false_auto_merge_severity'
|
|
646
|
+
),
|
|
647
|
+
'outcome' => classified,
|
|
648
|
+
'score_eligible' => eligible,
|
|
649
|
+
'provenance' => item['provenance'],
|
|
650
|
+
'process' => capture && { 'status' => capture[:status],
|
|
651
|
+
'exit_classification' => exit_class(capture[:status]) },
|
|
652
|
+
'raw' => {
|
|
653
|
+
'stdout' => raw_record(capture&.fetch(:stdout, '') || ''),
|
|
654
|
+
'stderr' => raw_record(capture&.fetch(:stderr, '') || ''),
|
|
655
|
+
'output' => raw_record(output)
|
|
656
|
+
},
|
|
657
|
+
'diagnostics' => diagnostics(capture&.fetch(:stderr, '') || '', unsupported_reason),
|
|
658
|
+
'conflict_regions' => region_evidence(item, markers),
|
|
659
|
+
'checks' => checks,
|
|
660
|
+
'independent_edit_evidence' => independent_edit_evidence(item, checks, classified),
|
|
661
|
+
'dimensions' => dimensions(item, classified, checks),
|
|
662
|
+
'runtime' => capture && { 'duration_ns' => capture[:duration_ns], 'comparable' => true }
|
|
663
|
+
}
|
|
664
|
+
end
|
|
665
|
+
|
|
666
|
+
def adapter_role(adapter)
|
|
667
|
+
return 'baseline' if adapter.start_with?('git.')
|
|
668
|
+
return 'candidate' if %w[ast-merge-git ast-merge-provider.diff2 structuredmerge.unsupported].include?(adapter)
|
|
669
|
+
|
|
670
|
+
'competitor'
|
|
671
|
+
end
|
|
672
|
+
|
|
673
|
+
def metamorphic_result(item, adapter, adapter_role, capture, edit_signal:)
|
|
674
|
+
checks = metamorphic_checks(item, edit_signal)
|
|
675
|
+
classified = if capture[:status].nil? || capture[:status] >= 2
|
|
676
|
+
'error'
|
|
677
|
+
elsif checks['acceptable']
|
|
678
|
+
'correct_clean'
|
|
679
|
+
else
|
|
680
|
+
'false_conflict'
|
|
681
|
+
end
|
|
682
|
+
eligible = item.dig('oracle', 'score_eligible') && classified != 'error'
|
|
683
|
+
{
|
|
684
|
+
'schema_version' => 'structuredmerge.benchmark.result/v1',
|
|
685
|
+
'id' => "result.#{safe_id(item['id'])}.#{safe_id(adapter)}",
|
|
686
|
+
'case_id' => item['id'],
|
|
687
|
+
'adapter_id' => adapter,
|
|
688
|
+
'adapter_role' => adapter_role,
|
|
689
|
+
'case_tags' => item.slice(
|
|
690
|
+
'operation', 'partition', 'family', 'provider', 'dialect',
|
|
691
|
+
'capabilities', 'false_auto_merge_severity'
|
|
692
|
+
),
|
|
693
|
+
'outcome' => classified,
|
|
694
|
+
'score_eligible' => eligible,
|
|
695
|
+
'provenance' => item['provenance'],
|
|
696
|
+
'process' => {
|
|
697
|
+
'status' => capture[:status],
|
|
698
|
+
'exit_classification' => metamorphic_exit_class(adapter_role, capture[:status])
|
|
699
|
+
},
|
|
700
|
+
'raw' => {
|
|
701
|
+
'stdout' => raw_record(capture[:stdout]),
|
|
702
|
+
'stderr' => raw_record(capture[:stderr]),
|
|
703
|
+
'output' => raw_record(capture[:output])
|
|
704
|
+
},
|
|
705
|
+
'diagnostics' => diagnostics(capture[:stderr], nil),
|
|
706
|
+
'conflict_regions' => region_evidence(item, []),
|
|
707
|
+
'checks' => checks,
|
|
708
|
+
'independent_edit_evidence' => {},
|
|
709
|
+
'dimensions' => dimensions(item, classified, checks),
|
|
710
|
+
'runtime' => { 'duration_ns' => capture[:duration_ns], 'comparable' => true }
|
|
711
|
+
}
|
|
712
|
+
end
|
|
713
|
+
|
|
714
|
+
def metamorphic_checks(item, edit_signal)
|
|
715
|
+
evaluations = item.fetch('expected_invariants').map do |invariant|
|
|
716
|
+
matched, evidence = evaluate_metamorphic_invariant(item, invariant, edit_signal)
|
|
717
|
+
{ 'id' => invariant, 'matched' => matched, 'evidence' => evidence }
|
|
718
|
+
end
|
|
719
|
+
accepted = evaluations.all? { |evaluation| evaluation['matched'] }
|
|
720
|
+
{
|
|
721
|
+
'declared_invariants' => evaluations,
|
|
722
|
+
'accepted_equivalence' => accepted ? 'declared_invariants' : nil,
|
|
723
|
+
'acceptable_equivalence_evaluations' => [{
|
|
724
|
+
'class' => 'declared_invariants',
|
|
725
|
+
'matched' => accepted
|
|
726
|
+
}],
|
|
727
|
+
'parse_validity_only_accepted' => false,
|
|
728
|
+
'generator_replay' => {
|
|
729
|
+
'generator_id' => item.dig('generator', 'id'),
|
|
730
|
+
'generator_version' => item.dig('generator', 'version'),
|
|
731
|
+
'seed' => item.dig('generator', 'seed'),
|
|
732
|
+
'transformation_ids' => item.fetch('transformations').map { |transformation| transformation['id'] },
|
|
733
|
+
'authored_bytes_digest_verified' => true
|
|
734
|
+
},
|
|
735
|
+
'preservation_violations' => [],
|
|
736
|
+
'acceptable' => accepted
|
|
737
|
+
}
|
|
738
|
+
end
|
|
739
|
+
|
|
740
|
+
def evaluate_metamorphic_invariant(item, invariant, edit_signal)
|
|
741
|
+
case invariant
|
|
742
|
+
when 'same-json-value', 'same-jsonc-value'
|
|
743
|
+
[json_values_equal?(item), 'JSON-family values parsed from source and transformed bytes']
|
|
744
|
+
when 'no-semantic-edit'
|
|
745
|
+
[!edit_signal, 'adapter emitted no edit units']
|
|
746
|
+
when 'comment-retained'
|
|
747
|
+
[json_comments_retained?(item), 'JSONC AST comment content retained']
|
|
748
|
+
else
|
|
749
|
+
[false, 'unsupported invariant']
|
|
750
|
+
end
|
|
751
|
+
end
|
|
752
|
+
|
|
753
|
+
def json_values_equal?(item)
|
|
754
|
+
require 'json/merge'
|
|
755
|
+
selector = item.fetch('selector')
|
|
756
|
+
source = Json::Merge.json_value_for_source(
|
|
757
|
+
item.dig('inputs', 'source', 'bytes'),
|
|
758
|
+
dialect: selector.fetch('dialect').to_sym,
|
|
759
|
+
backend: selector.fetch('backend')
|
|
760
|
+
)
|
|
761
|
+
transformed = Json::Merge.json_value_for_source(
|
|
762
|
+
item.dig('inputs', 'transformed', 'bytes'),
|
|
763
|
+
dialect: selector.fetch('dialect').to_sym,
|
|
764
|
+
backend: selector.fetch('backend')
|
|
765
|
+
)
|
|
766
|
+
source == transformed
|
|
767
|
+
rescue Json::Merge::ParseError, LoadError
|
|
768
|
+
false
|
|
769
|
+
end
|
|
770
|
+
|
|
771
|
+
def json_comments_retained?(item)
|
|
772
|
+
require 'json/merge'
|
|
773
|
+
selector = item.fetch('selector')
|
|
774
|
+
source = Json::Merge::FileAnalysis.new(
|
|
775
|
+
item.dig('inputs', 'source', 'bytes'),
|
|
776
|
+
dialect: selector.fetch('dialect').to_sym
|
|
777
|
+
)
|
|
778
|
+
transformed = Json::Merge::FileAnalysis.new(
|
|
779
|
+
item.dig('inputs', 'transformed', 'bytes'),
|
|
780
|
+
dialect: selector.fetch('dialect').to_sym
|
|
781
|
+
)
|
|
782
|
+
source.valid? && transformed.valid? &&
|
|
783
|
+
source.comment_nodes.map(&:normalized_content) == transformed.comment_nodes.map(&:normalized_content)
|
|
784
|
+
rescue TreeHaver::Error, LoadError
|
|
785
|
+
false
|
|
786
|
+
end
|
|
787
|
+
|
|
788
|
+
def metamorphic_exit_class(adapter_role, status)
|
|
789
|
+
return 'error' if status.nil? || status >= 2
|
|
790
|
+
return status == 1 ? 'differences' : 'no_differences' if adapter_role == 'baseline'
|
|
791
|
+
|
|
792
|
+
'analyzed'
|
|
793
|
+
end
|
|
794
|
+
|
|
795
|
+
def classify(item, status, checks)
|
|
796
|
+
expected = item.dig('expected', 'outcome')
|
|
797
|
+
return 'excluded_ambiguous' if expected == 'excluded_ambiguous'
|
|
798
|
+
return 'error' if status.nil? || status >= 2
|
|
799
|
+
return status == 1 ? 'false_conflict' : 'false_auto_merge' if expected == 'error'
|
|
800
|
+
return status == 1 ? 'true_conflict' : 'false_auto_merge' if expected == 'conflict'
|
|
801
|
+
return 'false_conflict' if status == 1
|
|
802
|
+
return 'correct_clean' if checks['acceptable']
|
|
803
|
+
|
|
804
|
+
'false_auto_merge'
|
|
805
|
+
end
|
|
806
|
+
|
|
807
|
+
def equivalence_checks(item, output)
|
|
808
|
+
expected = item.dig('expected', 'output', 'bytes')
|
|
809
|
+
exact = !expected.nil? && output == expected
|
|
810
|
+
structural = structural_equivalence(item, output, expected)
|
|
811
|
+
evaluations = item.fetch('acceptable_equivalence').map do |policy|
|
|
812
|
+
matched = case policy.fetch('class')
|
|
813
|
+
when 'exact_bytes' then exact
|
|
814
|
+
when 'structural_ast' then structural == true && structural_provider_matches?(item, policy)
|
|
815
|
+
else false
|
|
816
|
+
end
|
|
817
|
+
{ 'class' => policy.fetch('class'), 'matched' => matched }
|
|
818
|
+
end
|
|
819
|
+
selected = evaluations.find { |evaluation| evaluation['matched'] }
|
|
820
|
+
checks = {
|
|
821
|
+
'exact' => exact,
|
|
822
|
+
'structural' => structural,
|
|
823
|
+
'structural_provider' => item.dig('selector', 'provider_id'),
|
|
824
|
+
'acceptable_equivalence_evaluations' => evaluations,
|
|
825
|
+
'accepted_equivalence' => selected&.fetch('class'),
|
|
826
|
+
'parse_validity_only_accepted' => false
|
|
827
|
+
}
|
|
828
|
+
violations = preservation_violations(item, checks, output)
|
|
829
|
+
checks.merge('preservation_violations' => violations, 'acceptable' => !selected.nil? && violations.empty?)
|
|
830
|
+
end
|
|
831
|
+
|
|
832
|
+
def structural_equivalence(item, output, expected)
|
|
833
|
+
return nil unless expected
|
|
834
|
+
|
|
835
|
+
selector = item.fetch('selector')
|
|
836
|
+
require selector.fetch('require')
|
|
837
|
+
result = Ast::Merge.dispatch_provider(
|
|
838
|
+
:diff2,
|
|
839
|
+
{
|
|
840
|
+
provider_id: selector.fetch('provider_id'),
|
|
841
|
+
family: selector.fetch('family'),
|
|
842
|
+
dialect: selector.fetch('dialect'),
|
|
843
|
+
backend: selector.fetch('backend'),
|
|
844
|
+
profile_id: selector.fetch('profile'),
|
|
845
|
+
before_source: expected,
|
|
846
|
+
after_source: output,
|
|
847
|
+
path_name: "#{item['id']}.#{item['dialect']}"
|
|
848
|
+
}
|
|
849
|
+
)
|
|
850
|
+
result[:ok] == true && result.fetch(:changes).empty?
|
|
851
|
+
rescue Ast::Merge::Error, KeyError, LoadError
|
|
852
|
+
false
|
|
853
|
+
end
|
|
854
|
+
|
|
855
|
+
def structural_provider_matches?(item, policy)
|
|
856
|
+
policy['provider'].to_s == item.dig('selector', 'provider_id').to_s
|
|
857
|
+
end
|
|
858
|
+
|
|
859
|
+
def dimensions(item, outcome, checks)
|
|
860
|
+
eligible = item.dig('oracle', 'score_eligible')
|
|
861
|
+
{
|
|
862
|
+
'safety' => {
|
|
863
|
+
'eligible' => eligible,
|
|
864
|
+
'false_auto_merge' => outcome == 'false_auto_merge',
|
|
865
|
+
'severity' => item['false_auto_merge_severity'],
|
|
866
|
+
'compensable' => false
|
|
867
|
+
},
|
|
868
|
+
'effectiveness' => {
|
|
869
|
+
'eligible' => eligible,
|
|
870
|
+
'success' => %w[correct_clean true_conflict].include?(outcome)
|
|
871
|
+
},
|
|
872
|
+
'preservation' => {
|
|
873
|
+
'eligible' => eligible,
|
|
874
|
+
'requirements' => item['preservation_policy'],
|
|
875
|
+
'violations' => checks.fetch('preservation_violations')
|
|
876
|
+
},
|
|
877
|
+
'performance' => { 'quality_offset_allowed' => false }
|
|
878
|
+
}
|
|
879
|
+
end
|
|
880
|
+
|
|
881
|
+
def preservation_violations(item, checks, output)
|
|
882
|
+
return [] if checks['exact']
|
|
883
|
+
|
|
884
|
+
expected = item.dig('expected', 'output', 'bytes')
|
|
885
|
+
return [] unless expected
|
|
886
|
+
|
|
887
|
+
item['preservation_policy'].filter_map do |name, requirement|
|
|
888
|
+
next unless requirement == 'required'
|
|
889
|
+
next unless preservation_violation?(name, output, expected, checks)
|
|
890
|
+
|
|
891
|
+
name
|
|
892
|
+
end
|
|
893
|
+
end
|
|
894
|
+
|
|
895
|
+
def preservation_violation?(name, output, expected, checks)
|
|
896
|
+
case name
|
|
897
|
+
when 'encoding' then output.encoding != expected.encoding
|
|
898
|
+
when 'line_endings' then line_ending_style(output) != line_ending_style(expected)
|
|
899
|
+
when 'unknown_fields' then checks['structural'] != true
|
|
900
|
+
else output != expected
|
|
901
|
+
end
|
|
902
|
+
end
|
|
903
|
+
|
|
904
|
+
def line_ending_style(content)
|
|
905
|
+
return 'crlf' if content.include?("\r\n")
|
|
906
|
+
|
|
907
|
+
'lf'
|
|
908
|
+
end
|
|
909
|
+
|
|
910
|
+
def independent_edit_evidence(item, checks, outcome)
|
|
911
|
+
item['independent_edit_ids'].to_h do |id|
|
|
912
|
+
[id,
|
|
913
|
+
{ 'preserved' => outcome == 'correct_clean' && checks['acceptable'], 'method' => 'oracle equivalence' }]
|
|
914
|
+
end
|
|
915
|
+
end
|
|
916
|
+
|
|
917
|
+
def conflict_regions(output)
|
|
918
|
+
ranges = []
|
|
919
|
+
start_byte = nil
|
|
920
|
+
offset = 0
|
|
921
|
+
output.each_line do |line|
|
|
922
|
+
start_byte = offset if line.start_with?('<<<<<<<')
|
|
923
|
+
if start_byte && line.start_with?('>>>>>>>')
|
|
924
|
+
ranges << { 'start_byte' => start_byte, 'end_byte' => offset + line.bytesize }
|
|
925
|
+
start_byte = nil
|
|
926
|
+
end
|
|
927
|
+
offset += line.bytesize
|
|
928
|
+
end
|
|
929
|
+
ranges
|
|
930
|
+
end
|
|
931
|
+
|
|
932
|
+
def region_evidence(item, observed)
|
|
933
|
+
expected = item['expected_conflict_regions']
|
|
934
|
+
observed = observed.each_with_index.map { |region, index| region.merge('id' => "observed.#{index + 1}") }
|
|
935
|
+
localization_status = expected.empty? && observed.empty? ? 'not_applicable' : 'unknown'
|
|
936
|
+
{
|
|
937
|
+
'expected' => expected,
|
|
938
|
+
'observed' => observed,
|
|
939
|
+
'matched_region_ids' => [],
|
|
940
|
+
'missed_region_ids' => expected.map { |region| region['id'] },
|
|
941
|
+
'false_positive_region_ids' => observed.map { |region| region['id'] },
|
|
942
|
+
'localization_status' => localization_status,
|
|
943
|
+
'matching_basis' => 'unknown: expected role ranges/paths and observed output ranges have no proven mapping',
|
|
944
|
+
'localization_error_bytes' => nil
|
|
945
|
+
}
|
|
946
|
+
end
|
|
947
|
+
|
|
948
|
+
def raw_record(content)
|
|
949
|
+
{ 'inline' => content, 'bytes' => content.bytesize, 'sha256' => Digest::SHA256.hexdigest(content) }
|
|
950
|
+
end
|
|
951
|
+
|
|
952
|
+
def diagnostics(stderr, unsupported_reason)
|
|
953
|
+
lines = stderr.lines.map(&:strip).reject(&:empty?)
|
|
954
|
+
lines << unsupported_reason if unsupported_reason
|
|
955
|
+
lines.map do |line|
|
|
956
|
+
match = /\Aast-merge-git: ([a-z_]+):/.match(line)
|
|
957
|
+
category = match&.[](1) || (unsupported_reason == line ? 'unsupported' : 'process')
|
|
958
|
+
{ 'severity' => 'error', 'category' => category, 'message' => line }
|
|
959
|
+
end
|
|
960
|
+
end
|
|
961
|
+
|
|
962
|
+
def selector_env(selector)
|
|
963
|
+
{
|
|
964
|
+
'AST_MERGE_PROVIDER' => selector['provider_id'],
|
|
965
|
+
'AST_MERGE_FAMILY' => selector['family'],
|
|
966
|
+
'AST_MERGE_DIALECT' => selector['dialect'],
|
|
967
|
+
'AST_MERGE_BACKEND' => selector['backend'],
|
|
968
|
+
'AST_MERGE_PROFILE' => selector['profile'],
|
|
969
|
+
'AST_MERGE_REQUIRE' => selector['require']
|
|
970
|
+
}
|
|
971
|
+
end
|
|
972
|
+
|
|
973
|
+
def candidate_env(selector)
|
|
974
|
+
oracle_free_env.merge(selector_env(selector))
|
|
975
|
+
end
|
|
976
|
+
|
|
977
|
+
def oracle_free_env
|
|
978
|
+
ENV.keys.grep(/(?:ORACLE|EXPECTED)/i).to_h { |name| [name, nil] }
|
|
979
|
+
end
|
|
980
|
+
|
|
981
|
+
def cache_identity(selection)
|
|
982
|
+
source_sha, = Open3.capture2('git', '-C', Pathname(__dir__).join('..', '..', '..', '..').to_s,
|
|
983
|
+
'rev-parse', 'HEAD')
|
|
984
|
+
environment = {
|
|
985
|
+
'ruby' => RUBY_DESCRIPTION,
|
|
986
|
+
'platform' => RUBY_PLATFORM,
|
|
987
|
+
'host_os' => RbConfig::CONFIG['host_os'],
|
|
988
|
+
'host_cpu' => RbConfig::CONFIG['host_cpu']
|
|
989
|
+
}
|
|
990
|
+
configuration = selection['selected_case_ids'].map do |id|
|
|
991
|
+
@benchmark.case_by_id(id).slice('id', 'selector')
|
|
992
|
+
end
|
|
993
|
+
competitors = competitor_provenance
|
|
994
|
+
identity = {
|
|
995
|
+
'adapter_source_sha' => source_sha.strip,
|
|
996
|
+
'adapter_artifact_sha256' => Digest::SHA256.file(@driver_path).hexdigest,
|
|
997
|
+
'configuration_sha256' => Digest::SHA256.hexdigest(JSON.generate(configuration)),
|
|
998
|
+
'competitors' => competitors,
|
|
999
|
+
'corpus_sha256' => @benchmark.corpus_digest,
|
|
1000
|
+
'environment' => environment,
|
|
1001
|
+
'profile' => selection['profile'],
|
|
1002
|
+
'changed_paths' => selection['changed_paths'],
|
|
1003
|
+
'inferred_capabilities' => selection['inferred_capabilities'],
|
|
1004
|
+
'selection_explanation_sha256' => Digest::SHA256.hexdigest(
|
|
1005
|
+
JSON.generate(deep_sort(selection.fetch('explanation')))
|
|
1006
|
+
),
|
|
1007
|
+
'selected_case_ids' => selection['selected_case_ids']
|
|
1008
|
+
}
|
|
1009
|
+
identity.merge('sha256' => Digest::SHA256.hexdigest(JSON.generate(identity)))
|
|
1010
|
+
end
|
|
1011
|
+
|
|
1012
|
+
def competitor_provenance
|
|
1013
|
+
@competitor_provenance ||= @competitor_paths.to_h do |id, path|
|
|
1014
|
+
metadata = @benchmark.document.fetch('competitors').fetch(id)
|
|
1015
|
+
capture = timed_capture(oracle_free_env, path.to_s, '--version', chdir: path.dirname)
|
|
1016
|
+
error!("#{id} version probe failed: #{capture[:stderr].strip}") unless capture[:status].zero?
|
|
1017
|
+
expected = "#{metadata.fetch('adapter_id')} #{metadata.fetch('version')}"
|
|
1018
|
+
error!("#{id} version differs: #{capture[:stdout].strip}") unless capture[:stdout].strip == expected
|
|
1019
|
+
|
|
1020
|
+
[id, metadata.merge(
|
|
1021
|
+
'binary_path' => path.to_s,
|
|
1022
|
+
'binary_sha256' => Digest::SHA256.file(path).hexdigest,
|
|
1023
|
+
'reported_version' => capture[:stdout].strip
|
|
1024
|
+
)]
|
|
1025
|
+
end
|
|
1026
|
+
end
|
|
1027
|
+
|
|
1028
|
+
def deep_sort(value)
|
|
1029
|
+
case value
|
|
1030
|
+
when Hash then value.keys.sort.to_h { |key| [key, deep_sort(value.fetch(key))] }
|
|
1031
|
+
when Array then value.map { |item| deep_sort(item) }
|
|
1032
|
+
else value
|
|
1033
|
+
end
|
|
1034
|
+
end
|
|
1035
|
+
|
|
1036
|
+
def correctness_record(result)
|
|
1037
|
+
result.reject { |key, _value| %w[runtime deterministic_correctness_rerun].include?(key) }
|
|
1038
|
+
end
|
|
1039
|
+
|
|
1040
|
+
def exit_class(status)
|
|
1041
|
+
{ 0 => 'clean', 1 => 'conflict', 2 => 'error' }.fetch(status, 'error')
|
|
1042
|
+
end
|
|
1043
|
+
|
|
1044
|
+
def safe_id(value)
|
|
1045
|
+
value.gsub(/[^a-zA-Z0-9.-]/, '-')
|
|
1046
|
+
end
|
|
1047
|
+
|
|
1048
|
+
def error!(message)
|
|
1049
|
+
raise LocalBenchmark::Error, message
|
|
1050
|
+
end
|
|
1051
|
+
end
|
|
1052
|
+
|
|
1053
|
+
# Builds a non-scalar paired report from local benchmark raw evidence.
|
|
1054
|
+
class LocalBenchmarkReport
|
|
1055
|
+
SUCCESS = %w[correct_clean true_conflict].freeze
|
|
1056
|
+
|
|
1057
|
+
def self.build(run)
|
|
1058
|
+
new(run).build
|
|
1059
|
+
end
|
|
1060
|
+
|
|
1061
|
+
def initialize(run)
|
|
1062
|
+
@run = run
|
|
1063
|
+
@results = run.fetch('results')
|
|
1064
|
+
end
|
|
1065
|
+
|
|
1066
|
+
def build
|
|
1067
|
+
pairs = @results.group_by { |result| result['case_id'] }.values.map { |items| transition(items) }
|
|
1068
|
+
false_auto_merges = candidate_eligible.select { |item| item['outcome'] == 'false_auto_merge' }
|
|
1069
|
+
{
|
|
1070
|
+
'schema_version' => 'structuredmerge.benchmark.report/v1',
|
|
1071
|
+
'kind' => 'paired_aggregate_report',
|
|
1072
|
+
'corpus_id' => @run['corpus_id'],
|
|
1073
|
+
'corpus_digest' => @run['corpus_digest'],
|
|
1074
|
+
'selection' => @run['selection'],
|
|
1075
|
+
'cache_identity' => @run['cache_identity'],
|
|
1076
|
+
'dimensions' => {
|
|
1077
|
+
'safety' => {
|
|
1078
|
+
'eligible' => candidate_eligible.length,
|
|
1079
|
+
'false_auto_merge_result_ids' => false_auto_merges.map { |item| item['id'] },
|
|
1080
|
+
'gate' => false_auto_merges.empty? ? 'pass' : 'fail',
|
|
1081
|
+
'non_compensable' => true
|
|
1082
|
+
},
|
|
1083
|
+
'effectiveness' => outcome_counts,
|
|
1084
|
+
'preservation' => {
|
|
1085
|
+
'eligible' => eligible.length,
|
|
1086
|
+
'violation_result_ids' => eligible.filter_map do |item|
|
|
1087
|
+
item['id'] if item.dig('dimensions', 'preservation', 'violations').any?
|
|
1088
|
+
end
|
|
1089
|
+
},
|
|
1090
|
+
'performance' => performance,
|
|
1091
|
+
'reliability' => { 'error_result_ids' => @results.filter_map do |item|
|
|
1092
|
+
item['id'] if item['outcome'] == 'error'
|
|
1093
|
+
end },
|
|
1094
|
+
'coverage' => coverage,
|
|
1095
|
+
'competitive' => competitive
|
|
1096
|
+
},
|
|
1097
|
+
'strata' => strata,
|
|
1098
|
+
'transitions' => pairs,
|
|
1099
|
+
'newly_passing_case_ids' => pairs.filter_map { |pair| pair['case_id'] if pair['newly_passing'] },
|
|
1100
|
+
'newly_failing_case_ids' => pairs.filter_map { |pair| pair['case_id'] if pair['newly_failing'] },
|
|
1101
|
+
'changed_conflict_case_ids' => pairs.filter_map { |pair| pair['case_id'] if pair['changed_conflict'] },
|
|
1102
|
+
'hard_gate_failed' => false_auto_merges.any?,
|
|
1103
|
+
'scalar_score' => nil
|
|
1104
|
+
}
|
|
1105
|
+
end
|
|
1106
|
+
|
|
1107
|
+
private
|
|
1108
|
+
|
|
1109
|
+
def eligible
|
|
1110
|
+
@eligible ||= @results.select { |item| item['score_eligible'] }
|
|
1111
|
+
end
|
|
1112
|
+
|
|
1113
|
+
def candidate_eligible
|
|
1114
|
+
eligible.select { |item| item['adapter_role'] == 'candidate' }
|
|
1115
|
+
end
|
|
1116
|
+
|
|
1117
|
+
def outcome_counts
|
|
1118
|
+
%w[correct_clean false_conflict true_conflict false_auto_merge error unsupported
|
|
1119
|
+
excluded_ambiguous].to_h do |outcome|
|
|
1120
|
+
[outcome, @results.count { |item| item['outcome'] == outcome }]
|
|
1121
|
+
end
|
|
1122
|
+
end
|
|
1123
|
+
|
|
1124
|
+
def transition(items)
|
|
1125
|
+
baseline = items.find { |item| item['adapter_role'] == 'baseline' }
|
|
1126
|
+
candidate = items.find { |item| item['adapter_role'] == 'candidate' }
|
|
1127
|
+
{
|
|
1128
|
+
'case_id' => items.first['case_id'],
|
|
1129
|
+
'baseline_result_id' => baseline['id'],
|
|
1130
|
+
'candidate_result_id' => candidate['id'],
|
|
1131
|
+
'from' => baseline['outcome'],
|
|
1132
|
+
'to' => candidate['outcome'],
|
|
1133
|
+
'newly_passing' => !SUCCESS.include?(baseline['outcome']) && SUCCESS.include?(candidate['outcome']),
|
|
1134
|
+
'newly_failing' => SUCCESS.include?(baseline['outcome']) && !SUCCESS.include?(candidate['outcome']),
|
|
1135
|
+
'changed_conflict' => conflict?(baseline) && conflict?(candidate) &&
|
|
1136
|
+
baseline.dig('raw', 'output', 'sha256') != candidate.dig('raw', 'output', 'sha256')
|
|
1137
|
+
}
|
|
1138
|
+
end
|
|
1139
|
+
|
|
1140
|
+
def conflict?(result)
|
|
1141
|
+
%w[false_conflict true_conflict].include?(result['outcome'])
|
|
1142
|
+
end
|
|
1143
|
+
|
|
1144
|
+
def performance
|
|
1145
|
+
@results.group_by { |item| item['adapter_id'] }.transform_values do |items|
|
|
1146
|
+
values = items.filter_map { |item| item.dig('runtime', 'duration_ns') }
|
|
1147
|
+
{ 'samples' => values.length, 'total_ns' => values.sum, 'runtime_values_excluded_from_correctness' => true }
|
|
1148
|
+
end
|
|
1149
|
+
end
|
|
1150
|
+
|
|
1151
|
+
def coverage
|
|
1152
|
+
selected = @run.dig('selection', 'selected_case_ids').length
|
|
1153
|
+
unsupported = @results.select do |item|
|
|
1154
|
+
item['adapter_role'] == 'candidate' && item['outcome'] == 'unsupported'
|
|
1155
|
+
end
|
|
1156
|
+
{ 'selected_cases' => selected, 'executed_candidate_cases' => selected - unsupported.length,
|
|
1157
|
+
'unsupported_case_ids' => unsupported.map { |item| item['case_id'] },
|
|
1158
|
+
'unsupported_is_quality_failure' => false }
|
|
1159
|
+
end
|
|
1160
|
+
|
|
1161
|
+
def competitive
|
|
1162
|
+
results = @results.select { |item| item['adapter_role'] == 'competitor' }
|
|
1163
|
+
{
|
|
1164
|
+
'configured' => @run.fetch('competitors'),
|
|
1165
|
+
'outcomes' => results.group_by { |item| item['outcome'] }.transform_values(&:length),
|
|
1166
|
+
'unsupported_case_ids' => results.filter_map do |item|
|
|
1167
|
+
item['case_id'] if item['outcome'] == 'unsupported'
|
|
1168
|
+
end,
|
|
1169
|
+
'false_auto_merge_result_ids' => results.filter_map do |item|
|
|
1170
|
+
item['id'] if item['outcome'] == 'false_auto_merge'
|
|
1171
|
+
end,
|
|
1172
|
+
'affects_candidate_safety_gate' => false
|
|
1173
|
+
}
|
|
1174
|
+
end
|
|
1175
|
+
|
|
1176
|
+
def strata
|
|
1177
|
+
case_results = @results.select { |item| item['adapter_role'] == 'candidate' }
|
|
1178
|
+
tag_fields = %w[operation partition family provider dialect false_auto_merge_severity]
|
|
1179
|
+
tag_strata = tag_fields.to_h do |field|
|
|
1180
|
+
[field, case_results.group_by { |item| item.dig('case_tags', field) }.transform_values(&:length)]
|
|
1181
|
+
end
|
|
1182
|
+
transitions = @results.group_by { |item| item['case_id'] }.values
|
|
1183
|
+
tag_strata.merge(
|
|
1184
|
+
'capability' => case_results.each_with_object(Hash.new(0)) do |item, counts|
|
|
1185
|
+
item.dig('case_tags', 'capabilities').each { |capability| counts[capability] += 1 }
|
|
1186
|
+
end,
|
|
1187
|
+
'outcome' => @results.group_by { |item| item['outcome'] }.transform_values(&:length),
|
|
1188
|
+
'adapter' => @results.group_by { |item| item['adapter_id'] }.transform_values(&:length),
|
|
1189
|
+
'transition' => transitions.each_with_object(Hash.new(0)) do |items, counts|
|
|
1190
|
+
pair = transition(items)
|
|
1191
|
+
counts["#{pair['from']}->#{pair['to']}"] += 1
|
|
1192
|
+
end
|
|
1193
|
+
)
|
|
1194
|
+
end
|
|
1195
|
+
end
|
|
1196
|
+
# rubocop:enable Metrics/AbcSize, Metrics/ClassLength, Metrics/CyclomaticComplexity, Metrics/MethodLength, Metrics/PerceivedComplexity
|
|
1197
|
+
end
|
|
1198
|
+
end
|
|
1199
|
+
end
|