ast-merge-git 7.1.1 → 7.1.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1199 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'digest'
4
+ require 'fileutils'
5
+ require 'json'
6
+ require 'open3'
7
+ require 'rbconfig'
8
+
9
+ module Ast
10
+ module Merge
11
+ module Git
12
+ # Offline Slice 1022-compatible corpus validation and deterministic selection.
13
+ # rubocop:disable Metrics/AbcSize, Metrics/ClassLength, Metrics/CyclomaticComplexity, Metrics/MethodLength, Metrics/PerceivedComplexity -- benchmark evidence is intentionally explicit
14
+ class LocalBenchmark
15
+ Error = Class.new(StandardError)
16
+ SCHEMA = 'structuredmerge.benchmark.corpus/v1'
17
+ CASE_SCHEMA = 'structuredmerge.benchmark/v1'
18
+ OPERATIONS = %w[merge3 metamorphic diff].freeze
19
+ EXECUTABLE_OPERATIONS = %w[merge3 metamorphic].freeze
20
+ PARTITIONS = %w[sentinel gold metamorphic].freeze
21
+ EXPECTATIONS = %w[clean conflict error excluded_ambiguous].freeze
22
+ SEVERITIES = %w[none low high critical].freeze
23
+ PRESERVATION = %w[required allowed_to_change not_applicable].freeze
24
+ TRANSFORMATIONS = %w[
25
+ rename move reorder formatting comment independent_edit delete_modify duplicate_key_identity
26
+ schema_aware_mutation
27
+ ].freeze
28
+ METAMORPHIC_INVARIANTS = %w[
29
+ comment-retained no-semantic-edit same-json-value same-jsonc-value
30
+ ].freeze
31
+ ID_PATTERN = /\A[a-z0-9]+(?:[.-][a-z0-9]+)*\z/
32
+ PROVENANCE_FIELDS = %w[
33
+ origin_uri revision spdx_license license_evidence_uri authorship author_review reviewer derivation
34
+ ].freeze
35
+ SELECTOR_FIELDS = %w[provider_id family dialect backend profile require].freeze
36
+ INPUT_ROLES = {
37
+ 'merge3' => %w[base ours theirs],
38
+ 'metamorphic' => %w[source transformed],
39
+ 'diff' => %w[before after]
40
+ }.freeze
41
+
42
+ attr_reader :document, :corpus_digest, :path
43
+
44
+ def self.load(path)
45
+ source = File.binread(path)
46
+ new(JSON.parse(source), path: path, corpus_digest: Digest::SHA256.hexdigest(source)).tap(&:validate!)
47
+ rescue JSON::ParserError => e
48
+ raise Error, "invalid corpus JSON: #{e.message}"
49
+ rescue SystemCallError => e
50
+ raise Error, "cannot read corpus: #{e.message}"
51
+ end
52
+
53
+ def initialize(document, path: nil, corpus_digest: nil)
54
+ @document = document
55
+ @path = path && Pathname(path).expand_path
56
+ @corpus_digest = corpus_digest || digest(canonical_json(document))
57
+ end
58
+
59
+ def validate!
60
+ require_keys(document, %w[schema_version kind id version extends provenance profiles capability_map
61
+ selection competitors cases expected_summary], 'corpus')
62
+ error!('unsupported corpus schema') unless document['schema_version'] == SCHEMA
63
+ error!('corpus must extend Slice 1022 v1') unless document['extends'] == CASE_SCHEMA
64
+ error!('network must be denied') unless document['network_policy'] == 'denied'
65
+ error!('services must be empty') unless document['services'] == []
66
+ validate_provenance!(document['provenance'], 'corpus')
67
+ validate_profiles!
68
+ validate_capability_map!
69
+ validate_competitors!
70
+ validate_cases!
71
+ validate_summary!
72
+ true
73
+ end
74
+
75
+ def cases
76
+ validate!
77
+ document.fetch('cases')
78
+ end
79
+
80
+ def select(profile:, changed_paths: [])
81
+ validate!
82
+ profile = profile.to_s
83
+ definition = document.fetch('profiles')[profile]
84
+ error!("unknown profile: #{profile}") unless definition
85
+ paths = changed_paths.map(&:to_s).uniq.sort
86
+ inferred = paths.to_h { |changed| [changed, capabilities_for(changed)] }
87
+ capabilities = inferred.values.flatten.uniq.sort
88
+ direct = cases.select { |item| direct_case?(item, capabilities) }.map { |item| item.fetch('id') }
89
+ sentinels = definition.fetch('mandatory_sentinels')
90
+ selected = ordered(sentinels + direct)
91
+ population = cases.map { |item| item.fetch('id') } - selected
92
+ neighbors = neighbor_order(population).first(definition.fetch('neighbor_count'))
93
+ selected = ordered(selected + neighbors)
94
+ selected = sentinels if profile == 'micro'
95
+
96
+ {
97
+ 'profile' => profile,
98
+ 'seed' => document.dig('selection', 'seed'),
99
+ 'selected_case_ids' => selected,
100
+ 'excluded_case_ids' => cases.map { |item| item.fetch('id') } - selected,
101
+ 'changed_paths' => inferred.map { |changed, caps| { 'path' => changed, 'capabilities' => caps } },
102
+ 'inferred_capabilities' => capabilities,
103
+ 'direct_cases' => direct,
104
+ 'direct_case_reasons' => capabilities.to_h do |capability|
105
+ matching = cases.filter_map do |item|
106
+ item['id'] if direct_case?(item, [capability])
107
+ end
108
+ [capability, matching]
109
+ end,
110
+ 'sentinels' => sentinels,
111
+ 'neighbor_sample' => {
112
+ 'population' => population,
113
+ 'ordering_algorithm' => document.dig('selection', 'neighbor_order'),
114
+ 'seed' => document.dig('selection', 'seed'),
115
+ 'selected_case_ids' => neighbors
116
+ },
117
+ 'unsupported_selected_cases' => selected.reject do |id|
118
+ EXECUTABLE_OPERATIONS.include?(case_by_id(id)['operation'])
119
+ end,
120
+ 'budgets' => definition.fetch('budgets'),
121
+ 'explanation' => explanation(
122
+ profile,
123
+ inferred: inferred,
124
+ direct: direct,
125
+ sentinels: sentinels,
126
+ population: population,
127
+ neighbors: neighbors
128
+ )
129
+ }
130
+ end
131
+
132
+ def case_by_id(id)
133
+ document.fetch('cases').find { |item| item['id'] == id } || error!("unknown case: #{id}")
134
+ end
135
+
136
+ private
137
+
138
+ def validate_profiles!
139
+ error!('profiles must be exactly micro and dev') unless document['profiles'].keys.sort == %w[dev micro]
140
+ document['profiles'].each do |name, profile|
141
+ require_keys(profile, %w[mandatory_sentinels neighbor_count budgets], "profile #{name}")
142
+ require_keys(profile['budgets'], %w[wall_seconds case_count output_bytes], "profile #{name} budgets")
143
+ end
144
+ end
145
+
146
+ def validate_capability_map!
147
+ error!('capability_map must not be empty') unless document['capability_map'].is_a?(Array) &&
148
+ document['capability_map'].any?
149
+ document['capability_map'].each do |entry|
150
+ require_keys(entry, %w[path_prefix capabilities], 'capability map entry')
151
+ error!('capability map path must be relative') if Pathname(entry['path_prefix']).absolute?
152
+ unless entry['capabilities'] == entry['capabilities'].sort
153
+ error!('capability map capabilities must be sorted')
154
+ end
155
+ end
156
+ end
157
+
158
+ def validate_competitors!
159
+ document.fetch('competitors').each do |id, competitor|
160
+ require_keys(
161
+ competitor,
162
+ %w[adapter_id source_url source_revision version spdx_license reuse_posture toolchain
163
+ build_command operations dialects],
164
+ "competitor #{id}"
165
+ )
166
+ error!("competitor #{id}: adapter ID differs") unless competitor['adapter_id'] == id
167
+ unless /\A[0-9a-f]{40}\z/.match?(competitor['source_revision'])
168
+ error!("competitor #{id}: source revision must be a full SHA")
169
+ end
170
+ error!("competitor #{id}: only merge3 is admitted") unless competitor['operations'] == ['merge3']
171
+ next if competitor['dialects'] == competitor['dialects'].sort
172
+
173
+ error!("competitor #{id}: dialects must be sorted")
174
+ end
175
+ end
176
+
177
+ def validate_cases!
178
+ records = document['cases']
179
+ error!('cases must not be empty') unless records.is_a?(Array) && records.any?
180
+ ids = records.map { |item| validate_case!(item) }
181
+ error!('duplicate case ID') unless ids.uniq.length == ids.length
182
+ id_set = ids.to_h { |id| [id, true] }
183
+ records.select { |item| item['operation'] == 'metamorphic' }.each do |item|
184
+ error!("#{item['id']}: parent case is dangling") unless id_set[item['parent_case_id']]
185
+ end
186
+ sentinels = records.select { |item| item['partition'] == 'sentinel' }.map { |item| item['id'] }
187
+ document['profiles'].each_value do |profile|
188
+ error!('profile sentinels differ from corpus sentinels') unless profile['mandatory_sentinels'] == sentinels
189
+ end
190
+ end
191
+
192
+ def validate_case!(item)
193
+ require_keys(item, %w[schema_version kind id operation family provider dialect capabilities partition
194
+ provenance oracle acceptable_equivalence preservation_policy
195
+ false_auto_merge_severity selector inputs independent_edits independent_edit_ids
196
+ expected_conflict_regions], 'case')
197
+ id = item['id']
198
+ error!("#{id}: invalid stable case ID") unless ID_PATTERN.match?(id.to_s)
199
+ error!("#{id}: incompatible case schema") unless item['schema_version'] == CASE_SCHEMA
200
+ error!("#{id}: invalid kind") unless item['kind'] == 'benchmark_case'
201
+ error!("#{id}: unsupported operation") unless OPERATIONS.include?(item['operation'])
202
+ error!("#{id}: unsupported partition") unless PARTITIONS.include?(item['partition'])
203
+ unless item['capabilities'] == item['capabilities'].uniq.sort
204
+ error!("#{id}: capabilities must be unique and sorted")
205
+ end
206
+ error!("#{id}: unsupported severity") unless SEVERITIES.include?(item['false_auto_merge_severity'])
207
+ validate_provenance!(item['provenance'], id)
208
+ validate_oracle!(item, id)
209
+ validate_selector!(item['selector'], id)
210
+ validate_inputs!(item, id)
211
+ validate_edits!(item, id)
212
+ validate_preservation!(item, id)
213
+ validate_operation!(item, id)
214
+ id
215
+ end
216
+
217
+ def validate_provenance!(provenance, label)
218
+ require_keys(provenance, PROVENANCE_FIELDS, "#{label} provenance")
219
+ error!("#{label}: authorship must be reviewed") unless provenance['author_review'] == 'reviewed'
220
+ error!("#{label}: SPDX license is required") if provenance['spdx_license'].to_s.empty?
221
+ end
222
+
223
+ def validate_oracle!(item, id)
224
+ oracle = item['oracle']
225
+ require_keys(oracle, %w[class artifact admission score_eligible procedure], "#{id} oracle")
226
+ validate_inline!(oracle['artifact'], "#{id} oracle artifact")
227
+ error!("#{id}: oracle procedure cannot accept parse validity alone") if oracle['procedure'].to_s.empty?
228
+ error!("#{id}: acceptable equivalence must be explicit") unless item['acceptable_equivalence'].is_a?(Array) &&
229
+ item['acceptable_equivalence'].any?
230
+ return unless oracle['class'] == 'exact' && item.dig('expected', 'outcome') == 'clean'
231
+ return if oracle.dig('artifact', 'bytes') == item.dig('expected', 'output', 'bytes')
232
+
233
+ error!("#{id}: exact oracle artifact must match expected output bytes")
234
+ end
235
+
236
+ def validate_selector!(selector, id)
237
+ require_keys(selector, SELECTOR_FIELDS, "#{id} selector")
238
+ end
239
+
240
+ def validate_inputs!(item, id)
241
+ roles = INPUT_ROLES.fetch(item['operation'])
242
+ require_keys(item['inputs'], roles, "#{id} inputs")
243
+ item['inputs'].each { |role, record| validate_inline!(record, "#{id} #{role}") }
244
+ end
245
+
246
+ def validate_inline!(record, label)
247
+ require_keys(record, %w[mode bytes sha256], label)
248
+ error!("#{label}: only inline authored evidence is admitted") unless record['mode'] == 'inline'
249
+ error!("#{label}: input exceeds Slice 1022 inline limit") if record['bytes'].bytesize > 4096
250
+ error!("#{label}: SHA-256 does not match exact bytes") unless record['sha256'] == digest(record['bytes'])
251
+ end
252
+
253
+ def validate_edits!(item, id)
254
+ edits = item['independent_edits']
255
+ error!("#{id}: independent edits must be an array") unless edits.is_a?(Array)
256
+ edit_ids = edits.map { |edit| edit.fetch('id') }
257
+ error!("#{id}: duplicate independent edit ID") unless edit_ids.uniq.length == edit_ids.length
258
+ error!("#{id}: independent edit IDs differ") unless item['independent_edit_ids'] == edit_ids
259
+ end
260
+
261
+ def validate_preservation!(item, id)
262
+ required = %w[comments formatting order encoding line_endings unknown_fields source_regions]
263
+ require_keys(item['preservation_policy'], required, "#{id} preservation")
264
+ return if item['preservation_policy'].values.all? { |value| PRESERVATION.include?(value) }
265
+
266
+ error!("#{id}: invalid preservation requirement")
267
+ end
268
+
269
+ def validate_operation!(item, id)
270
+ if item['operation'] == 'merge3'
271
+ require_keys(item, %w[expected expected_conflicts], id)
272
+ expectation = item.dig('expected', 'outcome')
273
+ error!("#{id}: unsupported expected outcome") unless EXPECTATIONS.include?(expectation)
274
+ output = item.dig('expected', 'output')
275
+ validate_inline!(output, "#{id} expected output") if output
276
+ expected_conflict = expectation == 'conflict'
277
+ error!("#{id}: conflict expectation mismatch") unless item['expected_conflicts'] == expected_conflict
278
+ elsif item['operation'] == 'metamorphic'
279
+ validate_metamorphic!(item, id)
280
+ end
281
+ end
282
+
283
+ def validate_metamorphic!(item, id)
284
+ require_keys(item, %w[generator parent_case_id transformations expected_invariants], id)
285
+ require_keys(item['generator'], %w[id version sha256 seed], "#{id} generator")
286
+ digest = item.dig('generator', 'sha256')
287
+ error!("#{id}: generator SHA-256 is malformed") unless /\A[0-9a-f]{64}\z/.match?(digest)
288
+ error!("#{id}: expected invariants must not be empty") if item['expected_invariants'].empty?
289
+ unknown_invariants = item['expected_invariants'] - METAMORPHIC_INVARIANTS
290
+ error!("#{id}: unsupported expected invariants: #{unknown_invariants.join(', ')}") if unknown_invariants.any?
291
+ item['transformations'].each do |transformation|
292
+ require_keys(transformation, %w[id type parameters], "#{id} transformation")
293
+ error!("#{id}: unknown transformation") unless TRANSFORMATIONS.include?(transformation['type'])
294
+ next if transformation.dig('parameters', 'deterministic')
295
+
296
+ error!("#{id}: transformation must be deterministic")
297
+ end
298
+ end
299
+
300
+ def validate_summary!
301
+ summary = document['expected_summary']
302
+ error!('expected case count differs') unless summary['case_count'] == document['cases'].length
303
+ actual = document['cases'].group_by { |item| item['partition'] }.transform_values(&:length)
304
+ error!('expected partition counts differ') unless summary['partition_counts'] == actual
305
+ operations = document['cases'].group_by { |item| item['operation'] }.transform_values(&:length)
306
+ error!('expected operation counts differ') unless summary['operation_counts'] == operations
307
+ families = document['cases'].map { |item| item['family'] }.uniq.sort
308
+ error!('expected families differ') unless summary['families'] == families
309
+ expected_micro = document.dig('profiles', 'micro', 'mandatory_sentinels')
310
+ error!('expected micro case IDs differ') unless summary['micro_case_ids'] == expected_micro
311
+ end
312
+
313
+ def capabilities_for(path)
314
+ document['capability_map'].filter_map do |entry|
315
+ entry['capabilities'] if path.start_with?(entry['path_prefix'])
316
+ end.flatten.uniq.sort
317
+ end
318
+
319
+ def direct_case?(item, capabilities)
320
+ case_capabilities = item['capabilities'] + [item['family'], item['dialect']]
321
+ (case_capabilities & capabilities).any?
322
+ end
323
+
324
+ def ordered(ids)
325
+ order = document['cases'].map { |item| item['id'] }
326
+ ids.uniq.sort_by { |id| order.index(id) }
327
+ end
328
+
329
+ def neighbor_order(ids)
330
+ seed = document.dig('selection', 'seed')
331
+ ids.sort_by { |id| [digest("#{seed}\0#{id}"), id] }
332
+ end
333
+
334
+ def explanation(profile, details)
335
+ {
336
+ 'profile_rule' => profile == 'micro' ? 'mandatory sentinels only' : 'sentinels + direct + neighbors',
337
+ 'changed_paths' => details[:inferred].map { |path, caps| "#{path} => #{caps.join(',')}" },
338
+ 'direct_cases' => details[:direct],
339
+ 'sentinels' => details[:sentinels],
340
+ 'neighbors' => {
341
+ 'population' => details[:population],
342
+ 'algorithm' => document.dig('selection', 'neighbor_order'),
343
+ 'selected' => details[:neighbors]
344
+ },
345
+ 'budget_rule' => 'selection fits declared case budget; no silent extension or dropping'
346
+ }
347
+ end
348
+
349
+ def require_keys(hash, keys, label)
350
+ error!("#{label} must be an object") unless hash.is_a?(Hash)
351
+ missing = keys.reject { |key| hash.key?(key) }
352
+ error!("#{label} missing: #{missing.join(', ')}") if missing.any?
353
+ end
354
+
355
+ def digest(content)
356
+ Digest::SHA256.hexdigest(content)
357
+ end
358
+
359
+ def canonical_json(value)
360
+ JSON.generate(deep_sort(value))
361
+ end
362
+
363
+ def deep_sort(value)
364
+ case value
365
+ when Hash then value.keys.sort.to_h { |key| [key, deep_sort(value.fetch(key))] }
366
+ when Array then value.map { |item| deep_sort(item) }
367
+ else value
368
+ end
369
+ end
370
+
371
+ def error!(message)
372
+ raise Error, message
373
+ end
374
+ end
375
+
376
+ # Executes the same authored merge bytes through Git and the installed driver.
377
+ class LocalBenchmarkRunner
378
+ DEFAULT_TIMEOUT = 30
379
+ MERGIRAF_EXTENSIONS = {
380
+ 'bash' => 'sh',
381
+ 'html' => 'html',
382
+ 'json' => 'json',
383
+ 'markdown' => 'md',
384
+ 'ruby' => 'rb',
385
+ 'toml' => 'toml',
386
+ 'typescript' => 'ts',
387
+ 'yaml' => 'yaml'
388
+ }.freeze
389
+
390
+ def initialize(benchmark:, driver_path:, tmp_root:, timeout: DEFAULT_TIMEOUT, competitor_paths: {})
391
+ @benchmark = benchmark
392
+ @driver_path = Pathname(driver_path).expand_path
393
+ @tmp_root = Pathname(tmp_root).expand_path
394
+ @timeout = Integer(timeout)
395
+ @competitor_paths = competitor_paths.transform_values { |path| Pathname(path).expand_path }
396
+ end
397
+
398
+ def run(profile:, changed_paths: [])
399
+ verify_environment!
400
+ selection = @benchmark.select(profile: profile, changed_paths: changed_paths)
401
+ original_cwd = Dir.pwd
402
+ results = selection['selected_case_ids'].flat_map do |id|
403
+ item = @benchmark.case_by_id(id)
404
+ execute_case(item) + execute_competitors(item)
405
+ end
406
+ raise LocalBenchmark::Error, 'runner changed its working directory' unless Dir.pwd == original_cwd
407
+
408
+ {
409
+ 'schema_version' => 'structuredmerge.benchmark.run/v1',
410
+ 'kind' => 'paired_local_run',
411
+ 'corpus_id' => @benchmark.document['id'],
412
+ 'corpus_digest' => @benchmark.corpus_digest,
413
+ 'selection' => selection,
414
+ 'cache_identity' => cache_identity(selection),
415
+ 'competitors' => competitor_provenance,
416
+ 'results' => results
417
+ }
418
+ ensure
419
+ Dir.chdir(original_cwd) if original_cwd && Dir.pwd != original_cwd
420
+ end
421
+
422
+ private
423
+
424
+ def verify_environment!
425
+ error!("missing installed driver: #{@driver_path}") unless @driver_path.file? && @driver_path.executable?
426
+ @competitor_paths.each do |id, path|
427
+ error!("unknown competitor: #{id}") unless @benchmark.document.fetch('competitors').key?(id)
428
+ error!("missing competitor executable: #{path}") unless path.file? && path.executable?
429
+ end
430
+ root = Pathname(__dir__).join('..', '..', '..', '..').realpath
431
+ resolved = @tmp_root.exist? ? @tmp_root.realpath : @tmp_root.dirname.realpath.join(@tmp_root.basename)
432
+ error!('tmp_root must be inside the ast-merge-git repository') unless resolved.to_s.start_with?("#{root}/")
433
+ end
434
+
435
+ def execute_case(item)
436
+ return execute_merge3_case(item) if item['operation'] == 'merge3'
437
+ return execute_metamorphic_case(item) if item['operation'] == 'metamorphic'
438
+
439
+ unsupported_pair(item)
440
+ end
441
+
442
+ def execute_competitors(item)
443
+ @competitor_paths.map do |id, path|
444
+ metadata = @benchmark.document.dig('competitors', id)
445
+ unless metadata['operations'].include?(item['operation']) && metadata['dialects'].include?(item['dialect'])
446
+ next raw_result(
447
+ item,
448
+ id,
449
+ nil,
450
+ outcome: 'unsupported',
451
+ unsupported_reason: "#{id} does not support #{item['operation']}/#{item['dialect']}"
452
+ )
453
+ end
454
+
455
+ execute_mergiraf(item, path)
456
+ end
457
+ end
458
+
459
+ def execute_mergiraf(item, path)
460
+ workspace = @tmp_root.join("#{safe_id(item['id'])}-mergiraf-#{Process.pid}")
461
+ FileUtils.rm_rf(workspace)
462
+ FileUtils.mkdir_p(workspace)
463
+ write_roles(item, workspace)
464
+ output = workspace.join('output')
465
+ capture = timed_capture(
466
+ oracle_free_env,
467
+ path.to_s,
468
+ 'merge',
469
+ 'base',
470
+ 'ours',
471
+ 'theirs',
472
+ '--output',
473
+ output.to_s,
474
+ '--path-name',
475
+ "#{item['id']}.#{MERGIRAF_EXTENSIONS.fetch(item['dialect'])}",
476
+ '--conflict-marker-size',
477
+ '7',
478
+ chdir: workspace
479
+ )
480
+ capture[:output] = output.file? ? output.binread : ''
481
+ raw_result(item, 'mergiraf', capture)
482
+ ensure
483
+ FileUtils.rm_rf(workspace) if workspace
484
+ end
485
+
486
+ def execute_merge3_case(item)
487
+ workspace = @tmp_root.join("#{safe_id(item['id'])}-#{Process.pid}")
488
+ FileUtils.rm_rf(workspace)
489
+ FileUtils.mkdir_p(workspace)
490
+ baseline = execute_baseline(item, workspace)
491
+ candidate = execute_candidate(item, workspace)
492
+ rerun = execute_candidate(item, workspace)
493
+ candidate['deterministic_correctness_rerun'] = correctness_record(candidate) == correctness_record(rerun)
494
+ [baseline, candidate]
495
+ ensure
496
+ FileUtils.rm_rf(workspace) if workspace
497
+ end
498
+
499
+ def execute_metamorphic_case(item)
500
+ workspace = @tmp_root.join("#{safe_id(item['id'])}-#{Process.pid}")
501
+ FileUtils.rm_rf(workspace)
502
+ FileUtils.mkdir_p(workspace)
503
+ write_metamorphic_roles(item, workspace)
504
+ baseline = execute_metamorphic_baseline(item, workspace)
505
+ candidate = execute_metamorphic_candidate(item, workspace)
506
+ rerun = execute_metamorphic_candidate(item, workspace)
507
+ candidate['deterministic_correctness_rerun'] = correctness_record(candidate) == correctness_record(rerun)
508
+ [baseline, candidate]
509
+ ensure
510
+ FileUtils.rm_rf(workspace) if workspace
511
+ end
512
+
513
+ def unsupported_pair(item)
514
+ %w[git.unsupported structuredmerge.unsupported].map do |adapter|
515
+ raw_result(item, adapter, nil, outcome: 'unsupported',
516
+ unsupported_reason: "no installed #{item['operation']} adapter")
517
+ end
518
+ end
519
+
520
+ def execute_baseline(item, workspace)
521
+ write_roles(item, workspace)
522
+ capture = timed_capture({}, 'git', 'merge-file', '-p', 'ours', 'base', 'theirs', chdir: workspace)
523
+ capture[:output] = capture[:stdout]
524
+ raw_result(item, 'git.merge-file', capture)
525
+ end
526
+
527
+ def execute_candidate(item, workspace)
528
+ write_roles(item, workspace)
529
+ selector = item.fetch('selector')
530
+ capture = timed_capture(
531
+ candidate_env(selector),
532
+ @driver_path.to_s, 'base', 'ours', 'theirs', "#{item['id']}.#{item['dialect']}", '7',
533
+ chdir: workspace
534
+ )
535
+ capture[:output] = workspace.join('ours').binread
536
+ raw_result(item, 'ast-merge-git', capture)
537
+ end
538
+
539
+ def write_roles(item, workspace)
540
+ %w[base ours theirs].each do |role|
541
+ workspace.join(role).binwrite(item.dig('inputs', role, 'bytes'))
542
+ end
543
+ end
544
+
545
+ def write_metamorphic_roles(item, workspace)
546
+ %w[source transformed].each do |role|
547
+ workspace.join(role).binwrite(item.dig('inputs', role, 'bytes'))
548
+ end
549
+ end
550
+
551
+ def execute_metamorphic_baseline(item, workspace)
552
+ capture = timed_capture(
553
+ {},
554
+ 'git', 'diff', '--no-index', '--no-color', '--no-ext-diff', '--', 'source', 'transformed',
555
+ chdir: workspace
556
+ )
557
+ capture[:output] = capture[:stdout]
558
+ metamorphic_result(
559
+ item,
560
+ 'git.diff',
561
+ 'baseline',
562
+ capture,
563
+ edit_signal: capture[:status] == 1
564
+ )
565
+ end
566
+
567
+ def execute_metamorphic_candidate(item, workspace)
568
+ capture, changes = capture_provider_diff(item, workspace)
569
+ metamorphic_result(
570
+ item,
571
+ 'ast-merge-provider.diff2',
572
+ 'candidate',
573
+ capture,
574
+ edit_signal: changes.any?
575
+ )
576
+ end
577
+
578
+ def capture_provider_diff(item, workspace)
579
+ selector = item.fetch('selector')
580
+ capture = timed_capture(
581
+ candidate_env(selector),
582
+ @driver_path.to_s,
583
+ 'benchmark-provider-diff',
584
+ 'source',
585
+ 'transformed',
586
+ "#{item['id']}.#{item['dialect']}",
587
+ chdir: workspace
588
+ )
589
+ result = capture[:stdout].empty? ? {} : JSON.parse(capture[:stdout])
590
+ changes = capture[:status].zero? ? result.fetch('changes') : []
591
+ capture[:output] = JSON.generate(changes)
592
+ [capture, changes]
593
+ rescue JSON::ParserError, KeyError => e
594
+ capture ||= { stdout: '', stderr: '', status: 2, duration_ns: 0 }
595
+ capture[:stderr] = [capture[:stderr], e.message].reject(&:empty?).join("\n")
596
+ capture[:status] = 2
597
+ capture[:output] = ''
598
+ [capture, []]
599
+ end
600
+
601
+ def timed_capture(env, *command, chdir:)
602
+ started = Process.clock_gettime(Process::CLOCK_MONOTONIC, :nanosecond)
603
+ stdin, stdout_io, stderr_io, process = Open3.popen3(env, *command, chdir: chdir.to_s)
604
+ [stdin, stdout_io, stderr_io].each(&:binmode)
605
+ stdin.close
606
+ stdout_reader = Thread.new { stdout_io.read }
607
+ stderr_reader = Thread.new { stderr_io.read }
608
+ timed_out = process.join(@timeout).nil?
609
+ terminate_process(process) if timed_out
610
+ stdout = stdout_reader.value
611
+ stderr = stderr_reader.value
612
+ stderr = [stderr, "timeout after #{@timeout}s"].reject(&:empty?).join("\n") if timed_out
613
+ { stdout: stdout, stderr: stderr, status: timed_out ? 2 : process.value.exitstatus,
614
+ duration_ns: Process.clock_gettime(Process::CLOCK_MONOTONIC, :nanosecond) - started }
615
+ rescue Errno::ENOENT => e
616
+ { stdout: '', stderr: e.message, status: 2, output: '', duration_ns: 0 }
617
+ ensure
618
+ [stdin, stdout_io, stderr_io].compact.each { |io| io.close unless io.closed? }
619
+ end
620
+
621
+ def terminate_process(process)
622
+ Process.kill('TERM', process.pid)
623
+ return if process.join(1)
624
+
625
+ Process.kill('KILL', process.pid)
626
+ process.join
627
+ rescue Errno::ESRCH, Errno::ECHILD
628
+ process.join
629
+ end
630
+
631
+ def raw_result(item, adapter, capture, outcome: nil, unsupported_reason: nil)
632
+ output = capture&.fetch(:output, '') || ''
633
+ checks = equivalence_checks(item, output)
634
+ markers = conflict_regions(output)
635
+ classified = outcome || classify(item, capture.fetch(:status), checks)
636
+ eligible = item.dig('oracle', 'score_eligible') && !%w[unsupported excluded_ambiguous].include?(classified)
637
+ {
638
+ 'schema_version' => 'structuredmerge.benchmark.result/v1',
639
+ 'id' => "result.#{safe_id(item['id'])}.#{safe_id(adapter)}",
640
+ 'case_id' => item['id'],
641
+ 'adapter_id' => adapter,
642
+ 'adapter_role' => adapter_role(adapter),
643
+ 'case_tags' => item.slice(
644
+ 'operation', 'partition', 'family', 'provider', 'dialect',
645
+ 'capabilities', 'false_auto_merge_severity'
646
+ ),
647
+ 'outcome' => classified,
648
+ 'score_eligible' => eligible,
649
+ 'provenance' => item['provenance'],
650
+ 'process' => capture && { 'status' => capture[:status],
651
+ 'exit_classification' => exit_class(capture[:status]) },
652
+ 'raw' => {
653
+ 'stdout' => raw_record(capture&.fetch(:stdout, '') || ''),
654
+ 'stderr' => raw_record(capture&.fetch(:stderr, '') || ''),
655
+ 'output' => raw_record(output)
656
+ },
657
+ 'diagnostics' => diagnostics(capture&.fetch(:stderr, '') || '', unsupported_reason),
658
+ 'conflict_regions' => region_evidence(item, markers),
659
+ 'checks' => checks,
660
+ 'independent_edit_evidence' => independent_edit_evidence(item, checks, classified),
661
+ 'dimensions' => dimensions(item, classified, checks),
662
+ 'runtime' => capture && { 'duration_ns' => capture[:duration_ns], 'comparable' => true }
663
+ }
664
+ end
665
+
666
+ def adapter_role(adapter)
667
+ return 'baseline' if adapter.start_with?('git.')
668
+ return 'candidate' if %w[ast-merge-git ast-merge-provider.diff2 structuredmerge.unsupported].include?(adapter)
669
+
670
+ 'competitor'
671
+ end
672
+
673
+ def metamorphic_result(item, adapter, adapter_role, capture, edit_signal:)
674
+ checks = metamorphic_checks(item, edit_signal)
675
+ classified = if capture[:status].nil? || capture[:status] >= 2
676
+ 'error'
677
+ elsif checks['acceptable']
678
+ 'correct_clean'
679
+ else
680
+ 'false_conflict'
681
+ end
682
+ eligible = item.dig('oracle', 'score_eligible') && classified != 'error'
683
+ {
684
+ 'schema_version' => 'structuredmerge.benchmark.result/v1',
685
+ 'id' => "result.#{safe_id(item['id'])}.#{safe_id(adapter)}",
686
+ 'case_id' => item['id'],
687
+ 'adapter_id' => adapter,
688
+ 'adapter_role' => adapter_role,
689
+ 'case_tags' => item.slice(
690
+ 'operation', 'partition', 'family', 'provider', 'dialect',
691
+ 'capabilities', 'false_auto_merge_severity'
692
+ ),
693
+ 'outcome' => classified,
694
+ 'score_eligible' => eligible,
695
+ 'provenance' => item['provenance'],
696
+ 'process' => {
697
+ 'status' => capture[:status],
698
+ 'exit_classification' => metamorphic_exit_class(adapter_role, capture[:status])
699
+ },
700
+ 'raw' => {
701
+ 'stdout' => raw_record(capture[:stdout]),
702
+ 'stderr' => raw_record(capture[:stderr]),
703
+ 'output' => raw_record(capture[:output])
704
+ },
705
+ 'diagnostics' => diagnostics(capture[:stderr], nil),
706
+ 'conflict_regions' => region_evidence(item, []),
707
+ 'checks' => checks,
708
+ 'independent_edit_evidence' => {},
709
+ 'dimensions' => dimensions(item, classified, checks),
710
+ 'runtime' => { 'duration_ns' => capture[:duration_ns], 'comparable' => true }
711
+ }
712
+ end
713
+
714
+ def metamorphic_checks(item, edit_signal)
715
+ evaluations = item.fetch('expected_invariants').map do |invariant|
716
+ matched, evidence = evaluate_metamorphic_invariant(item, invariant, edit_signal)
717
+ { 'id' => invariant, 'matched' => matched, 'evidence' => evidence }
718
+ end
719
+ accepted = evaluations.all? { |evaluation| evaluation['matched'] }
720
+ {
721
+ 'declared_invariants' => evaluations,
722
+ 'accepted_equivalence' => accepted ? 'declared_invariants' : nil,
723
+ 'acceptable_equivalence_evaluations' => [{
724
+ 'class' => 'declared_invariants',
725
+ 'matched' => accepted
726
+ }],
727
+ 'parse_validity_only_accepted' => false,
728
+ 'generator_replay' => {
729
+ 'generator_id' => item.dig('generator', 'id'),
730
+ 'generator_version' => item.dig('generator', 'version'),
731
+ 'seed' => item.dig('generator', 'seed'),
732
+ 'transformation_ids' => item.fetch('transformations').map { |transformation| transformation['id'] },
733
+ 'authored_bytes_digest_verified' => true
734
+ },
735
+ 'preservation_violations' => [],
736
+ 'acceptable' => accepted
737
+ }
738
+ end
739
+
740
+ def evaluate_metamorphic_invariant(item, invariant, edit_signal)
741
+ case invariant
742
+ when 'same-json-value', 'same-jsonc-value'
743
+ [json_values_equal?(item), 'JSON-family values parsed from source and transformed bytes']
744
+ when 'no-semantic-edit'
745
+ [!edit_signal, 'adapter emitted no edit units']
746
+ when 'comment-retained'
747
+ [json_comments_retained?(item), 'JSONC AST comment content retained']
748
+ else
749
+ [false, 'unsupported invariant']
750
+ end
751
+ end
752
+
753
+ def json_values_equal?(item)
754
+ require 'json/merge'
755
+ selector = item.fetch('selector')
756
+ source = Json::Merge.json_value_for_source(
757
+ item.dig('inputs', 'source', 'bytes'),
758
+ dialect: selector.fetch('dialect').to_sym,
759
+ backend: selector.fetch('backend')
760
+ )
761
+ transformed = Json::Merge.json_value_for_source(
762
+ item.dig('inputs', 'transformed', 'bytes'),
763
+ dialect: selector.fetch('dialect').to_sym,
764
+ backend: selector.fetch('backend')
765
+ )
766
+ source == transformed
767
+ rescue Json::Merge::ParseError, LoadError
768
+ false
769
+ end
770
+
771
+ def json_comments_retained?(item)
772
+ require 'json/merge'
773
+ selector = item.fetch('selector')
774
+ source = Json::Merge::FileAnalysis.new(
775
+ item.dig('inputs', 'source', 'bytes'),
776
+ dialect: selector.fetch('dialect').to_sym
777
+ )
778
+ transformed = Json::Merge::FileAnalysis.new(
779
+ item.dig('inputs', 'transformed', 'bytes'),
780
+ dialect: selector.fetch('dialect').to_sym
781
+ )
782
+ source.valid? && transformed.valid? &&
783
+ source.comment_nodes.map(&:normalized_content) == transformed.comment_nodes.map(&:normalized_content)
784
+ rescue TreeHaver::Error, LoadError
785
+ false
786
+ end
787
+
788
+ def metamorphic_exit_class(adapter_role, status)
789
+ return 'error' if status.nil? || status >= 2
790
+ return status == 1 ? 'differences' : 'no_differences' if adapter_role == 'baseline'
791
+
792
+ 'analyzed'
793
+ end
794
+
795
+ def classify(item, status, checks)
796
+ expected = item.dig('expected', 'outcome')
797
+ return 'excluded_ambiguous' if expected == 'excluded_ambiguous'
798
+ return 'error' if status.nil? || status >= 2
799
+ return status == 1 ? 'false_conflict' : 'false_auto_merge' if expected == 'error'
800
+ return status == 1 ? 'true_conflict' : 'false_auto_merge' if expected == 'conflict'
801
+ return 'false_conflict' if status == 1
802
+ return 'correct_clean' if checks['acceptable']
803
+
804
+ 'false_auto_merge'
805
+ end
806
+
807
+ def equivalence_checks(item, output)
808
+ expected = item.dig('expected', 'output', 'bytes')
809
+ exact = !expected.nil? && output == expected
810
+ structural = structural_equivalence(item, output, expected)
811
+ evaluations = item.fetch('acceptable_equivalence').map do |policy|
812
+ matched = case policy.fetch('class')
813
+ when 'exact_bytes' then exact
814
+ when 'structural_ast' then structural == true && structural_provider_matches?(item, policy)
815
+ else false
816
+ end
817
+ { 'class' => policy.fetch('class'), 'matched' => matched }
818
+ end
819
+ selected = evaluations.find { |evaluation| evaluation['matched'] }
820
+ checks = {
821
+ 'exact' => exact,
822
+ 'structural' => structural,
823
+ 'structural_provider' => item.dig('selector', 'provider_id'),
824
+ 'acceptable_equivalence_evaluations' => evaluations,
825
+ 'accepted_equivalence' => selected&.fetch('class'),
826
+ 'parse_validity_only_accepted' => false
827
+ }
828
+ violations = preservation_violations(item, checks, output)
829
+ checks.merge('preservation_violations' => violations, 'acceptable' => !selected.nil? && violations.empty?)
830
+ end
831
+
832
+ def structural_equivalence(item, output, expected)
833
+ return nil unless expected
834
+
835
+ selector = item.fetch('selector')
836
+ require selector.fetch('require')
837
+ result = Ast::Merge.dispatch_provider(
838
+ :diff2,
839
+ {
840
+ provider_id: selector.fetch('provider_id'),
841
+ family: selector.fetch('family'),
842
+ dialect: selector.fetch('dialect'),
843
+ backend: selector.fetch('backend'),
844
+ profile_id: selector.fetch('profile'),
845
+ before_source: expected,
846
+ after_source: output,
847
+ path_name: "#{item['id']}.#{item['dialect']}"
848
+ }
849
+ )
850
+ result[:ok] == true && result.fetch(:changes).empty?
851
+ rescue Ast::Merge::Error, KeyError, LoadError
852
+ false
853
+ end
854
+
855
+ def structural_provider_matches?(item, policy)
856
+ policy['provider'].to_s == item.dig('selector', 'provider_id').to_s
857
+ end
858
+
859
+ def dimensions(item, outcome, checks)
860
+ eligible = item.dig('oracle', 'score_eligible')
861
+ {
862
+ 'safety' => {
863
+ 'eligible' => eligible,
864
+ 'false_auto_merge' => outcome == 'false_auto_merge',
865
+ 'severity' => item['false_auto_merge_severity'],
866
+ 'compensable' => false
867
+ },
868
+ 'effectiveness' => {
869
+ 'eligible' => eligible,
870
+ 'success' => %w[correct_clean true_conflict].include?(outcome)
871
+ },
872
+ 'preservation' => {
873
+ 'eligible' => eligible,
874
+ 'requirements' => item['preservation_policy'],
875
+ 'violations' => checks.fetch('preservation_violations')
876
+ },
877
+ 'performance' => { 'quality_offset_allowed' => false }
878
+ }
879
+ end
880
+
881
+ def preservation_violations(item, checks, output)
882
+ return [] if checks['exact']
883
+
884
+ expected = item.dig('expected', 'output', 'bytes')
885
+ return [] unless expected
886
+
887
+ item['preservation_policy'].filter_map do |name, requirement|
888
+ next unless requirement == 'required'
889
+ next unless preservation_violation?(name, output, expected, checks)
890
+
891
+ name
892
+ end
893
+ end
894
+
895
+ def preservation_violation?(name, output, expected, checks)
896
+ case name
897
+ when 'encoding' then output.encoding != expected.encoding
898
+ when 'line_endings' then line_ending_style(output) != line_ending_style(expected)
899
+ when 'unknown_fields' then checks['structural'] != true
900
+ else output != expected
901
+ end
902
+ end
903
+
904
+ def line_ending_style(content)
905
+ return 'crlf' if content.include?("\r\n")
906
+
907
+ 'lf'
908
+ end
909
+
910
+ def independent_edit_evidence(item, checks, outcome)
911
+ item['independent_edit_ids'].to_h do |id|
912
+ [id,
913
+ { 'preserved' => outcome == 'correct_clean' && checks['acceptable'], 'method' => 'oracle equivalence' }]
914
+ end
915
+ end
916
+
917
+ def conflict_regions(output)
918
+ ranges = []
919
+ start_byte = nil
920
+ offset = 0
921
+ output.each_line do |line|
922
+ start_byte = offset if line.start_with?('<<<<<<<')
923
+ if start_byte && line.start_with?('>>>>>>>')
924
+ ranges << { 'start_byte' => start_byte, 'end_byte' => offset + line.bytesize }
925
+ start_byte = nil
926
+ end
927
+ offset += line.bytesize
928
+ end
929
+ ranges
930
+ end
931
+
932
+ def region_evidence(item, observed)
933
+ expected = item['expected_conflict_regions']
934
+ observed = observed.each_with_index.map { |region, index| region.merge('id' => "observed.#{index + 1}") }
935
+ localization_status = expected.empty? && observed.empty? ? 'not_applicable' : 'unknown'
936
+ {
937
+ 'expected' => expected,
938
+ 'observed' => observed,
939
+ 'matched_region_ids' => [],
940
+ 'missed_region_ids' => expected.map { |region| region['id'] },
941
+ 'false_positive_region_ids' => observed.map { |region| region['id'] },
942
+ 'localization_status' => localization_status,
943
+ 'matching_basis' => 'unknown: expected role ranges/paths and observed output ranges have no proven mapping',
944
+ 'localization_error_bytes' => nil
945
+ }
946
+ end
947
+
948
+ def raw_record(content)
949
+ { 'inline' => content, 'bytes' => content.bytesize, 'sha256' => Digest::SHA256.hexdigest(content) }
950
+ end
951
+
952
+ def diagnostics(stderr, unsupported_reason)
953
+ lines = stderr.lines.map(&:strip).reject(&:empty?)
954
+ lines << unsupported_reason if unsupported_reason
955
+ lines.map do |line|
956
+ match = /\Aast-merge-git: ([a-z_]+):/.match(line)
957
+ category = match&.[](1) || (unsupported_reason == line ? 'unsupported' : 'process')
958
+ { 'severity' => 'error', 'category' => category, 'message' => line }
959
+ end
960
+ end
961
+
962
+ def selector_env(selector)
963
+ {
964
+ 'AST_MERGE_PROVIDER' => selector['provider_id'],
965
+ 'AST_MERGE_FAMILY' => selector['family'],
966
+ 'AST_MERGE_DIALECT' => selector['dialect'],
967
+ 'AST_MERGE_BACKEND' => selector['backend'],
968
+ 'AST_MERGE_PROFILE' => selector['profile'],
969
+ 'AST_MERGE_REQUIRE' => selector['require']
970
+ }
971
+ end
972
+
973
+ def candidate_env(selector)
974
+ oracle_free_env.merge(selector_env(selector))
975
+ end
976
+
977
+ def oracle_free_env
978
+ ENV.keys.grep(/(?:ORACLE|EXPECTED)/i).to_h { |name| [name, nil] }
979
+ end
980
+
981
+ def cache_identity(selection)
982
+ source_sha, = Open3.capture2('git', '-C', Pathname(__dir__).join('..', '..', '..', '..').to_s,
983
+ 'rev-parse', 'HEAD')
984
+ environment = {
985
+ 'ruby' => RUBY_DESCRIPTION,
986
+ 'platform' => RUBY_PLATFORM,
987
+ 'host_os' => RbConfig::CONFIG['host_os'],
988
+ 'host_cpu' => RbConfig::CONFIG['host_cpu']
989
+ }
990
+ configuration = selection['selected_case_ids'].map do |id|
991
+ @benchmark.case_by_id(id).slice('id', 'selector')
992
+ end
993
+ competitors = competitor_provenance
994
+ identity = {
995
+ 'adapter_source_sha' => source_sha.strip,
996
+ 'adapter_artifact_sha256' => Digest::SHA256.file(@driver_path).hexdigest,
997
+ 'configuration_sha256' => Digest::SHA256.hexdigest(JSON.generate(configuration)),
998
+ 'competitors' => competitors,
999
+ 'corpus_sha256' => @benchmark.corpus_digest,
1000
+ 'environment' => environment,
1001
+ 'profile' => selection['profile'],
1002
+ 'changed_paths' => selection['changed_paths'],
1003
+ 'inferred_capabilities' => selection['inferred_capabilities'],
1004
+ 'selection_explanation_sha256' => Digest::SHA256.hexdigest(
1005
+ JSON.generate(deep_sort(selection.fetch('explanation')))
1006
+ ),
1007
+ 'selected_case_ids' => selection['selected_case_ids']
1008
+ }
1009
+ identity.merge('sha256' => Digest::SHA256.hexdigest(JSON.generate(identity)))
1010
+ end
1011
+
1012
+ def competitor_provenance
1013
+ @competitor_provenance ||= @competitor_paths.to_h do |id, path|
1014
+ metadata = @benchmark.document.fetch('competitors').fetch(id)
1015
+ capture = timed_capture(oracle_free_env, path.to_s, '--version', chdir: path.dirname)
1016
+ error!("#{id} version probe failed: #{capture[:stderr].strip}") unless capture[:status].zero?
1017
+ expected = "#{metadata.fetch('adapter_id')} #{metadata.fetch('version')}"
1018
+ error!("#{id} version differs: #{capture[:stdout].strip}") unless capture[:stdout].strip == expected
1019
+
1020
+ [id, metadata.merge(
1021
+ 'binary_path' => path.to_s,
1022
+ 'binary_sha256' => Digest::SHA256.file(path).hexdigest,
1023
+ 'reported_version' => capture[:stdout].strip
1024
+ )]
1025
+ end
1026
+ end
1027
+
1028
+ def deep_sort(value)
1029
+ case value
1030
+ when Hash then value.keys.sort.to_h { |key| [key, deep_sort(value.fetch(key))] }
1031
+ when Array then value.map { |item| deep_sort(item) }
1032
+ else value
1033
+ end
1034
+ end
1035
+
1036
+ def correctness_record(result)
1037
+ result.reject { |key, _value| %w[runtime deterministic_correctness_rerun].include?(key) }
1038
+ end
1039
+
1040
+ def exit_class(status)
1041
+ { 0 => 'clean', 1 => 'conflict', 2 => 'error' }.fetch(status, 'error')
1042
+ end
1043
+
1044
+ def safe_id(value)
1045
+ value.gsub(/[^a-zA-Z0-9.-]/, '-')
1046
+ end
1047
+
1048
+ def error!(message)
1049
+ raise LocalBenchmark::Error, message
1050
+ end
1051
+ end
1052
+
1053
+ # Builds a non-scalar paired report from local benchmark raw evidence.
1054
+ class LocalBenchmarkReport
1055
+ SUCCESS = %w[correct_clean true_conflict].freeze
1056
+
1057
+ def self.build(run)
1058
+ new(run).build
1059
+ end
1060
+
1061
+ def initialize(run)
1062
+ @run = run
1063
+ @results = run.fetch('results')
1064
+ end
1065
+
1066
+ def build
1067
+ pairs = @results.group_by { |result| result['case_id'] }.values.map { |items| transition(items) }
1068
+ false_auto_merges = candidate_eligible.select { |item| item['outcome'] == 'false_auto_merge' }
1069
+ {
1070
+ 'schema_version' => 'structuredmerge.benchmark.report/v1',
1071
+ 'kind' => 'paired_aggregate_report',
1072
+ 'corpus_id' => @run['corpus_id'],
1073
+ 'corpus_digest' => @run['corpus_digest'],
1074
+ 'selection' => @run['selection'],
1075
+ 'cache_identity' => @run['cache_identity'],
1076
+ 'dimensions' => {
1077
+ 'safety' => {
1078
+ 'eligible' => candidate_eligible.length,
1079
+ 'false_auto_merge_result_ids' => false_auto_merges.map { |item| item['id'] },
1080
+ 'gate' => false_auto_merges.empty? ? 'pass' : 'fail',
1081
+ 'non_compensable' => true
1082
+ },
1083
+ 'effectiveness' => outcome_counts,
1084
+ 'preservation' => {
1085
+ 'eligible' => eligible.length,
1086
+ 'violation_result_ids' => eligible.filter_map do |item|
1087
+ item['id'] if item.dig('dimensions', 'preservation', 'violations').any?
1088
+ end
1089
+ },
1090
+ 'performance' => performance,
1091
+ 'reliability' => { 'error_result_ids' => @results.filter_map do |item|
1092
+ item['id'] if item['outcome'] == 'error'
1093
+ end },
1094
+ 'coverage' => coverage,
1095
+ 'competitive' => competitive
1096
+ },
1097
+ 'strata' => strata,
1098
+ 'transitions' => pairs,
1099
+ 'newly_passing_case_ids' => pairs.filter_map { |pair| pair['case_id'] if pair['newly_passing'] },
1100
+ 'newly_failing_case_ids' => pairs.filter_map { |pair| pair['case_id'] if pair['newly_failing'] },
1101
+ 'changed_conflict_case_ids' => pairs.filter_map { |pair| pair['case_id'] if pair['changed_conflict'] },
1102
+ 'hard_gate_failed' => false_auto_merges.any?,
1103
+ 'scalar_score' => nil
1104
+ }
1105
+ end
1106
+
1107
+ private
1108
+
1109
+ def eligible
1110
+ @eligible ||= @results.select { |item| item['score_eligible'] }
1111
+ end
1112
+
1113
+ def candidate_eligible
1114
+ eligible.select { |item| item['adapter_role'] == 'candidate' }
1115
+ end
1116
+
1117
+ def outcome_counts
1118
+ %w[correct_clean false_conflict true_conflict false_auto_merge error unsupported
1119
+ excluded_ambiguous].to_h do |outcome|
1120
+ [outcome, @results.count { |item| item['outcome'] == outcome }]
1121
+ end
1122
+ end
1123
+
1124
+ def transition(items)
1125
+ baseline = items.find { |item| item['adapter_role'] == 'baseline' }
1126
+ candidate = items.find { |item| item['adapter_role'] == 'candidate' }
1127
+ {
1128
+ 'case_id' => items.first['case_id'],
1129
+ 'baseline_result_id' => baseline['id'],
1130
+ 'candidate_result_id' => candidate['id'],
1131
+ 'from' => baseline['outcome'],
1132
+ 'to' => candidate['outcome'],
1133
+ 'newly_passing' => !SUCCESS.include?(baseline['outcome']) && SUCCESS.include?(candidate['outcome']),
1134
+ 'newly_failing' => SUCCESS.include?(baseline['outcome']) && !SUCCESS.include?(candidate['outcome']),
1135
+ 'changed_conflict' => conflict?(baseline) && conflict?(candidate) &&
1136
+ baseline.dig('raw', 'output', 'sha256') != candidate.dig('raw', 'output', 'sha256')
1137
+ }
1138
+ end
1139
+
1140
+ def conflict?(result)
1141
+ %w[false_conflict true_conflict].include?(result['outcome'])
1142
+ end
1143
+
1144
+ def performance
1145
+ @results.group_by { |item| item['adapter_id'] }.transform_values do |items|
1146
+ values = items.filter_map { |item| item.dig('runtime', 'duration_ns') }
1147
+ { 'samples' => values.length, 'total_ns' => values.sum, 'runtime_values_excluded_from_correctness' => true }
1148
+ end
1149
+ end
1150
+
1151
+ def coverage
1152
+ selected = @run.dig('selection', 'selected_case_ids').length
1153
+ unsupported = @results.select do |item|
1154
+ item['adapter_role'] == 'candidate' && item['outcome'] == 'unsupported'
1155
+ end
1156
+ { 'selected_cases' => selected, 'executed_candidate_cases' => selected - unsupported.length,
1157
+ 'unsupported_case_ids' => unsupported.map { |item| item['case_id'] },
1158
+ 'unsupported_is_quality_failure' => false }
1159
+ end
1160
+
1161
+ def competitive
1162
+ results = @results.select { |item| item['adapter_role'] == 'competitor' }
1163
+ {
1164
+ 'configured' => @run.fetch('competitors'),
1165
+ 'outcomes' => results.group_by { |item| item['outcome'] }.transform_values(&:length),
1166
+ 'unsupported_case_ids' => results.filter_map do |item|
1167
+ item['case_id'] if item['outcome'] == 'unsupported'
1168
+ end,
1169
+ 'false_auto_merge_result_ids' => results.filter_map do |item|
1170
+ item['id'] if item['outcome'] == 'false_auto_merge'
1171
+ end,
1172
+ 'affects_candidate_safety_gate' => false
1173
+ }
1174
+ end
1175
+
1176
+ def strata
1177
+ case_results = @results.select { |item| item['adapter_role'] == 'candidate' }
1178
+ tag_fields = %w[operation partition family provider dialect false_auto_merge_severity]
1179
+ tag_strata = tag_fields.to_h do |field|
1180
+ [field, case_results.group_by { |item| item.dig('case_tags', field) }.transform_values(&:length)]
1181
+ end
1182
+ transitions = @results.group_by { |item| item['case_id'] }.values
1183
+ tag_strata.merge(
1184
+ 'capability' => case_results.each_with_object(Hash.new(0)) do |item, counts|
1185
+ item.dig('case_tags', 'capabilities').each { |capability| counts[capability] += 1 }
1186
+ end,
1187
+ 'outcome' => @results.group_by { |item| item['outcome'] }.transform_values(&:length),
1188
+ 'adapter' => @results.group_by { |item| item['adapter_id'] }.transform_values(&:length),
1189
+ 'transition' => transitions.each_with_object(Hash.new(0)) do |items, counts|
1190
+ pair = transition(items)
1191
+ counts["#{pair['from']}->#{pair['to']}"] += 1
1192
+ end
1193
+ )
1194
+ end
1195
+ end
1196
+ # rubocop:enable Metrics/AbcSize, Metrics/ClassLength, Metrics/CyclomaticComplexity, Metrics/MethodLength, Metrics/PerceivedComplexity
1197
+ end
1198
+ end
1199
+ end