ast-merge-git 7.1.1 → 7.1.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,864 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'digest'
4
+ require 'fileutils'
5
+ require 'json'
6
+ require 'open3'
7
+ require 'rbconfig'
8
+
9
+ module Ast
10
+ module Merge
11
+ module Git
12
+ # Offline Slice 1022-compatible corpus validation and deterministic selection.
13
+ # rubocop:disable Metrics/AbcSize, Metrics/ClassLength, Metrics/CyclomaticComplexity, Metrics/MethodLength, Metrics/PerceivedComplexity -- benchmark evidence is intentionally explicit
14
+ class LocalBenchmark
15
+ Error = Class.new(StandardError)
16
+ SCHEMA = 'structuredmerge.benchmark.corpus/v1'
17
+ CASE_SCHEMA = 'structuredmerge.benchmark/v1'
18
+ OPERATIONS = %w[merge3 metamorphic diff].freeze
19
+ PARTITIONS = %w[sentinel gold metamorphic].freeze
20
+ EXPECTATIONS = %w[clean conflict error excluded_ambiguous].freeze
21
+ SEVERITIES = %w[none low high critical].freeze
22
+ PRESERVATION = %w[required allowed_to_change not_applicable].freeze
23
+ TRANSFORMATIONS = %w[
24
+ rename move reorder formatting comment independent_edit delete_modify duplicate_key_identity
25
+ schema_aware_mutation
26
+ ].freeze
27
+ ID_PATTERN = /\A[a-z0-9]+(?:[.-][a-z0-9]+)*\z/
28
+ PROVENANCE_FIELDS = %w[
29
+ origin_uri revision spdx_license license_evidence_uri authorship author_review reviewer derivation
30
+ ].freeze
31
+ SELECTOR_FIELDS = %w[provider_id family dialect backend profile require].freeze
32
+ INPUT_ROLES = {
33
+ 'merge3' => %w[base ours theirs],
34
+ 'metamorphic' => %w[source transformed],
35
+ 'diff' => %w[before after]
36
+ }.freeze
37
+
38
+ attr_reader :document, :corpus_digest, :path
39
+
40
+ def self.load(path)
41
+ source = File.binread(path)
42
+ new(JSON.parse(source), path: path, corpus_digest: Digest::SHA256.hexdigest(source)).tap(&:validate!)
43
+ rescue JSON::ParserError => e
44
+ raise Error, "invalid corpus JSON: #{e.message}"
45
+ rescue SystemCallError => e
46
+ raise Error, "cannot read corpus: #{e.message}"
47
+ end
48
+
49
+ def initialize(document, path: nil, corpus_digest: nil)
50
+ @document = document
51
+ @path = path && Pathname(path).expand_path
52
+ @corpus_digest = corpus_digest || digest(canonical_json(document))
53
+ end
54
+
55
+ def validate!
56
+ require_keys(document, %w[schema_version kind id version extends provenance profiles capability_map
57
+ selection cases expected_summary], 'corpus')
58
+ error!('unsupported corpus schema') unless document['schema_version'] == SCHEMA
59
+ error!('corpus must extend Slice 1022 v1') unless document['extends'] == CASE_SCHEMA
60
+ error!('network must be denied') unless document['network_policy'] == 'denied'
61
+ error!('services must be empty') unless document['services'] == []
62
+ validate_provenance!(document['provenance'], 'corpus')
63
+ validate_profiles!
64
+ validate_capability_map!
65
+ validate_cases!
66
+ validate_summary!
67
+ true
68
+ end
69
+
70
+ def cases
71
+ validate!
72
+ document.fetch('cases')
73
+ end
74
+
75
+ def select(profile:, changed_paths: [])
76
+ validate!
77
+ profile = profile.to_s
78
+ definition = document.fetch('profiles')[profile]
79
+ error!("unknown profile: #{profile}") unless definition
80
+ paths = changed_paths.map(&:to_s).uniq.sort
81
+ inferred = paths.to_h { |changed| [changed, capabilities_for(changed)] }
82
+ capabilities = inferred.values.flatten.uniq.sort
83
+ direct = cases.select { |item| direct_case?(item, capabilities) }.map { |item| item.fetch('id') }
84
+ sentinels = definition.fetch('mandatory_sentinels')
85
+ selected = ordered(sentinels + direct)
86
+ population = cases.map { |item| item.fetch('id') } - selected
87
+ neighbors = neighbor_order(population).first(definition.fetch('neighbor_count'))
88
+ selected = ordered(selected + neighbors)
89
+ selected = sentinels if profile == 'micro'
90
+
91
+ {
92
+ 'profile' => profile,
93
+ 'seed' => document.dig('selection', 'seed'),
94
+ 'selected_case_ids' => selected,
95
+ 'excluded_case_ids' => cases.map { |item| item.fetch('id') } - selected,
96
+ 'changed_paths' => inferred.map { |changed, caps| { 'path' => changed, 'capabilities' => caps } },
97
+ 'inferred_capabilities' => capabilities,
98
+ 'direct_cases' => direct,
99
+ 'direct_case_reasons' => capabilities.to_h do |capability|
100
+ matching = cases.filter_map do |item|
101
+ item['id'] if direct_case?(item, [capability])
102
+ end
103
+ [capability, matching]
104
+ end,
105
+ 'sentinels' => sentinels,
106
+ 'neighbor_sample' => {
107
+ 'population' => population,
108
+ 'ordering_algorithm' => document.dig('selection', 'neighbor_order'),
109
+ 'seed' => document.dig('selection', 'seed'),
110
+ 'selected_case_ids' => neighbors
111
+ },
112
+ 'unsupported_selected_cases' => selected.reject { |id| case_by_id(id)['operation'] == 'merge3' },
113
+ 'budgets' => definition.fetch('budgets'),
114
+ 'explanation' => explanation(
115
+ profile,
116
+ inferred: inferred,
117
+ direct: direct,
118
+ sentinels: sentinels,
119
+ population: population,
120
+ neighbors: neighbors
121
+ )
122
+ }
123
+ end
124
+
125
+ def case_by_id(id)
126
+ document.fetch('cases').find { |item| item['id'] == id } || error!("unknown case: #{id}")
127
+ end
128
+
129
+ private
130
+
131
+ def validate_profiles!
132
+ error!('profiles must be exactly micro and dev') unless document['profiles'].keys.sort == %w[dev micro]
133
+ document['profiles'].each do |name, profile|
134
+ require_keys(profile, %w[mandatory_sentinels neighbor_count budgets], "profile #{name}")
135
+ require_keys(profile['budgets'], %w[wall_seconds case_count output_bytes], "profile #{name} budgets")
136
+ end
137
+ end
138
+
139
+ def validate_capability_map!
140
+ error!('capability_map must not be empty') unless document['capability_map'].is_a?(Array) &&
141
+ document['capability_map'].any?
142
+ document['capability_map'].each do |entry|
143
+ require_keys(entry, %w[path_prefix capabilities], 'capability map entry')
144
+ error!('capability map path must be relative') if Pathname(entry['path_prefix']).absolute?
145
+ unless entry['capabilities'] == entry['capabilities'].sort
146
+ error!('capability map capabilities must be sorted')
147
+ end
148
+ end
149
+ end
150
+
151
+ def validate_cases!
152
+ records = document['cases']
153
+ error!('cases must not be empty') unless records.is_a?(Array) && records.any?
154
+ ids = records.map { |item| validate_case!(item) }
155
+ error!('duplicate case ID') unless ids.uniq.length == ids.length
156
+ id_set = ids.to_h { |id| [id, true] }
157
+ records.select { |item| item['operation'] == 'metamorphic' }.each do |item|
158
+ error!("#{item['id']}: parent case is dangling") unless id_set[item['parent_case_id']]
159
+ end
160
+ sentinels = records.select { |item| item['partition'] == 'sentinel' }.map { |item| item['id'] }
161
+ document['profiles'].each_value do |profile|
162
+ error!('profile sentinels differ from corpus sentinels') unless profile['mandatory_sentinels'] == sentinels
163
+ end
164
+ end
165
+
166
+ def validate_case!(item)
167
+ require_keys(item, %w[schema_version kind id operation family provider dialect capabilities partition
168
+ provenance oracle acceptable_equivalence preservation_policy
169
+ false_auto_merge_severity selector inputs independent_edits independent_edit_ids
170
+ expected_conflict_regions], 'case')
171
+ id = item['id']
172
+ error!("#{id}: invalid stable case ID") unless ID_PATTERN.match?(id.to_s)
173
+ error!("#{id}: incompatible case schema") unless item['schema_version'] == CASE_SCHEMA
174
+ error!("#{id}: invalid kind") unless item['kind'] == 'benchmark_case'
175
+ error!("#{id}: unsupported operation") unless OPERATIONS.include?(item['operation'])
176
+ error!("#{id}: unsupported partition") unless PARTITIONS.include?(item['partition'])
177
+ unless item['capabilities'] == item['capabilities'].uniq.sort
178
+ error!("#{id}: capabilities must be unique and sorted")
179
+ end
180
+ error!("#{id}: unsupported severity") unless SEVERITIES.include?(item['false_auto_merge_severity'])
181
+ validate_provenance!(item['provenance'], id)
182
+ validate_oracle!(item, id)
183
+ validate_selector!(item['selector'], id)
184
+ validate_inputs!(item, id)
185
+ validate_edits!(item, id)
186
+ validate_preservation!(item, id)
187
+ validate_operation!(item, id)
188
+ id
189
+ end
190
+
191
+ def validate_provenance!(provenance, label)
192
+ require_keys(provenance, PROVENANCE_FIELDS, "#{label} provenance")
193
+ error!("#{label}: authorship must be reviewed") unless provenance['author_review'] == 'reviewed'
194
+ error!("#{label}: SPDX license is required") if provenance['spdx_license'].to_s.empty?
195
+ end
196
+
197
+ def validate_oracle!(item, id)
198
+ oracle = item['oracle']
199
+ require_keys(oracle, %w[class artifact admission score_eligible procedure], "#{id} oracle")
200
+ validate_inline!(oracle['artifact'], "#{id} oracle artifact")
201
+ error!("#{id}: oracle procedure cannot accept parse validity alone") if oracle['procedure'].to_s.empty?
202
+ error!("#{id}: acceptable equivalence must be explicit") unless item['acceptable_equivalence'].is_a?(Array) &&
203
+ item['acceptable_equivalence'].any?
204
+ return unless oracle['class'] == 'exact' && item.dig('expected', 'outcome') == 'clean'
205
+ return if oracle.dig('artifact', 'bytes') == item.dig('expected', 'output', 'bytes')
206
+
207
+ error!("#{id}: exact oracle artifact must match expected output bytes")
208
+ end
209
+
210
+ def validate_selector!(selector, id)
211
+ require_keys(selector, SELECTOR_FIELDS, "#{id} selector")
212
+ end
213
+
214
+ def validate_inputs!(item, id)
215
+ roles = INPUT_ROLES.fetch(item['operation'])
216
+ require_keys(item['inputs'], roles, "#{id} inputs")
217
+ item['inputs'].each { |role, record| validate_inline!(record, "#{id} #{role}") }
218
+ end
219
+
220
+ def validate_inline!(record, label)
221
+ require_keys(record, %w[mode bytes sha256], label)
222
+ error!("#{label}: only inline authored evidence is admitted") unless record['mode'] == 'inline'
223
+ error!("#{label}: input exceeds Slice 1022 inline limit") if record['bytes'].bytesize > 4096
224
+ error!("#{label}: SHA-256 does not match exact bytes") unless record['sha256'] == digest(record['bytes'])
225
+ end
226
+
227
+ def validate_edits!(item, id)
228
+ edits = item['independent_edits']
229
+ error!("#{id}: independent edits must be an array") unless edits.is_a?(Array)
230
+ edit_ids = edits.map { |edit| edit.fetch('id') }
231
+ error!("#{id}: duplicate independent edit ID") unless edit_ids.uniq.length == edit_ids.length
232
+ error!("#{id}: independent edit IDs differ") unless item['independent_edit_ids'] == edit_ids
233
+ end
234
+
235
+ def validate_preservation!(item, id)
236
+ required = %w[comments formatting order encoding line_endings unknown_fields source_regions]
237
+ require_keys(item['preservation_policy'], required, "#{id} preservation")
238
+ return if item['preservation_policy'].values.all? { |value| PRESERVATION.include?(value) }
239
+
240
+ error!("#{id}: invalid preservation requirement")
241
+ end
242
+
243
+ def validate_operation!(item, id)
244
+ if item['operation'] == 'merge3'
245
+ require_keys(item, %w[expected expected_conflicts], id)
246
+ expectation = item.dig('expected', 'outcome')
247
+ error!("#{id}: unsupported expected outcome") unless EXPECTATIONS.include?(expectation)
248
+ output = item.dig('expected', 'output')
249
+ validate_inline!(output, "#{id} expected output") if output
250
+ expected_conflict = expectation == 'conflict'
251
+ error!("#{id}: conflict expectation mismatch") unless item['expected_conflicts'] == expected_conflict
252
+ elsif item['operation'] == 'metamorphic'
253
+ validate_metamorphic!(item, id)
254
+ end
255
+ end
256
+
257
+ def validate_metamorphic!(item, id)
258
+ require_keys(item, %w[generator parent_case_id transformations expected_invariants], id)
259
+ require_keys(item['generator'], %w[id version sha256 seed], "#{id} generator")
260
+ digest = item.dig('generator', 'sha256')
261
+ error!("#{id}: generator SHA-256 is malformed") unless /\A[0-9a-f]{64}\z/.match?(digest)
262
+ error!("#{id}: expected invariants must not be empty") if item['expected_invariants'].empty?
263
+ item['transformations'].each do |transformation|
264
+ require_keys(transformation, %w[id type parameters], "#{id} transformation")
265
+ error!("#{id}: unknown transformation") unless TRANSFORMATIONS.include?(transformation['type'])
266
+ next if transformation.dig('parameters', 'deterministic')
267
+
268
+ error!("#{id}: transformation must be deterministic")
269
+ end
270
+ end
271
+
272
+ def validate_summary!
273
+ summary = document['expected_summary']
274
+ error!('expected case count differs') unless summary['case_count'] == document['cases'].length
275
+ actual = document['cases'].group_by { |item| item['partition'] }.transform_values(&:length)
276
+ error!('expected partition counts differ') unless summary['partition_counts'] == actual
277
+ operations = document['cases'].group_by { |item| item['operation'] }.transform_values(&:length)
278
+ error!('expected operation counts differ') unless summary['operation_counts'] == operations
279
+ families = document['cases'].map { |item| item['family'] }.uniq.sort
280
+ error!('expected families differ') unless summary['families'] == families
281
+ expected_micro = document.dig('profiles', 'micro', 'mandatory_sentinels')
282
+ error!('expected micro case IDs differ') unless summary['micro_case_ids'] == expected_micro
283
+ end
284
+
285
+ def capabilities_for(path)
286
+ document['capability_map'].filter_map do |entry|
287
+ entry['capabilities'] if path.start_with?(entry['path_prefix'])
288
+ end.flatten.uniq.sort
289
+ end
290
+
291
+ def direct_case?(item, capabilities)
292
+ case_capabilities = item['capabilities'] + [item['family'], item['dialect']]
293
+ (case_capabilities & capabilities).any?
294
+ end
295
+
296
+ def ordered(ids)
297
+ order = document['cases'].map { |item| item['id'] }
298
+ ids.uniq.sort_by { |id| order.index(id) }
299
+ end
300
+
301
+ def neighbor_order(ids)
302
+ seed = document.dig('selection', 'seed')
303
+ ids.sort_by { |id| [digest("#{seed}\0#{id}"), id] }
304
+ end
305
+
306
+ def explanation(profile, details)
307
+ {
308
+ 'profile_rule' => profile == 'micro' ? 'mandatory sentinels only' : 'sentinels + direct + neighbors',
309
+ 'changed_paths' => details[:inferred].map { |path, caps| "#{path} => #{caps.join(',')}" },
310
+ 'direct_cases' => details[:direct],
311
+ 'sentinels' => details[:sentinels],
312
+ 'neighbors' => {
313
+ 'population' => details[:population],
314
+ 'algorithm' => document.dig('selection', 'neighbor_order'),
315
+ 'selected' => details[:neighbors]
316
+ },
317
+ 'budget_rule' => 'selection fits declared case budget; no silent extension or dropping'
318
+ }
319
+ end
320
+
321
+ def require_keys(hash, keys, label)
322
+ error!("#{label} must be an object") unless hash.is_a?(Hash)
323
+ missing = keys.reject { |key| hash.key?(key) }
324
+ error!("#{label} missing: #{missing.join(', ')}") if missing.any?
325
+ end
326
+
327
+ def digest(content)
328
+ Digest::SHA256.hexdigest(content)
329
+ end
330
+
331
+ def canonical_json(value)
332
+ JSON.generate(deep_sort(value))
333
+ end
334
+
335
+ def deep_sort(value)
336
+ case value
337
+ when Hash then value.keys.sort.to_h { |key| [key, deep_sort(value.fetch(key))] }
338
+ when Array then value.map { |item| deep_sort(item) }
339
+ else value
340
+ end
341
+ end
342
+
343
+ def error!(message)
344
+ raise Error, message
345
+ end
346
+ end
347
+
348
+ # Executes the same authored merge bytes through Git and the installed driver.
349
+ class LocalBenchmarkRunner
350
+ DEFAULT_TIMEOUT = 30
351
+
352
+ def initialize(benchmark:, driver_path:, tmp_root:, timeout: DEFAULT_TIMEOUT)
353
+ @benchmark = benchmark
354
+ @driver_path = Pathname(driver_path).expand_path
355
+ @tmp_root = Pathname(tmp_root).expand_path
356
+ @timeout = Integer(timeout)
357
+ end
358
+
359
+ def run(profile:, changed_paths: [])
360
+ verify_environment!
361
+ selection = @benchmark.select(profile: profile, changed_paths: changed_paths)
362
+ original_cwd = Dir.pwd
363
+ results = selection['selected_case_ids'].flat_map do |id|
364
+ execute_case(@benchmark.case_by_id(id))
365
+ end
366
+ raise LocalBenchmark::Error, 'runner changed its working directory' unless Dir.pwd == original_cwd
367
+
368
+ {
369
+ 'schema_version' => 'structuredmerge.benchmark.run/v1',
370
+ 'kind' => 'paired_local_run',
371
+ 'corpus_id' => @benchmark.document['id'],
372
+ 'corpus_digest' => @benchmark.corpus_digest,
373
+ 'selection' => selection,
374
+ 'cache_identity' => cache_identity(selection),
375
+ 'results' => results
376
+ }
377
+ ensure
378
+ Dir.chdir(original_cwd) if original_cwd && Dir.pwd != original_cwd
379
+ end
380
+
381
+ private
382
+
383
+ def verify_environment!
384
+ error!("missing installed driver: #{@driver_path}") unless @driver_path.file? && @driver_path.executable?
385
+ root = Pathname(__dir__).join('..', '..', '..', '..').realpath
386
+ resolved = @tmp_root.exist? ? @tmp_root.realpath : @tmp_root.dirname.realpath.join(@tmp_root.basename)
387
+ error!('tmp_root must be inside the ast-merge-git repository') unless resolved.to_s.start_with?("#{root}/")
388
+ end
389
+
390
+ def execute_case(item)
391
+ return unsupported_pair(item) unless item['operation'] == 'merge3'
392
+
393
+ workspace = @tmp_root.join("#{safe_id(item['id'])}-#{Process.pid}")
394
+ FileUtils.rm_rf(workspace)
395
+ FileUtils.mkdir_p(workspace)
396
+ baseline = execute_baseline(item, workspace)
397
+ candidate = execute_candidate(item, workspace)
398
+ rerun = execute_candidate(item, workspace)
399
+ candidate['deterministic_correctness_rerun'] = correctness_record(candidate) == correctness_record(rerun)
400
+ [baseline, candidate]
401
+ ensure
402
+ FileUtils.rm_rf(workspace) if workspace
403
+ end
404
+
405
+ def unsupported_pair(item)
406
+ %w[git.merge-file ast-merge-git].map do |adapter|
407
+ raw_result(item, adapter, nil, outcome: 'unsupported',
408
+ unsupported_reason: "no installed #{item['operation']} adapter")
409
+ end
410
+ end
411
+
412
+ def execute_baseline(item, workspace)
413
+ write_roles(item, workspace)
414
+ capture = timed_capture({}, 'git', 'merge-file', '-p', 'ours', 'base', 'theirs', chdir: workspace)
415
+ capture[:output] = capture[:stdout]
416
+ raw_result(item, 'git.merge-file', capture)
417
+ end
418
+
419
+ def execute_candidate(item, workspace)
420
+ write_roles(item, workspace)
421
+ selector = item.fetch('selector')
422
+ capture = timed_capture(
423
+ candidate_env(selector),
424
+ @driver_path.to_s, 'base', 'ours', 'theirs', "#{item['id']}.#{item['dialect']}", '7',
425
+ chdir: workspace
426
+ )
427
+ capture[:output] = workspace.join('ours').binread
428
+ raw_result(item, 'ast-merge-git', capture)
429
+ end
430
+
431
+ def write_roles(item, workspace)
432
+ %w[base ours theirs].each do |role|
433
+ workspace.join(role).binwrite(item.dig('inputs', role, 'bytes'))
434
+ end
435
+ end
436
+
437
+ def timed_capture(env, *command, chdir:)
438
+ started = Process.clock_gettime(Process::CLOCK_MONOTONIC, :nanosecond)
439
+ stdin, stdout_io, stderr_io, process = Open3.popen3(env, *command, chdir: chdir.to_s)
440
+ [stdin, stdout_io, stderr_io].each(&:binmode)
441
+ stdin.close
442
+ stdout_reader = Thread.new { stdout_io.read }
443
+ stderr_reader = Thread.new { stderr_io.read }
444
+ timed_out = process.join(@timeout).nil?
445
+ terminate_process(process) if timed_out
446
+ stdout = stdout_reader.value
447
+ stderr = stderr_reader.value
448
+ stderr = [stderr, "timeout after #{@timeout}s"].reject(&:empty?).join("\n") if timed_out
449
+ { stdout: stdout, stderr: stderr, status: timed_out ? 2 : process.value.exitstatus,
450
+ duration_ns: Process.clock_gettime(Process::CLOCK_MONOTONIC, :nanosecond) - started }
451
+ rescue Errno::ENOENT => e
452
+ { stdout: '', stderr: e.message, status: 2, output: '', duration_ns: 0 }
453
+ ensure
454
+ [stdin, stdout_io, stderr_io].compact.each { |io| io.close unless io.closed? }
455
+ end
456
+
457
+ def terminate_process(process)
458
+ Process.kill('TERM', process.pid)
459
+ return if process.join(1)
460
+
461
+ Process.kill('KILL', process.pid)
462
+ process.join
463
+ rescue Errno::ESRCH, Errno::ECHILD
464
+ process.join
465
+ end
466
+
467
+ def raw_result(item, adapter, capture, outcome: nil, unsupported_reason: nil)
468
+ output = capture&.fetch(:output, '') || ''
469
+ checks = equivalence_checks(item, output)
470
+ markers = conflict_regions(output)
471
+ classified = outcome || classify(item, capture.fetch(:status), checks)
472
+ eligible = item.dig('oracle', 'score_eligible') && !%w[unsupported excluded_ambiguous].include?(classified)
473
+ {
474
+ 'schema_version' => 'structuredmerge.benchmark.result/v1',
475
+ 'id' => "result.#{safe_id(item['id'])}.#{safe_id(adapter)}",
476
+ 'case_id' => item['id'],
477
+ 'adapter_id' => adapter,
478
+ 'case_tags' => item.slice(
479
+ 'operation', 'partition', 'family', 'provider', 'dialect',
480
+ 'capabilities', 'false_auto_merge_severity'
481
+ ),
482
+ 'outcome' => classified,
483
+ 'score_eligible' => eligible,
484
+ 'provenance' => item['provenance'],
485
+ 'process' => capture && { 'status' => capture[:status],
486
+ 'exit_classification' => exit_class(capture[:status]) },
487
+ 'raw' => {
488
+ 'stdout' => raw_record(capture&.fetch(:stdout, '') || ''),
489
+ 'stderr' => raw_record(capture&.fetch(:stderr, '') || ''),
490
+ 'output' => raw_record(output)
491
+ },
492
+ 'diagnostics' => diagnostics(capture&.fetch(:stderr, '') || '', unsupported_reason),
493
+ 'conflict_regions' => region_evidence(item, markers),
494
+ 'checks' => checks,
495
+ 'independent_edit_evidence' => independent_edit_evidence(item, checks, classified),
496
+ 'dimensions' => dimensions(item, classified, checks),
497
+ 'runtime' => capture && { 'duration_ns' => capture[:duration_ns], 'comparable' => true }
498
+ }
499
+ end
500
+
501
+ def classify(item, status, checks)
502
+ expected = item.dig('expected', 'outcome')
503
+ return 'excluded_ambiguous' if expected == 'excluded_ambiguous'
504
+ return 'error' if status.nil? || status >= 2
505
+ return status == 1 ? 'false_conflict' : 'false_auto_merge' if expected == 'error'
506
+ return status == 1 ? 'true_conflict' : 'false_auto_merge' if expected == 'conflict'
507
+ return 'false_conflict' if status == 1
508
+ return 'correct_clean' if checks['acceptable']
509
+
510
+ 'false_auto_merge'
511
+ end
512
+
513
+ def equivalence_checks(item, output)
514
+ expected = item.dig('expected', 'output', 'bytes')
515
+ exact = !expected.nil? && output == expected
516
+ structural = structural_equivalence(item, output, expected)
517
+ evaluations = item.fetch('acceptable_equivalence').map do |policy|
518
+ matched = case policy.fetch('class')
519
+ when 'exact_bytes' then exact
520
+ when 'structural_ast' then structural == true && structural_provider_matches?(item, policy)
521
+ else false
522
+ end
523
+ { 'class' => policy.fetch('class'), 'matched' => matched }
524
+ end
525
+ selected = evaluations.find { |evaluation| evaluation['matched'] }
526
+ checks = {
527
+ 'exact' => exact,
528
+ 'structural' => structural,
529
+ 'structural_provider' => item.dig('selector', 'provider_id'),
530
+ 'acceptable_equivalence_evaluations' => evaluations,
531
+ 'accepted_equivalence' => selected&.fetch('class'),
532
+ 'parse_validity_only_accepted' => false
533
+ }
534
+ violations = preservation_violations(item, checks, output)
535
+ checks.merge('preservation_violations' => violations, 'acceptable' => !selected.nil? && violations.empty?)
536
+ end
537
+
538
+ def structural_equivalence(item, output, expected)
539
+ return nil unless expected
540
+
541
+ selector = item.fetch('selector')
542
+ require selector.fetch('require')
543
+ result = Ast::Merge.dispatch_provider(
544
+ :diff2,
545
+ {
546
+ provider_id: selector.fetch('provider_id'),
547
+ family: selector.fetch('family'),
548
+ dialect: selector.fetch('dialect'),
549
+ backend: selector.fetch('backend'),
550
+ profile_id: selector.fetch('profile'),
551
+ before_source: expected,
552
+ after_source: output,
553
+ path_name: "#{item['id']}.#{item['dialect']}"
554
+ }
555
+ )
556
+ result[:ok] == true && result.fetch(:changes).empty?
557
+ rescue Ast::Merge::Error, KeyError, LoadError
558
+ false
559
+ end
560
+
561
+ def structural_provider_matches?(item, policy)
562
+ policy['provider'].to_s == item.dig('selector', 'provider_id').to_s
563
+ end
564
+
565
+ def dimensions(item, outcome, checks)
566
+ eligible = item.dig('oracle', 'score_eligible')
567
+ {
568
+ 'safety' => {
569
+ 'eligible' => eligible,
570
+ 'false_auto_merge' => outcome == 'false_auto_merge',
571
+ 'severity' => item['false_auto_merge_severity'],
572
+ 'compensable' => false
573
+ },
574
+ 'effectiveness' => {
575
+ 'eligible' => eligible,
576
+ 'success' => %w[correct_clean true_conflict].include?(outcome)
577
+ },
578
+ 'preservation' => {
579
+ 'eligible' => eligible,
580
+ 'requirements' => item['preservation_policy'],
581
+ 'violations' => checks.fetch('preservation_violations')
582
+ },
583
+ 'performance' => { 'quality_offset_allowed' => false }
584
+ }
585
+ end
586
+
587
+ def preservation_violations(item, checks, output)
588
+ return [] if checks['exact']
589
+
590
+ expected = item.dig('expected', 'output', 'bytes')
591
+ return [] unless expected
592
+
593
+ item['preservation_policy'].filter_map do |name, requirement|
594
+ next unless requirement == 'required'
595
+ next unless preservation_violation?(name, output, expected, checks)
596
+
597
+ name
598
+ end
599
+ end
600
+
601
+ def preservation_violation?(name, output, expected, checks)
602
+ case name
603
+ when 'encoding' then output.encoding != expected.encoding
604
+ when 'line_endings' then line_ending_style(output) != line_ending_style(expected)
605
+ when 'unknown_fields' then checks['structural'] != true
606
+ else output != expected
607
+ end
608
+ end
609
+
610
+ def line_ending_style(content)
611
+ return 'crlf' if content.include?("\r\n")
612
+
613
+ 'lf'
614
+ end
615
+
616
+ def independent_edit_evidence(item, checks, outcome)
617
+ item['independent_edit_ids'].to_h do |id|
618
+ [id,
619
+ { 'preserved' => outcome == 'correct_clean' && checks['acceptable'], 'method' => 'oracle equivalence' }]
620
+ end
621
+ end
622
+
623
+ def conflict_regions(output)
624
+ ranges = []
625
+ start_byte = nil
626
+ offset = 0
627
+ output.each_line do |line|
628
+ start_byte = offset if line.start_with?('<<<<<<<')
629
+ if start_byte && line.start_with?('>>>>>>>')
630
+ ranges << { 'start_byte' => start_byte, 'end_byte' => offset + line.bytesize }
631
+ start_byte = nil
632
+ end
633
+ offset += line.bytesize
634
+ end
635
+ ranges
636
+ end
637
+
638
+ def region_evidence(item, observed)
639
+ expected = item['expected_conflict_regions']
640
+ observed = observed.each_with_index.map { |region, index| region.merge('id' => "observed.#{index + 1}") }
641
+ localization_status = expected.empty? && observed.empty? ? 'not_applicable' : 'unknown'
642
+ {
643
+ 'expected' => expected,
644
+ 'observed' => observed,
645
+ 'matched_region_ids' => [],
646
+ 'missed_region_ids' => expected.map { |region| region['id'] },
647
+ 'false_positive_region_ids' => observed.map { |region| region['id'] },
648
+ 'localization_status' => localization_status,
649
+ 'matching_basis' => 'unknown: expected role ranges/paths and observed output ranges have no proven mapping',
650
+ 'localization_error_bytes' => nil
651
+ }
652
+ end
653
+
654
+ def raw_record(content)
655
+ { 'inline' => content, 'bytes' => content.bytesize, 'sha256' => Digest::SHA256.hexdigest(content) }
656
+ end
657
+
658
+ def diagnostics(stderr, unsupported_reason)
659
+ lines = stderr.lines.map(&:strip).reject(&:empty?)
660
+ lines << unsupported_reason if unsupported_reason
661
+ lines.map do |line|
662
+ match = /\Aast-merge-git: ([a-z_]+):/.match(line)
663
+ category = match&.[](1) || (unsupported_reason == line ? 'unsupported' : 'process')
664
+ { 'severity' => 'error', 'category' => category, 'message' => line }
665
+ end
666
+ end
667
+
668
+ def selector_env(selector)
669
+ {
670
+ 'AST_MERGE_PROVIDER' => selector['provider_id'],
671
+ 'AST_MERGE_FAMILY' => selector['family'],
672
+ 'AST_MERGE_DIALECT' => selector['dialect'],
673
+ 'AST_MERGE_BACKEND' => selector['backend'],
674
+ 'AST_MERGE_PROFILE' => selector['profile'],
675
+ 'AST_MERGE_REQUIRE' => selector['require']
676
+ }
677
+ end
678
+
679
+ def candidate_env(selector)
680
+ forbidden = ENV.keys.grep(/(?:ORACLE|EXPECTED)/i).to_h { |name| [name, nil] }
681
+ forbidden.merge(selector_env(selector))
682
+ end
683
+
684
+ def cache_identity(selection)
685
+ source_sha, = Open3.capture2('git', '-C', Pathname(__dir__).join('..', '..', '..', '..').to_s,
686
+ 'rev-parse', 'HEAD')
687
+ environment = {
688
+ 'ruby' => RUBY_DESCRIPTION,
689
+ 'platform' => RUBY_PLATFORM,
690
+ 'host_os' => RbConfig::CONFIG['host_os'],
691
+ 'host_cpu' => RbConfig::CONFIG['host_cpu']
692
+ }
693
+ configuration = selection['selected_case_ids'].map do |id|
694
+ @benchmark.case_by_id(id).slice('id', 'selector')
695
+ end
696
+ identity = {
697
+ 'adapter_source_sha' => source_sha.strip,
698
+ 'adapter_artifact_sha256' => Digest::SHA256.file(@driver_path).hexdigest,
699
+ 'configuration_sha256' => Digest::SHA256.hexdigest(JSON.generate(configuration)),
700
+ 'corpus_sha256' => @benchmark.corpus_digest,
701
+ 'environment' => environment,
702
+ 'profile' => selection['profile'],
703
+ 'changed_paths' => selection['changed_paths'],
704
+ 'inferred_capabilities' => selection['inferred_capabilities'],
705
+ 'selection_explanation_sha256' => Digest::SHA256.hexdigest(
706
+ JSON.generate(deep_sort(selection.fetch('explanation')))
707
+ ),
708
+ 'selected_case_ids' => selection['selected_case_ids']
709
+ }
710
+ identity.merge('sha256' => Digest::SHA256.hexdigest(JSON.generate(identity)))
711
+ end
712
+
713
+ def deep_sort(value)
714
+ case value
715
+ when Hash then value.keys.sort.to_h { |key| [key, deep_sort(value.fetch(key))] }
716
+ when Array then value.map { |item| deep_sort(item) }
717
+ else value
718
+ end
719
+ end
720
+
721
+ def correctness_record(result)
722
+ result.reject { |key, _value| %w[runtime deterministic_correctness_rerun].include?(key) }
723
+ end
724
+
725
+ def exit_class(status)
726
+ { 0 => 'clean', 1 => 'conflict', 2 => 'error' }.fetch(status, 'error')
727
+ end
728
+
729
+ def safe_id(value)
730
+ value.gsub(/[^a-zA-Z0-9.-]/, '-')
731
+ end
732
+
733
+ def error!(message)
734
+ raise LocalBenchmark::Error, message
735
+ end
736
+ end
737
+
738
+ # Builds a non-scalar paired report from local benchmark raw evidence.
739
+ class LocalBenchmarkReport
740
+ SUCCESS = %w[correct_clean true_conflict].freeze
741
+
742
+ def self.build(run)
743
+ new(run).build
744
+ end
745
+
746
+ def initialize(run)
747
+ @run = run
748
+ @results = run.fetch('results')
749
+ end
750
+
751
+ def build
752
+ pairs = @results.group_by { |result| result['case_id'] }.values.map { |items| transition(items) }
753
+ false_auto_merges = eligible.select { |item| item['outcome'] == 'false_auto_merge' }
754
+ {
755
+ 'schema_version' => 'structuredmerge.benchmark.report/v1',
756
+ 'kind' => 'paired_aggregate_report',
757
+ 'corpus_id' => @run['corpus_id'],
758
+ 'corpus_digest' => @run['corpus_digest'],
759
+ 'selection' => @run['selection'],
760
+ 'cache_identity' => @run['cache_identity'],
761
+ 'dimensions' => {
762
+ 'safety' => {
763
+ 'eligible' => eligible.length,
764
+ 'false_auto_merge_result_ids' => false_auto_merges.map { |item| item['id'] },
765
+ 'gate' => false_auto_merges.empty? ? 'pass' : 'fail',
766
+ 'non_compensable' => true
767
+ },
768
+ 'effectiveness' => outcome_counts,
769
+ 'preservation' => {
770
+ 'eligible' => eligible.length,
771
+ 'violation_result_ids' => eligible.filter_map do |item|
772
+ item['id'] if item.dig('dimensions', 'preservation', 'violations').any?
773
+ end
774
+ },
775
+ 'performance' => performance,
776
+ 'reliability' => { 'error_result_ids' => @results.filter_map do |item|
777
+ item['id'] if item['outcome'] == 'error'
778
+ end },
779
+ 'coverage' => coverage
780
+ },
781
+ 'strata' => strata,
782
+ 'transitions' => pairs,
783
+ 'newly_passing_case_ids' => pairs.filter_map { |pair| pair['case_id'] if pair['newly_passing'] },
784
+ 'newly_failing_case_ids' => pairs.filter_map { |pair| pair['case_id'] if pair['newly_failing'] },
785
+ 'changed_conflict_case_ids' => pairs.filter_map { |pair| pair['case_id'] if pair['changed_conflict'] },
786
+ 'hard_gate_failed' => false_auto_merges.any?,
787
+ 'scalar_score' => nil
788
+ }
789
+ end
790
+
791
+ private
792
+
793
+ def eligible
794
+ @eligible ||= @results.select { |item| item['score_eligible'] }
795
+ end
796
+
797
+ def outcome_counts
798
+ %w[correct_clean false_conflict true_conflict false_auto_merge error unsupported
799
+ excluded_ambiguous].to_h do |outcome|
800
+ [outcome, @results.count { |item| item['outcome'] == outcome }]
801
+ end
802
+ end
803
+
804
+ def transition(items)
805
+ baseline = items.find { |item| item['adapter_id'] == 'git.merge-file' }
806
+ candidate = items.find { |item| item['adapter_id'] == 'ast-merge-git' }
807
+ {
808
+ 'case_id' => items.first['case_id'],
809
+ 'baseline_result_id' => baseline['id'],
810
+ 'candidate_result_id' => candidate['id'],
811
+ 'from' => baseline['outcome'],
812
+ 'to' => candidate['outcome'],
813
+ 'newly_passing' => !SUCCESS.include?(baseline['outcome']) && SUCCESS.include?(candidate['outcome']),
814
+ 'newly_failing' => SUCCESS.include?(baseline['outcome']) && !SUCCESS.include?(candidate['outcome']),
815
+ 'changed_conflict' => conflict?(baseline) && conflict?(candidate) &&
816
+ baseline.dig('raw', 'output', 'sha256') != candidate.dig('raw', 'output', 'sha256')
817
+ }
818
+ end
819
+
820
+ def conflict?(result)
821
+ %w[false_conflict true_conflict].include?(result['outcome'])
822
+ end
823
+
824
+ def performance
825
+ @results.group_by { |item| item['adapter_id'] }.transform_values do |items|
826
+ values = items.filter_map { |item| item.dig('runtime', 'duration_ns') }
827
+ { 'samples' => values.length, 'total_ns' => values.sum, 'runtime_values_excluded_from_correctness' => true }
828
+ end
829
+ end
830
+
831
+ def coverage
832
+ selected = @run.dig('selection', 'selected_case_ids').length
833
+ unsupported = @results.select do |item|
834
+ item['adapter_id'] == 'ast-merge-git' && item['outcome'] == 'unsupported'
835
+ end
836
+ { 'selected_cases' => selected, 'executed_candidate_cases' => selected - unsupported.length,
837
+ 'unsupported_case_ids' => unsupported.map { |item| item['case_id'] },
838
+ 'unsupported_is_quality_failure' => false }
839
+ end
840
+
841
+ def strata
842
+ case_results = @results.select { |item| item['adapter_id'] == 'ast-merge-git' }
843
+ tag_fields = %w[operation partition family provider dialect false_auto_merge_severity]
844
+ tag_strata = tag_fields.to_h do |field|
845
+ [field, case_results.group_by { |item| item.dig('case_tags', field) }.transform_values(&:length)]
846
+ end
847
+ transitions = @results.group_by { |item| item['case_id'] }.values
848
+ tag_strata.merge(
849
+ 'capability' => case_results.each_with_object(Hash.new(0)) do |item, counts|
850
+ item.dig('case_tags', 'capabilities').each { |capability| counts[capability] += 1 }
851
+ end,
852
+ 'outcome' => @results.group_by { |item| item['outcome'] }.transform_values(&:length),
853
+ 'adapter' => @results.group_by { |item| item['adapter_id'] }.transform_values(&:length),
854
+ 'transition' => transitions.each_with_object(Hash.new(0)) do |items, counts|
855
+ pair = transition(items)
856
+ counts["#{pair['from']}->#{pair['to']}"] += 1
857
+ end
858
+ )
859
+ end
860
+ end
861
+ # rubocop:enable Metrics/AbcSize, Metrics/ClassLength, Metrics/CyclomaticComplexity, Metrics/MethodLength, Metrics/PerceivedComplexity
862
+ end
863
+ end
864
+ end