json-merge 7.0.0 → 7.1.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- checksums.yaml.gz.sig +0 -0
- data/LICENSE.md +13 -0
- data/README.md +600 -0
- data/lib/json/merge/comment_tracker.rb +27 -0
- data/lib/json/merge/conflict_resolver.rb +1379 -0
- data/lib/json/merge/debug_logger.rb +41 -0
- data/lib/json/merge/emitter.rb +227 -0
- data/lib/json/merge/file_analysis.rb +355 -0
- data/lib/json/merge/freeze_node.rb +62 -0
- data/lib/json/merge/merge_result.rb +156 -0
- data/lib/json/merge/node_wrapper.rb +258 -0
- data/lib/json/merge/object_match_refiner.rb +338 -0
- data/lib/json/merge/provider.rb +696 -0
- data/lib/json/merge/smart_merger.rb +276 -0
- data/lib/json/merge/source_locator.rb +68 -0
- data/lib/json/merge/three_way_decision.rb +168 -0
- data/lib/json/merge/version.rb +5 -3
- data/lib/json/merge.rb +393 -266
- data/lib/json-merge.rb +7 -1
- data/sig/json/merge.rbs +54 -0
- data.tar.gz.sig +0 -0
- metadata +264 -15
- metadata.gz.sig +0 -0
|
@@ -0,0 +1,338 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Json
|
|
4
|
+
module Merge
|
|
5
|
+
# Match refiner for JSON objects and array elements that didn't match by exact signature.
|
|
6
|
+
#
|
|
7
|
+
# This refiner uses fuzzy matching to pair JSON nodes that have:
|
|
8
|
+
# - Similar key names in objects (e.g., `databaseUrl` vs `database_url`)
|
|
9
|
+
# - Array elements with similar structure or content
|
|
10
|
+
# - Objects with overlapping keys but different values
|
|
11
|
+
#
|
|
12
|
+
# The matching algorithm considers:
|
|
13
|
+
# - Key name similarity for object pairs (Levenshtein distance)
|
|
14
|
+
# - Value type and content similarity
|
|
15
|
+
# - Structural overlap for nested objects
|
|
16
|
+
#
|
|
17
|
+
# @example Basic usage
|
|
18
|
+
# refiner = ObjectMatchRefiner.new(threshold: 0.6)
|
|
19
|
+
# matches = refiner.call(template_nodes, dest_nodes)
|
|
20
|
+
#
|
|
21
|
+
# @example With custom weights
|
|
22
|
+
# refiner = ObjectMatchRefiner.new(
|
|
23
|
+
# threshold: 0.5,
|
|
24
|
+
# key_weight: 0.6,
|
|
25
|
+
# value_weight: 0.4
|
|
26
|
+
# )
|
|
27
|
+
#
|
|
28
|
+
# @see Ast::Merge::MatchRefinerBase
|
|
29
|
+
class ObjectMatchRefiner < Ast::Merge::MatchRefinerBase
|
|
30
|
+
# Default weight for key similarity
|
|
31
|
+
DEFAULT_KEY_WEIGHT = 0.7
|
|
32
|
+
|
|
33
|
+
# Default weight for value similarity
|
|
34
|
+
DEFAULT_VALUE_WEIGHT = 0.3
|
|
35
|
+
|
|
36
|
+
# @return [Float] Weight for key similarity (0.0-1.0)
|
|
37
|
+
attr_reader :key_weight
|
|
38
|
+
|
|
39
|
+
# @return [Float] Weight for value similarity (0.0-1.0)
|
|
40
|
+
attr_reader :value_weight
|
|
41
|
+
|
|
42
|
+
# Initialize an object match refiner.
|
|
43
|
+
#
|
|
44
|
+
# @param threshold [Float] Minimum score to accept a match (default: 0.5)
|
|
45
|
+
# @param key_weight [Float] Weight for key similarity (default: 0.7)
|
|
46
|
+
# @param value_weight [Float] Weight for value similarity (default: 0.3)
|
|
47
|
+
def initialize(threshold: DEFAULT_THRESHOLD, key_weight: DEFAULT_KEY_WEIGHT, value_weight: DEFAULT_VALUE_WEIGHT,
|
|
48
|
+
**options)
|
|
49
|
+
super(threshold: threshold, **options)
|
|
50
|
+
@key_weight = key_weight
|
|
51
|
+
@value_weight = value_weight
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
# Find matches between unmatched JSON nodes.
|
|
55
|
+
#
|
|
56
|
+
# Handles both object key-value pairs and array elements.
|
|
57
|
+
#
|
|
58
|
+
# @param template_nodes [Array] Unmatched nodes from template
|
|
59
|
+
# @param dest_nodes [Array] Unmatched nodes from destination
|
|
60
|
+
# @param context [Hash] Additional context
|
|
61
|
+
# @return [Array<MatchResult>] Array of node matches
|
|
62
|
+
def call(template_nodes, dest_nodes, _context = {})
|
|
63
|
+
# Match object pairs (key-value entries)
|
|
64
|
+
pair_matches = match_pairs(template_nodes, dest_nodes)
|
|
65
|
+
|
|
66
|
+
# Match array elements (objects within arrays)
|
|
67
|
+
array_matches = match_array_objects(template_nodes, dest_nodes)
|
|
68
|
+
|
|
69
|
+
pair_matches + array_matches
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
private
|
|
73
|
+
|
|
74
|
+
# Match key-value pairs from JSON objects.
|
|
75
|
+
#
|
|
76
|
+
# @param template_nodes [Array] Template nodes
|
|
77
|
+
# @param dest_nodes [Array] Destination nodes
|
|
78
|
+
# @return [Array<MatchResult>]
|
|
79
|
+
def match_pairs(template_nodes, dest_nodes)
|
|
80
|
+
template_pairs = template_nodes.select { |n| pair_node?(n) }
|
|
81
|
+
dest_pairs = dest_nodes.select { |n| pair_node?(n) }
|
|
82
|
+
|
|
83
|
+
return [] if template_pairs.empty? || dest_pairs.empty?
|
|
84
|
+
|
|
85
|
+
greedy_match(template_pairs, dest_pairs) do |t_node, d_node|
|
|
86
|
+
compute_pair_similarity(t_node, d_node)
|
|
87
|
+
end
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
# Match object elements from JSON arrays.
|
|
91
|
+
#
|
|
92
|
+
# @param template_nodes [Array] Template nodes
|
|
93
|
+
# @param dest_nodes [Array] Destination nodes
|
|
94
|
+
# @return [Array<MatchResult>]
|
|
95
|
+
def match_array_objects(template_nodes, dest_nodes)
|
|
96
|
+
template_objects = template_nodes.select { |n| object_node?(n) && !pair_node?(n) }
|
|
97
|
+
dest_objects = dest_nodes.select { |n| object_node?(n) && !pair_node?(n) }
|
|
98
|
+
|
|
99
|
+
return [] if template_objects.empty? || dest_objects.empty?
|
|
100
|
+
|
|
101
|
+
greedy_match(template_objects, dest_objects) do |t_node, d_node|
|
|
102
|
+
compute_object_similarity(t_node, d_node)
|
|
103
|
+
end
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
# Check if a node is a key-value pair.
|
|
107
|
+
#
|
|
108
|
+
# @param node [Object] Node to check
|
|
109
|
+
# @return [Boolean]
|
|
110
|
+
def pair_node?(node)
|
|
111
|
+
return false unless node.respond_to?(:pair?)
|
|
112
|
+
|
|
113
|
+
node.pair?
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
# Check if a node is a JSON object.
|
|
117
|
+
#
|
|
118
|
+
# @param node [Object] Node to check
|
|
119
|
+
# @return [Boolean]
|
|
120
|
+
def object_node?(node)
|
|
121
|
+
return false unless node.respond_to?(:object?)
|
|
122
|
+
|
|
123
|
+
node.object?
|
|
124
|
+
end
|
|
125
|
+
|
|
126
|
+
# Compute similarity score between two key-value pairs.
|
|
127
|
+
#
|
|
128
|
+
# @param t_pair [NodeWrapper] Template pair
|
|
129
|
+
# @param d_pair [NodeWrapper] Destination pair
|
|
130
|
+
# @return [Float] Similarity score (0.0-1.0)
|
|
131
|
+
def compute_pair_similarity(t_pair, d_pair)
|
|
132
|
+
t_key = t_pair.key_name
|
|
133
|
+
d_key = d_pair.key_name
|
|
134
|
+
|
|
135
|
+
return 0.0 unless t_key && d_key
|
|
136
|
+
|
|
137
|
+
key_score = key_similarity(t_key, d_key)
|
|
138
|
+
value_score = value_similarity(t_pair.value_node, d_pair.value_node)
|
|
139
|
+
|
|
140
|
+
(key_score * key_weight) + (value_score * value_weight)
|
|
141
|
+
end
|
|
142
|
+
|
|
143
|
+
# Compute similarity score between two JSON objects.
|
|
144
|
+
#
|
|
145
|
+
# @param t_obj [NodeWrapper] Template object
|
|
146
|
+
# @param d_obj [NodeWrapper] Destination object
|
|
147
|
+
# @return [Float] Similarity score (0.0-1.0)
|
|
148
|
+
def compute_object_similarity(t_obj, d_obj)
|
|
149
|
+
t_keys = extract_keys(t_obj)
|
|
150
|
+
d_keys = extract_keys(d_obj)
|
|
151
|
+
|
|
152
|
+
return 1.0 if t_keys.empty? && d_keys.empty?
|
|
153
|
+
return 0.0 if t_keys.empty? || d_keys.empty?
|
|
154
|
+
|
|
155
|
+
# Compute key overlap
|
|
156
|
+
common_keys = (t_keys & d_keys).size
|
|
157
|
+
total_keys = (t_keys | d_keys).size
|
|
158
|
+
key_overlap = common_keys.to_f / total_keys
|
|
159
|
+
|
|
160
|
+
# Compute fuzzy key similarity for non-exact matches
|
|
161
|
+
fuzzy_score = compute_fuzzy_key_matches(t_keys - d_keys, d_keys - t_keys)
|
|
162
|
+
|
|
163
|
+
# Combine exact overlap and fuzzy matching
|
|
164
|
+
(key_overlap * 0.7) + (fuzzy_score * 0.3)
|
|
165
|
+
end
|
|
166
|
+
|
|
167
|
+
# Compute fuzzy matches between two sets of keys.
|
|
168
|
+
#
|
|
169
|
+
# @param keys1 [Array<String>] First set of keys
|
|
170
|
+
# @param keys2 [Array<String>] Second set of keys
|
|
171
|
+
# @return [Float] Fuzzy match score (0.0-1.0)
|
|
172
|
+
def compute_fuzzy_key_matches(keys1, keys2)
|
|
173
|
+
return 1.0 if keys1.empty? && keys2.empty?
|
|
174
|
+
return 0.0 if keys1.empty? || keys2.empty?
|
|
175
|
+
|
|
176
|
+
total_similarity = 0.0
|
|
177
|
+
keys1.each do |k1|
|
|
178
|
+
best_match = keys2.map { |k2| key_similarity(k1, k2) }.max || 0.0
|
|
179
|
+
total_similarity += best_match
|
|
180
|
+
end
|
|
181
|
+
|
|
182
|
+
total_similarity / keys1.size
|
|
183
|
+
end
|
|
184
|
+
|
|
185
|
+
# Extract keys from a JSON object.
|
|
186
|
+
#
|
|
187
|
+
# @param obj [NodeWrapper] Object node
|
|
188
|
+
# @return [Array<String>] Keys
|
|
189
|
+
def extract_keys(obj)
|
|
190
|
+
return [] unless obj.respond_to?(:pairs)
|
|
191
|
+
|
|
192
|
+
obj.pairs.map(&:key_name).compact
|
|
193
|
+
end
|
|
194
|
+
|
|
195
|
+
# Compute similarity between two keys.
|
|
196
|
+
#
|
|
197
|
+
# @param key1 [String] First key
|
|
198
|
+
# @param key2 [String] Second key
|
|
199
|
+
# @return [Float] Key similarity (0.0-1.0)
|
|
200
|
+
def key_similarity(key1, key2)
|
|
201
|
+
return 1.0 if key1 == key2
|
|
202
|
+
|
|
203
|
+
str1 = normalize_key(key1.to_s)
|
|
204
|
+
str2 = normalize_key(key2.to_s)
|
|
205
|
+
|
|
206
|
+
string_similarity(str1, str2)
|
|
207
|
+
end
|
|
208
|
+
|
|
209
|
+
# Normalize a key for comparison.
|
|
210
|
+
# Converts to lowercase and normalizes common naming conventions.
|
|
211
|
+
#
|
|
212
|
+
# @param key [String] Key to normalize
|
|
213
|
+
# @return [String] Normalized key
|
|
214
|
+
def normalize_key(key)
|
|
215
|
+
# Convert camelCase to snake_case first, then normalize
|
|
216
|
+
key.gsub(/([A-Z])/) { "_#{::Regexp.last_match(1).downcase}" }
|
|
217
|
+
.downcase
|
|
218
|
+
.gsub(/[-_]/, '')
|
|
219
|
+
end
|
|
220
|
+
|
|
221
|
+
# Compute similarity between two values.
|
|
222
|
+
#
|
|
223
|
+
# @param t_value [NodeWrapper, nil] Template value
|
|
224
|
+
# @param d_value [NodeWrapper, nil] Destination value
|
|
225
|
+
# @return [Float] Value similarity (0.0-1.0)
|
|
226
|
+
def value_similarity(t_value, d_value)
|
|
227
|
+
return 0.5 unless t_value && d_value
|
|
228
|
+
|
|
229
|
+
# Check if they're the same type
|
|
230
|
+
return 0.0 unless same_value_type?(t_value, d_value)
|
|
231
|
+
|
|
232
|
+
if t_value.string?
|
|
233
|
+
# Compare string values
|
|
234
|
+
t_text = extract_string_value(t_value)
|
|
235
|
+
d_text = extract_string_value(d_value)
|
|
236
|
+
string_similarity(t_text, d_text)
|
|
237
|
+
elsif t_value.object?
|
|
238
|
+
# Compare object structures
|
|
239
|
+
compute_object_similarity(t_value, d_value)
|
|
240
|
+
elsif t_value.array?
|
|
241
|
+
# Compare array lengths
|
|
242
|
+
array_similarity(t_value, d_value)
|
|
243
|
+
else
|
|
244
|
+
# Same type, consider similar
|
|
245
|
+
0.5
|
|
246
|
+
end
|
|
247
|
+
end
|
|
248
|
+
|
|
249
|
+
# Check if two values have the same JSON type.
|
|
250
|
+
#
|
|
251
|
+
# @param val1 [NodeWrapper] First value
|
|
252
|
+
# @param val2 [NodeWrapper] Second value
|
|
253
|
+
# @return [Boolean]
|
|
254
|
+
def same_value_type?(val1, val2)
|
|
255
|
+
val1.type == val2.type
|
|
256
|
+
end
|
|
257
|
+
|
|
258
|
+
# Extract string content from a string node.
|
|
259
|
+
#
|
|
260
|
+
# @param node [NodeWrapper] String node
|
|
261
|
+
# @return [String]
|
|
262
|
+
def extract_string_value(node)
|
|
263
|
+
# String nodes include quotes, remove them
|
|
264
|
+
text = node.respond_to?(:text) ? node.text : ''
|
|
265
|
+
text.gsub(/\A"|"\z/, '')
|
|
266
|
+
end
|
|
267
|
+
|
|
268
|
+
# Compute similarity between two arrays.
|
|
269
|
+
#
|
|
270
|
+
# @param arr1 [NodeWrapper] First array
|
|
271
|
+
# @param arr2 [NodeWrapper] Second array
|
|
272
|
+
# @return [Float] Array similarity (0.0-1.0)
|
|
273
|
+
def array_similarity(arr1, arr2)
|
|
274
|
+
len1 = arr1.respond_to?(:elements) ? arr1.elements.size : 0
|
|
275
|
+
len2 = arr2.respond_to?(:elements) ? arr2.elements.size : 0
|
|
276
|
+
|
|
277
|
+
return 1.0 if len1.zero? && len2.zero?
|
|
278
|
+
return 0.0 if len1.zero? || len2.zero?
|
|
279
|
+
|
|
280
|
+
[len1, len2].min.to_f / [len1, len2].max
|
|
281
|
+
end
|
|
282
|
+
|
|
283
|
+
# Compute string similarity using Levenshtein distance.
|
|
284
|
+
#
|
|
285
|
+
# @param str1 [String] First string
|
|
286
|
+
# @param str2 [String] Second string
|
|
287
|
+
# @return [Float] Similarity score (0.0-1.0)
|
|
288
|
+
def string_similarity(str1, str2)
|
|
289
|
+
return 1.0 if str1 == str2
|
|
290
|
+
return 0.0 if str1.to_s.empty? || str2.to_s.empty?
|
|
291
|
+
|
|
292
|
+
distance = levenshtein_distance(str1.to_s, str2.to_s)
|
|
293
|
+
max_len = [str1.to_s.length, str2.to_s.length].max
|
|
294
|
+
|
|
295
|
+
1.0 - (distance.to_f / max_len)
|
|
296
|
+
end
|
|
297
|
+
|
|
298
|
+
# Compute Levenshtein distance between two strings.
|
|
299
|
+
#
|
|
300
|
+
# Uses Wagner-Fischer algorithm with O(min(m,n)) space.
|
|
301
|
+
#
|
|
302
|
+
# @param str1 [String] First string
|
|
303
|
+
# @param str2 [String] Second string
|
|
304
|
+
# @return [Integer] Edit distance
|
|
305
|
+
def levenshtein_distance(str1, str2)
|
|
306
|
+
return str2.length if str1.empty?
|
|
307
|
+
return str1.length if str2.empty?
|
|
308
|
+
|
|
309
|
+
# Ensure str1 is the shorter string for space optimization
|
|
310
|
+
str1, str2 = str2, str1 if str1.length > str2.length
|
|
311
|
+
|
|
312
|
+
m = str1.length
|
|
313
|
+
n = str2.length
|
|
314
|
+
|
|
315
|
+
# Use two rows instead of full matrix
|
|
316
|
+
prev_row = (0..m).to_a
|
|
317
|
+
curr_row = Array.new(m + 1, 0)
|
|
318
|
+
|
|
319
|
+
(1..n).each do |j|
|
|
320
|
+
curr_row[0] = j
|
|
321
|
+
|
|
322
|
+
(1..m).each do |i|
|
|
323
|
+
cost = str1[i - 1] == str2[j - 1] ? 0 : 1
|
|
324
|
+
curr_row[i] = [
|
|
325
|
+
prev_row[i] + 1, # deletion
|
|
326
|
+
curr_row[i - 1] + 1, # insertion
|
|
327
|
+
prev_row[i - 1] + cost # substitution
|
|
328
|
+
].min
|
|
329
|
+
end
|
|
330
|
+
|
|
331
|
+
prev_row, curr_row = curr_row, prev_row
|
|
332
|
+
end
|
|
333
|
+
|
|
334
|
+
prev_row[m]
|
|
335
|
+
end
|
|
336
|
+
end
|
|
337
|
+
end
|
|
338
|
+
end
|