json-merge 7.0.0 → 7.1.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,338 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Json
4
+ module Merge
5
+ # Match refiner for JSON objects and array elements that didn't match by exact signature.
6
+ #
7
+ # This refiner uses fuzzy matching to pair JSON nodes that have:
8
+ # - Similar key names in objects (e.g., `databaseUrl` vs `database_url`)
9
+ # - Array elements with similar structure or content
10
+ # - Objects with overlapping keys but different values
11
+ #
12
+ # The matching algorithm considers:
13
+ # - Key name similarity for object pairs (Levenshtein distance)
14
+ # - Value type and content similarity
15
+ # - Structural overlap for nested objects
16
+ #
17
+ # @example Basic usage
18
+ # refiner = ObjectMatchRefiner.new(threshold: 0.6)
19
+ # matches = refiner.call(template_nodes, dest_nodes)
20
+ #
21
+ # @example With custom weights
22
+ # refiner = ObjectMatchRefiner.new(
23
+ # threshold: 0.5,
24
+ # key_weight: 0.6,
25
+ # value_weight: 0.4
26
+ # )
27
+ #
28
+ # @see Ast::Merge::MatchRefinerBase
29
+ class ObjectMatchRefiner < Ast::Merge::MatchRefinerBase
30
+ # Default weight for key similarity
31
+ DEFAULT_KEY_WEIGHT = 0.7
32
+
33
+ # Default weight for value similarity
34
+ DEFAULT_VALUE_WEIGHT = 0.3
35
+
36
+ # @return [Float] Weight for key similarity (0.0-1.0)
37
+ attr_reader :key_weight
38
+
39
+ # @return [Float] Weight for value similarity (0.0-1.0)
40
+ attr_reader :value_weight
41
+
42
+ # Initialize an object match refiner.
43
+ #
44
+ # @param threshold [Float] Minimum score to accept a match (default: 0.5)
45
+ # @param key_weight [Float] Weight for key similarity (default: 0.7)
46
+ # @param value_weight [Float] Weight for value similarity (default: 0.3)
47
+ def initialize(threshold: DEFAULT_THRESHOLD, key_weight: DEFAULT_KEY_WEIGHT, value_weight: DEFAULT_VALUE_WEIGHT,
48
+ **options)
49
+ super(threshold: threshold, **options)
50
+ @key_weight = key_weight
51
+ @value_weight = value_weight
52
+ end
53
+
54
+ # Find matches between unmatched JSON nodes.
55
+ #
56
+ # Handles both object key-value pairs and array elements.
57
+ #
58
+ # @param template_nodes [Array] Unmatched nodes from template
59
+ # @param dest_nodes [Array] Unmatched nodes from destination
60
+ # @param context [Hash] Additional context
61
+ # @return [Array<MatchResult>] Array of node matches
62
+ def call(template_nodes, dest_nodes, _context = {})
63
+ # Match object pairs (key-value entries)
64
+ pair_matches = match_pairs(template_nodes, dest_nodes)
65
+
66
+ # Match array elements (objects within arrays)
67
+ array_matches = match_array_objects(template_nodes, dest_nodes)
68
+
69
+ pair_matches + array_matches
70
+ end
71
+
72
+ private
73
+
74
+ # Match key-value pairs from JSON objects.
75
+ #
76
+ # @param template_nodes [Array] Template nodes
77
+ # @param dest_nodes [Array] Destination nodes
78
+ # @return [Array<MatchResult>]
79
+ def match_pairs(template_nodes, dest_nodes)
80
+ template_pairs = template_nodes.select { |n| pair_node?(n) }
81
+ dest_pairs = dest_nodes.select { |n| pair_node?(n) }
82
+
83
+ return [] if template_pairs.empty? || dest_pairs.empty?
84
+
85
+ greedy_match(template_pairs, dest_pairs) do |t_node, d_node|
86
+ compute_pair_similarity(t_node, d_node)
87
+ end
88
+ end
89
+
90
+ # Match object elements from JSON arrays.
91
+ #
92
+ # @param template_nodes [Array] Template nodes
93
+ # @param dest_nodes [Array] Destination nodes
94
+ # @return [Array<MatchResult>]
95
+ def match_array_objects(template_nodes, dest_nodes)
96
+ template_objects = template_nodes.select { |n| object_node?(n) && !pair_node?(n) }
97
+ dest_objects = dest_nodes.select { |n| object_node?(n) && !pair_node?(n) }
98
+
99
+ return [] if template_objects.empty? || dest_objects.empty?
100
+
101
+ greedy_match(template_objects, dest_objects) do |t_node, d_node|
102
+ compute_object_similarity(t_node, d_node)
103
+ end
104
+ end
105
+
106
+ # Check if a node is a key-value pair.
107
+ #
108
+ # @param node [Object] Node to check
109
+ # @return [Boolean]
110
+ def pair_node?(node)
111
+ return false unless node.respond_to?(:pair?)
112
+
113
+ node.pair?
114
+ end
115
+
116
+ # Check if a node is a JSON object.
117
+ #
118
+ # @param node [Object] Node to check
119
+ # @return [Boolean]
120
+ def object_node?(node)
121
+ return false unless node.respond_to?(:object?)
122
+
123
+ node.object?
124
+ end
125
+
126
+ # Compute similarity score between two key-value pairs.
127
+ #
128
+ # @param t_pair [NodeWrapper] Template pair
129
+ # @param d_pair [NodeWrapper] Destination pair
130
+ # @return [Float] Similarity score (0.0-1.0)
131
+ def compute_pair_similarity(t_pair, d_pair)
132
+ t_key = t_pair.key_name
133
+ d_key = d_pair.key_name
134
+
135
+ return 0.0 unless t_key && d_key
136
+
137
+ key_score = key_similarity(t_key, d_key)
138
+ value_score = value_similarity(t_pair.value_node, d_pair.value_node)
139
+
140
+ (key_score * key_weight) + (value_score * value_weight)
141
+ end
142
+
143
+ # Compute similarity score between two JSON objects.
144
+ #
145
+ # @param t_obj [NodeWrapper] Template object
146
+ # @param d_obj [NodeWrapper] Destination object
147
+ # @return [Float] Similarity score (0.0-1.0)
148
+ def compute_object_similarity(t_obj, d_obj)
149
+ t_keys = extract_keys(t_obj)
150
+ d_keys = extract_keys(d_obj)
151
+
152
+ return 1.0 if t_keys.empty? && d_keys.empty?
153
+ return 0.0 if t_keys.empty? || d_keys.empty?
154
+
155
+ # Compute key overlap
156
+ common_keys = (t_keys & d_keys).size
157
+ total_keys = (t_keys | d_keys).size
158
+ key_overlap = common_keys.to_f / total_keys
159
+
160
+ # Compute fuzzy key similarity for non-exact matches
161
+ fuzzy_score = compute_fuzzy_key_matches(t_keys - d_keys, d_keys - t_keys)
162
+
163
+ # Combine exact overlap and fuzzy matching
164
+ (key_overlap * 0.7) + (fuzzy_score * 0.3)
165
+ end
166
+
167
+ # Compute fuzzy matches between two sets of keys.
168
+ #
169
+ # @param keys1 [Array<String>] First set of keys
170
+ # @param keys2 [Array<String>] Second set of keys
171
+ # @return [Float] Fuzzy match score (0.0-1.0)
172
+ def compute_fuzzy_key_matches(keys1, keys2)
173
+ return 1.0 if keys1.empty? && keys2.empty?
174
+ return 0.0 if keys1.empty? || keys2.empty?
175
+
176
+ total_similarity = 0.0
177
+ keys1.each do |k1|
178
+ best_match = keys2.map { |k2| key_similarity(k1, k2) }.max || 0.0
179
+ total_similarity += best_match
180
+ end
181
+
182
+ total_similarity / keys1.size
183
+ end
184
+
185
+ # Extract keys from a JSON object.
186
+ #
187
+ # @param obj [NodeWrapper] Object node
188
+ # @return [Array<String>] Keys
189
+ def extract_keys(obj)
190
+ return [] unless obj.respond_to?(:pairs)
191
+
192
+ obj.pairs.map(&:key_name).compact
193
+ end
194
+
195
+ # Compute similarity between two keys.
196
+ #
197
+ # @param key1 [String] First key
198
+ # @param key2 [String] Second key
199
+ # @return [Float] Key similarity (0.0-1.0)
200
+ def key_similarity(key1, key2)
201
+ return 1.0 if key1 == key2
202
+
203
+ str1 = normalize_key(key1.to_s)
204
+ str2 = normalize_key(key2.to_s)
205
+
206
+ string_similarity(str1, str2)
207
+ end
208
+
209
+ # Normalize a key for comparison.
210
+ # Converts to lowercase and normalizes common naming conventions.
211
+ #
212
+ # @param key [String] Key to normalize
213
+ # @return [String] Normalized key
214
+ def normalize_key(key)
215
+ # Convert camelCase to snake_case first, then normalize
216
+ key.gsub(/([A-Z])/) { "_#{::Regexp.last_match(1).downcase}" }
217
+ .downcase
218
+ .gsub(/[-_]/, '')
219
+ end
220
+
221
+ # Compute similarity between two values.
222
+ #
223
+ # @param t_value [NodeWrapper, nil] Template value
224
+ # @param d_value [NodeWrapper, nil] Destination value
225
+ # @return [Float] Value similarity (0.0-1.0)
226
+ def value_similarity(t_value, d_value)
227
+ return 0.5 unless t_value && d_value
228
+
229
+ # Check if they're the same type
230
+ return 0.0 unless same_value_type?(t_value, d_value)
231
+
232
+ if t_value.string?
233
+ # Compare string values
234
+ t_text = extract_string_value(t_value)
235
+ d_text = extract_string_value(d_value)
236
+ string_similarity(t_text, d_text)
237
+ elsif t_value.object?
238
+ # Compare object structures
239
+ compute_object_similarity(t_value, d_value)
240
+ elsif t_value.array?
241
+ # Compare array lengths
242
+ array_similarity(t_value, d_value)
243
+ else
244
+ # Same type, consider similar
245
+ 0.5
246
+ end
247
+ end
248
+
249
+ # Check if two values have the same JSON type.
250
+ #
251
+ # @param val1 [NodeWrapper] First value
252
+ # @param val2 [NodeWrapper] Second value
253
+ # @return [Boolean]
254
+ def same_value_type?(val1, val2)
255
+ val1.type == val2.type
256
+ end
257
+
258
+ # Extract string content from a string node.
259
+ #
260
+ # @param node [NodeWrapper] String node
261
+ # @return [String]
262
+ def extract_string_value(node)
263
+ # String nodes include quotes, remove them
264
+ text = node.respond_to?(:text) ? node.text : ''
265
+ text.gsub(/\A"|"\z/, '')
266
+ end
267
+
268
+ # Compute similarity between two arrays.
269
+ #
270
+ # @param arr1 [NodeWrapper] First array
271
+ # @param arr2 [NodeWrapper] Second array
272
+ # @return [Float] Array similarity (0.0-1.0)
273
+ def array_similarity(arr1, arr2)
274
+ len1 = arr1.respond_to?(:elements) ? arr1.elements.size : 0
275
+ len2 = arr2.respond_to?(:elements) ? arr2.elements.size : 0
276
+
277
+ return 1.0 if len1.zero? && len2.zero?
278
+ return 0.0 if len1.zero? || len2.zero?
279
+
280
+ [len1, len2].min.to_f / [len1, len2].max
281
+ end
282
+
283
+ # Compute string similarity using Levenshtein distance.
284
+ #
285
+ # @param str1 [String] First string
286
+ # @param str2 [String] Second string
287
+ # @return [Float] Similarity score (0.0-1.0)
288
+ def string_similarity(str1, str2)
289
+ return 1.0 if str1 == str2
290
+ return 0.0 if str1.to_s.empty? || str2.to_s.empty?
291
+
292
+ distance = levenshtein_distance(str1.to_s, str2.to_s)
293
+ max_len = [str1.to_s.length, str2.to_s.length].max
294
+
295
+ 1.0 - (distance.to_f / max_len)
296
+ end
297
+
298
+ # Compute Levenshtein distance between two strings.
299
+ #
300
+ # Uses Wagner-Fischer algorithm with O(min(m,n)) space.
301
+ #
302
+ # @param str1 [String] First string
303
+ # @param str2 [String] Second string
304
+ # @return [Integer] Edit distance
305
+ def levenshtein_distance(str1, str2)
306
+ return str2.length if str1.empty?
307
+ return str1.length if str2.empty?
308
+
309
+ # Ensure str1 is the shorter string for space optimization
310
+ str1, str2 = str2, str1 if str1.length > str2.length
311
+
312
+ m = str1.length
313
+ n = str2.length
314
+
315
+ # Use two rows instead of full matrix
316
+ prev_row = (0..m).to_a
317
+ curr_row = Array.new(m + 1, 0)
318
+
319
+ (1..n).each do |j|
320
+ curr_row[0] = j
321
+
322
+ (1..m).each do |i|
323
+ cost = str1[i - 1] == str2[j - 1] ? 0 : 1
324
+ curr_row[i] = [
325
+ prev_row[i] + 1, # deletion
326
+ curr_row[i - 1] + 1, # insertion
327
+ prev_row[i - 1] + cost # substitution
328
+ ].min
329
+ end
330
+
331
+ prev_row, curr_row = curr_row, prev_row
332
+ end
333
+
334
+ prev_row[m]
335
+ end
336
+ end
337
+ end
338
+ end