exhale 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/.ruby-version +1 -0
- data/CHANGELOG.md +12 -0
- data/LICENSE.txt +21 -0
- data/README.md +149 -0
- data/Rakefile +96 -0
- data/exe/exhale +6 -0
- data/exhale.gemspec +43 -0
- data/lib/exhale/cli.rb +148 -0
- data/lib/exhale/contract/markdown.rb +107 -0
- data/lib/exhale/contract/reference.rb +63 -0
- data/lib/exhale/contract/resolver.rb +142 -0
- data/lib/exhale/contract.rb +194 -0
- data/lib/exhale/dry/check.rb +266 -0
- data/lib/exhale/dry/fingerprints.rb +105 -0
- data/lib/exhale/dry/gate.rb +425 -0
- data/lib/exhale/dry/index.rb +59 -0
- data/lib/exhale/dry/matcher.rb +468 -0
- data/lib/exhale/dry/normalizer/erb.rb +289 -0
- data/lib/exhale/dry/normalizer/ruby.rb +193 -0
- data/lib/exhale/dry/normalizer.rb +26 -0
- data/lib/exhale/errors.rb +27 -0
- data/lib/exhale/git.rb +279 -0
- data/lib/exhale/report.rb +203 -0
- data/lib/exhale/shape.rb +32 -0
- data/lib/exhale/source_files.rb +84 -0
- data/lib/exhale/unit.rb +27 -0
- data/lib/exhale/units/erb.rb +30 -0
- data/lib/exhale/units/ruby.rb +283 -0
- data/lib/exhale/version.rb +5 -0
- data/lib/exhale.rb +25 -0
- metadata +152 -0
|
@@ -0,0 +1,468 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "set"
|
|
4
|
+
require_relative "fingerprints"
|
|
5
|
+
require_relative "index"
|
|
6
|
+
|
|
7
|
+
module Exhale
|
|
8
|
+
module Dry
|
|
9
|
+
# Where one side of a match sits: a whole unit, or a fragment inside one.
|
|
10
|
+
# signature identifies a fragment's structure, so a fragment can be
|
|
11
|
+
# recognized at the base even after its lines move.
|
|
12
|
+
Location = Struct.new(:entry, :start_line, :end_line, :size, :set, :whole, :signature, :ordinal,
|
|
13
|
+
keyword_init: true) do
|
|
14
|
+
def unit
|
|
15
|
+
entry.unit
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
def path
|
|
19
|
+
unit.path
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
def lines
|
|
23
|
+
end_line - start_line + 1
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
# The file is part of the key: one identity defined in two files is
|
|
27
|
+
# two places, and a base pair keyed by one can't stand for the other.
|
|
28
|
+
def key
|
|
29
|
+
place = "#{path}##{unit.identity}"
|
|
30
|
+
whole ? place : "#{place}@#{signature.to_s(16)}##{ordinal}"
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
def structural_key
|
|
34
|
+
signature.to_s(16)
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
def contains?(other)
|
|
38
|
+
entry.equal?(other.entry) && start_line <= other.start_line && end_line >= other.end_line
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
# Sharing a line of the same file, whether or not the two sit in the
|
|
42
|
+
# same unit: `end; def second` puts two units on one line.
|
|
43
|
+
def overlaps?(other)
|
|
44
|
+
path == other.path && start_line <= other.end_line && other.start_line <= end_line
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
# Two locations are the same place when they cover the same lines of
|
|
48
|
+
# the same unit, whichever seeder found them.
|
|
49
|
+
def id
|
|
50
|
+
[entry.id, start_line, end_line]
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
def ==(other)
|
|
54
|
+
other.is_a?(Location) && id == other.id
|
|
55
|
+
end
|
|
56
|
+
alias_method :eql?, :==
|
|
57
|
+
|
|
58
|
+
def hash
|
|
59
|
+
id.hash
|
|
60
|
+
end
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
# A scored pair of locations. kind is :unit, :subtree or :run.
|
|
64
|
+
Match = Struct.new(:a, :b, :score, :kind, keyword_init: true) do
|
|
65
|
+
def size
|
|
66
|
+
[a.size, b.size].max
|
|
67
|
+
end
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
# Finds every pair of locations over the threshold. Two seeders propose
|
|
71
|
+
# candidates and one scorer judges them all:
|
|
72
|
+
#
|
|
73
|
+
# 1. Prefix filtering over rarest-first fingerprints finds every
|
|
74
|
+
# whole-unit pair over the threshold.
|
|
75
|
+
# 2. Exact subtree digests find fragments copied whole, as flay does.
|
|
76
|
+
# 3. Statement-run seeds find runs lifted out of the middle of a sequence,
|
|
77
|
+
# allowing one mismatched statement.
|
|
78
|
+
#
|
|
79
|
+
# Settings come per pair from the Contract, so the matcher generates
|
|
80
|
+
# candidates at the loosest floors any primitive asks for and judges each
|
|
81
|
+
# pair by its own.
|
|
82
|
+
class Matcher
|
|
83
|
+
RUN = 3
|
|
84
|
+
RUN_SEED_CAP = 50
|
|
85
|
+
STAR_ABOVE = 100
|
|
86
|
+
|
|
87
|
+
# Entry id => the id standing for its group of identical units, for
|
|
88
|
+
# every unit the matcher joined to at least one identical copy.
|
|
89
|
+
attr_reader :twins
|
|
90
|
+
|
|
91
|
+
def initialize(index, floors:, settings_for_pair:)
|
|
92
|
+
@index = index
|
|
93
|
+
@floors = floors
|
|
94
|
+
@settings_for_pair = settings_for_pair
|
|
95
|
+
@twins = {}
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
def matches
|
|
99
|
+
@twins = {}
|
|
100
|
+
eligible = @index.entries.select { |entry| eligible?(entry) }
|
|
101
|
+
found = unit_matches(eligible) + subtree_matches(eligible) + run_matches(eligible)
|
|
102
|
+
prune(found)
|
|
103
|
+
end
|
|
104
|
+
|
|
105
|
+
private
|
|
106
|
+
|
|
107
|
+
def eligible?(entry)
|
|
108
|
+
entry.unit.lines >= @floors[:min_lines] && entry.tree.size >= @floors[:min_nodes]
|
|
109
|
+
end
|
|
110
|
+
|
|
111
|
+
def whole(entry)
|
|
112
|
+
Location.new(entry: entry, start_line: entry.unit.start_line, end_line: entry.unit.end_line,
|
|
113
|
+
size: entry.tree.size, set: entry.set, whole: true, signature: entry.tree.digest)
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
# Units with identical trees are matched within their group, and only
|
|
117
|
+
# representatives enter the search for near copies. Eight hundred
|
|
118
|
+
# identical scaffold actions would otherwise make 320,000 pairs that
|
|
119
|
+
# all say the same thing.
|
|
120
|
+
#
|
|
121
|
+
# Identical units differ only in their lines and in the settings their
|
|
122
|
+
# primitives give them, so one representative stands for every member
|
|
123
|
+
# with the same tree, own settings and line count: those members are
|
|
124
|
+
# interchangeable. A near copy is compared with each representative.
|
|
125
|
+
def unit_matches(eligible)
|
|
126
|
+
identical = eligible.group_by { |entry| entry.tree.digest }.values.flat_map do |members|
|
|
127
|
+
class_star(members.map { |member| whole(member) }).filter_map do |a, b|
|
|
128
|
+
judge(a, b, :unit).tap { |match| join_twins(members.first, a, b) if match }
|
|
129
|
+
end
|
|
130
|
+
end
|
|
131
|
+
buckets = eligible.group_by { |entry| [entry.tree.digest, own_settings(whole(entry)), entry.unit.lines] }.values
|
|
132
|
+
stand_ins = buckets.to_h { |members| [members.first.id, members] }
|
|
133
|
+
identical + prefix_matches(buckets.map(&:first)).flat_map { |match| with_stand_ins(match, stand_ins) }
|
|
134
|
+
end
|
|
135
|
+
|
|
136
|
+
def join_twins(first, a, b)
|
|
137
|
+
@twins[a.entry.id] = first.id
|
|
138
|
+
@twins[b.entry.id] = first.id
|
|
139
|
+
end
|
|
140
|
+
|
|
141
|
+
# A representative joined to its identical copies carries them all into
|
|
142
|
+
# its finding. One that isn't joined, because its copies all miss their
|
|
143
|
+
# own floors, stands alone, so each copy pairs with the near copy itself.
|
|
144
|
+
def with_stand_ins(match, stand_ins)
|
|
145
|
+
a, b = match.a, match.b
|
|
146
|
+
extra = []
|
|
147
|
+
extra.concat(stand_ins.fetch(a.entry.id).drop(1).map { |entry| [whole(entry), b] }) unless @twins.key?(a.entry.id)
|
|
148
|
+
extra.concat(stand_ins.fetch(b.entry.id).drop(1).map { |entry| [a, whole(entry)] }) unless @twins.key?(b.entry.id)
|
|
149
|
+
[match] + extra.filter_map { |x, y| judge(x, y, :unit) }
|
|
150
|
+
end
|
|
151
|
+
|
|
152
|
+
# Connects every qualifying pair among identical locations without
|
|
153
|
+
# listing them all. Floors are two-dimensional and come per primitive,
|
|
154
|
+
# so no single center fits every pair. Copies are split by the settings
|
|
155
|
+
# their primitive gives them; for each pair of settings classes X and Y,
|
|
156
|
+
# the pair floor is the lower of the two in each dimension, every copy
|
|
157
|
+
# meeting it passes with every other (identical, so the score is 1), and
|
|
158
|
+
# a star across X and Y connects them. Cost is classes squared times
|
|
159
|
+
# copies.
|
|
160
|
+
def class_star(locations)
|
|
161
|
+
classes = locations.group_by { |location| own_settings(location) }.values
|
|
162
|
+
pairs = Set.new
|
|
163
|
+
classes.each_with_index do |x, at|
|
|
164
|
+
classes[at..].each do |y|
|
|
165
|
+
floor = @settings_for_pair.call(x.first.unit, y.first.unit)
|
|
166
|
+
xs = x.select { |location| big_enough?(location, floor) }
|
|
167
|
+
ys = x.equal?(y) ? xs : y.select { |location| big_enough?(location, floor) }
|
|
168
|
+
spokes(ys, xs).each { |pair| pairs << pair }
|
|
169
|
+
spokes(xs, ys).each { |pair| pairs << pair } unless x.equal?(y)
|
|
170
|
+
end
|
|
171
|
+
end
|
|
172
|
+
pairs.to_a.sort_by { |a, b| [a.entry.id, a.start_line, b.entry.id, b.start_line] }
|
|
173
|
+
end
|
|
174
|
+
|
|
175
|
+
# Each member paired with the first hub that isn't itself and shares no
|
|
176
|
+
# line with it, lower entry id first.
|
|
177
|
+
def spokes(hubs, members)
|
|
178
|
+
members.filter_map do |member|
|
|
179
|
+
hub = hubs.find { |candidate| !candidate.equal?(member) && !candidate.overlaps?(member) }
|
|
180
|
+
next unless hub
|
|
181
|
+
|
|
182
|
+
[hub, member].sort_by { |location| [location.entry.id, location.start_line] }
|
|
183
|
+
end
|
|
184
|
+
end
|
|
185
|
+
|
|
186
|
+
def own_settings(location)
|
|
187
|
+
@settings_for_pair.call(location.unit, location.unit)
|
|
188
|
+
end
|
|
189
|
+
|
|
190
|
+
def prefix_matches(entries)
|
|
191
|
+
postings = Hash.new { |hash, digest| hash[digest] = [] }
|
|
192
|
+
entries.each { |entry| entry.set.each { |digest| postings[digest] << entry } }
|
|
193
|
+
|
|
194
|
+
threshold = @floors[:threshold]
|
|
195
|
+
pairs = Set.new
|
|
196
|
+
entries.each do |entry|
|
|
197
|
+
prefix(entry, threshold).each do |digest|
|
|
198
|
+
postings.fetch(digest, []).each do |other|
|
|
199
|
+
next if other.equal?(entry)
|
|
200
|
+
|
|
201
|
+
pairs << (entry.id < other.id ? [entry, other] : [other, entry])
|
|
202
|
+
end
|
|
203
|
+
end
|
|
204
|
+
end
|
|
205
|
+
|
|
206
|
+
pairs.sort_by { |a, b| [a.id, b.id] }.filter_map do |a, b|
|
|
207
|
+
next if a.tree.digest == b.tree.digest
|
|
208
|
+
next unless comparable_totals?(a, b, threshold)
|
|
209
|
+
|
|
210
|
+
judge(whole(a), whole(b), :unit)
|
|
211
|
+
end
|
|
212
|
+
end
|
|
213
|
+
|
|
214
|
+
# A pair scoring at least t shares at least t of A's weight, so it must
|
|
215
|
+
# share one of the rarest fingerprints that together carry more than
|
|
216
|
+
# (1 - t) of that weight. Looking up only those finds every such pair,
|
|
217
|
+
# so candidate generation is exact, not a sample.
|
|
218
|
+
def prefix(entry, threshold)
|
|
219
|
+
budget = (1 - threshold) * entry.total
|
|
220
|
+
taken = 0
|
|
221
|
+
entry.set.sort_by { |digest| [@index.count(digest), digest] }.take_while do |digest|
|
|
222
|
+
next false if taken > budget
|
|
223
|
+
|
|
224
|
+
taken += @index.weights[digest]
|
|
225
|
+
true
|
|
226
|
+
end
|
|
227
|
+
end
|
|
228
|
+
|
|
229
|
+
# Jaccard can't exceed the smaller total over the larger one.
|
|
230
|
+
def comparable_totals?(a, b, threshold)
|
|
231
|
+
small, large = [a.total, b.total].minmax
|
|
232
|
+
small >= threshold * large
|
|
233
|
+
end
|
|
234
|
+
|
|
235
|
+
def subtree_matches(eligible)
|
|
236
|
+
groups = Hash.new { |hash, digest| hash[digest] = [] }
|
|
237
|
+
eligible.each do |entry|
|
|
238
|
+
entry.tree.each_node do |node|
|
|
239
|
+
next if node.equal?(entry.tree)
|
|
240
|
+
next if node.lines < @floors[:min_lines] || node.size < @floors[:min_nodes]
|
|
241
|
+
|
|
242
|
+
groups[node.digest] << [entry, node]
|
|
243
|
+
end
|
|
244
|
+
end
|
|
245
|
+
|
|
246
|
+
groups.each_value.flat_map do |occurrences|
|
|
247
|
+
next [] if occurrences.size < 2
|
|
248
|
+
|
|
249
|
+
locations = occurrences.map { |entry, node| fragment(entry, node) }
|
|
250
|
+
pairs = locations.size > STAR_ABOVE ? class_star(locations) : locations.combination(2)
|
|
251
|
+
pairs.filter_map { |a, b| judge(a, b, :subtree) }
|
|
252
|
+
end
|
|
253
|
+
end
|
|
254
|
+
|
|
255
|
+
def fragment(entry, node)
|
|
256
|
+
Location.new(entry: entry, start_line: node.start_line, end_line: node.end_line, size: node.size,
|
|
257
|
+
set: node.digests, whole: false, signature: node.digest,
|
|
258
|
+
ordinal: subtree_ordinal(entry, node))
|
|
259
|
+
end
|
|
260
|
+
|
|
261
|
+
# Which occurrence of this shape inside its unit, counted in preorder.
|
|
262
|
+
# Two copies of one fragment in the same method need different keys, or
|
|
263
|
+
# a new second copy would read as the old first one.
|
|
264
|
+
def subtree_ordinal(entry, node)
|
|
265
|
+
@subtrees ||= {}
|
|
266
|
+
by_digest = (@subtrees[entry.id] ||= entry.tree.each_node.group_by(&:digest))
|
|
267
|
+
by_digest.fetch(node.digest).index { |candidate| candidate.equal?(node) } + 1
|
|
268
|
+
end
|
|
269
|
+
|
|
270
|
+
# Every pair in a small group of run seeds; a star in a big one, which
|
|
271
|
+
# is enough to cluster them without a quadratic blowup. The hub has to
|
|
272
|
+
# be a copy that can pass: see #run_hub.
|
|
273
|
+
def connect(items, cap)
|
|
274
|
+
return items.combination(2).to_a if items.size <= cap
|
|
275
|
+
|
|
276
|
+
hub = yield(items)
|
|
277
|
+
items.reject { |item| item.equal?(hub) }.map { |other| [hub, other] }
|
|
278
|
+
end
|
|
279
|
+
|
|
280
|
+
# Run seeds aren't locations yet, so the hub is the seed whose window
|
|
281
|
+
# spans the most source lines. A copy squeezed onto one line would fail
|
|
282
|
+
# the line floor against every partner and take the group down with it.
|
|
283
|
+
def run_hub(seeds, sequences)
|
|
284
|
+
seeds.max_by do |id, i|
|
|
285
|
+
children = sequences[id][1].children
|
|
286
|
+
[children[i + RUN - 1].end_line - children[i].start_line, -id, -i]
|
|
287
|
+
end
|
|
288
|
+
end
|
|
289
|
+
|
|
290
|
+
def run_matches(eligible)
|
|
291
|
+
sequences = []
|
|
292
|
+
windows = Hash.new { |hash, key| hash[key] = [] }
|
|
293
|
+
eligible.each do |entry|
|
|
294
|
+
entry.tree.each_node do |node|
|
|
295
|
+
next unless node.sequence? && node.children.size >= RUN
|
|
296
|
+
|
|
297
|
+
id = sequences.size
|
|
298
|
+
sequences << [entry, node]
|
|
299
|
+
digests = node.children.map(&:digest)
|
|
300
|
+
(0..(digests.size - RUN)).each { |i| windows[digests[i, RUN]] << [id, i] }
|
|
301
|
+
end
|
|
302
|
+
end
|
|
303
|
+
|
|
304
|
+
seen = Set.new
|
|
305
|
+
windows.each_value.flat_map do |seeds|
|
|
306
|
+
next [] if seeds.size < 2
|
|
307
|
+
|
|
308
|
+
# A run copied past the cap still gets found, as a star from its
|
|
309
|
+
# widest occurrence, the same way big subtree groups are connected.
|
|
310
|
+
connect(seeds, RUN_SEED_CAP) { |group| run_hub(group, sequences) }.filter_map do |(id_a, i), (id_b, j)|
|
|
311
|
+
left = sequences[id_a][1].children
|
|
312
|
+
right = sequences[id_b][1].children
|
|
313
|
+
run = extend_run(left, right, i, j)
|
|
314
|
+
next if id_a == id_b && ranges_overlap?(run[0], run[1], run[2])
|
|
315
|
+
next unless seen.add?([id_a, run[0], id_b, run[1], run[2]])
|
|
316
|
+
|
|
317
|
+
judge_run(sequences, id_a, id_b, run) || judge_core(sequences, id_a, id_b, exact_core(left, right, i, j), run, seen)
|
|
318
|
+
end
|
|
319
|
+
end
|
|
320
|
+
end
|
|
321
|
+
|
|
322
|
+
def judge_run(sequences, id_a, id_b, run)
|
|
323
|
+
a = run_location(sequences[id_a][0], sequences[id_a][1].children, run[0], run[2])
|
|
324
|
+
b = run_location(sequences[id_b][0], sequences[id_b][1].children, run[1], run[2])
|
|
325
|
+
judge(a, b, :run)
|
|
326
|
+
end
|
|
327
|
+
|
|
328
|
+
# A run that spent its mismatch can score under the threshold when the
|
|
329
|
+
# mismatched statements are big. The exact core around the seed is then
|
|
330
|
+
# judged on its own.
|
|
331
|
+
def judge_core(sequences, id_a, id_b, core, run, seen)
|
|
332
|
+
return if core == run || (id_a == id_b && ranges_overlap?(core[0], core[1], core[2]))
|
|
333
|
+
return unless seen.add?([id_a, core[0], id_b, core[1], core[2]])
|
|
334
|
+
|
|
335
|
+
judge_run(sequences, id_a, id_b, core)
|
|
336
|
+
end
|
|
337
|
+
|
|
338
|
+
# The longest run of identical statements around a seed.
|
|
339
|
+
def exact_core(left, right, i, j)
|
|
340
|
+
length = RUN
|
|
341
|
+
length += 1 while i + length < left.size && j + length < right.size &&
|
|
342
|
+
left[i + length].digest == right[j + length].digest
|
|
343
|
+
back = 0
|
|
344
|
+
back += 1 while i - back - 1 >= 0 && j - back - 1 >= 0 && left[i - back - 1].digest == right[j - back - 1].digest
|
|
345
|
+
[i - back, j - back, length + back]
|
|
346
|
+
end
|
|
347
|
+
|
|
348
|
+
def ranges_overlap?(start_a, start_b, length)
|
|
349
|
+
start_a < start_b + length && start_b < start_a + length
|
|
350
|
+
end
|
|
351
|
+
|
|
352
|
+
# Grows a seed in both directions while statements keep matching,
|
|
353
|
+
# spending at most one mismatched statement, as in flay's fuzzy mode.
|
|
354
|
+
def extend_run(left, right, i, j)
|
|
355
|
+
budget = 1
|
|
356
|
+
length = RUN
|
|
357
|
+
while i + length < left.size && j + length < right.size
|
|
358
|
+
if left[i + length].digest == right[j + length].digest
|
|
359
|
+
length += 1
|
|
360
|
+
elsif budget.positive? && i + length + 1 < left.size && j + length + 1 < right.size &&
|
|
361
|
+
left[i + length + 1].digest == right[j + length + 1].digest
|
|
362
|
+
budget -= 1
|
|
363
|
+
length += 2
|
|
364
|
+
else
|
|
365
|
+
break
|
|
366
|
+
end
|
|
367
|
+
end
|
|
368
|
+
|
|
369
|
+
back = 0
|
|
370
|
+
while i - back - 1 >= 0 && j - back - 1 >= 0
|
|
371
|
+
if left[i - back - 1].digest == right[j - back - 1].digest
|
|
372
|
+
back += 1
|
|
373
|
+
elsif budget.positive? && i - back - 2 >= 0 && j - back - 2 >= 0 &&
|
|
374
|
+
left[i - back - 2].digest == right[j - back - 2].digest
|
|
375
|
+
budget -= 1
|
|
376
|
+
back += 2
|
|
377
|
+
else
|
|
378
|
+
break
|
|
379
|
+
end
|
|
380
|
+
end
|
|
381
|
+
|
|
382
|
+
[i - back, j - back, length + back]
|
|
383
|
+
end
|
|
384
|
+
|
|
385
|
+
def run_location(entry, children, start, length)
|
|
386
|
+
nodes = children[start, length]
|
|
387
|
+
set = Set.new
|
|
388
|
+
nodes.each { |node| set.merge(node.digests) }
|
|
389
|
+
signature = Fingerprints.run_digest(nodes)
|
|
390
|
+
Location.new(entry: entry, start_line: nodes.first.start_line, end_line: nodes.last.end_line,
|
|
391
|
+
size: nodes.sum(&:size), set: set, whole: false, signature: signature,
|
|
392
|
+
ordinal: run_ordinal(entry, children, start, nodes.map(&:digest)))
|
|
393
|
+
end
|
|
394
|
+
|
|
395
|
+
# Which occurrence of this run of statements inside its unit, counting
|
|
396
|
+
# every sequence in preorder, so the key doesn't depend on which other
|
|
397
|
+
# copies happened to match.
|
|
398
|
+
def run_ordinal(entry, children, start, digests)
|
|
399
|
+
count = 0
|
|
400
|
+
entry.tree.each_node do |node|
|
|
401
|
+
next unless node.sequence?
|
|
402
|
+
|
|
403
|
+
node_digests = node.children.map(&:digest)
|
|
404
|
+
(0..(node_digests.size - digests.size)).each do |i|
|
|
405
|
+
return count + 1 if node.children.equal?(children) && i == start
|
|
406
|
+
|
|
407
|
+
count += 1 if node_digests[i, digests.size] == digests
|
|
408
|
+
end
|
|
409
|
+
end
|
|
410
|
+
count + 1
|
|
411
|
+
end
|
|
412
|
+
|
|
413
|
+
def judge(a, b, kind)
|
|
414
|
+
return if a.overlaps?(b)
|
|
415
|
+
|
|
416
|
+
settings = @settings_for_pair.call(a.unit, b.unit)
|
|
417
|
+
return unless big_enough?(a, settings) && big_enough?(b, settings)
|
|
418
|
+
|
|
419
|
+
score = kind == :subtree ? Rational(1) : @index.score(a.set, total(a), b.set, total(b))
|
|
420
|
+
return if score < settings[:threshold]
|
|
421
|
+
|
|
422
|
+
Match.new(a: a, b: b, score: score, kind: kind)
|
|
423
|
+
end
|
|
424
|
+
|
|
425
|
+
def big_enough?(location, settings)
|
|
426
|
+
location.lines >= settings[:min_lines] && location.size >= settings[:min_nodes]
|
|
427
|
+
end
|
|
428
|
+
|
|
429
|
+
def total(location)
|
|
430
|
+
location.whole ? location.entry.total : @index.weight_of(location.set)
|
|
431
|
+
end
|
|
432
|
+
|
|
433
|
+
# Drops any match whose two sides sit inside the two sides of a larger
|
|
434
|
+
# match already kept. A whole-unit match swallows the fragments inside
|
|
435
|
+
# it, and a fragment swallows the smaller fragments inside it.
|
|
436
|
+
def prune(found)
|
|
437
|
+
ordered = found.sort_by do |match|
|
|
438
|
+
[match.a.whole && match.b.whole ? 0 : 1, -match.size, match.a.key, match.b.key, match.a.id, match.b.id]
|
|
439
|
+
end
|
|
440
|
+
|
|
441
|
+
kept_by_entries = Hash.new { |hash, key| hash[key] = [] }
|
|
442
|
+
ordered.each_with_object([]) do |match, kept|
|
|
443
|
+
key = [match.a.entry.id, match.b.entry.id].sort
|
|
444
|
+
next if twins?(match)
|
|
445
|
+
next if kept_by_entries[key].any? { |bigger| inside?(match, bigger) }
|
|
446
|
+
|
|
447
|
+
kept_by_entries[key] << match
|
|
448
|
+
kept << match
|
|
449
|
+
end
|
|
450
|
+
end
|
|
451
|
+
|
|
452
|
+
# Two identical units are already one finding through the star, so a
|
|
453
|
+
# fragment shared between them says nothing new. A fragment repeated
|
|
454
|
+
# inside a single unit still counts.
|
|
455
|
+
def twins?(match)
|
|
456
|
+
return false if match.a.whole || match.a.entry.equal?(match.b.entry)
|
|
457
|
+
|
|
458
|
+
twin_a = @twins && @twins[match.a.entry.id]
|
|
459
|
+
twin_a && twin_a == @twins[match.b.entry.id]
|
|
460
|
+
end
|
|
461
|
+
|
|
462
|
+
def inside?(match, bigger)
|
|
463
|
+
(bigger.a.contains?(match.a) && bigger.b.contains?(match.b)) ||
|
|
464
|
+
(bigger.a.contains?(match.b) && bigger.b.contains?(match.a))
|
|
465
|
+
end
|
|
466
|
+
end
|
|
467
|
+
end
|
|
468
|
+
end
|