relaton 3.0.0.pre.alpha.6 → 3.0.0.pre.alpha.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/lib/relaton/etsi/data_fetcher.rb +118 -9
- data/lib/relaton/iec/processor.rb +1 -1
- data/lib/relaton/index/config.rb +9 -1
- data/lib/relaton/index/file_io.rb +190 -0
- data/lib/relaton/index/type.rb +96 -23
- data/lib/relaton/version.rb +1 -1
- metadata +2 -2
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 21f068205ef6a80e718953c7989c7a51219a9f5f6e25a4fecc67ac93d9fa4da5
|
|
4
|
+
data.tar.gz: 7c438465496dc3b95370bc6990420073c63f41d5edba70022a71b39bd45f343f
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: af94d0ffa5ae9cbeefecf98d9aad8a04a7ea74814b92e470b6235b2bb1c2f19593db50e0f919b2ce3bbf1bdf44b8e637fa20000e6d59bc400736159fe0f9a2d1
|
|
7
|
+
data.tar.gz: 833bbe454eaa888a145fb9dcd5daff7d8bc8cf2e03f603c2100b0d19d981531a5b4bfce87b77bc00f3b5a5684904473f58656ebf820b2c7272b606f5c26afddc
|
|
@@ -45,22 +45,117 @@ module Relaton
|
|
|
45
45
|
report_errors
|
|
46
46
|
end
|
|
47
47
|
|
|
48
|
-
|
|
48
|
+
#
|
|
49
|
+
# Fetch pages 2..N. A page that stays bad after its retries is fetched
|
|
50
|
+
# again after the last page; if it is still bad, the crawl fails. A page
|
|
51
|
+
# is never skipped: relaton-data-etsi deletes data/ before the crawl and
|
|
52
|
+
# commits what it writes, so a skipped page would unpublish its documents.
|
|
53
|
+
#
|
|
54
|
+
# N comes from page 1's total_count, but the result set is sorted by
|
|
55
|
+
# deliverable number, so a document published during the crawl shifts the
|
|
56
|
+
# later pages by one. The crawl therefore reads on past N while the pages
|
|
57
|
+
# stay full (at most MAX_EXTRA_PAGES), and one page past a deferred last
|
|
58
|
+
# page; a record read twice overwrites its own file.
|
|
59
|
+
#
|
|
60
|
+
def fetch_remaining_pages(first_page) # rubocop:disable Metrics/MethodLength
|
|
61
|
+
total_pages = last = last_page(first_page)
|
|
62
|
+
deferred = []
|
|
63
|
+
page = 2
|
|
64
|
+
while page <= last
|
|
65
|
+
check_extra_pages page, total_pages
|
|
66
|
+
size = fetch_next_page(page, deferred)
|
|
67
|
+
break if size&.zero?
|
|
68
|
+
|
|
69
|
+
last = page + 1 if page == last && read_on?(size, page, total_pages)
|
|
70
|
+
page += 1
|
|
71
|
+
end
|
|
72
|
+
fetch_deferred_pages(deferred, total_pages)
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
def last_page(first_page)
|
|
49
76
|
total = first_page.first ? first_page.first["total_count"].to_i : 0
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
77
|
+
last = (total / PAGE_SIZE.to_f).ceil
|
|
78
|
+
first_page.size >= PAGE_SIZE ? [last, 2].max : last
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
#
|
|
82
|
+
# Whether to read the page after the last one: after a full page, or
|
|
83
|
+
# after a deferred page inside the total_count range. A deferred page
|
|
84
|
+
# past that range does not extend it, so a server that answers bad
|
|
85
|
+
# bodies past the end cannot keep the crawl going.
|
|
86
|
+
#
|
|
87
|
+
def read_on?(size, page, total_pages)
|
|
88
|
+
size ? size >= PAGE_SIZE : page <= total_pages
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
def check_extra_pages(page, total_pages)
|
|
92
|
+
return if page <= total_pages + MAX_EXTRA_PAGES
|
|
93
|
+
|
|
94
|
+
raise BadPage, "ETSI page #{page} is still full #{MAX_EXTRA_PAGES} " \
|
|
95
|
+
"pages past total_count (#{total_pages} pages)."
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
#
|
|
99
|
+
# @return [Integer, nil] the number of records on the page; nil for a
|
|
100
|
+
# deferred page
|
|
101
|
+
#
|
|
102
|
+
def fetch_next_page(page, deferred)
|
|
103
|
+
records = fetch_page(page)
|
|
104
|
+
process_records(records)
|
|
105
|
+
records.size
|
|
106
|
+
rescue BadPage => e
|
|
107
|
+
Util.warn "#{e.message} Fetching it again after the last page."
|
|
108
|
+
deferred << page
|
|
109
|
+
nil
|
|
110
|
+
end
|
|
54
111
|
|
|
55
|
-
|
|
112
|
+
def fetch_deferred_pages(pages, total_pages)
|
|
113
|
+
return if pages.empty?
|
|
114
|
+
|
|
115
|
+
sleep DEFERRED_DELAY
|
|
116
|
+
failed = pages.filter_map do |page|
|
|
117
|
+
refetch_page page, total_pages
|
|
118
|
+
rescue BadPage => e
|
|
119
|
+
e.message
|
|
56
120
|
end
|
|
121
|
+
raise BadPage, failed.join(" ") if failed.any?
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
# @return [String, nil] the failure message, nil on success
|
|
125
|
+
def refetch_page(page, total_pages)
|
|
126
|
+
records = fetch_page(page)
|
|
127
|
+
if records.empty? && page <= total_pages
|
|
128
|
+
return "ETSI page #{page} is empty on the re-fetch."
|
|
129
|
+
end
|
|
130
|
+
|
|
131
|
+
process_records records
|
|
132
|
+
nil
|
|
57
133
|
end
|
|
58
134
|
|
|
59
135
|
def fetch_page(page)
|
|
60
136
|
date = Time.now.to_date + 1
|
|
61
137
|
timestamp = (Time.now.to_f * 1000).to_i
|
|
62
138
|
url = format(SOURCEURL, page: page, date: date, timestamp: timestamp)
|
|
63
|
-
|
|
139
|
+
fetch_with_retry(url) { |body| parse_page(page, body) }
|
|
140
|
+
end
|
|
141
|
+
|
|
142
|
+
#
|
|
143
|
+
# ETSI's data.php sometimes answers with PHP's `Array` text in place of
|
|
144
|
+
# JSON (relaton-data-etsi crawl 36761804954); a re-fetch gets JSON.
|
|
145
|
+
#
|
|
146
|
+
# @raise [BadPage] when the body is not a JSON array
|
|
147
|
+
#
|
|
148
|
+
def parse_page(page, body)
|
|
149
|
+
records = JSON.parse(body)
|
|
150
|
+
return records if records.is_a?(Array)
|
|
151
|
+
|
|
152
|
+
raise BadPage, bad_page_message(page, body)
|
|
153
|
+
rescue JSON::ParserError
|
|
154
|
+
raise BadPage, bad_page_message(page, body)
|
|
155
|
+
end
|
|
156
|
+
|
|
157
|
+
def bad_page_message(page, body)
|
|
158
|
+
"ETSI page #{page} is not a JSON array: #{body.to_s[0, 200].inspect}."
|
|
64
159
|
end
|
|
65
160
|
|
|
66
161
|
def process_records(records)
|
|
@@ -92,16 +187,30 @@ module Relaton
|
|
|
92
187
|
"Published"
|
|
93
188
|
end
|
|
94
189
|
|
|
190
|
+
# A page body that is not a JSON array. See #parse_page.
|
|
191
|
+
class BadPage < StandardError; end
|
|
192
|
+
|
|
193
|
+
# Seconds to wait before a bad page is fetched again after the last page.
|
|
194
|
+
DEFERRED_DELAY = 60
|
|
195
|
+
|
|
196
|
+
# Full pages read past page 1's total_count before the crawl fails.
|
|
197
|
+
MAX_EXTRA_PAGES = 10
|
|
198
|
+
|
|
95
199
|
NETWORK_ERRORS = [
|
|
96
200
|
Mechanize::Error, Net::OpenTimeout, Net::ReadTimeout,
|
|
97
201
|
SocketError, Errno::ECONNRESET
|
|
98
202
|
].freeze
|
|
99
203
|
|
|
204
|
+
#
|
|
205
|
+
# @yield [String] the body; the block's result is returned, and a BadPage
|
|
206
|
+
# it raises is retried like a network error
|
|
207
|
+
#
|
|
100
208
|
def fetch_with_retry(url, retries: 3, delay: 2) # rubocop:disable Metrics/MethodLength
|
|
101
209
|
attempt = 0
|
|
102
210
|
begin
|
|
103
|
-
Mechanize.new.get(url).body
|
|
104
|
-
|
|
211
|
+
body = Mechanize.new.get(url).body
|
|
212
|
+
block_given? ? yield(body) : body
|
|
213
|
+
rescue *NETWORK_ERRORS, BadPage => e
|
|
105
214
|
attempt += 1
|
|
106
215
|
if attempt <= retries
|
|
107
216
|
Util.info "Fetch failed (#{e.message}), " \
|
|
@@ -7,7 +7,7 @@ module Relaton
|
|
|
7
7
|
@short = :relaton_iec
|
|
8
8
|
@prefix = "IEC"
|
|
9
9
|
@pubid_flavor = :Iec # global prefixes sourced from Pubid::Iec.prefixes
|
|
10
|
-
@defaultprefix = %r{^(IEC\s|CISPR\s|IEV($|\s))}
|
|
10
|
+
@defaultprefix = %r{^(IEC\s|CISPR\s|IEV($|\s)|CEI\s)} # CEI: IEC's French spelling (relaton#243)
|
|
11
11
|
@idtype = "IEC"
|
|
12
12
|
@datasets = %w[iec-harmonized-all iec-harmonized-latest]
|
|
13
13
|
end
|
data/lib/relaton/index/config.rb
CHANGED
|
@@ -4,7 +4,7 @@ module Relaton
|
|
|
4
4
|
# Configuration class for Relaton::Index
|
|
5
5
|
#
|
|
6
6
|
class Config
|
|
7
|
-
attr_reader :storage, :storage_dir, :filename
|
|
7
|
+
attr_reader :storage, :storage_dir, :filename, :build_sidecar_in_child
|
|
8
8
|
|
|
9
9
|
#
|
|
10
10
|
# Set default values
|
|
@@ -13,6 +13,14 @@ module Relaton
|
|
|
13
13
|
@storage = FileStorage
|
|
14
14
|
@storage_dir = Dir.home
|
|
15
15
|
@filename = "index.yaml"
|
|
16
|
+
@build_sidecar_in_child = true
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
# Build the sidecar in a forked child when available, so the one-time
|
|
20
|
+
# full materialization's memory dies with the child (relaton#242).
|
|
21
|
+
# Set to false to force an in-process build.
|
|
22
|
+
def build_sidecar_in_child=(flag)
|
|
23
|
+
@build_sidecar_in_child = flag
|
|
16
24
|
end
|
|
17
25
|
|
|
18
26
|
#
|
|
@@ -11,6 +11,10 @@ module Relaton
|
|
|
11
11
|
# wrong-structure handling (re-download, or stop and log).
|
|
12
12
|
class InvalidIndexError < StandardError; end
|
|
13
13
|
|
|
14
|
+
# Bump when the sidecar payload shape changes, so an older sidecar is
|
|
15
|
+
# discarded and rebuilt instead of misread.
|
|
16
|
+
SIDECAR_VERSION = 2 # v2: the payload carries the yaml byte-size for freshness
|
|
17
|
+
|
|
14
18
|
attr_reader :url, :pubid_class
|
|
15
19
|
attr_accessor :sorted
|
|
16
20
|
|
|
@@ -288,11 +292,197 @@ module Relaton
|
|
|
288
292
|
#
|
|
289
293
|
def remove
|
|
290
294
|
Index.config.storage.remove file
|
|
295
|
+
delete_sidecar
|
|
291
296
|
[]
|
|
292
297
|
end
|
|
293
298
|
|
|
299
|
+
#
|
|
300
|
+
# Raw-row read for the lazy search path (relaton#242 stopgap): returns
|
|
301
|
+
# precomputed root-number sort keys and the rows as plain hashes — no
|
|
302
|
+
# pubid objects. A Marshal sidecar next to the yaml holds the keys+rows
|
|
303
|
+
# so repeat loads skip the YAML parse and the one-time full
|
|
304
|
+
# materialization; the sidecar is rebuilt whenever the yaml is newer.
|
|
305
|
+
#
|
|
306
|
+
# @return [Array<Array<String>, Array<Hash>] sort keys and raw rows
|
|
307
|
+
#
|
|
308
|
+
def read_raw
|
|
309
|
+
case url
|
|
310
|
+
when String
|
|
311
|
+
with_file_lock do
|
|
312
|
+
check_file ? read_raw_file : fetch_raw_and_save
|
|
313
|
+
end
|
|
314
|
+
else
|
|
315
|
+
read_raw_file || [[], []]
|
|
316
|
+
end
|
|
317
|
+
end
|
|
318
|
+
|
|
319
|
+
# Deserialize raw rows into pubid rows. Only the caller knows which
|
|
320
|
+
# slice it needs, so materialization happens here rather than at load.
|
|
321
|
+
#
|
|
322
|
+
# @param [Array<Hash>] rows raw rows
|
|
323
|
+
# @return [Array<Hash>] rows with deserialized ids
|
|
324
|
+
def materialize(rows)
|
|
325
|
+
return rows unless @pubid_class
|
|
326
|
+
|
|
327
|
+
rows.map { |r| { id: deserialize_id(r[:id]), file: r[:file] } }
|
|
328
|
+
end
|
|
329
|
+
|
|
330
|
+
# Save raw rows (possibly alongside the keys they were loaded with);
|
|
331
|
+
# sorts by the same root-number key the object path sorts by.
|
|
332
|
+
#
|
|
333
|
+
# @param [Array<String>] keys sort keys, parallel to rows
|
|
334
|
+
# @param [Array<Hash>] rows raw rows
|
|
335
|
+
# @return [void]
|
|
336
|
+
def save_raw(keys, rows)
|
|
337
|
+
ordered = @pubid_class ? keys.zip(rows).sort_by { |k, _| k.to_s }
|
|
338
|
+
.map { |_, r| r } : rows
|
|
339
|
+
yaml = ordered.map do |item|
|
|
340
|
+
{ id: item[:id], file: item[:file] }
|
|
341
|
+
end.to_yaml
|
|
342
|
+
Index.config.storage.write file, yaml
|
|
343
|
+
delete_sidecar
|
|
344
|
+
end
|
|
345
|
+
|
|
294
346
|
private
|
|
295
347
|
|
|
348
|
+
def sidecar_file
|
|
349
|
+
"#{file}.ms"
|
|
350
|
+
end
|
|
351
|
+
|
|
352
|
+
def delete_sidecar
|
|
353
|
+
File.delete(sidecar_file) if File.file?(sidecar_file)
|
|
354
|
+
rescue Errno::EACCES
|
|
355
|
+
nil
|
|
356
|
+
end
|
|
357
|
+
|
|
358
|
+
def read_raw_file
|
|
359
|
+
# The sidecar is authoritative while it describes the yaml byte-for-byte
|
|
360
|
+
# (same size) — the yaml is not even parsed until the sidecar is stale
|
|
361
|
+
# or unreadable.
|
|
362
|
+
if sidecar_fresh?
|
|
363
|
+
loaded = sidecar
|
|
364
|
+
return loaded if loaded
|
|
365
|
+
end
|
|
366
|
+
|
|
367
|
+
yaml = Index.config.storage.read(file)
|
|
368
|
+
return unless yaml
|
|
369
|
+
|
|
370
|
+
begin
|
|
371
|
+
raw = YAML.safe_load(yaml, permitted_classes: [Symbol])
|
|
372
|
+
rescue Psych::SyntaxError
|
|
373
|
+
warn_local_index_error("YAML parsing error when reading")
|
|
374
|
+
delete_sidecar
|
|
375
|
+
return [[], []]
|
|
376
|
+
end
|
|
377
|
+
build_raw(raw)
|
|
378
|
+
end
|
|
379
|
+
|
|
380
|
+
# Size, not mtime: NTFS timestamp coarseness makes a rewritten yaml
|
|
381
|
+
# carry the sidecar's own mtime, so "yaml is newer" misses the rebuild.
|
|
382
|
+
def sidecar_fresh?
|
|
383
|
+
File.file?(sidecar_file) && File.file?(file) &&
|
|
384
|
+
File.size(file) == sidecar_yaml_size
|
|
385
|
+
end
|
|
386
|
+
|
|
387
|
+
def sidecar_yaml_size
|
|
388
|
+
Marshal.load(File.binread(sidecar_file))[4]
|
|
389
|
+
rescue TypeError, ArgumentError, EOFError
|
|
390
|
+
nil
|
|
391
|
+
end
|
|
392
|
+
|
|
393
|
+
def sidecar
|
|
394
|
+
version, sorted, keys, rows = Marshal.load(File.binread(sidecar_file))
|
|
395
|
+
return unless version == SIDECAR_VERSION
|
|
396
|
+
|
|
397
|
+
@sorted = sorted
|
|
398
|
+
[keys, rows]
|
|
399
|
+
rescue TypeError, ArgumentError, EOFError
|
|
400
|
+
nil
|
|
401
|
+
end
|
|
402
|
+
|
|
403
|
+
def build_raw(raw)
|
|
404
|
+
unless check_format(raw)
|
|
405
|
+
warn_local_index_error("Wrong structure of")
|
|
406
|
+
return [[], []]
|
|
407
|
+
end
|
|
408
|
+
|
|
409
|
+
return build_raw_in_child(raw) if build_sidecar_in_child?
|
|
410
|
+
|
|
411
|
+
build_raw_in_process(raw)
|
|
412
|
+
rescue InvalidIndexError
|
|
413
|
+
warn_local_index_error("Wrong structure of")
|
|
414
|
+
[[], []]
|
|
415
|
+
end
|
|
416
|
+
|
|
417
|
+
# The one-time key build materializes every row (~675 MB on the
|
|
418
|
+
# 79,993-row ISO index). Freed pages are often not returned to the OS,
|
|
419
|
+
# so an in-process build leaves the parent's RSS high — and containers
|
|
420
|
+
# OOM on RSS (relaton#242). Where fork exists, build the sidecar in a
|
|
421
|
+
# child: the graph dies with it and the parent never spikes.
|
|
422
|
+
def build_sidecar_in_child?
|
|
423
|
+
Process.respond_to?(:fork) && Index.config.build_sidecar_in_child != false
|
|
424
|
+
end
|
|
425
|
+
|
|
426
|
+
def build_raw_in_child(raw)
|
|
427
|
+
pid = Process.fork do
|
|
428
|
+
build_raw_in_process(raw)
|
|
429
|
+
exit!(0)
|
|
430
|
+
end
|
|
431
|
+
Process.wait(pid)
|
|
432
|
+
loaded = sidecar
|
|
433
|
+
return build_raw_in_process(raw) unless loaded # child failed
|
|
434
|
+
|
|
435
|
+
loaded
|
|
436
|
+
rescue Errno::ENOMEM, SystemCallError
|
|
437
|
+
build_raw_in_process(raw)
|
|
438
|
+
end
|
|
439
|
+
|
|
440
|
+
def build_raw_in_process(raw)
|
|
441
|
+
objects = deserialize_pubid(raw)
|
|
442
|
+
keys = objects.map { |r| r[:id].root.number.to_s }
|
|
443
|
+
rows = objects.map { |r| { id: raw_id(r), file: r[:file] } }
|
|
444
|
+
write_sidecar(keys, rows)
|
|
445
|
+
[keys, rows]
|
|
446
|
+
rescue InvalidIndexError
|
|
447
|
+
warn_local_index_error("Wrong structure of")
|
|
448
|
+
[[], []]
|
|
449
|
+
end
|
|
450
|
+
|
|
451
|
+
def raw_id(row)
|
|
452
|
+
id = row[:id]
|
|
453
|
+
id.respond_to?(:to_hash) ? id.to_hash : id
|
|
454
|
+
end
|
|
455
|
+
|
|
456
|
+
def write_sidecar(keys, rows)
|
|
457
|
+
File.binwrite(sidecar_file,
|
|
458
|
+
Marshal.dump([SIDECAR_VERSION, @sorted, keys, rows,
|
|
459
|
+
File.size(file)]))
|
|
460
|
+
rescue Errno::EACCES, Errno::ENOENT, Errno::EROFS
|
|
461
|
+
nil # the sidecar is an optimization; a read-only dir still works
|
|
462
|
+
end
|
|
463
|
+
|
|
464
|
+
def fetch_raw_and_save
|
|
465
|
+
uri = URI.parse(url)
|
|
466
|
+
body = Net::HTTP.get(uri)
|
|
467
|
+
yaml = nil
|
|
468
|
+
Zip::File.open_buffer(body) do |zip|
|
|
469
|
+
yaml = zip.entries.first.get_input_stream.read
|
|
470
|
+
end
|
|
471
|
+
Util.info "Downloaded index from `#{url}`", progname
|
|
472
|
+
raw = YAML.safe_load(yaml, permitted_classes: [Symbol])
|
|
473
|
+
if check_format(raw)
|
|
474
|
+
save raw
|
|
475
|
+
raw = nil # release the parsed copy; read_raw_file loads the sidecar's
|
|
476
|
+
read_raw_file
|
|
477
|
+
else
|
|
478
|
+
warn_remote_index_error "Wrong structure of"
|
|
479
|
+
[[], []]
|
|
480
|
+
end
|
|
481
|
+
rescue Psych::SyntaxError
|
|
482
|
+
warn_remote_index_error "YAML parsing error when reading"
|
|
483
|
+
[[], []]
|
|
484
|
+
end
|
|
485
|
+
|
|
296
486
|
def with_file_lock(&)
|
|
297
487
|
@@file_locks_mutex.synchronize do
|
|
298
488
|
@@file_locks[file] ||= Mutex.new
|
data/lib/relaton/index/type.rb
CHANGED
|
@@ -30,7 +30,24 @@ module Relaton
|
|
|
30
30
|
def index
|
|
31
31
|
return @source.whole_index if @source
|
|
32
32
|
|
|
33
|
-
@index ||= @file_io.
|
|
33
|
+
@index ||= @file_io.materialize(raw_index[1])
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
# Raw rows plus their precomputed root-number sort keys (relaton#242
|
|
37
|
+
# stopgap). The lazy search path materializes pubid objects only for
|
|
38
|
+
# the bucket it narrows to, so a process that answers one lookup holds
|
|
39
|
+
# raw hashes instead of the whole pubid graph. A caller-seeded `@index`
|
|
40
|
+
# (a spec fixture, a preloaded pool entry) is the source of truth: the
|
|
41
|
+
# raw side derives from it, in its order, so a seeder that sorts by the
|
|
42
|
+
# narrowing key keeps the bsearch valid.
|
|
43
|
+
def raw_index
|
|
44
|
+
@raw_index ||= if @index
|
|
45
|
+
keys = @index.map { |r| narrowing_key(r[:id]) }
|
|
46
|
+
rows = @index.map { |r| { id: raw_row_id(r[:id]), file: r[:file] } }
|
|
47
|
+
[keys, rows]
|
|
48
|
+
else
|
|
49
|
+
@file_io.read_raw
|
|
50
|
+
end
|
|
34
51
|
end
|
|
35
52
|
|
|
36
53
|
#
|
|
@@ -59,14 +76,23 @@ module Relaton
|
|
|
59
76
|
# @return [void]
|
|
60
77
|
#
|
|
61
78
|
def add_or_update(id, file)
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
79
|
+
# A seeded or materialized @index (spec fixtures, preloaded pool
|
|
80
|
+
# entries) and pubid-less indexes run the legacy object path
|
|
81
|
+
# bit-for-bit: their consumers match rows by object identity
|
|
82
|
+
# semantics (`==`, `exclude`, to_s dedup) that raw rows cannot
|
|
83
|
+
# reproduce. Only a pubid-backed FileIO-loaded index — the
|
|
84
|
+
# relaton#242 memory case — adds through raw rows.
|
|
85
|
+
return legacy_add(id, file) if @index || !@file_io.pubid_class
|
|
86
|
+
|
|
87
|
+
raw_hash = raw_row_id(id)
|
|
88
|
+
keys, rows = raw_index
|
|
89
|
+
pos = raw_position(raw_hash)
|
|
90
|
+
if pos
|
|
91
|
+
rows[pos] = { id: raw_hash, file: file }
|
|
66
92
|
else
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
93
|
+
rows << { id: raw_hash, file: file }
|
|
94
|
+
keys << narrowing_key(id)
|
|
95
|
+
@raw_lookup[raw_hash] = rows.size - 1
|
|
70
96
|
@file_io.sorted = false
|
|
71
97
|
end
|
|
72
98
|
end
|
|
@@ -110,7 +136,12 @@ module Relaton
|
|
|
110
136
|
# @return [void]
|
|
111
137
|
#
|
|
112
138
|
def save
|
|
113
|
-
@
|
|
139
|
+
if @index
|
|
140
|
+
@file_io.save(@index)
|
|
141
|
+
else
|
|
142
|
+
keys, rows = raw_index
|
|
143
|
+
@file_io.save_raw(keys, rows)
|
|
144
|
+
end
|
|
114
145
|
end
|
|
115
146
|
|
|
116
147
|
#
|
|
@@ -125,6 +156,8 @@ module Relaton
|
|
|
125
156
|
@source = new_source
|
|
126
157
|
@index = nil
|
|
127
158
|
@id_lookup = nil
|
|
159
|
+
@raw_index = nil
|
|
160
|
+
@raw_lookup = nil
|
|
128
161
|
end
|
|
129
162
|
|
|
130
163
|
#
|
|
@@ -135,6 +168,8 @@ module Relaton
|
|
|
135
168
|
def remove_all
|
|
136
169
|
@index = []
|
|
137
170
|
@id_lookup = nil
|
|
171
|
+
@raw_index = [[], []]
|
|
172
|
+
@raw_lookup = {}
|
|
138
173
|
@file_io.sorted = true
|
|
139
174
|
end
|
|
140
175
|
|
|
@@ -146,6 +181,44 @@ module Relaton
|
|
|
146
181
|
end
|
|
147
182
|
end
|
|
148
183
|
|
|
184
|
+
# Position of a raw row by its id hash — the raw-side equivalent of
|
|
185
|
+
# `id_lookup`, built once, so a crawl's add_or_update stays O(1) per row.
|
|
186
|
+
def raw_position(raw_hash)
|
|
187
|
+
@raw_lookup ||= raw_index[1].each_with_object({}) do |row, h|
|
|
188
|
+
h[row[:id]] = h.key?(row[:id]) ? h[row[:id]] : rows_index_of(h, row)
|
|
189
|
+
end
|
|
190
|
+
@raw_lookup[raw_hash]
|
|
191
|
+
end
|
|
192
|
+
|
|
193
|
+
def rows_index_of(lookup, row)
|
|
194
|
+
raw_index[1].index { |r| r[:id].equal?(row[:id]) || r[:id] == row[:id] }
|
|
195
|
+
end
|
|
196
|
+
# The pre-#242 add path, kept verbatim for seeded/materialized and
|
|
197
|
+
# pubid-less indexes.
|
|
198
|
+
def legacy_add(id, file)
|
|
199
|
+
key = id.to_s
|
|
200
|
+
item = id_lookup[key]
|
|
201
|
+
if item
|
|
202
|
+
item[:file] = file
|
|
203
|
+
else
|
|
204
|
+
new_item = { id: id, file: file }
|
|
205
|
+
index << new_item
|
|
206
|
+
id_lookup[key] = new_item
|
|
207
|
+
@file_io.sorted = false
|
|
208
|
+
end
|
|
209
|
+
end
|
|
210
|
+
|
|
211
|
+
# Same key expression the FileIO sidecar computes at build time; a
|
|
212
|
+
# String id (a no-pubid_class index) has no `.root`, and never narrows.
|
|
213
|
+
def narrowing_key(id)
|
|
214
|
+
id.respond_to?(:root) && id.root.respond_to?(:number) ? id.root.number.to_s : ""
|
|
215
|
+
end
|
|
216
|
+
|
|
217
|
+
def raw_row_id(id)
|
|
218
|
+
pc = @file_io.pubid_class
|
|
219
|
+
pc && id.is_a?(pc) ? id.to_hash : id
|
|
220
|
+
end
|
|
221
|
+
|
|
149
222
|
def new_source
|
|
150
223
|
ShardSource.new(@dir, @pages_url, @pubid_class) if @pages_url
|
|
151
224
|
end
|
|
@@ -155,13 +228,15 @@ module Relaton
|
|
|
155
228
|
# never refreshed, so the shards stay the fresher answer.
|
|
156
229
|
return @source.rows(id) if @source && id && !id.is_a?(String)
|
|
157
230
|
|
|
158
|
-
#
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
231
|
+
# The lazy path narrows on the precomputed keys and materializes only
|
|
232
|
+
# the bucket; the materialized `index` stays available to callers
|
|
233
|
+
# that need the whole graph (`#index`, a String query, a block).
|
|
234
|
+
# Load first: a pubid-backed load is what settles @file_io.sorted.
|
|
235
|
+
if id && !id.is_a?(String) && @file_io.pubid_class
|
|
236
|
+
raw_index
|
|
237
|
+
return candidates_by_number(id) if @file_io.sorted
|
|
164
238
|
end
|
|
239
|
+
index
|
|
165
240
|
end
|
|
166
241
|
|
|
167
242
|
# Narrowing key: the base *document's* number as a string. `#root` walks a
|
|
@@ -169,26 +244,24 @@ module Relaton
|
|
|
169
244
|
# returns self for a base document), so a document and all its wrappers
|
|
170
245
|
# share one key and cluster together. `.to_s` because the key is compared
|
|
171
246
|
# as a string (a pubid number Component is not `<`/`>`-comparable), and
|
|
172
|
-
# FileIO
|
|
247
|
+
# FileIO computes this exact same key once at sidecar-build time so the
|
|
248
|
+
# bsearch over the keys array stays valid.
|
|
173
249
|
def candidates_by_number(id)
|
|
174
250
|
target = id.root.number.to_s
|
|
251
|
+
keys, rows = raw_index
|
|
175
252
|
left = bsearch_left(target)
|
|
176
253
|
return [] unless left
|
|
177
254
|
|
|
178
255
|
right = bsearch_right(target)
|
|
179
|
-
|
|
256
|
+
@file_io.materialize(rows[left...right])
|
|
180
257
|
end
|
|
181
258
|
|
|
182
259
|
def bsearch_left(target)
|
|
183
|
-
|
|
184
|
-
item[:id].root.number.to_s >= target
|
|
185
|
-
end
|
|
260
|
+
raw_index[0].bsearch_index { |k| k >= target }
|
|
186
261
|
end
|
|
187
262
|
|
|
188
263
|
def bsearch_right(target)
|
|
189
|
-
|
|
190
|
-
item[:id].root.number.to_s > target
|
|
191
|
-
end || index.size
|
|
264
|
+
raw_index[0].bsearch_index { |k| k > target } || raw_index[0].size
|
|
192
265
|
end
|
|
193
266
|
|
|
194
267
|
# The query is the reference, so two identifiers take the subset match.
|
data/lib/relaton/version.rb
CHANGED
metadata
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: relaton
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 3.0.0.pre.alpha.
|
|
4
|
+
version: 3.0.0.pre.alpha.7
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Ribose Inc.
|
|
8
8
|
autorequire:
|
|
9
9
|
bindir: bin
|
|
10
10
|
cert_chain: []
|
|
11
|
-
date: 2026-10-
|
|
11
|
+
date: 2026-10-03 00:00:00.000000000 Z
|
|
12
12
|
dependencies:
|
|
13
13
|
- !ruby/object:Gem::Dependency
|
|
14
14
|
name: addressable
|