relaton 3.0.0.pre.alpha.6 → 3.0.0.pre.alpha.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/lib/relaton/bib/model/ics.rb +38 -4
- data/lib/relaton/etsi/data_fetcher.rb +118 -9
- data/lib/relaton/iec/processor.rb +1 -1
- data/lib/relaton/index/config.rb +19 -1
- data/lib/relaton/index/file_io.rb +334 -0
- data/lib/relaton/index/sqlite_backend.rb +99 -0
- data/lib/relaton/index/type.rb +107 -23
- data/lib/relaton/index.rb +3 -1
- data/lib/relaton/version.rb +1 -1
- metadata +17 -2
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 3fafde9870593aca737727dcb4cf8cb142da74d255be4e855eed22bda4bd0f1b
|
|
4
|
+
data.tar.gz: 60986309859651e6e1baa05635a1ebfb2fec313f908a5508c14a59ded9ec3f8f
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 2b075475ad0321b604674b8459c60ddc1e7ef10a1a0186e70042348c0d9d5bbe58166cb419548f8d98438040702ffd532204716a261bedaedf20888a5d6c08cf
|
|
7
|
+
data.tar.gz: 8b2b003b3b99918b96a42c622607feae3087e2778730a2d31dc94ebe9b51abea23a3f6fbdb254fc72a2daeeae47877feb4edb998d70aa6d0d46c86881db2812c
|
|
@@ -29,8 +29,7 @@ module Relaton
|
|
|
29
29
|
return unless val.is_a?(String) && !val.empty?
|
|
30
30
|
return if @text.is_a?(String) && !@text.empty?
|
|
31
31
|
|
|
32
|
-
|
|
33
|
-
self.text = description if description
|
|
32
|
+
populate_text_from_isoics
|
|
34
33
|
end
|
|
35
34
|
|
|
36
35
|
# When the deserializer reaches the end of the XML element and
|
|
@@ -38,9 +37,44 @@ module Relaton
|
|
|
38
37
|
# to mark the attribute as default-valued (suppressing serialization).
|
|
39
38
|
# Refuse that mark if we've already populated text from Isoics so the
|
|
40
39
|
# value survives round-trip. See #112.
|
|
40
|
+
# lutaml-model 0.8.92 (lutaml/lutaml-model#922): assignments made
|
|
41
|
+
# inside `code=` while from_xml is still mid-element do not register
|
|
42
|
+
# as explicit values. Populate once deserialization has finished.
|
|
43
|
+
def self.from_xml(node)
|
|
44
|
+
ics = super
|
|
45
|
+
# lutaml-model 0.8.92 (#922): a value assigned through a custom
|
|
46
|
+
# writer's `super` mid-deserialization is left default-suppressed.
|
|
47
|
+
# Register code as explicitly set, then populate text.
|
|
48
|
+
ics.value_set_for(:code) if ics.code.is_a?(String) && !ics.code.empty?
|
|
49
|
+
ics.populate_text_from_isoics
|
|
50
|
+
ics
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
def populate_text_from_isoics
|
|
54
|
+
code.is_a?(String) && !code.empty? or return
|
|
55
|
+
@text.is_a?(String) && !@text.empty? and return
|
|
56
|
+
|
|
57
|
+
description = Isoics.fetch(code)&.description
|
|
58
|
+
self.text = description if description
|
|
59
|
+
end
|
|
60
|
+
|
|
41
61
|
def using_default_for(attribute_name)
|
|
42
|
-
|
|
43
|
-
|
|
62
|
+
# lutaml-model 0.8.92 (#922) leaves values assigned through a
|
|
63
|
+
# custom writer's `super` default-suppressed. code, when present,
|
|
64
|
+
# is always explicit; text absent from the XML is filled from
|
|
65
|
+
# Isoics here — the element end — and emitted.
|
|
66
|
+
if attribute_name == :code
|
|
67
|
+
return value_set_for(:code) if @code.is_a?(String) && !@code.empty?
|
|
68
|
+
|
|
69
|
+
return super
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
return if attribute_name == :text && @text.is_a?(String) && !@text.empty?
|
|
73
|
+
|
|
74
|
+
if attribute_name == :text
|
|
75
|
+
populate_text_from_isoics
|
|
76
|
+
return value_set_for(:text) unless @text.nil? || @text.empty?
|
|
77
|
+
end
|
|
44
78
|
|
|
45
79
|
super
|
|
46
80
|
end
|
|
@@ -45,22 +45,117 @@ module Relaton
|
|
|
45
45
|
report_errors
|
|
46
46
|
end
|
|
47
47
|
|
|
48
|
-
|
|
48
|
+
#
|
|
49
|
+
# Fetch pages 2..N. A page that stays bad after its retries is fetched
|
|
50
|
+
# again after the last page; if it is still bad, the crawl fails. A page
|
|
51
|
+
# is never skipped: relaton-data-etsi deletes data/ before the crawl and
|
|
52
|
+
# commits what it writes, so a skipped page would unpublish its documents.
|
|
53
|
+
#
|
|
54
|
+
# N comes from page 1's total_count, but the result set is sorted by
|
|
55
|
+
# deliverable number, so a document published during the crawl shifts the
|
|
56
|
+
# later pages by one. The crawl therefore reads on past N while the pages
|
|
57
|
+
# stay full (at most MAX_EXTRA_PAGES), and one page past a deferred last
|
|
58
|
+
# page; a record read twice overwrites its own file.
|
|
59
|
+
#
|
|
60
|
+
def fetch_remaining_pages(first_page) # rubocop:disable Metrics/MethodLength
|
|
61
|
+
total_pages = last = last_page(first_page)
|
|
62
|
+
deferred = []
|
|
63
|
+
page = 2
|
|
64
|
+
while page <= last
|
|
65
|
+
check_extra_pages page, total_pages
|
|
66
|
+
size = fetch_next_page(page, deferred)
|
|
67
|
+
break if size&.zero?
|
|
68
|
+
|
|
69
|
+
last = page + 1 if page == last && read_on?(size, page, total_pages)
|
|
70
|
+
page += 1
|
|
71
|
+
end
|
|
72
|
+
fetch_deferred_pages(deferred, total_pages)
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
def last_page(first_page)
|
|
49
76
|
total = first_page.first ? first_page.first["total_count"].to_i : 0
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
77
|
+
last = (total / PAGE_SIZE.to_f).ceil
|
|
78
|
+
first_page.size >= PAGE_SIZE ? [last, 2].max : last
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
#
|
|
82
|
+
# Whether to read the page after the last one: after a full page, or
|
|
83
|
+
# after a deferred page inside the total_count range. A deferred page
|
|
84
|
+
# past that range does not extend it, so a server that answers bad
|
|
85
|
+
# bodies past the end cannot keep the crawl going.
|
|
86
|
+
#
|
|
87
|
+
def read_on?(size, page, total_pages)
|
|
88
|
+
size ? size >= PAGE_SIZE : page <= total_pages
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
def check_extra_pages(page, total_pages)
|
|
92
|
+
return if page <= total_pages + MAX_EXTRA_PAGES
|
|
93
|
+
|
|
94
|
+
raise BadPage, "ETSI page #{page} is still full #{MAX_EXTRA_PAGES} " \
|
|
95
|
+
"pages past total_count (#{total_pages} pages)."
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
#
|
|
99
|
+
# @return [Integer, nil] the number of records on the page; nil for a
|
|
100
|
+
# deferred page
|
|
101
|
+
#
|
|
102
|
+
def fetch_next_page(page, deferred)
|
|
103
|
+
records = fetch_page(page)
|
|
104
|
+
process_records(records)
|
|
105
|
+
records.size
|
|
106
|
+
rescue BadPage => e
|
|
107
|
+
Util.warn "#{e.message} Fetching it again after the last page."
|
|
108
|
+
deferred << page
|
|
109
|
+
nil
|
|
110
|
+
end
|
|
54
111
|
|
|
55
|
-
|
|
112
|
+
def fetch_deferred_pages(pages, total_pages)
|
|
113
|
+
return if pages.empty?
|
|
114
|
+
|
|
115
|
+
sleep DEFERRED_DELAY
|
|
116
|
+
failed = pages.filter_map do |page|
|
|
117
|
+
refetch_page page, total_pages
|
|
118
|
+
rescue BadPage => e
|
|
119
|
+
e.message
|
|
56
120
|
end
|
|
121
|
+
raise BadPage, failed.join(" ") if failed.any?
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
# @return [String, nil] the failure message, nil on success
|
|
125
|
+
def refetch_page(page, total_pages)
|
|
126
|
+
records = fetch_page(page)
|
|
127
|
+
if records.empty? && page <= total_pages
|
|
128
|
+
return "ETSI page #{page} is empty on the re-fetch."
|
|
129
|
+
end
|
|
130
|
+
|
|
131
|
+
process_records records
|
|
132
|
+
nil
|
|
57
133
|
end
|
|
58
134
|
|
|
59
135
|
def fetch_page(page)
|
|
60
136
|
date = Time.now.to_date + 1
|
|
61
137
|
timestamp = (Time.now.to_f * 1000).to_i
|
|
62
138
|
url = format(SOURCEURL, page: page, date: date, timestamp: timestamp)
|
|
63
|
-
|
|
139
|
+
fetch_with_retry(url) { |body| parse_page(page, body) }
|
|
140
|
+
end
|
|
141
|
+
|
|
142
|
+
#
|
|
143
|
+
# ETSI's data.php sometimes answers with PHP's `Array` text in place of
|
|
144
|
+
# JSON (relaton-data-etsi crawl 36761804954); a re-fetch gets JSON.
|
|
145
|
+
#
|
|
146
|
+
# @raise [BadPage] when the body is not a JSON array
|
|
147
|
+
#
|
|
148
|
+
def parse_page(page, body)
|
|
149
|
+
records = JSON.parse(body)
|
|
150
|
+
return records if records.is_a?(Array)
|
|
151
|
+
|
|
152
|
+
raise BadPage, bad_page_message(page, body)
|
|
153
|
+
rescue JSON::ParserError
|
|
154
|
+
raise BadPage, bad_page_message(page, body)
|
|
155
|
+
end
|
|
156
|
+
|
|
157
|
+
def bad_page_message(page, body)
|
|
158
|
+
"ETSI page #{page} is not a JSON array: #{body.to_s[0, 200].inspect}."
|
|
64
159
|
end
|
|
65
160
|
|
|
66
161
|
def process_records(records)
|
|
@@ -92,16 +187,30 @@ module Relaton
|
|
|
92
187
|
"Published"
|
|
93
188
|
end
|
|
94
189
|
|
|
190
|
+
# A page body that is not a JSON array. See #parse_page.
|
|
191
|
+
class BadPage < StandardError; end
|
|
192
|
+
|
|
193
|
+
# Seconds to wait before a bad page is fetched again after the last page.
|
|
194
|
+
DEFERRED_DELAY = 60
|
|
195
|
+
|
|
196
|
+
# Full pages read past page 1's total_count before the crawl fails.
|
|
197
|
+
MAX_EXTRA_PAGES = 10
|
|
198
|
+
|
|
95
199
|
NETWORK_ERRORS = [
|
|
96
200
|
Mechanize::Error, Net::OpenTimeout, Net::ReadTimeout,
|
|
97
201
|
SocketError, Errno::ECONNRESET
|
|
98
202
|
].freeze
|
|
99
203
|
|
|
204
|
+
#
|
|
205
|
+
# @yield [String] the body; the block's result is returned, and a BadPage
|
|
206
|
+
# it raises is retried like a network error
|
|
207
|
+
#
|
|
100
208
|
def fetch_with_retry(url, retries: 3, delay: 2) # rubocop:disable Metrics/MethodLength
|
|
101
209
|
attempt = 0
|
|
102
210
|
begin
|
|
103
|
-
Mechanize.new.get(url).body
|
|
104
|
-
|
|
211
|
+
body = Mechanize.new.get(url).body
|
|
212
|
+
block_given? ? yield(body) : body
|
|
213
|
+
rescue *NETWORK_ERRORS, BadPage => e
|
|
105
214
|
attempt += 1
|
|
106
215
|
if attempt <= retries
|
|
107
216
|
Util.info "Fetch failed (#{e.message}), " \
|
|
@@ -7,7 +7,7 @@ module Relaton
|
|
|
7
7
|
@short = :relaton_iec
|
|
8
8
|
@prefix = "IEC"
|
|
9
9
|
@pubid_flavor = :Iec # global prefixes sourced from Pubid::Iec.prefixes
|
|
10
|
-
@defaultprefix = %r{^(IEC\s|CISPR\s|IEV($|\s))}
|
|
10
|
+
@defaultprefix = %r{^(IEC\s|CISPR\s|IEV($|\s)|CEI\s)} # CEI: IEC's French spelling (relaton#243)
|
|
11
11
|
@idtype = "IEC"
|
|
12
12
|
@datasets = %w[iec-harmonized-all iec-harmonized-latest]
|
|
13
13
|
end
|
data/lib/relaton/index/config.rb
CHANGED
|
@@ -4,7 +4,8 @@ module Relaton
|
|
|
4
4
|
# Configuration class for Relaton::Index
|
|
5
5
|
#
|
|
6
6
|
class Config
|
|
7
|
-
attr_reader :storage, :storage_dir, :filename
|
|
7
|
+
attr_reader :storage, :storage_dir, :filename, :build_sidecar_in_child,
|
|
8
|
+
:sqlite_index
|
|
8
9
|
|
|
9
10
|
#
|
|
10
11
|
# Set default values
|
|
@@ -13,6 +14,23 @@ module Relaton
|
|
|
13
14
|
@storage = FileStorage
|
|
14
15
|
@storage_dir = Dir.home
|
|
15
16
|
@filename = "index.yaml"
|
|
17
|
+
@build_sidecar_in_child = true
|
|
18
|
+
@sqlite_index = true
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
# Build the sidecar in a forked child when available, so the one-time
|
|
22
|
+
# full materialization's memory dies with the child (relaton#242).
|
|
23
|
+
# Set to false to force an in-process build.
|
|
24
|
+
def build_sidecar_in_child=(flag)
|
|
25
|
+
@build_sidecar_in_child = flag
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
# Materialize a downloaded index into SQLite so a search answers from
|
|
29
|
+
# a bucket query and the process never holds the whole index
|
|
30
|
+
# (relaton#242 phase 2). Set to false to keep the raw-row sidecar path
|
|
31
|
+
# only.
|
|
32
|
+
def sqlite_index=(flag)
|
|
33
|
+
@sqlite_index = flag
|
|
16
34
|
end
|
|
17
35
|
|
|
18
36
|
#
|
|
@@ -11,6 +11,10 @@ module Relaton
|
|
|
11
11
|
# wrong-structure handling (re-download, or stop and log).
|
|
12
12
|
class InvalidIndexError < StandardError; end
|
|
13
13
|
|
|
14
|
+
# Bump when the sidecar payload shape changes, so an older sidecar is
|
|
15
|
+
# discarded and rebuilt instead of misread.
|
|
16
|
+
SIDECAR_VERSION = 2 # v2: the payload carries the yaml byte-size for freshness
|
|
17
|
+
|
|
14
18
|
attr_reader :url, :pubid_class
|
|
15
19
|
attr_accessor :sorted
|
|
16
20
|
|
|
@@ -271,6 +275,8 @@ module Relaton
|
|
|
271
275
|
end
|
|
272
276
|
end.to_yaml
|
|
273
277
|
Index.config.storage.write file, yaml
|
|
278
|
+
delete_sidecar
|
|
279
|
+
delete_sqlite
|
|
274
280
|
end
|
|
275
281
|
|
|
276
282
|
def sort_structured_index(index)
|
|
@@ -288,11 +294,339 @@ module Relaton
|
|
|
288
294
|
#
|
|
289
295
|
def remove
|
|
290
296
|
Index.config.storage.remove file
|
|
297
|
+
delete_sidecar
|
|
298
|
+
delete_sqlite
|
|
291
299
|
[]
|
|
292
300
|
end
|
|
293
301
|
|
|
302
|
+
#
|
|
303
|
+
# Raw-row read for the lazy search path (relaton#242 stopgap): returns
|
|
304
|
+
# precomputed root-number sort keys and the rows as plain hashes — no
|
|
305
|
+
# pubid objects. A Marshal sidecar next to the yaml holds them so
|
|
306
|
+
# repeat loads skip the YAML parse and the one-time full
|
|
307
|
+
# materialization; the sidecar is rebuilt whenever the yaml is newer.
|
|
308
|
+
#
|
|
309
|
+
# @return [Array<Array<String>, Array<Hash>] sort keys and raw rows
|
|
310
|
+
#
|
|
311
|
+
def read_raw
|
|
312
|
+
case url
|
|
313
|
+
when String
|
|
314
|
+
with_file_lock do
|
|
315
|
+
check_file ? read_raw_file : fetch_raw_and_save
|
|
316
|
+
end
|
|
317
|
+
else
|
|
318
|
+
read_raw_file || [[], []]
|
|
319
|
+
end
|
|
320
|
+
end
|
|
321
|
+
|
|
322
|
+
# ── SQLite backend (relaton#242 phase 2) ──
|
|
323
|
+
#
|
|
324
|
+
# The downloaded index is materialized once into a SQLite database
|
|
325
|
+
# keyed by the narrowing number. A search answers from a bucket query,
|
|
326
|
+
# so the process holds O(bucket) rows instead of the whole index. The
|
|
327
|
+
# build is pure hash transforms (no pubid objects), run in a forked
|
|
328
|
+
# child so even the one-time YAML parse never lands in this process.
|
|
329
|
+
|
|
330
|
+
def sqlite_ready?
|
|
331
|
+
url.is_a?(String) && Index.config.sqlite_index != false &&
|
|
332
|
+
File.file?(sqlite_db_path) && sqlite_fresh?
|
|
333
|
+
end
|
|
334
|
+
|
|
335
|
+
# Ensure the db exists and is current (24 h TTL, schema-versioned).
|
|
336
|
+
def ensure_sqlite
|
|
337
|
+
return false unless url.is_a?(String)
|
|
338
|
+
return true if sqlite_ready?
|
|
339
|
+
|
|
340
|
+
with_file_lock do
|
|
341
|
+
build_sqlite unless sqlite_ready?
|
|
342
|
+
end
|
|
343
|
+
sqlite_ready?
|
|
344
|
+
end
|
|
345
|
+
|
|
346
|
+
def sqlite_bucket(number)
|
|
347
|
+
sqlite_backend.bucket(number)
|
|
348
|
+
end
|
|
349
|
+
|
|
350
|
+
def sqlite_count
|
|
351
|
+
sqlite_backend.count
|
|
352
|
+
end
|
|
353
|
+
|
|
354
|
+
def close_sqlite
|
|
355
|
+
@sqlite_backend&.close
|
|
356
|
+
@sqlite_backend = nil
|
|
357
|
+
end
|
|
358
|
+
|
|
359
|
+
def delete_sqlite
|
|
360
|
+
close_sqlite
|
|
361
|
+
File.delete(sqlite_db_path) if File.file?(sqlite_db_path)
|
|
362
|
+
rescue Errno::EACCES
|
|
363
|
+
nil
|
|
364
|
+
end
|
|
365
|
+
|
|
366
|
+
def sqlite_db_path
|
|
367
|
+
"#{path_to_local_file}.db"
|
|
368
|
+
end
|
|
369
|
+
|
|
370
|
+
private
|
|
371
|
+
|
|
372
|
+
def sqlite_backend
|
|
373
|
+
@sqlite_backend ||= SqliteBackend.new(sqlite_db_path, pubid_class: @pubid_class)
|
|
374
|
+
end
|
|
375
|
+
|
|
376
|
+
def sqlite_fresh?
|
|
377
|
+
ctime = Index.config.storage.ctime(sqlite_db_path)
|
|
378
|
+
backend = sqlite_backend
|
|
379
|
+
ctime && ctime > Time.now - 86400 && !backend.stale_schema?
|
|
380
|
+
rescue SQLite3::SQLException
|
|
381
|
+
false
|
|
382
|
+
ensure
|
|
383
|
+
backend&.close
|
|
384
|
+
@sqlite_backend = nil
|
|
385
|
+
end
|
|
386
|
+
|
|
387
|
+
# The build downloads the published zip, converts its rows to
|
|
388
|
+
# `[number, sort_key, id_json, file]` tuples by hash transforms alone,
|
|
389
|
+
# and writes the db — in a forked child where the YAML parse's memory
|
|
390
|
+
# dies with the process.
|
|
391
|
+
def build_sqlite
|
|
392
|
+
# The whole build — download, YAML parse, db write — runs in a
|
|
393
|
+
# forked child where available: the parse's ~300 MB (per flavor,
|
|
394
|
+
# larger for IETF) dies with the child, and the serving process
|
|
395
|
+
# never allocates it.
|
|
396
|
+
if Process.respond_to?(:fork) && !ENV["RELATON_NO_FORK"]
|
|
397
|
+
pid = Process.fork do
|
|
398
|
+
tuples = download_tuples
|
|
399
|
+
write_sqlite(tuples) if tuples
|
|
400
|
+
exit!(0)
|
|
401
|
+
end
|
|
402
|
+
Process.wait(pid)
|
|
403
|
+
true
|
|
404
|
+
else
|
|
405
|
+
tuples = download_tuples
|
|
406
|
+
return false unless tuples
|
|
407
|
+
|
|
408
|
+
write_sqlite(tuples)
|
|
409
|
+
end
|
|
410
|
+
end
|
|
411
|
+
|
|
412
|
+
def write_sqlite(tuples)
|
|
413
|
+
File.dirname(sqlite_db_path).then { |d| require "fileutils"; FileUtils.mkdir_p(d) }
|
|
414
|
+
backend = SqliteBackend.new(sqlite_db_path, pubid_class: @pubid_class)
|
|
415
|
+
backend.build(tuples.each)
|
|
416
|
+
backend.close
|
|
417
|
+
end
|
|
418
|
+
|
|
419
|
+
# Enumerate `[number, sort_key, id_hash, file]` from the published zip.
|
|
420
|
+
# The root number — the base document's, walking `base` nesting — is
|
|
421
|
+
# computed on the raw hash; no pubid object is ever built.
|
|
422
|
+
def download_tuples
|
|
423
|
+
body = Net::HTTP.get(URI.parse(url))
|
|
424
|
+
# rubyzip may mutate the buffer it is handed; a StringIO over a copy
|
|
425
|
+
# keeps the response body intact for any later reader of it.
|
|
426
|
+
yaml = nil
|
|
427
|
+
Zip::File.open_buffer(StringIO.new(body.dup)) do |zip|
|
|
428
|
+
stream = zip.entries.first.get_input_stream
|
|
429
|
+
yaml = stream.read
|
|
430
|
+
end
|
|
431
|
+
yaml = yaml.read unless yaml.is_a?(String)
|
|
432
|
+
Util.info "Downloaded index from `#{url}` for sqlite materialization", progname
|
|
433
|
+
raw = YAML.safe_load(yaml, permitted_classes: [Symbol])
|
|
434
|
+
return nil unless check_format(raw)
|
|
435
|
+
|
|
436
|
+
Enumerator.new do |y|
|
|
437
|
+
raw.each do |r|
|
|
438
|
+
number = raw_root_number(r[:id]).to_s
|
|
439
|
+
y << [number, number, r[:id], r[:file]]
|
|
440
|
+
end
|
|
441
|
+
end
|
|
442
|
+
rescue Psych::SyntaxError, SocketError, OpenURI::HTTPError, Errno::ECONNRESET,
|
|
443
|
+
OpenSSL::SSL::SSLError => e
|
|
444
|
+
Util.info "SQLite index build failed (#{e.message}); falling back", progname
|
|
445
|
+
nil
|
|
446
|
+
end
|
|
447
|
+
|
|
448
|
+
# `#root` as a hash walk: a supplement nests its origin under `base`,
|
|
449
|
+
# and the flattened to_hash puts the supplement's own components first.
|
|
450
|
+
# Accepts string and symbol keys — a YAML round-trip with permitted
|
|
451
|
+
# Symbol may give either.
|
|
452
|
+
def raw_root_number(id_hash)
|
|
453
|
+
return nil unless id_hash.is_a?(Hash)
|
|
454
|
+
|
|
455
|
+
base = id_hash["base"] || id_hash[:base]
|
|
456
|
+
return raw_root_number(base) if base
|
|
457
|
+
|
|
458
|
+
id_hash["number"] || id_hash[:number]
|
|
459
|
+
end
|
|
460
|
+
|
|
461
|
+
public
|
|
462
|
+
|
|
463
|
+
# Deserialize raw rows into pubid rows. Only the caller knows which
|
|
464
|
+
# slice it needs, so materialization happens here rather than at load.
|
|
465
|
+
#
|
|
466
|
+
# @param [Array<Hash>] rows raw rows
|
|
467
|
+
# @return [Array<Hash>] rows with deserialized ids
|
|
468
|
+
def materialize(rows)
|
|
469
|
+
return rows unless @pubid_class
|
|
470
|
+
|
|
471
|
+
rows.map { |r| { id: deserialize_id(r[:id]), file: r[:file] } }
|
|
472
|
+
end
|
|
473
|
+
|
|
474
|
+
# Save raw rows (possibly alongside the keys they were loaded with);
|
|
475
|
+
# sorts by the same root-number key the object path sorts by.
|
|
476
|
+
#
|
|
477
|
+
# @param [Array<String>] keys sort keys, parallel to rows
|
|
478
|
+
# @param [Array<Hash>] rows raw rows
|
|
479
|
+
# @return [void]
|
|
480
|
+
def save_raw(keys, rows)
|
|
481
|
+
ordered = @pubid_class ? keys.zip(rows).sort_by { |k, _| k.to_s }
|
|
482
|
+
.map { |_, r| r } : rows
|
|
483
|
+
yaml = ordered.map do |item|
|
|
484
|
+
{ id: item[:id], file: item[:file] }
|
|
485
|
+
end.to_yaml
|
|
486
|
+
Index.config.storage.write file, yaml
|
|
487
|
+
delete_sidecar
|
|
488
|
+
end
|
|
489
|
+
|
|
294
490
|
private
|
|
295
491
|
|
|
492
|
+
def sidecar_file
|
|
493
|
+
"#{file}.ms"
|
|
494
|
+
end
|
|
495
|
+
|
|
496
|
+
def delete_sidecar
|
|
497
|
+
File.delete(sidecar_file) if File.file?(sidecar_file)
|
|
498
|
+
rescue Errno::EACCES
|
|
499
|
+
nil
|
|
500
|
+
end
|
|
501
|
+
|
|
502
|
+
def read_raw_file
|
|
503
|
+
# The sidecar is authoritative while it describes the yaml byte-for-byte
|
|
504
|
+
# (same size) — the yaml is not even parsed until the sidecar is stale
|
|
505
|
+
# or unreadable.
|
|
506
|
+
if sidecar_fresh?
|
|
507
|
+
loaded = sidecar
|
|
508
|
+
return loaded if loaded
|
|
509
|
+
end
|
|
510
|
+
|
|
511
|
+
yaml = Index.config.storage.read(file)
|
|
512
|
+
return unless yaml
|
|
513
|
+
|
|
514
|
+
begin
|
|
515
|
+
raw = YAML.safe_load(yaml, permitted_classes: [Symbol])
|
|
516
|
+
rescue Psych::SyntaxError
|
|
517
|
+
warn_local_index_error("YAML parsing error when reading")
|
|
518
|
+
delete_sidecar
|
|
519
|
+
return [[], []]
|
|
520
|
+
end
|
|
521
|
+
build_raw(raw)
|
|
522
|
+
end
|
|
523
|
+
|
|
524
|
+
# Size, not mtime: NTFS timestamp coarseness makes a rewritten yaml
|
|
525
|
+
# carry the sidecar's own mtime, so "yaml is newer" misses the rebuild.
|
|
526
|
+
def sidecar_fresh?
|
|
527
|
+
File.file?(sidecar_file) && File.file?(file) &&
|
|
528
|
+
File.size(file) == sidecar_yaml_size
|
|
529
|
+
end
|
|
530
|
+
|
|
531
|
+
def sidecar_yaml_size
|
|
532
|
+
Marshal.load(File.binread(sidecar_file))[4]
|
|
533
|
+
rescue TypeError, ArgumentError, EOFError
|
|
534
|
+
nil
|
|
535
|
+
end
|
|
536
|
+
|
|
537
|
+
def sidecar
|
|
538
|
+
version, sorted, keys, rows = Marshal.load(File.binread(sidecar_file))
|
|
539
|
+
return unless version == SIDECAR_VERSION
|
|
540
|
+
|
|
541
|
+
@sorted = sorted
|
|
542
|
+
[keys, rows]
|
|
543
|
+
rescue TypeError, ArgumentError, EOFError
|
|
544
|
+
nil
|
|
545
|
+
end
|
|
546
|
+
|
|
547
|
+
def build_raw(raw)
|
|
548
|
+
unless check_format(raw)
|
|
549
|
+
warn_local_index_error("Wrong structure of")
|
|
550
|
+
return [[], []]
|
|
551
|
+
end
|
|
552
|
+
|
|
553
|
+
return build_raw_in_child(raw) if build_sidecar_in_child?
|
|
554
|
+
|
|
555
|
+
build_raw_in_process(raw)
|
|
556
|
+
rescue InvalidIndexError
|
|
557
|
+
warn_local_index_error("Wrong structure of")
|
|
558
|
+
[[], []]
|
|
559
|
+
end
|
|
560
|
+
|
|
561
|
+
# The one-time key build materializes every row (~675 MB on the
|
|
562
|
+
# 79,993-row ISO index). Freed pages are often not returned to the OS,
|
|
563
|
+
# so an in-process build leaves the parent's RSS high — and containers
|
|
564
|
+
# OOM on RSS (relaton#242). Where fork exists, build the sidecar in a
|
|
565
|
+
# child: the graph dies with it and the parent never spikes.
|
|
566
|
+
def build_sidecar_in_child?
|
|
567
|
+
Process.respond_to?(:fork) && Index.config.build_sidecar_in_child != false
|
|
568
|
+
end
|
|
569
|
+
|
|
570
|
+
def build_raw_in_child(raw)
|
|
571
|
+
pid = Process.fork do
|
|
572
|
+
build_raw_in_process(raw)
|
|
573
|
+
exit!(0)
|
|
574
|
+
end
|
|
575
|
+
Process.wait(pid)
|
|
576
|
+
loaded = sidecar
|
|
577
|
+
return build_raw_in_process(raw) unless loaded # child failed
|
|
578
|
+
|
|
579
|
+
loaded
|
|
580
|
+
rescue Errno::ENOMEM, SystemCallError
|
|
581
|
+
build_raw_in_process(raw)
|
|
582
|
+
end
|
|
583
|
+
|
|
584
|
+
def build_raw_in_process(raw)
|
|
585
|
+
objects = deserialize_pubid(raw)
|
|
586
|
+
keys = objects.map { |r| r[:id].root.number.to_s }
|
|
587
|
+
rows = objects.map { |r| { id: raw_id(r), file: r[:file] } }
|
|
588
|
+
write_sidecar(keys, rows)
|
|
589
|
+
[keys, rows]
|
|
590
|
+
rescue InvalidIndexError
|
|
591
|
+
warn_local_index_error("Wrong structure of")
|
|
592
|
+
[[], []]
|
|
593
|
+
end
|
|
594
|
+
|
|
595
|
+
def raw_id(row)
|
|
596
|
+
id = row[:id]
|
|
597
|
+
id.respond_to?(:to_hash) ? id.to_hash : id
|
|
598
|
+
end
|
|
599
|
+
|
|
600
|
+
def write_sidecar(keys, rows)
|
|
601
|
+
File.binwrite(sidecar_file,
|
|
602
|
+
Marshal.dump([SIDECAR_VERSION, @sorted, keys, rows,
|
|
603
|
+
File.size(file)]))
|
|
604
|
+
rescue Errno::EACCES, Errno::ENOENT, Errno::EROFS
|
|
605
|
+
nil # the sidecar is an optimization; a read-only dir still works
|
|
606
|
+
end
|
|
607
|
+
|
|
608
|
+
def fetch_raw_and_save
|
|
609
|
+
uri = URI.parse(url)
|
|
610
|
+
body = Net::HTTP.get(uri)
|
|
611
|
+
yaml = nil
|
|
612
|
+
Zip::File.open_buffer(body) do |zip|
|
|
613
|
+
yaml = zip.entries.first.get_input_stream.read
|
|
614
|
+
end
|
|
615
|
+
Util.info "Downloaded index from `#{url}`", progname
|
|
616
|
+
raw = YAML.safe_load(yaml, permitted_classes: [Symbol])
|
|
617
|
+
if check_format(raw)
|
|
618
|
+
save raw
|
|
619
|
+
raw = nil # release the parsed copy; read_raw_file loads the sidecar's
|
|
620
|
+
read_raw_file
|
|
621
|
+
else
|
|
622
|
+
warn_remote_index_error "Wrong structure of"
|
|
623
|
+
[[], []]
|
|
624
|
+
end
|
|
625
|
+
rescue Psych::SyntaxError
|
|
626
|
+
warn_remote_index_error "YAML parsing error when reading"
|
|
627
|
+
[[], []]
|
|
628
|
+
end
|
|
629
|
+
|
|
296
630
|
def with_file_lock(&)
|
|
297
631
|
@@file_locks_mutex.synchronize do
|
|
298
632
|
@@file_locks[file] ||= Mutex.new
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "sqlite3"
|
|
4
|
+
|
|
5
|
+
module Relaton
|
|
6
|
+
module Index
|
|
7
|
+
#
|
|
8
|
+
# The SQLite materialization of one flavor index (relaton#242 phase 2).
|
|
9
|
+
#
|
|
10
|
+
# A downloaded index zip is built once into a local database whose rows
|
|
11
|
+
# are `(number, sort_key, id_json, file)` with an index on `number` — the
|
|
12
|
+
# narrowing key. A search materializes only its bucket; the whole index
|
|
13
|
+
# is never resident. WAL journaling lets one build and concurrent readers
|
|
14
|
+
# share the file across processes.
|
|
15
|
+
#
|
|
16
|
+
class SqliteBackend
|
|
17
|
+
SCHEMA_VERSION = "1"
|
|
18
|
+
|
|
19
|
+
def initialize(db_path, pubid_class: nil)
|
|
20
|
+
@db_path = db_path
|
|
21
|
+
@pubid_class = pubid_class
|
|
22
|
+
@db = SQLite3::Database.new(db_path)
|
|
23
|
+
@db.busy_timeout = 30_000
|
|
24
|
+
@db.results_as_hash = false
|
|
25
|
+
@db.execute("PRAGMA journal_mode=WAL")
|
|
26
|
+
@db.execute("PRAGMA synchronous=NORMAL")
|
|
27
|
+
create_schema
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
# Build from an enumerator of `[number, sort_key, id_hash, file]`
|
|
31
|
+
# tuples. Callers hand the child process's enumerator here; the tuples
|
|
32
|
+
# are written inside one transaction.
|
|
33
|
+
def build(rows_enum)
|
|
34
|
+
@db.transaction do
|
|
35
|
+
@db.execute("DELETE FROM index_rows")
|
|
36
|
+
@db.prepare("INSERT INTO index_rows (number, sort_key, id_json, file) VALUES (?, ?, ?, ?)") do |stmt|
|
|
37
|
+
rows_enum.each { |(number, sort_key, id_hash, file)| stmt.execute(number, sort_key, JSON.generate(id_hash), file) }
|
|
38
|
+
end
|
|
39
|
+
meta_set("schema_version", SCHEMA_VERSION)
|
|
40
|
+
end
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
# The bucket for a narrowing key: raw rows, ordered as written.
|
|
44
|
+
def bucket(number)
|
|
45
|
+
rows = []
|
|
46
|
+
@db.execute("SELECT id_json, file FROM index_rows WHERE number = ? ORDER BY rowid", [number]) do |row|
|
|
47
|
+
rows << { id: JSON.parse(row[0]), file: row[1] }
|
|
48
|
+
end
|
|
49
|
+
rows
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
def each_row
|
|
53
|
+
@db.execute("SELECT id_json, file FROM index_rows ORDER BY rowid") do |row|
|
|
54
|
+
yield({ id: JSON.parse(row[0]), file: row[1] })
|
|
55
|
+
end
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
def count
|
|
59
|
+
@db.get_first_value("SELECT COUNT(*) FROM index_rows").to_i
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
def stale_schema?
|
|
63
|
+
meta_get("schema_version") != SCHEMA_VERSION
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
def close
|
|
67
|
+
@db.close
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
private
|
|
71
|
+
|
|
72
|
+
def create_schema
|
|
73
|
+
@db.execute(<<~SQL)
|
|
74
|
+
CREATE TABLE IF NOT EXISTS index_rows (
|
|
75
|
+
number TEXT NOT NULL,
|
|
76
|
+
sort_key TEXT NOT NULL,
|
|
77
|
+
id_json TEXT NOT NULL,
|
|
78
|
+
file TEXT NOT NULL
|
|
79
|
+
)
|
|
80
|
+
SQL
|
|
81
|
+
@db.execute("CREATE INDEX IF NOT EXISTS idx_index_rows_number ON index_rows (number)")
|
|
82
|
+
@db.execute(<<~SQL)
|
|
83
|
+
CREATE TABLE IF NOT EXISTS index_meta (
|
|
84
|
+
key TEXT PRIMARY KEY,
|
|
85
|
+
value TEXT NOT NULL
|
|
86
|
+
)
|
|
87
|
+
SQL
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
def meta_set(key, value)
|
|
91
|
+
@db.execute("INSERT OR REPLACE INTO index_meta (key, value) VALUES (?, ?)", [key, value])
|
|
92
|
+
end
|
|
93
|
+
|
|
94
|
+
def meta_get(key)
|
|
95
|
+
@db.get_first_value("SELECT value FROM index_meta WHERE key = ?", [key])
|
|
96
|
+
end
|
|
97
|
+
end
|
|
98
|
+
end
|
|
99
|
+
end
|
data/lib/relaton/index/type.rb
CHANGED
|
@@ -30,7 +30,24 @@ module Relaton
|
|
|
30
30
|
def index
|
|
31
31
|
return @source.whole_index if @source
|
|
32
32
|
|
|
33
|
-
@index ||= @file_io.
|
|
33
|
+
@index ||= @file_io.materialize(raw_index[1])
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
# Raw rows plus their precomputed root-number sort keys (relaton#242
|
|
37
|
+
# stopgap). The lazy search path materializes pubid objects only for
|
|
38
|
+
# the bucket it narrows to, so a process that answers one lookup holds
|
|
39
|
+
# raw hashes instead of the whole pubid graph. A caller-seeded `@index`
|
|
40
|
+
# (a spec fixture, a preloaded pool entry) is the source of truth: the
|
|
41
|
+
# raw side derives from it, in its order, so a seeder that sorts by the
|
|
42
|
+
# narrowing key keeps the bsearch valid.
|
|
43
|
+
def raw_index
|
|
44
|
+
@raw_index ||= if @index
|
|
45
|
+
keys = @index.map { |r| narrowing_key(r[:id]) }
|
|
46
|
+
rows = @index.map { |r| { id: raw_row_id(r[:id]), file: r[:file] } }
|
|
47
|
+
[keys, rows]
|
|
48
|
+
else
|
|
49
|
+
@file_io.read_raw
|
|
50
|
+
end
|
|
34
51
|
end
|
|
35
52
|
|
|
36
53
|
#
|
|
@@ -59,14 +76,23 @@ module Relaton
|
|
|
59
76
|
# @return [void]
|
|
60
77
|
#
|
|
61
78
|
def add_or_update(id, file)
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
79
|
+
# A seeded or materialized @index (spec fixtures, preloaded pool
|
|
80
|
+
# entries) and pubid-less indexes run the legacy object path
|
|
81
|
+
# bit-for-bit: their consumers match rows by object identity
|
|
82
|
+
# semantics (`==`, `exclude`, to_s dedup) that raw rows cannot
|
|
83
|
+
# reproduce. Only a pubid-backed FileIO-loaded index — the
|
|
84
|
+
# relaton#242 memory case — adds through raw rows.
|
|
85
|
+
return legacy_add(id, file) if @index || !@file_io.pubid_class
|
|
86
|
+
|
|
87
|
+
raw_hash = raw_row_id(id)
|
|
88
|
+
keys, rows = raw_index
|
|
89
|
+
pos = raw_position(raw_hash)
|
|
90
|
+
if pos
|
|
91
|
+
rows[pos] = { id: raw_hash, file: file }
|
|
66
92
|
else
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
93
|
+
rows << { id: raw_hash, file: file }
|
|
94
|
+
keys << narrowing_key(id)
|
|
95
|
+
@raw_lookup[raw_hash] = rows.size - 1
|
|
70
96
|
@file_io.sorted = false
|
|
71
97
|
end
|
|
72
98
|
end
|
|
@@ -110,7 +136,12 @@ module Relaton
|
|
|
110
136
|
# @return [void]
|
|
111
137
|
#
|
|
112
138
|
def save
|
|
113
|
-
@
|
|
139
|
+
if @index
|
|
140
|
+
@file_io.save(@index)
|
|
141
|
+
else
|
|
142
|
+
keys, rows = raw_index
|
|
143
|
+
@file_io.save_raw(keys, rows)
|
|
144
|
+
end
|
|
114
145
|
end
|
|
115
146
|
|
|
116
147
|
#
|
|
@@ -125,6 +156,8 @@ module Relaton
|
|
|
125
156
|
@source = new_source
|
|
126
157
|
@index = nil
|
|
127
158
|
@id_lookup = nil
|
|
159
|
+
@raw_index = nil
|
|
160
|
+
@raw_lookup = nil
|
|
128
161
|
end
|
|
129
162
|
|
|
130
163
|
#
|
|
@@ -135,6 +168,8 @@ module Relaton
|
|
|
135
168
|
def remove_all
|
|
136
169
|
@index = []
|
|
137
170
|
@id_lookup = nil
|
|
171
|
+
@raw_index = [[], []]
|
|
172
|
+
@raw_lookup = {}
|
|
138
173
|
@file_io.sorted = true
|
|
139
174
|
end
|
|
140
175
|
|
|
@@ -146,6 +181,44 @@ module Relaton
|
|
|
146
181
|
end
|
|
147
182
|
end
|
|
148
183
|
|
|
184
|
+
# Position of a raw row by its id hash — the raw-side equivalent of
|
|
185
|
+
# `id_lookup`, built once, so a crawl's add_or_update stays O(1) per row.
|
|
186
|
+
def raw_position(raw_hash)
|
|
187
|
+
@raw_lookup ||= raw_index[1].each_with_object({}) do |row, h|
|
|
188
|
+
h[row[:id]] = h.key?(row[:id]) ? h[row[:id]] : rows_index_of(h, row)
|
|
189
|
+
end
|
|
190
|
+
@raw_lookup[raw_hash]
|
|
191
|
+
end
|
|
192
|
+
|
|
193
|
+
def rows_index_of(lookup, row)
|
|
194
|
+
raw_index[1].index { |r| r[:id].equal?(row[:id]) || r[:id] == row[:id] }
|
|
195
|
+
end
|
|
196
|
+
# The pre-#242 add path, kept verbatim for seeded/materialized and
|
|
197
|
+
# pubid-less indexes.
|
|
198
|
+
def legacy_add(id, file)
|
|
199
|
+
key = id.to_s
|
|
200
|
+
item = id_lookup[key]
|
|
201
|
+
if item
|
|
202
|
+
item[:file] = file
|
|
203
|
+
else
|
|
204
|
+
new_item = { id: id, file: file }
|
|
205
|
+
index << new_item
|
|
206
|
+
id_lookup[key] = new_item
|
|
207
|
+
@file_io.sorted = false
|
|
208
|
+
end
|
|
209
|
+
end
|
|
210
|
+
|
|
211
|
+
# Same key expression the FileIO sidecar computes at build time; a
|
|
212
|
+
# String id (a no-pubid_class index) has no `.root`, and never narrows.
|
|
213
|
+
def narrowing_key(id)
|
|
214
|
+
id.respond_to?(:root) && id.root.respond_to?(:number) ? id.root.number.to_s : ""
|
|
215
|
+
end
|
|
216
|
+
|
|
217
|
+
def raw_row_id(id)
|
|
218
|
+
pc = @file_io.pubid_class
|
|
219
|
+
pc && id.is_a?(pc) ? id.to_hash : id
|
|
220
|
+
end
|
|
221
|
+
|
|
149
222
|
def new_source
|
|
150
223
|
ShardSource.new(@dir, @pages_url, @pubid_class) if @pages_url
|
|
151
224
|
end
|
|
@@ -155,13 +228,26 @@ module Relaton
|
|
|
155
228
|
# never refreshed, so the shards stay the fresher answer.
|
|
156
229
|
return @source.rows(id) if @source && id && !id.is_a?(String)
|
|
157
230
|
|
|
158
|
-
#
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
231
|
+
# The SQLite backend answers a parsed query with a bucket query:
|
|
232
|
+
# O(bucket) rows ever resident, no sidecar load at all. Falls back
|
|
233
|
+
# to the raw-row path when the backend cannot build or is disabled.
|
|
234
|
+
if id && !id.is_a?(String) && @file_io.pubid_class
|
|
235
|
+
begin
|
|
236
|
+
return sqlite_candidates(id) if @file_io.ensure_sqlite
|
|
237
|
+
rescue SQLite3::SQLException
|
|
238
|
+
nil # fall back to the sidecar path below
|
|
239
|
+
end
|
|
240
|
+
|
|
241
|
+
raw_index
|
|
242
|
+
return candidates_by_number(id) if @file_io.sorted
|
|
164
243
|
end
|
|
244
|
+
index
|
|
245
|
+
end
|
|
246
|
+
|
|
247
|
+
def sqlite_candidates(id)
|
|
248
|
+
target = id.root.number.to_s
|
|
249
|
+
rows = @file_io.sqlite_bucket(target)
|
|
250
|
+
@file_io.materialize(rows)
|
|
165
251
|
end
|
|
166
252
|
|
|
167
253
|
# Narrowing key: the base *document's* number as a string. `#root` walks a
|
|
@@ -169,26 +255,24 @@ module Relaton
|
|
|
169
255
|
# returns self for a base document), so a document and all its wrappers
|
|
170
256
|
# share one key and cluster together. `.to_s` because the key is compared
|
|
171
257
|
# as a string (a pubid number Component is not `<`/`>`-comparable), and
|
|
172
|
-
# FileIO
|
|
258
|
+
# FileIO computes this exact same key once at sidecar-build time so the
|
|
259
|
+
# bsearch over the keys array stays valid.
|
|
173
260
|
def candidates_by_number(id)
|
|
174
261
|
target = id.root.number.to_s
|
|
262
|
+
keys, rows = raw_index
|
|
175
263
|
left = bsearch_left(target)
|
|
176
264
|
return [] unless left
|
|
177
265
|
|
|
178
266
|
right = bsearch_right(target)
|
|
179
|
-
|
|
267
|
+
@file_io.materialize(rows[left...right])
|
|
180
268
|
end
|
|
181
269
|
|
|
182
270
|
def bsearch_left(target)
|
|
183
|
-
|
|
184
|
-
item[:id].root.number.to_s >= target
|
|
185
|
-
end
|
|
271
|
+
raw_index[0].bsearch_index { |k| k >= target }
|
|
186
272
|
end
|
|
187
273
|
|
|
188
274
|
def bsearch_right(target)
|
|
189
|
-
|
|
190
|
-
item[:id].root.number.to_s > target
|
|
191
|
-
end || index.size
|
|
275
|
+
raw_index[0].bsearch_index { |k| k > target } || raw_index[0].size
|
|
192
276
|
end
|
|
193
277
|
|
|
194
278
|
# The query is the reference, so two identifiers take the subset match.
|
data/lib/relaton/index.rb
CHANGED
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
require "yaml"
|
|
4
4
|
require "zip"
|
|
5
|
+
require "open-uri"
|
|
5
6
|
require "relaton/logger"
|
|
6
7
|
|
|
7
8
|
require_relative "version"
|
|
@@ -10,9 +11,10 @@ require_relative "index/file_storage"
|
|
|
10
11
|
require_relative "index/config"
|
|
11
12
|
require_relative "index/util"
|
|
12
13
|
require_relative "index/pool"
|
|
13
|
-
require_relative "index/type"
|
|
14
14
|
require_relative "index/file_io"
|
|
15
15
|
require_relative "index/shard_source"
|
|
16
|
+
require_relative "index/sqlite_backend"
|
|
17
|
+
require_relative "index/type"
|
|
16
18
|
|
|
17
19
|
module Relaton
|
|
18
20
|
module Index
|
data/lib/relaton/version.rb
CHANGED
metadata
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: relaton
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 3.0.0.pre.alpha.
|
|
4
|
+
version: 3.0.0.pre.alpha.8
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Ribose Inc.
|
|
8
8
|
autorequire:
|
|
9
9
|
bindir: bin
|
|
10
10
|
cert_chain: []
|
|
11
|
-
date: 2026-10-
|
|
11
|
+
date: 2026-10-03 00:00:00.000000000 Z
|
|
12
12
|
dependencies:
|
|
13
13
|
- !ruby/object:Gem::Dependency
|
|
14
14
|
name: addressable
|
|
@@ -94,6 +94,20 @@ dependencies:
|
|
|
94
94
|
- - "~>"
|
|
95
95
|
- !ruby/object:Gem::Version
|
|
96
96
|
version: '1.0'
|
|
97
|
+
- !ruby/object:Gem::Dependency
|
|
98
|
+
name: sqlite3
|
|
99
|
+
requirement: !ruby/object:Gem::Requirement
|
|
100
|
+
requirements:
|
|
101
|
+
- - "~>"
|
|
102
|
+
- !ruby/object:Gem::Version
|
|
103
|
+
version: '1.7'
|
|
104
|
+
type: :runtime
|
|
105
|
+
prerelease: false
|
|
106
|
+
version_requirements: !ruby/object:Gem::Requirement
|
|
107
|
+
requirements:
|
|
108
|
+
- - "~>"
|
|
109
|
+
- !ruby/object:Gem::Version
|
|
110
|
+
version: '1.7'
|
|
97
111
|
- !ruby/object:Gem::Dependency
|
|
98
112
|
name: csv
|
|
99
113
|
requirement: !ruby/object:Gem::Requirement
|
|
@@ -933,6 +947,7 @@ files:
|
|
|
933
947
|
- lib/relaton/index/file_storage.rb
|
|
934
948
|
- lib/relaton/index/pool.rb
|
|
935
949
|
- lib/relaton/index/shard_source.rb
|
|
950
|
+
- lib/relaton/index/sqlite_backend.rb
|
|
936
951
|
- lib/relaton/index/type.rb
|
|
937
952
|
- lib/relaton/index/util.rb
|
|
938
953
|
- lib/relaton/isbn.rb
|