relaton 3.0.0.pre.alpha.6 → 3.0.0.pre.alpha.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: 5029468b22fa9cc5e65bf7d76f8bb6db597fb331bc1f0a89a21186063ea49963
4
- data.tar.gz: 6e572768f9ae838d04e682eb6e5cbc2a48dcb9e999fd2f4dce588952072475a6
3
+ metadata.gz: 3fafde9870593aca737727dcb4cf8cb142da74d255be4e855eed22bda4bd0f1b
4
+ data.tar.gz: 60986309859651e6e1baa05635a1ebfb2fec313f908a5508c14a59ded9ec3f8f
5
5
  SHA512:
6
- metadata.gz: 9ef70c48d70df0a8d7105b7195e0124f51a501eab815bca1ce6825a9113d338f3a1203c237d896443ae0f5ed1893efef97719bfd9a2938c53b58155171bf6c46
7
- data.tar.gz: bb362dea5df12c9021b0725132e4f365aa8f97d6a81e9b41c2190c66121b6948aa4f6aee9a52f0c51e218ea38b5655b6ad014ea97d8100f00a83d07098ebfa0c
6
+ metadata.gz: 2b075475ad0321b604674b8459c60ddc1e7ef10a1a0186e70042348c0d9d5bbe58166cb419548f8d98438040702ffd532204716a261bedaedf20888a5d6c08cf
7
+ data.tar.gz: 8b2b003b3b99918b96a42c622607feae3087e2778730a2d31dc94ebe9b51abea23a3f6fbdb254fc72a2daeeae47877feb4edb998d70aa6d0d46c86881db2812c
@@ -29,8 +29,7 @@ module Relaton
29
29
  return unless val.is_a?(String) && !val.empty?
30
30
  return if @text.is_a?(String) && !@text.empty?
31
31
 
32
- description = Isoics.fetch(val)&.description
33
- self.text = description if description
32
+ populate_text_from_isoics
34
33
  end
35
34
 
36
35
  # When the deserializer reaches the end of the XML element and
@@ -38,9 +37,44 @@ module Relaton
38
37
  # to mark the attribute as default-valued (suppressing serialization).
39
38
  # Refuse that mark if we've already populated text from Isoics so the
40
39
  # value survives round-trip. See #112.
40
+ # lutaml-model 0.8.92 (lutaml/lutaml-model#922): assignments made
41
+ # inside `code=` while from_xml is still mid-element do not register
42
+ # as explicit values. Populate once deserialization has finished.
43
+ def self.from_xml(node)
44
+ ics = super
45
+ # lutaml-model 0.8.92 (#922): a value assigned through a custom
46
+ # writer's `super` mid-deserialization is left default-suppressed.
47
+ # Register code as explicitly set, then populate text.
48
+ ics.value_set_for(:code) if ics.code.is_a?(String) && !ics.code.empty?
49
+ ics.populate_text_from_isoics
50
+ ics
51
+ end
52
+
53
+ def populate_text_from_isoics
54
+ code.is_a?(String) && !code.empty? or return
55
+ @text.is_a?(String) && !@text.empty? and return
56
+
57
+ description = Isoics.fetch(code)&.description
58
+ self.text = description if description
59
+ end
60
+
41
61
  def using_default_for(attribute_name)
42
- return if attribute_name == :text &&
43
- @text.is_a?(String) && !@text.empty?
62
+ # lutaml-model 0.8.92 (#922) leaves values assigned through a
63
+ # custom writer's `super` default-suppressed. code, when present,
64
+ # is always explicit; text absent from the XML is filled from
65
+ # Isoics here — the element end — and emitted.
66
+ if attribute_name == :code
67
+ return value_set_for(:code) if @code.is_a?(String) && !@code.empty?
68
+
69
+ return super
70
+ end
71
+
72
+ return if attribute_name == :text && @text.is_a?(String) && !@text.empty?
73
+
74
+ if attribute_name == :text
75
+ populate_text_from_isoics
76
+ return value_set_for(:text) unless @text.nil? || @text.empty?
77
+ end
44
78
 
45
79
  super
46
80
  end
@@ -45,22 +45,117 @@ module Relaton
45
45
  report_errors
46
46
  end
47
47
 
48
- def fetch_remaining_pages(first_page)
48
+ #
49
+ # Fetch pages 2..N. A page that stays bad after its retries is fetched
50
+ # again after the last page; if it is still bad, the crawl fails. A page
51
+ # is never skipped: relaton-data-etsi deletes data/ before the crawl and
52
+ # commits what it writes, so a skipped page would unpublish its documents.
53
+ #
54
+ # N comes from page 1's total_count, but the result set is sorted by
55
+ # deliverable number, so a document published during the crawl shifts the
56
+ # later pages by one. The crawl therefore reads on past N while the pages
57
+ # stay full (at most MAX_EXTRA_PAGES), and one page past a deferred last
58
+ # page; a record read twice overwrites its own file.
59
+ #
60
+ def fetch_remaining_pages(first_page) # rubocop:disable Metrics/MethodLength
61
+ total_pages = last = last_page(first_page)
62
+ deferred = []
63
+ page = 2
64
+ while page <= last
65
+ check_extra_pages page, total_pages
66
+ size = fetch_next_page(page, deferred)
67
+ break if size&.zero?
68
+
69
+ last = page + 1 if page == last && read_on?(size, page, total_pages)
70
+ page += 1
71
+ end
72
+ fetch_deferred_pages(deferred, total_pages)
73
+ end
74
+
75
+ def last_page(first_page)
49
76
  total = first_page.first ? first_page.first["total_count"].to_i : 0
50
- total_pages = (total / PAGE_SIZE.to_f).ceil
51
- (2..total_pages).each do |page|
52
- records = fetch_page(page)
53
- break if records.empty?
77
+ last = (total / PAGE_SIZE.to_f).ceil
78
+ first_page.size >= PAGE_SIZE ? [last, 2].max : last
79
+ end
80
+
81
+ #
82
+ # Whether to read the page after the last one: after a full page, or
83
+ # after a deferred page inside the total_count range. A deferred page
84
+ # past that range does not extend it, so a server that answers bad
85
+ # bodies past the end cannot keep the crawl going.
86
+ #
87
+ def read_on?(size, page, total_pages)
88
+ size ? size >= PAGE_SIZE : page <= total_pages
89
+ end
90
+
91
+ def check_extra_pages(page, total_pages)
92
+ return if page <= total_pages + MAX_EXTRA_PAGES
93
+
94
+ raise BadPage, "ETSI page #{page} is still full #{MAX_EXTRA_PAGES} " \
95
+ "pages past total_count (#{total_pages} pages)."
96
+ end
97
+
98
+ #
99
+ # @return [Integer, nil] the number of records on the page; nil for a
100
+ # deferred page
101
+ #
102
+ def fetch_next_page(page, deferred)
103
+ records = fetch_page(page)
104
+ process_records(records)
105
+ records.size
106
+ rescue BadPage => e
107
+ Util.warn "#{e.message} Fetching it again after the last page."
108
+ deferred << page
109
+ nil
110
+ end
54
111
 
55
- process_records(records)
112
+ def fetch_deferred_pages(pages, total_pages)
113
+ return if pages.empty?
114
+
115
+ sleep DEFERRED_DELAY
116
+ failed = pages.filter_map do |page|
117
+ refetch_page page, total_pages
118
+ rescue BadPage => e
119
+ e.message
56
120
  end
121
+ raise BadPage, failed.join(" ") if failed.any?
122
+ end
123
+
124
+ # @return [String, nil] the failure message, nil on success
125
+ def refetch_page(page, total_pages)
126
+ records = fetch_page(page)
127
+ if records.empty? && page <= total_pages
128
+ return "ETSI page #{page} is empty on the re-fetch."
129
+ end
130
+
131
+ process_records records
132
+ nil
57
133
  end
58
134
 
59
135
  def fetch_page(page)
60
136
  date = Time.now.to_date + 1
61
137
  timestamp = (Time.now.to_f * 1000).to_i
62
138
  url = format(SOURCEURL, page: page, date: date, timestamp: timestamp)
63
- JSON.parse(fetch_with_retry(url))
139
+ fetch_with_retry(url) { |body| parse_page(page, body) }
140
+ end
141
+
142
+ #
143
+ # ETSI's data.php sometimes answers with PHP's `Array` text in place of
144
+ # JSON (relaton-data-etsi crawl 36761804954); a re-fetch gets JSON.
145
+ #
146
+ # @raise [BadPage] when the body is not a JSON array
147
+ #
148
+ def parse_page(page, body)
149
+ records = JSON.parse(body)
150
+ return records if records.is_a?(Array)
151
+
152
+ raise BadPage, bad_page_message(page, body)
153
+ rescue JSON::ParserError
154
+ raise BadPage, bad_page_message(page, body)
155
+ end
156
+
157
+ def bad_page_message(page, body)
158
+ "ETSI page #{page} is not a JSON array: #{body.to_s[0, 200].inspect}."
64
159
  end
65
160
 
66
161
  def process_records(records)
@@ -92,16 +187,30 @@ module Relaton
92
187
  "Published"
93
188
  end
94
189
 
190
+ # A page body that is not a JSON array. See #parse_page.
191
+ class BadPage < StandardError; end
192
+
193
+ # Seconds to wait before a bad page is fetched again after the last page.
194
+ DEFERRED_DELAY = 60
195
+
196
+ # Full pages read past page 1's total_count before the crawl fails.
197
+ MAX_EXTRA_PAGES = 10
198
+
95
199
  NETWORK_ERRORS = [
96
200
  Mechanize::Error, Net::OpenTimeout, Net::ReadTimeout,
97
201
  SocketError, Errno::ECONNRESET
98
202
  ].freeze
99
203
 
204
+ #
205
+ # @yield [String] the body; the block's result is returned, and a BadPage
206
+ # it raises is retried like a network error
207
+ #
100
208
  def fetch_with_retry(url, retries: 3, delay: 2) # rubocop:disable Metrics/MethodLength
101
209
  attempt = 0
102
210
  begin
103
- Mechanize.new.get(url).body
104
- rescue *NETWORK_ERRORS => e
211
+ body = Mechanize.new.get(url).body
212
+ block_given? ? yield(body) : body
213
+ rescue *NETWORK_ERRORS, BadPage => e
105
214
  attempt += 1
106
215
  if attempt <= retries
107
216
  Util.info "Fetch failed (#{e.message}), " \
@@ -7,7 +7,7 @@ module Relaton
7
7
  @short = :relaton_iec
8
8
  @prefix = "IEC"
9
9
  @pubid_flavor = :Iec # global prefixes sourced from Pubid::Iec.prefixes
10
- @defaultprefix = %r{^(IEC\s|CISPR\s|IEV($|\s))}
10
+ @defaultprefix = %r{^(IEC\s|CISPR\s|IEV($|\s)|CEI\s)} # CEI: IEC's French spelling (relaton#243)
11
11
  @idtype = "IEC"
12
12
  @datasets = %w[iec-harmonized-all iec-harmonized-latest]
13
13
  end
@@ -4,7 +4,8 @@ module Relaton
4
4
  # Configuration class for Relaton::Index
5
5
  #
6
6
  class Config
7
- attr_reader :storage, :storage_dir, :filename
7
+ attr_reader :storage, :storage_dir, :filename, :build_sidecar_in_child,
8
+ :sqlite_index
8
9
 
9
10
  #
10
11
  # Set default values
@@ -13,6 +14,23 @@ module Relaton
13
14
  @storage = FileStorage
14
15
  @storage_dir = Dir.home
15
16
  @filename = "index.yaml"
17
+ @build_sidecar_in_child = true
18
+ @sqlite_index = true
19
+ end
20
+
21
+ # Build the sidecar in a forked child when available, so the one-time
22
+ # full materialization's memory dies with the child (relaton#242).
23
+ # Set to false to force an in-process build.
24
+ def build_sidecar_in_child=(flag)
25
+ @build_sidecar_in_child = flag
26
+ end
27
+
28
+ # Materialize a downloaded index into SQLite so a search answers from
29
+ # a bucket query and the process never holds the whole index
30
+ # (relaton#242 phase 2). Set to false to keep the raw-row sidecar path
31
+ # only.
32
+ def sqlite_index=(flag)
33
+ @sqlite_index = flag
16
34
  end
17
35
 
18
36
  #
@@ -11,6 +11,10 @@ module Relaton
11
11
  # wrong-structure handling (re-download, or stop and log).
12
12
  class InvalidIndexError < StandardError; end
13
13
 
14
+ # Bump when the sidecar payload shape changes, so an older sidecar is
15
+ # discarded and rebuilt instead of misread.
16
+ SIDECAR_VERSION = 2 # v2: the payload carries the yaml byte-size for freshness
17
+
14
18
  attr_reader :url, :pubid_class
15
19
  attr_accessor :sorted
16
20
 
@@ -271,6 +275,8 @@ module Relaton
271
275
  end
272
276
  end.to_yaml
273
277
  Index.config.storage.write file, yaml
278
+ delete_sidecar
279
+ delete_sqlite
274
280
  end
275
281
 
276
282
  def sort_structured_index(index)
@@ -288,11 +294,339 @@ module Relaton
288
294
  #
289
295
  def remove
290
296
  Index.config.storage.remove file
297
+ delete_sidecar
298
+ delete_sqlite
291
299
  []
292
300
  end
293
301
 
302
+ #
303
+ # Raw-row read for the lazy search path (relaton#242 stopgap): returns
304
+ # precomputed root-number sort keys and the rows as plain hashes — no
305
+ # pubid objects. A Marshal sidecar next to the yaml holds them so
306
+ # repeat loads skip the YAML parse and the one-time full
307
+ # materialization; the sidecar is rebuilt whenever the yaml is newer.
308
+ #
309
+ # @return [Array<Array<String>, Array<Hash>] sort keys and raw rows
310
+ #
311
+ def read_raw
312
+ case url
313
+ when String
314
+ with_file_lock do
315
+ check_file ? read_raw_file : fetch_raw_and_save
316
+ end
317
+ else
318
+ read_raw_file || [[], []]
319
+ end
320
+ end
321
+
322
+ # ── SQLite backend (relaton#242 phase 2) ──
323
+ #
324
+ # The downloaded index is materialized once into a SQLite database
325
+ # keyed by the narrowing number. A search answers from a bucket query,
326
+ # so the process holds O(bucket) rows instead of the whole index. The
327
+ # build is pure hash transforms (no pubid objects), run in a forked
328
+ # child so even the one-time YAML parse never lands in this process.
329
+
330
+ def sqlite_ready?
331
+ url.is_a?(String) && Index.config.sqlite_index != false &&
332
+ File.file?(sqlite_db_path) && sqlite_fresh?
333
+ end
334
+
335
+ # Ensure the db exists and is current (24 h TTL, schema-versioned).
336
+ def ensure_sqlite
337
+ return false unless url.is_a?(String)
338
+ return true if sqlite_ready?
339
+
340
+ with_file_lock do
341
+ build_sqlite unless sqlite_ready?
342
+ end
343
+ sqlite_ready?
344
+ end
345
+
346
+ def sqlite_bucket(number)
347
+ sqlite_backend.bucket(number)
348
+ end
349
+
350
+ def sqlite_count
351
+ sqlite_backend.count
352
+ end
353
+
354
+ def close_sqlite
355
+ @sqlite_backend&.close
356
+ @sqlite_backend = nil
357
+ end
358
+
359
+ def delete_sqlite
360
+ close_sqlite
361
+ File.delete(sqlite_db_path) if File.file?(sqlite_db_path)
362
+ rescue Errno::EACCES
363
+ nil
364
+ end
365
+
366
+ def sqlite_db_path
367
+ "#{path_to_local_file}.db"
368
+ end
369
+
370
+ private
371
+
372
+ def sqlite_backend
373
+ @sqlite_backend ||= SqliteBackend.new(sqlite_db_path, pubid_class: @pubid_class)
374
+ end
375
+
376
+ def sqlite_fresh?
377
+ ctime = Index.config.storage.ctime(sqlite_db_path)
378
+ backend = sqlite_backend
379
+ ctime && ctime > Time.now - 86400 && !backend.stale_schema?
380
+ rescue SQLite3::SQLException
381
+ false
382
+ ensure
383
+ backend&.close
384
+ @sqlite_backend = nil
385
+ end
386
+
387
+ # The build downloads the published zip, converts its rows to
388
+ # `[number, sort_key, id_json, file]` tuples by hash transforms alone,
389
+ # and writes the db — in a forked child where the YAML parse's memory
390
+ # dies with the process.
391
+ def build_sqlite
392
+ # The whole build — download, YAML parse, db write — runs in a
393
+ # forked child where available: the parse's ~300 MB (per flavor,
394
+ # larger for IETF) dies with the child, and the serving process
395
+ # never allocates it.
396
+ if Process.respond_to?(:fork) && !ENV["RELATON_NO_FORK"]
397
+ pid = Process.fork do
398
+ tuples = download_tuples
399
+ write_sqlite(tuples) if tuples
400
+ exit!(0)
401
+ end
402
+ Process.wait(pid)
403
+ true
404
+ else
405
+ tuples = download_tuples
406
+ return false unless tuples
407
+
408
+ write_sqlite(tuples)
409
+ end
410
+ end
411
+
412
+ def write_sqlite(tuples)
413
+ File.dirname(sqlite_db_path).then { |d| require "fileutils"; FileUtils.mkdir_p(d) }
414
+ backend = SqliteBackend.new(sqlite_db_path, pubid_class: @pubid_class)
415
+ backend.build(tuples.each)
416
+ backend.close
417
+ end
418
+
419
+ # Enumerate `[number, sort_key, id_hash, file]` from the published zip.
420
+ # The root number — the base document's, walking `base` nesting — is
421
+ # computed on the raw hash; no pubid object is ever built.
422
+ def download_tuples
423
+ body = Net::HTTP.get(URI.parse(url))
424
+ # rubyzip may mutate the buffer it is handed; a StringIO over a copy
425
+ # keeps the response body intact for any later reader of it.
426
+ yaml = nil
427
+ Zip::File.open_buffer(StringIO.new(body.dup)) do |zip|
428
+ stream = zip.entries.first.get_input_stream
429
+ yaml = stream.read
430
+ end
431
+ yaml = yaml.read unless yaml.is_a?(String)
432
+ Util.info "Downloaded index from `#{url}` for sqlite materialization", progname
433
+ raw = YAML.safe_load(yaml, permitted_classes: [Symbol])
434
+ return nil unless check_format(raw)
435
+
436
+ Enumerator.new do |y|
437
+ raw.each do |r|
438
+ number = raw_root_number(r[:id]).to_s
439
+ y << [number, number, r[:id], r[:file]]
440
+ end
441
+ end
442
+ rescue Psych::SyntaxError, SocketError, OpenURI::HTTPError, Errno::ECONNRESET,
443
+ OpenSSL::SSL::SSLError => e
444
+ Util.info "SQLite index build failed (#{e.message}); falling back", progname
445
+ nil
446
+ end
447
+
448
+ # `#root` as a hash walk: a supplement nests its origin under `base`,
449
+ # and the flattened to_hash puts the supplement's own components first.
450
+ # Accepts string and symbol keys — a YAML round-trip with permitted
451
+ # Symbol may give either.
452
+ def raw_root_number(id_hash)
453
+ return nil unless id_hash.is_a?(Hash)
454
+
455
+ base = id_hash["base"] || id_hash[:base]
456
+ return raw_root_number(base) if base
457
+
458
+ id_hash["number"] || id_hash[:number]
459
+ end
460
+
461
+ public
462
+
463
+ # Deserialize raw rows into pubid rows. Only the caller knows which
464
+ # slice it needs, so materialization happens here rather than at load.
465
+ #
466
+ # @param [Array<Hash>] rows raw rows
467
+ # @return [Array<Hash>] rows with deserialized ids
468
+ def materialize(rows)
469
+ return rows unless @pubid_class
470
+
471
+ rows.map { |r| { id: deserialize_id(r[:id]), file: r[:file] } }
472
+ end
473
+
474
+ # Save raw rows (possibly alongside the keys they were loaded with);
475
+ # sorts by the same root-number key the object path sorts by.
476
+ #
477
+ # @param [Array<String>] keys sort keys, parallel to rows
478
+ # @param [Array<Hash>] rows raw rows
479
+ # @return [void]
480
+ def save_raw(keys, rows)
481
+ ordered = @pubid_class ? keys.zip(rows).sort_by { |k, _| k.to_s }
482
+ .map { |_, r| r } : rows
483
+ yaml = ordered.map do |item|
484
+ { id: item[:id], file: item[:file] }
485
+ end.to_yaml
486
+ Index.config.storage.write file, yaml
487
+ delete_sidecar
488
+ end
489
+
294
490
  private
295
491
 
492
+ def sidecar_file
493
+ "#{file}.ms"
494
+ end
495
+
496
+ def delete_sidecar
497
+ File.delete(sidecar_file) if File.file?(sidecar_file)
498
+ rescue Errno::EACCES
499
+ nil
500
+ end
501
+
502
+ def read_raw_file
503
+ # The sidecar is authoritative while it describes the yaml byte-for-byte
504
+ # (same size) — the yaml is not even parsed until the sidecar is stale
505
+ # or unreadable.
506
+ if sidecar_fresh?
507
+ loaded = sidecar
508
+ return loaded if loaded
509
+ end
510
+
511
+ yaml = Index.config.storage.read(file)
512
+ return unless yaml
513
+
514
+ begin
515
+ raw = YAML.safe_load(yaml, permitted_classes: [Symbol])
516
+ rescue Psych::SyntaxError
517
+ warn_local_index_error("YAML parsing error when reading")
518
+ delete_sidecar
519
+ return [[], []]
520
+ end
521
+ build_raw(raw)
522
+ end
523
+
524
+ # Size, not mtime: NTFS timestamp coarseness makes a rewritten yaml
525
+ # carry the sidecar's own mtime, so "yaml is newer" misses the rebuild.
526
+ def sidecar_fresh?
527
+ File.file?(sidecar_file) && File.file?(file) &&
528
+ File.size(file) == sidecar_yaml_size
529
+ end
530
+
531
+ def sidecar_yaml_size
532
+ Marshal.load(File.binread(sidecar_file))[4]
533
+ rescue TypeError, ArgumentError, EOFError
534
+ nil
535
+ end
536
+
537
+ def sidecar
538
+ version, sorted, keys, rows = Marshal.load(File.binread(sidecar_file))
539
+ return unless version == SIDECAR_VERSION
540
+
541
+ @sorted = sorted
542
+ [keys, rows]
543
+ rescue TypeError, ArgumentError, EOFError
544
+ nil
545
+ end
546
+
547
+ def build_raw(raw)
548
+ unless check_format(raw)
549
+ warn_local_index_error("Wrong structure of")
550
+ return [[], []]
551
+ end
552
+
553
+ return build_raw_in_child(raw) if build_sidecar_in_child?
554
+
555
+ build_raw_in_process(raw)
556
+ rescue InvalidIndexError
557
+ warn_local_index_error("Wrong structure of")
558
+ [[], []]
559
+ end
560
+
561
+ # The one-time key build materializes every row (~675 MB on the
562
+ # 79,993-row ISO index). Freed pages are often not returned to the OS,
563
+ # so an in-process build leaves the parent's RSS high — and containers
564
+ # OOM on RSS (relaton#242). Where fork exists, build the sidecar in a
565
+ # child: the graph dies with it and the parent never spikes.
566
+ def build_sidecar_in_child?
567
+ Process.respond_to?(:fork) && Index.config.build_sidecar_in_child != false
568
+ end
569
+
570
+ def build_raw_in_child(raw)
571
+ pid = Process.fork do
572
+ build_raw_in_process(raw)
573
+ exit!(0)
574
+ end
575
+ Process.wait(pid)
576
+ loaded = sidecar
577
+ return build_raw_in_process(raw) unless loaded # child failed
578
+
579
+ loaded
580
+ rescue Errno::ENOMEM, SystemCallError
581
+ build_raw_in_process(raw)
582
+ end
583
+
584
+ def build_raw_in_process(raw)
585
+ objects = deserialize_pubid(raw)
586
+ keys = objects.map { |r| r[:id].root.number.to_s }
587
+ rows = objects.map { |r| { id: raw_id(r), file: r[:file] } }
588
+ write_sidecar(keys, rows)
589
+ [keys, rows]
590
+ rescue InvalidIndexError
591
+ warn_local_index_error("Wrong structure of")
592
+ [[], []]
593
+ end
594
+
595
+ def raw_id(row)
596
+ id = row[:id]
597
+ id.respond_to?(:to_hash) ? id.to_hash : id
598
+ end
599
+
600
+ def write_sidecar(keys, rows)
601
+ File.binwrite(sidecar_file,
602
+ Marshal.dump([SIDECAR_VERSION, @sorted, keys, rows,
603
+ File.size(file)]))
604
+ rescue Errno::EACCES, Errno::ENOENT, Errno::EROFS
605
+ nil # the sidecar is an optimization; a read-only dir still works
606
+ end
607
+
608
+ def fetch_raw_and_save
609
+ uri = URI.parse(url)
610
+ body = Net::HTTP.get(uri)
611
+ yaml = nil
612
+ Zip::File.open_buffer(body) do |zip|
613
+ yaml = zip.entries.first.get_input_stream.read
614
+ end
615
+ Util.info "Downloaded index from `#{url}`", progname
616
+ raw = YAML.safe_load(yaml, permitted_classes: [Symbol])
617
+ if check_format(raw)
618
+ save raw
619
+ raw = nil # release the parsed copy; read_raw_file loads the sidecar's
620
+ read_raw_file
621
+ else
622
+ warn_remote_index_error "Wrong structure of"
623
+ [[], []]
624
+ end
625
+ rescue Psych::SyntaxError
626
+ warn_remote_index_error "YAML parsing error when reading"
627
+ [[], []]
628
+ end
629
+
296
630
  def with_file_lock(&)
297
631
  @@file_locks_mutex.synchronize do
298
632
  @@file_locks[file] ||= Mutex.new
@@ -0,0 +1,99 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "sqlite3"
4
+
5
+ module Relaton
6
+ module Index
7
+ #
8
+ # The SQLite materialization of one flavor index (relaton#242 phase 2).
9
+ #
10
+ # A downloaded index zip is built once into a local database whose rows
11
+ # are `(number, sort_key, id_json, file)` with an index on `number` — the
12
+ # narrowing key. A search materializes only its bucket; the whole index
13
+ # is never resident. WAL journaling lets one build and concurrent readers
14
+ # share the file across processes.
15
+ #
16
+ class SqliteBackend
17
+ SCHEMA_VERSION = "1"
18
+
19
+ def initialize(db_path, pubid_class: nil)
20
+ @db_path = db_path
21
+ @pubid_class = pubid_class
22
+ @db = SQLite3::Database.new(db_path)
23
+ @db.busy_timeout = 30_000
24
+ @db.results_as_hash = false
25
+ @db.execute("PRAGMA journal_mode=WAL")
26
+ @db.execute("PRAGMA synchronous=NORMAL")
27
+ create_schema
28
+ end
29
+
30
+ # Build from an enumerator of `[number, sort_key, id_hash, file]`
31
+ # tuples. Callers hand the child process's enumerator here; the tuples
32
+ # are written inside one transaction.
33
+ def build(rows_enum)
34
+ @db.transaction do
35
+ @db.execute("DELETE FROM index_rows")
36
+ @db.prepare("INSERT INTO index_rows (number, sort_key, id_json, file) VALUES (?, ?, ?, ?)") do |stmt|
37
+ rows_enum.each { |(number, sort_key, id_hash, file)| stmt.execute(number, sort_key, JSON.generate(id_hash), file) }
38
+ end
39
+ meta_set("schema_version", SCHEMA_VERSION)
40
+ end
41
+ end
42
+
43
+ # The bucket for a narrowing key: raw rows, ordered as written.
44
+ def bucket(number)
45
+ rows = []
46
+ @db.execute("SELECT id_json, file FROM index_rows WHERE number = ? ORDER BY rowid", [number]) do |row|
47
+ rows << { id: JSON.parse(row[0]), file: row[1] }
48
+ end
49
+ rows
50
+ end
51
+
52
+ def each_row
53
+ @db.execute("SELECT id_json, file FROM index_rows ORDER BY rowid") do |row|
54
+ yield({ id: JSON.parse(row[0]), file: row[1] })
55
+ end
56
+ end
57
+
58
+ def count
59
+ @db.get_first_value("SELECT COUNT(*) FROM index_rows").to_i
60
+ end
61
+
62
+ def stale_schema?
63
+ meta_get("schema_version") != SCHEMA_VERSION
64
+ end
65
+
66
+ def close
67
+ @db.close
68
+ end
69
+
70
+ private
71
+
72
+ def create_schema
73
+ @db.execute(<<~SQL)
74
+ CREATE TABLE IF NOT EXISTS index_rows (
75
+ number TEXT NOT NULL,
76
+ sort_key TEXT NOT NULL,
77
+ id_json TEXT NOT NULL,
78
+ file TEXT NOT NULL
79
+ )
80
+ SQL
81
+ @db.execute("CREATE INDEX IF NOT EXISTS idx_index_rows_number ON index_rows (number)")
82
+ @db.execute(<<~SQL)
83
+ CREATE TABLE IF NOT EXISTS index_meta (
84
+ key TEXT PRIMARY KEY,
85
+ value TEXT NOT NULL
86
+ )
87
+ SQL
88
+ end
89
+
90
+ def meta_set(key, value)
91
+ @db.execute("INSERT OR REPLACE INTO index_meta (key, value) VALUES (?, ?)", [key, value])
92
+ end
93
+
94
+ def meta_get(key)
95
+ @db.get_first_value("SELECT value FROM index_meta WHERE key = ?", [key])
96
+ end
97
+ end
98
+ end
99
+ end
@@ -30,7 +30,24 @@ module Relaton
30
30
  def index
31
31
  return @source.whole_index if @source
32
32
 
33
- @index ||= @file_io.read
33
+ @index ||= @file_io.materialize(raw_index[1])
34
+ end
35
+
36
+ # Raw rows plus their precomputed root-number sort keys (relaton#242
37
+ # stopgap). The lazy search path materializes pubid objects only for
38
+ # the bucket it narrows to, so a process that answers one lookup holds
39
+ # raw hashes instead of the whole pubid graph. A caller-seeded `@index`
40
+ # (a spec fixture, a preloaded pool entry) is the source of truth: the
41
+ # raw side derives from it, in its order, so a seeder that sorts by the
42
+ # narrowing key keeps the bsearch valid.
43
+ def raw_index
44
+ @raw_index ||= if @index
45
+ keys = @index.map { |r| narrowing_key(r[:id]) }
46
+ rows = @index.map { |r| { id: raw_row_id(r[:id]), file: r[:file] } }
47
+ [keys, rows]
48
+ else
49
+ @file_io.read_raw
50
+ end
34
51
  end
35
52
 
36
53
  #
@@ -59,14 +76,23 @@ module Relaton
59
76
  # @return [void]
60
77
  #
61
78
  def add_or_update(id, file)
62
- key = id.to_s
63
- item = id_lookup[key]
64
- if item
65
- item[:file] = file
79
+ # A seeded or materialized @index (spec fixtures, preloaded pool
80
+ # entries) and pubid-less indexes run the legacy object path
81
+ # bit-for-bit: their consumers match rows by object identity
82
+ # semantics (`==`, `exclude`, to_s dedup) that raw rows cannot
83
+ # reproduce. Only a pubid-backed FileIO-loaded index — the
84
+ # relaton#242 memory case — adds through raw rows.
85
+ return legacy_add(id, file) if @index || !@file_io.pubid_class
86
+
87
+ raw_hash = raw_row_id(id)
88
+ keys, rows = raw_index
89
+ pos = raw_position(raw_hash)
90
+ if pos
91
+ rows[pos] = { id: raw_hash, file: file }
66
92
  else
67
- new_item = { id: id, file: file }
68
- index << new_item
69
- id_lookup[key] = new_item
93
+ rows << { id: raw_hash, file: file }
94
+ keys << narrowing_key(id)
95
+ @raw_lookup[raw_hash] = rows.size - 1
70
96
  @file_io.sorted = false
71
97
  end
72
98
  end
@@ -110,7 +136,12 @@ module Relaton
110
136
  # @return [void]
111
137
  #
112
138
  def save
113
- @file_io.save(@index || [])
139
+ if @index
140
+ @file_io.save(@index)
141
+ else
142
+ keys, rows = raw_index
143
+ @file_io.save_raw(keys, rows)
144
+ end
114
145
  end
115
146
 
116
147
  #
@@ -125,6 +156,8 @@ module Relaton
125
156
  @source = new_source
126
157
  @index = nil
127
158
  @id_lookup = nil
159
+ @raw_index = nil
160
+ @raw_lookup = nil
128
161
  end
129
162
 
130
163
  #
@@ -135,6 +168,8 @@ module Relaton
135
168
  def remove_all
136
169
  @index = []
137
170
  @id_lookup = nil
171
+ @raw_index = [[], []]
172
+ @raw_lookup = {}
138
173
  @file_io.sorted = true
139
174
  end
140
175
 
@@ -146,6 +181,44 @@ module Relaton
146
181
  end
147
182
  end
148
183
 
184
+ # Position of a raw row by its id hash — the raw-side equivalent of
185
+ # `id_lookup`, built once, so a crawl's add_or_update stays O(1) per row.
186
+ def raw_position(raw_hash)
187
+ @raw_lookup ||= raw_index[1].each_with_object({}) do |row, h|
188
+ h[row[:id]] = h.key?(row[:id]) ? h[row[:id]] : rows_index_of(h, row)
189
+ end
190
+ @raw_lookup[raw_hash]
191
+ end
192
+
193
+ def rows_index_of(lookup, row)
194
+ raw_index[1].index { |r| r[:id].equal?(row[:id]) || r[:id] == row[:id] }
195
+ end
196
+ # The pre-#242 add path, kept verbatim for seeded/materialized and
197
+ # pubid-less indexes.
198
+ def legacy_add(id, file)
199
+ key = id.to_s
200
+ item = id_lookup[key]
201
+ if item
202
+ item[:file] = file
203
+ else
204
+ new_item = { id: id, file: file }
205
+ index << new_item
206
+ id_lookup[key] = new_item
207
+ @file_io.sorted = false
208
+ end
209
+ end
210
+
211
+ # Same key expression the FileIO sidecar computes at build time; a
212
+ # String id (a no-pubid_class index) has no `.root`, and never narrows.
213
+ def narrowing_key(id)
214
+ id.respond_to?(:root) && id.root.respond_to?(:number) ? id.root.number.to_s : ""
215
+ end
216
+
217
+ def raw_row_id(id)
218
+ pc = @file_io.pubid_class
219
+ pc && id.is_a?(pc) ? id.to_hash : id
220
+ end
221
+
149
222
  def new_source
150
223
  ShardSource.new(@dir, @pages_url, @pubid_class) if @pages_url
151
224
  end
@@ -155,13 +228,26 @@ module Relaton
155
228
  # never refreshed, so the shards stay the fresher answer.
156
229
  return @source.rows(id) if @source && id && !id.is_a?(String)
157
230
 
158
- # index needs to be created to check if sorted
159
- idx = index
160
- if @file_io.sorted && id && !id.is_a?(String)
161
- candidates_by_number(id)
162
- else
163
- idx
231
+ # The SQLite backend answers a parsed query with a bucket query:
232
+ # O(bucket) rows ever resident, no sidecar load at all. Falls back
233
+ # to the raw-row path when the backend cannot build or is disabled.
234
+ if id && !id.is_a?(String) && @file_io.pubid_class
235
+ begin
236
+ return sqlite_candidates(id) if @file_io.ensure_sqlite
237
+ rescue SQLite3::SQLException
238
+ nil # fall back to the sidecar path below
239
+ end
240
+
241
+ raw_index
242
+ return candidates_by_number(id) if @file_io.sorted
164
243
  end
244
+ index
245
+ end
246
+
247
+ def sqlite_candidates(id)
248
+ target = id.root.number.to_s
249
+ rows = @file_io.sqlite_bucket(target)
250
+ @file_io.materialize(rows)
165
251
  end
166
252
 
167
253
  # Narrowing key: the base *document's* number as a string. `#root` walks a
@@ -169,26 +255,24 @@ module Relaton
169
255
  # returns self for a base document), so a document and all its wrappers
170
256
  # share one key and cluster together. `.to_s` because the key is compared
171
257
  # as a string (a pubid number Component is not `<`/`>`-comparable), and
172
- # FileIO sorts the index by this exact same key so bsearch stays valid.
258
+ # FileIO computes this exact same key once at sidecar-build time so the
259
+ # bsearch over the keys array stays valid.
173
260
  def candidates_by_number(id)
174
261
  target = id.root.number.to_s
262
+ keys, rows = raw_index
175
263
  left = bsearch_left(target)
176
264
  return [] unless left
177
265
 
178
266
  right = bsearch_right(target)
179
- index[left...right]
267
+ @file_io.materialize(rows[left...right])
180
268
  end
181
269
 
182
270
  def bsearch_left(target)
183
- index.bsearch_index do |item|
184
- item[:id].root.number.to_s >= target
185
- end
271
+ raw_index[0].bsearch_index { |k| k >= target }
186
272
  end
187
273
 
188
274
  def bsearch_right(target)
189
- index.bsearch_index do |item|
190
- item[:id].root.number.to_s > target
191
- end || index.size
275
+ raw_index[0].bsearch_index { |k| k > target } || raw_index[0].size
192
276
  end
193
277
 
194
278
  # The query is the reference, so two identifiers take the subset match.
data/lib/relaton/index.rb CHANGED
@@ -2,6 +2,7 @@
2
2
 
3
3
  require "yaml"
4
4
  require "zip"
5
+ require "open-uri"
5
6
  require "relaton/logger"
6
7
 
7
8
  require_relative "version"
@@ -10,9 +11,10 @@ require_relative "index/file_storage"
10
11
  require_relative "index/config"
11
12
  require_relative "index/util"
12
13
  require_relative "index/pool"
13
- require_relative "index/type"
14
14
  require_relative "index/file_io"
15
15
  require_relative "index/shard_source"
16
+ require_relative "index/sqlite_backend"
17
+ require_relative "index/type"
16
18
 
17
19
  module Relaton
18
20
  module Index
@@ -1,3 +1,3 @@
1
1
  module Relaton
2
- VERSION = "3.0.0.pre.alpha.6".freeze
2
+ VERSION = "3.0.0.pre.alpha.8".freeze
3
3
  end
metadata CHANGED
@@ -1,14 +1,14 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: relaton
3
3
  version: !ruby/object:Gem::Version
4
- version: 3.0.0.pre.alpha.6
4
+ version: 3.0.0.pre.alpha.8
5
5
  platform: ruby
6
6
  authors:
7
7
  - Ribose Inc.
8
8
  autorequire:
9
9
  bindir: bin
10
10
  cert_chain: []
11
- date: 2026-10-02 00:00:00.000000000 Z
11
+ date: 2026-10-03 00:00:00.000000000 Z
12
12
  dependencies:
13
13
  - !ruby/object:Gem::Dependency
14
14
  name: addressable
@@ -94,6 +94,20 @@ dependencies:
94
94
  - - "~>"
95
95
  - !ruby/object:Gem::Version
96
96
  version: '1.0'
97
+ - !ruby/object:Gem::Dependency
98
+ name: sqlite3
99
+ requirement: !ruby/object:Gem::Requirement
100
+ requirements:
101
+ - - "~>"
102
+ - !ruby/object:Gem::Version
103
+ version: '1.7'
104
+ type: :runtime
105
+ prerelease: false
106
+ version_requirements: !ruby/object:Gem::Requirement
107
+ requirements:
108
+ - - "~>"
109
+ - !ruby/object:Gem::Version
110
+ version: '1.7'
97
111
  - !ruby/object:Gem::Dependency
98
112
  name: csv
99
113
  requirement: !ruby/object:Gem::Requirement
@@ -933,6 +947,7 @@ files:
933
947
  - lib/relaton/index/file_storage.rb
934
948
  - lib/relaton/index/pool.rb
935
949
  - lib/relaton/index/shard_source.rb
950
+ - lib/relaton/index/sqlite_backend.rb
936
951
  - lib/relaton/index/type.rb
937
952
  - lib/relaton/index/util.rb
938
953
  - lib/relaton/isbn.rb