relaton 3.0.0.pre.alpha.5 → 3.0.0.pre.alpha.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. checksums.yaml +4 -4
  2. data/lib/relaton/3gpp/bibliography.rb +19 -6
  3. data/lib/relaton/3gpp/processor.rb +6 -0
  4. data/lib/relaton/bib/item_data.rb +14 -0
  5. data/lib/relaton/bipm/bibliography.rb +24 -21
  6. data/lib/relaton/bsi/bibliography.rb +26 -23
  7. data/lib/relaton/calconnect/bibliography.rb +6 -6
  8. data/lib/relaton/calconnect/hit_collection.rb +4 -1
  9. data/lib/relaton/ccsds/bibliography.rb +15 -10
  10. data/lib/relaton/ccsds/hit_collection.rb +1 -1
  11. data/lib/relaton/ccsds/processor.rb +3 -2
  12. data/lib/relaton/cen/bibliography.rb +27 -24
  13. data/lib/relaton/cie/bibliography.rb +11 -9
  14. data/lib/relaton/cie/scrapper.rb +15 -8
  15. data/lib/relaton/core/processor.rb +61 -8
  16. data/lib/relaton/core/request_error.rb +21 -3
  17. data/lib/relaton/db/cache.rb +49 -19
  18. data/lib/relaton/db/cache_entry.rb +9 -0
  19. data/lib/relaton/db/registry.rb +128 -28
  20. data/lib/relaton/db.rb +162 -48
  21. data/lib/relaton/ecma/bibliography.rb +18 -13
  22. data/lib/relaton/etsi/bibliography.rb +8 -7
  23. data/lib/relaton/etsi/data_fetcher.rb +118 -9
  24. data/lib/relaton/gb/bibliography.rb +20 -12
  25. data/lib/relaton/iala/bibliography.rb +15 -12
  26. data/lib/relaton/iana/bibliography.rb +12 -12
  27. data/lib/relaton/iana/parser.rb +5 -2
  28. data/lib/relaton/iec/bibliography.rb +18 -7
  29. data/lib/relaton/iec/processor.rb +6 -1
  30. data/lib/relaton/ieee/bibliography.rb +19 -13
  31. data/lib/relaton/ieee/processor.rb +16 -0
  32. data/lib/relaton/ieee/rawbib_id_parser.rb +2 -2
  33. data/lib/relaton/ietf/bibliography.rb +11 -9
  34. data/lib/relaton/ietf/scraper.rb +5 -4
  35. data/lib/relaton/index/config.rb +9 -1
  36. data/lib/relaton/index/file_io.rb +190 -0
  37. data/lib/relaton/index/type.rb +96 -23
  38. data/lib/relaton/iso/bibliography.rb +11 -9
  39. data/lib/relaton/itu/bibliography.rb +9 -4
  40. data/lib/relaton/jis/bibliography.rb +22 -11
  41. data/lib/relaton/nist/bibliography.rb +25 -21
  42. data/lib/relaton/oasis/bibliography.rb +16 -13
  43. data/lib/relaton/ogc/bibliography.rb +15 -14
  44. data/lib/relaton/ogc/hit_collection.rb +9 -7
  45. data/lib/relaton/ogc/processor.rb +7 -1
  46. data/lib/relaton/omg/bibliography.rb +9 -9
  47. data/lib/relaton/omg/scraper.rb +3 -2
  48. data/lib/relaton/plateau/bibliography.rb +9 -7
  49. data/lib/relaton/plateau/hit_collection.rb +2 -1
  50. data/lib/relaton/version.rb +1 -1
  51. data/lib/relaton/xsf/bibliography.rb +11 -6
  52. metadata +18 -4
data/lib/relaton/db.rb CHANGED
@@ -63,25 +63,23 @@ module Relaton
63
63
  # or "YYYY-MM-DD")
64
64
  #
65
65
  # @return [nil, RelatonBib::BibliographicItem, ...]
66
+ # @raise [Relaton::UnknownReferenceError] no flavor recognizes the
67
+ # reference (see Registry#route)
66
68
  ##
67
- def fetch(text, year = nil, opts = {})
69
+ def fetch(text, year = nil, opts = {}) # rubocop:disable Metrics/MethodLength
68
70
  reference = text.strip
69
- require "pubid"
70
- parsed = begin
71
- Pubid.parse(reference)
72
- rescue Pubid::Errors::Error, Parslet::ParseFailed
73
- nil
74
- end
75
- stdclass = (parsed && @registry.class_by_pubid(parsed)) ||
76
- @registry.class_by_ref(reference) || return
71
+ stdclass, pubid = @registry.route(reference)
77
72
  processor = @registry[stdclass]
78
73
  ref = if processor.respond_to?(:urn_to_code)
79
74
  processor.urn_to_code(reference)&.first
80
75
  else reference
81
76
  end
82
77
  ref ||= reference
78
+ # The routed pubid is the parse of `reference`; a URN rewritten to a
79
+ # code is parsed again by the flavor.
80
+ pubid = nil unless ref == reference
83
81
  result = combine_doc ref, year, opts, stdclass
84
- result || check_bibliocache(ref, year, opts, stdclass)
82
+ result || fetch_routed(ref, year, opts, stdclass, pubid)
85
83
  end
86
84
 
87
85
  # @see Relaton::Db#fetch
@@ -119,7 +117,7 @@ module Relaton
119
117
  # nil if document not found
120
118
  #
121
119
  def fetch_async(ref, year = nil, opts = {}, &block) # rubocop:disable Metrics/AbcSize,Metrics/MethodLength
122
- stdclass = @registry.class_by_ref ref
120
+ stdclass = async_route ref
123
121
  if stdclass
124
122
  unless @queues[stdclass]
125
123
  processor = @registry[stdclass]
@@ -141,23 +139,32 @@ module Relaton
141
139
  end
142
140
  end
143
141
 
142
+ #
143
+ # Fetch with the flavor the caller names, else the routed one.
144
+ #
145
+ # @param stdclass [Symbol, String, nil] a processor short name
146
+ # (`:relaton_iso`, as relaton-cli passes it) or a prefix (`"ISO"`)
147
+ # @raise [Relaton::UnknownReferenceError] no flavor is named and none
148
+ # recognizes the reference
149
+ #
144
150
  def fetch_std(code, year = nil, stdclass = nil, opts = {})
145
- std = nil
146
- @registry.processors.each do |name, processor|
147
- std = name if processor.prefix == stdclass
148
- end
149
- std = @registry.class_by_ref(code) or return nil unless std
151
+ std = named_class(stdclass)
152
+ # A flavor the caller names is the one asked: no co-publisher.
153
+ return check_bibliocache(code, year, opts, std) if std
150
154
 
151
- check_bibliocache(code, year, opts, std)
155
+ std, pubid = @registry.route(code)
156
+ fetch_routed(code, year, opts, std, pubid)
152
157
  end
153
158
 
154
159
  # The document identifier class corresponding to the given code
155
160
  # @param code [String]
156
161
  # @return [Array]
157
162
  def docid_type(code)
158
- stdclass = @registry.class_by_ref(code) or return [nil, code]
163
+ stdclass, = @registry.route(code)
159
164
  _, code = strip_id_wrapper(code, stdclass)
160
165
  [@registry[stdclass].idtype, code]
166
+ rescue UnknownReferenceError
167
+ [nil, code]
161
168
  end
162
169
 
163
170
  # @param key [String]
@@ -186,19 +193,42 @@ module Relaton
186
193
 
187
194
  private
188
195
 
189
- def fetch_doc(code, year, opts, processor)
190
- if Db.configuration.use_api then fetch_api(code, year, opts, processor)
191
- else processor.get(code, year, opts)
196
+ # The flavor routing picks for `fetch_async`'s per-flavor queue; nil
197
+ # (logged) when no flavor recognizes the reference.
198
+ def async_route(ref)
199
+ @registry.route(ref).first
200
+ rescue UnknownReferenceError => e
201
+ Util.info e.message, key: ref
202
+ nil
203
+ end
204
+
205
+ # The processor a caller names by short name (`:relaton_iso`) or prefix.
206
+ def named_class(stdclass)
207
+ return unless stdclass
208
+
209
+ short = stdclass.to_sym
210
+ return short if @registry.processors.key?(short)
211
+
212
+ @registry.processors.detect { |_, p| p.prefix == stdclass.to_s }&.first
213
+ end
214
+
215
+ # The flavor's `get` receives the parsed pubid when there is one
216
+ # (relaton#205), else the String. The API request sends the String the
217
+ # caller wrote.
218
+ def fetch_doc(code, year, opts, processor, query = nil)
219
+ if Db.configuration.use_api
220
+ fetch_api(code, year, opts, processor, query)
221
+ else processor.get(query || code, year, opts)
192
222
  end
193
223
  end
194
224
 
195
- def fetch_api(code, year, opts, processor)
225
+ def fetch_api(code, year, opts, processor, query = nil)
196
226
  url = "#{Db.configuration.api_host}" \
197
227
  "/api/v1/document?#{params(code, year, opts)}"
198
228
  rsp = Net::HTTP.get_response URI(url)
199
229
  processor.from_xml rsp.body if rsp.code == "200"
200
230
  rescue Errno::ECONNREFUSED
201
- processor.get(code, year, opts)
231
+ processor.get(query || code, year, opts)
202
232
  end
203
233
 
204
234
  def params(code, year, opts)
@@ -281,15 +311,36 @@ module Relaton
281
311
  # that query is not cached. A processor with no pubid class gets the
282
312
  # legacy string key.
283
313
  #
314
+ # @param pubid [Pubid::Identifier, nil] the reference as routing parsed
315
+ # it, so the processor does not parse it again
284
316
  # @return [Pubid::Identifier, String, nil]
285
317
  #
286
- def cache_key(code, year, opts, stdclass)
318
+ def cache_key(code, year, opts, stdclass, pubid = nil)
287
319
  processor = @registry[stdclass]
288
320
  unless processor.pubid_class
289
321
  return std_id(code, year, opts, stdclass).first
290
322
  end
291
323
 
292
- processor.cache_key(code, year, opts)
324
+ processor.cache_key(code, year, opts, pubid)
325
+ end
326
+
327
+ #
328
+ # The pubid `get` receives, and the cache key folded from it: one parse
329
+ # of the reference serves both (relaton#205). A processor with no pubid
330
+ # class gets the String and the legacy string key.
331
+ #
332
+ # @param pubid [Pubid::Identifier, nil] routing's parse
333
+ # @return [Array(Pubid::Identifier, Object)] the query pubid (nil: `get`
334
+ # gets the String) and the key (nil: not cached)
335
+ #
336
+ def query_and_key(code, year, opts, stdclass, pubid)
337
+ processor = @registry[stdclass]
338
+ unless processor.pubid_class
339
+ return [nil, std_id(code, year, opts, stdclass).first]
340
+ end
341
+
342
+ query = processor.query_pubid(code, opts, pubid)
343
+ [query, query && processor.cache_key(code, year, opts, query)]
293
344
  end
294
345
 
295
346
  # The key of a fetched document, from its primary identifier. The
@@ -317,53 +368,112 @@ module Relaton
317
368
  [prefix, code]
318
369
  end
319
370
 
320
- def bib_retval(entry, stdclass)
321
- if entry && !entry.match?(/^not_found/)
322
- @registry[stdclass].from_xml(entry)
371
+ # The routed flavor, then its co-publishers' flavors (relaton#205).
372
+ def fetch_routed(code, year, opts, stdclass, pubid)
373
+ check_bibliocache(code, year, opts, stdclass, pubid: pubid) ||
374
+ copublisher_fetch(code, year, opts, stdclass, pubid)
375
+ end
376
+
377
+ #
378
+ # A co-published document the lead flavor's catalog does not have:
379
+ # ask the co-publishers' flavors, in the order the parsed pubid holds
380
+ # them (relaton#205; pubid#469, #472). Each one keys, caches and fetches
381
+ # with its own flavor, so a second fetch is answered from the caches.
382
+ # A co-publisher whose flavor gives no pubid for the lead's printed form
383
+ # is skipped. Only that probe parse is rescued: an error from inside the
384
+ # co-publisher's fetch propagates.
385
+ #
386
+ # @return [Relaton::Bib::ItemData, nil]
387
+ #
388
+ def copublisher_fetch(code, year, opts, stdclass, pubid)
389
+ # The reference as the lead read it: no `PREFIX(...)` wrapper, no en dash.
390
+ _, code = strip_id_wrapper(code, stdclass)
391
+ copublisher_classes(code, opts, stdclass, pubid).each do |publisher, co|
392
+ Util.info "Not found; trying co-publisher `#{publisher}`", key: code
393
+ query = copublisher_query(code, opts, co, publisher) or next
394
+ bib = check_bibliocache(code, year, opts, co, pubid: query)
395
+ return bib if bib
396
+ end
397
+ nil
398
+ end
399
+
400
+ # @return [Array<Array(String, Symbol)>] each co-publisher that a flavor
401
+ # other than the lead serves, with that flavor
402
+ def copublisher_classes(code, opts, stdclass, pubid)
403
+ processor = @registry[stdclass]
404
+ query = processor.query_pubid(code, opts, pubid) or return []
405
+
406
+ processor.copublishers(query).filter_map do |publisher|
407
+ co = @registry.class_by_publisher(publisher)
408
+ [publisher, co] if co && co != stdclass
409
+ end.uniq(&:last)
410
+ end
411
+
412
+ # The co-publisher flavor's own parse of the reference, or nil when it
413
+ # gives none (a parse error, or a flavor that reads it as a miss: IEEE),
414
+ # which is logged.
415
+ #
416
+ # @return [Pubid::Identifier, nil]
417
+ def copublisher_query(code, opts, stdclass, publisher)
418
+ query = begin
419
+ @registry[stdclass].query_pubid(code, opts)
420
+ rescue ::Pubid::Errors::Error, Parslet::ParseFailed
421
+ nil
323
422
  end
423
+ return query if query
424
+
425
+ Util.info "Not found; co-publisher `#{publisher}` cannot read it",
426
+ key: code
427
+ nil
428
+ end
429
+
430
+ def bib_retval(entry, stdclass)
431
+ @registry[stdclass].from_xml(entry) if entry.is_a?(String)
324
432
  end
325
433
 
326
- def check_bibliocache(code, year, opts, stdclass) # rubocop:disable Metrics/AbcSize,Metrics/CyclomaticComplexity,Metrics/MethodLength,Metrics/PerceivedComplexity
434
+ def check_bibliocache(code, year, opts, stdclass, pubid: nil) # rubocop:disable Metrics/AbcSize,Metrics/CyclomaticComplexity,Metrics/MethodLength,Metrics/PerceivedComplexity
327
435
  _, searchcode = strip_id_wrapper(code, stdclass)
328
- id = cache_key(searchcode, year, opts, stdclass)
436
+ query, id = query_and_key(searchcode, year, opts, stdclass, pubid)
329
437
  db = id && (@local_db || @db)
330
438
  altdb = @local_db && @db ? @db : nil
331
439
  if db.nil?
332
440
  return if opts[:fetch_db]
333
441
 
334
- bibentry = new_bib_entry(searchcode, year, opts, stdclass)
442
+ bibentry = new_bib_entry(searchcode, year, opts, stdclass, query: query)
335
443
  return bib_retval(bibentry, stdclass)
336
444
  end
337
445
  if date_range?(opts)
338
- return check_date_range(searchcode, id, year, opts, stdclass)
446
+ return check_date_range(searchcode, id, year, opts, stdclass, query)
339
447
  end
340
448
 
341
449
  @semaphore.synchronize { db.expire id, year }
342
450
  if altdb
343
- return bib_retval(altdb[id], stdclass) if opts[:fetch_db]
451
+ return bib_retval(altdb.read(id), stdclass) if opts[:fetch_db]
344
452
 
345
453
  @semaphore.synchronize do
346
454
  db.clone_entry id, altdb if altdb.valid_entry? id, year
347
455
  end
348
- new_bib_entry(searchcode, year, opts, stdclass, db: db, id: id)
456
+ new_bib_entry(searchcode, year, opts, stdclass, db: db, id: id,
457
+ query: query)
349
458
  @semaphore.synchronize do
350
459
  altdb.clone_entry(id, db) if !altdb.valid_entry?(id, year)
351
460
  end
352
461
  else
353
- return bib_retval(db[id], stdclass) if opts[:fetch_db]
462
+ return bib_retval(db.read(id), stdclass) if opts[:fetch_db]
354
463
 
355
- new_bib_entry(searchcode, year, opts, stdclass, db: db, id: id)
464
+ new_bib_entry(searchcode, year, opts, stdclass, db: db, id: id,
465
+ query: query)
356
466
  end
357
- bib_retval(db[id], stdclass)
467
+ bib_retval(db.read(id), stdclass)
358
468
  end
359
469
 
360
470
  def new_bib_entry(code, year, opts, stdclass, **args)
361
- entry = @semaphore.synchronize { args[:db] && args[:db][args[:id]] }
471
+ entry = @semaphore.synchronize { args[:db]&.read(args[:id]) }
362
472
  if !entry || opts[:no_cache]
363
473
  return fetch_entry(code, year, opts, stdclass, **args)
364
474
  end
365
475
 
366
- if entry&.match?(/^not_found/)
476
+ if entry.is_a?(NotFound)
367
477
  Util.info "not found in cache, if you wish to " \
368
478
  "ignore cache please use `no-cache` option.", key: code
369
479
  return
@@ -373,15 +483,16 @@ module Relaton
373
483
 
374
484
  def fetch_entry(code, year, opts, stdclass, **args) # rubocop:disable Metrics/AbcSize
375
485
  processor = @registry[stdclass]
376
- bib = net_retry(code, year, opts, processor, opts.fetch(:retries, 1))
486
+ bib = net_retry(code, year, opts, processor, opts.fetch(:retries, 1),
487
+ args[:query])
377
488
  entry = bib_entry bib
378
489
  return entry if args[:db].nil?
379
490
 
380
491
  # `no_cache` refreshes a cached entry, but a failed fetch does not
381
- # replace a cached document with `not_found`.
492
+ # replace a cached document with a not-found entry.
382
493
  refresh = opts[:no_cache] && bib.respond_to?(:to_xml)
383
494
  @semaphore.synchronize do
384
- if refresh || !args[:db][args[:id]]
495
+ if refresh || !args[:db].read(args[:id])
385
496
  save_bib args[:db], args[:id], bib, entry, stdclass
386
497
  end
387
498
  end
@@ -404,7 +515,7 @@ module Relaton
404
515
  # with the range, and its answer is cached under its own identifier only:
405
516
  # the query row keeps pointing to the latest edition.
406
517
  #
407
- def check_date_range(code, key, year, opts, stdclass) # rubocop:disable Metrics/AbcSize,Metrics/MethodLength,Metrics/CyclomaticComplexity,Metrics/PerceivedComplexity
518
+ def check_date_range(code, key, year, opts, stdclass, query = nil) # rubocop:disable Metrics/AbcSize,Metrics/MethodLength,Metrics/CyclomaticComplexity,Metrics/PerceivedComplexity
408
519
  caches = [@local_db, @db].compact
409
520
  cached = caches.flat_map { |c| c.candidates(key) }.filter_map do |_, xml|
410
521
  date = published_date(xml)
@@ -414,7 +525,8 @@ module Relaton
414
525
  return if opts[:fetch_db]
415
526
 
416
527
  processor = @registry[stdclass]
417
- bib = net_retry(code, year, opts, processor, opts.fetch(:retries, 1))
528
+ bib = net_retry(code, year, opts, processor, opts.fetch(:retries, 1),
529
+ query)
418
530
  return unless bib.respond_to?(:to_xml)
419
531
 
420
532
  entry = bib_entry bib
@@ -427,19 +539,21 @@ module Relaton
427
539
  bib_retval entry, stdclass
428
540
  end
429
541
 
430
- def net_retry(code, year, opts, processor, retries)
431
- fetch_doc code, year, opts, processor
542
+ # @param query [Pubid::Identifier, nil] what `get` receives in place of
543
+ # the String +code+ (Db#query_and_key)
544
+ def net_retry(code, year, opts, processor, retries, query = nil)
545
+ fetch_doc code, year, opts, processor, query
432
546
  rescue Relaton::RequestError => e
433
547
  raise e unless retries > 1
434
548
 
435
- net_retry(code, year, opts, processor, retries - 1)
549
+ net_retry(code, year, opts, processor, retries - 1, query)
436
550
  end
437
551
 
438
552
  def bib_entry(bib)
439
553
  if bib.respond_to?(:to_xml)
440
554
  bib.to_xml(bibdata: true)
441
555
  else
442
- "not_found #{Date.today}"
556
+ NotFound.new(fetched: Date.today.to_s)
443
557
  end
444
558
  end
445
559
 
@@ -43,7 +43,8 @@ module Relaton
43
43
  # `ECMA-418` does not match `ECMA-418-1`. The class must be identical,
44
44
  # which keeps `ECMA-100` and `ECMA TR/100` apart.
45
45
  #
46
- # @param ref [String] the ECMA reference (e.g. "ECMA-6", "ECMA-269 ed3 vol2")
46
+ # @param ref [String, Pubid::Ecma::Identifier] the ECMA reference
47
+ # (e.g. "ECMA-6", "ECMA-269 ed3 vol2"), or its parse
47
48
  #
48
49
  # @return [Array<Hash>] matching index rows
49
50
  #
@@ -62,7 +63,7 @@ module Relaton
62
63
  # regex accepted, so every reference shape the flavor has to handle
63
64
  # parses without normalization here.
64
65
  #
65
- # @param ref [String]
66
+ # @param ref [String, Pubid::Ecma::Identifier]
66
67
  # @return [Pubid::Ecma::Identifier, nil]
67
68
  #
68
69
  # An unrecognized reference **raises**; like ISO, ETSI and 3GPP we let it
@@ -73,20 +74,24 @@ module Relaton
73
74
  # Rescuing here would collapse "this identifier is malformed" into "no
74
75
  # such document", leaving a caller unable to tell them apart.
75
76
  def parse_ref(ref)
77
+ # A pubid from Relaton::Db (relaton#205) is used as it is.
78
+ return ref unless ref.is_a?(String)
79
+
76
80
  ::Pubid::Ecma::Identifier.parse ref.to_s.strip
77
81
  end
78
82
 
79
- # @param code [String] the ECMA standard Code to look up (e..g "ECMA-6")
83
+ # @param ref [String, Pubid::Ecma::Identifier] the ECMA reference
84
+ # (e.g. "ECMA-6"), or its parse from Relaton::Db (relaton#205)
80
85
  # @param year [String] not used
81
86
  # @param opts [Hash] not used
82
87
  # @return [Relaton::Ecma::ItemData] Relaton of reference
83
- def get(code, _year = nil, _opts = {})
84
- Util.info "Fetching from Relaton repository ...", key: code
85
- result = fetch_doc(code)
88
+ def get(ref, _year = nil, _opts = {})
89
+ Util.info "Fetching from Relaton repository ...", key: ref.to_s
90
+ result = fetch_doc(ref)
86
91
  if result
87
- Util.info "Found: `#{result.docidentifier.first.content}`", key: code
92
+ Util.info "Found: `#{result.docidentifier.first.content}`", key: ref.to_s
88
93
  else
89
- Util.info "Not found.", key: code
94
+ Util.info "Not found.", key: ref.to_s
90
95
  end
91
96
  result
92
97
  end
@@ -106,7 +111,7 @@ module Relaton
106
111
  #
107
112
  # `r[:file]` breaks the tie, because the index sort is not stable.
108
113
  #
109
- # @param ref [String]
114
+ # @param ref [String, Pubid::Ecma::Identifier]
110
115
  # @return [Hash, nil]
111
116
  #
112
117
  def best_match(ref)
@@ -127,8 +132,8 @@ module Relaton
127
132
  edition.to_s.split(".").map(&:to_i)
128
133
  end
129
134
 
130
- def fetch_doc(code)
131
- row = best_match code
135
+ def fetch_doc(ref)
136
+ row = best_match ref
132
137
  return unless row
133
138
 
134
139
  url = "#{ENDPOINT}#{row[:file]}"
@@ -137,11 +142,11 @@ module Relaton
137
142
  rescue Mechanize::ResponseCodeError => e
138
143
  return if e.response_code == "404"
139
144
 
140
- raise Relaton::RequestError, "No document found for #{code} reference. #{e.message}"
145
+ raise Relaton::RequestError, "No document found for #{ref} reference. #{e.message}"
141
146
  rescue Mechanize::RedirectLimitReachedError, Timeout::Error,
142
147
  Mechanize::UnauthorizedError, Mechanize::UnsupportedSchemeError,
143
148
  Mechanize::ResponseReadError, Mechanize::ChunkedTerminationError => e
144
- raise Relaton::RequestError, "No document found for #{code} reference. #{e.message}"
149
+ raise Relaton::RequestError, "No document found for #{ref} reference. #{e.message}"
145
150
  end
146
151
  end
147
152
  end
@@ -6,14 +6,15 @@ module Relaton
6
6
  module Bibliography
7
7
  SOURCE = "https://raw.githubusercontent.com/relaton/relaton-data-etsi/refs/heads/v2/"
8
8
 
9
- # @param text [String]
9
+ # @param ref [String, ::Pubid::Etsi::Identifier]
10
10
  # @return [Relaton::Etsi::ItemData, nil]
11
- def search(text) # rubocop:disable Metrics/MethodLength, Metrics/AbcSize
11
+ def search(ref) # rubocop:disable Metrics/MethodLength, Metrics/AbcSize
12
12
  # An unrecognized reference raises Pubid::Errors::ParseError; like
13
13
  # ISO we let it propagate — the CLI turns it into a friendly message
14
14
  # and API callers rescue it themselves. Valid partial refs parse with
15
15
  # the omitted refinements (version/date/part) left blank.
16
- pubid = ::Pubid::Etsi.parse text
16
+ # A parsed pubid comes from Relaton::Db (relaton#205); it is only read.
17
+ pubid = ref.is_a?(String) ? ::Pubid::Etsi.parse(ref) : ref
17
18
 
18
19
  index = Relaton::Index.find_or_create :etsi, url: "#{SOURCE}#{INDEXFILE}.zip", file: "#{INDEXFILE}.yaml",
19
20
  pubid_class: ::Pubid::Etsi::Identifier
@@ -89,19 +90,19 @@ module Relaton
89
90
  [(id.version&.version).to_s.split(".").map(&:to_i), id.date.to_s]
90
91
  end
91
92
 
92
- # @param ref [String] the ETSI standard Code to look up
93
+ # @param ref [String, ::Pubid::Etsi::Identifier] the ETSI standard Code to look up
93
94
  # @param year [String, nil] year
94
95
  # @param opts [Hash] options
95
96
  # @return [Relaton::Etsi::ItemData, nil]
96
97
  def get(ref, _year = nil, _opts = {})
97
- Util.info "Fetching from Relaton repository ...", key: ref
98
+ Util.info "Fetching from Relaton repository ...", key: ref.to_s
98
99
  result = search(ref)
99
100
  unless result
100
- Util.info "Not found.", key: ref
101
+ Util.info "Not found.", key: ref.to_s
101
102
  return
102
103
  end
103
104
 
104
- Util.info "Found: `#{result.docidentifier[0].content}`", key: ref
105
+ Util.info "Found: `#{result.docidentifier[0].content}`", key: ref.to_s
105
106
  result
106
107
  end
107
108
 
@@ -45,22 +45,117 @@ module Relaton
45
45
  report_errors
46
46
  end
47
47
 
48
- def fetch_remaining_pages(first_page)
48
+ #
49
+ # Fetch pages 2..N. A page that stays bad after its retries is fetched
50
+ # again after the last page; if it is still bad, the crawl fails. A page
51
+ # is never skipped: relaton-data-etsi deletes data/ before the crawl and
52
+ # commits what it writes, so a skipped page would unpublish its documents.
53
+ #
54
+ # N comes from page 1's total_count, but the result set is sorted by
55
+ # deliverable number, so a document published during the crawl shifts the
56
+ # later pages by one. The crawl therefore reads on past N while the pages
57
+ # stay full (at most MAX_EXTRA_PAGES), and one page past a deferred last
58
+ # page; a record read twice overwrites its own file.
59
+ #
60
+ def fetch_remaining_pages(first_page) # rubocop:disable Metrics/MethodLength
61
+ total_pages = last = last_page(first_page)
62
+ deferred = []
63
+ page = 2
64
+ while page <= last
65
+ check_extra_pages page, total_pages
66
+ size = fetch_next_page(page, deferred)
67
+ break if size&.zero?
68
+
69
+ last = page + 1 if page == last && read_on?(size, page, total_pages)
70
+ page += 1
71
+ end
72
+ fetch_deferred_pages(deferred, total_pages)
73
+ end
74
+
75
+ def last_page(first_page)
49
76
  total = first_page.first ? first_page.first["total_count"].to_i : 0
50
- total_pages = (total / PAGE_SIZE.to_f).ceil
51
- (2..total_pages).each do |page|
52
- records = fetch_page(page)
53
- break if records.empty?
77
+ last = (total / PAGE_SIZE.to_f).ceil
78
+ first_page.size >= PAGE_SIZE ? [last, 2].max : last
79
+ end
80
+
81
+ #
82
+ # Whether to read the page after the last one: after a full page, or
83
+ # after a deferred page inside the total_count range. A deferred page
84
+ # past that range does not extend it, so a server that answers bad
85
+ # bodies past the end cannot keep the crawl going.
86
+ #
87
+ def read_on?(size, page, total_pages)
88
+ size ? size >= PAGE_SIZE : page <= total_pages
89
+ end
90
+
91
+ def check_extra_pages(page, total_pages)
92
+ return if page <= total_pages + MAX_EXTRA_PAGES
93
+
94
+ raise BadPage, "ETSI page #{page} is still full #{MAX_EXTRA_PAGES} " \
95
+ "pages past total_count (#{total_pages} pages)."
96
+ end
97
+
98
+ #
99
+ # @return [Integer, nil] the number of records on the page; nil for a
100
+ # deferred page
101
+ #
102
+ def fetch_next_page(page, deferred)
103
+ records = fetch_page(page)
104
+ process_records(records)
105
+ records.size
106
+ rescue BadPage => e
107
+ Util.warn "#{e.message} Fetching it again after the last page."
108
+ deferred << page
109
+ nil
110
+ end
54
111
 
55
- process_records(records)
112
+ def fetch_deferred_pages(pages, total_pages)
113
+ return if pages.empty?
114
+
115
+ sleep DEFERRED_DELAY
116
+ failed = pages.filter_map do |page|
117
+ refetch_page page, total_pages
118
+ rescue BadPage => e
119
+ e.message
56
120
  end
121
+ raise BadPage, failed.join(" ") if failed.any?
122
+ end
123
+
124
+ # @return [String, nil] the failure message, nil on success
125
+ def refetch_page(page, total_pages)
126
+ records = fetch_page(page)
127
+ if records.empty? && page <= total_pages
128
+ return "ETSI page #{page} is empty on the re-fetch."
129
+ end
130
+
131
+ process_records records
132
+ nil
57
133
  end
58
134
 
59
135
  def fetch_page(page)
60
136
  date = Time.now.to_date + 1
61
137
  timestamp = (Time.now.to_f * 1000).to_i
62
138
  url = format(SOURCEURL, page: page, date: date, timestamp: timestamp)
63
- JSON.parse(fetch_with_retry(url))
139
+ fetch_with_retry(url) { |body| parse_page(page, body) }
140
+ end
141
+
142
+ #
143
+ # ETSI's data.php sometimes answers with PHP's `Array` text in place of
144
+ # JSON (relaton-data-etsi crawl 36761804954); a re-fetch gets JSON.
145
+ #
146
+ # @raise [BadPage] when the body is not a JSON array
147
+ #
148
+ def parse_page(page, body)
149
+ records = JSON.parse(body)
150
+ return records if records.is_a?(Array)
151
+
152
+ raise BadPage, bad_page_message(page, body)
153
+ rescue JSON::ParserError
154
+ raise BadPage, bad_page_message(page, body)
155
+ end
156
+
157
+ def bad_page_message(page, body)
158
+ "ETSI page #{page} is not a JSON array: #{body.to_s[0, 200].inspect}."
64
159
  end
65
160
 
66
161
  def process_records(records)
@@ -92,16 +187,30 @@ module Relaton
92
187
  "Published"
93
188
  end
94
189
 
190
+ # A page body that is not a JSON array. See #parse_page.
191
+ class BadPage < StandardError; end
192
+
193
+ # Seconds to wait before a bad page is fetched again after the last page.
194
+ DEFERRED_DELAY = 60
195
+
196
+ # Full pages read past page 1's total_count before the crawl fails.
197
+ MAX_EXTRA_PAGES = 10
198
+
95
199
  NETWORK_ERRORS = [
96
200
  Mechanize::Error, Net::OpenTimeout, Net::ReadTimeout,
97
201
  SocketError, Errno::ECONNRESET
98
202
  ].freeze
99
203
 
204
+ #
205
+ # @yield [String] the body; the block's result is returned, and a BadPage
206
+ # it raises is retried like a network error
207
+ #
100
208
  def fetch_with_retry(url, retries: 3, delay: 2) # rubocop:disable Metrics/MethodLength
101
209
  attempt = 0
102
210
  begin
103
- Mechanize.new.get(url).body
104
- rescue *NETWORK_ERRORS => e
211
+ body = Mechanize.new.get(url).body
212
+ block_given? ? yield(body) : body
213
+ rescue *NETWORK_ERRORS, BadPage => e
105
214
  attempt += 1
106
215
  if attempt <= retries
107
216
  Util.info "Fetch failed (#{e.message}), " \