relaton 3.0.0.pre.alpha.4 → 3.0.0.pre.alpha.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. checksums.yaml +4 -4
  2. data/lib/relaton/adobe/processor.rb +9 -0
  3. data/lib/relaton/bib/converter/csl.rb +110 -0
  4. data/lib/relaton/bib/converter/ris.rb +104 -0
  5. data/lib/relaton/bib/converter/titles.rb +24 -0
  6. data/lib/relaton/bib/item_data.rb +9 -0
  7. data/lib/relaton/bib/model/item.rb +6 -6
  8. data/lib/relaton/bib/sanitizer.rb +71 -30
  9. data/lib/relaton/bib.rb +10 -0
  10. data/lib/relaton/bipm/processor.rb +1 -0
  11. data/lib/relaton/bipm/rawdata_bipm_metrologia/affiliations.rb +6 -6
  12. data/lib/relaton/bipm/si_brochure_parser.rb +2 -2
  13. data/lib/relaton/bsi/processor.rb +5 -0
  14. data/lib/relaton/ccsds/bibliography.rb +1 -0
  15. data/lib/relaton/ccsds/processor.rb +12 -0
  16. data/lib/relaton/cen/hit_collection.rb +1 -1
  17. data/lib/relaton/cen/processor.rb +5 -0
  18. data/lib/relaton/cen/scraper.rb +9 -9
  19. data/lib/relaton/cie/data_fetcher.rb +18 -18
  20. data/lib/relaton/cie/processor.rb +1 -0
  21. data/lib/relaton/cie.rb +1 -1
  22. data/lib/relaton/cloud.rb +127 -0
  23. data/lib/relaton/core/hit_collection.rb +6 -10
  24. data/lib/relaton/core/processor.rb +68 -0
  25. data/lib/relaton/db/cache.rb +444 -148
  26. data/lib/relaton/db/cache_entry.rb +33 -0
  27. data/lib/relaton/db/registry.rb +52 -0
  28. data/lib/relaton/db.rb +131 -72
  29. data/lib/relaton/doi/crossref.rb +23 -6
  30. data/lib/relaton/doi/processor.rb +1 -0
  31. data/lib/relaton/easc/processor.rb +1 -0
  32. data/lib/relaton/ecma/data_fetcher.rb +1 -1
  33. data/lib/relaton/ecma/data_parser.rb +1 -1
  34. data/lib/relaton/ecma/edition_parser.rb +2 -2
  35. data/lib/relaton/ecma/memento_parser.rb +4 -4
  36. data/lib/relaton/ecma/standard_parser.rb +6 -6
  37. data/lib/relaton/etsi/processor.rb +1 -0
  38. data/lib/relaton/gb/gb_scraper.rb +5 -5
  39. data/lib/relaton/gb/scraper.rb +16 -16
  40. data/lib/relaton/gb/sec_scraper.rb +8 -8
  41. data/lib/relaton/gb/t_scraper.rb +5 -5
  42. data/lib/relaton/gost/processor.rb +1 -0
  43. data/lib/relaton/iala/processor.rb +1 -0
  44. data/lib/relaton/iana/data_fetcher.rb +3 -3
  45. data/lib/relaton/iana/parser.rb +8 -4
  46. data/lib/relaton/iana/processor.rb +9 -0
  47. data/lib/relaton/iec/data_parser.rb +24 -9
  48. data/lib/relaton/iec/processor.rb +6 -0
  49. data/lib/relaton/iec.rb +1 -1
  50. data/lib/relaton/ieee/data_fetcher.rb +1 -1
  51. data/lib/relaton/ieee/processor.rb +8 -0
  52. data/lib/relaton/ietf/data_fetcher.rb +1 -1
  53. data/lib/relaton/ietf/processor.rb +1 -0
  54. data/lib/relaton/ietf/rfc/entry.rb +15 -19
  55. data/lib/relaton/iho/processor.rb +1 -0
  56. data/lib/relaton/index/file_io.rb +2 -2
  57. data/lib/relaton/index/pool.rb +4 -3
  58. data/lib/relaton/index/shard_source.rb +1 -1
  59. data/lib/relaton/index/type.rb +2 -5
  60. data/lib/relaton/isbn/open_library.rb +11 -7
  61. data/lib/relaton/isbn/processor.rb +10 -0
  62. data/lib/relaton/iso/data_parser.rb +2 -2
  63. data/lib/relaton/iso/processor.rb +5 -0
  64. data/lib/relaton/iso/scraper.rb +20 -20
  65. data/lib/relaton/itu/bibliography.rb +99 -13
  66. data/lib/relaton/itu/data_crawler_r.rb +2 -2
  67. data/lib/relaton/itu/hit_collection.rb +64 -52
  68. data/lib/relaton/itu/processor.rb +1 -0
  69. data/lib/relaton/itu/scraper.rb +13 -3
  70. data/lib/relaton/itu.rb +0 -2
  71. data/lib/relaton/jis/data_fetcher.rb +4 -4
  72. data/lib/relaton/jis/processor.rb +1 -0
  73. data/lib/relaton/jis/scraper.rb +8 -8
  74. data/lib/relaton/oasis/browser_agent.rb +2 -2
  75. data/lib/relaton/oasis/data_parser.rb +5 -5
  76. data/lib/relaton/oasis/data_parser_utils.rb +2 -2
  77. data/lib/relaton/oasis/data_part_parser.rb +7 -7
  78. data/lib/relaton/ogc/processor.rb +7 -0
  79. data/lib/relaton/oiml/processor.rb +1 -0
  80. data/lib/relaton/omg/scraper.rb +11 -11
  81. data/lib/relaton/omg.rb +1 -1
  82. data/lib/relaton/plateau/processor.rb +1 -0
  83. data/lib/relaton/un/bibliography.rb +21 -10
  84. data/lib/relaton/un/processor.rb +1 -0
  85. data/lib/relaton/version.rb +1 -1
  86. data/lib/relaton/w3c/processor.rb +1 -0
  87. data/lib/relaton.rb +13 -0
  88. metadata +48 -16
  89. data/lib/relaton/itu/pubid.rb +0 -199
@@ -0,0 +1,33 @@
1
+ require "lutaml/model"
2
+
3
+ module Relaton
4
+ class Db
5
+ # One row of the cache index. A row is keyed either by a pubid (`id`, its
6
+ # `to_hash`) or, for a processor with no pubid class, by a string (`key`).
7
+ # `file` names the document in the doc store; several rows can share one
8
+ # file (a query row and the row of the document it returned). A
9
+ # `not_found` row has no file.
10
+ class CacheEntry < Lutaml::Model::Serializable
11
+ DOC = "doc".freeze
12
+ NOT_FOUND = "not_found".freeze
13
+
14
+ attribute :id, :hash
15
+ attribute :key, :string
16
+ attribute :status, :string
17
+ attribute :file, :string
18
+ attribute :fetched, :string
19
+
20
+ key_value do
21
+ map "id", to: :id
22
+ map "key", to: :key
23
+ map "status", to: :status
24
+ map "file", to: :file
25
+ map "fetched", to: :fetched
26
+ end
27
+
28
+ def not_found?
29
+ status == NOT_FOUND
30
+ end
31
+ end
32
+ end
33
+ end
@@ -92,6 +92,58 @@ module Relaton
92
92
  end
93
93
 
94
94
  #
95
+ # The processor whose pubid class the identifier belongs to. Only the
96
+ # generic `Pubid::AllPartsIdentifier`, which belongs to no flavor, is
97
+ # matched through the document it wraps (`#root`). Any other identifier
98
+ # is matched by its own class: the `#root` of an adoption is the adopted
99
+ # document, so `CEN ISO/TS 21003-7` would otherwise be filed as ISO.
100
+ #
101
+ # @param pubid [Pubid::Identifier]
102
+ # @return [Relaton::Core::Processor, nil]
103
+ #
104
+ def processor_by_pubid(pubid)
105
+ generic = pubid.instance_of?(::Pubid::AllPartsIdentifier)
106
+ id = generic ? pubid.root : pubid
107
+ processors.values.detect do |processor|
108
+ klass = processor.pubid_class
109
+ klass && id.is_a?(klass)
110
+ end
111
+ end
112
+
113
+ #
114
+ # Pubid namespaces whose spelling differs from the processor short
115
+ # name.
116
+ PUBID_FLAVOR_ALIASES = { "cencenelec" => "cen", "tgpp" => "3gpp" }.freeze
117
+
118
+ # Flavors routed by the parsed Pubid class (relaton#205, pilot). Only
119
+ # processors on this list answer parse-first routing: an allowlist,
120
+ # not a guess. Pubid::Un is deliberately absent — its Document
121
+ # identifier parses DOI-shaped strings ("10.17487/RFC3986") that must
122
+ # not be hijacked away from their current handling. DOI and ISBN are
123
+ # present: their canonical forms carry their own token (`doi:…`,
124
+ # `ISBN …`), which pubid detects exactly as the prefix regex routes it.
125
+ PARSE_ROUTED_FLAVORS = %w[
126
+ bipm bs cencenelec calconnect cc ccsds cen cie csa doi
127
+ ecma ecs etsi gost iala iana iec ieee ietf isbn iso itu
128
+ jcgm jis nist oasis ogc oiml plateau w3c xsf 3gpp
129
+ ].freeze
130
+
131
+ # Find the processor that owns the parsed Pubid's flavor. The
132
+ # namespace is matched against the processor short name, with the
133
+ # alias table for the spellings that differ.
134
+ #
135
+ # @param pubid [Pubid::Core::Identifier] parsed query
136
+ # @return [Symbol, nil] standard class name
137
+ #
138
+ def class_by_pubid(pubid)
139
+ ns = pubid.class.name.split("::")[1]&.downcase
140
+ flavor = PUBID_FLAVOR_ALIASES.fetch(ns, ns)
141
+ return nil unless PARSE_ROUTED_FLAVORS.include?(flavor)
142
+
143
+ key = "relaton_#{flavor}".to_sym
144
+ processors.key?(key) ? key : nil
145
+ end
146
+
95
147
  # Find processor by refernce or prefix
96
148
  #
97
149
  # @param ref [String] reference or prefix
data/lib/relaton/db.rb CHANGED
@@ -1,11 +1,13 @@
1
1
  require "relaton/bib"
2
2
  require "yaml"
3
3
  require "net/http"
4
- require "nokogiri"
4
+ require "moxml"
5
5
  require "fileutils"
6
6
  require "date"
7
7
 
8
8
  module Relaton
9
+ autoload :Cloud, "relaton/cloud"
10
+
9
11
  class Db
10
12
  # @param global_cache [String] directory of global DB
11
13
  # @param local_cache [String] directory of local DB
@@ -64,7 +66,14 @@ module Relaton
64
66
  ##
65
67
  def fetch(text, year = nil, opts = {})
66
68
  reference = text.strip
67
- stdclass = @registry.class_by_ref(reference) || return
69
+ require "pubid"
70
+ parsed = begin
71
+ Pubid.parse(reference)
72
+ rescue Pubid::Errors::Error, Parslet::ParseFailed
73
+ nil
74
+ end
75
+ stdclass = (parsed && @registry.class_by_pubid(parsed)) ||
76
+ @registry.class_by_ref(reference) || return
68
77
  processor = @registry[stdclass]
69
78
  ref = if processor.respond_to?(:urn_to_code)
70
79
  processor.urn_to_code(reference)&.first
@@ -90,8 +99,8 @@ module Relaton
90
99
  result = []
91
100
  db = @db || @local_db
92
101
  if db
93
- result += db.all do |file, xml|
94
- search_xml file, xml, text, edition, year
102
+ result += db.all do |processor, xml|
103
+ search_xml processor, xml, text, edition, year
95
104
  end.compact
96
105
  end
97
106
  result
@@ -169,9 +178,10 @@ module Relaton
169
178
  # @return [String]
170
179
  def to_xml
171
180
  db = @local_db || @db || return
172
- Nokogiri::XML::Builder.new(encoding: "UTF-8") do |xml|
173
- xml.documents { xml.parent.add_child db.all.join(" ") }
174
- end.to_xml
181
+ parts = db.all.join(" ")
182
+ Moxml.parse(
183
+ "<?xml version=\"1.0\" encoding=\"UTF-8\"?>\n<documents>#{parts}</documents>",
184
+ ).to_xml(indent: 0, expand_empty: false)
175
185
  end
176
186
 
177
187
  private
@@ -195,17 +205,15 @@ module Relaton
195
205
  opts.merge(code: code, year: year).map { |k, v| "#{k}=#{v}" }.join "&"
196
206
  end
197
207
 
198
- def search_xml(file, xml, text, edition, year)
208
+ def search_xml(processor, xml, text, edition, year)
209
+ return unless processor
199
210
  return unless text.nil? || match_xml_text?(xml, text)
200
211
 
201
- search_edition_year(file, xml, edition, year)
212
+ search_edition_year(processor, xml, edition, year)
202
213
  end
203
214
 
204
- def search_edition_year(file, content, edition, year) # rubocop:disable Metrics/AbcSize,Metrics/CyclomaticComplexity,Metrics/PerceivedComplexity
205
- processor = @registry.processor_by_ref(file.split("/")[-2])
206
- item = if file.match?(/xml$/) then processor.from_xml(content)
207
- else processor.from_yaml(content)
208
- end
215
+ def search_edition_year(processor, content, edition, year) # rubocop:disable Metrics/CyclomaticComplexity,Metrics/PerceivedComplexity
216
+ item = processor.from_xml(content)
209
217
  item if (edition.nil? || item.edition.content == edition) && (year.nil? ||
210
218
  item.date.detect do |d|
211
219
  d.type == "published" && d.at.to_date.year.to_s == year.to_s
@@ -240,7 +248,9 @@ module Relaton
240
248
  if updates
241
249
  doc.relation << Bib::Relation.new(bibitem: updates, type: "updates")
242
250
  end
243
- divider = stdclass == :relaton_itu ? " " : "/"
251
+ # The supplement joins its base as the flavor's identifier spells it:
252
+ # `NIST SP 800-38A Add`, not `/Add`, which pubid does not parse.
253
+ divider = %i[relaton_itu relaton_nist].include?(stdclass) ? " " : "/"
244
254
  refs[1..].each_with_object(doc) do |c, d|
245
255
  bib = check_bibliocache(ref + divider + c, year, opts, stdclass)
246
256
  if bib
@@ -251,18 +261,52 @@ module Relaton
251
261
  end
252
262
  end
253
263
 
264
+ # The legacy string cache key, for a processor with no pubid class. The
265
+ # publication date range is not part of it: it filters, it is not
266
+ # identity.
254
267
  def std_id(code, year, opts, stdclass)
255
268
  prefix, code = strip_id_wrapper(code, stdclass)
256
269
  ret = code
257
270
  ret += (stdclass == :relaton_gb ? "-" : ":") + year if year
258
271
  ret += " (all parts)" if opts[:all_parts]
259
- after = opts[:publication_date_after]
260
- ret += " after-#{after}" if after
261
- before = opts[:publication_date_before]
262
- ret += " before-#{before}" if before
263
272
  ["#{prefix}(#{ret.strip})", code]
264
273
  end
265
274
 
275
+ #
276
+ # The cache key of a query: the flavor's parsed pubid, with the `year`
277
+ # and `all_parts` options folded in. A reference the flavor cannot parse
278
+ # raises `Pubid::Errors::ParseError`. Nil when the flavor has a pubid
279
+ # class but gives no key for this query (a miss by the flavor's own rule,
280
+ # or a query whose answer the cache cannot hold, such as a CCSDS format):
281
+ # that query is not cached. A processor with no pubid class gets the
282
+ # legacy string key.
283
+ #
284
+ # @return [Pubid::Identifier, String, nil]
285
+ #
286
+ def cache_key(code, year, opts, stdclass)
287
+ processor = @registry[stdclass]
288
+ unless processor.pubid_class
289
+ return std_id(code, year, opts, stdclass).first
290
+ end
291
+
292
+ processor.cache_key(code, year, opts)
293
+ end
294
+
295
+ # The key of a fetched document, from its primary identifier. The
296
+ # identifier is data, so an unparseable one gives no key.
297
+ def item_key(bib, stdclass)
298
+ docid = bib.docidentifier.detect(&:primary) || bib.docidentifier.first
299
+ return unless docid&.content
300
+
301
+ cache_key docid.content, nil, {}, stdclass
302
+ rescue ::Pubid::Errors::Error, Parslet::ParseFailed
303
+ nil
304
+ end
305
+
306
+ def date_range?(opts)
307
+ opts[:publication_date_before] || opts[:publication_date_after]
308
+ end
309
+
266
310
  def strip_id_wrapper(code, stdclass)
267
311
  prefix = @registry[stdclass].prefix
268
312
  code =
@@ -280,23 +324,9 @@ module Relaton
280
324
  end
281
325
 
282
326
  def check_bibliocache(code, year, opts, stdclass) # rubocop:disable Metrics/AbcSize,Metrics/CyclomaticComplexity,Metrics/MethodLength,Metrics/PerceivedComplexity
283
- if opts[:publication_date_before] || opts[:publication_date_after]
284
- base_opts = opts.except(
285
- :publication_date_before, :publication_date_after
286
- )
287
- base_id, = std_id(code, year, base_opts, stdclass)
288
- db = @local_db || @db
289
- if db&.valid_entry?(base_id, year)
290
- entry = db[base_id]
291
- if entry && !entry.match?(/^not_found/) &&
292
- pub_date_in_range?(entry, opts)
293
- return bib_retval(entry, stdclass)
294
- end
295
- end
296
- end
297
-
298
- id, searchcode = std_id(code, year, opts, stdclass)
299
- db = @local_db || @db
327
+ _, searchcode = strip_id_wrapper(code, stdclass)
328
+ id = cache_key(searchcode, year, opts, stdclass)
329
+ db = id && (@local_db || @db)
300
330
  altdb = @local_db && @db ? @db : nil
301
331
  if db.nil?
302
332
  return if opts[:fetch_db]
@@ -304,10 +334,11 @@ module Relaton
304
334
  bibentry = new_bib_entry(searchcode, year, opts, stdclass)
305
335
  return bib_retval(bibentry, stdclass)
306
336
  end
307
-
308
- @semaphore.synchronize do
309
- db.delete(id) unless db.valid_entry?(id, year)
337
+ if date_range?(opts)
338
+ return check_date_range(searchcode, id, year, opts, stdclass)
310
339
  end
340
+
341
+ @semaphore.synchronize { db.expire id, year }
311
342
  if altdb
312
343
  return bib_retval(altdb[id], stdclass) if opts[:fetch_db]
313
344
 
@@ -340,28 +371,60 @@ module Relaton
340
371
  entry
341
372
  end
342
373
 
343
- def fetch_entry(code, year, opts, stdclass, **args)
374
+ def fetch_entry(code, year, opts, stdclass, **args) # rubocop:disable Metrics/AbcSize
344
375
  processor = @registry[stdclass]
345
376
  bib = net_retry(code, year, opts, processor, opts.fetch(:retries, 1))
346
-
347
- entry = check_entry(bib, stdclass, **args)
377
+ entry = bib_entry bib
348
378
  return entry if args[:db].nil?
349
379
 
350
- @semaphore.synchronize { args[:db][args[:id]] ||= entry }
351
- args[:no_cache] ? bib_entry(bib) : entry
380
+ # `no_cache` refreshes a cached entry, but a failed fetch does not
381
+ # replace a cached document with `not_found`.
382
+ refresh = opts[:no_cache] && bib.respond_to?(:to_xml)
383
+ @semaphore.synchronize do
384
+ if refresh || !args[:db][args[:id]]
385
+ save_bib args[:db], args[:id], bib, entry, stdclass
386
+ end
387
+ end
388
+ entry
352
389
  end
353
390
 
354
- def check_entry(bib, stdclass, **args) # rubocop:disable Metrics/AbcSize,Metrics/CyclomaticComplexity,Metrics/PerceivedComplexity
355
- bib_id = bib && bib.docidentifier.first&.content
391
+ #
392
+ # Cache a fetched document. The document's own identifier gets a row;
393
+ # when the query key differs from it (an undated or incomplete query),
394
+ # the query gets a row that points to the same document.
395
+ #
396
+ def save_bib(db, key, bib, entry, stdclass)
397
+ item = bib.respond_to?(:docidentifier) && item_key(bib, stdclass)
398
+ db.store key, entry, item_key: item || nil
399
+ end
356
400
 
357
- quoted = Regexp.quote("(#{bib_id})")
358
- if args[:db] && args[:id] && bib_id &&
359
- args[:id] !~ /#{quoted}/
360
- bid = std_id(bib.docidentifier.first.content, nil, {}, stdclass).first
361
- @semaphore.synchronize { args[:db][bid] ||= bib_entry bib }
362
- "redirection #{bid}"
363
- else bib_entry bib
401
+ #
402
+ # A publication date range selects among the cached editions of the
403
+ # reference, so it is not part of the key. On a miss the flavor is asked
404
+ # with the range, and its answer is cached under its own identifier only:
405
+ # the query row keeps pointing to the latest edition.
406
+ #
407
+ def check_date_range(code, key, year, opts, stdclass) # rubocop:disable Metrics/AbcSize,Metrics/MethodLength,Metrics/CyclomaticComplexity,Metrics/PerceivedComplexity
408
+ caches = [@local_db, @db].compact
409
+ cached = caches.flat_map { |c| c.candidates(key) }.filter_map do |_, xml|
410
+ date = published_date(xml)
411
+ [date, xml] if date && pub_date_in_range?(xml, opts)
412
+ end.max_by(&:first)
413
+ return bib_retval(cached.last, stdclass) if cached && !opts[:no_cache]
414
+ return if opts[:fetch_db]
415
+
416
+ processor = @registry[stdclass]
417
+ bib = net_retry(code, year, opts, processor, opts.fetch(:retries, 1))
418
+ return unless bib.respond_to?(:to_xml)
419
+
420
+ entry = bib_entry bib
421
+ item = item_key(bib, stdclass)
422
+ if item
423
+ @semaphore.synchronize do
424
+ [@local_db, @db].compact.each { |c| c.store item, entry }
425
+ end
364
426
  end
427
+ bib_retval entry, stdclass
365
428
  end
366
429
 
367
430
  def net_retry(code, year, opts, processor, retries)
@@ -380,19 +443,25 @@ module Relaton
380
443
  end
381
444
  end
382
445
 
383
- def pub_date_in_range?(entry, opts) # rubocop:disable Metrics/CyclomaticComplexity,Metrics/PerceivedComplexity
384
- doc = Nokogiri::XML(entry)
385
- date_str = doc.at("//date[@type='published']/on")&.text
386
- return false unless date_str
446
+ # @param entry [String] document XML
447
+ # @return [Date, nil] the published date
448
+ def published_date(entry)
449
+ date_str = Moxml.parse(entry)
450
+ .at_xpath("//date[@type='published']/on")&.text
451
+ date_str && parse_pub_date(date_str)
452
+ end
387
453
 
388
- date = parse_pub_date(date_str)
454
+ def pub_date_in_range?(entry, opts) # rubocop:disable Metrics/CyclomaticComplexity
455
+ date = published_date(entry)
389
456
  return false unless date
390
457
 
458
+ # `parse_pub_date`, not `Date.parse`: a bound may be "YYYY" or
459
+ # "YYYY-MM", which `Date.parse` rejects.
391
460
  after = opts[:publication_date_after]
392
- return false if after && date < Date.parse(after.to_s)
461
+ return false if after && date < parse_pub_date(after.to_s)
393
462
 
394
463
  before = opts[:publication_date_before]
395
- return false if before && date >= Date.parse(before.to_s)
464
+ return false if before && date >= parse_pub_date(before.to_s)
396
465
 
397
466
  true
398
467
  end
@@ -407,18 +476,8 @@ module Relaton
407
476
  nil
408
477
  end
409
478
 
410
- def open_cache_biblio(dir) # rubocop:disable Metrics/MethodLength
411
- return nil if dir.nil?
412
-
413
- db = Cache.new dir
414
-
415
- Dir["#{dir}/*/"].each do |fdir|
416
- next if db.check_version?(fdir)
417
-
418
- FileUtils.rm_rf(fdir, secure: true)
419
- Util.info "cache #{fdir}: version is obsolete and cache is cleared."
420
- end
421
- db
479
+ def open_cache_biblio(dir)
480
+ dir && Cache.new(dir)
422
481
  end
423
482
 
424
483
  def process_queue(qwp)
@@ -10,25 +10,42 @@ module Relaton
10
10
  #
11
11
  # Get a document by DOI from the CrossRef API.
12
12
  #
13
- # @param [String] doi The DOI.
13
+ # @param [String, Pubid::Doi::Identifier] doi The DOI.
14
14
  #
15
15
  # @return [RelatonBib::BibliographicItem, RelatonIetf::IetfBibliographicItem,
16
16
  # RelatonBipm::BipmBibliographicItem, RelatonIeee::IeeeBibliographicItem,
17
17
  # RelatonNist::NistBibliographicItem] The bibitem.
18
18
  #
19
19
  def get(doi)
20
- Util.info "Fetching from search.crossref.org ...", key: doi
21
- id = doi.sub(%r{^doi:}, "")
22
- message = get_by_id id
20
+ key = doi.to_s
21
+ Util.info "Fetching from search.crossref.org ...", key: key
22
+ message = get_by_id doi_of(doi)
23
23
  if message
24
- Util.info "Found: `#{message['DOI']}`", key: doi
24
+ Util.info "Found: `#{message['DOI']}`", key: key
25
25
  Parser.parse message
26
26
  else
27
- Util.info "Not found.", key: doi
27
+ Util.info "Not found.", key: key
28
28
  nil
29
29
  end
30
30
  end
31
31
 
32
+ #
33
+ # The DOI itself (`<prefix>/<suffix>`), as the Crossref API takes it.
34
+ #
35
+ # A parsed pubid gives it from its components. A String may carry a
36
+ # `doi:` scheme (in any case) or a `doi.org` URL: Relaton::Db keys every
37
+ # form Pubid::Doi reads as one DOI, so all of them must reach the same
38
+ # DOI here, or a miss on one form is cached for the others.
39
+ #
40
+ # @param [String, Pubid::Doi::Identifier] doi
41
+ # @return [String]
42
+ #
43
+ def doi_of(doi)
44
+ return "#{doi.prefix}/#{doi.suffix}" unless doi.is_a?(String)
45
+
46
+ doi.sub(%r{\A(?:doi:|https?://(?:dx\.)?doi\.org/)}i, "")
47
+ end
48
+
32
49
  #
33
50
  # Get a document by DOI from the CrossRef API.
34
51
  #
@@ -8,6 +8,7 @@ module Relaton
8
8
  def initialize # rubocop:disable Lint/MissingSuper
9
9
  @short = :relaton_doi
10
10
  @prefix = "DOI"
11
+ @pubid_identifier = :Doi # Db cache key
11
12
  @defaultprefix = %r{^doi:}
12
13
  @idtype = "DOI"
13
14
  end
@@ -12,6 +12,7 @@ module Relaton
12
12
  def initialize
13
13
  @short = :relaton_easc
14
14
  @prefix = "EASC"
15
+ @pubid_identifier = :Easc # Db cache key
15
16
  # Both Cyrillic and Latin series prefixes route here.
16
17
  @defaultprefix = %r{^(?:ПМГ|РМГ|PMG|RMG)\b}
17
18
  @idtype = "EASC"
@@ -128,7 +128,7 @@ module Relaton
128
128
  def to_yaml(bib) = bib.to_yaml
129
129
  def to_bibxml(bib) = bib.to_rfcxml
130
130
 
131
- # @param hit [Nokogiri::XML::Element]
131
+ # @param hit [Moxml::Element]
132
132
  def parse_page(hit) # rubocop:disable Metrics/AbcSize, Metrics/MethodLength
133
133
  DataParser.new(hit, @errors).parse.each { |item| write_file item }
134
134
  end
@@ -6,7 +6,7 @@ module Relaton
6
6
  #
7
7
  # Initialize parser
8
8
  #
9
- # @param [Nokogiri::XML::Element] hit document hit
9
+ # @param [Moxml::Element] hit document hit
10
10
  # @param [Hash] errors error tracking hash
11
11
  #
12
12
  def initialize(hit, errors = {})
@@ -21,7 +21,7 @@ module Relaton
21
21
  docid = @bib[:docidentifier]
22
22
  @doc.xpath('//div[@id="main"]/div[1]/div/main/article/div/div/standard/div/ul/li').map do |hit|
23
23
  bib = @bib.dup
24
- id, ed, bib[:date], vol = edition_id_parts hit.at("./span", "./a").text
24
+ id, ed, bib[:date], vol = edition_id_parts hit.at_xpath("./span|./a").text
25
25
  bib[:source] = edition_source(hit) + edition_translation_source(ed)
26
26
  next if ed.nil? || ed.empty?
27
27
 
@@ -56,7 +56,7 @@ module Relaton
56
56
  end
57
57
 
58
58
  def edition_source(hit)
59
- es = { "src" => hit.at("./a"), "pdf" => hit.at("./span/a") }.map do |type, a|
59
+ es = { "src" => hit.at_xpath("./a"), "pdf" => hit.at_xpath("./span/a") }.map do |type, a|
60
60
  Bib::Uri.new(type: type, content: a[:href]) if a
61
61
  end.compact
62
62
  @errors[:edition_source] &&= es.empty?
@@ -5,7 +5,7 @@ module Relaton
5
5
 
6
6
  ATTRS = %i[docidentifier title date source ext].freeze
7
7
 
8
- # @param [Nokogiri::XML::Element] hit document hit
8
+ # @param [Moxml::Element] hit document hit
9
9
  # @param [Hash] errors error tracking hash
10
10
  def initialize(hit:, errors: {})
11
11
  @hit = hit
@@ -23,7 +23,7 @@ module Relaton
23
23
 
24
24
  # @return [Array<Relaton::Ecma::Docidentifier>]
25
25
  def fetch_docidentifier
26
- code = "ECMA MEM/#{@hit.at('div[1]//p').text}"
26
+ code = "ECMA MEM/#{@hit.at_xpath('div[1]//p').text}"
27
27
  docid = super(code)
28
28
  @errors[:memento_docidentifier] &&= docid.empty?
29
29
  docid
@@ -31,7 +31,7 @@ module Relaton
31
31
 
32
32
  # @return [Array<Relaton::Bib::Title>]
33
33
  def fetch_title
34
- year = @hit.at("div[1]//p").text
34
+ year = @hit.at_xpath("div[1]//p").text
35
35
  content = "\"Memento #{year}\" for year #{year}"
36
36
  result = [Bib::Title.new(content: content, language: "en", script: "Latn")]
37
37
  @errors[:memento_title] &&= result.empty?
@@ -40,7 +40,7 @@ module Relaton
40
40
 
41
41
  # @return [Array<Relaton::Bib::Date>]
42
42
  def fetch_date
43
- date = @hit.at("div[2]//p").text
43
+ date = @hit.at_xpath("div[2]//p").text
44
44
  on = Date.strptime(date, "%B %Y").strftime "%Y-%m"
45
45
  result = [Bib::Date.new(type: "published", at: on)]
46
46
  @errors[:memento_date] &&= result.empty?
@@ -5,7 +5,7 @@ module Relaton
5
5
 
6
6
  ATTRS = %i[docidentifier title date source abstract relation edition ext].freeze
7
7
 
8
- # @param [Nokogiri::XML::Element] hit document hit
8
+ # @param [Moxml::Element] hit document hit
9
9
  # @param [Mechanize::Page] doc fetched document page
10
10
  # @param [Hash] errors error tracking hash
11
11
  def initialize(hit:, doc:, errors: {})
@@ -68,7 +68,7 @@ module Relaton
68
68
  def fetch_source # rubocop:disable Metrics/AbcSize
69
69
  source = []
70
70
  source << Bib::Uri.new(type: "src", content: @hit[:href]) if @hit[:href]
71
- ref = @doc.at('//div[@class="ecma-item-content-wrapper"]/span/a',
71
+ ref = @doc.at_xpath('//div[@class="ecma-item-content-wrapper"]/span/a',
72
72
  '//div[@class="ecma-item-content-wrapper"]/a')
73
73
  source << Bib::Uri.new(type: "pdf", content: ref[:href]) if ref
74
74
  result = source + edition_translation_source(fetch_edition_content)
@@ -80,7 +80,7 @@ module Relaton
80
80
  def fetch_relation # rubocop:disable Metrics/AbcSize, Metrics/MethodLength, Metrics/CyclomaticComplexity
81
81
  edition_parser = EditionParser.new(doc: @doc, bib: {}, errors: @errors)
82
82
  result = @doc.xpath("//ul[@class='ecma-item-archives']/li").filter_map do |rel|
83
- ref, ed, date, vol = edition_parser.edition_id_parts rel.at("span").text
83
+ ref, ed, date, vol = edition_parser.edition_id_parts rel.at_xpath("span").text
84
84
  next if ed.nil? || ed.empty?
85
85
 
86
86
  docid = Docidentifier.new(type: "ECMA", content: ref, primary: true)
@@ -109,7 +109,7 @@ module Relaton
109
109
  private
110
110
 
111
111
  def fetch_edition_content
112
- @doc.at('//p[@class="ecma-item-edition"]')&.text&.match(/^\d+(?=(?:st|nd|th|rd))/)&.to_s
112
+ @doc.at_xpath('//p[@class="ecma-item-edition"]')&.text&.match(/^\d+(?=(?:st|nd|th|rd))/)&.to_s
113
113
  end
114
114
 
115
115
  def edition_translation_source(edition)
@@ -120,8 +120,8 @@ module Relaton
120
120
  return [] unless @doc
121
121
 
122
122
  @doc.xpath("//h2[.='Translations']/following-sibling::ul/li").map do |l|
123
- a = l.at("span/a")
124
- id = l.at("span").text
123
+ a = l.at_xpath("span/a")
124
+ id = l.at_xpath("span").text
125
125
  %r{\w+[\d-]+,\s(?<lang>\w+)\sversion,\s(?<ed>[\d.]+)(?:st|nd|rd|th)\sedition} =~ id
126
126
  case lang
127
127
  when "Japanese"
@@ -8,6 +8,7 @@ module Relaton
8
8
  def initialize # rubocop:disable Lint/MissingSuper
9
9
  @short = :relaton_etsi
10
10
  @prefix = "ETSI"
11
+ @pubid_identifier = :Etsi # Db cache key
11
12
  @defaultprefix = %r{^ETSI\s}
12
13
  @idtype = "ETSI"
13
14
  @datasets = %w[etsi-csv]
@@ -1,7 +1,7 @@
1
1
  # encoding: UTF-8
2
2
  # frozen_string_literal: true
3
3
 
4
- require "nokogiri"
4
+ require "moxml"
5
5
  require_relative "scraper"
6
6
 
7
7
  module Relaton
@@ -20,10 +20,10 @@ module Relaton
20
20
  hits = doc.xpath(
21
21
  "//table[contains(@class, 'result_list')]/tbody[2]/tr",
22
22
  ).map do |h|
23
- ref = h.at "./td[2]/a"
23
+ ref = h.at_xpath "./td[2]/a"
24
24
  pid = ref[:onclick].match(/[0-9A-F]+/).to_s
25
- status = h.at("./td[7]").text.strip
26
- rdate = h.at("./td[8]").text.strip
25
+ status = h.at_xpath("./td[7]").text.strip
26
+ rdate = h.at_xpath("./td[8]").text.strip
27
27
  Hit.new pid: pid, docref: ref.text, scraper: self,
28
28
  release_date: rdate, status: status
29
29
  end
@@ -52,7 +52,7 @@ module Relaton
52
52
  # * :type [String]
53
53
  # * :name [String]
54
54
  # def get_committee(doc, _ref)
55
- # name = doc.at("//div[contains(., '归口单位') or contains(., '归口部门')]/following-sibling::div")
55
+ # name = doc.at_xpath("//div[contains(., '归口单位') or contains(., '归口部门')]/following-sibling::div")
56
56
  # Committee.new(type: "technical", content: name.text.delete("\r\n\t\t"))
57
57
  # end
58
58
  end