relaton 3.0.0.pre.alpha.4 → 3.0.0.pre.alpha.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. checksums.yaml +4 -4
  2. data/lib/relaton/adobe/processor.rb +9 -0
  3. data/lib/relaton/bib/converter/csl.rb +110 -0
  4. data/lib/relaton/bib/converter/ris.rb +104 -0
  5. data/lib/relaton/bib/converter/titles.rb +24 -0
  6. data/lib/relaton/bib/item_data.rb +9 -0
  7. data/lib/relaton/bib/model/item.rb +6 -6
  8. data/lib/relaton/bib/sanitizer.rb +71 -30
  9. data/lib/relaton/bib.rb +10 -0
  10. data/lib/relaton/bipm/processor.rb +1 -0
  11. data/lib/relaton/bipm/rawdata_bipm_metrologia/affiliations.rb +6 -6
  12. data/lib/relaton/bipm/si_brochure_parser.rb +2 -2
  13. data/lib/relaton/bsi/processor.rb +5 -0
  14. data/lib/relaton/ccsds/bibliography.rb +1 -0
  15. data/lib/relaton/ccsds/processor.rb +12 -0
  16. data/lib/relaton/cen/hit_collection.rb +1 -1
  17. data/lib/relaton/cen/processor.rb +5 -0
  18. data/lib/relaton/cen/scraper.rb +9 -9
  19. data/lib/relaton/cie/data_fetcher.rb +18 -18
  20. data/lib/relaton/cie/processor.rb +1 -0
  21. data/lib/relaton/cie.rb +1 -1
  22. data/lib/relaton/cloud.rb +127 -0
  23. data/lib/relaton/core/hit_collection.rb +6 -10
  24. data/lib/relaton/core/processor.rb +68 -0
  25. data/lib/relaton/db/cache.rb +444 -148
  26. data/lib/relaton/db/cache_entry.rb +33 -0
  27. data/lib/relaton/db/registry.rb +52 -0
  28. data/lib/relaton/db.rb +131 -72
  29. data/lib/relaton/doi/crossref.rb +23 -6
  30. data/lib/relaton/doi/processor.rb +1 -0
  31. data/lib/relaton/easc/processor.rb +1 -0
  32. data/lib/relaton/ecma/data_fetcher.rb +1 -1
  33. data/lib/relaton/ecma/data_parser.rb +1 -1
  34. data/lib/relaton/ecma/edition_parser.rb +2 -2
  35. data/lib/relaton/ecma/memento_parser.rb +4 -4
  36. data/lib/relaton/ecma/standard_parser.rb +6 -6
  37. data/lib/relaton/etsi/processor.rb +1 -0
  38. data/lib/relaton/gb/gb_scraper.rb +5 -5
  39. data/lib/relaton/gb/scraper.rb +16 -16
  40. data/lib/relaton/gb/sec_scraper.rb +8 -8
  41. data/lib/relaton/gb/t_scraper.rb +5 -5
  42. data/lib/relaton/gost/processor.rb +1 -0
  43. data/lib/relaton/iala/processor.rb +1 -0
  44. data/lib/relaton/iana/data_fetcher.rb +3 -3
  45. data/lib/relaton/iana/parser.rb +8 -4
  46. data/lib/relaton/iana/processor.rb +9 -0
  47. data/lib/relaton/iec/data_parser.rb +24 -9
  48. data/lib/relaton/iec/processor.rb +6 -0
  49. data/lib/relaton/iec.rb +1 -1
  50. data/lib/relaton/ieee/data_fetcher.rb +1 -1
  51. data/lib/relaton/ieee/processor.rb +8 -0
  52. data/lib/relaton/ietf/data_fetcher.rb +1 -1
  53. data/lib/relaton/ietf/processor.rb +1 -0
  54. data/lib/relaton/ietf/rfc/entry.rb +15 -19
  55. data/lib/relaton/iho/processor.rb +1 -0
  56. data/lib/relaton/index/file_io.rb +2 -2
  57. data/lib/relaton/index/pool.rb +4 -3
  58. data/lib/relaton/index/shard_source.rb +1 -1
  59. data/lib/relaton/index/type.rb +2 -5
  60. data/lib/relaton/isbn/open_library.rb +11 -7
  61. data/lib/relaton/isbn/processor.rb +10 -0
  62. data/lib/relaton/iso/data_parser.rb +2 -2
  63. data/lib/relaton/iso/processor.rb +5 -0
  64. data/lib/relaton/iso/scraper.rb +20 -20
  65. data/lib/relaton/itu/bibliography.rb +99 -13
  66. data/lib/relaton/itu/data_crawler_r.rb +2 -2
  67. data/lib/relaton/itu/hit_collection.rb +64 -52
  68. data/lib/relaton/itu/processor.rb +1 -0
  69. data/lib/relaton/itu/scraper.rb +13 -3
  70. data/lib/relaton/itu.rb +0 -2
  71. data/lib/relaton/jis/data_fetcher.rb +4 -4
  72. data/lib/relaton/jis/processor.rb +1 -0
  73. data/lib/relaton/jis/scraper.rb +8 -8
  74. data/lib/relaton/oasis/browser_agent.rb +2 -2
  75. data/lib/relaton/oasis/data_parser.rb +5 -5
  76. data/lib/relaton/oasis/data_parser_utils.rb +2 -2
  77. data/lib/relaton/oasis/data_part_parser.rb +7 -7
  78. data/lib/relaton/ogc/processor.rb +7 -0
  79. data/lib/relaton/oiml/processor.rb +1 -0
  80. data/lib/relaton/omg/scraper.rb +11 -11
  81. data/lib/relaton/omg.rb +1 -1
  82. data/lib/relaton/plateau/processor.rb +1 -0
  83. data/lib/relaton/un/bibliography.rb +21 -10
  84. data/lib/relaton/un/processor.rb +1 -0
  85. data/lib/relaton/version.rb +1 -1
  86. data/lib/relaton/w3c/processor.rb +1 -0
  87. data/lib/relaton.rb +13 -0
  88. metadata +48 -16
  89. data/lib/relaton/itu/pubid.rb +0 -199
@@ -15,7 +15,7 @@ module Relaton
15
15
 
16
16
  @prefixes = nil
17
17
 
18
- # @param doc [Nokogiri::HTML::Document]
18
+ # @param doc [Moxml::Document]
19
19
  # @param src [String]
20
20
  # @param hit [RelatonGb::Hit]
21
21
  # @return [Hash]
@@ -41,7 +41,7 @@ module Relaton
41
41
  [Docidentifier.new(content: docref, type: "Chinese Standard", primary: true)]
42
42
  end
43
43
 
44
- # @param doc [Nokogiri::HTML::Document]
44
+ # @param doc [Moxml::Document]
45
45
  # @param docref [Strings]
46
46
  # @return [Array<Relaton::Bib::Contributor>]
47
47
  def get_contributors(doc, docref)
@@ -67,22 +67,22 @@ module Relaton
67
67
  Bib::TypedLocalizedString.new language: lang, content: content
68
68
  end
69
69
 
70
- # @param doc [Nokogiri::HTML::Document]
70
+ # @param doc [Moxml::Document]
71
71
  # @return [Array<Relaton::Bib::Title>]
72
72
  def get_titles(doc)
73
- tzh = doc.at("//td[contains(text(), '中文标准名称')]/b").text
73
+ tzh = doc.at_xpath("//td[contains(text(), '中文标准名称')]/b").text
74
74
  titles = Relaton::Bib::Title.from_string tzh, "zh", "Hans"
75
- ten = doc.at("//td[contains(text(), '英文标准名称')]").text.match(/[\w\s]+/).to_s
75
+ ten = doc.at_xpath("//td[contains(text(), '英文标准名称')]").text.match(/[\w\s]+/).to_s
76
76
  return titles if ten.empty?
77
77
 
78
78
  titles + Relaton::Bib::Title.from_string(ten, "en", "Latn")
79
79
  end
80
80
 
81
- # @param doc [Nokogiri::HTML::Document]
81
+ # @param doc [Moxml::Document]
82
82
  # @param status [String, NilClass]
83
83
  # @return [Relaton::Bib::Status]
84
84
  def get_status(doc, status = nil)
85
- status ||= doc.at("//td[contains(., '标准状态')]/span")&.text&.strip
85
+ status ||= doc.at_xpath("//td[contains(., '标准状态')]/span")&.text&.strip
86
86
  return unless STAGES[status]
87
87
 
88
88
  stage = Bib::Status::Stage.new content: STAGES[status]
@@ -91,17 +91,17 @@ module Relaton
91
91
 
92
92
  private
93
93
 
94
- # @param doc [Nokogiri::HTML::Document]
94
+ # @param doc [Moxml::Document]
95
95
  # @return [Array<String>]
96
96
  def get_ccs(doc)
97
- code = doc.at("//div[contains(text(), '中国标准分类号')]/following-sibling::div").text.strip
97
+ code = doc.at_xpath("//div[contains(text(), '中国标准分类号')]/following-sibling::div").text.strip
98
98
  [CCS.new(code: code)]
99
99
  end
100
100
 
101
- # @param doc [Nokogiri::HTML::Document]
101
+ # @param doc [Moxml::Document]
102
102
  # @return [Array<Relaton::Bib::ICS>]
103
103
  def get_ics(doc)
104
- ics = doc.at("//div[contains(text(), '国际标准分类号')]/following-sibling::div"\
104
+ ics = doc.at_xpath("//div[contains(text(), '国际标准分类号')]/following-sibling::div"\
105
105
  " | //dt[contains(text(), '国际标准分类号')]/following-sibling::dd")
106
106
  return [] unless ics
107
107
 
@@ -109,10 +109,10 @@ module Relaton
109
109
  [Bib::ICS.new(code: code)]
110
110
  end
111
111
 
112
- # @param doc [Nokogiri::HTML::Document]
112
+ # @param doc [Moxml::Document]
113
113
  # @return [String]
114
114
  def get_scope(doc)
115
- issued = doc.at("//div[contains(., '发布单位')]/following-sibling::div")
115
+ issued = doc.at_xpath("//div[contains(., '发布单位')]/following-sibling::div")
116
116
  case issued&.text
117
117
  when /国家标准/ then "national"
118
118
  when /^行业标准/ then "sector"
@@ -150,12 +150,12 @@ module Relaton
150
150
  (Bib::Uri.new(type: "src", content: src))
151
151
  end
152
152
 
153
- # @param doc [Nokogiri::HTML::Document]
153
+ # @param doc [Moxml::Document]
154
154
  # @return [Array<Hash>]
155
155
  # * :type [String] type of date
156
156
  # * :on [String] date
157
157
  def get_dates(doc)
158
- date = doc.at("//div[contains(text(), '发布日期')]/following-sibling::div"\
158
+ date = doc.at_xpath("//div[contains(text(), '发布日期')]/following-sibling::div"\
159
159
  " | //dt[contains(text(), '发布日期')]/following-sibling::dd")
160
160
  [Bib::Date.new(type: "published", at: date.text.delete("\r\n\t\t"))]
161
161
  end
@@ -177,7 +177,7 @@ module Relaton
177
177
  Doctype.new content: "standard"
178
178
  end
179
179
 
180
- # @param doc [Nokogiri::HTML::Document]
180
+ # @param doc [Moxml::Document]
181
181
  # @param ref [String]
182
182
  # @return [Relaton::Gb::GbType]
183
183
  def get_gbtype(doc, ref)
@@ -3,7 +3,7 @@
3
3
 
4
4
  require "net/http"
5
5
  require "json"
6
- require "nokogiri"
6
+ require "moxml"
7
7
  require_relative "scraper"
8
8
  require_relative "item"
9
9
  require_relative "hit_collection"
@@ -44,7 +44,7 @@ module Relaton
44
44
  def scrape_doc(hit)
45
45
  src = "https://hbba.sacinfo.org.cn/stdDetail/#{hit.pid}"
46
46
  page_uri = URI src
47
- doc = Nokogiri::HTML Net::HTTP.get(page_uri)
47
+ doc = Moxml.new.parse_html Net::HTTP.get(page_uri)
48
48
  ItemData.new(**scrapped_data(doc, src, hit))
49
49
  rescue SocketError, Timeout::Error, Errno::EINVAL, Errno::ECONNRESET, EOFError,
50
50
  Net::HTTPBadResponse, Net::HTTPHeaderSyntaxError, Net::ProtocolError,
@@ -54,14 +54,14 @@ module Relaton
54
54
 
55
55
  private
56
56
 
57
- # @param doc [Nokogiri::HTML::Document]
57
+ # @param doc [Moxml::Document]
58
58
  # @return [Array<Relaton::Bib::Title>]
59
59
  def get_titles(doc)
60
- tzh = doc.at("//h4").text.delete("\r\n\t")
60
+ tzh = doc.at_xpath("//h4").text.delete("\r\n\t")
61
61
  Bib::Title.from_string(tzh, "zh", "Hans")
62
62
  end
63
63
 
64
- # @param _doc [Nokogiri::HTML::Document]
64
+ # @param _doc [Moxml::Document]
65
65
  # @param ref [String]
66
66
  # @return [Hash]
67
67
  # * :type [String]
@@ -72,16 +72,16 @@ module Relaton
72
72
  # { type: "technical", name: name }
73
73
  # end
74
74
 
75
- # @param _doc [Nokogiri::HTML::Document]
75
+ # @param _doc [Moxml::Document]
76
76
  # @return [String]
77
77
  def get_scope(_doc)
78
78
  "sector"
79
79
  end
80
80
 
81
- # @param doc [Nokogiri::HTML::Document]
81
+ # @param doc [Moxml::Document]
82
82
  # @return [Array<String>]
83
83
  def get_ccs(doc)
84
- array(doc.at("//dt[contains(text(), '中国标准分类号')]/following-sibling::dd")).map do |cc|
84
+ array(doc.at_xpath("//dt[contains(text(), '中国标准分类号')]/following-sibling::dd")).map do |cc|
85
85
  text = Cnccs.fetch(cc.text.strip)&.description
86
86
  CCS.new code: cc.text, text: text
87
87
  end
@@ -1,7 +1,7 @@
1
1
  # encoding: UTF-8
2
2
  # frozen_string_literal: true
3
3
 
4
- require "nokogiri"
4
+ require "moxml"
5
5
  require_relative "scraper"
6
6
  require_relative "hit_collection"
7
7
  require_relative "hit"
@@ -23,8 +23,8 @@ module Relaton
23
23
  xpath = '//table[contains(@class, "standard_list_table")]/tr/td/a'
24
24
  t_xpath = "../preceding-sibling::td[4]"
25
25
  hits = doc.xpath(xpath).map do |h|
26
- docref = h.at(t_xpath).text.gsub(/â\u0080\u0094/, "-")
27
- status = h.at("../preceding-sibling::td[1]").text.delete "\r\n"
26
+ docref = h.at_xpath(t_xpath).text.gsub(/â\u0080\u0094/, "-")
27
+ status = h.at_xpath("../preceding-sibling::td[1]").text.delete "\r\n"
28
28
  pid = h[:href].sub(%r{/$}, "")
29
29
  Hit.new pid: pid, docref: docref, status: status, scraper: self
30
30
  end
@@ -55,7 +55,7 @@ module Relaton
55
55
  private
56
56
 
57
57
  # rubocop:disable Metrics/MethodLength
58
- # @param doc [Nokogiri::HTML::Document]
58
+ # @param doc [Moxml::Document]
59
59
  # @param src [String]
60
60
  # @param hit [RelatonGb::Hit]
61
61
  # @return [Hash]
@@ -89,7 +89,7 @@ module Relaton
89
89
 
90
90
  def get_titles(doc)
91
91
  xpz = '//td[contains(.,"中文标题")]/following-sibling::td[1]'
92
- titles = Bib::Title.from_string doc.at(xpz)
92
+ titles = Bib::Title.from_string doc.at_xpath(xpz)
93
93
  .text, "zh", "Hans"
94
94
  xpe = '//td[contains(.,"英文标题")]/following-sibling::td[1]'
95
95
  ten = doc.xpath(xpe).text
@@ -12,6 +12,7 @@ module Relaton
12
12
  def initialize
13
13
  @short = :relaton_gost
14
14
  @prefix = "GOST"
15
+ @pubid_identifier = :Gost # Db cache key
15
16
  # Both Latin "GOST" and Cyrillic "ГОСТ" route here. The trailing
16
17
  # \b keeps the prefix from swallowing longer tokens ("GOSTA …").
17
18
  @defaultprefix = %r{^(?:GOST|ГОСТ)\b}
@@ -8,6 +8,7 @@ module Relaton
8
8
  def initialize
9
9
  @short = :relaton_iala
10
10
  @prefix = "IALA"
11
+ @pubid_identifier = :Iala # Db cache key
11
12
  @defaultprefix = %r{^IALA\s}
12
13
  @idtype = "IALA"
13
14
  end
@@ -51,11 +51,11 @@ module Relaton
51
51
  end
52
52
 
53
53
  def parse(content)
54
- xml = Nokogiri::XML(content)
55
- registry = xml.at("/xmlns:registry")
54
+ xml = Moxml.parse(content)
55
+ registry = xml.at_xpath("/xmlns:registry", NS)
56
56
  doc = Parser.parse registry, nil, @errors
57
57
  save_doc doc
58
- registry.xpath("./xmlns:registry").each { |r| save_doc Parser.parse(r, doc, @errors) }
58
+ registry.xpath("./xmlns:registry", NS).each { |r| save_doc Parser.parse(r, doc, @errors) }
59
59
  end
60
60
 
61
61
  #
@@ -1,10 +1,14 @@
1
1
  module Relaton
2
2
  module Iana
3
+ # The IANA registry XML uses a default namespace, and moxml (unlike raw
4
+ # nokogiri) does not bind the literal `xmlns:` prefix to it — every
5
+ # namespaced query passes this binding explicitly.
6
+ NS = { "xmlns" => "http://www.iana.org/assignments" }.freeze
3
7
  class Parser
4
8
  #
5
9
  # Document parser initalization
6
10
  #
7
- # @param [Nokogiri::XML::Element] xml
11
+ # @param [Moxml::Element] xml
8
12
  #
9
13
  def initialize(xml, rootdoc, errors = {})
10
14
  @xml = xml
@@ -15,7 +19,7 @@ module Relaton
15
19
  #
16
20
  # Initialize document parser and run it
17
21
  #
18
- # @param [Nokogiri::XML::Element] xml
22
+ # @param [Moxml::Element] xml
19
23
  #
20
24
  # @return [Relaton::Iana::ItemData, nil] bibliographic item
21
25
  #
@@ -50,7 +54,7 @@ module Relaton
50
54
  # @return [Array<Relaton::Bib::Title>] title
51
55
  #
52
56
  def parse_title
53
- content = @xml.at("./xmlns:title")&.text || @xml[:id]
57
+ content = @xml.at_xpath("./xmlns:title", NS)&.text || @xml[:id]
54
58
  result = [Bib::Title.new(content: content, language: "en", script: "Latn")]
55
59
  @errors[:title] &&= result.empty?
56
60
  result
@@ -121,7 +125,7 @@ module Relaton
121
125
  # @return [Array<Relaton::Bib::Date>] date
122
126
  #
123
127
  def parse_date
124
- d = @xml.xpath("./xmlns:created|./xmlns:published|./xmlns:updated").map do |dt|
128
+ d = @xml.xpath("./xmlns:created|./xmlns:published|./xmlns:updated", NS).map do |dt|
125
129
  Bib::Date.new(type: dt.name, at: dt.text)
126
130
  end
127
131
  result = d.none? && @rootdoc ? @rootdoc.date : d
@@ -8,6 +8,7 @@ module Relaton
8
8
  def initialize # rubocop:disable Lint/MissingSuper
9
9
  @short = :relaton_iana
10
10
  @prefix = "IANA"
11
+ @pubid_identifier = :Iana # Db cache key
11
12
  @defaultprefix = %r{^IANA\s}
12
13
  @idtype = "IANA"
13
14
  @datasets = %w[iana-registries]
@@ -35,6 +36,14 @@ module Relaton
35
36
  DataFetcher.fetch(source, **opts)
36
37
  end
37
38
 
39
+ # `get` reads a reference pubid cannot parse as a miss, so it gets no key
40
+ # here either: it is not cached.
41
+ def cache_pubid(ref)
42
+ super
43
+ rescue StandardError
44
+ nil
45
+ end
46
+
38
47
  # @param xml [String]
39
48
  # @return [Relaton::Iana::ItemData]
40
49
  def from_xml(xml)
@@ -289,38 +289,53 @@ module Relaton
289
289
  #
290
290
  # @return [Array<Relaton::Bib::Relation>] relation
291
291
  #
292
- def relation # rubocop:disable Metrics/MethodLength
292
+ def relation
293
+ result = fetch_relations
294
+ @errors[:relation] &&= result.empty?
295
+ result
296
+ end
297
+
298
+ #
299
+ # Fetch relations from the webstore. Retry on network errors only.
300
+ # A non-success response (the legacy endpoint now redirects to a 404
301
+ # HTML page) or a malformed body gives no relations.
302
+ #
303
+ # @return [Array<Relaton::Bib::Relation>] relations
304
+ #
305
+ def fetch_relations # rubocop:disable Metrics/MethodLength
293
306
  try = 0
294
- result = begin
307
+ begin
295
308
  uri = URI "#{DOMAIN}/webstore/webstore.nsf/AjaxRequestXML?Openagent&url=#{urn_id}"
296
309
  resp = Net::HTTP.get_response uri
297
- doc = Nokogiri::XML resp.body
298
- create_relations doc
299
310
  rescue StandardError => e
300
311
  try += 1
301
312
  try < 3 ? retry : raise(e)
302
313
  end
303
- @errors[:relation] &&= result.empty?
304
- result
314
+ return [] if resp.is_a?(Net::HTTPResponse) && !resp.is_a?(Net::HTTPSuccess)
315
+
316
+ create_relations Moxml.parse(resp.body)
317
+ rescue Moxml::ParseError => e
318
+ Util.warn "Failed to parse relations for `#{urn_id}`: #{e.message.lines.first&.strip}"
319
+ []
305
320
  end
306
321
 
307
322
  #
308
323
  # Create relations.
309
324
  #
310
- # @param [Nokogiri::XML::Document] doc XML document
325
+ # @param [Moxml::Document] doc XML document
311
326
  #
312
327
  # @return [Array<Relaton::Bib::Relation>] relations
313
328
  #
314
329
  def create_relations(doc) # rubocop:disable Metrics/MethodLength
315
330
  doc.xpath('//ROW[STATUS[.!="PREPARING" and .!="PUBLISHED"]]')
316
331
  .map do |r|
317
- r_type = r.at("STATUS").text.downcase
332
+ r_type = r.at_xpath("STATUS").text.downcase
318
333
  type = case r_type
319
334
  when "revised", "replaced" then "updates"
320
335
  when "withdrawn" then "obsoletes"
321
336
  else r_type
322
337
  end
323
- ref = r.at("FULL_NAME").text
338
+ ref = r.at_xpath("FULL_NAME").text
324
339
  docid = Docidentifier.new(content: ref, type: "IEC", primary: true)
325
340
  bibitem = ItemData.new(formattedref: Bib::Formattedref.new(content: ref), docidentifier: [docid])
326
341
  Relation.new type: type, bibitem: bibitem
@@ -34,6 +34,12 @@ module Relaton
34
34
  DataFetcher.fetch(source, **opts)
35
35
  end
36
36
 
37
+ # `IEV` is not a document identifier (`get` answers it with the IEV
38
+ # vocabulary), so it gets no key: it is not cached.
39
+ def cache_pubid(ref)
40
+ ref.strip.casecmp?("IEV") ? nil : super
41
+ end
42
+
37
43
  # @param xml [String]
38
44
  # @return [Relaton::Iec::ItemData]
39
45
  def from_xml(xml)
data/lib/relaton/iec.rb CHANGED
@@ -1,6 +1,6 @@
1
1
  require "digest/md5"
2
2
  require "net/http"
3
- require "nokogiri"
3
+ require "moxml"
4
4
  require "pubid"
5
5
  require "zip"
6
6
  require "relaton/index"
@@ -147,7 +147,7 @@ module Relaton
147
147
  # IdamsParser#parse_relation, so mutates crossrefs under a mutex.
148
148
  #
149
149
  # @param [String] docnumber of main document
150
- # @param [Nokogiri::XML::Element] amsid relation data
150
+ # @param [Moxml::Element] amsid relation data
151
151
  #
152
152
  def add_crossref(docnumber, amsid)
153
153
  return if RELATION_TYPES[amsid.type] == false
@@ -36,6 +36,14 @@ module Relaton
36
36
  DataFetcher.fetch(source, **opts)
37
37
  end
38
38
 
39
+ # `get` reads a reference pubid cannot parse as a miss, so it gets no key
40
+ # here either: it is not cached.
41
+ def cache_pubid(ref)
42
+ super
43
+ rescue StandardError
44
+ nil
45
+ end
46
+
39
47
  # @param xml [String]
40
48
  # @return [Relaton::Ieee::ItemData]
41
49
  def from_xml(xml)
@@ -332,7 +332,7 @@ module Relaton
332
332
  #
333
333
  # Get RFC index
334
334
  #
335
- # @return [Nokogiri::XML::Document] RFC index
335
+ # @return [Moxml::Document] RFC index
336
336
  #
337
337
  def rfc_index
338
338
  uri = URI "https://www.rfc-editor.org/rfc-index.xml"
@@ -6,6 +6,7 @@ module Relaton
6
6
  def initialize # rubocop:disable Lint/MissingSuper
7
7
  @short = :relaton_ietf
8
8
  @prefix = "IETF"
9
+ @pubid_identifier = :Ietf # Db cache key
9
10
  @defaultprefix = /^((IETF|RFC|BCP|FYI|STD)\s|I-D[.\s])/
10
11
  @idtype = "IETF"
11
12
  @datasets = %w[ietf-rfcsubseries ietf-internet-drafts ietf-rfc-entries]
@@ -195,8 +195,7 @@ module Relaton
195
195
  formattedref: build_formattedref,
196
196
  date: build_subseries_date(rfc_index),
197
197
  relation: build_relations(rfc_index, wg_names: wg_names),
198
- series: build_series,
199
- ext: Ext.new(doctype: Doctype.new(content: "rfc"), stream: stream, flavor: "ietf"),
198
+ ext: Ext.new(doctype: Doctype.new(content: "rfc"), stream: canonical_stream, flavor: "ietf"),
200
199
  )
201
200
  end
202
201
 
@@ -259,13 +258,6 @@ module Relaton
259
258
  ItemData.new(formattedref: Bib::Formattedref.new(content: ref), docidentifier: [docid])
260
259
  end
261
260
 
262
- def build_series
263
- return [] unless stream
264
-
265
- t = Bib::Title.new(content: stream)
266
- [Bib::Series.new(type: "stream", title: [t])]
267
- end
268
-
269
261
  # --- RFC entry builders ---
270
262
 
271
263
  def build_rfc_docid
@@ -388,7 +380,7 @@ module Relaton
388
380
  def build_rfc_series
389
381
  series = build_rfc_is_also_series
390
382
  series << Bib::Series.new(title: [Bib::Title.new(content: "RFC")], number: shortnum)
391
- series + build_rfc_stream_series
383
+ series
392
384
  end
393
385
 
394
386
  def build_rfc_is_also_series
@@ -401,27 +393,31 @@ module Relaton
401
393
  end
402
394
  end
403
395
 
404
- def build_rfc_stream_series
405
- return [] unless stream
406
-
407
- t = Bib::Title.new(content: stream)
408
- [Bib::Series.new(type: "stream", title: [t])]
409
- end
410
-
411
396
  STREAM_ORGS = {
412
397
  "IETF" => ["IETF", "Internet Engineering Task Force"],
413
398
  "IRTF" => ["IRTF", "Internet Research Task Force"],
414
399
  "IAB" => ["IAB", "Internet Architecture Board"],
415
400
  }.freeze
416
401
 
402
+ # rfc-index.xml spells the independent stream "INDEPENDENT";
403
+ # Ext.stream's declared values use "Independent". Match case-insensitively
404
+ # against the declared values so both Ext.stream and STREAM_ORGS see the
405
+ # canonical spelling.
406
+ def canonical_stream
407
+ return unless stream
408
+
409
+ values = Ext.attributes[:stream].options[:values]
410
+ values.find { |v| v.casecmp?(stream) } || stream
411
+ end
412
+
417
413
  def build_rfc_ext
418
- Ext.new(doctype: Doctype.new(content: "rfc"), stream: stream, flavor: "ietf")
414
+ Ext.new(doctype: Doctype.new(content: "rfc"), stream: canonical_stream, flavor: "ietf")
419
415
  end
420
416
 
421
417
  def build_committee_contributor(wg_names = {})
422
418
  return if wg_acronym.nil? || wg_acronym == "NON WORKING GROUP"
423
419
 
424
- abbr, name = STREAM_ORGS[stream]
420
+ abbr, name = STREAM_ORGS[canonical_stream]
425
421
  org = if abbr
426
422
  Ietf::BibXMLParser.build_org(abbr, name)
427
423
  else
@@ -8,6 +8,7 @@ module Relaton
8
8
  def initialize
9
9
  @short = :relaton_iho
10
10
  @prefix = "IHO"
11
+ @pubid_identifier = :Iho # Db cache key
11
12
  @defaultprefix = %r{^IHO\s}
12
13
  @idtype = "IHO"
13
14
  end
@@ -28,11 +28,11 @@ module Relaton
28
28
  # if nil then the fiename is used to read and write file (used to create indes in GH actions)
29
29
  # @param [Pubid::Identifier] pubid class for deserialization
30
30
  #
31
- # `id_keys` is accepted for backward compatibility but no longer used: the
31
+ # The index format check round-trips each id through `pubid_class`;
32
32
  # index format is now validated by round-tripping a sample of ids through
33
33
  # the pubid class (see #check_serialization), which understands the pubid
34
34
  # v2 (lutaml) `_type` serialization that the old key-allowlist could not.
35
- def initialize(dir, url, filename, _id_keys = nil, pubid_class = nil)
35
+ def initialize(dir, url: nil, filename: nil, pubid_class: nil)
36
36
  @dir = dir
37
37
  @url = url
38
38
  @filename = filename
@@ -14,7 +14,6 @@ module Relaton
14
14
  # @param [String] type <description>
15
15
  # @param [String, nil] url external URL to index, used to fetch index for searching files
16
16
  # @param [String, nil] file output file name
17
- # @param [Array<Symbol>, nil] id_keys keys to check if index is correct
18
17
  # @param [String, nil] pages_url base URL of the Pages site that serves
19
18
  # the machine index (manifest + shards), see Relaton::Index::ShardSource
20
19
  #
@@ -24,9 +23,11 @@ module Relaton
24
23
  if @pool[type.upcase.to_sym]&.actual?(**args)
25
24
  @pool[type.upcase.to_sym]
26
25
  else
26
+ if args.key?(:id_keys)
27
+ Util.warn "id_keys is deprecated and ignored by Relaton::Index"
28
+ end
27
29
  @pool[type.upcase.to_sym] = Type.new(
28
- type, args[:url], args[:file], args[:id_keys], args[:pubid_class],
29
- pages_url: args[:pages_url]
30
+ type, **args.slice(:url, :file, :pubid_class, :pages_url)
30
31
  )
31
32
  end
32
33
  end
@@ -47,7 +47,7 @@ module Relaton
47
47
  @pages_url = pages_url.end_with?("/") ? pages_url : "#{pages_url}/"
48
48
  # Only the deserialization helpers are used: this FileIO reads and
49
49
  # writes no file.
50
- @file_io = FileIO.new(dir, nil, nil, nil, pubid_class)
50
+ @file_io = FileIO.new(dir, pubid_class: pubid_class)
51
51
  @mutex = Mutex.new
52
52
  @state = nil
53
53
  end
@@ -10,21 +10,18 @@ module Relaton
10
10
  # @param [String, Symbol] type type of index (ISO, IEC, etc.)
11
11
  # @param [String, nil] url external URL to index, used to fetch index for searching files
12
12
  # @param [String, nil] file output file name
13
- # @param [Array<Symbol>] id_keys keys of identifier to be used for sorting index
14
- # format of index file is checked if id_keys all is provided at least in one of the IDs
15
13
  # @param [Pubid::Identifier, nil] pubid class for deserialization
16
14
  # @param [String, nil] pages_url base URL of the Pages site that serves
17
15
  # the machine index. With it the type reads the index from there, in
18
16
  # memory (see ShardSource), and a parsed query fetches one shard only.
19
17
  #
20
- def initialize(type, url = nil, file = nil, id_keys = nil, pubid_class = nil, # rubocop:disable Metrics/ParameterLists
21
- pages_url: nil)
18
+ def initialize(type, url: nil, file: nil, pubid_class: nil, pages_url: nil)
22
19
  @file = file
23
20
  @dir = type.to_s.downcase
24
21
  @pubid_class = pubid_class
25
22
  @pages_url = pages_url
26
23
  filename = file || Index.config.filename
27
- @file_io = FileIO.new @dir, url, filename, id_keys, pubid_class
24
+ @file_io = FileIO.new @dir, url: url, filename: filename, pubid_class: pubid_class
28
25
  @source = new_source
29
26
  end
30
27
 
@@ -8,23 +8,27 @@ module Relaton
8
8
 
9
9
  ENDPOINT = "http://openlibrary.org/api/volumes/brief/isbn/".freeze
10
10
 
11
- def get(ref, _date = nil, _opts = {}) # rubocop:disable Metrics/MethodLength
12
- Util.info "Fetching from OpenLibrary ...", key: ref
13
-
14
- isbn = Isbn.new(ref).parse
11
+ # @param ref [String, Pubid::Isbn::Identifier] an ISBN-10 or ISBN-13
12
+ # @return [Relaton::Bib::ItemData, nil]
13
+ def get(ref, _date = nil, _opts = {}) # rubocop:disable Metrics/AbcSize,Metrics/MethodLength
14
+ Util.info "Fetching from OpenLibrary ...", key: ref.to_s
15
+
16
+ # A parsed pubid gives its digits (`raw`); `Isbn#parse` validates them
17
+ # and converts an ISBN-10 to ISBN-13, as for a String.
18
+ isbn = Isbn.new(ref.is_a?(String) ? ref : ref.raw).parse
15
19
  unless isbn
16
- Util.info "Incorrect ISBN.", key: ref
20
+ Util.info "Incorrect ISBN.", key: ref.to_s
17
21
  return
18
22
  end
19
23
 
20
24
  resp = request_api isbn
21
25
  unless resp
22
- Util.info "Not found.", key: ref
26
+ Util.info "Not found.", key: ref.to_s
23
27
  return
24
28
  end
25
29
 
26
30
  bib = Parser.parse resp
27
- Util.info "Found: `#{bib.docidentifier.first.content}`", key: ref
31
+ Util.info "Found: `#{bib.docidentifier.first.content}`", key: ref.to_s
28
32
  bib
29
33
  end
30
34
 
@@ -8,6 +8,7 @@ module Relaton
8
8
  def initialize # rubocop:disable Lint/MissingSuper
9
9
  @short = :relaton_isbn
10
10
  @prefix = "ISBN"
11
+ @pubid_identifier = :Isbn # Db cache key
11
12
  @defaultprefix = /^ISBN\s/
12
13
  @idtype = "ISBN"
13
14
  @datasets = %w[]
@@ -22,6 +23,15 @@ module Relaton
22
23
  ::Relaton::Isbn::OpenLibrary.get(code, date, opts)
23
24
  end
24
25
 
26
+ # `OpenLibrary.get` reads an incorrect ISBN as a miss, so a reference
27
+ # pubid cannot parse (it checks the check digit too) gets no key here
28
+ # either: it is not cached.
29
+ def cache_pubid(ref)
30
+ super
31
+ rescue StandardError
32
+ nil
33
+ end
34
+
25
35
  #
26
36
  # @param xml [String]
27
37
  # @return [Relaton::Bib::Bibitem]