relaton 3.0.0.pre.alpha.4 → 3.0.0.pre.alpha.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. checksums.yaml +4 -4
  2. data/lib/relaton/3gpp/bibliography.rb +19 -6
  3. data/lib/relaton/3gpp/processor.rb +6 -0
  4. data/lib/relaton/adobe/processor.rb +9 -0
  5. data/lib/relaton/bib/converter/csl.rb +110 -0
  6. data/lib/relaton/bib/converter/ris.rb +104 -0
  7. data/lib/relaton/bib/converter/titles.rb +24 -0
  8. data/lib/relaton/bib/item_data.rb +23 -0
  9. data/lib/relaton/bib/model/item.rb +6 -6
  10. data/lib/relaton/bib/sanitizer.rb +71 -30
  11. data/lib/relaton/bib.rb +10 -0
  12. data/lib/relaton/bipm/bibliography.rb +24 -21
  13. data/lib/relaton/bipm/processor.rb +1 -0
  14. data/lib/relaton/bipm/rawdata_bipm_metrologia/affiliations.rb +6 -6
  15. data/lib/relaton/bipm/si_brochure_parser.rb +2 -2
  16. data/lib/relaton/bsi/bibliography.rb +26 -23
  17. data/lib/relaton/bsi/processor.rb +5 -0
  18. data/lib/relaton/calconnect/bibliography.rb +6 -6
  19. data/lib/relaton/calconnect/hit_collection.rb +4 -1
  20. data/lib/relaton/ccsds/bibliography.rb +16 -10
  21. data/lib/relaton/ccsds/hit_collection.rb +1 -1
  22. data/lib/relaton/ccsds/processor.rb +13 -0
  23. data/lib/relaton/cen/bibliography.rb +27 -24
  24. data/lib/relaton/cen/hit_collection.rb +1 -1
  25. data/lib/relaton/cen/processor.rb +5 -0
  26. data/lib/relaton/cen/scraper.rb +9 -9
  27. data/lib/relaton/cie/bibliography.rb +11 -9
  28. data/lib/relaton/cie/data_fetcher.rb +18 -18
  29. data/lib/relaton/cie/processor.rb +1 -0
  30. data/lib/relaton/cie/scrapper.rb +15 -8
  31. data/lib/relaton/cie.rb +1 -1
  32. data/lib/relaton/cloud.rb +127 -0
  33. data/lib/relaton/core/hit_collection.rb +6 -10
  34. data/lib/relaton/core/processor.rb +121 -0
  35. data/lib/relaton/core/request_error.rb +21 -3
  36. data/lib/relaton/db/cache.rb +474 -148
  37. data/lib/relaton/db/cache_entry.rb +42 -0
  38. data/lib/relaton/db/registry.rb +152 -0
  39. data/lib/relaton/db.rb +275 -102
  40. data/lib/relaton/doi/crossref.rb +23 -6
  41. data/lib/relaton/doi/processor.rb +1 -0
  42. data/lib/relaton/easc/processor.rb +1 -0
  43. data/lib/relaton/ecma/bibliography.rb +18 -13
  44. data/lib/relaton/ecma/data_fetcher.rb +1 -1
  45. data/lib/relaton/ecma/data_parser.rb +1 -1
  46. data/lib/relaton/ecma/edition_parser.rb +2 -2
  47. data/lib/relaton/ecma/memento_parser.rb +4 -4
  48. data/lib/relaton/ecma/standard_parser.rb +6 -6
  49. data/lib/relaton/etsi/bibliography.rb +8 -7
  50. data/lib/relaton/etsi/processor.rb +1 -0
  51. data/lib/relaton/gb/bibliography.rb +20 -12
  52. data/lib/relaton/gb/gb_scraper.rb +5 -5
  53. data/lib/relaton/gb/scraper.rb +16 -16
  54. data/lib/relaton/gb/sec_scraper.rb +8 -8
  55. data/lib/relaton/gb/t_scraper.rb +5 -5
  56. data/lib/relaton/gost/processor.rb +1 -0
  57. data/lib/relaton/iala/bibliography.rb +15 -12
  58. data/lib/relaton/iala/processor.rb +1 -0
  59. data/lib/relaton/iana/bibliography.rb +12 -12
  60. data/lib/relaton/iana/data_fetcher.rb +3 -3
  61. data/lib/relaton/iana/parser.rb +12 -5
  62. data/lib/relaton/iana/processor.rb +9 -0
  63. data/lib/relaton/iec/bibliography.rb +18 -7
  64. data/lib/relaton/iec/data_parser.rb +24 -9
  65. data/lib/relaton/iec/processor.rb +11 -0
  66. data/lib/relaton/iec.rb +1 -1
  67. data/lib/relaton/ieee/bibliography.rb +19 -13
  68. data/lib/relaton/ieee/data_fetcher.rb +1 -1
  69. data/lib/relaton/ieee/processor.rb +24 -0
  70. data/lib/relaton/ieee/rawbib_id_parser.rb +2 -2
  71. data/lib/relaton/ietf/bibliography.rb +11 -9
  72. data/lib/relaton/ietf/data_fetcher.rb +1 -1
  73. data/lib/relaton/ietf/processor.rb +1 -0
  74. data/lib/relaton/ietf/rfc/entry.rb +15 -19
  75. data/lib/relaton/ietf/scraper.rb +5 -4
  76. data/lib/relaton/iho/processor.rb +1 -0
  77. data/lib/relaton/index/file_io.rb +2 -2
  78. data/lib/relaton/index/pool.rb +4 -3
  79. data/lib/relaton/index/shard_source.rb +1 -1
  80. data/lib/relaton/index/type.rb +2 -5
  81. data/lib/relaton/isbn/open_library.rb +11 -7
  82. data/lib/relaton/isbn/processor.rb +10 -0
  83. data/lib/relaton/iso/bibliography.rb +11 -9
  84. data/lib/relaton/iso/data_parser.rb +2 -2
  85. data/lib/relaton/iso/processor.rb +5 -0
  86. data/lib/relaton/iso/scraper.rb +20 -20
  87. data/lib/relaton/itu/bibliography.rb +107 -16
  88. data/lib/relaton/itu/data_crawler_r.rb +2 -2
  89. data/lib/relaton/itu/hit_collection.rb +64 -52
  90. data/lib/relaton/itu/processor.rb +1 -0
  91. data/lib/relaton/itu/scraper.rb +13 -3
  92. data/lib/relaton/itu.rb +0 -2
  93. data/lib/relaton/jis/bibliography.rb +22 -11
  94. data/lib/relaton/jis/data_fetcher.rb +4 -4
  95. data/lib/relaton/jis/processor.rb +1 -0
  96. data/lib/relaton/jis/scraper.rb +8 -8
  97. data/lib/relaton/nist/bibliography.rb +25 -21
  98. data/lib/relaton/oasis/bibliography.rb +16 -13
  99. data/lib/relaton/oasis/browser_agent.rb +2 -2
  100. data/lib/relaton/oasis/data_parser.rb +5 -5
  101. data/lib/relaton/oasis/data_parser_utils.rb +2 -2
  102. data/lib/relaton/oasis/data_part_parser.rb +7 -7
  103. data/lib/relaton/ogc/bibliography.rb +15 -14
  104. data/lib/relaton/ogc/hit_collection.rb +9 -7
  105. data/lib/relaton/ogc/processor.rb +13 -0
  106. data/lib/relaton/oiml/processor.rb +1 -0
  107. data/lib/relaton/omg/bibliography.rb +9 -9
  108. data/lib/relaton/omg/scraper.rb +14 -13
  109. data/lib/relaton/omg.rb +1 -1
  110. data/lib/relaton/plateau/bibliography.rb +9 -7
  111. data/lib/relaton/plateau/hit_collection.rb +2 -1
  112. data/lib/relaton/plateau/processor.rb +1 -0
  113. data/lib/relaton/un/bibliography.rb +21 -10
  114. data/lib/relaton/un/processor.rb +1 -0
  115. data/lib/relaton/version.rb +1 -1
  116. data/lib/relaton/w3c/processor.rb +1 -0
  117. data/lib/relaton/xsf/bibliography.rb +11 -6
  118. data/lib/relaton.rb +13 -0
  119. metadata +60 -14
  120. data/lib/relaton/itu/pubid.rb +0 -199
@@ -43,7 +43,8 @@ module Relaton
43
43
  # `ECMA-418` does not match `ECMA-418-1`. The class must be identical,
44
44
  # which keeps `ECMA-100` and `ECMA TR/100` apart.
45
45
  #
46
- # @param ref [String] the ECMA reference (e.g. "ECMA-6", "ECMA-269 ed3 vol2")
46
+ # @param ref [String, Pubid::Ecma::Identifier] the ECMA reference
47
+ # (e.g. "ECMA-6", "ECMA-269 ed3 vol2"), or its parse
47
48
  #
48
49
  # @return [Array<Hash>] matching index rows
49
50
  #
@@ -62,7 +63,7 @@ module Relaton
62
63
  # regex accepted, so every reference shape the flavor has to handle
63
64
  # parses without normalization here.
64
65
  #
65
- # @param ref [String]
66
+ # @param ref [String, Pubid::Ecma::Identifier]
66
67
  # @return [Pubid::Ecma::Identifier, nil]
67
68
  #
68
69
  # An unrecognized reference **raises**; like ISO, ETSI and 3GPP we let it
@@ -73,20 +74,24 @@ module Relaton
73
74
  # Rescuing here would collapse "this identifier is malformed" into "no
74
75
  # such document", leaving a caller unable to tell them apart.
75
76
  def parse_ref(ref)
77
+ # A pubid from Relaton::Db (relaton#205) is used as it is.
78
+ return ref unless ref.is_a?(String)
79
+
76
80
  ::Pubid::Ecma::Identifier.parse ref.to_s.strip
77
81
  end
78
82
 
79
- # @param code [String] the ECMA standard Code to look up (e..g "ECMA-6")
83
+ # @param ref [String, Pubid::Ecma::Identifier] the ECMA reference
84
+ # (e.g. "ECMA-6"), or its parse from Relaton::Db (relaton#205)
80
85
  # @param year [String] not used
81
86
  # @param opts [Hash] not used
82
87
  # @return [Relaton::Ecma::ItemData] Relaton of reference
83
- def get(code, _year = nil, _opts = {})
84
- Util.info "Fetching from Relaton repository ...", key: code
85
- result = fetch_doc(code)
88
+ def get(ref, _year = nil, _opts = {})
89
+ Util.info "Fetching from Relaton repository ...", key: ref.to_s
90
+ result = fetch_doc(ref)
86
91
  if result
87
- Util.info "Found: `#{result.docidentifier.first.content}`", key: code
92
+ Util.info "Found: `#{result.docidentifier.first.content}`", key: ref.to_s
88
93
  else
89
- Util.info "Not found.", key: code
94
+ Util.info "Not found.", key: ref.to_s
90
95
  end
91
96
  result
92
97
  end
@@ -106,7 +111,7 @@ module Relaton
106
111
  #
107
112
  # `r[:file]` breaks the tie, because the index sort is not stable.
108
113
  #
109
- # @param ref [String]
114
+ # @param ref [String, Pubid::Ecma::Identifier]
110
115
  # @return [Hash, nil]
111
116
  #
112
117
  def best_match(ref)
@@ -127,8 +132,8 @@ module Relaton
127
132
  edition.to_s.split(".").map(&:to_i)
128
133
  end
129
134
 
130
- def fetch_doc(code)
131
- row = best_match code
135
+ def fetch_doc(ref)
136
+ row = best_match ref
132
137
  return unless row
133
138
 
134
139
  url = "#{ENDPOINT}#{row[:file]}"
@@ -137,11 +142,11 @@ module Relaton
137
142
  rescue Mechanize::ResponseCodeError => e
138
143
  return if e.response_code == "404"
139
144
 
140
- raise Relaton::RequestError, "No document found for #{code} reference. #{e.message}"
145
+ raise Relaton::RequestError, "No document found for #{ref} reference. #{e.message}"
141
146
  rescue Mechanize::RedirectLimitReachedError, Timeout::Error,
142
147
  Mechanize::UnauthorizedError, Mechanize::UnsupportedSchemeError,
143
148
  Mechanize::ResponseReadError, Mechanize::ChunkedTerminationError => e
144
- raise Relaton::RequestError, "No document found for #{code} reference. #{e.message}"
149
+ raise Relaton::RequestError, "No document found for #{ref} reference. #{e.message}"
145
150
  end
146
151
  end
147
152
  end
@@ -128,7 +128,7 @@ module Relaton
128
128
  def to_yaml(bib) = bib.to_yaml
129
129
  def to_bibxml(bib) = bib.to_rfcxml
130
130
 
131
- # @param hit [Nokogiri::XML::Element]
131
+ # @param hit [Moxml::Element]
132
132
  def parse_page(hit) # rubocop:disable Metrics/AbcSize, Metrics/MethodLength
133
133
  DataParser.new(hit, @errors).parse.each { |item| write_file item }
134
134
  end
@@ -6,7 +6,7 @@ module Relaton
6
6
  #
7
7
  # Initialize parser
8
8
  #
9
- # @param [Nokogiri::XML::Element] hit document hit
9
+ # @param [Moxml::Element] hit document hit
10
10
  # @param [Hash] errors error tracking hash
11
11
  #
12
12
  def initialize(hit, errors = {})
@@ -21,7 +21,7 @@ module Relaton
21
21
  docid = @bib[:docidentifier]
22
22
  @doc.xpath('//div[@id="main"]/div[1]/div/main/article/div/div/standard/div/ul/li').map do |hit|
23
23
  bib = @bib.dup
24
- id, ed, bib[:date], vol = edition_id_parts hit.at("./span", "./a").text
24
+ id, ed, bib[:date], vol = edition_id_parts hit.at_xpath("./span|./a").text
25
25
  bib[:source] = edition_source(hit) + edition_translation_source(ed)
26
26
  next if ed.nil? || ed.empty?
27
27
 
@@ -56,7 +56,7 @@ module Relaton
56
56
  end
57
57
 
58
58
  def edition_source(hit)
59
- es = { "src" => hit.at("./a"), "pdf" => hit.at("./span/a") }.map do |type, a|
59
+ es = { "src" => hit.at_xpath("./a"), "pdf" => hit.at_xpath("./span/a") }.map do |type, a|
60
60
  Bib::Uri.new(type: type, content: a[:href]) if a
61
61
  end.compact
62
62
  @errors[:edition_source] &&= es.empty?
@@ -5,7 +5,7 @@ module Relaton
5
5
 
6
6
  ATTRS = %i[docidentifier title date source ext].freeze
7
7
 
8
- # @param [Nokogiri::XML::Element] hit document hit
8
+ # @param [Moxml::Element] hit document hit
9
9
  # @param [Hash] errors error tracking hash
10
10
  def initialize(hit:, errors: {})
11
11
  @hit = hit
@@ -23,7 +23,7 @@ module Relaton
23
23
 
24
24
  # @return [Array<Relaton::Ecma::Docidentifier>]
25
25
  def fetch_docidentifier
26
- code = "ECMA MEM/#{@hit.at('div[1]//p').text}"
26
+ code = "ECMA MEM/#{@hit.at_xpath('div[1]//p').text}"
27
27
  docid = super(code)
28
28
  @errors[:memento_docidentifier] &&= docid.empty?
29
29
  docid
@@ -31,7 +31,7 @@ module Relaton
31
31
 
32
32
  # @return [Array<Relaton::Bib::Title>]
33
33
  def fetch_title
34
- year = @hit.at("div[1]//p").text
34
+ year = @hit.at_xpath("div[1]//p").text
35
35
  content = "\"Memento #{year}\" for year #{year}"
36
36
  result = [Bib::Title.new(content: content, language: "en", script: "Latn")]
37
37
  @errors[:memento_title] &&= result.empty?
@@ -40,7 +40,7 @@ module Relaton
40
40
 
41
41
  # @return [Array<Relaton::Bib::Date>]
42
42
  def fetch_date
43
- date = @hit.at("div[2]//p").text
43
+ date = @hit.at_xpath("div[2]//p").text
44
44
  on = Date.strptime(date, "%B %Y").strftime "%Y-%m"
45
45
  result = [Bib::Date.new(type: "published", at: on)]
46
46
  @errors[:memento_date] &&= result.empty?
@@ -5,7 +5,7 @@ module Relaton
5
5
 
6
6
  ATTRS = %i[docidentifier title date source abstract relation edition ext].freeze
7
7
 
8
- # @param [Nokogiri::XML::Element] hit document hit
8
+ # @param [Moxml::Element] hit document hit
9
9
  # @param [Mechanize::Page] doc fetched document page
10
10
  # @param [Hash] errors error tracking hash
11
11
  def initialize(hit:, doc:, errors: {})
@@ -68,7 +68,7 @@ module Relaton
68
68
  def fetch_source # rubocop:disable Metrics/AbcSize
69
69
  source = []
70
70
  source << Bib::Uri.new(type: "src", content: @hit[:href]) if @hit[:href]
71
- ref = @doc.at('//div[@class="ecma-item-content-wrapper"]/span/a',
71
+ ref = @doc.at_xpath('//div[@class="ecma-item-content-wrapper"]/span/a',
72
72
  '//div[@class="ecma-item-content-wrapper"]/a')
73
73
  source << Bib::Uri.new(type: "pdf", content: ref[:href]) if ref
74
74
  result = source + edition_translation_source(fetch_edition_content)
@@ -80,7 +80,7 @@ module Relaton
80
80
  def fetch_relation # rubocop:disable Metrics/AbcSize, Metrics/MethodLength, Metrics/CyclomaticComplexity
81
81
  edition_parser = EditionParser.new(doc: @doc, bib: {}, errors: @errors)
82
82
  result = @doc.xpath("//ul[@class='ecma-item-archives']/li").filter_map do |rel|
83
- ref, ed, date, vol = edition_parser.edition_id_parts rel.at("span").text
83
+ ref, ed, date, vol = edition_parser.edition_id_parts rel.at_xpath("span").text
84
84
  next if ed.nil? || ed.empty?
85
85
 
86
86
  docid = Docidentifier.new(type: "ECMA", content: ref, primary: true)
@@ -109,7 +109,7 @@ module Relaton
109
109
  private
110
110
 
111
111
  def fetch_edition_content
112
- @doc.at('//p[@class="ecma-item-edition"]')&.text&.match(/^\d+(?=(?:st|nd|th|rd))/)&.to_s
112
+ @doc.at_xpath('//p[@class="ecma-item-edition"]')&.text&.match(/^\d+(?=(?:st|nd|th|rd))/)&.to_s
113
113
  end
114
114
 
115
115
  def edition_translation_source(edition)
@@ -120,8 +120,8 @@ module Relaton
120
120
  return [] unless @doc
121
121
 
122
122
  @doc.xpath("//h2[.='Translations']/following-sibling::ul/li").map do |l|
123
- a = l.at("span/a")
124
- id = l.at("span").text
123
+ a = l.at_xpath("span/a")
124
+ id = l.at_xpath("span").text
125
125
  %r{\w+[\d-]+,\s(?<lang>\w+)\sversion,\s(?<ed>[\d.]+)(?:st|nd|rd|th)\sedition} =~ id
126
126
  case lang
127
127
  when "Japanese"
@@ -6,14 +6,15 @@ module Relaton
6
6
  module Bibliography
7
7
  SOURCE = "https://raw.githubusercontent.com/relaton/relaton-data-etsi/refs/heads/v2/"
8
8
 
9
- # @param text [String]
9
+ # @param ref [String, ::Pubid::Etsi::Identifier]
10
10
  # @return [Relaton::Etsi::ItemData, nil]
11
- def search(text) # rubocop:disable Metrics/MethodLength, Metrics/AbcSize
11
+ def search(ref) # rubocop:disable Metrics/MethodLength, Metrics/AbcSize
12
12
  # An unrecognized reference raises Pubid::Errors::ParseError; like
13
13
  # ISO we let it propagate — the CLI turns it into a friendly message
14
14
  # and API callers rescue it themselves. Valid partial refs parse with
15
15
  # the omitted refinements (version/date/part) left blank.
16
- pubid = ::Pubid::Etsi.parse text
16
+ # A parsed pubid comes from Relaton::Db (relaton#205); it is only read.
17
+ pubid = ref.is_a?(String) ? ::Pubid::Etsi.parse(ref) : ref
17
18
 
18
19
  index = Relaton::Index.find_or_create :etsi, url: "#{SOURCE}#{INDEXFILE}.zip", file: "#{INDEXFILE}.yaml",
19
20
  pubid_class: ::Pubid::Etsi::Identifier
@@ -89,19 +90,19 @@ module Relaton
89
90
  [(id.version&.version).to_s.split(".").map(&:to_i), id.date.to_s]
90
91
  end
91
92
 
92
- # @param ref [String] the ETSI standard Code to look up
93
+ # @param ref [String, ::Pubid::Etsi::Identifier] the ETSI standard Code to look up
93
94
  # @param year [String, nil] year
94
95
  # @param opts [Hash] options
95
96
  # @return [Relaton::Etsi::ItemData, nil]
96
97
  def get(ref, _year = nil, _opts = {})
97
- Util.info "Fetching from Relaton repository ...", key: ref
98
+ Util.info "Fetching from Relaton repository ...", key: ref.to_s
98
99
  result = search(ref)
99
100
  unless result
100
- Util.info "Not found.", key: ref
101
+ Util.info "Not found.", key: ref.to_s
101
102
  return
102
103
  end
103
104
 
104
- Util.info "Found: `#{result.docidentifier[0].content}`", key: ref
105
+ Util.info "Found: `#{result.docidentifier[0].content}`", key: ref.to_s
105
106
  result
106
107
  end
107
108
 
@@ -8,6 +8,7 @@ module Relaton
8
8
  def initialize # rubocop:disable Lint/MissingSuper
9
9
  @short = :relaton_etsi
10
10
  @prefix = "ETSI"
11
+ @pubid_identifier = :Etsi # Db cache key
11
12
  @defaultprefix = %r{^ETSI\s}
12
13
  @idtype = "ETSI"
13
14
  @datasets = %w[etsi-csv]
@@ -9,15 +9,15 @@ module Relaton
9
9
  class Bibliography
10
10
  class << self
11
11
  # rubocop:disable Metrics/MethodLength
12
- # @param text [Strin] code of standard for search
12
+ # @param ref [Strin] code of standard for search
13
13
  # @return [RelatonGb::HitCollection]
14
- def search(text)
15
- case text
14
+ def search(ref)
15
+ case ref
16
16
  when /^(GB|GJ|GS)/
17
17
  # Scrape national standards.
18
- Util.info "Fetching from openstd.samr.gov.cn ...", key: text
18
+ Util.info "Fetching from openstd.samr.gov.cn ...", key: ref
19
19
  require_relative "gb_scraper"
20
- GbScraper.scrape_page text
20
+ GbScraper.scrape_page ref
21
21
  # when /^ZB/
22
22
  # Scrape proffesional.
23
23
  # when /^DB/
@@ -26,26 +26,34 @@ module Relaton
26
26
  # Enterprise standard
27
27
  when %r{^T/[^\s]{2,6}\s}
28
28
  # Scrape social standard.
29
- Util.info "Fetching from www.ttbz.org.cn ...", key: text
29
+ Util.info "Fetching from www.ttbz.org.cn ...", key: ref
30
30
  require_relative "t_scraper"
31
- TScraper.scrape_page text
31
+ TScraper.scrape_page ref
32
32
  else
33
33
  # Scrape sector standard.
34
34
  require "relaton/gb/sec_scraper"
35
- SecScraper.scrape_page text
35
+ SecScraper.scrape_page ref
36
36
  end
37
37
  end
38
38
  # rubocop:enable Metrics/MethodLength
39
39
 
40
- # @param code [String] the GB standard Code to look up (e..g "GB/T 20223")
40
+ # @param ref [String, Pubid::Gb::Identifier] the GB standard Code to
41
+ # look up (e.g. "GB/T 20223"), or its parse from Relaton::Db
42
+ # (relaton#205)
41
43
  # @param year [String] the year the standard was published (optional)
42
44
  # @param opts [Hash] options; restricted to :all_parts if all-parts reference is required
43
45
  # @return [Relaton::Gb::ItemData, nil]
44
- # @raise [Pubid::Errors::ParseError] when the code is not a GB
46
+ # @raise [Pubid::Errors::ParseError] when the reference is not a GB
45
47
  # identifier
46
- def get(code, year = nil, opts = {})
48
+ def get(ref, year = nil, opts = {}) # rubocop:disable Metrics/AbcSize,Metrics/CyclomaticComplexity,Metrics/PerceivedComplexity
47
49
  require "pubid"
48
- pubid = ::Pubid::Gb::Identifier.parse(code)
50
+ # A parsed pubid comes from Relaton::Db (relaton#205); it is never
51
+ # mutated (`exclude` below copies).
52
+ pubid = ref.is_a?(String) ? ::Pubid::Gb::Identifier.parse(ref) : ref
53
+ if pubid.all_parts?
54
+ opts = opts.merge(all_parts: true)
55
+ pubid = pubid.identifiers.first
56
+ end
49
57
  year = (year || pubid.year)&.to_s
50
58
  pubid = pubid.exclude(:year)
51
59
  pubid.part = "1" if opts[:all_parts]
@@ -1,7 +1,7 @@
1
1
  # encoding: UTF-8
2
2
  # frozen_string_literal: true
3
3
 
4
- require "nokogiri"
4
+ require "moxml"
5
5
  require_relative "scraper"
6
6
 
7
7
  module Relaton
@@ -20,10 +20,10 @@ module Relaton
20
20
  hits = doc.xpath(
21
21
  "//table[contains(@class, 'result_list')]/tbody[2]/tr",
22
22
  ).map do |h|
23
- ref = h.at "./td[2]/a"
23
+ ref = h.at_xpath "./td[2]/a"
24
24
  pid = ref[:onclick].match(/[0-9A-F]+/).to_s
25
- status = h.at("./td[7]").text.strip
26
- rdate = h.at("./td[8]").text.strip
25
+ status = h.at_xpath("./td[7]").text.strip
26
+ rdate = h.at_xpath("./td[8]").text.strip
27
27
  Hit.new pid: pid, docref: ref.text, scraper: self,
28
28
  release_date: rdate, status: status
29
29
  end
@@ -52,7 +52,7 @@ module Relaton
52
52
  # * :type [String]
53
53
  # * :name [String]
54
54
  # def get_committee(doc, _ref)
55
- # name = doc.at("//div[contains(., '归口单位') or contains(., '归口部门')]/following-sibling::div")
55
+ # name = doc.at_xpath("//div[contains(., '归口单位') or contains(., '归口部门')]/following-sibling::div")
56
56
  # Committee.new(type: "technical", content: name.text.delete("\r\n\t\t"))
57
57
  # end
58
58
  end
@@ -15,7 +15,7 @@ module Relaton
15
15
 
16
16
  @prefixes = nil
17
17
 
18
- # @param doc [Nokogiri::HTML::Document]
18
+ # @param doc [Moxml::Document]
19
19
  # @param src [String]
20
20
  # @param hit [RelatonGb::Hit]
21
21
  # @return [Hash]
@@ -41,7 +41,7 @@ module Relaton
41
41
  [Docidentifier.new(content: docref, type: "Chinese Standard", primary: true)]
42
42
  end
43
43
 
44
- # @param doc [Nokogiri::HTML::Document]
44
+ # @param doc [Moxml::Document]
45
45
  # @param docref [Strings]
46
46
  # @return [Array<Relaton::Bib::Contributor>]
47
47
  def get_contributors(doc, docref)
@@ -67,22 +67,22 @@ module Relaton
67
67
  Bib::TypedLocalizedString.new language: lang, content: content
68
68
  end
69
69
 
70
- # @param doc [Nokogiri::HTML::Document]
70
+ # @param doc [Moxml::Document]
71
71
  # @return [Array<Relaton::Bib::Title>]
72
72
  def get_titles(doc)
73
- tzh = doc.at("//td[contains(text(), '中文标准名称')]/b").text
73
+ tzh = doc.at_xpath("//td[contains(text(), '中文标准名称')]/b").text
74
74
  titles = Relaton::Bib::Title.from_string tzh, "zh", "Hans"
75
- ten = doc.at("//td[contains(text(), '英文标准名称')]").text.match(/[\w\s]+/).to_s
75
+ ten = doc.at_xpath("//td[contains(text(), '英文标准名称')]").text.match(/[\w\s]+/).to_s
76
76
  return titles if ten.empty?
77
77
 
78
78
  titles + Relaton::Bib::Title.from_string(ten, "en", "Latn")
79
79
  end
80
80
 
81
- # @param doc [Nokogiri::HTML::Document]
81
+ # @param doc [Moxml::Document]
82
82
  # @param status [String, NilClass]
83
83
  # @return [Relaton::Bib::Status]
84
84
  def get_status(doc, status = nil)
85
- status ||= doc.at("//td[contains(., '标准状态')]/span")&.text&.strip
85
+ status ||= doc.at_xpath("//td[contains(., '标准状态')]/span")&.text&.strip
86
86
  return unless STAGES[status]
87
87
 
88
88
  stage = Bib::Status::Stage.new content: STAGES[status]
@@ -91,17 +91,17 @@ module Relaton
91
91
 
92
92
  private
93
93
 
94
- # @param doc [Nokogiri::HTML::Document]
94
+ # @param doc [Moxml::Document]
95
95
  # @return [Array<String>]
96
96
  def get_ccs(doc)
97
- code = doc.at("//div[contains(text(), '中国标准分类号')]/following-sibling::div").text.strip
97
+ code = doc.at_xpath("//div[contains(text(), '中国标准分类号')]/following-sibling::div").text.strip
98
98
  [CCS.new(code: code)]
99
99
  end
100
100
 
101
- # @param doc [Nokogiri::HTML::Document]
101
+ # @param doc [Moxml::Document]
102
102
  # @return [Array<Relaton::Bib::ICS>]
103
103
  def get_ics(doc)
104
- ics = doc.at("//div[contains(text(), '国际标准分类号')]/following-sibling::div"\
104
+ ics = doc.at_xpath("//div[contains(text(), '国际标准分类号')]/following-sibling::div"\
105
105
  " | //dt[contains(text(), '国际标准分类号')]/following-sibling::dd")
106
106
  return [] unless ics
107
107
 
@@ -109,10 +109,10 @@ module Relaton
109
109
  [Bib::ICS.new(code: code)]
110
110
  end
111
111
 
112
- # @param doc [Nokogiri::HTML::Document]
112
+ # @param doc [Moxml::Document]
113
113
  # @return [String]
114
114
  def get_scope(doc)
115
- issued = doc.at("//div[contains(., '发布单位')]/following-sibling::div")
115
+ issued = doc.at_xpath("//div[contains(., '发布单位')]/following-sibling::div")
116
116
  case issued&.text
117
117
  when /国家标准/ then "national"
118
118
  when /^行业标准/ then "sector"
@@ -150,12 +150,12 @@ module Relaton
150
150
  (Bib::Uri.new(type: "src", content: src))
151
151
  end
152
152
 
153
- # @param doc [Nokogiri::HTML::Document]
153
+ # @param doc [Moxml::Document]
154
154
  # @return [Array<Hash>]
155
155
  # * :type [String] type of date
156
156
  # * :on [String] date
157
157
  def get_dates(doc)
158
- date = doc.at("//div[contains(text(), '发布日期')]/following-sibling::div"\
158
+ date = doc.at_xpath("//div[contains(text(), '发布日期')]/following-sibling::div"\
159
159
  " | //dt[contains(text(), '发布日期')]/following-sibling::dd")
160
160
  [Bib::Date.new(type: "published", at: date.text.delete("\r\n\t\t"))]
161
161
  end
@@ -177,7 +177,7 @@ module Relaton
177
177
  Doctype.new content: "standard"
178
178
  end
179
179
 
180
- # @param doc [Nokogiri::HTML::Document]
180
+ # @param doc [Moxml::Document]
181
181
  # @param ref [String]
182
182
  # @return [Relaton::Gb::GbType]
183
183
  def get_gbtype(doc, ref)
@@ -3,7 +3,7 @@
3
3
 
4
4
  require "net/http"
5
5
  require "json"
6
- require "nokogiri"
6
+ require "moxml"
7
7
  require_relative "scraper"
8
8
  require_relative "item"
9
9
  require_relative "hit_collection"
@@ -44,7 +44,7 @@ module Relaton
44
44
  def scrape_doc(hit)
45
45
  src = "https://hbba.sacinfo.org.cn/stdDetail/#{hit.pid}"
46
46
  page_uri = URI src
47
- doc = Nokogiri::HTML Net::HTTP.get(page_uri)
47
+ doc = Moxml.new.parse_html Net::HTTP.get(page_uri)
48
48
  ItemData.new(**scrapped_data(doc, src, hit))
49
49
  rescue SocketError, Timeout::Error, Errno::EINVAL, Errno::ECONNRESET, EOFError,
50
50
  Net::HTTPBadResponse, Net::HTTPHeaderSyntaxError, Net::ProtocolError,
@@ -54,14 +54,14 @@ module Relaton
54
54
 
55
55
  private
56
56
 
57
- # @param doc [Nokogiri::HTML::Document]
57
+ # @param doc [Moxml::Document]
58
58
  # @return [Array<Relaton::Bib::Title>]
59
59
  def get_titles(doc)
60
- tzh = doc.at("//h4").text.delete("\r\n\t")
60
+ tzh = doc.at_xpath("//h4").text.delete("\r\n\t")
61
61
  Bib::Title.from_string(tzh, "zh", "Hans")
62
62
  end
63
63
 
64
- # @param _doc [Nokogiri::HTML::Document]
64
+ # @param _doc [Moxml::Document]
65
65
  # @param ref [String]
66
66
  # @return [Hash]
67
67
  # * :type [String]
@@ -72,16 +72,16 @@ module Relaton
72
72
  # { type: "technical", name: name }
73
73
  # end
74
74
 
75
- # @param _doc [Nokogiri::HTML::Document]
75
+ # @param _doc [Moxml::Document]
76
76
  # @return [String]
77
77
  def get_scope(_doc)
78
78
  "sector"
79
79
  end
80
80
 
81
- # @param doc [Nokogiri::HTML::Document]
81
+ # @param doc [Moxml::Document]
82
82
  # @return [Array<String>]
83
83
  def get_ccs(doc)
84
- array(doc.at("//dt[contains(text(), '中国标准分类号')]/following-sibling::dd")).map do |cc|
84
+ array(doc.at_xpath("//dt[contains(text(), '中国标准分类号')]/following-sibling::dd")).map do |cc|
85
85
  text = Cnccs.fetch(cc.text.strip)&.description
86
86
  CCS.new code: cc.text, text: text
87
87
  end
@@ -1,7 +1,7 @@
1
1
  # encoding: UTF-8
2
2
  # frozen_string_literal: true
3
3
 
4
- require "nokogiri"
4
+ require "moxml"
5
5
  require_relative "scraper"
6
6
  require_relative "hit_collection"
7
7
  require_relative "hit"
@@ -23,8 +23,8 @@ module Relaton
23
23
  xpath = '//table[contains(@class, "standard_list_table")]/tr/td/a'
24
24
  t_xpath = "../preceding-sibling::td[4]"
25
25
  hits = doc.xpath(xpath).map do |h|
26
- docref = h.at(t_xpath).text.gsub(/â\u0080\u0094/, "-")
27
- status = h.at("../preceding-sibling::td[1]").text.delete "\r\n"
26
+ docref = h.at_xpath(t_xpath).text.gsub(/â\u0080\u0094/, "-")
27
+ status = h.at_xpath("../preceding-sibling::td[1]").text.delete "\r\n"
28
28
  pid = h[:href].sub(%r{/$}, "")
29
29
  Hit.new pid: pid, docref: docref, status: status, scraper: self
30
30
  end
@@ -55,7 +55,7 @@ module Relaton
55
55
  private
56
56
 
57
57
  # rubocop:disable Metrics/MethodLength
58
- # @param doc [Nokogiri::HTML::Document]
58
+ # @param doc [Moxml::Document]
59
59
  # @param src [String]
60
60
  # @param hit [RelatonGb::Hit]
61
61
  # @return [Hash]
@@ -89,7 +89,7 @@ module Relaton
89
89
 
90
90
  def get_titles(doc)
91
91
  xpz = '//td[contains(.,"中文标题")]/following-sibling::td[1]'
92
- titles = Bib::Title.from_string doc.at(xpz)
92
+ titles = Bib::Title.from_string doc.at_xpath(xpz)
93
93
  .text, "zh", "Hans"
94
94
  xpe = '//td[contains(.,"英文标题")]/following-sibling::td[1]'
95
95
  ten = doc.xpath(xpe).text
@@ -12,6 +12,7 @@ module Relaton
12
12
  def initialize
13
13
  @short = :relaton_gost
14
14
  @prefix = "GOST"
15
+ @pubid_identifier = :Gost # Db cache key
15
16
  # Both Latin "GOST" and Cyrillic "ГОСТ" route here. The trailing
16
17
  # \b keeps the prefix from swallowing longer tokens ("GOSTA …").
17
18
  @defaultprefix = %r{^(?:GOST|ГОСТ)\b}
@@ -8,15 +8,15 @@ module Relaton
8
8
  class << self
9
9
  # Search for an IALA publication by its identifier.
10
10
  #
11
- # @param text [String] the IALA reference to look up (e.g. "IALA S1070")
11
+ # @param ref [String] the IALA reference to look up (e.g. "IALA S1070")
12
12
  # @param _year [String, nil] optional edition/year filter
13
13
  # @param _opts [Hash] options (unused)
14
14
  # @return [Relaton::Iala::Item, nil]
15
- def search(text, _year = nil, _opts = {})
16
- Util.info "Fetching from Relaton repository ...", key: text
17
- row = best_match text
15
+ def search(ref, _year = nil, _opts = {})
16
+ Util.info "Fetching from Relaton repository ...", key: ref.to_s
17
+ row = best_match ref
18
18
  unless row
19
- Util.info "Not found.", key: text
19
+ Util.info "Not found.", key: ref.to_s
20
20
  return
21
21
  end
22
22
 
@@ -27,7 +27,7 @@ module Relaton
27
27
  end
28
28
 
29
29
  item = Relaton::Iala::Item.from_yaml resp.body
30
- Util.info "Found: `#{item.docidentifier.first&.content}`", key: text
30
+ Util.info "Found: `#{item.docidentifier.first&.content}`", key: ref.to_s
31
31
  item.tap { |i| i.fetched = Date.today.to_s }
32
32
  rescue SocketError, Errno::EINVAL, Errno::ECONNRESET, EOFError,
33
33
  Net::HTTPBadResponse, Net::HTTPHeaderSyntaxError,
@@ -59,7 +59,7 @@ module Relaton
59
59
  # must never match each other. An `Annex` row carries its base
60
60
  # identifier, which `===` compares with its own subset match.
61
61
  #
62
- # @param text [String]
62
+ # @param ref [String]
63
63
  # @return [Hash, nil]
64
64
  #
65
65
  # The substring-scan fallback this used to take when parsing failed is
@@ -68,8 +68,8 @@ module Relaton
68
68
  # ids, so an ambiguous reference silently resolved to whichever row
69
69
  # sorted first, instead of telling the caller the reference is not an
70
70
  # identifier. ecma, w3c and xsf never had one.
71
- def best_match(text)
72
- pubid = parse_ref text
71
+ def best_match(ref)
72
+ pubid = parse_ref ref
73
73
  rows = index.search(pubid)
74
74
  rows.max_by { |r| [edition_key(r[:id].edition), language_key(r[:id]), r[:file]] }
75
75
  end
@@ -114,7 +114,7 @@ module Relaton
114
114
  # zero-pads the number to its type's canonical width, so `IALA M1`,
115
115
  # `M0001` and `R1016:ed2.0(F)` all parse without normalization here.
116
116
  #
117
- # @param text [String]
117
+ # @param ref [String, Pubid::Iala::Identifier]
118
118
  # @return [Pubid::Iala::Identifier, nil]
119
119
  #
120
120
  # An unrecognized reference **raises**; like ISO, ETSI and 3GPP we let it
@@ -124,8 +124,11 @@ module Relaton
124
124
  # logs it through the `StandardError` arm at `lib/relaton/db.rb:122`.
125
125
  # Rescuing here would collapse "this identifier is malformed" into "no
126
126
  # such document", leaving a caller unable to tell them apart.
127
- def parse_ref(text)
128
- ::Pubid::Iala::Identifier.parse text.to_s.strip
127
+ def parse_ref(ref)
128
+ # A parsed pubid comes from Relaton::Db (relaton#205); it is used as it is.
129
+ return ref unless ref.is_a?(String)
130
+
131
+ ::Pubid::Iala::Identifier.parse ref.to_s.strip
129
132
  end
130
133
 
131
134
  # The index is pubid-backed: `pubid_class:` is what makes
@@ -8,6 +8,7 @@ module Relaton
8
8
  def initialize
9
9
  @short = :relaton_iala
10
10
  @prefix = "IALA"
11
+ @pubid_identifier = :Iala # Db cache key
11
12
  @defaultprefix = %r{^IALA\s}
12
13
  @idtype = "IALA"
13
14
  end