relaton 3.0.0.pre.alpha.4 → 3.0.0.pre.alpha.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/lib/relaton/3gpp/bibliography.rb +19 -6
- data/lib/relaton/3gpp/processor.rb +6 -0
- data/lib/relaton/adobe/processor.rb +9 -0
- data/lib/relaton/bib/converter/csl.rb +110 -0
- data/lib/relaton/bib/converter/ris.rb +104 -0
- data/lib/relaton/bib/converter/titles.rb +24 -0
- data/lib/relaton/bib/item_data.rb +23 -0
- data/lib/relaton/bib/model/item.rb +6 -6
- data/lib/relaton/bib/sanitizer.rb +71 -30
- data/lib/relaton/bib.rb +10 -0
- data/lib/relaton/bipm/bibliography.rb +24 -21
- data/lib/relaton/bipm/processor.rb +1 -0
- data/lib/relaton/bipm/rawdata_bipm_metrologia/affiliations.rb +6 -6
- data/lib/relaton/bipm/si_brochure_parser.rb +2 -2
- data/lib/relaton/bsi/bibliography.rb +26 -23
- data/lib/relaton/bsi/processor.rb +5 -0
- data/lib/relaton/calconnect/bibliography.rb +6 -6
- data/lib/relaton/calconnect/hit_collection.rb +4 -1
- data/lib/relaton/ccsds/bibliography.rb +16 -10
- data/lib/relaton/ccsds/hit_collection.rb +1 -1
- data/lib/relaton/ccsds/processor.rb +13 -0
- data/lib/relaton/cen/bibliography.rb +27 -24
- data/lib/relaton/cen/hit_collection.rb +1 -1
- data/lib/relaton/cen/processor.rb +5 -0
- data/lib/relaton/cen/scraper.rb +9 -9
- data/lib/relaton/cie/bibliography.rb +11 -9
- data/lib/relaton/cie/data_fetcher.rb +18 -18
- data/lib/relaton/cie/processor.rb +1 -0
- data/lib/relaton/cie/scrapper.rb +15 -8
- data/lib/relaton/cie.rb +1 -1
- data/lib/relaton/cloud.rb +127 -0
- data/lib/relaton/core/hit_collection.rb +6 -10
- data/lib/relaton/core/processor.rb +121 -0
- data/lib/relaton/core/request_error.rb +21 -3
- data/lib/relaton/db/cache.rb +474 -148
- data/lib/relaton/db/cache_entry.rb +42 -0
- data/lib/relaton/db/registry.rb +152 -0
- data/lib/relaton/db.rb +275 -102
- data/lib/relaton/doi/crossref.rb +23 -6
- data/lib/relaton/doi/processor.rb +1 -0
- data/lib/relaton/easc/processor.rb +1 -0
- data/lib/relaton/ecma/bibliography.rb +18 -13
- data/lib/relaton/ecma/data_fetcher.rb +1 -1
- data/lib/relaton/ecma/data_parser.rb +1 -1
- data/lib/relaton/ecma/edition_parser.rb +2 -2
- data/lib/relaton/ecma/memento_parser.rb +4 -4
- data/lib/relaton/ecma/standard_parser.rb +6 -6
- data/lib/relaton/etsi/bibliography.rb +8 -7
- data/lib/relaton/etsi/processor.rb +1 -0
- data/lib/relaton/gb/bibliography.rb +20 -12
- data/lib/relaton/gb/gb_scraper.rb +5 -5
- data/lib/relaton/gb/scraper.rb +16 -16
- data/lib/relaton/gb/sec_scraper.rb +8 -8
- data/lib/relaton/gb/t_scraper.rb +5 -5
- data/lib/relaton/gost/processor.rb +1 -0
- data/lib/relaton/iala/bibliography.rb +15 -12
- data/lib/relaton/iala/processor.rb +1 -0
- data/lib/relaton/iana/bibliography.rb +12 -12
- data/lib/relaton/iana/data_fetcher.rb +3 -3
- data/lib/relaton/iana/parser.rb +12 -5
- data/lib/relaton/iana/processor.rb +9 -0
- data/lib/relaton/iec/bibliography.rb +18 -7
- data/lib/relaton/iec/data_parser.rb +24 -9
- data/lib/relaton/iec/processor.rb +11 -0
- data/lib/relaton/iec.rb +1 -1
- data/lib/relaton/ieee/bibliography.rb +19 -13
- data/lib/relaton/ieee/data_fetcher.rb +1 -1
- data/lib/relaton/ieee/processor.rb +24 -0
- data/lib/relaton/ieee/rawbib_id_parser.rb +2 -2
- data/lib/relaton/ietf/bibliography.rb +11 -9
- data/lib/relaton/ietf/data_fetcher.rb +1 -1
- data/lib/relaton/ietf/processor.rb +1 -0
- data/lib/relaton/ietf/rfc/entry.rb +15 -19
- data/lib/relaton/ietf/scraper.rb +5 -4
- data/lib/relaton/iho/processor.rb +1 -0
- data/lib/relaton/index/file_io.rb +2 -2
- data/lib/relaton/index/pool.rb +4 -3
- data/lib/relaton/index/shard_source.rb +1 -1
- data/lib/relaton/index/type.rb +2 -5
- data/lib/relaton/isbn/open_library.rb +11 -7
- data/lib/relaton/isbn/processor.rb +10 -0
- data/lib/relaton/iso/bibliography.rb +11 -9
- data/lib/relaton/iso/data_parser.rb +2 -2
- data/lib/relaton/iso/processor.rb +5 -0
- data/lib/relaton/iso/scraper.rb +20 -20
- data/lib/relaton/itu/bibliography.rb +107 -16
- data/lib/relaton/itu/data_crawler_r.rb +2 -2
- data/lib/relaton/itu/hit_collection.rb +64 -52
- data/lib/relaton/itu/processor.rb +1 -0
- data/lib/relaton/itu/scraper.rb +13 -3
- data/lib/relaton/itu.rb +0 -2
- data/lib/relaton/jis/bibliography.rb +22 -11
- data/lib/relaton/jis/data_fetcher.rb +4 -4
- data/lib/relaton/jis/processor.rb +1 -0
- data/lib/relaton/jis/scraper.rb +8 -8
- data/lib/relaton/nist/bibliography.rb +25 -21
- data/lib/relaton/oasis/bibliography.rb +16 -13
- data/lib/relaton/oasis/browser_agent.rb +2 -2
- data/lib/relaton/oasis/data_parser.rb +5 -5
- data/lib/relaton/oasis/data_parser_utils.rb +2 -2
- data/lib/relaton/oasis/data_part_parser.rb +7 -7
- data/lib/relaton/ogc/bibliography.rb +15 -14
- data/lib/relaton/ogc/hit_collection.rb +9 -7
- data/lib/relaton/ogc/processor.rb +13 -0
- data/lib/relaton/oiml/processor.rb +1 -0
- data/lib/relaton/omg/bibliography.rb +9 -9
- data/lib/relaton/omg/scraper.rb +14 -13
- data/lib/relaton/omg.rb +1 -1
- data/lib/relaton/plateau/bibliography.rb +9 -7
- data/lib/relaton/plateau/hit_collection.rb +2 -1
- data/lib/relaton/plateau/processor.rb +1 -0
- data/lib/relaton/un/bibliography.rb +21 -10
- data/lib/relaton/un/processor.rb +1 -0
- data/lib/relaton/version.rb +1 -1
- data/lib/relaton/w3c/processor.rb +1 -0
- data/lib/relaton/xsf/bibliography.rb +11 -6
- data/lib/relaton.rb +13 -0
- metadata +60 -14
- data/lib/relaton/itu/pubid.rb +0 -199
|
@@ -43,7 +43,8 @@ module Relaton
|
|
|
43
43
|
# `ECMA-418` does not match `ECMA-418-1`. The class must be identical,
|
|
44
44
|
# which keeps `ECMA-100` and `ECMA TR/100` apart.
|
|
45
45
|
#
|
|
46
|
-
# @param ref [String] the ECMA reference
|
|
46
|
+
# @param ref [String, Pubid::Ecma::Identifier] the ECMA reference
|
|
47
|
+
# (e.g. "ECMA-6", "ECMA-269 ed3 vol2"), or its parse
|
|
47
48
|
#
|
|
48
49
|
# @return [Array<Hash>] matching index rows
|
|
49
50
|
#
|
|
@@ -62,7 +63,7 @@ module Relaton
|
|
|
62
63
|
# regex accepted, so every reference shape the flavor has to handle
|
|
63
64
|
# parses without normalization here.
|
|
64
65
|
#
|
|
65
|
-
# @param ref [String]
|
|
66
|
+
# @param ref [String, Pubid::Ecma::Identifier]
|
|
66
67
|
# @return [Pubid::Ecma::Identifier, nil]
|
|
67
68
|
#
|
|
68
69
|
# An unrecognized reference **raises**; like ISO, ETSI and 3GPP we let it
|
|
@@ -73,20 +74,24 @@ module Relaton
|
|
|
73
74
|
# Rescuing here would collapse "this identifier is malformed" into "no
|
|
74
75
|
# such document", leaving a caller unable to tell them apart.
|
|
75
76
|
def parse_ref(ref)
|
|
77
|
+
# A pubid from Relaton::Db (relaton#205) is used as it is.
|
|
78
|
+
return ref unless ref.is_a?(String)
|
|
79
|
+
|
|
76
80
|
::Pubid::Ecma::Identifier.parse ref.to_s.strip
|
|
77
81
|
end
|
|
78
82
|
|
|
79
|
-
# @param
|
|
83
|
+
# @param ref [String, Pubid::Ecma::Identifier] the ECMA reference
|
|
84
|
+
# (e.g. "ECMA-6"), or its parse from Relaton::Db (relaton#205)
|
|
80
85
|
# @param year [String] not used
|
|
81
86
|
# @param opts [Hash] not used
|
|
82
87
|
# @return [Relaton::Ecma::ItemData] Relaton of reference
|
|
83
|
-
def get(
|
|
84
|
-
Util.info "Fetching from Relaton repository ...", key:
|
|
85
|
-
result = fetch_doc(
|
|
88
|
+
def get(ref, _year = nil, _opts = {})
|
|
89
|
+
Util.info "Fetching from Relaton repository ...", key: ref.to_s
|
|
90
|
+
result = fetch_doc(ref)
|
|
86
91
|
if result
|
|
87
|
-
Util.info "Found: `#{result.docidentifier.first.content}`", key:
|
|
92
|
+
Util.info "Found: `#{result.docidentifier.first.content}`", key: ref.to_s
|
|
88
93
|
else
|
|
89
|
-
Util.info "Not found.", key:
|
|
94
|
+
Util.info "Not found.", key: ref.to_s
|
|
90
95
|
end
|
|
91
96
|
result
|
|
92
97
|
end
|
|
@@ -106,7 +111,7 @@ module Relaton
|
|
|
106
111
|
#
|
|
107
112
|
# `r[:file]` breaks the tie, because the index sort is not stable.
|
|
108
113
|
#
|
|
109
|
-
# @param ref [String]
|
|
114
|
+
# @param ref [String, Pubid::Ecma::Identifier]
|
|
110
115
|
# @return [Hash, nil]
|
|
111
116
|
#
|
|
112
117
|
def best_match(ref)
|
|
@@ -127,8 +132,8 @@ module Relaton
|
|
|
127
132
|
edition.to_s.split(".").map(&:to_i)
|
|
128
133
|
end
|
|
129
134
|
|
|
130
|
-
def fetch_doc(
|
|
131
|
-
row = best_match
|
|
135
|
+
def fetch_doc(ref)
|
|
136
|
+
row = best_match ref
|
|
132
137
|
return unless row
|
|
133
138
|
|
|
134
139
|
url = "#{ENDPOINT}#{row[:file]}"
|
|
@@ -137,11 +142,11 @@ module Relaton
|
|
|
137
142
|
rescue Mechanize::ResponseCodeError => e
|
|
138
143
|
return if e.response_code == "404"
|
|
139
144
|
|
|
140
|
-
raise Relaton::RequestError, "No document found for #{
|
|
145
|
+
raise Relaton::RequestError, "No document found for #{ref} reference. #{e.message}"
|
|
141
146
|
rescue Mechanize::RedirectLimitReachedError, Timeout::Error,
|
|
142
147
|
Mechanize::UnauthorizedError, Mechanize::UnsupportedSchemeError,
|
|
143
148
|
Mechanize::ResponseReadError, Mechanize::ChunkedTerminationError => e
|
|
144
|
-
raise Relaton::RequestError, "No document found for #{
|
|
149
|
+
raise Relaton::RequestError, "No document found for #{ref} reference. #{e.message}"
|
|
145
150
|
end
|
|
146
151
|
end
|
|
147
152
|
end
|
|
@@ -128,7 +128,7 @@ module Relaton
|
|
|
128
128
|
def to_yaml(bib) = bib.to_yaml
|
|
129
129
|
def to_bibxml(bib) = bib.to_rfcxml
|
|
130
130
|
|
|
131
|
-
# @param hit [
|
|
131
|
+
# @param hit [Moxml::Element]
|
|
132
132
|
def parse_page(hit) # rubocop:disable Metrics/AbcSize, Metrics/MethodLength
|
|
133
133
|
DataParser.new(hit, @errors).parse.each { |item| write_file item }
|
|
134
134
|
end
|
|
@@ -21,7 +21,7 @@ module Relaton
|
|
|
21
21
|
docid = @bib[:docidentifier]
|
|
22
22
|
@doc.xpath('//div[@id="main"]/div[1]/div/main/article/div/div/standard/div/ul/li').map do |hit|
|
|
23
23
|
bib = @bib.dup
|
|
24
|
-
id, ed, bib[:date], vol = edition_id_parts hit.
|
|
24
|
+
id, ed, bib[:date], vol = edition_id_parts hit.at_xpath("./span|./a").text
|
|
25
25
|
bib[:source] = edition_source(hit) + edition_translation_source(ed)
|
|
26
26
|
next if ed.nil? || ed.empty?
|
|
27
27
|
|
|
@@ -56,7 +56,7 @@ module Relaton
|
|
|
56
56
|
end
|
|
57
57
|
|
|
58
58
|
def edition_source(hit)
|
|
59
|
-
es = { "src" => hit.
|
|
59
|
+
es = { "src" => hit.at_xpath("./a"), "pdf" => hit.at_xpath("./span/a") }.map do |type, a|
|
|
60
60
|
Bib::Uri.new(type: type, content: a[:href]) if a
|
|
61
61
|
end.compact
|
|
62
62
|
@errors[:edition_source] &&= es.empty?
|
|
@@ -5,7 +5,7 @@ module Relaton
|
|
|
5
5
|
|
|
6
6
|
ATTRS = %i[docidentifier title date source ext].freeze
|
|
7
7
|
|
|
8
|
-
# @param [
|
|
8
|
+
# @param [Moxml::Element] hit document hit
|
|
9
9
|
# @param [Hash] errors error tracking hash
|
|
10
10
|
def initialize(hit:, errors: {})
|
|
11
11
|
@hit = hit
|
|
@@ -23,7 +23,7 @@ module Relaton
|
|
|
23
23
|
|
|
24
24
|
# @return [Array<Relaton::Ecma::Docidentifier>]
|
|
25
25
|
def fetch_docidentifier
|
|
26
|
-
code = "ECMA MEM/#{@hit.
|
|
26
|
+
code = "ECMA MEM/#{@hit.at_xpath('div[1]//p').text}"
|
|
27
27
|
docid = super(code)
|
|
28
28
|
@errors[:memento_docidentifier] &&= docid.empty?
|
|
29
29
|
docid
|
|
@@ -31,7 +31,7 @@ module Relaton
|
|
|
31
31
|
|
|
32
32
|
# @return [Array<Relaton::Bib::Title>]
|
|
33
33
|
def fetch_title
|
|
34
|
-
year = @hit.
|
|
34
|
+
year = @hit.at_xpath("div[1]//p").text
|
|
35
35
|
content = "\"Memento #{year}\" for year #{year}"
|
|
36
36
|
result = [Bib::Title.new(content: content, language: "en", script: "Latn")]
|
|
37
37
|
@errors[:memento_title] &&= result.empty?
|
|
@@ -40,7 +40,7 @@ module Relaton
|
|
|
40
40
|
|
|
41
41
|
# @return [Array<Relaton::Bib::Date>]
|
|
42
42
|
def fetch_date
|
|
43
|
-
date = @hit.
|
|
43
|
+
date = @hit.at_xpath("div[2]//p").text
|
|
44
44
|
on = Date.strptime(date, "%B %Y").strftime "%Y-%m"
|
|
45
45
|
result = [Bib::Date.new(type: "published", at: on)]
|
|
46
46
|
@errors[:memento_date] &&= result.empty?
|
|
@@ -5,7 +5,7 @@ module Relaton
|
|
|
5
5
|
|
|
6
6
|
ATTRS = %i[docidentifier title date source abstract relation edition ext].freeze
|
|
7
7
|
|
|
8
|
-
# @param [
|
|
8
|
+
# @param [Moxml::Element] hit document hit
|
|
9
9
|
# @param [Mechanize::Page] doc fetched document page
|
|
10
10
|
# @param [Hash] errors error tracking hash
|
|
11
11
|
def initialize(hit:, doc:, errors: {})
|
|
@@ -68,7 +68,7 @@ module Relaton
|
|
|
68
68
|
def fetch_source # rubocop:disable Metrics/AbcSize
|
|
69
69
|
source = []
|
|
70
70
|
source << Bib::Uri.new(type: "src", content: @hit[:href]) if @hit[:href]
|
|
71
|
-
ref = @doc.
|
|
71
|
+
ref = @doc.at_xpath('//div[@class="ecma-item-content-wrapper"]/span/a',
|
|
72
72
|
'//div[@class="ecma-item-content-wrapper"]/a')
|
|
73
73
|
source << Bib::Uri.new(type: "pdf", content: ref[:href]) if ref
|
|
74
74
|
result = source + edition_translation_source(fetch_edition_content)
|
|
@@ -80,7 +80,7 @@ module Relaton
|
|
|
80
80
|
def fetch_relation # rubocop:disable Metrics/AbcSize, Metrics/MethodLength, Metrics/CyclomaticComplexity
|
|
81
81
|
edition_parser = EditionParser.new(doc: @doc, bib: {}, errors: @errors)
|
|
82
82
|
result = @doc.xpath("//ul[@class='ecma-item-archives']/li").filter_map do |rel|
|
|
83
|
-
ref, ed, date, vol = edition_parser.edition_id_parts rel.
|
|
83
|
+
ref, ed, date, vol = edition_parser.edition_id_parts rel.at_xpath("span").text
|
|
84
84
|
next if ed.nil? || ed.empty?
|
|
85
85
|
|
|
86
86
|
docid = Docidentifier.new(type: "ECMA", content: ref, primary: true)
|
|
@@ -109,7 +109,7 @@ module Relaton
|
|
|
109
109
|
private
|
|
110
110
|
|
|
111
111
|
def fetch_edition_content
|
|
112
|
-
@doc.
|
|
112
|
+
@doc.at_xpath('//p[@class="ecma-item-edition"]')&.text&.match(/^\d+(?=(?:st|nd|th|rd))/)&.to_s
|
|
113
113
|
end
|
|
114
114
|
|
|
115
115
|
def edition_translation_source(edition)
|
|
@@ -120,8 +120,8 @@ module Relaton
|
|
|
120
120
|
return [] unless @doc
|
|
121
121
|
|
|
122
122
|
@doc.xpath("//h2[.='Translations']/following-sibling::ul/li").map do |l|
|
|
123
|
-
a = l.
|
|
124
|
-
id = l.
|
|
123
|
+
a = l.at_xpath("span/a")
|
|
124
|
+
id = l.at_xpath("span").text
|
|
125
125
|
%r{\w+[\d-]+,\s(?<lang>\w+)\sversion,\s(?<ed>[\d.]+)(?:st|nd|rd|th)\sedition} =~ id
|
|
126
126
|
case lang
|
|
127
127
|
when "Japanese"
|
|
@@ -6,14 +6,15 @@ module Relaton
|
|
|
6
6
|
module Bibliography
|
|
7
7
|
SOURCE = "https://raw.githubusercontent.com/relaton/relaton-data-etsi/refs/heads/v2/"
|
|
8
8
|
|
|
9
|
-
# @param
|
|
9
|
+
# @param ref [String, ::Pubid::Etsi::Identifier]
|
|
10
10
|
# @return [Relaton::Etsi::ItemData, nil]
|
|
11
|
-
def search(
|
|
11
|
+
def search(ref) # rubocop:disable Metrics/MethodLength, Metrics/AbcSize
|
|
12
12
|
# An unrecognized reference raises Pubid::Errors::ParseError; like
|
|
13
13
|
# ISO we let it propagate — the CLI turns it into a friendly message
|
|
14
14
|
# and API callers rescue it themselves. Valid partial refs parse with
|
|
15
15
|
# the omitted refinements (version/date/part) left blank.
|
|
16
|
-
pubid
|
|
16
|
+
# A parsed pubid comes from Relaton::Db (relaton#205); it is only read.
|
|
17
|
+
pubid = ref.is_a?(String) ? ::Pubid::Etsi.parse(ref) : ref
|
|
17
18
|
|
|
18
19
|
index = Relaton::Index.find_or_create :etsi, url: "#{SOURCE}#{INDEXFILE}.zip", file: "#{INDEXFILE}.yaml",
|
|
19
20
|
pubid_class: ::Pubid::Etsi::Identifier
|
|
@@ -89,19 +90,19 @@ module Relaton
|
|
|
89
90
|
[(id.version&.version).to_s.split(".").map(&:to_i), id.date.to_s]
|
|
90
91
|
end
|
|
91
92
|
|
|
92
|
-
# @param ref [String] the ETSI standard Code to look up
|
|
93
|
+
# @param ref [String, ::Pubid::Etsi::Identifier] the ETSI standard Code to look up
|
|
93
94
|
# @param year [String, nil] year
|
|
94
95
|
# @param opts [Hash] options
|
|
95
96
|
# @return [Relaton::Etsi::ItemData, nil]
|
|
96
97
|
def get(ref, _year = nil, _opts = {})
|
|
97
|
-
Util.info "Fetching from Relaton repository ...", key: ref
|
|
98
|
+
Util.info "Fetching from Relaton repository ...", key: ref.to_s
|
|
98
99
|
result = search(ref)
|
|
99
100
|
unless result
|
|
100
|
-
Util.info "Not found.", key: ref
|
|
101
|
+
Util.info "Not found.", key: ref.to_s
|
|
101
102
|
return
|
|
102
103
|
end
|
|
103
104
|
|
|
104
|
-
Util.info "Found: `#{result.docidentifier[0].content}`", key: ref
|
|
105
|
+
Util.info "Found: `#{result.docidentifier[0].content}`", key: ref.to_s
|
|
105
106
|
result
|
|
106
107
|
end
|
|
107
108
|
|
|
@@ -9,15 +9,15 @@ module Relaton
|
|
|
9
9
|
class Bibliography
|
|
10
10
|
class << self
|
|
11
11
|
# rubocop:disable Metrics/MethodLength
|
|
12
|
-
# @param
|
|
12
|
+
# @param ref [Strin] code of standard for search
|
|
13
13
|
# @return [RelatonGb::HitCollection]
|
|
14
|
-
def search(
|
|
15
|
-
case
|
|
14
|
+
def search(ref)
|
|
15
|
+
case ref
|
|
16
16
|
when /^(GB|GJ|GS)/
|
|
17
17
|
# Scrape national standards.
|
|
18
|
-
Util.info "Fetching from openstd.samr.gov.cn ...", key:
|
|
18
|
+
Util.info "Fetching from openstd.samr.gov.cn ...", key: ref
|
|
19
19
|
require_relative "gb_scraper"
|
|
20
|
-
GbScraper.scrape_page
|
|
20
|
+
GbScraper.scrape_page ref
|
|
21
21
|
# when /^ZB/
|
|
22
22
|
# Scrape proffesional.
|
|
23
23
|
# when /^DB/
|
|
@@ -26,26 +26,34 @@ module Relaton
|
|
|
26
26
|
# Enterprise standard
|
|
27
27
|
when %r{^T/[^\s]{2,6}\s}
|
|
28
28
|
# Scrape social standard.
|
|
29
|
-
Util.info "Fetching from www.ttbz.org.cn ...", key:
|
|
29
|
+
Util.info "Fetching from www.ttbz.org.cn ...", key: ref
|
|
30
30
|
require_relative "t_scraper"
|
|
31
|
-
TScraper.scrape_page
|
|
31
|
+
TScraper.scrape_page ref
|
|
32
32
|
else
|
|
33
33
|
# Scrape sector standard.
|
|
34
34
|
require "relaton/gb/sec_scraper"
|
|
35
|
-
SecScraper.scrape_page
|
|
35
|
+
SecScraper.scrape_page ref
|
|
36
36
|
end
|
|
37
37
|
end
|
|
38
38
|
# rubocop:enable Metrics/MethodLength
|
|
39
39
|
|
|
40
|
-
# @param
|
|
40
|
+
# @param ref [String, Pubid::Gb::Identifier] the GB standard Code to
|
|
41
|
+
# look up (e.g. "GB/T 20223"), or its parse from Relaton::Db
|
|
42
|
+
# (relaton#205)
|
|
41
43
|
# @param year [String] the year the standard was published (optional)
|
|
42
44
|
# @param opts [Hash] options; restricted to :all_parts if all-parts reference is required
|
|
43
45
|
# @return [Relaton::Gb::ItemData, nil]
|
|
44
|
-
# @raise [Pubid::Errors::ParseError] when the
|
|
46
|
+
# @raise [Pubid::Errors::ParseError] when the reference is not a GB
|
|
45
47
|
# identifier
|
|
46
|
-
def get(
|
|
48
|
+
def get(ref, year = nil, opts = {}) # rubocop:disable Metrics/AbcSize,Metrics/CyclomaticComplexity,Metrics/PerceivedComplexity
|
|
47
49
|
require "pubid"
|
|
48
|
-
pubid
|
|
50
|
+
# A parsed pubid comes from Relaton::Db (relaton#205); it is never
|
|
51
|
+
# mutated (`exclude` below copies).
|
|
52
|
+
pubid = ref.is_a?(String) ? ::Pubid::Gb::Identifier.parse(ref) : ref
|
|
53
|
+
if pubid.all_parts?
|
|
54
|
+
opts = opts.merge(all_parts: true)
|
|
55
|
+
pubid = pubid.identifiers.first
|
|
56
|
+
end
|
|
49
57
|
year = (year || pubid.year)&.to_s
|
|
50
58
|
pubid = pubid.exclude(:year)
|
|
51
59
|
pubid.part = "1" if opts[:all_parts]
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# encoding: UTF-8
|
|
2
2
|
# frozen_string_literal: true
|
|
3
3
|
|
|
4
|
-
require "
|
|
4
|
+
require "moxml"
|
|
5
5
|
require_relative "scraper"
|
|
6
6
|
|
|
7
7
|
module Relaton
|
|
@@ -20,10 +20,10 @@ module Relaton
|
|
|
20
20
|
hits = doc.xpath(
|
|
21
21
|
"//table[contains(@class, 'result_list')]/tbody[2]/tr",
|
|
22
22
|
).map do |h|
|
|
23
|
-
ref = h.
|
|
23
|
+
ref = h.at_xpath "./td[2]/a"
|
|
24
24
|
pid = ref[:onclick].match(/[0-9A-F]+/).to_s
|
|
25
|
-
status = h.
|
|
26
|
-
rdate = h.
|
|
25
|
+
status = h.at_xpath("./td[7]").text.strip
|
|
26
|
+
rdate = h.at_xpath("./td[8]").text.strip
|
|
27
27
|
Hit.new pid: pid, docref: ref.text, scraper: self,
|
|
28
28
|
release_date: rdate, status: status
|
|
29
29
|
end
|
|
@@ -52,7 +52,7 @@ module Relaton
|
|
|
52
52
|
# * :type [String]
|
|
53
53
|
# * :name [String]
|
|
54
54
|
# def get_committee(doc, _ref)
|
|
55
|
-
# name = doc.
|
|
55
|
+
# name = doc.at_xpath("//div[contains(., '归口单位') or contains(., '归口部门')]/following-sibling::div")
|
|
56
56
|
# Committee.new(type: "technical", content: name.text.delete("\r\n\t\t"))
|
|
57
57
|
# end
|
|
58
58
|
end
|
data/lib/relaton/gb/scraper.rb
CHANGED
|
@@ -15,7 +15,7 @@ module Relaton
|
|
|
15
15
|
|
|
16
16
|
@prefixes = nil
|
|
17
17
|
|
|
18
|
-
# @param doc [
|
|
18
|
+
# @param doc [Moxml::Document]
|
|
19
19
|
# @param src [String]
|
|
20
20
|
# @param hit [RelatonGb::Hit]
|
|
21
21
|
# @return [Hash]
|
|
@@ -41,7 +41,7 @@ module Relaton
|
|
|
41
41
|
[Docidentifier.new(content: docref, type: "Chinese Standard", primary: true)]
|
|
42
42
|
end
|
|
43
43
|
|
|
44
|
-
# @param doc [
|
|
44
|
+
# @param doc [Moxml::Document]
|
|
45
45
|
# @param docref [Strings]
|
|
46
46
|
# @return [Array<Relaton::Bib::Contributor>]
|
|
47
47
|
def get_contributors(doc, docref)
|
|
@@ -67,22 +67,22 @@ module Relaton
|
|
|
67
67
|
Bib::TypedLocalizedString.new language: lang, content: content
|
|
68
68
|
end
|
|
69
69
|
|
|
70
|
-
# @param doc [
|
|
70
|
+
# @param doc [Moxml::Document]
|
|
71
71
|
# @return [Array<Relaton::Bib::Title>]
|
|
72
72
|
def get_titles(doc)
|
|
73
|
-
tzh = doc.
|
|
73
|
+
tzh = doc.at_xpath("//td[contains(text(), '中文标准名称')]/b").text
|
|
74
74
|
titles = Relaton::Bib::Title.from_string tzh, "zh", "Hans"
|
|
75
|
-
ten = doc.
|
|
75
|
+
ten = doc.at_xpath("//td[contains(text(), '英文标准名称')]").text.match(/[\w\s]+/).to_s
|
|
76
76
|
return titles if ten.empty?
|
|
77
77
|
|
|
78
78
|
titles + Relaton::Bib::Title.from_string(ten, "en", "Latn")
|
|
79
79
|
end
|
|
80
80
|
|
|
81
|
-
# @param doc [
|
|
81
|
+
# @param doc [Moxml::Document]
|
|
82
82
|
# @param status [String, NilClass]
|
|
83
83
|
# @return [Relaton::Bib::Status]
|
|
84
84
|
def get_status(doc, status = nil)
|
|
85
|
-
status ||= doc.
|
|
85
|
+
status ||= doc.at_xpath("//td[contains(., '标准状态')]/span")&.text&.strip
|
|
86
86
|
return unless STAGES[status]
|
|
87
87
|
|
|
88
88
|
stage = Bib::Status::Stage.new content: STAGES[status]
|
|
@@ -91,17 +91,17 @@ module Relaton
|
|
|
91
91
|
|
|
92
92
|
private
|
|
93
93
|
|
|
94
|
-
# @param doc [
|
|
94
|
+
# @param doc [Moxml::Document]
|
|
95
95
|
# @return [Array<String>]
|
|
96
96
|
def get_ccs(doc)
|
|
97
|
-
code = doc.
|
|
97
|
+
code = doc.at_xpath("//div[contains(text(), '中国标准分类号')]/following-sibling::div").text.strip
|
|
98
98
|
[CCS.new(code: code)]
|
|
99
99
|
end
|
|
100
100
|
|
|
101
|
-
# @param doc [
|
|
101
|
+
# @param doc [Moxml::Document]
|
|
102
102
|
# @return [Array<Relaton::Bib::ICS>]
|
|
103
103
|
def get_ics(doc)
|
|
104
|
-
ics = doc.
|
|
104
|
+
ics = doc.at_xpath("//div[contains(text(), '国际标准分类号')]/following-sibling::div"\
|
|
105
105
|
" | //dt[contains(text(), '国际标准分类号')]/following-sibling::dd")
|
|
106
106
|
return [] unless ics
|
|
107
107
|
|
|
@@ -109,10 +109,10 @@ module Relaton
|
|
|
109
109
|
[Bib::ICS.new(code: code)]
|
|
110
110
|
end
|
|
111
111
|
|
|
112
|
-
# @param doc [
|
|
112
|
+
# @param doc [Moxml::Document]
|
|
113
113
|
# @return [String]
|
|
114
114
|
def get_scope(doc)
|
|
115
|
-
issued = doc.
|
|
115
|
+
issued = doc.at_xpath("//div[contains(., '发布单位')]/following-sibling::div")
|
|
116
116
|
case issued&.text
|
|
117
117
|
when /国家标准/ then "national"
|
|
118
118
|
when /^行业标准/ then "sector"
|
|
@@ -150,12 +150,12 @@ module Relaton
|
|
|
150
150
|
(Bib::Uri.new(type: "src", content: src))
|
|
151
151
|
end
|
|
152
152
|
|
|
153
|
-
# @param doc [
|
|
153
|
+
# @param doc [Moxml::Document]
|
|
154
154
|
# @return [Array<Hash>]
|
|
155
155
|
# * :type [String] type of date
|
|
156
156
|
# * :on [String] date
|
|
157
157
|
def get_dates(doc)
|
|
158
|
-
date = doc.
|
|
158
|
+
date = doc.at_xpath("//div[contains(text(), '发布日期')]/following-sibling::div"\
|
|
159
159
|
" | //dt[contains(text(), '发布日期')]/following-sibling::dd")
|
|
160
160
|
[Bib::Date.new(type: "published", at: date.text.delete("\r\n\t\t"))]
|
|
161
161
|
end
|
|
@@ -177,7 +177,7 @@ module Relaton
|
|
|
177
177
|
Doctype.new content: "standard"
|
|
178
178
|
end
|
|
179
179
|
|
|
180
|
-
# @param doc [
|
|
180
|
+
# @param doc [Moxml::Document]
|
|
181
181
|
# @param ref [String]
|
|
182
182
|
# @return [Relaton::Gb::GbType]
|
|
183
183
|
def get_gbtype(doc, ref)
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
|
|
4
4
|
require "net/http"
|
|
5
5
|
require "json"
|
|
6
|
-
require "
|
|
6
|
+
require "moxml"
|
|
7
7
|
require_relative "scraper"
|
|
8
8
|
require_relative "item"
|
|
9
9
|
require_relative "hit_collection"
|
|
@@ -44,7 +44,7 @@ module Relaton
|
|
|
44
44
|
def scrape_doc(hit)
|
|
45
45
|
src = "https://hbba.sacinfo.org.cn/stdDetail/#{hit.pid}"
|
|
46
46
|
page_uri = URI src
|
|
47
|
-
doc =
|
|
47
|
+
doc = Moxml.new.parse_html Net::HTTP.get(page_uri)
|
|
48
48
|
ItemData.new(**scrapped_data(doc, src, hit))
|
|
49
49
|
rescue SocketError, Timeout::Error, Errno::EINVAL, Errno::ECONNRESET, EOFError,
|
|
50
50
|
Net::HTTPBadResponse, Net::HTTPHeaderSyntaxError, Net::ProtocolError,
|
|
@@ -54,14 +54,14 @@ module Relaton
|
|
|
54
54
|
|
|
55
55
|
private
|
|
56
56
|
|
|
57
|
-
# @param doc [
|
|
57
|
+
# @param doc [Moxml::Document]
|
|
58
58
|
# @return [Array<Relaton::Bib::Title>]
|
|
59
59
|
def get_titles(doc)
|
|
60
|
-
tzh = doc.
|
|
60
|
+
tzh = doc.at_xpath("//h4").text.delete("\r\n\t")
|
|
61
61
|
Bib::Title.from_string(tzh, "zh", "Hans")
|
|
62
62
|
end
|
|
63
63
|
|
|
64
|
-
# @param _doc [
|
|
64
|
+
# @param _doc [Moxml::Document]
|
|
65
65
|
# @param ref [String]
|
|
66
66
|
# @return [Hash]
|
|
67
67
|
# * :type [String]
|
|
@@ -72,16 +72,16 @@ module Relaton
|
|
|
72
72
|
# { type: "technical", name: name }
|
|
73
73
|
# end
|
|
74
74
|
|
|
75
|
-
# @param _doc [
|
|
75
|
+
# @param _doc [Moxml::Document]
|
|
76
76
|
# @return [String]
|
|
77
77
|
def get_scope(_doc)
|
|
78
78
|
"sector"
|
|
79
79
|
end
|
|
80
80
|
|
|
81
|
-
# @param doc [
|
|
81
|
+
# @param doc [Moxml::Document]
|
|
82
82
|
# @return [Array<String>]
|
|
83
83
|
def get_ccs(doc)
|
|
84
|
-
array(doc.
|
|
84
|
+
array(doc.at_xpath("//dt[contains(text(), '中国标准分类号')]/following-sibling::dd")).map do |cc|
|
|
85
85
|
text = Cnccs.fetch(cc.text.strip)&.description
|
|
86
86
|
CCS.new code: cc.text, text: text
|
|
87
87
|
end
|
data/lib/relaton/gb/t_scraper.rb
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# encoding: UTF-8
|
|
2
2
|
# frozen_string_literal: true
|
|
3
3
|
|
|
4
|
-
require "
|
|
4
|
+
require "moxml"
|
|
5
5
|
require_relative "scraper"
|
|
6
6
|
require_relative "hit_collection"
|
|
7
7
|
require_relative "hit"
|
|
@@ -23,8 +23,8 @@ module Relaton
|
|
|
23
23
|
xpath = '//table[contains(@class, "standard_list_table")]/tr/td/a'
|
|
24
24
|
t_xpath = "../preceding-sibling::td[4]"
|
|
25
25
|
hits = doc.xpath(xpath).map do |h|
|
|
26
|
-
docref = h.
|
|
27
|
-
status = h.
|
|
26
|
+
docref = h.at_xpath(t_xpath).text.gsub(/â\u0080\u0094/, "-")
|
|
27
|
+
status = h.at_xpath("../preceding-sibling::td[1]").text.delete "\r\n"
|
|
28
28
|
pid = h[:href].sub(%r{/$}, "")
|
|
29
29
|
Hit.new pid: pid, docref: docref, status: status, scraper: self
|
|
30
30
|
end
|
|
@@ -55,7 +55,7 @@ module Relaton
|
|
|
55
55
|
private
|
|
56
56
|
|
|
57
57
|
# rubocop:disable Metrics/MethodLength
|
|
58
|
-
# @param doc [
|
|
58
|
+
# @param doc [Moxml::Document]
|
|
59
59
|
# @param src [String]
|
|
60
60
|
# @param hit [RelatonGb::Hit]
|
|
61
61
|
# @return [Hash]
|
|
@@ -89,7 +89,7 @@ module Relaton
|
|
|
89
89
|
|
|
90
90
|
def get_titles(doc)
|
|
91
91
|
xpz = '//td[contains(.,"中文标题")]/following-sibling::td[1]'
|
|
92
|
-
titles = Bib::Title.from_string doc.
|
|
92
|
+
titles = Bib::Title.from_string doc.at_xpath(xpz)
|
|
93
93
|
.text, "zh", "Hans"
|
|
94
94
|
xpe = '//td[contains(.,"英文标题")]/following-sibling::td[1]'
|
|
95
95
|
ten = doc.xpath(xpe).text
|
|
@@ -12,6 +12,7 @@ module Relaton
|
|
|
12
12
|
def initialize
|
|
13
13
|
@short = :relaton_gost
|
|
14
14
|
@prefix = "GOST"
|
|
15
|
+
@pubid_identifier = :Gost # Db cache key
|
|
15
16
|
# Both Latin "GOST" and Cyrillic "ГОСТ" route here. The trailing
|
|
16
17
|
# \b keeps the prefix from swallowing longer tokens ("GOSTA …").
|
|
17
18
|
@defaultprefix = %r{^(?:GOST|ГОСТ)\b}
|
|
@@ -8,15 +8,15 @@ module Relaton
|
|
|
8
8
|
class << self
|
|
9
9
|
# Search for an IALA publication by its identifier.
|
|
10
10
|
#
|
|
11
|
-
# @param
|
|
11
|
+
# @param ref [String] the IALA reference to look up (e.g. "IALA S1070")
|
|
12
12
|
# @param _year [String, nil] optional edition/year filter
|
|
13
13
|
# @param _opts [Hash] options (unused)
|
|
14
14
|
# @return [Relaton::Iala::Item, nil]
|
|
15
|
-
def search(
|
|
16
|
-
Util.info "Fetching from Relaton repository ...", key:
|
|
17
|
-
row = best_match
|
|
15
|
+
def search(ref, _year = nil, _opts = {})
|
|
16
|
+
Util.info "Fetching from Relaton repository ...", key: ref.to_s
|
|
17
|
+
row = best_match ref
|
|
18
18
|
unless row
|
|
19
|
-
Util.info "Not found.", key:
|
|
19
|
+
Util.info "Not found.", key: ref.to_s
|
|
20
20
|
return
|
|
21
21
|
end
|
|
22
22
|
|
|
@@ -27,7 +27,7 @@ module Relaton
|
|
|
27
27
|
end
|
|
28
28
|
|
|
29
29
|
item = Relaton::Iala::Item.from_yaml resp.body
|
|
30
|
-
Util.info "Found: `#{item.docidentifier.first&.content}`", key:
|
|
30
|
+
Util.info "Found: `#{item.docidentifier.first&.content}`", key: ref.to_s
|
|
31
31
|
item.tap { |i| i.fetched = Date.today.to_s }
|
|
32
32
|
rescue SocketError, Errno::EINVAL, Errno::ECONNRESET, EOFError,
|
|
33
33
|
Net::HTTPBadResponse, Net::HTTPHeaderSyntaxError,
|
|
@@ -59,7 +59,7 @@ module Relaton
|
|
|
59
59
|
# must never match each other. An `Annex` row carries its base
|
|
60
60
|
# identifier, which `===` compares with its own subset match.
|
|
61
61
|
#
|
|
62
|
-
# @param
|
|
62
|
+
# @param ref [String]
|
|
63
63
|
# @return [Hash, nil]
|
|
64
64
|
#
|
|
65
65
|
# The substring-scan fallback this used to take when parsing failed is
|
|
@@ -68,8 +68,8 @@ module Relaton
|
|
|
68
68
|
# ids, so an ambiguous reference silently resolved to whichever row
|
|
69
69
|
# sorted first, instead of telling the caller the reference is not an
|
|
70
70
|
# identifier. ecma, w3c and xsf never had one.
|
|
71
|
-
def best_match(
|
|
72
|
-
pubid = parse_ref
|
|
71
|
+
def best_match(ref)
|
|
72
|
+
pubid = parse_ref ref
|
|
73
73
|
rows = index.search(pubid)
|
|
74
74
|
rows.max_by { |r| [edition_key(r[:id].edition), language_key(r[:id]), r[:file]] }
|
|
75
75
|
end
|
|
@@ -114,7 +114,7 @@ module Relaton
|
|
|
114
114
|
# zero-pads the number to its type's canonical width, so `IALA M1`,
|
|
115
115
|
# `M0001` and `R1016:ed2.0(F)` all parse without normalization here.
|
|
116
116
|
#
|
|
117
|
-
# @param
|
|
117
|
+
# @param ref [String, Pubid::Iala::Identifier]
|
|
118
118
|
# @return [Pubid::Iala::Identifier, nil]
|
|
119
119
|
#
|
|
120
120
|
# An unrecognized reference **raises**; like ISO, ETSI and 3GPP we let it
|
|
@@ -124,8 +124,11 @@ module Relaton
|
|
|
124
124
|
# logs it through the `StandardError` arm at `lib/relaton/db.rb:122`.
|
|
125
125
|
# Rescuing here would collapse "this identifier is malformed" into "no
|
|
126
126
|
# such document", leaving a caller unable to tell them apart.
|
|
127
|
-
def parse_ref(
|
|
128
|
-
::
|
|
127
|
+
def parse_ref(ref)
|
|
128
|
+
# A parsed pubid comes from Relaton::Db (relaton#205); it is used as it is.
|
|
129
|
+
return ref unless ref.is_a?(String)
|
|
130
|
+
|
|
131
|
+
::Pubid::Iala::Identifier.parse ref.to_s.strip
|
|
129
132
|
end
|
|
130
133
|
|
|
131
134
|
# The index is pubid-backed: `pubid_class:` is what makes
|