relaton 3.0.0.pre.alpha.4 → 3.0.0.pre.alpha.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/lib/relaton/adobe/processor.rb +9 -0
- data/lib/relaton/bib/converter/csl.rb +110 -0
- data/lib/relaton/bib/converter/ris.rb +104 -0
- data/lib/relaton/bib/converter/titles.rb +24 -0
- data/lib/relaton/bib/item_data.rb +9 -0
- data/lib/relaton/bib/model/item.rb +6 -6
- data/lib/relaton/bib/sanitizer.rb +71 -30
- data/lib/relaton/bib.rb +10 -0
- data/lib/relaton/bipm/processor.rb +1 -0
- data/lib/relaton/bipm/rawdata_bipm_metrologia/affiliations.rb +6 -6
- data/lib/relaton/bipm/si_brochure_parser.rb +2 -2
- data/lib/relaton/bsi/processor.rb +5 -0
- data/lib/relaton/ccsds/bibliography.rb +1 -0
- data/lib/relaton/ccsds/processor.rb +12 -0
- data/lib/relaton/cen/hit_collection.rb +1 -1
- data/lib/relaton/cen/processor.rb +5 -0
- data/lib/relaton/cen/scraper.rb +9 -9
- data/lib/relaton/cie/data_fetcher.rb +18 -18
- data/lib/relaton/cie/processor.rb +1 -0
- data/lib/relaton/cie.rb +1 -1
- data/lib/relaton/cloud.rb +127 -0
- data/lib/relaton/core/hit_collection.rb +6 -10
- data/lib/relaton/core/processor.rb +68 -0
- data/lib/relaton/db/cache.rb +444 -148
- data/lib/relaton/db/cache_entry.rb +33 -0
- data/lib/relaton/db/registry.rb +52 -0
- data/lib/relaton/db.rb +131 -72
- data/lib/relaton/doi/crossref.rb +23 -6
- data/lib/relaton/doi/processor.rb +1 -0
- data/lib/relaton/easc/processor.rb +1 -0
- data/lib/relaton/ecma/data_fetcher.rb +1 -1
- data/lib/relaton/ecma/data_parser.rb +1 -1
- data/lib/relaton/ecma/edition_parser.rb +2 -2
- data/lib/relaton/ecma/memento_parser.rb +4 -4
- data/lib/relaton/ecma/standard_parser.rb +6 -6
- data/lib/relaton/etsi/processor.rb +1 -0
- data/lib/relaton/gb/gb_scraper.rb +5 -5
- data/lib/relaton/gb/scraper.rb +16 -16
- data/lib/relaton/gb/sec_scraper.rb +8 -8
- data/lib/relaton/gb/t_scraper.rb +5 -5
- data/lib/relaton/gost/processor.rb +1 -0
- data/lib/relaton/iala/processor.rb +1 -0
- data/lib/relaton/iana/data_fetcher.rb +3 -3
- data/lib/relaton/iana/parser.rb +8 -4
- data/lib/relaton/iana/processor.rb +9 -0
- data/lib/relaton/iec/data_parser.rb +24 -9
- data/lib/relaton/iec/processor.rb +6 -0
- data/lib/relaton/iec.rb +1 -1
- data/lib/relaton/ieee/data_fetcher.rb +1 -1
- data/lib/relaton/ieee/processor.rb +8 -0
- data/lib/relaton/ietf/data_fetcher.rb +1 -1
- data/lib/relaton/ietf/processor.rb +1 -0
- data/lib/relaton/ietf/rfc/entry.rb +15 -19
- data/lib/relaton/iho/processor.rb +1 -0
- data/lib/relaton/index/file_io.rb +2 -2
- data/lib/relaton/index/pool.rb +4 -3
- data/lib/relaton/index/shard_source.rb +1 -1
- data/lib/relaton/index/type.rb +2 -5
- data/lib/relaton/isbn/open_library.rb +11 -7
- data/lib/relaton/isbn/processor.rb +10 -0
- data/lib/relaton/iso/data_parser.rb +2 -2
- data/lib/relaton/iso/processor.rb +5 -0
- data/lib/relaton/iso/scraper.rb +20 -20
- data/lib/relaton/itu/bibliography.rb +99 -13
- data/lib/relaton/itu/data_crawler_r.rb +2 -2
- data/lib/relaton/itu/hit_collection.rb +64 -52
- data/lib/relaton/itu/processor.rb +1 -0
- data/lib/relaton/itu/scraper.rb +13 -3
- data/lib/relaton/itu.rb +0 -2
- data/lib/relaton/jis/data_fetcher.rb +4 -4
- data/lib/relaton/jis/processor.rb +1 -0
- data/lib/relaton/jis/scraper.rb +8 -8
- data/lib/relaton/oasis/browser_agent.rb +2 -2
- data/lib/relaton/oasis/data_parser.rb +5 -5
- data/lib/relaton/oasis/data_parser_utils.rb +2 -2
- data/lib/relaton/oasis/data_part_parser.rb +7 -7
- data/lib/relaton/ogc/processor.rb +7 -0
- data/lib/relaton/oiml/processor.rb +1 -0
- data/lib/relaton/omg/scraper.rb +11 -11
- data/lib/relaton/omg.rb +1 -1
- data/lib/relaton/plateau/processor.rb +1 -0
- data/lib/relaton/un/bibliography.rb +21 -10
- data/lib/relaton/un/processor.rb +1 -0
- data/lib/relaton/version.rb +1 -1
- data/lib/relaton/w3c/processor.rb +1 -0
- data/lib/relaton.rb +13 -0
- metadata +48 -16
- data/lib/relaton/itu/pubid.rb +0 -199
data/lib/relaton/gb/scraper.rb
CHANGED
|
@@ -15,7 +15,7 @@ module Relaton
|
|
|
15
15
|
|
|
16
16
|
@prefixes = nil
|
|
17
17
|
|
|
18
|
-
# @param doc [
|
|
18
|
+
# @param doc [Moxml::Document]
|
|
19
19
|
# @param src [String]
|
|
20
20
|
# @param hit [RelatonGb::Hit]
|
|
21
21
|
# @return [Hash]
|
|
@@ -41,7 +41,7 @@ module Relaton
|
|
|
41
41
|
[Docidentifier.new(content: docref, type: "Chinese Standard", primary: true)]
|
|
42
42
|
end
|
|
43
43
|
|
|
44
|
-
# @param doc [
|
|
44
|
+
# @param doc [Moxml::Document]
|
|
45
45
|
# @param docref [Strings]
|
|
46
46
|
# @return [Array<Relaton::Bib::Contributor>]
|
|
47
47
|
def get_contributors(doc, docref)
|
|
@@ -67,22 +67,22 @@ module Relaton
|
|
|
67
67
|
Bib::TypedLocalizedString.new language: lang, content: content
|
|
68
68
|
end
|
|
69
69
|
|
|
70
|
-
# @param doc [
|
|
70
|
+
# @param doc [Moxml::Document]
|
|
71
71
|
# @return [Array<Relaton::Bib::Title>]
|
|
72
72
|
def get_titles(doc)
|
|
73
|
-
tzh = doc.
|
|
73
|
+
tzh = doc.at_xpath("//td[contains(text(), '中文标准名称')]/b").text
|
|
74
74
|
titles = Relaton::Bib::Title.from_string tzh, "zh", "Hans"
|
|
75
|
-
ten = doc.
|
|
75
|
+
ten = doc.at_xpath("//td[contains(text(), '英文标准名称')]").text.match(/[\w\s]+/).to_s
|
|
76
76
|
return titles if ten.empty?
|
|
77
77
|
|
|
78
78
|
titles + Relaton::Bib::Title.from_string(ten, "en", "Latn")
|
|
79
79
|
end
|
|
80
80
|
|
|
81
|
-
# @param doc [
|
|
81
|
+
# @param doc [Moxml::Document]
|
|
82
82
|
# @param status [String, NilClass]
|
|
83
83
|
# @return [Relaton::Bib::Status]
|
|
84
84
|
def get_status(doc, status = nil)
|
|
85
|
-
status ||= doc.
|
|
85
|
+
status ||= doc.at_xpath("//td[contains(., '标准状态')]/span")&.text&.strip
|
|
86
86
|
return unless STAGES[status]
|
|
87
87
|
|
|
88
88
|
stage = Bib::Status::Stage.new content: STAGES[status]
|
|
@@ -91,17 +91,17 @@ module Relaton
|
|
|
91
91
|
|
|
92
92
|
private
|
|
93
93
|
|
|
94
|
-
# @param doc [
|
|
94
|
+
# @param doc [Moxml::Document]
|
|
95
95
|
# @return [Array<String>]
|
|
96
96
|
def get_ccs(doc)
|
|
97
|
-
code = doc.
|
|
97
|
+
code = doc.at_xpath("//div[contains(text(), '中国标准分类号')]/following-sibling::div").text.strip
|
|
98
98
|
[CCS.new(code: code)]
|
|
99
99
|
end
|
|
100
100
|
|
|
101
|
-
# @param doc [
|
|
101
|
+
# @param doc [Moxml::Document]
|
|
102
102
|
# @return [Array<Relaton::Bib::ICS>]
|
|
103
103
|
def get_ics(doc)
|
|
104
|
-
ics = doc.
|
|
104
|
+
ics = doc.at_xpath("//div[contains(text(), '国际标准分类号')]/following-sibling::div"\
|
|
105
105
|
" | //dt[contains(text(), '国际标准分类号')]/following-sibling::dd")
|
|
106
106
|
return [] unless ics
|
|
107
107
|
|
|
@@ -109,10 +109,10 @@ module Relaton
|
|
|
109
109
|
[Bib::ICS.new(code: code)]
|
|
110
110
|
end
|
|
111
111
|
|
|
112
|
-
# @param doc [
|
|
112
|
+
# @param doc [Moxml::Document]
|
|
113
113
|
# @return [String]
|
|
114
114
|
def get_scope(doc)
|
|
115
|
-
issued = doc.
|
|
115
|
+
issued = doc.at_xpath("//div[contains(., '发布单位')]/following-sibling::div")
|
|
116
116
|
case issued&.text
|
|
117
117
|
when /国家标准/ then "national"
|
|
118
118
|
when /^行业标准/ then "sector"
|
|
@@ -150,12 +150,12 @@ module Relaton
|
|
|
150
150
|
(Bib::Uri.new(type: "src", content: src))
|
|
151
151
|
end
|
|
152
152
|
|
|
153
|
-
# @param doc [
|
|
153
|
+
# @param doc [Moxml::Document]
|
|
154
154
|
# @return [Array<Hash>]
|
|
155
155
|
# * :type [String] type of date
|
|
156
156
|
# * :on [String] date
|
|
157
157
|
def get_dates(doc)
|
|
158
|
-
date = doc.
|
|
158
|
+
date = doc.at_xpath("//div[contains(text(), '发布日期')]/following-sibling::div"\
|
|
159
159
|
" | //dt[contains(text(), '发布日期')]/following-sibling::dd")
|
|
160
160
|
[Bib::Date.new(type: "published", at: date.text.delete("\r\n\t\t"))]
|
|
161
161
|
end
|
|
@@ -177,7 +177,7 @@ module Relaton
|
|
|
177
177
|
Doctype.new content: "standard"
|
|
178
178
|
end
|
|
179
179
|
|
|
180
|
-
# @param doc [
|
|
180
|
+
# @param doc [Moxml::Document]
|
|
181
181
|
# @param ref [String]
|
|
182
182
|
# @return [Relaton::Gb::GbType]
|
|
183
183
|
def get_gbtype(doc, ref)
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
|
|
4
4
|
require "net/http"
|
|
5
5
|
require "json"
|
|
6
|
-
require "
|
|
6
|
+
require "moxml"
|
|
7
7
|
require_relative "scraper"
|
|
8
8
|
require_relative "item"
|
|
9
9
|
require_relative "hit_collection"
|
|
@@ -44,7 +44,7 @@ module Relaton
|
|
|
44
44
|
def scrape_doc(hit)
|
|
45
45
|
src = "https://hbba.sacinfo.org.cn/stdDetail/#{hit.pid}"
|
|
46
46
|
page_uri = URI src
|
|
47
|
-
doc =
|
|
47
|
+
doc = Moxml.new.parse_html Net::HTTP.get(page_uri)
|
|
48
48
|
ItemData.new(**scrapped_data(doc, src, hit))
|
|
49
49
|
rescue SocketError, Timeout::Error, Errno::EINVAL, Errno::ECONNRESET, EOFError,
|
|
50
50
|
Net::HTTPBadResponse, Net::HTTPHeaderSyntaxError, Net::ProtocolError,
|
|
@@ -54,14 +54,14 @@ module Relaton
|
|
|
54
54
|
|
|
55
55
|
private
|
|
56
56
|
|
|
57
|
-
# @param doc [
|
|
57
|
+
# @param doc [Moxml::Document]
|
|
58
58
|
# @return [Array<Relaton::Bib::Title>]
|
|
59
59
|
def get_titles(doc)
|
|
60
|
-
tzh = doc.
|
|
60
|
+
tzh = doc.at_xpath("//h4").text.delete("\r\n\t")
|
|
61
61
|
Bib::Title.from_string(tzh, "zh", "Hans")
|
|
62
62
|
end
|
|
63
63
|
|
|
64
|
-
# @param _doc [
|
|
64
|
+
# @param _doc [Moxml::Document]
|
|
65
65
|
# @param ref [String]
|
|
66
66
|
# @return [Hash]
|
|
67
67
|
# * :type [String]
|
|
@@ -72,16 +72,16 @@ module Relaton
|
|
|
72
72
|
# { type: "technical", name: name }
|
|
73
73
|
# end
|
|
74
74
|
|
|
75
|
-
# @param _doc [
|
|
75
|
+
# @param _doc [Moxml::Document]
|
|
76
76
|
# @return [String]
|
|
77
77
|
def get_scope(_doc)
|
|
78
78
|
"sector"
|
|
79
79
|
end
|
|
80
80
|
|
|
81
|
-
# @param doc [
|
|
81
|
+
# @param doc [Moxml::Document]
|
|
82
82
|
# @return [Array<String>]
|
|
83
83
|
def get_ccs(doc)
|
|
84
|
-
array(doc.
|
|
84
|
+
array(doc.at_xpath("//dt[contains(text(), '中国标准分类号')]/following-sibling::dd")).map do |cc|
|
|
85
85
|
text = Cnccs.fetch(cc.text.strip)&.description
|
|
86
86
|
CCS.new code: cc.text, text: text
|
|
87
87
|
end
|
data/lib/relaton/gb/t_scraper.rb
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# encoding: UTF-8
|
|
2
2
|
# frozen_string_literal: true
|
|
3
3
|
|
|
4
|
-
require "
|
|
4
|
+
require "moxml"
|
|
5
5
|
require_relative "scraper"
|
|
6
6
|
require_relative "hit_collection"
|
|
7
7
|
require_relative "hit"
|
|
@@ -23,8 +23,8 @@ module Relaton
|
|
|
23
23
|
xpath = '//table[contains(@class, "standard_list_table")]/tr/td/a'
|
|
24
24
|
t_xpath = "../preceding-sibling::td[4]"
|
|
25
25
|
hits = doc.xpath(xpath).map do |h|
|
|
26
|
-
docref = h.
|
|
27
|
-
status = h.
|
|
26
|
+
docref = h.at_xpath(t_xpath).text.gsub(/â\u0080\u0094/, "-")
|
|
27
|
+
status = h.at_xpath("../preceding-sibling::td[1]").text.delete "\r\n"
|
|
28
28
|
pid = h[:href].sub(%r{/$}, "")
|
|
29
29
|
Hit.new pid: pid, docref: docref, status: status, scraper: self
|
|
30
30
|
end
|
|
@@ -55,7 +55,7 @@ module Relaton
|
|
|
55
55
|
private
|
|
56
56
|
|
|
57
57
|
# rubocop:disable Metrics/MethodLength
|
|
58
|
-
# @param doc [
|
|
58
|
+
# @param doc [Moxml::Document]
|
|
59
59
|
# @param src [String]
|
|
60
60
|
# @param hit [RelatonGb::Hit]
|
|
61
61
|
# @return [Hash]
|
|
@@ -89,7 +89,7 @@ module Relaton
|
|
|
89
89
|
|
|
90
90
|
def get_titles(doc)
|
|
91
91
|
xpz = '//td[contains(.,"中文标题")]/following-sibling::td[1]'
|
|
92
|
-
titles = Bib::Title.from_string doc.
|
|
92
|
+
titles = Bib::Title.from_string doc.at_xpath(xpz)
|
|
93
93
|
.text, "zh", "Hans"
|
|
94
94
|
xpe = '//td[contains(.,"英文标题")]/following-sibling::td[1]'
|
|
95
95
|
ten = doc.xpath(xpe).text
|
|
@@ -12,6 +12,7 @@ module Relaton
|
|
|
12
12
|
def initialize
|
|
13
13
|
@short = :relaton_gost
|
|
14
14
|
@prefix = "GOST"
|
|
15
|
+
@pubid_identifier = :Gost # Db cache key
|
|
15
16
|
# Both Latin "GOST" and Cyrillic "ГОСТ" route here. The trailing
|
|
16
17
|
# \b keeps the prefix from swallowing longer tokens ("GOSTA …").
|
|
17
18
|
@defaultprefix = %r{^(?:GOST|ГОСТ)\b}
|
|
@@ -51,11 +51,11 @@ module Relaton
|
|
|
51
51
|
end
|
|
52
52
|
|
|
53
53
|
def parse(content)
|
|
54
|
-
xml =
|
|
55
|
-
registry = xml.
|
|
54
|
+
xml = Moxml.parse(content)
|
|
55
|
+
registry = xml.at_xpath("/xmlns:registry", NS)
|
|
56
56
|
doc = Parser.parse registry, nil, @errors
|
|
57
57
|
save_doc doc
|
|
58
|
-
registry.xpath("./xmlns:registry").each { |r| save_doc Parser.parse(r, doc, @errors) }
|
|
58
|
+
registry.xpath("./xmlns:registry", NS).each { |r| save_doc Parser.parse(r, doc, @errors) }
|
|
59
59
|
end
|
|
60
60
|
|
|
61
61
|
#
|
data/lib/relaton/iana/parser.rb
CHANGED
|
@@ -1,10 +1,14 @@
|
|
|
1
1
|
module Relaton
|
|
2
2
|
module Iana
|
|
3
|
+
# The IANA registry XML uses a default namespace, and moxml (unlike raw
|
|
4
|
+
# nokogiri) does not bind the literal `xmlns:` prefix to it — every
|
|
5
|
+
# namespaced query passes this binding explicitly.
|
|
6
|
+
NS = { "xmlns" => "http://www.iana.org/assignments" }.freeze
|
|
3
7
|
class Parser
|
|
4
8
|
#
|
|
5
9
|
# Document parser initalization
|
|
6
10
|
#
|
|
7
|
-
# @param [
|
|
11
|
+
# @param [Moxml::Element] xml
|
|
8
12
|
#
|
|
9
13
|
def initialize(xml, rootdoc, errors = {})
|
|
10
14
|
@xml = xml
|
|
@@ -15,7 +19,7 @@ module Relaton
|
|
|
15
19
|
#
|
|
16
20
|
# Initialize document parser and run it
|
|
17
21
|
#
|
|
18
|
-
# @param [
|
|
22
|
+
# @param [Moxml::Element] xml
|
|
19
23
|
#
|
|
20
24
|
# @return [Relaton::Iana::ItemData, nil] bibliographic item
|
|
21
25
|
#
|
|
@@ -50,7 +54,7 @@ module Relaton
|
|
|
50
54
|
# @return [Array<Relaton::Bib::Title>] title
|
|
51
55
|
#
|
|
52
56
|
def parse_title
|
|
53
|
-
content = @xml.
|
|
57
|
+
content = @xml.at_xpath("./xmlns:title", NS)&.text || @xml[:id]
|
|
54
58
|
result = [Bib::Title.new(content: content, language: "en", script: "Latn")]
|
|
55
59
|
@errors[:title] &&= result.empty?
|
|
56
60
|
result
|
|
@@ -121,7 +125,7 @@ module Relaton
|
|
|
121
125
|
# @return [Array<Relaton::Bib::Date>] date
|
|
122
126
|
#
|
|
123
127
|
def parse_date
|
|
124
|
-
d = @xml.xpath("./xmlns:created|./xmlns:published|./xmlns:updated").map do |dt|
|
|
128
|
+
d = @xml.xpath("./xmlns:created|./xmlns:published|./xmlns:updated", NS).map do |dt|
|
|
125
129
|
Bib::Date.new(type: dt.name, at: dt.text)
|
|
126
130
|
end
|
|
127
131
|
result = d.none? && @rootdoc ? @rootdoc.date : d
|
|
@@ -8,6 +8,7 @@ module Relaton
|
|
|
8
8
|
def initialize # rubocop:disable Lint/MissingSuper
|
|
9
9
|
@short = :relaton_iana
|
|
10
10
|
@prefix = "IANA"
|
|
11
|
+
@pubid_identifier = :Iana # Db cache key
|
|
11
12
|
@defaultprefix = %r{^IANA\s}
|
|
12
13
|
@idtype = "IANA"
|
|
13
14
|
@datasets = %w[iana-registries]
|
|
@@ -35,6 +36,14 @@ module Relaton
|
|
|
35
36
|
DataFetcher.fetch(source, **opts)
|
|
36
37
|
end
|
|
37
38
|
|
|
39
|
+
# `get` reads a reference pubid cannot parse as a miss, so it gets no key
|
|
40
|
+
# here either: it is not cached.
|
|
41
|
+
def cache_pubid(ref)
|
|
42
|
+
super
|
|
43
|
+
rescue StandardError
|
|
44
|
+
nil
|
|
45
|
+
end
|
|
46
|
+
|
|
38
47
|
# @param xml [String]
|
|
39
48
|
# @return [Relaton::Iana::ItemData]
|
|
40
49
|
def from_xml(xml)
|
|
@@ -289,38 +289,53 @@ module Relaton
|
|
|
289
289
|
#
|
|
290
290
|
# @return [Array<Relaton::Bib::Relation>] relation
|
|
291
291
|
#
|
|
292
|
-
def relation
|
|
292
|
+
def relation
|
|
293
|
+
result = fetch_relations
|
|
294
|
+
@errors[:relation] &&= result.empty?
|
|
295
|
+
result
|
|
296
|
+
end
|
|
297
|
+
|
|
298
|
+
#
|
|
299
|
+
# Fetch relations from the webstore. Retry on network errors only.
|
|
300
|
+
# A non-success response (the legacy endpoint now redirects to a 404
|
|
301
|
+
# HTML page) or a malformed body gives no relations.
|
|
302
|
+
#
|
|
303
|
+
# @return [Array<Relaton::Bib::Relation>] relations
|
|
304
|
+
#
|
|
305
|
+
def fetch_relations # rubocop:disable Metrics/MethodLength
|
|
293
306
|
try = 0
|
|
294
|
-
|
|
307
|
+
begin
|
|
295
308
|
uri = URI "#{DOMAIN}/webstore/webstore.nsf/AjaxRequestXML?Openagent&url=#{urn_id}"
|
|
296
309
|
resp = Net::HTTP.get_response uri
|
|
297
|
-
doc = Nokogiri::XML resp.body
|
|
298
|
-
create_relations doc
|
|
299
310
|
rescue StandardError => e
|
|
300
311
|
try += 1
|
|
301
312
|
try < 3 ? retry : raise(e)
|
|
302
313
|
end
|
|
303
|
-
|
|
304
|
-
|
|
314
|
+
return [] if resp.is_a?(Net::HTTPResponse) && !resp.is_a?(Net::HTTPSuccess)
|
|
315
|
+
|
|
316
|
+
create_relations Moxml.parse(resp.body)
|
|
317
|
+
rescue Moxml::ParseError => e
|
|
318
|
+
Util.warn "Failed to parse relations for `#{urn_id}`: #{e.message.lines.first&.strip}"
|
|
319
|
+
[]
|
|
305
320
|
end
|
|
306
321
|
|
|
307
322
|
#
|
|
308
323
|
# Create relations.
|
|
309
324
|
#
|
|
310
|
-
# @param [
|
|
325
|
+
# @param [Moxml::Document] doc XML document
|
|
311
326
|
#
|
|
312
327
|
# @return [Array<Relaton::Bib::Relation>] relations
|
|
313
328
|
#
|
|
314
329
|
def create_relations(doc) # rubocop:disable Metrics/MethodLength
|
|
315
330
|
doc.xpath('//ROW[STATUS[.!="PREPARING" and .!="PUBLISHED"]]')
|
|
316
331
|
.map do |r|
|
|
317
|
-
r_type = r.
|
|
332
|
+
r_type = r.at_xpath("STATUS").text.downcase
|
|
318
333
|
type = case r_type
|
|
319
334
|
when "revised", "replaced" then "updates"
|
|
320
335
|
when "withdrawn" then "obsoletes"
|
|
321
336
|
else r_type
|
|
322
337
|
end
|
|
323
|
-
ref = r.
|
|
338
|
+
ref = r.at_xpath("FULL_NAME").text
|
|
324
339
|
docid = Docidentifier.new(content: ref, type: "IEC", primary: true)
|
|
325
340
|
bibitem = ItemData.new(formattedref: Bib::Formattedref.new(content: ref), docidentifier: [docid])
|
|
326
341
|
Relation.new type: type, bibitem: bibitem
|
|
@@ -34,6 +34,12 @@ module Relaton
|
|
|
34
34
|
DataFetcher.fetch(source, **opts)
|
|
35
35
|
end
|
|
36
36
|
|
|
37
|
+
# `IEV` is not a document identifier (`get` answers it with the IEV
|
|
38
|
+
# vocabulary), so it gets no key: it is not cached.
|
|
39
|
+
def cache_pubid(ref)
|
|
40
|
+
ref.strip.casecmp?("IEV") ? nil : super
|
|
41
|
+
end
|
|
42
|
+
|
|
37
43
|
# @param xml [String]
|
|
38
44
|
# @return [Relaton::Iec::ItemData]
|
|
39
45
|
def from_xml(xml)
|
data/lib/relaton/iec.rb
CHANGED
|
@@ -147,7 +147,7 @@ module Relaton
|
|
|
147
147
|
# IdamsParser#parse_relation, so mutates crossrefs under a mutex.
|
|
148
148
|
#
|
|
149
149
|
# @param [String] docnumber of main document
|
|
150
|
-
# @param [
|
|
150
|
+
# @param [Moxml::Element] amsid relation data
|
|
151
151
|
#
|
|
152
152
|
def add_crossref(docnumber, amsid)
|
|
153
153
|
return if RELATION_TYPES[amsid.type] == false
|
|
@@ -36,6 +36,14 @@ module Relaton
|
|
|
36
36
|
DataFetcher.fetch(source, **opts)
|
|
37
37
|
end
|
|
38
38
|
|
|
39
|
+
# `get` reads a reference pubid cannot parse as a miss, so it gets no key
|
|
40
|
+
# here either: it is not cached.
|
|
41
|
+
def cache_pubid(ref)
|
|
42
|
+
super
|
|
43
|
+
rescue StandardError
|
|
44
|
+
nil
|
|
45
|
+
end
|
|
46
|
+
|
|
39
47
|
# @param xml [String]
|
|
40
48
|
# @return [Relaton::Ieee::ItemData]
|
|
41
49
|
def from_xml(xml)
|
|
@@ -6,6 +6,7 @@ module Relaton
|
|
|
6
6
|
def initialize # rubocop:disable Lint/MissingSuper
|
|
7
7
|
@short = :relaton_ietf
|
|
8
8
|
@prefix = "IETF"
|
|
9
|
+
@pubid_identifier = :Ietf # Db cache key
|
|
9
10
|
@defaultprefix = /^((IETF|RFC|BCP|FYI|STD)\s|I-D[.\s])/
|
|
10
11
|
@idtype = "IETF"
|
|
11
12
|
@datasets = %w[ietf-rfcsubseries ietf-internet-drafts ietf-rfc-entries]
|
|
@@ -195,8 +195,7 @@ module Relaton
|
|
|
195
195
|
formattedref: build_formattedref,
|
|
196
196
|
date: build_subseries_date(rfc_index),
|
|
197
197
|
relation: build_relations(rfc_index, wg_names: wg_names),
|
|
198
|
-
|
|
199
|
-
ext: Ext.new(doctype: Doctype.new(content: "rfc"), stream: stream, flavor: "ietf"),
|
|
198
|
+
ext: Ext.new(doctype: Doctype.new(content: "rfc"), stream: canonical_stream, flavor: "ietf"),
|
|
200
199
|
)
|
|
201
200
|
end
|
|
202
201
|
|
|
@@ -259,13 +258,6 @@ module Relaton
|
|
|
259
258
|
ItemData.new(formattedref: Bib::Formattedref.new(content: ref), docidentifier: [docid])
|
|
260
259
|
end
|
|
261
260
|
|
|
262
|
-
def build_series
|
|
263
|
-
return [] unless stream
|
|
264
|
-
|
|
265
|
-
t = Bib::Title.new(content: stream)
|
|
266
|
-
[Bib::Series.new(type: "stream", title: [t])]
|
|
267
|
-
end
|
|
268
|
-
|
|
269
261
|
# --- RFC entry builders ---
|
|
270
262
|
|
|
271
263
|
def build_rfc_docid
|
|
@@ -388,7 +380,7 @@ module Relaton
|
|
|
388
380
|
def build_rfc_series
|
|
389
381
|
series = build_rfc_is_also_series
|
|
390
382
|
series << Bib::Series.new(title: [Bib::Title.new(content: "RFC")], number: shortnum)
|
|
391
|
-
series
|
|
383
|
+
series
|
|
392
384
|
end
|
|
393
385
|
|
|
394
386
|
def build_rfc_is_also_series
|
|
@@ -401,27 +393,31 @@ module Relaton
|
|
|
401
393
|
end
|
|
402
394
|
end
|
|
403
395
|
|
|
404
|
-
def build_rfc_stream_series
|
|
405
|
-
return [] unless stream
|
|
406
|
-
|
|
407
|
-
t = Bib::Title.new(content: stream)
|
|
408
|
-
[Bib::Series.new(type: "stream", title: [t])]
|
|
409
|
-
end
|
|
410
|
-
|
|
411
396
|
STREAM_ORGS = {
|
|
412
397
|
"IETF" => ["IETF", "Internet Engineering Task Force"],
|
|
413
398
|
"IRTF" => ["IRTF", "Internet Research Task Force"],
|
|
414
399
|
"IAB" => ["IAB", "Internet Architecture Board"],
|
|
415
400
|
}.freeze
|
|
416
401
|
|
|
402
|
+
# rfc-index.xml spells the independent stream "INDEPENDENT";
|
|
403
|
+
# Ext.stream's declared values use "Independent". Match case-insensitively
|
|
404
|
+
# against the declared values so both Ext.stream and STREAM_ORGS see the
|
|
405
|
+
# canonical spelling.
|
|
406
|
+
def canonical_stream
|
|
407
|
+
return unless stream
|
|
408
|
+
|
|
409
|
+
values = Ext.attributes[:stream].options[:values]
|
|
410
|
+
values.find { |v| v.casecmp?(stream) } || stream
|
|
411
|
+
end
|
|
412
|
+
|
|
417
413
|
def build_rfc_ext
|
|
418
|
-
Ext.new(doctype: Doctype.new(content: "rfc"), stream:
|
|
414
|
+
Ext.new(doctype: Doctype.new(content: "rfc"), stream: canonical_stream, flavor: "ietf")
|
|
419
415
|
end
|
|
420
416
|
|
|
421
417
|
def build_committee_contributor(wg_names = {})
|
|
422
418
|
return if wg_acronym.nil? || wg_acronym == "NON WORKING GROUP"
|
|
423
419
|
|
|
424
|
-
abbr, name = STREAM_ORGS[
|
|
420
|
+
abbr, name = STREAM_ORGS[canonical_stream]
|
|
425
421
|
org = if abbr
|
|
426
422
|
Ietf::BibXMLParser.build_org(abbr, name)
|
|
427
423
|
else
|
|
@@ -28,11 +28,11 @@ module Relaton
|
|
|
28
28
|
# if nil then the fiename is used to read and write file (used to create indes in GH actions)
|
|
29
29
|
# @param [Pubid::Identifier] pubid class for deserialization
|
|
30
30
|
#
|
|
31
|
-
#
|
|
31
|
+
# The index format check round-trips each id through `pubid_class`;
|
|
32
32
|
# index format is now validated by round-tripping a sample of ids through
|
|
33
33
|
# the pubid class (see #check_serialization), which understands the pubid
|
|
34
34
|
# v2 (lutaml) `_type` serialization that the old key-allowlist could not.
|
|
35
|
-
def initialize(dir, url, filename
|
|
35
|
+
def initialize(dir, url: nil, filename: nil, pubid_class: nil)
|
|
36
36
|
@dir = dir
|
|
37
37
|
@url = url
|
|
38
38
|
@filename = filename
|
data/lib/relaton/index/pool.rb
CHANGED
|
@@ -14,7 +14,6 @@ module Relaton
|
|
|
14
14
|
# @param [String] type <description>
|
|
15
15
|
# @param [String, nil] url external URL to index, used to fetch index for searching files
|
|
16
16
|
# @param [String, nil] file output file name
|
|
17
|
-
# @param [Array<Symbol>, nil] id_keys keys to check if index is correct
|
|
18
17
|
# @param [String, nil] pages_url base URL of the Pages site that serves
|
|
19
18
|
# the machine index (manifest + shards), see Relaton::Index::ShardSource
|
|
20
19
|
#
|
|
@@ -24,9 +23,11 @@ module Relaton
|
|
|
24
23
|
if @pool[type.upcase.to_sym]&.actual?(**args)
|
|
25
24
|
@pool[type.upcase.to_sym]
|
|
26
25
|
else
|
|
26
|
+
if args.key?(:id_keys)
|
|
27
|
+
Util.warn "id_keys is deprecated and ignored by Relaton::Index"
|
|
28
|
+
end
|
|
27
29
|
@pool[type.upcase.to_sym] = Type.new(
|
|
28
|
-
type, args
|
|
29
|
-
pages_url: args[:pages_url]
|
|
30
|
+
type, **args.slice(:url, :file, :pubid_class, :pages_url)
|
|
30
31
|
)
|
|
31
32
|
end
|
|
32
33
|
end
|
|
@@ -47,7 +47,7 @@ module Relaton
|
|
|
47
47
|
@pages_url = pages_url.end_with?("/") ? pages_url : "#{pages_url}/"
|
|
48
48
|
# Only the deserialization helpers are used: this FileIO reads and
|
|
49
49
|
# writes no file.
|
|
50
|
-
@file_io = FileIO.new(dir,
|
|
50
|
+
@file_io = FileIO.new(dir, pubid_class: pubid_class)
|
|
51
51
|
@mutex = Mutex.new
|
|
52
52
|
@state = nil
|
|
53
53
|
end
|
data/lib/relaton/index/type.rb
CHANGED
|
@@ -10,21 +10,18 @@ module Relaton
|
|
|
10
10
|
# @param [String, Symbol] type type of index (ISO, IEC, etc.)
|
|
11
11
|
# @param [String, nil] url external URL to index, used to fetch index for searching files
|
|
12
12
|
# @param [String, nil] file output file name
|
|
13
|
-
# @param [Array<Symbol>] id_keys keys of identifier to be used for sorting index
|
|
14
|
-
# format of index file is checked if id_keys all is provided at least in one of the IDs
|
|
15
13
|
# @param [Pubid::Identifier, nil] pubid class for deserialization
|
|
16
14
|
# @param [String, nil] pages_url base URL of the Pages site that serves
|
|
17
15
|
# the machine index. With it the type reads the index from there, in
|
|
18
16
|
# memory (see ShardSource), and a parsed query fetches one shard only.
|
|
19
17
|
#
|
|
20
|
-
def initialize(type, url
|
|
21
|
-
pages_url: nil)
|
|
18
|
+
def initialize(type, url: nil, file: nil, pubid_class: nil, pages_url: nil)
|
|
22
19
|
@file = file
|
|
23
20
|
@dir = type.to_s.downcase
|
|
24
21
|
@pubid_class = pubid_class
|
|
25
22
|
@pages_url = pages_url
|
|
26
23
|
filename = file || Index.config.filename
|
|
27
|
-
@file_io = FileIO.new @dir, url, filename
|
|
24
|
+
@file_io = FileIO.new @dir, url: url, filename: filename, pubid_class: pubid_class
|
|
28
25
|
@source = new_source
|
|
29
26
|
end
|
|
30
27
|
|
|
@@ -8,23 +8,27 @@ module Relaton
|
|
|
8
8
|
|
|
9
9
|
ENDPOINT = "http://openlibrary.org/api/volumes/brief/isbn/".freeze
|
|
10
10
|
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
11
|
+
# @param ref [String, Pubid::Isbn::Identifier] an ISBN-10 or ISBN-13
|
|
12
|
+
# @return [Relaton::Bib::ItemData, nil]
|
|
13
|
+
def get(ref, _date = nil, _opts = {}) # rubocop:disable Metrics/AbcSize,Metrics/MethodLength
|
|
14
|
+
Util.info "Fetching from OpenLibrary ...", key: ref.to_s
|
|
15
|
+
|
|
16
|
+
# A parsed pubid gives its digits (`raw`); `Isbn#parse` validates them
|
|
17
|
+
# and converts an ISBN-10 to ISBN-13, as for a String.
|
|
18
|
+
isbn = Isbn.new(ref.is_a?(String) ? ref : ref.raw).parse
|
|
15
19
|
unless isbn
|
|
16
|
-
Util.info "Incorrect ISBN.", key: ref
|
|
20
|
+
Util.info "Incorrect ISBN.", key: ref.to_s
|
|
17
21
|
return
|
|
18
22
|
end
|
|
19
23
|
|
|
20
24
|
resp = request_api isbn
|
|
21
25
|
unless resp
|
|
22
|
-
Util.info "Not found.", key: ref
|
|
26
|
+
Util.info "Not found.", key: ref.to_s
|
|
23
27
|
return
|
|
24
28
|
end
|
|
25
29
|
|
|
26
30
|
bib = Parser.parse resp
|
|
27
|
-
Util.info "Found: `#{bib.docidentifier.first.content}`", key: ref
|
|
31
|
+
Util.info "Found: `#{bib.docidentifier.first.content}`", key: ref.to_s
|
|
28
32
|
bib
|
|
29
33
|
end
|
|
30
34
|
|
|
@@ -8,6 +8,7 @@ module Relaton
|
|
|
8
8
|
def initialize # rubocop:disable Lint/MissingSuper
|
|
9
9
|
@short = :relaton_isbn
|
|
10
10
|
@prefix = "ISBN"
|
|
11
|
+
@pubid_identifier = :Isbn # Db cache key
|
|
11
12
|
@defaultprefix = /^ISBN\s/
|
|
12
13
|
@idtype = "ISBN"
|
|
13
14
|
@datasets = %w[]
|
|
@@ -22,6 +23,15 @@ module Relaton
|
|
|
22
23
|
::Relaton::Isbn::OpenLibrary.get(code, date, opts)
|
|
23
24
|
end
|
|
24
25
|
|
|
26
|
+
# `OpenLibrary.get` reads an incorrect ISBN as a miss, so a reference
|
|
27
|
+
# pubid cannot parse (it checks the check digit too) gets no key here
|
|
28
|
+
# either: it is not cached.
|
|
29
|
+
def cache_pubid(ref)
|
|
30
|
+
super
|
|
31
|
+
rescue StandardError
|
|
32
|
+
nil
|
|
33
|
+
end
|
|
34
|
+
|
|
25
35
|
#
|
|
26
36
|
# @param xml [String]
|
|
27
37
|
# @return [Relaton::Bib::Bibitem]
|