relaton 3.0.0.pre.alpha.4 → 3.0.0.pre.alpha.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/lib/relaton/adobe/processor.rb +9 -0
- data/lib/relaton/bib/converter/csl.rb +110 -0
- data/lib/relaton/bib/converter/ris.rb +104 -0
- data/lib/relaton/bib/converter/titles.rb +24 -0
- data/lib/relaton/bib/item_data.rb +9 -0
- data/lib/relaton/bib/model/item.rb +6 -6
- data/lib/relaton/bib/sanitizer.rb +71 -30
- data/lib/relaton/bib.rb +10 -0
- data/lib/relaton/bipm/processor.rb +1 -0
- data/lib/relaton/bipm/rawdata_bipm_metrologia/affiliations.rb +6 -6
- data/lib/relaton/bipm/si_brochure_parser.rb +2 -2
- data/lib/relaton/bsi/processor.rb +5 -0
- data/lib/relaton/ccsds/bibliography.rb +1 -0
- data/lib/relaton/ccsds/processor.rb +12 -0
- data/lib/relaton/cen/hit_collection.rb +1 -1
- data/lib/relaton/cen/processor.rb +5 -0
- data/lib/relaton/cen/scraper.rb +9 -9
- data/lib/relaton/cie/data_fetcher.rb +18 -18
- data/lib/relaton/cie/processor.rb +1 -0
- data/lib/relaton/cie.rb +1 -1
- data/lib/relaton/cloud.rb +127 -0
- data/lib/relaton/core/hit_collection.rb +6 -10
- data/lib/relaton/core/processor.rb +68 -0
- data/lib/relaton/db/cache.rb +444 -148
- data/lib/relaton/db/cache_entry.rb +33 -0
- data/lib/relaton/db/registry.rb +52 -0
- data/lib/relaton/db.rb +131 -72
- data/lib/relaton/doi/crossref.rb +23 -6
- data/lib/relaton/doi/processor.rb +1 -0
- data/lib/relaton/easc/processor.rb +1 -0
- data/lib/relaton/ecma/data_fetcher.rb +1 -1
- data/lib/relaton/ecma/data_parser.rb +1 -1
- data/lib/relaton/ecma/edition_parser.rb +2 -2
- data/lib/relaton/ecma/memento_parser.rb +4 -4
- data/lib/relaton/ecma/standard_parser.rb +6 -6
- data/lib/relaton/etsi/processor.rb +1 -0
- data/lib/relaton/gb/gb_scraper.rb +5 -5
- data/lib/relaton/gb/scraper.rb +16 -16
- data/lib/relaton/gb/sec_scraper.rb +8 -8
- data/lib/relaton/gb/t_scraper.rb +5 -5
- data/lib/relaton/gost/processor.rb +1 -0
- data/lib/relaton/iala/processor.rb +1 -0
- data/lib/relaton/iana/data_fetcher.rb +3 -3
- data/lib/relaton/iana/parser.rb +8 -4
- data/lib/relaton/iana/processor.rb +9 -0
- data/lib/relaton/iec/data_parser.rb +24 -9
- data/lib/relaton/iec/processor.rb +6 -0
- data/lib/relaton/iec.rb +1 -1
- data/lib/relaton/ieee/data_fetcher.rb +1 -1
- data/lib/relaton/ieee/processor.rb +8 -0
- data/lib/relaton/ietf/data_fetcher.rb +1 -1
- data/lib/relaton/ietf/processor.rb +1 -0
- data/lib/relaton/ietf/rfc/entry.rb +15 -19
- data/lib/relaton/iho/processor.rb +1 -0
- data/lib/relaton/index/file_io.rb +2 -2
- data/lib/relaton/index/pool.rb +4 -3
- data/lib/relaton/index/shard_source.rb +1 -1
- data/lib/relaton/index/type.rb +2 -5
- data/lib/relaton/isbn/open_library.rb +11 -7
- data/lib/relaton/isbn/processor.rb +10 -0
- data/lib/relaton/iso/data_parser.rb +2 -2
- data/lib/relaton/iso/processor.rb +5 -0
- data/lib/relaton/iso/scraper.rb +20 -20
- data/lib/relaton/itu/bibliography.rb +99 -13
- data/lib/relaton/itu/data_crawler_r.rb +2 -2
- data/lib/relaton/itu/hit_collection.rb +64 -52
- data/lib/relaton/itu/processor.rb +1 -0
- data/lib/relaton/itu/scraper.rb +13 -3
- data/lib/relaton/itu.rb +0 -2
- data/lib/relaton/jis/data_fetcher.rb +4 -4
- data/lib/relaton/jis/processor.rb +1 -0
- data/lib/relaton/jis/scraper.rb +8 -8
- data/lib/relaton/oasis/browser_agent.rb +2 -2
- data/lib/relaton/oasis/data_parser.rb +5 -5
- data/lib/relaton/oasis/data_parser_utils.rb +2 -2
- data/lib/relaton/oasis/data_part_parser.rb +7 -7
- data/lib/relaton/ogc/processor.rb +7 -0
- data/lib/relaton/oiml/processor.rb +1 -0
- data/lib/relaton/omg/scraper.rb +11 -11
- data/lib/relaton/omg.rb +1 -1
- data/lib/relaton/plateau/processor.rb +1 -0
- data/lib/relaton/un/bibliography.rb +21 -10
- data/lib/relaton/un/processor.rb +1 -0
- data/lib/relaton/version.rb +1 -1
- data/lib/relaton/w3c/processor.rb +1 -0
- data/lib/relaton.rb +13 -0
- metadata +48 -16
- data/lib/relaton/itu/pubid.rb +0 -199
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: b34a85f74b2b8b07d61b51a20d976f1204240dbc353f9e7d73eeeb7262912337
|
|
4
|
+
data.tar.gz: 6e51891af83e32391abb096a788685f42d7cc648fc6a3e3483683868daa8e4b3
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 6530a4f026cf099bcd3a1c1b359d307089b57bbe14f94a77362adabafa329abbcb4f64349feaa9061e9a6eda9ff413645af6b14812bdcb3ac9bcb4e22e6ad0a1
|
|
7
|
+
data.tar.gz: 7c555c9ddb34285255f22766b7b9c56be98b657018bfb4503bb66bc316eea7a1c2fd796f21e1910ad41615a0079d8db32f76d2b46bc051a2a4e7c6ed4c0316cc
|
|
@@ -12,6 +12,7 @@ module Relaton
|
|
|
12
12
|
def initialize
|
|
13
13
|
@short = :relaton_adobe
|
|
14
14
|
@prefix = "Adobe"
|
|
15
|
+
@pubid_identifier = :Adobe # Db cache key
|
|
15
16
|
@defaultprefix = %r{^(?:Adobe|ATN)}
|
|
16
17
|
@idtype = "Adobe"
|
|
17
18
|
end
|
|
@@ -21,6 +22,14 @@ module Relaton
|
|
|
21
22
|
Bibliography.get(code, date, opts)
|
|
22
23
|
end
|
|
23
24
|
|
|
25
|
+
# `Bibliography.get` reads a reference pubid cannot parse as a miss
|
|
26
|
+
# (`pubid_for`), so it gets no key here either: it is not cached.
|
|
27
|
+
def cache_pubid(ref)
|
|
28
|
+
super
|
|
29
|
+
rescue StandardError
|
|
30
|
+
nil
|
|
31
|
+
end
|
|
32
|
+
|
|
24
33
|
def from_xml(xml)
|
|
25
34
|
require_relative "../adobe"
|
|
26
35
|
Item.from_xml xml
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "json"
|
|
4
|
+
|
|
5
|
+
module Relaton
|
|
6
|
+
module Bib
|
|
7
|
+
module Converter
|
|
8
|
+
# CSL-JSON serialization — the Citation Style Language interchange
|
|
9
|
+
# format consumed by Zotero, Mendeley, and every citeproc.
|
|
10
|
+
module Csl
|
|
11
|
+
ENTRY_TYPES = {
|
|
12
|
+
"standard" => "standard", "book" => "book",
|
|
13
|
+
"article" => "article-journal", "inbook" => "chapter",
|
|
14
|
+
"inproceedings" => "paper-conference", "report" => "report",
|
|
15
|
+
"thesis" => "thesis", "website" => "webpage",
|
|
16
|
+
"webresource" => "webpage",
|
|
17
|
+
}.freeze
|
|
18
|
+
|
|
19
|
+
def self.from_item(item)
|
|
20
|
+
Renderer.new(item).to_s
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
class Renderer
|
|
24
|
+
def initialize(item)
|
|
25
|
+
@item = item
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
def to_s
|
|
29
|
+
out = {}
|
|
30
|
+
docid = primary_docid
|
|
31
|
+
out[:id] = docid.empty? ? "relaton" : docid
|
|
32
|
+
out[:type] = ENTRY_TYPES[@item.type.to_s] || "standard"
|
|
33
|
+
out[:title] = primary_title unless primary_title.empty?
|
|
34
|
+
authors = csl_authors
|
|
35
|
+
out[:author] = authors unless authors.empty?
|
|
36
|
+
out[:issued] = { "date-parts" => [[published_year]] } unless published_year.empty?
|
|
37
|
+
publisher = publisher_name
|
|
38
|
+
out[:publisher] = publisher unless publisher.empty?
|
|
39
|
+
out[:number] = docid unless docid.empty?
|
|
40
|
+
out[:edition] = @item.edition.to_s unless @item.edition.to_s.empty?
|
|
41
|
+
lang = Array(@item.language).first
|
|
42
|
+
out[:language] = lang.to_s if lang
|
|
43
|
+
keywords = Array(@item.keyword).map { |k| k.content.to_s }.reject(&:empty?)
|
|
44
|
+
out[:keyword] = keywords unless keywords.empty?
|
|
45
|
+
out[:URL] = source_uri if source_uri
|
|
46
|
+
[out].to_json + "\n"
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
private
|
|
50
|
+
|
|
51
|
+
def csl_authors
|
|
52
|
+
Array(@item.contributor).filter_map do |c|
|
|
53
|
+
roles = Array(c.role).map(&:type).compact
|
|
54
|
+
next unless (roles & %w[author performer editor]).any?
|
|
55
|
+
|
|
56
|
+
if c.person
|
|
57
|
+
name = c.person.name
|
|
58
|
+
fore = name ? Array(name.forename).map { |f| f.respond_to?(:content) ? f.content.to_s : f.to_s }.join(" ") : ""
|
|
59
|
+
sur = name && name.surname.respond_to?(:content) ? name.surname.content.to_s : name&.surname.to_s
|
|
60
|
+
complete = name && name.completename.respond_to?(:content) ? name.completename.content.to_s : name&.completename.to_s
|
|
61
|
+
if sur.to_s.empty?
|
|
62
|
+
complete.to_s.empty? ? nil : { "literal" => complete }
|
|
63
|
+
else
|
|
64
|
+
entry = { "family" => sur }
|
|
65
|
+
entry["given"] = fore unless fore.empty?
|
|
66
|
+
entry
|
|
67
|
+
end
|
|
68
|
+
elsif c.organization
|
|
69
|
+
literal = org_name(c.organization)
|
|
70
|
+
literal.empty? ? nil : { "literal" => literal }
|
|
71
|
+
end
|
|
72
|
+
end
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
def org_name(org)
|
|
76
|
+
name = Array(org.name).map { |n| n.respond_to?(:content) ? n.content.to_s : n.to_s }.find(&:itself)
|
|
77
|
+
name || org.abbreviation.to_s
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
def publisher_name
|
|
81
|
+
pub = Array(@item.contributor).find do |c|
|
|
82
|
+
Array(c.role).map(&:type).include?("publisher") && c.organization
|
|
83
|
+
end
|
|
84
|
+
pub ? org_name(pub.organization) : ""
|
|
85
|
+
end
|
|
86
|
+
|
|
87
|
+
def primary_title
|
|
88
|
+
Titles.of(@item)
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
def primary_docid
|
|
92
|
+
ids = Array(@item.docidentifier)
|
|
93
|
+
(ids.find(&:primary) || ids.first)&.content.to_s
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
def published_year
|
|
97
|
+
date = Array(@item.date).find { |d| %w[published issued].include?(d.type.to_s) } || @item.date.first
|
|
98
|
+
value = date && (date.at || date.from || date.to)
|
|
99
|
+
value.to_s[/\d{4}/]
|
|
100
|
+
end
|
|
101
|
+
|
|
102
|
+
def source_uri
|
|
103
|
+
Array(@item.source).map { |s| s.respond_to?(:content) ? s.content.to_s : s.to_s }
|
|
104
|
+
.find { |u| u.start_with?("http") }
|
|
105
|
+
end
|
|
106
|
+
end
|
|
107
|
+
end
|
|
108
|
+
end
|
|
109
|
+
end
|
|
110
|
+
end
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
module Relaton
|
|
5
|
+
module Bib
|
|
6
|
+
module Converter
|
|
7
|
+
# RIS serialization — the interchange format for EndNote, Zotero,
|
|
8
|
+
# and Reference Manager. Standards map to TY - STD.
|
|
9
|
+
module Ris
|
|
10
|
+
ENTRY_TYPES = {
|
|
11
|
+
"standard" => "STD", "book" => "BOOK", "article" => "JOUR",
|
|
12
|
+
"inbook" => "CHAP", "inproceedings" => "CONF", "report" => "RPRT",
|
|
13
|
+
"thesis" => "THES", "website" => "ELEC", "webresource" => "ELEC",
|
|
14
|
+
}.freeze
|
|
15
|
+
|
|
16
|
+
def self.from_item(item)
|
|
17
|
+
Renderer.new(item).to_s
|
|
18
|
+
end
|
|
19
|
+
|
|
20
|
+
class Renderer
|
|
21
|
+
def initialize(item)
|
|
22
|
+
@item = item
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
def to_s
|
|
26
|
+
lines = ["TY - #{ENTRY_TYPES[@item.type.to_s] || 'STD'}"]
|
|
27
|
+
add_authors(lines)
|
|
28
|
+
add_field(lines, "TI", primary_title)
|
|
29
|
+
add_field(lines, "ID", primary_docid)
|
|
30
|
+
add_field(lines, "PY", published_year)
|
|
31
|
+
add_field(lines, "ET", @item.edition&.to_s)
|
|
32
|
+
lang = Array(@item.language).first
|
|
33
|
+
add_field(lines, "LA", lang.to_s)
|
|
34
|
+
Array(@item.keyword).each { |k| add_field(lines, "KW", k.content.to_s) }
|
|
35
|
+
add_field(lines, "UR", source_uri)
|
|
36
|
+
lines << "ER - "
|
|
37
|
+
lines.join("\r\n") + "\r\n"
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
private
|
|
41
|
+
|
|
42
|
+
def add_field(lines, tag, value)
|
|
43
|
+
lines << "#{tag} - #{value}" if value && !value.empty?
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
def add_authors(lines)
|
|
47
|
+
Array(@item.contributor).each do |c|
|
|
48
|
+
roles = Array(c.role).map(&:type).compact
|
|
49
|
+
authorish = (roles & %w[author performer editor]).any?
|
|
50
|
+
publisher = roles.include?("publisher")
|
|
51
|
+
if authorish && c.person
|
|
52
|
+
out = person_display_name(c.person.name)
|
|
53
|
+
lines << "AU - #{out}" unless out.empty?
|
|
54
|
+
elsif authorish && c.organization
|
|
55
|
+
out = org_name(c.organization)
|
|
56
|
+
lines << "AU - #{out}" unless out.empty?
|
|
57
|
+
elsif publisher && c.organization
|
|
58
|
+
out = org_name(c.organization)
|
|
59
|
+
lines << "PB - #{out}" unless out.empty? || lines.any? { |l| l.start_with?("PB - ") }
|
|
60
|
+
end
|
|
61
|
+
end
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
def person_display_name(name)
|
|
65
|
+
return "" unless name
|
|
66
|
+
|
|
67
|
+
fore = Array(name.forename).map { |f| f.respond_to?(:content) ? f.content.to_s : f.to_s }.join(" ")
|
|
68
|
+
sur = name.surname.respond_to?(:content) ? name.surname.content.to_s : name.surname.to_s
|
|
69
|
+
complete = name.completename.respond_to?(:content) ? name.completename.content.to_s : name.completename.to_s
|
|
70
|
+
return "#{sur}, #{fore}" unless fore.empty? || sur.empty?
|
|
71
|
+
return sur unless sur.empty?
|
|
72
|
+
|
|
73
|
+
complete
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
def org_name(org)
|
|
77
|
+
name = Array(org.name).map { |n| n.respond_to?(:content) ? n.content.to_s : n.to_s }.find(&:itself)
|
|
78
|
+
name || org.abbreviation.to_s
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
def primary_title
|
|
82
|
+
Titles.of(@item)
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
def primary_docid
|
|
86
|
+
ids = Array(@item.docidentifier)
|
|
87
|
+
(ids.find(&:primary) || ids.first)&.content.to_s
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
def published_year
|
|
91
|
+
date = Array(@item.date).find { |d| %w[published issued].include?(d.type.to_s) } || @item.date.first
|
|
92
|
+
value = date && (date.at || date.from || date.to)
|
|
93
|
+
value.to_s[/\d{4}/]
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
def source_uri
|
|
97
|
+
Array(@item.source).map { |s| s.respond_to?(:content) ? s.content.to_s : s.to_s }
|
|
98
|
+
.find { |u| u.start_with?("http") }
|
|
99
|
+
end
|
|
100
|
+
end
|
|
101
|
+
end
|
|
102
|
+
end
|
|
103
|
+
end
|
|
104
|
+
end
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Relaton
|
|
4
|
+
module Bib
|
|
5
|
+
module Converter
|
|
6
|
+
# Single citation language; decomposed titles compose, composite or
|
|
7
|
+
# plain titles stand alone. Shared by every export converter.
|
|
8
|
+
module Titles
|
|
9
|
+
def self.of(item)
|
|
10
|
+
titles = Array(item.title).select { |t| t.content.to_s != "" }
|
|
11
|
+
content = ->(t) { t&.content.to_s }
|
|
12
|
+
intro = titles.find { |t| t.type == "title-intro" }
|
|
13
|
+
main = titles.find { |t| t.type == "title-main" }
|
|
14
|
+
part = titles.find { |t| t.type == "title-part" }
|
|
15
|
+
composite = titles.find { |t| t.type == "main" }
|
|
16
|
+
base = [content.call(intro), content.call(main)].reject(&:empty?)
|
|
17
|
+
base = [content.call(composite)].reject(&:empty?) if base.empty?
|
|
18
|
+
base = [content.call(titles.first)].reject(&:empty?) if base.empty?
|
|
19
|
+
[*base, content.call(part)].reject(&:empty?).join(" — ")
|
|
20
|
+
end
|
|
21
|
+
end
|
|
22
|
+
end
|
|
23
|
+
end
|
|
24
|
+
end
|
|
@@ -152,6 +152,15 @@ module Relaton
|
|
|
152
152
|
Converter::Bibtex.from_item(self).to_s
|
|
153
153
|
end
|
|
154
154
|
|
|
155
|
+
def to_ris
|
|
156
|
+
Converter::Ris.from_item(self)
|
|
157
|
+
end
|
|
158
|
+
|
|
159
|
+
def to_csl_json
|
|
160
|
+
Converter::Csl.from_item(self)
|
|
161
|
+
end
|
|
162
|
+
|
|
163
|
+
|
|
155
164
|
def to_asciibib
|
|
156
165
|
Converter::Asciibib.from_item(self)
|
|
157
166
|
end
|
|
@@ -46,10 +46,6 @@ require_relative "ext"
|
|
|
46
46
|
require_relative "item_shared"
|
|
47
47
|
require_relative "type/plain_date"
|
|
48
48
|
|
|
49
|
-
Lutaml::Model::Config.configure do |config|
|
|
50
|
-
config.xml_adapter_type = :nokogiri
|
|
51
|
-
end
|
|
52
|
-
|
|
53
49
|
module Relaton
|
|
54
50
|
module Bib
|
|
55
51
|
class Relation < Lutaml::Model::Serializable
|
|
@@ -69,8 +65,12 @@ module Relaton
|
|
|
69
65
|
|
|
70
66
|
# lutaml-model has no built-in dispatch on root element name
|
|
71
67
|
# (polymorphic_map only works on attribute discriminators), so we
|
|
72
|
-
# peek at the root tag with
|
|
73
|
-
root_name =
|
|
68
|
+
# peek at the root tag with moxml and forward to the right class.
|
|
69
|
+
root_name = begin
|
|
70
|
+
Moxml.parse(xml.to_s).root&.name
|
|
71
|
+
rescue Moxml::ParseError
|
|
72
|
+
nil
|
|
73
|
+
end
|
|
74
74
|
klass = root_name == "bibdata" ? namespace::Bibdata : namespace::Bibitem
|
|
75
75
|
klass.from_xml(xml, options)
|
|
76
76
|
end
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
require "
|
|
1
|
+
require "moxml"
|
|
2
2
|
|
|
3
3
|
module Relaton
|
|
4
4
|
module Bib
|
|
@@ -63,11 +63,6 @@ module Relaton
|
|
|
63
63
|
# Reserved prefixes. XML declares both, so the content must not.
|
|
64
64
|
NS_RESERVED = %w[xml xmlns].freeze
|
|
65
65
|
|
|
66
|
-
# Serialise without the FORMAT option, so the sanitiser keeps the
|
|
67
|
-
# shape of element-only content instead of adding newlines and
|
|
68
|
-
# indent.
|
|
69
|
-
SAVE_OPTS = Nokogiri::XML::Node::SaveOptions::AS_XML
|
|
70
|
-
|
|
71
66
|
def self.sanitize(content)
|
|
72
67
|
return content unless sanitizable?(content)
|
|
73
68
|
|
|
@@ -75,9 +70,7 @@ module Relaton
|
|
|
75
70
|
return content if node.nil?
|
|
76
71
|
|
|
77
72
|
sanitize_children(node)
|
|
78
|
-
node.children.map
|
|
79
|
-
c.to_xml(encoding: "UTF-8", save_with: SAVE_OPTS)
|
|
80
|
-
end.join
|
|
73
|
+
node.children.map { |c| c.to_xml(encoding: "UTF-8", indent: 0, expand_empty: false) }.join
|
|
81
74
|
end
|
|
82
75
|
|
|
83
76
|
#
|
|
@@ -85,13 +78,23 @@ module Relaton
|
|
|
85
78
|
#
|
|
86
79
|
# @param [String] content The raw marked-up content.
|
|
87
80
|
#
|
|
88
|
-
# @return [
|
|
89
|
-
# content does not parse.
|
|
81
|
+
# @return [Moxml::Element, nil] The wrapper element, or nil when
|
|
82
|
+
# the content does not parse.
|
|
90
83
|
#
|
|
84
|
+
# moxml's parse_fragment returns detached nodes without a parent
|
|
85
|
+
# chain, so the document that owns their C memory can be collected
|
|
86
|
+
# while the node wrappers live on. Parse under a synthetic root
|
|
87
|
+
# instead — the same wrapper machinery parse_with_prefixes uses —
|
|
88
|
+
# which keeps the document reachable from the root.
|
|
91
89
|
def self.parse(content)
|
|
92
|
-
|
|
93
|
-
|
|
90
|
+
# leptris accepts undeclared prefixes instead of failing the
|
|
91
|
+
# parse, so detect them by scan and take the placeholder path
|
|
92
|
+
# directly.
|
|
93
|
+
return parse_with_prefixes(content) if placeholder_declarations(content)
|
|
94
94
|
|
|
95
|
+
name = wrapper_name(content)
|
|
96
|
+
Moxml.parse("<#{name}>#{content}</#{name}>").root
|
|
97
|
+
rescue Moxml::ParseError
|
|
95
98
|
parse_with_prefixes(content)
|
|
96
99
|
end
|
|
97
100
|
private_class_method :parse
|
|
@@ -112,16 +115,16 @@ module Relaton
|
|
|
112
115
|
#
|
|
113
116
|
# @param [String] content The raw marked-up content.
|
|
114
117
|
#
|
|
115
|
-
# @return [
|
|
118
|
+
# @return [Moxml::Element, nil] The wrapper element, or nil
|
|
116
119
|
# when the content uses no prefix or does not parse.
|
|
117
120
|
#
|
|
118
121
|
def self.parse_with_prefixes(content)
|
|
119
122
|
decl = placeholder_declarations(content) or return
|
|
120
123
|
name = wrapper_name(content)
|
|
121
|
-
doc =
|
|
122
|
-
return unless doc.errors.empty?
|
|
123
|
-
|
|
124
|
+
doc = Moxml.parse "<#{name} #{decl}>#{content}</#{name}>"
|
|
124
125
|
drop_placeholder_namespaces doc.root
|
|
126
|
+
rescue Moxml::ParseError
|
|
127
|
+
nil
|
|
125
128
|
end
|
|
126
129
|
private_class_method :parse_with_prefixes
|
|
127
130
|
|
|
@@ -183,20 +186,41 @@ module Relaton
|
|
|
183
186
|
# reach the output, so the declarations never leak. Do not
|
|
184
187
|
# serialise the root itself.
|
|
185
188
|
#
|
|
186
|
-
# @param [
|
|
189
|
+
# @param [Moxml::Element] root The wrapper element.
|
|
187
190
|
#
|
|
188
|
-
# @return [
|
|
191
|
+
# @return [Moxml::Element] The same element.
|
|
189
192
|
#
|
|
190
193
|
def self.drop_placeholder_namespaces(root)
|
|
191
194
|
placeholders = root.namespace_definitions
|
|
192
|
-
root
|
|
193
|
-
node.
|
|
195
|
+
traverse(root) do |node|
|
|
196
|
+
if node.element? && placeholder_bound?(node, placeholders)
|
|
197
|
+
# moxml keeps the prefix in the C node name; renaming to the
|
|
198
|
+
# local name is the un-prefix operation.
|
|
199
|
+
node.name = node.name
|
|
200
|
+
end
|
|
194
201
|
next unless node.element?
|
|
195
202
|
|
|
196
203
|
drop_attribute_namespaces node, placeholders
|
|
197
204
|
end
|
|
198
205
|
root
|
|
199
206
|
end
|
|
207
|
+
|
|
208
|
+
# Is the node bound to a wrapper placeholder rather than to a
|
|
209
|
+
# namespace it declares itself? A content element may re-declare
|
|
210
|
+
# the same prefix and URI (spec: "keeps a namespace that reuses
|
|
211
|
+
# the placeholder URI"); that declaration is its own, not the
|
|
212
|
+
# wrapper's, and must survive.
|
|
213
|
+
def self.placeholder_bound?(node, placeholders)
|
|
214
|
+
ns = node.namespace
|
|
215
|
+
return false unless placeholders.include?(ns)
|
|
216
|
+
|
|
217
|
+
# declared_namespaces yields Namespace wrappers or [prefix,
|
|
218
|
+
# uri] pairs depending on the adapter path.
|
|
219
|
+
node.declared_namespaces.none? do |own|
|
|
220
|
+
prefix, uri = own.respond_to?(:prefix) ? [own.prefix, own.uri] : own
|
|
221
|
+
prefix == ns.prefix && uri == ns.uri
|
|
222
|
+
end
|
|
223
|
+
end
|
|
200
224
|
private_class_method :drop_placeholder_namespaces
|
|
201
225
|
|
|
202
226
|
#
|
|
@@ -208,21 +232,23 @@ module Relaton
|
|
|
208
232
|
# target redefined" -- the unparseable output this whole path
|
|
209
233
|
# exists to prevent. Drop the prefixed one instead.
|
|
210
234
|
#
|
|
211
|
-
# @param [
|
|
212
|
-
# @param [Array
|
|
213
|
-
# wrapper's own declarations.
|
|
235
|
+
# @param [Moxml::Element] node The element.
|
|
236
|
+
# @param [Array] placeholders The wrapper's own declarations.
|
|
214
237
|
#
|
|
215
238
|
# @return [void]
|
|
216
239
|
#
|
|
217
240
|
def self.drop_attribute_namespaces(node, placeholders)
|
|
218
|
-
node.
|
|
241
|
+
node.attributes.each do |attr|
|
|
219
242
|
next unless placeholders.include?(attr.namespace)
|
|
220
243
|
|
|
221
|
-
if plain_attribute?(node, attr.name) then attr.
|
|
222
|
-
else attr.
|
|
244
|
+
if plain_attribute?(node, attr.name) then attr.remove
|
|
245
|
+
else attr.name = attr.name
|
|
223
246
|
end
|
|
224
247
|
end
|
|
225
248
|
end
|
|
249
|
+
# (attribute prefixes: a content element that re-declares the
|
|
250
|
+
# placeholder keeps its prefixed attributes — the placeholder is
|
|
251
|
+
# out of scope for them the same way it is for element names)
|
|
226
252
|
private_class_method :drop_attribute_namespaces
|
|
227
253
|
|
|
228
254
|
#
|
|
@@ -232,13 +258,13 @@ module Relaton
|
|
|
232
258
|
# the prefixed attribute itself and every un-prefixing would look
|
|
233
259
|
# like a collision. Match on the namespace as well.
|
|
234
260
|
#
|
|
235
|
-
# @param [
|
|
261
|
+
# @param [Moxml::Element] node The element.
|
|
236
262
|
# @param [String] name The un-prefixed attribute name.
|
|
237
263
|
#
|
|
238
264
|
# @return [Boolean] Whether the element carries it.
|
|
239
265
|
#
|
|
240
266
|
def self.plain_attribute?(node, name)
|
|
241
|
-
node.
|
|
267
|
+
node.attributes.any? { |a| a.namespace.nil? && a.name == name }
|
|
242
268
|
end
|
|
243
269
|
private_class_method :plain_attribute?
|
|
244
270
|
|
|
@@ -255,10 +281,25 @@ module Relaton
|
|
|
255
281
|
next if OPAQUE.include?(child.name)
|
|
256
282
|
|
|
257
283
|
sanitize_children(child)
|
|
258
|
-
|
|
284
|
+
unwrap(child) unless ALLOWED.include?(child.name)
|
|
259
285
|
end
|
|
260
286
|
end
|
|
261
287
|
private_class_method :sanitize_children
|
|
288
|
+
|
|
289
|
+
# Replace an element with its own children (Nokogiri's
|
|
290
|
+
# Node#replace(children) has no single-node moxml counterpart).
|
|
291
|
+
def self.unwrap(child)
|
|
292
|
+
child.children.to_a.each { |c| child.add_previous_sibling(c) }
|
|
293
|
+
child.remove
|
|
294
|
+
end
|
|
295
|
+
private_class_method :unwrap
|
|
296
|
+
|
|
297
|
+
# Depth-first walk over node and all descendants.
|
|
298
|
+
def self.traverse(node, &block)
|
|
299
|
+
block.call(node)
|
|
300
|
+
node.children.to_a.each { |c| traverse(c, &block) }
|
|
301
|
+
end
|
|
302
|
+
private_class_method :traverse
|
|
262
303
|
end
|
|
263
304
|
end
|
|
264
305
|
end
|
data/lib/relaton/bib.rb
CHANGED
|
@@ -19,6 +19,16 @@ require_relative "bib/model/bibdata"
|
|
|
19
19
|
require_relative "bib/converter/bibxml"
|
|
20
20
|
require_relative "bib/converter/bibtex"
|
|
21
21
|
require_relative "bib/converter/asciibib"
|
|
22
|
+
|
|
23
|
+
module Relaton
|
|
24
|
+
module Bib
|
|
25
|
+
module Converter
|
|
26
|
+
autoload :Ris, "relaton/bib/converter/ris"
|
|
27
|
+
autoload :Csl, "relaton/bib/converter/csl"
|
|
28
|
+
autoload :Titles, "relaton/bib/converter/titles"
|
|
29
|
+
end
|
|
30
|
+
end
|
|
31
|
+
end
|
|
22
32
|
require_relative "bib/model/relation"
|
|
23
33
|
|
|
24
34
|
module Relaton
|
|
@@ -8,6 +8,7 @@ module Relaton
|
|
|
8
8
|
def initialize
|
|
9
9
|
@short = :relaton_bipm
|
|
10
10
|
@prefix = "BIPM"
|
|
11
|
+
@pubid_identifier = :Bipm # Db cache key
|
|
11
12
|
@defaultprefix = %r{^(?:BIPM|CCTF|CCDS|CGPM|CIPM|JCRB)(?!\w)}
|
|
12
13
|
@idtype = "BIPM"
|
|
13
14
|
@datasets = %w[bipm-data-outcomes bipm-si-brochure rawdata-bipm-metrologia]
|
|
@@ -19,9 +19,9 @@ module Relaton::Bipm
|
|
|
19
19
|
#
|
|
20
20
|
def self.parse(dir)
|
|
21
21
|
affiliations = Dir["#{dir}/*.xml"].each_with_object([]) do |path, m|
|
|
22
|
-
doc =
|
|
22
|
+
doc = Moxml.parse(File.read(path, encoding: "UTF-8"))
|
|
23
23
|
doc.xpath("//aff").each do |aff|
|
|
24
|
-
m << parse_affiliation(aff) if aff.
|
|
24
|
+
m << parse_affiliation(aff) if aff.at_xpath("institution")
|
|
25
25
|
end
|
|
26
26
|
end.uniq { |a| a.organization.name.first.content }
|
|
27
27
|
new affiliations
|
|
@@ -31,18 +31,18 @@ module Relaton::Bipm
|
|
|
31
31
|
# Parse affiliation organization
|
|
32
32
|
# https://github.com/relaton/relaton-data-bipm/issues/17#issuecomment-1367035444
|
|
33
33
|
#
|
|
34
|
-
# @param [
|
|
34
|
+
# @param [Moxml::Element] aff
|
|
35
35
|
#
|
|
36
36
|
# @return [Relaton::Bib::Affiliation] Organization name, country, division, street address
|
|
37
37
|
#
|
|
38
38
|
def self.parse_affiliation(aff)
|
|
39
|
-
text = aff.
|
|
39
|
+
text = aff.at_xpath("text()").text
|
|
40
40
|
return if text.include? "Permanent address:" || text.include?("1005 Southover Lane") ||
|
|
41
41
|
text == "Germany" || text.starts_with?("Guest") || text.starts_with?("Deceased") ||
|
|
42
42
|
text.include?("Author to whom any correspondence should be addressed")
|
|
43
43
|
|
|
44
44
|
args = {}
|
|
45
|
-
institution = aff.
|
|
45
|
+
institution = aff.at_xpath('institution')
|
|
46
46
|
if institution
|
|
47
47
|
name = institution.text
|
|
48
48
|
return if name == "1005 Southover Lane"
|
|
@@ -72,7 +72,7 @@ module Relaton::Bipm
|
|
|
72
72
|
address = []
|
|
73
73
|
addr = aff.xpath("text()[preceding-sibling::institution]").text.gsub(/^\W*|\W*$/, "")
|
|
74
74
|
address << addr unless addr.empty?
|
|
75
|
-
country = aff.
|
|
75
|
+
country = aff.at_xpath('country')
|
|
76
76
|
address << country.text if country && !country.text.empty?
|
|
77
77
|
address = address.join(", ")
|
|
78
78
|
return [] if address.empty?
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
require "
|
|
1
|
+
require "moxml"
|
|
2
2
|
|
|
3
3
|
module Relaton::Bipm
|
|
4
4
|
class SiBrochureParser
|
|
@@ -129,7 +129,7 @@ module Relaton::Bipm
|
|
|
129
129
|
end
|
|
130
130
|
|
|
131
131
|
def extract_editorialgroup(xml)
|
|
132
|
-
doc =
|
|
132
|
+
doc = Moxml.parse(xml)
|
|
133
133
|
doc.xpath("//editorialgroup/committee").map do |committee|
|
|
134
134
|
acronym = committee["acronym"]
|
|
135
135
|
names = committee.xpath("variant").map do |v|
|
|
@@ -19,6 +19,11 @@ module Relaton::Bsi
|
|
|
19
19
|
::Relaton::Bsi::Bibliography.get(code, date, opts)
|
|
20
20
|
end
|
|
21
21
|
|
|
22
|
+
# `Bibliography.get` applies a year to the document the reference names.
|
|
23
|
+
def fold_year(pubid, year)
|
|
24
|
+
fold_year_on_root pubid, year
|
|
25
|
+
end
|
|
26
|
+
|
|
22
27
|
# @param xml [String]
|
|
23
28
|
# @return [Relaton::Bsi::ItemData]
|
|
24
29
|
def from_xml(xml)
|
|
@@ -8,6 +8,7 @@ module Relaton
|
|
|
8
8
|
def initialize # rubocop:disable Lint/MissingSuper
|
|
9
9
|
@short = :relaton_ccsds
|
|
10
10
|
@prefix = "CCSDS"
|
|
11
|
+
@pubid_identifier = :Ccsds # Db cache key
|
|
11
12
|
@defaultprefix = %r{^CCSDS(?!\w)}
|
|
12
13
|
@idtype = "CCSDS"
|
|
13
14
|
@datasets = %w[ccsds]
|
|
@@ -22,6 +23,17 @@ module Relaton
|
|
|
22
23
|
Bibliography.get(code, date, opts)
|
|
23
24
|
end
|
|
24
25
|
|
|
26
|
+
# A format (` (DOC)` or `opts[:format]`) keeps only the item's sources
|
|
27
|
+
# of that format, a filter the Db cache cannot apply to a cached item:
|
|
28
|
+
# such a query gets no key, so it is not cached.
|
|
29
|
+
def cache_key(ref, year, opts)
|
|
30
|
+
require_relative "../ccsds"
|
|
31
|
+
_, format_opts = Bibliography.parse_format(ref, opts.dup)
|
|
32
|
+
return if format_opts[:format]
|
|
33
|
+
|
|
34
|
+
super
|
|
35
|
+
end
|
|
36
|
+
|
|
25
37
|
#
|
|
26
38
|
# Fetch all the documents from a source
|
|
27
39
|
#
|
|
@@ -105,7 +105,7 @@ module Relaton
|
|
|
105
105
|
# @return [Array<RelatonCen::Hit>]
|
|
106
106
|
def hits(resp)
|
|
107
107
|
resp.xpath("//table[@class='dashlist']/tbody/tr/td[2]").map do |h|
|
|
108
|
-
ref = h.
|
|
108
|
+
ref = h.at_xpath("strong/a")
|
|
109
109
|
code = ref.text.strip
|
|
110
110
|
url = ref[:href]
|
|
111
111
|
Hit.new({ code: code, url: url }, self)
|
|
@@ -20,6 +20,11 @@ module Relaton
|
|
|
20
20
|
::Relaton::Cen::Bibliography.get(code, date, opts)
|
|
21
21
|
end
|
|
22
22
|
|
|
23
|
+
# `Bibliography.get` applies a year to the document the reference names.
|
|
24
|
+
def fold_year(pubid, year)
|
|
25
|
+
fold_year_on_root pubid, year
|
|
26
|
+
end
|
|
27
|
+
|
|
23
28
|
# @param xml [String]
|
|
24
29
|
# @return [Relaton::Cen::ItemData]
|
|
25
30
|
def from_xml(xml)
|