relaton 3.0.0.pre.alpha.4 → 3.0.0.pre.alpha.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. checksums.yaml +4 -4
  2. data/lib/relaton/adobe/processor.rb +9 -0
  3. data/lib/relaton/bib/converter/csl.rb +110 -0
  4. data/lib/relaton/bib/converter/ris.rb +104 -0
  5. data/lib/relaton/bib/converter/titles.rb +24 -0
  6. data/lib/relaton/bib/item_data.rb +9 -0
  7. data/lib/relaton/bib/model/item.rb +6 -6
  8. data/lib/relaton/bib/sanitizer.rb +71 -30
  9. data/lib/relaton/bib.rb +10 -0
  10. data/lib/relaton/bipm/processor.rb +1 -0
  11. data/lib/relaton/bipm/rawdata_bipm_metrologia/affiliations.rb +6 -6
  12. data/lib/relaton/bipm/si_brochure_parser.rb +2 -2
  13. data/lib/relaton/bsi/processor.rb +5 -0
  14. data/lib/relaton/ccsds/bibliography.rb +1 -0
  15. data/lib/relaton/ccsds/processor.rb +12 -0
  16. data/lib/relaton/cen/hit_collection.rb +1 -1
  17. data/lib/relaton/cen/processor.rb +5 -0
  18. data/lib/relaton/cen/scraper.rb +9 -9
  19. data/lib/relaton/cie/data_fetcher.rb +18 -18
  20. data/lib/relaton/cie/processor.rb +1 -0
  21. data/lib/relaton/cie.rb +1 -1
  22. data/lib/relaton/cloud.rb +127 -0
  23. data/lib/relaton/core/hit_collection.rb +6 -10
  24. data/lib/relaton/core/processor.rb +68 -0
  25. data/lib/relaton/db/cache.rb +444 -148
  26. data/lib/relaton/db/cache_entry.rb +33 -0
  27. data/lib/relaton/db/registry.rb +52 -0
  28. data/lib/relaton/db.rb +131 -72
  29. data/lib/relaton/doi/crossref.rb +23 -6
  30. data/lib/relaton/doi/processor.rb +1 -0
  31. data/lib/relaton/easc/processor.rb +1 -0
  32. data/lib/relaton/ecma/data_fetcher.rb +1 -1
  33. data/lib/relaton/ecma/data_parser.rb +1 -1
  34. data/lib/relaton/ecma/edition_parser.rb +2 -2
  35. data/lib/relaton/ecma/memento_parser.rb +4 -4
  36. data/lib/relaton/ecma/standard_parser.rb +6 -6
  37. data/lib/relaton/etsi/processor.rb +1 -0
  38. data/lib/relaton/gb/gb_scraper.rb +5 -5
  39. data/lib/relaton/gb/scraper.rb +16 -16
  40. data/lib/relaton/gb/sec_scraper.rb +8 -8
  41. data/lib/relaton/gb/t_scraper.rb +5 -5
  42. data/lib/relaton/gost/processor.rb +1 -0
  43. data/lib/relaton/iala/processor.rb +1 -0
  44. data/lib/relaton/iana/data_fetcher.rb +3 -3
  45. data/lib/relaton/iana/parser.rb +8 -4
  46. data/lib/relaton/iana/processor.rb +9 -0
  47. data/lib/relaton/iec/data_parser.rb +24 -9
  48. data/lib/relaton/iec/processor.rb +6 -0
  49. data/lib/relaton/iec.rb +1 -1
  50. data/lib/relaton/ieee/data_fetcher.rb +1 -1
  51. data/lib/relaton/ieee/processor.rb +8 -0
  52. data/lib/relaton/ietf/data_fetcher.rb +1 -1
  53. data/lib/relaton/ietf/processor.rb +1 -0
  54. data/lib/relaton/ietf/rfc/entry.rb +15 -19
  55. data/lib/relaton/iho/processor.rb +1 -0
  56. data/lib/relaton/index/file_io.rb +2 -2
  57. data/lib/relaton/index/pool.rb +4 -3
  58. data/lib/relaton/index/shard_source.rb +1 -1
  59. data/lib/relaton/index/type.rb +2 -5
  60. data/lib/relaton/isbn/open_library.rb +11 -7
  61. data/lib/relaton/isbn/processor.rb +10 -0
  62. data/lib/relaton/iso/data_parser.rb +2 -2
  63. data/lib/relaton/iso/processor.rb +5 -0
  64. data/lib/relaton/iso/scraper.rb +20 -20
  65. data/lib/relaton/itu/bibliography.rb +99 -13
  66. data/lib/relaton/itu/data_crawler_r.rb +2 -2
  67. data/lib/relaton/itu/hit_collection.rb +64 -52
  68. data/lib/relaton/itu/processor.rb +1 -0
  69. data/lib/relaton/itu/scraper.rb +13 -3
  70. data/lib/relaton/itu.rb +0 -2
  71. data/lib/relaton/jis/data_fetcher.rb +4 -4
  72. data/lib/relaton/jis/processor.rb +1 -0
  73. data/lib/relaton/jis/scraper.rb +8 -8
  74. data/lib/relaton/oasis/browser_agent.rb +2 -2
  75. data/lib/relaton/oasis/data_parser.rb +5 -5
  76. data/lib/relaton/oasis/data_parser_utils.rb +2 -2
  77. data/lib/relaton/oasis/data_part_parser.rb +7 -7
  78. data/lib/relaton/ogc/processor.rb +7 -0
  79. data/lib/relaton/oiml/processor.rb +1 -0
  80. data/lib/relaton/omg/scraper.rb +11 -11
  81. data/lib/relaton/omg.rb +1 -1
  82. data/lib/relaton/plateau/processor.rb +1 -0
  83. data/lib/relaton/un/bibliography.rb +21 -10
  84. data/lib/relaton/un/processor.rb +1 -0
  85. data/lib/relaton/version.rb +1 -1
  86. data/lib/relaton/w3c/processor.rb +1 -0
  87. data/lib/relaton.rb +13 -0
  88. metadata +48 -16
  89. data/lib/relaton/itu/pubid.rb +0 -199
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: d2c4c35d668277f91fb1ad11eafe4da0b25f49fe9110ac81df83d8bae5bebcac
4
- data.tar.gz: 6d0e6ce6134990d786cb67fc20c1a90a1992ced44907191e03340f3a34702c0c
3
+ metadata.gz: b34a85f74b2b8b07d61b51a20d976f1204240dbc353f9e7d73eeeb7262912337
4
+ data.tar.gz: 6e51891af83e32391abb096a788685f42d7cc648fc6a3e3483683868daa8e4b3
5
5
  SHA512:
6
- metadata.gz: dff41b1c067bc5eb8a5737b1fa6dc927dc9ef0d0b04a8e33a7bc4d637c6c78a99bbb8f4ec7cd3aeee5682cab61f099b4d15952884b990da8020db6fe9122df0d
7
- data.tar.gz: 77c8b1aba1fb16829882b3e311416db6e2481d3800c898148810203d0805a4bd0caef2960831a464b44f9e357aff1cdad6cc9573dc02473518179621acf7f1b0
6
+ metadata.gz: 6530a4f026cf099bcd3a1c1b359d307089b57bbe14f94a77362adabafa329abbcb4f64349feaa9061e9a6eda9ff413645af6b14812bdcb3ac9bcb4e22e6ad0a1
7
+ data.tar.gz: 7c555c9ddb34285255f22766b7b9c56be98b657018bfb4503bb66bc316eea7a1c2fd796f21e1910ad41615a0079d8db32f76d2b46bc051a2a4e7c6ed4c0316cc
@@ -12,6 +12,7 @@ module Relaton
12
12
  def initialize
13
13
  @short = :relaton_adobe
14
14
  @prefix = "Adobe"
15
+ @pubid_identifier = :Adobe # Db cache key
15
16
  @defaultprefix = %r{^(?:Adobe|ATN)}
16
17
  @idtype = "Adobe"
17
18
  end
@@ -21,6 +22,14 @@ module Relaton
21
22
  Bibliography.get(code, date, opts)
22
23
  end
23
24
 
25
+ # `Bibliography.get` reads a reference pubid cannot parse as a miss
26
+ # (`pubid_for`), so it gets no key here either: it is not cached.
27
+ def cache_pubid(ref)
28
+ super
29
+ rescue StandardError
30
+ nil
31
+ end
32
+
24
33
  def from_xml(xml)
25
34
  require_relative "../adobe"
26
35
  Item.from_xml xml
@@ -0,0 +1,110 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "json"
4
+
5
+ module Relaton
6
+ module Bib
7
+ module Converter
8
+ # CSL-JSON serialization — the Citation Style Language interchange
9
+ # format consumed by Zotero, Mendeley, and every citeproc.
10
+ module Csl
11
+ ENTRY_TYPES = {
12
+ "standard" => "standard", "book" => "book",
13
+ "article" => "article-journal", "inbook" => "chapter",
14
+ "inproceedings" => "paper-conference", "report" => "report",
15
+ "thesis" => "thesis", "website" => "webpage",
16
+ "webresource" => "webpage",
17
+ }.freeze
18
+
19
+ def self.from_item(item)
20
+ Renderer.new(item).to_s
21
+ end
22
+
23
+ class Renderer
24
+ def initialize(item)
25
+ @item = item
26
+ end
27
+
28
+ def to_s
29
+ out = {}
30
+ docid = primary_docid
31
+ out[:id] = docid.empty? ? "relaton" : docid
32
+ out[:type] = ENTRY_TYPES[@item.type.to_s] || "standard"
33
+ out[:title] = primary_title unless primary_title.empty?
34
+ authors = csl_authors
35
+ out[:author] = authors unless authors.empty?
36
+ out[:issued] = { "date-parts" => [[published_year]] } unless published_year.empty?
37
+ publisher = publisher_name
38
+ out[:publisher] = publisher unless publisher.empty?
39
+ out[:number] = docid unless docid.empty?
40
+ out[:edition] = @item.edition.to_s unless @item.edition.to_s.empty?
41
+ lang = Array(@item.language).first
42
+ out[:language] = lang.to_s if lang
43
+ keywords = Array(@item.keyword).map { |k| k.content.to_s }.reject(&:empty?)
44
+ out[:keyword] = keywords unless keywords.empty?
45
+ out[:URL] = source_uri if source_uri
46
+ [out].to_json + "\n"
47
+ end
48
+
49
+ private
50
+
51
+ def csl_authors
52
+ Array(@item.contributor).filter_map do |c|
53
+ roles = Array(c.role).map(&:type).compact
54
+ next unless (roles & %w[author performer editor]).any?
55
+
56
+ if c.person
57
+ name = c.person.name
58
+ fore = name ? Array(name.forename).map { |f| f.respond_to?(:content) ? f.content.to_s : f.to_s }.join(" ") : ""
59
+ sur = name && name.surname.respond_to?(:content) ? name.surname.content.to_s : name&.surname.to_s
60
+ complete = name && name.completename.respond_to?(:content) ? name.completename.content.to_s : name&.completename.to_s
61
+ if sur.to_s.empty?
62
+ complete.to_s.empty? ? nil : { "literal" => complete }
63
+ else
64
+ entry = { "family" => sur }
65
+ entry["given"] = fore unless fore.empty?
66
+ entry
67
+ end
68
+ elsif c.organization
69
+ literal = org_name(c.organization)
70
+ literal.empty? ? nil : { "literal" => literal }
71
+ end
72
+ end
73
+ end
74
+
75
+ def org_name(org)
76
+ name = Array(org.name).map { |n| n.respond_to?(:content) ? n.content.to_s : n.to_s }.find(&:itself)
77
+ name || org.abbreviation.to_s
78
+ end
79
+
80
+ def publisher_name
81
+ pub = Array(@item.contributor).find do |c|
82
+ Array(c.role).map(&:type).include?("publisher") && c.organization
83
+ end
84
+ pub ? org_name(pub.organization) : ""
85
+ end
86
+
87
+ def primary_title
88
+ Titles.of(@item)
89
+ end
90
+
91
+ def primary_docid
92
+ ids = Array(@item.docidentifier)
93
+ (ids.find(&:primary) || ids.first)&.content.to_s
94
+ end
95
+
96
+ def published_year
97
+ date = Array(@item.date).find { |d| %w[published issued].include?(d.type.to_s) } || @item.date.first
98
+ value = date && (date.at || date.from || date.to)
99
+ value.to_s[/\d{4}/]
100
+ end
101
+
102
+ def source_uri
103
+ Array(@item.source).map { |s| s.respond_to?(:content) ? s.content.to_s : s.to_s }
104
+ .find { |u| u.start_with?("http") }
105
+ end
106
+ end
107
+ end
108
+ end
109
+ end
110
+ end
@@ -0,0 +1,104 @@
1
+ # frozen_string_literal: true
2
+
3
+
4
+ module Relaton
5
+ module Bib
6
+ module Converter
7
+ # RIS serialization — the interchange format for EndNote, Zotero,
8
+ # and Reference Manager. Standards map to TY - STD.
9
+ module Ris
10
+ ENTRY_TYPES = {
11
+ "standard" => "STD", "book" => "BOOK", "article" => "JOUR",
12
+ "inbook" => "CHAP", "inproceedings" => "CONF", "report" => "RPRT",
13
+ "thesis" => "THES", "website" => "ELEC", "webresource" => "ELEC",
14
+ }.freeze
15
+
16
+ def self.from_item(item)
17
+ Renderer.new(item).to_s
18
+ end
19
+
20
+ class Renderer
21
+ def initialize(item)
22
+ @item = item
23
+ end
24
+
25
+ def to_s
26
+ lines = ["TY - #{ENTRY_TYPES[@item.type.to_s] || 'STD'}"]
27
+ add_authors(lines)
28
+ add_field(lines, "TI", primary_title)
29
+ add_field(lines, "ID", primary_docid)
30
+ add_field(lines, "PY", published_year)
31
+ add_field(lines, "ET", @item.edition&.to_s)
32
+ lang = Array(@item.language).first
33
+ add_field(lines, "LA", lang.to_s)
34
+ Array(@item.keyword).each { |k| add_field(lines, "KW", k.content.to_s) }
35
+ add_field(lines, "UR", source_uri)
36
+ lines << "ER - "
37
+ lines.join("\r\n") + "\r\n"
38
+ end
39
+
40
+ private
41
+
42
+ def add_field(lines, tag, value)
43
+ lines << "#{tag} - #{value}" if value && !value.empty?
44
+ end
45
+
46
+ def add_authors(lines)
47
+ Array(@item.contributor).each do |c|
48
+ roles = Array(c.role).map(&:type).compact
49
+ authorish = (roles & %w[author performer editor]).any?
50
+ publisher = roles.include?("publisher")
51
+ if authorish && c.person
52
+ out = person_display_name(c.person.name)
53
+ lines << "AU - #{out}" unless out.empty?
54
+ elsif authorish && c.organization
55
+ out = org_name(c.organization)
56
+ lines << "AU - #{out}" unless out.empty?
57
+ elsif publisher && c.organization
58
+ out = org_name(c.organization)
59
+ lines << "PB - #{out}" unless out.empty? || lines.any? { |l| l.start_with?("PB - ") }
60
+ end
61
+ end
62
+ end
63
+
64
+ def person_display_name(name)
65
+ return "" unless name
66
+
67
+ fore = Array(name.forename).map { |f| f.respond_to?(:content) ? f.content.to_s : f.to_s }.join(" ")
68
+ sur = name.surname.respond_to?(:content) ? name.surname.content.to_s : name.surname.to_s
69
+ complete = name.completename.respond_to?(:content) ? name.completename.content.to_s : name.completename.to_s
70
+ return "#{sur}, #{fore}" unless fore.empty? || sur.empty?
71
+ return sur unless sur.empty?
72
+
73
+ complete
74
+ end
75
+
76
+ def org_name(org)
77
+ name = Array(org.name).map { |n| n.respond_to?(:content) ? n.content.to_s : n.to_s }.find(&:itself)
78
+ name || org.abbreviation.to_s
79
+ end
80
+
81
+ def primary_title
82
+ Titles.of(@item)
83
+ end
84
+
85
+ def primary_docid
86
+ ids = Array(@item.docidentifier)
87
+ (ids.find(&:primary) || ids.first)&.content.to_s
88
+ end
89
+
90
+ def published_year
91
+ date = Array(@item.date).find { |d| %w[published issued].include?(d.type.to_s) } || @item.date.first
92
+ value = date && (date.at || date.from || date.to)
93
+ value.to_s[/\d{4}/]
94
+ end
95
+
96
+ def source_uri
97
+ Array(@item.source).map { |s| s.respond_to?(:content) ? s.content.to_s : s.to_s }
98
+ .find { |u| u.start_with?("http") }
99
+ end
100
+ end
101
+ end
102
+ end
103
+ end
104
+ end
@@ -0,0 +1,24 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Relaton
4
+ module Bib
5
+ module Converter
6
+ # Single citation language; decomposed titles compose, composite or
7
+ # plain titles stand alone. Shared by every export converter.
8
+ module Titles
9
+ def self.of(item)
10
+ titles = Array(item.title).select { |t| t.content.to_s != "" }
11
+ content = ->(t) { t&.content.to_s }
12
+ intro = titles.find { |t| t.type == "title-intro" }
13
+ main = titles.find { |t| t.type == "title-main" }
14
+ part = titles.find { |t| t.type == "title-part" }
15
+ composite = titles.find { |t| t.type == "main" }
16
+ base = [content.call(intro), content.call(main)].reject(&:empty?)
17
+ base = [content.call(composite)].reject(&:empty?) if base.empty?
18
+ base = [content.call(titles.first)].reject(&:empty?) if base.empty?
19
+ [*base, content.call(part)].reject(&:empty?).join(" — ")
20
+ end
21
+ end
22
+ end
23
+ end
24
+ end
@@ -152,6 +152,15 @@ module Relaton
152
152
  Converter::Bibtex.from_item(self).to_s
153
153
  end
154
154
 
155
+ def to_ris
156
+ Converter::Ris.from_item(self)
157
+ end
158
+
159
+ def to_csl_json
160
+ Converter::Csl.from_item(self)
161
+ end
162
+
163
+
155
164
  def to_asciibib
156
165
  Converter::Asciibib.from_item(self)
157
166
  end
@@ -46,10 +46,6 @@ require_relative "ext"
46
46
  require_relative "item_shared"
47
47
  require_relative "type/plain_date"
48
48
 
49
- Lutaml::Model::Config.configure do |config|
50
- config.xml_adapter_type = :nokogiri
51
- end
52
-
53
49
  module Relaton
54
50
  module Bib
55
51
  class Relation < Lutaml::Model::Serializable
@@ -69,8 +65,12 @@ module Relaton
69
65
 
70
66
  # lutaml-model has no built-in dispatch on root element name
71
67
  # (polymorphic_map only works on attribute discriminators), so we
72
- # peek at the root tag with Nokogiri and forward to the right class.
73
- root_name = Nokogiri::XML(xml.to_s).root&.name
68
+ # peek at the root tag with moxml and forward to the right class.
69
+ root_name = begin
70
+ Moxml.parse(xml.to_s).root&.name
71
+ rescue Moxml::ParseError
72
+ nil
73
+ end
74
74
  klass = root_name == "bibdata" ? namespace::Bibdata : namespace::Bibitem
75
75
  klass.from_xml(xml, options)
76
76
  end
@@ -1,4 +1,4 @@
1
- require "nokogiri"
1
+ require "moxml"
2
2
 
3
3
  module Relaton
4
4
  module Bib
@@ -63,11 +63,6 @@ module Relaton
63
63
  # Reserved prefixes. XML declares both, so the content must not.
64
64
  NS_RESERVED = %w[xml xmlns].freeze
65
65
 
66
- # Serialise without the FORMAT option, so the sanitiser keeps the
67
- # shape of element-only content instead of adding newlines and
68
- # indent.
69
- SAVE_OPTS = Nokogiri::XML::Node::SaveOptions::AS_XML
70
-
71
66
  def self.sanitize(content)
72
67
  return content unless sanitizable?(content)
73
68
 
@@ -75,9 +70,7 @@ module Relaton
75
70
  return content if node.nil?
76
71
 
77
72
  sanitize_children(node)
78
- node.children.map do |c|
79
- c.to_xml(encoding: "UTF-8", save_with: SAVE_OPTS)
80
- end.join
73
+ node.children.map { |c| c.to_xml(encoding: "UTF-8", indent: 0, expand_empty: false) }.join
81
74
  end
82
75
 
83
76
  #
@@ -85,13 +78,23 @@ module Relaton
85
78
  #
86
79
  # @param [String] content The raw marked-up content.
87
80
  #
88
- # @return [Nokogiri::XML::Node, nil] The node, or nil when the
89
- # content does not parse.
81
+ # @return [Moxml::Element, nil] The wrapper element, or nil when
82
+ # the content does not parse.
90
83
  #
84
+ # moxml's parse_fragment returns detached nodes without a parent
85
+ # chain, so the document that owns their C memory can be collected
86
+ # while the node wrappers live on. Parse under a synthetic root
87
+ # instead — the same wrapper machinery parse_with_prefixes uses —
88
+ # which keeps the document reachable from the root.
91
89
  def self.parse(content)
92
- fragment = Nokogiri::XML::DocumentFragment.parse(content)
93
- return fragment if fragment.errors.empty?
90
+ # leptris accepts undeclared prefixes instead of failing the
91
+ # parse, so detect them by scan and take the placeholder path
92
+ # directly.
93
+ return parse_with_prefixes(content) if placeholder_declarations(content)
94
94
 
95
+ name = wrapper_name(content)
96
+ Moxml.parse("<#{name}>#{content}</#{name}>").root
97
+ rescue Moxml::ParseError
95
98
  parse_with_prefixes(content)
96
99
  end
97
100
  private_class_method :parse
@@ -112,16 +115,16 @@ module Relaton
112
115
  #
113
116
  # @param [String] content The raw marked-up content.
114
117
  #
115
- # @return [Nokogiri::XML::Element, nil] The wrapper element, or nil
118
+ # @return [Moxml::Element, nil] The wrapper element, or nil
116
119
  # when the content uses no prefix or does not parse.
117
120
  #
118
121
  def self.parse_with_prefixes(content)
119
122
  decl = placeholder_declarations(content) or return
120
123
  name = wrapper_name(content)
121
- doc = Nokogiri::XML "<#{name} #{decl}>#{content}</#{name}>"
122
- return unless doc.errors.empty?
123
-
124
+ doc = Moxml.parse "<#{name} #{decl}>#{content}</#{name}>"
124
125
  drop_placeholder_namespaces doc.root
126
+ rescue Moxml::ParseError
127
+ nil
125
128
  end
126
129
  private_class_method :parse_with_prefixes
127
130
 
@@ -183,20 +186,41 @@ module Relaton
183
186
  # reach the output, so the declarations never leak. Do not
184
187
  # serialise the root itself.
185
188
  #
186
- # @param [Nokogiri::XML::Element] root The wrapper element.
189
+ # @param [Moxml::Element] root The wrapper element.
187
190
  #
188
- # @return [Nokogiri::XML::Element] The same element.
191
+ # @return [Moxml::Element] The same element.
189
192
  #
190
193
  def self.drop_placeholder_namespaces(root)
191
194
  placeholders = root.namespace_definitions
192
- root.traverse do |node|
193
- node.namespace = nil if placeholders.include?(node.namespace)
195
+ traverse(root) do |node|
196
+ if node.element? && placeholder_bound?(node, placeholders)
197
+ # moxml keeps the prefix in the C node name; renaming to the
198
+ # local name is the un-prefix operation.
199
+ node.name = node.name
200
+ end
194
201
  next unless node.element?
195
202
 
196
203
  drop_attribute_namespaces node, placeholders
197
204
  end
198
205
  root
199
206
  end
207
+
208
+ # Is the node bound to a wrapper placeholder rather than to a
209
+ # namespace it declares itself? A content element may re-declare
210
+ # the same prefix and URI (spec: "keeps a namespace that reuses
211
+ # the placeholder URI"); that declaration is its own, not the
212
+ # wrapper's, and must survive.
213
+ def self.placeholder_bound?(node, placeholders)
214
+ ns = node.namespace
215
+ return false unless placeholders.include?(ns)
216
+
217
+ # declared_namespaces yields Namespace wrappers or [prefix,
218
+ # uri] pairs depending on the adapter path.
219
+ node.declared_namespaces.none? do |own|
220
+ prefix, uri = own.respond_to?(:prefix) ? [own.prefix, own.uri] : own
221
+ prefix == ns.prefix && uri == ns.uri
222
+ end
223
+ end
200
224
  private_class_method :drop_placeholder_namespaces
201
225
 
202
226
  #
@@ -208,21 +232,23 @@ module Relaton
208
232
  # target redefined" -- the unparseable output this whole path
209
233
  # exists to prevent. Drop the prefixed one instead.
210
234
  #
211
- # @param [Nokogiri::XML::Element] node The element.
212
- # @param [Array<Nokogiri::XML::Namespace>] placeholders The
213
- # wrapper's own declarations.
235
+ # @param [Moxml::Element] node The element.
236
+ # @param [Array] placeholders The wrapper's own declarations.
214
237
  #
215
238
  # @return [void]
216
239
  #
217
240
  def self.drop_attribute_namespaces(node, placeholders)
218
- node.attribute_nodes.each do |attr|
241
+ node.attributes.each do |attr|
219
242
  next unless placeholders.include?(attr.namespace)
220
243
 
221
- if plain_attribute?(node, attr.name) then attr.unlink
222
- else attr.namespace = nil
244
+ if plain_attribute?(node, attr.name) then attr.remove
245
+ else attr.name = attr.name
223
246
  end
224
247
  end
225
248
  end
249
+ # (attribute prefixes: a content element that re-declares the
250
+ # placeholder keeps its prefixed attributes — the placeholder is
251
+ # out of scope for them the same way it is for element names)
226
252
  private_class_method :drop_attribute_namespaces
227
253
 
228
254
  #
@@ -232,13 +258,13 @@ module Relaton
232
258
  # the prefixed attribute itself and every un-prefixing would look
233
259
  # like a collision. Match on the namespace as well.
234
260
  #
235
- # @param [Nokogiri::XML::Element] node The element.
261
+ # @param [Moxml::Element] node The element.
236
262
  # @param [String] name The un-prefixed attribute name.
237
263
  #
238
264
  # @return [Boolean] Whether the element carries it.
239
265
  #
240
266
  def self.plain_attribute?(node, name)
241
- node.attribute_nodes.any? { |a| a.namespace.nil? && a.name == name }
267
+ node.attributes.any? { |a| a.namespace.nil? && a.name == name }
242
268
  end
243
269
  private_class_method :plain_attribute?
244
270
 
@@ -255,10 +281,25 @@ module Relaton
255
281
  next if OPAQUE.include?(child.name)
256
282
 
257
283
  sanitize_children(child)
258
- child.replace(child.children) unless ALLOWED.include?(child.name)
284
+ unwrap(child) unless ALLOWED.include?(child.name)
259
285
  end
260
286
  end
261
287
  private_class_method :sanitize_children
288
+
289
+ # Replace an element with its own children (Nokogiri's
290
+ # Node#replace(children) has no single-node moxml counterpart).
291
+ def self.unwrap(child)
292
+ child.children.to_a.each { |c| child.add_previous_sibling(c) }
293
+ child.remove
294
+ end
295
+ private_class_method :unwrap
296
+
297
+ # Depth-first walk over node and all descendants.
298
+ def self.traverse(node, &block)
299
+ block.call(node)
300
+ node.children.to_a.each { |c| traverse(c, &block) }
301
+ end
302
+ private_class_method :traverse
262
303
  end
263
304
  end
264
305
  end
data/lib/relaton/bib.rb CHANGED
@@ -19,6 +19,16 @@ require_relative "bib/model/bibdata"
19
19
  require_relative "bib/converter/bibxml"
20
20
  require_relative "bib/converter/bibtex"
21
21
  require_relative "bib/converter/asciibib"
22
+
23
+ module Relaton
24
+ module Bib
25
+ module Converter
26
+ autoload :Ris, "relaton/bib/converter/ris"
27
+ autoload :Csl, "relaton/bib/converter/csl"
28
+ autoload :Titles, "relaton/bib/converter/titles"
29
+ end
30
+ end
31
+ end
22
32
  require_relative "bib/model/relation"
23
33
 
24
34
  module Relaton
@@ -8,6 +8,7 @@ module Relaton
8
8
  def initialize
9
9
  @short = :relaton_bipm
10
10
  @prefix = "BIPM"
11
+ @pubid_identifier = :Bipm # Db cache key
11
12
  @defaultprefix = %r{^(?:BIPM|CCTF|CCDS|CGPM|CIPM|JCRB)(?!\w)}
12
13
  @idtype = "BIPM"
13
14
  @datasets = %w[bipm-data-outcomes bipm-si-brochure rawdata-bipm-metrologia]
@@ -19,9 +19,9 @@ module Relaton::Bipm
19
19
  #
20
20
  def self.parse(dir)
21
21
  affiliations = Dir["#{dir}/*.xml"].each_with_object([]) do |path, m|
22
- doc = Nokogiri::XML(File.read(path, encoding: "UTF-8"))
22
+ doc = Moxml.parse(File.read(path, encoding: "UTF-8"))
23
23
  doc.xpath("//aff").each do |aff|
24
- m << parse_affiliation(aff) if aff.at("institution")
24
+ m << parse_affiliation(aff) if aff.at_xpath("institution")
25
25
  end
26
26
  end.uniq { |a| a.organization.name.first.content }
27
27
  new affiliations
@@ -31,18 +31,18 @@ module Relaton::Bipm
31
31
  # Parse affiliation organization
32
32
  # https://github.com/relaton/relaton-data-bipm/issues/17#issuecomment-1367035444
33
33
  #
34
- # @param [Nokogiri::XML::Element] aff
34
+ # @param [Moxml::Element] aff
35
35
  #
36
36
  # @return [Relaton::Bib::Affiliation] Organization name, country, division, street address
37
37
  #
38
38
  def self.parse_affiliation(aff)
39
- text = aff.at("text()").text
39
+ text = aff.at_xpath("text()").text
40
40
  return if text.include? "Permanent address:" || text.include?("1005 Southover Lane") ||
41
41
  text == "Germany" || text.starts_with?("Guest") || text.starts_with?("Deceased") ||
42
42
  text.include?("Author to whom any correspondence should be addressed")
43
43
 
44
44
  args = {}
45
- institution = aff.at('institution')
45
+ institution = aff.at_xpath('institution')
46
46
  if institution
47
47
  name = institution.text
48
48
  return if name == "1005 Southover Lane"
@@ -72,7 +72,7 @@ module Relaton::Bipm
72
72
  address = []
73
73
  addr = aff.xpath("text()[preceding-sibling::institution]").text.gsub(/^\W*|\W*$/, "")
74
74
  address << addr unless addr.empty?
75
- country = aff.at('country')
75
+ country = aff.at_xpath('country')
76
76
  address << country.text if country && !country.text.empty?
77
77
  address = address.join(", ")
78
78
  return [] if address.empty?
@@ -1,4 +1,4 @@
1
- require "nokogiri"
1
+ require "moxml"
2
2
 
3
3
  module Relaton::Bipm
4
4
  class SiBrochureParser
@@ -129,7 +129,7 @@ module Relaton::Bipm
129
129
  end
130
130
 
131
131
  def extract_editorialgroup(xml)
132
- doc = Nokogiri::XML(xml)
132
+ doc = Moxml.parse(xml)
133
133
  doc.xpath("//editorialgroup/committee").map do |committee|
134
134
  acronym = committee["acronym"]
135
135
  names = committee.xpath("variant").map do |v|
@@ -19,6 +19,11 @@ module Relaton::Bsi
19
19
  ::Relaton::Bsi::Bibliography.get(code, date, opts)
20
20
  end
21
21
 
22
+ # `Bibliography.get` applies a year to the document the reference names.
23
+ def fold_year(pubid, year)
24
+ fold_year_on_root pubid, year
25
+ end
26
+
22
27
  # @param xml [String]
23
28
  # @return [Relaton::Bsi::ItemData]
24
29
  def from_xml(xml)
@@ -46,6 +46,7 @@ module Relaton
46
46
  opts[:format] ||= Regexp.last_match(1)
47
47
  [ref, opts]
48
48
  end
49
+ public :parse_format
49
50
 
50
51
  def fetch_item(ref)
51
52
  hit = search(ref).first
@@ -8,6 +8,7 @@ module Relaton
8
8
  def initialize # rubocop:disable Lint/MissingSuper
9
9
  @short = :relaton_ccsds
10
10
  @prefix = "CCSDS"
11
+ @pubid_identifier = :Ccsds # Db cache key
11
12
  @defaultprefix = %r{^CCSDS(?!\w)}
12
13
  @idtype = "CCSDS"
13
14
  @datasets = %w[ccsds]
@@ -22,6 +23,17 @@ module Relaton
22
23
  Bibliography.get(code, date, opts)
23
24
  end
24
25
 
26
+ # A format (` (DOC)` or `opts[:format]`) keeps only the item's sources
27
+ # of that format, a filter the Db cache cannot apply to a cached item:
28
+ # such a query gets no key, so it is not cached.
29
+ def cache_key(ref, year, opts)
30
+ require_relative "../ccsds"
31
+ _, format_opts = Bibliography.parse_format(ref, opts.dup)
32
+ return if format_opts[:format]
33
+
34
+ super
35
+ end
36
+
25
37
  #
26
38
  # Fetch all the documents from a source
27
39
  #
@@ -105,7 +105,7 @@ module Relaton
105
105
  # @return [Array<RelatonCen::Hit>]
106
106
  def hits(resp)
107
107
  resp.xpath("//table[@class='dashlist']/tbody/tr/td[2]").map do |h|
108
- ref = h.at("strong/a")
108
+ ref = h.at_xpath("strong/a")
109
109
  code = ref.text.strip
110
110
  url = ref[:href]
111
111
  Hit.new({ code: code, url: url }, self)
@@ -20,6 +20,11 @@ module Relaton
20
20
  ::Relaton::Cen::Bibliography.get(code, date, opts)
21
21
  end
22
22
 
23
+ # `Bibliography.get` applies a year to the document the reference names.
24
+ def fold_year(pubid, year)
25
+ fold_year_on_root pubid, year
26
+ end
27
+
23
28
  # @param xml [String]
24
29
  # @return [Relaton::Cen::ItemData]
25
30
  def from_xml(xml)