relaton 3.0.0.pre.alpha.4 → 3.0.0.pre.alpha.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/lib/relaton/adobe/processor.rb +9 -0
- data/lib/relaton/bib/converter/csl.rb +110 -0
- data/lib/relaton/bib/converter/ris.rb +104 -0
- data/lib/relaton/bib/converter/titles.rb +24 -0
- data/lib/relaton/bib/item_data.rb +9 -0
- data/lib/relaton/bib/model/item.rb +6 -6
- data/lib/relaton/bib/sanitizer.rb +71 -30
- data/lib/relaton/bib.rb +10 -0
- data/lib/relaton/bipm/processor.rb +1 -0
- data/lib/relaton/bipm/rawdata_bipm_metrologia/affiliations.rb +6 -6
- data/lib/relaton/bipm/si_brochure_parser.rb +2 -2
- data/lib/relaton/bsi/processor.rb +5 -0
- data/lib/relaton/ccsds/bibliography.rb +1 -0
- data/lib/relaton/ccsds/processor.rb +12 -0
- data/lib/relaton/cen/hit_collection.rb +1 -1
- data/lib/relaton/cen/processor.rb +5 -0
- data/lib/relaton/cen/scraper.rb +9 -9
- data/lib/relaton/cie/data_fetcher.rb +18 -18
- data/lib/relaton/cie/processor.rb +1 -0
- data/lib/relaton/cie.rb +1 -1
- data/lib/relaton/cloud.rb +127 -0
- data/lib/relaton/core/hit_collection.rb +6 -10
- data/lib/relaton/core/processor.rb +68 -0
- data/lib/relaton/db/cache.rb +444 -148
- data/lib/relaton/db/cache_entry.rb +33 -0
- data/lib/relaton/db/registry.rb +52 -0
- data/lib/relaton/db.rb +131 -72
- data/lib/relaton/doi/crossref.rb +23 -6
- data/lib/relaton/doi/processor.rb +1 -0
- data/lib/relaton/easc/processor.rb +1 -0
- data/lib/relaton/ecma/data_fetcher.rb +1 -1
- data/lib/relaton/ecma/data_parser.rb +1 -1
- data/lib/relaton/ecma/edition_parser.rb +2 -2
- data/lib/relaton/ecma/memento_parser.rb +4 -4
- data/lib/relaton/ecma/standard_parser.rb +6 -6
- data/lib/relaton/etsi/processor.rb +1 -0
- data/lib/relaton/gb/gb_scraper.rb +5 -5
- data/lib/relaton/gb/scraper.rb +16 -16
- data/lib/relaton/gb/sec_scraper.rb +8 -8
- data/lib/relaton/gb/t_scraper.rb +5 -5
- data/lib/relaton/gost/processor.rb +1 -0
- data/lib/relaton/iala/processor.rb +1 -0
- data/lib/relaton/iana/data_fetcher.rb +3 -3
- data/lib/relaton/iana/parser.rb +8 -4
- data/lib/relaton/iana/processor.rb +9 -0
- data/lib/relaton/iec/data_parser.rb +24 -9
- data/lib/relaton/iec/processor.rb +6 -0
- data/lib/relaton/iec.rb +1 -1
- data/lib/relaton/ieee/data_fetcher.rb +1 -1
- data/lib/relaton/ieee/processor.rb +8 -0
- data/lib/relaton/ietf/data_fetcher.rb +1 -1
- data/lib/relaton/ietf/processor.rb +1 -0
- data/lib/relaton/ietf/rfc/entry.rb +15 -19
- data/lib/relaton/iho/processor.rb +1 -0
- data/lib/relaton/index/file_io.rb +2 -2
- data/lib/relaton/index/pool.rb +4 -3
- data/lib/relaton/index/shard_source.rb +1 -1
- data/lib/relaton/index/type.rb +2 -5
- data/lib/relaton/isbn/open_library.rb +11 -7
- data/lib/relaton/isbn/processor.rb +10 -0
- data/lib/relaton/iso/data_parser.rb +2 -2
- data/lib/relaton/iso/processor.rb +5 -0
- data/lib/relaton/iso/scraper.rb +20 -20
- data/lib/relaton/itu/bibliography.rb +99 -13
- data/lib/relaton/itu/data_crawler_r.rb +2 -2
- data/lib/relaton/itu/hit_collection.rb +64 -52
- data/lib/relaton/itu/processor.rb +1 -0
- data/lib/relaton/itu/scraper.rb +13 -3
- data/lib/relaton/itu.rb +0 -2
- data/lib/relaton/jis/data_fetcher.rb +4 -4
- data/lib/relaton/jis/processor.rb +1 -0
- data/lib/relaton/jis/scraper.rb +8 -8
- data/lib/relaton/oasis/browser_agent.rb +2 -2
- data/lib/relaton/oasis/data_parser.rb +5 -5
- data/lib/relaton/oasis/data_parser_utils.rb +2 -2
- data/lib/relaton/oasis/data_part_parser.rb +7 -7
- data/lib/relaton/ogc/processor.rb +7 -0
- data/lib/relaton/oiml/processor.rb +1 -0
- data/lib/relaton/omg/scraper.rb +11 -11
- data/lib/relaton/omg.rb +1 -1
- data/lib/relaton/plateau/processor.rb +1 -0
- data/lib/relaton/un/bibliography.rb +21 -10
- data/lib/relaton/un/processor.rb +1 -0
- data/lib/relaton/version.rb +1 -1
- data/lib/relaton/w3c/processor.rb +1 -0
- data/lib/relaton.rb +13 -0
- metadata +48 -16
- data/lib/relaton/itu/pubid.rb +0 -199
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
require "lutaml/model"
|
|
2
|
+
|
|
3
|
+
module Relaton
|
|
4
|
+
class Db
|
|
5
|
+
# One row of the cache index. A row is keyed either by a pubid (`id`, its
|
|
6
|
+
# `to_hash`) or, for a processor with no pubid class, by a string (`key`).
|
|
7
|
+
# `file` names the document in the doc store; several rows can share one
|
|
8
|
+
# file (a query row and the row of the document it returned). A
|
|
9
|
+
# `not_found` row has no file.
|
|
10
|
+
class CacheEntry < Lutaml::Model::Serializable
|
|
11
|
+
DOC = "doc".freeze
|
|
12
|
+
NOT_FOUND = "not_found".freeze
|
|
13
|
+
|
|
14
|
+
attribute :id, :hash
|
|
15
|
+
attribute :key, :string
|
|
16
|
+
attribute :status, :string
|
|
17
|
+
attribute :file, :string
|
|
18
|
+
attribute :fetched, :string
|
|
19
|
+
|
|
20
|
+
key_value do
|
|
21
|
+
map "id", to: :id
|
|
22
|
+
map "key", to: :key
|
|
23
|
+
map "status", to: :status
|
|
24
|
+
map "file", to: :file
|
|
25
|
+
map "fetched", to: :fetched
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
def not_found?
|
|
29
|
+
status == NOT_FOUND
|
|
30
|
+
end
|
|
31
|
+
end
|
|
32
|
+
end
|
|
33
|
+
end
|
data/lib/relaton/db/registry.rb
CHANGED
|
@@ -92,6 +92,58 @@ module Relaton
|
|
|
92
92
|
end
|
|
93
93
|
|
|
94
94
|
#
|
|
95
|
+
# The processor whose pubid class the identifier belongs to. Only the
|
|
96
|
+
# generic `Pubid::AllPartsIdentifier`, which belongs to no flavor, is
|
|
97
|
+
# matched through the document it wraps (`#root`). Any other identifier
|
|
98
|
+
# is matched by its own class: the `#root` of an adoption is the adopted
|
|
99
|
+
# document, so `CEN ISO/TS 21003-7` would otherwise be filed as ISO.
|
|
100
|
+
#
|
|
101
|
+
# @param pubid [Pubid::Identifier]
|
|
102
|
+
# @return [Relaton::Core::Processor, nil]
|
|
103
|
+
#
|
|
104
|
+
def processor_by_pubid(pubid)
|
|
105
|
+
generic = pubid.instance_of?(::Pubid::AllPartsIdentifier)
|
|
106
|
+
id = generic ? pubid.root : pubid
|
|
107
|
+
processors.values.detect do |processor|
|
|
108
|
+
klass = processor.pubid_class
|
|
109
|
+
klass && id.is_a?(klass)
|
|
110
|
+
end
|
|
111
|
+
end
|
|
112
|
+
|
|
113
|
+
#
|
|
114
|
+
# Pubid namespaces whose spelling differs from the processor short
|
|
115
|
+
# name.
|
|
116
|
+
PUBID_FLAVOR_ALIASES = { "cencenelec" => "cen", "tgpp" => "3gpp" }.freeze
|
|
117
|
+
|
|
118
|
+
# Flavors routed by the parsed Pubid class (relaton#205, pilot). Only
|
|
119
|
+
# processors on this list answer parse-first routing: an allowlist,
|
|
120
|
+
# not a guess. Pubid::Un is deliberately absent — its Document
|
|
121
|
+
# identifier parses DOI-shaped strings ("10.17487/RFC3986") that must
|
|
122
|
+
# not be hijacked away from their current handling. DOI and ISBN are
|
|
123
|
+
# present: their canonical forms carry their own token (`doi:…`,
|
|
124
|
+
# `ISBN …`), which pubid detects exactly as the prefix regex routes it.
|
|
125
|
+
PARSE_ROUTED_FLAVORS = %w[
|
|
126
|
+
bipm bs cencenelec calconnect cc ccsds cen cie csa doi
|
|
127
|
+
ecma ecs etsi gost iala iana iec ieee ietf isbn iso itu
|
|
128
|
+
jcgm jis nist oasis ogc oiml plateau w3c xsf 3gpp
|
|
129
|
+
].freeze
|
|
130
|
+
|
|
131
|
+
# Find the processor that owns the parsed Pubid's flavor. The
|
|
132
|
+
# namespace is matched against the processor short name, with the
|
|
133
|
+
# alias table for the spellings that differ.
|
|
134
|
+
#
|
|
135
|
+
# @param pubid [Pubid::Core::Identifier] parsed query
|
|
136
|
+
# @return [Symbol, nil] standard class name
|
|
137
|
+
#
|
|
138
|
+
def class_by_pubid(pubid)
|
|
139
|
+
ns = pubid.class.name.split("::")[1]&.downcase
|
|
140
|
+
flavor = PUBID_FLAVOR_ALIASES.fetch(ns, ns)
|
|
141
|
+
return nil unless PARSE_ROUTED_FLAVORS.include?(flavor)
|
|
142
|
+
|
|
143
|
+
key = "relaton_#{flavor}".to_sym
|
|
144
|
+
processors.key?(key) ? key : nil
|
|
145
|
+
end
|
|
146
|
+
|
|
95
147
|
# Find processor by refernce or prefix
|
|
96
148
|
#
|
|
97
149
|
# @param ref [String] reference or prefix
|
data/lib/relaton/db.rb
CHANGED
|
@@ -1,11 +1,13 @@
|
|
|
1
1
|
require "relaton/bib"
|
|
2
2
|
require "yaml"
|
|
3
3
|
require "net/http"
|
|
4
|
-
require "
|
|
4
|
+
require "moxml"
|
|
5
5
|
require "fileutils"
|
|
6
6
|
require "date"
|
|
7
7
|
|
|
8
8
|
module Relaton
|
|
9
|
+
autoload :Cloud, "relaton/cloud"
|
|
10
|
+
|
|
9
11
|
class Db
|
|
10
12
|
# @param global_cache [String] directory of global DB
|
|
11
13
|
# @param local_cache [String] directory of local DB
|
|
@@ -64,7 +66,14 @@ module Relaton
|
|
|
64
66
|
##
|
|
65
67
|
def fetch(text, year = nil, opts = {})
|
|
66
68
|
reference = text.strip
|
|
67
|
-
|
|
69
|
+
require "pubid"
|
|
70
|
+
parsed = begin
|
|
71
|
+
Pubid.parse(reference)
|
|
72
|
+
rescue Pubid::Errors::Error, Parslet::ParseFailed
|
|
73
|
+
nil
|
|
74
|
+
end
|
|
75
|
+
stdclass = (parsed && @registry.class_by_pubid(parsed)) ||
|
|
76
|
+
@registry.class_by_ref(reference) || return
|
|
68
77
|
processor = @registry[stdclass]
|
|
69
78
|
ref = if processor.respond_to?(:urn_to_code)
|
|
70
79
|
processor.urn_to_code(reference)&.first
|
|
@@ -90,8 +99,8 @@ module Relaton
|
|
|
90
99
|
result = []
|
|
91
100
|
db = @db || @local_db
|
|
92
101
|
if db
|
|
93
|
-
result += db.all do |
|
|
94
|
-
search_xml
|
|
102
|
+
result += db.all do |processor, xml|
|
|
103
|
+
search_xml processor, xml, text, edition, year
|
|
95
104
|
end.compact
|
|
96
105
|
end
|
|
97
106
|
result
|
|
@@ -169,9 +178,10 @@ module Relaton
|
|
|
169
178
|
# @return [String]
|
|
170
179
|
def to_xml
|
|
171
180
|
db = @local_db || @db || return
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
181
|
+
parts = db.all.join(" ")
|
|
182
|
+
Moxml.parse(
|
|
183
|
+
"<?xml version=\"1.0\" encoding=\"UTF-8\"?>\n<documents>#{parts}</documents>",
|
|
184
|
+
).to_xml(indent: 0, expand_empty: false)
|
|
175
185
|
end
|
|
176
186
|
|
|
177
187
|
private
|
|
@@ -195,17 +205,15 @@ module Relaton
|
|
|
195
205
|
opts.merge(code: code, year: year).map { |k, v| "#{k}=#{v}" }.join "&"
|
|
196
206
|
end
|
|
197
207
|
|
|
198
|
-
def search_xml(
|
|
208
|
+
def search_xml(processor, xml, text, edition, year)
|
|
209
|
+
return unless processor
|
|
199
210
|
return unless text.nil? || match_xml_text?(xml, text)
|
|
200
211
|
|
|
201
|
-
search_edition_year(
|
|
212
|
+
search_edition_year(processor, xml, edition, year)
|
|
202
213
|
end
|
|
203
214
|
|
|
204
|
-
def search_edition_year(
|
|
205
|
-
|
|
206
|
-
item = if file.match?(/xml$/) then processor.from_xml(content)
|
|
207
|
-
else processor.from_yaml(content)
|
|
208
|
-
end
|
|
215
|
+
def search_edition_year(processor, content, edition, year) # rubocop:disable Metrics/CyclomaticComplexity,Metrics/PerceivedComplexity
|
|
216
|
+
item = processor.from_xml(content)
|
|
209
217
|
item if (edition.nil? || item.edition.content == edition) && (year.nil? ||
|
|
210
218
|
item.date.detect do |d|
|
|
211
219
|
d.type == "published" && d.at.to_date.year.to_s == year.to_s
|
|
@@ -240,7 +248,9 @@ module Relaton
|
|
|
240
248
|
if updates
|
|
241
249
|
doc.relation << Bib::Relation.new(bibitem: updates, type: "updates")
|
|
242
250
|
end
|
|
243
|
-
|
|
251
|
+
# The supplement joins its base as the flavor's identifier spells it:
|
|
252
|
+
# `NIST SP 800-38A Add`, not `/Add`, which pubid does not parse.
|
|
253
|
+
divider = %i[relaton_itu relaton_nist].include?(stdclass) ? " " : "/"
|
|
244
254
|
refs[1..].each_with_object(doc) do |c, d|
|
|
245
255
|
bib = check_bibliocache(ref + divider + c, year, opts, stdclass)
|
|
246
256
|
if bib
|
|
@@ -251,18 +261,52 @@ module Relaton
|
|
|
251
261
|
end
|
|
252
262
|
end
|
|
253
263
|
|
|
264
|
+
# The legacy string cache key, for a processor with no pubid class. The
|
|
265
|
+
# publication date range is not part of it: it filters, it is not
|
|
266
|
+
# identity.
|
|
254
267
|
def std_id(code, year, opts, stdclass)
|
|
255
268
|
prefix, code = strip_id_wrapper(code, stdclass)
|
|
256
269
|
ret = code
|
|
257
270
|
ret += (stdclass == :relaton_gb ? "-" : ":") + year if year
|
|
258
271
|
ret += " (all parts)" if opts[:all_parts]
|
|
259
|
-
after = opts[:publication_date_after]
|
|
260
|
-
ret += " after-#{after}" if after
|
|
261
|
-
before = opts[:publication_date_before]
|
|
262
|
-
ret += " before-#{before}" if before
|
|
263
272
|
["#{prefix}(#{ret.strip})", code]
|
|
264
273
|
end
|
|
265
274
|
|
|
275
|
+
#
|
|
276
|
+
# The cache key of a query: the flavor's parsed pubid, with the `year`
|
|
277
|
+
# and `all_parts` options folded in. A reference the flavor cannot parse
|
|
278
|
+
# raises `Pubid::Errors::ParseError`. Nil when the flavor has a pubid
|
|
279
|
+
# class but gives no key for this query (a miss by the flavor's own rule,
|
|
280
|
+
# or a query whose answer the cache cannot hold, such as a CCSDS format):
|
|
281
|
+
# that query is not cached. A processor with no pubid class gets the
|
|
282
|
+
# legacy string key.
|
|
283
|
+
#
|
|
284
|
+
# @return [Pubid::Identifier, String, nil]
|
|
285
|
+
#
|
|
286
|
+
def cache_key(code, year, opts, stdclass)
|
|
287
|
+
processor = @registry[stdclass]
|
|
288
|
+
unless processor.pubid_class
|
|
289
|
+
return std_id(code, year, opts, stdclass).first
|
|
290
|
+
end
|
|
291
|
+
|
|
292
|
+
processor.cache_key(code, year, opts)
|
|
293
|
+
end
|
|
294
|
+
|
|
295
|
+
# The key of a fetched document, from its primary identifier. The
|
|
296
|
+
# identifier is data, so an unparseable one gives no key.
|
|
297
|
+
def item_key(bib, stdclass)
|
|
298
|
+
docid = bib.docidentifier.detect(&:primary) || bib.docidentifier.first
|
|
299
|
+
return unless docid&.content
|
|
300
|
+
|
|
301
|
+
cache_key docid.content, nil, {}, stdclass
|
|
302
|
+
rescue ::Pubid::Errors::Error, Parslet::ParseFailed
|
|
303
|
+
nil
|
|
304
|
+
end
|
|
305
|
+
|
|
306
|
+
def date_range?(opts)
|
|
307
|
+
opts[:publication_date_before] || opts[:publication_date_after]
|
|
308
|
+
end
|
|
309
|
+
|
|
266
310
|
def strip_id_wrapper(code, stdclass)
|
|
267
311
|
prefix = @registry[stdclass].prefix
|
|
268
312
|
code =
|
|
@@ -280,23 +324,9 @@ module Relaton
|
|
|
280
324
|
end
|
|
281
325
|
|
|
282
326
|
def check_bibliocache(code, year, opts, stdclass) # rubocop:disable Metrics/AbcSize,Metrics/CyclomaticComplexity,Metrics/MethodLength,Metrics/PerceivedComplexity
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
)
|
|
287
|
-
base_id, = std_id(code, year, base_opts, stdclass)
|
|
288
|
-
db = @local_db || @db
|
|
289
|
-
if db&.valid_entry?(base_id, year)
|
|
290
|
-
entry = db[base_id]
|
|
291
|
-
if entry && !entry.match?(/^not_found/) &&
|
|
292
|
-
pub_date_in_range?(entry, opts)
|
|
293
|
-
return bib_retval(entry, stdclass)
|
|
294
|
-
end
|
|
295
|
-
end
|
|
296
|
-
end
|
|
297
|
-
|
|
298
|
-
id, searchcode = std_id(code, year, opts, stdclass)
|
|
299
|
-
db = @local_db || @db
|
|
327
|
+
_, searchcode = strip_id_wrapper(code, stdclass)
|
|
328
|
+
id = cache_key(searchcode, year, opts, stdclass)
|
|
329
|
+
db = id && (@local_db || @db)
|
|
300
330
|
altdb = @local_db && @db ? @db : nil
|
|
301
331
|
if db.nil?
|
|
302
332
|
return if opts[:fetch_db]
|
|
@@ -304,10 +334,11 @@ module Relaton
|
|
|
304
334
|
bibentry = new_bib_entry(searchcode, year, opts, stdclass)
|
|
305
335
|
return bib_retval(bibentry, stdclass)
|
|
306
336
|
end
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
db.delete(id) unless db.valid_entry?(id, year)
|
|
337
|
+
if date_range?(opts)
|
|
338
|
+
return check_date_range(searchcode, id, year, opts, stdclass)
|
|
310
339
|
end
|
|
340
|
+
|
|
341
|
+
@semaphore.synchronize { db.expire id, year }
|
|
311
342
|
if altdb
|
|
312
343
|
return bib_retval(altdb[id], stdclass) if opts[:fetch_db]
|
|
313
344
|
|
|
@@ -340,28 +371,60 @@ module Relaton
|
|
|
340
371
|
entry
|
|
341
372
|
end
|
|
342
373
|
|
|
343
|
-
def fetch_entry(code, year, opts, stdclass, **args)
|
|
374
|
+
def fetch_entry(code, year, opts, stdclass, **args) # rubocop:disable Metrics/AbcSize
|
|
344
375
|
processor = @registry[stdclass]
|
|
345
376
|
bib = net_retry(code, year, opts, processor, opts.fetch(:retries, 1))
|
|
346
|
-
|
|
347
|
-
entry = check_entry(bib, stdclass, **args)
|
|
377
|
+
entry = bib_entry bib
|
|
348
378
|
return entry if args[:db].nil?
|
|
349
379
|
|
|
350
|
-
|
|
351
|
-
|
|
380
|
+
# `no_cache` refreshes a cached entry, but a failed fetch does not
|
|
381
|
+
# replace a cached document with `not_found`.
|
|
382
|
+
refresh = opts[:no_cache] && bib.respond_to?(:to_xml)
|
|
383
|
+
@semaphore.synchronize do
|
|
384
|
+
if refresh || !args[:db][args[:id]]
|
|
385
|
+
save_bib args[:db], args[:id], bib, entry, stdclass
|
|
386
|
+
end
|
|
387
|
+
end
|
|
388
|
+
entry
|
|
352
389
|
end
|
|
353
390
|
|
|
354
|
-
|
|
355
|
-
|
|
391
|
+
#
|
|
392
|
+
# Cache a fetched document. The document's own identifier gets a row;
|
|
393
|
+
# when the query key differs from it (an undated or incomplete query),
|
|
394
|
+
# the query gets a row that points to the same document.
|
|
395
|
+
#
|
|
396
|
+
def save_bib(db, key, bib, entry, stdclass)
|
|
397
|
+
item = bib.respond_to?(:docidentifier) && item_key(bib, stdclass)
|
|
398
|
+
db.store key, entry, item_key: item || nil
|
|
399
|
+
end
|
|
356
400
|
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
401
|
+
#
|
|
402
|
+
# A publication date range selects among the cached editions of the
|
|
403
|
+
# reference, so it is not part of the key. On a miss the flavor is asked
|
|
404
|
+
# with the range, and its answer is cached under its own identifier only:
|
|
405
|
+
# the query row keeps pointing to the latest edition.
|
|
406
|
+
#
|
|
407
|
+
def check_date_range(code, key, year, opts, stdclass) # rubocop:disable Metrics/AbcSize,Metrics/MethodLength,Metrics/CyclomaticComplexity,Metrics/PerceivedComplexity
|
|
408
|
+
caches = [@local_db, @db].compact
|
|
409
|
+
cached = caches.flat_map { |c| c.candidates(key) }.filter_map do |_, xml|
|
|
410
|
+
date = published_date(xml)
|
|
411
|
+
[date, xml] if date && pub_date_in_range?(xml, opts)
|
|
412
|
+
end.max_by(&:first)
|
|
413
|
+
return bib_retval(cached.last, stdclass) if cached && !opts[:no_cache]
|
|
414
|
+
return if opts[:fetch_db]
|
|
415
|
+
|
|
416
|
+
processor = @registry[stdclass]
|
|
417
|
+
bib = net_retry(code, year, opts, processor, opts.fetch(:retries, 1))
|
|
418
|
+
return unless bib.respond_to?(:to_xml)
|
|
419
|
+
|
|
420
|
+
entry = bib_entry bib
|
|
421
|
+
item = item_key(bib, stdclass)
|
|
422
|
+
if item
|
|
423
|
+
@semaphore.synchronize do
|
|
424
|
+
[@local_db, @db].compact.each { |c| c.store item, entry }
|
|
425
|
+
end
|
|
364
426
|
end
|
|
427
|
+
bib_retval entry, stdclass
|
|
365
428
|
end
|
|
366
429
|
|
|
367
430
|
def net_retry(code, year, opts, processor, retries)
|
|
@@ -380,19 +443,25 @@ module Relaton
|
|
|
380
443
|
end
|
|
381
444
|
end
|
|
382
445
|
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
446
|
+
# @param entry [String] document XML
|
|
447
|
+
# @return [Date, nil] the published date
|
|
448
|
+
def published_date(entry)
|
|
449
|
+
date_str = Moxml.parse(entry)
|
|
450
|
+
.at_xpath("//date[@type='published']/on")&.text
|
|
451
|
+
date_str && parse_pub_date(date_str)
|
|
452
|
+
end
|
|
387
453
|
|
|
388
|
-
|
|
454
|
+
def pub_date_in_range?(entry, opts) # rubocop:disable Metrics/CyclomaticComplexity
|
|
455
|
+
date = published_date(entry)
|
|
389
456
|
return false unless date
|
|
390
457
|
|
|
458
|
+
# `parse_pub_date`, not `Date.parse`: a bound may be "YYYY" or
|
|
459
|
+
# "YYYY-MM", which `Date.parse` rejects.
|
|
391
460
|
after = opts[:publication_date_after]
|
|
392
|
-
return false if after && date <
|
|
461
|
+
return false if after && date < parse_pub_date(after.to_s)
|
|
393
462
|
|
|
394
463
|
before = opts[:publication_date_before]
|
|
395
|
-
return false if before && date >=
|
|
464
|
+
return false if before && date >= parse_pub_date(before.to_s)
|
|
396
465
|
|
|
397
466
|
true
|
|
398
467
|
end
|
|
@@ -407,18 +476,8 @@ module Relaton
|
|
|
407
476
|
nil
|
|
408
477
|
end
|
|
409
478
|
|
|
410
|
-
def open_cache_biblio(dir)
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
db = Cache.new dir
|
|
414
|
-
|
|
415
|
-
Dir["#{dir}/*/"].each do |fdir|
|
|
416
|
-
next if db.check_version?(fdir)
|
|
417
|
-
|
|
418
|
-
FileUtils.rm_rf(fdir, secure: true)
|
|
419
|
-
Util.info "cache #{fdir}: version is obsolete and cache is cleared."
|
|
420
|
-
end
|
|
421
|
-
db
|
|
479
|
+
def open_cache_biblio(dir)
|
|
480
|
+
dir && Cache.new(dir)
|
|
422
481
|
end
|
|
423
482
|
|
|
424
483
|
def process_queue(qwp)
|
data/lib/relaton/doi/crossref.rb
CHANGED
|
@@ -10,25 +10,42 @@ module Relaton
|
|
|
10
10
|
#
|
|
11
11
|
# Get a document by DOI from the CrossRef API.
|
|
12
12
|
#
|
|
13
|
-
# @param [String] doi The DOI.
|
|
13
|
+
# @param [String, Pubid::Doi::Identifier] doi The DOI.
|
|
14
14
|
#
|
|
15
15
|
# @return [RelatonBib::BibliographicItem, RelatonIetf::IetfBibliographicItem,
|
|
16
16
|
# RelatonBipm::BipmBibliographicItem, RelatonIeee::IeeeBibliographicItem,
|
|
17
17
|
# RelatonNist::NistBibliographicItem] The bibitem.
|
|
18
18
|
#
|
|
19
19
|
def get(doi)
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
message = get_by_id
|
|
20
|
+
key = doi.to_s
|
|
21
|
+
Util.info "Fetching from search.crossref.org ...", key: key
|
|
22
|
+
message = get_by_id doi_of(doi)
|
|
23
23
|
if message
|
|
24
|
-
Util.info "Found: `#{message['DOI']}`", key:
|
|
24
|
+
Util.info "Found: `#{message['DOI']}`", key: key
|
|
25
25
|
Parser.parse message
|
|
26
26
|
else
|
|
27
|
-
Util.info "Not found.", key:
|
|
27
|
+
Util.info "Not found.", key: key
|
|
28
28
|
nil
|
|
29
29
|
end
|
|
30
30
|
end
|
|
31
31
|
|
|
32
|
+
#
|
|
33
|
+
# The DOI itself (`<prefix>/<suffix>`), as the Crossref API takes it.
|
|
34
|
+
#
|
|
35
|
+
# A parsed pubid gives it from its components. A String may carry a
|
|
36
|
+
# `doi:` scheme (in any case) or a `doi.org` URL: Relaton::Db keys every
|
|
37
|
+
# form Pubid::Doi reads as one DOI, so all of them must reach the same
|
|
38
|
+
# DOI here, or a miss on one form is cached for the others.
|
|
39
|
+
#
|
|
40
|
+
# @param [String, Pubid::Doi::Identifier] doi
|
|
41
|
+
# @return [String]
|
|
42
|
+
#
|
|
43
|
+
def doi_of(doi)
|
|
44
|
+
return "#{doi.prefix}/#{doi.suffix}" unless doi.is_a?(String)
|
|
45
|
+
|
|
46
|
+
doi.sub(%r{\A(?:doi:|https?://(?:dx\.)?doi\.org/)}i, "")
|
|
47
|
+
end
|
|
48
|
+
|
|
32
49
|
#
|
|
33
50
|
# Get a document by DOI from the CrossRef API.
|
|
34
51
|
#
|
|
@@ -128,7 +128,7 @@ module Relaton
|
|
|
128
128
|
def to_yaml(bib) = bib.to_yaml
|
|
129
129
|
def to_bibxml(bib) = bib.to_rfcxml
|
|
130
130
|
|
|
131
|
-
# @param hit [
|
|
131
|
+
# @param hit [Moxml::Element]
|
|
132
132
|
def parse_page(hit) # rubocop:disable Metrics/AbcSize, Metrics/MethodLength
|
|
133
133
|
DataParser.new(hit, @errors).parse.each { |item| write_file item }
|
|
134
134
|
end
|
|
@@ -21,7 +21,7 @@ module Relaton
|
|
|
21
21
|
docid = @bib[:docidentifier]
|
|
22
22
|
@doc.xpath('//div[@id="main"]/div[1]/div/main/article/div/div/standard/div/ul/li').map do |hit|
|
|
23
23
|
bib = @bib.dup
|
|
24
|
-
id, ed, bib[:date], vol = edition_id_parts hit.
|
|
24
|
+
id, ed, bib[:date], vol = edition_id_parts hit.at_xpath("./span|./a").text
|
|
25
25
|
bib[:source] = edition_source(hit) + edition_translation_source(ed)
|
|
26
26
|
next if ed.nil? || ed.empty?
|
|
27
27
|
|
|
@@ -56,7 +56,7 @@ module Relaton
|
|
|
56
56
|
end
|
|
57
57
|
|
|
58
58
|
def edition_source(hit)
|
|
59
|
-
es = { "src" => hit.
|
|
59
|
+
es = { "src" => hit.at_xpath("./a"), "pdf" => hit.at_xpath("./span/a") }.map do |type, a|
|
|
60
60
|
Bib::Uri.new(type: type, content: a[:href]) if a
|
|
61
61
|
end.compact
|
|
62
62
|
@errors[:edition_source] &&= es.empty?
|
|
@@ -5,7 +5,7 @@ module Relaton
|
|
|
5
5
|
|
|
6
6
|
ATTRS = %i[docidentifier title date source ext].freeze
|
|
7
7
|
|
|
8
|
-
# @param [
|
|
8
|
+
# @param [Moxml::Element] hit document hit
|
|
9
9
|
# @param [Hash] errors error tracking hash
|
|
10
10
|
def initialize(hit:, errors: {})
|
|
11
11
|
@hit = hit
|
|
@@ -23,7 +23,7 @@ module Relaton
|
|
|
23
23
|
|
|
24
24
|
# @return [Array<Relaton::Ecma::Docidentifier>]
|
|
25
25
|
def fetch_docidentifier
|
|
26
|
-
code = "ECMA MEM/#{@hit.
|
|
26
|
+
code = "ECMA MEM/#{@hit.at_xpath('div[1]//p').text}"
|
|
27
27
|
docid = super(code)
|
|
28
28
|
@errors[:memento_docidentifier] &&= docid.empty?
|
|
29
29
|
docid
|
|
@@ -31,7 +31,7 @@ module Relaton
|
|
|
31
31
|
|
|
32
32
|
# @return [Array<Relaton::Bib::Title>]
|
|
33
33
|
def fetch_title
|
|
34
|
-
year = @hit.
|
|
34
|
+
year = @hit.at_xpath("div[1]//p").text
|
|
35
35
|
content = "\"Memento #{year}\" for year #{year}"
|
|
36
36
|
result = [Bib::Title.new(content: content, language: "en", script: "Latn")]
|
|
37
37
|
@errors[:memento_title] &&= result.empty?
|
|
@@ -40,7 +40,7 @@ module Relaton
|
|
|
40
40
|
|
|
41
41
|
# @return [Array<Relaton::Bib::Date>]
|
|
42
42
|
def fetch_date
|
|
43
|
-
date = @hit.
|
|
43
|
+
date = @hit.at_xpath("div[2]//p").text
|
|
44
44
|
on = Date.strptime(date, "%B %Y").strftime "%Y-%m"
|
|
45
45
|
result = [Bib::Date.new(type: "published", at: on)]
|
|
46
46
|
@errors[:memento_date] &&= result.empty?
|
|
@@ -5,7 +5,7 @@ module Relaton
|
|
|
5
5
|
|
|
6
6
|
ATTRS = %i[docidentifier title date source abstract relation edition ext].freeze
|
|
7
7
|
|
|
8
|
-
# @param [
|
|
8
|
+
# @param [Moxml::Element] hit document hit
|
|
9
9
|
# @param [Mechanize::Page] doc fetched document page
|
|
10
10
|
# @param [Hash] errors error tracking hash
|
|
11
11
|
def initialize(hit:, doc:, errors: {})
|
|
@@ -68,7 +68,7 @@ module Relaton
|
|
|
68
68
|
def fetch_source # rubocop:disable Metrics/AbcSize
|
|
69
69
|
source = []
|
|
70
70
|
source << Bib::Uri.new(type: "src", content: @hit[:href]) if @hit[:href]
|
|
71
|
-
ref = @doc.
|
|
71
|
+
ref = @doc.at_xpath('//div[@class="ecma-item-content-wrapper"]/span/a',
|
|
72
72
|
'//div[@class="ecma-item-content-wrapper"]/a')
|
|
73
73
|
source << Bib::Uri.new(type: "pdf", content: ref[:href]) if ref
|
|
74
74
|
result = source + edition_translation_source(fetch_edition_content)
|
|
@@ -80,7 +80,7 @@ module Relaton
|
|
|
80
80
|
def fetch_relation # rubocop:disable Metrics/AbcSize, Metrics/MethodLength, Metrics/CyclomaticComplexity
|
|
81
81
|
edition_parser = EditionParser.new(doc: @doc, bib: {}, errors: @errors)
|
|
82
82
|
result = @doc.xpath("//ul[@class='ecma-item-archives']/li").filter_map do |rel|
|
|
83
|
-
ref, ed, date, vol = edition_parser.edition_id_parts rel.
|
|
83
|
+
ref, ed, date, vol = edition_parser.edition_id_parts rel.at_xpath("span").text
|
|
84
84
|
next if ed.nil? || ed.empty?
|
|
85
85
|
|
|
86
86
|
docid = Docidentifier.new(type: "ECMA", content: ref, primary: true)
|
|
@@ -109,7 +109,7 @@ module Relaton
|
|
|
109
109
|
private
|
|
110
110
|
|
|
111
111
|
def fetch_edition_content
|
|
112
|
-
@doc.
|
|
112
|
+
@doc.at_xpath('//p[@class="ecma-item-edition"]')&.text&.match(/^\d+(?=(?:st|nd|th|rd))/)&.to_s
|
|
113
113
|
end
|
|
114
114
|
|
|
115
115
|
def edition_translation_source(edition)
|
|
@@ -120,8 +120,8 @@ module Relaton
|
|
|
120
120
|
return [] unless @doc
|
|
121
121
|
|
|
122
122
|
@doc.xpath("//h2[.='Translations']/following-sibling::ul/li").map do |l|
|
|
123
|
-
a = l.
|
|
124
|
-
id = l.
|
|
123
|
+
a = l.at_xpath("span/a")
|
|
124
|
+
id = l.at_xpath("span").text
|
|
125
125
|
%r{\w+[\d-]+,\s(?<lang>\w+)\sversion,\s(?<ed>[\d.]+)(?:st|nd|rd|th)\sedition} =~ id
|
|
126
126
|
case lang
|
|
127
127
|
when "Japanese"
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# encoding: UTF-8
|
|
2
2
|
# frozen_string_literal: true
|
|
3
3
|
|
|
4
|
-
require "
|
|
4
|
+
require "moxml"
|
|
5
5
|
require_relative "scraper"
|
|
6
6
|
|
|
7
7
|
module Relaton
|
|
@@ -20,10 +20,10 @@ module Relaton
|
|
|
20
20
|
hits = doc.xpath(
|
|
21
21
|
"//table[contains(@class, 'result_list')]/tbody[2]/tr",
|
|
22
22
|
).map do |h|
|
|
23
|
-
ref = h.
|
|
23
|
+
ref = h.at_xpath "./td[2]/a"
|
|
24
24
|
pid = ref[:onclick].match(/[0-9A-F]+/).to_s
|
|
25
|
-
status = h.
|
|
26
|
-
rdate = h.
|
|
25
|
+
status = h.at_xpath("./td[7]").text.strip
|
|
26
|
+
rdate = h.at_xpath("./td[8]").text.strip
|
|
27
27
|
Hit.new pid: pid, docref: ref.text, scraper: self,
|
|
28
28
|
release_date: rdate, status: status
|
|
29
29
|
end
|
|
@@ -52,7 +52,7 @@ module Relaton
|
|
|
52
52
|
# * :type [String]
|
|
53
53
|
# * :name [String]
|
|
54
54
|
# def get_committee(doc, _ref)
|
|
55
|
-
# name = doc.
|
|
55
|
+
# name = doc.at_xpath("//div[contains(., '归口单位') or contains(., '归口部门')]/following-sibling::div")
|
|
56
56
|
# Committee.new(type: "technical", content: name.text.delete("\r\n\t\t"))
|
|
57
57
|
# end
|
|
58
58
|
end
|