relaton 3.0.0.pre.alpha.5 → 3.0.0.pre.alpha.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/lib/relaton/3gpp/bibliography.rb +19 -6
- data/lib/relaton/3gpp/processor.rb +6 -0
- data/lib/relaton/bib/item_data.rb +14 -0
- data/lib/relaton/bipm/bibliography.rb +24 -21
- data/lib/relaton/bsi/bibliography.rb +26 -23
- data/lib/relaton/calconnect/bibliography.rb +6 -6
- data/lib/relaton/calconnect/hit_collection.rb +4 -1
- data/lib/relaton/ccsds/bibliography.rb +15 -10
- data/lib/relaton/ccsds/hit_collection.rb +1 -1
- data/lib/relaton/ccsds/processor.rb +3 -2
- data/lib/relaton/cen/bibliography.rb +27 -24
- data/lib/relaton/cie/bibliography.rb +11 -9
- data/lib/relaton/cie/scrapper.rb +15 -8
- data/lib/relaton/core/processor.rb +61 -8
- data/lib/relaton/core/request_error.rb +21 -3
- data/lib/relaton/db/cache.rb +49 -19
- data/lib/relaton/db/cache_entry.rb +9 -0
- data/lib/relaton/db/registry.rb +128 -28
- data/lib/relaton/db.rb +162 -48
- data/lib/relaton/ecma/bibliography.rb +18 -13
- data/lib/relaton/etsi/bibliography.rb +8 -7
- data/lib/relaton/etsi/data_fetcher.rb +118 -9
- data/lib/relaton/gb/bibliography.rb +20 -12
- data/lib/relaton/iala/bibliography.rb +15 -12
- data/lib/relaton/iana/bibliography.rb +12 -12
- data/lib/relaton/iana/parser.rb +5 -2
- data/lib/relaton/iec/bibliography.rb +18 -7
- data/lib/relaton/iec/processor.rb +6 -1
- data/lib/relaton/ieee/bibliography.rb +19 -13
- data/lib/relaton/ieee/processor.rb +16 -0
- data/lib/relaton/ieee/rawbib_id_parser.rb +2 -2
- data/lib/relaton/ietf/bibliography.rb +11 -9
- data/lib/relaton/ietf/scraper.rb +5 -4
- data/lib/relaton/index/config.rb +9 -1
- data/lib/relaton/index/file_io.rb +190 -0
- data/lib/relaton/index/type.rb +96 -23
- data/lib/relaton/iso/bibliography.rb +11 -9
- data/lib/relaton/itu/bibliography.rb +9 -4
- data/lib/relaton/jis/bibliography.rb +22 -11
- data/lib/relaton/nist/bibliography.rb +25 -21
- data/lib/relaton/oasis/bibliography.rb +16 -13
- data/lib/relaton/ogc/bibliography.rb +15 -14
- data/lib/relaton/ogc/hit_collection.rb +9 -7
- data/lib/relaton/ogc/processor.rb +7 -1
- data/lib/relaton/omg/bibliography.rb +9 -9
- data/lib/relaton/omg/scraper.rb +3 -2
- data/lib/relaton/plateau/bibliography.rb +9 -7
- data/lib/relaton/plateau/hit_collection.rb +2 -1
- data/lib/relaton/version.rb +1 -1
- data/lib/relaton/xsf/bibliography.rb +11 -6
- metadata +18 -4
|
@@ -11,6 +11,10 @@ module Relaton
|
|
|
11
11
|
# wrong-structure handling (re-download, or stop and log).
|
|
12
12
|
class InvalidIndexError < StandardError; end
|
|
13
13
|
|
|
14
|
+
# Bump when the sidecar payload shape changes, so an older sidecar is
|
|
15
|
+
# discarded and rebuilt instead of misread.
|
|
16
|
+
SIDECAR_VERSION = 2 # v2: the payload carries the yaml byte-size for freshness
|
|
17
|
+
|
|
14
18
|
attr_reader :url, :pubid_class
|
|
15
19
|
attr_accessor :sorted
|
|
16
20
|
|
|
@@ -288,11 +292,197 @@ module Relaton
|
|
|
288
292
|
#
|
|
289
293
|
def remove
|
|
290
294
|
Index.config.storage.remove file
|
|
295
|
+
delete_sidecar
|
|
291
296
|
[]
|
|
292
297
|
end
|
|
293
298
|
|
|
299
|
+
#
|
|
300
|
+
# Raw-row read for the lazy search path (relaton#242 stopgap): returns
|
|
301
|
+
# precomputed root-number sort keys and the rows as plain hashes — no
|
|
302
|
+
# pubid objects. A Marshal sidecar next to the yaml holds the keys+rows
|
|
303
|
+
# so repeat loads skip the YAML parse and the one-time full
|
|
304
|
+
# materialization; the sidecar is rebuilt whenever the yaml is newer.
|
|
305
|
+
#
|
|
306
|
+
# @return [Array<Array<String>, Array<Hash>] sort keys and raw rows
|
|
307
|
+
#
|
|
308
|
+
def read_raw
|
|
309
|
+
case url
|
|
310
|
+
when String
|
|
311
|
+
with_file_lock do
|
|
312
|
+
check_file ? read_raw_file : fetch_raw_and_save
|
|
313
|
+
end
|
|
314
|
+
else
|
|
315
|
+
read_raw_file || [[], []]
|
|
316
|
+
end
|
|
317
|
+
end
|
|
318
|
+
|
|
319
|
+
# Deserialize raw rows into pubid rows. Only the caller knows which
|
|
320
|
+
# slice it needs, so materialization happens here rather than at load.
|
|
321
|
+
#
|
|
322
|
+
# @param [Array<Hash>] rows raw rows
|
|
323
|
+
# @return [Array<Hash>] rows with deserialized ids
|
|
324
|
+
def materialize(rows)
|
|
325
|
+
return rows unless @pubid_class
|
|
326
|
+
|
|
327
|
+
rows.map { |r| { id: deserialize_id(r[:id]), file: r[:file] } }
|
|
328
|
+
end
|
|
329
|
+
|
|
330
|
+
# Save raw rows (possibly alongside the keys they were loaded with);
|
|
331
|
+
# sorts by the same root-number key the object path sorts by.
|
|
332
|
+
#
|
|
333
|
+
# @param [Array<String>] keys sort keys, parallel to rows
|
|
334
|
+
# @param [Array<Hash>] rows raw rows
|
|
335
|
+
# @return [void]
|
|
336
|
+
def save_raw(keys, rows)
|
|
337
|
+
ordered = @pubid_class ? keys.zip(rows).sort_by { |k, _| k.to_s }
|
|
338
|
+
.map { |_, r| r } : rows
|
|
339
|
+
yaml = ordered.map do |item|
|
|
340
|
+
{ id: item[:id], file: item[:file] }
|
|
341
|
+
end.to_yaml
|
|
342
|
+
Index.config.storage.write file, yaml
|
|
343
|
+
delete_sidecar
|
|
344
|
+
end
|
|
345
|
+
|
|
294
346
|
private
|
|
295
347
|
|
|
348
|
+
def sidecar_file
|
|
349
|
+
"#{file}.ms"
|
|
350
|
+
end
|
|
351
|
+
|
|
352
|
+
def delete_sidecar
|
|
353
|
+
File.delete(sidecar_file) if File.file?(sidecar_file)
|
|
354
|
+
rescue Errno::EACCES
|
|
355
|
+
nil
|
|
356
|
+
end
|
|
357
|
+
|
|
358
|
+
def read_raw_file
|
|
359
|
+
# The sidecar is authoritative while it describes the yaml byte-for-byte
|
|
360
|
+
# (same size) — the yaml is not even parsed until the sidecar is stale
|
|
361
|
+
# or unreadable.
|
|
362
|
+
if sidecar_fresh?
|
|
363
|
+
loaded = sidecar
|
|
364
|
+
return loaded if loaded
|
|
365
|
+
end
|
|
366
|
+
|
|
367
|
+
yaml = Index.config.storage.read(file)
|
|
368
|
+
return unless yaml
|
|
369
|
+
|
|
370
|
+
begin
|
|
371
|
+
raw = YAML.safe_load(yaml, permitted_classes: [Symbol])
|
|
372
|
+
rescue Psych::SyntaxError
|
|
373
|
+
warn_local_index_error("YAML parsing error when reading")
|
|
374
|
+
delete_sidecar
|
|
375
|
+
return [[], []]
|
|
376
|
+
end
|
|
377
|
+
build_raw(raw)
|
|
378
|
+
end
|
|
379
|
+
|
|
380
|
+
# Size, not mtime: NTFS timestamp coarseness makes a rewritten yaml
|
|
381
|
+
# carry the sidecar's own mtime, so "yaml is newer" misses the rebuild.
|
|
382
|
+
def sidecar_fresh?
|
|
383
|
+
File.file?(sidecar_file) && File.file?(file) &&
|
|
384
|
+
File.size(file) == sidecar_yaml_size
|
|
385
|
+
end
|
|
386
|
+
|
|
387
|
+
def sidecar_yaml_size
|
|
388
|
+
Marshal.load(File.binread(sidecar_file))[4]
|
|
389
|
+
rescue TypeError, ArgumentError, EOFError
|
|
390
|
+
nil
|
|
391
|
+
end
|
|
392
|
+
|
|
393
|
+
def sidecar
|
|
394
|
+
version, sorted, keys, rows = Marshal.load(File.binread(sidecar_file))
|
|
395
|
+
return unless version == SIDECAR_VERSION
|
|
396
|
+
|
|
397
|
+
@sorted = sorted
|
|
398
|
+
[keys, rows]
|
|
399
|
+
rescue TypeError, ArgumentError, EOFError
|
|
400
|
+
nil
|
|
401
|
+
end
|
|
402
|
+
|
|
403
|
+
def build_raw(raw)
|
|
404
|
+
unless check_format(raw)
|
|
405
|
+
warn_local_index_error("Wrong structure of")
|
|
406
|
+
return [[], []]
|
|
407
|
+
end
|
|
408
|
+
|
|
409
|
+
return build_raw_in_child(raw) if build_sidecar_in_child?
|
|
410
|
+
|
|
411
|
+
build_raw_in_process(raw)
|
|
412
|
+
rescue InvalidIndexError
|
|
413
|
+
warn_local_index_error("Wrong structure of")
|
|
414
|
+
[[], []]
|
|
415
|
+
end
|
|
416
|
+
|
|
417
|
+
# The one-time key build materializes every row (~675 MB on the
|
|
418
|
+
# 79,993-row ISO index). Freed pages are often not returned to the OS,
|
|
419
|
+
# so an in-process build leaves the parent's RSS high — and containers
|
|
420
|
+
# OOM on RSS (relaton#242). Where fork exists, build the sidecar in a
|
|
421
|
+
# child: the graph dies with it and the parent never spikes.
|
|
422
|
+
def build_sidecar_in_child?
|
|
423
|
+
Process.respond_to?(:fork) && Index.config.build_sidecar_in_child != false
|
|
424
|
+
end
|
|
425
|
+
|
|
426
|
+
def build_raw_in_child(raw)
|
|
427
|
+
pid = Process.fork do
|
|
428
|
+
build_raw_in_process(raw)
|
|
429
|
+
exit!(0)
|
|
430
|
+
end
|
|
431
|
+
Process.wait(pid)
|
|
432
|
+
loaded = sidecar
|
|
433
|
+
return build_raw_in_process(raw) unless loaded # child failed
|
|
434
|
+
|
|
435
|
+
loaded
|
|
436
|
+
rescue Errno::ENOMEM, SystemCallError
|
|
437
|
+
build_raw_in_process(raw)
|
|
438
|
+
end
|
|
439
|
+
|
|
440
|
+
def build_raw_in_process(raw)
|
|
441
|
+
objects = deserialize_pubid(raw)
|
|
442
|
+
keys = objects.map { |r| r[:id].root.number.to_s }
|
|
443
|
+
rows = objects.map { |r| { id: raw_id(r), file: r[:file] } }
|
|
444
|
+
write_sidecar(keys, rows)
|
|
445
|
+
[keys, rows]
|
|
446
|
+
rescue InvalidIndexError
|
|
447
|
+
warn_local_index_error("Wrong structure of")
|
|
448
|
+
[[], []]
|
|
449
|
+
end
|
|
450
|
+
|
|
451
|
+
def raw_id(row)
|
|
452
|
+
id = row[:id]
|
|
453
|
+
id.respond_to?(:to_hash) ? id.to_hash : id
|
|
454
|
+
end
|
|
455
|
+
|
|
456
|
+
def write_sidecar(keys, rows)
|
|
457
|
+
File.binwrite(sidecar_file,
|
|
458
|
+
Marshal.dump([SIDECAR_VERSION, @sorted, keys, rows,
|
|
459
|
+
File.size(file)]))
|
|
460
|
+
rescue Errno::EACCES, Errno::ENOENT, Errno::EROFS
|
|
461
|
+
nil # the sidecar is an optimization; a read-only dir still works
|
|
462
|
+
end
|
|
463
|
+
|
|
464
|
+
def fetch_raw_and_save
|
|
465
|
+
uri = URI.parse(url)
|
|
466
|
+
body = Net::HTTP.get(uri)
|
|
467
|
+
yaml = nil
|
|
468
|
+
Zip::File.open_buffer(body) do |zip|
|
|
469
|
+
yaml = zip.entries.first.get_input_stream.read
|
|
470
|
+
end
|
|
471
|
+
Util.info "Downloaded index from `#{url}`", progname
|
|
472
|
+
raw = YAML.safe_load(yaml, permitted_classes: [Symbol])
|
|
473
|
+
if check_format(raw)
|
|
474
|
+
save raw
|
|
475
|
+
raw = nil # release the parsed copy; read_raw_file loads the sidecar's
|
|
476
|
+
read_raw_file
|
|
477
|
+
else
|
|
478
|
+
warn_remote_index_error "Wrong structure of"
|
|
479
|
+
[[], []]
|
|
480
|
+
end
|
|
481
|
+
rescue Psych::SyntaxError
|
|
482
|
+
warn_remote_index_error "YAML parsing error when reading"
|
|
483
|
+
[[], []]
|
|
484
|
+
end
|
|
485
|
+
|
|
296
486
|
def with_file_lock(&)
|
|
297
487
|
@@file_locks_mutex.synchronize do
|
|
298
488
|
@@file_locks[file] ||= Mutex.new
|
data/lib/relaton/index/type.rb
CHANGED
|
@@ -30,7 +30,24 @@ module Relaton
|
|
|
30
30
|
def index
|
|
31
31
|
return @source.whole_index if @source
|
|
32
32
|
|
|
33
|
-
@index ||= @file_io.
|
|
33
|
+
@index ||= @file_io.materialize(raw_index[1])
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
# Raw rows plus their precomputed root-number sort keys (relaton#242
|
|
37
|
+
# stopgap). The lazy search path materializes pubid objects only for
|
|
38
|
+
# the bucket it narrows to, so a process that answers one lookup holds
|
|
39
|
+
# raw hashes instead of the whole pubid graph. A caller-seeded `@index`
|
|
40
|
+
# (a spec fixture, a preloaded pool entry) is the source of truth: the
|
|
41
|
+
# raw side derives from it, in its order, so a seeder that sorts by the
|
|
42
|
+
# narrowing key keeps the bsearch valid.
|
|
43
|
+
def raw_index
|
|
44
|
+
@raw_index ||= if @index
|
|
45
|
+
keys = @index.map { |r| narrowing_key(r[:id]) }
|
|
46
|
+
rows = @index.map { |r| { id: raw_row_id(r[:id]), file: r[:file] } }
|
|
47
|
+
[keys, rows]
|
|
48
|
+
else
|
|
49
|
+
@file_io.read_raw
|
|
50
|
+
end
|
|
34
51
|
end
|
|
35
52
|
|
|
36
53
|
#
|
|
@@ -59,14 +76,23 @@ module Relaton
|
|
|
59
76
|
# @return [void]
|
|
60
77
|
#
|
|
61
78
|
def add_or_update(id, file)
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
79
|
+
# A seeded or materialized @index (spec fixtures, preloaded pool
|
|
80
|
+
# entries) and pubid-less indexes run the legacy object path
|
|
81
|
+
# bit-for-bit: their consumers match rows by object identity
|
|
82
|
+
# semantics (`==`, `exclude`, to_s dedup) that raw rows cannot
|
|
83
|
+
# reproduce. Only a pubid-backed FileIO-loaded index — the
|
|
84
|
+
# relaton#242 memory case — adds through raw rows.
|
|
85
|
+
return legacy_add(id, file) if @index || !@file_io.pubid_class
|
|
86
|
+
|
|
87
|
+
raw_hash = raw_row_id(id)
|
|
88
|
+
keys, rows = raw_index
|
|
89
|
+
pos = raw_position(raw_hash)
|
|
90
|
+
if pos
|
|
91
|
+
rows[pos] = { id: raw_hash, file: file }
|
|
66
92
|
else
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
93
|
+
rows << { id: raw_hash, file: file }
|
|
94
|
+
keys << narrowing_key(id)
|
|
95
|
+
@raw_lookup[raw_hash] = rows.size - 1
|
|
70
96
|
@file_io.sorted = false
|
|
71
97
|
end
|
|
72
98
|
end
|
|
@@ -110,7 +136,12 @@ module Relaton
|
|
|
110
136
|
# @return [void]
|
|
111
137
|
#
|
|
112
138
|
def save
|
|
113
|
-
@
|
|
139
|
+
if @index
|
|
140
|
+
@file_io.save(@index)
|
|
141
|
+
else
|
|
142
|
+
keys, rows = raw_index
|
|
143
|
+
@file_io.save_raw(keys, rows)
|
|
144
|
+
end
|
|
114
145
|
end
|
|
115
146
|
|
|
116
147
|
#
|
|
@@ -125,6 +156,8 @@ module Relaton
|
|
|
125
156
|
@source = new_source
|
|
126
157
|
@index = nil
|
|
127
158
|
@id_lookup = nil
|
|
159
|
+
@raw_index = nil
|
|
160
|
+
@raw_lookup = nil
|
|
128
161
|
end
|
|
129
162
|
|
|
130
163
|
#
|
|
@@ -135,6 +168,8 @@ module Relaton
|
|
|
135
168
|
def remove_all
|
|
136
169
|
@index = []
|
|
137
170
|
@id_lookup = nil
|
|
171
|
+
@raw_index = [[], []]
|
|
172
|
+
@raw_lookup = {}
|
|
138
173
|
@file_io.sorted = true
|
|
139
174
|
end
|
|
140
175
|
|
|
@@ -146,6 +181,44 @@ module Relaton
|
|
|
146
181
|
end
|
|
147
182
|
end
|
|
148
183
|
|
|
184
|
+
# Position of a raw row by its id hash — the raw-side equivalent of
|
|
185
|
+
# `id_lookup`, built once, so a crawl's add_or_update stays O(1) per row.
|
|
186
|
+
def raw_position(raw_hash)
|
|
187
|
+
@raw_lookup ||= raw_index[1].each_with_object({}) do |row, h|
|
|
188
|
+
h[row[:id]] = h.key?(row[:id]) ? h[row[:id]] : rows_index_of(h, row)
|
|
189
|
+
end
|
|
190
|
+
@raw_lookup[raw_hash]
|
|
191
|
+
end
|
|
192
|
+
|
|
193
|
+
def rows_index_of(lookup, row)
|
|
194
|
+
raw_index[1].index { |r| r[:id].equal?(row[:id]) || r[:id] == row[:id] }
|
|
195
|
+
end
|
|
196
|
+
# The pre-#242 add path, kept verbatim for seeded/materialized and
|
|
197
|
+
# pubid-less indexes.
|
|
198
|
+
def legacy_add(id, file)
|
|
199
|
+
key = id.to_s
|
|
200
|
+
item = id_lookup[key]
|
|
201
|
+
if item
|
|
202
|
+
item[:file] = file
|
|
203
|
+
else
|
|
204
|
+
new_item = { id: id, file: file }
|
|
205
|
+
index << new_item
|
|
206
|
+
id_lookup[key] = new_item
|
|
207
|
+
@file_io.sorted = false
|
|
208
|
+
end
|
|
209
|
+
end
|
|
210
|
+
|
|
211
|
+
# Same key expression the FileIO sidecar computes at build time; a
|
|
212
|
+
# String id (a no-pubid_class index) has no `.root`, and never narrows.
|
|
213
|
+
def narrowing_key(id)
|
|
214
|
+
id.respond_to?(:root) && id.root.respond_to?(:number) ? id.root.number.to_s : ""
|
|
215
|
+
end
|
|
216
|
+
|
|
217
|
+
def raw_row_id(id)
|
|
218
|
+
pc = @file_io.pubid_class
|
|
219
|
+
pc && id.is_a?(pc) ? id.to_hash : id
|
|
220
|
+
end
|
|
221
|
+
|
|
149
222
|
def new_source
|
|
150
223
|
ShardSource.new(@dir, @pages_url, @pubid_class) if @pages_url
|
|
151
224
|
end
|
|
@@ -155,13 +228,15 @@ module Relaton
|
|
|
155
228
|
# never refreshed, so the shards stay the fresher answer.
|
|
156
229
|
return @source.rows(id) if @source && id && !id.is_a?(String)
|
|
157
230
|
|
|
158
|
-
#
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
231
|
+
# The lazy path narrows on the precomputed keys and materializes only
|
|
232
|
+
# the bucket; the materialized `index` stays available to callers
|
|
233
|
+
# that need the whole graph (`#index`, a String query, a block).
|
|
234
|
+
# Load first: a pubid-backed load is what settles @file_io.sorted.
|
|
235
|
+
if id && !id.is_a?(String) && @file_io.pubid_class
|
|
236
|
+
raw_index
|
|
237
|
+
return candidates_by_number(id) if @file_io.sorted
|
|
164
238
|
end
|
|
239
|
+
index
|
|
165
240
|
end
|
|
166
241
|
|
|
167
242
|
# Narrowing key: the base *document's* number as a string. `#root` walks a
|
|
@@ -169,26 +244,24 @@ module Relaton
|
|
|
169
244
|
# returns self for a base document), so a document and all its wrappers
|
|
170
245
|
# share one key and cluster together. `.to_s` because the key is compared
|
|
171
246
|
# as a string (a pubid number Component is not `<`/`>`-comparable), and
|
|
172
|
-
# FileIO
|
|
247
|
+
# FileIO computes this exact same key once at sidecar-build time so the
|
|
248
|
+
# bsearch over the keys array stays valid.
|
|
173
249
|
def candidates_by_number(id)
|
|
174
250
|
target = id.root.number.to_s
|
|
251
|
+
keys, rows = raw_index
|
|
175
252
|
left = bsearch_left(target)
|
|
176
253
|
return [] unless left
|
|
177
254
|
|
|
178
255
|
right = bsearch_right(target)
|
|
179
|
-
|
|
256
|
+
@file_io.materialize(rows[left...right])
|
|
180
257
|
end
|
|
181
258
|
|
|
182
259
|
def bsearch_left(target)
|
|
183
|
-
|
|
184
|
-
item[:id].root.number.to_s >= target
|
|
185
|
-
end
|
|
260
|
+
raw_index[0].bsearch_index { |k| k >= target }
|
|
186
261
|
end
|
|
187
262
|
|
|
188
263
|
def bsearch_right(target)
|
|
189
|
-
|
|
190
|
-
item[:id].root.number.to_s > target
|
|
191
|
-
end || index.size
|
|
264
|
+
raw_index[0].bsearch_index { |k| k > target } || raw_index[0].size
|
|
192
265
|
end
|
|
193
266
|
|
|
194
267
|
# The query is the reference, so two identifiers take the subset match.
|
|
@@ -24,7 +24,9 @@ module Relaton
|
|
|
24
24
|
raise Relaton::RequestError, e.message
|
|
25
25
|
end
|
|
26
26
|
|
|
27
|
-
# @param ref [String] the ISO standard Code to
|
|
27
|
+
# @param ref [String, Pubid::Iso::Identifier] the ISO standard Code to
|
|
28
|
+
# look up (e..g "ISO 9000"), or the parse that Relaton::Db routed with
|
|
29
|
+
# (relaton#205); a pubid is never mutated
|
|
28
30
|
# @param year [String, NilClass] the year the standard was published
|
|
29
31
|
# @param opts [Hash] options; restricted to :all_parts if all-parts
|
|
30
32
|
# @option opts [Boolean] :all_parts if all-parts reference is required
|
|
@@ -33,21 +35,21 @@ module Relaton
|
|
|
33
35
|
#
|
|
34
36
|
# @return [RelatonIsoBib::IsoBibliographicItem] Bibliographic item
|
|
35
37
|
def get(ref, year = nil, opts = {}) # rubocop:disable Metrics/CyclomaticComplexity,Metrics/MethodLength,Metrics/PerceivedComplexity,Metrics/AbcSize
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
query_pubid = ::Pubid::Iso::Identifier.parse(code)
|
|
38
|
+
query_pubid = if ref.is_a?(String)
|
|
39
|
+
::Pubid::Iso::Identifier.parse(ref.gsub("\u2013", "-"))
|
|
40
|
+
else
|
|
41
|
+
ref
|
|
42
|
+
end
|
|
43
43
|
if year&.respond_to?(:to_i)
|
|
44
|
+
# Copy first: a caller's pubid is also its Db cache key.
|
|
45
|
+
query_pubid = query_pubid.class.from_hash(query_pubid.to_hash)
|
|
44
46
|
query_pubid.root.date = ::Pubid::Components::Date.new(year: year.to_s)
|
|
45
47
|
end
|
|
46
48
|
query_pubid = query_pubid.to_all_parts if opts[:all_parts]
|
|
47
49
|
Util.info "Fetching from Relaton repository ...", key: query_pubid.to_s
|
|
48
50
|
|
|
49
51
|
hits, missed_year_ids = isobib_search_filter(query_pubid, opts)
|
|
50
|
-
tip_ids = look_up_with_any_types_stages(hits, ref, opts)
|
|
52
|
+
tip_ids = look_up_with_any_types_stages(hits, ref.to_s, opts)
|
|
51
53
|
|
|
52
54
|
date_filter = opts[:publication_date_before] || opts[:publication_date_after]
|
|
53
55
|
if date_filter && !query_pubid.all_parts
|
|
@@ -31,13 +31,18 @@ module Relaton
|
|
|
31
31
|
HitCollection.new(refid).tap(&:search)
|
|
32
32
|
end
|
|
33
33
|
|
|
34
|
-
# @param
|
|
34
|
+
# @param ref [String, Pubid::Itu::Identifier] the ITU standard Code to
|
|
35
|
+
# look up, or its parse from Relaton::Db (relaton#205); #with_year
|
|
36
|
+
# copies, so the pubid is not changed
|
|
35
37
|
# @param year [String] the year the standard was published (optional)
|
|
36
38
|
# @param opts [Hash] options
|
|
37
39
|
# @return [Relaton::Bib::ItemData, nil]
|
|
38
|
-
def get(
|
|
39
|
-
|
|
40
|
-
|
|
40
|
+
def get(ref, year = nil, opts = {})
|
|
41
|
+
if ref.is_a? String
|
|
42
|
+
warn_incorrect_ref(ref)
|
|
43
|
+
ref = ::Pubid::Itu.parse(ref)
|
|
44
|
+
end
|
|
45
|
+
refid = with_year ref, year
|
|
41
46
|
|
|
42
47
|
ret = itubib_get1(refid)
|
|
43
48
|
return nil if ret.nil?
|
|
@@ -13,7 +13,8 @@ module Relaton
|
|
|
13
13
|
# the reference's series and number (a supplement is filed under its base
|
|
14
14
|
# number), so an edition and its amendments are returned together.
|
|
15
15
|
#
|
|
16
|
-
# @param [String]
|
|
16
|
+
# @param [String, Pubid::Jis::Identifier] ref JIS document reference; a
|
|
17
|
+
# pubid is never mutated
|
|
17
18
|
# @param [String, nil] year JIS document year
|
|
18
19
|
#
|
|
19
20
|
# @return [Relaton::Jis::HitCollection] search result
|
|
@@ -21,16 +22,21 @@ module Relaton
|
|
|
21
22
|
# @raise [Pubid::Errors::ParseError] when the reference is not a JIS
|
|
22
23
|
# identifier, so a malformed reference is not reported as "not found"
|
|
23
24
|
#
|
|
24
|
-
def search(
|
|
25
|
-
pubid = ::Pubid::Jis::Identifier.parse
|
|
26
|
-
|
|
25
|
+
def search(ref, year = nil)
|
|
26
|
+
pubid = ref.is_a?(String) ? ::Pubid::Jis::Identifier.parse(ref) : ref
|
|
27
|
+
if year && pubid.year.nil?
|
|
28
|
+
# Copy first: a caller's pubid is also its Db cache key.
|
|
29
|
+
pubid = pubid.class.from_hash(pubid.to_hash)
|
|
30
|
+
pubid.year = year.to_i
|
|
31
|
+
end
|
|
27
32
|
HitCollection.new pubid
|
|
28
33
|
end
|
|
29
34
|
|
|
30
35
|
#
|
|
31
36
|
# Get JIS document by reference
|
|
32
37
|
#
|
|
33
|
-
# @param [String] ref JIS document reference
|
|
38
|
+
# @param [String, Pubid::Jis::Identifier] ref JIS document reference, or
|
|
39
|
+
# the parse that Relaton::Db routed with (relaton#205)
|
|
34
40
|
# @param [String, nil] year JIS document year
|
|
35
41
|
# @param [Hash] opts options
|
|
36
42
|
# @option opts [Boolean] :all_parts return all parts of document
|
|
@@ -41,16 +47,21 @@ module Relaton
|
|
|
41
47
|
# identifier
|
|
42
48
|
#
|
|
43
49
|
def get(ref, year = nil, opts = {}) # rubocop:disable Metrics/AbcSize
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
50
|
+
if ref.is_a?(String)
|
|
51
|
+
query = ref.sub(/\s\((all parts|規格群)\)/, "")
|
|
52
|
+
opts[:all_parts] ||= !$1.nil?
|
|
53
|
+
else
|
|
54
|
+
opts[:all_parts] ||= ref.all_parts?
|
|
55
|
+
query = ref.all_parts? ? ref.identifiers.first : ref
|
|
56
|
+
end
|
|
57
|
+
Util.info "Fetching from webdesk.jsa.or.jp ...", key: ref.to_s
|
|
58
|
+
hits = search(query, year)
|
|
48
59
|
result = opts[:all_parts] ? hits.find_all_parts : hits.find
|
|
49
60
|
if result.is_a? Bib::ItemData
|
|
50
|
-
Util.info "Found: `#{result.docidentifier[0].content}`", key: ref
|
|
61
|
+
Util.info "Found: `#{result.docidentifier[0].content}`", key: ref.to_s
|
|
51
62
|
return result
|
|
52
63
|
end
|
|
53
|
-
hint result, ref, year
|
|
64
|
+
hint result, ref.to_s, year
|
|
54
65
|
end
|
|
55
66
|
|
|
56
67
|
#
|
|
@@ -7,18 +7,18 @@ module Relaton
|
|
|
7
7
|
#
|
|
8
8
|
# Search NIST documents by reference
|
|
9
9
|
#
|
|
10
|
-
# @param
|
|
10
|
+
# @param ref [String] reference
|
|
11
11
|
#
|
|
12
12
|
# @return [Relaton::Nist::HitCollection] search result
|
|
13
13
|
#
|
|
14
|
-
def search(
|
|
15
|
-
|
|
14
|
+
def search(ref, year = nil, opts = {})
|
|
15
|
+
query = ref.sub(/^NISTIR/, "NIST IR").sub(/\/Add/, " Add")
|
|
16
16
|
# pubid 2.x only recognizes the addendum marker with a trailing
|
|
17
17
|
# period, and only splits an uppercase part letter — canonicalize
|
|
18
18
|
# both ("800-38a Add" -> "800-38A Add.") so @reference parses to the
|
|
19
19
|
# same pubid the index/CSRC carry.
|
|
20
|
-
|
|
21
|
-
HitCollection.search
|
|
20
|
+
query = query.sub(/\bAdd\b\.?/i, "Add.").sub(/([0-9])([a-z])(?=\s+Add\.)/) { "#{$1}#{$2.upcase}" }
|
|
21
|
+
HitCollection.search query, year, opts
|
|
22
22
|
rescue OpenURI::HTTPError, SocketError, OpenSSL::SSL::SSLError => e
|
|
23
23
|
raise Relaton::RequestError, e.message
|
|
24
24
|
end
|
|
@@ -26,34 +26,38 @@ module Relaton
|
|
|
26
26
|
#
|
|
27
27
|
# Get NIST document by reference
|
|
28
28
|
#
|
|
29
|
-
# @param
|
|
29
|
+
# @param ref [String, Pubid::Nist::Identifier] the NIST standard Code
|
|
30
|
+
# to look up, or its parse from Relaton::Db (relaton#205)
|
|
30
31
|
# @param year [String] the year the standard was published (optional)
|
|
31
32
|
# @param opts [Hash] options
|
|
32
33
|
# @option opts [Boolean] :all_parts restricted to all parts
|
|
33
34
|
#
|
|
34
35
|
# @return [Relaton::Nist::ItemData, nil] bibliographic item
|
|
35
36
|
#
|
|
36
|
-
def get(
|
|
37
|
-
|
|
37
|
+
def get(ref, year = nil, opts = {}) # rubocop:disable Metrics/AbcSize, Metrics/MethodLength, Metrics/CyclomaticComplexity, Metrics/PerceivedComplexity
|
|
38
|
+
# The lookup below works on text, so a pubid from Relaton::Db is
|
|
39
|
+
# read back as its printed form; the pubid itself is not changed.
|
|
40
|
+
ref = ref.to_s unless ref.is_a?(String)
|
|
41
|
+
return fetch_ref_err(ref, year, []) if ref.match?(/\sEP$/)
|
|
38
42
|
|
|
39
|
-
/^(?<code2>[^(]+)(?:\((?<date2>\w+\s(?:\d{2},\s)?\d{4})\))?\s?\(?(?:(?<=\()(?<stage>(?:I|F|\d)PD))?/ =~
|
|
40
|
-
stage ||= /(?<=\.)PD-\w+(?=\.)/.match(
|
|
43
|
+
/^(?<code2>[^(]+)(?:\((?<date2>\w+\s(?:\d{2},\s)?\d{4})\))?\s?\(?(?:(?<=\()(?<stage>(?:I|F|\d)PD))?/ =~ ref
|
|
44
|
+
stage ||= /(?<=\.)PD-\w+(?=\.)/.match(ref)&.to_s
|
|
41
45
|
if code2
|
|
42
|
-
|
|
46
|
+
ref = code2.strip
|
|
43
47
|
opts[:date] = parse_date(date2, str: false) if date2
|
|
44
48
|
opts[:stage] = stage if stage
|
|
45
49
|
end
|
|
46
50
|
|
|
47
51
|
if year.nil?
|
|
48
|
-
/^(?<code1>[^:]+):(?<year1>[^:]+)$/ =~
|
|
52
|
+
/^(?<code1>[^:]+):(?<year1>[^:]+)$/ =~ ref
|
|
49
53
|
unless code1.nil?
|
|
50
|
-
|
|
54
|
+
ref = code1
|
|
51
55
|
year = year1
|
|
52
56
|
end
|
|
53
57
|
end
|
|
54
58
|
|
|
55
|
-
|
|
56
|
-
nistbib_get(
|
|
59
|
+
ref += "-1" if opts[:all_parts]
|
|
60
|
+
nistbib_get(ref, year, opts)
|
|
57
61
|
end
|
|
58
62
|
|
|
59
63
|
private
|
|
@@ -61,14 +65,14 @@ module Relaton
|
|
|
61
65
|
#
|
|
62
66
|
# Get NIST document by reference
|
|
63
67
|
#
|
|
64
|
-
# @param [String]
|
|
68
|
+
# @param [String] ref reference
|
|
65
69
|
# @param [String] year year
|
|
66
70
|
# @param [Hash] opts options
|
|
67
71
|
#
|
|
68
72
|
# @return [Relaton::Nist::ItemData, nil] bibliographic item
|
|
69
73
|
#
|
|
70
|
-
def nistbib_get(
|
|
71
|
-
result = nistbib_search_filter(
|
|
74
|
+
def nistbib_get(ref, year, opts)
|
|
75
|
+
result = nistbib_search_filter(ref, year, opts) || (return nil)
|
|
72
76
|
ret = nistbib_results_filter(result, year, opts)
|
|
73
77
|
if ret[:ret]
|
|
74
78
|
Util.info "Found: `#{ret[:ret].docidentifier.first.content}`", key: result.reference
|
|
@@ -153,14 +157,14 @@ module Relaton
|
|
|
153
157
|
#
|
|
154
158
|
# Get search results and filter them by code and year
|
|
155
159
|
#
|
|
156
|
-
# @param
|
|
160
|
+
# @param ref [String] reference
|
|
157
161
|
# @param year [String, nil] year
|
|
158
162
|
# @param opts [Hash] options
|
|
159
163
|
#
|
|
160
164
|
# @return [Relaton::Nist::HitCollection] hits collection
|
|
161
165
|
#
|
|
162
|
-
def nistbib_search_filter(
|
|
163
|
-
result = search(
|
|
166
|
+
def nistbib_search_filter(ref, year, opts)
|
|
167
|
+
result = search(ref, year, opts)
|
|
164
168
|
result.search_filter
|
|
165
169
|
end
|
|
166
170
|
|