relaton 3.0.0.pre.alpha.5 → 3.0.0.pre.alpha.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. checksums.yaml +4 -4
  2. data/lib/relaton/3gpp/bibliography.rb +19 -6
  3. data/lib/relaton/3gpp/processor.rb +6 -0
  4. data/lib/relaton/bib/item_data.rb +14 -0
  5. data/lib/relaton/bipm/bibliography.rb +24 -21
  6. data/lib/relaton/bsi/bibliography.rb +26 -23
  7. data/lib/relaton/calconnect/bibliography.rb +6 -6
  8. data/lib/relaton/calconnect/hit_collection.rb +4 -1
  9. data/lib/relaton/ccsds/bibliography.rb +15 -10
  10. data/lib/relaton/ccsds/hit_collection.rb +1 -1
  11. data/lib/relaton/ccsds/processor.rb +3 -2
  12. data/lib/relaton/cen/bibliography.rb +27 -24
  13. data/lib/relaton/cie/bibliography.rb +11 -9
  14. data/lib/relaton/cie/scrapper.rb +15 -8
  15. data/lib/relaton/core/processor.rb +61 -8
  16. data/lib/relaton/core/request_error.rb +21 -3
  17. data/lib/relaton/db/cache.rb +49 -19
  18. data/lib/relaton/db/cache_entry.rb +9 -0
  19. data/lib/relaton/db/registry.rb +128 -28
  20. data/lib/relaton/db.rb +162 -48
  21. data/lib/relaton/ecma/bibliography.rb +18 -13
  22. data/lib/relaton/etsi/bibliography.rb +8 -7
  23. data/lib/relaton/etsi/data_fetcher.rb +118 -9
  24. data/lib/relaton/gb/bibliography.rb +20 -12
  25. data/lib/relaton/iala/bibliography.rb +15 -12
  26. data/lib/relaton/iana/bibliography.rb +12 -12
  27. data/lib/relaton/iana/parser.rb +5 -2
  28. data/lib/relaton/iec/bibliography.rb +18 -7
  29. data/lib/relaton/iec/processor.rb +6 -1
  30. data/lib/relaton/ieee/bibliography.rb +19 -13
  31. data/lib/relaton/ieee/processor.rb +16 -0
  32. data/lib/relaton/ieee/rawbib_id_parser.rb +2 -2
  33. data/lib/relaton/ietf/bibliography.rb +11 -9
  34. data/lib/relaton/ietf/scraper.rb +5 -4
  35. data/lib/relaton/index/config.rb +9 -1
  36. data/lib/relaton/index/file_io.rb +190 -0
  37. data/lib/relaton/index/type.rb +96 -23
  38. data/lib/relaton/iso/bibliography.rb +11 -9
  39. data/lib/relaton/itu/bibliography.rb +9 -4
  40. data/lib/relaton/jis/bibliography.rb +22 -11
  41. data/lib/relaton/nist/bibliography.rb +25 -21
  42. data/lib/relaton/oasis/bibliography.rb +16 -13
  43. data/lib/relaton/ogc/bibliography.rb +15 -14
  44. data/lib/relaton/ogc/hit_collection.rb +9 -7
  45. data/lib/relaton/ogc/processor.rb +7 -1
  46. data/lib/relaton/omg/bibliography.rb +9 -9
  47. data/lib/relaton/omg/scraper.rb +3 -2
  48. data/lib/relaton/plateau/bibliography.rb +9 -7
  49. data/lib/relaton/plateau/hit_collection.rb +2 -1
  50. data/lib/relaton/version.rb +1 -1
  51. data/lib/relaton/xsf/bibliography.rb +11 -6
  52. metadata +18 -4
@@ -11,6 +11,10 @@ module Relaton
11
11
  # wrong-structure handling (re-download, or stop and log).
12
12
  class InvalidIndexError < StandardError; end
13
13
 
14
+ # Bump when the sidecar payload shape changes, so an older sidecar is
15
+ # discarded and rebuilt instead of misread.
16
+ SIDECAR_VERSION = 2 # v2: the payload carries the yaml byte-size for freshness
17
+
14
18
  attr_reader :url, :pubid_class
15
19
  attr_accessor :sorted
16
20
 
@@ -288,11 +292,197 @@ module Relaton
288
292
  #
289
293
  def remove
290
294
  Index.config.storage.remove file
295
+ delete_sidecar
291
296
  []
292
297
  end
293
298
 
299
+ #
300
+ # Raw-row read for the lazy search path (relaton#242 stopgap): returns
301
+ # precomputed root-number sort keys and the rows as plain hashes — no
302
+ # pubid objects. A Marshal sidecar next to the yaml holds the keys+rows
303
+ # so repeat loads skip the YAML parse and the one-time full
304
+ # materialization; the sidecar is rebuilt whenever the yaml is newer.
305
+ #
306
+ # @return [Array<Array<String>, Array<Hash>] sort keys and raw rows
307
+ #
308
+ def read_raw
309
+ case url
310
+ when String
311
+ with_file_lock do
312
+ check_file ? read_raw_file : fetch_raw_and_save
313
+ end
314
+ else
315
+ read_raw_file || [[], []]
316
+ end
317
+ end
318
+
319
+ # Deserialize raw rows into pubid rows. Only the caller knows which
320
+ # slice it needs, so materialization happens here rather than at load.
321
+ #
322
+ # @param [Array<Hash>] rows raw rows
323
+ # @return [Array<Hash>] rows with deserialized ids
324
+ def materialize(rows)
325
+ return rows unless @pubid_class
326
+
327
+ rows.map { |r| { id: deserialize_id(r[:id]), file: r[:file] } }
328
+ end
329
+
330
+ # Save raw rows (possibly alongside the keys they were loaded with);
331
+ # sorts by the same root-number key the object path sorts by.
332
+ #
333
+ # @param [Array<String>] keys sort keys, parallel to rows
334
+ # @param [Array<Hash>] rows raw rows
335
+ # @return [void]
336
+ def save_raw(keys, rows)
337
+ ordered = @pubid_class ? keys.zip(rows).sort_by { |k, _| k.to_s }
338
+ .map { |_, r| r } : rows
339
+ yaml = ordered.map do |item|
340
+ { id: item[:id], file: item[:file] }
341
+ end.to_yaml
342
+ Index.config.storage.write file, yaml
343
+ delete_sidecar
344
+ end
345
+
294
346
  private
295
347
 
348
+ def sidecar_file
349
+ "#{file}.ms"
350
+ end
351
+
352
+ def delete_sidecar
353
+ File.delete(sidecar_file) if File.file?(sidecar_file)
354
+ rescue Errno::EACCES
355
+ nil
356
+ end
357
+
358
+ def read_raw_file
359
+ # The sidecar is authoritative while it describes the yaml byte-for-byte
360
+ # (same size) — the yaml is not even parsed until the sidecar is stale
361
+ # or unreadable.
362
+ if sidecar_fresh?
363
+ loaded = sidecar
364
+ return loaded if loaded
365
+ end
366
+
367
+ yaml = Index.config.storage.read(file)
368
+ return unless yaml
369
+
370
+ begin
371
+ raw = YAML.safe_load(yaml, permitted_classes: [Symbol])
372
+ rescue Psych::SyntaxError
373
+ warn_local_index_error("YAML parsing error when reading")
374
+ delete_sidecar
375
+ return [[], []]
376
+ end
377
+ build_raw(raw)
378
+ end
379
+
380
+ # Size, not mtime: NTFS timestamp coarseness makes a rewritten yaml
381
+ # carry the sidecar's own mtime, so "yaml is newer" misses the rebuild.
382
+ def sidecar_fresh?
383
+ File.file?(sidecar_file) && File.file?(file) &&
384
+ File.size(file) == sidecar_yaml_size
385
+ end
386
+
387
+ def sidecar_yaml_size
388
+ Marshal.load(File.binread(sidecar_file))[4]
389
+ rescue TypeError, ArgumentError, EOFError
390
+ nil
391
+ end
392
+
393
+ def sidecar
394
+ version, sorted, keys, rows = Marshal.load(File.binread(sidecar_file))
395
+ return unless version == SIDECAR_VERSION
396
+
397
+ @sorted = sorted
398
+ [keys, rows]
399
+ rescue TypeError, ArgumentError, EOFError
400
+ nil
401
+ end
402
+
403
+ def build_raw(raw)
404
+ unless check_format(raw)
405
+ warn_local_index_error("Wrong structure of")
406
+ return [[], []]
407
+ end
408
+
409
+ return build_raw_in_child(raw) if build_sidecar_in_child?
410
+
411
+ build_raw_in_process(raw)
412
+ rescue InvalidIndexError
413
+ warn_local_index_error("Wrong structure of")
414
+ [[], []]
415
+ end
416
+
417
+ # The one-time key build materializes every row (~675 MB on the
418
+ # 79,993-row ISO index). Freed pages are often not returned to the OS,
419
+ # so an in-process build leaves the parent's RSS high — and containers
420
+ # OOM on RSS (relaton#242). Where fork exists, build the sidecar in a
421
+ # child: the graph dies with it and the parent never spikes.
422
+ def build_sidecar_in_child?
423
+ Process.respond_to?(:fork) && Index.config.build_sidecar_in_child != false
424
+ end
425
+
426
+ def build_raw_in_child(raw)
427
+ pid = Process.fork do
428
+ build_raw_in_process(raw)
429
+ exit!(0)
430
+ end
431
+ Process.wait(pid)
432
+ loaded = sidecar
433
+ return build_raw_in_process(raw) unless loaded # child failed
434
+
435
+ loaded
436
+ rescue Errno::ENOMEM, SystemCallError
437
+ build_raw_in_process(raw)
438
+ end
439
+
440
+ def build_raw_in_process(raw)
441
+ objects = deserialize_pubid(raw)
442
+ keys = objects.map { |r| r[:id].root.number.to_s }
443
+ rows = objects.map { |r| { id: raw_id(r), file: r[:file] } }
444
+ write_sidecar(keys, rows)
445
+ [keys, rows]
446
+ rescue InvalidIndexError
447
+ warn_local_index_error("Wrong structure of")
448
+ [[], []]
449
+ end
450
+
451
+ def raw_id(row)
452
+ id = row[:id]
453
+ id.respond_to?(:to_hash) ? id.to_hash : id
454
+ end
455
+
456
+ def write_sidecar(keys, rows)
457
+ File.binwrite(sidecar_file,
458
+ Marshal.dump([SIDECAR_VERSION, @sorted, keys, rows,
459
+ File.size(file)]))
460
+ rescue Errno::EACCES, Errno::ENOENT, Errno::EROFS
461
+ nil # the sidecar is an optimization; a read-only dir still works
462
+ end
463
+
464
+ def fetch_raw_and_save
465
+ uri = URI.parse(url)
466
+ body = Net::HTTP.get(uri)
467
+ yaml = nil
468
+ Zip::File.open_buffer(body) do |zip|
469
+ yaml = zip.entries.first.get_input_stream.read
470
+ end
471
+ Util.info "Downloaded index from `#{url}`", progname
472
+ raw = YAML.safe_load(yaml, permitted_classes: [Symbol])
473
+ if check_format(raw)
474
+ save raw
475
+ raw = nil # release the parsed copy; read_raw_file loads the sidecar's
476
+ read_raw_file
477
+ else
478
+ warn_remote_index_error "Wrong structure of"
479
+ [[], []]
480
+ end
481
+ rescue Psych::SyntaxError
482
+ warn_remote_index_error "YAML parsing error when reading"
483
+ [[], []]
484
+ end
485
+
296
486
  def with_file_lock(&)
297
487
  @@file_locks_mutex.synchronize do
298
488
  @@file_locks[file] ||= Mutex.new
@@ -30,7 +30,24 @@ module Relaton
30
30
  def index
31
31
  return @source.whole_index if @source
32
32
 
33
- @index ||= @file_io.read
33
+ @index ||= @file_io.materialize(raw_index[1])
34
+ end
35
+
36
+ # Raw rows plus their precomputed root-number sort keys (relaton#242
37
+ # stopgap). The lazy search path materializes pubid objects only for
38
+ # the bucket it narrows to, so a process that answers one lookup holds
39
+ # raw hashes instead of the whole pubid graph. A caller-seeded `@index`
40
+ # (a spec fixture, a preloaded pool entry) is the source of truth: the
41
+ # raw side derives from it, in its order, so a seeder that sorts by the
42
+ # narrowing key keeps the bsearch valid.
43
+ def raw_index
44
+ @raw_index ||= if @index
45
+ keys = @index.map { |r| narrowing_key(r[:id]) }
46
+ rows = @index.map { |r| { id: raw_row_id(r[:id]), file: r[:file] } }
47
+ [keys, rows]
48
+ else
49
+ @file_io.read_raw
50
+ end
34
51
  end
35
52
 
36
53
  #
@@ -59,14 +76,23 @@ module Relaton
59
76
  # @return [void]
60
77
  #
61
78
  def add_or_update(id, file)
62
- key = id.to_s
63
- item = id_lookup[key]
64
- if item
65
- item[:file] = file
79
+ # A seeded or materialized @index (spec fixtures, preloaded pool
80
+ # entries) and pubid-less indexes run the legacy object path
81
+ # bit-for-bit: their consumers match rows by object identity
82
+ # semantics (`==`, `exclude`, to_s dedup) that raw rows cannot
83
+ # reproduce. Only a pubid-backed FileIO-loaded index — the
84
+ # relaton#242 memory case — adds through raw rows.
85
+ return legacy_add(id, file) if @index || !@file_io.pubid_class
86
+
87
+ raw_hash = raw_row_id(id)
88
+ keys, rows = raw_index
89
+ pos = raw_position(raw_hash)
90
+ if pos
91
+ rows[pos] = { id: raw_hash, file: file }
66
92
  else
67
- new_item = { id: id, file: file }
68
- index << new_item
69
- id_lookup[key] = new_item
93
+ rows << { id: raw_hash, file: file }
94
+ keys << narrowing_key(id)
95
+ @raw_lookup[raw_hash] = rows.size - 1
70
96
  @file_io.sorted = false
71
97
  end
72
98
  end
@@ -110,7 +136,12 @@ module Relaton
110
136
  # @return [void]
111
137
  #
112
138
  def save
113
- @file_io.save(@index || [])
139
+ if @index
140
+ @file_io.save(@index)
141
+ else
142
+ keys, rows = raw_index
143
+ @file_io.save_raw(keys, rows)
144
+ end
114
145
  end
115
146
 
116
147
  #
@@ -125,6 +156,8 @@ module Relaton
125
156
  @source = new_source
126
157
  @index = nil
127
158
  @id_lookup = nil
159
+ @raw_index = nil
160
+ @raw_lookup = nil
128
161
  end
129
162
 
130
163
  #
@@ -135,6 +168,8 @@ module Relaton
135
168
  def remove_all
136
169
  @index = []
137
170
  @id_lookup = nil
171
+ @raw_index = [[], []]
172
+ @raw_lookup = {}
138
173
  @file_io.sorted = true
139
174
  end
140
175
 
@@ -146,6 +181,44 @@ module Relaton
146
181
  end
147
182
  end
148
183
 
184
+ # Position of a raw row by its id hash — the raw-side equivalent of
185
+ # `id_lookup`, built once, so a crawl's add_or_update stays O(1) per row.
186
+ def raw_position(raw_hash)
187
+ @raw_lookup ||= raw_index[1].each_with_object({}) do |row, h|
188
+ h[row[:id]] = h.key?(row[:id]) ? h[row[:id]] : rows_index_of(h, row)
189
+ end
190
+ @raw_lookup[raw_hash]
191
+ end
192
+
193
+ def rows_index_of(lookup, row)
194
+ raw_index[1].index { |r| r[:id].equal?(row[:id]) || r[:id] == row[:id] }
195
+ end
196
+ # The pre-#242 add path, kept verbatim for seeded/materialized and
197
+ # pubid-less indexes.
198
+ def legacy_add(id, file)
199
+ key = id.to_s
200
+ item = id_lookup[key]
201
+ if item
202
+ item[:file] = file
203
+ else
204
+ new_item = { id: id, file: file }
205
+ index << new_item
206
+ id_lookup[key] = new_item
207
+ @file_io.sorted = false
208
+ end
209
+ end
210
+
211
+ # Same key expression the FileIO sidecar computes at build time; a
212
+ # String id (a no-pubid_class index) has no `.root`, and never narrows.
213
+ def narrowing_key(id)
214
+ id.respond_to?(:root) && id.root.respond_to?(:number) ? id.root.number.to_s : ""
215
+ end
216
+
217
+ def raw_row_id(id)
218
+ pc = @file_io.pubid_class
219
+ pc && id.is_a?(pc) ? id.to_hash : id
220
+ end
221
+
149
222
  def new_source
150
223
  ShardSource.new(@dir, @pages_url, @pubid_class) if @pages_url
151
224
  end
@@ -155,13 +228,15 @@ module Relaton
155
228
  # never refreshed, so the shards stay the fresher answer.
156
229
  return @source.rows(id) if @source && id && !id.is_a?(String)
157
230
 
158
- # index needs to be created to check if sorted
159
- idx = index
160
- if @file_io.sorted && id && !id.is_a?(String)
161
- candidates_by_number(id)
162
- else
163
- idx
231
+ # The lazy path narrows on the precomputed keys and materializes only
232
+ # the bucket; the materialized `index` stays available to callers
233
+ # that need the whole graph (`#index`, a String query, a block).
234
+ # Load first: a pubid-backed load is what settles @file_io.sorted.
235
+ if id && !id.is_a?(String) && @file_io.pubid_class
236
+ raw_index
237
+ return candidates_by_number(id) if @file_io.sorted
164
238
  end
239
+ index
165
240
  end
166
241
 
167
242
  # Narrowing key: the base *document's* number as a string. `#root` walks a
@@ -169,26 +244,24 @@ module Relaton
169
244
  # returns self for a base document), so a document and all its wrappers
170
245
  # share one key and cluster together. `.to_s` because the key is compared
171
246
  # as a string (a pubid number Component is not `<`/`>`-comparable), and
172
- # FileIO sorts the index by this exact same key so bsearch stays valid.
247
+ # FileIO computes this exact same key once at sidecar-build time so the
248
+ # bsearch over the keys array stays valid.
173
249
  def candidates_by_number(id)
174
250
  target = id.root.number.to_s
251
+ keys, rows = raw_index
175
252
  left = bsearch_left(target)
176
253
  return [] unless left
177
254
 
178
255
  right = bsearch_right(target)
179
- index[left...right]
256
+ @file_io.materialize(rows[left...right])
180
257
  end
181
258
 
182
259
  def bsearch_left(target)
183
- index.bsearch_index do |item|
184
- item[:id].root.number.to_s >= target
185
- end
260
+ raw_index[0].bsearch_index { |k| k >= target }
186
261
  end
187
262
 
188
263
  def bsearch_right(target)
189
- index.bsearch_index do |item|
190
- item[:id].root.number.to_s > target
191
- end || index.size
264
+ raw_index[0].bsearch_index { |k| k > target } || raw_index[0].size
192
265
  end
193
266
 
194
267
  # The query is the reference, so two identifiers take the subset match.
@@ -24,7 +24,9 @@ module Relaton
24
24
  raise Relaton::RequestError, e.message
25
25
  end
26
26
 
27
- # @param ref [String] the ISO standard Code to look up (e..g "ISO 9000")
27
+ # @param ref [String, Pubid::Iso::Identifier] the ISO standard Code to
28
+ # look up (e..g "ISO 9000"), or the parse that Relaton::Db routed with
29
+ # (relaton#205); a pubid is never mutated
28
30
  # @param year [String, NilClass] the year the standard was published
29
31
  # @param opts [Hash] options; restricted to :all_parts if all-parts
30
32
  # @option opts [Boolean] :all_parts if all-parts reference is required
@@ -33,21 +35,21 @@ module Relaton
33
35
  #
34
36
  # @return [RelatonIsoBib::IsoBibliographicItem] Bibliographic item
35
37
  def get(ref, year = nil, opts = {}) # rubocop:disable Metrics/CyclomaticComplexity,Metrics/MethodLength,Metrics/PerceivedComplexity,Metrics/AbcSize
36
- code = ref.gsub("\u2013", "-")
37
-
38
- # parse "all parts" request
39
- # code.sub! " (all parts)", ""
40
- # opts[:all_parts] ||= $~ && opts[:all_parts].nil?
41
-
42
- query_pubid = ::Pubid::Iso::Identifier.parse(code)
38
+ query_pubid = if ref.is_a?(String)
39
+ ::Pubid::Iso::Identifier.parse(ref.gsub("\u2013", "-"))
40
+ else
41
+ ref
42
+ end
43
43
  if year&.respond_to?(:to_i)
44
+ # Copy first: a caller's pubid is also its Db cache key.
45
+ query_pubid = query_pubid.class.from_hash(query_pubid.to_hash)
44
46
  query_pubid.root.date = ::Pubid::Components::Date.new(year: year.to_s)
45
47
  end
46
48
  query_pubid = query_pubid.to_all_parts if opts[:all_parts]
47
49
  Util.info "Fetching from Relaton repository ...", key: query_pubid.to_s
48
50
 
49
51
  hits, missed_year_ids = isobib_search_filter(query_pubid, opts)
50
- tip_ids = look_up_with_any_types_stages(hits, ref, opts)
52
+ tip_ids = look_up_with_any_types_stages(hits, ref.to_s, opts)
51
53
 
52
54
  date_filter = opts[:publication_date_before] || opts[:publication_date_after]
53
55
  if date_filter && !query_pubid.all_parts
@@ -31,13 +31,18 @@ module Relaton
31
31
  HitCollection.new(refid).tap(&:search)
32
32
  end
33
33
 
34
- # @param code [String] the ITU standard Code to look up
34
+ # @param ref [String, Pubid::Itu::Identifier] the ITU standard Code to
35
+ # look up, or its parse from Relaton::Db (relaton#205); #with_year
36
+ # copies, so the pubid is not changed
35
37
  # @param year [String] the year the standard was published (optional)
36
38
  # @param opts [Hash] options
37
39
  # @return [Relaton::Bib::ItemData, nil]
38
- def get(code, year = nil, opts = {})
39
- warn_incorrect_ref(code)
40
- refid = with_year ::Pubid::Itu.parse(code), year
40
+ def get(ref, year = nil, opts = {})
41
+ if ref.is_a? String
42
+ warn_incorrect_ref(ref)
43
+ ref = ::Pubid::Itu.parse(ref)
44
+ end
45
+ refid = with_year ref, year
41
46
 
42
47
  ret = itubib_get1(refid)
43
48
  return nil if ret.nil?
@@ -13,7 +13,8 @@ module Relaton
13
13
  # the reference's series and number (a supplement is filed under its base
14
14
  # number), so an edition and its amendments are returned together.
15
15
  #
16
- # @param [String] code JIS document code
16
+ # @param [String, Pubid::Jis::Identifier] ref JIS document reference; a
17
+ # pubid is never mutated
17
18
  # @param [String, nil] year JIS document year
18
19
  #
19
20
  # @return [Relaton::Jis::HitCollection] search result
@@ -21,16 +22,21 @@ module Relaton
21
22
  # @raise [Pubid::Errors::ParseError] when the reference is not a JIS
22
23
  # identifier, so a malformed reference is not reported as "not found"
23
24
  #
24
- def search(code, year = nil)
25
- pubid = ::Pubid::Jis::Identifier.parse code
26
- pubid.year ||= year.to_i if year
25
+ def search(ref, year = nil)
26
+ pubid = ref.is_a?(String) ? ::Pubid::Jis::Identifier.parse(ref) : ref
27
+ if year && pubid.year.nil?
28
+ # Copy first: a caller's pubid is also its Db cache key.
29
+ pubid = pubid.class.from_hash(pubid.to_hash)
30
+ pubid.year = year.to_i
31
+ end
27
32
  HitCollection.new pubid
28
33
  end
29
34
 
30
35
  #
31
36
  # Get JIS document by reference
32
37
  #
33
- # @param [String] ref JIS document reference
38
+ # @param [String, Pubid::Jis::Identifier] ref JIS document reference, or
39
+ # the parse that Relaton::Db routed with (relaton#205)
34
40
  # @param [String, nil] year JIS document year
35
41
  # @param [Hash] opts options
36
42
  # @option opts [Boolean] :all_parts return all parts of document
@@ -41,16 +47,21 @@ module Relaton
41
47
  # identifier
42
48
  #
43
49
  def get(ref, year = nil, opts = {}) # rubocop:disable Metrics/AbcSize
44
- code = ref.sub(/\s\((all parts|規格群)\)/, "")
45
- opts[:all_parts] ||= !$1.nil?
46
- Util.info "Fetching from webdesk.jsa.or.jp ...", key: ref
47
- hits = search(code, year)
50
+ if ref.is_a?(String)
51
+ query = ref.sub(/\s\((all parts|規格群)\)/, "")
52
+ opts[:all_parts] ||= !$1.nil?
53
+ else
54
+ opts[:all_parts] ||= ref.all_parts?
55
+ query = ref.all_parts? ? ref.identifiers.first : ref
56
+ end
57
+ Util.info "Fetching from webdesk.jsa.or.jp ...", key: ref.to_s
58
+ hits = search(query, year)
48
59
  result = opts[:all_parts] ? hits.find_all_parts : hits.find
49
60
  if result.is_a? Bib::ItemData
50
- Util.info "Found: `#{result.docidentifier[0].content}`", key: ref
61
+ Util.info "Found: `#{result.docidentifier[0].content}`", key: ref.to_s
51
62
  return result
52
63
  end
53
- hint result, ref, year
64
+ hint result, ref.to_s, year
54
65
  end
55
66
 
56
67
  #
@@ -7,18 +7,18 @@ module Relaton
7
7
  #
8
8
  # Search NIST documents by reference
9
9
  #
10
- # @param text [String] reference
10
+ # @param ref [String] reference
11
11
  #
12
12
  # @return [Relaton::Nist::HitCollection] search result
13
13
  #
14
- def search(text, year = nil, opts = {})
15
- ref = text.sub(/^NISTIR/, "NIST IR").sub(/\/Add/, " Add")
14
+ def search(ref, year = nil, opts = {})
15
+ query = ref.sub(/^NISTIR/, "NIST IR").sub(/\/Add/, " Add")
16
16
  # pubid 2.x only recognizes the addendum marker with a trailing
17
17
  # period, and only splits an uppercase part letter — canonicalize
18
18
  # both ("800-38a Add" -> "800-38A Add.") so @reference parses to the
19
19
  # same pubid the index/CSRC carry.
20
- ref = ref.sub(/\bAdd\b\.?/i, "Add.").sub(/([0-9])([a-z])(?=\s+Add\.)/) { "#{$1}#{$2.upcase}" }
21
- HitCollection.search ref, year, opts
20
+ query = query.sub(/\bAdd\b\.?/i, "Add.").sub(/([0-9])([a-z])(?=\s+Add\.)/) { "#{$1}#{$2.upcase}" }
21
+ HitCollection.search query, year, opts
22
22
  rescue OpenURI::HTTPError, SocketError, OpenSSL::SSL::SSLError => e
23
23
  raise Relaton::RequestError, e.message
24
24
  end
@@ -26,34 +26,38 @@ module Relaton
26
26
  #
27
27
  # Get NIST document by reference
28
28
  #
29
- # @param code [String] the NIST standard Code to look up
29
+ # @param ref [String, Pubid::Nist::Identifier] the NIST standard Code
30
+ # to look up, or its parse from Relaton::Db (relaton#205)
30
31
  # @param year [String] the year the standard was published (optional)
31
32
  # @param opts [Hash] options
32
33
  # @option opts [Boolean] :all_parts restricted to all parts
33
34
  #
34
35
  # @return [Relaton::Nist::ItemData, nil] bibliographic item
35
36
  #
36
- def get(code, year = nil, opts = {}) # rubocop:disable Metrics/AbcSize, Metrics/MethodLength, Metrics/CyclomaticComplexity, Metrics/PerceivedComplexity
37
- return fetch_ref_err(code, year, []) if code.match?(/\sEP$/)
37
+ def get(ref, year = nil, opts = {}) # rubocop:disable Metrics/AbcSize, Metrics/MethodLength, Metrics/CyclomaticComplexity, Metrics/PerceivedComplexity
38
+ # The lookup below works on text, so a pubid from Relaton::Db is
39
+ # read back as its printed form; the pubid itself is not changed.
40
+ ref = ref.to_s unless ref.is_a?(String)
41
+ return fetch_ref_err(ref, year, []) if ref.match?(/\sEP$/)
38
42
 
39
- /^(?<code2>[^(]+)(?:\((?<date2>\w+\s(?:\d{2},\s)?\d{4})\))?\s?\(?(?:(?<=\()(?<stage>(?:I|F|\d)PD))?/ =~ code
40
- stage ||= /(?<=\.)PD-\w+(?=\.)/.match(code)&.to_s
43
+ /^(?<code2>[^(]+)(?:\((?<date2>\w+\s(?:\d{2},\s)?\d{4})\))?\s?\(?(?:(?<=\()(?<stage>(?:I|F|\d)PD))?/ =~ ref
44
+ stage ||= /(?<=\.)PD-\w+(?=\.)/.match(ref)&.to_s
41
45
  if code2
42
- code = code2.strip
46
+ ref = code2.strip
43
47
  opts[:date] = parse_date(date2, str: false) if date2
44
48
  opts[:stage] = stage if stage
45
49
  end
46
50
 
47
51
  if year.nil?
48
- /^(?<code1>[^:]+):(?<year1>[^:]+)$/ =~ code
52
+ /^(?<code1>[^:]+):(?<year1>[^:]+)$/ =~ ref
49
53
  unless code1.nil?
50
- code = code1
54
+ ref = code1
51
55
  year = year1
52
56
  end
53
57
  end
54
58
 
55
- code += "-1" if opts[:all_parts]
56
- nistbib_get(code, year, opts)
59
+ ref += "-1" if opts[:all_parts]
60
+ nistbib_get(ref, year, opts)
57
61
  end
58
62
 
59
63
  private
@@ -61,14 +65,14 @@ module Relaton
61
65
  #
62
66
  # Get NIST document by reference
63
67
  #
64
- # @param [String] code reference
68
+ # @param [String] ref reference
65
69
  # @param [String] year year
66
70
  # @param [Hash] opts options
67
71
  #
68
72
  # @return [Relaton::Nist::ItemData, nil] bibliographic item
69
73
  #
70
- def nistbib_get(code, year, opts)
71
- result = nistbib_search_filter(code, year, opts) || (return nil)
74
+ def nistbib_get(ref, year, opts)
75
+ result = nistbib_search_filter(ref, year, opts) || (return nil)
72
76
  ret = nistbib_results_filter(result, year, opts)
73
77
  if ret[:ret]
74
78
  Util.info "Found: `#{ret[:ret].docidentifier.first.content}`", key: result.reference
@@ -153,14 +157,14 @@ module Relaton
153
157
  #
154
158
  # Get search results and filter them by code and year
155
159
  #
156
- # @param code [String] reference
160
+ # @param ref [String] reference
157
161
  # @param year [String, nil] year
158
162
  # @param opts [Hash] options
159
163
  #
160
164
  # @return [Relaton::Nist::HitCollection] hits collection
161
165
  #
162
- def nistbib_search_filter(code, year, opts)
163
- result = search(code, year, opts)
166
+ def nistbib_search_filter(ref, year, opts)
167
+ result = search(ref, year, opts)
164
168
  result.search_filter
165
169
  end
166
170