relaton 3.0.0.pre.alpha.4 → 3.0.0.pre.alpha.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. checksums.yaml +4 -4
  2. data/lib/relaton/adobe/processor.rb +9 -0
  3. data/lib/relaton/bib/converter/csl.rb +110 -0
  4. data/lib/relaton/bib/converter/ris.rb +104 -0
  5. data/lib/relaton/bib/converter/titles.rb +24 -0
  6. data/lib/relaton/bib/item_data.rb +9 -0
  7. data/lib/relaton/bib/model/item.rb +6 -6
  8. data/lib/relaton/bib/sanitizer.rb +71 -30
  9. data/lib/relaton/bib.rb +10 -0
  10. data/lib/relaton/bipm/processor.rb +1 -0
  11. data/lib/relaton/bipm/rawdata_bipm_metrologia/affiliations.rb +6 -6
  12. data/lib/relaton/bipm/si_brochure_parser.rb +2 -2
  13. data/lib/relaton/bsi/processor.rb +5 -0
  14. data/lib/relaton/ccsds/bibliography.rb +1 -0
  15. data/lib/relaton/ccsds/processor.rb +12 -0
  16. data/lib/relaton/cen/hit_collection.rb +1 -1
  17. data/lib/relaton/cen/processor.rb +5 -0
  18. data/lib/relaton/cen/scraper.rb +9 -9
  19. data/lib/relaton/cie/data_fetcher.rb +18 -18
  20. data/lib/relaton/cie/processor.rb +1 -0
  21. data/lib/relaton/cie.rb +1 -1
  22. data/lib/relaton/cloud.rb +127 -0
  23. data/lib/relaton/core/hit_collection.rb +6 -10
  24. data/lib/relaton/core/processor.rb +68 -0
  25. data/lib/relaton/db/cache.rb +444 -148
  26. data/lib/relaton/db/cache_entry.rb +33 -0
  27. data/lib/relaton/db/registry.rb +52 -0
  28. data/lib/relaton/db.rb +131 -72
  29. data/lib/relaton/doi/crossref.rb +23 -6
  30. data/lib/relaton/doi/processor.rb +1 -0
  31. data/lib/relaton/easc/processor.rb +1 -0
  32. data/lib/relaton/ecma/data_fetcher.rb +1 -1
  33. data/lib/relaton/ecma/data_parser.rb +1 -1
  34. data/lib/relaton/ecma/edition_parser.rb +2 -2
  35. data/lib/relaton/ecma/memento_parser.rb +4 -4
  36. data/lib/relaton/ecma/standard_parser.rb +6 -6
  37. data/lib/relaton/etsi/processor.rb +1 -0
  38. data/lib/relaton/gb/gb_scraper.rb +5 -5
  39. data/lib/relaton/gb/scraper.rb +16 -16
  40. data/lib/relaton/gb/sec_scraper.rb +8 -8
  41. data/lib/relaton/gb/t_scraper.rb +5 -5
  42. data/lib/relaton/gost/processor.rb +1 -0
  43. data/lib/relaton/iala/processor.rb +1 -0
  44. data/lib/relaton/iana/data_fetcher.rb +3 -3
  45. data/lib/relaton/iana/parser.rb +8 -4
  46. data/lib/relaton/iana/processor.rb +9 -0
  47. data/lib/relaton/iec/data_parser.rb +24 -9
  48. data/lib/relaton/iec/processor.rb +6 -0
  49. data/lib/relaton/iec.rb +1 -1
  50. data/lib/relaton/ieee/data_fetcher.rb +1 -1
  51. data/lib/relaton/ieee/processor.rb +8 -0
  52. data/lib/relaton/ietf/data_fetcher.rb +1 -1
  53. data/lib/relaton/ietf/processor.rb +1 -0
  54. data/lib/relaton/ietf/rfc/entry.rb +15 -19
  55. data/lib/relaton/iho/processor.rb +1 -0
  56. data/lib/relaton/index/file_io.rb +2 -2
  57. data/lib/relaton/index/pool.rb +4 -3
  58. data/lib/relaton/index/shard_source.rb +1 -1
  59. data/lib/relaton/index/type.rb +2 -5
  60. data/lib/relaton/isbn/open_library.rb +11 -7
  61. data/lib/relaton/isbn/processor.rb +10 -0
  62. data/lib/relaton/iso/data_parser.rb +2 -2
  63. data/lib/relaton/iso/processor.rb +5 -0
  64. data/lib/relaton/iso/scraper.rb +20 -20
  65. data/lib/relaton/itu/bibliography.rb +99 -13
  66. data/lib/relaton/itu/data_crawler_r.rb +2 -2
  67. data/lib/relaton/itu/hit_collection.rb +64 -52
  68. data/lib/relaton/itu/processor.rb +1 -0
  69. data/lib/relaton/itu/scraper.rb +13 -3
  70. data/lib/relaton/itu.rb +0 -2
  71. data/lib/relaton/jis/data_fetcher.rb +4 -4
  72. data/lib/relaton/jis/processor.rb +1 -0
  73. data/lib/relaton/jis/scraper.rb +8 -8
  74. data/lib/relaton/oasis/browser_agent.rb +2 -2
  75. data/lib/relaton/oasis/data_parser.rb +5 -5
  76. data/lib/relaton/oasis/data_parser_utils.rb +2 -2
  77. data/lib/relaton/oasis/data_part_parser.rb +7 -7
  78. data/lib/relaton/ogc/processor.rb +7 -0
  79. data/lib/relaton/oiml/processor.rb +1 -0
  80. data/lib/relaton/omg/scraper.rb +11 -11
  81. data/lib/relaton/omg.rb +1 -1
  82. data/lib/relaton/plateau/processor.rb +1 -0
  83. data/lib/relaton/un/bibliography.rb +21 -10
  84. data/lib/relaton/un/processor.rb +1 -0
  85. data/lib/relaton/version.rb +1 -1
  86. data/lib/relaton/w3c/processor.rb +1 -0
  87. data/lib/relaton.rb +13 -0
  88. metadata +48 -16
  89. data/lib/relaton/itu/pubid.rb +0 -199
@@ -1,24 +1,62 @@
1
1
  require "fileutils"
2
- require "timeout"
2
+ require "digest"
3
+ require "json"
4
+ require "date"
5
+ require "lutaml/store"
6
+ require_relative "cache_entry"
3
7
 
4
8
  module Relaton
5
9
  class Db
10
+ #
11
+ # The document cache of one directory, on two lutaml-store FileSystem
12
+ # stores under `<dir>/v2/`:
13
+ #
14
+ # - `rows/` holds the index. One store key per bucket, `<flavor>/<root
15
+ # number>` (e.g. `iso/19115`), whose value is the list of the bucket's
16
+ # CacheEntry rows. A lookup reads one small bucket, and a write is one
17
+ # atomic `update` of it.
18
+ # - `docs/` holds the XML documents. A document is named after the row it
19
+ # was stored for, and several rows can point to it.
20
+ #
21
+ # A key is a parsed pubid, a wrapped string (`ISO(ISO 19115-1)`, parsed
22
+ # through the prefix's processor), or a plain string that no processor
23
+ # owns. Locking and atomic writes come from lutaml-store.
24
+ #
6
25
  class Cache
26
+ LAYOUT = "v2".freeze
27
+ VERSIONS_KEY = "_versions".freeze
28
+ STRING_FLAVOR = "_key".freeze
29
+ PUBID_FLAVOR = "_pubid".freeze
30
+ NOT_FOUND = /\Anot_found/
31
+ UNDATED_TTL = 60
32
+ WRAPPED_KEY = /\A(?<prefix>[^(\s]+)\((?<code>.+)\)\z/m
33
+ # What a row may add to a dated key and still answer it: `===` also
34
+ # reads an omitted `part` as "any part" (`ISO 19115:2003 ===
35
+ # ISO 19115-1:2003`), which names another document.
36
+ LANGUAGE_COMPONENTS = %w[language languages].freeze
37
+ # What a row may add to a key and still be one of its editions.
38
+ EDITION_COMPONENTS = (%w[year date month] + LANGUAGE_COMPONENTS).freeze
39
+
7
40
  # @return [String]
8
41
  attr_reader :dir
9
42
 
10
- # @param dir [String] DB directory
11
- def initialize(dir, ext = "xml")
43
+ # @param dir [String] cache directory
44
+ def initialize(dir)
12
45
  @dir = dir
13
- @ext = ext
14
- FileUtils::mkdir_p dir
46
+ # A stale file at the cache path (e.g. a placeholder left by cache
47
+ # migration flows) makes every write raise Errno::EEXIST; replace
48
+ # it with the cache directory.
49
+ FileUtils.rm_rf dir if File.exist?(dir) && !File.directory?(dir)
50
+ archive_old_layout
51
+ open_stores
52
+ check_versions
15
53
  end
16
54
 
17
- # Move caches to anothe dir
55
+ # Move the cache to another directory.
18
56
  # @param new_dir [String, nil]
19
- # @return [String, nil]
57
+ # @return [String, nil] the new directory
20
58
  def mv(new_dir)
21
- return unless new_dir && @ext == "xml"
59
+ return unless new_dir
22
60
 
23
61
  if File.exist? new_dir
24
62
  Util.info "target directory exists `#{new_dir}`"
@@ -27,206 +65,464 @@ module Relaton
27
65
 
28
66
  FileUtils.mv dir, new_dir
29
67
  @dir = new_dir
68
+ open_stores
69
+ @dir
30
70
  end
31
71
 
32
- # Clear database
72
+ # Remove every row and document.
33
73
  def clear
34
- FileUtils.rm_rf Dir.glob "#{dir}/*"
74
+ @rows.clear
75
+ @docs.clear
76
+ @versions = {}
77
+ end
78
+
79
+ # Read the document (or `not_found <date>`) for a key: the row with
80
+ # this key, else, for a dated pubid, a row the key is a subset of.
81
+ #
82
+ # @param key [Pubid::Identifier, String]
83
+ # @return [String, nil]
84
+ def [](key)
85
+ found = lookup(key)
86
+ found && read_entry(found.last)
35
87
  end
88
+ alias get []
36
89
 
37
- # Save item
38
- # @param key [String]
39
- # @param value [String] Bibitem xml serialization
90
+ # @param key [Pubid::Identifier, String]
91
+ # @param value [String, nil] document XML or `not_found <date>`; nil
92
+ # deletes the row
40
93
  def []=(key, value)
41
- if value.nil?
42
- delete key
43
- return
94
+ store key, value
95
+ end
96
+
97
+ #
98
+ # Save a document for a query key. When the document's own identifier
99
+ # (`item_key`) differs from the query, both get a row, and both rows
100
+ # point to one document file.
101
+ #
102
+ # @param key [Pubid::Identifier, String] query key
103
+ # @param value [String, nil] document XML or `not_found <date>`
104
+ # @param item_key [Pubid::Identifier, String, nil] the document's key
105
+ # @return [String, nil] value
106
+ #
107
+ def store(key, value, item_key: nil) # rubocop:disable Metrics/AbcSize,Metrics/MethodLength
108
+ return delete(key) if value.nil?
109
+
110
+ bucket, key = resolve key
111
+ fetched = fetched_of value
112
+ record_version bucket
113
+ if value.match? NOT_FOUND
114
+ upsert bucket, entry(key, CacheEntry::NOT_FOUND, nil, fetched)
115
+ return value
116
+ end
117
+
118
+ item_bucket, item_key = item_key ? resolve(item_key) : [bucket, key]
119
+ file = doc_name item_bucket, item_key
120
+ record_version item_bucket
121
+ @rows.adapter.transaction do
122
+ @docs.set file, value
123
+ upsert item_bucket, entry(item_key, CacheEntry::DOC, file, fetched)
124
+ unless canonical(key) == canonical(item_key)
125
+ upsert bucket, entry(key, CacheEntry::DOC, file, fetched)
126
+ end
44
127
  end
128
+ value
129
+ end
130
+
131
+ # Delete the row of a key. Its document goes too, when no other row
132
+ # points to it.
133
+ # @param key [Pubid::Identifier, String]
134
+ def delete(key)
135
+ bucket, key = resolve key
136
+ target = canonical key
137
+ return unless read_bucket(bucket).any? { |row| row_key(row) == target }
45
138
 
46
- prefix_dir = "#{@dir}/#{prefix(key)}"
47
- FileUtils::mkdir_p prefix_dir
48
- set_version prefix_dir
49
- file_safe_write "#{filename(key)}.#{ext(value)}", value
139
+ delete_row bucket, target
50
140
  end
51
141
 
52
- # @param value [String]
53
- # @return [String]
54
- def ext(value)
55
- case value
56
- when /^not_found/ then "notfound"
57
- when /^redirection/ then "redirect"
58
- else @ext
59
- end
142
+ #
143
+ # Delete the row a lookup of the key finds (its own, or the one a dated
144
+ # key is a subset of) when it is no longer valid.
145
+ #
146
+ # @param key [Pubid::Identifier, String]
147
+ # @param year [String, nil]
148
+ #
149
+ def expire(key, year)
150
+ found = lookup(key) or return
151
+ return if valid_row?(found.last, year)
152
+
153
+ delete_row resolve(found.first).first, row_key(found.last)
60
154
  end
61
155
 
62
- # Read item
63
- # @param key [String]
64
- # @return [String]
65
- def [](key)
66
- value = get(key)
67
- if (code = redirect_code value)
68
- self[code]
69
- else
70
- value
156
+ #
157
+ # Save the row of a key, and its document, from another cache.
158
+ #
159
+ # @param key [Pubid::Identifier, String]
160
+ # @param other [Relaton::Db::Cache]
161
+ #
162
+ def clone_entry(key, other)
163
+ found = other.lookup(key) or return
164
+ row_id, row = found
165
+ store row_id, other.read_entry(row)
166
+ end
167
+
168
+ # @param key [Pubid::Identifier, String]
169
+ # @return [String, nil] the date the entry was fetched
170
+ def fetched(key)
171
+ lookup(key)&.last&.dig("fetched")
172
+ end
173
+
174
+ # An undated entry expires after 60 days, a dated one never.
175
+ # @param key [Pubid::Identifier, String]
176
+ # @param year [String, nil]
177
+ def valid_entry?(key, year)
178
+ found = lookup(key)
179
+ found ? valid_row?(found.last, year) : false
180
+ end
181
+
182
+ #
183
+ # The cached editions of the key: documents whose pubid the key is a
184
+ # subset of (`key === id`) and that add only a year, a date or a
185
+ # language to it. For a query that selects among editions (e.g. by
186
+ # publication date).
187
+ #
188
+ # @param key [Pubid::Identifier]
189
+ # @return [Array<Array(Pubid::Identifier, String)>]
190
+ #
191
+ def candidates(key) # rubocop:disable Metrics/AbcSize,Metrics/CyclomaticComplexity
192
+ bucket, key = resolve key
193
+ return [] if key.is_a? String
194
+
195
+ read_bucket(bucket).filter_map do |row|
196
+ next if row["status"] != CacheEntry::DOC || !row["id"]
197
+
198
+ id = pubid_from row["id"]
199
+ next unless subset_of? key, id, row["id"], EDITION_COMPONENTS
200
+
201
+ xml = @docs.get row["file"]
202
+ [id, xml] if xml
71
203
  end
72
204
  end
73
205
 
74
206
  #
75
- # Save entry from cache of `db` to this cache.
207
+ # Every cached document once.
76
208
  #
77
- # @param [String] key key of the entry
78
- # @param [Relaton::Db] db database
209
+ # @yieldparam processor [Relaton::Core::Processor, nil] the owning flavor
210
+ # @yieldparam xml [String]
211
+ # @return [Array] the documents, or the block results
79
212
  #
80
- def clone_entry(key, db)
81
- self[key] ||= db.get(key)
82
- if (code = redirect_code get(key))
83
- clone_entry code, db
213
+ def all
214
+ files = {}
215
+ each_row do |bucket, row|
216
+ files[row["file"]] ||= bucket if row["file"]
217
+ end
218
+ files.filter_map do |file, bucket|
219
+ xml = @docs.get(file) or next
220
+ block_given? ? yield(processor_for(bucket), xml) : xml
84
221
  end
85
222
  end
86
223
 
87
- # Return fetched date
88
- # @param key [String]
89
- # @return [String]
90
- def fetched(key)
91
- value = self[key]
92
- return unless value
224
+ # @return [Array<Hash>] every row
225
+ def rows
226
+ list = []
227
+ each_row { |_, row| list << row }
228
+ list
229
+ end
93
230
 
94
- if value.match?(/^not_found/)
95
- value.match(/\d{4}-\d{2}-\d{2}/).to_s
96
- else
97
- doc = Nokogiri::XML value
98
- doc.at("/bibitem/fetched|bibdata/fetched")&.text
231
+ # The row of a key, with the key it is stored under.
232
+ # @param key [Pubid::Identifier, String]
233
+ # @return [Array(Object, Hash), nil]
234
+ def lookup(key)
235
+ bucket, key = resolve key
236
+ rows = read_bucket bucket
237
+ target = canonical key
238
+ row = rows.detect { |r| row_key(r) == target }
239
+ return [key, row] if row
240
+ return unless dated? key
241
+
242
+ subset_match rows, key
243
+ end
244
+
245
+ # @param row [Hash]
246
+ # @return [String, nil] document XML or `not_found <date>`
247
+ def read_entry(row)
248
+ return "not_found #{row['fetched']}" if row["status"] == CacheEntry::NOT_FOUND
249
+
250
+ @docs.get row["file"]
251
+ end
252
+
253
+ private
254
+
255
+ def open_stores
256
+ base = File.join dir, LAYOUT
257
+ @rows = new_store File.join(base, "rows"), ".json"
258
+ @docs = new_store File.join(base, "docs"), ".xml"
259
+ end
260
+
261
+ def new_store(path, extension)
262
+ Lutaml::Store::BasicStore.new(
263
+ adapter_type: :filesystem,
264
+ adapter_options: { path: path, extension: extension,
265
+ integrity_checks: false },
266
+ cache: { enabled: false },
267
+ )
268
+ end
269
+
270
+ # A cache written by the file-per-key layout (before v2) is moved to
271
+ # `<dir>-v1.bak`, never deleted.
272
+ def archive_old_layout # rubocop:disable Metrics/AbcSize,Metrics/MethodLength
273
+ return unless Dir.exist? dir
274
+
275
+ old = Dir.children(dir) - [LAYOUT]
276
+ bak = "#{dir}-v1.bak"
277
+ # Once only: an older relaton that shares the directory writes the old
278
+ # layout again, and v2 ignores it.
279
+ return if old.empty? || File.exist?(bak)
280
+
281
+ FileUtils.mkdir_p bak
282
+ old.each do |name|
283
+ FileUtils.mv File.join(dir, name), bak
284
+ rescue Errno::ENOENT
285
+ next # another process moved it first
99
286
  end
287
+ Util.info "cache #{dir}: the old cache is moved to #{bak}"
100
288
  end
101
289
 
102
- # Returns all items
103
- # @return [Array<String>]
104
- def all(&block)
105
- Dir.glob("#{@dir}/**/*.{xml,yml,yaml}").map do |f|
106
- content = File.read(f, encoding: "utf-8")
107
- block ? yield(f, content) : content
290
+ # Drop the rows and documents of a flavor whose grammar changed since
291
+ # they were written.
292
+ def check_versions
293
+ @versions = @rows.get(VERSIONS_KEY) || {}
294
+ @versions.each do |flavor, hash|
295
+ processor = Registry.instance[:"relaton_#{flavor}"]
296
+ next if processor && processor.grammar_hash == hash
297
+
298
+ drop_flavor flavor
299
+ Util.info "cache #{dir}: version of `#{flavor}` is obsolete " \
300
+ "and its entries are cleared."
108
301
  end
109
302
  end
110
303
 
111
- # Delete item
112
- # @param key [String]
113
- def delete(key)
114
- file = filename key
115
- f = search_ext file
116
- return unless f
304
+ def drop_flavor(flavor)
305
+ @rows.adapter.transaction do
306
+ [@rows, @docs].each do |store|
307
+ store.each_key { |k| store.delete k if k.start_with? "#{flavor}/" }
308
+ end
309
+ @versions = @rows.update(VERSIONS_KEY) { |v| (v || {}).except flavor }
310
+ end
311
+ end
312
+
313
+ def record_version(bucket)
314
+ flavor = bucket.split("/", 2).first
315
+ return if flavor.start_with?("_") || @versions.key?(flavor)
117
316
 
118
- if File.extname(f) == ".redirect"
119
- code = redirect_code get(key)
120
- delete code if code
317
+ processor = Registry.instance[:"relaton_#{flavor}"] or return
318
+ hash = processor.grammar_hash
319
+ @versions = @rows.update(VERSIONS_KEY) do |v|
320
+ (v || {}).merge(flavor => hash)
121
321
  end
122
- File.delete f
123
322
  end
124
323
 
125
- # Check if version of the DB match to the gem grammar hash.
126
- # @param fdir [String] dir pathe to flover cache
127
- # @return [Boolean]
128
- def check_version?(fdir)
129
- version_dir = "#{fdir}/version"
130
- return false unless File.exist? version_dir
324
+ # @return [Array(String, Object)] the bucket and the resolved key
325
+ def resolve(key)
326
+ key = wrapped_key key if key.is_a? String
327
+ if key.is_a? String
328
+ ["#{STRING_FLAVOR}/#{key}", key]
329
+ else
330
+ [bucket_for(key), key]
331
+ end
332
+ end
131
333
 
132
- v = File.read version_dir, encoding: "utf-8"
133
- v.strip == self.class.grammar_hash(fdir)
334
+ # `ISO(ISO 19115-1)` -> the ISO processor's pubid for `ISO 19115-1`.
335
+ def wrapped_key(key)
336
+ match = key.match(WRAPPED_KEY) or return key
337
+ processor = Registry.instance.by_type(match[:prefix]) or return key
338
+ processor.cache_key(match[:code], nil, {}) || key
134
339
  end
135
340
 
136
- # if cached reference is undated, expire it after 60 days
137
- # @param key [String]
138
- # @param year [String]
139
- def valid_entry?(key, year)
140
- datestr = fetched key
141
- return false unless datestr
341
+ def bucket_for(pubid)
342
+ processor = Registry.instance.processor_by_pubid pubid
343
+ flavor = processor ? flavor_name(processor) : PUBID_FLAVOR
344
+ number = pubid.root.number.to_s
345
+ # No number (DOI, ISBN, the SI Brochure): 256 digest buckets, not one
346
+ # bucket for the whole flavor.
347
+ number = "~#{digest(pubid)[0, 2]}" if number.empty?
348
+ "#{flavor}/#{number}"
349
+ end
142
350
 
143
- date = Date.parse datestr
144
- year || Date.today - date < 60
351
+ def flavor_name(processor)
352
+ processor.short.to_s.delete_prefix "relaton_"
145
353
  end
146
354
 
147
- # Reads file by a key
148
- #
149
- # @param key [String]
150
- # @return [String, NilClass]
151
- def get(key)
152
- file = filename key
153
- return unless (f = search_ext(file))
355
+ def processor_for(bucket)
356
+ flavor = bucket.split("/", 2).first
357
+ if flavor == STRING_FLAVOR
358
+ Registry.instance.processor_by_ref bucket.split("/", 2).last
359
+ else
360
+ Registry.instance[:"relaton_#{flavor}"]
361
+ end
362
+ end
154
363
 
155
- File.read(f, encoding: "utf-8")
364
+ def read_bucket(bucket)
365
+ Array @rows.get(bucket)
156
366
  end
157
367
 
158
- # @param fdir [String] dir pathe to flover cache
159
- # @return [String]
160
- def self.grammar_hash(fdir)
161
- type = fdir.split("/").last
162
- Registry.instance.by_type(type)&.grammar_hash
368
+ def each_row
369
+ @rows.each_key do |bucket|
370
+ next if bucket == VERSIONS_KEY
371
+
372
+ read_bucket(bucket).each { |row| yield bucket, row }
373
+ end
163
374
  end
164
375
 
165
- private
376
+ # Replace or add a row. A document the replaced row pointed to goes
377
+ # when no row points to it any more.
378
+ def upsert(bucket, row) # rubocop:disable Metrics/AbcSize,Metrics/MethodLength
379
+ target = row_key row
380
+ replaced = nil
381
+ @rows.adapter.transaction do
382
+ @rows.update(bucket) do |old|
383
+ old = Array(old)
384
+ replaced = old.detect { |r| row_key(r) == target }
385
+ (old - [replaced]) << row
386
+ end
387
+ old_file = replaced&.dig("file")
388
+ remove_doc old_file, bucket if old_file && old_file != row["file"]
389
+ end
390
+ end
166
391
 
167
- # @param value [String]
168
- # @return [String]
169
- def filename(key)
170
- prefcode = key.downcase.match(/^(?<prefix>[^(]+)\((?<code>[^)]+)/)
171
- fn = if prefcode
172
- "#{prefcode[:prefix]}/#{prefcode[:code].gsub(/[:\s\/()]/,
173
- '_').squeeze('_')}"
174
- else
175
- key.gsub(/[-:\s]/, "_")
176
- end
177
- "#{@dir}/#{fn.sub(/(,|_$)/, '')}"
392
+ def delete_row(bucket, target)
393
+ @rows.adapter.transaction do
394
+ removed = nil
395
+ rows = @rows.update(bucket) do |old|
396
+ old = Array(old)
397
+ removed = old.detect { |row| row_key(row) == target }
398
+ old - [removed]
399
+ end
400
+ @rows.delete bucket if rows.empty?
401
+ remove_doc removed["file"], bucket if removed&.dig("file")
402
+ end
403
+ end
404
+
405
+ def valid_row?(row, year)
406
+ return false unless read_entry(row)
407
+
408
+ year || Date.today - Date.parse(row["fetched"]) < UNDATED_TTL
178
409
  end
179
410
 
180
411
  #
181
- # Checks if there is file with xml or txt extension and return filename with
182
- # the extension.
412
+ # Whether a row answers a key: the key is a subset of it (`===`), and
413
+ # every component the row adds is one of `allowed`.
183
414
  #
184
- # @param file [String]
185
- # @return [String, NilClass]
186
- def search_ext(file)
187
- if File.exist?("#{file}.#{@ext}")
188
- "#{file}.#{@ext}"
189
- elsif File.exist? "#{file}.notfound"
190
- "#{file}.notfound"
191
- elsif File.exist? "#{file}.redirect"
192
- "#{file}.redirect"
415
+ # @param key [Pubid::Identifier]
416
+ # @param id [Pubid::Identifier] the row's pubid
417
+ # @param row_id [Hash] the row's stored `id`
418
+ # @param allowed [Array<String>]
419
+ #
420
+ def subset_of?(key, id, row_id, allowed)
421
+ return false unless key === id # rubocop:disable Style/CaseEquality
422
+
423
+ (added_components(canonical(key), row_id) - allowed).empty?
424
+ end
425
+
426
+ # The names of the components `row` has and `query` has not, at any
427
+ # depth (a supplement's base, an adoption's adopted document).
428
+ def added_components(query, row) # rubocop:disable Metrics/AbcSize,Metrics/CyclomaticComplexity,Metrics/MethodLength,Metrics/PerceivedComplexity
429
+ case row
430
+ when Hash
431
+ row.flat_map do |name, value|
432
+ next [] if value.nil?
433
+ next [name] unless query.is_a?(Hash) && query.key?(name)
434
+
435
+ added_components query[name], value
436
+ end
437
+ when Array
438
+ return ["[]"] unless query.is_a?(Array) && query.size == row.size
439
+
440
+ row.each_with_index.flat_map { |v, i| added_components query[i], v }
441
+ else []
193
442
  end
194
443
  end
195
444
 
196
- # Set version of the DB to the gem grammar hash.
197
- # @param fdir [String] dir pathe to flover cache
198
- def set_version(fdir)
199
- file_version = "#{fdir}/version"
200
- unless File.exist? file_version
201
- file_safe_write file_version, self.class.grammar_hash(fdir)
445
+ def entry(key, status, file, fetched)
446
+ row = CacheEntry.new(status: status, file: file, fetched: fetched)
447
+ if key.is_a?(String) then row.key = key
448
+ else row.id = canonical(key)
202
449
  end
450
+ row.to_hash
203
451
  end
204
452
 
205
- # Return item's file name
206
- # @param key [String]
207
- # @return [String]
208
- def prefix(key)
209
- key.downcase.match(/^[^(]+(?=\()/).to_s
453
+ # A key as it is stored: a pubid's `to_hash` after a JSON round trip,
454
+ # so a fresh key compares equal to a stored one.
455
+ def canonical(key)
456
+ return key if key.is_a? String
457
+
458
+ JSON.parse JSON.generate(key.to_hash)
210
459
  end
211
460
 
212
- # Check if a file content is redirection
213
- #
214
- # @prarm value [String] file content
215
- # @return [String, NilClass] redirection code or nil
216
- def redirect_code(value)
217
- %r{redirection\s(?<code>.*)} =~ value
218
- code
461
+ def row_key(row)
462
+ row["id"] || row["key"]
463
+ end
464
+
465
+ def dated?(key)
466
+ !key.is_a?(String) && key.respond_to?(:year) && !key.year.nil?
467
+ end
468
+
469
+ # The newest row whose pubid the key is a subset of. A row whose
470
+ # document is gone does not count.
471
+ def subset_match(rows, key) # rubocop:disable Metrics/CyclomaticComplexity
472
+ rows.filter_map do |row|
473
+ next unless row["id"]
474
+
475
+ id = pubid_from row["id"]
476
+ next unless subset_of? key, id, row["id"], LANGUAGE_COMPONENTS
477
+ next if row["file"] && !@docs.exists?(row["file"])
478
+
479
+ [id, row]
480
+ end.max_by { |_, row| row["fetched"].to_s }
219
481
  end
220
482
 
221
- # @param file [String]
222
- # @content [String]
223
- def file_safe_write(file, content)
224
- File.open file, File::RDWR | File::CREAT, encoding: "UTF-8" do |f|
225
- Timeout.timeout(10) { f.flock File::LOCK_EX }
226
- f.write content
227
- f.flock File::LOCK_UN
483
+ def pubid_from(hash)
484
+ require "pubid"
485
+ ::Pubid.from_hash hash
486
+ end
487
+
488
+ def doc_name(bucket, key)
489
+ "#{bucket}/#{digest(key)[0, 16]}"
490
+ end
491
+
492
+ # A digest of the canonical key, independent of the hash order.
493
+ def digest(key)
494
+ Digest::SHA256.hexdigest JSON.generate(canonical_sorted(key))
495
+ end
496
+
497
+ def canonical_sorted(key)
498
+ value = canonical key
499
+ value.is_a?(Hash) ? deep_sort(value) : value
500
+ end
501
+
502
+ def deep_sort(value)
503
+ case value
504
+ when Hash then value.sort.to_h { |k, v| [k, deep_sort(v)] }
505
+ when Array then value.map { |v| deep_sort v }
506
+ else value
228
507
  end
229
508
  end
509
+
510
+ # Delete a document unless a row still points to it. A row that points
511
+ # to it lives in the deleted row's bucket or in the document's own one.
512
+ def remove_doc(file, bucket)
513
+ file_bucket = file.rpartition("/").first
514
+ used = [bucket, file_bucket].uniq.any? do |b|
515
+ read_bucket(b).any? { |row| row["file"] == file }
516
+ end
517
+ @docs.delete file unless used
518
+ end
519
+
520
+ def fetched_of(value)
521
+ date = if value.match? NOT_FOUND then value[/\d{4}-\d{2}-\d{2}/]
522
+ else value[%r{<fetched>\s*([^<\s]+)\s*</fetched>}, 1]
523
+ end
524
+ date || Date.today.to_s
525
+ end
230
526
  end
231
527
  end
232
528
  end