relaton 3.0.0.pre.alpha.4 → 3.0.0.pre.alpha.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. checksums.yaml +4 -4
  2. data/lib/relaton/3gpp/bibliography.rb +19 -6
  3. data/lib/relaton/3gpp/processor.rb +6 -0
  4. data/lib/relaton/adobe/processor.rb +9 -0
  5. data/lib/relaton/bib/converter/csl.rb +110 -0
  6. data/lib/relaton/bib/converter/ris.rb +104 -0
  7. data/lib/relaton/bib/converter/titles.rb +24 -0
  8. data/lib/relaton/bib/item_data.rb +23 -0
  9. data/lib/relaton/bib/model/item.rb +6 -6
  10. data/lib/relaton/bib/sanitizer.rb +71 -30
  11. data/lib/relaton/bib.rb +10 -0
  12. data/lib/relaton/bipm/bibliography.rb +24 -21
  13. data/lib/relaton/bipm/processor.rb +1 -0
  14. data/lib/relaton/bipm/rawdata_bipm_metrologia/affiliations.rb +6 -6
  15. data/lib/relaton/bipm/si_brochure_parser.rb +2 -2
  16. data/lib/relaton/bsi/bibliography.rb +26 -23
  17. data/lib/relaton/bsi/processor.rb +5 -0
  18. data/lib/relaton/calconnect/bibliography.rb +6 -6
  19. data/lib/relaton/calconnect/hit_collection.rb +4 -1
  20. data/lib/relaton/ccsds/bibliography.rb +16 -10
  21. data/lib/relaton/ccsds/hit_collection.rb +1 -1
  22. data/lib/relaton/ccsds/processor.rb +13 -0
  23. data/lib/relaton/cen/bibliography.rb +27 -24
  24. data/lib/relaton/cen/hit_collection.rb +1 -1
  25. data/lib/relaton/cen/processor.rb +5 -0
  26. data/lib/relaton/cen/scraper.rb +9 -9
  27. data/lib/relaton/cie/bibliography.rb +11 -9
  28. data/lib/relaton/cie/data_fetcher.rb +18 -18
  29. data/lib/relaton/cie/processor.rb +1 -0
  30. data/lib/relaton/cie/scrapper.rb +15 -8
  31. data/lib/relaton/cie.rb +1 -1
  32. data/lib/relaton/cloud.rb +127 -0
  33. data/lib/relaton/core/hit_collection.rb +6 -10
  34. data/lib/relaton/core/processor.rb +121 -0
  35. data/lib/relaton/core/request_error.rb +21 -3
  36. data/lib/relaton/db/cache.rb +474 -148
  37. data/lib/relaton/db/cache_entry.rb +42 -0
  38. data/lib/relaton/db/registry.rb +152 -0
  39. data/lib/relaton/db.rb +275 -102
  40. data/lib/relaton/doi/crossref.rb +23 -6
  41. data/lib/relaton/doi/processor.rb +1 -0
  42. data/lib/relaton/easc/processor.rb +1 -0
  43. data/lib/relaton/ecma/bibliography.rb +18 -13
  44. data/lib/relaton/ecma/data_fetcher.rb +1 -1
  45. data/lib/relaton/ecma/data_parser.rb +1 -1
  46. data/lib/relaton/ecma/edition_parser.rb +2 -2
  47. data/lib/relaton/ecma/memento_parser.rb +4 -4
  48. data/lib/relaton/ecma/standard_parser.rb +6 -6
  49. data/lib/relaton/etsi/bibliography.rb +8 -7
  50. data/lib/relaton/etsi/processor.rb +1 -0
  51. data/lib/relaton/gb/bibliography.rb +20 -12
  52. data/lib/relaton/gb/gb_scraper.rb +5 -5
  53. data/lib/relaton/gb/scraper.rb +16 -16
  54. data/lib/relaton/gb/sec_scraper.rb +8 -8
  55. data/lib/relaton/gb/t_scraper.rb +5 -5
  56. data/lib/relaton/gost/processor.rb +1 -0
  57. data/lib/relaton/iala/bibliography.rb +15 -12
  58. data/lib/relaton/iala/processor.rb +1 -0
  59. data/lib/relaton/iana/bibliography.rb +12 -12
  60. data/lib/relaton/iana/data_fetcher.rb +3 -3
  61. data/lib/relaton/iana/parser.rb +12 -5
  62. data/lib/relaton/iana/processor.rb +9 -0
  63. data/lib/relaton/iec/bibliography.rb +18 -7
  64. data/lib/relaton/iec/data_parser.rb +24 -9
  65. data/lib/relaton/iec/processor.rb +11 -0
  66. data/lib/relaton/iec.rb +1 -1
  67. data/lib/relaton/ieee/bibliography.rb +19 -13
  68. data/lib/relaton/ieee/data_fetcher.rb +1 -1
  69. data/lib/relaton/ieee/processor.rb +24 -0
  70. data/lib/relaton/ieee/rawbib_id_parser.rb +2 -2
  71. data/lib/relaton/ietf/bibliography.rb +11 -9
  72. data/lib/relaton/ietf/data_fetcher.rb +1 -1
  73. data/lib/relaton/ietf/processor.rb +1 -0
  74. data/lib/relaton/ietf/rfc/entry.rb +15 -19
  75. data/lib/relaton/ietf/scraper.rb +5 -4
  76. data/lib/relaton/iho/processor.rb +1 -0
  77. data/lib/relaton/index/file_io.rb +2 -2
  78. data/lib/relaton/index/pool.rb +4 -3
  79. data/lib/relaton/index/shard_source.rb +1 -1
  80. data/lib/relaton/index/type.rb +2 -5
  81. data/lib/relaton/isbn/open_library.rb +11 -7
  82. data/lib/relaton/isbn/processor.rb +10 -0
  83. data/lib/relaton/iso/bibliography.rb +11 -9
  84. data/lib/relaton/iso/data_parser.rb +2 -2
  85. data/lib/relaton/iso/processor.rb +5 -0
  86. data/lib/relaton/iso/scraper.rb +20 -20
  87. data/lib/relaton/itu/bibliography.rb +107 -16
  88. data/lib/relaton/itu/data_crawler_r.rb +2 -2
  89. data/lib/relaton/itu/hit_collection.rb +64 -52
  90. data/lib/relaton/itu/processor.rb +1 -0
  91. data/lib/relaton/itu/scraper.rb +13 -3
  92. data/lib/relaton/itu.rb +0 -2
  93. data/lib/relaton/jis/bibliography.rb +22 -11
  94. data/lib/relaton/jis/data_fetcher.rb +4 -4
  95. data/lib/relaton/jis/processor.rb +1 -0
  96. data/lib/relaton/jis/scraper.rb +8 -8
  97. data/lib/relaton/nist/bibliography.rb +25 -21
  98. data/lib/relaton/oasis/bibliography.rb +16 -13
  99. data/lib/relaton/oasis/browser_agent.rb +2 -2
  100. data/lib/relaton/oasis/data_parser.rb +5 -5
  101. data/lib/relaton/oasis/data_parser_utils.rb +2 -2
  102. data/lib/relaton/oasis/data_part_parser.rb +7 -7
  103. data/lib/relaton/ogc/bibliography.rb +15 -14
  104. data/lib/relaton/ogc/hit_collection.rb +9 -7
  105. data/lib/relaton/ogc/processor.rb +13 -0
  106. data/lib/relaton/oiml/processor.rb +1 -0
  107. data/lib/relaton/omg/bibliography.rb +9 -9
  108. data/lib/relaton/omg/scraper.rb +14 -13
  109. data/lib/relaton/omg.rb +1 -1
  110. data/lib/relaton/plateau/bibliography.rb +9 -7
  111. data/lib/relaton/plateau/hit_collection.rb +2 -1
  112. data/lib/relaton/plateau/processor.rb +1 -0
  113. data/lib/relaton/un/bibliography.rb +21 -10
  114. data/lib/relaton/un/processor.rb +1 -0
  115. data/lib/relaton/version.rb +1 -1
  116. data/lib/relaton/w3c/processor.rb +1 -0
  117. data/lib/relaton/xsf/bibliography.rb +11 -6
  118. data/lib/relaton.rb +13 -0
  119. metadata +60 -14
  120. data/lib/relaton/itu/pubid.rb +0 -199
@@ -1,24 +1,64 @@
1
1
  require "fileutils"
2
- require "timeout"
2
+ require "digest"
3
+ require "json"
4
+ require "date"
5
+ require "lutaml/store"
6
+ require_relative "cache_entry"
3
7
 
4
8
  module Relaton
5
9
  class Db
10
+ #
11
+ # The document cache of one directory, on two lutaml-store FileSystem
12
+ # stores under `<dir>/v2/`:
13
+ #
14
+ # - `rows/` holds the index. One store key per bucket, `<flavor>/<root
15
+ # number>` (e.g. `iso/19115`), whose value is the list of the bucket's
16
+ # CacheEntry rows. A lookup reads one small bucket, and a write is one
17
+ # atomic `update` of it.
18
+ # - `docs/` holds the XML documents. A document is named after the row it
19
+ # was stored for, and several rows can point to it.
20
+ #
21
+ # A key is a parsed pubid, a wrapped string (`ISO(ISO 19115-1)`, parsed
22
+ # through the prefix's processor), or a plain string that no processor
23
+ # owns. Locking and atomic writes come from lutaml-store.
24
+ #
6
25
  class Cache
26
+ LAYOUT = "v2".freeze
27
+ VERSIONS_KEY = "_versions".freeze
28
+ STRING_FLAVOR = "_key".freeze
29
+ PUBID_FLAVOR = "_pubid".freeze
30
+ # The legacy string form of a not-found entry (`not_found <date>`).
31
+ NOT_FOUND = /\Anot_found/
32
+ # Days an undated entry, or any not_found entry, stays valid.
33
+ UNDATED_TTL = 60
34
+ WRAPPED_KEY = /\A(?<prefix>[^(\s]+)\((?<code>.+)\)\z/m
35
+ # What a row may add to a dated key and still answer it: `===` also
36
+ # reads an omitted `part` as "any part" (`ISO 19115:2003 ===
37
+ # ISO 19115-1:2003`), which names another document.
38
+ LANGUAGE_COMPONENTS = %w[language languages].freeze
39
+ # What a row may add to a key and still be one of its editions.
40
+ EDITION_COMPONENTS = (%w[year date month] + LANGUAGE_COMPONENTS).freeze
41
+
7
42
  # @return [String]
8
43
  attr_reader :dir
9
44
 
10
- # @param dir [String] DB directory
11
- def initialize(dir, ext = "xml")
45
+ # @param dir [String] cache directory
46
+ def initialize(dir)
12
47
  @dir = dir
13
- @ext = ext
14
- FileUtils::mkdir_p dir
48
+ # A stale file at the cache path (e.g. a placeholder left by cache
49
+ # migration flows) makes every write raise Errno::EEXIST; replace
50
+ # it with the cache directory.
51
+ FileUtils.rm_rf dir if File.exist?(dir) && !File.directory?(dir)
52
+ archive_old_layout
53
+ open_stores
54
+ check_versions
15
55
  end
16
56
 
17
- # Move caches to anothe dir
57
+ # Move the cache to another directory.
18
58
  # @param new_dir [String, nil]
19
- # @return [String, nil]
59
+ # @return [String, nil] the new directory
20
60
  def mv(new_dir)
21
- return unless new_dir && @ext == "xml"
61
+ return unless new_dir
22
62
 
23
63
  if File.exist? new_dir
24
64
  Util.info "target directory exists `#{new_dir}`"
@@ -27,205 +67,491 @@ module Relaton
27
67
 
28
68
  FileUtils.mv dir, new_dir
29
69
  @dir = new_dir
70
+ open_stores
71
+ @dir
30
72
  end
31
73
 
32
- # Clear database
74
+ # Remove every row and document.
33
75
  def clear
34
- FileUtils.rm_rf Dir.glob "#{dir}/*"
76
+ @rows.clear
77
+ @docs.clear
78
+ @versions = {}
79
+ end
80
+
81
+ # Read the entry for a key: the row with this key, else, for a dated
82
+ # pubid, a row the key is a subset of.
83
+ #
84
+ # @param key [Pubid::Identifier, String]
85
+ # @return [String, Relaton::Db::NotFound, nil] document XML or not found
86
+ def read(key)
87
+ found = lookup(key)
88
+ found && read_entry(found.last)
89
+ end
90
+
91
+ # The entry for a key as a string, a not-found entry as
92
+ # `not_found <date>`.
93
+ #
94
+ # @param key [Pubid::Identifier, String]
95
+ # @return [String, nil]
96
+ def [](key)
97
+ read(key)&.to_s
35
98
  end
99
+ alias get []
36
100
 
37
- # Save item
38
- # @param key [String]
39
- # @param value [String] Bibitem xml serialization
101
+ # @param key [Pubid::Identifier, String]
102
+ # @param value [String, Relaton::Db::NotFound, nil] document XML, a
103
+ # not-found entry (also as `not_found <date>`); nil deletes the row
40
104
  def []=(key, value)
41
- if value.nil?
42
- delete key
43
- return
105
+ store key, value
106
+ end
107
+
108
+ #
109
+ # Save a document for a query key. When the document's own identifier
110
+ # (`item_key`) differs from the query, both get a row, and both rows
111
+ # point to one document file.
112
+ #
113
+ # @param key [Pubid::Identifier, String] query key
114
+ # @param value [String, Relaton::Db::NotFound, nil] document XML, a
115
+ # not-found entry (also as `not_found <date>`); nil deletes the row
116
+ # @param item_key [Pubid::Identifier, String, nil] the document's key
117
+ # @return [String, Relaton::Db::NotFound, nil] value
118
+ #
119
+ def store(key, value, item_key: nil) # rubocop:disable Metrics/AbcSize,Metrics/MethodLength
120
+ return delete(key) if value.nil?
121
+
122
+ bucket, key = resolve key
123
+ value = coerce value
124
+ fetched = fetched_of value
125
+ record_version bucket
126
+ if value.is_a? NotFound
127
+ upsert bucket, new_row(key, CacheEntry::NOT_FOUND, nil, fetched)
128
+ return value
44
129
  end
45
130
 
46
- prefix_dir = "#{@dir}/#{prefix(key)}"
47
- FileUtils::mkdir_p prefix_dir
48
- set_version prefix_dir
49
- file_safe_write "#{filename(key)}.#{ext(value)}", value
131
+ item_bucket, item_key = item_key ? resolve(item_key) : [bucket, key]
132
+ file = doc_name item_bucket, item_key
133
+ record_version item_bucket
134
+ @rows.adapter.transaction do
135
+ @docs.set file, value
136
+ upsert item_bucket, new_row(item_key, CacheEntry::DOC, file, fetched)
137
+ unless canonical(key) == canonical(item_key)
138
+ upsert bucket, new_row(key, CacheEntry::DOC, file, fetched)
139
+ end
140
+ end
141
+ value
50
142
  end
51
143
 
52
- # @param value [String]
53
- # @return [String]
54
- def ext(value)
55
- case value
56
- when /^not_found/ then "notfound"
57
- when /^redirection/ then "redirect"
58
- else @ext
59
- end
144
+ # Delete the row of a key. Its document goes too, when no other row
145
+ # points to it.
146
+ # @param key [Pubid::Identifier, String]
147
+ def delete(key)
148
+ bucket, key = resolve key
149
+ target = canonical key
150
+ return unless read_bucket(bucket).any? { |row| row_key(row) == target }
151
+
152
+ delete_row bucket, target
60
153
  end
61
154
 
62
- # Read item
63
- # @param key [String]
64
- # @return [String]
65
- def [](key)
66
- value = get(key)
67
- if (code = redirect_code value)
68
- self[code]
69
- else
70
- value
155
+ #
156
+ # Delete the row a lookup of the key finds (its own, or the one a dated
157
+ # key is a subset of) when it is no longer valid.
158
+ #
159
+ # @param key [Pubid::Identifier, String]
160
+ # @param year [String, nil]
161
+ #
162
+ def expire(key, year)
163
+ found = lookup(key) or return
164
+ return if valid_row?(found.last, year)
165
+
166
+ delete_row resolve(found.first).first, row_key(found.last)
167
+ end
168
+
169
+ #
170
+ # Save the row of a key, and its document, from another cache. A row
171
+ # the cache already holds is kept: a local cache keeps its own value
172
+ # for a key the global cache also holds (local over global).
173
+ #
174
+ # @param key [Pubid::Identifier, String]
175
+ # @param other [Relaton::Db::Cache]
176
+ #
177
+ def clone_entry(key, other)
178
+ return if lookup(key)
179
+
180
+ found = other.lookup(key) or return
181
+ row_id, row = found
182
+ store row_id, other.read_entry(row)
183
+ end
184
+
185
+ # @param key [Pubid::Identifier, String]
186
+ # @return [String, nil] the date the entry was fetched
187
+ def fetched(key)
188
+ lookup(key)&.last&.dig("fetched")
189
+ end
190
+
191
+ # An undated entry expires after 60 days, a dated document never. A
192
+ # not_found entry expires after 60 days even for a dated query: the
193
+ # document can be published later.
194
+ # @param key [Pubid::Identifier, String]
195
+ # @param year [String, nil]
196
+ def valid_entry?(key, year)
197
+ found = lookup(key)
198
+ found ? valid_row?(found.last, year) : false
199
+ end
200
+
201
+ #
202
+ # The cached editions of the key: documents whose pubid the key is a
203
+ # subset of (`key === id`) and that add only a year, a date or a
204
+ # language to it. For a query that selects among editions (e.g. by
205
+ # publication date).
206
+ #
207
+ # @param key [Pubid::Identifier]
208
+ # @return [Array<Array(Pubid::Identifier, String)>]
209
+ #
210
+ def candidates(key) # rubocop:disable Metrics/AbcSize,Metrics/CyclomaticComplexity
211
+ bucket, key = resolve key
212
+ return [] if key.is_a? String
213
+
214
+ read_bucket(bucket).filter_map do |row|
215
+ next if row["status"] != CacheEntry::DOC || !row["id"]
216
+
217
+ id = pubid_from row["id"]
218
+ next unless subset_of? key, id, row["id"], EDITION_COMPONENTS
219
+
220
+ xml = @docs.get row["file"]
221
+ [id, xml] if xml
71
222
  end
72
223
  end
73
224
 
74
225
  #
75
- # Save entry from cache of `db` to this cache.
226
+ # Every cached document once.
76
227
  #
77
- # @param [String] key key of the entry
78
- # @param [Relaton::Db] db database
228
+ # @yieldparam processor [Relaton::Core::Processor, nil] the owning flavor
229
+ # @yieldparam xml [String]
230
+ # @return [Array] the documents, or the block results
79
231
  #
80
- def clone_entry(key, db)
81
- self[key] ||= db.get(key)
82
- if (code = redirect_code get(key))
83
- clone_entry code, db
232
+ def all
233
+ files = {}
234
+ each_row do |bucket, row|
235
+ files[row["file"]] ||= bucket if row["file"]
236
+ end
237
+ files.filter_map do |file, bucket|
238
+ xml = @docs.get(file) or next
239
+ block_given? ? yield(processor_for(bucket), xml) : xml
84
240
  end
85
241
  end
86
242
 
87
- # Return fetched date
88
- # @param key [String]
89
- # @return [String]
90
- def fetched(key)
91
- value = self[key]
92
- return unless value
243
+ # @return [Array<Hash>] every row
244
+ def rows
245
+ list = []
246
+ each_row { |_, row| list << row }
247
+ list
248
+ end
93
249
 
94
- if value.match?(/^not_found/)
95
- value.match(/\d{4}-\d{2}-\d{2}/).to_s
96
- else
97
- doc = Nokogiri::XML value
98
- doc.at("/bibitem/fetched|bibdata/fetched")&.text
250
+ # The row of a key, with the key it is stored under.
251
+ # @param key [Pubid::Identifier, String]
252
+ # @return [Array(Object, Hash), nil]
253
+ def lookup(key)
254
+ bucket, key = resolve key
255
+ rows = read_bucket bucket
256
+ target = canonical key
257
+ row = rows.detect { |r| row_key(r) == target }
258
+ return [key, row] if row
259
+ return unless dated? key
260
+
261
+ subset_match rows, key
262
+ end
263
+
264
+ # @param row [Hash]
265
+ # @return [String, Relaton::Db::NotFound, nil] document XML or not found
266
+ def read_entry(row)
267
+ if row["status"] == CacheEntry::NOT_FOUND
268
+ return NotFound.new(fetched: row["fetched"])
269
+ end
270
+
271
+ @docs.get row["file"]
272
+ end
273
+
274
+ private
275
+
276
+ def open_stores
277
+ base = File.join dir, LAYOUT
278
+ @rows = new_store File.join(base, "rows"), ".json"
279
+ @docs = new_store File.join(base, "docs"), ".xml"
280
+ end
281
+
282
+ def new_store(path, extension)
283
+ Lutaml::Store::BasicStore.new(
284
+ adapter_type: :filesystem,
285
+ adapter_options: { path: path, extension: extension,
286
+ integrity_checks: false },
287
+ cache: { enabled: false },
288
+ )
289
+ end
290
+
291
+ # A cache written by the file-per-key layout (before v2) is moved to
292
+ # `<dir>-v1.bak`, never deleted.
293
+ def archive_old_layout # rubocop:disable Metrics/AbcSize,Metrics/MethodLength
294
+ return unless Dir.exist? dir
295
+
296
+ old = Dir.children(dir) - [LAYOUT]
297
+ bak = "#{dir}-v1.bak"
298
+ # Once only: an older relaton that shares the directory writes the old
299
+ # layout again, and v2 ignores it.
300
+ return if old.empty? || File.exist?(bak)
301
+
302
+ FileUtils.mkdir_p bak
303
+ old.each do |name|
304
+ FileUtils.mv File.join(dir, name), bak
305
+ rescue Errno::ENOENT
306
+ next # another process moved it first
99
307
  end
308
+ Util.info "cache #{dir}: the old cache is moved to #{bak}"
100
309
  end
101
310
 
102
- # Returns all items
103
- # @return [Array<String>]
104
- def all(&block)
105
- Dir.glob("#{@dir}/**/*.{xml,yml,yaml}").map do |f|
106
- content = File.read(f, encoding: "utf-8")
107
- block ? yield(f, content) : content
311
+ # Drop the rows and documents of a flavor whose grammar changed since
312
+ # they were written.
313
+ def check_versions
314
+ @versions = @rows.get(VERSIONS_KEY) || {}
315
+ @versions.each do |flavor, hash|
316
+ processor = Registry.instance[:"relaton_#{flavor}"]
317
+ next if processor && processor.grammar_hash == hash
318
+
319
+ drop_flavor flavor
320
+ Util.info "cache #{dir}: version of `#{flavor}` is obsolete " \
321
+ "and its entries are cleared."
108
322
  end
109
323
  end
110
324
 
111
- # Delete item
112
- # @param key [String]
113
- def delete(key)
114
- file = filename key
115
- f = search_ext file
116
- return unless f
325
+ def drop_flavor(flavor)
326
+ @rows.adapter.transaction do
327
+ [@rows, @docs].each do |store|
328
+ store.each_key { |k| store.delete k if k.start_with? "#{flavor}/" }
329
+ end
330
+ @versions = @rows.update(VERSIONS_KEY) { |v| (v || {}).except flavor }
331
+ end
332
+ end
117
333
 
118
- if File.extname(f) == ".redirect"
119
- code = redirect_code get(key)
120
- delete code if code
334
+ def record_version(bucket)
335
+ flavor = bucket.split("/", 2).first
336
+ return if flavor.start_with?("_") || @versions.key?(flavor)
337
+
338
+ processor = Registry.instance[:"relaton_#{flavor}"] or return
339
+ hash = processor.grammar_hash
340
+ @versions = @rows.update(VERSIONS_KEY) do |v|
341
+ (v || {}).merge(flavor => hash)
121
342
  end
122
- File.delete f
123
343
  end
124
344
 
125
- # Check if version of the DB match to the gem grammar hash.
126
- # @param fdir [String] dir pathe to flover cache
127
- # @return [Boolean]
128
- def check_version?(fdir)
129
- version_dir = "#{fdir}/version"
130
- return false unless File.exist? version_dir
345
+ # @return [Array(String, Object)] the bucket and the resolved key
346
+ def resolve(key)
347
+ key = wrapped_key key if key.is_a? String
348
+ if key.is_a? String
349
+ ["#{STRING_FLAVOR}/#{key}", key]
350
+ else
351
+ [bucket_for(key), key]
352
+ end
353
+ end
131
354
 
132
- v = File.read version_dir, encoding: "utf-8"
133
- v.strip == self.class.grammar_hash(fdir)
355
+ # `ISO(ISO 19115-1)` -> the ISO processor's pubid for `ISO 19115-1`.
356
+ def wrapped_key(key)
357
+ match = key.match(WRAPPED_KEY) or return key
358
+ processor = Registry.instance.by_type(match[:prefix]) or return key
359
+ processor.cache_key(match[:code], nil, {}) || key
134
360
  end
135
361
 
136
- # if cached reference is undated, expire it after 60 days
137
- # @param key [String]
138
- # @param year [String]
139
- def valid_entry?(key, year)
140
- datestr = fetched key
141
- return false unless datestr
362
+ def bucket_for(pubid)
363
+ processor = Registry.instance.processor_by_pubid pubid
364
+ flavor = processor ? flavor_name(processor) : PUBID_FLAVOR
365
+ number = pubid.root.number.to_s
366
+ # No number (DOI, ISBN, the SI Brochure): 256 digest buckets, not one
367
+ # bucket for the whole flavor.
368
+ number = "~#{digest(pubid)[0, 2]}" if number.empty?
369
+ "#{flavor}/#{number}"
370
+ end
142
371
 
143
- date = Date.parse datestr
144
- year || Date.today - date < 60
372
+ def flavor_name(processor)
373
+ processor.short.to_s.delete_prefix "relaton_"
145
374
  end
146
375
 
147
- # Reads file by a key
148
- #
149
- # @param key [String]
150
- # @return [String, NilClass]
151
- def get(key)
152
- file = filename key
153
- return unless (f = search_ext(file))
376
+ def processor_for(bucket)
377
+ flavor = bucket.split("/", 2).first
378
+ if flavor == STRING_FLAVOR
379
+ Registry.instance.processor_by_ref bucket.split("/", 2).last
380
+ else
381
+ Registry.instance[:"relaton_#{flavor}"]
382
+ end
383
+ end
154
384
 
155
- File.read(f, encoding: "utf-8")
385
+ def read_bucket(bucket)
386
+ Array @rows.get(bucket)
156
387
  end
157
388
 
158
- # @param fdir [String] dir pathe to flover cache
159
- # @return [String]
160
- def self.grammar_hash(fdir)
161
- type = fdir.split("/").last
162
- Registry.instance.by_type(type)&.grammar_hash
389
+ def each_row
390
+ @rows.each_key do |bucket|
391
+ next if bucket == VERSIONS_KEY
392
+
393
+ read_bucket(bucket).each { |row| yield bucket, row }
394
+ end
163
395
  end
164
396
 
165
- private
397
+ # Replace or add a row. A document the replaced row pointed to goes
398
+ # when no row points to it any more.
399
+ def upsert(bucket, row) # rubocop:disable Metrics/AbcSize,Metrics/MethodLength
400
+ target = row_key row
401
+ replaced = nil
402
+ @rows.adapter.transaction do
403
+ @rows.update(bucket) do |old|
404
+ old = Array(old)
405
+ replaced = old.detect { |r| row_key(r) == target }
406
+ (old - [replaced]) << row
407
+ end
408
+ old_file = replaced&.dig("file")
409
+ remove_doc old_file, bucket if old_file && old_file != row["file"]
410
+ end
411
+ end
166
412
 
167
- # @param value [String]
168
- # @return [String]
169
- def filename(key)
170
- prefcode = key.downcase.match(/^(?<prefix>[^(]+)\((?<code>[^)]+)/)
171
- fn = if prefcode
172
- "#{prefcode[:prefix]}/#{prefcode[:code].gsub(/[:\s\/()]/,
173
- '_').squeeze('_')}"
174
- else
175
- key.gsub(/[-:\s]/, "_")
176
- end
177
- "#{@dir}/#{fn.sub(/(,|_$)/, '')}"
413
+ def delete_row(bucket, target)
414
+ @rows.adapter.transaction do
415
+ removed = nil
416
+ rows = @rows.update(bucket) do |old|
417
+ old = Array(old)
418
+ removed = old.detect { |row| row_key(row) == target }
419
+ old - [removed]
420
+ end
421
+ @rows.delete bucket if rows.empty?
422
+ remove_doc removed["file"], bucket if removed&.dig("file")
423
+ end
424
+ end
425
+
426
+ def valid_row?(row, year)
427
+ return false unless read_entry(row)
428
+
429
+ return year if year && row["status"] == CacheEntry::DOC
430
+
431
+ Date.today - Date.parse(row["fetched"]) < UNDATED_TTL
178
432
  end
179
433
 
180
434
  #
181
- # Checks if there is file with xml or txt extension and return filename with
182
- # the extension.
435
+ # Whether a row answers a key: the key is a subset of it (`===`), and
436
+ # every component the row adds is one of `allowed`.
183
437
  #
184
- # @param file [String]
185
- # @return [String, NilClass]
186
- def search_ext(file)
187
- if File.exist?("#{file}.#{@ext}")
188
- "#{file}.#{@ext}"
189
- elsif File.exist? "#{file}.notfound"
190
- "#{file}.notfound"
191
- elsif File.exist? "#{file}.redirect"
192
- "#{file}.redirect"
438
+ # @param key [Pubid::Identifier]
439
+ # @param id [Pubid::Identifier] the row's pubid
440
+ # @param row_id [Hash] the row's stored `id`
441
+ # @param allowed [Array<String>]
442
+ #
443
+ def subset_of?(key, id, row_id, allowed)
444
+ return false unless key === id # rubocop:disable Style/CaseEquality
445
+
446
+ (added_components(canonical(key), row_id) - allowed).empty?
447
+ end
448
+
449
+ # The names of the components `row` has and `query` has not, at any
450
+ # depth (a supplement's base, an adoption's adopted document).
451
+ def added_components(query, row) # rubocop:disable Metrics/AbcSize,Metrics/CyclomaticComplexity,Metrics/MethodLength,Metrics/PerceivedComplexity
452
+ case row
453
+ when Hash
454
+ row.flat_map do |name, value|
455
+ next [] if value.nil?
456
+ next [name] unless query.is_a?(Hash) && query.key?(name)
457
+
458
+ added_components query[name], value
459
+ end
460
+ when Array
461
+ return ["[]"] unless query.is_a?(Array) && query.size == row.size
462
+
463
+ row.each_with_index.flat_map { |v, i| added_components query[i], v }
464
+ else []
193
465
  end
194
466
  end
195
467
 
196
- # Set version of the DB to the gem grammar hash.
197
- # @param fdir [String] dir pathe to flover cache
198
- def set_version(fdir)
199
- file_version = "#{fdir}/version"
200
- unless File.exist? file_version
201
- file_safe_write file_version, self.class.grammar_hash(fdir)
468
+ def new_row(key, status, file, fetched)
469
+ row = CacheEntry.new(status: status, file: file, fetched: fetched)
470
+ if key.is_a?(String) then row.key = key
471
+ else row.id = canonical(key)
202
472
  end
473
+ row.to_hash
203
474
  end
204
475
 
205
- # Return item's file name
206
- # @param key [String]
207
- # @return [String]
208
- def prefix(key)
209
- key.downcase.match(/^[^(]+(?=\()/).to_s
476
+ # A key as it is stored: a pubid's `to_hash` after a JSON round trip,
477
+ # so a fresh key compares equal to a stored one.
478
+ def canonical(key)
479
+ return key if key.is_a? String
480
+
481
+ JSON.parse JSON.generate(key.to_hash)
210
482
  end
211
483
 
212
- # Check if a file content is redirection
213
- #
214
- # @prarm value [String] file content
215
- # @return [String, NilClass] redirection code or nil
216
- def redirect_code(value)
217
- %r{redirection\s(?<code>.*)} =~ value
218
- code
484
+ def row_key(row)
485
+ row["id"] || row["key"]
486
+ end
487
+
488
+ def dated?(key)
489
+ !key.is_a?(String) && key.respond_to?(:year) && !key.year.nil?
490
+ end
491
+
492
+ # The newest row whose pubid the key is a subset of. A row whose
493
+ # document is gone does not count.
494
+ def subset_match(rows, key) # rubocop:disable Metrics/CyclomaticComplexity
495
+ rows.filter_map do |row|
496
+ next unless row["id"]
497
+
498
+ id = pubid_from row["id"]
499
+ next unless subset_of? key, id, row["id"], LANGUAGE_COMPONENTS
500
+ next if row["file"] && !@docs.exists?(row["file"])
501
+
502
+ [id, row]
503
+ end.max_by { |_, row| row["fetched"].to_s }
504
+ end
505
+
506
+ def pubid_from(hash)
507
+ require "pubid"
508
+ ::Pubid.from_hash hash
509
+ end
510
+
511
+ def doc_name(bucket, key)
512
+ "#{bucket}/#{digest(key)[0, 16]}"
513
+ end
514
+
515
+ # A digest of the canonical key, independent of the hash order.
516
+ def digest(key)
517
+ Digest::SHA256.hexdigest JSON.generate(canonical_sorted(key))
518
+ end
519
+
520
+ def canonical_sorted(key)
521
+ value = canonical key
522
+ value.is_a?(Hash) ? deep_sort(value) : value
523
+ end
524
+
525
+ def deep_sort(value)
526
+ case value
527
+ when Hash then value.sort.to_h { |k, v| [k, deep_sort(v)] }
528
+ when Array then value.map { |v| deep_sort v }
529
+ else value
530
+ end
219
531
  end
220
532
 
221
- # @param file [String]
222
- # @content [String]
223
- def file_safe_write(file, content)
224
- File.open file, File::RDWR | File::CREAT, encoding: "UTF-8" do |f|
225
- Timeout.timeout(10) { f.flock File::LOCK_EX }
226
- f.write content
227
- f.flock File::LOCK_UN
533
+ # Delete a document unless a row still points to it. A row that points
534
+ # to it lives in the deleted row's bucket or in the document's own one.
535
+ def remove_doc(file, bucket)
536
+ file_bucket = file.rpartition("/").first
537
+ used = [bucket, file_bucket].uniq.any? do |b|
538
+ read_bucket(b).any? { |row| row["file"] == file }
228
539
  end
540
+ @docs.delete file unless used
541
+ end
542
+
543
+ # A legacy `not_found <date>` string becomes a NotFound value.
544
+ def coerce(value)
545
+ return value unless value.is_a?(String) && value.match?(NOT_FOUND)
546
+
547
+ NotFound.new(fetched: value[/\d{4}-\d{2}-\d{2}/] || Date.today.to_s)
548
+ end
549
+
550
+ def fetched_of(value)
551
+ date = if value.is_a? NotFound then value.fetched
552
+ else value[%r{<fetched>\s*([^<\s]+)\s*</fetched>}, 1]
553
+ end
554
+ date || Date.today.to_s
229
555
  end
230
556
  end
231
557
  end