relaton 3.0.0.pre.alpha.4 → 3.0.0.pre.alpha.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/lib/relaton/3gpp/bibliography.rb +19 -6
- data/lib/relaton/3gpp/processor.rb +6 -0
- data/lib/relaton/adobe/processor.rb +9 -0
- data/lib/relaton/bib/converter/csl.rb +110 -0
- data/lib/relaton/bib/converter/ris.rb +104 -0
- data/lib/relaton/bib/converter/titles.rb +24 -0
- data/lib/relaton/bib/item_data.rb +23 -0
- data/lib/relaton/bib/model/item.rb +6 -6
- data/lib/relaton/bib/sanitizer.rb +71 -30
- data/lib/relaton/bib.rb +10 -0
- data/lib/relaton/bipm/bibliography.rb +24 -21
- data/lib/relaton/bipm/processor.rb +1 -0
- data/lib/relaton/bipm/rawdata_bipm_metrologia/affiliations.rb +6 -6
- data/lib/relaton/bipm/si_brochure_parser.rb +2 -2
- data/lib/relaton/bsi/bibliography.rb +26 -23
- data/lib/relaton/bsi/processor.rb +5 -0
- data/lib/relaton/calconnect/bibliography.rb +6 -6
- data/lib/relaton/calconnect/hit_collection.rb +4 -1
- data/lib/relaton/ccsds/bibliography.rb +16 -10
- data/lib/relaton/ccsds/hit_collection.rb +1 -1
- data/lib/relaton/ccsds/processor.rb +13 -0
- data/lib/relaton/cen/bibliography.rb +27 -24
- data/lib/relaton/cen/hit_collection.rb +1 -1
- data/lib/relaton/cen/processor.rb +5 -0
- data/lib/relaton/cen/scraper.rb +9 -9
- data/lib/relaton/cie/bibliography.rb +11 -9
- data/lib/relaton/cie/data_fetcher.rb +18 -18
- data/lib/relaton/cie/processor.rb +1 -0
- data/lib/relaton/cie/scrapper.rb +15 -8
- data/lib/relaton/cie.rb +1 -1
- data/lib/relaton/cloud.rb +127 -0
- data/lib/relaton/core/hit_collection.rb +6 -10
- data/lib/relaton/core/processor.rb +121 -0
- data/lib/relaton/core/request_error.rb +21 -3
- data/lib/relaton/db/cache.rb +474 -148
- data/lib/relaton/db/cache_entry.rb +42 -0
- data/lib/relaton/db/registry.rb +152 -0
- data/lib/relaton/db.rb +275 -102
- data/lib/relaton/doi/crossref.rb +23 -6
- data/lib/relaton/doi/processor.rb +1 -0
- data/lib/relaton/easc/processor.rb +1 -0
- data/lib/relaton/ecma/bibliography.rb +18 -13
- data/lib/relaton/ecma/data_fetcher.rb +1 -1
- data/lib/relaton/ecma/data_parser.rb +1 -1
- data/lib/relaton/ecma/edition_parser.rb +2 -2
- data/lib/relaton/ecma/memento_parser.rb +4 -4
- data/lib/relaton/ecma/standard_parser.rb +6 -6
- data/lib/relaton/etsi/bibliography.rb +8 -7
- data/lib/relaton/etsi/processor.rb +1 -0
- data/lib/relaton/gb/bibliography.rb +20 -12
- data/lib/relaton/gb/gb_scraper.rb +5 -5
- data/lib/relaton/gb/scraper.rb +16 -16
- data/lib/relaton/gb/sec_scraper.rb +8 -8
- data/lib/relaton/gb/t_scraper.rb +5 -5
- data/lib/relaton/gost/processor.rb +1 -0
- data/lib/relaton/iala/bibliography.rb +15 -12
- data/lib/relaton/iala/processor.rb +1 -0
- data/lib/relaton/iana/bibliography.rb +12 -12
- data/lib/relaton/iana/data_fetcher.rb +3 -3
- data/lib/relaton/iana/parser.rb +12 -5
- data/lib/relaton/iana/processor.rb +9 -0
- data/lib/relaton/iec/bibliography.rb +18 -7
- data/lib/relaton/iec/data_parser.rb +24 -9
- data/lib/relaton/iec/processor.rb +11 -0
- data/lib/relaton/iec.rb +1 -1
- data/lib/relaton/ieee/bibliography.rb +19 -13
- data/lib/relaton/ieee/data_fetcher.rb +1 -1
- data/lib/relaton/ieee/processor.rb +24 -0
- data/lib/relaton/ieee/rawbib_id_parser.rb +2 -2
- data/lib/relaton/ietf/bibliography.rb +11 -9
- data/lib/relaton/ietf/data_fetcher.rb +1 -1
- data/lib/relaton/ietf/processor.rb +1 -0
- data/lib/relaton/ietf/rfc/entry.rb +15 -19
- data/lib/relaton/ietf/scraper.rb +5 -4
- data/lib/relaton/iho/processor.rb +1 -0
- data/lib/relaton/index/file_io.rb +2 -2
- data/lib/relaton/index/pool.rb +4 -3
- data/lib/relaton/index/shard_source.rb +1 -1
- data/lib/relaton/index/type.rb +2 -5
- data/lib/relaton/isbn/open_library.rb +11 -7
- data/lib/relaton/isbn/processor.rb +10 -0
- data/lib/relaton/iso/bibliography.rb +11 -9
- data/lib/relaton/iso/data_parser.rb +2 -2
- data/lib/relaton/iso/processor.rb +5 -0
- data/lib/relaton/iso/scraper.rb +20 -20
- data/lib/relaton/itu/bibliography.rb +107 -16
- data/lib/relaton/itu/data_crawler_r.rb +2 -2
- data/lib/relaton/itu/hit_collection.rb +64 -52
- data/lib/relaton/itu/processor.rb +1 -0
- data/lib/relaton/itu/scraper.rb +13 -3
- data/lib/relaton/itu.rb +0 -2
- data/lib/relaton/jis/bibliography.rb +22 -11
- data/lib/relaton/jis/data_fetcher.rb +4 -4
- data/lib/relaton/jis/processor.rb +1 -0
- data/lib/relaton/jis/scraper.rb +8 -8
- data/lib/relaton/nist/bibliography.rb +25 -21
- data/lib/relaton/oasis/bibliography.rb +16 -13
- data/lib/relaton/oasis/browser_agent.rb +2 -2
- data/lib/relaton/oasis/data_parser.rb +5 -5
- data/lib/relaton/oasis/data_parser_utils.rb +2 -2
- data/lib/relaton/oasis/data_part_parser.rb +7 -7
- data/lib/relaton/ogc/bibliography.rb +15 -14
- data/lib/relaton/ogc/hit_collection.rb +9 -7
- data/lib/relaton/ogc/processor.rb +13 -0
- data/lib/relaton/oiml/processor.rb +1 -0
- data/lib/relaton/omg/bibliography.rb +9 -9
- data/lib/relaton/omg/scraper.rb +14 -13
- data/lib/relaton/omg.rb +1 -1
- data/lib/relaton/plateau/bibliography.rb +9 -7
- data/lib/relaton/plateau/hit_collection.rb +2 -1
- data/lib/relaton/plateau/processor.rb +1 -0
- data/lib/relaton/un/bibliography.rb +21 -10
- data/lib/relaton/un/processor.rb +1 -0
- data/lib/relaton/version.rb +1 -1
- data/lib/relaton/w3c/processor.rb +1 -0
- data/lib/relaton/xsf/bibliography.rb +11 -6
- data/lib/relaton.rb +13 -0
- metadata +60 -14
- data/lib/relaton/itu/pubid.rb +0 -199
data/lib/relaton/db/cache.rb
CHANGED
|
@@ -1,24 +1,64 @@
|
|
|
1
1
|
require "fileutils"
|
|
2
|
-
require "
|
|
2
|
+
require "digest"
|
|
3
|
+
require "json"
|
|
4
|
+
require "date"
|
|
5
|
+
require "lutaml/store"
|
|
6
|
+
require_relative "cache_entry"
|
|
3
7
|
|
|
4
8
|
module Relaton
|
|
5
9
|
class Db
|
|
10
|
+
#
|
|
11
|
+
# The document cache of one directory, on two lutaml-store FileSystem
|
|
12
|
+
# stores under `<dir>/v2/`:
|
|
13
|
+
#
|
|
14
|
+
# - `rows/` holds the index. One store key per bucket, `<flavor>/<root
|
|
15
|
+
# number>` (e.g. `iso/19115`), whose value is the list of the bucket's
|
|
16
|
+
# CacheEntry rows. A lookup reads one small bucket, and a write is one
|
|
17
|
+
# atomic `update` of it.
|
|
18
|
+
# - `docs/` holds the XML documents. A document is named after the row it
|
|
19
|
+
# was stored for, and several rows can point to it.
|
|
20
|
+
#
|
|
21
|
+
# A key is a parsed pubid, a wrapped string (`ISO(ISO 19115-1)`, parsed
|
|
22
|
+
# through the prefix's processor), or a plain string that no processor
|
|
23
|
+
# owns. Locking and atomic writes come from lutaml-store.
|
|
24
|
+
#
|
|
6
25
|
class Cache
|
|
26
|
+
LAYOUT = "v2".freeze
|
|
27
|
+
VERSIONS_KEY = "_versions".freeze
|
|
28
|
+
STRING_FLAVOR = "_key".freeze
|
|
29
|
+
PUBID_FLAVOR = "_pubid".freeze
|
|
30
|
+
# The legacy string form of a not-found entry (`not_found <date>`).
|
|
31
|
+
NOT_FOUND = /\Anot_found/
|
|
32
|
+
# Days an undated entry, or any not_found entry, stays valid.
|
|
33
|
+
UNDATED_TTL = 60
|
|
34
|
+
WRAPPED_KEY = /\A(?<prefix>[^(\s]+)\((?<code>.+)\)\z/m
|
|
35
|
+
# What a row may add to a dated key and still answer it: `===` also
|
|
36
|
+
# reads an omitted `part` as "any part" (`ISO 19115:2003 ===
|
|
37
|
+
# ISO 19115-1:2003`), which names another document.
|
|
38
|
+
LANGUAGE_COMPONENTS = %w[language languages].freeze
|
|
39
|
+
# What a row may add to a key and still be one of its editions.
|
|
40
|
+
EDITION_COMPONENTS = (%w[year date month] + LANGUAGE_COMPONENTS).freeze
|
|
41
|
+
|
|
7
42
|
# @return [String]
|
|
8
43
|
attr_reader :dir
|
|
9
44
|
|
|
10
|
-
# @param dir [String]
|
|
11
|
-
def initialize(dir
|
|
45
|
+
# @param dir [String] cache directory
|
|
46
|
+
def initialize(dir)
|
|
12
47
|
@dir = dir
|
|
13
|
-
|
|
14
|
-
|
|
48
|
+
# A stale file at the cache path (e.g. a placeholder left by cache
|
|
49
|
+
# migration flows) makes every write raise Errno::EEXIST; replace
|
|
50
|
+
# it with the cache directory.
|
|
51
|
+
FileUtils.rm_rf dir if File.exist?(dir) && !File.directory?(dir)
|
|
52
|
+
archive_old_layout
|
|
53
|
+
open_stores
|
|
54
|
+
check_versions
|
|
15
55
|
end
|
|
16
56
|
|
|
17
|
-
# Move
|
|
57
|
+
# Move the cache to another directory.
|
|
18
58
|
# @param new_dir [String, nil]
|
|
19
|
-
# @return [String, nil]
|
|
59
|
+
# @return [String, nil] the new directory
|
|
20
60
|
def mv(new_dir)
|
|
21
|
-
return unless new_dir
|
|
61
|
+
return unless new_dir
|
|
22
62
|
|
|
23
63
|
if File.exist? new_dir
|
|
24
64
|
Util.info "target directory exists `#{new_dir}`"
|
|
@@ -27,205 +67,491 @@ module Relaton
|
|
|
27
67
|
|
|
28
68
|
FileUtils.mv dir, new_dir
|
|
29
69
|
@dir = new_dir
|
|
70
|
+
open_stores
|
|
71
|
+
@dir
|
|
30
72
|
end
|
|
31
73
|
|
|
32
|
-
#
|
|
74
|
+
# Remove every row and document.
|
|
33
75
|
def clear
|
|
34
|
-
|
|
76
|
+
@rows.clear
|
|
77
|
+
@docs.clear
|
|
78
|
+
@versions = {}
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
# Read the entry for a key: the row with this key, else, for a dated
|
|
82
|
+
# pubid, a row the key is a subset of.
|
|
83
|
+
#
|
|
84
|
+
# @param key [Pubid::Identifier, String]
|
|
85
|
+
# @return [String, Relaton::Db::NotFound, nil] document XML or not found
|
|
86
|
+
def read(key)
|
|
87
|
+
found = lookup(key)
|
|
88
|
+
found && read_entry(found.last)
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
# The entry for a key as a string, a not-found entry as
|
|
92
|
+
# `not_found <date>`.
|
|
93
|
+
#
|
|
94
|
+
# @param key [Pubid::Identifier, String]
|
|
95
|
+
# @return [String, nil]
|
|
96
|
+
def [](key)
|
|
97
|
+
read(key)&.to_s
|
|
35
98
|
end
|
|
99
|
+
alias get []
|
|
36
100
|
|
|
37
|
-
#
|
|
38
|
-
# @param
|
|
39
|
-
#
|
|
101
|
+
# @param key [Pubid::Identifier, String]
|
|
102
|
+
# @param value [String, Relaton::Db::NotFound, nil] document XML, a
|
|
103
|
+
# not-found entry (also as `not_found <date>`); nil deletes the row
|
|
40
104
|
def []=(key, value)
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
105
|
+
store key, value
|
|
106
|
+
end
|
|
107
|
+
|
|
108
|
+
#
|
|
109
|
+
# Save a document for a query key. When the document's own identifier
|
|
110
|
+
# (`item_key`) differs from the query, both get a row, and both rows
|
|
111
|
+
# point to one document file.
|
|
112
|
+
#
|
|
113
|
+
# @param key [Pubid::Identifier, String] query key
|
|
114
|
+
# @param value [String, Relaton::Db::NotFound, nil] document XML, a
|
|
115
|
+
# not-found entry (also as `not_found <date>`); nil deletes the row
|
|
116
|
+
# @param item_key [Pubid::Identifier, String, nil] the document's key
|
|
117
|
+
# @return [String, Relaton::Db::NotFound, nil] value
|
|
118
|
+
#
|
|
119
|
+
def store(key, value, item_key: nil) # rubocop:disable Metrics/AbcSize,Metrics/MethodLength
|
|
120
|
+
return delete(key) if value.nil?
|
|
121
|
+
|
|
122
|
+
bucket, key = resolve key
|
|
123
|
+
value = coerce value
|
|
124
|
+
fetched = fetched_of value
|
|
125
|
+
record_version bucket
|
|
126
|
+
if value.is_a? NotFound
|
|
127
|
+
upsert bucket, new_row(key, CacheEntry::NOT_FOUND, nil, fetched)
|
|
128
|
+
return value
|
|
44
129
|
end
|
|
45
130
|
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
131
|
+
item_bucket, item_key = item_key ? resolve(item_key) : [bucket, key]
|
|
132
|
+
file = doc_name item_bucket, item_key
|
|
133
|
+
record_version item_bucket
|
|
134
|
+
@rows.adapter.transaction do
|
|
135
|
+
@docs.set file, value
|
|
136
|
+
upsert item_bucket, new_row(item_key, CacheEntry::DOC, file, fetched)
|
|
137
|
+
unless canonical(key) == canonical(item_key)
|
|
138
|
+
upsert bucket, new_row(key, CacheEntry::DOC, file, fetched)
|
|
139
|
+
end
|
|
140
|
+
end
|
|
141
|
+
value
|
|
50
142
|
end
|
|
51
143
|
|
|
52
|
-
#
|
|
53
|
-
#
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
144
|
+
# Delete the row of a key. Its document goes too, when no other row
|
|
145
|
+
# points to it.
|
|
146
|
+
# @param key [Pubid::Identifier, String]
|
|
147
|
+
def delete(key)
|
|
148
|
+
bucket, key = resolve key
|
|
149
|
+
target = canonical key
|
|
150
|
+
return unless read_bucket(bucket).any? { |row| row_key(row) == target }
|
|
151
|
+
|
|
152
|
+
delete_row bucket, target
|
|
60
153
|
end
|
|
61
154
|
|
|
62
|
-
#
|
|
63
|
-
#
|
|
64
|
-
#
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
155
|
+
#
|
|
156
|
+
# Delete the row a lookup of the key finds (its own, or the one a dated
|
|
157
|
+
# key is a subset of) when it is no longer valid.
|
|
158
|
+
#
|
|
159
|
+
# @param key [Pubid::Identifier, String]
|
|
160
|
+
# @param year [String, nil]
|
|
161
|
+
#
|
|
162
|
+
def expire(key, year)
|
|
163
|
+
found = lookup(key) or return
|
|
164
|
+
return if valid_row?(found.last, year)
|
|
165
|
+
|
|
166
|
+
delete_row resolve(found.first).first, row_key(found.last)
|
|
167
|
+
end
|
|
168
|
+
|
|
169
|
+
#
|
|
170
|
+
# Save the row of a key, and its document, from another cache. A row
|
|
171
|
+
# the cache already holds is kept: a local cache keeps its own value
|
|
172
|
+
# for a key the global cache also holds (local over global).
|
|
173
|
+
#
|
|
174
|
+
# @param key [Pubid::Identifier, String]
|
|
175
|
+
# @param other [Relaton::Db::Cache]
|
|
176
|
+
#
|
|
177
|
+
def clone_entry(key, other)
|
|
178
|
+
return if lookup(key)
|
|
179
|
+
|
|
180
|
+
found = other.lookup(key) or return
|
|
181
|
+
row_id, row = found
|
|
182
|
+
store row_id, other.read_entry(row)
|
|
183
|
+
end
|
|
184
|
+
|
|
185
|
+
# @param key [Pubid::Identifier, String]
|
|
186
|
+
# @return [String, nil] the date the entry was fetched
|
|
187
|
+
def fetched(key)
|
|
188
|
+
lookup(key)&.last&.dig("fetched")
|
|
189
|
+
end
|
|
190
|
+
|
|
191
|
+
# An undated entry expires after 60 days, a dated document never. A
|
|
192
|
+
# not_found entry expires after 60 days even for a dated query: the
|
|
193
|
+
# document can be published later.
|
|
194
|
+
# @param key [Pubid::Identifier, String]
|
|
195
|
+
# @param year [String, nil]
|
|
196
|
+
def valid_entry?(key, year)
|
|
197
|
+
found = lookup(key)
|
|
198
|
+
found ? valid_row?(found.last, year) : false
|
|
199
|
+
end
|
|
200
|
+
|
|
201
|
+
#
|
|
202
|
+
# The cached editions of the key: documents whose pubid the key is a
|
|
203
|
+
# subset of (`key === id`) and that add only a year, a date or a
|
|
204
|
+
# language to it. For a query that selects among editions (e.g. by
|
|
205
|
+
# publication date).
|
|
206
|
+
#
|
|
207
|
+
# @param key [Pubid::Identifier]
|
|
208
|
+
# @return [Array<Array(Pubid::Identifier, String)>]
|
|
209
|
+
#
|
|
210
|
+
def candidates(key) # rubocop:disable Metrics/AbcSize,Metrics/CyclomaticComplexity
|
|
211
|
+
bucket, key = resolve key
|
|
212
|
+
return [] if key.is_a? String
|
|
213
|
+
|
|
214
|
+
read_bucket(bucket).filter_map do |row|
|
|
215
|
+
next if row["status"] != CacheEntry::DOC || !row["id"]
|
|
216
|
+
|
|
217
|
+
id = pubid_from row["id"]
|
|
218
|
+
next unless subset_of? key, id, row["id"], EDITION_COMPONENTS
|
|
219
|
+
|
|
220
|
+
xml = @docs.get row["file"]
|
|
221
|
+
[id, xml] if xml
|
|
71
222
|
end
|
|
72
223
|
end
|
|
73
224
|
|
|
74
225
|
#
|
|
75
|
-
#
|
|
226
|
+
# Every cached document once.
|
|
76
227
|
#
|
|
77
|
-
# @
|
|
78
|
-
# @
|
|
228
|
+
# @yieldparam processor [Relaton::Core::Processor, nil] the owning flavor
|
|
229
|
+
# @yieldparam xml [String]
|
|
230
|
+
# @return [Array] the documents, or the block results
|
|
79
231
|
#
|
|
80
|
-
def
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
232
|
+
def all
|
|
233
|
+
files = {}
|
|
234
|
+
each_row do |bucket, row|
|
|
235
|
+
files[row["file"]] ||= bucket if row["file"]
|
|
236
|
+
end
|
|
237
|
+
files.filter_map do |file, bucket|
|
|
238
|
+
xml = @docs.get(file) or next
|
|
239
|
+
block_given? ? yield(processor_for(bucket), xml) : xml
|
|
84
240
|
end
|
|
85
241
|
end
|
|
86
242
|
|
|
87
|
-
#
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
243
|
+
# @return [Array<Hash>] every row
|
|
244
|
+
def rows
|
|
245
|
+
list = []
|
|
246
|
+
each_row { |_, row| list << row }
|
|
247
|
+
list
|
|
248
|
+
end
|
|
93
249
|
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
250
|
+
# The row of a key, with the key it is stored under.
|
|
251
|
+
# @param key [Pubid::Identifier, String]
|
|
252
|
+
# @return [Array(Object, Hash), nil]
|
|
253
|
+
def lookup(key)
|
|
254
|
+
bucket, key = resolve key
|
|
255
|
+
rows = read_bucket bucket
|
|
256
|
+
target = canonical key
|
|
257
|
+
row = rows.detect { |r| row_key(r) == target }
|
|
258
|
+
return [key, row] if row
|
|
259
|
+
return unless dated? key
|
|
260
|
+
|
|
261
|
+
subset_match rows, key
|
|
262
|
+
end
|
|
263
|
+
|
|
264
|
+
# @param row [Hash]
|
|
265
|
+
# @return [String, Relaton::Db::NotFound, nil] document XML or not found
|
|
266
|
+
def read_entry(row)
|
|
267
|
+
if row["status"] == CacheEntry::NOT_FOUND
|
|
268
|
+
return NotFound.new(fetched: row["fetched"])
|
|
269
|
+
end
|
|
270
|
+
|
|
271
|
+
@docs.get row["file"]
|
|
272
|
+
end
|
|
273
|
+
|
|
274
|
+
private
|
|
275
|
+
|
|
276
|
+
def open_stores
|
|
277
|
+
base = File.join dir, LAYOUT
|
|
278
|
+
@rows = new_store File.join(base, "rows"), ".json"
|
|
279
|
+
@docs = new_store File.join(base, "docs"), ".xml"
|
|
280
|
+
end
|
|
281
|
+
|
|
282
|
+
def new_store(path, extension)
|
|
283
|
+
Lutaml::Store::BasicStore.new(
|
|
284
|
+
adapter_type: :filesystem,
|
|
285
|
+
adapter_options: { path: path, extension: extension,
|
|
286
|
+
integrity_checks: false },
|
|
287
|
+
cache: { enabled: false },
|
|
288
|
+
)
|
|
289
|
+
end
|
|
290
|
+
|
|
291
|
+
# A cache written by the file-per-key layout (before v2) is moved to
|
|
292
|
+
# `<dir>-v1.bak`, never deleted.
|
|
293
|
+
def archive_old_layout # rubocop:disable Metrics/AbcSize,Metrics/MethodLength
|
|
294
|
+
return unless Dir.exist? dir
|
|
295
|
+
|
|
296
|
+
old = Dir.children(dir) - [LAYOUT]
|
|
297
|
+
bak = "#{dir}-v1.bak"
|
|
298
|
+
# Once only: an older relaton that shares the directory writes the old
|
|
299
|
+
# layout again, and v2 ignores it.
|
|
300
|
+
return if old.empty? || File.exist?(bak)
|
|
301
|
+
|
|
302
|
+
FileUtils.mkdir_p bak
|
|
303
|
+
old.each do |name|
|
|
304
|
+
FileUtils.mv File.join(dir, name), bak
|
|
305
|
+
rescue Errno::ENOENT
|
|
306
|
+
next # another process moved it first
|
|
99
307
|
end
|
|
308
|
+
Util.info "cache #{dir}: the old cache is moved to #{bak}"
|
|
100
309
|
end
|
|
101
310
|
|
|
102
|
-
#
|
|
103
|
-
#
|
|
104
|
-
def
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
311
|
+
# Drop the rows and documents of a flavor whose grammar changed since
|
|
312
|
+
# they were written.
|
|
313
|
+
def check_versions
|
|
314
|
+
@versions = @rows.get(VERSIONS_KEY) || {}
|
|
315
|
+
@versions.each do |flavor, hash|
|
|
316
|
+
processor = Registry.instance[:"relaton_#{flavor}"]
|
|
317
|
+
next if processor && processor.grammar_hash == hash
|
|
318
|
+
|
|
319
|
+
drop_flavor flavor
|
|
320
|
+
Util.info "cache #{dir}: version of `#{flavor}` is obsolete " \
|
|
321
|
+
"and its entries are cleared."
|
|
108
322
|
end
|
|
109
323
|
end
|
|
110
324
|
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
325
|
+
def drop_flavor(flavor)
|
|
326
|
+
@rows.adapter.transaction do
|
|
327
|
+
[@rows, @docs].each do |store|
|
|
328
|
+
store.each_key { |k| store.delete k if k.start_with? "#{flavor}/" }
|
|
329
|
+
end
|
|
330
|
+
@versions = @rows.update(VERSIONS_KEY) { |v| (v || {}).except flavor }
|
|
331
|
+
end
|
|
332
|
+
end
|
|
117
333
|
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
334
|
+
def record_version(bucket)
|
|
335
|
+
flavor = bucket.split("/", 2).first
|
|
336
|
+
return if flavor.start_with?("_") || @versions.key?(flavor)
|
|
337
|
+
|
|
338
|
+
processor = Registry.instance[:"relaton_#{flavor}"] or return
|
|
339
|
+
hash = processor.grammar_hash
|
|
340
|
+
@versions = @rows.update(VERSIONS_KEY) do |v|
|
|
341
|
+
(v || {}).merge(flavor => hash)
|
|
121
342
|
end
|
|
122
|
-
File.delete f
|
|
123
343
|
end
|
|
124
344
|
|
|
125
|
-
#
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
345
|
+
# @return [Array(String, Object)] the bucket and the resolved key
|
|
346
|
+
def resolve(key)
|
|
347
|
+
key = wrapped_key key if key.is_a? String
|
|
348
|
+
if key.is_a? String
|
|
349
|
+
["#{STRING_FLAVOR}/#{key}", key]
|
|
350
|
+
else
|
|
351
|
+
[bucket_for(key), key]
|
|
352
|
+
end
|
|
353
|
+
end
|
|
131
354
|
|
|
132
|
-
|
|
133
|
-
|
|
355
|
+
# `ISO(ISO 19115-1)` -> the ISO processor's pubid for `ISO 19115-1`.
|
|
356
|
+
def wrapped_key(key)
|
|
357
|
+
match = key.match(WRAPPED_KEY) or return key
|
|
358
|
+
processor = Registry.instance.by_type(match[:prefix]) or return key
|
|
359
|
+
processor.cache_key(match[:code], nil, {}) || key
|
|
134
360
|
end
|
|
135
361
|
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
362
|
+
def bucket_for(pubid)
|
|
363
|
+
processor = Registry.instance.processor_by_pubid pubid
|
|
364
|
+
flavor = processor ? flavor_name(processor) : PUBID_FLAVOR
|
|
365
|
+
number = pubid.root.number.to_s
|
|
366
|
+
# No number (DOI, ISBN, the SI Brochure): 256 digest buckets, not one
|
|
367
|
+
# bucket for the whole flavor.
|
|
368
|
+
number = "~#{digest(pubid)[0, 2]}" if number.empty?
|
|
369
|
+
"#{flavor}/#{number}"
|
|
370
|
+
end
|
|
142
371
|
|
|
143
|
-
|
|
144
|
-
|
|
372
|
+
def flavor_name(processor)
|
|
373
|
+
processor.short.to_s.delete_prefix "relaton_"
|
|
145
374
|
end
|
|
146
375
|
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
376
|
+
def processor_for(bucket)
|
|
377
|
+
flavor = bucket.split("/", 2).first
|
|
378
|
+
if flavor == STRING_FLAVOR
|
|
379
|
+
Registry.instance.processor_by_ref bucket.split("/", 2).last
|
|
380
|
+
else
|
|
381
|
+
Registry.instance[:"relaton_#{flavor}"]
|
|
382
|
+
end
|
|
383
|
+
end
|
|
154
384
|
|
|
155
|
-
|
|
385
|
+
def read_bucket(bucket)
|
|
386
|
+
Array @rows.get(bucket)
|
|
156
387
|
end
|
|
157
388
|
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
389
|
+
def each_row
|
|
390
|
+
@rows.each_key do |bucket|
|
|
391
|
+
next if bucket == VERSIONS_KEY
|
|
392
|
+
|
|
393
|
+
read_bucket(bucket).each { |row| yield bucket, row }
|
|
394
|
+
end
|
|
163
395
|
end
|
|
164
396
|
|
|
165
|
-
|
|
397
|
+
# Replace or add a row. A document the replaced row pointed to goes
|
|
398
|
+
# when no row points to it any more.
|
|
399
|
+
def upsert(bucket, row) # rubocop:disable Metrics/AbcSize,Metrics/MethodLength
|
|
400
|
+
target = row_key row
|
|
401
|
+
replaced = nil
|
|
402
|
+
@rows.adapter.transaction do
|
|
403
|
+
@rows.update(bucket) do |old|
|
|
404
|
+
old = Array(old)
|
|
405
|
+
replaced = old.detect { |r| row_key(r) == target }
|
|
406
|
+
(old - [replaced]) << row
|
|
407
|
+
end
|
|
408
|
+
old_file = replaced&.dig("file")
|
|
409
|
+
remove_doc old_file, bucket if old_file && old_file != row["file"]
|
|
410
|
+
end
|
|
411
|
+
end
|
|
166
412
|
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
413
|
+
def delete_row(bucket, target)
|
|
414
|
+
@rows.adapter.transaction do
|
|
415
|
+
removed = nil
|
|
416
|
+
rows = @rows.update(bucket) do |old|
|
|
417
|
+
old = Array(old)
|
|
418
|
+
removed = old.detect { |row| row_key(row) == target }
|
|
419
|
+
old - [removed]
|
|
420
|
+
end
|
|
421
|
+
@rows.delete bucket if rows.empty?
|
|
422
|
+
remove_doc removed["file"], bucket if removed&.dig("file")
|
|
423
|
+
end
|
|
424
|
+
end
|
|
425
|
+
|
|
426
|
+
def valid_row?(row, year)
|
|
427
|
+
return false unless read_entry(row)
|
|
428
|
+
|
|
429
|
+
return year if year && row["status"] == CacheEntry::DOC
|
|
430
|
+
|
|
431
|
+
Date.today - Date.parse(row["fetched"]) < UNDATED_TTL
|
|
178
432
|
end
|
|
179
433
|
|
|
180
434
|
#
|
|
181
|
-
#
|
|
182
|
-
# the
|
|
435
|
+
# Whether a row answers a key: the key is a subset of it (`===`), and
|
|
436
|
+
# every component the row adds is one of `allowed`.
|
|
183
437
|
#
|
|
184
|
-
# @param
|
|
185
|
-
# @
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
438
|
+
# @param key [Pubid::Identifier]
|
|
439
|
+
# @param id [Pubid::Identifier] the row's pubid
|
|
440
|
+
# @param row_id [Hash] the row's stored `id`
|
|
441
|
+
# @param allowed [Array<String>]
|
|
442
|
+
#
|
|
443
|
+
def subset_of?(key, id, row_id, allowed)
|
|
444
|
+
return false unless key === id # rubocop:disable Style/CaseEquality
|
|
445
|
+
|
|
446
|
+
(added_components(canonical(key), row_id) - allowed).empty?
|
|
447
|
+
end
|
|
448
|
+
|
|
449
|
+
# The names of the components `row` has and `query` has not, at any
|
|
450
|
+
# depth (a supplement's base, an adoption's adopted document).
|
|
451
|
+
def added_components(query, row) # rubocop:disable Metrics/AbcSize,Metrics/CyclomaticComplexity,Metrics/MethodLength,Metrics/PerceivedComplexity
|
|
452
|
+
case row
|
|
453
|
+
when Hash
|
|
454
|
+
row.flat_map do |name, value|
|
|
455
|
+
next [] if value.nil?
|
|
456
|
+
next [name] unless query.is_a?(Hash) && query.key?(name)
|
|
457
|
+
|
|
458
|
+
added_components query[name], value
|
|
459
|
+
end
|
|
460
|
+
when Array
|
|
461
|
+
return ["[]"] unless query.is_a?(Array) && query.size == row.size
|
|
462
|
+
|
|
463
|
+
row.each_with_index.flat_map { |v, i| added_components query[i], v }
|
|
464
|
+
else []
|
|
193
465
|
end
|
|
194
466
|
end
|
|
195
467
|
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
unless File.exist? file_version
|
|
201
|
-
file_safe_write file_version, self.class.grammar_hash(fdir)
|
|
468
|
+
def new_row(key, status, file, fetched)
|
|
469
|
+
row = CacheEntry.new(status: status, file: file, fetched: fetched)
|
|
470
|
+
if key.is_a?(String) then row.key = key
|
|
471
|
+
else row.id = canonical(key)
|
|
202
472
|
end
|
|
473
|
+
row.to_hash
|
|
203
474
|
end
|
|
204
475
|
|
|
205
|
-
#
|
|
206
|
-
#
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
476
|
+
# A key as it is stored: a pubid's `to_hash` after a JSON round trip,
|
|
477
|
+
# so a fresh key compares equal to a stored one.
|
|
478
|
+
def canonical(key)
|
|
479
|
+
return key if key.is_a? String
|
|
480
|
+
|
|
481
|
+
JSON.parse JSON.generate(key.to_hash)
|
|
210
482
|
end
|
|
211
483
|
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
def
|
|
217
|
-
|
|
218
|
-
|
|
484
|
+
def row_key(row)
|
|
485
|
+
row["id"] || row["key"]
|
|
486
|
+
end
|
|
487
|
+
|
|
488
|
+
def dated?(key)
|
|
489
|
+
!key.is_a?(String) && key.respond_to?(:year) && !key.year.nil?
|
|
490
|
+
end
|
|
491
|
+
|
|
492
|
+
# The newest row whose pubid the key is a subset of. A row whose
|
|
493
|
+
# document is gone does not count.
|
|
494
|
+
def subset_match(rows, key) # rubocop:disable Metrics/CyclomaticComplexity
|
|
495
|
+
rows.filter_map do |row|
|
|
496
|
+
next unless row["id"]
|
|
497
|
+
|
|
498
|
+
id = pubid_from row["id"]
|
|
499
|
+
next unless subset_of? key, id, row["id"], LANGUAGE_COMPONENTS
|
|
500
|
+
next if row["file"] && !@docs.exists?(row["file"])
|
|
501
|
+
|
|
502
|
+
[id, row]
|
|
503
|
+
end.max_by { |_, row| row["fetched"].to_s }
|
|
504
|
+
end
|
|
505
|
+
|
|
506
|
+
def pubid_from(hash)
|
|
507
|
+
require "pubid"
|
|
508
|
+
::Pubid.from_hash hash
|
|
509
|
+
end
|
|
510
|
+
|
|
511
|
+
def doc_name(bucket, key)
|
|
512
|
+
"#{bucket}/#{digest(key)[0, 16]}"
|
|
513
|
+
end
|
|
514
|
+
|
|
515
|
+
# A digest of the canonical key, independent of the hash order.
|
|
516
|
+
def digest(key)
|
|
517
|
+
Digest::SHA256.hexdigest JSON.generate(canonical_sorted(key))
|
|
518
|
+
end
|
|
519
|
+
|
|
520
|
+
def canonical_sorted(key)
|
|
521
|
+
value = canonical key
|
|
522
|
+
value.is_a?(Hash) ? deep_sort(value) : value
|
|
523
|
+
end
|
|
524
|
+
|
|
525
|
+
def deep_sort(value)
|
|
526
|
+
case value
|
|
527
|
+
when Hash then value.sort.to_h { |k, v| [k, deep_sort(v)] }
|
|
528
|
+
when Array then value.map { |v| deep_sort v }
|
|
529
|
+
else value
|
|
530
|
+
end
|
|
219
531
|
end
|
|
220
532
|
|
|
221
|
-
#
|
|
222
|
-
#
|
|
223
|
-
def
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
f.flock File::LOCK_UN
|
|
533
|
+
# Delete a document unless a row still points to it. A row that points
|
|
534
|
+
# to it lives in the deleted row's bucket or in the document's own one.
|
|
535
|
+
def remove_doc(file, bucket)
|
|
536
|
+
file_bucket = file.rpartition("/").first
|
|
537
|
+
used = [bucket, file_bucket].uniq.any? do |b|
|
|
538
|
+
read_bucket(b).any? { |row| row["file"] == file }
|
|
228
539
|
end
|
|
540
|
+
@docs.delete file unless used
|
|
541
|
+
end
|
|
542
|
+
|
|
543
|
+
# A legacy `not_found <date>` string becomes a NotFound value.
|
|
544
|
+
def coerce(value)
|
|
545
|
+
return value unless value.is_a?(String) && value.match?(NOT_FOUND)
|
|
546
|
+
|
|
547
|
+
NotFound.new(fetched: value[/\d{4}-\d{2}-\d{2}/] || Date.today.to_s)
|
|
548
|
+
end
|
|
549
|
+
|
|
550
|
+
def fetched_of(value)
|
|
551
|
+
date = if value.is_a? NotFound then value.fetched
|
|
552
|
+
else value[%r{<fetched>\s*([^<\s]+)\s*</fetched>}, 1]
|
|
553
|
+
end
|
|
554
|
+
date || Date.today.to_s
|
|
229
555
|
end
|
|
230
556
|
end
|
|
231
557
|
end
|