relaton 3.0.0.pre.alpha.1 → 3.0.0.pre.alpha.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (270) hide show
  1. checksums.yaml +4 -4
  2. data/README.adoc +57 -1
  3. data/bin/console +0 -1
  4. data/lib/relaton/3gpp/bibliography.rb +82 -7
  5. data/lib/relaton/3gpp/data_fetcher.rb +51 -3
  6. data/lib/relaton/3gpp/docidentifier.rb +114 -0
  7. data/lib/relaton/3gpp/item.rb +6 -0
  8. data/lib/relaton/3gpp/parser.rb +1 -1
  9. data/lib/relaton/3gpp/processor.rb +4 -1
  10. data/lib/relaton/3gpp.rb +5 -1
  11. data/lib/relaton/adobe/bibdata.rb +8 -0
  12. data/lib/relaton/adobe/bibitem.rb +8 -0
  13. data/lib/relaton/adobe/bibliography.rb +92 -0
  14. data/lib/relaton/adobe/docidentifier.rb +49 -0
  15. data/lib/relaton/adobe/doctype.rb +14 -0
  16. data/lib/relaton/adobe/ext.rb +32 -0
  17. data/lib/relaton/adobe/item.rb +15 -0
  18. data/lib/relaton/adobe/item_base.rb +18 -0
  19. data/lib/relaton/adobe/item_data.rb +6 -0
  20. data/lib/relaton/adobe/processor.rb +45 -0
  21. data/lib/relaton/adobe/util.rb +8 -0
  22. data/lib/relaton/adobe.rb +37 -0
  23. data/lib/relaton/bib/model/address.rb +2 -2
  24. data/lib/relaton/bib/model/docidentifier.rb +24 -9
  25. data/lib/relaton/bib/model/localized_string.rb +1 -1
  26. data/lib/relaton/bib/model/structured_identifier.rb +10 -9
  27. data/lib/relaton/bib/sanitizer.rb +202 -6
  28. data/lib/relaton/bib.rb +0 -2
  29. data/lib/relaton/bipm/bibliography.rb +159 -10
  30. data/lib/relaton/bipm/data_fetcher.rb +26 -2
  31. data/lib/relaton/bipm/data_outcomes_parser.rb +3 -4
  32. data/lib/relaton/bipm/id_parser.rb +5 -4
  33. data/lib/relaton/bipm/model/structured_identifier.rb +21 -0
  34. data/lib/relaton/bipm/processor.rb +2 -2
  35. data/lib/relaton/bipm/rawdata_bipm_metrologia/fetcher.rb +2 -4
  36. data/lib/relaton/bipm/si_brochure_parser.rb +5 -3
  37. data/lib/relaton/bipm.rb +7 -1
  38. data/lib/relaton/bsi/bibliography.rb +115 -43
  39. data/lib/relaton/bsi/hit.rb +14 -0
  40. data/lib/relaton/bsi/hit_collection.rb +15 -16
  41. data/lib/relaton/bsi/model/docidentifier.rb +99 -1
  42. data/lib/relaton/bsi/processor.rb +1 -0
  43. data/lib/relaton/calconnect/bibliography.rb +12 -14
  44. data/lib/relaton/calconnect/data_fetcher.rb +77 -9
  45. data/lib/relaton/calconnect/docidentifier.rb +80 -0
  46. data/lib/relaton/calconnect/hit_collection.rb +65 -57
  47. data/lib/relaton/calconnect/model/item.rb +7 -0
  48. data/lib/relaton/calconnect/processor.rb +7 -1
  49. data/lib/relaton/calconnect.rb +11 -1
  50. data/lib/relaton/ccsds/data/fetcher.rb +17 -12
  51. data/lib/relaton/ccsds/data/parser.rb +1 -1
  52. data/lib/relaton/ccsds/hit_collection.rb +6 -1
  53. data/lib/relaton/ccsds/model/docidentifier.rb +121 -0
  54. data/lib/relaton/ccsds/model/item.rb +2 -0
  55. data/lib/relaton/cen/bibliography.rb +75 -47
  56. data/lib/relaton/cen/hit.rb +16 -1
  57. data/lib/relaton/cen/hit_collection.rb +59 -11
  58. data/lib/relaton/cen/model/docidentifier.rb +92 -1
  59. data/lib/relaton/cen/processor.rb +13 -8
  60. data/lib/relaton/cen/scraper.rb +13 -5
  61. data/lib/relaton/cen.rb +1 -0
  62. data/lib/relaton/cie/data_fetcher.rb +215 -30
  63. data/lib/relaton/cie/processor.rb +3 -1
  64. data/lib/relaton/cie/scrapper.rb +15 -2
  65. data/lib/relaton/cie.rb +2 -1
  66. data/lib/relaton/core/data_fetcher.rb +150 -3
  67. data/lib/relaton/core/governor.rb +320 -0
  68. data/lib/relaton/core/pacer.rb +134 -0
  69. data/lib/relaton/core/processor.rb +19 -0
  70. data/lib/relaton/core/request_error.rb +14 -0
  71. data/lib/relaton/core.rb +3 -0
  72. data/lib/relaton/db/registry.rb +41 -1
  73. data/lib/relaton/doi/crossref.rb +19 -2
  74. data/lib/relaton/doi/parser.rb +109 -15
  75. data/lib/relaton/easc/bibdata.rb +8 -0
  76. data/lib/relaton/easc/bibitem.rb +8 -0
  77. data/lib/relaton/easc/bibliography.rb +95 -0
  78. data/lib/relaton/easc/docidentifier.rb +100 -0
  79. data/lib/relaton/easc/doctype.rb +14 -0
  80. data/lib/relaton/easc/ext.rb +44 -0
  81. data/lib/relaton/easc/item.rb +13 -0
  82. data/lib/relaton/easc/item_base.rb +18 -0
  83. data/lib/relaton/easc/item_data.rb +6 -0
  84. data/lib/relaton/easc/processor.rb +46 -0
  85. data/lib/relaton/easc/util.rb +8 -0
  86. data/lib/relaton/easc.rb +35 -0
  87. data/lib/relaton/ecma/bibliography.rb +93 -25
  88. data/lib/relaton/ecma/data_fetcher.rb +71 -12
  89. data/lib/relaton/ecma/docidentifier.rb +124 -0
  90. data/lib/relaton/ecma/item.rb +2 -0
  91. data/lib/relaton/ecma/memento_parser.rb +1 -1
  92. data/lib/relaton/ecma/page_fetcher.rb +15 -3
  93. data/lib/relaton/ecma/parser_common.rb +2 -2
  94. data/lib/relaton/ecma/processor.rb +4 -1
  95. data/lib/relaton/ecma/standard_parser.rb +2 -2
  96. data/lib/relaton/ecma.rb +10 -1
  97. data/lib/relaton/etsi/bibliography.rb +67 -2
  98. data/lib/relaton/etsi/data_fetcher.rb +43 -4
  99. data/lib/relaton/etsi/processor.rb +3 -1
  100. data/lib/relaton/etsi.rb +2 -1
  101. data/lib/relaton/gb/bibliography.rb +55 -29
  102. data/lib/relaton/gb/docidentifier.rb +58 -9
  103. data/lib/relaton/gb/processor.rb +3 -0
  104. data/lib/relaton/gb/scraper.rb +27 -10
  105. data/lib/relaton/gost/bibdata.rb +8 -0
  106. data/lib/relaton/gost/bibitem.rb +8 -0
  107. data/lib/relaton/gost/bibliography.rb +107 -0
  108. data/lib/relaton/gost/docidentifier.rb +80 -0
  109. data/lib/relaton/gost/doctype.rb +16 -0
  110. data/lib/relaton/gost/ext.rb +46 -0
  111. data/lib/relaton/gost/item.rb +15 -0
  112. data/lib/relaton/gost/item_base.rb +18 -0
  113. data/lib/relaton/gost/item_data.rb +6 -0
  114. data/lib/relaton/gost/processor.rb +49 -0
  115. data/lib/relaton/gost/util.rb +8 -0
  116. data/lib/relaton/gost.rb +36 -0
  117. data/lib/relaton/iala/bibdata.rb +8 -0
  118. data/lib/relaton/iala/bibitem.rb +8 -0
  119. data/lib/relaton/iala/bibliography.rb +146 -0
  120. data/lib/relaton/iala/docidentifier.rb +89 -0
  121. data/lib/relaton/iala/doctype.rb +18 -0
  122. data/lib/relaton/iala/ext.rb +32 -0
  123. data/lib/relaton/iala/item.rb +21 -0
  124. data/lib/relaton/iala/item_base.rb +18 -0
  125. data/lib/relaton/iala/item_data.rb +6 -0
  126. data/lib/relaton/iala/processor.rb +43 -0
  127. data/lib/relaton/iala/relation.rb +7 -0
  128. data/lib/relaton/iala/util.rb +8 -0
  129. data/lib/relaton/iala.rb +35 -0
  130. data/lib/relaton/iana/bibliography.rb +67 -14
  131. data/lib/relaton/iana/data_fetcher.rb +35 -5
  132. data/lib/relaton/iana/processor.rb +3 -1
  133. data/lib/relaton/iana.rb +12 -1
  134. data/lib/relaton/iec/data_fetcher.rb +7 -1
  135. data/lib/relaton/iec/hit_collection.rb +1 -1
  136. data/lib/relaton/iec/model/docidentifier.rb +9 -5
  137. data/lib/relaton/iec/model/ext.rb +2 -2
  138. data/lib/relaton/iec/processor.rb +1 -0
  139. data/lib/relaton/ieee/bibliography.rb +25 -3
  140. data/lib/relaton/ieee/data_fetcher.rb +158 -17
  141. data/lib/relaton/ieee/idams_parser.rb +18 -11
  142. data/lib/relaton/ieee/processor.rb +4 -1
  143. data/lib/relaton/ieee/rawbib_id_parser.rb +291 -86
  144. data/lib/relaton/ieee.rb +2 -1
  145. data/lib/relaton/ietf/data_fetcher.rb +295 -12
  146. data/lib/relaton/ietf/processor.rb +7 -3
  147. data/lib/relaton/ietf/rfc/entry.rb +39 -3
  148. data/lib/relaton/ietf/scraper.rb +69 -36
  149. data/lib/relaton/ietf.rb +4 -1
  150. data/lib/relaton/iho/bibliography.rb +1 -1
  151. data/lib/relaton/iho/docidentifier.rb +1 -1
  152. data/lib/relaton/index/file_io.rb +11 -11
  153. data/lib/relaton/index/file_storage.rb +6 -1
  154. data/lib/relaton/index/pool.rb +6 -1
  155. data/lib/relaton/index/shard_source.rb +201 -0
  156. data/lib/relaton/index/type.rb +63 -12
  157. data/lib/relaton/index.rb +2 -1
  158. data/lib/relaton/iso/bibliography.rb +20 -15
  159. data/lib/relaton/iso/data_fetcher.rb +3 -3
  160. data/lib/relaton/iso/data_parser.rb +17 -3
  161. data/lib/relaton/iso/hit_collection.rb +27 -15
  162. data/lib/relaton/iso/item_data.rb +22 -0
  163. data/lib/relaton/iso/model/docidentifier.rb +24 -12
  164. data/lib/relaton/iso/processor.rb +1 -0
  165. data/lib/relaton/iso/scraper.rb +19 -3
  166. data/lib/relaton/itu/bibliography.rb +9 -4
  167. data/lib/relaton/itu/data_crawler_r.rb +664 -0
  168. data/lib/relaton/itu/data_fetcher.rb +496 -50
  169. data/lib/relaton/itu/data_merge_r.rb +149 -0
  170. data/lib/relaton/itu/data_parser_r.rb +163 -89
  171. data/lib/relaton/itu/data_parser_t.rb +228 -0
  172. data/lib/relaton/itu/family_cache.rb +177 -0
  173. data/lib/relaton/itu/governor.rb +56 -0
  174. data/lib/relaton/itu/hit.rb +9 -3
  175. data/lib/relaton/itu/hit_collection.rb +258 -86
  176. data/lib/relaton/itu/model/docidentifier.rb +67 -1
  177. data/lib/relaton/itu/model/structured_identifier.rb +19 -0
  178. data/lib/relaton/itu/processor.rb +10 -4
  179. data/lib/relaton/itu/pubid.rb +27 -5
  180. data/lib/relaton/itu/recommendation_fields.rb +334 -0
  181. data/lib/relaton/itu/recommendation_parser.rb +18 -149
  182. data/lib/relaton/itu/scraper.rb +13 -3
  183. data/lib/relaton/itu.rb +2 -1
  184. data/lib/relaton/jcgm/bibdata.rb +8 -0
  185. data/lib/relaton/jcgm/bibitem.rb +8 -0
  186. data/lib/relaton/jcgm/bibliography.rb +97 -0
  187. data/lib/relaton/jcgm/data_fetcher.rb +81 -0
  188. data/lib/relaton/jcgm/docidentifier.rb +102 -0
  189. data/lib/relaton/jcgm/doctype.rb +12 -0
  190. data/lib/relaton/jcgm/ext.rb +23 -0
  191. data/lib/relaton/jcgm/item.rb +20 -0
  192. data/lib/relaton/jcgm/item_base.rb +18 -0
  193. data/lib/relaton/jcgm/item_data.rb +6 -0
  194. data/lib/relaton/jcgm/meetings_parser.rb +175 -0
  195. data/lib/relaton/jcgm/processor.rb +71 -0
  196. data/lib/relaton/jcgm/relation.rb +9 -0
  197. data/lib/relaton/jcgm/structured_identifier.rb +40 -0
  198. data/lib/relaton/jcgm/util.rb +8 -0
  199. data/lib/relaton/jcgm.rb +24 -0
  200. data/lib/relaton/jis/bibliography.rb +8 -10
  201. data/lib/relaton/jis/data_fetcher.rb +21 -19
  202. data/lib/relaton/jis/docidentifier.rb +104 -5
  203. data/lib/relaton/jis/hit.rb +18 -23
  204. data/lib/relaton/jis/hit_collection.rb +19 -18
  205. data/lib/relaton/jis/processor.rb +1 -1
  206. data/lib/relaton/jis.rb +2 -3
  207. data/lib/relaton/logger/channels/gh_issue.rb +78 -13
  208. data/lib/relaton/nist/data_fetcher.rb +63 -13
  209. data/lib/relaton/nist/docidentifier.rb +165 -0
  210. data/lib/relaton/nist/item.rb +2 -0
  211. data/lib/relaton/nist/item_base.rb +16 -0
  212. data/lib/relaton/nist/mods_parser.rb +38 -12
  213. data/lib/relaton/nist/processor.rb +2 -1
  214. data/lib/relaton/nist/relation.rb +3 -0
  215. data/lib/relaton/nist/scraper.rb +6 -3
  216. data/lib/relaton/oasis/bibliography.rb +147 -6
  217. data/lib/relaton/oasis/data_fetcher.rb +41 -5
  218. data/lib/relaton/oasis/data_parser_utils.rb +37 -3
  219. data/lib/relaton/oasis/docidentifier.rb +54 -0
  220. data/lib/relaton/oasis/item.rb +3 -0
  221. data/lib/relaton/oasis/processor.rb +7 -1
  222. data/lib/relaton/oasis.rb +14 -1
  223. data/lib/relaton/ogc/data_fetcher.rb +23 -2
  224. data/lib/relaton/ogc/docidentifier.rb +105 -0
  225. data/lib/relaton/ogc/hit_collection.rb +78 -3
  226. data/lib/relaton/ogc/processor.rb +2 -1
  227. data/lib/relaton/ogc.rb +5 -1
  228. data/lib/relaton/oiml/bibliography.rb +90 -15
  229. data/lib/relaton/oiml/docidentifier.rb +18 -3
  230. data/lib/relaton/omg/docidentifier.rb +67 -0
  231. data/lib/relaton/omg/item.rb +1 -0
  232. data/lib/relaton/omg/processor.rb +1 -0
  233. data/lib/relaton/omg/scraper.rb +61 -16
  234. data/lib/relaton/omg.rb +1 -0
  235. data/lib/relaton/plateau/bibliography.rb +10 -3
  236. data/lib/relaton/plateau/data_fetcher.rb +25 -2
  237. data/lib/relaton/plateau/handbook_parser.rb +8 -1
  238. data/lib/relaton/plateau/hit.rb +10 -2
  239. data/lib/relaton/plateau/hit_collection.rb +31 -11
  240. data/lib/relaton/plateau/processor.rb +3 -1
  241. data/lib/relaton/plateau/technical_report_parser.rb +8 -1
  242. data/lib/relaton/plateau.rb +2 -1
  243. data/lib/relaton/sdo/config.rb +34 -0
  244. data/lib/relaton/sdo/fetcher.rb +52 -0
  245. data/lib/relaton/sdo/logo.rb +95 -0
  246. data/lib/relaton/sdo/name.rb +26 -0
  247. data/lib/relaton/sdo/organization.rb +71 -0
  248. data/lib/relaton/sdo/store.rb +49 -0
  249. data/lib/relaton/sdo.rb +29 -0
  250. data/lib/relaton/version.rb +1 -1
  251. data/lib/relaton/w3c/bibliography.rb +132 -12
  252. data/lib/relaton/w3c/data_fetcher.rb +194 -16
  253. data/lib/relaton/w3c/data_parser.rb +3 -3
  254. data/lib/relaton/w3c/docidentifier.rb +48 -0
  255. data/lib/relaton/w3c/governor.rb +32 -0
  256. data/lib/relaton/w3c/item.rb +3 -0
  257. data/lib/relaton/w3c/pubid.rb +12 -0
  258. data/lib/relaton/w3c/safe_realize.rb +110 -21
  259. data/lib/relaton/w3c.rb +12 -1
  260. data/lib/relaton/xsf/bibliography.rb +61 -1
  261. data/lib/relaton/xsf/data_fetcher.rb +55 -5
  262. data/lib/relaton/xsf/docidentifier.rb +46 -0
  263. data/lib/relaton/xsf/hit_collection.rb +31 -3
  264. data/lib/relaton/xsf/item.rb +6 -0
  265. data/lib/relaton/xsf/processor.rb +1 -0
  266. data/lib/relaton/xsf.rb +5 -1
  267. data/lib/relaton.rb +42 -0
  268. metadata +135 -24
  269. data/lib/relaton/ieee/pub_id.rb +0 -161
  270. data/lib/relaton/index/id_number.rb +0 -30
@@ -6,8 +6,6 @@ module Relaton
6
6
  # In index mode url should be nil.
7
7
  #
8
8
  class FileIO
9
- include IdNumber
10
-
11
9
  # Raised internally when a deserialized id cannot be parsed or is not
12
10
  # understood by the pubid class; `#load_index` rescues it to trigger the
13
11
  # wrong-structure handling (re-download, or stop and log).
@@ -152,11 +150,13 @@ module Relaton
152
150
  load_index(yaml) || []
153
151
  end
154
152
 
155
- # Deserialize and sort by the same narrowing key Type#search bsearches
156
- # on, so binary search always has a consistent total order. The published
157
- # index is only approximately sorted (generated under pubid 1.x base
158
- # semantics); merely detecting sortedness left bsearch disabled and every
159
- # search a full O(n) scan. Sorting here is one-time per load.
153
+ # Deserialize and sort by the same narrowing key Type#search bsearches on
154
+ # — the base document's number, `id.root.number.to_s` (see
155
+ # Type#candidates_by_number) — so binary search always has a consistent
156
+ # total order. The published index is only approximately sorted (generated
157
+ # under pubid 1.x base semantics); merely detecting sortedness left bsearch
158
+ # disabled and every search a full O(n) scan. Sorting here is one-time per
159
+ # load.
160
160
  def deserialize_pubid(index)
161
161
  return index unless @pubid_class
162
162
 
@@ -164,7 +164,7 @@ module Relaton
164
164
  { id: deserialize_id(r[:id]), file: r[:file] }
165
165
  end
166
166
  warn_unless_sorted(deserialized)
167
- deserialized.sort_by! { |r| get_id_number(r[:id]) }
167
+ deserialized.sort_by! { |r| r[:id].root.number.to_s }
168
168
  @sorted = true
169
169
  deserialized
170
170
  end
@@ -184,13 +184,13 @@ module Relaton
184
184
  raise InvalidIndexError, "unsupported id #{raw.inspect}"
185
185
  end
186
186
 
187
- # Log when the loaded index is not already in get_id_number order, so the
187
+ # Log when the loaded index is not already in narrowing-key order, so the
188
188
  # in-memory sort above (and the underlying not-sorted index file) is
189
189
  # visible. Stops at the first out-of-order pair.
190
190
  def warn_unless_sorted(index)
191
191
  prev = nil
192
192
  index.each do |r|
193
- num = get_id_number(r[:id])
193
+ num = r[:id].root.number.to_s
194
194
  if prev && prev > num
195
195
  Util.warn "Index file `#{file}` is not sorted by id number; " \
196
196
  "sorting #{index.size} entries in memory.", progname
@@ -275,7 +275,7 @@ module Relaton
275
275
 
276
276
  def sort_structured_index(index)
277
277
  if @pubid_class && index.first&.dig(:id).is_a?(@pubid_class)
278
- index.sort_by { |item| get_id_number item[:id] }
278
+ index.sort_by { |item| item[:id].root.number.to_s }
279
279
  else
280
280
  index
281
281
  end
@@ -39,7 +39,12 @@ module Relaton
39
39
  def write(file, data)
40
40
  dir = File.dirname file
41
41
  FileUtils.mkdir_p dir
42
- File.write file, data, encoding: "UTF-8"
42
+ # Write the bytes verbatim. `data` is often a raw Net::HTTP body, an
43
+ # ASCII-8BIT string; passing `encoding: "UTF-8"` would *transcode* it and
44
+ # raise Encoding::UndefinedConversionError on any non-ASCII byte (e.g. a
45
+ # UTF-8 `\xC3` in "électrotechnique"). The bytes are already valid UTF-8
46
+ # and `read` decodes them as UTF-8, so a binary write round-trips.
47
+ File.binwrite file, data
43
48
  end
44
49
 
45
50
  #
@@ -15,6 +15,8 @@ module Relaton
15
15
  # @param [String, nil] url external URL to index, used to fetch index for searching files
16
16
  # @param [String, nil] file output file name
17
17
  # @param [Array<Symbol>, nil] id_keys keys to check if index is correct
18
+ # @param [String, nil] pages_url base URL of the Pages site that serves
19
+ # the machine index (manifest + shards), see Relaton::Index::ShardSource
18
20
  #
19
21
  # @return [Relaton::Index::Type] typed index
20
22
  #
@@ -22,7 +24,10 @@ module Relaton
22
24
  if @pool[type.upcase.to_sym]&.actual?(**args)
23
25
  @pool[type.upcase.to_sym]
24
26
  else
25
- @pool[type.upcase.to_sym] = Type.new(type, args[:url], args[:file], args[:id_keys], args[:pubid_class])
27
+ @pool[type.upcase.to_sym] = Type.new(
28
+ type, args[:url], args[:file], args[:id_keys], args[:pubid_class],
29
+ pages_url: args[:pages_url]
30
+ )
26
31
  end
27
32
  end
28
33
 
@@ -0,0 +1,201 @@
1
+ require "json"
2
+ require "net/http"
3
+ require "openssl"
4
+ require "zlib"
5
+
6
+ module Relaton
7
+ module Index
8
+ #
9
+ # Reads a flavor's machine index from its GitHub Pages site (relaton#189).
10
+ #
11
+ # `index/manifest.json` gives the lookup rule, and a query reads only the
12
+ # one `index/shard-NNNNN.json` its document family lives in, in place of the
13
+ # whole `index-vN.zip`. The shard key is `crc32(id.root.number.to_s) %
14
+ # shards`, the expression the relaton-cli producer (`MachineIndex`) writes
15
+ # and `Type#candidates_by_number` narrows on. See
16
+ # `docs/data-repository-format.adoc`, "The machine index on the Pages site".
17
+ #
18
+ # Everything is held in memory for TTL seconds and never written to the
19
+ # storage. A shard 404 is a definitive "not found" (empty buckets are not
20
+ # published). A transport failure raises `Relaton::RequestError`, which
21
+ # `Relaton::Db#net_retry` retries; a machine index this client cannot read
22
+ # raises `Relaton::Index::Error`. Neither falls back to the monolith: the
23
+ # whole index is read only when the manifest publishes no shards, or when
24
+ # the query has no root number to key on.
25
+ #
26
+ class ShardSource
27
+ TTL = 24 * 60 * 60
28
+ VERSION = 2
29
+ KEY = "root-number".freeze
30
+ ALGORITHM = "crc32".freeze
31
+
32
+ NET_ERRORS = [
33
+ SocketError, Timeout::Error, IOError, SystemCallError, OpenSSL::SSL::SSLError,
34
+ Net::HTTPBadResponse, Net::HTTPHeaderSyntaxError, Net::ProtocolError,
35
+ Zlib::Error # a truncated gzip body, which Net::HTTP inflates
36
+
37
+ ].freeze
38
+
39
+ attr_reader :pages_url
40
+
41
+ #
42
+ # @param [String] dir flavor directory name, used in log messages
43
+ # @param [String] pages_url base URL of the Pages site, with a trailing `/`
44
+ # @param [Class] pubid_class the flavor's pubid identifier class
45
+ #
46
+ def initialize(dir, pages_url, pubid_class)
47
+ @pages_url = pages_url.end_with?("/") ? pages_url : "#{pages_url}/"
48
+ # Only the deserialization helpers are used: this FileIO reads and
49
+ # writes no file.
50
+ @file_io = FileIO.new(dir, nil, nil, nil, pubid_class)
51
+ @mutex = Mutex.new
52
+ @state = nil
53
+ end
54
+
55
+ #
56
+ # Rows of the shard that holds the document family of `id`.
57
+ #
58
+ # @param [Pubid::Identifier] id parsed query
59
+ #
60
+ # @return [Array<Hash>] `{ id:, file: }` rows, empty when not found
61
+ #
62
+ def rows(id)
63
+ number = id.root.number.to_s
64
+ state = current_state
65
+ shards = manifest(state)["shards"]
66
+ return load_whole_index(state) if shards.zero? || number.empty?
67
+
68
+ n = Zlib.crc32(number) % shards
69
+ once(state, n) { fetch_shard(n) }
70
+ end
71
+
72
+ #
73
+ # Every row, from the monolith the manifest names (`<index>.zip` on the
74
+ # Pages site). Needed by a String query and a block-only search.
75
+ #
76
+ # @return [Array<Hash>] `{ id:, file: }` rows, sorted by root number
77
+ #
78
+ def whole_index
79
+ load_whole_index current_state
80
+ end
81
+
82
+ private
83
+
84
+ # One generation of cached data. It is replaced as a whole when it
85
+ # expires, so a fetch that finishes late stores into its own generation
86
+ # and never mixes an old shard with a new manifest.
87
+ def new_state
88
+ { created: Time.now, data: {}, locks: {} }
89
+ end
90
+
91
+ def current_state
92
+ @mutex.synchronize do
93
+ @state = new_state if @state.nil? || Time.now - @state[:created] > TTL
94
+ @state
95
+ end
96
+ end
97
+
98
+ # Runs the block once for each key in a generation. The network call is
99
+ # made under a lock for that key only, so lookups of other shards (from
100
+ # the `Relaton::Db` workers pool) do not wait for it. A raise stores
101
+ # nothing, so the next call tries again.
102
+ def once(state, key)
103
+ lock = @mutex.synchronize { state[:locks][key] ||= Mutex.new }
104
+ lock.synchronize do
105
+ data = state[:data]
106
+ return @mutex.synchronize { data[key] } if @mutex.synchronize { data.key?(key) }
107
+
108
+ value = yield
109
+ @mutex.synchronize { data[key] = value }
110
+ end
111
+ end
112
+
113
+ def manifest(state)
114
+ once(state, :manifest) do
115
+ body = get("index/manifest.json")
116
+ raise Error, "no machine index manifest at `#{url('index/manifest.json')}`" unless body
117
+
118
+ check_manifest parse_json(body, "index/manifest.json")
119
+ end
120
+ end
121
+
122
+ def check_manifest(data)
123
+ unless data.is_a?(Hash) && data["version"] == VERSION &&
124
+ data["key"] == KEY && data["algorithm"] == ALGORITHM &&
125
+ data["shards"].is_a?(Integer) && !data["shards"].negative? &&
126
+ data["index"].is_a?(String) && !data["index"].empty?
127
+ raise Error, "unsupported machine index manifest at " \
128
+ "`#{url('index/manifest.json')}`: #{data.inspect}"
129
+ end
130
+
131
+ data
132
+ end
133
+
134
+ def fetch_shard(num)
135
+ path = format("index/shard-%05d.json", num)
136
+ body = get(path)
137
+ return [] unless body
138
+
139
+ data = parse_json(body, path)
140
+ raise Error, "shard `#{url(path)}` is not an array" unless data.is_a?(Array)
141
+
142
+ data.map { |r| { id: deserialize(r["id"], path), file: r["file"] } }
143
+ end
144
+
145
+ def load_whole_index(state)
146
+ path = "#{manifest(state)['index']}.zip"
147
+ once(state, :whole_index) do
148
+ body = get(path)
149
+ raise Error, "no whole index at `#{url(path)}`" unless body
150
+
151
+ yaml = unzip(body)
152
+ raise Error, "empty archive at `#{url(path)}`" unless yaml
153
+
154
+ index = YAML.safe_load(yaml, permitted_classes: [Symbol])
155
+ raise Error, "wrong structure of `#{url(path)}`" unless @file_io.check_basic_format(index)
156
+
157
+ @file_io.deserialize_pubid(index)
158
+ rescue FileIO::InvalidIndexError, Psych::SyntaxError, Zip::Error, Zlib::Error => e
159
+ raise Error, "cannot read `#{url(path)}`: #{e.message}"
160
+ end
161
+ end
162
+
163
+ def unzip(body)
164
+ yaml = nil
165
+ # `open_buffer` does not return the block's value.
166
+ Zip::File.open_buffer(body) { |zip| yaml = zip.entries.first&.get_input_stream&.read }
167
+ yaml
168
+ end
169
+
170
+ def deserialize(raw, path)
171
+ @file_io.deserialize_id(raw)
172
+ rescue FileIO::InvalidIndexError => e
173
+ raise Error, "cannot read shard `#{url(path)}`: #{e.message}"
174
+ end
175
+
176
+ def parse_json(body, path)
177
+ JSON.parse(body)
178
+ rescue JSON::ParserError => e
179
+ raise Error, "cannot parse `#{url(path)}`: #{e.message}"
180
+ end
181
+
182
+ def url(path)
183
+ "#{@pages_url}#{path}"
184
+ end
185
+
186
+ #
187
+ # @return [String, nil] the body, or nil for a 404
188
+ #
189
+ def get(path)
190
+ resp = Net::HTTP.get_response(URI.parse(url(path)))
191
+ case resp.code
192
+ when "200" then resp.body
193
+ when "404" then nil
194
+ else raise Relaton::RequestError, "Could not access #{url(path)}: HTTP #{resp.code}"
195
+ end
196
+ rescue *NET_ERRORS => e
197
+ raise Relaton::RequestError, "Could not access #{url(path)}: #{e.message}"
198
+ end
199
+ end
200
+ end
201
+ end
@@ -4,8 +4,6 @@ module Relaton
4
4
  # Relaton::Index::Type is a class for indexing Relaton files.
5
5
  #
6
6
  class Type
7
- include IdNumber
8
-
9
7
  #
10
8
  # Initialize a new Relaton::Index::Type object
11
9
  #
@@ -15,14 +13,26 @@ module Relaton
15
13
  # @param [Array<Symbol>] id_keys keys of identifier to be used for sorting index
16
14
  # format of index file is checked if id_keys all is provided at least in one of the IDs
17
15
  # @param [Pubid::Identifier, nil] pubid class for deserialization
16
+ # @param [String, nil] pages_url base URL of the Pages site that serves
17
+ # the machine index. With it the type reads the index from there, in
18
+ # memory (see ShardSource), and a parsed query fetches one shard only.
18
19
  #
19
- def initialize(type, url = nil, file = nil, id_keys = nil, pubid_class = nil) # rubocop:disable Metrics/ParameterLists
20
+ def initialize(type, url = nil, file = nil, id_keys = nil, pubid_class = nil, # rubocop:disable Metrics/ParameterLists
21
+ pages_url: nil)
20
22
  @file = file
23
+ @dir = type.to_s.downcase
24
+ @pubid_class = pubid_class
25
+ @pages_url = pages_url
21
26
  filename = file || Index.config.filename
22
- @file_io = FileIO.new type.to_s.downcase, url, filename, id_keys, pubid_class
27
+ @file_io = FileIO.new @dir, url, filename, id_keys, pubid_class
28
+ @source = new_source
23
29
  end
24
30
 
31
+ # With a Pages source the whole index is not kept here: the source holds
32
+ # it, and drops it when its 24 h expire.
25
33
  def index
34
+ return @source.whole_index if @source
35
+
26
36
  @index ||= @file_io.read
27
37
  end
28
38
 
@@ -33,11 +43,14 @@ module Relaton
33
43
  # @param [Hash] **args arguments
34
44
  # @option args [String, nil] :url external URL to index, used to fetch index for searching files
35
45
  # @option args [String, nil] :file output file name
46
+ # @option args [String, nil] :pages_url base URL of the Pages site
36
47
  #
37
48
  # @return [Boolean] true if index is actual, false otherwise
38
49
  #
39
50
  def actual?(**args)
40
- (!args.key?(:url) || args[:url] == @file_io.url) && (!args.key?(:file) || args[:file] == @file)
51
+ (!args.key?(:url) || args[:url] == @file_io.url) &&
52
+ (!args.key?(:file) || args[:file] == @file) &&
53
+ (!args.key?(:pages_url) || args[:pages_url] == @pages_url)
41
54
  end
42
55
 
43
56
  #
@@ -62,15 +75,34 @@ module Relaton
62
75
  end
63
76
 
64
77
  #
65
- # Search index for a given ID
78
+ # Search index for a given ID.
79
+ #
80
+ # Without a block, two identifiers are matched with pubid's asymmetric
81
+ # subset match `id === item[:id]` (`Pubid::SubsetMatch`): the query is the
82
+ # reference, and a component it leaves nil or empty matches any value, so
83
+ # `OGC 12-128` answers with `12-128r19`. That is what a search by
84
+ # reference means. A caller that needs **exact** equality passes
85
+ # `exact: true`, which selects the rows with `item[:id] == id`. A nil
86
+ # component is a wildcard for `===`, not "the document has none": over
87
+ # the fixture indexes the subset match accepts 260 CCSDS pairs (the `-S`
88
+ # suffix) and 20,513 IETF pairs (a draft slug matches each of its
89
+ # versions) that `==` rejects. `Relaton::Ccsds` and `Relaton::Ietf` are
90
+ # the two callers that pass it.
91
+ #
92
+ # A String query keeps the substring match it always had, unless it is
93
+ # `exact:`.
66
94
  #
67
95
  # @param [String, Pubid::Identifier] id ID to search for
96
+ # @param [Boolean] exact match with `==` instead of the subset match
68
97
  #
69
98
  # @return [Array<Hash>] search results
70
99
  #
71
- def search(id = nil, &block)
100
+ def search(id = nil, exact: false, &block)
101
+ raise ArgumentError, "search takes exact: or a block, not both" if exact && block
102
+
72
103
  items = search_candidates(id)
73
104
  return items.select(&block) if block
105
+ return items.select { |i| i[:id] == id } if exact
74
106
 
75
107
  items.select { |i| match_item(i, id) }
76
108
  end
@@ -90,7 +122,10 @@ module Relaton
90
122
  # @return [void]
91
123
  #
92
124
  def remove_file
93
- @file_io.remove
125
+ # A Pages source has no file: its FileIO would resolve the default
126
+ # filename against the working directory.
127
+ @file_io.remove unless @source
128
+ @source = new_source
94
129
  @index = nil
95
130
  @id_lookup = nil
96
131
  end
@@ -114,7 +149,15 @@ module Relaton
114
149
  end
115
150
  end
116
151
 
152
+ def new_source
153
+ ShardSource.new(@dir, @pages_url, @pubid_class) if @pages_url
154
+ end
155
+
117
156
  def search_candidates(id)
157
+ # A parsed query reads only its own shard. `@index`, once loaded, is
158
+ # never refreshed, so the shards stay the fresher answer.
159
+ return @source.rows(id) if @source && id && !id.is_a?(String)
160
+
118
161
  # index needs to be created to check if sorted
119
162
  idx = index
120
163
  if @file_io.sorted && id && !id.is_a?(String)
@@ -124,8 +167,14 @@ module Relaton
124
167
  end
125
168
  end
126
169
 
170
+ # Narrowing key: the base *document's* number as a string. `#root` walks a
171
+ # supplement/amendment/corrigendum's `.base` chain to the origin (and
172
+ # returns self for a base document), so a document and all its wrappers
173
+ # share one key and cluster together. `.to_s` because the key is compared
174
+ # as a string (a pubid number Component is not `<`/`>`-comparable), and
175
+ # FileIO sorts the index by this exact same key so bsearch stays valid.
127
176
  def candidates_by_number(id)
128
- target = get_id_number(id)
177
+ target = id.root.number.to_s
129
178
  left = bsearch_left(target)
130
179
  return [] unless left
131
180
 
@@ -135,23 +184,25 @@ module Relaton
135
184
 
136
185
  def bsearch_left(target)
137
186
  index.bsearch_index do |item|
138
- get_id_number(item[:id]) >= target
187
+ item[:id].root.number.to_s >= target
139
188
  end
140
189
  end
141
190
 
142
191
  def bsearch_right(target)
143
192
  index.bsearch_index do |item|
144
- get_id_number(item[:id]) > target
193
+ item[:id].root.number.to_s > target
145
194
  end || index.size
146
195
  end
147
196
 
197
+ # The query is the reference, so two identifiers take the subset match.
198
+ # See #search for the exact-equality escape hatch.
148
199
  def match_item(item, id)
149
200
  if item[:id].is_a?(String)
150
201
  item[:id].include?(id.is_a?(String) ? id : id.to_s)
151
202
  elsif id.is_a?(String)
152
203
  item[:id].to_s.include?(id)
153
204
  else
154
- item[:id] == id
205
+ id === item[:id]
155
206
  end
156
207
  end
157
208
  end
data/lib/relaton/index.rb CHANGED
@@ -5,13 +5,14 @@ require "zip"
5
5
  require "relaton/logger"
6
6
 
7
7
  require_relative "version"
8
+ require_relative "core/request_error"
8
9
  require_relative "index/file_storage"
9
10
  require_relative "index/config"
10
11
  require_relative "index/util"
11
- require_relative "index/id_number"
12
12
  require_relative "index/pool"
13
13
  require_relative "index/type"
14
14
  require_relative "index/file_io"
15
+ require_relative "index/shard_source"
15
16
 
16
17
  module Relaton
17
18
  module Index
@@ -43,14 +43,14 @@ module Relaton
43
43
  if year&.respond_to?(:to_i)
44
44
  query_pubid.root.date = ::Pubid::Components::Date.new(year: year.to_s)
45
45
  end
46
- query_pubid.root.all_parts = opts[:all_parts] if opts[:all_parts]
46
+ query_pubid = query_pubid.to_all_parts if opts[:all_parts]
47
47
  Util.info "Fetching from Relaton repository ...", key: query_pubid.to_s
48
48
 
49
49
  hits, missed_year_ids = isobib_search_filter(query_pubid, opts)
50
50
  tip_ids = look_up_with_any_types_stages(hits, ref, opts)
51
51
 
52
52
  date_filter = opts[:publication_date_before] || opts[:publication_date_after]
53
- if date_filter && !query_pubid.root.all_parts
53
+ if date_filter && !query_pubid.all_parts
54
54
  ret = find_match_by_date(hits, query_pubid, opts)
55
55
  else
56
56
  ret = hits.fetch_doc(date_filter ? opts : {})
@@ -69,9 +69,6 @@ module Relaton
69
69
  return ret if get_all
70
70
 
71
71
  ret.to_most_recent_reference
72
- rescue Parslet::ParseFailed
73
- Util.warn "Is not recognized as a standards identifier.", key: code
74
- nil
75
72
  end
76
73
 
77
74
  # @param query_pubid [Pubid::Iso::Identifier]
@@ -285,8 +282,8 @@ module Relaton
285
282
  def check_year(year, hit) # rubocop:disable Metrics/AbcSize
286
283
  pub = hit.pubid
287
284
  own_year = pub.date&.year.to_s
288
- base_year = pub.base_identifier&.date&.year.to_s
289
- if pub.base_identifier.nil?
285
+ base_year = pub.base&.date&.year.to_s
286
+ if pub.base.nil?
290
287
  own_year == year.to_s
291
288
  else
292
289
  base_year == year.to_s || own_year == year.to_s
@@ -307,7 +304,11 @@ module Relaton
307
304
  Util.info "TIP: Matches exist for #{ids}.", key: pubid.to_s
308
305
  end
309
306
 
310
- if pubid.part
307
+ # `pubid` may itself be the all-parts wrapper (an all-parts query that
308
+ # still found nothing): its own `part` is always nil (never delegated),
309
+ # so read the underlying document via `#root`, and skip suggesting
310
+ # "(all parts)" to a caller who already asked for it.
311
+ if pubid.all_parts || pubid.root.part
311
312
  Util.info "TIP: If it cannot be found, the document may no longer be published in parts.", key: pubid.to_s
312
313
  else
313
314
  Util.info "TIP: If you wish to cite all document parts for the reference, " \
@@ -324,6 +325,12 @@ module Relaton
324
325
  pubid = ::Pubid::Iso::Identifier.parse(ref_no_type_stage)
325
326
  resp, = isobib_search_filter(pubid, opts, any_types_stages: true)
326
327
  resp.map &:pubid
328
+ rescue ::Pubid::Errors::ParseError
329
+ # The type/stage-stripped variant is a machine-derived probe, not the
330
+ # user's identifier; if it doesn't parse there are simply no
331
+ # alternative-type/stage candidates. The original reference already
332
+ # parsed in #get, so a failure here must not abort the lookup.
333
+ []
327
334
  end
328
335
 
329
336
  #
@@ -354,11 +361,11 @@ module Relaton
354
361
  #
355
362
  def filter_hits(hit_collection, query_pubid, any_types_stages) # rubocop:disable Metrics/AbcSize
356
363
  # filter out
357
- excludings = build_excludings(query_pubid.root.all_parts, any_types_stages)
364
+ excludings = build_excludings(query_pubid.all_parts, any_types_stages)
358
365
  no_year_ref = hit_collection.ref_pubid_no_year.exclude(*excludings)
359
366
  hit_collection.select! do |i|
360
367
  pubid_match?(i.pubid, query_pubid, excludings, no_year_ref) &&
361
- !(query_pubid.root.all_parts && i.pubid.part.nil?)
368
+ !(query_pubid.all_parts && i.pubid.part.nil?)
362
369
  end
363
370
 
364
371
  filter_hits_by_year(hit_collection, query_pubid.root.date&.year)
@@ -380,7 +387,7 @@ module Relaton
380
387
  if pubid.is_a? String then pubid == query_pubid.to_s
381
388
  else
382
389
  pubid = pubid.dup
383
- pubid.base_identifier = pubid.base_identifier.exclude(:date, :edition) if pubid.base_identifier
390
+ pubid.base = pubid.base.exclude(:date, :edition) if pubid.base
384
391
  normalize_compound_part(pubid.exclude(*excludings)) == no_year_ref
385
392
  end
386
393
  end
@@ -394,12 +401,10 @@ module Relaton
394
401
  # before comparing. `exclude` returns a fresh instance, so mutating this
395
402
  # copy is safe. Remove once pubid create() splits compound parts itself.
396
403
  def normalize_compound_part(pubid)
397
- num = pubid.part&.value.to_s
404
+ num = pubid.part.to_s
398
405
  return pubid unless pubid.subpart.nil? && num.include?("-")
399
406
 
400
- head, tail = num.split("-", 2)
401
- pubid.part = ::Pubid::Iso::Components::Code.new(value: head)
402
- pubid.subpart = ::Pubid::Iso::Components::Code.new(value: tail)
407
+ pubid.part, pubid.subpart = num.split("-", 2)
403
408
  pubid
404
409
  end
405
410
  end
@@ -190,9 +190,9 @@ module Relaton
190
190
 
191
191
  def amend_base(ref)
192
192
  pubid = ::Pubid::Iso::Identifier.parse(ref)
193
- return nil unless pubid.base_identifier
193
+ return nil unless pubid.base
194
194
 
195
- pubid.base_identifier.to_s
195
+ pubid.base.to_s
196
196
  rescue StandardError
197
197
  nil
198
198
  end
@@ -284,7 +284,7 @@ module Relaton
284
284
  # is expected to parse; if one does not (`docid.pubid` is nil) record it
285
285
  # so `report_errors` raises a tracked GitHub issue at the end, and skip
286
286
  # the index entry rather than indexing a raw string (which would crash
287
- # the index sort: `get_id_number` calls `.number` on the id). The data
287
+ # the index sort: it calls `.root.number` on the id). The data
288
288
  # file is still written, so the document is not lost — only unindexed
289
289
  # until its id parses.
290
290
  def index_primary(docid, file)
@@ -100,7 +100,11 @@ module Relaton
100
100
  def docidentifier
101
101
  ids = []
102
102
  if pubid
103
- ids << Docidentifier.new(content: pubid, type: "ISO", primary: true)
103
+ primary = Docidentifier.new(content: pubid, type: "ISO", primary: true)
104
+ ids << primary
105
+ if (undated = undated_docid(primary))
106
+ ids << undated
107
+ end
104
108
  if (ref = iso_reference_pubid)
105
109
  ids << Docidentifier.new(content: ref, type: "iso-reference")
106
110
  end
@@ -114,6 +118,16 @@ module Relaton
114
118
  ids
115
119
  end
116
120
 
121
+ # Build the undated reference docid (e.g. `ISO 10303-52` from
122
+ # `ISO 10303-52:2011`). Returns nil when the reference carries no year
123
+ # (the undated form would just duplicate the primary).
124
+ def undated_docid(primary)
125
+ undated = Docidentifier.new(content: pubid, type: "iso-undated")
126
+ undated unless undated.to_s == primary.to_s
127
+ rescue StandardError
128
+ nil
129
+ end
130
+
117
131
  def safe_urn_docid
118
132
  return nil unless urn_pubid
119
133
 
@@ -385,9 +399,9 @@ module Relaton
385
399
  end
386
400
 
387
401
  def base_relation
388
- return [] unless pubid&.base_identifier
402
+ return [] unless pubid&.base
389
403
 
390
- [relation_for(pubid.base_identifier.to_s, "updates")]
404
+ [relation_for(pubid.base.to_s, "updates")]
391
405
  end
392
406
 
393
407
  def relation_for(ref, type)