relaton 3.0.0.pre.alpha.1 → 3.0.0.pre.alpha.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (270) hide show
  1. checksums.yaml +4 -4
  2. data/README.adoc +57 -1
  3. data/bin/console +0 -1
  4. data/lib/relaton/3gpp/bibliography.rb +82 -7
  5. data/lib/relaton/3gpp/data_fetcher.rb +51 -3
  6. data/lib/relaton/3gpp/docidentifier.rb +114 -0
  7. data/lib/relaton/3gpp/item.rb +6 -0
  8. data/lib/relaton/3gpp/parser.rb +1 -1
  9. data/lib/relaton/3gpp/processor.rb +4 -1
  10. data/lib/relaton/3gpp.rb +5 -1
  11. data/lib/relaton/adobe/bibdata.rb +8 -0
  12. data/lib/relaton/adobe/bibitem.rb +8 -0
  13. data/lib/relaton/adobe/bibliography.rb +92 -0
  14. data/lib/relaton/adobe/docidentifier.rb +49 -0
  15. data/lib/relaton/adobe/doctype.rb +14 -0
  16. data/lib/relaton/adobe/ext.rb +32 -0
  17. data/lib/relaton/adobe/item.rb +15 -0
  18. data/lib/relaton/adobe/item_base.rb +18 -0
  19. data/lib/relaton/adobe/item_data.rb +6 -0
  20. data/lib/relaton/adobe/processor.rb +45 -0
  21. data/lib/relaton/adobe/util.rb +8 -0
  22. data/lib/relaton/adobe.rb +37 -0
  23. data/lib/relaton/bib/model/address.rb +2 -2
  24. data/lib/relaton/bib/model/docidentifier.rb +24 -9
  25. data/lib/relaton/bib/model/localized_string.rb +1 -1
  26. data/lib/relaton/bib/model/structured_identifier.rb +10 -9
  27. data/lib/relaton/bib/sanitizer.rb +202 -6
  28. data/lib/relaton/bib.rb +0 -2
  29. data/lib/relaton/bipm/bibliography.rb +159 -10
  30. data/lib/relaton/bipm/data_fetcher.rb +26 -2
  31. data/lib/relaton/bipm/data_outcomes_parser.rb +3 -4
  32. data/lib/relaton/bipm/id_parser.rb +5 -4
  33. data/lib/relaton/bipm/model/structured_identifier.rb +21 -0
  34. data/lib/relaton/bipm/processor.rb +2 -2
  35. data/lib/relaton/bipm/rawdata_bipm_metrologia/fetcher.rb +2 -4
  36. data/lib/relaton/bipm/si_brochure_parser.rb +5 -3
  37. data/lib/relaton/bipm.rb +7 -1
  38. data/lib/relaton/bsi/bibliography.rb +115 -43
  39. data/lib/relaton/bsi/hit.rb +14 -0
  40. data/lib/relaton/bsi/hit_collection.rb +15 -16
  41. data/lib/relaton/bsi/model/docidentifier.rb +99 -1
  42. data/lib/relaton/bsi/processor.rb +1 -0
  43. data/lib/relaton/calconnect/bibliography.rb +12 -14
  44. data/lib/relaton/calconnect/data_fetcher.rb +77 -9
  45. data/lib/relaton/calconnect/docidentifier.rb +80 -0
  46. data/lib/relaton/calconnect/hit_collection.rb +65 -57
  47. data/lib/relaton/calconnect/model/item.rb +7 -0
  48. data/lib/relaton/calconnect/processor.rb +7 -1
  49. data/lib/relaton/calconnect.rb +11 -1
  50. data/lib/relaton/ccsds/data/fetcher.rb +17 -12
  51. data/lib/relaton/ccsds/data/parser.rb +1 -1
  52. data/lib/relaton/ccsds/hit_collection.rb +6 -1
  53. data/lib/relaton/ccsds/model/docidentifier.rb +121 -0
  54. data/lib/relaton/ccsds/model/item.rb +2 -0
  55. data/lib/relaton/cen/bibliography.rb +75 -47
  56. data/lib/relaton/cen/hit.rb +16 -1
  57. data/lib/relaton/cen/hit_collection.rb +59 -11
  58. data/lib/relaton/cen/model/docidentifier.rb +92 -1
  59. data/lib/relaton/cen/processor.rb +13 -8
  60. data/lib/relaton/cen/scraper.rb +13 -5
  61. data/lib/relaton/cen.rb +1 -0
  62. data/lib/relaton/cie/data_fetcher.rb +215 -30
  63. data/lib/relaton/cie/processor.rb +3 -1
  64. data/lib/relaton/cie/scrapper.rb +15 -2
  65. data/lib/relaton/cie.rb +2 -1
  66. data/lib/relaton/core/data_fetcher.rb +150 -3
  67. data/lib/relaton/core/governor.rb +320 -0
  68. data/lib/relaton/core/pacer.rb +134 -0
  69. data/lib/relaton/core/processor.rb +19 -0
  70. data/lib/relaton/core/request_error.rb +14 -0
  71. data/lib/relaton/core.rb +3 -0
  72. data/lib/relaton/db/registry.rb +41 -1
  73. data/lib/relaton/doi/crossref.rb +19 -2
  74. data/lib/relaton/doi/parser.rb +109 -15
  75. data/lib/relaton/easc/bibdata.rb +8 -0
  76. data/lib/relaton/easc/bibitem.rb +8 -0
  77. data/lib/relaton/easc/bibliography.rb +95 -0
  78. data/lib/relaton/easc/docidentifier.rb +100 -0
  79. data/lib/relaton/easc/doctype.rb +14 -0
  80. data/lib/relaton/easc/ext.rb +44 -0
  81. data/lib/relaton/easc/item.rb +13 -0
  82. data/lib/relaton/easc/item_base.rb +18 -0
  83. data/lib/relaton/easc/item_data.rb +6 -0
  84. data/lib/relaton/easc/processor.rb +46 -0
  85. data/lib/relaton/easc/util.rb +8 -0
  86. data/lib/relaton/easc.rb +35 -0
  87. data/lib/relaton/ecma/bibliography.rb +93 -25
  88. data/lib/relaton/ecma/data_fetcher.rb +71 -12
  89. data/lib/relaton/ecma/docidentifier.rb +124 -0
  90. data/lib/relaton/ecma/item.rb +2 -0
  91. data/lib/relaton/ecma/memento_parser.rb +1 -1
  92. data/lib/relaton/ecma/page_fetcher.rb +15 -3
  93. data/lib/relaton/ecma/parser_common.rb +2 -2
  94. data/lib/relaton/ecma/processor.rb +4 -1
  95. data/lib/relaton/ecma/standard_parser.rb +2 -2
  96. data/lib/relaton/ecma.rb +10 -1
  97. data/lib/relaton/etsi/bibliography.rb +67 -2
  98. data/lib/relaton/etsi/data_fetcher.rb +43 -4
  99. data/lib/relaton/etsi/processor.rb +3 -1
  100. data/lib/relaton/etsi.rb +2 -1
  101. data/lib/relaton/gb/bibliography.rb +55 -29
  102. data/lib/relaton/gb/docidentifier.rb +58 -9
  103. data/lib/relaton/gb/processor.rb +3 -0
  104. data/lib/relaton/gb/scraper.rb +27 -10
  105. data/lib/relaton/gost/bibdata.rb +8 -0
  106. data/lib/relaton/gost/bibitem.rb +8 -0
  107. data/lib/relaton/gost/bibliography.rb +107 -0
  108. data/lib/relaton/gost/docidentifier.rb +80 -0
  109. data/lib/relaton/gost/doctype.rb +16 -0
  110. data/lib/relaton/gost/ext.rb +46 -0
  111. data/lib/relaton/gost/item.rb +15 -0
  112. data/lib/relaton/gost/item_base.rb +18 -0
  113. data/lib/relaton/gost/item_data.rb +6 -0
  114. data/lib/relaton/gost/processor.rb +49 -0
  115. data/lib/relaton/gost/util.rb +8 -0
  116. data/lib/relaton/gost.rb +36 -0
  117. data/lib/relaton/iala/bibdata.rb +8 -0
  118. data/lib/relaton/iala/bibitem.rb +8 -0
  119. data/lib/relaton/iala/bibliography.rb +146 -0
  120. data/lib/relaton/iala/docidentifier.rb +89 -0
  121. data/lib/relaton/iala/doctype.rb +18 -0
  122. data/lib/relaton/iala/ext.rb +32 -0
  123. data/lib/relaton/iala/item.rb +21 -0
  124. data/lib/relaton/iala/item_base.rb +18 -0
  125. data/lib/relaton/iala/item_data.rb +6 -0
  126. data/lib/relaton/iala/processor.rb +43 -0
  127. data/lib/relaton/iala/relation.rb +7 -0
  128. data/lib/relaton/iala/util.rb +8 -0
  129. data/lib/relaton/iala.rb +35 -0
  130. data/lib/relaton/iana/bibliography.rb +67 -14
  131. data/lib/relaton/iana/data_fetcher.rb +35 -5
  132. data/lib/relaton/iana/processor.rb +3 -1
  133. data/lib/relaton/iana.rb +12 -1
  134. data/lib/relaton/iec/data_fetcher.rb +7 -1
  135. data/lib/relaton/iec/hit_collection.rb +1 -1
  136. data/lib/relaton/iec/model/docidentifier.rb +9 -5
  137. data/lib/relaton/iec/model/ext.rb +2 -2
  138. data/lib/relaton/iec/processor.rb +1 -0
  139. data/lib/relaton/ieee/bibliography.rb +25 -3
  140. data/lib/relaton/ieee/data_fetcher.rb +158 -17
  141. data/lib/relaton/ieee/idams_parser.rb +18 -11
  142. data/lib/relaton/ieee/processor.rb +4 -1
  143. data/lib/relaton/ieee/rawbib_id_parser.rb +291 -86
  144. data/lib/relaton/ieee.rb +2 -1
  145. data/lib/relaton/ietf/data_fetcher.rb +295 -12
  146. data/lib/relaton/ietf/processor.rb +7 -3
  147. data/lib/relaton/ietf/rfc/entry.rb +39 -3
  148. data/lib/relaton/ietf/scraper.rb +69 -36
  149. data/lib/relaton/ietf.rb +4 -1
  150. data/lib/relaton/iho/bibliography.rb +1 -1
  151. data/lib/relaton/iho/docidentifier.rb +1 -1
  152. data/lib/relaton/index/file_io.rb +11 -11
  153. data/lib/relaton/index/file_storage.rb +6 -1
  154. data/lib/relaton/index/pool.rb +6 -1
  155. data/lib/relaton/index/shard_source.rb +201 -0
  156. data/lib/relaton/index/type.rb +63 -12
  157. data/lib/relaton/index.rb +2 -1
  158. data/lib/relaton/iso/bibliography.rb +20 -15
  159. data/lib/relaton/iso/data_fetcher.rb +3 -3
  160. data/lib/relaton/iso/data_parser.rb +17 -3
  161. data/lib/relaton/iso/hit_collection.rb +27 -15
  162. data/lib/relaton/iso/item_data.rb +22 -0
  163. data/lib/relaton/iso/model/docidentifier.rb +24 -12
  164. data/lib/relaton/iso/processor.rb +1 -0
  165. data/lib/relaton/iso/scraper.rb +19 -3
  166. data/lib/relaton/itu/bibliography.rb +9 -4
  167. data/lib/relaton/itu/data_crawler_r.rb +664 -0
  168. data/lib/relaton/itu/data_fetcher.rb +496 -50
  169. data/lib/relaton/itu/data_merge_r.rb +149 -0
  170. data/lib/relaton/itu/data_parser_r.rb +163 -89
  171. data/lib/relaton/itu/data_parser_t.rb +228 -0
  172. data/lib/relaton/itu/family_cache.rb +177 -0
  173. data/lib/relaton/itu/governor.rb +56 -0
  174. data/lib/relaton/itu/hit.rb +9 -3
  175. data/lib/relaton/itu/hit_collection.rb +258 -86
  176. data/lib/relaton/itu/model/docidentifier.rb +67 -1
  177. data/lib/relaton/itu/model/structured_identifier.rb +19 -0
  178. data/lib/relaton/itu/processor.rb +10 -4
  179. data/lib/relaton/itu/pubid.rb +27 -5
  180. data/lib/relaton/itu/recommendation_fields.rb +334 -0
  181. data/lib/relaton/itu/recommendation_parser.rb +18 -149
  182. data/lib/relaton/itu/scraper.rb +13 -3
  183. data/lib/relaton/itu.rb +2 -1
  184. data/lib/relaton/jcgm/bibdata.rb +8 -0
  185. data/lib/relaton/jcgm/bibitem.rb +8 -0
  186. data/lib/relaton/jcgm/bibliography.rb +97 -0
  187. data/lib/relaton/jcgm/data_fetcher.rb +81 -0
  188. data/lib/relaton/jcgm/docidentifier.rb +102 -0
  189. data/lib/relaton/jcgm/doctype.rb +12 -0
  190. data/lib/relaton/jcgm/ext.rb +23 -0
  191. data/lib/relaton/jcgm/item.rb +20 -0
  192. data/lib/relaton/jcgm/item_base.rb +18 -0
  193. data/lib/relaton/jcgm/item_data.rb +6 -0
  194. data/lib/relaton/jcgm/meetings_parser.rb +175 -0
  195. data/lib/relaton/jcgm/processor.rb +71 -0
  196. data/lib/relaton/jcgm/relation.rb +9 -0
  197. data/lib/relaton/jcgm/structured_identifier.rb +40 -0
  198. data/lib/relaton/jcgm/util.rb +8 -0
  199. data/lib/relaton/jcgm.rb +24 -0
  200. data/lib/relaton/jis/bibliography.rb +8 -10
  201. data/lib/relaton/jis/data_fetcher.rb +21 -19
  202. data/lib/relaton/jis/docidentifier.rb +104 -5
  203. data/lib/relaton/jis/hit.rb +18 -23
  204. data/lib/relaton/jis/hit_collection.rb +19 -18
  205. data/lib/relaton/jis/processor.rb +1 -1
  206. data/lib/relaton/jis.rb +2 -3
  207. data/lib/relaton/logger/channels/gh_issue.rb +78 -13
  208. data/lib/relaton/nist/data_fetcher.rb +63 -13
  209. data/lib/relaton/nist/docidentifier.rb +165 -0
  210. data/lib/relaton/nist/item.rb +2 -0
  211. data/lib/relaton/nist/item_base.rb +16 -0
  212. data/lib/relaton/nist/mods_parser.rb +38 -12
  213. data/lib/relaton/nist/processor.rb +2 -1
  214. data/lib/relaton/nist/relation.rb +3 -0
  215. data/lib/relaton/nist/scraper.rb +6 -3
  216. data/lib/relaton/oasis/bibliography.rb +147 -6
  217. data/lib/relaton/oasis/data_fetcher.rb +41 -5
  218. data/lib/relaton/oasis/data_parser_utils.rb +37 -3
  219. data/lib/relaton/oasis/docidentifier.rb +54 -0
  220. data/lib/relaton/oasis/item.rb +3 -0
  221. data/lib/relaton/oasis/processor.rb +7 -1
  222. data/lib/relaton/oasis.rb +14 -1
  223. data/lib/relaton/ogc/data_fetcher.rb +23 -2
  224. data/lib/relaton/ogc/docidentifier.rb +105 -0
  225. data/lib/relaton/ogc/hit_collection.rb +78 -3
  226. data/lib/relaton/ogc/processor.rb +2 -1
  227. data/lib/relaton/ogc.rb +5 -1
  228. data/lib/relaton/oiml/bibliography.rb +90 -15
  229. data/lib/relaton/oiml/docidentifier.rb +18 -3
  230. data/lib/relaton/omg/docidentifier.rb +67 -0
  231. data/lib/relaton/omg/item.rb +1 -0
  232. data/lib/relaton/omg/processor.rb +1 -0
  233. data/lib/relaton/omg/scraper.rb +61 -16
  234. data/lib/relaton/omg.rb +1 -0
  235. data/lib/relaton/plateau/bibliography.rb +10 -3
  236. data/lib/relaton/plateau/data_fetcher.rb +25 -2
  237. data/lib/relaton/plateau/handbook_parser.rb +8 -1
  238. data/lib/relaton/plateau/hit.rb +10 -2
  239. data/lib/relaton/plateau/hit_collection.rb +31 -11
  240. data/lib/relaton/plateau/processor.rb +3 -1
  241. data/lib/relaton/plateau/technical_report_parser.rb +8 -1
  242. data/lib/relaton/plateau.rb +2 -1
  243. data/lib/relaton/sdo/config.rb +34 -0
  244. data/lib/relaton/sdo/fetcher.rb +52 -0
  245. data/lib/relaton/sdo/logo.rb +95 -0
  246. data/lib/relaton/sdo/name.rb +26 -0
  247. data/lib/relaton/sdo/organization.rb +71 -0
  248. data/lib/relaton/sdo/store.rb +49 -0
  249. data/lib/relaton/sdo.rb +29 -0
  250. data/lib/relaton/version.rb +1 -1
  251. data/lib/relaton/w3c/bibliography.rb +132 -12
  252. data/lib/relaton/w3c/data_fetcher.rb +194 -16
  253. data/lib/relaton/w3c/data_parser.rb +3 -3
  254. data/lib/relaton/w3c/docidentifier.rb +48 -0
  255. data/lib/relaton/w3c/governor.rb +32 -0
  256. data/lib/relaton/w3c/item.rb +3 -0
  257. data/lib/relaton/w3c/pubid.rb +12 -0
  258. data/lib/relaton/w3c/safe_realize.rb +110 -21
  259. data/lib/relaton/w3c.rb +12 -1
  260. data/lib/relaton/xsf/bibliography.rb +61 -1
  261. data/lib/relaton/xsf/data_fetcher.rb +55 -5
  262. data/lib/relaton/xsf/docidentifier.rb +46 -0
  263. data/lib/relaton/xsf/hit_collection.rb +31 -3
  264. data/lib/relaton/xsf/item.rb +6 -0
  265. data/lib/relaton/xsf/processor.rb +1 -0
  266. data/lib/relaton/xsf.rb +5 -1
  267. data/lib/relaton.rb +42 -0
  268. metadata +135 -24
  269. data/lib/relaton/ieee/pub_id.rb +0 -161
  270. data/lib/relaton/index/id_number.rb +0 -30
@@ -5,14 +5,20 @@ module Relaton
5
5
  class HitCollection < Relaton::Core::HitCollection
6
6
  ENDPOINT = "https://raw.githubusercontent.com/relaton/relaton-data-plateau/v2/"
7
7
 
8
+ # An edition suffix in either the canonical Japanese form (`第1.0版`) or the
9
+ # legacy Latin form (`1.0`), stripped in #to_all_editions to derive the
10
+ # edition-less family id from a relaton-bib docidentifier *string* (the
11
+ # fetched document is not a pubid). Matching itself is done on pubid objects.
12
+ EDITION_SUFFIX = / (?:第[\d.]+版|\d+\.\d+)$/
13
+
14
+ # An edition-less reference selects every edition of the document with
15
+ # pubid's subset match `pubid_ref === row`: the omitted edition matches
16
+ # any value, and pubid declares `annex` strict for PLATEAU, so
17
+ # `PLATEAU Handbook #03` does not reach `#03-1`. A reference with an
18
+ # edition selects that one row with `==`.
8
19
  def find
9
- @array = index.search do |row|
10
- if all_editions?
11
- row[:id].sub(/ \d+\.\d+$/, "") == @ref
12
- else
13
- row[:id] == @ref
14
- end
15
- end.map { |row| Hit.new(row, self) }
20
+ @array = index.search(pubid_ref, exact: !all_editions?)
21
+ .map { |row| Hit.new(row, self) }
16
22
  self
17
23
  end
18
24
 
@@ -24,14 +30,28 @@ module Relaton
24
30
 
25
31
  def index
26
32
  @index ||= Relaton::Index.find_or_create(
27
- :plateau, url: "#{ENDPOINT}#{INDEXFILE}.zip", file: "#{INDEXFILE}.yaml"
33
+ :plateau, url: "#{ENDPOINT}#{INDEXFILE}.zip", file: "#{INDEXFILE}.yaml",
34
+ pubid_class: ::Pubid::Plateau::Identifier
28
35
  )
29
36
  end
30
37
 
31
38
  private
32
39
 
40
+ # `#ref` (the raw query string, from Core::HitCollection) parsed into a
41
+ # canonical Pubid::Plateau::Identifier (memoized). Canonical (`第1.0版`),
42
+ # edition-less (`#00`), and legacy Latin (`#00 1.0`) references all
43
+ # parse — pubid normalizes Latin input to the canonical id
44
+ # (metanorma/pubid #269) — so search accepts both forms. An unrecognized
45
+ # reference raises; like ISO and 3GPP we let it propagate — relaton-cli
46
+ # rescues Pubid::Errors::Error. See lib/relaton/plateau/CLAUDE.md.
47
+ def pubid_ref
48
+ @pubid_ref ||= ::Pubid::Plateau.parse(ref)
49
+ end
50
+
51
+ # A reference with no edition (`PLATEAU Handbook #00`, or any Technical
52
+ # Report — TRs carry no edition) asks for all editions of the document.
33
53
  def all_editions?
34
- @ref.match?(/ #\d+$/)
54
+ pubid_ref.edition.nil?
35
55
  end
36
56
 
37
57
  def to_all_editions
@@ -43,12 +63,12 @@ module Relaton
43
63
  end
44
64
  docid = bibitem.docidentifier.map do |d|
45
65
  Bib::Docidentifier.new(
46
- content: d.content.sub(/ \d+\.\d+$/, ""), type: d.type, primary: d.primary
66
+ content: d.content.sub(EDITION_SUFFIX, ""), type: d.type, primary: d.primary
47
67
  )
48
68
  end
49
69
  ItemData.new(
50
70
  docidentifier: docid,
51
- docnumber: bibitem.docnumber.sub(/ \d+\.\d+$/, ""),
71
+ docnumber: bibitem.docnumber.sub(EDITION_SUFFIX, ""),
52
72
  title: bibitem.title,
53
73
  contributor: bibitem.contributor,
54
74
  relation: relations
@@ -42,7 +42,9 @@ module Relaton
42
42
 
43
43
  def remove_index_file
44
44
  require_relative "../plateau"
45
- Relaton::Index.find_or_create(:plateau, url: true, file: "#{INDEXFILE}.yaml").remove_file
45
+ Relaton::Index.find_or_create(
46
+ :plateau, url: true, file: "#{INDEXFILE}.yaml"
47
+ ).remove_file
46
48
  end
47
49
  end
48
50
  end
@@ -10,7 +10,14 @@ module Relaton
10
10
 
11
11
  def parse_docnumber
12
12
  @errors[:tr_docnumber] &&= @entry["slug"].nil? || @entry["slug"].to_s.empty?
13
- "Technical Report ##{@entry["slug"]} #{edition_number}"
13
+ "Technical Report ##{canonical_slug}"
14
+ end
15
+
16
+ # Canonical sub-number separator is "-" (Pubid::Plateau); source slugs use
17
+ # "_" (e.g. "46_1" -> "46-1"). PLATEAU Technical Report ids carry no
18
+ # edition in canonical form, so #edition_number is not part of the id.
19
+ def canonical_slug
20
+ @entry["slug"].to_s.tr("_", "-")
14
21
  end
15
22
 
16
23
  def parse_abstract
@@ -1,5 +1,6 @@
1
1
  require "net/http"
2
2
  require "uri"
3
+ require "pubid"
3
4
  require "relaton/index"
4
5
  require "relaton/iso"
5
6
  require_relative "version"
@@ -17,7 +18,7 @@ require_relative "plateau/processor"
17
18
 
18
19
  module Relaton
19
20
  module Plateau
20
- INDEXFILE = "index-v1"
21
+ INDEXFILE = "index-v2"
21
22
 
22
23
  class Error < StandardError; end
23
24
 
@@ -0,0 +1,34 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Relaton
4
+ module Sdo
5
+ # Configuration for the SDO org/logo store: where the index manifest lives,
6
+ # where to cache it, and for how long. Mirrors Relaton::Index::Config; the
7
+ # storage backend is swappable (e.g. an S3-backed object) via `storage=`.
8
+ class Config
9
+ # Default published index of the relaton-data-sdo data repo.
10
+ DEFAULT_URL =
11
+ "https://raw.githubusercontent.com/relaton/relaton-data-sdo/main/index.yaml"
12
+
13
+ attr_accessor :url, :storage, :storage_dir, :ttl
14
+
15
+ def initialize
16
+ @url = DEFAULT_URL
17
+ @storage = Relaton::Index::FileStorage
18
+ @storage_dir = File.join(Dir.home, ".relaton", "sdo")
19
+ @ttl = 24 * 60 * 60 # seconds
20
+ end
21
+ end
22
+
23
+ class << self
24
+ def configuration
25
+ @configuration ||= Config.new
26
+ end
27
+
28
+ def configure
29
+ yield configuration if block_given?
30
+ configuration
31
+ end
32
+ end
33
+ end
34
+ end
@@ -0,0 +1,52 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Relaton
4
+ module Sdo
5
+ # Download helper. The index manifest is text-cached under the configured
6
+ # storage dir with a TTL (reusing Relaton::Index::FileStorage); logo binaries
7
+ # are fetched on demand and left to in-process memoization on the Logo.
8
+ module Fetcher
9
+ module_function
10
+
11
+ # Fetch the index manifest, serving a fresh on-disk copy when present and
12
+ # falling back to a stale cache if the network fetch fails.
13
+ def fetch_index
14
+ config = Sdo.configuration
15
+ cache = File.join(config.storage_dir, "index.yaml")
16
+
17
+ if fresh?(cache, config)
18
+ cached = config.storage.read(cache)
19
+ return cached if cached
20
+ end
21
+
22
+ data = download(config.url)
23
+ return config.storage.read(cache) if data.nil? # stale cache or nil
24
+
25
+ config.storage.write(cache, data)
26
+ data
27
+ end
28
+
29
+ # GET a URL and return the body, or nil on failure. A URL with no scheme
30
+ # (or a file:// URL) is read straight off disk — handy for local fixtures.
31
+ def download(url)
32
+ uri = url.is_a?(URI) ? url : URI.parse(url.to_s)
33
+ return read_file(uri, url) if uri.scheme.nil? || uri.scheme == "file"
34
+
35
+ response = Net::HTTP.get_response(uri)
36
+ response.is_a?(Net::HTTPSuccess) ? response.body : nil
37
+ rescue StandardError
38
+ nil
39
+ end
40
+
41
+ def fresh?(file, config)
42
+ ctime = config.storage.ctime(file)
43
+ ctime && (Time.now - ctime) < config.ttl
44
+ end
45
+
46
+ def read_file(uri, url)
47
+ path = uri.path.to_s.empty? ? url.to_s : uri.path
48
+ File.exist?(path) ? File.binread(path) : nil
49
+ end
50
+ end
51
+ end
52
+ end
@@ -0,0 +1,95 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Relaton
4
+ module Sdo
5
+ # One logo variant for an organization. Variants are differentiated by
6
+ # `style` (the primary discriminator — e.g. "default", "red", "iso_1972"),
7
+ # `format` (file suffix) and `size`. `applicability` is opaque, informational
8
+ # text (e.g. "stage>=60"); the store never interprets it — consumers do.
9
+ class Logo
10
+ MIME_TYPES = {
11
+ "png" => "image/png",
12
+ "jpg" => "image/jpeg",
13
+ "jpeg" => "image/jpeg",
14
+ "gif" => "image/gif",
15
+ "svg" => "image/svg+xml",
16
+ "webp" => "image/webp",
17
+ "eps" => "application/postscript",
18
+ "pdf" => "application/pdf",
19
+ }.freeze
20
+
21
+ attr_reader :style, :format, :size, :url, :applicability
22
+
23
+ def initialize(style: nil, format: nil, size: nil, url: nil, applicability: nil)
24
+ @style = style
25
+ @format = format
26
+ @size = size
27
+ @url = url
28
+ @applicability = applicability
29
+ end
30
+
31
+ def self.from_hash(hash)
32
+ new(style: hash["style"], format: hash["format"], size: hash["size"],
33
+ url: hash["url"], applicability: hash["applicability"])
34
+ end
35
+
36
+ # Raw binary content of the logo, fetched lazily from `url` and memoized.
37
+ # Raises Sdo::Error (naming the URL) if the fetch fails, rather than
38
+ # returning nil and crashing cryptically inside data_uri/save downstream.
39
+ def content
40
+ @content ||= begin
41
+ data = Fetcher.download(url)
42
+ raise Error, "could not fetch logo from #{url.inspect}" if data.nil?
43
+
44
+ data
45
+ end
46
+ end
47
+
48
+ # The logo as a base64-encoded data URI, ready to embed in XML/HTML.
49
+ def data_uri
50
+ "data:#{mimetype};base64,#{Base64.strict_encode64(content)}"
51
+ end
52
+
53
+ def mimetype
54
+ MIME_TYPES[format.to_s.downcase] || "application/octet-stream"
55
+ end
56
+
57
+ # Write the logo to disk. With no argument, uses a filename derived from
58
+ # the source URL (or from style/size/format); returns the path written.
59
+ def save(path = nil)
60
+ path ||= default_filename
61
+ File.binwrite(path, content)
62
+ path
63
+ end
64
+
65
+ def to_h
66
+ { style: style, format: format, size: size, url: url,
67
+ applicability: applicability }.compact
68
+ end
69
+
70
+ def describe
71
+ to_h.map { |k, v| "#{k}=#{v}" }.join(", ")
72
+ end
73
+
74
+ private
75
+
76
+ def default_filename
77
+ base = File.basename(url_path)
78
+ return base unless base.empty? || File.extname(base).empty?
79
+
80
+ stem = [style, size].reject { |s| s.to_s.empty? }.join("-")
81
+ stem = "logo" if stem.empty?
82
+ ext = format.to_s.empty? ? "bin" : format
83
+ "#{stem}.#{ext}"
84
+ end
85
+
86
+ # The URL's path component, tolerating URLs that URI can't parse (e.g. a
87
+ # stray space) by falling back to the raw string sans query/fragment.
88
+ def url_path
89
+ URI.parse(url.to_s).path
90
+ rescue URI::InvalidURIError
91
+ url.to_s.split(/[?#]/, 2).first
92
+ end
93
+ end
94
+ end
95
+ end
@@ -0,0 +1,26 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Relaton
4
+ module Sdo
5
+ # A single organization name, optionally tagged with a language. A name with
6
+ # no language is the default (untranslated) name.
7
+ class Name
8
+ attr_reader :content, :language
9
+
10
+ def initialize(content:, language: nil)
11
+ @content = content
12
+ @language = language
13
+ end
14
+
15
+ def self.from_hash(hash)
16
+ return new(content: hash) unless hash.is_a?(Hash)
17
+
18
+ new(content: hash["content"], language: hash["language"])
19
+ end
20
+
21
+ def default?
22
+ language.nil? || language.to_s.empty?
23
+ end
24
+ end
25
+ end
26
+ end
@@ -0,0 +1,71 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Relaton
4
+ module Sdo
5
+ # A standards-development organization: its names (with translations) and its
6
+ # logo variants. Built from one entry of the index manifest.
7
+ class Organization
8
+ attr_reader :abbreviation, :names, :logos
9
+
10
+ def initialize(abbreviation:, names: [], logos: [])
11
+ @abbreviation = abbreviation
12
+ @names = names
13
+ @logos = logos
14
+ end
15
+
16
+ def self.from_hash(abbreviation, hash)
17
+ hash ||= {}
18
+ new(
19
+ abbreviation: abbreviation,
20
+ names: Array(hash["name"]).map { |n| Name.from_hash(n) },
21
+ logos: Array(hash["logo"]).map { |l| Logo.from_hash(l) },
22
+ )
23
+ end
24
+
25
+ # The organization's name. With no argument (or a nil language) returns the
26
+ # default, untranslated name; otherwise the translation for the requested
27
+ # language — passed positionally (`name("fr")`) or as the `language:`
28
+ # keyword (`name(language: "fr")`) — or nil if there is none. The keyword
29
+ # wins if both are given.
30
+ def name(lang = nil, language: nil)
31
+ language ||= lang
32
+ selected =
33
+ if language.nil?
34
+ names.find(&:default?) || names.first
35
+ else
36
+ names.find { |n| n.language == language }
37
+ end
38
+ selected&.content
39
+ end
40
+
41
+ # All logo variants matching the given filters. Every argument is optional;
42
+ # omitting one matches any value for it. Returns an array (possibly empty).
43
+ def logo_query(format: nil, size: nil, style: nil)
44
+ logos.select do |logo|
45
+ match?(logo.format, format) &&
46
+ match?(logo.size, size) &&
47
+ match?(logo.style, style)
48
+ end
49
+ end
50
+
51
+ # A single logo matching the filters. Returns nil when nothing matches; the
52
+ # sole match when exactly one does; raises when the filter is ambiguous so
53
+ # the caller narrows it by format/size/style.
54
+ def logo(format: nil, size: nil, style: nil)
55
+ matches = logo_query(format: format, size: size, style: style)
56
+ return matches.first if matches.size <= 1
57
+
58
+ raise Error, "#{matches.size} logos match #{abbreviation}; narrow by " \
59
+ "format/size/style: #{matches.map(&:describe).join('; ')}"
60
+ end
61
+
62
+ private
63
+
64
+ def match?(value, query)
65
+ return true if query.nil?
66
+
67
+ !value.nil? && value.to_s.casecmp?(query.to_s)
68
+ end
69
+ end
70
+ end
71
+ end
@@ -0,0 +1,49 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Relaton
4
+ module Sdo
5
+ # Singleton registry of organizations, built lazily from the index manifest
6
+ # and memoized for the process. `reset!` drops the memo (mainly for specs).
7
+ class Store
8
+ include Singleton
9
+
10
+ def initialize
11
+ @mutex = Mutex.new
12
+ end
13
+
14
+ # Look up an organization by abbreviation (case-insensitive). Returns an
15
+ # Organization or nil.
16
+ def organization(abbreviation)
17
+ index[normalize(abbreviation)]
18
+ end
19
+
20
+ # Lazily built and memoized; the mutex keeps a concurrent first access from
21
+ # downloading/parsing the index twice.
22
+ def index
23
+ @index || @mutex.synchronize { @index ||= build_index }
24
+ end
25
+
26
+ def reset!
27
+ @mutex.synchronize { @index = nil }
28
+ end
29
+
30
+ private
31
+
32
+ def normalize(abbreviation)
33
+ abbreviation.to_s.strip.upcase
34
+ end
35
+
36
+ def build_index
37
+ yaml = Fetcher.fetch_index
38
+ data = yaml.nil? || yaml.empty? ? {} : YAML.safe_load(yaml)
39
+ # A corrupt/unexpected index (non-mapping YAML) degrades to "no orgs"
40
+ # rather than crashing every lookup with a TypeError.
41
+ organizations = data.is_a?(Hash) ? (data["organizations"] || {}) : {}
42
+ organizations.each_with_object({}) do |(abbreviation, attrs), acc|
43
+ acc[normalize(abbreviation)] =
44
+ Organization.from_hash(abbreviation, attrs)
45
+ end
46
+ end
47
+ end
48
+ end
49
+ end
@@ -0,0 +1,29 @@
1
+ # frozen_string_literal: true
2
+
3
+ # Relaton::Sdo — the standards-body organization & logo store (metanorma#346,
4
+ # relaton-db#132). A NON-flavor, lazily-autoloaded top-level component: it has
5
+ # no processor and no Db::Registry entry. It fetches a small `index.yaml`
6
+ # manifest published by the `relaton-data-sdo` data repo, caches it under
7
+ # `~/.relaton/sdo/`, and exposes org names/translations and logo variants
8
+ # (differentiated by style/format/size) via `Relaton.organization`.
9
+
10
+ require "net/http"
11
+ require "uri"
12
+ require "yaml"
13
+ require "base64"
14
+ require "fileutils"
15
+ require "singleton"
16
+
17
+ require_relative "index/file_storage"
18
+ require_relative "sdo/config"
19
+ require_relative "sdo/name"
20
+ require_relative "sdo/logo"
21
+ require_relative "sdo/organization"
22
+ require_relative "sdo/fetcher"
23
+ require_relative "sdo/store"
24
+
25
+ module Relaton
26
+ module Sdo
27
+ class Error < StandardError; end
28
+ end
29
+ end
@@ -1,3 +1,3 @@
1
1
  module Relaton
2
- VERSION = "3.0.0.pre.alpha.1".freeze
2
+ VERSION = "3.0.0.pre.alpha.2".freeze
3
3
  end
@@ -2,23 +2,26 @@
2
2
 
3
3
  require "net/http"
4
4
  require "relaton/bib/hash_parser_v1"
5
- require_relative "pubid"
6
5
 
7
6
  module Relaton
8
7
  module W3c
9
8
  # Class methods for search W3C standards.
10
9
  class Bibliography
11
10
  SOURCE = "https://raw.githubusercontent.com/relaton/relaton-data-w3c/v2/"
11
+ # The Pages site of the data repo: it serves the machine index (manifest
12
+ # and shards) that `#index` reads. The documents still come from SOURCE.
13
+ PAGES_URL = "https://relaton.github.io/relaton-data-w3c/"
12
14
 
13
15
  class << self
14
- # @param text [String]
16
+ # @param ref [String, Pubid::W3c::Identifier] a reference, or one
17
+ # already parsed (relaton#189: `Relaton::Db` parses it once)
15
18
  # @return [Relaton::W3c::ItemData]
16
- def search(text) # rubocop:disable Metrics/MethodLength, Metrics/AbcSize
17
- pubid = PubId.parse text.sub(/^W3C\s/, "")
18
- index = Relaton::Index.find_or_create(
19
- :W3C, url: "#{SOURCE}#{INDEXFILE}.zip", file: "#{INDEXFILE}.yaml", id_keys: PubId::PARTS,
20
- )
21
- row = index.search { |r| pubid == r[:id] }.sort_by { |r| (r[:id][:date] || r[:id][:year]).to_i }.last
19
+ def search(ref)
20
+ # A reference pubid rejects raises `Pubid::Errors::ParseError`, which
21
+ # the transport rescue below does not catch.
22
+ pubid = ref.is_a?(::Pubid::W3c::Identifier) ? ref : parse_ref(ref)
23
+
24
+ row = best_match pubid
22
25
  return unless row
23
26
 
24
27
  url = "#{SOURCE}#{row[:file]}"
@@ -32,20 +35,137 @@ module Relaton
32
35
  raise Relaton::RequestError, "Could not access #{url}: #{e.message}"
33
36
  end
34
37
 
35
- # @param ref [String] the W3C standard Code to look up
38
+ #
39
+ # Find the index row for a reference, newest edition first.
40
+ #
41
+ # Passing the pubid to `Index::Type#search` is what enables the binary
42
+ # search on `id.root.number` — with a block alone the whole index is
43
+ # scanned. Selection is `Type#search`'s default with no block — pubid's
44
+ # asymmetric subset match (`Pubid::SubsetMatch`): the reference on the
45
+ # left, the row on the right, and a component it omits matches any
46
+ # value. `date`
47
+ # is W3C's only optional component, so an undated `REC-xml-names` still
48
+ # finds the dated row, while a dated reference reaches only its own.
49
+ # The maturity level is deliberately NOT a wildcard: it is the
50
+ # identifier's class, and `===` requires the same class, so `WD-`,
51
+ # `REC-` and a bare slug never match each other — the same contract the
52
+ # bespoke `PubId#==` had for its `stage`/`type`.
53
+ #
54
+ # @param pubid [Pubid::W3c::Identifier]
55
+ # @return [Hash, nil]
56
+ #
57
+ def best_match(pubid)
58
+ rows = index.search(pubid)
59
+ rows = index.search { |r| loose_match? r[:id], pubid } if rows.empty?
60
+
61
+ # Newest edition wins. Undated rows all score 0, so the file path
62
+ # breaks the tie and a repeated lookup returns the same document
63
+ # (the index sort is not stable).
64
+ rows.max_by { |r| [date_key(r[:id].date), r[:file]] }
65
+ end
66
+
67
+ #
68
+ # Order key for a W3C publication date.
69
+ #
70
+ # The dates are opaque digit runs of varying width, so a plain `to_i`
71
+ # does not order them: a legacy 6-digit `YYMMDD` always loses to an
72
+ # 8-digit `YYYYMMDD`, however much later it is (`980619` is June 1998,
73
+ # `19980512` is May). Restoring the century fixes that — all 63
74
+ # 6-digit dates in the corpus are 1990s.
75
+ #
76
+ # Everything else keeps `to_i`, deliberately. The 21 legacy 4-digit
77
+ # `MMDD` dates carry no year and cannot be ordered against a real one
78
+ # at all; as small integers they land below every dated row and above
79
+ # an undated one, which is where `to_i` already put them.
80
+ #
81
+ # @param date [String, nil]
82
+ # @return [Integer]
83
+ #
84
+ def date_key(date)
85
+ str = date.to_s
86
+ str.length == 6 ? "19#{str}".to_i : str.to_i
87
+ end
88
+
89
+ #
90
+ # The narrowed range cannot serve a reference whose slug differs from
91
+ # the row's only by case: the bsearch key is case-sensitive. The
92
+ # bespoke `PubId#==` compared its `code` with `casecmp?`, so a full
93
+ # scan repeats the match case-insensitively rather than lose that.
94
+ # (The BIPM `search_index` precedent.)
95
+ #
96
+ # @param row_id [Pubid::W3c::Identifier]
97
+ # @param pubid [Pubid::W3c::Identifier]
98
+ # @return [Boolean]
99
+ #
100
+ def loose_match?(row_id, pubid)
101
+ row_id.instance_of?(pubid.class) &&
102
+ row_id.number.to_s.casecmp?(pubid.number.to_s) &&
103
+ (pubid.date.nil? || row_id.date == pubid.date)
104
+ end
105
+
106
+ #
107
+ # The machine index on the Pages site (relaton#189, W3C is the pilot).
108
+ # A parsed query reads only its own shard; the whole index is read, in
109
+ # memory, only by the case-insensitive fallback in `#best_match`.
110
+ # A Pages failure raises `Relaton::RequestError`, with no fallback to
111
+ # the `index-v2.zip` in the data repo.
112
+ #
113
+ def index
114
+ Relaton::Index.find_or_create(
115
+ :W3C, pages_url: PAGES_URL, pubid_class: ::Pubid::W3c::Identifier
116
+ )
117
+ end
118
+
119
+ #
120
+ # Parse a user reference into a `Pubid::W3c::Identifier`, or nil.
121
+ #
122
+ # A search string is a query, not a document identifier field, so it
123
+ # does not go through `Docidentifier`: it has to absorb two forms the
124
+ # bespoke regex accepted and a pubid grammar should not. A URL is not
125
+ # an identifier (`https://www.w3.org/TR/xml-names/`), and `TR` is a
126
+ # path segment of that URL rather than a maturity level, so
127
+ # `TR-vocab-adms` means the document `vocab-adms`. The publisher
128
+ # prefix is added when absent, because `Pubid::W3c` requires it.
129
+ #
130
+ # @param text [String]
131
+ # @return [Pubid::W3c::Identifier, nil]
132
+ #
133
+ # An unrecognized reference **raises**; like ISO, ETSI and 3GPP we let it
134
+ # propagate. relaton-cli rescues `Pubid::Errors::Error` and renders
135
+ # `"..." is not a recognized standards identifier`
136
+ # (`gems/relaton-cli/lib/relaton/cli/command.rb:324`), and `Db#fetch`
137
+ # logs it through the `StandardError` arm at `lib/relaton/db.rb:122`.
138
+ # Rescuing here would collapse "this identifier is malformed" into "no
139
+ # such document", leaving a caller unable to tell them apart.
140
+ def parse_ref(text)
141
+ ::Pubid::W3c::Identifier.parse normalize_ref(text)
142
+ end
143
+
144
+ def normalize_ref(text)
145
+ ref = text.to_s.strip
146
+ .sub(%r{\Ahttps?://[^/]+/}i, "") # a URL is not an identifier
147
+ .sub(/\AW3C\s+/i, "")
148
+ .sub(%r{\ATR[/-]}i, "") # URL path segment, not a stage
149
+ .sub(%r{/\z}, "")
150
+ "W3C #{ref}"
151
+ end
152
+
153
+ # @param ref [String, Pubid::W3c::Identifier] the W3C standard Code
154
+ # to look up, or its parsed identifier
36
155
  # @param year [String, NilClass] not used
37
156
  # @param opts [Hash] options
38
157
  # @return [Relaton::W3c::ItemData]
39
158
  def get(ref, _year = nil, _opts = {})
40
- Util.info "Fetching from Relaton repository ...", key: ref
159
+ key = ref.to_s
160
+ Util.info "Fetching from Relaton repository ...", key: key
41
161
  result = search(ref)
42
162
  unless result
43
- Util.info "Not found.", key: ref
163
+ Util.info "Not found.", key: key
44
164
  return
45
165
  end
46
166
 
47
167
  found = result.docidentifier.first.content
48
- Util.info "Found: `#{found}`", key: ref
168
+ Util.info "Found: `#{found}`", key: key
49
169
  result
50
170
  end
51
171
  end