relaton 3.0.0.pre.alpha.1 → 3.0.0.pre.alpha.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (270) hide show
  1. checksums.yaml +4 -4
  2. data/README.adoc +57 -1
  3. data/bin/console +0 -1
  4. data/lib/relaton/3gpp/bibliography.rb +82 -7
  5. data/lib/relaton/3gpp/data_fetcher.rb +51 -3
  6. data/lib/relaton/3gpp/docidentifier.rb +114 -0
  7. data/lib/relaton/3gpp/item.rb +6 -0
  8. data/lib/relaton/3gpp/parser.rb +1 -1
  9. data/lib/relaton/3gpp/processor.rb +4 -1
  10. data/lib/relaton/3gpp.rb +5 -1
  11. data/lib/relaton/adobe/bibdata.rb +8 -0
  12. data/lib/relaton/adobe/bibitem.rb +8 -0
  13. data/lib/relaton/adobe/bibliography.rb +92 -0
  14. data/lib/relaton/adobe/docidentifier.rb +49 -0
  15. data/lib/relaton/adobe/doctype.rb +14 -0
  16. data/lib/relaton/adobe/ext.rb +32 -0
  17. data/lib/relaton/adobe/item.rb +15 -0
  18. data/lib/relaton/adobe/item_base.rb +18 -0
  19. data/lib/relaton/adobe/item_data.rb +6 -0
  20. data/lib/relaton/adobe/processor.rb +45 -0
  21. data/lib/relaton/adobe/util.rb +8 -0
  22. data/lib/relaton/adobe.rb +37 -0
  23. data/lib/relaton/bib/model/address.rb +2 -2
  24. data/lib/relaton/bib/model/docidentifier.rb +24 -9
  25. data/lib/relaton/bib/model/localized_string.rb +1 -1
  26. data/lib/relaton/bib/model/structured_identifier.rb +10 -9
  27. data/lib/relaton/bib/sanitizer.rb +202 -6
  28. data/lib/relaton/bib.rb +0 -2
  29. data/lib/relaton/bipm/bibliography.rb +159 -10
  30. data/lib/relaton/bipm/data_fetcher.rb +26 -2
  31. data/lib/relaton/bipm/data_outcomes_parser.rb +3 -4
  32. data/lib/relaton/bipm/id_parser.rb +5 -4
  33. data/lib/relaton/bipm/model/structured_identifier.rb +21 -0
  34. data/lib/relaton/bipm/processor.rb +2 -2
  35. data/lib/relaton/bipm/rawdata_bipm_metrologia/fetcher.rb +2 -4
  36. data/lib/relaton/bipm/si_brochure_parser.rb +5 -3
  37. data/lib/relaton/bipm.rb +7 -1
  38. data/lib/relaton/bsi/bibliography.rb +115 -43
  39. data/lib/relaton/bsi/hit.rb +14 -0
  40. data/lib/relaton/bsi/hit_collection.rb +15 -16
  41. data/lib/relaton/bsi/model/docidentifier.rb +99 -1
  42. data/lib/relaton/bsi/processor.rb +1 -0
  43. data/lib/relaton/calconnect/bibliography.rb +12 -14
  44. data/lib/relaton/calconnect/data_fetcher.rb +77 -9
  45. data/lib/relaton/calconnect/docidentifier.rb +80 -0
  46. data/lib/relaton/calconnect/hit_collection.rb +65 -57
  47. data/lib/relaton/calconnect/model/item.rb +7 -0
  48. data/lib/relaton/calconnect/processor.rb +7 -1
  49. data/lib/relaton/calconnect.rb +11 -1
  50. data/lib/relaton/ccsds/data/fetcher.rb +17 -12
  51. data/lib/relaton/ccsds/data/parser.rb +1 -1
  52. data/lib/relaton/ccsds/hit_collection.rb +6 -1
  53. data/lib/relaton/ccsds/model/docidentifier.rb +121 -0
  54. data/lib/relaton/ccsds/model/item.rb +2 -0
  55. data/lib/relaton/cen/bibliography.rb +75 -47
  56. data/lib/relaton/cen/hit.rb +16 -1
  57. data/lib/relaton/cen/hit_collection.rb +59 -11
  58. data/lib/relaton/cen/model/docidentifier.rb +92 -1
  59. data/lib/relaton/cen/processor.rb +13 -8
  60. data/lib/relaton/cen/scraper.rb +13 -5
  61. data/lib/relaton/cen.rb +1 -0
  62. data/lib/relaton/cie/data_fetcher.rb +215 -30
  63. data/lib/relaton/cie/processor.rb +3 -1
  64. data/lib/relaton/cie/scrapper.rb +15 -2
  65. data/lib/relaton/cie.rb +2 -1
  66. data/lib/relaton/core/data_fetcher.rb +150 -3
  67. data/lib/relaton/core/governor.rb +320 -0
  68. data/lib/relaton/core/pacer.rb +134 -0
  69. data/lib/relaton/core/processor.rb +19 -0
  70. data/lib/relaton/core/request_error.rb +14 -0
  71. data/lib/relaton/core.rb +3 -0
  72. data/lib/relaton/db/registry.rb +41 -1
  73. data/lib/relaton/doi/crossref.rb +19 -2
  74. data/lib/relaton/doi/parser.rb +109 -15
  75. data/lib/relaton/easc/bibdata.rb +8 -0
  76. data/lib/relaton/easc/bibitem.rb +8 -0
  77. data/lib/relaton/easc/bibliography.rb +95 -0
  78. data/lib/relaton/easc/docidentifier.rb +100 -0
  79. data/lib/relaton/easc/doctype.rb +14 -0
  80. data/lib/relaton/easc/ext.rb +44 -0
  81. data/lib/relaton/easc/item.rb +13 -0
  82. data/lib/relaton/easc/item_base.rb +18 -0
  83. data/lib/relaton/easc/item_data.rb +6 -0
  84. data/lib/relaton/easc/processor.rb +46 -0
  85. data/lib/relaton/easc/util.rb +8 -0
  86. data/lib/relaton/easc.rb +35 -0
  87. data/lib/relaton/ecma/bibliography.rb +93 -25
  88. data/lib/relaton/ecma/data_fetcher.rb +71 -12
  89. data/lib/relaton/ecma/docidentifier.rb +124 -0
  90. data/lib/relaton/ecma/item.rb +2 -0
  91. data/lib/relaton/ecma/memento_parser.rb +1 -1
  92. data/lib/relaton/ecma/page_fetcher.rb +15 -3
  93. data/lib/relaton/ecma/parser_common.rb +2 -2
  94. data/lib/relaton/ecma/processor.rb +4 -1
  95. data/lib/relaton/ecma/standard_parser.rb +2 -2
  96. data/lib/relaton/ecma.rb +10 -1
  97. data/lib/relaton/etsi/bibliography.rb +67 -2
  98. data/lib/relaton/etsi/data_fetcher.rb +43 -4
  99. data/lib/relaton/etsi/processor.rb +3 -1
  100. data/lib/relaton/etsi.rb +2 -1
  101. data/lib/relaton/gb/bibliography.rb +55 -29
  102. data/lib/relaton/gb/docidentifier.rb +58 -9
  103. data/lib/relaton/gb/processor.rb +3 -0
  104. data/lib/relaton/gb/scraper.rb +27 -10
  105. data/lib/relaton/gost/bibdata.rb +8 -0
  106. data/lib/relaton/gost/bibitem.rb +8 -0
  107. data/lib/relaton/gost/bibliography.rb +107 -0
  108. data/lib/relaton/gost/docidentifier.rb +80 -0
  109. data/lib/relaton/gost/doctype.rb +16 -0
  110. data/lib/relaton/gost/ext.rb +46 -0
  111. data/lib/relaton/gost/item.rb +15 -0
  112. data/lib/relaton/gost/item_base.rb +18 -0
  113. data/lib/relaton/gost/item_data.rb +6 -0
  114. data/lib/relaton/gost/processor.rb +49 -0
  115. data/lib/relaton/gost/util.rb +8 -0
  116. data/lib/relaton/gost.rb +36 -0
  117. data/lib/relaton/iala/bibdata.rb +8 -0
  118. data/lib/relaton/iala/bibitem.rb +8 -0
  119. data/lib/relaton/iala/bibliography.rb +146 -0
  120. data/lib/relaton/iala/docidentifier.rb +89 -0
  121. data/lib/relaton/iala/doctype.rb +18 -0
  122. data/lib/relaton/iala/ext.rb +32 -0
  123. data/lib/relaton/iala/item.rb +21 -0
  124. data/lib/relaton/iala/item_base.rb +18 -0
  125. data/lib/relaton/iala/item_data.rb +6 -0
  126. data/lib/relaton/iala/processor.rb +43 -0
  127. data/lib/relaton/iala/relation.rb +7 -0
  128. data/lib/relaton/iala/util.rb +8 -0
  129. data/lib/relaton/iala.rb +35 -0
  130. data/lib/relaton/iana/bibliography.rb +67 -14
  131. data/lib/relaton/iana/data_fetcher.rb +35 -5
  132. data/lib/relaton/iana/processor.rb +3 -1
  133. data/lib/relaton/iana.rb +12 -1
  134. data/lib/relaton/iec/data_fetcher.rb +7 -1
  135. data/lib/relaton/iec/hit_collection.rb +1 -1
  136. data/lib/relaton/iec/model/docidentifier.rb +9 -5
  137. data/lib/relaton/iec/model/ext.rb +2 -2
  138. data/lib/relaton/iec/processor.rb +1 -0
  139. data/lib/relaton/ieee/bibliography.rb +25 -3
  140. data/lib/relaton/ieee/data_fetcher.rb +158 -17
  141. data/lib/relaton/ieee/idams_parser.rb +18 -11
  142. data/lib/relaton/ieee/processor.rb +4 -1
  143. data/lib/relaton/ieee/rawbib_id_parser.rb +291 -86
  144. data/lib/relaton/ieee.rb +2 -1
  145. data/lib/relaton/ietf/data_fetcher.rb +295 -12
  146. data/lib/relaton/ietf/processor.rb +7 -3
  147. data/lib/relaton/ietf/rfc/entry.rb +39 -3
  148. data/lib/relaton/ietf/scraper.rb +69 -36
  149. data/lib/relaton/ietf.rb +4 -1
  150. data/lib/relaton/iho/bibliography.rb +1 -1
  151. data/lib/relaton/iho/docidentifier.rb +1 -1
  152. data/lib/relaton/index/file_io.rb +11 -11
  153. data/lib/relaton/index/file_storage.rb +6 -1
  154. data/lib/relaton/index/pool.rb +6 -1
  155. data/lib/relaton/index/shard_source.rb +201 -0
  156. data/lib/relaton/index/type.rb +63 -12
  157. data/lib/relaton/index.rb +2 -1
  158. data/lib/relaton/iso/bibliography.rb +20 -15
  159. data/lib/relaton/iso/data_fetcher.rb +3 -3
  160. data/lib/relaton/iso/data_parser.rb +17 -3
  161. data/lib/relaton/iso/hit_collection.rb +27 -15
  162. data/lib/relaton/iso/item_data.rb +22 -0
  163. data/lib/relaton/iso/model/docidentifier.rb +24 -12
  164. data/lib/relaton/iso/processor.rb +1 -0
  165. data/lib/relaton/iso/scraper.rb +19 -3
  166. data/lib/relaton/itu/bibliography.rb +9 -4
  167. data/lib/relaton/itu/data_crawler_r.rb +664 -0
  168. data/lib/relaton/itu/data_fetcher.rb +496 -50
  169. data/lib/relaton/itu/data_merge_r.rb +149 -0
  170. data/lib/relaton/itu/data_parser_r.rb +163 -89
  171. data/lib/relaton/itu/data_parser_t.rb +228 -0
  172. data/lib/relaton/itu/family_cache.rb +177 -0
  173. data/lib/relaton/itu/governor.rb +56 -0
  174. data/lib/relaton/itu/hit.rb +9 -3
  175. data/lib/relaton/itu/hit_collection.rb +258 -86
  176. data/lib/relaton/itu/model/docidentifier.rb +67 -1
  177. data/lib/relaton/itu/model/structured_identifier.rb +19 -0
  178. data/lib/relaton/itu/processor.rb +10 -4
  179. data/lib/relaton/itu/pubid.rb +27 -5
  180. data/lib/relaton/itu/recommendation_fields.rb +334 -0
  181. data/lib/relaton/itu/recommendation_parser.rb +18 -149
  182. data/lib/relaton/itu/scraper.rb +13 -3
  183. data/lib/relaton/itu.rb +2 -1
  184. data/lib/relaton/jcgm/bibdata.rb +8 -0
  185. data/lib/relaton/jcgm/bibitem.rb +8 -0
  186. data/lib/relaton/jcgm/bibliography.rb +97 -0
  187. data/lib/relaton/jcgm/data_fetcher.rb +81 -0
  188. data/lib/relaton/jcgm/docidentifier.rb +102 -0
  189. data/lib/relaton/jcgm/doctype.rb +12 -0
  190. data/lib/relaton/jcgm/ext.rb +23 -0
  191. data/lib/relaton/jcgm/item.rb +20 -0
  192. data/lib/relaton/jcgm/item_base.rb +18 -0
  193. data/lib/relaton/jcgm/item_data.rb +6 -0
  194. data/lib/relaton/jcgm/meetings_parser.rb +175 -0
  195. data/lib/relaton/jcgm/processor.rb +71 -0
  196. data/lib/relaton/jcgm/relation.rb +9 -0
  197. data/lib/relaton/jcgm/structured_identifier.rb +40 -0
  198. data/lib/relaton/jcgm/util.rb +8 -0
  199. data/lib/relaton/jcgm.rb +24 -0
  200. data/lib/relaton/jis/bibliography.rb +8 -10
  201. data/lib/relaton/jis/data_fetcher.rb +21 -19
  202. data/lib/relaton/jis/docidentifier.rb +104 -5
  203. data/lib/relaton/jis/hit.rb +18 -23
  204. data/lib/relaton/jis/hit_collection.rb +19 -18
  205. data/lib/relaton/jis/processor.rb +1 -1
  206. data/lib/relaton/jis.rb +2 -3
  207. data/lib/relaton/logger/channels/gh_issue.rb +78 -13
  208. data/lib/relaton/nist/data_fetcher.rb +63 -13
  209. data/lib/relaton/nist/docidentifier.rb +165 -0
  210. data/lib/relaton/nist/item.rb +2 -0
  211. data/lib/relaton/nist/item_base.rb +16 -0
  212. data/lib/relaton/nist/mods_parser.rb +38 -12
  213. data/lib/relaton/nist/processor.rb +2 -1
  214. data/lib/relaton/nist/relation.rb +3 -0
  215. data/lib/relaton/nist/scraper.rb +6 -3
  216. data/lib/relaton/oasis/bibliography.rb +147 -6
  217. data/lib/relaton/oasis/data_fetcher.rb +41 -5
  218. data/lib/relaton/oasis/data_parser_utils.rb +37 -3
  219. data/lib/relaton/oasis/docidentifier.rb +54 -0
  220. data/lib/relaton/oasis/item.rb +3 -0
  221. data/lib/relaton/oasis/processor.rb +7 -1
  222. data/lib/relaton/oasis.rb +14 -1
  223. data/lib/relaton/ogc/data_fetcher.rb +23 -2
  224. data/lib/relaton/ogc/docidentifier.rb +105 -0
  225. data/lib/relaton/ogc/hit_collection.rb +78 -3
  226. data/lib/relaton/ogc/processor.rb +2 -1
  227. data/lib/relaton/ogc.rb +5 -1
  228. data/lib/relaton/oiml/bibliography.rb +90 -15
  229. data/lib/relaton/oiml/docidentifier.rb +18 -3
  230. data/lib/relaton/omg/docidentifier.rb +67 -0
  231. data/lib/relaton/omg/item.rb +1 -0
  232. data/lib/relaton/omg/processor.rb +1 -0
  233. data/lib/relaton/omg/scraper.rb +61 -16
  234. data/lib/relaton/omg.rb +1 -0
  235. data/lib/relaton/plateau/bibliography.rb +10 -3
  236. data/lib/relaton/plateau/data_fetcher.rb +25 -2
  237. data/lib/relaton/plateau/handbook_parser.rb +8 -1
  238. data/lib/relaton/plateau/hit.rb +10 -2
  239. data/lib/relaton/plateau/hit_collection.rb +31 -11
  240. data/lib/relaton/plateau/processor.rb +3 -1
  241. data/lib/relaton/plateau/technical_report_parser.rb +8 -1
  242. data/lib/relaton/plateau.rb +2 -1
  243. data/lib/relaton/sdo/config.rb +34 -0
  244. data/lib/relaton/sdo/fetcher.rb +52 -0
  245. data/lib/relaton/sdo/logo.rb +95 -0
  246. data/lib/relaton/sdo/name.rb +26 -0
  247. data/lib/relaton/sdo/organization.rb +71 -0
  248. data/lib/relaton/sdo/store.rb +49 -0
  249. data/lib/relaton/sdo.rb +29 -0
  250. data/lib/relaton/version.rb +1 -1
  251. data/lib/relaton/w3c/bibliography.rb +132 -12
  252. data/lib/relaton/w3c/data_fetcher.rb +194 -16
  253. data/lib/relaton/w3c/data_parser.rb +3 -3
  254. data/lib/relaton/w3c/docidentifier.rb +48 -0
  255. data/lib/relaton/w3c/governor.rb +32 -0
  256. data/lib/relaton/w3c/item.rb +3 -0
  257. data/lib/relaton/w3c/pubid.rb +12 -0
  258. data/lib/relaton/w3c/safe_realize.rb +110 -21
  259. data/lib/relaton/w3c.rb +12 -1
  260. data/lib/relaton/xsf/bibliography.rb +61 -1
  261. data/lib/relaton/xsf/data_fetcher.rb +55 -5
  262. data/lib/relaton/xsf/docidentifier.rb +46 -0
  263. data/lib/relaton/xsf/hit_collection.rb +31 -3
  264. data/lib/relaton/xsf/item.rb +6 -0
  265. data/lib/relaton/xsf/processor.rb +1 -0
  266. data/lib/relaton/xsf.rb +5 -1
  267. data/lib/relaton.rb +42 -0
  268. metadata +135 -24
  269. data/lib/relaton/ieee/pub_id.rb +0 -161
  270. data/lib/relaton/index/id_number.rb +0 -30
@@ -10,7 +10,7 @@ module Relaton
10
10
  def find
11
11
  return self if ref.nil? || ref.empty?
12
12
 
13
- row = index.search(ref).min_by { |r| r[:id] }
13
+ row = best_match ref
14
14
  return self unless row
15
15
 
16
16
  url = "#{ENDPOINT}#{row[:file]}"
@@ -25,12 +25,87 @@ module Relaton
25
25
  self
26
26
  end
27
27
 
28
- # @return [Relaton::Index]
28
+ # @return [Relaton::Index::Type]
29
29
  def index
30
30
  @index ||= Relaton::Index.find_or_create(
31
- :ogc, url: "#{ENDPOINT}index-v1.zip", file: "#{INDEXFILE}.yaml",
31
+ :ogc, url: "#{ENDPOINT}#{INDEXFILE}.zip", file: "#{INDEXFILE}.yaml",
32
+ pubid_class: ::Pubid::Ogc::Identifier
32
33
  )
33
34
  end
35
+
36
+ private
37
+
38
+ #
39
+ # Find the index row for a reference, latest revision first.
40
+ #
41
+ # Passing the pubid to `Index::Type#search` is what enables the binary
42
+ # search on `id.root.number` — with the plain string this used to pass,
43
+ # the whole index is scanned however the index was built. Selection is
44
+ # `Type#search`'s default with no block — pubid's asymmetric subset match
45
+ # (`Pubid::SubsetMatch`): the reference on the left, the row on the right,
46
+ # and a component it omits matches any
47
+ # value. `revision` is OGC's only optional component, so a
48
+ # bare `OGC 12-128` still finds the `r19` row, while `OGC 12-128r19`
49
+ # reaches only its own. `year` is always stated — the bsearch key is the
50
+ # `<nnn>` field alone, so one bucket holds every year that reused the
51
+ # number (`05-007` and `17-007` share bucket `007`).
52
+ #
53
+ # @param text [String]
54
+ # @return [Hash, nil]
55
+ #
56
+ # The substring-scan fallback this used to take when parsing failed is
57
+ # gone with the rescue. It was the same conflation the convention
58
+ # rejects: an unparseable string matched a SUBSTRING of the rendered
59
+ # ids, so an ambiguous reference silently resolved to whichever row
60
+ # sorted first, instead of telling the caller the reference is not an
61
+ # identifier. ecma, w3c and xsf never had one.
62
+ def best_match(text)
63
+ pubid = parse_ref text
64
+ rows = index.search(pubid)
65
+ rows.max_by { |r| [revision_key(r[:id].revision), r[:file]] }
66
+ end
67
+
68
+ #
69
+ # Order key for an OGC revision suffix.
70
+ #
71
+ # Revisions are `r<n>` with optional letter suffixes (`r3a`, `r12a`) plus
72
+ # the odd `a` and `c1`, so the number must be compared as an integer:
73
+ # `r2` beats `r14` as text but is the older document. The token itself
74
+ # breaks the tie, so `r3a` sorts above `r3` and a repeated lookup returns
75
+ # the same row (the index sort is not stable). An absent revision sorts
76
+ # below every present one.
77
+ #
78
+ # Highest revision is the newest published document in 105 of the 106
79
+ # multi-revision documents in the index, scored against each document's
80
+ # own `date[0].at`; the previous `min_by` on the rendered id matched 1.
81
+ #
82
+ # @param revision [String, nil]
83
+ # @return [Array]
84
+ #
85
+ def revision_key(revision)
86
+ [revision.to_s[/\d+/].to_i, revision.to_s]
87
+ end
88
+
89
+ #
90
+ # Parse a user reference into a `Pubid::Ogc::Identifier`, or nil.
91
+ #
92
+ # `Pubid::Ogc` takes the `OGC ` publisher token as optional and
93
+ # lowercases the revision, so `OGC 19-025r1`, `19-025r1`, `11-038R2` and
94
+ # the revision-less `16-079` all parse without normalization here.
95
+ #
96
+ # @param text [String]
97
+ # @return [Pubid::Ogc::Identifier, nil]
98
+ #
99
+ # An unrecognized reference **raises**; like ISO, ETSI and 3GPP we let it
100
+ # propagate. relaton-cli rescues `Pubid::Errors::Error` and renders
101
+ # `"..." is not a recognized standards identifier`
102
+ # (`gems/relaton-cli/lib/relaton/cli/command.rb:324`), and `Db#fetch`
103
+ # logs it through the `StandardError` arm at `lib/relaton/db.rb:122`.
104
+ # Rescuing here would collapse "this identifier is malformed" into "no
105
+ # such document", leaving a caller unable to tell them apart.
106
+ def parse_ref(text)
107
+ ::Pubid::Ogc::Identifier.parse text.to_s.strip
108
+ end
34
109
  end
35
110
  end
36
111
  end
@@ -9,6 +9,7 @@ module Relaton
9
9
  @defaultprefix = %r{^OGC\s}
10
10
  @idtype = "OGC"
11
11
  @datasets = %w[ogc-naming-authority]
12
+ @pubid_flavor = :Ogc
12
13
  end
13
14
 
14
15
  # @param code [String]
@@ -50,7 +51,7 @@ module Relaton
50
51
  def remove_index_file
51
52
  require_relative "../ogc"
52
53
  Relaton::Index.find_or_create(
53
- :ogc, url: true, file: "#{INDEXFILE}.yaml",
54
+ :ogc, url: true, file: "#{INDEXFILE}.yaml"
54
55
  ).remove_file
55
56
  end
56
57
  end
data/lib/relaton/ogc.rb CHANGED
@@ -1,3 +1,7 @@
1
+ # pubid loads with the flavor, for the ::Pubid::Ogc::Identifier that the
2
+ # index code names. Processor#remove_index_file names no pubid class: the
3
+ # delete never reads the index (see lib/relaton/index/CLAUDE.md).
4
+ require "pubid"
1
5
  require "relaton/index"
2
6
  require "relaton/iso"
3
7
  require_relative "version"
@@ -11,7 +15,7 @@ require_relative "ogc/bibliography"
11
15
 
12
16
  module Relaton
13
17
  module Ogc
14
- INDEXFILE = "index-v1".freeze
18
+ INDEXFILE = "index-v2".freeze
15
19
  class Error < StandardError; end
16
20
 
17
21
  # Returns hash of XML reammar
@@ -20,13 +20,7 @@ module Relaton
20
20
  def search(text, year = nil, _opts = {}) # rubocop:disable Metrics/MethodLength, Metrics/AbcSize
21
21
  pubid = text.is_a?(String) ? ::Pubid::Oiml.parse(text) : text
22
22
  Util.info "Fetching from Relaton repository ...", key: pubid.to_s
23
- # Pass the pubid so Relaton::Index narrows candidates by number via
24
- # binary search before applying the block. Every row's `:id` is a
25
- # Pubid::Oiml::Identifier (Relaton::Index deserialized it via the
26
- # `pubid_class` passed in `#index`), so the block compares pubids and
27
- # the result picks the latest edition.
28
- row = index.search(pubid) { |r| pubid_match?(r[:id], pubid, year) }
29
- .max_by { |r| r[:id].year.to_i }
23
+ row = best_row(pubid, year)
30
24
  unless row
31
25
  Util.info "Not found.", key: pubid.to_s
32
26
  return
@@ -48,9 +42,29 @@ module Relaton
48
42
  raise Relaton::RequestError, "Could not access #{uri}: #{e.message}"
49
43
  end
50
44
 
45
+ # Fetch an OIML publication, suppressing the edition year for an undated
46
+ # citation so the designation renders undated (`OIML B 18`), the way the
47
+ # ISO fetcher does. A dated citation (`OIML B 18:2022`, or an explicit
48
+ # `year`) still pins that edition. Mirrors `Relaton::Iso::Bibliography#get`.
49
+ #
50
+ # @param opts [Hash] options
51
+ # @option opts [Boolean] :keep_year retain the edition year even for an
52
+ # undated citation (or, when false, strip it even for a dated one)
53
+ #
51
54
  # @see #search
52
55
  def get(ref, year = nil, opts = {})
53
- search(ref, year, opts)
56
+ item = search(ref, year, opts)
57
+ return item unless item
58
+
59
+ pubid = ref.is_a?(String) ? ::Pubid::Oiml.parse(ref) : ref
60
+ dated = oiml_side(pubid).year || year
61
+ # Keep the year only when the citation genuinely asks for it: a resolved
62
+ # year (unless keep_year is explicitly false), or keep_year truthy. (ISO
63
+ # also keeps for :all_parts; OIML has no all-parts retrieval, so search
64
+ # ignores opts and there is nothing to mirror here.)
65
+ return item if (dated && opts[:keep_year].nil?) || opts[:keep_year]
66
+
67
+ item.to_most_recent_reference
54
68
  end
55
69
 
56
70
  private
@@ -64,6 +78,60 @@ module Relaton
64
78
  )
65
79
  end
66
80
 
81
+ # The index row for a reference: the latest edition among the rows it
82
+ # matches.
83
+ #
84
+ # The pubid is passed to `Index::Type#search`, so the index narrows
85
+ # candidates by number via binary search, and each row's `:id` is a
86
+ # Pubid::Oiml::Identifier (deserialized via the `pubid_class` in
87
+ # `#index`).
88
+ #
89
+ # A Bulletin matches exactly (`exact: true`). Its year is part of its
90
+ # locator, not an edition: the issue and the sequence render after the
91
+ # year, so the stem below reduces every article of one year to
92
+ # `OIML Bulletin`, and a lookup used to return any one of them.
93
+ #
94
+ # This flavor does not use pubid's subset match `===` yet. Measured
95
+ # over the full `relaton-data-oiml` index, `===` lets a language-less
96
+ # Amendment, Annex or Errata reference reach its translations
97
+ # (`language` is not strict), lets `OIML R 137-1 (F)` reach
98
+ # `OIML R 137-1-2:2012 (F)` (`subpart` is not strict), and rejects
99
+ # `OIML R 102:1995 Annex B-C` for `OIML R 102 Annex B-C` (the
100
+ # `year_on_base` render flag is compared).
101
+ #
102
+ # @param query [Pubid::Oiml::Identifier]
103
+ # @param year [String, nil]
104
+ # @return [Hash, nil] the index row (`{ id:, file: }`)
105
+ def best_row(query, year)
106
+ return index.search(query, exact: true).first if bulletin?(query)
107
+
108
+ index.search(query) { |r| pubid_match?(r[:id], query, year) }
109
+ .max_by { |r| oiml_side(r[:id]).year.to_i }
110
+ end
111
+
112
+ # @return [Boolean] true for an identifier of an OIML Bulletin
113
+ def bulletin?(pubid)
114
+ pubid.is_a?(::Pubid::Oiml::Identifiers::Bulletin)
115
+ end
116
+
117
+ # A document OIML co-publishes with another SDO (ISO confirmed so far)
118
+ # is a Pubid::Oiml::Identifiers::DualPublished — a "|"-joined pair
119
+ # (pubid#437), e.g. `ISO 4064-1:2024|OIML R 49-1:2024`. It delegates
120
+ # #root/#code/#type/#stage/#iteration/#publisher to whichever side is
121
+ # OIML, but NOT #year or #language — those read nil on the wrapper
122
+ # itself regardless of what either side holds. This unwraps to the
123
+ # OIML side for those reads (a plain, non-dual pubid is returned
124
+ # unchanged), so a query and a row agree on year/language/stem
125
+ # whether either one is dual-published or not, and regardless of
126
+ # which side print order named first.
127
+ #
128
+ # @param pubid [Pubid::Oiml::Identifier] a plain or dual-published pubid
129
+ # @return [Pubid::Oiml::Identifier] the OIML-flavored side (itself, for
130
+ # a plain pubid)
131
+ def oiml_side(pubid)
132
+ pubid.is_a?(::Pubid::Oiml::Identifiers::DualPublished) ? pubid.oiml_identifier : pubid
133
+ end
134
+
67
135
  # Both `row_id` and `query` are Pubid::Oiml::Identifier instances.
68
136
  # Matching is on the year/language-stripped "stem" (e.g. `OIML R 138`),
69
137
  # which keeps the type letter and any amendment suffix — so an amendment
@@ -71,13 +139,15 @@ module Relaton
71
139
  # does not expose the suffix as its own attribute. Language must match
72
140
  # exactly (a language-less query targets the language-less abstract
73
141
  # record); year is nil-tolerant so an unqualified query finds the latest
74
- # edition (selected by `max_by` in #search). The `year` argument lets a
75
- # caller pin an edition the reference string omitted.
142
+ # edition (selected by `max_by` in #best_row). The `year` argument lets
143
+ # a caller pin an edition the reference string omitted.
76
144
  def pubid_match?(row_id, query, year)
77
- wanted_year = (query.year || year)&.to_s
78
- stem(row_id) == stem(query) &&
79
- row_id.language.to_s == query.language.to_s &&
80
- (wanted_year.nil? || row_id.year.to_s == wanted_year)
145
+ row = oiml_side(row_id)
146
+ q = oiml_side(query)
147
+ wanted_year = (q.year || year)&.to_s
148
+ stem(row) == stem(q) &&
149
+ row.language.to_s == q.language.to_s &&
150
+ (wanted_year.nil? || row.year.to_s == wanted_year)
81
151
  end
82
152
 
83
153
  # The identifier without its edition year or language, e.g.
@@ -85,8 +155,13 @@ module Relaton
85
155
  # via #exclude (returns a copy, so the cached index id is untouched)
86
156
  # rather than string surgery on #to_s. The amendment suffix is kept, so
87
157
  # an amendment (`OIML R 138-Amend`) never reduces to the base record.
158
+ #
159
+ # Reduced through #oiml_side first: `#exclude` recurses correctly into
160
+ # both sides of a DualPublished (unlike a direct #year/#language read),
161
+ # but a plain query has no external side to reduce to, so comparing
162
+ # full dual-published stems against a plain one would never match.
88
163
  def stem(pubid)
89
- pubid.exclude(:year, :language).to_s
164
+ oiml_side(pubid).exclude(:year, :language).to_s
90
165
  end
91
166
  end
92
167
  end
@@ -17,16 +17,31 @@ module Relaton
17
17
  @pubid = nil
18
18
  end
19
19
 
20
+ # Both mutators go through #exclude (an immutable copy) rather than
21
+ # mutating `@pubid` in place: for a plain identifier the two are
22
+ # equivalent, but for a Pubid::Oiml::Identifiers::DualPublished
23
+ # (pubid#437, e.g. `ISO 4064-1:2024|OIML R 49-1:2024`) a direct
24
+ # `@pubid.part = nil`/`@pubid.date = nil` is a silent no-op — neither
25
+ # attribute is delegated to either side — while `#exclude` correctly
26
+ # recurses into both sides. `content=` re-parses the rendered string,
27
+ # which resyncs `@pubid` too, so there is nothing left to reassign here.
20
28
  def remove_part!
21
- @pubid&.part = nil
29
+ return unless @pubid
30
+
31
+ self.content = @pubid.exclude(:part).to_s
22
32
  end
23
33
 
24
34
  def remove_date!
25
- @pubid&.date = nil
35
+ return unless @pubid
36
+
37
+ # Re-sync `content` from the now-dateless pubid: it is the string that
38
+ # actually renders (e.g. `OIML R 138`, or `OIML R 138 (E)` keeping the
39
+ # language). Mutating `@pubid` alone leaves the dated `content` in place.
40
+ self.content = @pubid.exclude(:date).to_s
26
41
  end
27
42
 
28
43
  def to_all_parts!
29
- @pubid&.all_parts = true
44
+ @pubid &&= @pubid.to_all_parts
30
45
  end
31
46
  end
32
47
  end
@@ -0,0 +1,67 @@
1
+ module Relaton
2
+ module Omg
3
+ # An OMG document identifier backed by `Pubid::Omg`.
4
+ #
5
+ # `content` stays a plain string, so serialization is unchanged; the parsed
6
+ # identifier lives beside it in `@pubid` and drives the mutators. Follows
7
+ # the IALA and CEN shape (`lib/relaton/iala/docidentifier.rb`,
8
+ # `lib/relaton/cen/model/docidentifier.rb`).
9
+ class Docidentifier < Bib::Docidentifier
10
+ # @return [Pubid::Omg::Identifier, nil] nil when the content is not an
11
+ # OMG identifier, or the grammar cannot read it
12
+ attr_reader :pubid
13
+
14
+ # Capture the inherited (LocalizedMarkedUpString) content setter before
15
+ # overriding #content=, so #refresh_content! writes the re-rendered string
16
+ # back WITHOUT re-parsing it and discarding the mutation.
17
+ alias_method :store_content, :content=
18
+
19
+ def content=(value)
20
+ super
21
+ return unless value
22
+
23
+ @pubid = begin
24
+ # `pubid` is required lazily because deserialization reaches this
25
+ # class without the flavor entry file having been loaded. LoadError
26
+ # degrades to a plain string; StandardError covers a non-OMG value
27
+ # and a title the grammar rejects, both of which are DATA and must
28
+ # not raise. A malformed *query* raises — see Scraper.scrape_page.
29
+ require "pubid"
30
+ ::Pubid::Omg::Identifier.parse(value)
31
+ rescue LoadError, StandardError
32
+ nil
33
+ end
34
+ end
35
+
36
+ # OMG identifiers carry no date. The version is OMG's discriminator
37
+ # (`OMG AMI4CCM 1.0` and `OMG AMI4CCM 1.1` are two editions of one
38
+ # specification), so the version-agnostic ("most recent") reference drops
39
+ # the version — as IALA maps `remove_date!` onto its edition.
40
+ def remove_date!
41
+ return unless @pubid
42
+
43
+ replace_pubid @pubid.exclude(:version)
44
+ end
45
+
46
+ # The part is the volume or format segment after the version, e.g.
47
+ # `Superstructure` in `OMG UML 2.1.1 Superstructure`.
48
+ def remove_part!
49
+ return unless @pubid
50
+
51
+ replace_pubid @pubid.exclude(:part)
52
+ end
53
+
54
+ # `to_all_parts!` stays the inherited no-op. An OMG part is a volume or a
55
+ # format name, not a numbered part, so there is no "all parts" form.
56
+
57
+ private
58
+
59
+ # `Pubid#exclude` returns a COPY, so the new identifier replaces the old
60
+ # one.
61
+ def replace_pubid(new_pubid)
62
+ @pubid = new_pubid
63
+ store_content @pubid.to_s
64
+ end
65
+ end
66
+ end
67
+ end
@@ -2,6 +2,7 @@ module Relaton
2
2
  module Omg
3
3
  class Item < Bib::Item
4
4
  model ItemData
5
+ attribute :docidentifier, Docidentifier, collection: true, initialize_empty: true
5
6
  attribute :ext, Ext
6
7
  end
7
8
  end
@@ -6,6 +6,7 @@ module Relaton
6
6
  def initialize # rubocop:disable Lint/MissingSuper
7
7
  @short = :relaton_omg
8
8
  @prefix = "OMG"
9
+ @pubid_flavor = :Omg # Pubid::Omg.prefixes is ["OMG"], the same as @prefix
9
10
  @defaultprefix = /^OMG /
10
11
  @idtype = "OMG"
11
12
  end
@@ -1,23 +1,31 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require "json"
3
4
  require "mechanize"
5
+ require "pubid"
4
6
 
5
7
  module Relaton
6
8
  module Omg
7
9
  class Scraper
8
10
  URL_PATTERN = "https://www.omg.org/spec/"
11
+ LD_DATE = "https://www.omg.org/techprocess/ab/SpecificationMetadata/publicationDate"
9
12
 
10
- def initialize(acronym, version = nil, spec = nil)
13
+ # @param acronym [String] the specification acronym, e.g. "UML"
14
+ # @param version [String, nil] the version, e.g. "2.1.1" or "2.5 beta 1"
15
+ # @param part [String, nil] the document part, e.g. "Superstructure"
16
+ def initialize(acronym, version = nil, part = nil)
11
17
  @acronym = acronym
12
18
  @version = version
13
- @spec = spec
19
+ @part = part
14
20
  end
15
21
 
22
+ # @param ref [String] the OMG reference, e.g. "OMG UML 2.1.1 Superstructure"
23
+ # @return [Relaton::Omg::ItemData, nil] nil when the page is not found
24
+ # @raise [Pubid::Errors::ParseError] when the reference is not an OMG
25
+ # identifier
16
26
  def self.scrape_page(ref)
17
- %r{^OMG (?<acronym>[^\s]+)(?:[\s/](?<version>[\d.]+(?:\sbeta(?:\s\d)?)?))?(?:[\s/](?<spec>\w+))?$} =~ ref
18
- return unless acronym
19
-
20
- scraper = new(acronym, version, spec)
27
+ pubid = ::Pubid::Omg::Identifier.parse(ref)
28
+ scraper = new(pubid.acronym, pubid.version, pubid.part)
21
29
  doc = scraper.get_doc
22
30
  return if doc.nil? || scraper.fetch_link.empty?
23
31
 
@@ -56,15 +64,15 @@ module Relaton
56
64
 
57
65
  def fetch_title
58
66
  content = @doc.at('//dt[.="Title:"]/following-sibling::dd').text
59
- content += ": #{@spec}" if @spec
67
+ content += ": #{@part}" if @part
60
68
  [Bib::Title.new(type: "main", content: content, language: "en", script: "Latn")]
61
69
  end
62
70
 
71
+ # The version comes from the page, not from the query, so a versionless
72
+ # query (`OMG AMI4CCM`) answers with the version that the page shows.
63
73
  def fetch_docid
64
- id = ["OMG", @acronym]
65
- id << doc_version if doc_version
66
- id << @spec if @spec
67
- [Bib::Docidentifier.new(content: id.join(" "), type: "OMG", primary: true)]
74
+ id = render_id(@acronym, doc_version, @part)
75
+ [Docidentifier.new(content: id, type: "OMG", primary: true)]
68
76
  end
69
77
 
70
78
  def fetch_abstract
@@ -81,11 +89,39 @@ module Relaton
81
89
  end
82
90
 
83
91
  def fetch_date
92
+ return [] unless pub_date
93
+
84
94
  [Bib::Date.new(type: "published", at: pub_date.to_s)]
85
95
  end
86
96
 
87
97
  def pub_date
88
- ::Date.parse @doc.at('//dt[.="Publication Date:"]/following-sibling::dd').text.strip
98
+ return @pub_date if defined? @pub_date
99
+
100
+ @pub_date = jsonld_date || dd_date
101
+ end
102
+
103
+ # The visible date renders the month name in the locale of the server, and
104
+ # the CDN caches that variant. Parse the machine-readable value first.
105
+ #
106
+ # @return [Date, nil]
107
+ def jsonld_date
108
+ script = @doc.at('//script[@type="application/ld+json"]')
109
+ return unless script
110
+
111
+ node = JSON.parse(script.text).find { |e| e.is_a?(Hash) && e[LD_DATE] }
112
+ value = node && node[LD_DATE].first["@value"]
113
+ value && ::Date.parse(value)
114
+ rescue JSON::ParserError, ::Date::Error
115
+ nil
116
+ end
117
+
118
+ # @return [Date, nil]
119
+ def dd_date
120
+ text = @doc.at('//dt[.="Publication Date:"]/following-sibling::dd').text.strip
121
+ ::Date.parse text
122
+ rescue ::Date::Error
123
+ Util.warn "Cannot parse the publication date `#{text}`."
124
+ nil
89
125
  end
90
126
 
91
127
  def fetch_status
@@ -98,8 +134,8 @@ module Relaton
98
134
  return @links if @links
99
135
 
100
136
  @links = []
101
- if @spec
102
- a = @doc.at("//a[@href='#{@url}/#{@spec}/PDF']")
137
+ if @part
138
+ a = @doc.at("//a[@href='#{@url}/#{@part}/PDF']")
103
139
  @links << Bib::Uri.new(type: "src", content: a[:href]) if a
104
140
  else
105
141
  a = @doc.at('//dt[.="This Document:"]/following-sibling::dd/a')
@@ -116,8 +152,8 @@ module Relaton
116
152
  ver = row.at("td").text
117
153
  unless ver == doc_version
118
154
  acronym = row.at("td[3]/a")[:href].split("/")[4]
119
- id = ["OMG", acronym, ver].join(" ")
120
- docid = Bib::Docidentifier.new(content: id, type: "OMG")
155
+ id = render_id(acronym, ver)
156
+ docid = Docidentifier.new(content: id, type: "OMG")
121
157
  bibitem = Bib::ItemBase.new(formattedref: Bib::Formattedref.new(content: id), docidentifier: [docid])
122
158
  mem << Bib::Relation.new(type: "obsoletes", bibitem: bibitem)
123
159
  end
@@ -136,6 +172,15 @@ module Relaton
136
172
  '//dt/span/a[contains(., "IPR Mode")]/../../following-sibling::dd/span',
137
173
  ).map { |l| l.text.match(/[\w\s-]+/).to_s.strip }
138
174
  end
175
+
176
+ private
177
+
178
+ # @return [String] the identifier, rendered by pubid
179
+ def render_id(acronym, version, part = nil)
180
+ ::Pubid::Omg::Identifiers::Specification.new(
181
+ acronym: acronym, version: version, part: part,
182
+ ).to_s
183
+ end
139
184
  end
140
185
  end
141
186
  end
data/lib/relaton/omg.rb CHANGED
@@ -3,6 +3,7 @@ require "relaton/bib"
3
3
  require_relative "version"
4
4
  require_relative "omg/util"
5
5
  require_relative "omg/ext"
6
+ require_relative "omg/docidentifier"
6
7
  require_relative "omg/item_data"
7
8
  require_relative "omg/item"
8
9
  require_relative "omg/bibitem"
@@ -7,7 +7,11 @@ module Relaton
7
7
  HitCollection.new(code).find
8
8
  end
9
9
 
10
- def get(code, _year = nil, _opts = {})
10
+ # Only a transport failure is rescued, and it becomes the
11
+ # Relaton::RequestError that Relaton::Db retries. Anything else keeps
12
+ # its own class: an unrecognized reference raises Pubid::Errors::ParseError
13
+ # (relaton-cli reports it), and a bug keeps its backtrace.
14
+ def get(code, _year = nil, _opts = {}) # rubocop:disable Metrics/MethodLength
11
15
  Util.info "Fetching ...", key: code
12
16
  result = search(code).fetch_doc
13
17
  if result
@@ -16,8 +20,11 @@ module Relaton
16
20
  else
17
21
  Util.warn "Not found.", key: code
18
22
  end
19
- rescue StandardError => e
20
- raise Error, e.message
23
+ rescue SocketError, Errno::EINVAL, Errno::ECONNRESET, EOFError,
24
+ Net::HTTPBadResponse, Net::HTTPHeaderSyntaxError,
25
+ Net::ProtocolError, Net::ReadTimeout, OpenSSL::SSL::SSLError,
26
+ Errno::ETIMEDOUT => e
27
+ raise Relaton::RequestError, e.message
21
28
  end
22
29
  end
23
30
  end
@@ -13,7 +13,9 @@ module Relaton
13
13
  TECHNICAL_REPORTS_URL = "https://www.mlit.go.jp/plateau/_next/data/1.3.0/libraries/technical-reports.json".freeze
14
14
 
15
15
  def index
16
- @index ||= Relaton::Index.find_or_create :plateau, file: "#{INDEXFILE}.yaml"
16
+ @index ||= Relaton::Index.find_or_create(
17
+ :plateau, file: "#{INDEXFILE}.yaml", pubid_class: ::Pubid::Plateau::Identifier
18
+ )
17
19
  end
18
20
 
19
21
  def log_error(msg)
@@ -123,10 +125,31 @@ module Relaton
123
125
  else
124
126
  File.write(file, serialize(item))
125
127
  @files << file
126
- index.add_or_update id, file
128
+ pid = pubid id
129
+ if pid
130
+ index.add_or_update pid, file
131
+ else
132
+ Util.warn "Unparseable id `#{id}` was not indexed (#{file})", key: id
133
+ end
127
134
  end
128
135
  end
129
136
 
137
+ # Parse a canonical PLATEAU docidentifier into a Pubid::Plateau::Identifier,
138
+ # or nil if pubid can't parse it or it doesn't round-trip through
139
+ # `from_hash(to_hash)` — so a single bad id never aborts the crawl or
140
+ # corrupts index-v2 (the read side rejects an index whose rows don't
141
+ # deserialize). The parser emits canonical ids, so this should not skip
142
+ # anything; the guard is defensive.
143
+ def pubid(id)
144
+ pid = ::Pubid::Plateau.parse id
145
+ hash = pid.to_hash
146
+ return nil unless ::Pubid::Plateau::Identifier.from_hash(hash).to_hash == hash
147
+
148
+ pid
149
+ rescue StandardError
150
+ nil
151
+ end
152
+
130
153
  def file_name(id)
131
154
  name = id.gsub(/\s+/, "-").gsub(/[^\w-]+/, "").downcase
132
155
  if id.match?(/民間活用編/)
@@ -16,13 +16,20 @@ module Relaton
16
16
  @edition ||= @version["title"].split.first.match(/[\d.]+/).to_s
17
17
  end
18
18
 
19
+ # Canonical PLATEAU handbook edition label, e.g. "第3.0版" — the form
20
+ # Pubid::Plateau parses. The Latinized #edition is kept for the structured
21
+ # Bib::Edition/number and the ext structuredidentifier.
22
+ def edition_label
23
+ @edition_label ||= @version["title"].to_s[/第[\d.]+版/].to_s
24
+ end
25
+
19
26
  def slug_number
20
27
  @slug_number ||= @entry["slug"]&.to_s&.split("_")&.first
21
28
  end
22
29
 
23
30
  def parse_docnumber
24
31
  @errors[:hb_docnumber] &&= @entry["slug"].nil? || @entry["slug"].to_s.empty?
25
- ["Handbook ##{slug_number}", edition].compact.join(" ")
32
+ ["Handbook ##{slug_number}", edition_label].reject { |s| s.to_s.empty? }.join(" ")
26
33
  end
27
34
 
28
35
  def parse_abstract
@@ -1,10 +1,18 @@
1
1
  module Relaton
2
2
  module Plateau
3
3
  class Hit < Relaton::Core::Hit
4
+ # The index names the file, so a non-200 is an access failure, not a
5
+ # miss. Without the check the error page body reached Item.from_yaml,
6
+ # which returned an item with no docidentifier.
4
7
  def item
5
8
  @item ||= begin
6
- url = "#{HitCollection::ENDPOINT}#{hit[:file]}"
7
- resp = Net::HTTP.get_response(URI(url))
9
+ uri = URI("#{HitCollection::ENDPOINT}#{hit[:file]}")
10
+ resp = Net::HTTP.get_response(uri)
11
+ unless resp.code == "200"
12
+ raise Relaton::RequestError,
13
+ "Could not access #{uri}: HTTP #{resp.code}"
14
+ end
15
+
8
16
  Item.from_yaml(resp.body)
9
17
  end
10
18
  end