pubid 2.0.0.pre.alpha.9 → 2.0.0.pre.alpha.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (432) hide show
  1. checksums.yaml +4 -4
  2. data/README.adoc +33 -11
  3. data/conformance/pending.yaml +4 -0
  4. data/lib/pubid/adobe/identifier.rb +1 -3
  5. data/lib/pubid/adobe/parser.rb +10 -2
  6. data/lib/pubid/adobe.rb +1 -0
  7. data/lib/pubid/amca/builder.rb +2 -2
  8. data/lib/pubid/amca/identifiers/base.rb +49 -27
  9. data/lib/pubid/amca/identifiers/interpretation.rb +17 -9
  10. data/lib/pubid/amca/identifiers/publication.rb +13 -9
  11. data/lib/pubid/amca/identifiers/standard.rb +2 -0
  12. data/lib/pubid/amca/parser.rb +1 -1
  13. data/lib/pubid/amca/renderer.rb +4 -4
  14. data/lib/pubid/amca/urn_generator.rb +2 -2
  15. data/lib/pubid/amca.rb +2 -1
  16. data/lib/pubid/ansi/builder.rb +2 -2
  17. data/lib/pubid/ansi/identifier.rb +18 -0
  18. data/lib/pubid/ansi/parser.rb +1 -1
  19. data/lib/pubid/ansi/renderer.rb +2 -2
  20. data/lib/pubid/ansi.rb +5 -5
  21. data/lib/pubid/api/builder.rb +16 -2
  22. data/lib/pubid/api/identifier.rb +25 -4
  23. data/lib/pubid/api/identifiers/mpms.rb +10 -8
  24. data/lib/pubid/api/parser.rb +1 -1
  25. data/lib/pubid/api/renderer.rb +16 -7
  26. data/lib/pubid/api/single_identifier.rb +5 -12
  27. data/lib/pubid/api.rb +1 -0
  28. data/lib/pubid/ashrae/builder.rb +31 -10
  29. data/lib/pubid/ashrae/identifiers/addenda_package.rb +6 -0
  30. data/lib/pubid/ashrae/identifiers/addendum.rb +8 -0
  31. data/lib/pubid/ashrae/identifiers/base.rb +81 -5
  32. data/lib/pubid/ashrae/identifiers/combined_addenda.rb +7 -0
  33. data/lib/pubid/ashrae/identifiers/errata.rb +9 -0
  34. data/lib/pubid/ashrae/identifiers/guideline.rb +3 -0
  35. data/lib/pubid/ashrae/identifiers/interpretation.rb +16 -0
  36. data/lib/pubid/ashrae/identifiers/standard.rb +3 -0
  37. data/lib/pubid/ashrae/parser.rb +23 -8
  38. data/lib/pubid/ashrae/renderer.rb +6 -6
  39. data/lib/pubid/ashrae/supplement_identifier.rb +17 -2
  40. data/lib/pubid/ashrae/urn_generator.rb +8 -2
  41. data/lib/pubid/ashrae.rb +7 -1
  42. data/lib/pubid/asme/builder.rb +11 -2
  43. data/lib/pubid/asme/identifier.rb +9 -0
  44. data/lib/pubid/asme/identifiers/standard.rb +86 -2
  45. data/lib/pubid/asme/parser.rb +1 -1
  46. data/lib/pubid/asme/renderer.rb +3 -3
  47. data/lib/pubid/asme/single_identifier.rb +8 -1
  48. data/lib/pubid/asme/urn_generator.rb +2 -2
  49. data/lib/pubid/asme.rb +1 -0
  50. data/lib/pubid/astm/builder.rb +10 -3
  51. data/lib/pubid/astm/identifier.rb +9 -0
  52. data/lib/pubid/astm/identifiers/adjunct.rb +16 -1
  53. data/lib/pubid/astm/identifiers/code_number.rb +81 -0
  54. data/lib/pubid/astm/identifiers/data_series.rb +2 -0
  55. data/lib/pubid/astm/identifiers/iso_dual_published.rb +16 -0
  56. data/lib/pubid/astm/identifiers/manual.rb +10 -0
  57. data/lib/pubid/astm/identifiers/monograph.rb +2 -0
  58. data/lib/pubid/astm/identifiers/research_report.rb +10 -0
  59. data/lib/pubid/astm/identifiers/standard.rb +10 -0
  60. data/lib/pubid/astm/identifiers/technical_report.rb +2 -0
  61. data/lib/pubid/astm/identifiers/work_in_progress.rb +2 -0
  62. data/lib/pubid/astm/identifiers.rb +1 -0
  63. data/lib/pubid/astm/parser.rb +1 -1
  64. data/lib/pubid/astm/renderer.rb +1 -1
  65. data/lib/pubid/astm/single_identifier.rb +31 -1
  66. data/lib/pubid/astm/urn_generator.rb +7 -5
  67. data/lib/pubid/astm.rb +1 -0
  68. data/lib/pubid/bipm/builder.rb +66 -10
  69. data/lib/pubid/bipm/identifier.rb +134 -9
  70. data/lib/pubid/bipm/identifiers/committee_document.rb +12 -0
  71. data/lib/pubid/bipm/identifiers/guide.rb +11 -0
  72. data/lib/pubid/bipm/identifiers/meeting.rb +12 -0
  73. data/lib/pubid/bipm/identifiers/mep.rb +12 -0
  74. data/lib/pubid/bipm/identifiers/metrologia_article.rb +24 -0
  75. data/lib/pubid/bipm/identifiers/si_brochure.rb +13 -0
  76. data/lib/pubid/bipm/parser.rb +65 -12
  77. data/lib/pubid/bipm/renderer.rb +8 -0
  78. data/lib/pubid/bipm/urn_parser.rb +11 -3
  79. data/lib/pubid/bipm.rb +7 -1
  80. data/lib/pubid/bsi/builder.rb +48 -38
  81. data/lib/pubid/bsi/identifiers/adopted_european_norm.rb +36 -7
  82. data/lib/pubid/bsi/identifiers/adopted_international_standard.rb +7 -6
  83. data/lib/pubid/bsi/identifiers/british_industrial_practice.rb +0 -1
  84. data/lib/pubid/bsi/identifiers/bundled_identifier.rb +10 -0
  85. data/lib/pubid/bsi/identifiers/committee_document.rb +10 -1
  86. data/lib/pubid/bsi/identifiers/handbook.rb +0 -2
  87. data/lib/pubid/bsi/identifiers/practice_guide.rb +0 -1
  88. data/lib/pubid/bsi/identifiers/set.rb +11 -0
  89. data/lib/pubid/bsi/identifiers/standalone_amendment.rb +10 -1
  90. data/lib/pubid/bsi/parser.rb +1 -1
  91. data/lib/pubid/bsi/renderer.rb +63 -51
  92. data/lib/pubid/bsi/single_identifier.rb +16 -7
  93. data/lib/pubid/bsi/urn_generator.rb +11 -2
  94. data/lib/pubid/bsi.rb +7 -1
  95. data/lib/pubid/builder/base.rb +18 -8
  96. data/lib/pubid/bundled_identifier.rb +16 -6
  97. data/lib/pubid/calconnect/identifier.rb +9 -5
  98. data/lib/pubid/calconnect/parser.rb +1 -1
  99. data/lib/pubid/calconnect.rb +7 -1
  100. data/lib/pubid/ccsds/identifier.rb +9 -2
  101. data/lib/pubid/ccsds/identifiers/base.rb +4 -2
  102. data/lib/pubid/ccsds/identifiers/corrigendum.rb +2 -2
  103. data/lib/pubid/ccsds/parser.rb +1 -1
  104. data/lib/pubid/ccsds/single_identifier.rb +2 -2
  105. data/lib/pubid/ccsds.rb +1 -0
  106. data/lib/pubid/cen_cenelec/builder.rb +101 -76
  107. data/lib/pubid/cen_cenelec/identifier.rb +115 -3
  108. data/lib/pubid/cen_cenelec/identifiers/adopted_european_norm.rb +27 -16
  109. data/lib/pubid/cen_cenelec/identifiers/amendment.rb +46 -6
  110. data/lib/pubid/cen_cenelec/identifiers/base.rb +8 -9
  111. data/lib/pubid/cen_cenelec/identifiers/cen_report.rb +1 -1
  112. data/lib/pubid/cen_cenelec/identifiers/cen_workshop_agreement.rb +1 -1
  113. data/lib/pubid/cen_cenelec/identifiers/cenelec_harmonization_document.rb +1 -1
  114. data/lib/pubid/cen_cenelec/identifiers/consolidated_identifier.rb +29 -20
  115. data/lib/pubid/cen_cenelec/identifiers/corrigendum.rb +47 -8
  116. data/lib/pubid/cen_cenelec/identifiers/european_prestandard.rb +27 -2
  117. data/lib/pubid/cen_cenelec/identifiers/european_specification.rb +1 -1
  118. data/lib/pubid/cen_cenelec/identifiers/fragment.rb +15 -5
  119. data/lib/pubid/cen_cenelec/identifiers/guide.rb +1 -1
  120. data/lib/pubid/cen_cenelec/identifiers/harmonization_document.rb +1 -1
  121. data/lib/pubid/cen_cenelec/identifiers/technical_report.rb +1 -1
  122. data/lib/pubid/cen_cenelec/identifiers/technical_specification.rb +1 -1
  123. data/lib/pubid/cen_cenelec/parser.rb +1 -1
  124. data/lib/pubid/cen_cenelec/renderer.rb +55 -60
  125. data/lib/pubid/cen_cenelec/single_identifier.rb +20 -2
  126. data/lib/pubid/cen_cenelec/urn_generator.rb +66 -44
  127. data/lib/pubid/cen_cenelec.rb +28 -10
  128. data/lib/pubid/cie/builder.rb +10 -0
  129. data/lib/pubid/cie/identifier.rb +6 -1
  130. data/lib/pubid/cie/identifiers/bundle.rb +4 -2
  131. data/lib/pubid/cie/identifiers/conference.rb +2 -2
  132. data/lib/pubid/cie/identifiers/corrigendum.rb +2 -2
  133. data/lib/pubid/cie/identifiers/dual_published.rb +2 -2
  134. data/lib/pubid/cie/identifiers/identical.rb +2 -2
  135. data/lib/pubid/cie/identifiers/joint_published.rb +2 -2
  136. data/lib/pubid/cie/identifiers/proceedings.rb +8 -6
  137. data/lib/pubid/cie/identifiers/standard.rb +22 -13
  138. data/lib/pubid/cie/identifiers/supplement.rb +2 -2
  139. data/lib/pubid/cie/identifiers/tutorial_bundle.rb +2 -2
  140. data/lib/pubid/cie/parser.rb +1 -1
  141. data/lib/pubid/cie.rb +7 -1
  142. data/lib/pubid/conformance/checks.rb +72 -0
  143. data/lib/pubid/conformance/corpus/case.rb +55 -0
  144. data/lib/pubid/conformance/corpus.rb +36 -0
  145. data/lib/pubid/conformance/generator.rb +212 -0
  146. data/lib/pubid/conformance/pending.rb +63 -0
  147. data/lib/pubid/conformance/runner.rb +153 -0
  148. data/lib/pubid/conformance.rb +55 -0
  149. data/lib/pubid/core.rb +2 -0
  150. data/lib/pubid/csa/builder.rb +47 -47
  151. data/lib/pubid/csa/composite_identifier.rb +14 -19
  152. data/lib/pubid/csa/identifier.rb +143 -81
  153. data/lib/pubid/csa/identifiers/bundled.rb +23 -13
  154. data/lib/pubid/csa/identifiers/canadian_adopted.rb +13 -13
  155. data/lib/pubid/csa/identifiers/cec.rb +44 -4
  156. data/lib/pubid/csa/identifiers/combined.rb +95 -73
  157. data/lib/pubid/csa/identifiers/csa_adopted.rb +4 -4
  158. data/lib/pubid/csa/identifiers/package.rb +2 -2
  159. data/lib/pubid/csa/parser.rb +6 -2
  160. data/lib/pubid/csa/renderer.rb +13 -5
  161. data/lib/pubid/csa/single_identifier.rb +93 -4
  162. data/lib/pubid/csa/urn_generator.rb +9 -2
  163. data/lib/pubid/csa/wrapper_identifier.rb +39 -23
  164. data/lib/pubid/csa.rb +7 -1
  165. data/lib/pubid/doi/identifier.rb +6 -3
  166. data/lib/pubid/doi/parser.rb +1 -1
  167. data/lib/pubid/doi.rb +1 -0
  168. data/lib/pubid/easc/identifier.rb +0 -2
  169. data/lib/pubid/easc/parser.rb +10 -2
  170. data/lib/pubid/easc.rb +1 -0
  171. data/lib/pubid/ecma/builder.rb +23 -9
  172. data/lib/pubid/ecma/identifier.rb +81 -10
  173. data/lib/pubid/ecma/parser.rb +39 -4
  174. data/lib/pubid/ecma/renderer.rb +29 -8
  175. data/lib/pubid/ecma/urn_generator.rb +34 -13
  176. data/lib/pubid/ecma/urn_parser.rb +30 -14
  177. data/lib/pubid/ecma.rb +1 -0
  178. data/lib/pubid/errors.rb +80 -0
  179. data/lib/pubid/etsi/builder.rb +6 -1
  180. data/lib/pubid/etsi/identifiers/base.rb +34 -20
  181. data/lib/pubid/etsi/identifiers/etsi_standard.rb +120 -41
  182. data/lib/pubid/etsi/identifiers/supplement_identifier.rb +10 -0
  183. data/lib/pubid/etsi/parser.rb +1 -1
  184. data/lib/pubid/etsi.rb +1 -0
  185. data/lib/pubid/evs/builder.rb +26 -0
  186. data/lib/pubid/evs/identifier.rb +37 -0
  187. data/lib/pubid/evs/identifiers/national_adoption.rb +17 -0
  188. data/lib/pubid/evs/identifiers.rb +9 -0
  189. data/lib/pubid/evs/parser.rb +27 -0
  190. data/lib/pubid/evs/renderer.rb +30 -0
  191. data/lib/pubid/evs/urn_generator.rb +30 -0
  192. data/lib/pubid/evs/urn_parser.rb +80 -0
  193. data/lib/pubid/evs.rb +90 -0
  194. data/lib/pubid/export.rb +1 -0
  195. data/lib/pubid/format_detector.rb +1 -0
  196. data/lib/pubid/gb/identifier.rb +6 -3
  197. data/lib/pubid/gb/parser.rb +1 -1
  198. data/lib/pubid/gb.rb +1 -0
  199. data/lib/pubid/gost/builder.rb +46 -9
  200. data/lib/pubid/gost/identifier.rb +0 -2
  201. data/lib/pubid/gost/identifiers/foreign_reference.rb +1 -1
  202. data/lib/pubid/gost/parser.rb +10 -2
  203. data/lib/pubid/gost.rb +1 -0
  204. data/lib/pubid/iala/identifier.rb +9 -6
  205. data/lib/pubid/iala/parser.rb +10 -2
  206. data/lib/pubid/iala/urn_parser.rb +6 -2
  207. data/lib/pubid/iala.rb +1 -0
  208. data/lib/pubid/iana/builder.rb +3 -1
  209. data/lib/pubid/iana/identifier.rb +74 -6
  210. data/lib/pubid/iana/identifiers/registry.rb +36 -0
  211. data/lib/pubid/iana/parser.rb +1 -1
  212. data/lib/pubid/iana/renderer.rb +3 -0
  213. data/lib/pubid/iana/urn_generator.rb +5 -1
  214. data/lib/pubid/iana.rb +1 -0
  215. data/lib/pubid/identifier.rb +594 -10
  216. data/lib/pubid/identifier_metadata.rb +1 -0
  217. data/lib/pubid/idf/builder.rb +3 -3
  218. data/lib/pubid/idf/identifier.rb +18 -9
  219. data/lib/pubid/idf/parser.rb +1 -1
  220. data/lib/pubid/idf/renderer.rb +4 -4
  221. data/lib/pubid/idf/urn_generator.rb +1 -1
  222. data/lib/pubid/idf.rb +7 -1
  223. data/lib/pubid/iec/builder.rb +14 -7
  224. data/lib/pubid/iec/components.rb +0 -1
  225. data/lib/pubid/iec/identifier.rb +73 -83
  226. data/lib/pubid/iec/identifiers/amendment.rb +6 -2
  227. data/lib/pubid/iec/identifiers/component_specification.rb +0 -10
  228. data/lib/pubid/iec/identifiers/conformity_assessment.rb +0 -10
  229. data/lib/pubid/iec/identifiers/corrigendum.rb +2 -2
  230. data/lib/pubid/iec/identifiers/fragment_identifier.rb +2 -4
  231. data/lib/pubid/iec/identifiers/guide.rb +0 -16
  232. data/lib/pubid/iec/identifiers/international_standard.rb +15 -2
  233. data/lib/pubid/iec/identifiers/operational_document.rb +0 -10
  234. data/lib/pubid/iec/identifiers/publicly_available_specification.rb +0 -12
  235. data/lib/pubid/iec/identifiers/societal_technology_trend_report.rb +0 -10
  236. data/lib/pubid/iec/identifiers/systems_reference_document.rb +0 -10
  237. data/lib/pubid/iec/identifiers/technical_report.rb +0 -17
  238. data/lib/pubid/iec/identifiers/technical_specification.rb +0 -17
  239. data/lib/pubid/iec/identifiers/technology_report.rb +0 -10
  240. data/lib/pubid/iec/identifiers/test_report_form.rb +0 -11
  241. data/lib/pubid/iec/identifiers/white_paper.rb +0 -10
  242. data/lib/pubid/iec/identifiers/working_document.rb +24 -10
  243. data/lib/pubid/iec/parser.rb +2 -18
  244. data/lib/pubid/iec/renderer.rb +13 -10
  245. data/lib/pubid/iec/single_identifier.rb +33 -10
  246. data/lib/pubid/iec/urn_generator.rb +227 -61
  247. data/lib/pubid/iec/urn_parser.rb +42 -4
  248. data/lib/pubid/iec.rb +1 -0
  249. data/lib/pubid/ieee/aiee/identifier.rb +2 -2
  250. data/lib/pubid/ieee/aiee/parser.rb +1 -1
  251. data/lib/pubid/ieee/builder.rb +45 -1
  252. data/lib/pubid/ieee/identifiers/adopted_standard.rb +32 -0
  253. data/lib/pubid/ieee/identifiers/base.rb +42 -5
  254. data/lib/pubid/ieee/identifiers/csa_dual_published.rb +21 -0
  255. data/lib/pubid/ieee/identifiers/dual_published.rb +55 -0
  256. data/lib/pubid/ieee/identifiers/interpretation_identifier.rb +23 -0
  257. data/lib/pubid/ieee/identifiers/joint_development.rb +11 -7
  258. data/lib/pubid/ieee/identifiers/multi_numbered_identifier.rb +57 -5
  259. data/lib/pubid/ieee/identifiers/nesc/base.rb +2 -2
  260. data/lib/pubid/ieee/identifiers/nesc/draft.rb +2 -2
  261. data/lib/pubid/ieee/identifiers/nesc/handbook.rb +2 -2
  262. data/lib/pubid/ieee/identifiers/nesc/redline.rb +2 -2
  263. data/lib/pubid/ieee/identifiers/nesc/standard.rb +2 -2
  264. data/lib/pubid/ieee/ire/identifier.rb +2 -2
  265. data/lib/pubid/ieee/ire/parser.rb +1 -1
  266. data/lib/pubid/ieee/nesc/parser.rb +1 -1
  267. data/lib/pubid/ieee/parser.rb +1 -1
  268. data/lib/pubid/ieee/renderer.rb +5 -1
  269. data/lib/pubid/ieee/urn_generator.rb +32 -3
  270. data/lib/pubid/ieee.rb +7 -1
  271. data/lib/pubid/ieee_debug.rb +1 -0
  272. data/lib/pubid/ietf/builder.rb +24 -8
  273. data/lib/pubid/ietf/identifiers/base.rb +25 -28
  274. data/lib/pubid/ietf/identifiers/bcp.rb +11 -0
  275. data/lib/pubid/ietf/identifiers/fyi.rb +11 -0
  276. data/lib/pubid/ietf/identifiers/internet_draft.rb +18 -2
  277. data/lib/pubid/ietf/identifiers/rfc.rb +2 -0
  278. data/lib/pubid/ietf/identifiers/serialization.rb +51 -0
  279. data/lib/pubid/ietf/identifiers/std.rb +11 -0
  280. data/lib/pubid/ietf/identifiers.rb +1 -0
  281. data/lib/pubid/ietf/parser.rb +27 -9
  282. data/lib/pubid/ietf/renderer.rb +44 -6
  283. data/lib/pubid/ietf/urn_generator.rb +20 -6
  284. data/lib/pubid/ietf/urn_parser.rb +10 -2
  285. data/lib/pubid/ietf.rb +12 -2
  286. data/lib/pubid/iho/identifiers/base.rb +10 -3
  287. data/lib/pubid/iho/parser.rb +1 -1
  288. data/lib/pubid/iho.rb +1 -0
  289. data/lib/pubid/isbn/identifier.rb +25 -7
  290. data/lib/pubid/isbn/parser.rb +1 -1
  291. data/lib/pubid/isbn.rb +1 -0
  292. data/lib/pubid/iso/bundled_identifier.rb +5 -3
  293. data/lib/pubid/iso/combined_identifier.rb +5 -3
  294. data/lib/pubid/iso/identifier.rb +12 -7
  295. data/lib/pubid/iso/identifiers/directives_supplement.rb +4 -4
  296. data/lib/pubid/iso/parser.rb +1 -1
  297. data/lib/pubid/iso.rb +7 -1
  298. data/lib/pubid/itu/identifiers/addendum.rb +2 -2
  299. data/lib/pubid/itu/identifiers/amendment.rb +2 -2
  300. data/lib/pubid/itu/identifiers/base.rb +19 -5
  301. data/lib/pubid/itu/identifiers/corrigendum.rb +2 -2
  302. data/lib/pubid/itu/identifiers/errata.rb +2 -2
  303. data/lib/pubid/itu/identifiers/supplement.rb +2 -2
  304. data/lib/pubid/itu/parser.rb +1 -1
  305. data/lib/pubid/itu.rb +1 -0
  306. data/lib/pubid/jcgm/builder.rb +1 -1
  307. data/lib/pubid/jcgm/identifier.rb +10 -0
  308. data/lib/pubid/jcgm/identifiers/meeting.rb +1 -1
  309. data/lib/pubid/jcgm/parser.rb +10 -3
  310. data/lib/pubid/jcgm/renderer.rb +10 -3
  311. data/lib/pubid/jcgm/single_identifier.rb +24 -15
  312. data/lib/pubid/jcgm/urn_generator.rb +1 -1
  313. data/lib/pubid/jcgm/urn_parser.rb +156 -14
  314. data/lib/pubid/jcgm.rb +10 -0
  315. data/lib/pubid/jis/identifier.rb +10 -3
  316. data/lib/pubid/jis/parser.rb +1 -1
  317. data/lib/pubid/jis.rb +1 -0
  318. data/lib/pubid/nist/components/stage.rb +2 -2
  319. data/lib/pubid/nist/configuration.rb +6 -8
  320. data/lib/pubid/nist/identifiers/base.rb +12 -2
  321. data/lib/pubid/nist/identifiers/circular.rb +9 -2
  322. data/lib/pubid/nist/identifiers/circular_supplement.rb +14 -5
  323. data/lib/pubid/nist/identifiers/commercial_standards_monthly.rb +7 -1
  324. data/lib/pubid/nist/identifiers/crpl_report.rb +7 -1
  325. data/lib/pubid/nist/identifiers/handbook.rb +9 -2
  326. data/lib/pubid/nist/identifiers/internal_report.rb +12 -5
  327. data/lib/pubid/nist/identifiers/miscellaneous_publication.rb +14 -6
  328. data/lib/pubid/nist/identifiers/monograph.rb +14 -7
  329. data/lib/pubid/nist/identifiers/report.rb +14 -6
  330. data/lib/pubid/nist/parser.rb +1 -1
  331. data/lib/pubid/nist/supplement_identifier.rb +16 -1
  332. data/lib/pubid/nist.rb +7 -1
  333. data/lib/pubid/oasis/builder.rb +12 -10
  334. data/lib/pubid/oasis/identifier.rb +89 -11
  335. data/lib/pubid/oasis/parser.rb +1 -1
  336. data/lib/pubid/oasis.rb +7 -1
  337. data/lib/pubid/ogc/identifier.rb +9 -5
  338. data/lib/pubid/ogc/parser.rb +1 -1
  339. data/lib/pubid/ogc.rb +1 -0
  340. data/lib/pubid/oiml/builder.rb +5 -1
  341. data/lib/pubid/oiml/identifiers/basic_publication.rb +2 -0
  342. data/lib/pubid/oiml/identifiers/bulletin.rb +36 -0
  343. data/lib/pubid/oiml/identifiers/code_number.rb +90 -0
  344. data/lib/pubid/oiml/identifiers/document.rb +2 -0
  345. data/lib/pubid/oiml/identifiers/expert_report.rb +2 -0
  346. data/lib/pubid/oiml/identifiers/guide.rb +2 -0
  347. data/lib/pubid/oiml/identifiers/recommendation.rb +2 -0
  348. data/lib/pubid/oiml/identifiers/seminar_report.rb +2 -0
  349. data/lib/pubid/oiml/identifiers/vocabulary.rb +2 -0
  350. data/lib/pubid/oiml/identifiers.rb +1 -0
  351. data/lib/pubid/oiml/parser.rb +1 -1
  352. data/lib/pubid/oiml/single_identifier.rb +26 -42
  353. data/lib/pubid/oiml/supplement_identifier.rb +27 -0
  354. data/lib/pubid/oiml.rb +7 -1
  355. data/lib/pubid/omg/builder.rb +1 -0
  356. data/lib/pubid/omg/identifier.rb +20 -3
  357. data/lib/pubid/omg/parser.rb +64 -11
  358. data/lib/pubid/omg/renderer.rb +13 -1
  359. data/lib/pubid/omg.rb +2 -1
  360. data/lib/pubid/parser/grammar.rb +74 -0
  361. data/lib/pubid/parser.rb +2 -0
  362. data/lib/pubid/parsers/mr_string.rb +19 -0
  363. data/lib/pubid/plateau/parser.rb +1 -1
  364. data/lib/pubid/plateau.rb +10 -0
  365. data/lib/pubid/prefixes_support.rb +1 -0
  366. data/lib/pubid/renderers/annotator.rb +233 -0
  367. data/lib/pubid/renderers/base.rb +13 -0
  368. data/lib/pubid/renderers/directives_renderer.rb +3 -3
  369. data/lib/pubid/renderers/human_readable.rb +3 -3
  370. data/lib/pubid/renderers/mr_string.rb +15 -9
  371. data/lib/pubid/renderers.rb +1 -0
  372. data/lib/pubid/rendering/numbering.rb +25 -7
  373. data/lib/pubid/rendering.rb +1 -0
  374. data/lib/pubid/sae/identifiers/base.rb +9 -2
  375. data/lib/pubid/sae/parser.rb +1 -1
  376. data/lib/pubid/sae.rb +1 -0
  377. data/lib/pubid/schema/declaration.rb +33 -0
  378. data/lib/pubid/schema/error.rb +14 -0
  379. data/lib/pubid/schema/identifier_type.rb +23 -0
  380. data/lib/pubid/schema/loader.rb +117 -0
  381. data/lib/pubid/schema/typed_stage.rb +24 -0
  382. data/lib/pubid/schema.rb +25 -0
  383. data/lib/pubid/tgpp/builder.rb +1 -1
  384. data/lib/pubid/tgpp/identifier.rb +21 -5
  385. data/lib/pubid/tgpp/identifiers/technical_report.rb +2 -2
  386. data/lib/pubid/tgpp/identifiers/technical_specification.rb +2 -2
  387. data/lib/pubid/tgpp/parser.rb +22 -6
  388. data/lib/pubid/tgpp/renderer.rb +21 -5
  389. data/lib/pubid/tgpp/urn_generator.rb +27 -8
  390. data/lib/pubid/tgpp/urn_parser.rb +8 -6
  391. data/lib/pubid/tgpp.rb +7 -1
  392. data/lib/pubid/type_resolver.rb +16 -1
  393. data/lib/pubid/un/identifier.rb +6 -3
  394. data/lib/pubid/un/parser.rb +1 -1
  395. data/lib/pubid/un.rb +1 -0
  396. data/lib/pubid/urn_parser/errors.rb +8 -1
  397. data/lib/pubid/urn_parser.rb +1 -0
  398. data/lib/pubid/utils.rb +1 -0
  399. data/lib/pubid/version.rb +1 -1
  400. data/lib/pubid/w3c/builder.rb +5 -5
  401. data/lib/pubid/w3c/identifier.rb +38 -9
  402. data/lib/pubid/w3c/parser.rb +1 -1
  403. data/lib/pubid/w3c/renderer.rb +1 -1
  404. data/lib/pubid/w3c/urn_generator.rb +2 -2
  405. data/lib/pubid/w3c.rb +1 -0
  406. data/lib/pubid/xsf/identifier.rb +9 -5
  407. data/lib/pubid/xsf/parser.rb +17 -4
  408. data/lib/pubid/xsf.rb +7 -1
  409. data/lib/pubid.rb +399 -21
  410. data/lib/tasks/conformance.rake +34 -0
  411. data/lib/tasks/docs.rake +3 -3
  412. data/lib/tasks/schema.rake +88 -0
  413. data/schema/core/joint_prefixes.yaml +24 -0
  414. data/schema/iec.yaml +1017 -0
  415. data/schema/iso.yaml +1586 -0
  416. data/schema/schema.schema.yaml +96 -0
  417. metadata +38 -17
  418. data/archived-gems/pubid-ccsds/update_codes.yaml +0 -1
  419. data/archived-gems/pubid-iec/stages.yaml +0 -129
  420. data/archived-gems/pubid-iec/update_codes.yaml +0 -67
  421. data/archived-gems/pubid-ieee/update_codes.yaml +0 -104
  422. data/archived-gems/pubid-iso/stages.yaml +0 -106
  423. data/archived-gems/pubid-iso/update_codes.yaml +0 -4
  424. data/archived-gems/pubid-itu/i18n.yaml +0 -13
  425. data/archived-gems/pubid-itu/series.yaml +0 -42
  426. data/archived-gems/pubid-nist/publishers.yaml +0 -6
  427. data/archived-gems/pubid-nist/series.yaml +0 -121
  428. data/archived-gems/pubid-nist/update_codes.yaml +0 -93
  429. data/archived-gems/pubid-plateau/update_codes.yaml +0 -6
  430. data/lib/pubid/cen_cenelec/supplement_identifier.rb +0 -48
  431. data/lib/pubid/iec/components/code.rb +0 -36
  432. /data/{archived-gems/pubid-nist → data/nist}/stages.yaml +0 -0
data/lib/pubid/oiml.rb CHANGED
@@ -1,5 +1,6 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require "pubid"
3
4
  module Pubid
4
5
  module Oiml
5
6
  extend Pubid::PrefixesSupport
@@ -19,8 +20,13 @@ module Pubid
19
20
  autoload :UrnParser, "#{__dir__}/oiml/urn_parser"
20
21
 
21
22
  def self.parse(identifier)
23
+ unless identifier.is_a?(String)
24
+ raise Pubid::Errors::InvalidInputError,
25
+ Pubid::INPUT_NOT_A_STRING_MESSAGE
26
+ end
27
+
22
28
  if identifier.length > Pubid::MAX_INPUT_LENGTH
23
- raise ArgumentError, Pubid::INPUT_TOO_LONG_MESSAGE
29
+ raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
24
30
  end
25
31
 
26
32
  parser = Parser.new
@@ -12,6 +12,7 @@ module Pubid
12
12
  Identifiers::Specification.new(
13
13
  acronym: data[:acronym].to_s,
14
14
  version: data[:version]&.to_s,
15
+ part: data[:part]&.to_s,
15
16
  )
16
17
  end
17
18
  end
@@ -13,6 +13,19 @@ module Pubid
13
13
  # rendering.
14
14
  attribute :version, :string
15
15
 
16
+ # The document part: the volume or format segment after the version,
17
+ # e.g. "Superstructure", "Infrastructure", "PDF".
18
+ #
19
+ # This retypes the `part` that ::Pubid::Identifier declares as a
20
+ # Components::Code, because an OMG part is a name and not a numbered
21
+ # part. Reusing the inherited name is what gives relaton `remove_part!`
22
+ # as `exclude(:part)`, and what lets Renderers::Annotator wrap the value
23
+ # from its own TOKENS table. The declaration sits once, on the class
24
+ # every OMG identifier inherits from, whose body lives in this one file
25
+ # and is never reopened — the placement the number-retype tranches
26
+ # require.
27
+ attribute :part, :string
28
+
16
29
  OMG_TYPE_MAP = {
17
30
  "pubid:omg:specification" => "Pubid::Omg::Identifiers::Specification",
18
31
  }.freeze
@@ -21,6 +34,7 @@ module Pubid
21
34
  map "_type", to: :_type, polymorphic_map: OMG_TYPE_MAP
22
35
  map "acronym", to: :acronym
23
36
  map "version", to: :version
37
+ map "part", to: :part
24
38
  end
25
39
 
26
40
  PUBLISHER = "OMG"
@@ -30,14 +44,17 @@ module Pubid
30
44
  end
31
45
 
32
46
  def self.parse(identifier)
47
+ unless identifier.is_a?(String)
48
+ raise Pubid::Errors::InvalidInputError,
49
+ Pubid::INPUT_NOT_A_STRING_MESSAGE
50
+ end
51
+
33
52
  if identifier.length > Pubid::MAX_INPUT_LENGTH
34
- raise ArgumentError, Pubid::INPUT_TOO_LONG_MESSAGE
53
+ raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
35
54
  end
36
55
 
37
56
  parsed = Parser.parse(identifier)
38
57
  Builder.build(parsed)
39
- rescue Parslet::ParseFailed => e
40
- raise "Failed to parse OMG identifier '#{identifier}': #{e.message}"
41
58
  end
42
59
  end
43
60
  end
@@ -7,27 +7,80 @@ module Pubid
7
7
  # Parslet grammar for OMG specification identifiers.
8
8
  #
9
9
  # Accepts:
10
- # OMG {ACRONYM}[ {VERSION}]
10
+ # OMG {ACRONYM}[ {VERSION}][ {PART}]
11
11
  #
12
- # ACRONYM is uppercase letters/digits, at least 1 char.
13
- # VERSION is digits/dots/spaces/beta+space+digit, e.g. "1.0", "2.5.1",
14
- # "5 beta 3".
15
- class Parser < Parslet::Parser
12
+ # ACRONYM is the URL segment of https://www.omg.org/spec/<ACRONYM>/, e.g.
13
+ # "UML", "DDS-XTypes", "EDMC-FIBO/BE", "VSIPL++", "smartant".
14
+ # VERSION is digits/dots, optionally followed by a beta label, e.g. "1.0",
15
+ # "2.5.1", "5 beta 3", "2.5 beta".
16
+ # PART is the volume or format segment, e.g. "Superstructure", "PDF". A
17
+ # space or a slash separates it.
18
+ class Parser < ::Pubid::Parser::Grammar
16
19
  rule(:space) { str(" ") }
17
20
 
18
- # Acronym: starts with uppercase, may contain uppercase + lowercase +
19
- # digits (covers "SysML", "AMI4CCM", "UML", "CORBA", "BMM", ...).
20
- rule(:acronym) { (match("[A-Z]") >> match("[A-Za-z0-9]").repeat).as(:acronym) }
21
+ # Acronym: the URL segment OMG gives the specification, kept verbatim,
22
+ # because a consumer builds the URL from it. It starts with a letter,
23
+ # which may be lower case ("smartant"). Letters, digits and "+" follow
24
+ # ("SysML", "AMI4CCM", "VSIPL++"). A hyphen or a slash joins further
25
+ # segments ("DDS-PSM-Cxx", "EDMC-FIBO/BE"). Each segment is non-empty,
26
+ # so a hyphen or a slash never ends the acronym.
27
+ #
28
+ # The slash in "EDMC-FIBO/BE" belongs to the acronym: OMG serves the
29
+ # document at /spec/EDMC-FIBO/BE/. A slash after the version separates
30
+ # the document part instead (see the identifier rule).
31
+ rule(:acronym_char) { match("[A-Za-z0-9+]") }
32
+
33
+ rule(:acronym) do
34
+ (match("[A-Za-z]") >> acronym_char.repeat >>
35
+ (match("[-/]") >> acronym_char.repeat(1)).repeat).as(:acronym)
36
+ end
37
+
38
+ # Version: digits with optional dots, optionally followed by " beta" and
39
+ # an optional beta number. OMG writes the label both ways: the document
40
+ # at /spec/UML/2.5/Beta1 gives its own version as "2.5 beta", and DDS 1.4
41
+ # supersedes /spec/DDS/1.4/Beta2.
42
+ #
43
+ # The beta number must stay optional here. The document part below would
44
+ # otherwise swallow a bare "beta" and report the version as "2.5" — a
45
+ # silent wrong answer, not a parse failure.
46
+ #
47
+ # Both halves of the beta label end at a word boundary. Parslet never
48
+ # backtracks into a `.maybe` that already succeeded, so an unanchored
49
+ # literal makes the version eat the front of a document part and then
50
+ # reject the whole identifier: "OMG DDS 1.4 beta2" and
51
+ # "OMG DDS 1.4 betawave" would raise instead of reading the part.
52
+ rule(:word_boundary) { match("[A-Za-z0-9]").absent? }
53
+
54
+ rule(:beta) do
55
+ str(" beta") >> word_boundary >>
56
+ (space >> match("[0-9]").repeat(1) >> word_boundary).maybe
57
+ end
21
58
 
22
- # Version: digits with optional dots, optionally followed by " beta N".
23
59
  rule(:version) do
24
60
  (match("[0-9]").repeat(1) >>
25
61
  (str(".") >> match("[0-9]").repeat(1)).repeat >>
26
- (str(" beta ") >> match("[0-9]").repeat(1)).maybe).as(:version)
62
+ beta.maybe).as(:version)
27
63
  end
28
64
 
65
+ # After the version, OMG separates the document part with either a
66
+ # space or a slash. The renderer prints a space, so the two spellings of
67
+ # one document stay equal.
68
+ rule(:part_separator) { space | str("/") }
69
+
70
+ # Document part: the volume or format segment OMG puts after the
71
+ # version. UML 2.1.1 is two documents, Superstructure and
72
+ # Infrastructure, and the URL carries the same segment
73
+ # (/spec/UML/2.1.1/Superstructure). A format name (/spec/DDS/1.4/PDF)
74
+ # occupies the same position.
75
+ rule(:part) { match("[A-Za-z0-9]").repeat(1).as(:part) }
76
+
77
+ # Only a space separates a part that follows the acronym directly. The
78
+ # acronym rule takes a slash there, so a slash separator is legal only
79
+ # after a version.
29
80
  rule(:identifier) do
30
- str("OMG") >> space >> acronym >> (space >> version).maybe
81
+ str("OMG") >> space >> acronym >>
82
+ ((space >> version >> (part_separator >> part).maybe) |
83
+ (space >> part)).maybe
31
84
  end
32
85
 
33
86
  rule(:root) { identifier }
@@ -7,13 +7,25 @@ module Pubid
7
7
  # Produces:
8
8
  # "OMG AMI4CCM 1.0"
9
9
  # "OMG UML 2.5.1"
10
+ # "OMG UML 2.1.1 Superstructure"
10
11
  # "OMG CORBA"
12
+ #
13
+ # The document part always prints behind a space. OMG writes it behind
14
+ # either a space or a slash, and the parser takes both, so normalizing
15
+ # here is what keeps the two spellings of one document equal.
11
16
  class Renderer < ::Pubid::Renderers::Base
12
17
  def render(**_opts)
13
18
  result = "OMG #{@id.acronym}"
14
- result += " #{@id.version}" if @id.version && !@id.version.empty?
19
+ result += " #{@id.version}" if present?(@id.version)
20
+ result += " #{@id.part}" if present?(@id.part)
15
21
  result
16
22
  end
23
+
24
+ private
25
+
26
+ def present?(value)
27
+ value && !value.empty?
28
+ end
17
29
  end
18
30
  end
19
31
  end
data/lib/pubid/omg.rb CHANGED
@@ -1,10 +1,11 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require "pubid"
3
4
  module Pubid
4
5
  # OMG (Object Management Group) specification flavor.
5
6
  #
6
7
  # Covers formal OMG specifications identified by an acronym (UML, SysML,
7
- # CORBA, AMI4CCM, ...) with optional version (`1.0`, `2.5.1`,
8
+ # CORBA, DDS-XTypes, EDMC-FIBO/BE, ...) with optional version (`1.0`, `2.5.1`,
8
9
  # `5 beta 3`).
9
10
  module Omg
10
11
  extend Pubid::PrefixesSupport
@@ -0,0 +1,74 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "parslet"
4
+ require_relative "../errors"
5
+
6
+ module Pubid
7
+ module Parser
8
+ # The superclass of every flavor grammar.
9
+ #
10
+ # Its only job is to keep parslet out of pubid's public contract. Every
11
+ # grammar failure originates in `Parslet::Atoms::Base#parse`, which
12
+ # `Parslet::Parser` does not override, so one override here covers all 46
13
+ # grammars, both public entry points of every flavor, and the cross-flavor
14
+ # delegation routes (`Pubid::Iso.parse("ITU-T G.711")` runs ITU's grammar)
15
+ # — without touching the 131 `parse` methods.
16
+ #
17
+ # The name is `Grammar`, not `Base`, because {Pubid::Parsers::Base} already
18
+ # exists and means something unrelated (the MR-string parser).
19
+ #
20
+ # Known limit: a rule atom parsed directly, `Parser.new.<rule>.parse(str)`,
21
+ # bypasses this and raises a bare `Parslet::ParseFailed`. Nothing in the
22
+ # gem does that.
23
+ class Grammar < ::Parslet::Parser
24
+ # @param io [String, IO]
25
+ # @param options [Hash] passed through to parslet
26
+ # @raise [Pubid::Errors::ParseError]
27
+ def parse(io, options = {})
28
+ super
29
+ rescue ::Pubid::Errors::ParseError
30
+ # A nested grammar already wrapped it. Keep the inner flavor and input.
31
+ raise
32
+ rescue ::Parslet::ParseFailed => e
33
+ raise wrap_parse_failure(e, io)
34
+ end
35
+
36
+ private
37
+
38
+ # @param error [Parslet::ParseFailed]
39
+ # @param io [String, IO] what was handed to {#parse}
40
+ # @return [Pubid::Errors::ParseError]
41
+ def wrap_parse_failure(error, io)
42
+ ::Pubid::Errors::ParseError.new(
43
+ error.message,
44
+ error.parse_failure_cause,
45
+ input: io.is_a?(String) ? io : nil,
46
+ flavor: pubid_flavor_name,
47
+ )
48
+ end
49
+
50
+ # "Pubid::Iso::Parser" -> "iso"; "Pubid::Ieee::Aiee::Parser" -> "ieee".
51
+ #
52
+ # Resolved through {Pubid::Registry} rather than taken from the module
53
+ # name, because the two disagree: `Pubid::Tgpp` registers as `"3gpp"`,
54
+ # and `Pubid::CenCenelec` registers twice (`"cen_cenelec"` first, then
55
+ # the `"cen"` alias). Reporting the registered name is what lets a caller
56
+ # feed `error.flavor` straight back to `Pubid::Registry.get`.
57
+ # @return [String, nil]
58
+ def pubid_flavor_name
59
+ parts = self.class.name.to_s.split("::")
60
+ return nil unless parts[0] == "Pubid" && parts.length > 1
61
+
62
+ registered_flavor_name(parts[1]) || parts[1].downcase
63
+ end
64
+
65
+ # @param module_name [String] e.g. "Tgpp"
66
+ # @return [String, nil] the registered name, nil if not registered
67
+ def registered_flavor_name(module_name)
68
+ ::Pubid::Registry.canonical_name(::Pubid.const_get(module_name))
69
+ rescue ::NameError
70
+ nil
71
+ end
72
+ end
73
+ end
74
+ end
data/lib/pubid/parser.rb CHANGED
@@ -1,7 +1,9 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require "pubid"
3
4
  module Pubid
4
5
  module Parser
6
+ autoload :Grammar, "pubid/parser/grammar"
5
7
  autoload :CommonParseMethods, "pubid/parser/common_parse_methods"
6
8
  autoload :CommonParseRules, "pubid/parser/common_parse_rules"
7
9
  end
@@ -101,6 +101,25 @@ module Pubid
101
101
  raise ArgumentError, "Unknown flavor: #{flavor}" unless flavor_module
102
102
 
103
103
  identifier_string = convert_to_human_readable(mr_string)
104
+
105
+ # Refuse to hand the flavor a string this parser did not change.
106
+ #
107
+ # `detect_flavor` falls back to `:iso` for any publisher it does not
108
+ # know, and `Pubid::Iso.parse` sends an MR-shaped string straight back
109
+ # here — so a string that survives conversion unchanged recurses
110
+ # forever. `ECMA-426 ed1` is the standing example: "ECMA" is absent
111
+ # from FLAVOR_MAP, the string matches the MR shape heuristic
112
+ # (`/\A[A-Z]{2,}[.-]/`), and conversion is a fixed point.
113
+ # `Pubid::Iso.parse("ECMA-426 ed1")` raised SystemStackError.
114
+ #
115
+ # A fixed point means "this was never an MR string", which is a parse
116
+ # failure — the class the cross-flavor contract requires, and the one
117
+ # `Pubid.parse` treats as "try the next flavor".
118
+ if identifier_string == mr_string && flavor_module == Pubid::Iso
119
+ raise Parslet::ParseFailed,
120
+ "#{mr_string.inspect} is not an MR string"
121
+ end
122
+
104
123
  flavor_module.parse(identifier_string)
105
124
  end
106
125
 
@@ -2,7 +2,7 @@
2
2
 
3
3
  module Pubid
4
4
  module Plateau
5
- class Parser < Parslet::Parser
5
+ class Parser < ::Pubid::Parser::Grammar
6
6
  rule(:space) { str(" ") }
7
7
  rule(:dash) { str("-") }
8
8
  rule(:hash) { str("#") }
data/lib/pubid/plateau.rb CHANGED
@@ -1,5 +1,6 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require "pubid"
3
4
  require "parslet"
4
5
 
5
6
  module Pubid
@@ -19,6 +20,15 @@ module Pubid
19
20
  autoload :UrnParser, "#{__dir__}/plateau/urn_parser"
20
21
 
21
22
  def self.parse(input)
23
+ unless input.is_a?(String)
24
+ raise Pubid::Errors::InvalidInputError,
25
+ Pubid::INPUT_NOT_A_STRING_MESSAGE
26
+ end
27
+
28
+ if input.length > Pubid::MAX_INPUT_LENGTH
29
+ raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
30
+ end
31
+
22
32
  # Apply legacy update_codes normalization first
23
33
  normalized = Core::UpdateCodes.apply(input, :plateau)
24
34
  parser = Parser.new
@@ -1,5 +1,6 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require "pubid"
3
4
  module Pubid
4
5
  # Mixin providing the uniform, static +prefixes+ class method that every
5
6
  # registered flavor exposes. relaton uses this to build a global prefix
@@ -0,0 +1,233 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Pubid
4
+ module Renderers
5
+ # Adds semantic <span class="..."> markers to an already-rendered
6
+ # identifier string.
7
+ #
8
+ # In pubid 1.x annotation was applied in ONE place — `prerender_params`
9
+ # wrapped every value in the param hash — so all 40-odd flavors got it for
10
+ # free. pubid 2 replaced that hash with typed components and per-flavor
11
+ # renderers, and the annotation went with it: `Renderers::Base#annotate`
12
+ # exists, but only the HumanReadable family calls it. All 39 flavor
13
+ # renderers receive `context.annotated` and ignore it, so
14
+ # `to_s(annotated: true)` returned plain text for every flavor but ISO.
15
+ #
16
+ # Re-annotating inside 39 renderers would mean 39 chances to drift. This
17
+ # class recovers the markup from the OUTSIDE instead: it asks the
18
+ # identifier for its own component values and wraps each one where it
19
+ # appears in the rendered string. That is one implementation, and a flavor
20
+ # added tomorrow is covered without touching its renderer.
21
+ #
22
+ # It is a fallback, not a replacement. `Identifier#render` uses it only
23
+ # when the renderer produced no span of its own, so ISO's exact,
24
+ # render-time placement still wins.
25
+ #
26
+ # Deliberate limits:
27
+ #
28
+ # * A token that does not appear verbatim in the output is skipped. A
29
+ # renderer may transform a value (abbreviate it, change its case), and a
30
+ # missing span is a far better outcome than a wrong one or a crash.
31
+ # * Matching walks left to right behind a cursor, so a later token can
32
+ # never match text an earlier one already claimed — that is what stops
33
+ # the part "1" of `ISO 1234-1` from matching inside "1234".
34
+ class Annotator
35
+ # Semantic class for each token, in the order tokens appear in a printed
36
+ # identifier. Order matters: it is the sequence the cursor walks.
37
+ TOKENS = [
38
+ [:publisher, "publisher"],
39
+ [:copublishers, "publisher"],
40
+ %i[typed_stage typed_stage_css],
41
+ [:type, "doctype"],
42
+ [:stage, "stage"],
43
+ [:number, "docnumber"],
44
+ [:part, "part"],
45
+ [:subpart, "part"],
46
+ [:stage_iteration, "iteration"],
47
+ [:year, "year"],
48
+ [:edition, "edition"],
49
+ [:languages, "language"],
50
+ ].freeze
51
+
52
+ # Characters that may not sit directly against a match, so a token never
53
+ # binds to the middle of a longer run of the same character class.
54
+ WORD_CHAR = /[A-Za-z0-9]/
55
+
56
+ # How deep to follow nested identifiers. A wrapper around a wrapper is
57
+ # real (BSI adopts a CEN prestandard that adopts an ISO standard); four
58
+ # levels is past anything the corpus holds, and the cap is here so a
59
+ # cyclic `base` cannot hang a rendering call.
60
+ MAX_NESTING = 4
61
+
62
+ def initialize(identifier, context = nil)
63
+ @id = identifier
64
+ @context = context
65
+ end
66
+
67
+ # @param rendered [String] the plain rendering of {@id}
68
+ # @return [String] the same string with semantic spans inserted
69
+ def annotate(rendered)
70
+ return rendered unless rendered.is_a?(String) && !rendered.empty?
71
+
72
+ cursor = 0
73
+ result = +""
74
+
75
+ ordered_tokens(rendered).each do |text, css_class|
76
+ index = find_token(rendered, text, cursor)
77
+ next if index.nil?
78
+
79
+ result << rendered[cursor...index]
80
+ result << %(<span class="#{css_class}">#{text}</span>)
81
+ cursor = index + text.length
82
+ end
83
+
84
+ result << rendered[cursor..]
85
+ result
86
+ end
87
+
88
+ private
89
+
90
+ # Tokens sorted by where they actually appear, not by the order
91
+ # {TOKENS} lists them.
92
+ #
93
+ # Flavors disagree about layout: ISO prints the number before the type
94
+ # ("ISO/IEC TR 2131"), CCSDS prints it after ("CCSDS 121.0-B-2"). Walking
95
+ # in declaration order pushed the cursor past the number for CCSDS, so
96
+ # "121" was never annotated. Sorting by first occurrence makes the walk
97
+ # follow the printed identifier instead of a fixed idea of one.
98
+ def ordered_tokens(rendered)
99
+ tokens = []
100
+ each_token { |text, css_class| tokens << [text, css_class] }
101
+
102
+ decorated = tokens.each_with_index.map do |token, i|
103
+ position = find_token(rendered, token.first, 0) || rendered.length
104
+ [position, i, token]
105
+ end
106
+
107
+ decorated.sort_by { |position, i, _| [position, i] }.map(&:last)
108
+ end
109
+
110
+ # Yields [text, css_class] for every annotatable token this identifier
111
+ # actually carries, then every one its nested identifiers carry.
112
+ def each_token(&)
113
+ emit_tokens(@id, 0, &)
114
+ end
115
+
116
+ # A wrapper — an adoption, a supplement, a bundle — carries no number of
117
+ # its own and prints the document it wraps, so the annotatable tokens in
118
+ # its string belong to `base`. Reading only the wrapper's own attributes
119
+ # is why every CSA container and CIE's supplement rendered plain while
120
+ # accepting the flag.
121
+ #
122
+ # Order does not matter here: `ordered_tokens` sorts by first occurrence,
123
+ # so a nested token lands where it actually appears in the string.
124
+ def emit_tokens(id, depth, &)
125
+ return if depth > MAX_NESTING
126
+
127
+ tokens_for(id).each do |attr_name, css_class|
128
+ Array(token_values(id, attr_name)).each do |value|
129
+ text = token_text(value)
130
+ next if text.nil? || text.empty?
131
+
132
+ yield text, resolve_class(css_class, value)
133
+ end
134
+ end
135
+
136
+ nested_identifiers(id).each { |nested| emit_tokens(nested, depth + 1, &) }
137
+ end
138
+
139
+ # {TOKENS} unless the identifier names more of its own — see
140
+ # `Pubid::Identifier#annotation_tokens`.
141
+ def tokens_for(id)
142
+ id.respond_to?(:annotation_tokens) ? id.annotation_tokens : TOKENS
143
+ end
144
+
145
+ # The identifiers this one wraps, under any of the four names the gem
146
+ # uses: `base` (the uniform parent accessor), the `ids` / `identifiers`
147
+ # collections that bundles and consolidated identifiers hold instead, and
148
+ # CSA `Bundled`'s `bundled_with` — which holds the amendments a
149
+ # consolidation prints, and whose values it composes into the string from
150
+ # their components rather than from their own `to_s`.
151
+ def nested_identifiers(id)
152
+ %i[base ids identifiers bundled_with].flat_map do |name|
153
+ next [] unless id.respond_to?(name)
154
+
155
+ begin
156
+ Array(id.public_send(name))
157
+ rescue StandardError
158
+ []
159
+ end
160
+ end.grep(::Pubid::Identifier)
161
+ end
162
+
163
+ def token_values(id, attr_name)
164
+ return nil unless id.respond_to?(attr_name)
165
+
166
+ id.public_send(attr_name)
167
+ rescue StandardError
168
+ # A derived reader may assume state a partial identifier lacks. A
169
+ # missing span is not worth an exception on a rendering path.
170
+ nil
171
+ end
172
+
173
+ # The printed form of one component. Components render themselves through
174
+ # the context; a bare scalar is already its own text.
175
+ def token_text(value)
176
+ return nil if value.nil?
177
+
178
+ text = if value.respond_to?(:render)
179
+ value.render(context: @context)
180
+ else
181
+ value
182
+ end
183
+ text.to_s.strip
184
+ rescue StandardError
185
+ nil
186
+ end
187
+
188
+ def resolve_class(css_class, value)
189
+ return css_class unless css_class == :typed_stage_css
190
+
191
+ TypedStageClass.for(value)
192
+ end
193
+
194
+ # First occurrence of +text+ at or after +cursor+ that is not embedded in
195
+ # a longer word — so "1" matches the part in "ISO 1234-1", never the "1"
196
+ # inside "1234".
197
+ def find_token(rendered, text, cursor)
198
+ at = cursor
199
+ while (index = rendered.index(text, at))
200
+ return index if standalone?(rendered, index, text.length)
201
+
202
+ at = index + 1
203
+ end
204
+ nil
205
+ end
206
+
207
+ def standalone?(rendered, index, length)
208
+ before = index.zero? ? nil : rendered[index - 1]
209
+ after = rendered[index + length]
210
+
211
+ !WORD_CHAR.match?(before.to_s) && !WORD_CHAR.match?(after.to_s)
212
+ end
213
+
214
+ # The typed-stage class depends on the stage's type code, the same
215
+ # mapping `Renderers::Base#typed_stage_css` applies. Kept here rather
216
+ # than reached for through a private method on another object.
217
+ module TypedStageClass
218
+ MAP = {
219
+ "amd" => "amendment",
220
+ "cor" => "corrigendum",
221
+ "add" => "addendum",
222
+ }.freeze
223
+
224
+ def self.for(typed_stage)
225
+ code = typed_stage.respond_to?(:type_code) ? typed_stage.type_code.to_s : ""
226
+ return "stage" if code.empty? || code == "is"
227
+
228
+ MAP[code] || "doctype"
229
+ end
230
+ end
231
+ end
232
+ end
233
+ end
@@ -41,6 +41,19 @@ module Pubid
41
41
  %(#{lead}<span class="#{css_class}">#{core}</span>#{trail})
42
42
  end
43
43
 
44
+ # Render a value that may be a component or a bare scalar.
45
+ #
46
+ # `number`, `part` and `subpart` are `Components::Code` on
47
+ # ::Pubid::Identifier but a plain `:string` in a growing number of
48
+ # flavors, so a shared renderer cannot assume the format-aware
49
+ # `#render(context:)` seam is there. This keeps both shapes working while
50
+ # the flavors convert one tranche at a time.
51
+ def render_component(value, context)
52
+ return nil if value.nil?
53
+
54
+ value.respond_to?(:render) ? value.render(context: context) : value.to_s
55
+ end
56
+
44
57
  # Choose between "stage" and a type/supplement class for a typed stage.
45
58
  def typed_stage_css(typed_stage)
46
59
  code = typed_stage&.type_code.to_s
@@ -50,15 +50,15 @@ module Pubid
50
50
  ann = context.annotated
51
51
  parts = []
52
52
  if @id.number
53
- parts << annotate(@id.number.render(context:), "docnumber",
53
+ parts << annotate(render_component(@id.number, context), "docnumber",
54
54
  annotated: ann)
55
55
  end
56
56
  if @id.part
57
- parts << " #{annotate(@id.part.render(context:), 'part',
57
+ parts << " #{annotate(render_component(@id.part, context), 'part',
58
58
  annotated: ann)}"
59
59
  end
60
60
  if @id.subpart
61
- parts << "-#{annotate(@id.subpart.render(context:), 'part',
61
+ parts << "-#{annotate(render_component(@id.subpart, context), 'part',
62
62
  annotated: ann)}"
63
63
  end
64
64
  if @id.stage_iteration
@@ -41,15 +41,15 @@ module Pubid
41
41
  ann = context.annotated
42
42
  parts = []
43
43
  if @id.number
44
- parts << annotate(@id.number.render(context:), "docnumber",
44
+ parts << annotate(render_component(@id.number, context), "docnumber",
45
45
  annotated: ann)
46
46
  end
47
47
  if @id.part
48
- parts << "-#{annotate(@id.part.render(context:), 'part',
48
+ parts << "-#{annotate(render_component(@id.part, context), 'part',
49
49
  annotated: ann)}"
50
50
  end
51
51
  if @id.subpart
52
- parts << "-#{annotate(@id.subpart.render(context:), 'part',
52
+ parts << "-#{annotate(render_component(@id.subpart, context), 'part',
53
53
  annotated: ann)}"
54
54
  end
55
55
  if @id.stage_iteration