pubid 2.0.0.pre.alpha.8 → 2.0.0.pre.alpha.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (484) hide show
  1. checksums.yaml +4 -4
  2. data/data/bipm/update_codes.yaml +9 -0
  3. data/data/iec/update_codes.yaml +3 -0
  4. data/data/ieee/update_codes.yaml +107 -1
  5. data/data/nist/series.yaml +155 -0
  6. data/data/nist/update_codes.yaml +8 -0
  7. data/lib/pubid/adobe/builder.rb +52 -0
  8. data/lib/pubid/adobe/identifier.rb +49 -0
  9. data/lib/pubid/adobe/identifiers/publication.rb +31 -0
  10. data/lib/pubid/adobe/identifiers/tech_note.rb +32 -0
  11. data/lib/pubid/adobe/identifiers.rb +10 -0
  12. data/lib/pubid/adobe/parser.rb +128 -0
  13. data/lib/pubid/adobe/renderer.rb +39 -0
  14. data/lib/pubid/adobe/urn_generator.rb +42 -0
  15. data/lib/pubid/adobe/urn_parser.rb +69 -0
  16. data/lib/pubid/adobe.rb +60 -0
  17. data/lib/pubid/amca/builder.rb +2 -2
  18. data/lib/pubid/amca/identifier.rb +1 -3
  19. data/lib/pubid/amca/identifiers/base.rb +1 -6
  20. data/lib/pubid/amca/identifiers/interpretation.rb +1 -1
  21. data/lib/pubid/amca/identifiers/publication.rb +1 -1
  22. data/lib/pubid/amca/identifiers/standard.rb +1 -1
  23. data/lib/pubid/amca/identifiers.rb +0 -1
  24. data/lib/pubid/amca/single_identifier.rb +1 -1
  25. data/lib/pubid/amca.rb +3 -3
  26. data/lib/pubid/ashrae/builder.rb +22 -22
  27. data/lib/pubid/ashrae/identifier.rb +3 -4
  28. data/lib/pubid/ashrae/identifiers/addenda_package.rb +1 -1
  29. data/lib/pubid/ashrae/identifiers/addendum.rb +1 -1
  30. data/lib/pubid/ashrae/identifiers/base.rb +14 -8
  31. data/lib/pubid/ashrae/identifiers/combined_addenda.rb +1 -1
  32. data/lib/pubid/ashrae/identifiers.rb +0 -1
  33. data/lib/pubid/ashrae/parser.rb +14 -9
  34. data/lib/pubid/ashrae/renderer.rb +19 -19
  35. data/lib/pubid/ashrae/single_identifier.rb +1 -1
  36. data/lib/pubid/ashrae/supplement_identifier.rb +6 -6
  37. data/lib/pubid/ashrae.rb +3 -3
  38. data/lib/pubid/bipm/builder.rb +107 -0
  39. data/lib/pubid/bipm/identifier.rb +160 -0
  40. data/lib/pubid/bipm/identifiers/committee_document.rb +24 -0
  41. data/lib/pubid/bipm/identifiers/guide.rb +24 -0
  42. data/lib/pubid/bipm/identifiers/meeting.rb +41 -0
  43. data/lib/pubid/bipm/identifiers/mep.rb +21 -0
  44. data/lib/pubid/bipm/identifiers/metrologia_article.rb +24 -0
  45. data/lib/pubid/bipm/identifiers/si_brochure.rb +20 -0
  46. data/lib/pubid/bipm/identifiers.rb +15 -0
  47. data/lib/pubid/bipm/parser.rb +184 -0
  48. data/lib/pubid/bipm/renderer.rb +110 -0
  49. data/lib/pubid/bipm/urn_generator.rb +58 -0
  50. data/lib/pubid/bipm/urn_parser.rb +52 -0
  51. data/lib/pubid/bipm.rb +86 -0
  52. data/lib/pubid/bsi/builder.rb +14 -14
  53. data/lib/pubid/bsi/identifier.rb +0 -1
  54. data/lib/pubid/bsi/identifiers/addendum_document.rb +2 -2
  55. data/lib/pubid/bsi/identifiers/amendment.rb +4 -4
  56. data/lib/pubid/bsi/identifiers/consolidated_identifier.rb +6 -0
  57. data/lib/pubid/bsi/identifiers/corrigendum.rb +4 -4
  58. data/lib/pubid/bsi/identifiers/expert_commentary.rb +5 -5
  59. data/lib/pubid/bsi/identifiers/supplement_document.rb +2 -2
  60. data/lib/pubid/bsi/identifiers/value_added_publication.rb +6 -6
  61. data/lib/pubid/bsi/renderer.rb +8 -8
  62. data/lib/pubid/bsi/single_identifier.rb +48 -0
  63. data/lib/pubid/builder/base.rb +5 -3
  64. data/lib/pubid/bundled_identifier.rb +3 -3
  65. data/lib/pubid/calconnect/builder.rb +38 -0
  66. data/lib/pubid/calconnect/identifier.rb +117 -0
  67. data/lib/pubid/calconnect/identifiers/standard.rb +29 -0
  68. data/lib/pubid/calconnect/identifiers.rb +9 -0
  69. data/lib/pubid/calconnect/parser.rb +54 -0
  70. data/lib/pubid/calconnect/renderer.rb +36 -0
  71. data/lib/pubid/calconnect/urn_generator.rb +33 -0
  72. data/lib/pubid/calconnect/urn_parser.rb +34 -0
  73. data/lib/pubid/calconnect.rb +77 -0
  74. data/lib/pubid/ccsds/builder.rb +1 -1
  75. data/lib/pubid/ccsds/identifier.rb +1 -1
  76. data/lib/pubid/ccsds/identifiers/corrigendum.rb +2 -2
  77. data/lib/pubid/ccsds/supplement_identifier.rb +6 -6
  78. data/lib/pubid/ccsds/urn_generator.rb +3 -3
  79. data/lib/pubid/cen_cenelec/builder.rb +46 -12
  80. data/lib/pubid/cen_cenelec/identifiers/amendment.rb +2 -2
  81. data/lib/pubid/cen_cenelec/identifiers/consolidated_identifier.rb +6 -0
  82. data/lib/pubid/cen_cenelec/identifiers/corrigendum.rb +2 -2
  83. data/lib/pubid/cen_cenelec/identifiers/european_norm.rb +60 -3
  84. data/lib/pubid/cen_cenelec/identifiers/fragment.rb +2 -2
  85. data/lib/pubid/cen_cenelec/renderer.rb +5 -5
  86. data/lib/pubid/cen_cenelec/supplement_identifier.rb +5 -5
  87. data/lib/pubid/cen_cenelec/urn_generator.rb +1 -1
  88. data/lib/pubid/cen_cenelec.rb +76 -72
  89. data/lib/pubid/cie/builder.rb +139 -24
  90. data/lib/pubid/cie/identifier.rb +7 -3
  91. data/lib/pubid/cie/identifiers/bundle.rb +32 -4
  92. data/lib/pubid/cie/identifiers/code_attributes.rb +52 -0
  93. data/lib/pubid/cie/identifiers/conference.rb +9 -4
  94. data/lib/pubid/cie/identifiers/corrigendum.rb +17 -23
  95. data/lib/pubid/cie/identifiers/dual_published.rb +4 -4
  96. data/lib/pubid/cie/identifiers/identical.rb +7 -7
  97. data/lib/pubid/cie/identifiers/joint_published.rb +8 -8
  98. data/lib/pubid/cie/identifiers/proceedings.rb +39 -0
  99. data/lib/pubid/cie/identifiers/standard.rb +33 -7
  100. data/lib/pubid/cie/identifiers/supplement.rb +24 -22
  101. data/lib/pubid/cie/identifiers/tutorial_bundle.rb +6 -2
  102. data/lib/pubid/cie/identifiers.rb +2 -0
  103. data/lib/pubid/cie/parser.rb +54 -5
  104. data/lib/pubid/cie/single_identifier.rb +40 -1
  105. data/lib/pubid/cie/supplement_identifier.rb +29 -16
  106. data/lib/pubid/cie/urn_generator.rb +11 -8
  107. data/lib/pubid/cie.rb +0 -1
  108. data/lib/pubid/components/date.rb +21 -5
  109. data/lib/pubid/csa/builder.rb +1 -1
  110. data/lib/pubid/csa/composite_identifier.rb +17 -1
  111. data/lib/pubid/csa/identifier.rb +48 -3
  112. data/lib/pubid/csa/identifiers/bundled.rb +6 -0
  113. data/lib/pubid/csa/identifiers/combined.rb +6 -0
  114. data/lib/pubid/csa/identifiers/package.rb +2 -2
  115. data/lib/pubid/csa/parser.rb +6 -6
  116. data/lib/pubid/csa/wrapper_identifier.rb +17 -0
  117. data/lib/pubid/doi/builder.rb +18 -0
  118. data/lib/pubid/doi/identifier.rb +46 -0
  119. data/lib/pubid/doi/identifiers/resource.rb +30 -0
  120. data/lib/pubid/doi/identifiers.rb +9 -0
  121. data/lib/pubid/doi/parser.rb +52 -0
  122. data/lib/pubid/doi/renderer.rb +13 -0
  123. data/lib/pubid/doi.rb +55 -0
  124. data/lib/pubid/easc/builder.rb +55 -0
  125. data/lib/pubid/easc/identifier.rb +61 -0
  126. data/lib/pubid/easc/identifiers/pmg.rb +20 -0
  127. data/lib/pubid/easc/identifiers/rmg.rb +19 -0
  128. data/lib/pubid/easc/identifiers.rb +10 -0
  129. data/lib/pubid/easc/parser.rb +80 -0
  130. data/lib/pubid/easc/renderer.rb +48 -0
  131. data/lib/pubid/easc/urn_generator.rb +46 -0
  132. data/lib/pubid/easc/urn_parser.rb +57 -0
  133. data/lib/pubid/easc.rb +52 -0
  134. data/lib/pubid/ecma/builder.rb +31 -0
  135. data/lib/pubid/ecma/identifier.rb +76 -0
  136. data/lib/pubid/ecma/identifiers/memento.rb +30 -0
  137. data/lib/pubid/ecma/identifiers/standard.rb +31 -0
  138. data/lib/pubid/ecma/identifiers/technical_report.rb +31 -0
  139. data/lib/pubid/ecma/identifiers.rb +11 -0
  140. data/lib/pubid/ecma/parser.rb +41 -0
  141. data/lib/pubid/ecma/renderer.rb +46 -0
  142. data/lib/pubid/ecma/urn_generator.rb +25 -0
  143. data/lib/pubid/ecma/urn_parser.rb +35 -0
  144. data/lib/pubid/ecma.rb +67 -0
  145. data/lib/pubid/etsi/builder.rb +7 -3
  146. data/lib/pubid/etsi/identifier.rb +1 -3
  147. data/lib/pubid/etsi/identifiers/base.rb +28 -10
  148. data/lib/pubid/etsi/identifiers/etsi_standard.rb +128 -4
  149. data/lib/pubid/etsi/identifiers/supplement_identifier.rb +35 -3
  150. data/lib/pubid/etsi/identifiers.rb +0 -1
  151. data/lib/pubid/etsi/parser.rb +8 -2
  152. data/lib/pubid/etsi/renderer.rb +8 -2
  153. data/lib/pubid/etsi.rb +5 -5
  154. data/lib/pubid/export/exporter.rb +2 -1
  155. data/lib/pubid/export/flavor_exporter.rb +18 -3
  156. data/lib/pubid/gb/builder.rb +45 -0
  157. data/lib/pubid/gb/identifier.rb +64 -0
  158. data/lib/pubid/gb/identifiers/standard.rb +34 -0
  159. data/lib/pubid/gb/identifiers.rb +9 -0
  160. data/lib/pubid/gb/parser.rb +63 -0
  161. data/lib/pubid/gb/renderer.rb +35 -0
  162. data/lib/pubid/gb.rb +87 -0
  163. data/lib/pubid/gost/builder.rb +148 -0
  164. data/lib/pubid/gost/identifier.rb +49 -0
  165. data/lib/pubid/gost/identifiers/foreign_reference.rb +25 -0
  166. data/lib/pubid/gost/identifiers/harmonized.rb +39 -0
  167. data/lib/pubid/gost/identifiers/identical_adoption.rb +39 -0
  168. data/lib/pubid/gost/identifiers/interstate_standard.rb +20 -0
  169. data/lib/pubid/gost/identifiers/national_standard.rb +19 -0
  170. data/lib/pubid/gost/identifiers.rb +13 -0
  171. data/lib/pubid/gost/parser.rb +97 -0
  172. data/lib/pubid/gost/renderer.rb +48 -0
  173. data/lib/pubid/gost/urn_generator.rb +46 -0
  174. data/lib/pubid/gost/urn_parser.rb +52 -0
  175. data/lib/pubid/gost.rb +49 -0
  176. data/lib/pubid/iala/builder.rb +22 -27
  177. data/lib/pubid/iala/identifier.rb +27 -2
  178. data/lib/pubid/iala/identifiers/advice.rb +1 -1
  179. data/lib/pubid/iala/identifiers/annex.rb +10 -10
  180. data/lib/pubid/iala/identifiers/general_assembly.rb +5 -2
  181. data/lib/pubid/iala/identifiers/guideline.rb +3 -1
  182. data/lib/pubid/iala/identifiers/letter.rb +1 -1
  183. data/lib/pubid/iala/identifiers/manual.rb +3 -1
  184. data/lib/pubid/iala/identifiers/model_course.rb +3 -1
  185. data/lib/pubid/iala/identifiers/recommendation.rb +3 -1
  186. data/lib/pubid/iala/identifiers/report.rb +1 -1
  187. data/lib/pubid/iala/identifiers/resolution.rb +1 -1
  188. data/lib/pubid/iala/identifiers/standard.rb +3 -1
  189. data/lib/pubid/iala/identifiers.rb +0 -1
  190. data/lib/pubid/iala/renderer.rb +1 -1
  191. data/lib/pubid/iala/urn_generator.rb +1 -1
  192. data/lib/pubid/iala/urn_parser.rb +19 -2
  193. data/lib/pubid/iala.rb +4 -4
  194. data/lib/pubid/iana/builder.rb +20 -0
  195. data/lib/pubid/iana/identifier.rb +72 -0
  196. data/lib/pubid/iana/identifiers/registry.rb +33 -0
  197. data/lib/pubid/iana/identifiers.rb +9 -0
  198. data/lib/pubid/iana/parser.rb +38 -0
  199. data/lib/pubid/iana/renderer.rb +29 -0
  200. data/lib/pubid/iana/urn_generator.rb +15 -0
  201. data/lib/pubid/iana/urn_parser.rb +24 -0
  202. data/lib/pubid/iana.rb +69 -0
  203. data/lib/pubid/identifier.rb +210 -23
  204. data/lib/pubid/idf/builder.rb +1 -1
  205. data/lib/pubid/idf/parser.rb +2 -2
  206. data/lib/pubid/idf/renderer.rb +1 -1
  207. data/lib/pubid/idf/supplement_identifier.rb +3 -3
  208. data/lib/pubid/iec/builder.rb +35 -25
  209. data/lib/pubid/iec/identifier.rb +47 -2
  210. data/lib/pubid/iec/identifiers/base.rb +2 -2
  211. data/lib/pubid/iec/identifiers/consolidated_identifier.rb +6 -0
  212. data/lib/pubid/iec/identifiers/fragment_identifier.rb +11 -11
  213. data/lib/pubid/iec/identifiers/international_standard.rb +8 -5
  214. data/lib/pubid/iec/identifiers/sheet_identifier.rb +12 -12
  215. data/lib/pubid/iec/identifiers/technical_group.rb +30 -0
  216. data/lib/pubid/iec/identifiers/technical_report.rb +1 -1
  217. data/lib/pubid/iec/identifiers/technical_specification.rb +1 -1
  218. data/lib/pubid/iec/identifiers/vap_identifier.rb +12 -12
  219. data/lib/pubid/iec/identifiers.rb +1 -0
  220. data/lib/pubid/iec/parser.rb +51 -10
  221. data/lib/pubid/iec/renderer.rb +18 -8
  222. data/lib/pubid/iec/single_identifier.rb +1 -1
  223. data/lib/pubid/iec/supplement_identifier.rb +32 -19
  224. data/lib/pubid/iec.rb +1 -6
  225. data/lib/pubid/ieee/aiee/builder.rb +17 -3
  226. data/lib/pubid/ieee/aiee/identifier.rb +45 -61
  227. data/lib/pubid/ieee/builder.rb +341 -68
  228. data/lib/pubid/ieee/compaction.rb +121 -0
  229. data/lib/pubid/ieee/components/draft.rb +47 -7
  230. data/lib/pubid/ieee/identifier.rb +1 -3
  231. data/lib/pubid/ieee/identifiers/adopted_standard.rb +30 -5
  232. data/lib/pubid/ieee/identifiers/amendment.rb +40 -0
  233. data/lib/pubid/ieee/identifiers/base.rb +106 -13
  234. data/lib/pubid/ieee/identifiers/code_number.rb +82 -0
  235. data/lib/pubid/ieee/identifiers/conformance_identifier.rb +6 -2
  236. data/lib/pubid/ieee/identifiers/corrigendum.rb +17 -2
  237. data/lib/pubid/ieee/identifiers/csa_dual_published.rb +3 -3
  238. data/lib/pubid/ieee/identifiers/dual_identifier.rb +3 -3
  239. data/lib/pubid/ieee/identifiers/dual_published.rb +3 -3
  240. data/lib/pubid/ieee/identifiers/iec_ieee_copublished.rb +102 -5
  241. data/lib/pubid/ieee/identifiers/interpretation_identifier.rb +2 -1
  242. data/lib/pubid/ieee/identifiers/joint_development.rb +56 -11
  243. data/lib/pubid/ieee/identifiers/multi_numbered_identifier.rb +1 -1
  244. data/lib/pubid/ieee/identifiers/nesc/base.rb +90 -19
  245. data/lib/pubid/ieee/identifiers/nesc/draft.rb +20 -5
  246. data/lib/pubid/ieee/identifiers/nesc/edition.rb +32 -0
  247. data/lib/pubid/ieee/identifiers/nesc/handbook.rb +16 -5
  248. data/lib/pubid/ieee/identifiers/nesc/redline.rb +11 -2
  249. data/lib/pubid/ieee/identifiers/nesc/standard.rb +18 -3
  250. data/lib/pubid/ieee/identifiers/nesc.rb +1 -0
  251. data/lib/pubid/ieee/identifiers/parenthetical_identifier.rb +3 -3
  252. data/lib/pubid/ieee/identifiers/project_draft_identifier.rb +14 -6
  253. data/lib/pubid/ieee/identifiers/redlined_standard.rb +4 -4
  254. data/lib/pubid/ieee/identifiers/si_standard.rb +4 -1
  255. data/lib/pubid/ieee/identifiers/standard.rb +4 -1
  256. data/lib/pubid/ieee/identifiers/supplement_identifier.rb +26 -7
  257. data/lib/pubid/ieee/identifiers.rb +2 -1
  258. data/lib/pubid/ieee/ire/builder.rb +7 -6
  259. data/lib/pubid/ieee/ire/identifier.rb +48 -33
  260. data/lib/pubid/ieee/nesc/builder.rb +33 -20
  261. data/lib/pubid/ieee/nesc/parser.rb +15 -14
  262. data/lib/pubid/ieee/parser.rb +461 -43
  263. data/lib/pubid/ieee/pre_parser.rb +1 -1
  264. data/lib/pubid/ieee/renderer.rb +154 -57
  265. data/lib/pubid/ieee/typed_stages.rb +12 -1
  266. data/lib/pubid/ieee/urn_generator.rb +3 -5
  267. data/lib/pubid/ieee.rb +56 -3
  268. data/lib/pubid/ietf/builder.rb +62 -0
  269. data/lib/pubid/ietf/identifier.rb +5 -0
  270. data/lib/pubid/ietf/identifiers/base.rb +75 -0
  271. data/lib/pubid/ietf/identifiers/bcp.rb +25 -0
  272. data/lib/pubid/ietf/identifiers/fyi.rb +25 -0
  273. data/lib/pubid/ietf/identifiers/internet_draft.rb +29 -0
  274. data/lib/pubid/ietf/identifiers/rfc.rb +25 -0
  275. data/lib/pubid/ietf/identifiers/std.rb +25 -0
  276. data/lib/pubid/ietf/identifiers.rb +13 -0
  277. data/lib/pubid/ietf/parser.rb +43 -0
  278. data/lib/pubid/ietf/renderer.rb +26 -0
  279. data/lib/pubid/ietf/urn_generator.rb +23 -0
  280. data/lib/pubid/ietf/urn_parser.rb +35 -0
  281. data/lib/pubid/ietf.rb +76 -0
  282. data/lib/pubid/iho/identifier.rb +1 -3
  283. data/lib/pubid/iho/identifiers/base.rb +1 -7
  284. data/lib/pubid/iho/identifiers/bibliographic.rb +1 -1
  285. data/lib/pubid/iho/identifiers/circular_letter.rb +1 -1
  286. data/lib/pubid/iho/identifiers/miscellaneous.rb +1 -1
  287. data/lib/pubid/iho/identifiers/publication.rb +1 -1
  288. data/lib/pubid/iho/identifiers/standard.rb +1 -1
  289. data/lib/pubid/iho/identifiers.rb +0 -1
  290. data/lib/pubid/iho.rb +7 -7
  291. data/lib/pubid/isbn/builder.rb +44 -0
  292. data/lib/pubid/isbn/check_digit.rb +53 -0
  293. data/lib/pubid/isbn/identifier.rb +63 -0
  294. data/lib/pubid/isbn/identifiers/book.rb +30 -0
  295. data/lib/pubid/isbn/identifiers.rb +9 -0
  296. data/lib/pubid/isbn/parser.rb +45 -0
  297. data/lib/pubid/isbn/renderer.rb +16 -0
  298. data/lib/pubid/isbn.rb +52 -0
  299. data/lib/pubid/iso/builder.rb +9 -4
  300. data/lib/pubid/iso/combined_identifier.rb +2 -2
  301. data/lib/pubid/iso/identifier.rb +17 -1
  302. data/lib/pubid/iso/identifiers/directives_supplement.rb +8 -8
  303. data/lib/pubid/iso/normalizer.rb +2 -2
  304. data/lib/pubid/iso/parser.rb +16 -12
  305. data/lib/pubid/iso/supplement_identifier.rb +19 -6
  306. data/lib/pubid/iso/urn_generator.rb +4 -1
  307. data/lib/pubid/iso/urn_parser.rb +1 -1
  308. data/lib/pubid/itu/builder.rb +230 -29
  309. data/lib/pubid/itu/components/code.rb +48 -5
  310. data/lib/pubid/itu/components/designation.rb +35 -0
  311. data/lib/pubid/itu/components.rb +1 -0
  312. data/lib/pubid/itu/identifier.rb +1 -3
  313. data/lib/pubid/itu/identifiers/addendum.rb +15 -0
  314. data/lib/pubid/itu/identifiers/amendment.rb +6 -28
  315. data/lib/pubid/itu/identifiers/annex.rb +13 -2
  316. data/lib/pubid/itu/identifiers/annex_of_recommendation.rb +87 -0
  317. data/lib/pubid/itu/identifiers/appendix_of_recommendation.rb +92 -0
  318. data/lib/pubid/itu/identifiers/base.rb +341 -23
  319. data/lib/pubid/itu/identifiers/combined_identifier.rb +99 -23
  320. data/lib/pubid/itu/identifiers/corrigendum.rb +16 -23
  321. data/lib/pubid/itu/identifiers/errata.rb +15 -0
  322. data/lib/pubid/itu/identifiers/handbook.rb +38 -0
  323. data/lib/pubid/itu/identifiers/question.rb +63 -0
  324. data/lib/pubid/itu/identifiers/recommendation.rb +3 -2
  325. data/lib/pubid/itu/identifiers/report.rb +39 -0
  326. data/lib/pubid/itu/identifiers/special_publication.rb +3 -1
  327. data/lib/pubid/itu/identifiers/standard_serialization.rb +58 -0
  328. data/lib/pubid/itu/identifiers/supplement.rb +119 -14
  329. data/lib/pubid/itu/identifiers.rb +10 -1
  330. data/lib/pubid/itu/model.rb +1 -1
  331. data/lib/pubid/itu/parser.rb +480 -26
  332. data/lib/pubid/itu/urn_generator.rb +36 -6
  333. data/lib/pubid/itu/urn_parser.rb +8 -0
  334. data/lib/pubid/itu.rb +11 -3
  335. data/lib/pubid/jcgm/builder.rb +23 -13
  336. data/lib/pubid/jcgm/identifiers/amendment.rb +0 -2
  337. data/lib/pubid/jcgm/identifiers/corrigendum.rb +34 -0
  338. data/lib/pubid/jcgm/identifiers/gum_guide.rb +3 -3
  339. data/lib/pubid/jcgm/identifiers/meeting.rb +47 -0
  340. data/lib/pubid/jcgm/identifiers.rb +2 -0
  341. data/lib/pubid/jcgm/parser.rb +66 -6
  342. data/lib/pubid/jcgm/renderer.rb +26 -12
  343. data/lib/pubid/jcgm/single_identifier.rb +74 -5
  344. data/lib/pubid/jcgm/supplement_identifier.rb +39 -4
  345. data/lib/pubid/jcgm/urn_generator.rb +21 -13
  346. data/lib/pubid/jcgm/urn_parser.rb +15 -3
  347. data/lib/pubid/jis/identifier.rb +52 -0
  348. data/lib/pubid/nist/builder.rb +29 -9
  349. data/lib/pubid/nist/caster.rb +7 -45
  350. data/lib/pubid/nist/circular_supplement_builder.rb +7 -5
  351. data/lib/pubid/nist/components/supplement.rb +5 -2
  352. data/lib/pubid/nist/configuration.rb +7 -1
  353. data/lib/pubid/nist/identifier.rb +1 -3
  354. data/lib/pubid/nist/identifiers/base.rb +121 -37
  355. data/lib/pubid/nist/identifiers/circular.rb +1 -1
  356. data/lib/pubid/nist/identifiers/commercial_standard.rb +1 -1
  357. data/lib/pubid/nist/identifiers/commercial_standard_emergency.rb +1 -1
  358. data/lib/pubid/nist/identifiers/commercial_standards_monthly.rb +1 -1
  359. data/lib/pubid/nist/identifiers/crpl_report.rb +1 -1
  360. data/lib/pubid/nist/identifiers/dated_document.rb +1 -1
  361. data/lib/pubid/nist/identifiers/federal_information_processing_standards.rb +1 -1
  362. data/lib/pubid/nist/identifiers/grant_contractor_report.rb +1 -1
  363. data/lib/pubid/nist/identifiers/handbook.rb +1 -1
  364. data/lib/pubid/nist/identifiers/internal_report.rb +1 -1
  365. data/lib/pubid/nist/identifiers/letter_circular.rb +1 -1
  366. data/lib/pubid/nist/identifiers/miscellaneous_publication.rb +1 -1
  367. data/lib/pubid/nist/identifiers/monograph.rb +1 -1
  368. data/lib/pubid/nist/identifiers/ncstar.rb +2 -2
  369. data/lib/pubid/nist/identifiers/nsrds.rb +1 -1
  370. data/lib/pubid/nist/identifiers/owmwp.rb +1 -1
  371. data/lib/pubid/nist/identifiers/report.rb +1 -1
  372. data/lib/pubid/nist/identifiers/special_publication.rb +10 -1
  373. data/lib/pubid/nist/identifiers/technical_note.rb +1 -1
  374. data/lib/pubid/nist/identifiers.rb +0 -1
  375. data/lib/pubid/nist/parser.rb +26 -3
  376. data/lib/pubid/nist/preprocessor.rb +15 -0
  377. data/lib/pubid/nist/router.rb +2 -2
  378. data/lib/pubid/nist/series/ir.rb +0 -3
  379. data/lib/pubid/nist/supplement_identifier.rb +9 -9
  380. data/lib/pubid/nist.rb +12 -12
  381. data/lib/pubid/oasis/builder.rb +77 -0
  382. data/lib/pubid/oasis/identifier.rb +82 -0
  383. data/lib/pubid/oasis/identifiers/standard.rb +29 -0
  384. data/lib/pubid/oasis/identifiers.rb +9 -0
  385. data/lib/pubid/oasis/parser.rb +34 -0
  386. data/lib/pubid/oasis/renderer.rb +29 -0
  387. data/lib/pubid/oasis/urn_generator.rb +25 -0
  388. data/lib/pubid/oasis/urn_parser.rb +22 -0
  389. data/lib/pubid/oasis.rb +74 -0
  390. data/lib/pubid/ogc/builder.rb +33 -0
  391. data/lib/pubid/ogc/identifier.rb +77 -0
  392. data/lib/pubid/ogc/identifiers/document.rb +26 -0
  393. data/lib/pubid/ogc/identifiers.rb +10 -0
  394. data/lib/pubid/ogc/parser.rb +40 -0
  395. data/lib/pubid/ogc/renderer.rb +33 -0
  396. data/lib/pubid/ogc/urn_generator.rb +14 -0
  397. data/lib/pubid/ogc/urn_parser.rb +22 -0
  398. data/lib/pubid/ogc.rb +71 -0
  399. data/lib/pubid/oiml/builder.rb +31 -7
  400. data/lib/pubid/oiml/identifier.rb +9 -1
  401. data/lib/pubid/oiml/parser.rb +11 -12
  402. data/lib/pubid/oiml/renderer.rb +13 -13
  403. data/lib/pubid/oiml/single_identifier.rb +20 -0
  404. data/lib/pubid/oiml/supplement_identifier.rb +9 -9
  405. data/lib/pubid/oiml/urn_parser.rb +23 -6
  406. data/lib/pubid/oiml.rb +4 -0
  407. data/lib/pubid/omg/builder.rb +19 -0
  408. data/lib/pubid/omg/identifier.rb +44 -0
  409. data/lib/pubid/omg/identifiers/specification.rb +32 -0
  410. data/lib/pubid/omg/identifiers.rb +9 -0
  411. data/lib/pubid/omg/parser.rb +40 -0
  412. data/lib/pubid/omg/renderer.rb +19 -0
  413. data/lib/pubid/omg.rb +51 -0
  414. data/lib/pubid/parsers/mr_string.rb +101 -7
  415. data/lib/pubid/plateau/builder.rb +8 -4
  416. data/lib/pubid/plateau/identifier.rb +1 -3
  417. data/lib/pubid/plateau/identifiers/base.rb +1 -6
  418. data/lib/pubid/plateau/identifiers/handbook.rb +1 -1
  419. data/lib/pubid/plateau/identifiers/technical_report.rb +1 -1
  420. data/lib/pubid/plateau/identifiers.rb +0 -1
  421. data/lib/pubid/plateau/parser.rb +15 -5
  422. data/lib/pubid/plateau/renderer.rb +1 -1
  423. data/lib/pubid/plateau/supplement_identifier.rb +11 -11
  424. data/lib/pubid/plateau.rb +4 -4
  425. data/lib/pubid/renderers/human_readable.rb +3 -1
  426. data/lib/pubid/renderers/mr_string.rb +46 -4
  427. data/lib/pubid/renderers/supplement_renderer.rb +1 -1
  428. data/lib/pubid/rendering/supplement.rb +4 -4
  429. data/lib/pubid/sae/builder.rb +1 -1
  430. data/lib/pubid/sae/identifier.rb +1 -3
  431. data/lib/pubid/sae/identifiers/base.rb +10 -7
  432. data/lib/pubid/sae/identifiers.rb +0 -1
  433. data/lib/pubid/sae.rb +3 -3
  434. data/lib/pubid/tgpp/builder.rb +41 -0
  435. data/lib/pubid/tgpp/identifier.rb +91 -0
  436. data/lib/pubid/tgpp/identifiers/technical_report.rb +32 -0
  437. data/lib/pubid/tgpp/identifiers/technical_specification.rb +33 -0
  438. data/lib/pubid/tgpp/identifiers.rb +11 -0
  439. data/lib/pubid/tgpp/parser.rb +62 -0
  440. data/lib/pubid/tgpp/renderer.rb +36 -0
  441. data/lib/pubid/tgpp/urn_generator.rb +20 -0
  442. data/lib/pubid/tgpp/urn_parser.rb +28 -0
  443. data/lib/pubid/tgpp.rb +78 -0
  444. data/lib/pubid/type_resolver.rb +59 -0
  445. data/lib/pubid/un/builder.rb +42 -0
  446. data/lib/pubid/un/identifier.rb +37 -0
  447. data/lib/pubid/un/identifiers/document.rb +24 -0
  448. data/lib/pubid/un/identifiers.rb +9 -0
  449. data/lib/pubid/un/parser.rb +25 -0
  450. data/lib/pubid/un/renderer.rb +11 -0
  451. data/lib/pubid/un.rb +46 -0
  452. data/lib/pubid/version.rb +1 -1
  453. data/lib/pubid/w3c/builder.rb +55 -0
  454. data/lib/pubid/w3c/identifier.rb +101 -0
  455. data/lib/pubid/w3c/identifiers/candidate_recommendation.rb +31 -0
  456. data/lib/pubid/w3c/identifiers/candidate_recommendation_draft.rb +31 -0
  457. data/lib/pubid/w3c/identifiers/draft_note.rb +31 -0
  458. data/lib/pubid/w3c/identifiers/note.rb +30 -0
  459. data/lib/pubid/w3c/identifiers/obsolete_recommendation.rb +31 -0
  460. data/lib/pubid/w3c/identifiers/proposed_edited_recommendation.rb +31 -0
  461. data/lib/pubid/w3c/identifiers/proposed_recommendation.rb +31 -0
  462. data/lib/pubid/w3c/identifiers/recommendation.rb +31 -0
  463. data/lib/pubid/w3c/identifiers/standard.rb +12 -0
  464. data/lib/pubid/w3c/identifiers/superseded_recommendation.rb +31 -0
  465. data/lib/pubid/w3c/identifiers/working_draft.rb +30 -0
  466. data/lib/pubid/w3c/identifiers.rb +25 -0
  467. data/lib/pubid/w3c/parser.rb +43 -0
  468. data/lib/pubid/w3c/renderer.rb +35 -0
  469. data/lib/pubid/w3c/urn_generator.rb +18 -0
  470. data/lib/pubid/w3c/urn_parser.rb +36 -0
  471. data/lib/pubid/w3c.rb +67 -0
  472. data/lib/pubid/xsf/builder.rb +17 -0
  473. data/lib/pubid/xsf/identifier.rb +62 -0
  474. data/lib/pubid/xsf/identifiers/xep.rb +31 -0
  475. data/lib/pubid/xsf/identifiers.rb +9 -0
  476. data/lib/pubid/xsf/parser.rb +25 -0
  477. data/lib/pubid/xsf/renderer.rb +23 -0
  478. data/lib/pubid/xsf/urn_generator.rb +13 -0
  479. data/lib/pubid/xsf/urn_parser.rb +18 -0
  480. data/lib/pubid/xsf.rb +76 -0
  481. data/lib/pubid.rb +25 -1
  482. metadata +205 -4
  483. data/lib/pubid/cie/components/code.rb +0 -80
  484. data/lib/pubid/iala/identifiers/base.rb +0 -14
@@ -57,6 +57,24 @@ module Pubid
57
57
  date_with_month_text | date_with_month_numeric | date_year_only
58
58
  end
59
59
 
60
+ # Trailing print/reaffirm date "<sep>Month YYYY" — captured under the
61
+ # distinct :trailing_month/:trailing_year keys so it never collides with a
62
+ # base -YYYY identity year (the builder promotes it only when no base year
63
+ # exists). Accepts BOTH a month name ("May 2014") and a month-first numeric
64
+ # month ("05 2014"). The two variants are unambiguous: month_name never
65
+ # starts with a digit, and month_numeric (01-12) never matches a bare
66
+ # 19xx/20xx year — so a bare trailing year still falls through to callers'
67
+ # own bare-year clause.
68
+ rule(:trailing_month_year) do
69
+ # The numeric branch accepts `-` as well as space before the year:
70
+ # preprocessing (the `(\d)\s+(\d{4})` gsub) rewrites a trailing
71
+ # "05 2014" → "05-2014", so a numeric month reaches the grammar
72
+ # dash-joined; the month-name branch escapes that gsub (the char before
73
+ # the space is a letter) and stays space-joined.
74
+ ((comma | space) >> month_name.as(:trailing_month) >> space >> year_digits.as(:trailing_year)) |
75
+ ((comma | space) >> month_numeric.as(:trailing_month) >> (space | dash) >> year_digits.as(:trailing_year))
76
+ end
77
+
60
78
  # Month patterns
61
79
  rule(:month_name) do
62
80
  # Period-suffixed abbreviations (longest first)
@@ -75,9 +93,17 @@ module Pubid
75
93
  # Organizations
76
94
  rule(:organization) do
77
95
  str("IEEE") | str("AIEE") | str("ANSI") | str("ASA") |
96
+ # ANS (American Nuclear Society) — a third co-publisher on nuclear
97
+ # standards ("ANSI/IEEE/ANS 7.4-3-2-1982"). Listed AFTER ANSI so it
98
+ # never shadows the longer token.
99
+ str("ANS") |
78
100
  str("IEC") | str("ISO") | str("ASTM") | str("CSA") | str("ASME") |
79
101
  str("NACE") | str("NSF") | str("ASHRAE") | str("NCTA") | str("AESC") |
80
- str("EIA") # NEW Session 224: Add EIA support
102
+ str("EIA") | # NEW Session 224: Add EIA support
103
+ # Historical / foreign co-publishers seen in relaton-data-ieee
104
+ # (bucket 4): AMPP, USAS (both standalone), and the IEEE sub-board /
105
+ # partner co-publishers USEMCSC, EAB, MPAI.
106
+ str("USEMCSC") | str("AMPP") | str("USAS") | str("EAB") | str("MPAI")
81
107
  end
82
108
 
83
109
  # Complex organization prefixes (Category 5: ANSI Complex)
@@ -130,12 +156,42 @@ module Pubid
130
156
  (slash >> str("C") >> digits >> dot >> digits >> dot >> digits >> dash >> year_digits).as(:ieee_crossref)
131
157
  end
132
158
 
159
+ # IPCEA co-designation suffix (/IPCEA P-46-426-1962). Captured verbatim
160
+ # (leading slash included) so it round-trips through the `crossref`
161
+ # attribute. Used only by the S-designation rule below.
162
+ rule(:ipcea_copub) do
163
+ (slash >> str("IPCEA") >> space >>
164
+ match('[A-Za-z0-9.\-]').repeat(1)).as(:ipcea_copub)
165
+ end
166
+
167
+ # Historical IEEE/IPCEA co-published cable designation (e.g. S-135).
168
+ # The "S-<digits>" number has a dash between the letter series and the
169
+ # digits, which the shared `number` rule deliberately rejects (that
170
+ # tightening is what keeps a bare "IEEE S" from parsing). So this one-off
171
+ # family gets its own rule rather than loosening `number` and risking the
172
+ # 12k-row corpus. Requiring `str("S") >> dash >> digits` keeps "IEEE S"
173
+ # (no dash+digit) rejected. Handles: "IEEE Std S-135", "IEEE S-135",
174
+ # bare "S-135", the "/IPCEA …" slash co-designation, and the
175
+ # "(IPCEA …)" parenthetical variant.
176
+ rule(:s_designation) do
177
+ (publisher >> space).maybe >>
178
+ (type_word.as(:type) >> space?).maybe >>
179
+ (str("S") >> dash >> digits).as(:s_number) >>
180
+ ipcea_copub.maybe >>
181
+ parenthetical.maybe
182
+ end
183
+
133
184
  # Document number - support letters and digits, with optional prefix P
134
185
  # Complex multi-part numbers like P11073-10404-10419 should be fully captured
135
186
  # But simple cases like "623-1976" should not consume the dash before year
136
187
  rule(:number) do
137
188
  (str("P").maybe >>
138
- (digits | upper).repeat(1) >> # The first component must be at least one digit
189
+ # The numeric core must contain at least one digit: an optional letter
190
+ # prefix (C, S, …) then a required `digits` run, then any digit/letter
191
+ # tail. The old `(digits | upper).repeat(1)` wrongly accepted an
192
+ # all-letter token like a bare "S" (a prefix with no number — e.g.
193
+ # "IEEE S"), which is not a valid IEEE identifier.
194
+ upper.repeat(0) >> digits >> (digits | upper).repeat(0) >>
139
195
  # Only consume dash+digits if followed by another dash+digits (multi-part pattern)
140
196
  # OR if the digits don't look like a year (not 4 digits starting with 19/20)
141
197
  # This prevents consuming "623-1976" as a number but allows "P11073-10404-10419"
@@ -170,7 +226,14 @@ module Pubid
170
226
 
171
227
  # Draft patterns
172
228
  rule(:draft_status) do
173
- (str("Active Unapproved") | str("Unapproved") | str("Approved")) >> space
229
+ # "Active" (bucket 3) joins the generic draft-status path so
230
+ # "IEEE Active Std P… /D…" parses like the "Unapproved" forms, without
231
+ # touching ieee_approved_draft_identifier (which would break the
232
+ # issue-#209 unapproved-drops-Std rendering). "Active Approved" completes
233
+ # the two-word matrix alongside "Active Unapproved" (draft-grammar
234
+ # coverage: the "<status> Draft P…/D…" prefix forms). Longest token first.
235
+ (str("Active Unapproved") | str("Active Approved") | str("Unapproved") |
236
+ str("Approved") | str("Active")) >> space
174
237
  end
175
238
 
176
239
  rule(:draft_prefix) do
@@ -180,7 +243,11 @@ module Pubid
180
243
  rule(:draft_version) do
181
244
  # Enhanced to handle multiple draft notation patterns
182
245
  # D is optional to handle /08 style drafts (e.g., IEEE P1052/08)
183
- (str("D") >> str("IS").absent?).maybe >> # Avoid matching "DIS" (ISO stage)
246
+ # A draft never begins with "R-" — that is the revision suffix
247
+ # (revision_suffix rule); guard so the D-less path doesn't swallow a
248
+ # bare "/R-<id>" (e.g. the no-draft "P1722/R-1") as a draft.
249
+ (str("R") >> dash).absent? >>
250
+ (str("D") >> str("IS").absent?).maybe >> # Avoid matching "DIS" (ISO stage)
184
251
  (
185
252
  # Pattern: D3.1 (decimal with 1-2 digits on each side) - MOST COMMON, put first
186
253
  # Also handles trailing letter: D7.3A, D2.0E
@@ -202,8 +269,17 @@ module Pubid
202
269
  rule(:draft_date) do
203
270
  # Enhanced to handle: ", Sept 2008" or " Sept 2008" or ", Month Year"
204
271
  ((comma | space) >> month_name.as(:month) >> space >> year_digits.as(:year)) |
272
+ # Numeric-month form ", 05 2007" / " 05 2007" — no text month name. The
273
+ # year separator may be a dash: preprocessing (parser.rb ~1278) rewrites
274
+ # a trailing " <digits> <year>" to "<digits>-<year>", so "05 2007"
275
+ # reaches the grammar as "05-2007".
276
+ ((comma | space) >> month_numeric.as(:month) >> (space | dash) >> year_digits.as(:year)) |
205
277
  (((space? >> comma >> space?) | space) >> month_name.as(:month) >>
206
278
  (
279
+ # "Month DD, Year" (day then comma) and "Month DD Year" (day, no
280
+ # comma) — the day/year separator may be a dash for the same
281
+ # preprocessing reason ("July 15 2012" → "July 15-2012").
282
+ ((space >> digits.as(:day)) >> ((comma | space | dash) >> year_digits.as(:year))) |
207
283
  ((space >> digits.as(:day)).maybe >> comma >> year_digits.as(:year)) |
208
284
  (comma >> space? >> year_digits.as(:year)) |
209
285
  (space >> year_digits.as(:year))
@@ -229,13 +305,32 @@ module Pubid
229
305
  draft_date.maybe).as(:draft)
230
306
  end
231
307
 
308
+ # Trailing revision suffix "/R-<id>". Normalization funnels every revision
309
+ # spelling here — IEEE's native inline "Rev<n>" (repositioned) and
310
+ # relaton's synthetic "/R-<x>" (kept in place) — so a single trailing rule
311
+ # captures them all. The id is alphanumeric ("2", "18", "i").
312
+ rule(:revision_suffix) do
313
+ slash >> str("R") >> dash >> match('[0-9A-Za-z]').repeat(1).as(:revision)
314
+ end
315
+
232
316
  # Edition - enhanced to support IEC formats like "Edition 1.0 2015-03"
317
+ #
318
+ # The edition's own year is captured as :edition_year (NOT :year). The
319
+ # generic Std branch carries a base "-YYYY" slot AND a separate trailing
320
+ # `edition.maybe`; a bare :year here would collide with the base :year
321
+ # when both fire, so Parslet warns "Duplicate subtrees … keys: [:year]"
322
+ # and drops the base identity year (the same failure #299 fixed for the
323
+ # trailing month/year clause). The builder promotes :edition_year to the
324
+ # identity year only when there is no base year.
233
325
  rule(:edition) do
234
- (comma >> year_digits.as(:year) >> str(" Edition")) |
326
+ (comma >> year_digits.as(:edition_year) >> str(" Edition")) |
235
327
  ((space | dash) >> str("Edition ") >>
236
328
  (digits >> dot >> digits).as(:edition) >>
237
- (space | str(" - ")) >>
238
- year_digits.as(:year) >>
329
+ # Year separator: a space, " - ", or a bare dash (preprocessing
330
+ # rewrites "Edition 3.0 2016" -> "Edition 3.0-2016", the shape the
331
+ # normalized "/E-<n>-YYYY" suffix produces — nil-residue item 1).
332
+ (str(" - ") | space | dash) >>
333
+ year_digits.as(:edition_year) >>
239
334
  (dash >> digit.repeat(2, 2).as(:edition_month)).maybe) # Capture -MM as edition_month
240
335
  end
241
336
 
@@ -261,10 +356,16 @@ module Pubid
261
356
  ((dash | str(":") | space) >> year_digits.as(:cor_year)).maybe).as(:corrigendum)
262
357
  end
263
358
 
264
- # Amendment
359
+ # Amendment — mirrors the corrigendum rule's separator flexibility so
360
+ # IEEE-format "/Amd 2-2004" and "/Amd2-2004" parse the same way as
361
+ # their Cor counterparts (issue #210).
265
362
  rule(:amendment) do
266
- (slash >> str("Amd") >> digits.as(:amd_number) >>
267
- (dash >> year_digits.as(:amd_year)).maybe).as(:amendment)
363
+ ((str("_") | slash | dash | space) >>
364
+ (str("Amendment") | str("Amd")) >>
365
+ (dash | dot | space).maybe >>
366
+ space? >>
367
+ digits.as(:amd_number).maybe >>
368
+ ((dash | str(":") | space) >> year_digits.as(:amd_year)).maybe).as(:amendment)
268
369
  end
269
370
 
270
371
  # Interpretation notation (/INT)
@@ -283,9 +384,14 @@ module Pubid
283
384
  ).as(:reaffirmed)
284
385
  end
285
386
 
286
- # Redline
387
+ # Redline suffix at the very end. Accepts relaton's canonical " Redline"
388
+ # (space, no dash) and pubid's older " - Redline" (space-dash-space),
389
+ # case-insensitive. Captured (presence only) so the builder sets a
390
+ # `redline: true` flag the renderer restores — a redline is a distinct
391
+ # document and must not collapse to its base standard.
287
392
  rule(:redline) do
288
- str(" - Redline").as(:redline)
393
+ (space >> (dash >> space).maybe >>
394
+ (str("Redline") | str("REDLINE") | str("redline"))).as(:redline)
289
395
  end
290
396
 
291
397
  # Book nickname (e.g., "[The Orange Book]", "[IEEE Gold Book]")
@@ -346,8 +452,11 @@ module Pubid
346
452
  str(" and ").absent? >>
347
453
  str(", ").absent? >>
348
454
  str(" as amended by ").absent? >>
349
- str(" / ").absent? >>
350
- str("; ").absent? >>
455
+ # Stop at any "/", ";", or "-" that introduces another relationship
456
+ # (look-ahead: separator + relationship_type keyword). This lets the
457
+ # slash inside "IEEE Std 525-2007/Cor 1-2015" stay part of the
458
+ # identifier, while the slash before "Incorporates ..." splits.
459
+ relationship_break.absent? >>
351
460
  str(")").absent? >>
352
461
  match(".")
353
462
  ).repeat(1)
@@ -372,15 +481,38 @@ module Pubid
372
481
  str(" and its approved amendments").as(:approved_amendments)
373
482
  end
374
483
 
484
+ # A character sequence that may separate two relationships inside the
485
+ # parenthetical: a "/", ";", or "-" with optional surrounding spaces.
486
+ # Standalone it is permissive — the actual decision to split is gated by
487
+ # `relationship_break`, which requires another relationship_type to
488
+ # follow. This way the slash in "IEEE Std 525-2007/Cor 1-2015" stays
489
+ # part of the related identifier.
490
+ rule(:relationship_separator) do
491
+ (space.maybe >> str("/") >> space.maybe) |
492
+ (space.maybe >> str(";") >> space.maybe) |
493
+ (space.maybe >> str("-") >> space.maybe)
494
+ end
495
+
496
+ # Look-ahead: a separator followed by another relationship_type keyword.
497
+ # Used as an absent? guard in identifier_string so the parser stops
498
+ # consuming characters right before a new relationship begins.
499
+ rule(:relationship_break) do
500
+ relationship_separator >> space.maybe >> relationship_type
501
+ end
502
+
375
503
  # Relationship clause (handles all relationship types)
376
504
  rule(:relationship_clause) do
377
505
  space.maybe >> str("(") >>
378
506
  relationship_type.as(:relationship_type) >>
379
507
  identifier_list.as(:related_ids) >>
380
508
  as_amended_by_clause.maybe >>
381
- # Handle multiple relationships separated by " / " OR "; "
509
+ # Additional relationships separated by "/", ";", or "-" (optionally
510
+ # surrounded by spaces). The separator is only honored when followed
511
+ # by another relationship_type — identifier-internal slashes like
512
+ # "/Cor 1-2015" are not mistaken for relationship breaks because
513
+ # identifier_string stops at relationship_break ahead.
382
514
  (
383
- (str(" / ") | str("; ")) >> # Support both separators
515
+ relationship_separator >>
384
516
  relationship_type.as(:relationship_type) >>
385
517
  identifier_list.as(:related_ids) >>
386
518
  as_amended_by_clause.maybe
@@ -435,6 +567,11 @@ module Pubid
435
567
  str("IEC/IEEE") >>
436
568
  space >>
437
569
  str("P").absent? >> # NOT a P prefix (would be joint development)
570
+ # The copublished number must contain at least one digit. This rejects
571
+ # an all-letter placeholder like "IEC/IEEE TR" (no real document
572
+ # number) — the same "require a digit" tightening the Standard number
573
+ # rule got, applied to the copublished number grammar.
574
+ (match("[^0-9\n]").repeat >> digit).present? >>
438
575
  match("[^\n]").repeat(1).as(:content)
439
576
  end
440
577
 
@@ -456,20 +593,100 @@ module Pubid
456
593
  # Variant 2: , CDV1 notation (comma before stage code)
457
594
  (comma >> (str("CDV") | str("FDIS") | str("CD") | str("DIS")).as(:iec_stage) >> digits.maybe.as(:stage_iteration))
458
595
  ).maybe >>
596
+ # Optional edition, from relaton's "/E-<n>" suffix normalized to
597
+ # "Edition <n>.0[ YYYY]" (nil-residue hand-off item 1). The edition
598
+ # rule carries its own year, so the year clause below simply doesn't
599
+ # fire when an edition is present.
600
+ edition.maybe >>
459
601
  ((dash >> year_digits.as(:year)) | # Either -YEAR
460
- (comma.maybe >> space >> month_name.as(:month) >> space.maybe >> year_digits.as(:year))).maybe # Or Month YEAR (with optional comma)
602
+ (comma.maybe >> space >> month_name.as(:month) >> space.maybe >> year_digits.as(:year))).maybe >> # Or Month YEAR (with optional comma)
603
+ revision_suffix.maybe
461
604
  end
462
605
 
463
606
  rule(:joint_development_iso_format) do
464
- # ISO/IEC/IEEE FDIS 26511:2018 (ISO-led format)
465
- (str("ISO/IEC/IEEE") | str("ISO/IEEE") | str("IEC/IEEE")).as(:joint_publishers) >>
607
+ # ISO-led stage designations. Two spellings:
608
+ # colon form : "ISO/IEC/IEEE FDIS 26511:2018" (already used)
609
+ # corpus form : "ISO/IEC/IEEE FDIS P26515-2018-05" (historical)
610
+ # The corpus form adds a leading "P" on the number, a trailing
611
+ # "-YYYY[-MM]" date (instead of ":YYYY"), multi-digit committee-draft
612
+ # stage codes (CD1..CD4) plus CDV, and a wider set of joint publishers.
613
+ # (roadmap items 2/3, phase 1). longest publisher token first.
614
+ (str("ISO/IEC/IEEE") | str("IEEE/ISO/IEC") | str("IEEE/IEC/ISO") |
615
+ str("ISO/IEEE") | str("IEC/IEEE") | str("IEEE/IEC") | str("ISO/IEC") |
616
+ str("IEEE")).as(:joint_publishers) >>
466
617
  space >>
467
- # ISO stage codes
468
- (str("FDIS") | str("DIS") | str("CD") | str("WD") | str("PWI") | str("NP")).as(:iso_stage) >>
618
+ # ISO stage codes: FDIS, FCD, CDV; DIS/CD with an optional round digit
619
+ # (DIS2, CD1..CD4); WD/PWI/NP. (FCD before FDIS is fine — distinct.)
620
+ (str("FDIS") | str("FCD") | str("CDV") |
621
+ (str("DIS") >> digit.maybe) |
622
+ (str("CD") >> digit.maybe) |
623
+ str("WD") | str("PWI") | str("NP")).as(:iso_stage) >>
624
+ # optional " Std" noise word after the stage (e.g. "FDIS Std P15288")
625
+ (space >> str("Std")).maybe >>
469
626
  space >>
627
+ str("P").maybe >> # optional project marker on the number
470
628
  digits.as(:number) >>
471
- ((dot | dash) >> digits.as(:part)).maybe >> # Optional part
472
- (str(":") >> year_digits.as(:year)).maybe
629
+ # part must not swallow the trailing year (year_digits.absent?)
630
+ ((dot | dash) >> year_digits.absent? >> digits.as(:part)).maybe >>
631
+ (
632
+ (str(":") >> year_digits.as(:year)) |
633
+ (dash >> year_digits.as(:year) >>
634
+ (dash >> month_numeric.as(:month)).maybe) |
635
+ # Trailing text date ", April 2015" / " April 2015" (build_joint_development
636
+ # already reads :month/:year). No :year collision — the iso rule has no
637
+ # other :year capture (edition uses :edition_year).
638
+ ((comma | space) >> month_name.as(:month) >> space >> year_digits.as(:year))
639
+ ).maybe >>
640
+ # Optional /D<draft> tail. normalize_relaton_suffixes repositions the
641
+ # historical "…/D-3-2017" onto the number as "…-2017/D3", so by the
642
+ # time this rule runs the draft usually trails the date (bucket 5);
643
+ # a date-less "/D-4" keeps its hyphen (bucket 7), hence dash.maybe.
644
+ (slash >> str("D") >> dash.maybe >>
645
+ match('[0-9.]').repeat(1).as(:draft_version)).maybe >>
646
+ # Optional edition, from relaton's "/E-<n>" suffix normalized to
647
+ # "Edition <n>.0[ YYYY]" (nil-residue hand-off item 1).
648
+ edition.maybe >>
649
+ revision_suffix.maybe
650
+ end
651
+
652
+ # Embedded (stage-LAST) ISO-led designations: the corpus writes the ISO
653
+ # stage AFTER the (dotted or dashed) part —
654
+ # "ISO/IEC/IEEE 29119.4.FDIS, April 2015" (dot part .4, stage .FDIS)
655
+ # "ISO/IEC/IEEE 24748-5.CD3, February 2015" (dash part -5, stage .CD3)
656
+ # "IEEE P24748.5.CD3, July 2015" (bare IEEE + P)
657
+ # — rather than before the number (joint_development_iso_format). Same
658
+ # meaning as the stage-first form: `29119.4`/`29119-4` = "29119 part 4" and
659
+ # the trailing `.FDIS` is the ISO stage, NOT a second part. Routes through
660
+ # the same build_joint_development (via :joint_publishers/:iso_stage), so the
661
+ # stage is modeled correctly and the numeric part stays separate.
662
+ rule(:joint_development_embedded_stage) do
663
+ # publisher set mirrors joint_development_iso_format (incl. bare IEEE);
664
+ # longest token first.
665
+ (str("ISO/IEC/IEEE") | str("IEEE/ISO/IEC") | str("IEEE/IEC/ISO") |
666
+ str("ISO/IEEE") | str("IEC/IEEE") | str("IEEE/IEC") | str("ISO/IEC") |
667
+ str("IEEE")).as(:joint_publishers) >>
668
+ space >>
669
+ str("P").maybe >> # optional project marker on the number
670
+ digits.as(:number) >>
671
+ # optional numeric part (dot or dash); must not swallow a year
672
+ ((dot | dash) >> year_digits.absent? >> digits.as(:part)).maybe >>
673
+ # the embedded stage, dot-separated, closed vocab (same as iso_stage).
674
+ # The (digit|letter) look-ahead keeps it at a token boundary so "CD"
675
+ # can't match inside a longer token and a glued no-space date
676
+ # (".DISMay2013") is left to fall through.
677
+ dot >>
678
+ (str("FDIS") | str("FCD") | str("CDV") |
679
+ (str("DIS") >> digit.maybe) |
680
+ (str("CD") >> digit.maybe) |
681
+ str("WD") | str("PWI") | str("NP")).as(:iso_stage) >>
682
+ (digit | match("[A-Za-z]")).absent? >>
683
+ # optional trailing date: :YYYY, -YYYY[-MM], or text "Month YYYY"
684
+ (
685
+ (str(":") >> year_digits.as(:year)) |
686
+ (dash >> year_digits.as(:year) >>
687
+ (dash >> month_numeric.as(:month)).maybe) |
688
+ ((comma | space) >> month_name.as(:month) >> space >> year_digits.as(:year))
689
+ ).maybe
473
690
  end
474
691
 
475
692
  # Number-first pattern: "1873-2015 IEEE Standard..."
@@ -493,12 +710,25 @@ module Pubid
493
710
  (slash >> digits.as(:draft_version)).as(:digit_draft).maybe >>
494
711
  # FDIS and other ISO stage codes without D prefix (Pattern 3)
495
712
  fdraft.maybe >>
496
- # Enhanced: Accept both comma and space before month/year
497
- ((comma | space) >> month_name.as(:month) >> space >> year_digits.as(:year)).maybe >>
713
+ # Trailing "Month YYYY" print/reaffirm date. Captured under distinct
714
+ # keys so it never collides with the base -YYYY identity year (a
715
+ # collision made Parslet drop the base year and warn "Duplicate
716
+ # subtrees … keys: [:year]"). The builder promotes it to the identity
717
+ # only when there is no base year.
718
+ trailing_month_year.maybe >>
498
719
  corrigendum.maybe >>
499
720
  draft.maybe >>
721
+ # Revision trails the draft (before any date), matching normalization's
722
+ # ".../D<n>/R-<x>" repositioning of "P802.16Rev2/D3 Feb 2008".
723
+ revision_suffix.maybe >>
500
724
  # ALSO accept month/year after draft (some patterns like /DX, Month YEAR)
501
- ((comma | space) >> month_name.as(:month) >> space >> year_digits.as(:year)).maybe >>
725
+ trailing_month_year.maybe >>
726
+ # Trailing corrigendum AFTER the draft+date ("…/D1, Jan 2007/Cor. 1").
727
+ # The draft's own draft_date has already consumed the date, leaving
728
+ # "/Cor. 1" here; the resulting flat corrigendum+draft tree routes to
729
+ # build_flat_corrigendum (disjoint from the pre-draft corrigendum slot,
730
+ # which only fires when no draft precedes it).
731
+ corrigendum.maybe >>
502
732
  parenthetical.maybe
503
733
  end
504
734
 
@@ -508,27 +738,50 @@ module Pubid
508
738
  str("P") >> space.maybe >> # Make space after P optional
509
739
  number >>
510
740
  (part_subpart_year | edition).maybe >>
511
- # Enhanced: Accept both comma and space before month/year
512
- ((comma | space) >> month_name.as(:month) >> space >> year_digits.as(:year)).maybe >>
741
+ # Trailing "Month YYYY"/bare-year date under distinct keys so they
742
+ # never collide with the base -YYYY identity year (see ieee_p_identifier).
743
+ trailing_month_year.maybe >>
513
744
  corrigendum.maybe >>
514
745
  draft.maybe >>
746
+ revision_suffix.maybe >>
515
747
  # ALSO accept month/year after draft
516
- ((comma | space) >> month_name.as(:month) >> space >> year_digits.as(:year)).maybe >>
748
+ trailing_month_year.maybe >>
517
749
  # Accept bare year after draft: ", 2015"
518
- ((comma | space) >> year_digits.as(:year)).maybe >>
750
+ ((comma | space) >> year_digits.as(:trailing_year)).maybe >>
751
+ # Trailing corrigendum after the draft+date (parity with ieee_p_identifier).
752
+ corrigendum.maybe >>
519
753
  parenthetical.maybe
520
754
  end
521
755
 
522
756
  # IEEE Draft P pattern: "IEEE Draft P802.11..." OR "Draft P802.11..." (IEEE prefix optional)
757
+ # An optional status phrase may precede "Draft" — "IEEE Unapproved Draft P…",
758
+ # "IEEE Active Approved Draft P…" — the largest index-v2 draft bucket. The
759
+ # status is captured (draft_status ends in a space, so "Approved Draft"
760
+ # splits cleanly) and round-trips as the draft_status attribute; the literal
761
+ # "Draft" stays a bare marker (dropped on render, like the plain form).
523
762
  rule(:ieee_draft_p_identifier) do
524
763
  (str("IEEE").as(:publisher) >> space).maybe >> # Make IEEE prefix optional
764
+ draft_status.as(:draft_status).maybe >>
525
765
  str("Draft") >> space >>
526
- str("P") >>
766
+ # `P` is optional — a status-word draft may carry a bare number
767
+ # ("IEEE Unapproved Draft 802.1ah/D4.2"), mirroring
768
+ # ieee_approved_draft_identifier's str("P").maybe. The `number` rule
769
+ # already accepts the bare forms (802.1ah, C57.15, 11073-10471).
770
+ str("P").maybe >>
527
771
  number >>
528
772
  (part_subpart_year | edition).maybe >>
529
- # Enhanced: Accept month/year after draft number
530
- (space >> month_name.as(:month) >> space >> year_digits.as(:year)).maybe >>
773
+ # Trailing "Month YYYY" date under distinct keys so it never collides
774
+ # with the base -YYYY identity year (see ieee_p_identifier).
775
+ ((space >> month_name.as(:trailing_month) >> space >> year_digits.as(:trailing_year)) |
776
+ (space >> month_numeric.as(:trailing_month) >> (space | dash) >> year_digits.as(:trailing_year))).maybe >>
531
777
  draft.maybe >>
778
+ revision_suffix.maybe >>
779
+ # Trailing corrigendum after the draft ("…/D2.0/Cor. 1", or
780
+ # "…/D1.0, Dec 2007/Cor. 1" where the draft's own draft_date consumes
781
+ # the date, leaving "/Cor. N"). The flat corrigendum+draft tree (no
782
+ # :base) routes to build_flat_corrigendum, which rebuilds the base
783
+ # standard (carrying the draft + draft_status) and wraps it.
784
+ corrigendum.maybe >>
532
785
  parenthetical.maybe
533
786
  end
534
787
 
@@ -542,6 +795,7 @@ module Pubid
542
795
  number >>
543
796
  (part_subpart_year | edition).maybe >>
544
797
  draft.maybe >>
798
+ revision_suffix.maybe >>
545
799
  parenthetical.maybe
546
800
  end
547
801
 
@@ -656,7 +910,7 @@ module Pubid
656
910
  (type_word.as(:type) >> space?).maybe >>
657
911
  number >>
658
912
  part_subpart_year.maybe # This captures the full identifier before /Cor
659
- ).as(:base_identifier) >>
913
+ ).as(:base) >>
660
914
  # Now match the corrigendum portion
661
915
  (slash | dash | space) >>
662
916
  str("Cor") >>
@@ -677,7 +931,7 @@ module Pubid
677
931
  (type_word.as(:type) >> space?).maybe >>
678
932
  number >>
679
933
  part_subpart_year.maybe
680
- ).as(:base_identifier) >>
934
+ ).as(:base) >>
681
935
  # Now match the interpretation portion
682
936
  (slash | dash | space) >>
683
937
  str("INT") >>
@@ -695,7 +949,7 @@ module Pubid
695
949
  (type_word.as(:type) >> space?).maybe >>
696
950
  number >>
697
951
  part_subpart_year.maybe
698
- ).as(:base_identifier) >>
952
+ ).as(:base) >>
699
953
  # Now match the conformance portion
700
954
  (slash | dash | space) >>
701
955
  str("Conformance") >>
@@ -757,11 +1011,13 @@ module Pubid
757
1011
  ieee_astm_si_psi | # NEW Session 171: Add IEEE/ASTM SI/PSI support
758
1012
  multi_numbered_identifier | # NEW: Try multi-numbered identifiers before generic patterns
759
1013
  csa_dual_published | # NEW: Try CSA dual published before generic patterns
1014
+ s_designation | # Historical IEEE/IPCEA cable designation (S-135)
760
1015
  corrigendum_identifier | # NEW: Try corrigendum before generic patterns
761
1016
  interpretation_identifier | # NEW: Try interpretation identifier before generic patterns
762
1017
  conformance_identifier | # NEW: Try conformance identifier before generic patterns
763
1018
  joint_development_ieee_format |
764
1019
  joint_development_iso_format |
1020
+ joint_development_embedded_stage | # stage-LAST embedded form (before generic)
765
1021
  iec_ieee_copublished |
766
1022
  number_first_identifier |
767
1023
  ieee_approved_draft_identifier |
@@ -781,8 +1037,16 @@ module Pubid
781
1037
  ashrae_copub.maybe >> # NEW: Add /ASHRAE Guideline support
782
1038
  ieee_crossref.maybe >> # NEW: Add /C62.22.1-1996 cross-reference support
783
1039
  draft.maybe >>
784
- # Enhanced: Accept both comma and space before month/year
785
- ((comma | space) >> month_name.as(:month) >> space >> year_digits.as(:year)).maybe >>
1040
+ revision_suffix.maybe >>
1041
+ # Trailing "Month YYYY" print/reaffirm date under distinct keys so it
1042
+ # never collides with the base -YYYY identity year (see
1043
+ # ieee_p_identifier). The builder promotes it only when no base year.
1044
+ trailing_month_year.maybe >>
1045
+ # Trailing corrigendum after the draft+date ("IEEE Approved P1015/D1,
1046
+ # Jan 2007/Cor. 1"): the generic bucket is the only path a status-word
1047
+ # form reaches. Routes via build_flat_corrigendum (disjoint from the
1048
+ # pre-draft corrigendum slot at line 954).
1049
+ corrigendum.maybe >>
786
1050
  edition.maybe >>
787
1051
  parenthetical.maybe >> # REVERT: Back to single parenthetical
788
1052
  book_nickname.maybe >> # NEW: Add book nickname support
@@ -794,6 +1058,133 @@ module Pubid
794
1058
 
795
1059
  root(:identifier)
796
1060
 
1061
+ # Rewrite relaton's historical IEEE serialization into canonical pubid
1062
+ # spellings. relaton's own formatter (Relaton::Ieee::PubId::Id#to_s) emits
1063
+ # suffix tokens that differ from pubid's grammar:
1064
+ #
1065
+ # /D-N-YYYY[-MM] draft + trailing numeric date (the dominant form)
1066
+ # /E-N[-YYYY[-MM]] edition
1067
+ # /R-N[-YYYY] revision (pubid has no revision suffix)
1068
+ # " Redline" redline suffix without the " - " pubid expects
1069
+ #
1070
+ # The draft/edition trailing date is repositioned onto the document number
1071
+ # as a base year/month (a form pubid already parses), which also keeps the
1072
+ # draft component clean so it round-trips through to_hash/from_hash.
1073
+ def self.normalize_relaton_suffixes(cleaned)
1074
+ # NOTE: the trailing " Redline"/" - Redline" suffix is NO LONGER stripped
1075
+ # here — the grammar's `redline` rule captures it into a redline flag so
1076
+ # a redline id stays distinct from its base standard.
1077
+
1078
+ # Combined draft + corrigendum: relaton emits "…/D-N/CorM-YYYY" (draft
1079
+ # then corrigendum), but pubid's grammar accepts the corrigendum first.
1080
+ # Swap them so the corrigendum keeps its own year and the draft trails.
1081
+ # The hyphen after "D" is mandatory here: relaton's formatter always
1082
+ # emits "/D-<draft>", whereas pubid's own canonical joint-development
1083
+ # form is "/D<draft>-<year>" (no hyphen, year kept on the draft) — which
1084
+ # already parses and must not be repositioned. A trailing corrigendum
1085
+ # month (the "-MM" in "/CorM-YYYY-MM") is intentionally dropped: pubid's
1086
+ # corrigendum model carries only a year.
1087
+ cleaned = cleaned.sub(
1088
+ %r{\A(.*)/D-([0-9A-Za-z][0-9A-Za-z.+]*?)/Cor\.?[ ]?(\d+)(?:-((?:19|20)\d\d))?(?:-\d\d)?\z},
1089
+ ) do
1090
+ base, draft, cor, year = Regexp.last_match.captures
1091
+ "#{base}/Cor #{cor}#{year ? "-#{year}" : ''}/D#{draft}"
1092
+ end
1093
+
1094
+ # Combined draft + revision, and the empty-draft revision-only form:
1095
+ # "…/D-<d>/R-<x>-YYYY[-MM]" and "…/D-/R-<x>-YYYY" (nil-residue #2).
1096
+ # Reposition the base publication date onto the number (pubid's
1097
+ # "-YYYY[-MM]" shape), keep the draft as "/D<d>" (dropped when the draft
1098
+ # is empty), and leave a trailing "/R-<x>" the grammar captures as the
1099
+ # revision. Runs before the plain "/D-…" reposition, which the embedded
1100
+ # "/R-" would otherwise defeat.
1101
+ cleaned = cleaned.sub(
1102
+ %r{\A(.*?)/D-([0-9A-Za-z.+]*)/R-([0-9A-Za-z]+)(?:-((?:19|20)\d\d)(?:-(0[1-9]|1[0-2]))?)?\z},
1103
+ ) do
1104
+ base, draft, rev, year, month = Regexp.last_match.captures
1105
+ date = year ? "-#{year}#{month ? "-#{month}" : ''}" : ""
1106
+ draft_part = draft.to_s.empty? ? "" : "/D#{draft}"
1107
+ "#{base}#{date}#{draft_part}/R-#{rev}"
1108
+ end
1109
+
1110
+ # /D-N drafts with a trailing numeric date, when the draft is the last
1111
+ # suffix: reposition the -YYYY[-MM] date onto the number. A following
1112
+ # /Cor, /Amd, /R or /E suffix carries its own year, so the `\z` anchor
1113
+ # keeps this from firing on those combined forms.
1114
+ cleaned = cleaned.sub(
1115
+ %r{\A(.*)/D-([0-9A-Za-z][0-9A-Za-z.+]*?)-((?:19|20)\d\d)(?:-(0[1-9]|1[0-2]))?\z},
1116
+ ) do
1117
+ base, draft, year, month = Regexp.last_match.captures
1118
+ "#{base}-#{year}#{month ? "-#{month}" : ''}/D#{draft}"
1119
+ end
1120
+
1121
+ # /E-N editions: relaton's "/E-2-2023-02" → pubid's "Edition 2.0 2023-02".
1122
+ cleaned = cleaned.sub(
1123
+ %r{\A(.*?)/E-(\d+)(?:-((?:19|20)\d\d)(?:-(0[1-9]|1[0-2]))?)?\z},
1124
+ ) do
1125
+ base, edition, year, month = Regexp.last_match.captures
1126
+ date = year ? " #{year}#{month ? "-#{month}" : ''}" : ""
1127
+ "#{base} Edition #{edition}.0#{date}"
1128
+ end
1129
+
1130
+ # /R-N revisions: PRESERVE them (the grammar's revision_suffix rule now
1131
+ # captures a trailing "/R-<x>" into the `revision` attribute). Just
1132
+ # reposition any trailing publication year onto the number, keeping the
1133
+ # "/R-<x>" in place for the grammar.
1134
+ cleaned.sub(
1135
+ %r{\A(.*?)/R-([0-9A-Za-z]+)(?:-((?:19|20)\d\d))?\z},
1136
+ ) do
1137
+ base, rev, year = Regexp.last_match.captures
1138
+ "#{year ? "#{base}-#{year}" : base}/R-#{rev}"
1139
+ end
1140
+ end
1141
+
1142
+ # Strip the IEEE rawbib revision-notation dialects. `REV`/`Rev`
1143
+ # (case-insensitive) + a trailing revision id `[A-Za-z0-9]+`, glued to the
1144
+ # number or separated by `-`, `/`, `_`, `.`, or a space, and preceding the
1145
+ # draft. pubid's canonical "<num>/D<n>/R-<x>" form already drops the
1146
+ # revision on render (normalize_relaton_suffixes strips a trailing /R-x),
1147
+ # so the revision-less result is *the same identifier* — and stripping
1148
+ # (rather than reordering) leaves any trailing date/parenthetical intact,
1149
+ # which is why forms that already parse (`Draft P…-REVmb/D3.0, Mar 2010`)
1150
+ # are NOT disturbed. Examples:
1151
+ # "P802.16.2-REVa/D8" -> "P802.16.2/D8"
1152
+ # "P802.16/REVd/D5" -> "P802.16/D5"
1153
+ # "P802.15.1REVa/D5" -> "P802.15.1/D5"
1154
+ # "P802.11REVmb" -> "P802.11" (no draft)
1155
+ def self.normalize_revision_notation(cleaned)
1156
+ # NUMBERED revisions ("Rev<digits>") are PRESERVED — repositioned to a
1157
+ # trailing "/R-<n>" suffix the grammar captures as the `revision`
1158
+ # attribute (IEEE's native inline spelling; numbered-revision hand-off).
1159
+ # A "\d+" right after "Rev" both selects the numbered subset and keeps
1160
+ # these off the English word "Revision". Three source positions:
1161
+ # after a draft : "PC37.30.2/D043 Rev 18" -> ".../D043/R-18"
1162
+ cleaned = cleaned.sub(
1163
+ %r{(/D[0-9A-Za-z.]*)\s+[Rr][Ee][Vv]\s*(\d+)}, '\1/R-\2'
1164
+ )
1165
+ # before a draft: "P802.16Rev2/D3" -> "P802.16/D3/R-2"
1166
+ cleaned = cleaned.sub(
1167
+ %r{[-/_.]?\s?[Rr][Ee][Vv][-\s]?(\d+)(/D[0-9A-Za-z.]*)}, '\2/R-\1'
1168
+ )
1169
+ # no draft, trailing: "P1722-rev1" -> "P1722/R-1"
1170
+ cleaned = cleaned.sub(
1171
+ %r{(\d)[-._]?\s?[Rr][Ee][Vv]\s*(\d+)\s*\z}, '\1/R-\2'
1172
+ )
1173
+
1174
+ # LETTERED inline revisions ("REVa", "REVmb") have no pubid model and are
1175
+ # still STRIPPED (unchanged behaviour). The numbered forms above already
1176
+ # became "/R-<n>", so these regexes only see the lettered residue.
1177
+ # Revision token that PRECEDES a draft: drop it (keep the /D…).
1178
+ cleaned = cleaned.sub(
1179
+ %r{[-/_.]?\s?[Rr][Ee][Vv][-\s]?[A-Za-z0-9]+(?=/D[0-9])},
1180
+ "",
1181
+ )
1182
+ # Trailing revision glued to the number with no draft ("P802.11REVmb");
1183
+ # a digit must immediately precede REV so a trailing English word like
1184
+ # "…Revision" can't match.
1185
+ cleaned.sub(%r{(\d)[Rr][Ee][Vv][A-Za-z0-9]+\s*\z}, '\1')
1186
+ end
1187
+
797
1188
  def self.parse(string)
798
1189
  # Strip .pdf extension if present (Pattern 3: File Extensions)
799
1190
  cleaned = string.sub(/\.pdf$/i, "")
@@ -811,6 +1202,24 @@ module Pubid
811
1202
  # No valid IEEE identifier pattern needs more than 1 space
812
1203
  cleaned = cleaned.gsub(/\s+/, " ")
813
1204
 
1205
+ # A joint ISO-led publisher list is sometimes crawled with a stray slash
1206
+ # (or slash+space) before the ISO stage code — "ISO/IEC/IEEE/ FDIS …" or
1207
+ # "ISO/IEC/IEEE/FDIS …". Restore the space separator so the stage parses
1208
+ # (bucket 7).
1209
+ cleaned = cleaned.gsub(
1210
+ %r{\b(ISO/IEC/IEEE|IEEE/ISO/IEC|IEEE/IEC/ISO|ISO/IEEE|IEC/IEEE|IEEE/IEC|ISO/IEC)/ ?(FDIS|FCD|CDV|DIS\d?|CD\d?|WD|PWI|NP)\b},
1211
+ '\1 \2',
1212
+ )
1213
+
1214
+ # Rewrite the rawbib revision-notation dialects (REVa/REVd/glued) into
1215
+ # the canonical /R-<x> form before the suffix normalization below.
1216
+ cleaned = normalize_revision_notation(cleaned)
1217
+
1218
+ # Normalize relaton's bespoke historical serialization (the spellings
1219
+ # emitted by Relaton::Ieee::PubId::Id#to_s) into canonical pubid forms
1220
+ # so `relaton-data-ieee` parses. See #normalize_relaton_suffixes.
1221
+ cleaned = normalize_relaton_suffixes(cleaned)
1222
+
814
1223
  # NEW Session 171: CONSERVATIVE data quality fixes for TODO.IEEE-MUST-DO.txt
815
1224
  # Only fix clear typos: space before dash + 4-digit year, OR dash + space + 4-digit year
816
1225
  # Do NOT touch " - " (space-dash-space) which is valid formatting
@@ -859,10 +1268,21 @@ module Pubid
859
1268
  # NEW: Convert IEC/IEEE space-separated to semicolon format
860
1269
  # Pattern: "IEC 61523-3 First edition 2004-09; IEEE 1497" → already semicolon
861
1270
  # Pattern: "IEC 62539 First Edition 2007-07 IEEE 930" → needs semicolon
862
- # Match: IEC identifier (with edition) + space + IEEE identifier
863
- # Be conservative: only convert if IEC has "First edition" or similar and followed by IEEE
1271
+ # Pattern: "IEC 60076-21:2011 Edition 1.0 2011-12 IEEE Std C57.15" → needs semicolon (issue #202)
1272
+ # Match: IEC identifier (with optional colon-year, optional "First"/numeric
1273
+ # Edition + YYYY-MM) + space + IEEE identifier.
1274
+ cleaned = cleaned.gsub(
1275
+ /(IEC\s+\d+(?:-\d+)?(?::\d{4})?(?:\s+(?:First\s+)?[Ee]dition\s+\d+(?:\.\d+)?\s+\d{4}-\d{2})?)\s+(IEEE\s+Std\s+\S+|IEEE\s+\S+)/,
1276
+ '\1; \2'
1277
+ )
1278
+
1279
+ # Strip ":YYYY" from IEC numbers when an Edition clause follows — the
1280
+ # IEEE parser's number rule doesn't accept the colon-year form, but
1281
+ # the year is preserved in the "Edition N.M YYYY-MM" suffix.
1282
+ # (issue #202)
864
1283
  cleaned = cleaned.gsub(
865
- /(IEC\s+\d+(?:-\d+)?(?:\s+First?\s+Edition\s+\d{4}-\d{2})?)\s+(IEEE\s+\S+)/, '\1; \2'
1284
+ /^(IEC\s+\d+(?:-\d+)?):\d{4}(\s+(?:First\s+)?[Ee]dition\s+\d+(?:\.\d+)?\s+\d{4}-\d{2})/,
1285
+ '\1\2'
866
1286
  )
867
1287
 
868
1288
  # NEW Phase 1 (Session 141): Remove literal trademark symbol
@@ -1078,8 +1498,6 @@ module Pubid
1078
1498
  # Remove period after "Std": "IEEE Std." -> "IEEE Std"
1079
1499
  cleaned = cleaned.gsub(/\bStd\.\s+/, "Std ")
1080
1500
 
1081
- # Redline Suffix Removal: " - Redline" at end
1082
- cleaned = cleaned.gsub(/\s+-\s+Redline\b.*$/, "")
1083
1501
 
1084
1502
  # Title portion removal after year: "YYYY - IEEE Standard for..."
1085
1503
  cleaned = cleaned.gsub(