pubid 2.0.0.pre.alpha.12 → 2.0.0.pre.alpha.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (285) hide show
  1. checksums.yaml +4 -4
  2. data/README.adoc +43 -1
  3. data/data/ieee/update_codes.yaml +17 -4
  4. data/data/nist/update_codes.yaml +7 -3
  5. data/data/parg/tables/bipm_groups.yaml +14 -0
  6. data/data/parg/tables/bipm_type_codes.yaml +5 -0
  7. data/data/parg/tables/bipm_type_names_en.yaml +6 -0
  8. data/data/parg/tables/bipm_type_names_fr.yaml +6 -0
  9. data/data/parg/tables/directives_supplements_typed_stages.yaml +3 -0
  10. data/data/parg/tables/directives_typed_stages.yaml +5 -0
  11. data/data/parg/tables/idf_typed_stages.yaml +27 -0
  12. data/data/parg/tables/idf_typed_stages_supplements.yaml +2 -0
  13. data/data/parg/tables/iec_typed_stages.yaml +130 -0
  14. data/data/parg/tables/iso_publishers.yaml +4 -0
  15. data/data/parg/tables/organizations.yaml +12 -0
  16. data/data/parg/tables/tc_types.yaml +42 -0
  17. data/data/parg/tables/typed_stages.yaml +114 -0
  18. data/data/parg/tables/typed_stages_supplements.yaml +64 -0
  19. data/data/parg/tables/wg_types.yaml +21 -0
  20. data/lib/pubid/adobe/builder.rb +2 -0
  21. data/lib/pubid/adobe/identifier.rb +11 -1
  22. data/lib/pubid/all_parts.rb +201 -0
  23. data/lib/pubid/all_parts_identifier.rb +19 -0
  24. data/lib/pubid/amca/CLAUDE.md +47 -0
  25. data/lib/pubid/amca/builder.rb +3 -5
  26. data/lib/pubid/amca/identifiers/base.rb +11 -1
  27. data/lib/pubid/amca/identifiers/publication.rb +13 -0
  28. data/lib/pubid/amca/parser.rb +2 -1
  29. data/lib/pubid/amca/renderer.rb +22 -33
  30. data/lib/pubid/amca/urn_generator.rb +21 -2
  31. data/lib/pubid/amca/urn_parser.rb +36 -10
  32. data/lib/pubid/ansi/builder.rb +6 -0
  33. data/lib/pubid/ansi/identifier.rb +1 -1
  34. data/lib/pubid/api/CLAUDE.md +23 -0
  35. data/lib/pubid/api/builder.rb +2 -0
  36. data/lib/pubid/api/identifier.rb +1 -1
  37. data/lib/pubid/api/parser.rb +8 -4
  38. data/lib/pubid/ashrae/CLAUDE.md +13 -0
  39. data/lib/pubid/ashrae/builder.rb +58 -14
  40. data/lib/pubid/ashrae/identifiers/base.rb +10 -1
  41. data/lib/pubid/ashrae/identifiers/errata.rb +14 -2
  42. data/lib/pubid/ashrae/identifiers/interpretation.rb +2 -10
  43. data/lib/pubid/ashrae/parser.rb +80 -39
  44. data/lib/pubid/ashrae/renderer.rb +32 -1
  45. data/lib/pubid/ashrae/urn_generator.rb +32 -9
  46. data/lib/pubid/asme/CLAUDE.md +25 -0
  47. data/lib/pubid/asme/builder.rb +16 -9
  48. data/lib/pubid/asme/components/code.rb +2 -0
  49. data/lib/pubid/asme/identifier.rb +1 -1
  50. data/lib/pubid/asme/identifiers/standard.rb +6 -1
  51. data/lib/pubid/asme/parser.rb +41 -14
  52. data/lib/pubid/astm/CLAUDE.md +9 -0
  53. data/lib/pubid/astm/builder.rb +2 -0
  54. data/lib/pubid/astm/components/code.rb +2 -0
  55. data/lib/pubid/astm/identifier.rb +1 -1
  56. data/lib/pubid/astm/parser.rb +4 -1
  57. data/lib/pubid/bipm/CLAUDE.md +11 -0
  58. data/lib/pubid/bipm/builder.rb +2 -0
  59. data/lib/pubid/bipm/identifier.rb +1 -1
  60. data/lib/pubid/bsi/CLAUDE.md +93 -0
  61. data/lib/pubid/bsi/builder.rb +13 -11
  62. data/lib/pubid/bsi/identifiers/addendum_document.rb +2 -0
  63. data/lib/pubid/bsi/identifiers/adopted_european_norm.rb +6 -54
  64. data/lib/pubid/bsi/identifiers/adopted_international_standard.rb +5 -22
  65. data/lib/pubid/bsi/identifiers/amendment.rb +36 -12
  66. data/lib/pubid/bsi/identifiers/bundled_identifier.rb +2 -0
  67. data/lib/pubid/bsi/identifiers/consolidated_identifier.rb +23 -26
  68. data/lib/pubid/bsi/identifiers/corrigendum.rb +29 -12
  69. data/lib/pubid/bsi/identifiers/expert_commentary.rb +6 -7
  70. data/lib/pubid/bsi/identifiers/national_annex.rb +18 -20
  71. data/lib/pubid/bsi/identifiers/root_identity.rb +31 -0
  72. data/lib/pubid/bsi/identifiers/set.rb +2 -0
  73. data/lib/pubid/bsi/identifiers/supplement_document.rb +2 -0
  74. data/lib/pubid/bsi/identifiers.rb +1 -0
  75. data/lib/pubid/bsi/parser.rb +8 -8
  76. data/lib/pubid/bsi/renderer.rb +20 -20
  77. data/lib/pubid/bsi/single_identifier.rb +1 -3
  78. data/lib/pubid/bsi/urn_generator.rb +28 -18
  79. data/lib/pubid/builder/base.rb +27 -0
  80. data/lib/pubid/calconnect/builder.rb +2 -0
  81. data/lib/pubid/calconnect/identifier.rb +5 -1
  82. data/lib/pubid/ccsds/builder.rb +2 -0
  83. data/lib/pubid/ccsds/identifier.rb +9 -1
  84. data/lib/pubid/cen_cenelec/CLAUDE.md +59 -0
  85. data/lib/pubid/cen_cenelec/builder.rb +6 -1
  86. data/lib/pubid/cen_cenelec/identifier.rb +12 -28
  87. data/lib/pubid/cen_cenelec/identifiers/amendment.rb +3 -10
  88. data/lib/pubid/cen_cenelec/identifiers/corrigendum.rb +3 -10
  89. data/lib/pubid/cen_cenelec/parser.rb +20 -5
  90. data/lib/pubid/cie/CLAUDE.md +58 -0
  91. data/lib/pubid/cie/builder.rb +2 -0
  92. data/lib/pubid/cie/components/language.rb +2 -0
  93. data/lib/pubid/cie/identifier.rb +1 -1
  94. data/lib/pubid/cie/parser.rb +9 -2
  95. data/lib/pubid/components/adoption.rb +2 -0
  96. data/lib/pubid/components/code.rb +2 -0
  97. data/lib/pubid/components/date.rb +8 -6
  98. data/lib/pubid/components/edition.rb +2 -0
  99. data/lib/pubid/components/iteration.rb +2 -0
  100. data/lib/pubid/components/language.rb +2 -0
  101. data/lib/pubid/components/locality.rb +2 -0
  102. data/lib/pubid/components/publisher.rb +2 -0
  103. data/lib/pubid/components/relationship.rb +2 -0
  104. data/lib/pubid/components/stage.rb +2 -0
  105. data/lib/pubid/components/supplement.rb +2 -0
  106. data/lib/pubid/components/type.rb +2 -0
  107. data/lib/pubid/components/typed_stage.rb +8 -0
  108. data/lib/pubid/conformance/checks.rb +1 -1
  109. data/lib/pubid/csa/CLAUDE.md +41 -0
  110. data/lib/pubid/csa/builder.rb +2 -0
  111. data/lib/pubid/csa/identifier.rb +19 -3
  112. data/lib/pubid/csa/parser.rb +25 -8
  113. data/lib/pubid/csa/renderer.rb +12 -12
  114. data/lib/pubid/csa/single_identifier.rb +17 -0
  115. data/lib/pubid/doi/builder.rb +2 -0
  116. data/lib/pubid/doi/identifier.rb +1 -1
  117. data/lib/pubid/easc/builder.rb +2 -0
  118. data/lib/pubid/easc/identifier.rb +10 -1
  119. data/lib/pubid/ecma/CLAUDE.md +28 -0
  120. data/lib/pubid/ecma/builder.rb +2 -0
  121. data/lib/pubid/ecma/identifier.rb +8 -1
  122. data/lib/pubid/etsi/CLAUDE.md +34 -0
  123. data/lib/pubid/etsi/builder.rb +2 -0
  124. data/lib/pubid/etsi/components/code.rb +6 -0
  125. data/lib/pubid/etsi/components/version.rb +2 -0
  126. data/lib/pubid/etsi/identifiers/base.rb +1 -1
  127. data/lib/pubid/etsi/identifiers/etsi_standard.rb +7 -0
  128. data/lib/pubid/evs/CLAUDE.md +58 -0
  129. data/lib/pubid/evs/builder.rb +2 -0
  130. data/lib/pubid/evs.rb +1 -1
  131. data/lib/pubid/gb/CLAUDE.md +140 -0
  132. data/lib/pubid/gb/builder.rb +7 -2
  133. data/lib/pubid/gb/identifier.rb +6 -4
  134. data/lib/pubid/gb/identifiers/all_parts.rb +17 -0
  135. data/lib/pubid/gb/identifiers.rb +1 -0
  136. data/lib/pubid/gb/renderer.rb +0 -1
  137. data/lib/pubid/gost/CLAUDE.md +64 -0
  138. data/lib/pubid/gost/builder.rb +3 -1
  139. data/lib/pubid/gost/identifier.rb +16 -1
  140. data/lib/pubid/gost/parser.rb +8 -1
  141. data/lib/pubid/iala/CLAUDE.md +82 -0
  142. data/lib/pubid/iala/builder.rb +2 -0
  143. data/lib/pubid/iala/identifier.rb +10 -1
  144. data/lib/pubid/iana/CLAUDE.md +7 -0
  145. data/lib/pubid/iana/builder.rb +2 -0
  146. data/lib/pubid/iana/identifier.rb +1 -1
  147. data/lib/pubid/identifier.rb +161 -17
  148. data/lib/pubid/idf/builder.rb +11 -1
  149. data/lib/pubid/idf/identifier.rb +5 -0
  150. data/lib/pubid/idf/identifiers/all_parts.rb +17 -0
  151. data/lib/pubid/idf/identifiers.rb +1 -0
  152. data/lib/pubid/iec/CLAUDE.md +31 -0
  153. data/lib/pubid/iec/builder.rb +7 -1
  154. data/lib/pubid/iec/components/consolidated_amendment.rb +4 -0
  155. data/lib/pubid/iec/components/sheet.rb +2 -0
  156. data/lib/pubid/iec/components/trf_info.rb +2 -0
  157. data/lib/pubid/iec/components/vap_suffix.rb +2 -0
  158. data/lib/pubid/iec/identifier.rb +8 -3
  159. data/lib/pubid/iec/identifiers/all_parts.rb +19 -0
  160. data/lib/pubid/iec/identifiers.rb +1 -0
  161. data/lib/pubid/iec/parser.rb +9 -4
  162. data/lib/pubid/iec/renderer.rb +0 -1
  163. data/lib/pubid/iec/urn_generator.rb +9 -1
  164. data/lib/pubid/iec/urn_parser.rb +3 -2
  165. data/lib/pubid/ieee/CLAUDE.md +97 -0
  166. data/lib/pubid/ieee/builder.rb +194 -27
  167. data/lib/pubid/ieee/components/code.rb +2 -0
  168. data/lib/pubid/ieee/components/draft.rb +35 -2
  169. data/lib/pubid/ieee/components/typed_stage.rb +2 -0
  170. data/lib/pubid/ieee/identifiers/base.rb +21 -1
  171. data/lib/pubid/ieee/identifiers/iec_ieee_copublished.rb +9 -0
  172. data/lib/pubid/ieee/identifiers/joint_development.rb +66 -23
  173. data/lib/pubid/ieee/identifiers/project_draft_identifier.rb +8 -1
  174. data/lib/pubid/ieee/parser.rb +156 -29
  175. data/lib/pubid/ieee/renderer.rb +42 -7
  176. data/lib/pubid/ieee/urn_generator.rb +31 -0
  177. data/lib/pubid/ietf/CLAUDE.md +7 -0
  178. data/lib/pubid/ietf/builder.rb +2 -0
  179. data/lib/pubid/ietf/identifiers/base.rb +1 -1
  180. data/lib/pubid/iho/builder.rb +2 -0
  181. data/lib/pubid/isbn/builder.rb +2 -0
  182. data/lib/pubid/isbn/identifier.rb +1 -1
  183. data/lib/pubid/iso/CLAUDE.md +47 -0
  184. data/lib/pubid/iso/builder.rb +19 -5
  185. data/lib/pubid/iso/components/publisher.rb +2 -0
  186. data/lib/pubid/iso/identifier.rb +10 -15
  187. data/lib/pubid/iso/identifiers/all_parts.rb +19 -0
  188. data/lib/pubid/iso/identifiers/directives_supplement.rb +4 -2
  189. data/lib/pubid/iso/identifiers.rb +1 -0
  190. data/lib/pubid/iso/normalizer.rb +4 -1
  191. data/lib/pubid/iso/rendering_style.rb +0 -1
  192. data/lib/pubid/itu/CLAUDE.md +115 -0
  193. data/lib/pubid/itu/builder.rb +26 -4
  194. data/lib/pubid/itu/components/code.rb +2 -0
  195. data/lib/pubid/itu/components/designation.rb +2 -0
  196. data/lib/pubid/itu/components/sector.rb +2 -0
  197. data/lib/pubid/itu/components/series.rb +2 -0
  198. data/lib/pubid/itu/identifiers/base.rb +11 -18
  199. data/lib/pubid/itu/identifiers/radio_regulations.rb +27 -0
  200. data/lib/pubid/itu/identifiers/special_publication.rb +48 -14
  201. data/lib/pubid/itu/identifiers/standard_serialization.rb +2 -0
  202. data/lib/pubid/itu/identifiers/supplement.rb +15 -0
  203. data/lib/pubid/itu/identifiers.rb +1 -0
  204. data/lib/pubid/itu/parser.rb +108 -22
  205. data/lib/pubid/itu/urn_generator.rb +9 -2
  206. data/lib/pubid/jcgm/CLAUDE.md +7 -0
  207. data/lib/pubid/jcgm/builder.rb +2 -0
  208. data/lib/pubid/jcgm/components/publisher.rb +2 -0
  209. data/lib/pubid/jcgm.rb +1 -1
  210. data/lib/pubid/jis/builder.rb +5 -1
  211. data/lib/pubid/jis/identifier.rb +6 -18
  212. data/lib/pubid/jis/identifiers/all_parts.rb +19 -0
  213. data/lib/pubid/jis/identifiers.rb +1 -0
  214. data/lib/pubid/jis/renderer.rb +0 -2
  215. data/lib/pubid/jis/urn_generator.rb +0 -1
  216. data/lib/pubid/nist/CLAUDE.md +56 -0
  217. data/lib/pubid/nist/builder.rb +3 -0
  218. data/lib/pubid/nist/components/edition.rb +2 -0
  219. data/lib/pubid/nist/components/issue_number.rb +2 -0
  220. data/lib/pubid/nist/components/part.rb +2 -0
  221. data/lib/pubid/nist/components/stage.rb +2 -0
  222. data/lib/pubid/nist/components/supplement.rb +2 -0
  223. data/lib/pubid/nist/components/translation.rb +2 -0
  224. data/lib/pubid/nist/components/update.rb +2 -0
  225. data/lib/pubid/nist/components/version.rb +2 -0
  226. data/lib/pubid/nist/components/volume.rb +2 -0
  227. data/lib/pubid/nist/identifiers/base.rb +39 -7
  228. data/lib/pubid/nist/parser.rb +24 -2
  229. data/lib/pubid/nist/preprocessor.rb +53 -2
  230. data/lib/pubid/nist/urn_parser.rb +10 -1
  231. data/lib/pubid/oasis/CLAUDE.md +19 -0
  232. data/lib/pubid/oasis/builder.rb +2 -0
  233. data/lib/pubid/oasis/identifier.rb +20 -1
  234. data/lib/pubid/ogc/CLAUDE.md +34 -0
  235. data/lib/pubid/ogc/builder.rb +2 -0
  236. data/lib/pubid/ogc/identifier.rb +12 -1
  237. data/lib/pubid/oiml/CLAUDE.md +189 -0
  238. data/lib/pubid/oiml/builder.rb +20 -0
  239. data/lib/pubid/oiml/components/code.rb +6 -0
  240. data/lib/pubid/oiml/identifier.rb +13 -0
  241. data/lib/pubid/oiml/identifiers/annex.rb +4 -0
  242. data/lib/pubid/oiml/identifiers/certification_system.rb +34 -0
  243. data/lib/pubid/oiml/identifiers/code_number.rb +8 -0
  244. data/lib/pubid/oiml/identifiers/dual_published.rb +174 -0
  245. data/lib/pubid/oiml/identifiers.rb +2 -0
  246. data/lib/pubid/oiml/parser.rb +35 -4
  247. data/lib/pubid/oiml/renderer.rb +23 -1
  248. data/lib/pubid/oiml/single_identifier.rb +4 -0
  249. data/lib/pubid/oiml/supplement_identifier.rb +7 -0
  250. data/lib/pubid/oiml/urn_generator.rb +28 -0
  251. data/lib/pubid/oiml.rb +6 -1
  252. data/lib/pubid/omg/CLAUDE.md +15 -0
  253. data/lib/pubid/omg/builder.rb +2 -0
  254. data/lib/pubid/omg/identifier.rb +1 -1
  255. data/lib/pubid/parg/artifact.rb +46 -0
  256. data/lib/pubid/parg/backend.rb +92 -0
  257. data/lib/pubid/parg.rb +8 -0
  258. data/lib/pubid/parser/grammar.rb +23 -0
  259. data/lib/pubid/pg.rb +8 -0
  260. data/lib/pubid/plateau/builder.rb +2 -0
  261. data/lib/pubid/plateau/identifiers/base.rb +4 -0
  262. data/lib/pubid/plateau/supplement_identifier.rb +14 -2
  263. data/lib/pubid/plateau/urn_generator.rb +7 -1
  264. data/lib/pubid/plateau.rb +1 -2
  265. data/lib/pubid/renderers/human_readable.rb +0 -1
  266. data/lib/pubid/sae/builder.rb +2 -0
  267. data/lib/pubid/sae/components/date.rb +2 -0
  268. data/lib/pubid/sae/components/type.rb +2 -0
  269. data/lib/pubid/sae/identifiers/base.rb +1 -1
  270. data/lib/pubid/subset_match.rb +197 -0
  271. data/lib/pubid/tgpp/CLAUDE.md +43 -0
  272. data/lib/pubid/tgpp/builder.rb +2 -0
  273. data/lib/pubid/tgpp/identifier.rb +15 -1
  274. data/lib/pubid/type_resolver.rb +14 -2
  275. data/lib/pubid/un/builder.rb +2 -0
  276. data/lib/pubid/un/identifier.rb +1 -1
  277. data/lib/pubid/version.rb +1 -1
  278. data/lib/pubid/w3c/CLAUDE.md +7 -0
  279. data/lib/pubid/w3c/builder.rb +2 -0
  280. data/lib/pubid/w3c/identifier.rb +1 -1
  281. data/lib/pubid/xsf/CLAUDE.md +11 -0
  282. data/lib/pubid/xsf/builder.rb +2 -0
  283. data/lib/pubid/xsf/identifier.rb +1 -1
  284. data/lib/pubid.rb +17 -3
  285. metadata +78 -2
@@ -7,6 +7,11 @@ module Pubid
7
7
  # Pubid::Jis::Identifiers descend from this class, so a parsed JIS id is an
8
8
  # instance of Pubid::Jis::Identifier.
9
9
  class Identifier < ::Pubid::Identifier
10
+ # JIS prints its own all-parts suffix, "(規格群)".
11
+ def self.all_parts_class
12
+ Identifiers::AllParts
13
+ end
14
+
10
15
  # JIS keeps its number flat at the top level (string, to preserve leading
11
16
  # zeros like "0205"), with the division letter in `series` and any
12
17
  # multi-level part numbers in `parts`. Supplements override `number` with
@@ -16,9 +21,6 @@ module Pubid
16
21
  attribute :parts, :string, collection: true # Optional multi-level parts
17
22
  attribute :year, :integer
18
23
  attribute :language, :string # "E" or "J"
19
- # Boolean flags carry no default, so they stay nil (and are omitted from
20
- # the serialized hash) unless actually set true.
21
- attribute :all_parts, :boolean
22
24
  # Reaffirmation (再確認): a trailing "R" on the year marks an edition
23
25
  # that was reaffirmed without revision (e.g. ":2019R").
24
26
  attribute :reaffirmed, :boolean
@@ -50,7 +52,6 @@ module Pubid
50
52
  map "parts", to: :parts
51
53
  map "year", to: :year
52
54
  map "language", to: :language
53
- map "all_parts", to: :all_parts
54
55
  map "reaffirmed", to: :reaffirmed
55
56
  # render_empty keeps a bare "SYMBOL" (empty-string value) in the hash so
56
57
  # it round-trips distinctly from "no symbol" (nil).
@@ -62,10 +63,6 @@ module Pubid
62
63
  # would otherwise fail serialization type validation.
63
64
  PUBLISHER = "JIS"
64
65
 
65
- def all_parts?
66
- all_parts == true
67
- end
68
-
69
66
  def reaffirmed?
70
67
  reaffirmed == true
71
68
  end
@@ -91,23 +88,14 @@ module Pubid
91
88
  result
92
89
  end
93
90
 
94
- # Comparison with all_parts logic
95
- # When either identifier has all_parts=true, compare only series and number
96
91
  def ==(other)
97
92
  return false unless other.is_a?(Identifier)
98
93
 
99
- if all_parts? || other.all_parts?
100
- # Compare only series and number, ignore year, parts, all_parts
101
- return series == other.series && number == other.number
102
- end
103
-
104
- # Normal full comparison
105
94
  series == other.series &&
106
95
  number == other.number &&
107
96
  (parts || []) == (other.parts || []) &&
108
97
  year == other.year &&
109
98
  language == other.language &&
110
- all_parts? == other.all_parts? &&
111
99
  reaffirmed? == other.reaffirmed? &&
112
100
  symbol == other.symbol
113
101
  end
@@ -186,7 +174,7 @@ module Pubid
186
174
  raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
187
175
  end
188
176
 
189
- parsed = Parser.parse(identifier)
177
+ parsed = Pubid::Parg::Backend.parse(:jis, identifier)
190
178
  Builder.build(parsed)
191
179
  end
192
180
  end
@@ -0,0 +1,19 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Pubid
4
+ module Jis
5
+ module Identifiers
6
+ # Every part of one JIS document: "JIS C 0617(規格群)".
7
+ class AllParts < ::Pubid::Jis::Identifier
8
+ include ::Pubid::AllParts
9
+
10
+ SUFFIX = "(規格群)"
11
+
12
+ # The URN of the document plus the "all" slot.
13
+ def to_urn
14
+ "#{identity.to_urn}:all"
15
+ end
16
+ end
17
+ end
18
+ end
19
+ end
@@ -3,6 +3,7 @@
3
3
  module Pubid
4
4
  module Jis
5
5
  module Identifiers
6
+ autoload :AllParts, "#{__dir__}/identifiers/all_parts"
6
7
  autoload :Amendment, "#{__dir__}/identifiers/amendment"
7
8
  autoload :Corrigendum, "#{__dir__}/identifiers/corrigendum"
8
9
  autoload :Explanation, "#{__dir__}/identifiers/explanation"
@@ -38,7 +38,6 @@ module Pubid
38
38
  result = "#{PUBLISHER} #{id.code}"
39
39
  result += ":#{id.year_with_reaffirmation}" if id.year
40
40
  result += "(#{id.language})" if id.language
41
- result += "(規格群)" if id.all_parts?
42
41
  result + id.symbol_suffix
43
42
  end
44
43
 
@@ -52,7 +51,6 @@ module Pubid
52
51
  result += id.code.to_s
53
52
  result += ":#{id.year_with_reaffirmation}" if id.year
54
53
  result += "(#{id.language})" if id.language
55
- result += "(規格群)" if id.all_parts?
56
54
  result += id.symbol_suffix
57
55
  result
58
56
  end
@@ -12,7 +12,6 @@ module Pubid
12
12
 
13
13
  parts << identifier.language.to_s.downcase if identifier.language
14
14
 
15
- parts << "all" if identifier.all_parts?
16
15
 
17
16
  if identifier.is_a?(SupplementIdentifier) && identifier.supplement_notation
18
17
  parts << identifier.supplement_notation.to_s.downcase
@@ -0,0 +1,56 @@
1
+ # NIST flavor notes
2
+
3
+ NIST rendering formats and annotated output.
4
+
5
+ Read this before you change `lib/pubid/nist/` or `spec/pubid/nist/`. The root
6
+ `CLAUDE.md` keeps the cross-flavor contract that every flavor obeys.
7
+
8
+ - **Six NIST types rendered plain under `to_s(annotated: true)`, and two
9
+ more carried a latent corruption.** NIST's `to_s` takes a **positional**
10
+ `format`, which Ruby fills with a Hash when a caller passes `annotated:`;
11
+ `Identifiers::Base#to_s` already handles that and annotates. The six that
12
+ did not — `CommercialStandardsMonthly`, `CrplReport`, `InteragencyReport`,
13
+ `MiscellaneousPublication`, `Monograph`, `Report` — either compose their
14
+ own string or hand `super` a **Symbol**, which drops the flag. Each now
15
+ pulls `annotated` out of the Hash itself and wraps its result.
16
+
17
+ **`Circular` and `Handbook` are the interesting half.** Both did
18
+ `result = super` — which annotates — and then rewrote the edition with a
19
+ **`$`-anchored** regex. Once the string ends in `</span>` the anchor
20
+ cannot match, so the rewrite silently stops applying. Both now strip the
21
+ flag before `super`, rewrite, then annotate.
22
+
23
+ **An output assertion cannot catch that**, and this is the lesson: for
24
+ both fixture references (`NBS CIRC 11e2-1915`, `NBS HB 44e2-1955`) the
25
+ annotator matches only the leading publisher, so the tail is bare and the
26
+ wrong order still produces the right string. The spec asserts the
27
+ **ordering** instead — it wraps `annotate_plain_render` and checks the
28
+ string handed to it is already rewritten and carries no span — and that
29
+ assertion was verified to fail when the order is reverted.
30
+
31
+ `CircularSupplement` and `SupplementIdentifier` were also normalised:
32
+ both had a date-range branch that returned a hand-composed string without
33
+ ever reaching `super`. `SupplementIdentifier` has four exits, so its
34
+ composition moved into a private `render_plain`; **`super` is not
35
+ reachable from a private method**, so `to_s` hands it in as a block —
36
+ `render_plain(format) { super(format) }`.
37
+
38
+ - **`all_parts_edition_keys` needed `update`/`update_component`, not
39
+ `edition_year`.** `Identifier.all_parts_edition_keys` defaults to
40
+ `%i[date year edition version]`; NIST's primary edition carrier
41
+ (`edition`, a `Components::Edition`) is already covered, but the Letter
42
+ Circular / Circular "rJun1992"-style revision (`Builder` around the
43
+ "Convert revision with month+year to update component" comment) parses
44
+ into a **separate** attribute, `update`/`update_component`
45
+ (`Components::Update`: number+year+month), which the default list
46
+ missed entirely — `"NBS LC 800 rJun1992"` and `"NBS LC 800 rJul1995"`
47
+ failed to collapse under `#to_all_parts`/`#===`. Fixed with
48
+ `Pubid::Nist::Identifier.all_parts_edition_keys` (`super + %i[update
49
+ update_component]`). **`edition_year` and `revision_year`/
50
+ `revision_month` were investigated and are NOT added**: `Builder` only
51
+ ever sets `edition_year` alongside the real `edition` component (never
52
+ as its sole carrier, e.g. the TechnicalNote "date IS edition" branch),
53
+ and `revision_year`/`revision_month` are transient — converted into
54
+ `update`/`update_component` and cleared to `nil` before the object is
55
+ returned. Neither carries live information `edition`/`update` doesn't
56
+ already cover. Locked by `spec/pubid/all_parts_edition_keys_audit_spec.rb`.
@@ -87,6 +87,7 @@ module Pubid
87
87
  # Note: :base_portion is lost during parser merge, so check for supplement indicators
88
88
  if parsed_hash[:supplement_date_range] || parsed_hash[:supplement_slash_year] ||
89
89
  parsed_hash[:supplement_month_year] || parsed_hash[:supplement_year] ||
90
+ parsed_hash[:supplement_empty] ||
90
91
  parsed_hash[:supplement] || parsed_hash[:base_portion]
91
92
  return build_circular_supplement(parsed_hash)
92
93
  end
@@ -479,3 +480,5 @@ module Pubid
479
480
  end
480
481
  end
481
482
  end
483
+
484
+ Pubid::Nist::Builder.prepend(Pubid::Builder::AllPartsWrap)
@@ -28,6 +28,8 @@ module Pubid
28
28
  # Edition.new(type: "r", id: "5").to_s # => "r5"
29
29
  # Edition.new(type: "r", id: "5", original_prefix: " Rev. ").to_s # => "Rev. 5"
30
30
  class Edition < Lutaml::Model::Serializable
31
+ include ::Pubid::SubsetMatch
32
+
31
33
  attribute :type, :string # "-", "e", or "r"
32
34
  attribute :id, :string # Edition ID (number or year)
33
35
  attribute :additional_text, :string # Text after "rev" (WITHOUT "rev" prefix)
@@ -8,6 +8,8 @@ module Pubid
8
8
  # IssueNumber component for NIST identifiers
9
9
  # Represents the issue/number designation (e.g., "No. 12" in "Vol. 6, No. 12")
10
10
  class IssueNumber < Lutaml::Model::Serializable
11
+ include ::Pubid::SubsetMatch
12
+
11
13
  attribute :number, :string
12
14
 
13
15
  # Short form rendering: "n12"
@@ -23,6 +23,8 @@ module Pubid
23
23
  # - SP: Part number (pt1)
24
24
  # - Letter suffixes (A, B, C, etc.)
25
25
  class Part < Lutaml::Model::Serializable
26
+ include ::Pubid::SubsetMatch
27
+
26
28
  attribute :type, :string # "pt" for part notation, "n" for issue, "" for letter suffix
27
29
  attribute :value, :string # Part number or letter (1, 2, A, B, etc.)
28
30
 
@@ -13,6 +13,8 @@ module Pubid
13
13
  # Stage.new(id: "i", type: "pd").to_s(:short) # => "ipd"
14
14
  # Stage.new(id: "f", type: "pd").to_s(:long) # => "(Final Public Draft)"
15
15
  class Stage < Lutaml::Model::Serializable
16
+ include ::Pubid::SubsetMatch
17
+
16
18
  attribute :id, :string # i, f, 1-9
17
19
  attribute :type, :string # pd, wd, prd
18
20
 
@@ -15,6 +15,8 @@ module Pubid
15
15
  # Supplement.new(month: "Jan", year: "1924").to_s(:short) # => "supJan1924"
16
16
  # Supplement.new(has_revision: true).to_s(:short) # => "suprev"
17
17
  class Supplement < Lutaml::Model::Serializable
18
+ include ::Pubid::SubsetMatch
19
+
18
20
  attribute :number, :string # Supplement number (e.g., "2" in "supp2")
19
21
  attribute :year, :string # Year (4 digits); range START year
20
22
  attribute :month, :string # Month abbreviation; range START month
@@ -13,6 +13,8 @@ module Pubid
13
13
  # Translation.new(code: "por").to_s(:mr) # => ".por"
14
14
  # Translation.new(code: "ind").to_s(:short) # => " ind"
15
15
  class Translation < Lutaml::Model::Serializable
16
+ include ::Pubid::SubsetMatch
17
+
16
18
  attribute :code, :string # 3-letter ISO 639-2 code: spa, por, ind, etc.
17
19
 
18
20
  # Backward compatibility: language method returns code
@@ -15,6 +15,8 @@ module Pubid
15
15
  # Update.new(number: 1, year: 2021, month: 2).to_s(:mr) # => "-upd1-202102"
16
16
  # Update.new(number: 1, prefix: "dash").to_s(:short) # => "-upd1" (preserves original prefix)
17
17
  class Update < Lutaml::Model::Serializable
18
+ include ::Pubid::SubsetMatch
19
+
18
20
  attribute :number, :string # Update number as string
19
21
  attribute :year, :string # Year (4 digits as string)
20
22
  attribute :month, :string # Month (01-12 as string, optional)
@@ -12,6 +12,8 @@ module Pubid
12
12
  # Version.new(value: "1.0.2").to_s(:short) # => "ver1.0.2"
13
13
  # Version.new(value: "2.0").to_s(:long) # => "Version 2.0"
14
14
  class Version < Lutaml::Model::Serializable
15
+ include ::Pubid::SubsetMatch
16
+
15
17
  attribute :value, :string # Dotted notation: "1.0.2"
16
18
 
17
19
  # Render version in specified format
@@ -16,6 +16,8 @@ module Pubid
16
16
  # - CSM (Commercial Standards Monthly): Volume 6, Issue 1
17
17
  # - CIRC (Circular): Volume 539
18
18
  class Volume < Lutaml::Model::Serializable
19
+ include ::Pubid::SubsetMatch
20
+
19
21
  attribute :value, :string
20
22
 
21
23
  def to_s
@@ -223,6 +223,30 @@ module Pubid
223
223
 
224
224
  alias eql? ==
225
225
 
226
+ # A subset match skips the same attributes: `to_hash` drops the build
227
+ # artifacts, so an index row rebuilt by `from_hash` never has them.
228
+ def self.subset_ignored_attributes
229
+ EQUALITY_IGNORED_ATTRS
230
+ end
231
+
232
+ # `edition` is the primary edition/revision carrier and is already in
233
+ # the default %i[date year edition version] list. `update`/
234
+ # `update_component` (Components::Update: number+year+month) is a
235
+ # SEPARATE, currently-live discriminator the default list misses — the
236
+ # Letter Circular / Circular "rJun1992"-style revision parses into it,
237
+ # not into `edition` (see Builder#build_dated_identifier), so e.g.
238
+ # "NBS LC 800 rJun1992" and "NBS LC 800 rJul1995" failed to collapse
239
+ # under #to_all_parts/#=== before this override. (`edition_year` and
240
+ # `revision_year`/`revision_month` are NOT added here: the builder
241
+ # always sets `edition_year` alongside the real `edition` component
242
+ # (never alone), and `revision_year`/`revision_month` are transient —
243
+ # converted into `update`/`update_component` and cleared to nil before
244
+ # the object is returned — so neither carries live information outside
245
+ # what `edition`/`update` already do.)
246
+ def self.all_parts_edition_keys
247
+ super + %i[update update_component]
248
+ end
249
+
226
250
  def hash
227
251
  vals = self.class.attributes.each_key.reject do |name|
228
252
  EQUALITY_IGNORED_ATTRS.include?(name)
@@ -413,8 +437,6 @@ module Pubid
413
437
  match ? match[1].to_i : nil
414
438
  end
415
439
 
416
- public
417
-
418
440
  def to_full_style
419
441
  # "National Institute of Standards and Technology Special Publication 800-27, Revision A"
420
442
  result = publisher_full_name
@@ -674,14 +696,25 @@ module Pubid
674
696
  result += "#{vol_str}n#{issue_number.number}"
675
697
  end
676
698
 
677
- # Use edition component - NO space before edition in MR format (per NIST spec)
678
- result += edition.to_s if edition
699
+ # With a number, the edition glues to it per the NIST spec
700
+ # ("800-53r5"); series-only editions take a dot separator
701
+ # ("NBS.CIRC.e2" — the attested raw spelling; testsuite#5 C4/C5).
702
+ result += if edition
703
+ number ? edition.to_s : ".#{edition}"
704
+ else
705
+ ""
706
+ end
679
707
 
680
708
  # Use version_component
681
709
  result += version_component.to_s(:mr) if version_component
682
710
 
683
- # Supplement (e.g. ".9981sup7") - keep distinct documents distinct
684
- result += supplement_short
711
+ # Supplement (e.g. ".9981sup7") - keep distinct documents distinct;
712
+ # a series-only supplement ("NBS.CIRC.sup") takes the dot itself.
713
+ result += if supplement
714
+ number ? supplement_short : ".#{supplement_short}"
715
+ else
716
+ ""
717
+ end
685
718
 
686
719
  # Use update_component
687
720
  result += update_component.to_s(:mr) if update_component
@@ -746,6 +779,5 @@ module Pubid
746
779
  "NIST"
747
780
  end
748
781
  end
749
-
750
782
  end
751
783
  end
@@ -13,8 +13,20 @@ module Pubid
13
13
  # feeds the cleaned string to the Parslet grammar and stamps the
14
14
  # detected format onto the parse tree.
15
15
  def self.class_parse_with_preprocessing(input)
16
+ # The shared Grammar strips a trailing "(all parts)", but the NIST
17
+ # preprocessor runs first and would swallow it — strip before it and
18
+ # carry the marker to the tree like Grammar#parse does.
19
+ all_parts = input.is_a?(String) && input.match?(::Pubid::Parser::Grammar::ALL_PARTS_SUFFIX)
20
+ input = input.sub(::Pubid::Parser::Grammar::ALL_PARTS_SUFFIX, "") if all_parts
16
21
  result = Preprocessor.new(input).call
17
22
  parsed = new.parse(result.cleaned)
23
+ if all_parts
24
+ parsed = case parsed
25
+ when Hash then parsed.merge(all_parts: true)
26
+ when Array then parsed.map { |h| h.merge(all_parts: true) }
27
+ else parsed
28
+ end
29
+ end
18
30
 
19
31
  if parsed.is_a?(Hash)
20
32
  parsed.merge(parsed_format: result.format)
@@ -651,8 +663,9 @@ module Pubid
651
663
  rule(:mr_identifier) do
652
664
  hash_prefix.maybe >>
653
665
  publisher >> dot >>
654
- simple_series >> dot >>
655
- report_number >>
666
+ # The catalogue also lists bare series identities with no
667
+ # report number ("NBS.CIRC").
668
+ simple_series >> dot.maybe >> report_number.maybe >>
656
669
  # Edition with underscore separator (MR format: 1648_2009)
657
670
  (str("_") >> digits.as(:edition_year)).maybe >>
658
671
  # Support letter suffix before update (e.g., 8286C-upd1) - Session 219
@@ -713,6 +726,15 @@ module Pubid
713
726
  # 4-digit years so it can't swallow "sup3/1926" or a base number.
714
727
  ((str("supp") | str("sup")) >> match("[0-9]").repeat(4, 4).as(:supp_year_start) >>
715
728
  dash >> match("[0-9]").repeat(4, 4).as(:supp_year_end)).as(:supplement_date_range) |
729
+ # Bare supplement marker to the whole series, no base number
730
+ # ("NBS.CIRC.sup" — testsuite#5 C4): the marker alone is the
731
+ # supplement.
732
+ ((str("supp") | str("sup")) >>
733
+ (
734
+ (month_abbrev >> digits).as(:supplement_month_year) |
735
+ (digits.as(:supp_number) >> slash >> digits.as(:supp_year)).as(:supplement_slash_year) |
736
+ str("").as(:supplement_empty)
737
+ ).maybe) |
716
738
  # With base identifier + supplement
717
739
  (
718
740
  # Capture base portion (everything before "supp" or "sup" or slash+year)
@@ -37,6 +37,11 @@ module Pubid
37
37
  def initialize(input)
38
38
  @input = input.to_s.strip
39
39
  @cleaned = Core::UpdateCodes.apply(@input, :nist)
40
+ # The format describes the string the parser will see: capture
41
+ # it right after the update-codes remap, before the stages'
42
+ # cosmetic spacing (a dotted catalogue alias that remaps to the
43
+ # space form renders short; the dotted originals stay :mr).
44
+ @format = @cleaned.include?(".") && !@cleaned.match?(/\s/) ? :mr : :short
40
45
  end
41
46
 
42
47
  # Run every normalization stage and return a Result.
@@ -53,6 +58,14 @@ module Pubid
53
58
  # Extracted so rubocop can scope length/ABC metrics narrowly.
54
59
  # rubocop:disable Metrics/MethodLength, Metrics/AbcSize
55
60
  def run_stages
61
+ # Short-form "supprev" is the catalogue spelling of the plain
62
+ # supplement ("NBS CIRC 154supprev" ≡ "NBS CIRC 154sup"; the
63
+ # revision-bearing identity is the mr spelling "154suprev").
64
+ # Rewrite before the supplement/revision stages; the mr form is
65
+ # untouched.
66
+ if detected_format == :short
67
+ @cleaned = @cleaned.gsub("supprev", "sup")
68
+ end
56
69
  normalize_spurious_u_suffix!
57
70
  normalize_publisher_and_series!
58
71
  normalize_lcirc_supplement_contexts!
@@ -75,12 +88,14 @@ module Pubid
75
88
  normalize_part_notation!
76
89
  normalize_series_specific_spacing!
77
90
  normalize_verbose_keywords!
91
+ normalize_legacy_corpus_spellings!
78
92
  end
79
93
  # rubocop:enable Metrics/MethodLength, Metrics/AbcSize
80
94
 
81
- # Detect input format: :mr (dot-separated machine-readable) or :short.
95
+ # Detect input format: :mr (dot-separated machine-readable) or
96
+ # :short. Frozen in #initialize (post update-codes, pre-stages).
82
97
  def detected_format
83
- @input.include?(".") && !@input.match?(/\s/) ? :mr : :short
98
+ @format
84
99
  end
85
100
 
86
101
  private
@@ -167,12 +182,21 @@ module Pubid
167
182
  # Trailing "-a" → "-A" at end of identifier.
168
183
  def uppercase_dash_letter!
169
184
  @cleaned = @cleaned.gsub(/(\d)-([a-z])$/) { "#{$1}-#{$2.upcase}" }
185
+ # Same letter when a part tail follows ("-add", " Add."), so the
186
+ # canonical case survives; lowercase update/translation codes
187
+ # ("-upd", ".uppl") never match (next char is a letter).
188
+ @cleaned = @cleaned.gsub(/(\d)-([a-z])(?=[-.\s])/) { "#{$1}-#{$2.upcase}" }
170
189
  end
171
190
 
172
191
  # Trailing "a" → "A" when attached directly to a digit (excludes
173
192
  # "r" to preserve revision+year patterns like "73-197r").
174
193
  def uppercase_trailing_letter!
175
194
  @cleaned = @cleaned.gsub(/(\d)([a-z&&[^r]])$/) { "#{$1}#{$2.upcase}" }
195
+ # Same letter when an addendum tail follows ("-add", " Add."),
196
+ # so the canonical part case survives ("800-38a-add" ->
197
+ # "800-38A"); part digits ("800-85a-1") and update/translation
198
+ # codes ("-upd", ".uppl") never match.
199
+ @cleaned = @cleaned.gsub(/(\d)([a-z&&[^r]])(?=\s*[-.\s]\s*[aA]dd)/) { "#{$1}#{$2.upcase}" }
176
200
  end
177
201
 
178
202
  # Letter suffix on revision: "22r1a" → "22r1A".
@@ -336,6 +360,33 @@ module Pubid
336
360
  end
337
361
  end
338
362
 
363
+ # Legacy corpus spellings: catalogue forms the mr grammar and the
364
+ # renderers cannot round-trip on their own. Each rule mirrors an
365
+ # already-parseable spelling of the same document.
366
+ def normalize_legacy_corpus_spellings!
367
+ # The mr renderer prints the translation code with its own
368
+ # leading dot ("955-S..uppl"); the parser wants one dot.
369
+ @cleaned = @cleaned.gsub("..", ".")
370
+ # Trailing-dot addendum spelling; the canonical render comes from
371
+ # the lowercase ".add" parse ("NBS.TN.467pt1.Add." -> ".add").
372
+ @cleaned = @cleaned.gsub(/\.Add\.\z/, ".add")
373
+ # Edition glued to the series in mr form ("NBS.CIRCe2" is
374
+ # "NBS.CIRC.e2"; digits before "e" are untouched: "24e7", and
375
+ # the short form never glues an edition to a letter part:
376
+ # "150-1Ae2009").
377
+ if detected_format == :mr
378
+ @cleaned = @cleaned.gsub(/([A-Z])e(\d)/, '\1.e\2')
379
+ end
380
+ # Update markers without a number render a phantom "1";
381
+ # "…-upd" is the numberless spelling of "…-upd1".
382
+ @cleaned = @cleaned.gsub(/-upd\z/, "-upd1")
383
+ # Handbook legacy renumbering in the all-dash spelling: the
384
+ # e-form stage already eats the prefix generically ("HB
385
+ # 150-1e2017" -> "HB 1-2017"); the all-dash form carries the
386
+ # same 105-/150- prefixes.
387
+ @cleaned = @cleaned.gsub(/\b(NIST HB\s+)(?:105|150)-(\d+)-(\d{4})(?=\s|\z)/, '\1\2-\3')
388
+ end
389
+
339
390
  # Series-specific reverts: HB handbooks, OWMWP dates, and RPT year
340
391
  # ranges use dash-year structurally (not as an edition marker), so
341
392
  # the broad convert_dashyear_to_edition! rule would corrupt them.
@@ -42,13 +42,22 @@ module Pubid
42
42
  type_token, payload = parts
43
43
  code, revision = parse_payload(payload)
44
44
 
45
- text = "NIST #{type_label(type_token)} #{code}"
45
+ text = "#{publisher_for(type_token)} #{type_label(type_token)} #{code}"
46
46
  text += revision if revision
47
47
  flavor_parse(text)
48
48
  end
49
49
 
50
50
  private
51
51
 
52
+ # The NBS-era series carry the NBS imprint in their canonical form
53
+ # ("NBS CSM 1") — the rebuild must name the publisher the document
54
+ # carries, or the flavor parse rejects its own URN's rebuild.
55
+ NBS_SERIES = ["csm"].freeze
56
+
57
+ def publisher_for(type_token)
58
+ NBS_SERIES.include?(type_token.downcase) ? "NBS" : "NIST"
59
+ end
60
+
52
61
  def parse_payload(payload)
53
62
  # Strip the trailing ".supp" supplement marker.
54
63
  stripped = payload.sub(/\.supp\z/, "")
@@ -0,0 +1,19 @@
1
+ # OASIS flavor notes
2
+
3
+ OASIS verbatim slugs, the index key, the MR slug and partial-reference matching.
4
+
5
+ Read them before you change `lib/pubid/oasis/` or `spec/pubid/oasis/`. The root file keeps the cross-flavor contract that every flavor obeys.
6
+
7
+ - **The identity is the verbatim slug; the decomposition is a lossy projection of it**: an OASIS identifier is a free-form slug with no fixed internal order (`OSLC-CoreShapes-3.0-PS01-Pt8`, `OSLC-AM-3.0-Part1-PS01`, `amqp-core`), three part spellings and mixed case. The grammar therefore captures the whole slug with `any.repeat(1)` and only strips the `OASIS ` prefix; `Builder#decompose` then classifies each *whole* dash-separated fragment into `number` / `version` / `stage` / `part` / `label`. **`original` holds the printed slug verbatim and alone drives `to_s` and `to_urn`** — `Renderer#render` is a pure echo of it — so the printed form round-trips byte-exactly whatever the classifier makes of it. The decomposition is **not** an identity and cannot be one: it loses fragment order (`x-1.0-os-Pt1` and `x-1.0-Pt1-os` decompose alike) and keeps only the first fragment of each recognized kind, so a repeated one is dropped. That is not hypothetical — **7 of the 605 published `relaton-data-oasis` ids already lose a fragment** (`OASIS xacml-3.0-hierarchical-v1.0-CS02` keeps `3.0` and drops `v1.0`). Every design decision below follows from that asymmetry.
8
+
9
+ - **`spec` → `number`, the index key (PR #359)**: `Relaton::Index::Type#candidates_by_number` sorts every row and binary-searches it on `id.root.number.to_s`. OASIS kept the specification name in a bespoke `attribute :spec, :string` and never set the `number` it inherits from `::Pubid::Identifier`, so **all 605 published rows shared the empty key `""`** and the search degraded to a linear scan, silently. The attribute is **renamed**, not shadowed by a derived reader: `attribute :number, :string` plus `map "number", to: :number`, and `Builder#decompose` returns `number:`. **`spec` is dropped with no alias and no reader** — one name for one value, the W3C `code` → `number` precedent. A derived `#spec` was written first and then removed: nothing in `lib/` read it, and `relaton-oasis` has no pubid dependency at all, so it had no consumer to preserve. That is what separates this case from IANA's `#registry`, IETF's `#series` and BIPM's `#volume`, which do have one. `root_number_spec.rb` asserts the identifier does **not** respond to `spec`, so the name cannot creep back. The key **clusters**: every version, stage and part of one specification shares it — 309 buckets over 605 ids, median 1, max 25 (`STIX`) — the shape of an IETF draft slug or an IANA registry slug, not an exact key.
10
+
11
+ - **Why the `number` declaration sits on the base here**: `attribute :number, :string` redefines the parent's `Components::Code number` on a class `Identifiers::Standard` inherits from, which is the recorded determinism landmine. It is safe for the same reason W3C's is: lutaml deep-dups the parent attribute table into each subclass at class-definition time, so a subclass holds a snapshot, and Ruby resolves the superclass constant to completion before opening a leaf body. **`Pubid::Oasis::Identifier`'s class body lives in one file and is never reopened**, so the snapshot is always complete. `stage` and `part` in that same body have relied on this since the flavor landed, overriding `Components::Stage` and `Components::Code`. **Split this class across two files and the landmine comes back** — IEEE is the counter-shape, its base reachable through two paths. `spec/pubid/oasis/root_number_spec.rb` carries the structural tripwire (the base *and* the leaf must resolve `number` to `Lutaml::Model::Type::String`) and is only meaningful under the full `bundle exec rake`.
12
+
13
+ - **MR slug**: OASIS supplied no `mr_*` hook, and every base hook looks for something the flavor does not use — the publisher is the `PUBLISHER` constant rather than a lutaml attribute, there is no `Components::Date` and no `typed_stage`. With `number` nil as well, **`to_mr_string` was `""` for all 605 ids**, and `to_slug` is what consumers use as an output **filename**. The base now supplies `mr_publisher` (`"oasis"`) and `mr_number_with_part`, which returns the sanitized **`original`** — not the five decomposed fields, because those assemble `x-1.0-os-Pt1` and `x-1.0-Pt1-os` into one slug, while `original` cannot collide. A private `mr_sanitize` filters **by charset**, so a field added later cannot leak an unsafe character. It uses **BIPM's `[^a-z0-9]+` form, not ETSI's `[^a-z0-9-]+`**: a run of non-alphanumerics collapses to one `-`, without which `v3.0]-PS01` prints a bare `--`. Three characters need the filter today — the `.` in every version (`Renderers::MrString` joins *segments* with `.`, so a dot inside one breaks the documented structure), plus the space and `]` of a few malformed records. Result: **605 distinct slugs for the 605 published ids, 0 raises, 0 characters outside `[a-z0-9._-]`**. One caveat, the same one BIPM records: `-` is both the intra-slug join and the substitute, so two slugs differing only in which non-slug character they use collapse. The corpus has exactly one such pair — the malformed `OpenC2-MQTT-v1.0] -CS01` spelled once with a normal space and once with a non-breaking one, two spellings of one document.
14
+
15
+ - **`#exclude` clears `original`, so a partial reference can widen**: `#matches?` is `exclude(*ignore) == other.exclude(*ignore)`, and `original` spells out verbatim the very component the caller is ignoring. So it survived the exclusion and a bare `OASIS WSDM` could **never** match `OASIS WSDM-v1.1`, however much was ignored — relaton would narrow to the right index bucket and then match nothing in it. `Pubid::Oasis::Identifier#exclude` now nils `original` when the excluded set intersects `DECOMPOSITION_KEYS` (`number version stage part label`), the "reset the whole cluster" rule CSA applies to its year-format siblings. **A plain `==` still compares `original`, deliberately**: an excluded copy is a comparison token, never a rendered document, so widening happens only when the caller asks for it. Dropping `original` from `==` outright was considered and rejected — it costs nothing on today's corpus (no two ids collide) but it exposes those 7 lossy ids to a silent collision with a future sibling, and it would let two different printed identifiers be equal. `spec/pubid/oasis/partial_ref_spec.rb` locks both halves.
16
+
17
+ - **relaton note**: nothing published needs migrating. `relaton-oasis` has no pubid dependency and its `index-v1.yaml` is string-keyed (`:id: OASIS amqp-core`), which is why the serialized shape was changed properly rather than patched. There is **no alias** for the old `spec` key, and a pre-`number` row would deserialize with a nil `number` and no error (lutaml ignores unknown keys), so a future pubid index must be crawled *after* this. Rendering is the loud failure if it is not.
18
+
19
+ - **Verification and the fixture net**: the change was verified by replaying a `main` baseline over all 605 published ids plus the fixture lines — `to_s`, `to_urn`, identifier class and parse errors **byte-identical**, `to_hash` differing **only** by the key rename, empty index keys 607 → 0, distinct MR slugs 1 → 606 of 607, and `from_hash(to_hash) == id` for all 605 with **0** pairs of distinct ids equal. `spec/pubid/oasis/fixtures_spec.rb` is live — its glob is correct, so OASIS is not on the `ten-dead-fixture-specs` list — and `root_number_spec.rb` adds a corpus sweep over the same file. **Do not run `rake "validation:classify[oasis]"`**: there is no `spec/fixtures/oasis/identifiers/full/`, so the generator would delete `pass/oasis.txt` and rebuild nothing. (hand-off: oasis-index-number.)
@@ -77,3 +77,5 @@ module Pubid
77
77
  end
78
78
  end
79
79
  end
80
+
81
+ Pubid::Oasis::Builder.prepend(Pubid::Builder::AllPartsWrap)
@@ -36,6 +36,18 @@ module Pubid
36
36
  # building the table before `Identifiers::Standard` snapshots it — DO NOT
37
37
  # split this class across two files.
38
38
  attribute :number, :string
39
+
40
+ # `original` alone drives #to_s, and the shared exclude-copy loses it
41
+ # (the redefined attribute does not survive the rebuild). An all-parts
42
+ # copy is therefore dup-based: nil the part and edition-ish attributes
43
+ # in place, keeping the printed slug verbatim.
44
+ def without_parts(*extra)
45
+ copy = dup
46
+ (::Pubid::Identifier::PART_ATTRIBUTES + extra).each do |name|
47
+ copy.public_send(:"#{name}=", nil) if copy.respond_to?(:"#{name}=")
48
+ end
49
+ copy
50
+ end
39
51
  attribute :version, :string
40
52
  attribute :stage, :string
41
53
  attribute :part, :string
@@ -91,6 +103,13 @@ module Pubid
91
103
  result
92
104
  end
93
105
 
106
+ # The same reason applies to a subset match: `original` spells out the
107
+ # parts a partial reference omits, so `===` compares the decomposition
108
+ # only. Two slugs that decompose alike therefore match each other.
109
+ def self.subset_ignored_attributes
110
+ %i[original]
111
+ end
112
+
94
113
  # MR string hooks. `to_slug` delegates to `to_mr_string` and consumers
95
114
  # use it as an output FILENAME, so the slug must be unique per document.
96
115
  #
@@ -127,7 +146,7 @@ module Pubid
127
146
  raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
128
147
  end
129
148
 
130
- parsed = Parser.parse(identifier)
149
+ parsed = Pubid::Parg::Backend.parse(:oasis, identifier)
131
150
  Builder.build(parsed)
132
151
  end
133
152