pubid 2.0.0.pre.alpha.12 → 2.0.0.pre.alpha.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.adoc +43 -1
- data/data/ieee/update_codes.yaml +17 -4
- data/data/nist/update_codes.yaml +7 -3
- data/data/parg/tables/bipm_groups.yaml +14 -0
- data/data/parg/tables/bipm_type_codes.yaml +5 -0
- data/data/parg/tables/bipm_type_names_en.yaml +6 -0
- data/data/parg/tables/bipm_type_names_fr.yaml +6 -0
- data/data/parg/tables/directives_supplements_typed_stages.yaml +3 -0
- data/data/parg/tables/directives_typed_stages.yaml +5 -0
- data/data/parg/tables/idf_typed_stages.yaml +27 -0
- data/data/parg/tables/idf_typed_stages_supplements.yaml +2 -0
- data/data/parg/tables/iec_typed_stages.yaml +130 -0
- data/data/parg/tables/iso_publishers.yaml +4 -0
- data/data/parg/tables/organizations.yaml +12 -0
- data/data/parg/tables/tc_types.yaml +42 -0
- data/data/parg/tables/typed_stages.yaml +114 -0
- data/data/parg/tables/typed_stages_supplements.yaml +64 -0
- data/data/parg/tables/wg_types.yaml +21 -0
- data/lib/pubid/adobe/builder.rb +2 -0
- data/lib/pubid/adobe/identifier.rb +11 -1
- data/lib/pubid/all_parts.rb +201 -0
- data/lib/pubid/all_parts_identifier.rb +19 -0
- data/lib/pubid/amca/CLAUDE.md +47 -0
- data/lib/pubid/amca/builder.rb +3 -5
- data/lib/pubid/amca/identifiers/base.rb +11 -1
- data/lib/pubid/amca/identifiers/publication.rb +13 -0
- data/lib/pubid/amca/parser.rb +2 -1
- data/lib/pubid/amca/renderer.rb +22 -33
- data/lib/pubid/amca/urn_generator.rb +21 -2
- data/lib/pubid/amca/urn_parser.rb +36 -10
- data/lib/pubid/ansi/builder.rb +6 -0
- data/lib/pubid/ansi/identifier.rb +1 -1
- data/lib/pubid/api/CLAUDE.md +23 -0
- data/lib/pubid/api/builder.rb +2 -0
- data/lib/pubid/api/identifier.rb +1 -1
- data/lib/pubid/api/parser.rb +8 -4
- data/lib/pubid/ashrae/CLAUDE.md +13 -0
- data/lib/pubid/ashrae/builder.rb +58 -14
- data/lib/pubid/ashrae/identifiers/base.rb +10 -1
- data/lib/pubid/ashrae/identifiers/errata.rb +14 -2
- data/lib/pubid/ashrae/identifiers/interpretation.rb +2 -10
- data/lib/pubid/ashrae/parser.rb +80 -39
- data/lib/pubid/ashrae/renderer.rb +32 -1
- data/lib/pubid/ashrae/urn_generator.rb +32 -9
- data/lib/pubid/asme/CLAUDE.md +25 -0
- data/lib/pubid/asme/builder.rb +16 -9
- data/lib/pubid/asme/components/code.rb +2 -0
- data/lib/pubid/asme/identifier.rb +1 -1
- data/lib/pubid/asme/identifiers/standard.rb +6 -1
- data/lib/pubid/asme/parser.rb +41 -14
- data/lib/pubid/astm/CLAUDE.md +9 -0
- data/lib/pubid/astm/builder.rb +2 -0
- data/lib/pubid/astm/components/code.rb +2 -0
- data/lib/pubid/astm/identifier.rb +1 -1
- data/lib/pubid/astm/parser.rb +4 -1
- data/lib/pubid/bipm/CLAUDE.md +11 -0
- data/lib/pubid/bipm/builder.rb +2 -0
- data/lib/pubid/bipm/identifier.rb +1 -1
- data/lib/pubid/bsi/CLAUDE.md +93 -0
- data/lib/pubid/bsi/builder.rb +13 -11
- data/lib/pubid/bsi/identifiers/addendum_document.rb +2 -0
- data/lib/pubid/bsi/identifiers/adopted_european_norm.rb +6 -54
- data/lib/pubid/bsi/identifiers/adopted_international_standard.rb +5 -22
- data/lib/pubid/bsi/identifiers/amendment.rb +36 -12
- data/lib/pubid/bsi/identifiers/bundled_identifier.rb +2 -0
- data/lib/pubid/bsi/identifiers/consolidated_identifier.rb +23 -26
- data/lib/pubid/bsi/identifiers/corrigendum.rb +29 -12
- data/lib/pubid/bsi/identifiers/expert_commentary.rb +6 -7
- data/lib/pubid/bsi/identifiers/national_annex.rb +18 -20
- data/lib/pubid/bsi/identifiers/root_identity.rb +31 -0
- data/lib/pubid/bsi/identifiers/set.rb +2 -0
- data/lib/pubid/bsi/identifiers/supplement_document.rb +2 -0
- data/lib/pubid/bsi/identifiers.rb +1 -0
- data/lib/pubid/bsi/parser.rb +8 -8
- data/lib/pubid/bsi/renderer.rb +20 -20
- data/lib/pubid/bsi/single_identifier.rb +1 -3
- data/lib/pubid/bsi/urn_generator.rb +28 -18
- data/lib/pubid/builder/base.rb +27 -0
- data/lib/pubid/calconnect/builder.rb +2 -0
- data/lib/pubid/calconnect/identifier.rb +5 -1
- data/lib/pubid/ccsds/builder.rb +2 -0
- data/lib/pubid/ccsds/identifier.rb +9 -1
- data/lib/pubid/cen_cenelec/CLAUDE.md +59 -0
- data/lib/pubid/cen_cenelec/builder.rb +6 -1
- data/lib/pubid/cen_cenelec/identifier.rb +12 -28
- data/lib/pubid/cen_cenelec/identifiers/amendment.rb +3 -10
- data/lib/pubid/cen_cenelec/identifiers/corrigendum.rb +3 -10
- data/lib/pubid/cen_cenelec/parser.rb +20 -5
- data/lib/pubid/cie/CLAUDE.md +58 -0
- data/lib/pubid/cie/builder.rb +2 -0
- data/lib/pubid/cie/components/language.rb +2 -0
- data/lib/pubid/cie/identifier.rb +1 -1
- data/lib/pubid/cie/parser.rb +9 -2
- data/lib/pubid/components/adoption.rb +2 -0
- data/lib/pubid/components/code.rb +2 -0
- data/lib/pubid/components/date.rb +8 -6
- data/lib/pubid/components/edition.rb +2 -0
- data/lib/pubid/components/iteration.rb +2 -0
- data/lib/pubid/components/language.rb +2 -0
- data/lib/pubid/components/locality.rb +2 -0
- data/lib/pubid/components/publisher.rb +2 -0
- data/lib/pubid/components/relationship.rb +2 -0
- data/lib/pubid/components/stage.rb +2 -0
- data/lib/pubid/components/supplement.rb +2 -0
- data/lib/pubid/components/type.rb +2 -0
- data/lib/pubid/components/typed_stage.rb +8 -0
- data/lib/pubid/conformance/checks.rb +1 -1
- data/lib/pubid/csa/CLAUDE.md +41 -0
- data/lib/pubid/csa/builder.rb +2 -0
- data/lib/pubid/csa/identifier.rb +19 -3
- data/lib/pubid/csa/parser.rb +25 -8
- data/lib/pubid/csa/renderer.rb +12 -12
- data/lib/pubid/csa/single_identifier.rb +17 -0
- data/lib/pubid/doi/builder.rb +2 -0
- data/lib/pubid/doi/identifier.rb +1 -1
- data/lib/pubid/easc/builder.rb +2 -0
- data/lib/pubid/easc/identifier.rb +10 -1
- data/lib/pubid/ecma/CLAUDE.md +28 -0
- data/lib/pubid/ecma/builder.rb +2 -0
- data/lib/pubid/ecma/identifier.rb +8 -1
- data/lib/pubid/etsi/CLAUDE.md +34 -0
- data/lib/pubid/etsi/builder.rb +2 -0
- data/lib/pubid/etsi/components/code.rb +6 -0
- data/lib/pubid/etsi/components/version.rb +2 -0
- data/lib/pubid/etsi/identifiers/base.rb +1 -1
- data/lib/pubid/etsi/identifiers/etsi_standard.rb +7 -0
- data/lib/pubid/evs/CLAUDE.md +58 -0
- data/lib/pubid/evs/builder.rb +2 -0
- data/lib/pubid/evs.rb +1 -1
- data/lib/pubid/gb/CLAUDE.md +140 -0
- data/lib/pubid/gb/builder.rb +7 -2
- data/lib/pubid/gb/identifier.rb +6 -4
- data/lib/pubid/gb/identifiers/all_parts.rb +17 -0
- data/lib/pubid/gb/identifiers.rb +1 -0
- data/lib/pubid/gb/renderer.rb +0 -1
- data/lib/pubid/gost/CLAUDE.md +64 -0
- data/lib/pubid/gost/builder.rb +3 -1
- data/lib/pubid/gost/identifier.rb +16 -1
- data/lib/pubid/gost/parser.rb +8 -1
- data/lib/pubid/iala/CLAUDE.md +82 -0
- data/lib/pubid/iala/builder.rb +2 -0
- data/lib/pubid/iala/identifier.rb +10 -1
- data/lib/pubid/iana/CLAUDE.md +7 -0
- data/lib/pubid/iana/builder.rb +2 -0
- data/lib/pubid/iana/identifier.rb +1 -1
- data/lib/pubid/identifier.rb +161 -17
- data/lib/pubid/idf/builder.rb +11 -1
- data/lib/pubid/idf/identifier.rb +5 -0
- data/lib/pubid/idf/identifiers/all_parts.rb +17 -0
- data/lib/pubid/idf/identifiers.rb +1 -0
- data/lib/pubid/iec/CLAUDE.md +31 -0
- data/lib/pubid/iec/builder.rb +7 -1
- data/lib/pubid/iec/components/consolidated_amendment.rb +4 -0
- data/lib/pubid/iec/components/sheet.rb +2 -0
- data/lib/pubid/iec/components/trf_info.rb +2 -0
- data/lib/pubid/iec/components/vap_suffix.rb +2 -0
- data/lib/pubid/iec/identifier.rb +8 -3
- data/lib/pubid/iec/identifiers/all_parts.rb +19 -0
- data/lib/pubid/iec/identifiers.rb +1 -0
- data/lib/pubid/iec/parser.rb +9 -4
- data/lib/pubid/iec/renderer.rb +0 -1
- data/lib/pubid/iec/urn_generator.rb +9 -1
- data/lib/pubid/iec/urn_parser.rb +3 -2
- data/lib/pubid/ieee/CLAUDE.md +97 -0
- data/lib/pubid/ieee/builder.rb +194 -27
- data/lib/pubid/ieee/components/code.rb +2 -0
- data/lib/pubid/ieee/components/draft.rb +35 -2
- data/lib/pubid/ieee/components/typed_stage.rb +2 -0
- data/lib/pubid/ieee/identifiers/base.rb +21 -1
- data/lib/pubid/ieee/identifiers/iec_ieee_copublished.rb +9 -0
- data/lib/pubid/ieee/identifiers/joint_development.rb +66 -23
- data/lib/pubid/ieee/identifiers/project_draft_identifier.rb +8 -1
- data/lib/pubid/ieee/parser.rb +156 -29
- data/lib/pubid/ieee/renderer.rb +42 -7
- data/lib/pubid/ieee/urn_generator.rb +31 -0
- data/lib/pubid/ietf/CLAUDE.md +7 -0
- data/lib/pubid/ietf/builder.rb +2 -0
- data/lib/pubid/ietf/identifiers/base.rb +1 -1
- data/lib/pubid/iho/builder.rb +2 -0
- data/lib/pubid/isbn/builder.rb +2 -0
- data/lib/pubid/isbn/identifier.rb +1 -1
- data/lib/pubid/iso/CLAUDE.md +47 -0
- data/lib/pubid/iso/builder.rb +19 -5
- data/lib/pubid/iso/components/publisher.rb +2 -0
- data/lib/pubid/iso/identifier.rb +10 -15
- data/lib/pubid/iso/identifiers/all_parts.rb +19 -0
- data/lib/pubid/iso/identifiers/directives_supplement.rb +4 -2
- data/lib/pubid/iso/identifiers.rb +1 -0
- data/lib/pubid/iso/normalizer.rb +4 -1
- data/lib/pubid/iso/rendering_style.rb +0 -1
- data/lib/pubid/itu/CLAUDE.md +115 -0
- data/lib/pubid/itu/builder.rb +26 -4
- data/lib/pubid/itu/components/code.rb +2 -0
- data/lib/pubid/itu/components/designation.rb +2 -0
- data/lib/pubid/itu/components/sector.rb +2 -0
- data/lib/pubid/itu/components/series.rb +2 -0
- data/lib/pubid/itu/identifiers/base.rb +11 -18
- data/lib/pubid/itu/identifiers/radio_regulations.rb +27 -0
- data/lib/pubid/itu/identifiers/special_publication.rb +48 -14
- data/lib/pubid/itu/identifiers/standard_serialization.rb +2 -0
- data/lib/pubid/itu/identifiers/supplement.rb +15 -0
- data/lib/pubid/itu/identifiers.rb +1 -0
- data/lib/pubid/itu/parser.rb +108 -22
- data/lib/pubid/itu/urn_generator.rb +9 -2
- data/lib/pubid/jcgm/CLAUDE.md +7 -0
- data/lib/pubid/jcgm/builder.rb +2 -0
- data/lib/pubid/jcgm/components/publisher.rb +2 -0
- data/lib/pubid/jcgm.rb +1 -1
- data/lib/pubid/jis/builder.rb +5 -1
- data/lib/pubid/jis/identifier.rb +6 -18
- data/lib/pubid/jis/identifiers/all_parts.rb +19 -0
- data/lib/pubid/jis/identifiers.rb +1 -0
- data/lib/pubid/jis/renderer.rb +0 -2
- data/lib/pubid/jis/urn_generator.rb +0 -1
- data/lib/pubid/nist/CLAUDE.md +56 -0
- data/lib/pubid/nist/builder.rb +3 -0
- data/lib/pubid/nist/components/edition.rb +2 -0
- data/lib/pubid/nist/components/issue_number.rb +2 -0
- data/lib/pubid/nist/components/part.rb +2 -0
- data/lib/pubid/nist/components/stage.rb +2 -0
- data/lib/pubid/nist/components/supplement.rb +2 -0
- data/lib/pubid/nist/components/translation.rb +2 -0
- data/lib/pubid/nist/components/update.rb +2 -0
- data/lib/pubid/nist/components/version.rb +2 -0
- data/lib/pubid/nist/components/volume.rb +2 -0
- data/lib/pubid/nist/identifiers/base.rb +39 -7
- data/lib/pubid/nist/parser.rb +24 -2
- data/lib/pubid/nist/preprocessor.rb +53 -2
- data/lib/pubid/nist/urn_parser.rb +10 -1
- data/lib/pubid/oasis/CLAUDE.md +19 -0
- data/lib/pubid/oasis/builder.rb +2 -0
- data/lib/pubid/oasis/identifier.rb +20 -1
- data/lib/pubid/ogc/CLAUDE.md +34 -0
- data/lib/pubid/ogc/builder.rb +2 -0
- data/lib/pubid/ogc/identifier.rb +12 -1
- data/lib/pubid/oiml/CLAUDE.md +189 -0
- data/lib/pubid/oiml/builder.rb +20 -0
- data/lib/pubid/oiml/components/code.rb +6 -0
- data/lib/pubid/oiml/identifier.rb +13 -0
- data/lib/pubid/oiml/identifiers/annex.rb +4 -0
- data/lib/pubid/oiml/identifiers/certification_system.rb +34 -0
- data/lib/pubid/oiml/identifiers/code_number.rb +8 -0
- data/lib/pubid/oiml/identifiers/dual_published.rb +174 -0
- data/lib/pubid/oiml/identifiers.rb +2 -0
- data/lib/pubid/oiml/parser.rb +35 -4
- data/lib/pubid/oiml/renderer.rb +23 -1
- data/lib/pubid/oiml/single_identifier.rb +4 -0
- data/lib/pubid/oiml/supplement_identifier.rb +7 -0
- data/lib/pubid/oiml/urn_generator.rb +28 -0
- data/lib/pubid/oiml.rb +6 -1
- data/lib/pubid/omg/CLAUDE.md +15 -0
- data/lib/pubid/omg/builder.rb +2 -0
- data/lib/pubid/omg/identifier.rb +1 -1
- data/lib/pubid/parg/artifact.rb +46 -0
- data/lib/pubid/parg/backend.rb +92 -0
- data/lib/pubid/parg.rb +8 -0
- data/lib/pubid/parser/grammar.rb +23 -0
- data/lib/pubid/pg.rb +8 -0
- data/lib/pubid/plateau/builder.rb +2 -0
- data/lib/pubid/plateau/identifiers/base.rb +4 -0
- data/lib/pubid/plateau/supplement_identifier.rb +14 -2
- data/lib/pubid/plateau/urn_generator.rb +7 -1
- data/lib/pubid/plateau.rb +1 -2
- data/lib/pubid/renderers/human_readable.rb +0 -1
- data/lib/pubid/sae/builder.rb +2 -0
- data/lib/pubid/sae/components/date.rb +2 -0
- data/lib/pubid/sae/components/type.rb +2 -0
- data/lib/pubid/sae/identifiers/base.rb +1 -1
- data/lib/pubid/subset_match.rb +197 -0
- data/lib/pubid/tgpp/CLAUDE.md +43 -0
- data/lib/pubid/tgpp/builder.rb +2 -0
- data/lib/pubid/tgpp/identifier.rb +15 -1
- data/lib/pubid/type_resolver.rb +14 -2
- data/lib/pubid/un/builder.rb +2 -0
- data/lib/pubid/un/identifier.rb +1 -1
- data/lib/pubid/version.rb +1 -1
- data/lib/pubid/w3c/CLAUDE.md +7 -0
- data/lib/pubid/w3c/builder.rb +2 -0
- data/lib/pubid/w3c/identifier.rb +1 -1
- data/lib/pubid/xsf/CLAUDE.md +11 -0
- data/lib/pubid/xsf/builder.rb +2 -0
- data/lib/pubid/xsf/identifier.rb +1 -1
- data/lib/pubid.rb +17 -3
- metadata +78 -2
data/lib/pubid/jis/identifier.rb
CHANGED
|
@@ -7,6 +7,11 @@ module Pubid
|
|
|
7
7
|
# Pubid::Jis::Identifiers descend from this class, so a parsed JIS id is an
|
|
8
8
|
# instance of Pubid::Jis::Identifier.
|
|
9
9
|
class Identifier < ::Pubid::Identifier
|
|
10
|
+
# JIS prints its own all-parts suffix, "(規格群)".
|
|
11
|
+
def self.all_parts_class
|
|
12
|
+
Identifiers::AllParts
|
|
13
|
+
end
|
|
14
|
+
|
|
10
15
|
# JIS keeps its number flat at the top level (string, to preserve leading
|
|
11
16
|
# zeros like "0205"), with the division letter in `series` and any
|
|
12
17
|
# multi-level part numbers in `parts`. Supplements override `number` with
|
|
@@ -16,9 +21,6 @@ module Pubid
|
|
|
16
21
|
attribute :parts, :string, collection: true # Optional multi-level parts
|
|
17
22
|
attribute :year, :integer
|
|
18
23
|
attribute :language, :string # "E" or "J"
|
|
19
|
-
# Boolean flags carry no default, so they stay nil (and are omitted from
|
|
20
|
-
# the serialized hash) unless actually set true.
|
|
21
|
-
attribute :all_parts, :boolean
|
|
22
24
|
# Reaffirmation (再確認): a trailing "R" on the year marks an edition
|
|
23
25
|
# that was reaffirmed without revision (e.g. ":2019R").
|
|
24
26
|
attribute :reaffirmed, :boolean
|
|
@@ -50,7 +52,6 @@ module Pubid
|
|
|
50
52
|
map "parts", to: :parts
|
|
51
53
|
map "year", to: :year
|
|
52
54
|
map "language", to: :language
|
|
53
|
-
map "all_parts", to: :all_parts
|
|
54
55
|
map "reaffirmed", to: :reaffirmed
|
|
55
56
|
# render_empty keeps a bare "SYMBOL" (empty-string value) in the hash so
|
|
56
57
|
# it round-trips distinctly from "no symbol" (nil).
|
|
@@ -62,10 +63,6 @@ module Pubid
|
|
|
62
63
|
# would otherwise fail serialization type validation.
|
|
63
64
|
PUBLISHER = "JIS"
|
|
64
65
|
|
|
65
|
-
def all_parts?
|
|
66
|
-
all_parts == true
|
|
67
|
-
end
|
|
68
|
-
|
|
69
66
|
def reaffirmed?
|
|
70
67
|
reaffirmed == true
|
|
71
68
|
end
|
|
@@ -91,23 +88,14 @@ module Pubid
|
|
|
91
88
|
result
|
|
92
89
|
end
|
|
93
90
|
|
|
94
|
-
# Comparison with all_parts logic
|
|
95
|
-
# When either identifier has all_parts=true, compare only series and number
|
|
96
91
|
def ==(other)
|
|
97
92
|
return false unless other.is_a?(Identifier)
|
|
98
93
|
|
|
99
|
-
if all_parts? || other.all_parts?
|
|
100
|
-
# Compare only series and number, ignore year, parts, all_parts
|
|
101
|
-
return series == other.series && number == other.number
|
|
102
|
-
end
|
|
103
|
-
|
|
104
|
-
# Normal full comparison
|
|
105
94
|
series == other.series &&
|
|
106
95
|
number == other.number &&
|
|
107
96
|
(parts || []) == (other.parts || []) &&
|
|
108
97
|
year == other.year &&
|
|
109
98
|
language == other.language &&
|
|
110
|
-
all_parts? == other.all_parts? &&
|
|
111
99
|
reaffirmed? == other.reaffirmed? &&
|
|
112
100
|
symbol == other.symbol
|
|
113
101
|
end
|
|
@@ -186,7 +174,7 @@ module Pubid
|
|
|
186
174
|
raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
|
|
187
175
|
end
|
|
188
176
|
|
|
189
|
-
parsed =
|
|
177
|
+
parsed = Pubid::Parg::Backend.parse(:jis, identifier)
|
|
190
178
|
Builder.build(parsed)
|
|
191
179
|
end
|
|
192
180
|
end
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Pubid
|
|
4
|
+
module Jis
|
|
5
|
+
module Identifiers
|
|
6
|
+
# Every part of one JIS document: "JIS C 0617(規格群)".
|
|
7
|
+
class AllParts < ::Pubid::Jis::Identifier
|
|
8
|
+
include ::Pubid::AllParts
|
|
9
|
+
|
|
10
|
+
SUFFIX = "(規格群)"
|
|
11
|
+
|
|
12
|
+
# The URN of the document plus the "all" slot.
|
|
13
|
+
def to_urn
|
|
14
|
+
"#{identity.to_urn}:all"
|
|
15
|
+
end
|
|
16
|
+
end
|
|
17
|
+
end
|
|
18
|
+
end
|
|
19
|
+
end
|
|
@@ -3,6 +3,7 @@
|
|
|
3
3
|
module Pubid
|
|
4
4
|
module Jis
|
|
5
5
|
module Identifiers
|
|
6
|
+
autoload :AllParts, "#{__dir__}/identifiers/all_parts"
|
|
6
7
|
autoload :Amendment, "#{__dir__}/identifiers/amendment"
|
|
7
8
|
autoload :Corrigendum, "#{__dir__}/identifiers/corrigendum"
|
|
8
9
|
autoload :Explanation, "#{__dir__}/identifiers/explanation"
|
data/lib/pubid/jis/renderer.rb
CHANGED
|
@@ -38,7 +38,6 @@ module Pubid
|
|
|
38
38
|
result = "#{PUBLISHER} #{id.code}"
|
|
39
39
|
result += ":#{id.year_with_reaffirmation}" if id.year
|
|
40
40
|
result += "(#{id.language})" if id.language
|
|
41
|
-
result += "(規格群)" if id.all_parts?
|
|
42
41
|
result + id.symbol_suffix
|
|
43
42
|
end
|
|
44
43
|
|
|
@@ -52,7 +51,6 @@ module Pubid
|
|
|
52
51
|
result += id.code.to_s
|
|
53
52
|
result += ":#{id.year_with_reaffirmation}" if id.year
|
|
54
53
|
result += "(#{id.language})" if id.language
|
|
55
|
-
result += "(規格群)" if id.all_parts?
|
|
56
54
|
result += id.symbol_suffix
|
|
57
55
|
result
|
|
58
56
|
end
|
|
@@ -12,7 +12,6 @@ module Pubid
|
|
|
12
12
|
|
|
13
13
|
parts << identifier.language.to_s.downcase if identifier.language
|
|
14
14
|
|
|
15
|
-
parts << "all" if identifier.all_parts?
|
|
16
15
|
|
|
17
16
|
if identifier.is_a?(SupplementIdentifier) && identifier.supplement_notation
|
|
18
17
|
parts << identifier.supplement_notation.to_s.downcase
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
# NIST flavor notes
|
|
2
|
+
|
|
3
|
+
NIST rendering formats and annotated output.
|
|
4
|
+
|
|
5
|
+
Read this before you change `lib/pubid/nist/` or `spec/pubid/nist/`. The root
|
|
6
|
+
`CLAUDE.md` keeps the cross-flavor contract that every flavor obeys.
|
|
7
|
+
|
|
8
|
+
- **Six NIST types rendered plain under `to_s(annotated: true)`, and two
|
|
9
|
+
more carried a latent corruption.** NIST's `to_s` takes a **positional**
|
|
10
|
+
`format`, which Ruby fills with a Hash when a caller passes `annotated:`;
|
|
11
|
+
`Identifiers::Base#to_s` already handles that and annotates. The six that
|
|
12
|
+
did not — `CommercialStandardsMonthly`, `CrplReport`, `InteragencyReport`,
|
|
13
|
+
`MiscellaneousPublication`, `Monograph`, `Report` — either compose their
|
|
14
|
+
own string or hand `super` a **Symbol**, which drops the flag. Each now
|
|
15
|
+
pulls `annotated` out of the Hash itself and wraps its result.
|
|
16
|
+
|
|
17
|
+
**`Circular` and `Handbook` are the interesting half.** Both did
|
|
18
|
+
`result = super` — which annotates — and then rewrote the edition with a
|
|
19
|
+
**`$`-anchored** regex. Once the string ends in `</span>` the anchor
|
|
20
|
+
cannot match, so the rewrite silently stops applying. Both now strip the
|
|
21
|
+
flag before `super`, rewrite, then annotate.
|
|
22
|
+
|
|
23
|
+
**An output assertion cannot catch that**, and this is the lesson: for
|
|
24
|
+
both fixture references (`NBS CIRC 11e2-1915`, `NBS HB 44e2-1955`) the
|
|
25
|
+
annotator matches only the leading publisher, so the tail is bare and the
|
|
26
|
+
wrong order still produces the right string. The spec asserts the
|
|
27
|
+
**ordering** instead — it wraps `annotate_plain_render` and checks the
|
|
28
|
+
string handed to it is already rewritten and carries no span — and that
|
|
29
|
+
assertion was verified to fail when the order is reverted.
|
|
30
|
+
|
|
31
|
+
`CircularSupplement` and `SupplementIdentifier` were also normalised:
|
|
32
|
+
both had a date-range branch that returned a hand-composed string without
|
|
33
|
+
ever reaching `super`. `SupplementIdentifier` has four exits, so its
|
|
34
|
+
composition moved into a private `render_plain`; **`super` is not
|
|
35
|
+
reachable from a private method**, so `to_s` hands it in as a block —
|
|
36
|
+
`render_plain(format) { super(format) }`.
|
|
37
|
+
|
|
38
|
+
- **`all_parts_edition_keys` needed `update`/`update_component`, not
|
|
39
|
+
`edition_year`.** `Identifier.all_parts_edition_keys` defaults to
|
|
40
|
+
`%i[date year edition version]`; NIST's primary edition carrier
|
|
41
|
+
(`edition`, a `Components::Edition`) is already covered, but the Letter
|
|
42
|
+
Circular / Circular "rJun1992"-style revision (`Builder` around the
|
|
43
|
+
"Convert revision with month+year to update component" comment) parses
|
|
44
|
+
into a **separate** attribute, `update`/`update_component`
|
|
45
|
+
(`Components::Update`: number+year+month), which the default list
|
|
46
|
+
missed entirely — `"NBS LC 800 rJun1992"` and `"NBS LC 800 rJul1995"`
|
|
47
|
+
failed to collapse under `#to_all_parts`/`#===`. Fixed with
|
|
48
|
+
`Pubid::Nist::Identifier.all_parts_edition_keys` (`super + %i[update
|
|
49
|
+
update_component]`). **`edition_year` and `revision_year`/
|
|
50
|
+
`revision_month` were investigated and are NOT added**: `Builder` only
|
|
51
|
+
ever sets `edition_year` alongside the real `edition` component (never
|
|
52
|
+
as its sole carrier, e.g. the TechnicalNote "date IS edition" branch),
|
|
53
|
+
and `revision_year`/`revision_month` are transient — converted into
|
|
54
|
+
`update`/`update_component` and cleared to `nil` before the object is
|
|
55
|
+
returned. Neither carries live information `edition`/`update` doesn't
|
|
56
|
+
already cover. Locked by `spec/pubid/all_parts_edition_keys_audit_spec.rb`.
|
data/lib/pubid/nist/builder.rb
CHANGED
|
@@ -87,6 +87,7 @@ module Pubid
|
|
|
87
87
|
# Note: :base_portion is lost during parser merge, so check for supplement indicators
|
|
88
88
|
if parsed_hash[:supplement_date_range] || parsed_hash[:supplement_slash_year] ||
|
|
89
89
|
parsed_hash[:supplement_month_year] || parsed_hash[:supplement_year] ||
|
|
90
|
+
parsed_hash[:supplement_empty] ||
|
|
90
91
|
parsed_hash[:supplement] || parsed_hash[:base_portion]
|
|
91
92
|
return build_circular_supplement(parsed_hash)
|
|
92
93
|
end
|
|
@@ -479,3 +480,5 @@ module Pubid
|
|
|
479
480
|
end
|
|
480
481
|
end
|
|
481
482
|
end
|
|
483
|
+
|
|
484
|
+
Pubid::Nist::Builder.prepend(Pubid::Builder::AllPartsWrap)
|
|
@@ -28,6 +28,8 @@ module Pubid
|
|
|
28
28
|
# Edition.new(type: "r", id: "5").to_s # => "r5"
|
|
29
29
|
# Edition.new(type: "r", id: "5", original_prefix: " Rev. ").to_s # => "Rev. 5"
|
|
30
30
|
class Edition < Lutaml::Model::Serializable
|
|
31
|
+
include ::Pubid::SubsetMatch
|
|
32
|
+
|
|
31
33
|
attribute :type, :string # "-", "e", or "r"
|
|
32
34
|
attribute :id, :string # Edition ID (number or year)
|
|
33
35
|
attribute :additional_text, :string # Text after "rev" (WITHOUT "rev" prefix)
|
|
@@ -8,6 +8,8 @@ module Pubid
|
|
|
8
8
|
# IssueNumber component for NIST identifiers
|
|
9
9
|
# Represents the issue/number designation (e.g., "No. 12" in "Vol. 6, No. 12")
|
|
10
10
|
class IssueNumber < Lutaml::Model::Serializable
|
|
11
|
+
include ::Pubid::SubsetMatch
|
|
12
|
+
|
|
11
13
|
attribute :number, :string
|
|
12
14
|
|
|
13
15
|
# Short form rendering: "n12"
|
|
@@ -23,6 +23,8 @@ module Pubid
|
|
|
23
23
|
# - SP: Part number (pt1)
|
|
24
24
|
# - Letter suffixes (A, B, C, etc.)
|
|
25
25
|
class Part < Lutaml::Model::Serializable
|
|
26
|
+
include ::Pubid::SubsetMatch
|
|
27
|
+
|
|
26
28
|
attribute :type, :string # "pt" for part notation, "n" for issue, "" for letter suffix
|
|
27
29
|
attribute :value, :string # Part number or letter (1, 2, A, B, etc.)
|
|
28
30
|
|
|
@@ -13,6 +13,8 @@ module Pubid
|
|
|
13
13
|
# Stage.new(id: "i", type: "pd").to_s(:short) # => "ipd"
|
|
14
14
|
# Stage.new(id: "f", type: "pd").to_s(:long) # => "(Final Public Draft)"
|
|
15
15
|
class Stage < Lutaml::Model::Serializable
|
|
16
|
+
include ::Pubid::SubsetMatch
|
|
17
|
+
|
|
16
18
|
attribute :id, :string # i, f, 1-9
|
|
17
19
|
attribute :type, :string # pd, wd, prd
|
|
18
20
|
|
|
@@ -15,6 +15,8 @@ module Pubid
|
|
|
15
15
|
# Supplement.new(month: "Jan", year: "1924").to_s(:short) # => "supJan1924"
|
|
16
16
|
# Supplement.new(has_revision: true).to_s(:short) # => "suprev"
|
|
17
17
|
class Supplement < Lutaml::Model::Serializable
|
|
18
|
+
include ::Pubid::SubsetMatch
|
|
19
|
+
|
|
18
20
|
attribute :number, :string # Supplement number (e.g., "2" in "supp2")
|
|
19
21
|
attribute :year, :string # Year (4 digits); range START year
|
|
20
22
|
attribute :month, :string # Month abbreviation; range START month
|
|
@@ -13,6 +13,8 @@ module Pubid
|
|
|
13
13
|
# Translation.new(code: "por").to_s(:mr) # => ".por"
|
|
14
14
|
# Translation.new(code: "ind").to_s(:short) # => " ind"
|
|
15
15
|
class Translation < Lutaml::Model::Serializable
|
|
16
|
+
include ::Pubid::SubsetMatch
|
|
17
|
+
|
|
16
18
|
attribute :code, :string # 3-letter ISO 639-2 code: spa, por, ind, etc.
|
|
17
19
|
|
|
18
20
|
# Backward compatibility: language method returns code
|
|
@@ -15,6 +15,8 @@ module Pubid
|
|
|
15
15
|
# Update.new(number: 1, year: 2021, month: 2).to_s(:mr) # => "-upd1-202102"
|
|
16
16
|
# Update.new(number: 1, prefix: "dash").to_s(:short) # => "-upd1" (preserves original prefix)
|
|
17
17
|
class Update < Lutaml::Model::Serializable
|
|
18
|
+
include ::Pubid::SubsetMatch
|
|
19
|
+
|
|
18
20
|
attribute :number, :string # Update number as string
|
|
19
21
|
attribute :year, :string # Year (4 digits as string)
|
|
20
22
|
attribute :month, :string # Month (01-12 as string, optional)
|
|
@@ -12,6 +12,8 @@ module Pubid
|
|
|
12
12
|
# Version.new(value: "1.0.2").to_s(:short) # => "ver1.0.2"
|
|
13
13
|
# Version.new(value: "2.0").to_s(:long) # => "Version 2.0"
|
|
14
14
|
class Version < Lutaml::Model::Serializable
|
|
15
|
+
include ::Pubid::SubsetMatch
|
|
16
|
+
|
|
15
17
|
attribute :value, :string # Dotted notation: "1.0.2"
|
|
16
18
|
|
|
17
19
|
# Render version in specified format
|
|
@@ -223,6 +223,30 @@ module Pubid
|
|
|
223
223
|
|
|
224
224
|
alias eql? ==
|
|
225
225
|
|
|
226
|
+
# A subset match skips the same attributes: `to_hash` drops the build
|
|
227
|
+
# artifacts, so an index row rebuilt by `from_hash` never has them.
|
|
228
|
+
def self.subset_ignored_attributes
|
|
229
|
+
EQUALITY_IGNORED_ATTRS
|
|
230
|
+
end
|
|
231
|
+
|
|
232
|
+
# `edition` is the primary edition/revision carrier and is already in
|
|
233
|
+
# the default %i[date year edition version] list. `update`/
|
|
234
|
+
# `update_component` (Components::Update: number+year+month) is a
|
|
235
|
+
# SEPARATE, currently-live discriminator the default list misses — the
|
|
236
|
+
# Letter Circular / Circular "rJun1992"-style revision parses into it,
|
|
237
|
+
# not into `edition` (see Builder#build_dated_identifier), so e.g.
|
|
238
|
+
# "NBS LC 800 rJun1992" and "NBS LC 800 rJul1995" failed to collapse
|
|
239
|
+
# under #to_all_parts/#=== before this override. (`edition_year` and
|
|
240
|
+
# `revision_year`/`revision_month` are NOT added here: the builder
|
|
241
|
+
# always sets `edition_year` alongside the real `edition` component
|
|
242
|
+
# (never alone), and `revision_year`/`revision_month` are transient —
|
|
243
|
+
# converted into `update`/`update_component` and cleared to nil before
|
|
244
|
+
# the object is returned — so neither carries live information outside
|
|
245
|
+
# what `edition`/`update` already do.)
|
|
246
|
+
def self.all_parts_edition_keys
|
|
247
|
+
super + %i[update update_component]
|
|
248
|
+
end
|
|
249
|
+
|
|
226
250
|
def hash
|
|
227
251
|
vals = self.class.attributes.each_key.reject do |name|
|
|
228
252
|
EQUALITY_IGNORED_ATTRS.include?(name)
|
|
@@ -413,8 +437,6 @@ module Pubid
|
|
|
413
437
|
match ? match[1].to_i : nil
|
|
414
438
|
end
|
|
415
439
|
|
|
416
|
-
public
|
|
417
|
-
|
|
418
440
|
def to_full_style
|
|
419
441
|
# "National Institute of Standards and Technology Special Publication 800-27, Revision A"
|
|
420
442
|
result = publisher_full_name
|
|
@@ -674,14 +696,25 @@ module Pubid
|
|
|
674
696
|
result += "#{vol_str}n#{issue_number.number}"
|
|
675
697
|
end
|
|
676
698
|
|
|
677
|
-
#
|
|
678
|
-
|
|
699
|
+
# With a number, the edition glues to it per the NIST spec
|
|
700
|
+
# ("800-53r5"); series-only editions take a dot separator
|
|
701
|
+
# ("NBS.CIRC.e2" — the attested raw spelling; testsuite#5 C4/C5).
|
|
702
|
+
result += if edition
|
|
703
|
+
number ? edition.to_s : ".#{edition}"
|
|
704
|
+
else
|
|
705
|
+
""
|
|
706
|
+
end
|
|
679
707
|
|
|
680
708
|
# Use version_component
|
|
681
709
|
result += version_component.to_s(:mr) if version_component
|
|
682
710
|
|
|
683
|
-
# Supplement (e.g. ".9981sup7") - keep distinct documents distinct
|
|
684
|
-
|
|
711
|
+
# Supplement (e.g. ".9981sup7") - keep distinct documents distinct;
|
|
712
|
+
# a series-only supplement ("NBS.CIRC.sup") takes the dot itself.
|
|
713
|
+
result += if supplement
|
|
714
|
+
number ? supplement_short : ".#{supplement_short}"
|
|
715
|
+
else
|
|
716
|
+
""
|
|
717
|
+
end
|
|
685
718
|
|
|
686
719
|
# Use update_component
|
|
687
720
|
result += update_component.to_s(:mr) if update_component
|
|
@@ -746,6 +779,5 @@ module Pubid
|
|
|
746
779
|
"NIST"
|
|
747
780
|
end
|
|
748
781
|
end
|
|
749
|
-
|
|
750
782
|
end
|
|
751
783
|
end
|
data/lib/pubid/nist/parser.rb
CHANGED
|
@@ -13,8 +13,20 @@ module Pubid
|
|
|
13
13
|
# feeds the cleaned string to the Parslet grammar and stamps the
|
|
14
14
|
# detected format onto the parse tree.
|
|
15
15
|
def self.class_parse_with_preprocessing(input)
|
|
16
|
+
# The shared Grammar strips a trailing "(all parts)", but the NIST
|
|
17
|
+
# preprocessor runs first and would swallow it — strip before it and
|
|
18
|
+
# carry the marker to the tree like Grammar#parse does.
|
|
19
|
+
all_parts = input.is_a?(String) && input.match?(::Pubid::Parser::Grammar::ALL_PARTS_SUFFIX)
|
|
20
|
+
input = input.sub(::Pubid::Parser::Grammar::ALL_PARTS_SUFFIX, "") if all_parts
|
|
16
21
|
result = Preprocessor.new(input).call
|
|
17
22
|
parsed = new.parse(result.cleaned)
|
|
23
|
+
if all_parts
|
|
24
|
+
parsed = case parsed
|
|
25
|
+
when Hash then parsed.merge(all_parts: true)
|
|
26
|
+
when Array then parsed.map { |h| h.merge(all_parts: true) }
|
|
27
|
+
else parsed
|
|
28
|
+
end
|
|
29
|
+
end
|
|
18
30
|
|
|
19
31
|
if parsed.is_a?(Hash)
|
|
20
32
|
parsed.merge(parsed_format: result.format)
|
|
@@ -651,8 +663,9 @@ module Pubid
|
|
|
651
663
|
rule(:mr_identifier) do
|
|
652
664
|
hash_prefix.maybe >>
|
|
653
665
|
publisher >> dot >>
|
|
654
|
-
|
|
655
|
-
|
|
666
|
+
# The catalogue also lists bare series identities with no
|
|
667
|
+
# report number ("NBS.CIRC").
|
|
668
|
+
simple_series >> dot.maybe >> report_number.maybe >>
|
|
656
669
|
# Edition with underscore separator (MR format: 1648_2009)
|
|
657
670
|
(str("_") >> digits.as(:edition_year)).maybe >>
|
|
658
671
|
# Support letter suffix before update (e.g., 8286C-upd1) - Session 219
|
|
@@ -713,6 +726,15 @@ module Pubid
|
|
|
713
726
|
# 4-digit years so it can't swallow "sup3/1926" or a base number.
|
|
714
727
|
((str("supp") | str("sup")) >> match("[0-9]").repeat(4, 4).as(:supp_year_start) >>
|
|
715
728
|
dash >> match("[0-9]").repeat(4, 4).as(:supp_year_end)).as(:supplement_date_range) |
|
|
729
|
+
# Bare supplement marker to the whole series, no base number
|
|
730
|
+
# ("NBS.CIRC.sup" — testsuite#5 C4): the marker alone is the
|
|
731
|
+
# supplement.
|
|
732
|
+
((str("supp") | str("sup")) >>
|
|
733
|
+
(
|
|
734
|
+
(month_abbrev >> digits).as(:supplement_month_year) |
|
|
735
|
+
(digits.as(:supp_number) >> slash >> digits.as(:supp_year)).as(:supplement_slash_year) |
|
|
736
|
+
str("").as(:supplement_empty)
|
|
737
|
+
).maybe) |
|
|
716
738
|
# With base identifier + supplement
|
|
717
739
|
(
|
|
718
740
|
# Capture base portion (everything before "supp" or "sup" or slash+year)
|
|
@@ -37,6 +37,11 @@ module Pubid
|
|
|
37
37
|
def initialize(input)
|
|
38
38
|
@input = input.to_s.strip
|
|
39
39
|
@cleaned = Core::UpdateCodes.apply(@input, :nist)
|
|
40
|
+
# The format describes the string the parser will see: capture
|
|
41
|
+
# it right after the update-codes remap, before the stages'
|
|
42
|
+
# cosmetic spacing (a dotted catalogue alias that remaps to the
|
|
43
|
+
# space form renders short; the dotted originals stay :mr).
|
|
44
|
+
@format = @cleaned.include?(".") && !@cleaned.match?(/\s/) ? :mr : :short
|
|
40
45
|
end
|
|
41
46
|
|
|
42
47
|
# Run every normalization stage and return a Result.
|
|
@@ -53,6 +58,14 @@ module Pubid
|
|
|
53
58
|
# Extracted so rubocop can scope length/ABC metrics narrowly.
|
|
54
59
|
# rubocop:disable Metrics/MethodLength, Metrics/AbcSize
|
|
55
60
|
def run_stages
|
|
61
|
+
# Short-form "supprev" is the catalogue spelling of the plain
|
|
62
|
+
# supplement ("NBS CIRC 154supprev" ≡ "NBS CIRC 154sup"; the
|
|
63
|
+
# revision-bearing identity is the mr spelling "154suprev").
|
|
64
|
+
# Rewrite before the supplement/revision stages; the mr form is
|
|
65
|
+
# untouched.
|
|
66
|
+
if detected_format == :short
|
|
67
|
+
@cleaned = @cleaned.gsub("supprev", "sup")
|
|
68
|
+
end
|
|
56
69
|
normalize_spurious_u_suffix!
|
|
57
70
|
normalize_publisher_and_series!
|
|
58
71
|
normalize_lcirc_supplement_contexts!
|
|
@@ -75,12 +88,14 @@ module Pubid
|
|
|
75
88
|
normalize_part_notation!
|
|
76
89
|
normalize_series_specific_spacing!
|
|
77
90
|
normalize_verbose_keywords!
|
|
91
|
+
normalize_legacy_corpus_spellings!
|
|
78
92
|
end
|
|
79
93
|
# rubocop:enable Metrics/MethodLength, Metrics/AbcSize
|
|
80
94
|
|
|
81
|
-
# Detect input format: :mr (dot-separated machine-readable) or
|
|
95
|
+
# Detect input format: :mr (dot-separated machine-readable) or
|
|
96
|
+
# :short. Frozen in #initialize (post update-codes, pre-stages).
|
|
82
97
|
def detected_format
|
|
83
|
-
@
|
|
98
|
+
@format
|
|
84
99
|
end
|
|
85
100
|
|
|
86
101
|
private
|
|
@@ -167,12 +182,21 @@ module Pubid
|
|
|
167
182
|
# Trailing "-a" → "-A" at end of identifier.
|
|
168
183
|
def uppercase_dash_letter!
|
|
169
184
|
@cleaned = @cleaned.gsub(/(\d)-([a-z])$/) { "#{$1}-#{$2.upcase}" }
|
|
185
|
+
# Same letter when a part tail follows ("-add", " Add."), so the
|
|
186
|
+
# canonical case survives; lowercase update/translation codes
|
|
187
|
+
# ("-upd", ".uppl") never match (next char is a letter).
|
|
188
|
+
@cleaned = @cleaned.gsub(/(\d)-([a-z])(?=[-.\s])/) { "#{$1}-#{$2.upcase}" }
|
|
170
189
|
end
|
|
171
190
|
|
|
172
191
|
# Trailing "a" → "A" when attached directly to a digit (excludes
|
|
173
192
|
# "r" to preserve revision+year patterns like "73-197r").
|
|
174
193
|
def uppercase_trailing_letter!
|
|
175
194
|
@cleaned = @cleaned.gsub(/(\d)([a-z&&[^r]])$/) { "#{$1}#{$2.upcase}" }
|
|
195
|
+
# Same letter when an addendum tail follows ("-add", " Add."),
|
|
196
|
+
# so the canonical part case survives ("800-38a-add" ->
|
|
197
|
+
# "800-38A"); part digits ("800-85a-1") and update/translation
|
|
198
|
+
# codes ("-upd", ".uppl") never match.
|
|
199
|
+
@cleaned = @cleaned.gsub(/(\d)([a-z&&[^r]])(?=\s*[-.\s]\s*[aA]dd)/) { "#{$1}#{$2.upcase}" }
|
|
176
200
|
end
|
|
177
201
|
|
|
178
202
|
# Letter suffix on revision: "22r1a" → "22r1A".
|
|
@@ -336,6 +360,33 @@ module Pubid
|
|
|
336
360
|
end
|
|
337
361
|
end
|
|
338
362
|
|
|
363
|
+
# Legacy corpus spellings: catalogue forms the mr grammar and the
|
|
364
|
+
# renderers cannot round-trip on their own. Each rule mirrors an
|
|
365
|
+
# already-parseable spelling of the same document.
|
|
366
|
+
def normalize_legacy_corpus_spellings!
|
|
367
|
+
# The mr renderer prints the translation code with its own
|
|
368
|
+
# leading dot ("955-S..uppl"); the parser wants one dot.
|
|
369
|
+
@cleaned = @cleaned.gsub("..", ".")
|
|
370
|
+
# Trailing-dot addendum spelling; the canonical render comes from
|
|
371
|
+
# the lowercase ".add" parse ("NBS.TN.467pt1.Add." -> ".add").
|
|
372
|
+
@cleaned = @cleaned.gsub(/\.Add\.\z/, ".add")
|
|
373
|
+
# Edition glued to the series in mr form ("NBS.CIRCe2" is
|
|
374
|
+
# "NBS.CIRC.e2"; digits before "e" are untouched: "24e7", and
|
|
375
|
+
# the short form never glues an edition to a letter part:
|
|
376
|
+
# "150-1Ae2009").
|
|
377
|
+
if detected_format == :mr
|
|
378
|
+
@cleaned = @cleaned.gsub(/([A-Z])e(\d)/, '\1.e\2')
|
|
379
|
+
end
|
|
380
|
+
# Update markers without a number render a phantom "1";
|
|
381
|
+
# "…-upd" is the numberless spelling of "…-upd1".
|
|
382
|
+
@cleaned = @cleaned.gsub(/-upd\z/, "-upd1")
|
|
383
|
+
# Handbook legacy renumbering in the all-dash spelling: the
|
|
384
|
+
# e-form stage already eats the prefix generically ("HB
|
|
385
|
+
# 150-1e2017" -> "HB 1-2017"); the all-dash form carries the
|
|
386
|
+
# same 105-/150- prefixes.
|
|
387
|
+
@cleaned = @cleaned.gsub(/\b(NIST HB\s+)(?:105|150)-(\d+)-(\d{4})(?=\s|\z)/, '\1\2-\3')
|
|
388
|
+
end
|
|
389
|
+
|
|
339
390
|
# Series-specific reverts: HB handbooks, OWMWP dates, and RPT year
|
|
340
391
|
# ranges use dash-year structurally (not as an edition marker), so
|
|
341
392
|
# the broad convert_dashyear_to_edition! rule would corrupt them.
|
|
@@ -42,13 +42,22 @@ module Pubid
|
|
|
42
42
|
type_token, payload = parts
|
|
43
43
|
code, revision = parse_payload(payload)
|
|
44
44
|
|
|
45
|
-
text = "
|
|
45
|
+
text = "#{publisher_for(type_token)} #{type_label(type_token)} #{code}"
|
|
46
46
|
text += revision if revision
|
|
47
47
|
flavor_parse(text)
|
|
48
48
|
end
|
|
49
49
|
|
|
50
50
|
private
|
|
51
51
|
|
|
52
|
+
# The NBS-era series carry the NBS imprint in their canonical form
|
|
53
|
+
# ("NBS CSM 1") — the rebuild must name the publisher the document
|
|
54
|
+
# carries, or the flavor parse rejects its own URN's rebuild.
|
|
55
|
+
NBS_SERIES = ["csm"].freeze
|
|
56
|
+
|
|
57
|
+
def publisher_for(type_token)
|
|
58
|
+
NBS_SERIES.include?(type_token.downcase) ? "NBS" : "NIST"
|
|
59
|
+
end
|
|
60
|
+
|
|
52
61
|
def parse_payload(payload)
|
|
53
62
|
# Strip the trailing ".supp" supplement marker.
|
|
54
63
|
stripped = payload.sub(/\.supp\z/, "")
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
# OASIS flavor notes
|
|
2
|
+
|
|
3
|
+
OASIS verbatim slugs, the index key, the MR slug and partial-reference matching.
|
|
4
|
+
|
|
5
|
+
Read them before you change `lib/pubid/oasis/` or `spec/pubid/oasis/`. The root file keeps the cross-flavor contract that every flavor obeys.
|
|
6
|
+
|
|
7
|
+
- **The identity is the verbatim slug; the decomposition is a lossy projection of it**: an OASIS identifier is a free-form slug with no fixed internal order (`OSLC-CoreShapes-3.0-PS01-Pt8`, `OSLC-AM-3.0-Part1-PS01`, `amqp-core`), three part spellings and mixed case. The grammar therefore captures the whole slug with `any.repeat(1)` and only strips the `OASIS ` prefix; `Builder#decompose` then classifies each *whole* dash-separated fragment into `number` / `version` / `stage` / `part` / `label`. **`original` holds the printed slug verbatim and alone drives `to_s` and `to_urn`** — `Renderer#render` is a pure echo of it — so the printed form round-trips byte-exactly whatever the classifier makes of it. The decomposition is **not** an identity and cannot be one: it loses fragment order (`x-1.0-os-Pt1` and `x-1.0-Pt1-os` decompose alike) and keeps only the first fragment of each recognized kind, so a repeated one is dropped. That is not hypothetical — **7 of the 605 published `relaton-data-oasis` ids already lose a fragment** (`OASIS xacml-3.0-hierarchical-v1.0-CS02` keeps `3.0` and drops `v1.0`). Every design decision below follows from that asymmetry.
|
|
8
|
+
|
|
9
|
+
- **`spec` → `number`, the index key (PR #359)**: `Relaton::Index::Type#candidates_by_number` sorts every row and binary-searches it on `id.root.number.to_s`. OASIS kept the specification name in a bespoke `attribute :spec, :string` and never set the `number` it inherits from `::Pubid::Identifier`, so **all 605 published rows shared the empty key `""`** and the search degraded to a linear scan, silently. The attribute is **renamed**, not shadowed by a derived reader: `attribute :number, :string` plus `map "number", to: :number`, and `Builder#decompose` returns `number:`. **`spec` is dropped with no alias and no reader** — one name for one value, the W3C `code` → `number` precedent. A derived `#spec` was written first and then removed: nothing in `lib/` read it, and `relaton-oasis` has no pubid dependency at all, so it had no consumer to preserve. That is what separates this case from IANA's `#registry`, IETF's `#series` and BIPM's `#volume`, which do have one. `root_number_spec.rb` asserts the identifier does **not** respond to `spec`, so the name cannot creep back. The key **clusters**: every version, stage and part of one specification shares it — 309 buckets over 605 ids, median 1, max 25 (`STIX`) — the shape of an IETF draft slug or an IANA registry slug, not an exact key.
|
|
10
|
+
|
|
11
|
+
- **Why the `number` declaration sits on the base here**: `attribute :number, :string` redefines the parent's `Components::Code number` on a class `Identifiers::Standard` inherits from, which is the recorded determinism landmine. It is safe for the same reason W3C's is: lutaml deep-dups the parent attribute table into each subclass at class-definition time, so a subclass holds a snapshot, and Ruby resolves the superclass constant to completion before opening a leaf body. **`Pubid::Oasis::Identifier`'s class body lives in one file and is never reopened**, so the snapshot is always complete. `stage` and `part` in that same body have relied on this since the flavor landed, overriding `Components::Stage` and `Components::Code`. **Split this class across two files and the landmine comes back** — IEEE is the counter-shape, its base reachable through two paths. `spec/pubid/oasis/root_number_spec.rb` carries the structural tripwire (the base *and* the leaf must resolve `number` to `Lutaml::Model::Type::String`) and is only meaningful under the full `bundle exec rake`.
|
|
12
|
+
|
|
13
|
+
- **MR slug**: OASIS supplied no `mr_*` hook, and every base hook looks for something the flavor does not use — the publisher is the `PUBLISHER` constant rather than a lutaml attribute, there is no `Components::Date` and no `typed_stage`. With `number` nil as well, **`to_mr_string` was `""` for all 605 ids**, and `to_slug` is what consumers use as an output **filename**. The base now supplies `mr_publisher` (`"oasis"`) and `mr_number_with_part`, which returns the sanitized **`original`** — not the five decomposed fields, because those assemble `x-1.0-os-Pt1` and `x-1.0-Pt1-os` into one slug, while `original` cannot collide. A private `mr_sanitize` filters **by charset**, so a field added later cannot leak an unsafe character. It uses **BIPM's `[^a-z0-9]+` form, not ETSI's `[^a-z0-9-]+`**: a run of non-alphanumerics collapses to one `-`, without which `v3.0]-PS01` prints a bare `--`. Three characters need the filter today — the `.` in every version (`Renderers::MrString` joins *segments* with `.`, so a dot inside one breaks the documented structure), plus the space and `]` of a few malformed records. Result: **605 distinct slugs for the 605 published ids, 0 raises, 0 characters outside `[a-z0-9._-]`**. One caveat, the same one BIPM records: `-` is both the intra-slug join and the substitute, so two slugs differing only in which non-slug character they use collapse. The corpus has exactly one such pair — the malformed `OpenC2-MQTT-v1.0] -CS01` spelled once with a normal space and once with a non-breaking one, two spellings of one document.
|
|
14
|
+
|
|
15
|
+
- **`#exclude` clears `original`, so a partial reference can widen**: `#matches?` is `exclude(*ignore) == other.exclude(*ignore)`, and `original` spells out verbatim the very component the caller is ignoring. So it survived the exclusion and a bare `OASIS WSDM` could **never** match `OASIS WSDM-v1.1`, however much was ignored — relaton would narrow to the right index bucket and then match nothing in it. `Pubid::Oasis::Identifier#exclude` now nils `original` when the excluded set intersects `DECOMPOSITION_KEYS` (`number version stage part label`), the "reset the whole cluster" rule CSA applies to its year-format siblings. **A plain `==` still compares `original`, deliberately**: an excluded copy is a comparison token, never a rendered document, so widening happens only when the caller asks for it. Dropping `original` from `==` outright was considered and rejected — it costs nothing on today's corpus (no two ids collide) but it exposes those 7 lossy ids to a silent collision with a future sibling, and it would let two different printed identifiers be equal. `spec/pubid/oasis/partial_ref_spec.rb` locks both halves.
|
|
16
|
+
|
|
17
|
+
- **relaton note**: nothing published needs migrating. `relaton-oasis` has no pubid dependency and its `index-v1.yaml` is string-keyed (`:id: OASIS amqp-core`), which is why the serialized shape was changed properly rather than patched. There is **no alias** for the old `spec` key, and a pre-`number` row would deserialize with a nil `number` and no error (lutaml ignores unknown keys), so a future pubid index must be crawled *after* this. Rendering is the loud failure if it is not.
|
|
18
|
+
|
|
19
|
+
- **Verification and the fixture net**: the change was verified by replaying a `main` baseline over all 605 published ids plus the fixture lines — `to_s`, `to_urn`, identifier class and parse errors **byte-identical**, `to_hash` differing **only** by the key rename, empty index keys 607 → 0, distinct MR slugs 1 → 606 of 607, and `from_hash(to_hash) == id` for all 605 with **0** pairs of distinct ids equal. `spec/pubid/oasis/fixtures_spec.rb` is live — its glob is correct, so OASIS is not on the `ten-dead-fixture-specs` list — and `root_number_spec.rb` adds a corpus sweep over the same file. **Do not run `rake "validation:classify[oasis]"`**: there is no `spec/fixtures/oasis/identifiers/full/`, so the generator would delete `pass/oasis.txt` and rebuild nothing. (hand-off: oasis-index-number.)
|
data/lib/pubid/oasis/builder.rb
CHANGED
|
@@ -36,6 +36,18 @@ module Pubid
|
|
|
36
36
|
# building the table before `Identifiers::Standard` snapshots it — DO NOT
|
|
37
37
|
# split this class across two files.
|
|
38
38
|
attribute :number, :string
|
|
39
|
+
|
|
40
|
+
# `original` alone drives #to_s, and the shared exclude-copy loses it
|
|
41
|
+
# (the redefined attribute does not survive the rebuild). An all-parts
|
|
42
|
+
# copy is therefore dup-based: nil the part and edition-ish attributes
|
|
43
|
+
# in place, keeping the printed slug verbatim.
|
|
44
|
+
def without_parts(*extra)
|
|
45
|
+
copy = dup
|
|
46
|
+
(::Pubid::Identifier::PART_ATTRIBUTES + extra).each do |name|
|
|
47
|
+
copy.public_send(:"#{name}=", nil) if copy.respond_to?(:"#{name}=")
|
|
48
|
+
end
|
|
49
|
+
copy
|
|
50
|
+
end
|
|
39
51
|
attribute :version, :string
|
|
40
52
|
attribute :stage, :string
|
|
41
53
|
attribute :part, :string
|
|
@@ -91,6 +103,13 @@ module Pubid
|
|
|
91
103
|
result
|
|
92
104
|
end
|
|
93
105
|
|
|
106
|
+
# The same reason applies to a subset match: `original` spells out the
|
|
107
|
+
# parts a partial reference omits, so `===` compares the decomposition
|
|
108
|
+
# only. Two slugs that decompose alike therefore match each other.
|
|
109
|
+
def self.subset_ignored_attributes
|
|
110
|
+
%i[original]
|
|
111
|
+
end
|
|
112
|
+
|
|
94
113
|
# MR string hooks. `to_slug` delegates to `to_mr_string` and consumers
|
|
95
114
|
# use it as an output FILENAME, so the slug must be unique per document.
|
|
96
115
|
#
|
|
@@ -127,7 +146,7 @@ module Pubid
|
|
|
127
146
|
raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
|
|
128
147
|
end
|
|
129
148
|
|
|
130
|
-
parsed =
|
|
149
|
+
parsed = Pubid::Parg::Backend.parse(:oasis, identifier)
|
|
131
150
|
Builder.build(parsed)
|
|
132
151
|
end
|
|
133
152
|
|