pubid 2.0.0.pre.alpha.9 → 2.0.0.pre.alpha.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.adoc +33 -11
- data/conformance/pending.yaml +4 -0
- data/lib/pubid/adobe/identifier.rb +1 -3
- data/lib/pubid/adobe/parser.rb +10 -2
- data/lib/pubid/adobe.rb +1 -0
- data/lib/pubid/amca/builder.rb +2 -2
- data/lib/pubid/amca/identifiers/base.rb +49 -27
- data/lib/pubid/amca/identifiers/interpretation.rb +17 -9
- data/lib/pubid/amca/identifiers/publication.rb +13 -9
- data/lib/pubid/amca/identifiers/standard.rb +2 -0
- data/lib/pubid/amca/parser.rb +1 -1
- data/lib/pubid/amca/renderer.rb +4 -4
- data/lib/pubid/amca/urn_generator.rb +2 -2
- data/lib/pubid/amca.rb +2 -1
- data/lib/pubid/ansi/builder.rb +2 -2
- data/lib/pubid/ansi/identifier.rb +18 -0
- data/lib/pubid/ansi/parser.rb +1 -1
- data/lib/pubid/ansi/renderer.rb +2 -2
- data/lib/pubid/ansi.rb +5 -5
- data/lib/pubid/api/builder.rb +16 -2
- data/lib/pubid/api/identifier.rb +25 -4
- data/lib/pubid/api/identifiers/mpms.rb +10 -8
- data/lib/pubid/api/parser.rb +1 -1
- data/lib/pubid/api/renderer.rb +16 -7
- data/lib/pubid/api/single_identifier.rb +5 -12
- data/lib/pubid/api.rb +1 -0
- data/lib/pubid/ashrae/builder.rb +31 -10
- data/lib/pubid/ashrae/identifiers/addenda_package.rb +6 -0
- data/lib/pubid/ashrae/identifiers/addendum.rb +8 -0
- data/lib/pubid/ashrae/identifiers/base.rb +81 -5
- data/lib/pubid/ashrae/identifiers/combined_addenda.rb +7 -0
- data/lib/pubid/ashrae/identifiers/errata.rb +9 -0
- data/lib/pubid/ashrae/identifiers/guideline.rb +3 -0
- data/lib/pubid/ashrae/identifiers/interpretation.rb +16 -0
- data/lib/pubid/ashrae/identifiers/standard.rb +3 -0
- data/lib/pubid/ashrae/parser.rb +23 -8
- data/lib/pubid/ashrae/renderer.rb +6 -6
- data/lib/pubid/ashrae/supplement_identifier.rb +17 -2
- data/lib/pubid/ashrae/urn_generator.rb +8 -2
- data/lib/pubid/ashrae.rb +7 -1
- data/lib/pubid/asme/builder.rb +11 -2
- data/lib/pubid/asme/identifier.rb +9 -0
- data/lib/pubid/asme/identifiers/standard.rb +86 -2
- data/lib/pubid/asme/parser.rb +1 -1
- data/lib/pubid/asme/renderer.rb +3 -3
- data/lib/pubid/asme/single_identifier.rb +8 -1
- data/lib/pubid/asme/urn_generator.rb +2 -2
- data/lib/pubid/asme.rb +1 -0
- data/lib/pubid/astm/builder.rb +10 -3
- data/lib/pubid/astm/identifier.rb +9 -0
- data/lib/pubid/astm/identifiers/adjunct.rb +16 -1
- data/lib/pubid/astm/identifiers/code_number.rb +81 -0
- data/lib/pubid/astm/identifiers/data_series.rb +2 -0
- data/lib/pubid/astm/identifiers/iso_dual_published.rb +16 -0
- data/lib/pubid/astm/identifiers/manual.rb +10 -0
- data/lib/pubid/astm/identifiers/monograph.rb +2 -0
- data/lib/pubid/astm/identifiers/research_report.rb +10 -0
- data/lib/pubid/astm/identifiers/standard.rb +10 -0
- data/lib/pubid/astm/identifiers/technical_report.rb +2 -0
- data/lib/pubid/astm/identifiers/work_in_progress.rb +2 -0
- data/lib/pubid/astm/identifiers.rb +1 -0
- data/lib/pubid/astm/parser.rb +1 -1
- data/lib/pubid/astm/renderer.rb +1 -1
- data/lib/pubid/astm/single_identifier.rb +31 -1
- data/lib/pubid/astm/urn_generator.rb +7 -5
- data/lib/pubid/astm.rb +1 -0
- data/lib/pubid/bipm/builder.rb +66 -10
- data/lib/pubid/bipm/identifier.rb +134 -9
- data/lib/pubid/bipm/identifiers/committee_document.rb +12 -0
- data/lib/pubid/bipm/identifiers/guide.rb +11 -0
- data/lib/pubid/bipm/identifiers/meeting.rb +12 -0
- data/lib/pubid/bipm/identifiers/mep.rb +12 -0
- data/lib/pubid/bipm/identifiers/metrologia_article.rb +24 -0
- data/lib/pubid/bipm/identifiers/si_brochure.rb +13 -0
- data/lib/pubid/bipm/parser.rb +65 -12
- data/lib/pubid/bipm/renderer.rb +8 -0
- data/lib/pubid/bipm/urn_parser.rb +11 -3
- data/lib/pubid/bipm.rb +7 -1
- data/lib/pubid/bsi/builder.rb +48 -38
- data/lib/pubid/bsi/identifiers/adopted_european_norm.rb +36 -7
- data/lib/pubid/bsi/identifiers/adopted_international_standard.rb +7 -6
- data/lib/pubid/bsi/identifiers/british_industrial_practice.rb +0 -1
- data/lib/pubid/bsi/identifiers/bundled_identifier.rb +10 -0
- data/lib/pubid/bsi/identifiers/committee_document.rb +10 -1
- data/lib/pubid/bsi/identifiers/handbook.rb +0 -2
- data/lib/pubid/bsi/identifiers/practice_guide.rb +0 -1
- data/lib/pubid/bsi/identifiers/set.rb +11 -0
- data/lib/pubid/bsi/identifiers/standalone_amendment.rb +10 -1
- data/lib/pubid/bsi/parser.rb +1 -1
- data/lib/pubid/bsi/renderer.rb +63 -51
- data/lib/pubid/bsi/single_identifier.rb +16 -7
- data/lib/pubid/bsi/urn_generator.rb +11 -2
- data/lib/pubid/bsi.rb +7 -1
- data/lib/pubid/builder/base.rb +18 -8
- data/lib/pubid/bundled_identifier.rb +16 -6
- data/lib/pubid/calconnect/identifier.rb +9 -5
- data/lib/pubid/calconnect/parser.rb +1 -1
- data/lib/pubid/calconnect.rb +7 -1
- data/lib/pubid/ccsds/identifier.rb +9 -2
- data/lib/pubid/ccsds/identifiers/base.rb +4 -2
- data/lib/pubid/ccsds/identifiers/corrigendum.rb +2 -2
- data/lib/pubid/ccsds/parser.rb +1 -1
- data/lib/pubid/ccsds/single_identifier.rb +2 -2
- data/lib/pubid/ccsds.rb +1 -0
- data/lib/pubid/cen_cenelec/builder.rb +101 -76
- data/lib/pubid/cen_cenelec/identifier.rb +115 -3
- data/lib/pubid/cen_cenelec/identifiers/adopted_european_norm.rb +27 -16
- data/lib/pubid/cen_cenelec/identifiers/amendment.rb +46 -6
- data/lib/pubid/cen_cenelec/identifiers/base.rb +8 -9
- data/lib/pubid/cen_cenelec/identifiers/cen_report.rb +1 -1
- data/lib/pubid/cen_cenelec/identifiers/cen_workshop_agreement.rb +1 -1
- data/lib/pubid/cen_cenelec/identifiers/cenelec_harmonization_document.rb +1 -1
- data/lib/pubid/cen_cenelec/identifiers/consolidated_identifier.rb +29 -20
- data/lib/pubid/cen_cenelec/identifiers/corrigendum.rb +47 -8
- data/lib/pubid/cen_cenelec/identifiers/european_prestandard.rb +27 -2
- data/lib/pubid/cen_cenelec/identifiers/european_specification.rb +1 -1
- data/lib/pubid/cen_cenelec/identifiers/fragment.rb +15 -5
- data/lib/pubid/cen_cenelec/identifiers/guide.rb +1 -1
- data/lib/pubid/cen_cenelec/identifiers/harmonization_document.rb +1 -1
- data/lib/pubid/cen_cenelec/identifiers/technical_report.rb +1 -1
- data/lib/pubid/cen_cenelec/identifiers/technical_specification.rb +1 -1
- data/lib/pubid/cen_cenelec/parser.rb +1 -1
- data/lib/pubid/cen_cenelec/renderer.rb +55 -60
- data/lib/pubid/cen_cenelec/single_identifier.rb +20 -2
- data/lib/pubid/cen_cenelec/urn_generator.rb +66 -44
- data/lib/pubid/cen_cenelec.rb +28 -10
- data/lib/pubid/cie/builder.rb +10 -0
- data/lib/pubid/cie/identifier.rb +6 -1
- data/lib/pubid/cie/identifiers/bundle.rb +4 -2
- data/lib/pubid/cie/identifiers/conference.rb +2 -2
- data/lib/pubid/cie/identifiers/corrigendum.rb +2 -2
- data/lib/pubid/cie/identifiers/dual_published.rb +2 -2
- data/lib/pubid/cie/identifiers/identical.rb +2 -2
- data/lib/pubid/cie/identifiers/joint_published.rb +2 -2
- data/lib/pubid/cie/identifiers/proceedings.rb +8 -6
- data/lib/pubid/cie/identifiers/standard.rb +22 -13
- data/lib/pubid/cie/identifiers/supplement.rb +2 -2
- data/lib/pubid/cie/identifiers/tutorial_bundle.rb +2 -2
- data/lib/pubid/cie/parser.rb +1 -1
- data/lib/pubid/cie.rb +7 -1
- data/lib/pubid/conformance/checks.rb +72 -0
- data/lib/pubid/conformance/corpus/case.rb +55 -0
- data/lib/pubid/conformance/corpus.rb +36 -0
- data/lib/pubid/conformance/generator.rb +212 -0
- data/lib/pubid/conformance/pending.rb +63 -0
- data/lib/pubid/conformance/runner.rb +153 -0
- data/lib/pubid/conformance.rb +55 -0
- data/lib/pubid/core.rb +2 -0
- data/lib/pubid/csa/builder.rb +47 -47
- data/lib/pubid/csa/composite_identifier.rb +14 -19
- data/lib/pubid/csa/identifier.rb +143 -81
- data/lib/pubid/csa/identifiers/bundled.rb +23 -13
- data/lib/pubid/csa/identifiers/canadian_adopted.rb +13 -13
- data/lib/pubid/csa/identifiers/cec.rb +44 -4
- data/lib/pubid/csa/identifiers/combined.rb +95 -73
- data/lib/pubid/csa/identifiers/csa_adopted.rb +4 -4
- data/lib/pubid/csa/identifiers/package.rb +2 -2
- data/lib/pubid/csa/parser.rb +6 -2
- data/lib/pubid/csa/renderer.rb +13 -5
- data/lib/pubid/csa/single_identifier.rb +93 -4
- data/lib/pubid/csa/urn_generator.rb +9 -2
- data/lib/pubid/csa/wrapper_identifier.rb +39 -23
- data/lib/pubid/csa.rb +7 -1
- data/lib/pubid/doi/identifier.rb +6 -3
- data/lib/pubid/doi/parser.rb +1 -1
- data/lib/pubid/doi.rb +1 -0
- data/lib/pubid/easc/identifier.rb +0 -2
- data/lib/pubid/easc/parser.rb +10 -2
- data/lib/pubid/easc.rb +1 -0
- data/lib/pubid/ecma/builder.rb +23 -9
- data/lib/pubid/ecma/identifier.rb +81 -10
- data/lib/pubid/ecma/parser.rb +39 -4
- data/lib/pubid/ecma/renderer.rb +29 -8
- data/lib/pubid/ecma/urn_generator.rb +34 -13
- data/lib/pubid/ecma/urn_parser.rb +30 -14
- data/lib/pubid/ecma.rb +1 -0
- data/lib/pubid/errors.rb +80 -0
- data/lib/pubid/etsi/builder.rb +6 -1
- data/lib/pubid/etsi/identifiers/base.rb +34 -20
- data/lib/pubid/etsi/identifiers/etsi_standard.rb +120 -41
- data/lib/pubid/etsi/identifiers/supplement_identifier.rb +10 -0
- data/lib/pubid/etsi/parser.rb +1 -1
- data/lib/pubid/etsi.rb +1 -0
- data/lib/pubid/evs/builder.rb +26 -0
- data/lib/pubid/evs/identifier.rb +37 -0
- data/lib/pubid/evs/identifiers/national_adoption.rb +17 -0
- data/lib/pubid/evs/identifiers.rb +9 -0
- data/lib/pubid/evs/parser.rb +27 -0
- data/lib/pubid/evs/renderer.rb +30 -0
- data/lib/pubid/evs/urn_generator.rb +30 -0
- data/lib/pubid/evs/urn_parser.rb +80 -0
- data/lib/pubid/evs.rb +90 -0
- data/lib/pubid/export.rb +1 -0
- data/lib/pubid/format_detector.rb +1 -0
- data/lib/pubid/gb/identifier.rb +6 -3
- data/lib/pubid/gb/parser.rb +1 -1
- data/lib/pubid/gb.rb +1 -0
- data/lib/pubid/gost/builder.rb +46 -9
- data/lib/pubid/gost/identifier.rb +0 -2
- data/lib/pubid/gost/identifiers/foreign_reference.rb +1 -1
- data/lib/pubid/gost/parser.rb +10 -2
- data/lib/pubid/gost.rb +1 -0
- data/lib/pubid/iala/identifier.rb +9 -6
- data/lib/pubid/iala/parser.rb +10 -2
- data/lib/pubid/iala/urn_parser.rb +6 -2
- data/lib/pubid/iala.rb +1 -0
- data/lib/pubid/iana/builder.rb +3 -1
- data/lib/pubid/iana/identifier.rb +74 -6
- data/lib/pubid/iana/identifiers/registry.rb +36 -0
- data/lib/pubid/iana/parser.rb +1 -1
- data/lib/pubid/iana/renderer.rb +3 -0
- data/lib/pubid/iana/urn_generator.rb +5 -1
- data/lib/pubid/iana.rb +1 -0
- data/lib/pubid/identifier.rb +594 -10
- data/lib/pubid/identifier_metadata.rb +1 -0
- data/lib/pubid/idf/builder.rb +3 -3
- data/lib/pubid/idf/identifier.rb +18 -9
- data/lib/pubid/idf/parser.rb +1 -1
- data/lib/pubid/idf/renderer.rb +4 -4
- data/lib/pubid/idf/urn_generator.rb +1 -1
- data/lib/pubid/idf.rb +7 -1
- data/lib/pubid/iec/builder.rb +14 -7
- data/lib/pubid/iec/components.rb +0 -1
- data/lib/pubid/iec/identifier.rb +73 -83
- data/lib/pubid/iec/identifiers/amendment.rb +6 -2
- data/lib/pubid/iec/identifiers/component_specification.rb +0 -10
- data/lib/pubid/iec/identifiers/conformity_assessment.rb +0 -10
- data/lib/pubid/iec/identifiers/corrigendum.rb +2 -2
- data/lib/pubid/iec/identifiers/fragment_identifier.rb +2 -4
- data/lib/pubid/iec/identifiers/guide.rb +0 -16
- data/lib/pubid/iec/identifiers/international_standard.rb +15 -2
- data/lib/pubid/iec/identifiers/operational_document.rb +0 -10
- data/lib/pubid/iec/identifiers/publicly_available_specification.rb +0 -12
- data/lib/pubid/iec/identifiers/societal_technology_trend_report.rb +0 -10
- data/lib/pubid/iec/identifiers/systems_reference_document.rb +0 -10
- data/lib/pubid/iec/identifiers/technical_report.rb +0 -17
- data/lib/pubid/iec/identifiers/technical_specification.rb +0 -17
- data/lib/pubid/iec/identifiers/technology_report.rb +0 -10
- data/lib/pubid/iec/identifiers/test_report_form.rb +0 -11
- data/lib/pubid/iec/identifiers/white_paper.rb +0 -10
- data/lib/pubid/iec/identifiers/working_document.rb +24 -10
- data/lib/pubid/iec/parser.rb +2 -18
- data/lib/pubid/iec/renderer.rb +13 -10
- data/lib/pubid/iec/single_identifier.rb +33 -10
- data/lib/pubid/iec/urn_generator.rb +227 -61
- data/lib/pubid/iec/urn_parser.rb +42 -4
- data/lib/pubid/iec.rb +1 -0
- data/lib/pubid/ieee/aiee/identifier.rb +2 -2
- data/lib/pubid/ieee/aiee/parser.rb +1 -1
- data/lib/pubid/ieee/builder.rb +45 -1
- data/lib/pubid/ieee/identifiers/adopted_standard.rb +32 -0
- data/lib/pubid/ieee/identifiers/base.rb +42 -5
- data/lib/pubid/ieee/identifiers/csa_dual_published.rb +21 -0
- data/lib/pubid/ieee/identifiers/dual_published.rb +55 -0
- data/lib/pubid/ieee/identifiers/interpretation_identifier.rb +23 -0
- data/lib/pubid/ieee/identifiers/joint_development.rb +11 -7
- data/lib/pubid/ieee/identifiers/multi_numbered_identifier.rb +57 -5
- data/lib/pubid/ieee/identifiers/nesc/base.rb +2 -2
- data/lib/pubid/ieee/identifiers/nesc/draft.rb +2 -2
- data/lib/pubid/ieee/identifiers/nesc/handbook.rb +2 -2
- data/lib/pubid/ieee/identifiers/nesc/redline.rb +2 -2
- data/lib/pubid/ieee/identifiers/nesc/standard.rb +2 -2
- data/lib/pubid/ieee/ire/identifier.rb +2 -2
- data/lib/pubid/ieee/ire/parser.rb +1 -1
- data/lib/pubid/ieee/nesc/parser.rb +1 -1
- data/lib/pubid/ieee/parser.rb +1 -1
- data/lib/pubid/ieee/renderer.rb +5 -1
- data/lib/pubid/ieee/urn_generator.rb +32 -3
- data/lib/pubid/ieee.rb +7 -1
- data/lib/pubid/ieee_debug.rb +1 -0
- data/lib/pubid/ietf/builder.rb +24 -8
- data/lib/pubid/ietf/identifiers/base.rb +25 -28
- data/lib/pubid/ietf/identifiers/bcp.rb +11 -0
- data/lib/pubid/ietf/identifiers/fyi.rb +11 -0
- data/lib/pubid/ietf/identifiers/internet_draft.rb +18 -2
- data/lib/pubid/ietf/identifiers/rfc.rb +2 -0
- data/lib/pubid/ietf/identifiers/serialization.rb +51 -0
- data/lib/pubid/ietf/identifiers/std.rb +11 -0
- data/lib/pubid/ietf/identifiers.rb +1 -0
- data/lib/pubid/ietf/parser.rb +27 -9
- data/lib/pubid/ietf/renderer.rb +44 -6
- data/lib/pubid/ietf/urn_generator.rb +20 -6
- data/lib/pubid/ietf/urn_parser.rb +10 -2
- data/lib/pubid/ietf.rb +12 -2
- data/lib/pubid/iho/identifiers/base.rb +10 -3
- data/lib/pubid/iho/parser.rb +1 -1
- data/lib/pubid/iho.rb +1 -0
- data/lib/pubid/isbn/identifier.rb +25 -7
- data/lib/pubid/isbn/parser.rb +1 -1
- data/lib/pubid/isbn.rb +1 -0
- data/lib/pubid/iso/bundled_identifier.rb +5 -3
- data/lib/pubid/iso/combined_identifier.rb +5 -3
- data/lib/pubid/iso/identifier.rb +12 -7
- data/lib/pubid/iso/identifiers/directives_supplement.rb +4 -4
- data/lib/pubid/iso/parser.rb +1 -1
- data/lib/pubid/iso.rb +7 -1
- data/lib/pubid/itu/identifiers/addendum.rb +2 -2
- data/lib/pubid/itu/identifiers/amendment.rb +2 -2
- data/lib/pubid/itu/identifiers/base.rb +19 -5
- data/lib/pubid/itu/identifiers/corrigendum.rb +2 -2
- data/lib/pubid/itu/identifiers/errata.rb +2 -2
- data/lib/pubid/itu/identifiers/supplement.rb +2 -2
- data/lib/pubid/itu/parser.rb +1 -1
- data/lib/pubid/itu.rb +1 -0
- data/lib/pubid/jcgm/builder.rb +1 -1
- data/lib/pubid/jcgm/identifier.rb +10 -0
- data/lib/pubid/jcgm/identifiers/meeting.rb +1 -1
- data/lib/pubid/jcgm/parser.rb +10 -3
- data/lib/pubid/jcgm/renderer.rb +10 -3
- data/lib/pubid/jcgm/single_identifier.rb +24 -15
- data/lib/pubid/jcgm/urn_generator.rb +1 -1
- data/lib/pubid/jcgm/urn_parser.rb +156 -14
- data/lib/pubid/jcgm.rb +10 -0
- data/lib/pubid/jis/identifier.rb +10 -3
- data/lib/pubid/jis/parser.rb +1 -1
- data/lib/pubid/jis.rb +1 -0
- data/lib/pubid/nist/components/stage.rb +2 -2
- data/lib/pubid/nist/configuration.rb +6 -8
- data/lib/pubid/nist/identifiers/base.rb +12 -2
- data/lib/pubid/nist/identifiers/circular.rb +9 -2
- data/lib/pubid/nist/identifiers/circular_supplement.rb +14 -5
- data/lib/pubid/nist/identifiers/commercial_standards_monthly.rb +7 -1
- data/lib/pubid/nist/identifiers/crpl_report.rb +7 -1
- data/lib/pubid/nist/identifiers/handbook.rb +9 -2
- data/lib/pubid/nist/identifiers/internal_report.rb +12 -5
- data/lib/pubid/nist/identifiers/miscellaneous_publication.rb +14 -6
- data/lib/pubid/nist/identifiers/monograph.rb +14 -7
- data/lib/pubid/nist/identifiers/report.rb +14 -6
- data/lib/pubid/nist/parser.rb +1 -1
- data/lib/pubid/nist/supplement_identifier.rb +16 -1
- data/lib/pubid/nist.rb +7 -1
- data/lib/pubid/oasis/builder.rb +12 -10
- data/lib/pubid/oasis/identifier.rb +89 -11
- data/lib/pubid/oasis/parser.rb +1 -1
- data/lib/pubid/oasis.rb +7 -1
- data/lib/pubid/ogc/identifier.rb +9 -5
- data/lib/pubid/ogc/parser.rb +1 -1
- data/lib/pubid/ogc.rb +1 -0
- data/lib/pubid/oiml/builder.rb +5 -1
- data/lib/pubid/oiml/identifiers/basic_publication.rb +2 -0
- data/lib/pubid/oiml/identifiers/bulletin.rb +36 -0
- data/lib/pubid/oiml/identifiers/code_number.rb +90 -0
- data/lib/pubid/oiml/identifiers/document.rb +2 -0
- data/lib/pubid/oiml/identifiers/expert_report.rb +2 -0
- data/lib/pubid/oiml/identifiers/guide.rb +2 -0
- data/lib/pubid/oiml/identifiers/recommendation.rb +2 -0
- data/lib/pubid/oiml/identifiers/seminar_report.rb +2 -0
- data/lib/pubid/oiml/identifiers/vocabulary.rb +2 -0
- data/lib/pubid/oiml/identifiers.rb +1 -0
- data/lib/pubid/oiml/parser.rb +1 -1
- data/lib/pubid/oiml/single_identifier.rb +26 -42
- data/lib/pubid/oiml/supplement_identifier.rb +27 -0
- data/lib/pubid/oiml.rb +7 -1
- data/lib/pubid/omg/builder.rb +1 -0
- data/lib/pubid/omg/identifier.rb +20 -3
- data/lib/pubid/omg/parser.rb +64 -11
- data/lib/pubid/omg/renderer.rb +13 -1
- data/lib/pubid/omg.rb +2 -1
- data/lib/pubid/parser/grammar.rb +74 -0
- data/lib/pubid/parser.rb +2 -0
- data/lib/pubid/parsers/mr_string.rb +19 -0
- data/lib/pubid/plateau/parser.rb +1 -1
- data/lib/pubid/plateau.rb +10 -0
- data/lib/pubid/prefixes_support.rb +1 -0
- data/lib/pubid/renderers/annotator.rb +233 -0
- data/lib/pubid/renderers/base.rb +13 -0
- data/lib/pubid/renderers/directives_renderer.rb +3 -3
- data/lib/pubid/renderers/human_readable.rb +3 -3
- data/lib/pubid/renderers/mr_string.rb +15 -9
- data/lib/pubid/renderers.rb +1 -0
- data/lib/pubid/rendering/numbering.rb +25 -7
- data/lib/pubid/rendering.rb +1 -0
- data/lib/pubid/sae/identifiers/base.rb +9 -2
- data/lib/pubid/sae/parser.rb +1 -1
- data/lib/pubid/sae.rb +1 -0
- data/lib/pubid/schema/declaration.rb +33 -0
- data/lib/pubid/schema/error.rb +14 -0
- data/lib/pubid/schema/identifier_type.rb +23 -0
- data/lib/pubid/schema/loader.rb +117 -0
- data/lib/pubid/schema/typed_stage.rb +24 -0
- data/lib/pubid/schema.rb +25 -0
- data/lib/pubid/tgpp/builder.rb +1 -1
- data/lib/pubid/tgpp/identifier.rb +21 -5
- data/lib/pubid/tgpp/identifiers/technical_report.rb +2 -2
- data/lib/pubid/tgpp/identifiers/technical_specification.rb +2 -2
- data/lib/pubid/tgpp/parser.rb +22 -6
- data/lib/pubid/tgpp/renderer.rb +21 -5
- data/lib/pubid/tgpp/urn_generator.rb +27 -8
- data/lib/pubid/tgpp/urn_parser.rb +8 -6
- data/lib/pubid/tgpp.rb +7 -1
- data/lib/pubid/type_resolver.rb +16 -1
- data/lib/pubid/un/identifier.rb +6 -3
- data/lib/pubid/un/parser.rb +1 -1
- data/lib/pubid/un.rb +1 -0
- data/lib/pubid/urn_parser/errors.rb +8 -1
- data/lib/pubid/urn_parser.rb +1 -0
- data/lib/pubid/utils.rb +1 -0
- data/lib/pubid/version.rb +1 -1
- data/lib/pubid/w3c/builder.rb +5 -5
- data/lib/pubid/w3c/identifier.rb +38 -9
- data/lib/pubid/w3c/parser.rb +1 -1
- data/lib/pubid/w3c/renderer.rb +1 -1
- data/lib/pubid/w3c/urn_generator.rb +2 -2
- data/lib/pubid/w3c.rb +1 -0
- data/lib/pubid/xsf/identifier.rb +9 -5
- data/lib/pubid/xsf/parser.rb +17 -4
- data/lib/pubid/xsf.rb +7 -1
- data/lib/pubid.rb +399 -21
- data/lib/tasks/conformance.rake +34 -0
- data/lib/tasks/docs.rake +3 -3
- data/lib/tasks/schema.rake +88 -0
- data/schema/core/joint_prefixes.yaml +24 -0
- data/schema/iec.yaml +1017 -0
- data/schema/iso.yaml +1586 -0
- data/schema/schema.schema.yaml +96 -0
- metadata +38 -17
- data/archived-gems/pubid-ccsds/update_codes.yaml +0 -1
- data/archived-gems/pubid-iec/stages.yaml +0 -129
- data/archived-gems/pubid-iec/update_codes.yaml +0 -67
- data/archived-gems/pubid-ieee/update_codes.yaml +0 -104
- data/archived-gems/pubid-iso/stages.yaml +0 -106
- data/archived-gems/pubid-iso/update_codes.yaml +0 -4
- data/archived-gems/pubid-itu/i18n.yaml +0 -13
- data/archived-gems/pubid-itu/series.yaml +0 -42
- data/archived-gems/pubid-nist/publishers.yaml +0 -6
- data/archived-gems/pubid-nist/series.yaml +0 -121
- data/archived-gems/pubid-nist/update_codes.yaml +0 -93
- data/archived-gems/pubid-plateau/update_codes.yaml +0 -6
- data/lib/pubid/cen_cenelec/supplement_identifier.rb +0 -48
- data/lib/pubid/iec/components/code.rb +0 -36
- /data/{archived-gems/pubid-nist → data/nist}/stages.yaml +0 -0
data/lib/pubid/oiml.rb
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
|
+
require "pubid"
|
|
3
4
|
module Pubid
|
|
4
5
|
module Oiml
|
|
5
6
|
extend Pubid::PrefixesSupport
|
|
@@ -19,8 +20,13 @@ module Pubid
|
|
|
19
20
|
autoload :UrnParser, "#{__dir__}/oiml/urn_parser"
|
|
20
21
|
|
|
21
22
|
def self.parse(identifier)
|
|
23
|
+
unless identifier.is_a?(String)
|
|
24
|
+
raise Pubid::Errors::InvalidInputError,
|
|
25
|
+
Pubid::INPUT_NOT_A_STRING_MESSAGE
|
|
26
|
+
end
|
|
27
|
+
|
|
22
28
|
if identifier.length > Pubid::MAX_INPUT_LENGTH
|
|
23
|
-
raise
|
|
29
|
+
raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
|
|
24
30
|
end
|
|
25
31
|
|
|
26
32
|
parser = Parser.new
|
data/lib/pubid/omg/builder.rb
CHANGED
data/lib/pubid/omg/identifier.rb
CHANGED
|
@@ -13,6 +13,19 @@ module Pubid
|
|
|
13
13
|
# rendering.
|
|
14
14
|
attribute :version, :string
|
|
15
15
|
|
|
16
|
+
# The document part: the volume or format segment after the version,
|
|
17
|
+
# e.g. "Superstructure", "Infrastructure", "PDF".
|
|
18
|
+
#
|
|
19
|
+
# This retypes the `part` that ::Pubid::Identifier declares as a
|
|
20
|
+
# Components::Code, because an OMG part is a name and not a numbered
|
|
21
|
+
# part. Reusing the inherited name is what gives relaton `remove_part!`
|
|
22
|
+
# as `exclude(:part)`, and what lets Renderers::Annotator wrap the value
|
|
23
|
+
# from its own TOKENS table. The declaration sits once, on the class
|
|
24
|
+
# every OMG identifier inherits from, whose body lives in this one file
|
|
25
|
+
# and is never reopened — the placement the number-retype tranches
|
|
26
|
+
# require.
|
|
27
|
+
attribute :part, :string
|
|
28
|
+
|
|
16
29
|
OMG_TYPE_MAP = {
|
|
17
30
|
"pubid:omg:specification" => "Pubid::Omg::Identifiers::Specification",
|
|
18
31
|
}.freeze
|
|
@@ -21,6 +34,7 @@ module Pubid
|
|
|
21
34
|
map "_type", to: :_type, polymorphic_map: OMG_TYPE_MAP
|
|
22
35
|
map "acronym", to: :acronym
|
|
23
36
|
map "version", to: :version
|
|
37
|
+
map "part", to: :part
|
|
24
38
|
end
|
|
25
39
|
|
|
26
40
|
PUBLISHER = "OMG"
|
|
@@ -30,14 +44,17 @@ module Pubid
|
|
|
30
44
|
end
|
|
31
45
|
|
|
32
46
|
def self.parse(identifier)
|
|
47
|
+
unless identifier.is_a?(String)
|
|
48
|
+
raise Pubid::Errors::InvalidInputError,
|
|
49
|
+
Pubid::INPUT_NOT_A_STRING_MESSAGE
|
|
50
|
+
end
|
|
51
|
+
|
|
33
52
|
if identifier.length > Pubid::MAX_INPUT_LENGTH
|
|
34
|
-
raise
|
|
53
|
+
raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
|
|
35
54
|
end
|
|
36
55
|
|
|
37
56
|
parsed = Parser.parse(identifier)
|
|
38
57
|
Builder.build(parsed)
|
|
39
|
-
rescue Parslet::ParseFailed => e
|
|
40
|
-
raise "Failed to parse OMG identifier '#{identifier}': #{e.message}"
|
|
41
58
|
end
|
|
42
59
|
end
|
|
43
60
|
end
|
data/lib/pubid/omg/parser.rb
CHANGED
|
@@ -7,27 +7,80 @@ module Pubid
|
|
|
7
7
|
# Parslet grammar for OMG specification identifiers.
|
|
8
8
|
#
|
|
9
9
|
# Accepts:
|
|
10
|
-
# OMG {ACRONYM}[ {VERSION}]
|
|
10
|
+
# OMG {ACRONYM}[ {VERSION}][ {PART}]
|
|
11
11
|
#
|
|
12
|
-
# ACRONYM is
|
|
13
|
-
#
|
|
14
|
-
#
|
|
15
|
-
|
|
12
|
+
# ACRONYM is the URL segment of https://www.omg.org/spec/<ACRONYM>/, e.g.
|
|
13
|
+
# "UML", "DDS-XTypes", "EDMC-FIBO/BE", "VSIPL++", "smartant".
|
|
14
|
+
# VERSION is digits/dots, optionally followed by a beta label, e.g. "1.0",
|
|
15
|
+
# "2.5.1", "5 beta 3", "2.5 beta".
|
|
16
|
+
# PART is the volume or format segment, e.g. "Superstructure", "PDF". A
|
|
17
|
+
# space or a slash separates it.
|
|
18
|
+
class Parser < ::Pubid::Parser::Grammar
|
|
16
19
|
rule(:space) { str(" ") }
|
|
17
20
|
|
|
18
|
-
# Acronym:
|
|
19
|
-
#
|
|
20
|
-
|
|
21
|
+
# Acronym: the URL segment OMG gives the specification, kept verbatim,
|
|
22
|
+
# because a consumer builds the URL from it. It starts with a letter,
|
|
23
|
+
# which may be lower case ("smartant"). Letters, digits and "+" follow
|
|
24
|
+
# ("SysML", "AMI4CCM", "VSIPL++"). A hyphen or a slash joins further
|
|
25
|
+
# segments ("DDS-PSM-Cxx", "EDMC-FIBO/BE"). Each segment is non-empty,
|
|
26
|
+
# so a hyphen or a slash never ends the acronym.
|
|
27
|
+
#
|
|
28
|
+
# The slash in "EDMC-FIBO/BE" belongs to the acronym: OMG serves the
|
|
29
|
+
# document at /spec/EDMC-FIBO/BE/. A slash after the version separates
|
|
30
|
+
# the document part instead (see the identifier rule).
|
|
31
|
+
rule(:acronym_char) { match("[A-Za-z0-9+]") }
|
|
32
|
+
|
|
33
|
+
rule(:acronym) do
|
|
34
|
+
(match("[A-Za-z]") >> acronym_char.repeat >>
|
|
35
|
+
(match("[-/]") >> acronym_char.repeat(1)).repeat).as(:acronym)
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
# Version: digits with optional dots, optionally followed by " beta" and
|
|
39
|
+
# an optional beta number. OMG writes the label both ways: the document
|
|
40
|
+
# at /spec/UML/2.5/Beta1 gives its own version as "2.5 beta", and DDS 1.4
|
|
41
|
+
# supersedes /spec/DDS/1.4/Beta2.
|
|
42
|
+
#
|
|
43
|
+
# The beta number must stay optional here. The document part below would
|
|
44
|
+
# otherwise swallow a bare "beta" and report the version as "2.5" — a
|
|
45
|
+
# silent wrong answer, not a parse failure.
|
|
46
|
+
#
|
|
47
|
+
# Both halves of the beta label end at a word boundary. Parslet never
|
|
48
|
+
# backtracks into a `.maybe` that already succeeded, so an unanchored
|
|
49
|
+
# literal makes the version eat the front of a document part and then
|
|
50
|
+
# reject the whole identifier: "OMG DDS 1.4 beta2" and
|
|
51
|
+
# "OMG DDS 1.4 betawave" would raise instead of reading the part.
|
|
52
|
+
rule(:word_boundary) { match("[A-Za-z0-9]").absent? }
|
|
53
|
+
|
|
54
|
+
rule(:beta) do
|
|
55
|
+
str(" beta") >> word_boundary >>
|
|
56
|
+
(space >> match("[0-9]").repeat(1) >> word_boundary).maybe
|
|
57
|
+
end
|
|
21
58
|
|
|
22
|
-
# Version: digits with optional dots, optionally followed by " beta N".
|
|
23
59
|
rule(:version) do
|
|
24
60
|
(match("[0-9]").repeat(1) >>
|
|
25
61
|
(str(".") >> match("[0-9]").repeat(1)).repeat >>
|
|
26
|
-
|
|
62
|
+
beta.maybe).as(:version)
|
|
27
63
|
end
|
|
28
64
|
|
|
65
|
+
# After the version, OMG separates the document part with either a
|
|
66
|
+
# space or a slash. The renderer prints a space, so the two spellings of
|
|
67
|
+
# one document stay equal.
|
|
68
|
+
rule(:part_separator) { space | str("/") }
|
|
69
|
+
|
|
70
|
+
# Document part: the volume or format segment OMG puts after the
|
|
71
|
+
# version. UML 2.1.1 is two documents, Superstructure and
|
|
72
|
+
# Infrastructure, and the URL carries the same segment
|
|
73
|
+
# (/spec/UML/2.1.1/Superstructure). A format name (/spec/DDS/1.4/PDF)
|
|
74
|
+
# occupies the same position.
|
|
75
|
+
rule(:part) { match("[A-Za-z0-9]").repeat(1).as(:part) }
|
|
76
|
+
|
|
77
|
+
# Only a space separates a part that follows the acronym directly. The
|
|
78
|
+
# acronym rule takes a slash there, so a slash separator is legal only
|
|
79
|
+
# after a version.
|
|
29
80
|
rule(:identifier) do
|
|
30
|
-
str("OMG") >> space >> acronym >>
|
|
81
|
+
str("OMG") >> space >> acronym >>
|
|
82
|
+
((space >> version >> (part_separator >> part).maybe) |
|
|
83
|
+
(space >> part)).maybe
|
|
31
84
|
end
|
|
32
85
|
|
|
33
86
|
rule(:root) { identifier }
|
data/lib/pubid/omg/renderer.rb
CHANGED
|
@@ -7,13 +7,25 @@ module Pubid
|
|
|
7
7
|
# Produces:
|
|
8
8
|
# "OMG AMI4CCM 1.0"
|
|
9
9
|
# "OMG UML 2.5.1"
|
|
10
|
+
# "OMG UML 2.1.1 Superstructure"
|
|
10
11
|
# "OMG CORBA"
|
|
12
|
+
#
|
|
13
|
+
# The document part always prints behind a space. OMG writes it behind
|
|
14
|
+
# either a space or a slash, and the parser takes both, so normalizing
|
|
15
|
+
# here is what keeps the two spellings of one document equal.
|
|
11
16
|
class Renderer < ::Pubid::Renderers::Base
|
|
12
17
|
def render(**_opts)
|
|
13
18
|
result = "OMG #{@id.acronym}"
|
|
14
|
-
result += " #{@id.version}" if @id.version
|
|
19
|
+
result += " #{@id.version}" if present?(@id.version)
|
|
20
|
+
result += " #{@id.part}" if present?(@id.part)
|
|
15
21
|
result
|
|
16
22
|
end
|
|
23
|
+
|
|
24
|
+
private
|
|
25
|
+
|
|
26
|
+
def present?(value)
|
|
27
|
+
value && !value.empty?
|
|
28
|
+
end
|
|
17
29
|
end
|
|
18
30
|
end
|
|
19
31
|
end
|
data/lib/pubid/omg.rb
CHANGED
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
|
+
require "pubid"
|
|
3
4
|
module Pubid
|
|
4
5
|
# OMG (Object Management Group) specification flavor.
|
|
5
6
|
#
|
|
6
7
|
# Covers formal OMG specifications identified by an acronym (UML, SysML,
|
|
7
|
-
# CORBA,
|
|
8
|
+
# CORBA, DDS-XTypes, EDMC-FIBO/BE, ...) with optional version (`1.0`, `2.5.1`,
|
|
8
9
|
# `5 beta 3`).
|
|
9
10
|
module Omg
|
|
10
11
|
extend Pubid::PrefixesSupport
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "parslet"
|
|
4
|
+
require_relative "../errors"
|
|
5
|
+
|
|
6
|
+
module Pubid
|
|
7
|
+
module Parser
|
|
8
|
+
# The superclass of every flavor grammar.
|
|
9
|
+
#
|
|
10
|
+
# Its only job is to keep parslet out of pubid's public contract. Every
|
|
11
|
+
# grammar failure originates in `Parslet::Atoms::Base#parse`, which
|
|
12
|
+
# `Parslet::Parser` does not override, so one override here covers all 46
|
|
13
|
+
# grammars, both public entry points of every flavor, and the cross-flavor
|
|
14
|
+
# delegation routes (`Pubid::Iso.parse("ITU-T G.711")` runs ITU's grammar)
|
|
15
|
+
# — without touching the 131 `parse` methods.
|
|
16
|
+
#
|
|
17
|
+
# The name is `Grammar`, not `Base`, because {Pubid::Parsers::Base} already
|
|
18
|
+
# exists and means something unrelated (the MR-string parser).
|
|
19
|
+
#
|
|
20
|
+
# Known limit: a rule atom parsed directly, `Parser.new.<rule>.parse(str)`,
|
|
21
|
+
# bypasses this and raises a bare `Parslet::ParseFailed`. Nothing in the
|
|
22
|
+
# gem does that.
|
|
23
|
+
class Grammar < ::Parslet::Parser
|
|
24
|
+
# @param io [String, IO]
|
|
25
|
+
# @param options [Hash] passed through to parslet
|
|
26
|
+
# @raise [Pubid::Errors::ParseError]
|
|
27
|
+
def parse(io, options = {})
|
|
28
|
+
super
|
|
29
|
+
rescue ::Pubid::Errors::ParseError
|
|
30
|
+
# A nested grammar already wrapped it. Keep the inner flavor and input.
|
|
31
|
+
raise
|
|
32
|
+
rescue ::Parslet::ParseFailed => e
|
|
33
|
+
raise wrap_parse_failure(e, io)
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
private
|
|
37
|
+
|
|
38
|
+
# @param error [Parslet::ParseFailed]
|
|
39
|
+
# @param io [String, IO] what was handed to {#parse}
|
|
40
|
+
# @return [Pubid::Errors::ParseError]
|
|
41
|
+
def wrap_parse_failure(error, io)
|
|
42
|
+
::Pubid::Errors::ParseError.new(
|
|
43
|
+
error.message,
|
|
44
|
+
error.parse_failure_cause,
|
|
45
|
+
input: io.is_a?(String) ? io : nil,
|
|
46
|
+
flavor: pubid_flavor_name,
|
|
47
|
+
)
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
# "Pubid::Iso::Parser" -> "iso"; "Pubid::Ieee::Aiee::Parser" -> "ieee".
|
|
51
|
+
#
|
|
52
|
+
# Resolved through {Pubid::Registry} rather than taken from the module
|
|
53
|
+
# name, because the two disagree: `Pubid::Tgpp` registers as `"3gpp"`,
|
|
54
|
+
# and `Pubid::CenCenelec` registers twice (`"cen_cenelec"` first, then
|
|
55
|
+
# the `"cen"` alias). Reporting the registered name is what lets a caller
|
|
56
|
+
# feed `error.flavor` straight back to `Pubid::Registry.get`.
|
|
57
|
+
# @return [String, nil]
|
|
58
|
+
def pubid_flavor_name
|
|
59
|
+
parts = self.class.name.to_s.split("::")
|
|
60
|
+
return nil unless parts[0] == "Pubid" && parts.length > 1
|
|
61
|
+
|
|
62
|
+
registered_flavor_name(parts[1]) || parts[1].downcase
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
# @param module_name [String] e.g. "Tgpp"
|
|
66
|
+
# @return [String, nil] the registered name, nil if not registered
|
|
67
|
+
def registered_flavor_name(module_name)
|
|
68
|
+
::Pubid::Registry.canonical_name(::Pubid.const_get(module_name))
|
|
69
|
+
rescue ::NameError
|
|
70
|
+
nil
|
|
71
|
+
end
|
|
72
|
+
end
|
|
73
|
+
end
|
|
74
|
+
end
|
data/lib/pubid/parser.rb
CHANGED
|
@@ -101,6 +101,25 @@ module Pubid
|
|
|
101
101
|
raise ArgumentError, "Unknown flavor: #{flavor}" unless flavor_module
|
|
102
102
|
|
|
103
103
|
identifier_string = convert_to_human_readable(mr_string)
|
|
104
|
+
|
|
105
|
+
# Refuse to hand the flavor a string this parser did not change.
|
|
106
|
+
#
|
|
107
|
+
# `detect_flavor` falls back to `:iso` for any publisher it does not
|
|
108
|
+
# know, and `Pubid::Iso.parse` sends an MR-shaped string straight back
|
|
109
|
+
# here — so a string that survives conversion unchanged recurses
|
|
110
|
+
# forever. `ECMA-426 ed1` is the standing example: "ECMA" is absent
|
|
111
|
+
# from FLAVOR_MAP, the string matches the MR shape heuristic
|
|
112
|
+
# (`/\A[A-Z]{2,}[.-]/`), and conversion is a fixed point.
|
|
113
|
+
# `Pubid::Iso.parse("ECMA-426 ed1")` raised SystemStackError.
|
|
114
|
+
#
|
|
115
|
+
# A fixed point means "this was never an MR string", which is a parse
|
|
116
|
+
# failure — the class the cross-flavor contract requires, and the one
|
|
117
|
+
# `Pubid.parse` treats as "try the next flavor".
|
|
118
|
+
if identifier_string == mr_string && flavor_module == Pubid::Iso
|
|
119
|
+
raise Parslet::ParseFailed,
|
|
120
|
+
"#{mr_string.inspect} is not an MR string"
|
|
121
|
+
end
|
|
122
|
+
|
|
104
123
|
flavor_module.parse(identifier_string)
|
|
105
124
|
end
|
|
106
125
|
|
data/lib/pubid/plateau/parser.rb
CHANGED
data/lib/pubid/plateau.rb
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
|
+
require "pubid"
|
|
3
4
|
require "parslet"
|
|
4
5
|
|
|
5
6
|
module Pubid
|
|
@@ -19,6 +20,15 @@ module Pubid
|
|
|
19
20
|
autoload :UrnParser, "#{__dir__}/plateau/urn_parser"
|
|
20
21
|
|
|
21
22
|
def self.parse(input)
|
|
23
|
+
unless input.is_a?(String)
|
|
24
|
+
raise Pubid::Errors::InvalidInputError,
|
|
25
|
+
Pubid::INPUT_NOT_A_STRING_MESSAGE
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
if input.length > Pubid::MAX_INPUT_LENGTH
|
|
29
|
+
raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
|
|
30
|
+
end
|
|
31
|
+
|
|
22
32
|
# Apply legacy update_codes normalization first
|
|
23
33
|
normalized = Core::UpdateCodes.apply(input, :plateau)
|
|
24
34
|
parser = Parser.new
|
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Pubid
|
|
4
|
+
module Renderers
|
|
5
|
+
# Adds semantic <span class="..."> markers to an already-rendered
|
|
6
|
+
# identifier string.
|
|
7
|
+
#
|
|
8
|
+
# In pubid 1.x annotation was applied in ONE place — `prerender_params`
|
|
9
|
+
# wrapped every value in the param hash — so all 40-odd flavors got it for
|
|
10
|
+
# free. pubid 2 replaced that hash with typed components and per-flavor
|
|
11
|
+
# renderers, and the annotation went with it: `Renderers::Base#annotate`
|
|
12
|
+
# exists, but only the HumanReadable family calls it. All 39 flavor
|
|
13
|
+
# renderers receive `context.annotated` and ignore it, so
|
|
14
|
+
# `to_s(annotated: true)` returned plain text for every flavor but ISO.
|
|
15
|
+
#
|
|
16
|
+
# Re-annotating inside 39 renderers would mean 39 chances to drift. This
|
|
17
|
+
# class recovers the markup from the OUTSIDE instead: it asks the
|
|
18
|
+
# identifier for its own component values and wraps each one where it
|
|
19
|
+
# appears in the rendered string. That is one implementation, and a flavor
|
|
20
|
+
# added tomorrow is covered without touching its renderer.
|
|
21
|
+
#
|
|
22
|
+
# It is a fallback, not a replacement. `Identifier#render` uses it only
|
|
23
|
+
# when the renderer produced no span of its own, so ISO's exact,
|
|
24
|
+
# render-time placement still wins.
|
|
25
|
+
#
|
|
26
|
+
# Deliberate limits:
|
|
27
|
+
#
|
|
28
|
+
# * A token that does not appear verbatim in the output is skipped. A
|
|
29
|
+
# renderer may transform a value (abbreviate it, change its case), and a
|
|
30
|
+
# missing span is a far better outcome than a wrong one or a crash.
|
|
31
|
+
# * Matching walks left to right behind a cursor, so a later token can
|
|
32
|
+
# never match text an earlier one already claimed — that is what stops
|
|
33
|
+
# the part "1" of `ISO 1234-1` from matching inside "1234".
|
|
34
|
+
class Annotator
|
|
35
|
+
# Semantic class for each token, in the order tokens appear in a printed
|
|
36
|
+
# identifier. Order matters: it is the sequence the cursor walks.
|
|
37
|
+
TOKENS = [
|
|
38
|
+
[:publisher, "publisher"],
|
|
39
|
+
[:copublishers, "publisher"],
|
|
40
|
+
%i[typed_stage typed_stage_css],
|
|
41
|
+
[:type, "doctype"],
|
|
42
|
+
[:stage, "stage"],
|
|
43
|
+
[:number, "docnumber"],
|
|
44
|
+
[:part, "part"],
|
|
45
|
+
[:subpart, "part"],
|
|
46
|
+
[:stage_iteration, "iteration"],
|
|
47
|
+
[:year, "year"],
|
|
48
|
+
[:edition, "edition"],
|
|
49
|
+
[:languages, "language"],
|
|
50
|
+
].freeze
|
|
51
|
+
|
|
52
|
+
# Characters that may not sit directly against a match, so a token never
|
|
53
|
+
# binds to the middle of a longer run of the same character class.
|
|
54
|
+
WORD_CHAR = /[A-Za-z0-9]/
|
|
55
|
+
|
|
56
|
+
# How deep to follow nested identifiers. A wrapper around a wrapper is
|
|
57
|
+
# real (BSI adopts a CEN prestandard that adopts an ISO standard); four
|
|
58
|
+
# levels is past anything the corpus holds, and the cap is here so a
|
|
59
|
+
# cyclic `base` cannot hang a rendering call.
|
|
60
|
+
MAX_NESTING = 4
|
|
61
|
+
|
|
62
|
+
def initialize(identifier, context = nil)
|
|
63
|
+
@id = identifier
|
|
64
|
+
@context = context
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
# @param rendered [String] the plain rendering of {@id}
|
|
68
|
+
# @return [String] the same string with semantic spans inserted
|
|
69
|
+
def annotate(rendered)
|
|
70
|
+
return rendered unless rendered.is_a?(String) && !rendered.empty?
|
|
71
|
+
|
|
72
|
+
cursor = 0
|
|
73
|
+
result = +""
|
|
74
|
+
|
|
75
|
+
ordered_tokens(rendered).each do |text, css_class|
|
|
76
|
+
index = find_token(rendered, text, cursor)
|
|
77
|
+
next if index.nil?
|
|
78
|
+
|
|
79
|
+
result << rendered[cursor...index]
|
|
80
|
+
result << %(<span class="#{css_class}">#{text}</span>)
|
|
81
|
+
cursor = index + text.length
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
result << rendered[cursor..]
|
|
85
|
+
result
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
private
|
|
89
|
+
|
|
90
|
+
# Tokens sorted by where they actually appear, not by the order
|
|
91
|
+
# {TOKENS} lists them.
|
|
92
|
+
#
|
|
93
|
+
# Flavors disagree about layout: ISO prints the number before the type
|
|
94
|
+
# ("ISO/IEC TR 2131"), CCSDS prints it after ("CCSDS 121.0-B-2"). Walking
|
|
95
|
+
# in declaration order pushed the cursor past the number for CCSDS, so
|
|
96
|
+
# "121" was never annotated. Sorting by first occurrence makes the walk
|
|
97
|
+
# follow the printed identifier instead of a fixed idea of one.
|
|
98
|
+
def ordered_tokens(rendered)
|
|
99
|
+
tokens = []
|
|
100
|
+
each_token { |text, css_class| tokens << [text, css_class] }
|
|
101
|
+
|
|
102
|
+
decorated = tokens.each_with_index.map do |token, i|
|
|
103
|
+
position = find_token(rendered, token.first, 0) || rendered.length
|
|
104
|
+
[position, i, token]
|
|
105
|
+
end
|
|
106
|
+
|
|
107
|
+
decorated.sort_by { |position, i, _| [position, i] }.map(&:last)
|
|
108
|
+
end
|
|
109
|
+
|
|
110
|
+
# Yields [text, css_class] for every annotatable token this identifier
|
|
111
|
+
# actually carries, then every one its nested identifiers carry.
|
|
112
|
+
def each_token(&)
|
|
113
|
+
emit_tokens(@id, 0, &)
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
# A wrapper — an adoption, a supplement, a bundle — carries no number of
|
|
117
|
+
# its own and prints the document it wraps, so the annotatable tokens in
|
|
118
|
+
# its string belong to `base`. Reading only the wrapper's own attributes
|
|
119
|
+
# is why every CSA container and CIE's supplement rendered plain while
|
|
120
|
+
# accepting the flag.
|
|
121
|
+
#
|
|
122
|
+
# Order does not matter here: `ordered_tokens` sorts by first occurrence,
|
|
123
|
+
# so a nested token lands where it actually appears in the string.
|
|
124
|
+
def emit_tokens(id, depth, &)
|
|
125
|
+
return if depth > MAX_NESTING
|
|
126
|
+
|
|
127
|
+
tokens_for(id).each do |attr_name, css_class|
|
|
128
|
+
Array(token_values(id, attr_name)).each do |value|
|
|
129
|
+
text = token_text(value)
|
|
130
|
+
next if text.nil? || text.empty?
|
|
131
|
+
|
|
132
|
+
yield text, resolve_class(css_class, value)
|
|
133
|
+
end
|
|
134
|
+
end
|
|
135
|
+
|
|
136
|
+
nested_identifiers(id).each { |nested| emit_tokens(nested, depth + 1, &) }
|
|
137
|
+
end
|
|
138
|
+
|
|
139
|
+
# {TOKENS} unless the identifier names more of its own — see
|
|
140
|
+
# `Pubid::Identifier#annotation_tokens`.
|
|
141
|
+
def tokens_for(id)
|
|
142
|
+
id.respond_to?(:annotation_tokens) ? id.annotation_tokens : TOKENS
|
|
143
|
+
end
|
|
144
|
+
|
|
145
|
+
# The identifiers this one wraps, under any of the four names the gem
|
|
146
|
+
# uses: `base` (the uniform parent accessor), the `ids` / `identifiers`
|
|
147
|
+
# collections that bundles and consolidated identifiers hold instead, and
|
|
148
|
+
# CSA `Bundled`'s `bundled_with` — which holds the amendments a
|
|
149
|
+
# consolidation prints, and whose values it composes into the string from
|
|
150
|
+
# their components rather than from their own `to_s`.
|
|
151
|
+
def nested_identifiers(id)
|
|
152
|
+
%i[base ids identifiers bundled_with].flat_map do |name|
|
|
153
|
+
next [] unless id.respond_to?(name)
|
|
154
|
+
|
|
155
|
+
begin
|
|
156
|
+
Array(id.public_send(name))
|
|
157
|
+
rescue StandardError
|
|
158
|
+
[]
|
|
159
|
+
end
|
|
160
|
+
end.grep(::Pubid::Identifier)
|
|
161
|
+
end
|
|
162
|
+
|
|
163
|
+
def token_values(id, attr_name)
|
|
164
|
+
return nil unless id.respond_to?(attr_name)
|
|
165
|
+
|
|
166
|
+
id.public_send(attr_name)
|
|
167
|
+
rescue StandardError
|
|
168
|
+
# A derived reader may assume state a partial identifier lacks. A
|
|
169
|
+
# missing span is not worth an exception on a rendering path.
|
|
170
|
+
nil
|
|
171
|
+
end
|
|
172
|
+
|
|
173
|
+
# The printed form of one component. Components render themselves through
|
|
174
|
+
# the context; a bare scalar is already its own text.
|
|
175
|
+
def token_text(value)
|
|
176
|
+
return nil if value.nil?
|
|
177
|
+
|
|
178
|
+
text = if value.respond_to?(:render)
|
|
179
|
+
value.render(context: @context)
|
|
180
|
+
else
|
|
181
|
+
value
|
|
182
|
+
end
|
|
183
|
+
text.to_s.strip
|
|
184
|
+
rescue StandardError
|
|
185
|
+
nil
|
|
186
|
+
end
|
|
187
|
+
|
|
188
|
+
def resolve_class(css_class, value)
|
|
189
|
+
return css_class unless css_class == :typed_stage_css
|
|
190
|
+
|
|
191
|
+
TypedStageClass.for(value)
|
|
192
|
+
end
|
|
193
|
+
|
|
194
|
+
# First occurrence of +text+ at or after +cursor+ that is not embedded in
|
|
195
|
+
# a longer word — so "1" matches the part in "ISO 1234-1", never the "1"
|
|
196
|
+
# inside "1234".
|
|
197
|
+
def find_token(rendered, text, cursor)
|
|
198
|
+
at = cursor
|
|
199
|
+
while (index = rendered.index(text, at))
|
|
200
|
+
return index if standalone?(rendered, index, text.length)
|
|
201
|
+
|
|
202
|
+
at = index + 1
|
|
203
|
+
end
|
|
204
|
+
nil
|
|
205
|
+
end
|
|
206
|
+
|
|
207
|
+
def standalone?(rendered, index, length)
|
|
208
|
+
before = index.zero? ? nil : rendered[index - 1]
|
|
209
|
+
after = rendered[index + length]
|
|
210
|
+
|
|
211
|
+
!WORD_CHAR.match?(before.to_s) && !WORD_CHAR.match?(after.to_s)
|
|
212
|
+
end
|
|
213
|
+
|
|
214
|
+
# The typed-stage class depends on the stage's type code, the same
|
|
215
|
+
# mapping `Renderers::Base#typed_stage_css` applies. Kept here rather
|
|
216
|
+
# than reached for through a private method on another object.
|
|
217
|
+
module TypedStageClass
|
|
218
|
+
MAP = {
|
|
219
|
+
"amd" => "amendment",
|
|
220
|
+
"cor" => "corrigendum",
|
|
221
|
+
"add" => "addendum",
|
|
222
|
+
}.freeze
|
|
223
|
+
|
|
224
|
+
def self.for(typed_stage)
|
|
225
|
+
code = typed_stage.respond_to?(:type_code) ? typed_stage.type_code.to_s : ""
|
|
226
|
+
return "stage" if code.empty? || code == "is"
|
|
227
|
+
|
|
228
|
+
MAP[code] || "doctype"
|
|
229
|
+
end
|
|
230
|
+
end
|
|
231
|
+
end
|
|
232
|
+
end
|
|
233
|
+
end
|
data/lib/pubid/renderers/base.rb
CHANGED
|
@@ -41,6 +41,19 @@ module Pubid
|
|
|
41
41
|
%(#{lead}<span class="#{css_class}">#{core}</span>#{trail})
|
|
42
42
|
end
|
|
43
43
|
|
|
44
|
+
# Render a value that may be a component or a bare scalar.
|
|
45
|
+
#
|
|
46
|
+
# `number`, `part` and `subpart` are `Components::Code` on
|
|
47
|
+
# ::Pubid::Identifier but a plain `:string` in a growing number of
|
|
48
|
+
# flavors, so a shared renderer cannot assume the format-aware
|
|
49
|
+
# `#render(context:)` seam is there. This keeps both shapes working while
|
|
50
|
+
# the flavors convert one tranche at a time.
|
|
51
|
+
def render_component(value, context)
|
|
52
|
+
return nil if value.nil?
|
|
53
|
+
|
|
54
|
+
value.respond_to?(:render) ? value.render(context: context) : value.to_s
|
|
55
|
+
end
|
|
56
|
+
|
|
44
57
|
# Choose between "stage" and a type/supplement class for a typed stage.
|
|
45
58
|
def typed_stage_css(typed_stage)
|
|
46
59
|
code = typed_stage&.type_code.to_s
|
|
@@ -50,15 +50,15 @@ module Pubid
|
|
|
50
50
|
ann = context.annotated
|
|
51
51
|
parts = []
|
|
52
52
|
if @id.number
|
|
53
|
-
parts << annotate(@id.number
|
|
53
|
+
parts << annotate(render_component(@id.number, context), "docnumber",
|
|
54
54
|
annotated: ann)
|
|
55
55
|
end
|
|
56
56
|
if @id.part
|
|
57
|
-
parts << " #{annotate(@id.part
|
|
57
|
+
parts << " #{annotate(render_component(@id.part, context), 'part',
|
|
58
58
|
annotated: ann)}"
|
|
59
59
|
end
|
|
60
60
|
if @id.subpart
|
|
61
|
-
parts << "-#{annotate(@id.subpart
|
|
61
|
+
parts << "-#{annotate(render_component(@id.subpart, context), 'part',
|
|
62
62
|
annotated: ann)}"
|
|
63
63
|
end
|
|
64
64
|
if @id.stage_iteration
|
|
@@ -41,15 +41,15 @@ module Pubid
|
|
|
41
41
|
ann = context.annotated
|
|
42
42
|
parts = []
|
|
43
43
|
if @id.number
|
|
44
|
-
parts << annotate(@id.number
|
|
44
|
+
parts << annotate(render_component(@id.number, context), "docnumber",
|
|
45
45
|
annotated: ann)
|
|
46
46
|
end
|
|
47
47
|
if @id.part
|
|
48
|
-
parts << "-#{annotate(@id.part
|
|
48
|
+
parts << "-#{annotate(render_component(@id.part, context), 'part',
|
|
49
49
|
annotated: ann)}"
|
|
50
50
|
end
|
|
51
51
|
if @id.subpart
|
|
52
|
-
parts << "-#{annotate(@id.subpart
|
|
52
|
+
parts << "-#{annotate(render_component(@id.subpart, context), 'part',
|
|
53
53
|
annotated: ann)}"
|
|
54
54
|
end
|
|
55
55
|
if @id.stage_iteration
|