pubid 2.0.0.pre.alpha.13 → 2.0.0.pre.alpha.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/data/parg/tables/bipm_groups.yaml +14 -0
- data/data/parg/tables/bipm_type_codes.yaml +5 -0
- data/data/parg/tables/bipm_type_names_en.yaml +6 -0
- data/data/parg/tables/bipm_type_names_fr.yaml +6 -0
- data/data/parg/tables/directives_supplements_typed_stages.yaml +3 -0
- data/data/parg/tables/directives_typed_stages.yaml +5 -0
- data/data/parg/tables/idf_typed_stages.yaml +27 -0
- data/data/parg/tables/idf_typed_stages_supplements.yaml +2 -0
- data/data/parg/tables/iec_typed_stages.yaml +130 -0
- data/data/parg/tables/iso_publishers.yaml +4 -0
- data/data/parg/tables/organizations.yaml +12 -0
- data/data/parg/tables/tc_types.yaml +42 -0
- data/data/parg/tables/typed_stages.yaml +114 -0
- data/data/parg/tables/typed_stages_supplements.yaml +64 -0
- data/data/parg/tables/wg_types.yaml +21 -0
- data/lib/pubid/adobe/identifier.rb +11 -1
- data/lib/pubid/amca/identifiers/base.rb +1 -1
- data/lib/pubid/ansi/identifier.rb +1 -1
- data/lib/pubid/api/identifier.rb +1 -1
- data/lib/pubid/api/parser.rb +8 -4
- data/lib/pubid/ashrae/identifiers/base.rb +10 -1
- data/lib/pubid/ashrae/parser.rb +18 -11
- data/lib/pubid/asme/identifier.rb +1 -1
- data/lib/pubid/astm/identifier.rb +1 -1
- data/lib/pubid/bipm/identifier.rb +1 -1
- data/lib/pubid/bsi/single_identifier.rb +1 -3
- data/lib/pubid/calconnect/identifier.rb +1 -1
- data/lib/pubid/ccsds/identifier.rb +1 -1
- data/lib/pubid/cen_cenelec/identifier.rb +2 -1
- data/lib/pubid/cen_cenelec/parser.rb +9 -2
- data/lib/pubid/cie/identifier.rb +1 -1
- data/lib/pubid/cie/parser.rb +9 -2
- data/lib/pubid/conformance/checks.rb +1 -1
- data/lib/pubid/csa/identifier.rb +6 -2
- data/lib/pubid/csa/parser.rb +25 -8
- data/lib/pubid/doi/identifier.rb +1 -1
- data/lib/pubid/easc/identifier.rb +10 -1
- data/lib/pubid/ecma/identifier.rb +1 -1
- data/lib/pubid/etsi/identifiers/base.rb +1 -1
- data/lib/pubid/evs.rb +1 -1
- data/lib/pubid/gb/identifier.rb +1 -1
- data/lib/pubid/gost/identifier.rb +11 -1
- data/lib/pubid/gost/parser.rb +8 -1
- data/lib/pubid/iala/identifier.rb +10 -1
- data/lib/pubid/iana/identifier.rb +1 -1
- data/lib/pubid/idf/builder.rb +5 -0
- data/lib/pubid/iec/identifier.rb +1 -1
- data/lib/pubid/iec/parser.rb +9 -4
- data/lib/pubid/ieee/builder.rb +61 -23
- data/lib/pubid/ieee/identifiers/base.rb +1 -1
- data/lib/pubid/ieee/identifiers/joint_development.rb +55 -19
- data/lib/pubid/ieee/parser.rb +25 -11
- data/lib/pubid/ieee/renderer.rb +5 -3
- data/lib/pubid/ietf/identifiers/base.rb +1 -1
- data/lib/pubid/isbn/identifier.rb +1 -1
- data/lib/pubid/iso/identifier.rb +4 -1
- data/lib/pubid/iso/normalizer.rb +4 -1
- data/lib/pubid/itu/CLAUDE.md +46 -0
- data/lib/pubid/itu/builder.rb +24 -4
- data/lib/pubid/itu/identifiers/base.rb +11 -18
- data/lib/pubid/itu/identifiers/radio_regulations.rb +27 -0
- data/lib/pubid/itu/identifiers/special_publication.rb +48 -14
- data/lib/pubid/itu/identifiers/standard_serialization.rb +2 -0
- data/lib/pubid/itu/identifiers.rb +1 -0
- data/lib/pubid/itu/parser.rb +108 -22
- data/lib/pubid/itu/urn_generator.rb +9 -2
- data/lib/pubid/jcgm.rb +1 -1
- data/lib/pubid/jis/identifier.rb +1 -1
- data/lib/pubid/nist/builder.rb +1 -0
- data/lib/pubid/nist/identifiers/base.rb +15 -4
- data/lib/pubid/nist/parser.rb +9 -0
- data/lib/pubid/nist/urn_parser.rb +10 -1
- data/lib/pubid/oasis/identifier.rb +1 -1
- data/lib/pubid/ogc/identifier.rb +1 -1
- data/lib/pubid/oiml.rb +1 -1
- data/lib/pubid/omg/identifier.rb +1 -1
- data/lib/pubid/parg/artifact.rb +46 -0
- data/lib/pubid/parg/backend.rb +92 -0
- data/lib/pubid/parg.rb +8 -0
- data/lib/pubid/pg.rb +8 -0
- data/lib/pubid/plateau.rb +1 -2
- data/lib/pubid/sae/identifiers/base.rb +1 -1
- data/lib/pubid/tgpp/identifier.rb +1 -1
- data/lib/pubid/un/identifier.rb +1 -1
- data/lib/pubid/version.rb +1 -1
- data/lib/pubid/w3c/identifier.rb +1 -1
- data/lib/pubid/xsf/identifier.rb +1 -1
- data/lib/pubid.rb +1 -0
- metadata +36 -2
data/lib/pubid/itu/parser.rb
CHANGED
|
@@ -125,9 +125,48 @@ module Pubid
|
|
|
125
125
|
dash >> (letter.repeat(1, 3) >> dot >> digits).as(:range_end)
|
|
126
126
|
end
|
|
127
127
|
|
|
128
|
-
#
|
|
128
|
+
# A "-YYYYMM" approval date — the "200307" of "T-REC-T.4-200307-I" and
|
|
129
|
+
# "ITU-T T.4-200307". ITU's own edition suffix is short ("-5"), so six
|
|
130
|
+
# digits that read as a plausible year (19xx/20xx) and month (01-12) are
|
|
131
|
+
# the date, never a part. A six-digit run that fails either test
|
|
132
|
+
# ("-200313", "-180001") is still a part, as it was before.
|
|
133
|
+
rule(:yyyymm_year) { (str("19") | str("20")) >> digit >> digit }
|
|
134
|
+
rule(:yyyymm_month) do
|
|
135
|
+
(str("0") >> match["1-9"]) | (str("1") >> match["0-2"])
|
|
136
|
+
end
|
|
137
|
+
rule(:yyyymm_shape) { yyyymm_year >> yyyymm_month >> digit.absent? }
|
|
138
|
+
|
|
139
|
+
# The status letter that trails the date in a publication id — "I" (in
|
|
140
|
+
# force) or "S" (superseded). It names the state of the edition, not the
|
|
141
|
+
# edition, so it is parsed and dropped. "S" is also the Spanish language
|
|
142
|
+
# suffix, so it is a status ONLY in the full "T-REC-…" id, where ITU
|
|
143
|
+
# always writes one; after an "ITU-T …-YYYYMM" print form only "I" is,
|
|
144
|
+
# and "-S" stays the language ("ITU-T Z.100-199911-S").
|
|
145
|
+
rule(:id_status) do
|
|
146
|
+
dash >> match["IS"] >> match["A-Za-z0-9"].absent?
|
|
147
|
+
end
|
|
148
|
+
|
|
149
|
+
rule(:print_id_status) do
|
|
150
|
+
dash >> str("I") >> match["A-Za-z0-9"].absent?
|
|
151
|
+
end
|
|
152
|
+
|
|
153
|
+
rule(:yyyymm_date) do
|
|
154
|
+
dash >> yyyymm_year.as(:year) >> yyyymm_month.as(:month) >>
|
|
155
|
+
digit.absent?
|
|
156
|
+
end
|
|
157
|
+
|
|
158
|
+
rule(:id_date) { yyyymm_date >> print_id_status.maybe }
|
|
159
|
+
|
|
160
|
+
# Either date spelling of a Recommendation.
|
|
161
|
+
rule(:document_date) { date_part | id_date }
|
|
162
|
+
|
|
163
|
+
# "ITU-T REC T.4", "ITU-T REC-T.4" — the redundant type word of ITU's
|
|
164
|
+
# own URLs. Not captured: a Recommendation is the default type.
|
|
165
|
+
rule(:rec_word) { str("REC") >> (space | dash) }
|
|
166
|
+
|
|
167
|
+
# Parts. The yyyymm guard keeps the approval date out of the part list.
|
|
129
168
|
rule(:part) do
|
|
130
|
-
dash >> digits.as(:part)
|
|
169
|
+
dash >> yyyymm_shape.absent? >> digits.as(:part)
|
|
131
170
|
end
|
|
132
171
|
|
|
133
172
|
rule(:parts) { part.repeat(0).as(:parts) }
|
|
@@ -273,6 +312,7 @@ module Pubid
|
|
|
273
312
|
itu_prefix >>
|
|
274
313
|
sector >>
|
|
275
314
|
space >>
|
|
315
|
+
rec_word.maybe >>
|
|
276
316
|
series >> dot >>
|
|
277
317
|
code >>
|
|
278
318
|
range_end.maybe >>
|
|
@@ -281,7 +321,7 @@ module Pubid
|
|
|
281
321
|
series_word.maybe >>
|
|
282
322
|
attachment.maybe >>
|
|
283
323
|
version_part.maybe >>
|
|
284
|
-
|
|
324
|
+
document_date.maybe
|
|
285
325
|
end
|
|
286
326
|
|
|
287
327
|
rule(:base_without_series) do
|
|
@@ -292,7 +332,7 @@ module Pubid
|
|
|
292
332
|
code_suffixes >>
|
|
293
333
|
attachment.maybe >>
|
|
294
334
|
version_part.maybe >>
|
|
295
|
-
|
|
335
|
+
document_date.maybe
|
|
296
336
|
end
|
|
297
337
|
|
|
298
338
|
# A series-code document — "EMC-5", "MES-2", "QOS-2", "IMPL-8",
|
|
@@ -311,12 +351,10 @@ module Pubid
|
|
|
311
351
|
# The number stays in `code.number`, so `root.number` — the field
|
|
312
352
|
# relaton-index bsearches on — is "5" for EMC-5 and "QKD" for SEC-QKD
|
|
313
353
|
# rather than nil.
|
|
314
|
-
# The OB guard keeps "ITU-T OB-1" a clean parse failure
|
|
315
|
-
#
|
|
316
|
-
#
|
|
317
|
-
#
|
|
318
|
-
# input into a crash for callers. It guards "OB" + dash specifically, so
|
|
319
|
-
# a genuine two-letter mnemonic starting "OB" would still parse.
|
|
354
|
+
# The OB guard keeps "ITU-T OB-1" a clean parse failure rather than a
|
|
355
|
+
# Recommendation of a series "OB" — the Operational Bulletin's series
|
|
356
|
+
# name. It guards "OB" + dash specifically, so a genuine two-letter
|
|
357
|
+
# mnemonic starting "OB" would still parse.
|
|
320
358
|
rule(:series_code_body) do
|
|
321
359
|
(str("OB") >> dash).absent? >>
|
|
322
360
|
letter.repeat(2).as(:series) >> dash.as(:series_dash) >>
|
|
@@ -365,10 +403,8 @@ module Pubid
|
|
|
365
403
|
# Builder#build's `combined` branch, which builds a CombinedIdentifier and
|
|
366
404
|
# would silently drop the marker — a clean parse failure is better than a
|
|
367
405
|
# Report that comes back as a Recommendation. The OB guard mirrors
|
|
368
|
-
# series_code_body's:
|
|
369
|
-
#
|
|
370
|
-
# dropped) or hit validate_ob_no_sector!, whose ArgumentError escapes
|
|
371
|
-
# Identifier.parse's Parslet::ParseFailed rescue.
|
|
406
|
+
# series_code_body's: "Report ITU-T OB.1" would otherwise route to
|
|
407
|
+
# SpecialPublication with the marker dropped.
|
|
372
408
|
rule(:report_body) do
|
|
373
409
|
(str("OB") >> dot).absent? >>
|
|
374
410
|
(series >> dot).maybe >>
|
|
@@ -535,6 +571,7 @@ module Pubid
|
|
|
535
571
|
itu_prefix >>
|
|
536
572
|
sector >>
|
|
537
573
|
space >>
|
|
574
|
+
rec_word.maybe >>
|
|
538
575
|
series >> dot >>
|
|
539
576
|
code >>
|
|
540
577
|
range_end.maybe >>
|
|
@@ -543,7 +580,7 @@ module Pubid
|
|
|
543
580
|
series_word.maybe >>
|
|
544
581
|
attachment.maybe >>
|
|
545
582
|
version_part.maybe >>
|
|
546
|
-
|
|
583
|
+
document_date.maybe >>
|
|
547
584
|
language.maybe
|
|
548
585
|
end
|
|
549
586
|
|
|
@@ -556,24 +593,71 @@ module Pubid
|
|
|
556
593
|
code_suffixes >>
|
|
557
594
|
attachment.maybe >>
|
|
558
595
|
version_part.maybe >>
|
|
559
|
-
|
|
596
|
+
document_date.maybe >>
|
|
597
|
+
language.maybe
|
|
598
|
+
end
|
|
599
|
+
|
|
600
|
+
# ITU's publication id — "T-REC-T.4-200307-I",
|
|
601
|
+
# "R-REC-BO.1130-5-202602-I": <sector>-REC-<number>[-<edition>]-<YYYYMM>
|
|
602
|
+
# [-<status>], the name ITU gives each edition in its URLs and PDF
|
|
603
|
+
# files. It builds the plain Recommendation it names and renders in the
|
|
604
|
+
# print form ("ITU-T T.4 (07/2003)"). The date is required: without it
|
|
605
|
+
# the string names no edition. No other rule starts with a bare sector
|
|
606
|
+
# letter, so the slot is free.
|
|
607
|
+
rule(:publication_id) do
|
|
608
|
+
sector >> dash >> str("REC") >> dash >>
|
|
609
|
+
series >> dot >> code >> yyyymm_date >> id_status.maybe >>
|
|
560
610
|
language.maybe
|
|
561
611
|
end
|
|
562
612
|
|
|
613
|
+
# The Radio Regulations — "ITU-R RR", "ITU-R RR (2020)", and the URL
|
|
614
|
+
# spelling "ITU-R RR-2020". Always ITU-R. The trailing any.absent? is
|
|
615
|
+
# load-bearing: PEG ordered choice never re-enters the alternation once
|
|
616
|
+
# an alternative succeeds, so a partial match on "ITU-R RR.1" must fail
|
|
617
|
+
# here and fall through to with_series.
|
|
618
|
+
rule(:radio_regulations) do
|
|
619
|
+
itu_prefix >> str("R").as(:sector) >> space >>
|
|
620
|
+
str("RR").as(:radio_regulations) >>
|
|
621
|
+
(date_part | (dash >> digit.repeat(4, 4).as(:year))).maybe >>
|
|
622
|
+
language.maybe >> any.absent?
|
|
623
|
+
end
|
|
624
|
+
|
|
563
625
|
# OB (Operational Bulletin) — Special Publication.
|
|
564
|
-
# OB is a cross-bureau ITU publication
|
|
565
|
-
#
|
|
626
|
+
# OB is a cross-bureau ITU publication. The TSB spelling carries a
|
|
627
|
+
# sector ("ITU-T OB.1096 (2016)"); the builder keeps it, and it renders
|
|
628
|
+
# back, but it is not part of the bulletin's identity.
|
|
566
629
|
rule(:ob_series) { str("OB").as(:series) }
|
|
567
630
|
|
|
568
631
|
rule(:ob_dot_body) { dot >> number }
|
|
569
632
|
rule(:ob_no_body) { space >> str("No.") >> space >> number }
|
|
633
|
+
# "ITU OB 1000" — metanorma-itu's docidentifier ("Annex to ITU OB %").
|
|
634
|
+
# Accepted as an input spelling only; it renders "ITU OB No. 1000", the
|
|
635
|
+
# form ITU's own bulletin site uses.
|
|
636
|
+
rule(:ob_bare_body) { space >> number }
|
|
637
|
+
|
|
638
|
+
# The date as a bulletin prints it — "ITU-T OB.1096 - 15.III.2016": day,
|
|
639
|
+
# Roman month, year. The months are tried longest first, because PEG
|
|
640
|
+
# takes the first alternative that matches and "I" would otherwise win
|
|
641
|
+
# on "III"; "XIII" matches "XII", then fails on the required dot.
|
|
642
|
+
rule(:roman_month) do
|
|
643
|
+
%w[XII XI X IX VIII VII VI V IV III II I]
|
|
644
|
+
.map { |m| str(m) }.reduce(:|)
|
|
645
|
+
end
|
|
646
|
+
|
|
647
|
+
rule(:ob_roman_date) do
|
|
648
|
+
str(" - ") >> digit.repeat(2, 2).as(:day) >> dot >>
|
|
649
|
+
roman_month.as(:roman_month) >> dot >>
|
|
650
|
+
digit.repeat(4, 4).as(:year)
|
|
651
|
+
end
|
|
652
|
+
|
|
653
|
+
rule(:ob_date) { date_part | ob_roman_date }
|
|
570
654
|
|
|
571
655
|
rule(:ob_with_sector) do
|
|
572
656
|
itu_prefix >>
|
|
573
657
|
(sector >> space).maybe >>
|
|
574
658
|
ob_series >>
|
|
575
|
-
(ob_dot_body | ob_no_body) >>
|
|
576
|
-
|
|
659
|
+
(ob_dot_body | ob_no_body | ob_bare_body) >>
|
|
660
|
+
ob_date.maybe >>
|
|
577
661
|
language.maybe
|
|
578
662
|
end
|
|
579
663
|
|
|
@@ -585,7 +669,7 @@ module Pubid
|
|
|
585
669
|
str("Operational Bulletin").as(:_op_bull) >>
|
|
586
670
|
space >> str("No.") >> space >>
|
|
587
671
|
number >>
|
|
588
|
-
|
|
672
|
+
ob_date.maybe >>
|
|
589
673
|
language.maybe
|
|
590
674
|
end
|
|
591
675
|
|
|
@@ -671,13 +755,15 @@ module Pubid
|
|
|
671
755
|
handbook |
|
|
672
756
|
numeric_question |
|
|
673
757
|
letter_question |
|
|
758
|
+
radio_regulations |
|
|
674
759
|
with_series |
|
|
675
760
|
contribution |
|
|
676
761
|
# Unreachable earlier: special_publication needs the literal "OB",
|
|
677
762
|
# handbook/numeric_question need leading digits, letter_question
|
|
678
763
|
# needs series >> dot, and contribution needs "-C" after the series.
|
|
679
764
|
series_code_identifier |
|
|
680
|
-
without_series
|
|
765
|
+
without_series |
|
|
766
|
+
publication_id
|
|
681
767
|
end
|
|
682
768
|
|
|
683
769
|
# Common-text form: an ITU identifier followed by "| ISO/IEC ...".
|
|
@@ -16,7 +16,10 @@ module Pubid
|
|
|
16
16
|
def generate_base_urn
|
|
17
17
|
parts = ["urn", "itu"]
|
|
18
18
|
|
|
19
|
-
|
|
19
|
+
# An Operational Bulletin is cross-bureau: its sector is a spelling,
|
|
20
|
+
# not identity (see SpecialPublication#==), so it stays out of the URN.
|
|
21
|
+
if identifier.sector &&
|
|
22
|
+
!identifier.is_a?(Identifiers::SpecialPublication)
|
|
20
23
|
sector = identifier.sector.to_s
|
|
21
24
|
parts << sector.to_s.downcase
|
|
22
25
|
else
|
|
@@ -56,7 +59,11 @@ module Pubid
|
|
|
56
59
|
|
|
57
60
|
if identifier.date
|
|
58
61
|
date = identifier.date
|
|
59
|
-
|
|
62
|
+
# Only an Operational Bulletin's printed date carries a day
|
|
63
|
+
# ("15.III.2016"); it is in `==`, so it must reach the URN too.
|
|
64
|
+
if date&.year && date.month && date.day
|
|
65
|
+
parts << "#{date.day}/#{date.month}/#{date.year}"
|
|
66
|
+
elsif date&.year && date.month
|
|
60
67
|
parts << "#{date.month}/#{date.year}"
|
|
61
68
|
elsif date&.year
|
|
62
69
|
parts << date.year.to_s
|
data/lib/pubid/jcgm.rb
CHANGED
data/lib/pubid/jis/identifier.rb
CHANGED
data/lib/pubid/nist/builder.rb
CHANGED
|
@@ -87,6 +87,7 @@ module Pubid
|
|
|
87
87
|
# Note: :base_portion is lost during parser merge, so check for supplement indicators
|
|
88
88
|
if parsed_hash[:supplement_date_range] || parsed_hash[:supplement_slash_year] ||
|
|
89
89
|
parsed_hash[:supplement_month_year] || parsed_hash[:supplement_year] ||
|
|
90
|
+
parsed_hash[:supplement_empty] ||
|
|
90
91
|
parsed_hash[:supplement] || parsed_hash[:base_portion]
|
|
91
92
|
return build_circular_supplement(parsed_hash)
|
|
92
93
|
end
|
|
@@ -696,14 +696,25 @@ module Pubid
|
|
|
696
696
|
result += "#{vol_str}n#{issue_number.number}"
|
|
697
697
|
end
|
|
698
698
|
|
|
699
|
-
#
|
|
700
|
-
|
|
699
|
+
# With a number, the edition glues to it per the NIST spec
|
|
700
|
+
# ("800-53r5"); series-only editions take a dot separator
|
|
701
|
+
# ("NBS.CIRC.e2" — the attested raw spelling; testsuite#5 C4/C5).
|
|
702
|
+
result += if edition
|
|
703
|
+
number ? edition.to_s : ".#{edition}"
|
|
704
|
+
else
|
|
705
|
+
""
|
|
706
|
+
end
|
|
701
707
|
|
|
702
708
|
# Use version_component
|
|
703
709
|
result += version_component.to_s(:mr) if version_component
|
|
704
710
|
|
|
705
|
-
# Supplement (e.g. ".9981sup7") - keep distinct documents distinct
|
|
706
|
-
|
|
711
|
+
# Supplement (e.g. ".9981sup7") - keep distinct documents distinct;
|
|
712
|
+
# a series-only supplement ("NBS.CIRC.sup") takes the dot itself.
|
|
713
|
+
result += if supplement
|
|
714
|
+
number ? supplement_short : ".#{supplement_short}"
|
|
715
|
+
else
|
|
716
|
+
""
|
|
717
|
+
end
|
|
707
718
|
|
|
708
719
|
# Use update_component
|
|
709
720
|
result += update_component.to_s(:mr) if update_component
|
data/lib/pubid/nist/parser.rb
CHANGED
|
@@ -726,6 +726,15 @@ module Pubid
|
|
|
726
726
|
# 4-digit years so it can't swallow "sup3/1926" or a base number.
|
|
727
727
|
((str("supp") | str("sup")) >> match("[0-9]").repeat(4, 4).as(:supp_year_start) >>
|
|
728
728
|
dash >> match("[0-9]").repeat(4, 4).as(:supp_year_end)).as(:supplement_date_range) |
|
|
729
|
+
# Bare supplement marker to the whole series, no base number
|
|
730
|
+
# ("NBS.CIRC.sup" — testsuite#5 C4): the marker alone is the
|
|
731
|
+
# supplement.
|
|
732
|
+
((str("supp") | str("sup")) >>
|
|
733
|
+
(
|
|
734
|
+
(month_abbrev >> digits).as(:supplement_month_year) |
|
|
735
|
+
(digits.as(:supp_number) >> slash >> digits.as(:supp_year)).as(:supplement_slash_year) |
|
|
736
|
+
str("").as(:supplement_empty)
|
|
737
|
+
).maybe) |
|
|
729
738
|
# With base identifier + supplement
|
|
730
739
|
(
|
|
731
740
|
# Capture base portion (everything before "supp" or "sup" or slash+year)
|
|
@@ -42,13 +42,22 @@ module Pubid
|
|
|
42
42
|
type_token, payload = parts
|
|
43
43
|
code, revision = parse_payload(payload)
|
|
44
44
|
|
|
45
|
-
text = "
|
|
45
|
+
text = "#{publisher_for(type_token)} #{type_label(type_token)} #{code}"
|
|
46
46
|
text += revision if revision
|
|
47
47
|
flavor_parse(text)
|
|
48
48
|
end
|
|
49
49
|
|
|
50
50
|
private
|
|
51
51
|
|
|
52
|
+
# The NBS-era series carry the NBS imprint in their canonical form
|
|
53
|
+
# ("NBS CSM 1") — the rebuild must name the publisher the document
|
|
54
|
+
# carries, or the flavor parse rejects its own URN's rebuild.
|
|
55
|
+
NBS_SERIES = ["csm"].freeze
|
|
56
|
+
|
|
57
|
+
def publisher_for(type_token)
|
|
58
|
+
NBS_SERIES.include?(type_token.downcase) ? "NBS" : "NIST"
|
|
59
|
+
end
|
|
60
|
+
|
|
52
61
|
def parse_payload(payload)
|
|
53
62
|
# Strip the trailing ".supp" supplement marker.
|
|
54
63
|
stripped = payload.sub(/\.supp\z/, "")
|
data/lib/pubid/ogc/identifier.rb
CHANGED
data/lib/pubid/oiml.rb
CHANGED
data/lib/pubid/omg/identifier.rb
CHANGED
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "parsanol"
|
|
4
|
+
|
|
5
|
+
module Pubid
|
|
6
|
+
module Parg
|
|
7
|
+
# A baked, checksum-verified PG artifact for one flavor. The artifact
|
|
8
|
+
# is the parser of record for the flavor: its grammar, tests, entries,
|
|
9
|
+
# and embedded tables travel together and are verified on load.
|
|
10
|
+
class Artifact
|
|
11
|
+
DATA_DIR = File.expand_path("../../../data/parg", __dir__)
|
|
12
|
+
# The deferred entity atoms resolve from_table references at parse
|
|
13
|
+
# time from this dir, so the table YAMLs vendor alongside the
|
|
14
|
+
# baked artifacts.
|
|
15
|
+
TABLES_DIR = File.join(DATA_DIR, "tables")
|
|
16
|
+
|
|
17
|
+
@artifacts = {}
|
|
18
|
+
|
|
19
|
+
class << self
|
|
20
|
+
# Load and memoize the artifact for a flavor (e.g. :iso).
|
|
21
|
+
def for(flavor)
|
|
22
|
+
@artifacts[flavor] ||=
|
|
23
|
+
begin
|
|
24
|
+
path = File.join(DATA_DIR, "#{flavor}.json")
|
|
25
|
+
new(Parsanol::PARG::Artifact.load(path, tables_dir: TABLES_DIR))
|
|
26
|
+
rescue Parsanol::PARG::Error => e
|
|
27
|
+
raise Pubid::Errors::ParseError,
|
|
28
|
+
"PG artifact #{path} rejected: #{e.message}"
|
|
29
|
+
end
|
|
30
|
+
end
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
def initialize(artifact)
|
|
34
|
+
@artifact = artifact
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
def parse(entry, input)
|
|
38
|
+
@artifact.parse(entry, input)
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
def checksum
|
|
42
|
+
@artifact.envelope["checksum"]
|
|
43
|
+
end
|
|
44
|
+
end
|
|
45
|
+
end
|
|
46
|
+
end
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Pubid
|
|
4
|
+
module Parg
|
|
5
|
+
# The PG-artifact parse backend: runs the flavor's baked artifact on
|
|
6
|
+
# an input and returns the builder-ready attribute hash the flavor's
|
|
7
|
+
# Builder expects — the same hash its parslet parser produced.
|
|
8
|
+
module Backend
|
|
9
|
+
LEAF_KEYS = %i[value line column offset length].freeze
|
|
10
|
+
|
|
11
|
+
module_function
|
|
12
|
+
|
|
13
|
+
ALL_PARTS_SUFFIX = "(all parts)"
|
|
14
|
+
|
|
15
|
+
def parse(flavor, input, entry: "identifier", merge_top_sequence: true)
|
|
16
|
+
# The parslet grammar base strips the "(all parts)" suffix before
|
|
17
|
+
# parsing and marks the tree (Grammar#parse / #mark_all_parts);
|
|
18
|
+
# the artifact backend carries the same contract so every flavor
|
|
19
|
+
# whose grammar does not itself consume the suffix keeps working.
|
|
20
|
+
if input.end_with?(ALL_PARTS_SUFFIX)
|
|
21
|
+
base = input.sub(/\s*\(all parts\)\s*\z/, "")
|
|
22
|
+
tree = to_builder_hash(Artifact.for(flavor).parse(entry, base), merge: merge_top_sequence)
|
|
23
|
+
return mark_all_parts(tree)
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
to_builder_hash(Artifact.for(flavor).parse(entry, input), merge: merge_top_sequence)
|
|
27
|
+
rescue Parsanol::ParseFailed => e
|
|
28
|
+
raise Pubid::Errors::ParseError.new(e.message, nil,
|
|
29
|
+
input: input,
|
|
30
|
+
flavor: registered_name(flavor))
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
# The registered flavor name (Registry-resolvable; "3gpp" for
|
|
34
|
+
# Pubid::Tgpp), falling back to the internal symbol when the
|
|
35
|
+
# flavor is not registered.
|
|
36
|
+
def registered_name(flavor)
|
|
37
|
+
mod = Pubid.const_get(flavor.to_s.split("_").map(&:capitalize).join)
|
|
38
|
+
names = Pubid::Registry.flavor_names.select { |name| Pubid::Registry.get(name) == mod }
|
|
39
|
+
# A module may register under several names; the longest is the
|
|
40
|
+
# canonical one ("cen_cenelec", not the "cen" alias).
|
|
41
|
+
names.max_by(&:length)
|
|
42
|
+
rescue NameError
|
|
43
|
+
nil
|
|
44
|
+
end || flavor.to_s
|
|
45
|
+
|
|
46
|
+
def mark_all_parts(tree)
|
|
47
|
+
case tree
|
|
48
|
+
when Hash then tree.merge(all_parts: true)
|
|
49
|
+
when Array then tree.map { |t| t.merge(all_parts: true) }
|
|
50
|
+
else tree
|
|
51
|
+
end
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
# The artifact emits the parsanol-tree wire shape: capture leaves
|
|
55
|
+
# ({value, line, column, offset, length}) carry their text, and the
|
|
56
|
+
# top-level sequence is a list of capture hashes. The builder-ready
|
|
57
|
+
# form scalarizes leaves and folds that top-level list into one
|
|
58
|
+
# attribute hash (last key wins) — parslet's sequence fold.
|
|
59
|
+
def to_builder_hash(shape, merge: true)
|
|
60
|
+
normalized = normalize(shape)
|
|
61
|
+
return normalized.reduce(:merge) if merge && mergeable_sequence?(normalized)
|
|
62
|
+
|
|
63
|
+
normalized
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
def normalize(node)
|
|
67
|
+
case node
|
|
68
|
+
when Parsanol::Slice then node.content
|
|
69
|
+
when Hash then normalize_hash(node)
|
|
70
|
+
when Array then node.map { |item| normalize(item) }
|
|
71
|
+
else node
|
|
72
|
+
end
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
def normalize_hash(node)
|
|
76
|
+
return node[:value] if wire_leaf?(node)
|
|
77
|
+
|
|
78
|
+
node.transform_values { |value| normalize(value) }
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
def mergeable_sequence?(value)
|
|
82
|
+
value.is_a?(Array) && value.all?(Hash)
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
def wire_leaf?(node)
|
|
86
|
+
return false unless node.size == LEAF_KEYS.size
|
|
87
|
+
|
|
88
|
+
LEAF_KEYS.all? { |key| node.key?(key) }
|
|
89
|
+
end
|
|
90
|
+
end
|
|
91
|
+
end
|
|
92
|
+
end
|
data/lib/pubid/parg.rb
ADDED
data/lib/pubid/pg.rb
ADDED
data/lib/pubid/plateau.rb
CHANGED
|
@@ -31,8 +31,7 @@ module Pubid
|
|
|
31
31
|
|
|
32
32
|
# Apply legacy update_codes normalization first
|
|
33
33
|
normalized = Core::UpdateCodes.apply(input, :plateau)
|
|
34
|
-
|
|
35
|
-
parsed = parser.parse(normalized)
|
|
34
|
+
parsed = Pubid::Parg::Backend.parse(:plateau, normalized)
|
|
36
35
|
Builder.build(parsed)
|
|
37
36
|
end
|
|
38
37
|
|
data/lib/pubid/un/identifier.rb
CHANGED
data/lib/pubid/version.rb
CHANGED
data/lib/pubid/w3c/identifier.rb
CHANGED
data/lib/pubid/xsf/identifier.rb
CHANGED
data/lib/pubid.rb
CHANGED
|
@@ -17,6 +17,7 @@ require "pubid/prefixes_support"
|
|
|
17
17
|
module Pubid
|
|
18
18
|
autoload :Schema, "pubid/schema"
|
|
19
19
|
autoload :Conformance, "pubid/conformance"
|
|
20
|
+
autoload :Parg, "pubid/parg"
|
|
20
21
|
# Upper bound on the length of an identifier string accepted by any +parse+
|
|
21
22
|
# entry point. Real-world standards identifiers are well under 200 characters;
|
|
22
23
|
# this limit exists purely to keep pathological, attacker-controlled inputs
|