pubid 2.0.0.pre.alpha.13 → 2.0.0.pre.alpha.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. checksums.yaml +4 -4
  2. data/data/parg/tables/bipm_groups.yaml +14 -0
  3. data/data/parg/tables/bipm_type_codes.yaml +5 -0
  4. data/data/parg/tables/bipm_type_names_en.yaml +6 -0
  5. data/data/parg/tables/bipm_type_names_fr.yaml +6 -0
  6. data/data/parg/tables/directives_supplements_typed_stages.yaml +3 -0
  7. data/data/parg/tables/directives_typed_stages.yaml +5 -0
  8. data/data/parg/tables/idf_typed_stages.yaml +27 -0
  9. data/data/parg/tables/idf_typed_stages_supplements.yaml +2 -0
  10. data/data/parg/tables/iec_typed_stages.yaml +130 -0
  11. data/data/parg/tables/iso_publishers.yaml +4 -0
  12. data/data/parg/tables/organizations.yaml +12 -0
  13. data/data/parg/tables/tc_types.yaml +42 -0
  14. data/data/parg/tables/typed_stages.yaml +114 -0
  15. data/data/parg/tables/typed_stages_supplements.yaml +64 -0
  16. data/data/parg/tables/wg_types.yaml +21 -0
  17. data/lib/pubid/adobe/identifier.rb +11 -1
  18. data/lib/pubid/amca/identifiers/base.rb +1 -1
  19. data/lib/pubid/ansi/identifier.rb +1 -1
  20. data/lib/pubid/api/identifier.rb +1 -1
  21. data/lib/pubid/api/parser.rb +8 -4
  22. data/lib/pubid/ashrae/identifiers/base.rb +10 -1
  23. data/lib/pubid/ashrae/parser.rb +18 -11
  24. data/lib/pubid/asme/identifier.rb +1 -1
  25. data/lib/pubid/astm/identifier.rb +1 -1
  26. data/lib/pubid/bipm/identifier.rb +1 -1
  27. data/lib/pubid/bsi/single_identifier.rb +1 -3
  28. data/lib/pubid/calconnect/identifier.rb +1 -1
  29. data/lib/pubid/ccsds/identifier.rb +1 -1
  30. data/lib/pubid/cen_cenelec/identifier.rb +2 -1
  31. data/lib/pubid/cen_cenelec/parser.rb +9 -2
  32. data/lib/pubid/cie/identifier.rb +1 -1
  33. data/lib/pubid/cie/parser.rb +9 -2
  34. data/lib/pubid/conformance/checks.rb +1 -1
  35. data/lib/pubid/csa/identifier.rb +6 -2
  36. data/lib/pubid/csa/parser.rb +25 -8
  37. data/lib/pubid/doi/identifier.rb +1 -1
  38. data/lib/pubid/easc/identifier.rb +10 -1
  39. data/lib/pubid/ecma/identifier.rb +1 -1
  40. data/lib/pubid/etsi/identifiers/base.rb +1 -1
  41. data/lib/pubid/evs.rb +1 -1
  42. data/lib/pubid/gb/identifier.rb +1 -1
  43. data/lib/pubid/gost/identifier.rb +11 -1
  44. data/lib/pubid/gost/parser.rb +8 -1
  45. data/lib/pubid/iala/identifier.rb +10 -1
  46. data/lib/pubid/iana/identifier.rb +1 -1
  47. data/lib/pubid/idf/builder.rb +5 -0
  48. data/lib/pubid/iec/identifier.rb +1 -1
  49. data/lib/pubid/iec/parser.rb +9 -4
  50. data/lib/pubid/ieee/builder.rb +61 -23
  51. data/lib/pubid/ieee/identifiers/base.rb +1 -1
  52. data/lib/pubid/ieee/identifiers/joint_development.rb +55 -19
  53. data/lib/pubid/ieee/parser.rb +25 -11
  54. data/lib/pubid/ieee/renderer.rb +5 -3
  55. data/lib/pubid/ietf/identifiers/base.rb +1 -1
  56. data/lib/pubid/isbn/identifier.rb +1 -1
  57. data/lib/pubid/iso/identifier.rb +4 -1
  58. data/lib/pubid/iso/normalizer.rb +4 -1
  59. data/lib/pubid/itu/CLAUDE.md +46 -0
  60. data/lib/pubid/itu/builder.rb +24 -4
  61. data/lib/pubid/itu/identifiers/base.rb +11 -18
  62. data/lib/pubid/itu/identifiers/radio_regulations.rb +27 -0
  63. data/lib/pubid/itu/identifiers/special_publication.rb +48 -14
  64. data/lib/pubid/itu/identifiers/standard_serialization.rb +2 -0
  65. data/lib/pubid/itu/identifiers.rb +1 -0
  66. data/lib/pubid/itu/parser.rb +108 -22
  67. data/lib/pubid/itu/urn_generator.rb +9 -2
  68. data/lib/pubid/jcgm.rb +1 -1
  69. data/lib/pubid/jis/identifier.rb +1 -1
  70. data/lib/pubid/nist/builder.rb +1 -0
  71. data/lib/pubid/nist/identifiers/base.rb +15 -4
  72. data/lib/pubid/nist/parser.rb +9 -0
  73. data/lib/pubid/nist/urn_parser.rb +10 -1
  74. data/lib/pubid/oasis/identifier.rb +1 -1
  75. data/lib/pubid/ogc/identifier.rb +1 -1
  76. data/lib/pubid/oiml.rb +1 -1
  77. data/lib/pubid/omg/identifier.rb +1 -1
  78. data/lib/pubid/parg/artifact.rb +46 -0
  79. data/lib/pubid/parg/backend.rb +92 -0
  80. data/lib/pubid/parg.rb +8 -0
  81. data/lib/pubid/pg.rb +8 -0
  82. data/lib/pubid/plateau.rb +1 -2
  83. data/lib/pubid/sae/identifiers/base.rb +1 -1
  84. data/lib/pubid/tgpp/identifier.rb +1 -1
  85. data/lib/pubid/un/identifier.rb +1 -1
  86. data/lib/pubid/version.rb +1 -1
  87. data/lib/pubid/w3c/identifier.rb +1 -1
  88. data/lib/pubid/xsf/identifier.rb +1 -1
  89. data/lib/pubid.rb +1 -0
  90. metadata +36 -2
@@ -125,9 +125,48 @@ module Pubid
125
125
  dash >> (letter.repeat(1, 3) >> dot >> digits).as(:range_end)
126
126
  end
127
127
 
128
- # Parts
128
+ # A "-YYYYMM" approval date — the "200307" of "T-REC-T.4-200307-I" and
129
+ # "ITU-T T.4-200307". ITU's own edition suffix is short ("-5"), so six
130
+ # digits that read as a plausible year (19xx/20xx) and month (01-12) are
131
+ # the date, never a part. A six-digit run that fails either test
132
+ # ("-200313", "-180001") is still a part, as it was before.
133
+ rule(:yyyymm_year) { (str("19") | str("20")) >> digit >> digit }
134
+ rule(:yyyymm_month) do
135
+ (str("0") >> match["1-9"]) | (str("1") >> match["0-2"])
136
+ end
137
+ rule(:yyyymm_shape) { yyyymm_year >> yyyymm_month >> digit.absent? }
138
+
139
+ # The status letter that trails the date in a publication id — "I" (in
140
+ # force) or "S" (superseded). It names the state of the edition, not the
141
+ # edition, so it is parsed and dropped. "S" is also the Spanish language
142
+ # suffix, so it is a status ONLY in the full "T-REC-…" id, where ITU
143
+ # always writes one; after an "ITU-T …-YYYYMM" print form only "I" is,
144
+ # and "-S" stays the language ("ITU-T Z.100-199911-S").
145
+ rule(:id_status) do
146
+ dash >> match["IS"] >> match["A-Za-z0-9"].absent?
147
+ end
148
+
149
+ rule(:print_id_status) do
150
+ dash >> str("I") >> match["A-Za-z0-9"].absent?
151
+ end
152
+
153
+ rule(:yyyymm_date) do
154
+ dash >> yyyymm_year.as(:year) >> yyyymm_month.as(:month) >>
155
+ digit.absent?
156
+ end
157
+
158
+ rule(:id_date) { yyyymm_date >> print_id_status.maybe }
159
+
160
+ # Either date spelling of a Recommendation.
161
+ rule(:document_date) { date_part | id_date }
162
+
163
+ # "ITU-T REC T.4", "ITU-T REC-T.4" — the redundant type word of ITU's
164
+ # own URLs. Not captured: a Recommendation is the default type.
165
+ rule(:rec_word) { str("REC") >> (space | dash) }
166
+
167
+ # Parts. The yyyymm guard keeps the approval date out of the part list.
129
168
  rule(:part) do
130
- dash >> digits.as(:part)
169
+ dash >> yyyymm_shape.absent? >> digits.as(:part)
131
170
  end
132
171
 
133
172
  rule(:parts) { part.repeat(0).as(:parts) }
@@ -273,6 +312,7 @@ module Pubid
273
312
  itu_prefix >>
274
313
  sector >>
275
314
  space >>
315
+ rec_word.maybe >>
276
316
  series >> dot >>
277
317
  code >>
278
318
  range_end.maybe >>
@@ -281,7 +321,7 @@ module Pubid
281
321
  series_word.maybe >>
282
322
  attachment.maybe >>
283
323
  version_part.maybe >>
284
- date_part.maybe
324
+ document_date.maybe
285
325
  end
286
326
 
287
327
  rule(:base_without_series) do
@@ -292,7 +332,7 @@ module Pubid
292
332
  code_suffixes >>
293
333
  attachment.maybe >>
294
334
  version_part.maybe >>
295
- date_part.maybe
335
+ document_date.maybe
296
336
  end
297
337
 
298
338
  # A series-code document — "EMC-5", "MES-2", "QOS-2", "IMPL-8",
@@ -311,12 +351,10 @@ module Pubid
311
351
  # The number stays in `code.number`, so `root.number` — the field
312
352
  # relaton-index bsearches on — is "5" for EMC-5 and "QKD" for SEC-QKD
313
353
  # rather than nil.
314
- # The OB guard keeps "ITU-T OB-1" a clean parse failure. Without it the
315
- # string reaches Builder#build's Recommendation fallback, whose
316
- # validate_ob_no_sector! raises an ArgumentError that escapes
317
- # Identifier.parse's Parslet::ParseFailed rescue — turning a rejected
318
- # input into a crash for callers. It guards "OB" + dash specifically, so
319
- # a genuine two-letter mnemonic starting "OB" would still parse.
354
+ # The OB guard keeps "ITU-T OB-1" a clean parse failure rather than a
355
+ # Recommendation of a series "OB" — the Operational Bulletin's series
356
+ # name. It guards "OB" + dash specifically, so a genuine two-letter
357
+ # mnemonic starting "OB" would still parse.
320
358
  rule(:series_code_body) do
321
359
  (str("OB") >> dash).absent? >>
322
360
  letter.repeat(2).as(:series) >> dash.as(:series_dash) >>
@@ -365,10 +403,8 @@ module Pubid
365
403
  # Builder#build's `combined` branch, which builds a CombinedIdentifier and
366
404
  # would silently drop the marker — a clean parse failure is better than a
367
405
  # Report that comes back as a Recommendation. The OB guard mirrors
368
- # series_code_body's: an Operational Bulletin is cross-bureau, and
369
- # "Report ITU-T OB.1" would otherwise route to SpecialPublication (marker
370
- # dropped) or hit validate_ob_no_sector!, whose ArgumentError escapes
371
- # Identifier.parse's Parslet::ParseFailed rescue.
406
+ # series_code_body's: "Report ITU-T OB.1" would otherwise route to
407
+ # SpecialPublication with the marker dropped.
372
408
  rule(:report_body) do
373
409
  (str("OB") >> dot).absent? >>
374
410
  (series >> dot).maybe >>
@@ -535,6 +571,7 @@ module Pubid
535
571
  itu_prefix >>
536
572
  sector >>
537
573
  space >>
574
+ rec_word.maybe >>
538
575
  series >> dot >>
539
576
  code >>
540
577
  range_end.maybe >>
@@ -543,7 +580,7 @@ module Pubid
543
580
  series_word.maybe >>
544
581
  attachment.maybe >>
545
582
  version_part.maybe >>
546
- date_part.maybe >>
583
+ document_date.maybe >>
547
584
  language.maybe
548
585
  end
549
586
 
@@ -556,24 +593,71 @@ module Pubid
556
593
  code_suffixes >>
557
594
  attachment.maybe >>
558
595
  version_part.maybe >>
559
- date_part.maybe >>
596
+ document_date.maybe >>
597
+ language.maybe
598
+ end
599
+
600
+ # ITU's publication id — "T-REC-T.4-200307-I",
601
+ # "R-REC-BO.1130-5-202602-I": <sector>-REC-<number>[-<edition>]-<YYYYMM>
602
+ # [-<status>], the name ITU gives each edition in its URLs and PDF
603
+ # files. It builds the plain Recommendation it names and renders in the
604
+ # print form ("ITU-T T.4 (07/2003)"). The date is required: without it
605
+ # the string names no edition. No other rule starts with a bare sector
606
+ # letter, so the slot is free.
607
+ rule(:publication_id) do
608
+ sector >> dash >> str("REC") >> dash >>
609
+ series >> dot >> code >> yyyymm_date >> id_status.maybe >>
560
610
  language.maybe
561
611
  end
562
612
 
613
+ # The Radio Regulations — "ITU-R RR", "ITU-R RR (2020)", and the URL
614
+ # spelling "ITU-R RR-2020". Always ITU-R. The trailing any.absent? is
615
+ # load-bearing: PEG ordered choice never re-enters the alternation once
616
+ # an alternative succeeds, so a partial match on "ITU-R RR.1" must fail
617
+ # here and fall through to with_series.
618
+ rule(:radio_regulations) do
619
+ itu_prefix >> str("R").as(:sector) >> space >>
620
+ str("RR").as(:radio_regulations) >>
621
+ (date_part | (dash >> digit.repeat(4, 4).as(:year))).maybe >>
622
+ language.maybe >> any.absent?
623
+ end
624
+
563
625
  # OB (Operational Bulletin) — Special Publication.
564
- # OB is a cross-bureau ITU publication; sector, when present in legacy
565
- # strings like "ITU-T OB.1096", is silently dropped by the builder.
626
+ # OB is a cross-bureau ITU publication. The TSB spelling carries a
627
+ # sector ("ITU-T OB.1096 (2016)"); the builder keeps it, and it renders
628
+ # back, but it is not part of the bulletin's identity.
566
629
  rule(:ob_series) { str("OB").as(:series) }
567
630
 
568
631
  rule(:ob_dot_body) { dot >> number }
569
632
  rule(:ob_no_body) { space >> str("No.") >> space >> number }
633
+ # "ITU OB 1000" — metanorma-itu's docidentifier ("Annex to ITU OB %").
634
+ # Accepted as an input spelling only; it renders "ITU OB No. 1000", the
635
+ # form ITU's own bulletin site uses.
636
+ rule(:ob_bare_body) { space >> number }
637
+
638
+ # The date as a bulletin prints it — "ITU-T OB.1096 - 15.III.2016": day,
639
+ # Roman month, year. The months are tried longest first, because PEG
640
+ # takes the first alternative that matches and "I" would otherwise win
641
+ # on "III"; "XIII" matches "XII", then fails on the required dot.
642
+ rule(:roman_month) do
643
+ %w[XII XI X IX VIII VII VI V IV III II I]
644
+ .map { |m| str(m) }.reduce(:|)
645
+ end
646
+
647
+ rule(:ob_roman_date) do
648
+ str(" - ") >> digit.repeat(2, 2).as(:day) >> dot >>
649
+ roman_month.as(:roman_month) >> dot >>
650
+ digit.repeat(4, 4).as(:year)
651
+ end
652
+
653
+ rule(:ob_date) { date_part | ob_roman_date }
570
654
 
571
655
  rule(:ob_with_sector) do
572
656
  itu_prefix >>
573
657
  (sector >> space).maybe >>
574
658
  ob_series >>
575
- (ob_dot_body | ob_no_body) >>
576
- date_part.maybe >>
659
+ (ob_dot_body | ob_no_body | ob_bare_body) >>
660
+ ob_date.maybe >>
577
661
  language.maybe
578
662
  end
579
663
 
@@ -585,7 +669,7 @@ module Pubid
585
669
  str("Operational Bulletin").as(:_op_bull) >>
586
670
  space >> str("No.") >> space >>
587
671
  number >>
588
- date_part.maybe >>
672
+ ob_date.maybe >>
589
673
  language.maybe
590
674
  end
591
675
 
@@ -671,13 +755,15 @@ module Pubid
671
755
  handbook |
672
756
  numeric_question |
673
757
  letter_question |
758
+ radio_regulations |
674
759
  with_series |
675
760
  contribution |
676
761
  # Unreachable earlier: special_publication needs the literal "OB",
677
762
  # handbook/numeric_question need leading digits, letter_question
678
763
  # needs series >> dot, and contribution needs "-C" after the series.
679
764
  series_code_identifier |
680
- without_series
765
+ without_series |
766
+ publication_id
681
767
  end
682
768
 
683
769
  # Common-text form: an ITU identifier followed by "| ISO/IEC ...".
@@ -16,7 +16,10 @@ module Pubid
16
16
  def generate_base_urn
17
17
  parts = ["urn", "itu"]
18
18
 
19
- if identifier.sector
19
+ # An Operational Bulletin is cross-bureau: its sector is a spelling,
20
+ # not identity (see SpecialPublication#==), so it stays out of the URN.
21
+ if identifier.sector &&
22
+ !identifier.is_a?(Identifiers::SpecialPublication)
20
23
  sector = identifier.sector.to_s
21
24
  parts << sector.to_s.downcase
22
25
  else
@@ -56,7 +59,11 @@ module Pubid
56
59
 
57
60
  if identifier.date
58
61
  date = identifier.date
59
- if date&.year && date.month
62
+ # Only an Operational Bulletin's printed date carries a day
63
+ # ("15.III.2016"); it is in `==`, so it must reach the URN too.
64
+ if date&.year && date.month && date.day
65
+ parts << "#{date.day}/#{date.month}/#{date.year}"
66
+ elsif date&.year && date.month
60
67
  parts << "#{date.month}/#{date.year}"
61
68
  elsif date&.year
62
69
  parts << date.year.to_s
data/lib/pubid/jcgm.rb CHANGED
@@ -35,7 +35,7 @@ module Pubid
35
35
  parser = Parser.new
36
36
  builder = Builder.new
37
37
 
38
- parsed = parser.parse(identifier)
38
+ parsed = Pubid::Parg::Backend.parse(:jcgm, identifier)
39
39
  builder.build(parsed)
40
40
  end
41
41
 
@@ -174,7 +174,7 @@ module Pubid
174
174
  raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
175
175
  end
176
176
 
177
- parsed = Parser.parse(identifier)
177
+ parsed = Pubid::Parg::Backend.parse(:jis, identifier)
178
178
  Builder.build(parsed)
179
179
  end
180
180
  end
@@ -87,6 +87,7 @@ module Pubid
87
87
  # Note: :base_portion is lost during parser merge, so check for supplement indicators
88
88
  if parsed_hash[:supplement_date_range] || parsed_hash[:supplement_slash_year] ||
89
89
  parsed_hash[:supplement_month_year] || parsed_hash[:supplement_year] ||
90
+ parsed_hash[:supplement_empty] ||
90
91
  parsed_hash[:supplement] || parsed_hash[:base_portion]
91
92
  return build_circular_supplement(parsed_hash)
92
93
  end
@@ -696,14 +696,25 @@ module Pubid
696
696
  result += "#{vol_str}n#{issue_number.number}"
697
697
  end
698
698
 
699
- # Use edition component - NO space before edition in MR format (per NIST spec)
700
- result += edition.to_s if edition
699
+ # With a number, the edition glues to it per the NIST spec
700
+ # ("800-53r5"); series-only editions take a dot separator
701
+ # ("NBS.CIRC.e2" — the attested raw spelling; testsuite#5 C4/C5).
702
+ result += if edition
703
+ number ? edition.to_s : ".#{edition}"
704
+ else
705
+ ""
706
+ end
701
707
 
702
708
  # Use version_component
703
709
  result += version_component.to_s(:mr) if version_component
704
710
 
705
- # Supplement (e.g. ".9981sup7") - keep distinct documents distinct
706
- result += supplement_short
711
+ # Supplement (e.g. ".9981sup7") - keep distinct documents distinct;
712
+ # a series-only supplement ("NBS.CIRC.sup") takes the dot itself.
713
+ result += if supplement
714
+ number ? supplement_short : ".#{supplement_short}"
715
+ else
716
+ ""
717
+ end
707
718
 
708
719
  # Use update_component
709
720
  result += update_component.to_s(:mr) if update_component
@@ -726,6 +726,15 @@ module Pubid
726
726
  # 4-digit years so it can't swallow "sup3/1926" or a base number.
727
727
  ((str("supp") | str("sup")) >> match("[0-9]").repeat(4, 4).as(:supp_year_start) >>
728
728
  dash >> match("[0-9]").repeat(4, 4).as(:supp_year_end)).as(:supplement_date_range) |
729
+ # Bare supplement marker to the whole series, no base number
730
+ # ("NBS.CIRC.sup" — testsuite#5 C4): the marker alone is the
731
+ # supplement.
732
+ ((str("supp") | str("sup")) >>
733
+ (
734
+ (month_abbrev >> digits).as(:supplement_month_year) |
735
+ (digits.as(:supp_number) >> slash >> digits.as(:supp_year)).as(:supplement_slash_year) |
736
+ str("").as(:supplement_empty)
737
+ ).maybe) |
729
738
  # With base identifier + supplement
730
739
  (
731
740
  # Capture base portion (everything before "supp" or "sup" or slash+year)
@@ -42,13 +42,22 @@ module Pubid
42
42
  type_token, payload = parts
43
43
  code, revision = parse_payload(payload)
44
44
 
45
- text = "NIST #{type_label(type_token)} #{code}"
45
+ text = "#{publisher_for(type_token)} #{type_label(type_token)} #{code}"
46
46
  text += revision if revision
47
47
  flavor_parse(text)
48
48
  end
49
49
 
50
50
  private
51
51
 
52
+ # The NBS-era series carry the NBS imprint in their canonical form
53
+ # ("NBS CSM 1") — the rebuild must name the publisher the document
54
+ # carries, or the flavor parse rejects its own URN's rebuild.
55
+ NBS_SERIES = ["csm"].freeze
56
+
57
+ def publisher_for(type_token)
58
+ NBS_SERIES.include?(type_token.downcase) ? "NBS" : "NIST"
59
+ end
60
+
52
61
  def parse_payload(payload)
53
62
  # Strip the trailing ".supp" supplement marker.
54
63
  stripped = payload.sub(/\.supp\z/, "")
@@ -146,7 +146,7 @@ module Pubid
146
146
  raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
147
147
  end
148
148
 
149
- parsed = Parser.parse(identifier)
149
+ parsed = Pubid::Parg::Backend.parse(:oasis, identifier)
150
150
  Builder.build(parsed)
151
151
  end
152
152
 
@@ -85,7 +85,7 @@ module Pubid
85
85
  raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
86
86
  end
87
87
 
88
- Builder.build(Parser.parse(identifier))
88
+ Builder.build(Pubid::Parg::Backend.parse(:ogc, identifier))
89
89
  end
90
90
  end
91
91
  end
data/lib/pubid/oiml.rb CHANGED
@@ -37,7 +37,7 @@ module Pubid
37
37
  parser = Parser.new
38
38
  builder = Builder.new
39
39
 
40
- parsed = parser.parse(identifier)
40
+ parsed = Pubid::Parg::Backend.parse(:oiml, identifier)
41
41
  builder.build(parsed)
42
42
  end
43
43
 
@@ -53,7 +53,7 @@ module Pubid
53
53
  raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
54
54
  end
55
55
 
56
- parsed = Parser.parse(identifier)
56
+ parsed = Pubid::Parg::Backend.parse(:omg, identifier)
57
57
  Builder.build(parsed)
58
58
  end
59
59
  end
@@ -0,0 +1,46 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "parsanol"
4
+
5
+ module Pubid
6
+ module Parg
7
+ # A baked, checksum-verified PG artifact for one flavor. The artifact
8
+ # is the parser of record for the flavor: its grammar, tests, entries,
9
+ # and embedded tables travel together and are verified on load.
10
+ class Artifact
11
+ DATA_DIR = File.expand_path("../../../data/parg", __dir__)
12
+ # The deferred entity atoms resolve from_table references at parse
13
+ # time from this dir, so the table YAMLs vendor alongside the
14
+ # baked artifacts.
15
+ TABLES_DIR = File.join(DATA_DIR, "tables")
16
+
17
+ @artifacts = {}
18
+
19
+ class << self
20
+ # Load and memoize the artifact for a flavor (e.g. :iso).
21
+ def for(flavor)
22
+ @artifacts[flavor] ||=
23
+ begin
24
+ path = File.join(DATA_DIR, "#{flavor}.json")
25
+ new(Parsanol::PARG::Artifact.load(path, tables_dir: TABLES_DIR))
26
+ rescue Parsanol::PARG::Error => e
27
+ raise Pubid::Errors::ParseError,
28
+ "PG artifact #{path} rejected: #{e.message}"
29
+ end
30
+ end
31
+ end
32
+
33
+ def initialize(artifact)
34
+ @artifact = artifact
35
+ end
36
+
37
+ def parse(entry, input)
38
+ @artifact.parse(entry, input)
39
+ end
40
+
41
+ def checksum
42
+ @artifact.envelope["checksum"]
43
+ end
44
+ end
45
+ end
46
+ end
@@ -0,0 +1,92 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Pubid
4
+ module Parg
5
+ # The PG-artifact parse backend: runs the flavor's baked artifact on
6
+ # an input and returns the builder-ready attribute hash the flavor's
7
+ # Builder expects — the same hash its parslet parser produced.
8
+ module Backend
9
+ LEAF_KEYS = %i[value line column offset length].freeze
10
+
11
+ module_function
12
+
13
+ ALL_PARTS_SUFFIX = "(all parts)"
14
+
15
+ def parse(flavor, input, entry: "identifier", merge_top_sequence: true)
16
+ # The parslet grammar base strips the "(all parts)" suffix before
17
+ # parsing and marks the tree (Grammar#parse / #mark_all_parts);
18
+ # the artifact backend carries the same contract so every flavor
19
+ # whose grammar does not itself consume the suffix keeps working.
20
+ if input.end_with?(ALL_PARTS_SUFFIX)
21
+ base = input.sub(/\s*\(all parts\)\s*\z/, "")
22
+ tree = to_builder_hash(Artifact.for(flavor).parse(entry, base), merge: merge_top_sequence)
23
+ return mark_all_parts(tree)
24
+ end
25
+
26
+ to_builder_hash(Artifact.for(flavor).parse(entry, input), merge: merge_top_sequence)
27
+ rescue Parsanol::ParseFailed => e
28
+ raise Pubid::Errors::ParseError.new(e.message, nil,
29
+ input: input,
30
+ flavor: registered_name(flavor))
31
+ end
32
+
33
+ # The registered flavor name (Registry-resolvable; "3gpp" for
34
+ # Pubid::Tgpp), falling back to the internal symbol when the
35
+ # flavor is not registered.
36
+ def registered_name(flavor)
37
+ mod = Pubid.const_get(flavor.to_s.split("_").map(&:capitalize).join)
38
+ names = Pubid::Registry.flavor_names.select { |name| Pubid::Registry.get(name) == mod }
39
+ # A module may register under several names; the longest is the
40
+ # canonical one ("cen_cenelec", not the "cen" alias).
41
+ names.max_by(&:length)
42
+ rescue NameError
43
+ nil
44
+ end || flavor.to_s
45
+
46
+ def mark_all_parts(tree)
47
+ case tree
48
+ when Hash then tree.merge(all_parts: true)
49
+ when Array then tree.map { |t| t.merge(all_parts: true) }
50
+ else tree
51
+ end
52
+ end
53
+
54
+ # The artifact emits the parsanol-tree wire shape: capture leaves
55
+ # ({value, line, column, offset, length}) carry their text, and the
56
+ # top-level sequence is a list of capture hashes. The builder-ready
57
+ # form scalarizes leaves and folds that top-level list into one
58
+ # attribute hash (last key wins) — parslet's sequence fold.
59
+ def to_builder_hash(shape, merge: true)
60
+ normalized = normalize(shape)
61
+ return normalized.reduce(:merge) if merge && mergeable_sequence?(normalized)
62
+
63
+ normalized
64
+ end
65
+
66
+ def normalize(node)
67
+ case node
68
+ when Parsanol::Slice then node.content
69
+ when Hash then normalize_hash(node)
70
+ when Array then node.map { |item| normalize(item) }
71
+ else node
72
+ end
73
+ end
74
+
75
+ def normalize_hash(node)
76
+ return node[:value] if wire_leaf?(node)
77
+
78
+ node.transform_values { |value| normalize(value) }
79
+ end
80
+
81
+ def mergeable_sequence?(value)
82
+ value.is_a?(Array) && value.all?(Hash)
83
+ end
84
+
85
+ def wire_leaf?(node)
86
+ return false unless node.size == LEAF_KEYS.size
87
+
88
+ LEAF_KEYS.all? { |key| node.key?(key) }
89
+ end
90
+ end
91
+ end
92
+ end
data/lib/pubid/parg.rb ADDED
@@ -0,0 +1,8 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Pubid
4
+ module Parg
5
+ autoload :Artifact, "pubid/parg/artifact"
6
+ autoload :Backend, "pubid/parg/backend"
7
+ end
8
+ end
data/lib/pubid/pg.rb ADDED
@@ -0,0 +1,8 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Pubid
4
+ module Pg
5
+ autoload :Artifact, "pubid/pg/artifact"
6
+ autoload :Backend, "pubid/pg/backend"
7
+ end
8
+ end
data/lib/pubid/plateau.rb CHANGED
@@ -31,8 +31,7 @@ module Pubid
31
31
 
32
32
  # Apply legacy update_codes normalization first
33
33
  normalized = Core::UpdateCodes.apply(input, :plateau)
34
- parser = Parser.new
35
- parsed = parser.parse(normalized)
34
+ parsed = Pubid::Parg::Backend.parse(:plateau, normalized)
36
35
  Builder.build(parsed)
37
36
  end
38
37
 
@@ -16,7 +16,7 @@ module Pubid
16
16
  raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
17
17
  end
18
18
 
19
- parsed = Parser.parse(input)
19
+ parsed = Pubid::Parg::Backend.parse(:sae, input)
20
20
  Builder.build(parsed)
21
21
  end
22
22
 
@@ -113,7 +113,7 @@ module Pubid
113
113
  raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
114
114
  end
115
115
 
116
- parsed = Parser.parse(identifier)
116
+ parsed = Pubid::Parg::Backend.parse(:tgpp, identifier)
117
117
  Builder.build(parsed)
118
118
  end
119
119
  end
@@ -32,7 +32,7 @@ module Pubid
32
32
  raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
33
33
  end
34
34
 
35
- parsed = Parser.parse(identifier)
35
+ parsed = Pubid::Parg::Backend.parse(:un, identifier, merge_top_sequence: false)
36
36
  Builder.build(parsed)
37
37
  end
38
38
  end
data/lib/pubid/version.rb CHANGED
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Pubid
4
- VERSION = "2.0.0.pre.alpha.13"
4
+ VERSION = "2.0.0.pre.alpha.14"
5
5
  end
@@ -122,7 +122,7 @@ module Pubid
122
122
  raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
123
123
  end
124
124
 
125
- parsed = Parser.parse(identifier)
125
+ parsed = Pubid::Parg::Backend.parse(:w3c, identifier)
126
126
  Builder.build(parsed)
127
127
  end
128
128
  end
@@ -58,7 +58,7 @@ module Pubid
58
58
  raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
59
59
  end
60
60
 
61
- parsed = Parser.parse(identifier)
61
+ parsed = Pubid::Parg::Backend.parse(:xsf, identifier)
62
62
  Builder.build(parsed)
63
63
  end
64
64
  end
data/lib/pubid.rb CHANGED
@@ -17,6 +17,7 @@ require "pubid/prefixes_support"
17
17
  module Pubid
18
18
  autoload :Schema, "pubid/schema"
19
19
  autoload :Conformance, "pubid/conformance"
20
+ autoload :Parg, "pubid/parg"
20
21
  # Upper bound on the length of an identifier string accepted by any +parse+
21
22
  # entry point. Real-world standards identifiers are well under 200 characters;
22
23
  # this limit exists purely to keep pathological, attacker-controlled inputs