pubid 2.0.0.pre.alpha.13 → 2.0.0.pre.alpha.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (137) hide show
  1. checksums.yaml +4 -4
  2. data/data/parg/adobe.json +1 -0
  3. data/data/parg/aiee.parg +37 -0
  4. data/data/parg/amca.json +1 -0
  5. data/data/parg/ansi.json +1 -0
  6. data/data/parg/api.json +1 -0
  7. data/data/parg/ashrae.json +1 -0
  8. data/data/parg/asme.json +1 -0
  9. data/data/parg/astm.json +1 -0
  10. data/data/parg/bipm.json +1 -0
  11. data/data/parg/bsi.json +1 -0
  12. data/data/parg/calconnect.json +1 -0
  13. data/data/parg/ccsds.json +1 -0
  14. data/data/parg/cen_cenelec.json +1 -0
  15. data/data/parg/cen_cenelec.parg +42 -0
  16. data/data/parg/cie.json +1 -0
  17. data/data/parg/csa.json +1 -0
  18. data/data/parg/doi.json +1 -0
  19. data/data/parg/easc.json +1 -0
  20. data/data/parg/ecma.json +1 -0
  21. data/data/parg/etsi.json +1 -0
  22. data/data/parg/evs.json +1 -0
  23. data/data/parg/gb.json +1 -0
  24. data/data/parg/gost.json +1 -0
  25. data/data/parg/iala.json +1 -0
  26. data/data/parg/iana.json +1 -0
  27. data/data/parg/idf.parg +39 -0
  28. data/data/parg/iec.json +1 -0
  29. data/data/parg/ieee.json +1 -0
  30. data/data/parg/ietf.json +1 -0
  31. data/data/parg/ire.parg +39 -0
  32. data/data/parg/isbn.json +1 -0
  33. data/data/parg/iso.json +1 -0
  34. data/data/parg/itu.json +1 -0
  35. data/data/parg/jcgm.json +1 -0
  36. data/data/parg/jis.json +1 -0
  37. data/data/parg/nesc.parg +38 -0
  38. data/data/parg/nist.json +1 -0
  39. data/data/parg/oasis.json +1 -0
  40. data/data/parg/ogc.json +1 -0
  41. data/data/parg/oiml.json +1 -0
  42. data/data/parg/omg.json +1 -0
  43. data/data/parg/plateau.json +1 -0
  44. data/data/parg/sae.json +1 -0
  45. data/data/parg/tables/bipm_groups.yaml +14 -0
  46. data/data/parg/tables/bipm_type_codes.yaml +5 -0
  47. data/data/parg/tables/bipm_type_names_en.yaml +6 -0
  48. data/data/parg/tables/bipm_type_names_fr.yaml +6 -0
  49. data/data/parg/tables/directives_supplements_typed_stages.yaml +3 -0
  50. data/data/parg/tables/directives_typed_stages.yaml +5 -0
  51. data/data/parg/tables/idf_typed_stages.yaml +27 -0
  52. data/data/parg/tables/idf_typed_stages_supplements.yaml +2 -0
  53. data/data/parg/tables/iec_typed_stages.yaml +130 -0
  54. data/data/parg/tables/iso_publishers.yaml +4 -0
  55. data/data/parg/tables/organizations.yaml +12 -0
  56. data/data/parg/tables/tc_types.yaml +42 -0
  57. data/data/parg/tables/typed_stages.yaml +114 -0
  58. data/data/parg/tables/typed_stages_supplements.yaml +64 -0
  59. data/data/parg/tables/wg_types.yaml +21 -0
  60. data/data/parg/tgpp.json +1 -0
  61. data/data/parg/un.json +1 -0
  62. data/data/parg/w3c.json +1 -0
  63. data/data/parg/xsf.json +1 -0
  64. data/lib/pubid/adobe/identifier.rb +11 -1
  65. data/lib/pubid/amca/identifiers/base.rb +1 -1
  66. data/lib/pubid/ansi/identifier.rb +1 -1
  67. data/lib/pubid/api/identifier.rb +1 -1
  68. data/lib/pubid/api/parser.rb +8 -4
  69. data/lib/pubid/ashrae/identifiers/base.rb +10 -1
  70. data/lib/pubid/ashrae/parser.rb +18 -11
  71. data/lib/pubid/asme/identifier.rb +1 -1
  72. data/lib/pubid/astm/identifier.rb +1 -1
  73. data/lib/pubid/bipm/identifier.rb +1 -1
  74. data/lib/pubid/bsi/single_identifier.rb +1 -3
  75. data/lib/pubid/calconnect/identifier.rb +1 -1
  76. data/lib/pubid/ccsds/identifier.rb +1 -1
  77. data/lib/pubid/cen_cenelec/identifier.rb +2 -1
  78. data/lib/pubid/cen_cenelec/parser.rb +9 -2
  79. data/lib/pubid/cie/identifier.rb +1 -1
  80. data/lib/pubid/cie/parser.rb +9 -2
  81. data/lib/pubid/conformance/checks.rb +1 -1
  82. data/lib/pubid/csa/identifier.rb +6 -2
  83. data/lib/pubid/csa/parser.rb +25 -8
  84. data/lib/pubid/doi/identifier.rb +1 -1
  85. data/lib/pubid/easc/identifier.rb +10 -1
  86. data/lib/pubid/ecma/identifier.rb +1 -1
  87. data/lib/pubid/etsi/identifiers/base.rb +1 -1
  88. data/lib/pubid/evs.rb +1 -1
  89. data/lib/pubid/gb/identifier.rb +1 -1
  90. data/lib/pubid/gost/identifier.rb +11 -1
  91. data/lib/pubid/gost/parser.rb +8 -1
  92. data/lib/pubid/iala/identifier.rb +10 -1
  93. data/lib/pubid/iana/identifier.rb +1 -1
  94. data/lib/pubid/idf/builder.rb +5 -0
  95. data/lib/pubid/iec/identifier.rb +1 -1
  96. data/lib/pubid/iec/parser.rb +9 -4
  97. data/lib/pubid/ieee/builder.rb +61 -23
  98. data/lib/pubid/ieee/identifiers/base.rb +1 -1
  99. data/lib/pubid/ieee/identifiers/joint_development.rb +55 -19
  100. data/lib/pubid/ieee/parser.rb +25 -11
  101. data/lib/pubid/ieee/renderer.rb +5 -3
  102. data/lib/pubid/ietf/identifiers/base.rb +1 -1
  103. data/lib/pubid/isbn/identifier.rb +1 -1
  104. data/lib/pubid/iso/identifier.rb +4 -1
  105. data/lib/pubid/iso/normalizer.rb +4 -1
  106. data/lib/pubid/itu/CLAUDE.md +46 -0
  107. data/lib/pubid/itu/builder.rb +24 -4
  108. data/lib/pubid/itu/identifiers/base.rb +11 -18
  109. data/lib/pubid/itu/identifiers/radio_regulations.rb +27 -0
  110. data/lib/pubid/itu/identifiers/special_publication.rb +48 -14
  111. data/lib/pubid/itu/identifiers/standard_serialization.rb +2 -0
  112. data/lib/pubid/itu/identifiers.rb +1 -0
  113. data/lib/pubid/itu/parser.rb +108 -22
  114. data/lib/pubid/itu/urn_generator.rb +9 -2
  115. data/lib/pubid/jcgm.rb +1 -1
  116. data/lib/pubid/jis/identifier.rb +1 -1
  117. data/lib/pubid/nist/builder.rb +1 -0
  118. data/lib/pubid/nist/identifiers/base.rb +15 -4
  119. data/lib/pubid/nist/parser.rb +9 -0
  120. data/lib/pubid/nist/urn_parser.rb +10 -1
  121. data/lib/pubid/oasis/identifier.rb +1 -1
  122. data/lib/pubid/ogc/identifier.rb +1 -1
  123. data/lib/pubid/oiml.rb +1 -1
  124. data/lib/pubid/omg/identifier.rb +1 -1
  125. data/lib/pubid/parg/artifact.rb +46 -0
  126. data/lib/pubid/parg/backend.rb +92 -0
  127. data/lib/pubid/parg.rb +8 -0
  128. data/lib/pubid/pg.rb +8 -0
  129. data/lib/pubid/plateau.rb +1 -2
  130. data/lib/pubid/sae/identifiers/base.rb +1 -1
  131. data/lib/pubid/tgpp/identifier.rb +1 -1
  132. data/lib/pubid/un/identifier.rb +1 -1
  133. data/lib/pubid/version.rb +1 -1
  134. data/lib/pubid/w3c/identifier.rb +1 -1
  135. data/lib/pubid/xsf/identifier.rb +1 -1
  136. data/lib/pubid.rb +1 -0
  137. metadata +83 -2
@@ -595,7 +595,10 @@ module Pubid
595
595
  # ISO/IEC/IEEE P26511/D8-2018 or ISO/IEEE P1003.1-2008 or IEC/IEEE P62582-1-2011
596
596
  # ALSO handle: IEC/IEEE P60780-323, CDV1 2014 (comma before stage code)
597
597
  # ALSO handle: IEEE/CSA P844.1/293.1/D2 (CSA dual numbering)
598
- (str("ISO/IEC/IEEE") | str("ISO/IEEE") | str("IEC/IEEE") | str("IEEE/CSA")).as(:joint_publishers) >>
598
+ # ALSO handle: IEEE/ISO/IEC P42010/D=CD-2020 — the printed publisher
599
+ # order is the organization's perspective (pubid#469) and parses as
600
+ # printed; it is no longer rewritten to ISO-first.
601
+ (str("ISO/IEC/IEEE") | str("IEEE/ISO/IEC") | str("ISO/IEEE") | str("IEC/IEEE") | str("IEEE/CSA")).as(:joint_publishers) >>
599
602
  space >>
600
603
  # P = project (the document is a draft): identity-bearing, so it
601
604
  # is captured and preserved, never silently consumed.
@@ -605,14 +608,19 @@ module Pubid
605
608
  # CSA dual numbering: /293.1 (second number)
606
609
  (slash >> digits >> (dot >> digits).maybe >> (dash >> digits.as(:draft_version)).maybe).maybe >>
607
610
  (
608
- # Variant 1b: the ordinal-less stage draft "D=CDV[:2020]" -
611
+ # Variant 1b: the ordinal-less stage draft "D=CDV-2020" -
609
612
  # D (draft) = CDV (the IEC stage it drafts). The year rides in
610
613
  # the draft clause (a distinct key, so the builder keeps the
611
- # date inside the designator).
614
+ # date inside the designator). The year follows IEEE format:
615
+ # dash-joined ("D=CD-2020", the ruled canonical); the colon
616
+ # spelling ("D=CDV:2020") stays accepted as the alias it was
617
+ # frozen with. A dash-year is guarded to 19xx/20xx so a
618
+ # YYMM monthcode tail can never be read as a year.
612
619
  (slash >> str("D") >> str("=") >>
613
620
  (str("CDV") | str("FDIS") | str("PWI") | str("WD") |
614
621
  str("NP") | str("DIS") | str("CD")).as(:draft_iso_stage) >>
615
- (str(":") >> year_digits.as(:draft_stage_year)).maybe) |
622
+ ((str(":") | str("-")) >>
623
+ ((str("19") | str("20")) >> digit >> digit).as(:draft_stage_year)).maybe) |
616
624
  # Variant 1: /D8 notation (original), with the compound
617
625
  # both-systems suffix "=DDIS.3" (docs/IEEE-DRAFT-STAGES.md §1.3)
618
626
  (slash >> str("D") >> digits.as(:draft_version) >>
@@ -1468,7 +1476,10 @@ module Pubid
1468
1476
  cleaned
1469
1477
  end
1470
1478
 
1471
- def self.parse(string)
1479
+ # Pre-parse ingestion normalizations (R2): every parse path —
1480
+ # parslet and PG artifact alike — feeds the grammar the same
1481
+ # normalized string.
1482
+ def self.normalize_input(string)
1472
1483
  # Strip .pdf extension if present (Pattern 3: File Extensions)
1473
1484
  cleaned = string.sub(/\.pdf$/i, "")
1474
1485
 
@@ -1822,11 +1833,10 @@ module Pubid
1822
1833
  # Fix 2H: "IEC XXXX First edition YYYY-MM; IEEE NNNN" -> normalize semicolon
1823
1834
  # Already handled by earlier semicolon normalization
1824
1835
 
1825
- # Fix 2I: "IEEE/ISO/IEC PXXX/DIS" -> normalize to "ISO/IEC/IEEE PXXX/DIS"
1826
- cleaned = cleaned.gsub(/^IEEE\/ISO\/IEC\s+(P[\w.-]+)/,
1827
- 'ISO/IEC/IEEE \1')
1828
- cleaned = cleaned.gsub(/^IEEE\/IEC\/ISO\s+(P[\w.-]+)/,
1829
- 'IEC/ISO/IEEE \1')
1836
+ # (Fix 2I removed: "IEEE/ISO/IEC PXXX/…" is no longer rewritten to
1837
+ # ISO-first — the printed publisher order is the organization's
1838
+ # perspective (pubid#469) and the joint P-form rule parses it as
1839
+ # printed.)
1830
1840
 
1831
1841
  # Fix 2J: "IEEE/IEC PXXX D5" -> normalize space to slash before D
1832
1842
  cleaned = cleaned.gsub(/^(IEEE\/IEC P[\w.-]+)\s+D(\d)/, '\1/D\2')
@@ -1925,7 +1935,11 @@ module Pubid
1925
1935
  # Fix 2AF: "IEEE Std 1003.1/2003.l/lNT" -> fix typos
1926
1936
  # .l -> .1 and lNT -> INT handled by existing fixes
1927
1937
 
1928
- new.parse(cleaned)
1938
+ cleaned
1939
+ end
1940
+
1941
+ def self.parse(string)
1942
+ new.parse(normalize_input(string))
1929
1943
  end
1930
1944
  end
1931
1945
  end
@@ -169,9 +169,11 @@ module Pubid
169
169
  # ("D08, September, 2018").
170
170
  if id.draft_obj
171
171
  printed = id.draft_obj.to_s
172
- if id.publisher == "IEC" && id.copublisher == ["IEEE"]
173
- printed = printed.split(", ").first
174
- elsif id.publisher == "IEEE" && id.draft_status.to_s.empty?
172
+ # The draft designator carries its date on every lead
173
+ # (docs/IEEE-DRAFT-STAGES.md §3: "the draft designator with its
174
+ # date" — the IEC/IEEE-led date-drop contradicted the DCD
175
+ # convention and the raw records).
176
+ if id.publisher == "IEEE" && id.draft_status.to_s.empty?
175
177
  # The long comma form is for dated project drafts; an
176
178
  # unapproved-draft render keeps its pinned single-comma form
177
179
  # (pubid#318 idempotence).
@@ -64,7 +64,7 @@ module Pubid
64
64
  raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
65
65
  end
66
66
 
67
- parsed = Parser.parse(identifier)
67
+ parsed = Pubid::Parg::Backend.parse(:ietf, identifier)
68
68
  Builder.build(parsed)
69
69
  end
70
70
  end
@@ -64,7 +64,7 @@ module Pubid
64
64
 
65
65
  # @raise [Pubid::Errors::ParseError] if the string is not a valid ISBN
66
66
  def self.build_identifier(identifier)
67
- parsed = Parser.parse(identifier)
67
+ parsed = Pubid::Parg::Backend.parse(:isbn, identifier)
68
68
  Builder.build(parsed)
69
69
  rescue ArgumentError => e
70
70
  # The Builder validates length and check digit. Surface that as a parse
@@ -304,7 +304,10 @@ module Pubid
304
304
  when :mr_string
305
305
  Pubid::Parsers::MrString.parse(string)
306
306
  else
307
- parsed = Pubid::Iso::Parser.new.parse(string)
307
+ # R1 parser swap: the baked PG artifact is the identifier
308
+ # parser of record; the Builder consumes the same attribute
309
+ # hash it always has.
310
+ parsed = Pubid::Parg::Backend.parse(:iso, string)
308
311
  Pubid::Iso::Builder.new.build(parsed)
309
312
  end
310
313
  end
@@ -27,7 +27,10 @@ module Pubid
27
27
  private
28
28
 
29
29
  def parse_with_builder(string)
30
- parsed = Pubid::Iso.parser.parse(string)
30
+ # R2 ingestion hook: normalized strings reach the model through
31
+ # the same parser of record as Identifier.parse (the baked PG
32
+ # artifact), so every Tier-3 normalization feeds one grammar.
33
+ parsed = Pubid::Parg::Backend.parse(:iso, string)
31
34
  Pubid::Iso.builder.build(parsed)
32
35
  end
33
36
 
@@ -67,3 +67,49 @@ on any call. The base now derives the key from the class name
67
67
  `underscore` is not a dependency. metanorma-itu constructs through this
68
68
  lookup; its flavor-local `pubid_contribution.rb` render override can be
69
69
  deleted once it migrates.
70
+
71
+ ## relaton's query forms — RR, OB sector, publication ids
72
+
73
+ Four forms relaton's `Relaton::Itu::Pubid` parsed and `Pubid::Itu` did not
74
+ (hand-off itu-relaton-query-forms). None of them occurs in the published
75
+ `relaton-data-itu` index — no `series: RR`, no OB row, no six-digit part — and
76
+ a replay of all 24,382 rows and every ITU pass fixture showed 0 changes.
77
+
78
+ - **Radio Regulations** — `Identifiers::RadioRegulations`
79
+ (`pubid:itu:radio-regulations`): `ITU-R RR`, `ITU-R RR (2020)`, and the URL
80
+ spelling `ITU-R RR-2020`, which used to build a *wrong* Recommendation
81
+ (series `RR`, number `2020`) with no error. "RR" is the series; there is no
82
+ code, so `#number` returns the series and `root.number` is `"RR"`. The rule
83
+ ends in `any.absent?`: it sits before `with_series`, and PEG never re-enters
84
+ the alternation, so a partial match on `ITU-R RR.1` must fail inside it.
85
+ - **Operational Bulletins keep their sector.** This reverses the old
86
+ "cross-bureau, sector must not be set" rule: `validate_ob_no_sector!` is
87
+ gone, `ITU-T OB.1096 (2016)` renders back as it is, and the sector-less
88
+ `ITU OB No. 1096` is unchanged. The long forms with a sector (`ITU-T OB No.
89
+ 1096`) now normalise to `ITU-T OB.1096`. `No.` stays in the default
90
+ render — it is how ITU's bulletin site and pubid v1 write it — and
91
+ metanorma-itu's `ITU OB 1000` / `Annex to ITU OB 1000` (its i18n template
92
+ omits `No.`) is accepted on parse (`ob_bare_body`) and rendered with it.
93
+ The sector is a spelling, not
94
+ identity: `SpecialPublication#==` skips it, the URN keeps `urn:itu:itu:…`
95
+ and `mr_type` stays nil, so both spellings are one bulletin on every
96
+ surface except `to_s`/`to_hash`. The printed date `- 15.III.2016` sets
97
+ `date.day`, which only this form does, so `render_ob_date` uses the day as
98
+ the spelling marker; `day_to_kv` emits only then.
99
+ - **`-YYYYMM` is a date, never a part.** `part` refuses a six-digit run with a
100
+ 19xx/20xx year and a 01–12 month (`yyyymm_shape`), and `id_date` reads it as
101
+ year+month. `-200313`, `-180001` stay parts. `ITU-T REC T.4` drops the
102
+ uncaptured `rec_word`; `T-REC-T.4-200307-I` is `publication_id`, last in
103
+ `identifier` (nothing else starts with a bare sector letter), and requires
104
+ the date. The trailing status letter (`I` in force, `S` superseded) is
105
+ parsed and dropped: it names the state of an edition, not the edition.
106
+ **`S` is also the Spanish language suffix**, so it is a status only inside
107
+ the full `T-REC-…` id (`id_status`); after an `ITU-T …-YYYYMM` print form
108
+ only `-I` is (`print_id_status`), and `ITU-T Z.100-199911-S` keeps language
109
+ `S`. The `-YYYYMM` date and `REC` word are also in `base_with_series`/
110
+ `base_without_series`: the part guard applies there too, so without them
111
+ `ITU-T G.989-200307 Amd 1` — a (wrong) Amendment on `main` — stopped parsing.
112
+ The day of an OB date reaches the URN (`…:15/03/2016`), since it is in `==`.
113
+ **Not done:** an RR supplement (`ITU-R RR (2020) Amd 1` fails; the
114
+ base-less `ITU-R RR Amd 1` still builds an Amendment on series `RR`), and
115
+ `ITU-R RR-E` is still a Recommendation numbered `E` — both as on `main`.
@@ -50,6 +50,16 @@ module Pubid
50
50
  return sp
51
51
  end
52
52
 
53
+ # Radio Regulations — "ITU-R RR (2020)"
54
+ if data[:radio_regulations]
55
+ return Identifiers::RadioRegulations.new(
56
+ sector: Components::Sector.new(sector: data[:sector].to_s),
57
+ series: Components::Series.new(series: "RR"),
58
+ date: data[:year] ? build_date(data) : nil,
59
+ language: data[:language]&.to_s,
60
+ )
61
+ end
62
+
53
63
  # Check if this is a supplement identifier
54
64
  if data[:supplement_type]
55
65
  supp = build_supplement(data)
@@ -153,11 +163,12 @@ module Pubid
153
163
  nil
154
164
  end
155
165
 
156
- # Build Special Publication (OB). Sector is silently dropped — OB is a
157
- # cross-bureau publication and `Identifier` rejects sector+OB
158
- # in its constructor.
166
+ # Build Special Publication (OB). The sector of the TSB spelling
167
+ # ("ITU-T OB.1096") is kept so the bulletin renders back as it was cited;
168
+ # SpecialPublication#== ignores it, since OB is cross-bureau.
159
169
  def build_special_publication(data)
160
170
  Identifiers::SpecialPublication.new(
171
+ sector: (Components::Sector.new(sector: data[:sector].to_s) if data[:sector]),
161
172
  series: Components::Series.new(series: "OB"),
162
173
  code: data[:number] ? build_code(data) : nil,
163
174
  date: data[:year] ? build_date(data) : nil,
@@ -344,10 +355,19 @@ module Pubid
344
355
  )
345
356
  end
346
357
 
358
+ # The Roman month of a bulletin date ("15.III.2016") is stored as the
359
+ # two-digit month every other ITU date uses; the day marks the spelling.
347
360
  def build_date(data)
361
+ month = if data[:roman_month]
362
+ (Identifiers::SpecialPublication::ROMAN_MONTHS.index(data[:roman_month].to_s) + 1)
363
+ .to_s.rjust(2, "0")
364
+ else
365
+ data[:month]&.to_s
366
+ end
348
367
  Pubid::Components::Date.new(
349
368
  year: data[:year].to_s,
350
- month: data[:month]&.to_s,
369
+ month: month,
370
+ day: data[:day]&.to_s,
351
371
  )
352
372
  end
353
373
 
@@ -15,7 +15,7 @@ module Pubid
15
15
  raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
16
16
  end
17
17
 
18
- parsed = Parser.parse(normalize_whitespace(identifier))
18
+ parsed = Pubid::Parg::Backend.parse(:itu, normalize_whitespace(identifier))
19
19
  Builder.build(parsed)
20
20
  end
21
21
 
@@ -101,8 +101,6 @@ module Pubid
101
101
  end
102
102
 
103
103
  super
104
-
105
- validate_ob_no_sector!
106
104
  end
107
105
 
108
106
  # The document number lives on the `code` component for ITU; surface it at
@@ -462,6 +460,16 @@ module Pubid
462
460
  model.code ||= Components::Code.new
463
461
  end
464
462
 
463
+ # The day is set only by the printed bulletin date ("15.III.2016"), so
464
+ # it emits only there and no existing index row gains a key.
465
+ def day_to_kv(model, doc)
466
+ emit_kv(doc, "day", model.date&.day)
467
+ end
468
+
469
+ def day_from_kv(model, value)
470
+ date_for(model).day = value.to_s
471
+ end
472
+
465
473
  def date_for(model)
466
474
  model.date ||= Pubid::Components::Date.new
467
475
  end
@@ -472,21 +480,6 @@ module Pubid
472
480
  str = value.to_s
473
481
  LANGUAGES[str] || str
474
482
  end
475
-
476
- # OB (Operational Bulletin) is a cross-bureau ITU publication and
477
- # must not have a sector. Direct construction with both raises;
478
- # the parser silently drops sector for legacy strings like
479
- # "ITU-T OB.X" (handled in Builder).
480
- def validate_ob_no_sector!
481
- return unless series&.series == "OB"
482
- return if sector.nil?
483
- return if sector.is_a?(Components::Sector) && (sector.sector.nil? || sector.sector.to_s.empty?)
484
-
485
- raise ArgumentError,
486
- "OB (Operational Bulletin) is a cross-bureau ITU publication; " \
487
- "sector must not be set"
488
- end
489
483
  end
490
-
491
484
  end
492
485
  end
@@ -0,0 +1,27 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Pubid
4
+ module Itu
5
+ module Identifiers
6
+ # The ITU Radio Regulations — the treaty text revised by each World
7
+ # Radiocommunication Conference.
8
+ # Format: ITU-R RR [(YYYY)]
9
+ # Example: ITU-R RR (2020)
10
+ #
11
+ # It has no document number: "RR" is the whole designation, stored as
12
+ # the series. `#number` returns it, so `root.number` (the relaton-index
13
+ # key) is not empty.
14
+ class RadioRegulations < Identifier
15
+ include StandardSerialization
16
+
17
+ def number
18
+ series&.series
19
+ end
20
+
21
+ def render_base(**_opts)
22
+ "#{publisher}-#{sector} #{series}#{render_date_suffix}"
23
+ end
24
+ end
25
+ end
26
+ end
27
+ end
@@ -4,26 +4,60 @@ module Pubid
4
4
  module Itu
5
5
  module Identifiers
6
6
  # ITU Special Publication — currently models the Operational Bulletin (OB).
7
- # OB is a cross-bureau publication (no sector) rendered as
8
- # "ITU OB No. {number}" with optional date.
9
- #
10
- # Pattern: "ITU OB No. 1283 (01/2024)"
7
+ # OB is a cross-bureau publication. ITU prints it without a sector
8
+ # ("ITU OB No. 1283 (01/2024)"); the TSB spelling carries one
9
+ # ("ITU-T OB.1096 (2016)"), which is kept and rendered back. The sector
10
+ # is not part of the identity: `==`, the URN and the MR slug ignore it,
11
+ # so both spellings name one bulletin.
11
12
  class SpecialPublication < Identifier
12
13
  include StandardSerialization
13
14
 
15
+ ROMAN_MONTHS = %w[I II III IV V VI VII VIII IX X XI XII].freeze
16
+
14
17
  def render_base(**_opts)
15
18
  number = code&.number
16
- result = "#{publisher} #{series} No. #{number}"
17
-
18
- if date
19
- result += if date.month
20
- " (#{date.month.to_s.rjust(2, '0')}/#{date.year})"
21
- else
22
- " (#{date.year})"
23
- end
24
- end
19
+ result = if sector
20
+ "#{publisher}-#{sector} #{series}.#{number}"
21
+ else
22
+ "#{publisher} #{series} No. #{number}"
23
+ end
24
+
25
+ result + render_ob_date
26
+ end
27
+
28
+ # Cross-bureau: the sector is how one bureau cites the bulletin, not
29
+ # which bulletin it is.
30
+ def ==(other)
31
+ return false unless other.instance_of?(self.class)
32
+
33
+ series == other.series &&
34
+ code == other.code &&
35
+ date == other.date &&
36
+ language == other.language &&
37
+ common_text_twin == other.common_text_twin
38
+ end
39
+
40
+ # Keep the MR slug sector-free, like `==` (it was always so, because
41
+ # the sector used to be dropped).
42
+ def mr_type
43
+ nil
44
+ end
25
45
 
26
- result
46
+ private
47
+
48
+ # " (MM/YYYY)" or " (YYYY)"; the day-bearing date is the printed
49
+ # bulletin form " - 15.III.2016", the only spelling that sets a day.
50
+ def render_ob_date
51
+ return "" unless date
52
+
53
+ if date.day && date.month
54
+ roman = ROMAN_MONTHS[date.month.to_i - 1]
55
+ " - #{date.day.to_s.rjust(2, '0')}.#{roman}.#{date.year}"
56
+ elsif date.month
57
+ " (#{date.month.to_s.rjust(2, '0')}/#{date.year})"
58
+ else
59
+ " (#{date.year})"
60
+ end
27
61
  end
28
62
  end
29
63
  end
@@ -46,6 +46,8 @@ module Pubid
46
46
  with: { to: :year_to_kv, from: :year_from_kv }
47
47
  map "month",
48
48
  with: { to: :month_to_kv, from: :month_from_kv }
49
+ map "day",
50
+ with: { to: :day_to_kv, from: :day_from_kv }
49
51
  map "language", to: :language
50
52
  map "common_text_twin",
51
53
  with: { to: :common_text_twin_to_kv,
@@ -16,6 +16,7 @@ module Pubid
16
16
  autoload :Errata, "#{__dir__}/identifiers/errata"
17
17
  autoload :Handbook, "#{__dir__}/identifiers/handbook"
18
18
  autoload :Question, "#{__dir__}/identifiers/question"
19
+ autoload :RadioRegulations, "#{__dir__}/identifiers/radio_regulations"
19
20
  autoload :Recommendation, "#{__dir__}/identifiers/recommendation"
20
21
  autoload :Report, "#{__dir__}/identifiers/report"
21
22
  autoload :SpecialPublication, "#{__dir__}/identifiers/special_publication"
@@ -125,9 +125,48 @@ module Pubid
125
125
  dash >> (letter.repeat(1, 3) >> dot >> digits).as(:range_end)
126
126
  end
127
127
 
128
- # Parts
128
+ # A "-YYYYMM" approval date — the "200307" of "T-REC-T.4-200307-I" and
129
+ # "ITU-T T.4-200307". ITU's own edition suffix is short ("-5"), so six
130
+ # digits that read as a plausible year (19xx/20xx) and month (01-12) are
131
+ # the date, never a part. A six-digit run that fails either test
132
+ # ("-200313", "-180001") is still a part, as it was before.
133
+ rule(:yyyymm_year) { (str("19") | str("20")) >> digit >> digit }
134
+ rule(:yyyymm_month) do
135
+ (str("0") >> match["1-9"]) | (str("1") >> match["0-2"])
136
+ end
137
+ rule(:yyyymm_shape) { yyyymm_year >> yyyymm_month >> digit.absent? }
138
+
139
+ # The status letter that trails the date in a publication id — "I" (in
140
+ # force) or "S" (superseded). It names the state of the edition, not the
141
+ # edition, so it is parsed and dropped. "S" is also the Spanish language
142
+ # suffix, so it is a status ONLY in the full "T-REC-…" id, where ITU
143
+ # always writes one; after an "ITU-T …-YYYYMM" print form only "I" is,
144
+ # and "-S" stays the language ("ITU-T Z.100-199911-S").
145
+ rule(:id_status) do
146
+ dash >> match["IS"] >> match["A-Za-z0-9"].absent?
147
+ end
148
+
149
+ rule(:print_id_status) do
150
+ dash >> str("I") >> match["A-Za-z0-9"].absent?
151
+ end
152
+
153
+ rule(:yyyymm_date) do
154
+ dash >> yyyymm_year.as(:year) >> yyyymm_month.as(:month) >>
155
+ digit.absent?
156
+ end
157
+
158
+ rule(:id_date) { yyyymm_date >> print_id_status.maybe }
159
+
160
+ # Either date spelling of a Recommendation.
161
+ rule(:document_date) { date_part | id_date }
162
+
163
+ # "ITU-T REC T.4", "ITU-T REC-T.4" — the redundant type word of ITU's
164
+ # own URLs. Not captured: a Recommendation is the default type.
165
+ rule(:rec_word) { str("REC") >> (space | dash) }
166
+
167
+ # Parts. The yyyymm guard keeps the approval date out of the part list.
129
168
  rule(:part) do
130
- dash >> digits.as(:part)
169
+ dash >> yyyymm_shape.absent? >> digits.as(:part)
131
170
  end
132
171
 
133
172
  rule(:parts) { part.repeat(0).as(:parts) }
@@ -273,6 +312,7 @@ module Pubid
273
312
  itu_prefix >>
274
313
  sector >>
275
314
  space >>
315
+ rec_word.maybe >>
276
316
  series >> dot >>
277
317
  code >>
278
318
  range_end.maybe >>
@@ -281,7 +321,7 @@ module Pubid
281
321
  series_word.maybe >>
282
322
  attachment.maybe >>
283
323
  version_part.maybe >>
284
- date_part.maybe
324
+ document_date.maybe
285
325
  end
286
326
 
287
327
  rule(:base_without_series) do
@@ -292,7 +332,7 @@ module Pubid
292
332
  code_suffixes >>
293
333
  attachment.maybe >>
294
334
  version_part.maybe >>
295
- date_part.maybe
335
+ document_date.maybe
296
336
  end
297
337
 
298
338
  # A series-code document — "EMC-5", "MES-2", "QOS-2", "IMPL-8",
@@ -311,12 +351,10 @@ module Pubid
311
351
  # The number stays in `code.number`, so `root.number` — the field
312
352
  # relaton-index bsearches on — is "5" for EMC-5 and "QKD" for SEC-QKD
313
353
  # rather than nil.
314
- # The OB guard keeps "ITU-T OB-1" a clean parse failure. Without it the
315
- # string reaches Builder#build's Recommendation fallback, whose
316
- # validate_ob_no_sector! raises an ArgumentError that escapes
317
- # Identifier.parse's Parslet::ParseFailed rescue — turning a rejected
318
- # input into a crash for callers. It guards "OB" + dash specifically, so
319
- # a genuine two-letter mnemonic starting "OB" would still parse.
354
+ # The OB guard keeps "ITU-T OB-1" a clean parse failure rather than a
355
+ # Recommendation of a series "OB" — the Operational Bulletin's series
356
+ # name. It guards "OB" + dash specifically, so a genuine two-letter
357
+ # mnemonic starting "OB" would still parse.
320
358
  rule(:series_code_body) do
321
359
  (str("OB") >> dash).absent? >>
322
360
  letter.repeat(2).as(:series) >> dash.as(:series_dash) >>
@@ -365,10 +403,8 @@ module Pubid
365
403
  # Builder#build's `combined` branch, which builds a CombinedIdentifier and
366
404
  # would silently drop the marker — a clean parse failure is better than a
367
405
  # Report that comes back as a Recommendation. The OB guard mirrors
368
- # series_code_body's: an Operational Bulletin is cross-bureau, and
369
- # "Report ITU-T OB.1" would otherwise route to SpecialPublication (marker
370
- # dropped) or hit validate_ob_no_sector!, whose ArgumentError escapes
371
- # Identifier.parse's Parslet::ParseFailed rescue.
406
+ # series_code_body's: "Report ITU-T OB.1" would otherwise route to
407
+ # SpecialPublication with the marker dropped.
372
408
  rule(:report_body) do
373
409
  (str("OB") >> dot).absent? >>
374
410
  (series >> dot).maybe >>
@@ -535,6 +571,7 @@ module Pubid
535
571
  itu_prefix >>
536
572
  sector >>
537
573
  space >>
574
+ rec_word.maybe >>
538
575
  series >> dot >>
539
576
  code >>
540
577
  range_end.maybe >>
@@ -543,7 +580,7 @@ module Pubid
543
580
  series_word.maybe >>
544
581
  attachment.maybe >>
545
582
  version_part.maybe >>
546
- date_part.maybe >>
583
+ document_date.maybe >>
547
584
  language.maybe
548
585
  end
549
586
 
@@ -556,24 +593,71 @@ module Pubid
556
593
  code_suffixes >>
557
594
  attachment.maybe >>
558
595
  version_part.maybe >>
559
- date_part.maybe >>
596
+ document_date.maybe >>
597
+ language.maybe
598
+ end
599
+
600
+ # ITU's publication id — "T-REC-T.4-200307-I",
601
+ # "R-REC-BO.1130-5-202602-I": <sector>-REC-<number>[-<edition>]-<YYYYMM>
602
+ # [-<status>], the name ITU gives each edition in its URLs and PDF
603
+ # files. It builds the plain Recommendation it names and renders in the
604
+ # print form ("ITU-T T.4 (07/2003)"). The date is required: without it
605
+ # the string names no edition. No other rule starts with a bare sector
606
+ # letter, so the slot is free.
607
+ rule(:publication_id) do
608
+ sector >> dash >> str("REC") >> dash >>
609
+ series >> dot >> code >> yyyymm_date >> id_status.maybe >>
560
610
  language.maybe
561
611
  end
562
612
 
613
+ # The Radio Regulations — "ITU-R RR", "ITU-R RR (2020)", and the URL
614
+ # spelling "ITU-R RR-2020". Always ITU-R. The trailing any.absent? is
615
+ # load-bearing: PEG ordered choice never re-enters the alternation once
616
+ # an alternative succeeds, so a partial match on "ITU-R RR.1" must fail
617
+ # here and fall through to with_series.
618
+ rule(:radio_regulations) do
619
+ itu_prefix >> str("R").as(:sector) >> space >>
620
+ str("RR").as(:radio_regulations) >>
621
+ (date_part | (dash >> digit.repeat(4, 4).as(:year))).maybe >>
622
+ language.maybe >> any.absent?
623
+ end
624
+
563
625
  # OB (Operational Bulletin) — Special Publication.
564
- # OB is a cross-bureau ITU publication; sector, when present in legacy
565
- # strings like "ITU-T OB.1096", is silently dropped by the builder.
626
+ # OB is a cross-bureau ITU publication. The TSB spelling carries a
627
+ # sector ("ITU-T OB.1096 (2016)"); the builder keeps it, and it renders
628
+ # back, but it is not part of the bulletin's identity.
566
629
  rule(:ob_series) { str("OB").as(:series) }
567
630
 
568
631
  rule(:ob_dot_body) { dot >> number }
569
632
  rule(:ob_no_body) { space >> str("No.") >> space >> number }
633
+ # "ITU OB 1000" — metanorma-itu's docidentifier ("Annex to ITU OB %").
634
+ # Accepted as an input spelling only; it renders "ITU OB No. 1000", the
635
+ # form ITU's own bulletin site uses.
636
+ rule(:ob_bare_body) { space >> number }
637
+
638
+ # The date as a bulletin prints it — "ITU-T OB.1096 - 15.III.2016": day,
639
+ # Roman month, year. The months are tried longest first, because PEG
640
+ # takes the first alternative that matches and "I" would otherwise win
641
+ # on "III"; "XIII" matches "XII", then fails on the required dot.
642
+ rule(:roman_month) do
643
+ %w[XII XI X IX VIII VII VI V IV III II I]
644
+ .map { |m| str(m) }.reduce(:|)
645
+ end
646
+
647
+ rule(:ob_roman_date) do
648
+ str(" - ") >> digit.repeat(2, 2).as(:day) >> dot >>
649
+ roman_month.as(:roman_month) >> dot >>
650
+ digit.repeat(4, 4).as(:year)
651
+ end
652
+
653
+ rule(:ob_date) { date_part | ob_roman_date }
570
654
 
571
655
  rule(:ob_with_sector) do
572
656
  itu_prefix >>
573
657
  (sector >> space).maybe >>
574
658
  ob_series >>
575
- (ob_dot_body | ob_no_body) >>
576
- date_part.maybe >>
659
+ (ob_dot_body | ob_no_body | ob_bare_body) >>
660
+ ob_date.maybe >>
577
661
  language.maybe
578
662
  end
579
663
 
@@ -585,7 +669,7 @@ module Pubid
585
669
  str("Operational Bulletin").as(:_op_bull) >>
586
670
  space >> str("No.") >> space >>
587
671
  number >>
588
- date_part.maybe >>
672
+ ob_date.maybe >>
589
673
  language.maybe
590
674
  end
591
675
 
@@ -671,13 +755,15 @@ module Pubid
671
755
  handbook |
672
756
  numeric_question |
673
757
  letter_question |
758
+ radio_regulations |
674
759
  with_series |
675
760
  contribution |
676
761
  # Unreachable earlier: special_publication needs the literal "OB",
677
762
  # handbook/numeric_question need leading digits, letter_question
678
763
  # needs series >> dot, and contribution needs "-C" after the series.
679
764
  series_code_identifier |
680
- without_series
765
+ without_series |
766
+ publication_id
681
767
  end
682
768
 
683
769
  # Common-text form: an ITU identifier followed by "| ISO/IEC ...".