pubid 2.0.0.pre.alpha.13 → 2.0.0.pre.alpha.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. checksums.yaml +4 -4
  2. data/data/parg/tables/bipm_groups.yaml +14 -0
  3. data/data/parg/tables/bipm_type_codes.yaml +5 -0
  4. data/data/parg/tables/bipm_type_names_en.yaml +6 -0
  5. data/data/parg/tables/bipm_type_names_fr.yaml +6 -0
  6. data/data/parg/tables/directives_supplements_typed_stages.yaml +3 -0
  7. data/data/parg/tables/directives_typed_stages.yaml +5 -0
  8. data/data/parg/tables/idf_typed_stages.yaml +27 -0
  9. data/data/parg/tables/idf_typed_stages_supplements.yaml +2 -0
  10. data/data/parg/tables/iec_typed_stages.yaml +130 -0
  11. data/data/parg/tables/iso_publishers.yaml +4 -0
  12. data/data/parg/tables/organizations.yaml +12 -0
  13. data/data/parg/tables/tc_types.yaml +42 -0
  14. data/data/parg/tables/typed_stages.yaml +114 -0
  15. data/data/parg/tables/typed_stages_supplements.yaml +64 -0
  16. data/data/parg/tables/wg_types.yaml +21 -0
  17. data/lib/pubid/adobe/identifier.rb +11 -1
  18. data/lib/pubid/amca/identifiers/base.rb +1 -1
  19. data/lib/pubid/ansi/identifier.rb +1 -1
  20. data/lib/pubid/api/identifier.rb +1 -1
  21. data/lib/pubid/api/parser.rb +8 -4
  22. data/lib/pubid/ashrae/identifiers/base.rb +10 -1
  23. data/lib/pubid/ashrae/parser.rb +18 -11
  24. data/lib/pubid/asme/identifier.rb +1 -1
  25. data/lib/pubid/astm/identifier.rb +1 -1
  26. data/lib/pubid/bipm/identifier.rb +1 -1
  27. data/lib/pubid/bsi/single_identifier.rb +1 -3
  28. data/lib/pubid/calconnect/identifier.rb +1 -1
  29. data/lib/pubid/ccsds/identifier.rb +1 -1
  30. data/lib/pubid/cen_cenelec/identifier.rb +2 -1
  31. data/lib/pubid/cen_cenelec/parser.rb +9 -2
  32. data/lib/pubid/cie/identifier.rb +1 -1
  33. data/lib/pubid/cie/parser.rb +9 -2
  34. data/lib/pubid/conformance/checks.rb +1 -1
  35. data/lib/pubid/csa/identifier.rb +6 -2
  36. data/lib/pubid/csa/parser.rb +25 -8
  37. data/lib/pubid/doi/identifier.rb +1 -1
  38. data/lib/pubid/easc/identifier.rb +10 -1
  39. data/lib/pubid/ecma/identifier.rb +1 -1
  40. data/lib/pubid/etsi/identifiers/base.rb +1 -1
  41. data/lib/pubid/evs.rb +1 -1
  42. data/lib/pubid/gb/identifier.rb +1 -1
  43. data/lib/pubid/gost/identifier.rb +11 -1
  44. data/lib/pubid/gost/parser.rb +8 -1
  45. data/lib/pubid/iala/identifier.rb +10 -1
  46. data/lib/pubid/iana/identifier.rb +1 -1
  47. data/lib/pubid/idf/builder.rb +5 -0
  48. data/lib/pubid/iec/identifier.rb +1 -1
  49. data/lib/pubid/iec/parser.rb +9 -4
  50. data/lib/pubid/ieee/builder.rb +61 -23
  51. data/lib/pubid/ieee/identifiers/base.rb +1 -1
  52. data/lib/pubid/ieee/identifiers/joint_development.rb +55 -19
  53. data/lib/pubid/ieee/parser.rb +25 -11
  54. data/lib/pubid/ieee/renderer.rb +5 -3
  55. data/lib/pubid/ietf/identifiers/base.rb +1 -1
  56. data/lib/pubid/isbn/identifier.rb +1 -1
  57. data/lib/pubid/iso/identifier.rb +4 -1
  58. data/lib/pubid/iso/normalizer.rb +4 -1
  59. data/lib/pubid/itu/CLAUDE.md +46 -0
  60. data/lib/pubid/itu/builder.rb +24 -4
  61. data/lib/pubid/itu/identifiers/base.rb +11 -18
  62. data/lib/pubid/itu/identifiers/radio_regulations.rb +27 -0
  63. data/lib/pubid/itu/identifiers/special_publication.rb +48 -14
  64. data/lib/pubid/itu/identifiers/standard_serialization.rb +2 -0
  65. data/lib/pubid/itu/identifiers.rb +1 -0
  66. data/lib/pubid/itu/parser.rb +108 -22
  67. data/lib/pubid/itu/urn_generator.rb +9 -2
  68. data/lib/pubid/jcgm.rb +1 -1
  69. data/lib/pubid/jis/identifier.rb +1 -1
  70. data/lib/pubid/nist/builder.rb +1 -0
  71. data/lib/pubid/nist/identifiers/base.rb +15 -4
  72. data/lib/pubid/nist/parser.rb +9 -0
  73. data/lib/pubid/nist/urn_parser.rb +10 -1
  74. data/lib/pubid/oasis/identifier.rb +1 -1
  75. data/lib/pubid/ogc/identifier.rb +1 -1
  76. data/lib/pubid/oiml.rb +1 -1
  77. data/lib/pubid/omg/identifier.rb +1 -1
  78. data/lib/pubid/parg/artifact.rb +46 -0
  79. data/lib/pubid/parg/backend.rb +92 -0
  80. data/lib/pubid/parg.rb +8 -0
  81. data/lib/pubid/pg.rb +8 -0
  82. data/lib/pubid/plateau.rb +1 -2
  83. data/lib/pubid/sae/identifiers/base.rb +1 -1
  84. data/lib/pubid/tgpp/identifier.rb +1 -1
  85. data/lib/pubid/un/identifier.rb +1 -1
  86. data/lib/pubid/version.rb +1 -1
  87. data/lib/pubid/w3c/identifier.rb +1 -1
  88. data/lib/pubid/xsf/identifier.rb +1 -1
  89. data/lib/pubid.rb +1 -0
  90. metadata +36 -2
@@ -79,6 +79,11 @@ module Pubid
79
79
  # Canonical format based on lead party
80
80
  # @return [Symbol] :ieee or :iso
81
81
  def canonical_format
82
+ # A stage-tracked D= designator is IEEE draft notation — its
83
+ # canonical face is the designator spelling (§1.3 spelling 7:
84
+ # "JOINT PNUMBER/D=<STAGE>[:year]"), whatever the lead party.
85
+ return :ieee if ieee_draft.to_s.start_with?("D=")
86
+
82
87
  case lead_party
83
88
  when "IEEE", "AIEE"
84
89
  :ieee
@@ -130,28 +135,39 @@ module Pubid
130
135
  # Publishers (slash-separated)
131
136
  parts << publishers.join("/") if publishers && !publishers.empty?
132
137
 
133
- # ISO stage code (only if this was originally ISO-led)
134
- # For IEEE-led conversions, we skip the stage since we don't have ISO equivalent
135
- if lead_party == "ISO" && (typed_stage || iso_stage)
136
- if typed_stage
137
- parts << typed_stage.to_iso_format
138
- elsif iso_stage
139
- parts << iso_stage
140
- end
138
+ # The stage word is the ISO format's own position convention —
139
+ # it prints whenever the identifier carries one, regardless of
140
+ # lead party (the format face decides, not the arrangement). A
141
+ # stage-tracked D= designator decomposes into its stage word
142
+ # and date ("D=CD.2-2020" → "CD2", 2020); the iteration glues
143
+ # onto the word only for the stages whose printed ISO spellings
144
+ # carry it (CD2, DIS2 — the grammar's multi-digit families);
145
+ # the P project stage is IEEE convention and never prints here.
146
+ # The printed word is always the bare stage ("CD4" prints "CD",
147
+ # "DIS3" prints "DIS" — every iso_stage_spellings_spec
148
+ # expectation); the iteration is draft machinery, not ISO-face
149
+ # identity.
150
+ stage_word = iso_stage.to_s.sub(/\A([A-Z]+?)\d+\z/, '\1')
151
+ stage_word = nil if stage_word.empty?
152
+ draft_year = nil
153
+ if stage_word.nil? &&
154
+ ieee_draft.to_s.match(/\AD=([A-Z]+)(?:\.(\d+[a-z]?))?(?:[-:](\d{4}))?\z/)
155
+ stage_word = Regexp.last_match(1)
156
+ draft_year = Regexp.last_match(3)
141
157
  end
158
+ parts << stage_word if stage_word
142
159
 
143
- # IEEE semantics: P = project (a draft); no P = a standard. The
144
- # P-state is identity-bearing and prints as spelled — it is never
145
- # added or stripped here.
160
+ # The P project marker is identity-bearing and prints on both
161
+ # faces (the standing trademark_leaf_to_s_spec contract:
162
+ # "ISO/IEC/IEEE P26511:2018").
146
163
  code_str = code.to_s
147
164
  code_str += mark unless code_str.empty?
148
165
  parts << code_str if code_str && !code_str.empty?
149
166
 
150
- # Join with space and add year with colon; the joint stage-draft
151
- # clause ("D=WD.5") rides after the year in the ISO-led print.
167
+ # Join with space and add year with colon (the ISO position
168
+ # convention; a decomposed D= date rides here as the year).
152
169
  result = parts.join(" ")
153
- result += ":#{year}" if year
154
- result += "/#{ieee_draft}" if ieee_draft && ieee_draft.start_with?("D=")
170
+ result += ":#{year || draft_year}" if year || draft_year
155
171
  # Only the language/edition marker ("(E)", "(E/F)") prints; a
156
172
  # trailing relationship narrative is metadata, not identity.
157
173
  if parenthetical_content&.match?(%r{\A[A-Z](?:\s*[/&]\s*[A-Z])*\z})
@@ -177,18 +193,38 @@ module Pubid
177
193
  # Mark after the number, before the draft and the year
178
194
  code_str += mark unless code_str.empty?
179
195
 
180
- # Add IEEE draft notation if available (e.g., /D8)
181
- if ieee_draft
196
+ # Add IEEE draft notation if available (e.g., /D8). An ISO stage
197
+ # word renders in IEEE's stage-tracked position as the ordinal-less
198
+ # stage draft (docs/IEEE-DRAFT-STAGES.md §1.3, spelling 7):
199
+ # "IEEE/ISO/IEC CD P42010:2020" → "IEEE/ISO/IEC P42010/D=CD-2020".
200
+ # The iso_stage branch is checked FIRST — the typed_stage registry
201
+ # lookup for a stage word answers a draft-equivalent ordinal ("D2"),
202
+ # which is not the canonical stage-tracked spelling.
203
+ if ieee_draft.to_s.start_with?("D=") && year
204
+ # The ordinal-less stage draft's canonical face (the
205
+ # UpdateCodes rewrite): the publication year colon-joins the
206
+ # code and the designator trails — "P16326:2017/D=WD.5".
207
+ code_str += ":#{year}"
208
+ code_str += "/#{ieee_draft}"
209
+ @designator_carries_year = true
210
+ elsif ieee_draft
182
211
  code_str += "/#{ieee_draft}"
212
+ elsif iso_stage
213
+ # The ordinal-less stage draft: an iteration glued onto the
214
+ # printed word ("CD2") renders in the doctrine's ".iter" slot
215
+ # ("D=CD.2" — docs/IEEE-DRAFT-STAGES.md §1.3).
216
+ stage = iso_stage.match(/\A([A-Z]+?)(\d+)\z/)
217
+ code_str += stage ? "/D=#{stage[1]}.#{stage[2]}" : "/D=#{iso_stage}"
183
218
  elsif typed_stage&.ieee_draft_equivalent
184
219
  code_str += "/#{typed_stage.ieee_draft_equivalent}"
185
220
  end
186
221
 
187
222
  parts << code_str if code_str && !code_str.empty?
188
223
 
189
- # Join with space and add year with dash
224
+ # Join with space and add year with dash — unless the D=
225
+ # designator face already carried it above.
190
226
  result = parts.join(" ")
191
- result += "-#{year}" if year
227
+ result += "-#{year}" if year && !@designator_carries_year
192
228
 
193
229
  result
194
230
  end
@@ -595,7 +595,10 @@ module Pubid
595
595
  # ISO/IEC/IEEE P26511/D8-2018 or ISO/IEEE P1003.1-2008 or IEC/IEEE P62582-1-2011
596
596
  # ALSO handle: IEC/IEEE P60780-323, CDV1 2014 (comma before stage code)
597
597
  # ALSO handle: IEEE/CSA P844.1/293.1/D2 (CSA dual numbering)
598
- (str("ISO/IEC/IEEE") | str("ISO/IEEE") | str("IEC/IEEE") | str("IEEE/CSA")).as(:joint_publishers) >>
598
+ # ALSO handle: IEEE/ISO/IEC P42010/D=CD-2020 — the printed publisher
599
+ # order is the organization's perspective (pubid#469) and parses as
600
+ # printed; it is no longer rewritten to ISO-first.
601
+ (str("ISO/IEC/IEEE") | str("IEEE/ISO/IEC") | str("ISO/IEEE") | str("IEC/IEEE") | str("IEEE/CSA")).as(:joint_publishers) >>
599
602
  space >>
600
603
  # P = project (the document is a draft): identity-bearing, so it
601
604
  # is captured and preserved, never silently consumed.
@@ -605,14 +608,19 @@ module Pubid
605
608
  # CSA dual numbering: /293.1 (second number)
606
609
  (slash >> digits >> (dot >> digits).maybe >> (dash >> digits.as(:draft_version)).maybe).maybe >>
607
610
  (
608
- # Variant 1b: the ordinal-less stage draft "D=CDV[:2020]" -
611
+ # Variant 1b: the ordinal-less stage draft "D=CDV-2020" -
609
612
  # D (draft) = CDV (the IEC stage it drafts). The year rides in
610
613
  # the draft clause (a distinct key, so the builder keeps the
611
- # date inside the designator).
614
+ # date inside the designator). The year follows IEEE format:
615
+ # dash-joined ("D=CD-2020", the ruled canonical); the colon
616
+ # spelling ("D=CDV:2020") stays accepted as the alias it was
617
+ # frozen with. A dash-year is guarded to 19xx/20xx so a
618
+ # YYMM monthcode tail can never be read as a year.
612
619
  (slash >> str("D") >> str("=") >>
613
620
  (str("CDV") | str("FDIS") | str("PWI") | str("WD") |
614
621
  str("NP") | str("DIS") | str("CD")).as(:draft_iso_stage) >>
615
- (str(":") >> year_digits.as(:draft_stage_year)).maybe) |
622
+ ((str(":") | str("-")) >>
623
+ ((str("19") | str("20")) >> digit >> digit).as(:draft_stage_year)).maybe) |
616
624
  # Variant 1: /D8 notation (original), with the compound
617
625
  # both-systems suffix "=DDIS.3" (docs/IEEE-DRAFT-STAGES.md §1.3)
618
626
  (slash >> str("D") >> digits.as(:draft_version) >>
@@ -1468,7 +1476,10 @@ module Pubid
1468
1476
  cleaned
1469
1477
  end
1470
1478
 
1471
- def self.parse(string)
1479
+ # Pre-parse ingestion normalizations (R2): every parse path —
1480
+ # parslet and PG artifact alike — feeds the grammar the same
1481
+ # normalized string.
1482
+ def self.normalize_input(string)
1472
1483
  # Strip .pdf extension if present (Pattern 3: File Extensions)
1473
1484
  cleaned = string.sub(/\.pdf$/i, "")
1474
1485
 
@@ -1822,11 +1833,10 @@ module Pubid
1822
1833
  # Fix 2H: "IEC XXXX First edition YYYY-MM; IEEE NNNN" -> normalize semicolon
1823
1834
  # Already handled by earlier semicolon normalization
1824
1835
 
1825
- # Fix 2I: "IEEE/ISO/IEC PXXX/DIS" -> normalize to "ISO/IEC/IEEE PXXX/DIS"
1826
- cleaned = cleaned.gsub(/^IEEE\/ISO\/IEC\s+(P[\w.-]+)/,
1827
- 'ISO/IEC/IEEE \1')
1828
- cleaned = cleaned.gsub(/^IEEE\/IEC\/ISO\s+(P[\w.-]+)/,
1829
- 'IEC/ISO/IEEE \1')
1836
+ # (Fix 2I removed: "IEEE/ISO/IEC PXXX/…" is no longer rewritten to
1837
+ # ISO-first — the printed publisher order is the organization's
1838
+ # perspective (pubid#469) and the joint P-form rule parses it as
1839
+ # printed.)
1830
1840
 
1831
1841
  # Fix 2J: "IEEE/IEC PXXX D5" -> normalize space to slash before D
1832
1842
  cleaned = cleaned.gsub(/^(IEEE\/IEC P[\w.-]+)\s+D(\d)/, '\1/D\2')
@@ -1925,7 +1935,11 @@ module Pubid
1925
1935
  # Fix 2AF: "IEEE Std 1003.1/2003.l/lNT" -> fix typos
1926
1936
  # .l -> .1 and lNT -> INT handled by existing fixes
1927
1937
 
1928
- new.parse(cleaned)
1938
+ cleaned
1939
+ end
1940
+
1941
+ def self.parse(string)
1942
+ new.parse(normalize_input(string))
1929
1943
  end
1930
1944
  end
1931
1945
  end
@@ -169,9 +169,11 @@ module Pubid
169
169
  # ("D08, September, 2018").
170
170
  if id.draft_obj
171
171
  printed = id.draft_obj.to_s
172
- if id.publisher == "IEC" && id.copublisher == ["IEEE"]
173
- printed = printed.split(", ").first
174
- elsif id.publisher == "IEEE" && id.draft_status.to_s.empty?
172
+ # The draft designator carries its date on every lead
173
+ # (docs/IEEE-DRAFT-STAGES.md §3: "the draft designator with its
174
+ # date" — the IEC/IEEE-led date-drop contradicted the DCD
175
+ # convention and the raw records).
176
+ if id.publisher == "IEEE" && id.draft_status.to_s.empty?
175
177
  # The long comma form is for dated project drafts; an
176
178
  # unapproved-draft render keeps its pinned single-comma form
177
179
  # (pubid#318 idempotence).
@@ -64,7 +64,7 @@ module Pubid
64
64
  raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
65
65
  end
66
66
 
67
- parsed = Parser.parse(identifier)
67
+ parsed = Pubid::Parg::Backend.parse(:ietf, identifier)
68
68
  Builder.build(parsed)
69
69
  end
70
70
  end
@@ -64,7 +64,7 @@ module Pubid
64
64
 
65
65
  # @raise [Pubid::Errors::ParseError] if the string is not a valid ISBN
66
66
  def self.build_identifier(identifier)
67
- parsed = Parser.parse(identifier)
67
+ parsed = Pubid::Parg::Backend.parse(:isbn, identifier)
68
68
  Builder.build(parsed)
69
69
  rescue ArgumentError => e
70
70
  # The Builder validates length and check digit. Surface that as a parse
@@ -304,7 +304,10 @@ module Pubid
304
304
  when :mr_string
305
305
  Pubid::Parsers::MrString.parse(string)
306
306
  else
307
- parsed = Pubid::Iso::Parser.new.parse(string)
307
+ # R1 parser swap: the baked PG artifact is the identifier
308
+ # parser of record; the Builder consumes the same attribute
309
+ # hash it always has.
310
+ parsed = Pubid::Parg::Backend.parse(:iso, string)
308
311
  Pubid::Iso::Builder.new.build(parsed)
309
312
  end
310
313
  end
@@ -27,7 +27,10 @@ module Pubid
27
27
  private
28
28
 
29
29
  def parse_with_builder(string)
30
- parsed = Pubid::Iso.parser.parse(string)
30
+ # R2 ingestion hook: normalized strings reach the model through
31
+ # the same parser of record as Identifier.parse (the baked PG
32
+ # artifact), so every Tier-3 normalization feeds one grammar.
33
+ parsed = Pubid::Parg::Backend.parse(:iso, string)
31
34
  Pubid::Iso.builder.build(parsed)
32
35
  end
33
36
 
@@ -67,3 +67,49 @@ on any call. The base now derives the key from the class name
67
67
  `underscore` is not a dependency. metanorma-itu constructs through this
68
68
  lookup; its flavor-local `pubid_contribution.rb` render override can be
69
69
  deleted once it migrates.
70
+
71
+ ## relaton's query forms — RR, OB sector, publication ids
72
+
73
+ Four forms relaton's `Relaton::Itu::Pubid` parsed and `Pubid::Itu` did not
74
+ (hand-off itu-relaton-query-forms). None of them occurs in the published
75
+ `relaton-data-itu` index — no `series: RR`, no OB row, no six-digit part — and
76
+ a replay of all 24,382 rows and every ITU pass fixture showed 0 changes.
77
+
78
+ - **Radio Regulations** — `Identifiers::RadioRegulations`
79
+ (`pubid:itu:radio-regulations`): `ITU-R RR`, `ITU-R RR (2020)`, and the URL
80
+ spelling `ITU-R RR-2020`, which used to build a *wrong* Recommendation
81
+ (series `RR`, number `2020`) with no error. "RR" is the series; there is no
82
+ code, so `#number` returns the series and `root.number` is `"RR"`. The rule
83
+ ends in `any.absent?`: it sits before `with_series`, and PEG never re-enters
84
+ the alternation, so a partial match on `ITU-R RR.1` must fail inside it.
85
+ - **Operational Bulletins keep their sector.** This reverses the old
86
+ "cross-bureau, sector must not be set" rule: `validate_ob_no_sector!` is
87
+ gone, `ITU-T OB.1096 (2016)` renders back as it is, and the sector-less
88
+ `ITU OB No. 1096` is unchanged. The long forms with a sector (`ITU-T OB No.
89
+ 1096`) now normalise to `ITU-T OB.1096`. `No.` stays in the default
90
+ render — it is how ITU's bulletin site and pubid v1 write it — and
91
+ metanorma-itu's `ITU OB 1000` / `Annex to ITU OB 1000` (its i18n template
92
+ omits `No.`) is accepted on parse (`ob_bare_body`) and rendered with it.
93
+ The sector is a spelling, not
94
+ identity: `SpecialPublication#==` skips it, the URN keeps `urn:itu:itu:…`
95
+ and `mr_type` stays nil, so both spellings are one bulletin on every
96
+ surface except `to_s`/`to_hash`. The printed date `- 15.III.2016` sets
97
+ `date.day`, which only this form does, so `render_ob_date` uses the day as
98
+ the spelling marker; `day_to_kv` emits only then.
99
+ - **`-YYYYMM` is a date, never a part.** `part` refuses a six-digit run with a
100
+ 19xx/20xx year and a 01–12 month (`yyyymm_shape`), and `id_date` reads it as
101
+ year+month. `-200313`, `-180001` stay parts. `ITU-T REC T.4` drops the
102
+ uncaptured `rec_word`; `T-REC-T.4-200307-I` is `publication_id`, last in
103
+ `identifier` (nothing else starts with a bare sector letter), and requires
104
+ the date. The trailing status letter (`I` in force, `S` superseded) is
105
+ parsed and dropped: it names the state of an edition, not the edition.
106
+ **`S` is also the Spanish language suffix**, so it is a status only inside
107
+ the full `T-REC-…` id (`id_status`); after an `ITU-T …-YYYYMM` print form
108
+ only `-I` is (`print_id_status`), and `ITU-T Z.100-199911-S` keeps language
109
+ `S`. The `-YYYYMM` date and `REC` word are also in `base_with_series`/
110
+ `base_without_series`: the part guard applies there too, so without them
111
+ `ITU-T G.989-200307 Amd 1` — a (wrong) Amendment on `main` — stopped parsing.
112
+ The day of an OB date reaches the URN (`…:15/03/2016`), since it is in `==`.
113
+ **Not done:** an RR supplement (`ITU-R RR (2020) Amd 1` fails; the
114
+ base-less `ITU-R RR Amd 1` still builds an Amendment on series `RR`), and
115
+ `ITU-R RR-E` is still a Recommendation numbered `E` — both as on `main`.
@@ -50,6 +50,16 @@ module Pubid
50
50
  return sp
51
51
  end
52
52
 
53
+ # Radio Regulations — "ITU-R RR (2020)"
54
+ if data[:radio_regulations]
55
+ return Identifiers::RadioRegulations.new(
56
+ sector: Components::Sector.new(sector: data[:sector].to_s),
57
+ series: Components::Series.new(series: "RR"),
58
+ date: data[:year] ? build_date(data) : nil,
59
+ language: data[:language]&.to_s,
60
+ )
61
+ end
62
+
53
63
  # Check if this is a supplement identifier
54
64
  if data[:supplement_type]
55
65
  supp = build_supplement(data)
@@ -153,11 +163,12 @@ module Pubid
153
163
  nil
154
164
  end
155
165
 
156
- # Build Special Publication (OB). Sector is silently dropped — OB is a
157
- # cross-bureau publication and `Identifier` rejects sector+OB
158
- # in its constructor.
166
+ # Build Special Publication (OB). The sector of the TSB spelling
167
+ # ("ITU-T OB.1096") is kept so the bulletin renders back as it was cited;
168
+ # SpecialPublication#== ignores it, since OB is cross-bureau.
159
169
  def build_special_publication(data)
160
170
  Identifiers::SpecialPublication.new(
171
+ sector: (Components::Sector.new(sector: data[:sector].to_s) if data[:sector]),
161
172
  series: Components::Series.new(series: "OB"),
162
173
  code: data[:number] ? build_code(data) : nil,
163
174
  date: data[:year] ? build_date(data) : nil,
@@ -344,10 +355,19 @@ module Pubid
344
355
  )
345
356
  end
346
357
 
358
+ # The Roman month of a bulletin date ("15.III.2016") is stored as the
359
+ # two-digit month every other ITU date uses; the day marks the spelling.
347
360
  def build_date(data)
361
+ month = if data[:roman_month]
362
+ (Identifiers::SpecialPublication::ROMAN_MONTHS.index(data[:roman_month].to_s) + 1)
363
+ .to_s.rjust(2, "0")
364
+ else
365
+ data[:month]&.to_s
366
+ end
348
367
  Pubid::Components::Date.new(
349
368
  year: data[:year].to_s,
350
- month: data[:month]&.to_s,
369
+ month: month,
370
+ day: data[:day]&.to_s,
351
371
  )
352
372
  end
353
373
 
@@ -15,7 +15,7 @@ module Pubid
15
15
  raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
16
16
  end
17
17
 
18
- parsed = Parser.parse(normalize_whitespace(identifier))
18
+ parsed = Pubid::Parg::Backend.parse(:itu, normalize_whitespace(identifier))
19
19
  Builder.build(parsed)
20
20
  end
21
21
 
@@ -101,8 +101,6 @@ module Pubid
101
101
  end
102
102
 
103
103
  super
104
-
105
- validate_ob_no_sector!
106
104
  end
107
105
 
108
106
  # The document number lives on the `code` component for ITU; surface it at
@@ -462,6 +460,16 @@ module Pubid
462
460
  model.code ||= Components::Code.new
463
461
  end
464
462
 
463
+ # The day is set only by the printed bulletin date ("15.III.2016"), so
464
+ # it emits only there and no existing index row gains a key.
465
+ def day_to_kv(model, doc)
466
+ emit_kv(doc, "day", model.date&.day)
467
+ end
468
+
469
+ def day_from_kv(model, value)
470
+ date_for(model).day = value.to_s
471
+ end
472
+
465
473
  def date_for(model)
466
474
  model.date ||= Pubid::Components::Date.new
467
475
  end
@@ -472,21 +480,6 @@ module Pubid
472
480
  str = value.to_s
473
481
  LANGUAGES[str] || str
474
482
  end
475
-
476
- # OB (Operational Bulletin) is a cross-bureau ITU publication and
477
- # must not have a sector. Direct construction with both raises;
478
- # the parser silently drops sector for legacy strings like
479
- # "ITU-T OB.X" (handled in Builder).
480
- def validate_ob_no_sector!
481
- return unless series&.series == "OB"
482
- return if sector.nil?
483
- return if sector.is_a?(Components::Sector) && (sector.sector.nil? || sector.sector.to_s.empty?)
484
-
485
- raise ArgumentError,
486
- "OB (Operational Bulletin) is a cross-bureau ITU publication; " \
487
- "sector must not be set"
488
- end
489
483
  end
490
-
491
484
  end
492
485
  end
@@ -0,0 +1,27 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Pubid
4
+ module Itu
5
+ module Identifiers
6
+ # The ITU Radio Regulations — the treaty text revised by each World
7
+ # Radiocommunication Conference.
8
+ # Format: ITU-R RR [(YYYY)]
9
+ # Example: ITU-R RR (2020)
10
+ #
11
+ # It has no document number: "RR" is the whole designation, stored as
12
+ # the series. `#number` returns it, so `root.number` (the relaton-index
13
+ # key) is not empty.
14
+ class RadioRegulations < Identifier
15
+ include StandardSerialization
16
+
17
+ def number
18
+ series&.series
19
+ end
20
+
21
+ def render_base(**_opts)
22
+ "#{publisher}-#{sector} #{series}#{render_date_suffix}"
23
+ end
24
+ end
25
+ end
26
+ end
27
+ end
@@ -4,26 +4,60 @@ module Pubid
4
4
  module Itu
5
5
  module Identifiers
6
6
  # ITU Special Publication — currently models the Operational Bulletin (OB).
7
- # OB is a cross-bureau publication (no sector) rendered as
8
- # "ITU OB No. {number}" with optional date.
9
- #
10
- # Pattern: "ITU OB No. 1283 (01/2024)"
7
+ # OB is a cross-bureau publication. ITU prints it without a sector
8
+ # ("ITU OB No. 1283 (01/2024)"); the TSB spelling carries one
9
+ # ("ITU-T OB.1096 (2016)"), which is kept and rendered back. The sector
10
+ # is not part of the identity: `==`, the URN and the MR slug ignore it,
11
+ # so both spellings name one bulletin.
11
12
  class SpecialPublication < Identifier
12
13
  include StandardSerialization
13
14
 
15
+ ROMAN_MONTHS = %w[I II III IV V VI VII VIII IX X XI XII].freeze
16
+
14
17
  def render_base(**_opts)
15
18
  number = code&.number
16
- result = "#{publisher} #{series} No. #{number}"
17
-
18
- if date
19
- result += if date.month
20
- " (#{date.month.to_s.rjust(2, '0')}/#{date.year})"
21
- else
22
- " (#{date.year})"
23
- end
24
- end
19
+ result = if sector
20
+ "#{publisher}-#{sector} #{series}.#{number}"
21
+ else
22
+ "#{publisher} #{series} No. #{number}"
23
+ end
24
+
25
+ result + render_ob_date
26
+ end
27
+
28
+ # Cross-bureau: the sector is how one bureau cites the bulletin, not
29
+ # which bulletin it is.
30
+ def ==(other)
31
+ return false unless other.instance_of?(self.class)
32
+
33
+ series == other.series &&
34
+ code == other.code &&
35
+ date == other.date &&
36
+ language == other.language &&
37
+ common_text_twin == other.common_text_twin
38
+ end
39
+
40
+ # Keep the MR slug sector-free, like `==` (it was always so, because
41
+ # the sector used to be dropped).
42
+ def mr_type
43
+ nil
44
+ end
25
45
 
26
- result
46
+ private
47
+
48
+ # " (MM/YYYY)" or " (YYYY)"; the day-bearing date is the printed
49
+ # bulletin form " - 15.III.2016", the only spelling that sets a day.
50
+ def render_ob_date
51
+ return "" unless date
52
+
53
+ if date.day && date.month
54
+ roman = ROMAN_MONTHS[date.month.to_i - 1]
55
+ " - #{date.day.to_s.rjust(2, '0')}.#{roman}.#{date.year}"
56
+ elsif date.month
57
+ " (#{date.month.to_s.rjust(2, '0')}/#{date.year})"
58
+ else
59
+ " (#{date.year})"
60
+ end
27
61
  end
28
62
  end
29
63
  end
@@ -46,6 +46,8 @@ module Pubid
46
46
  with: { to: :year_to_kv, from: :year_from_kv }
47
47
  map "month",
48
48
  with: { to: :month_to_kv, from: :month_from_kv }
49
+ map "day",
50
+ with: { to: :day_to_kv, from: :day_from_kv }
49
51
  map "language", to: :language
50
52
  map "common_text_twin",
51
53
  with: { to: :common_text_twin_to_kv,
@@ -16,6 +16,7 @@ module Pubid
16
16
  autoload :Errata, "#{__dir__}/identifiers/errata"
17
17
  autoload :Handbook, "#{__dir__}/identifiers/handbook"
18
18
  autoload :Question, "#{__dir__}/identifiers/question"
19
+ autoload :RadioRegulations, "#{__dir__}/identifiers/radio_regulations"
19
20
  autoload :Recommendation, "#{__dir__}/identifiers/recommendation"
20
21
  autoload :Report, "#{__dir__}/identifiers/report"
21
22
  autoload :SpecialPublication, "#{__dir__}/identifiers/special_publication"