pubid 2.0.0.pre.alpha.13 → 2.0.0.pre.alpha.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. checksums.yaml +4 -4
  2. data/data/parg/tables/bipm_groups.yaml +14 -0
  3. data/data/parg/tables/bipm_type_codes.yaml +5 -0
  4. data/data/parg/tables/bipm_type_names_en.yaml +6 -0
  5. data/data/parg/tables/bipm_type_names_fr.yaml +6 -0
  6. data/data/parg/tables/directives_supplements_typed_stages.yaml +3 -0
  7. data/data/parg/tables/directives_typed_stages.yaml +5 -0
  8. data/data/parg/tables/idf_typed_stages.yaml +27 -0
  9. data/data/parg/tables/idf_typed_stages_supplements.yaml +2 -0
  10. data/data/parg/tables/iec_typed_stages.yaml +130 -0
  11. data/data/parg/tables/iso_publishers.yaml +4 -0
  12. data/data/parg/tables/organizations.yaml +12 -0
  13. data/data/parg/tables/tc_types.yaml +42 -0
  14. data/data/parg/tables/typed_stages.yaml +114 -0
  15. data/data/parg/tables/typed_stages_supplements.yaml +64 -0
  16. data/data/parg/tables/wg_types.yaml +21 -0
  17. data/lib/pubid/adobe/identifier.rb +11 -1
  18. data/lib/pubid/amca/identifiers/base.rb +1 -1
  19. data/lib/pubid/ansi/identifier.rb +1 -1
  20. data/lib/pubid/api/identifier.rb +1 -1
  21. data/lib/pubid/api/parser.rb +8 -4
  22. data/lib/pubid/ashrae/identifiers/base.rb +10 -1
  23. data/lib/pubid/ashrae/parser.rb +18 -11
  24. data/lib/pubid/asme/identifier.rb +1 -1
  25. data/lib/pubid/astm/identifier.rb +1 -1
  26. data/lib/pubid/bipm/identifier.rb +1 -1
  27. data/lib/pubid/bsi/single_identifier.rb +1 -3
  28. data/lib/pubid/calconnect/identifier.rb +1 -1
  29. data/lib/pubid/ccsds/identifier.rb +1 -1
  30. data/lib/pubid/cen_cenelec/identifier.rb +2 -1
  31. data/lib/pubid/cen_cenelec/parser.rb +9 -2
  32. data/lib/pubid/cie/identifier.rb +1 -1
  33. data/lib/pubid/cie/parser.rb +9 -2
  34. data/lib/pubid/conformance/checks.rb +1 -1
  35. data/lib/pubid/csa/identifier.rb +6 -2
  36. data/lib/pubid/csa/parser.rb +25 -8
  37. data/lib/pubid/doi/identifier.rb +1 -1
  38. data/lib/pubid/easc/identifier.rb +10 -1
  39. data/lib/pubid/ecma/identifier.rb +1 -1
  40. data/lib/pubid/etsi/identifiers/base.rb +1 -1
  41. data/lib/pubid/evs.rb +1 -1
  42. data/lib/pubid/gb/identifier.rb +1 -1
  43. data/lib/pubid/gost/identifier.rb +11 -1
  44. data/lib/pubid/gost/parser.rb +8 -1
  45. data/lib/pubid/iala/identifier.rb +10 -1
  46. data/lib/pubid/iana/identifier.rb +1 -1
  47. data/lib/pubid/idf/builder.rb +5 -0
  48. data/lib/pubid/iec/identifier.rb +1 -1
  49. data/lib/pubid/iec/parser.rb +9 -4
  50. data/lib/pubid/ieee/builder.rb +61 -23
  51. data/lib/pubid/ieee/identifiers/base.rb +1 -1
  52. data/lib/pubid/ieee/identifiers/joint_development.rb +55 -19
  53. data/lib/pubid/ieee/parser.rb +25 -11
  54. data/lib/pubid/ieee/renderer.rb +5 -3
  55. data/lib/pubid/ietf/identifiers/base.rb +1 -1
  56. data/lib/pubid/isbn/identifier.rb +1 -1
  57. data/lib/pubid/iso/identifier.rb +4 -1
  58. data/lib/pubid/iso/normalizer.rb +4 -1
  59. data/lib/pubid/itu/CLAUDE.md +46 -0
  60. data/lib/pubid/itu/builder.rb +24 -4
  61. data/lib/pubid/itu/identifiers/base.rb +11 -18
  62. data/lib/pubid/itu/identifiers/radio_regulations.rb +27 -0
  63. data/lib/pubid/itu/identifiers/special_publication.rb +48 -14
  64. data/lib/pubid/itu/identifiers/standard_serialization.rb +2 -0
  65. data/lib/pubid/itu/identifiers.rb +1 -0
  66. data/lib/pubid/itu/parser.rb +108 -22
  67. data/lib/pubid/itu/urn_generator.rb +9 -2
  68. data/lib/pubid/jcgm.rb +1 -1
  69. data/lib/pubid/jis/identifier.rb +1 -1
  70. data/lib/pubid/nist/builder.rb +1 -0
  71. data/lib/pubid/nist/identifiers/base.rb +15 -4
  72. data/lib/pubid/nist/parser.rb +9 -0
  73. data/lib/pubid/nist/urn_parser.rb +10 -1
  74. data/lib/pubid/oasis/identifier.rb +1 -1
  75. data/lib/pubid/ogc/identifier.rb +1 -1
  76. data/lib/pubid/oiml.rb +1 -1
  77. data/lib/pubid/omg/identifier.rb +1 -1
  78. data/lib/pubid/parg/artifact.rb +46 -0
  79. data/lib/pubid/parg/backend.rb +92 -0
  80. data/lib/pubid/parg.rb +8 -0
  81. data/lib/pubid/pg.rb +8 -0
  82. data/lib/pubid/plateau.rb +1 -2
  83. data/lib/pubid/sae/identifiers/base.rb +1 -1
  84. data/lib/pubid/tgpp/identifier.rb +1 -1
  85. data/lib/pubid/un/identifier.rb +1 -1
  86. data/lib/pubid/version.rb +1 -1
  87. data/lib/pubid/w3c/identifier.rb +1 -1
  88. data/lib/pubid/xsf/identifier.rb +1 -1
  89. data/lib/pubid.rb +1 -0
  90. metadata +36 -2
@@ -77,7 +77,7 @@ module Pubid
77
77
 
78
78
  # Apply legacy update_codes normalization first
79
79
  normalized = Core::UpdateCodes.apply(identifier, :ccsds)
80
- parsed = Pubid::Ccsds::Parser.parse(normalized)
80
+ parsed = Pubid::Parg::Backend.parse(:ccsds, normalized)
81
81
  Pubid::Ccsds::Builder.build(parsed)
82
82
  end
83
83
 
@@ -102,7 +102,8 @@ module Pubid
102
102
  raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
103
103
  end
104
104
 
105
- parsed = Parser.parse(identifier)
105
+ normalized = Parser.normalize_input(identifier)
106
+ parsed = Pubid::Parg::Backend.parse(:cen_cenelec, normalized)
106
107
  Builder.new.build(parsed)
107
108
  end
108
109
 
@@ -131,7 +131,10 @@ module Pubid
131
131
 
132
132
  rule(:root) { identifier }
133
133
 
134
- def self.parse(input)
134
+ # Pre-parse ingestion normalizations (R2): every parse path —
135
+ # parslet and PG artifact alike — feeds the grammar the same
136
+ # normalized string.
137
+ def self.normalize_input(input)
135
138
  # Normalize special dash characters
136
139
  normalized = input.gsub(/[\u2011\u00AD]/, "-")
137
140
 
@@ -145,7 +148,11 @@ module Pubid
145
148
  normalized = normalized.gsub("CEN-CLC", "CEN/CLC")
146
149
  .gsub("CLC-CEN", "CLC/CEN")
147
150
  .gsub("GUIDE", "Guide")
148
- new.parse(normalized)
151
+ normalized
152
+ end
153
+
154
+ def self.parse(input)
155
+ new.parse(normalize_input(input))
149
156
  end
150
157
  end
151
158
  end
@@ -24,7 +24,7 @@ module Pubid
24
24
  raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
25
25
  end
26
26
 
27
- parsed = Parser.parse(input)
27
+ parsed = Pubid::Parg::Backend.parse(:cie, Parser.normalize_input(input))
28
28
  builder = Builder.new
29
29
  builder.build(parsed, original_string: input)
30
30
  end
@@ -375,7 +375,10 @@ module Pubid
375
375
  end
376
376
 
377
377
  # Class method for parsing with preprocessing
378
- def self.parse(string)
378
+ # Pre-parse ingestion normalizations (R2): every parse path —
379
+ # parslet and PG artifact alike — feeds the grammar the same
380
+ # normalized string.
381
+ def self.normalize_input(string)
379
382
  # Minimal preprocessing for data quality
380
383
  cleaned = string.strip
381
384
 
@@ -389,7 +392,11 @@ module Pubid
389
392
  # This is a data quality fix - correct format always has colon
390
393
  cleaned = cleaned.gsub(%r{/(E|F|G|DE|ES|CN|RU|FR)(\d{4})}, '/\1:\2')
391
394
 
392
- new.parse(cleaned)
395
+ cleaned
396
+ end
397
+
398
+ def self.parse(string)
399
+ new.parse(normalize_input(string))
393
400
  end
394
401
  end
395
402
  end
@@ -18,7 +18,7 @@ module Pubid
18
18
  mismatchers(identifier, test_case, flavor_module)
19
19
  .flat_map(&:call)
20
20
  rescue StandardError => e
21
- ["#{test_case.id} raised #{e.class}"]
21
+ ["#{test_case.id} raised #{e.class}: #{e.message}"]
22
22
  end
23
23
 
24
24
  # Each checker returns the case's mismatch list for its concern.
@@ -61,7 +61,9 @@ module Pubid
61
61
  normalized = normalized.gsub(/\s+/, " ").strip
62
62
 
63
63
  # Parse normally (will create Bundled or Combined identifier)
64
- tree = Parser.new.parse(normalized)
64
+ tree = Pubid::Parg::Backend.parse(:csa, Parser.normalize_input(normalized))
65
+ prefix = Parser.publisher_prefix_for(normalized)
66
+ Parser.inject_publisher_prefix(tree, prefix) if prefix && tree.is_a?(Hash)
65
67
  result = build!(tree, input)
66
68
 
67
69
  # Apply CAN/CSA- prefix to the appropriate parts
@@ -419,7 +421,9 @@ module Pubid
419
421
  # Normalize CAN3- to CSA (historical prefix)
420
422
  normalized = normalized.gsub("CAN3-", "CSA ")
421
423
 
422
- tree = Parser.new.parse(normalized)
424
+ tree = Pubid::Parg::Backend.parse(:csa, Parser.normalize_input(normalized))
425
+ prefix = Parser.publisher_prefix_for(normalized)
426
+ Parser.inject_publisher_prefix(tree, prefix) if prefix && tree.is_a?(Hash)
423
427
  result = build!(tree, input)
424
428
 
425
429
  # Set publisher prefix if detected
@@ -366,13 +366,28 @@ module Pubid
366
366
 
367
367
  # Preprocessing to normalize input
368
368
  def parse(input)
369
+ prefix = self.class.publisher_prefix_for(input)
370
+ result = super(self.class.normalize_input(input))
371
+ if prefix && result.is_a?(Hash)
372
+ self.class.inject_publisher_prefix(result, prefix)
373
+ end
374
+ result
375
+ end
376
+
377
+ # Pre-parse ingestion normalizations (R2): every parse path —
378
+ # parslet and PG artifact alike — feeds the grammar the same
379
+ # normalized string.
380
+ def self.normalize_input(input)
369
381
  # Skip comment lines
370
382
  if input.strip.start_with?("#")
371
383
  raise Pubid::Errors::ParseError.new(
372
384
  "Comment line", input: input, flavor: "csa"
373
385
  )
374
386
  end
387
+ __normalize_input(input)
388
+ end
375
389
 
390
+ def self.__normalize_input(input)
376
391
  # Remove CONSOLIDATED notation FIRST (before other processing)
377
392
  normalized = input.gsub(/\s*\(\s*CONSOLIDATED\s*\)\s*/, " ")
378
393
  normalized = normalized.gsub(/\s*\bCONSOLIDATED\b\s*/, " ")
@@ -404,20 +419,22 @@ module Pubid
404
419
  # Clean up extra spaces
405
420
  normalized = normalized.gsub(/\s+/, " ").strip
406
421
 
407
- # Parse and inject publisher_prefix into result
408
- result = super(normalized)
422
+ normalized
423
+ end
409
424
 
410
- # Inject publisher_prefix if we have one
411
- if publisher_prefix && result.is_a?(Hash)
412
- inject_publisher_prefix(result, publisher_prefix)
425
+ def self.publisher_prefix_for(input)
426
+ if input.start_with?("CAN/CSA-")
427
+ "CAN/CSA-"
428
+ elsif input.start_with?("CAN3-")
429
+ "CAN3-"
430
+ elsif input.start_with?("CSA ")
431
+ "CSA"
413
432
  end
414
-
415
- result
416
433
  end
417
434
 
418
435
  private
419
436
 
420
- def inject_publisher_prefix(hash, prefix)
437
+ def self.inject_publisher_prefix(hash, prefix)
421
438
  # For combined identifiers
422
439
  if hash[:first]
423
440
  hash[:first][:publisher_prefix] = prefix
@@ -41,7 +41,7 @@ module Pubid
41
41
  raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
42
42
  end
43
43
 
44
- parsed = Parser.parse(identifier)
44
+ parsed = Pubid::Parg::Backend.parse(:doi, identifier)
45
45
  Builder.build(parsed)
46
46
  end
47
47
  end
@@ -28,7 +28,16 @@ module Pubid
28
28
  # РМГ 29-2013
29
29
  class Identifier < ::Pubid::Identifier
30
30
  def self.parse(identifier)
31
- parsed = Parser.parse(identifier)
31
+ unless identifier.is_a?(String)
32
+ raise Pubid::Errors::InvalidInputError,
33
+ Pubid::INPUT_NOT_A_STRING_MESSAGE
34
+ end
35
+
36
+ if identifier.length > Pubid::MAX_INPUT_LENGTH
37
+ raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
38
+ end
39
+
40
+ parsed = Pubid::Parg::Backend.parse(:easc, identifier)
32
41
  Builder.build(parsed)
33
42
  end
34
43
 
@@ -129,7 +129,7 @@ module Pubid
129
129
  raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
130
130
  end
131
131
 
132
- parsed = Parser.parse(identifier)
132
+ parsed = Pubid::Parg::Backend.parse(:ecma, identifier)
133
133
  Builder.build(parsed)
134
134
  end
135
135
 
@@ -18,7 +18,7 @@ module Pubid
18
18
  raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
19
19
  end
20
20
 
21
- parsed = Parser.parse(identifier)
21
+ parsed = Pubid::Parg::Backend.parse(:etsi, identifier)
22
22
  Builder.build(parsed)
23
23
  end
24
24
 
data/lib/pubid/evs.rb CHANGED
@@ -33,7 +33,7 @@ module Pubid
33
33
  raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
34
34
  end
35
35
 
36
- parsed = Parser.new.parse(identifier)
36
+ parsed = Pubid::Parg::Backend.parse(:evs, identifier)
37
37
  Builder.new.build(parsed)
38
38
  end
39
39
 
@@ -80,7 +80,7 @@ module Pubid
80
80
  raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
81
81
  end
82
82
 
83
- parsed = Parser.parse(identifier)
83
+ parsed = Pubid::Parg::Backend.parse(:gb, identifier)
84
84
  Builder.build(parsed)
85
85
  end
86
86
  end
@@ -13,7 +13,17 @@ module Pubid
13
13
  # Harmonized). A bare InterstateStandard/NationalStandard has neither.
14
14
  class Identifier < ::Pubid::Identifier
15
15
  def self.parse(identifier)
16
- parsed = Parser.parse(identifier)
16
+ unless identifier.is_a?(String)
17
+ raise ::Pubid::Errors::InvalidInputError,
18
+ ::Pubid::INPUT_NOT_A_STRING_MESSAGE
19
+ end
20
+
21
+ if identifier.length > ::Pubid::MAX_INPUT_LENGTH
22
+ raise ::Pubid::Errors::InvalidInputError,
23
+ ::Pubid::INPUT_TOO_LONG_MESSAGE
24
+ end
25
+
26
+ parsed = Pubid::Parg::Backend.parse(:gost, Parser.normalize_input(identifier))
17
27
  Builder.build(parsed)
18
28
  end
19
29
 
@@ -87,6 +87,13 @@ module Pubid
87
87
  adopted_part.maybe >> adopted_reference_part.maybe
88
88
  end
89
89
 
90
+ # Pre-parse ingestion normalizations (R2): every parse path —
91
+ # parslet and PG artifact alike — feeds the grammar the same
92
+ # normalized string.
93
+ def self.normalize_input(string)
94
+ string.strip
95
+ end
96
+
90
97
  def self.parse(string)
91
98
  unless string.is_a?(String)
92
99
  raise ::Pubid::Errors::InvalidInputError,
@@ -98,7 +105,7 @@ module Pubid
98
105
  ::Pubid::INPUT_TOO_LONG_MESSAGE
99
106
  end
100
107
 
101
- new.parse(string.strip)
108
+ new.parse(normalize_input(string))
102
109
  end
103
110
  end
104
111
  end
@@ -21,10 +21,19 @@ module Pubid
21
21
  # @return [Pubid::Iala::Identifier]
22
22
  # @raise [Pubid::Errors::ParseError] If parsing fails
23
23
  def self.parse(identifier)
24
+ unless identifier.is_a?(String)
25
+ raise Pubid::Errors::InvalidInputError,
26
+ Pubid::INPUT_NOT_A_STRING_MESSAGE
27
+ end
28
+
29
+ if identifier.length > Pubid::MAX_INPUT_LENGTH
30
+ raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
31
+ end
32
+
24
33
  if FormatDetector.detect(identifier) == :urn
25
34
  UrnParser.parse(identifier)
26
35
  else
27
- parsed = Parser.parse(identifier)
36
+ parsed = Pubid::Parg::Backend.parse(:iala, identifier)
28
37
  Builder.build(parsed)
29
38
  end
30
39
  end
@@ -132,7 +132,7 @@ module Pubid
132
132
  raise Pubid::Errors::InvalidInputError, Pubid::INPUT_TOO_LONG_MESSAGE
133
133
  end
134
134
 
135
- parsed = Parser.parse(identifier)
135
+ parsed = Pubid::Parg::Backend.parse(:iana, identifier)
136
136
  Builder.build(parsed)
137
137
  end
138
138
  end
@@ -22,6 +22,11 @@ module Pubid
22
22
  end
23
23
 
24
24
  assign_attributes(identifier, parsed_hash)
25
+ # The locate-stage pass above selects the class; the typed_stage
26
+ # attribute itself is set by the :type_with_stage key. Subtrees
27
+ # built without that key (e.g. a joint identifier from a PG
28
+ # artifact shape) would render with a nil typed_stage otherwise.
29
+ identifier.typed_stage ||= typed_stage if typed_stage
25
30
  # "(all parts)" names every part of the document, so it wraps the
26
31
  # document, which holds no mark itself.
27
32
  all_parts ? identifier.to_all_parts : identifier
@@ -323,7 +323,7 @@ module Pubid
323
323
 
324
324
  # Apply legacy update_codes normalization first, before any other preprocessing
325
325
  normalized = Core::UpdateCodes.apply(string, :iec)
326
- parsed = Pubid::Iec::Parser.new.parse(normalized)
326
+ parsed = Pubid::Parg::Backend.parse(:iec, Pubid::Iec::Parser.normalize_input(normalized))
327
327
  Pubid::Iec::Builder.new.build(parsed)
328
328
  end
329
329
 
@@ -428,15 +428,20 @@ module Pubid
428
428
 
429
429
  IEV_SHORTHAND = /\AIEV(?=\z|[\s-])/
430
430
 
431
- # Preprocess input to normalize tab-separated editions and other formats
432
- def parse(input)
431
+ # Pre-parse ingestion normalizations (R2): every parse path —
432
+ # parslet and PG artifact alike — feeds the grammar the same
433
+ # normalized string.
434
+ def self.normalize_input(input)
433
435
  # Normalize tab-separated editions: "IECEE AD-001\tED1.6" -> "IECEE AD-001 ED1.6"
434
436
  normalized = input.gsub("\t", " ")
435
437
  # Normalize comma-separated editions: "IEC CAB-G01:2025-02, Ed. 2.1" -> "IEC CAB-G01:2025-02 Ed. 2.1"
436
438
  normalized = normalized.gsub(/,\s+Ed\./, " Ed.")
437
439
  # Expand IEV shorthand: "IEV" / "IEV-351" -> "IEC 60050" / "IEC 60050-351"
438
- normalized = normalized.sub(IEV_SHORTHAND, "IEC 60050")
439
- super(normalized)
440
+ normalized.sub(IEV_SHORTHAND, "IEC 60050")
441
+ end
442
+
443
+ def parse(input)
444
+ super(self.class.normalize_input(input))
440
445
  end
441
446
  end
442
447
  end
@@ -88,6 +88,14 @@ module Pubid
88
88
  date_info = nil
89
89
 
90
90
  if content
91
+ # A bare (unparenthesised) revision narrative printed after the
92
+ # code is the same relationship prose as the parenthetical form
93
+ # (C3 ruling) — peel it before the number tokenisation so both
94
+ # spellings collapse onto one canonical.
95
+ if (m = content.match(/\A(.+?) (Revis(?:ion|on) (?:of|to) IEEE Std .+)\z/))
96
+ content = "#{m[1]} (#{m[2]})"
97
+ end
98
+
91
99
  # Extract copublished number (everything before IEC: or comma or parenthesis)
92
100
  copublished_number = if content.include?("IEC:")
93
101
  content.split(" IEC:").first.strip
@@ -114,21 +122,42 @@ module Pubid
114
122
  end
115
123
  end
116
124
 
117
- # Extract date info if present
125
+ # The parenthetical tail is classified, not swallowed (C3 ruling:
126
+ # a "(Revision of IEEE Std …)" narrative is a relationship, not
127
+ # identity — it lands in `relationships` so the URN carries it as
128
+ # the `rel.` segment and the canonical human drops it, exactly the
129
+ # joint ISO route's model; an "(MM/DD)" print date is non-identity
130
+ # and dropped; anything else keeps the historical date_info
131
+ # rendering).
118
132
  if content.include?(" (")
119
- date_part = content.split(" (")[1]
120
- if date_part&.include?(")")
121
- date_info = date_part.split(")")[0]
133
+ tail = content.split(" (")[1].to_s.split(")")[0]
134
+ if (m = tail.match(/\ARevis(?:ion|on) (?:of|to) IEEE Std (.+)\z/))
135
+ related =
136
+ begin
137
+ Identifier.parse(m[1])
138
+ rescue Parslet::ParseFailed
139
+ Identifier.new(parenthetical_content: m[1])
140
+ end
141
+ relationships = [Components::Relationship.new(
142
+ relationship_type: Components::Relationship::REVISION_OF,
143
+ related_identifiers: [related],
144
+ )]
145
+ elsif tail.match?(/\A\d{2}\/\d{2}\z/)
146
+ # print date — non-identity, dropped
147
+ else
148
+ date_info = tail
122
149
  end
123
150
  end
124
151
  end
125
152
 
126
- Identifiers::IecIeeeCopublished.new(
153
+ kwargs = {
127
154
  draft_info: draft_info,
128
155
  iec_year: iec_year,
129
156
  date_info: date_info,
130
157
  **copublished_structured(copublished_number),
131
- )
158
+ }
159
+ kwargs[:relationships] = relationships if relationships
160
+ Identifiers::IecIeeeCopublished.new(**kwargs)
132
161
  end
133
162
 
134
163
  # Decompose an IEC/IEEE `copublished_number` into the split index columns
@@ -372,12 +401,14 @@ module Pubid
372
401
  # Build the complete code string: "802.1AC-2016" or "535-2013" or "C37.41-2016"
373
402
  number_str = extract_value(base_data[:number])
374
403
 
375
- # Add part if present (e.g., ".1AC" or ".41")
404
+ # Part and subpart captures carry their own separator (".15",
405
+ # "-10417", ".4j" — the grammar mirrors the original parser's
406
+ # `((dot | dash) >> words_digits).as(:part)`), so they join the
407
+ # code string verbatim; re-adding a separator here would double
408
+ # it ("802..15").
376
409
  if base_data[:part]
377
410
  part_val = extract_value(base_data[:part])
378
- # Determine separator: dot for most cases, dash for some
379
- separator = number_str.match?(/^[A-Z]/) ? "." : "." # Letter prefix uses dot
380
- number_str += separator + part_val
411
+ number_str += part_val if part_val
381
412
  end
382
413
 
383
414
  # Add subpart if present
@@ -386,7 +417,7 @@ module Pubid
386
417
  subparts = [subparts] unless subparts.is_a?(Array)
387
418
  subparts.each do |sp|
388
419
  subpart_val = extract_value(sp)
389
- number_str += ".#{subpart_val}" if subpart_val
420
+ number_str += subpart_val if subpart_val
390
421
  end
391
422
  end
392
423
 
@@ -545,11 +576,10 @@ module Pubid
545
576
  # Extract number with parts and year
546
577
  number_str = extract_value(base_data[:number])
547
578
 
548
- # Add part if present
579
+ # Part/subpart captures carry their own separator — join verbatim.
549
580
  if base_data[:part]
550
581
  part_val = extract_value(base_data[:part])
551
- separator = number_str.match?(/^[A-Z]/) ? "." : "."
552
- number_str += separator + part_val
582
+ number_str += part_val if part_val
553
583
  end
554
584
 
555
585
  # Add subpart if present
@@ -558,7 +588,7 @@ module Pubid
558
588
  subparts = [subparts] unless subparts.is_a?(Array)
559
589
  subparts.each do |sp|
560
590
  subpart_val = extract_value(sp)
561
- number_str += ".#{subpart_val}" if subpart_val
591
+ number_str += subpart_val if subpart_val
562
592
  end
563
593
  end
564
594
 
@@ -619,11 +649,10 @@ module Pubid
619
649
  # Extract number with parts and year
620
650
  number_str = extract_value(base_data[:number])
621
651
 
622
- # Add part if present
652
+ # Part/subpart captures carry their own separator — join verbatim.
623
653
  if base_data[:part]
624
654
  part_val = extract_value(base_data[:part])
625
- separator = number_str.match?(/^[A-Z]/) ? "." : "."
626
- number_str += separator + part_val
655
+ number_str += part_val if part_val
627
656
  end
628
657
 
629
658
  # Add subpart if present
@@ -632,7 +661,7 @@ module Pubid
632
661
  subparts = [subparts] unless subparts.is_a?(Array)
633
662
  subparts.each do |sp|
634
663
  subpart_val = extract_value(sp)
635
- number_str += ".#{subpart_val}" if subpart_val
664
+ number_str += subpart_val if subpart_val
636
665
  end
637
666
  end
638
667
 
@@ -754,8 +783,13 @@ module Pubid
754
783
  "D=#{stage}"
755
784
  end
756
785
  if parsed[:draft_iso_iteration]
757
- joint_draft += ".#{extract_value(parsed[:draft_iso_iteration])}"
786
+ iter = extract_value(parsed[:draft_iso_iteration]).to_s.sub(/\A\./, "")
787
+ joint_draft += ".#{iter}" unless iter.empty?
758
788
  end
789
+ # The date rides inside the designator colon-joined
790
+ # ("D=CDV:2020" — the standing joint_stage_draft_spec contract;
791
+ # the dash spelling the grammar newly accepts is an alias that
792
+ # normalizes to the colon face).
759
793
  if parsed[:draft_stage_year]
760
794
  joint_draft += ":#{extract_value(parsed[:draft_stage_year])}"
761
795
  end
@@ -827,8 +861,11 @@ module Pubid
827
861
 
828
862
  # Detect lead party based on pattern
829
863
  if parsed[:iso_stage]
830
- # ISO format - lead party is ISO
831
- attributes[:lead_party] = "ISO"
864
+ # ISO stage word present. The lead party is the PRINTED first
865
+ # publisher (pubid#469: the arrangement is the organization's
866
+ # perspective, shown by the print) — "IEEE/ISO/IEC CD P42010"
867
+ # is IEEE-led even though the stage word is ISO's.
868
+ attributes[:lead_party] = attributes[:publishers]&.first || "ISO"
832
869
  attributes[:iso_stage] = extract_value(parsed[:iso_stage])
833
870
 
834
871
  # Create typed_stage for ISO stage
@@ -1157,7 +1194,8 @@ module Pubid
1157
1194
  end
1158
1195
 
1159
1196
  if code_str && !code_parts.empty?
1160
- code_str += ".#{code_parts.join('.')}"
1197
+ # captures carry their own separator — join verbatim
1198
+ code_str += code_parts.join
1161
1199
  end
1162
1200
 
1163
1201
  # Year detection: If code_str ends with year-like pattern (dash + 4 digits, 1884-2099)
@@ -371,7 +371,7 @@ module Pubid
371
371
  result = PreParser.preprocess(normalized)
372
372
  return build_dual(result.parts) if result.dispatch == :dual_semicolon
373
373
  end
374
- parsed = Parser.parse(normalized) # Use class method for preprocessing
374
+ parsed = Pubid::Parg::Backend.parse(:ieee, Parser.normalize_input(normalized))
375
375
  builder = Builder.new(Identifier)
376
376
  # Pass the original input string to builder for context
377
377
  builder.original_input = input