pubid 2.0.0.pre.alpha.24 → 2.0.0.pre.alpha.26
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/data/parg/ieee.json +1 -1
- data/lib/pubid/iec/builder.rb +4 -1
- data/lib/pubid/iec/parser.rb +2 -1
- data/lib/pubid/iec/single_identifier.rb +4 -1
- data/lib/pubid/ieee/builder.rb +75 -6
- data/lib/pubid/ieee/identifiers/joint_development.rb +2 -1
- data/lib/pubid/ieee/parser.rb +5 -169
- data/lib/pubid/version.rb +1 -1
- metadata +2 -2
data/lib/pubid/iec/builder.rb
CHANGED
|
@@ -362,7 +362,10 @@ module Pubid
|
|
|
362
362
|
build(value)
|
|
363
363
|
|
|
364
364
|
when :publisher
|
|
365
|
-
|
|
365
|
+
# CEI is IEC's French spelling (pubid#488): a French-printed
|
|
366
|
+
# "CEI …" identifier is the IEC publisher; the canonical
|
|
367
|
+
# rendering prints IEC.
|
|
368
|
+
Components::Publisher.new(body: value == "CEI" ? "IEC" : value)
|
|
366
369
|
|
|
367
370
|
when :copublishers
|
|
368
371
|
if value.nil? || value.empty?
|
data/lib/pubid/iec/parser.rb
CHANGED
|
@@ -223,11 +223,12 @@ module Pubid
|
|
|
223
223
|
|
|
224
224
|
rule(:language) do
|
|
225
225
|
# IEC 60038:2009(en,fr)
|
|
226
|
+
# IEC 60038:2009(en-fr) (house style, pubid#491)
|
|
226
227
|
# IEC 60038:2009(E/F)
|
|
227
228
|
str("(") >>
|
|
228
229
|
(
|
|
229
230
|
# parse 2-char language codes: ru,en,fr,de,ar,es
|
|
230
|
-
(match["a-z"].repeat(1) >> str(",").maybe) |
|
|
231
|
+
(match["a-z"].repeat(1) >> (str(",") | str("-")).maybe) |
|
|
231
232
|
# parse single language codes: R/E/F
|
|
232
233
|
(match["EFARDS"] >> str("/").maybe)
|
|
233
234
|
).repeat.as(:languages) >>
|
|
@@ -78,11 +78,14 @@ module Pubid
|
|
|
78
78
|
def language_portion(lang_single: false)
|
|
79
79
|
return "" unless languages&.any?
|
|
80
80
|
|
|
81
|
+
# IEC house style joins multiple language codes with '-'
|
|
82
|
+
# ("IEC 60050-103:2020(en-fr)"), not ',' (pubid#491; pubid-iec 1.x
|
|
83
|
+
# behavior, which metanorma-iec's fixtures encode).
|
|
81
84
|
[
|
|
82
85
|
"(",
|
|
83
86
|
languages.map do |lang|
|
|
84
87
|
lang.to_s(lang_single: lang_single)
|
|
85
|
-
end.join(lang_single ? "/" : "
|
|
88
|
+
end.join(lang_single ? "/" : "-"),
|
|
86
89
|
")",
|
|
87
90
|
].join
|
|
88
91
|
end
|
data/lib/pubid/ieee/builder.rb
CHANGED
|
@@ -272,6 +272,51 @@ module Pubid
|
|
|
272
272
|
# Parslet can return array of hashes - merge them
|
|
273
273
|
parsed_hash = parsed.is_a?(Array) ? merge_parsed_array(parsed) : parsed
|
|
274
274
|
|
|
275
|
+
# Grammar-captured dash dates (relaton's "/D-3-2017-07" and
|
|
276
|
+
# "/R-x-2011-04" tails) land on the identifier's year/month exactly
|
|
277
|
+
# like the number's own date clause. The revision-suffix form sits at
|
|
278
|
+
# the top level; the draft tail nests under :draft — either as its
|
|
279
|
+
# own element or wrapped in a draft_version hash.
|
|
280
|
+
drafts = parsed_hash[:draft].is_a?(Array) ? parsed_hash[:draft] : [parsed_hash[:draft]].compact
|
|
281
|
+
drafts.each do |d|
|
|
282
|
+
next unless d.is_a?(Hash) && d[:draft_version].is_a?(Hash)
|
|
283
|
+
dv = d[:draft_version]
|
|
284
|
+
d[:dash_year] = dv[:dash_year] if dv[:dash_year]
|
|
285
|
+
d[:dash_month] = dv[:dash_month] if dv[:dash_month]
|
|
286
|
+
d.delete(:draft_version)
|
|
287
|
+
end
|
|
288
|
+
dash = parsed_hash[:dash_year] ? parsed_hash : nil
|
|
289
|
+
dash ||= drafts.find { |d| d.is_a?(Hash) && d[:dash_year] }
|
|
290
|
+
if dash && parsed_hash[:year].nil?
|
|
291
|
+
joined = drafts.filter_map { |d| d.is_a?(Hash) ? d[:draft_version] : nil }
|
|
292
|
+
.map { |v| v.is_a?(Hash) ? v[:value] : v.to_s }.join
|
|
293
|
+
bare_joint_draft = parsed_hash[:joint_publishers] &&
|
|
294
|
+
parsed_hash[:part].nil? &&
|
|
295
|
+
parsed_hash[:iso_stage].nil? &&
|
|
296
|
+
drafts.any? { |d| d.is_a?(Hash) && d[:draft_version] }
|
|
297
|
+
dash_led = joined.start_with?("-")
|
|
298
|
+
if bare_joint_draft && dash_led
|
|
299
|
+
parsed_hash[:part] = { value: dash[:dash_year] }
|
|
300
|
+
elsif dash[:dash_month] || dash_led
|
|
301
|
+
parsed_hash[:year] = dash[:dash_year]
|
|
302
|
+
parsed_hash[:month] = dash[:dash_month]
|
|
303
|
+
else
|
|
304
|
+
# Year-only tail embeds in the draft face ("D9-2006"), the model
|
|
305
|
+
# the relaton-pinned renders use for corrupted update_codes
|
|
306
|
+
# drafts: standard class, stage from the type word (the compound
|
|
307
|
+
# draft version misses the D-registry), no separate year.
|
|
308
|
+
host = drafts.find { |d| d.is_a?(Hash) && d[:draft_version] }
|
|
309
|
+
if host
|
|
310
|
+
host[:draft_version] = "#{joined}-#{dash[:dash_year]}"
|
|
311
|
+
dash.delete(:dash_year)
|
|
312
|
+
dash.delete(:dash_month)
|
|
313
|
+
else
|
|
314
|
+
parsed_hash[:year] = dash[:dash_year]
|
|
315
|
+
parsed_hash[:month] = dash[:dash_month]
|
|
316
|
+
end
|
|
317
|
+
end
|
|
318
|
+
end
|
|
319
|
+
|
|
275
320
|
# Handle multi-numbered identifiers (cross-reference and joint standards)
|
|
276
321
|
# CRITICAL: Don't recurse when building secondary identifier to prevent infinite loop
|
|
277
322
|
if !building_secondary && parsed_hash[:primary_identifier] && (parsed_hash[:secondary_crossref] || parsed_hash[:secondary_joint])
|
|
@@ -739,8 +784,27 @@ module Pubid
|
|
|
739
784
|
attributes[:copublisher] = pubs.drop(1)
|
|
740
785
|
end
|
|
741
786
|
|
|
742
|
-
# Build code with parts if present
|
|
787
|
+
# Build code with parts if present. A bare-P joint draft's dash
|
|
788
|
+
# date ("P42010/D-4-2019") rides the code as a year-part
|
|
789
|
+
# ("P42010.2019/D4") — the face the relaton fixtures pin — while
|
|
790
|
+
# stage-first joints keep the top-level year ("FDIS P15289:2017").
|
|
743
791
|
code_parts = []
|
|
792
|
+
if parsed[:part].nil? && parsed[:iso_stage].nil? && parsed[:draft_version] &&
|
|
793
|
+
parsed[:printed_dash_year] && parsed[:year] &&
|
|
794
|
+
(parsed[:year].is_a?(Hash) ? parsed[:year][:value] : parsed[:year]).to_s.match?(/\A(19|20)\d\d\z/)
|
|
795
|
+
# A dash-led draft ("P42010/D-4-2019") takes the year-part face;
|
|
796
|
+
# an undashed one ("P26511/D8-2018") embeds the date in the draft
|
|
797
|
+
# face.
|
|
798
|
+
if extract_value(parsed[:draft_dash]).to_s == "-"
|
|
799
|
+
code_parts << extract_value(parsed[:year])
|
|
800
|
+
parsed[:year] = nil
|
|
801
|
+
else
|
|
802
|
+
# The year stays: the IEEE face suppresses it while a draft is
|
|
803
|
+
# attached, and the ISO face renders it ("P26511:2018").
|
|
804
|
+
parsed[:draft_version] =
|
|
805
|
+
"#{extract_value(parsed[:draft_version])}-#{extract_value(parsed[:year])}"
|
|
806
|
+
end
|
|
807
|
+
end
|
|
744
808
|
code_parts << extract_value(parsed[:part]) if parsed[:part]
|
|
745
809
|
|
|
746
810
|
code_str = extract_value(parsed[:number])
|
|
@@ -785,9 +849,13 @@ module Pubid
|
|
|
785
849
|
end
|
|
786
850
|
attributes[:parenthetical_content] ||= extract_value(parsed[:edition_marker]) if parsed[:edition_marker]
|
|
787
851
|
|
|
788
|
-
# Extract edition, from relaton's "/E-<n>" suffix
|
|
789
|
-
#
|
|
790
|
-
|
|
852
|
+
# Extract edition, from relaton's "/E-<n>" suffix. The grammar's /E
|
|
853
|
+
# alternative captures the bare ordinal (nil-residue hand-off item 1).
|
|
854
|
+
if parsed[:edition]
|
|
855
|
+
attributes[:edition] = extract_value(parsed[:edition])
|
|
856
|
+
elsif parsed[:edition_e]
|
|
857
|
+
attributes[:edition] = "#{extract_value(parsed[:edition_e])}.0"
|
|
858
|
+
end
|
|
791
859
|
if parsed[:edition_month]
|
|
792
860
|
attributes[:edition_month] = extract_value(parsed[:edition_month])
|
|
793
861
|
end
|
|
@@ -1458,9 +1526,10 @@ module Pubid
|
|
|
1458
1526
|
extract_value(dv)
|
|
1459
1527
|
end
|
|
1460
1528
|
|
|
1461
|
-
# Construct draft notation like "D1", "D2", etc.
|
|
1462
1529
|
if version
|
|
1463
|
-
|
|
1530
|
+
# relaton's hyphenated "/D-2" spelling captures the dash inside
|
|
1531
|
+
# draft_version; the stage registry keys on "D2".
|
|
1532
|
+
draft_abbr = "D#{version.to_s.sub(/\A-/, "")}"
|
|
1464
1533
|
# Check if this specific draft stage is in registry
|
|
1465
1534
|
stage = Pubid::Ieee.locate_stage(draft_abbr)
|
|
1466
1535
|
return draft_abbr if stage&.abbr&.include?(draft_abbr)
|
|
@@ -239,7 +239,8 @@ module Pubid
|
|
|
239
239
|
# Join with space and add year with dash — unless the D=
|
|
240
240
|
# designator face already carried it above.
|
|
241
241
|
result = parts.join(" ")
|
|
242
|
-
result += "-#{year}" if year && !@designator_carries_year
|
|
242
|
+
result += "-#{year}" if year && !@designator_carries_year &&
|
|
243
|
+
!ieee_draft.to_s.end_with?("-#{year}")
|
|
243
244
|
|
|
244
245
|
result
|
|
245
246
|
end
|
data/lib/pubid/ieee/parser.rb
CHANGED
|
@@ -701,9 +701,8 @@ module Pubid
|
|
|
701
701
|
((comma | space) >> month_name.as(:month) >> space >>
|
|
702
702
|
year_digits.as(:year) >> str("").as(:printed_month_year))
|
|
703
703
|
).maybe >>
|
|
704
|
-
# Optional /D<draft> tail
|
|
705
|
-
#
|
|
706
|
-
# time this rule runs the draft usually trails the date (bucket 5);
|
|
704
|
+
# Optional /D<draft> tail; the historical hyphenated form
|
|
705
|
+
# "…/D-3-2017" carries its date inside the draft clause (bucket 5);
|
|
707
706
|
# a date-less "/D-4" keeps its hyphen (bucket 7), hence dash.maybe.
|
|
708
707
|
# A text date may trail the DRAFT itself (pubid#216:
|
|
709
708
|
# "CD P26515/D1, March 2017", "FDIS P15289/D3, 2017") — distinct
|
|
@@ -912,7 +911,7 @@ module Pubid
|
|
|
912
911
|
revision_suffix.maybe >>
|
|
913
912
|
# …and the print date may also trail the repositioned revision
|
|
914
913
|
# ("IEEE Unapproved Draft P802.16Rev2/D9a, March 2009" reaches the
|
|
915
|
-
# grammar
|
|
914
|
+
# grammar with its inline revision parsed natively).
|
|
916
915
|
(comma >> space? >> month_name.as(:trailing_month) >> space >>
|
|
917
916
|
year_digits.as(:trailing_year)).maybe >>
|
|
918
917
|
# Trailing corrigendum after the draft ("…/D2.0/Cor. 1", or
|
|
@@ -1046,9 +1045,8 @@ module Pubid
|
|
|
1046
1045
|
(str(".") >> year_digits.as(:year)).maybe >>
|
|
1047
1046
|
# The date and the draft appear in EITHER order: the legacy
|
|
1048
1047
|
# spelling puts the draft first ("PSI 10/D2, October 2015"),
|
|
1049
|
-
#
|
|
1050
|
-
#
|
|
1051
|
-
# "PSI 10-2010/D3"). Draft-first is tried first so the legacy
|
|
1048
|
+
# the rawbib hyphenated form ("PSI 10/D-3-2010") parses in place
|
|
1049
|
+
# with its date on the draft clause. Draft-first is tried first so the legacy
|
|
1052
1050
|
# comma-date keeps its original match; each side is optional so
|
|
1053
1051
|
# a date-only or draft-less form still parses.
|
|
1054
1052
|
(
|
|
@@ -1243,140 +1241,6 @@ module Pubid
|
|
|
1243
1241
|
end
|
|
1244
1242
|
|
|
1245
1243
|
root(:identifier)
|
|
1246
|
-
|
|
1247
|
-
# Rewrite relaton's historical IEEE serialization into canonical pubid
|
|
1248
|
-
# spellings. relaton's own formatter (Relaton::Ieee::PubId::Id#to_s) emits
|
|
1249
|
-
# suffix tokens that differ from pubid's grammar:
|
|
1250
|
-
#
|
|
1251
|
-
# /D-N-YYYY[-MM] draft + trailing numeric date (the dominant form)
|
|
1252
|
-
# /E-N[-YYYY[-MM]] edition
|
|
1253
|
-
# /R-N[-YYYY] revision (pubid has no revision suffix)
|
|
1254
|
-
# " Redline" redline suffix without the " - " pubid expects
|
|
1255
|
-
#
|
|
1256
|
-
# The draft/edition trailing date is repositioned onto the document number
|
|
1257
|
-
# as a base year/month (a form pubid already parses), which also keeps the
|
|
1258
|
-
# draft component clean so it round-trips through to_hash/from_hash.
|
|
1259
|
-
def self.normalize_relaton_suffixes(cleaned)
|
|
1260
|
-
# NOTE: the trailing " Redline"/" - Redline" suffix is NO LONGER stripped
|
|
1261
|
-
# here — the grammar's `redline` rule captures it into a redline flag so
|
|
1262
|
-
# a redline id stays distinct from its base standard.
|
|
1263
|
-
|
|
1264
|
-
# Combined draft + corrigendum: relaton emits "…/D-N/CorM-YYYY" (draft
|
|
1265
|
-
# then corrigendum), but pubid's grammar accepts the corrigendum first.
|
|
1266
|
-
# Swap them so the corrigendum keeps its own year and the draft trails.
|
|
1267
|
-
# The hyphen after "D" is mandatory here: relaton's formatter always
|
|
1268
|
-
# emits "/D-<draft>", whereas pubid's own canonical joint-development
|
|
1269
|
-
# form is "/D<draft>-<year>" (no hyphen, year kept on the draft) — which
|
|
1270
|
-
# already parses and must not be repositioned. A trailing corrigendum
|
|
1271
|
-
# month (the "-MM" in "/CorM-YYYY-MM") is intentionally dropped: pubid's
|
|
1272
|
-
# corrigendum model carries only a year.
|
|
1273
|
-
cleaned = cleaned.sub(
|
|
1274
|
-
%r{\A(.*)/D-([0-9A-Za-z][0-9A-Za-z.+]*?)/Cor\.?[ ]?(\d+)(?:-((?:19|20)\d\d))?(?:-\d\d)?\z},
|
|
1275
|
-
) do
|
|
1276
|
-
base, draft, cor, year = Regexp.last_match.captures
|
|
1277
|
-
"#{base}/Cor #{cor}#{year ? "-#{year}" : ''}/D#{draft}"
|
|
1278
|
-
end
|
|
1279
|
-
|
|
1280
|
-
# Combined draft + revision, and the empty-draft revision-only form:
|
|
1281
|
-
# "…/D-<d>/R-<x>-YYYY[-MM]" and "…/D-/R-<x>-YYYY" (nil-residue #2).
|
|
1282
|
-
# Reposition the base publication date onto the number (pubid's
|
|
1283
|
-
# "-YYYY[-MM]" shape), keep the draft as "/D<d>" (dropped when the draft
|
|
1284
|
-
# is empty), and leave a trailing "/R-<x>" the grammar captures as the
|
|
1285
|
-
# revision. Runs before the plain "/D-…" reposition, which the embedded
|
|
1286
|
-
# "/R-" would otherwise defeat.
|
|
1287
|
-
cleaned = cleaned.sub(
|
|
1288
|
-
%r{\A(.*?)/D-([0-9A-Za-z.+]*)/R-([0-9A-Za-z]+)(?:-((?:19|20)\d\d)(?:-(0[1-9]|1[0-2]))?)?\z},
|
|
1289
|
-
) do
|
|
1290
|
-
base, draft, rev, year, month = Regexp.last_match.captures
|
|
1291
|
-
date = year ? "-#{year}#{month ? "-#{month}" : ''}" : ""
|
|
1292
|
-
draft_part = draft.to_s.empty? ? "" : "/D#{draft}"
|
|
1293
|
-
"#{base}#{date}#{draft_part}/R-#{rev}"
|
|
1294
|
-
end
|
|
1295
|
-
|
|
1296
|
-
# /D-N drafts with a trailing numeric date, when the draft is the last
|
|
1297
|
-
# suffix: reposition the -YYYY[-MM] date onto the number. A following
|
|
1298
|
-
# /Cor, /Amd, /R or /E suffix carries its own year, so the `\z` anchor
|
|
1299
|
-
# keeps this from firing on those combined forms.
|
|
1300
|
-
cleaned = cleaned.sub(
|
|
1301
|
-
%r{\A(.*)/D-([0-9A-Za-z][0-9A-Za-z.+]*?)-((?:19|20)\d\d)(?:-(0[1-9]|1[0-2]))?\z},
|
|
1302
|
-
) do
|
|
1303
|
-
base, draft, year, month = Regexp.last_match.captures
|
|
1304
|
-
"#{base}-#{year}#{month ? "-#{month}" : ''}/D#{draft}"
|
|
1305
|
-
end
|
|
1306
|
-
|
|
1307
|
-
# /E-N editions: relaton's "/E-2-2023-02" → pubid's "Edition 2.0 2023-02".
|
|
1308
|
-
cleaned = cleaned.sub(
|
|
1309
|
-
%r{\A(.*?)/E-(\d+)(?:-((?:19|20)\d\d)(?:-(0[1-9]|1[0-2]))?)?\z},
|
|
1310
|
-
) do
|
|
1311
|
-
base, edition, year, month = Regexp.last_match.captures
|
|
1312
|
-
date = year ? " #{year}#{month ? "-#{month}" : ''}" : ""
|
|
1313
|
-
"#{base} Edition #{edition}.0#{date}"
|
|
1314
|
-
end
|
|
1315
|
-
|
|
1316
|
-
# /R-N revisions: PRESERVE them (the grammar's revision_suffix rule now
|
|
1317
|
-
# captures a trailing "/R-<x>" into the `revision` attribute). Just
|
|
1318
|
-
# reposition any trailing publication year onto the number, keeping the
|
|
1319
|
-
# "/R-<x>" in place for the grammar.
|
|
1320
|
-
cleaned.sub(
|
|
1321
|
-
%r{\A(.*?)/R-([0-9A-Za-z]+)(?:-((?:19|20)\d\d))?\z},
|
|
1322
|
-
) do
|
|
1323
|
-
base, rev, year = Regexp.last_match.captures
|
|
1324
|
-
"#{year ? "#{base}-#{year}" : base}/R-#{rev}"
|
|
1325
|
-
end
|
|
1326
|
-
end
|
|
1327
|
-
|
|
1328
|
-
# Strip the IEEE rawbib revision-notation dialects. `REV`/`Rev`
|
|
1329
|
-
# (case-insensitive) + a trailing revision id `[A-Za-z0-9]+`, glued to the
|
|
1330
|
-
# number or separated by `-`, `/`, `_`, `.`, or a space, and preceding the
|
|
1331
|
-
# draft. pubid's canonical "<num>/D<n>/R-<x>" form already drops the
|
|
1332
|
-
# revision on render (normalize_relaton_suffixes strips a trailing /R-x),
|
|
1333
|
-
# so the revision-less result is *the same identifier* — and stripping
|
|
1334
|
-
# (rather than reordering) leaves any trailing date/parenthetical intact,
|
|
1335
|
-
# which is why forms that already parse (`Draft P…-REVmb/D3.0, Mar 2010`)
|
|
1336
|
-
# are NOT disturbed. Examples:
|
|
1337
|
-
# "P802.16.2-REVa/D8" -> "P802.16.2/D8"
|
|
1338
|
-
# "P802.16/REVd/D5" -> "P802.16/D5"
|
|
1339
|
-
# "P802.15.1REVa/D5" -> "P802.15.1/D5"
|
|
1340
|
-
# "P802.11REVmb" -> "P802.11" (no draft)
|
|
1341
|
-
def self.normalize_revision_notation(cleaned)
|
|
1342
|
-
# NUMBERED revisions ("Rev<digits>") are PRESERVED — repositioned to a
|
|
1343
|
-
# trailing "/R-<n>" suffix the grammar captures as the `revision`
|
|
1344
|
-
# attribute (IEEE's native inline spelling; numbered-revision hand-off).
|
|
1345
|
-
# A "\d+" right after "Rev" both selects the numbered subset and keeps
|
|
1346
|
-
# these off the English word "Revision". Three source positions:
|
|
1347
|
-
# after a draft : "PC37.30.2/D043 Rev 18" -> ".../D043/R-18"
|
|
1348
|
-
# The draft captures accept relaton's hyphenated "/D-<n>" spelling
|
|
1349
|
-
# (pubid#316): without the "-?" the before-draft regex matched only
|
|
1350
|
-
# "/D" of "/D-3" and emitted the garbage "/D/R-2-3-2008-02".
|
|
1351
|
-
cleaned = cleaned.sub(
|
|
1352
|
-
%r{(/D-?[0-9A-Za-z.]+)\s+[Rr][Ee][Vv]\s*(\d+)}, '\1/R-\2'
|
|
1353
|
-
)
|
|
1354
|
-
# before a draft: "P802.16Rev2/D3" -> "P802.16/D3/R-2"
|
|
1355
|
-
cleaned = cleaned.sub(
|
|
1356
|
-
%r{[-/_.]?\s?[Rr][Ee][Vv][-\s]?(\d+)(/D-?[0-9A-Za-z.]+)}, '\2/R-\1'
|
|
1357
|
-
)
|
|
1358
|
-
# no draft, trailing: "P1722-rev1" -> "P1722/R-1"
|
|
1359
|
-
cleaned = cleaned.sub(
|
|
1360
|
-
%r{(\d)[-._]?\s?[Rr][Ee][Vv]\s*(\d+)\s*\z}, '\1/R-\2'
|
|
1361
|
-
)
|
|
1362
|
-
|
|
1363
|
-
# LETTERED inline revisions ("REVa", "REVmb") have no pubid model and are
|
|
1364
|
-
# still STRIPPED (unchanged behaviour). The numbered forms above already
|
|
1365
|
-
# became "/R-<n>", so these regexes only see the lettered residue.
|
|
1366
|
-
# Revision token that PRECEDES a draft: drop it (keep the /D…).
|
|
1367
|
-
cleaned = cleaned.sub(
|
|
1368
|
-
%r{[-/_.]?\s?[Rr][Ee][Vv][-\s]?[A-Za-z0-9]+(?=/D-?[0-9A-Za-z])},
|
|
1369
|
-
"",
|
|
1370
|
-
)
|
|
1371
|
-
# Trailing revision glued to the number with no draft ("P802.11REVmb");
|
|
1372
|
-
# a digit must immediately precede REV so a trailing English word like
|
|
1373
|
-
# "…Revision" can't match.
|
|
1374
|
-
cleaned.sub(%r{(\d)[Rr][Ee][Vv][A-Za-z0-9]+\s*\z}, '\1')
|
|
1375
|
-
end
|
|
1376
|
-
|
|
1377
|
-
# Pre-parse ingestion normalizations (R2): every parse path —
|
|
1378
|
-
# parslet and PG artifact alike — feeds the grammar the same
|
|
1379
|
-
# normalized string.
|
|
1380
1244
|
def self.normalize_input(string)
|
|
1381
1245
|
# Strip .pdf extension if present (Pattern 3: File Extensions)
|
|
1382
1246
|
cleaned = string.sub(/\.pdf$/i, "")
|
|
@@ -1390,20 +1254,6 @@ module Pubid
|
|
|
1390
1254
|
# No valid IEEE identifier pattern needs more than 1 space
|
|
1391
1255
|
cleaned = cleaned.gsub(/\s+/, " ")
|
|
1392
1256
|
|
|
1393
|
-
# Rewrite the rawbib revision-notation dialects (REVa/REVd/glued) into
|
|
1394
|
-
# the canonical /R-<x> form before the suffix normalization below.
|
|
1395
|
-
cleaned = normalize_revision_notation(cleaned)
|
|
1396
|
-
|
|
1397
|
-
# Normalize relaton's bespoke historical serialization (the spellings
|
|
1398
|
-
# emitted by Relaton::Ieee::PubId::Id#to_s) into canonical pubid forms
|
|
1399
|
-
# so `relaton-data-ieee` parses. See #normalize_relaton_suffixes.
|
|
1400
|
-
cleaned = normalize_relaton_suffixes(cleaned)
|
|
1401
|
-
|
|
1402
|
-
# NEW Session 171: CONSERVATIVE data quality fixes for TODO.IEEE-MUST-DO.txt
|
|
1403
|
-
# Only fix clear typos: space before dash + 4-digit year, OR dash + space + 4-digit year
|
|
1404
|
-
# Do NOT touch " - " (space-dash-space) which is valid formatting
|
|
1405
|
-
cleaned = cleaned.gsub(/(\d)\s+-(\d{4})\b/, '\1-\2') # "C37.101 -2006" → "C37.101-2006"
|
|
1406
|
-
cleaned = cleaned.gsub(/(\d)-\s+(\d{4})\b/, '\1-\2') # "C62.35- 2010" → "C62.35-2010"
|
|
1407
1257
|
|
|
1408
1258
|
# NEW Session 171: Remove wrong ! prefix
|
|
1409
1259
|
cleaned = cleaned.gsub(/^!IEEE /, "IEEE ")
|
|
@@ -1414,10 +1264,6 @@ module Pubid
|
|
|
1414
1264
|
cleaned = cleaned.gsub("&amp;", "&") # Double-encoded ampersand
|
|
1415
1265
|
cleaned = cleaned.gsub("&", "&") # Single-encoded ampersand
|
|
1416
1266
|
|
|
1417
|
-
# NEW: Wrap P&V notation in parentheses (Paper & Video, etc.)
|
|
1418
|
-
# Pattern: "IEEE Std 500-1984 P&V" → "IEEE Std 500-1984 (P&V)"
|
|
1419
|
-
cleaned = cleaned.gsub(/\s+(P&V)\s*$/, ' (\1)')
|
|
1420
|
-
|
|
1421
1267
|
# NEW Phase 1: Fix number spacing issues (e.g., "C57.1 2.25" → "C57.12.25")
|
|
1422
1268
|
# This handles cases where a space appears in the middle of a number
|
|
1423
1269
|
cleaned = cleaned.gsub(/(\d+\.\d+)\s+(\d+\.)/, '\1\2')
|
|
@@ -1426,16 +1272,6 @@ module Pubid
|
|
|
1426
1272
|
# Remove spaces within 4-digit years
|
|
1427
1273
|
cleaned = cleaned.gsub(/\b(1|2)\s+(\d{3})\b/, '\1\2')
|
|
1428
1274
|
|
|
1429
|
-
# NEW: Fix month+year spacing (e.g., "March2016" → "March 2016")
|
|
1430
|
-
# Add space between month name and 4-digit year when they're concatenated
|
|
1431
|
-
cleaned = cleaned.gsub(
|
|
1432
|
-
/\b(January|February|March|April|May|June|July|August|September|October|November|December)(\d{4})\b/, '\1 \2'
|
|
1433
|
-
)
|
|
1434
|
-
# Also handle abbreviated months
|
|
1435
|
-
cleaned = cleaned.gsub(
|
|
1436
|
-
/\b(Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Sept|Oct|Nov|Dec)(\d{4})\b/, '\1 \2'
|
|
1437
|
-
)
|
|
1438
|
-
|
|
1439
1275
|
# NEW: Convert IEC/IEEE space-separated to semicolon format
|
|
1440
1276
|
# Pattern: "IEC 61523-3 First edition 2004-09; IEEE 1497" → already semicolon
|
|
1441
1277
|
# Pattern: "IEC 62539 First Edition 2007-07 IEEE 930" → needs semicolon
|
data/lib/pubid/version.rb
CHANGED
metadata
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: pubid
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 2.0.0.pre.alpha.
|
|
4
|
+
version: 2.0.0.pre.alpha.26
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Ribose Inc.
|
|
8
8
|
autorequire:
|
|
9
9
|
bindir: exe
|
|
10
10
|
cert_chain: []
|
|
11
|
-
date: 2026-10-
|
|
11
|
+
date: 2026-10-04 00:00:00.000000000 Z
|
|
12
12
|
dependencies:
|
|
13
13
|
- !ruby/object:Gem::Dependency
|
|
14
14
|
name: lutaml-model
|