pubid 2.0.0.pre.alpha.23 → 2.0.0.pre.alpha.25
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/data/parg/ieee.json +1 -1
- data/lib/pubid/iec/builder.rb +4 -1
- data/lib/pubid/iec/parser.rb +2 -1
- data/lib/pubid/iec/single_identifier.rb +4 -1
- data/lib/pubid/ieee/builder.rb +75 -6
- data/lib/pubid/ieee/identifiers/joint_development.rb +2 -1
- data/lib/pubid/ieee/parser.rb +5 -256
- data/lib/pubid/version.rb +1 -1
- metadata +1 -1
data/lib/pubid/iec/builder.rb
CHANGED
|
@@ -362,7 +362,10 @@ module Pubid
|
|
|
362
362
|
build(value)
|
|
363
363
|
|
|
364
364
|
when :publisher
|
|
365
|
-
|
|
365
|
+
# CEI is IEC's French spelling (pubid#488): a French-printed
|
|
366
|
+
# "CEI …" identifier is the IEC publisher; the canonical
|
|
367
|
+
# rendering prints IEC.
|
|
368
|
+
Components::Publisher.new(body: value == "CEI" ? "IEC" : value)
|
|
366
369
|
|
|
367
370
|
when :copublishers
|
|
368
371
|
if value.nil? || value.empty?
|
data/lib/pubid/iec/parser.rb
CHANGED
|
@@ -223,11 +223,12 @@ module Pubid
|
|
|
223
223
|
|
|
224
224
|
rule(:language) do
|
|
225
225
|
# IEC 60038:2009(en,fr)
|
|
226
|
+
# IEC 60038:2009(en-fr) (house style, pubid#491)
|
|
226
227
|
# IEC 60038:2009(E/F)
|
|
227
228
|
str("(") >>
|
|
228
229
|
(
|
|
229
230
|
# parse 2-char language codes: ru,en,fr,de,ar,es
|
|
230
|
-
(match["a-z"].repeat(1) >> str(",").maybe) |
|
|
231
|
+
(match["a-z"].repeat(1) >> (str(",") | str("-")).maybe) |
|
|
231
232
|
# parse single language codes: R/E/F
|
|
232
233
|
(match["EFARDS"] >> str("/").maybe)
|
|
233
234
|
).repeat.as(:languages) >>
|
|
@@ -78,11 +78,14 @@ module Pubid
|
|
|
78
78
|
def language_portion(lang_single: false)
|
|
79
79
|
return "" unless languages&.any?
|
|
80
80
|
|
|
81
|
+
# IEC house style joins multiple language codes with '-'
|
|
82
|
+
# ("IEC 60050-103:2020(en-fr)"), not ',' (pubid#491; pubid-iec 1.x
|
|
83
|
+
# behavior, which metanorma-iec's fixtures encode).
|
|
81
84
|
[
|
|
82
85
|
"(",
|
|
83
86
|
languages.map do |lang|
|
|
84
87
|
lang.to_s(lang_single: lang_single)
|
|
85
|
-
end.join(lang_single ? "/" : "
|
|
88
|
+
end.join(lang_single ? "/" : "-"),
|
|
86
89
|
")",
|
|
87
90
|
].join
|
|
88
91
|
end
|
data/lib/pubid/ieee/builder.rb
CHANGED
|
@@ -272,6 +272,51 @@ module Pubid
|
|
|
272
272
|
# Parslet can return array of hashes - merge them
|
|
273
273
|
parsed_hash = parsed.is_a?(Array) ? merge_parsed_array(parsed) : parsed
|
|
274
274
|
|
|
275
|
+
# Grammar-captured dash dates (relaton's "/D-3-2017-07" and
|
|
276
|
+
# "/R-x-2011-04" tails) land on the identifier's year/month exactly
|
|
277
|
+
# like the number's own date clause. The revision-suffix form sits at
|
|
278
|
+
# the top level; the draft tail nests under :draft — either as its
|
|
279
|
+
# own element or wrapped in a draft_version hash.
|
|
280
|
+
drafts = parsed_hash[:draft].is_a?(Array) ? parsed_hash[:draft] : [parsed_hash[:draft]].compact
|
|
281
|
+
drafts.each do |d|
|
|
282
|
+
next unless d.is_a?(Hash) && d[:draft_version].is_a?(Hash)
|
|
283
|
+
dv = d[:draft_version]
|
|
284
|
+
d[:dash_year] = dv[:dash_year] if dv[:dash_year]
|
|
285
|
+
d[:dash_month] = dv[:dash_month] if dv[:dash_month]
|
|
286
|
+
d.delete(:draft_version)
|
|
287
|
+
end
|
|
288
|
+
dash = parsed_hash[:dash_year] ? parsed_hash : nil
|
|
289
|
+
dash ||= drafts.find { |d| d.is_a?(Hash) && d[:dash_year] }
|
|
290
|
+
if dash && parsed_hash[:year].nil?
|
|
291
|
+
joined = drafts.filter_map { |d| d.is_a?(Hash) ? d[:draft_version] : nil }
|
|
292
|
+
.map { |v| v.is_a?(Hash) ? v[:value] : v.to_s }.join
|
|
293
|
+
bare_joint_draft = parsed_hash[:joint_publishers] &&
|
|
294
|
+
parsed_hash[:part].nil? &&
|
|
295
|
+
parsed_hash[:iso_stage].nil? &&
|
|
296
|
+
drafts.any? { |d| d.is_a?(Hash) && d[:draft_version] }
|
|
297
|
+
dash_led = joined.start_with?("-")
|
|
298
|
+
if bare_joint_draft && dash_led
|
|
299
|
+
parsed_hash[:part] = { value: dash[:dash_year] }
|
|
300
|
+
elsif dash[:dash_month] || dash_led
|
|
301
|
+
parsed_hash[:year] = dash[:dash_year]
|
|
302
|
+
parsed_hash[:month] = dash[:dash_month]
|
|
303
|
+
else
|
|
304
|
+
# Year-only tail embeds in the draft face ("D9-2006"), the model
|
|
305
|
+
# the relaton-pinned renders use for corrupted update_codes
|
|
306
|
+
# drafts: standard class, stage from the type word (the compound
|
|
307
|
+
# draft version misses the D-registry), no separate year.
|
|
308
|
+
host = drafts.find { |d| d.is_a?(Hash) && d[:draft_version] }
|
|
309
|
+
if host
|
|
310
|
+
host[:draft_version] = "#{joined}-#{dash[:dash_year]}"
|
|
311
|
+
dash.delete(:dash_year)
|
|
312
|
+
dash.delete(:dash_month)
|
|
313
|
+
else
|
|
314
|
+
parsed_hash[:year] = dash[:dash_year]
|
|
315
|
+
parsed_hash[:month] = dash[:dash_month]
|
|
316
|
+
end
|
|
317
|
+
end
|
|
318
|
+
end
|
|
319
|
+
|
|
275
320
|
# Handle multi-numbered identifiers (cross-reference and joint standards)
|
|
276
321
|
# CRITICAL: Don't recurse when building secondary identifier to prevent infinite loop
|
|
277
322
|
if !building_secondary && parsed_hash[:primary_identifier] && (parsed_hash[:secondary_crossref] || parsed_hash[:secondary_joint])
|
|
@@ -739,8 +784,27 @@ module Pubid
|
|
|
739
784
|
attributes[:copublisher] = pubs.drop(1)
|
|
740
785
|
end
|
|
741
786
|
|
|
742
|
-
# Build code with parts if present
|
|
787
|
+
# Build code with parts if present. A bare-P joint draft's dash
|
|
788
|
+
# date ("P42010/D-4-2019") rides the code as a year-part
|
|
789
|
+
# ("P42010.2019/D4") — the face the relaton fixtures pin — while
|
|
790
|
+
# stage-first joints keep the top-level year ("FDIS P15289:2017").
|
|
743
791
|
code_parts = []
|
|
792
|
+
if parsed[:part].nil? && parsed[:iso_stage].nil? && parsed[:draft_version] &&
|
|
793
|
+
parsed[:printed_dash_year] && parsed[:year] &&
|
|
794
|
+
(parsed[:year].is_a?(Hash) ? parsed[:year][:value] : parsed[:year]).to_s.match?(/\A(19|20)\d\d\z/)
|
|
795
|
+
# A dash-led draft ("P42010/D-4-2019") takes the year-part face;
|
|
796
|
+
# an undashed one ("P26511/D8-2018") embeds the date in the draft
|
|
797
|
+
# face.
|
|
798
|
+
if extract_value(parsed[:draft_dash]).to_s == "-"
|
|
799
|
+
code_parts << extract_value(parsed[:year])
|
|
800
|
+
parsed[:year] = nil
|
|
801
|
+
else
|
|
802
|
+
# The year stays: the IEEE face suppresses it while a draft is
|
|
803
|
+
# attached, and the ISO face renders it ("P26511:2018").
|
|
804
|
+
parsed[:draft_version] =
|
|
805
|
+
"#{extract_value(parsed[:draft_version])}-#{extract_value(parsed[:year])}"
|
|
806
|
+
end
|
|
807
|
+
end
|
|
744
808
|
code_parts << extract_value(parsed[:part]) if parsed[:part]
|
|
745
809
|
|
|
746
810
|
code_str = extract_value(parsed[:number])
|
|
@@ -785,9 +849,13 @@ module Pubid
|
|
|
785
849
|
end
|
|
786
850
|
attributes[:parenthetical_content] ||= extract_value(parsed[:edition_marker]) if parsed[:edition_marker]
|
|
787
851
|
|
|
788
|
-
# Extract edition, from relaton's "/E-<n>" suffix
|
|
789
|
-
#
|
|
790
|
-
|
|
852
|
+
# Extract edition, from relaton's "/E-<n>" suffix. The grammar's /E
|
|
853
|
+
# alternative captures the bare ordinal (nil-residue hand-off item 1).
|
|
854
|
+
if parsed[:edition]
|
|
855
|
+
attributes[:edition] = extract_value(parsed[:edition])
|
|
856
|
+
elsif parsed[:edition_e]
|
|
857
|
+
attributes[:edition] = "#{extract_value(parsed[:edition_e])}.0"
|
|
858
|
+
end
|
|
791
859
|
if parsed[:edition_month]
|
|
792
860
|
attributes[:edition_month] = extract_value(parsed[:edition_month])
|
|
793
861
|
end
|
|
@@ -1458,9 +1526,10 @@ module Pubid
|
|
|
1458
1526
|
extract_value(dv)
|
|
1459
1527
|
end
|
|
1460
1528
|
|
|
1461
|
-
# Construct draft notation like "D1", "D2", etc.
|
|
1462
1529
|
if version
|
|
1463
|
-
|
|
1530
|
+
# relaton's hyphenated "/D-2" spelling captures the dash inside
|
|
1531
|
+
# draft_version; the stage registry keys on "D2".
|
|
1532
|
+
draft_abbr = "D#{version.to_s.sub(/\A-/, "")}"
|
|
1464
1533
|
# Check if this specific draft stage is in registry
|
|
1465
1534
|
stage = Pubid::Ieee.locate_stage(draft_abbr)
|
|
1466
1535
|
return draft_abbr if stage&.abbr&.include?(draft_abbr)
|
|
@@ -239,7 +239,8 @@ module Pubid
|
|
|
239
239
|
# Join with space and add year with dash — unless the D=
|
|
240
240
|
# designator face already carried it above.
|
|
241
241
|
result = parts.join(" ")
|
|
242
|
-
result += "-#{year}" if year && !@designator_carries_year
|
|
242
|
+
result += "-#{year}" if year && !@designator_carries_year &&
|
|
243
|
+
!ieee_draft.to_s.end_with?("-#{year}")
|
|
243
244
|
|
|
244
245
|
result
|
|
245
246
|
end
|
data/lib/pubid/ieee/parser.rb
CHANGED
|
@@ -701,9 +701,8 @@ module Pubid
|
|
|
701
701
|
((comma | space) >> month_name.as(:month) >> space >>
|
|
702
702
|
year_digits.as(:year) >> str("").as(:printed_month_year))
|
|
703
703
|
).maybe >>
|
|
704
|
-
# Optional /D<draft> tail
|
|
705
|
-
#
|
|
706
|
-
# time this rule runs the draft usually trails the date (bucket 5);
|
|
704
|
+
# Optional /D<draft> tail; the historical hyphenated form
|
|
705
|
+
# "…/D-3-2017" carries its date inside the draft clause (bucket 5);
|
|
707
706
|
# a date-less "/D-4" keeps its hyphen (bucket 7), hence dash.maybe.
|
|
708
707
|
# A text date may trail the DRAFT itself (pubid#216:
|
|
709
708
|
# "CD P26515/D1, March 2017", "FDIS P15289/D3, 2017") — distinct
|
|
@@ -912,7 +911,7 @@ module Pubid
|
|
|
912
911
|
revision_suffix.maybe >>
|
|
913
912
|
# …and the print date may also trail the repositioned revision
|
|
914
913
|
# ("IEEE Unapproved Draft P802.16Rev2/D9a, March 2009" reaches the
|
|
915
|
-
# grammar
|
|
914
|
+
# grammar with its inline revision parsed natively).
|
|
916
915
|
(comma >> space? >> month_name.as(:trailing_month) >> space >>
|
|
917
916
|
year_digits.as(:trailing_year)).maybe >>
|
|
918
917
|
# Trailing corrigendum after the draft ("…/D2.0/Cor. 1", or
|
|
@@ -1046,9 +1045,8 @@ module Pubid
|
|
|
1046
1045
|
(str(".") >> year_digits.as(:year)).maybe >>
|
|
1047
1046
|
# The date and the draft appear in EITHER order: the legacy
|
|
1048
1047
|
# spelling puts the draft first ("PSI 10/D2, October 2015"),
|
|
1049
|
-
#
|
|
1050
|
-
#
|
|
1051
|
-
# "PSI 10-2010/D3"). Draft-first is tried first so the legacy
|
|
1048
|
+
# the rawbib hyphenated form ("PSI 10/D-3-2010") parses in place
|
|
1049
|
+
# with its date on the draft clause. Draft-first is tried first so the legacy
|
|
1052
1050
|
# comma-date keeps its original match; each side is optional so
|
|
1053
1051
|
# a date-only or draft-less form still parses.
|
|
1054
1052
|
(
|
|
@@ -1243,242 +1241,6 @@ module Pubid
|
|
|
1243
1241
|
end
|
|
1244
1242
|
|
|
1245
1243
|
root(:identifier)
|
|
1246
|
-
|
|
1247
|
-
# Rewrite relaton's historical IEEE serialization into canonical pubid
|
|
1248
|
-
# spellings. relaton's own formatter (Relaton::Ieee::PubId::Id#to_s) emits
|
|
1249
|
-
# suffix tokens that differ from pubid's grammar:
|
|
1250
|
-
#
|
|
1251
|
-
# /D-N-YYYY[-MM] draft + trailing numeric date (the dominant form)
|
|
1252
|
-
# /E-N[-YYYY[-MM]] edition
|
|
1253
|
-
# /R-N[-YYYY] revision (pubid has no revision suffix)
|
|
1254
|
-
# " Redline" redline suffix without the " - " pubid expects
|
|
1255
|
-
#
|
|
1256
|
-
# The draft/edition trailing date is repositioned onto the document number
|
|
1257
|
-
# as a base year/month (a form pubid already parses), which also keeps the
|
|
1258
|
-
# draft component clean so it round-trips through to_hash/from_hash.
|
|
1259
|
-
def self.normalize_relaton_suffixes(cleaned)
|
|
1260
|
-
# NOTE: the trailing " Redline"/" - Redline" suffix is NO LONGER stripped
|
|
1261
|
-
# here — the grammar's `redline` rule captures it into a redline flag so
|
|
1262
|
-
# a redline id stays distinct from its base standard.
|
|
1263
|
-
|
|
1264
|
-
# Combined draft + corrigendum: relaton emits "…/D-N/CorM-YYYY" (draft
|
|
1265
|
-
# then corrigendum), but pubid's grammar accepts the corrigendum first.
|
|
1266
|
-
# Swap them so the corrigendum keeps its own year and the draft trails.
|
|
1267
|
-
# The hyphen after "D" is mandatory here: relaton's formatter always
|
|
1268
|
-
# emits "/D-<draft>", whereas pubid's own canonical joint-development
|
|
1269
|
-
# form is "/D<draft>-<year>" (no hyphen, year kept on the draft) — which
|
|
1270
|
-
# already parses and must not be repositioned. A trailing corrigendum
|
|
1271
|
-
# month (the "-MM" in "/CorM-YYYY-MM") is intentionally dropped: pubid's
|
|
1272
|
-
# corrigendum model carries only a year.
|
|
1273
|
-
cleaned = cleaned.sub(
|
|
1274
|
-
%r{\A(.*)/D-([0-9A-Za-z][0-9A-Za-z.+]*?)/Cor\.?[ ]?(\d+)(?:-((?:19|20)\d\d))?(?:-\d\d)?\z},
|
|
1275
|
-
) do
|
|
1276
|
-
base, draft, cor, year = Regexp.last_match.captures
|
|
1277
|
-
"#{base}/Cor #{cor}#{year ? "-#{year}" : ''}/D#{draft}"
|
|
1278
|
-
end
|
|
1279
|
-
|
|
1280
|
-
# Combined draft + revision, and the empty-draft revision-only form:
|
|
1281
|
-
# "…/D-<d>/R-<x>-YYYY[-MM]" and "…/D-/R-<x>-YYYY" (nil-residue #2).
|
|
1282
|
-
# Reposition the base publication date onto the number (pubid's
|
|
1283
|
-
# "-YYYY[-MM]" shape), keep the draft as "/D<d>" (dropped when the draft
|
|
1284
|
-
# is empty), and leave a trailing "/R-<x>" the grammar captures as the
|
|
1285
|
-
# revision. Runs before the plain "/D-…" reposition, which the embedded
|
|
1286
|
-
# "/R-" would otherwise defeat.
|
|
1287
|
-
cleaned = cleaned.sub(
|
|
1288
|
-
%r{\A(.*?)/D-([0-9A-Za-z.+]*)/R-([0-9A-Za-z]+)(?:-((?:19|20)\d\d)(?:-(0[1-9]|1[0-2]))?)?\z},
|
|
1289
|
-
) do
|
|
1290
|
-
base, draft, rev, year, month = Regexp.last_match.captures
|
|
1291
|
-
date = year ? "-#{year}#{month ? "-#{month}" : ''}" : ""
|
|
1292
|
-
draft_part = draft.to_s.empty? ? "" : "/D#{draft}"
|
|
1293
|
-
"#{base}#{date}#{draft_part}/R-#{rev}"
|
|
1294
|
-
end
|
|
1295
|
-
|
|
1296
|
-
# /D-N drafts with a trailing numeric date, when the draft is the last
|
|
1297
|
-
# suffix: reposition the -YYYY[-MM] date onto the number. A following
|
|
1298
|
-
# /Cor, /Amd, /R or /E suffix carries its own year, so the `\z` anchor
|
|
1299
|
-
# keeps this from firing on those combined forms.
|
|
1300
|
-
cleaned = cleaned.sub(
|
|
1301
|
-
%r{\A(.*)/D-([0-9A-Za-z][0-9A-Za-z.+]*?)-((?:19|20)\d\d)(?:-(0[1-9]|1[0-2]))?\z},
|
|
1302
|
-
) do
|
|
1303
|
-
base, draft, year, month = Regexp.last_match.captures
|
|
1304
|
-
"#{base}-#{year}#{month ? "-#{month}" : ''}/D#{draft}"
|
|
1305
|
-
end
|
|
1306
|
-
|
|
1307
|
-
# /E-N editions: relaton's "/E-2-2023-02" → pubid's "Edition 2.0 2023-02".
|
|
1308
|
-
cleaned = cleaned.sub(
|
|
1309
|
-
%r{\A(.*?)/E-(\d+)(?:-((?:19|20)\d\d)(?:-(0[1-9]|1[0-2]))?)?\z},
|
|
1310
|
-
) do
|
|
1311
|
-
base, edition, year, month = Regexp.last_match.captures
|
|
1312
|
-
date = year ? " #{year}#{month ? "-#{month}" : ''}" : ""
|
|
1313
|
-
"#{base} Edition #{edition}.0#{date}"
|
|
1314
|
-
end
|
|
1315
|
-
|
|
1316
|
-
# /R-N revisions: PRESERVE them (the grammar's revision_suffix rule now
|
|
1317
|
-
# captures a trailing "/R-<x>" into the `revision` attribute). Just
|
|
1318
|
-
# reposition any trailing publication year onto the number, keeping the
|
|
1319
|
-
# "/R-<x>" in place for the grammar.
|
|
1320
|
-
cleaned.sub(
|
|
1321
|
-
%r{\A(.*?)/R-([0-9A-Za-z]+)(?:-((?:19|20)\d\d))?\z},
|
|
1322
|
-
) do
|
|
1323
|
-
base, rev, year = Regexp.last_match.captures
|
|
1324
|
-
"#{year ? "#{base}-#{year}" : base}/R-#{rev}"
|
|
1325
|
-
end
|
|
1326
|
-
end
|
|
1327
|
-
|
|
1328
|
-
# Strip the IEEE rawbib revision-notation dialects. `REV`/`Rev`
|
|
1329
|
-
# (case-insensitive) + a trailing revision id `[A-Za-z0-9]+`, glued to the
|
|
1330
|
-
# number or separated by `-`, `/`, `_`, `.`, or a space, and preceding the
|
|
1331
|
-
# draft. pubid's canonical "<num>/D<n>/R-<x>" form already drops the
|
|
1332
|
-
# revision on render (normalize_relaton_suffixes strips a trailing /R-x),
|
|
1333
|
-
# so the revision-less result is *the same identifier* — and stripping
|
|
1334
|
-
# (rather than reordering) leaves any trailing date/parenthetical intact,
|
|
1335
|
-
# which is why forms that already parse (`Draft P…-REVmb/D3.0, Mar 2010`)
|
|
1336
|
-
# are NOT disturbed. Examples:
|
|
1337
|
-
# "P802.16.2-REVa/D8" -> "P802.16.2/D8"
|
|
1338
|
-
# "P802.16/REVd/D5" -> "P802.16/D5"
|
|
1339
|
-
# "P802.15.1REVa/D5" -> "P802.15.1/D5"
|
|
1340
|
-
# "P802.11REVmb" -> "P802.11" (no draft)
|
|
1341
|
-
def self.normalize_revision_notation(cleaned)
|
|
1342
|
-
# NUMBERED revisions ("Rev<digits>") are PRESERVED — repositioned to a
|
|
1343
|
-
# trailing "/R-<n>" suffix the grammar captures as the `revision`
|
|
1344
|
-
# attribute (IEEE's native inline spelling; numbered-revision hand-off).
|
|
1345
|
-
# A "\d+" right after "Rev" both selects the numbered subset and keeps
|
|
1346
|
-
# these off the English word "Revision". Three source positions:
|
|
1347
|
-
# after a draft : "PC37.30.2/D043 Rev 18" -> ".../D043/R-18"
|
|
1348
|
-
# The draft captures accept relaton's hyphenated "/D-<n>" spelling
|
|
1349
|
-
# (pubid#316): without the "-?" the before-draft regex matched only
|
|
1350
|
-
# "/D" of "/D-3" and emitted the garbage "/D/R-2-3-2008-02".
|
|
1351
|
-
cleaned = cleaned.sub(
|
|
1352
|
-
%r{(/D-?[0-9A-Za-z.]+)\s+[Rr][Ee][Vv]\s*(\d+)}, '\1/R-\2'
|
|
1353
|
-
)
|
|
1354
|
-
# before a draft: "P802.16Rev2/D3" -> "P802.16/D3/R-2"
|
|
1355
|
-
cleaned = cleaned.sub(
|
|
1356
|
-
%r{[-/_.]?\s?[Rr][Ee][Vv][-\s]?(\d+)(/D-?[0-9A-Za-z.]+)}, '\2/R-\1'
|
|
1357
|
-
)
|
|
1358
|
-
# no draft, trailing: "P1722-rev1" -> "P1722/R-1"
|
|
1359
|
-
cleaned = cleaned.sub(
|
|
1360
|
-
%r{(\d)[-._]?\s?[Rr][Ee][Vv]\s*(\d+)\s*\z}, '\1/R-\2'
|
|
1361
|
-
)
|
|
1362
|
-
|
|
1363
|
-
# LETTERED inline revisions ("REVa", "REVmb") have no pubid model and are
|
|
1364
|
-
# still STRIPPED (unchanged behaviour). The numbered forms above already
|
|
1365
|
-
# became "/R-<n>", so these regexes only see the lettered residue.
|
|
1366
|
-
# Revision token that PRECEDES a draft: drop it (keep the /D…).
|
|
1367
|
-
cleaned = cleaned.sub(
|
|
1368
|
-
%r{[-/_.]?\s?[Rr][Ee][Vv][-\s]?[A-Za-z0-9]+(?=/D-?[0-9A-Za-z])},
|
|
1369
|
-
"",
|
|
1370
|
-
)
|
|
1371
|
-
# Trailing revision glued to the number with no draft ("P802.11REVmb");
|
|
1372
|
-
# a digit must immediately precede REV so a trailing English word like
|
|
1373
|
-
# "…Revision" can't match.
|
|
1374
|
-
cleaned.sub(%r{(\d)[Rr][Ee][Vv][A-Za-z0-9]+\s*\z}, '\1')
|
|
1375
|
-
end
|
|
1376
|
-
|
|
1377
|
-
# Rewrite the mechanical spellings of the joint ISO-stage forms that no
|
|
1378
|
-
# grammar branch reaches (pubid#216 residuals). Two safety rules keep
|
|
1379
|
-
# this from stealing inputs other rules already parse:
|
|
1380
|
-
# 1. only BROKEN separators fire — underscore/space glue between
|
|
1381
|
-
# number, draft and stage, a slash-SPACE before the stage, a
|
|
1382
|
-
# dash/underscore month after it. A plain "/FDIS" tail is
|
|
1383
|
-
# natively accepted by ieee_p_identifier (fdraft) and the
|
|
1384
|
-
# stage-LAST embedded rule, so it is never rewritten.
|
|
1385
|
-
# 2. the number carries at most ONE part — joint_development_iso_format
|
|
1386
|
-
# reads a single optional part, while ieee_p_identifier takes
|
|
1387
|
-
# multi-part numbers ("P62271-37-013"); a rewrite that moved a
|
|
1388
|
-
# multi-part number onto the joint rule would break it.
|
|
1389
|
-
def self.normalize_joint_stage_spellings(cleaned)
|
|
1390
|
-
# The separator (slash-optional-space or a space) is INSIDE the
|
|
1391
|
-
# capture so a rewrite re-emits it — an uncaptured separator was
|
|
1392
|
-
# silently eaten, gluing the publisher to the stage.
|
|
1393
|
-
pubs = %r{((?:ISO/IEC/IEEE|IEEE/ISO/IEC|IEEE/IEC/ISO|ISO/IEEE|IEC/IEEE|IEEE/IEC|ISO/IEC|IEEE)(?:/ ?| ))}
|
|
1394
|
-
stage = /(FDIS|FCD|CDV|DIS\d?|CD\d?|WD|PWI|NP)/
|
|
1395
|
-
num = /(P?\d+(?:[.-]\d+)?)/
|
|
1396
|
-
# What the joint grammar can finish reading AFTER the stage: the end
|
|
1397
|
-
# of the string, a (comma-)month-year date, or a dash-year[-month].
|
|
1398
|
-
# Captured and re-emitted so a rewrite never drops the date.
|
|
1399
|
-
tail = /(,? (?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Sept|Oct|Nov|Dec)[a-z]* \d{4}|\z|-\d{4}(?:-\d\d)?)/
|
|
1400
|
-
|
|
1401
|
-
# Draft glued to the number with an underscore: "P15289_D3" →
|
|
1402
|
-
# "P15289/D3" (a digit must precede the underscore; "_FDIS" is the
|
|
1403
|
-
# stage, not a draft, and is already slash-rewritten earlier).
|
|
1404
|
-
cleaned = cleaned.gsub(%r{(\d)_D(\d)}, '\1/D\2')
|
|
1405
|
-
|
|
1406
|
-
# Crawl typo: a comma with no space before a month (",March 2021").
|
|
1407
|
-
# In these identifiers a comma is always a date separator, so
|
|
1408
|
-
# spacing it is safe.
|
|
1409
|
-
cleaned = cleaned.sub(/,(?=[A-Za-z])/, ", ")
|
|
1410
|
-
|
|
1411
|
-
# Stage AFTER the draft, both glued with underscores (the early
|
|
1412
|
-
# stage gsub already turned "_CD" into "/CD"):
|
|
1413
|
-
# "…P24748-3/D3/FDIS, April 2020 (E)" → "…FDIS P24748-3/D3, …"
|
|
1414
|
-
cleaned = cleaned.sub(
|
|
1415
|
-
%r{\A#{pubs}#{num}/(D[\d.]+)/ ?#{stage}\b},
|
|
1416
|
-
'\1\4 \2/\3',
|
|
1417
|
-
)
|
|
1418
|
-
|
|
1419
|
-
# Slash-SPACE before the stage ("…P16085/ FDIS, August 2020" —
|
|
1420
|
-
# ieee_p's fdraft has no inner space, so this spelling has no other
|
|
1421
|
-
# home): → "…FDIS P16085, August 2020"
|
|
1422
|
-
cleaned = cleaned.sub(
|
|
1423
|
-
%r{\A#{pubs}#{num}/ #{stage}#{tail}},
|
|
1424
|
-
'\1\3 \2\4',
|
|
1425
|
-
)
|
|
1426
|
-
|
|
1427
|
-
# A space-year after a slash-stage ("…P26511.2/FDIS 2018" — no rule
|
|
1428
|
-
# reads "/FDIS 2018"): → "…FDIS P26511.2-2018" (dash-year clause).
|
|
1429
|
-
cleaned = cleaned.sub(
|
|
1430
|
-
%r{\A#{pubs}#{num}/#{stage} (\d{4})\b},
|
|
1431
|
-
'\1\3 \2-\4',
|
|
1432
|
-
)
|
|
1433
|
-
|
|
1434
|
-
# Stage separated from the number by a SPACE ("IEC/IEEE P63113 CD4,
|
|
1435
|
-
# April 2019"): → "IEC/IEEE CD4 P63113, April 2019". The tail guard
|
|
1436
|
-
# keeps "…FDIS March 2019" (space-month-year, readable after the
|
|
1437
|
-
# swap) while leaving prose suffixes alone.
|
|
1438
|
-
cleaned = cleaned.sub(
|
|
1439
|
-
%r{\A#{pubs}#{num} #{stage}#{tail}},
|
|
1440
|
-
'\1\3 \2\4',
|
|
1441
|
-
)
|
|
1442
|
-
|
|
1443
|
-
# The same space-stage with a space-DRAFT chained after it
|
|
1444
|
-
# ("IEC/IEEE P60980-344 CDV D1, June 2019"): → stage-first with a
|
|
1445
|
-
# slash draft ("…CDV P60980-344/D1, June 2019").
|
|
1446
|
-
cleaned = cleaned.sub(
|
|
1447
|
-
%r{\A#{pubs}#{num} #{stage} (D\d+)\b},
|
|
1448
|
-
'\1\3 \2/\4',
|
|
1449
|
-
)
|
|
1450
|
-
|
|
1451
|
-
# Slash-stage carrying a dash-date ("…P15288/CD2-2013-09 …" — the
|
|
1452
|
-
# dash-year tail is what the joint grammar reads after the number, so
|
|
1453
|
-
# this cannot steal the plain "/FDIS, June 2021" spellings
|
|
1454
|
-
# ieee_p_identifier finishes): → "…CD2 P15288-2013-09 …". The date
|
|
1455
|
-
# must be a plausible year — a 4-digit monthcode (YYMM, e.g.
|
|
1456
|
-
# "…P24748-4/DIS-1404" = April 2014) is a DRAFT designator, not a
|
|
1457
|
-
# dash-date, and rewriting it onto the number breaks the parse.
|
|
1458
|
-
cleaned = cleaned.sub(
|
|
1459
|
-
%r{\A#{pubs}#{num}/ ?#{stage}(-(?:19|20)\d\d(?:-\d\d)?)},
|
|
1460
|
-
'\1\3 \2\4',
|
|
1461
|
-
)
|
|
1462
|
-
|
|
1463
|
-
# Draft separated from a stage-first number by a SPACE:
|
|
1464
|
-
# "…DIS P11073-10418 D13, January 2011" → "…DIS P11073-10418/D13, …"
|
|
1465
|
-
cleaned = cleaned.sub(
|
|
1466
|
-
%r{\A#{pubs}#{stage} #{num}/? ?(D\d+)\b},
|
|
1467
|
-
'\1\2 \3/\4',
|
|
1468
|
-
)
|
|
1469
|
-
|
|
1470
|
-
# NOT rewritten: a month glued to the stage with a dash/underscore
|
|
1471
|
-
# ("…/FDIS_Dec 2012", "…/FDIS-Dec 2015"). Even routed onto the joint
|
|
1472
|
-
# grammar, the joint render of a month+draft is itself not yet
|
|
1473
|
-
# idempotent, and the bare-IEEE fdraft path drops the stage and the
|
|
1474
|
-
# date from to_hash — rewriting would trade a visible parse failure
|
|
1475
|
-
# for silent garbage. Deferred with the pubid#216 semantic residue.
|
|
1476
|
-
cleaned
|
|
1477
|
-
end
|
|
1478
|
-
|
|
1479
|
-
# Pre-parse ingestion normalizations (R2): every parse path —
|
|
1480
|
-
# parslet and PG artifact alike — feeds the grammar the same
|
|
1481
|
-
# normalized string.
|
|
1482
1244
|
def self.normalize_input(string)
|
|
1483
1245
|
# Strip .pdf extension if present (Pattern 3: File Extensions)
|
|
1484
1246
|
cleaned = string.sub(/\.pdf$/i, "")
|
|
@@ -1492,14 +1254,6 @@ module Pubid
|
|
|
1492
1254
|
# No valid IEEE identifier pattern needs more than 1 space
|
|
1493
1255
|
cleaned = cleaned.gsub(/\s+/, " ")
|
|
1494
1256
|
|
|
1495
|
-
# Rewrite the rawbib revision-notation dialects (REVa/REVd/glued) into
|
|
1496
|
-
# the canonical /R-<x> form before the suffix normalization below.
|
|
1497
|
-
cleaned = normalize_revision_notation(cleaned)
|
|
1498
|
-
|
|
1499
|
-
# Normalize relaton's bespoke historical serialization (the spellings
|
|
1500
|
-
# emitted by Relaton::Ieee::PubId::Id#to_s) into canonical pubid forms
|
|
1501
|
-
# so `relaton-data-ieee` parses. See #normalize_relaton_suffixes.
|
|
1502
|
-
cleaned = normalize_relaton_suffixes(cleaned)
|
|
1503
1257
|
|
|
1504
1258
|
# NEW Session 171: CONSERVATIVE data quality fixes for TODO.IEEE-MUST-DO.txt
|
|
1505
1259
|
# Only fix clear typos: space before dash + 4-digit year, OR dash + space + 4-digit year
|
|
@@ -1538,11 +1292,6 @@ module Pubid
|
|
|
1538
1292
|
/\b(Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Sept|Oct|Nov|Dec)(\d{4})\b/, '\1 \2'
|
|
1539
1293
|
)
|
|
1540
1294
|
|
|
1541
|
-
# Rewrite the joint ISO-stage glued spellings (pubid#216) AFTER the
|
|
1542
|
-
# month unglue above, so "Dec2015" is already spaced when the
|
|
1543
|
-
# dash-month rewrite looks for it.
|
|
1544
|
-
cleaned = normalize_joint_stage_spellings(cleaned)
|
|
1545
|
-
|
|
1546
1295
|
# NEW: Convert IEC/IEEE space-separated to semicolon format
|
|
1547
1296
|
# Pattern: "IEC 61523-3 First edition 2004-09; IEEE 1497" → already semicolon
|
|
1548
1297
|
# Pattern: "IEC 62539 First Edition 2007-07 IEEE 930" → needs semicolon
|
data/lib/pubid/version.rb
CHANGED