relaton 3.0.0.pre.alpha.1 → 3.0.0.pre.alpha.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.adoc +57 -1
- data/bin/console +0 -1
- data/lib/relaton/3gpp/bibliography.rb +82 -7
- data/lib/relaton/3gpp/data_fetcher.rb +51 -3
- data/lib/relaton/3gpp/docidentifier.rb +114 -0
- data/lib/relaton/3gpp/item.rb +6 -0
- data/lib/relaton/3gpp/parser.rb +1 -1
- data/lib/relaton/3gpp/processor.rb +4 -1
- data/lib/relaton/3gpp.rb +5 -1
- data/lib/relaton/adobe/bibdata.rb +8 -0
- data/lib/relaton/adobe/bibitem.rb +8 -0
- data/lib/relaton/adobe/bibliography.rb +92 -0
- data/lib/relaton/adobe/docidentifier.rb +49 -0
- data/lib/relaton/adobe/doctype.rb +14 -0
- data/lib/relaton/adobe/ext.rb +32 -0
- data/lib/relaton/adobe/item.rb +15 -0
- data/lib/relaton/adobe/item_base.rb +18 -0
- data/lib/relaton/adobe/item_data.rb +6 -0
- data/lib/relaton/adobe/processor.rb +45 -0
- data/lib/relaton/adobe/util.rb +8 -0
- data/lib/relaton/adobe.rb +37 -0
- data/lib/relaton/bib/model/address.rb +2 -2
- data/lib/relaton/bib/model/docidentifier.rb +24 -9
- data/lib/relaton/bib/model/localized_string.rb +1 -1
- data/lib/relaton/bib/model/structured_identifier.rb +10 -9
- data/lib/relaton/bib/sanitizer.rb +202 -6
- data/lib/relaton/bib.rb +0 -2
- data/lib/relaton/bipm/bibliography.rb +159 -10
- data/lib/relaton/bipm/data_fetcher.rb +26 -2
- data/lib/relaton/bipm/data_outcomes_parser.rb +3 -4
- data/lib/relaton/bipm/id_parser.rb +5 -4
- data/lib/relaton/bipm/model/structured_identifier.rb +21 -0
- data/lib/relaton/bipm/processor.rb +2 -2
- data/lib/relaton/bipm/rawdata_bipm_metrologia/fetcher.rb +2 -4
- data/lib/relaton/bipm/si_brochure_parser.rb +5 -3
- data/lib/relaton/bipm.rb +7 -1
- data/lib/relaton/bsi/bibliography.rb +115 -43
- data/lib/relaton/bsi/hit.rb +14 -0
- data/lib/relaton/bsi/hit_collection.rb +15 -16
- data/lib/relaton/bsi/model/docidentifier.rb +99 -1
- data/lib/relaton/bsi/processor.rb +1 -0
- data/lib/relaton/calconnect/bibliography.rb +12 -14
- data/lib/relaton/calconnect/data_fetcher.rb +77 -9
- data/lib/relaton/calconnect/docidentifier.rb +80 -0
- data/lib/relaton/calconnect/hit_collection.rb +65 -57
- data/lib/relaton/calconnect/model/item.rb +7 -0
- data/lib/relaton/calconnect/processor.rb +7 -1
- data/lib/relaton/calconnect.rb +11 -1
- data/lib/relaton/ccsds/data/fetcher.rb +17 -12
- data/lib/relaton/ccsds/data/parser.rb +1 -1
- data/lib/relaton/ccsds/hit_collection.rb +6 -1
- data/lib/relaton/ccsds/model/docidentifier.rb +121 -0
- data/lib/relaton/ccsds/model/item.rb +2 -0
- data/lib/relaton/cen/bibliography.rb +75 -47
- data/lib/relaton/cen/hit.rb +16 -1
- data/lib/relaton/cen/hit_collection.rb +59 -11
- data/lib/relaton/cen/model/docidentifier.rb +92 -1
- data/lib/relaton/cen/processor.rb +13 -8
- data/lib/relaton/cen/scraper.rb +13 -5
- data/lib/relaton/cen.rb +1 -0
- data/lib/relaton/cie/data_fetcher.rb +215 -30
- data/lib/relaton/cie/processor.rb +3 -1
- data/lib/relaton/cie/scrapper.rb +15 -2
- data/lib/relaton/cie.rb +2 -1
- data/lib/relaton/core/data_fetcher.rb +150 -3
- data/lib/relaton/core/governor.rb +320 -0
- data/lib/relaton/core/pacer.rb +134 -0
- data/lib/relaton/core/processor.rb +19 -0
- data/lib/relaton/core/request_error.rb +14 -0
- data/lib/relaton/core.rb +3 -0
- data/lib/relaton/db/registry.rb +41 -1
- data/lib/relaton/doi/crossref.rb +19 -2
- data/lib/relaton/doi/parser.rb +109 -15
- data/lib/relaton/easc/bibdata.rb +8 -0
- data/lib/relaton/easc/bibitem.rb +8 -0
- data/lib/relaton/easc/bibliography.rb +95 -0
- data/lib/relaton/easc/docidentifier.rb +100 -0
- data/lib/relaton/easc/doctype.rb +14 -0
- data/lib/relaton/easc/ext.rb +44 -0
- data/lib/relaton/easc/item.rb +13 -0
- data/lib/relaton/easc/item_base.rb +18 -0
- data/lib/relaton/easc/item_data.rb +6 -0
- data/lib/relaton/easc/processor.rb +46 -0
- data/lib/relaton/easc/util.rb +8 -0
- data/lib/relaton/easc.rb +35 -0
- data/lib/relaton/ecma/bibliography.rb +93 -25
- data/lib/relaton/ecma/data_fetcher.rb +71 -12
- data/lib/relaton/ecma/docidentifier.rb +124 -0
- data/lib/relaton/ecma/item.rb +2 -0
- data/lib/relaton/ecma/memento_parser.rb +1 -1
- data/lib/relaton/ecma/page_fetcher.rb +15 -3
- data/lib/relaton/ecma/parser_common.rb +2 -2
- data/lib/relaton/ecma/processor.rb +4 -1
- data/lib/relaton/ecma/standard_parser.rb +2 -2
- data/lib/relaton/ecma.rb +10 -1
- data/lib/relaton/etsi/bibliography.rb +67 -2
- data/lib/relaton/etsi/data_fetcher.rb +43 -4
- data/lib/relaton/etsi/processor.rb +3 -1
- data/lib/relaton/etsi.rb +2 -1
- data/lib/relaton/gb/bibliography.rb +55 -29
- data/lib/relaton/gb/docidentifier.rb +58 -9
- data/lib/relaton/gb/processor.rb +3 -0
- data/lib/relaton/gb/scraper.rb +27 -10
- data/lib/relaton/gost/bibdata.rb +8 -0
- data/lib/relaton/gost/bibitem.rb +8 -0
- data/lib/relaton/gost/bibliography.rb +107 -0
- data/lib/relaton/gost/docidentifier.rb +80 -0
- data/lib/relaton/gost/doctype.rb +16 -0
- data/lib/relaton/gost/ext.rb +46 -0
- data/lib/relaton/gost/item.rb +15 -0
- data/lib/relaton/gost/item_base.rb +18 -0
- data/lib/relaton/gost/item_data.rb +6 -0
- data/lib/relaton/gost/processor.rb +49 -0
- data/lib/relaton/gost/util.rb +8 -0
- data/lib/relaton/gost.rb +36 -0
- data/lib/relaton/iala/bibdata.rb +8 -0
- data/lib/relaton/iala/bibitem.rb +8 -0
- data/lib/relaton/iala/bibliography.rb +146 -0
- data/lib/relaton/iala/docidentifier.rb +89 -0
- data/lib/relaton/iala/doctype.rb +18 -0
- data/lib/relaton/iala/ext.rb +32 -0
- data/lib/relaton/iala/item.rb +21 -0
- data/lib/relaton/iala/item_base.rb +18 -0
- data/lib/relaton/iala/item_data.rb +6 -0
- data/lib/relaton/iala/processor.rb +43 -0
- data/lib/relaton/iala/relation.rb +7 -0
- data/lib/relaton/iala/util.rb +8 -0
- data/lib/relaton/iala.rb +35 -0
- data/lib/relaton/iana/bibliography.rb +67 -14
- data/lib/relaton/iana/data_fetcher.rb +35 -5
- data/lib/relaton/iana/processor.rb +3 -1
- data/lib/relaton/iana.rb +12 -1
- data/lib/relaton/iec/data_fetcher.rb +7 -1
- data/lib/relaton/iec/hit_collection.rb +1 -1
- data/lib/relaton/iec/model/docidentifier.rb +9 -5
- data/lib/relaton/iec/model/ext.rb +2 -2
- data/lib/relaton/iec/processor.rb +1 -0
- data/lib/relaton/ieee/bibliography.rb +25 -3
- data/lib/relaton/ieee/data_fetcher.rb +158 -17
- data/lib/relaton/ieee/idams_parser.rb +18 -11
- data/lib/relaton/ieee/processor.rb +4 -1
- data/lib/relaton/ieee/rawbib_id_parser.rb +291 -86
- data/lib/relaton/ieee.rb +2 -1
- data/lib/relaton/ietf/data_fetcher.rb +295 -12
- data/lib/relaton/ietf/processor.rb +7 -3
- data/lib/relaton/ietf/rfc/entry.rb +39 -3
- data/lib/relaton/ietf/scraper.rb +69 -36
- data/lib/relaton/ietf.rb +4 -1
- data/lib/relaton/iho/bibliography.rb +1 -1
- data/lib/relaton/iho/docidentifier.rb +1 -1
- data/lib/relaton/index/file_io.rb +11 -11
- data/lib/relaton/index/file_storage.rb +6 -1
- data/lib/relaton/index/pool.rb +6 -1
- data/lib/relaton/index/shard_source.rb +201 -0
- data/lib/relaton/index/type.rb +63 -12
- data/lib/relaton/index.rb +2 -1
- data/lib/relaton/iso/bibliography.rb +20 -15
- data/lib/relaton/iso/data_fetcher.rb +3 -3
- data/lib/relaton/iso/data_parser.rb +17 -3
- data/lib/relaton/iso/hit_collection.rb +27 -15
- data/lib/relaton/iso/item_data.rb +22 -0
- data/lib/relaton/iso/model/docidentifier.rb +24 -12
- data/lib/relaton/iso/processor.rb +1 -0
- data/lib/relaton/iso/scraper.rb +19 -3
- data/lib/relaton/itu/bibliography.rb +9 -4
- data/lib/relaton/itu/data_crawler_r.rb +664 -0
- data/lib/relaton/itu/data_fetcher.rb +496 -50
- data/lib/relaton/itu/data_merge_r.rb +149 -0
- data/lib/relaton/itu/data_parser_r.rb +163 -89
- data/lib/relaton/itu/data_parser_t.rb +228 -0
- data/lib/relaton/itu/family_cache.rb +177 -0
- data/lib/relaton/itu/governor.rb +56 -0
- data/lib/relaton/itu/hit.rb +9 -3
- data/lib/relaton/itu/hit_collection.rb +258 -86
- data/lib/relaton/itu/model/docidentifier.rb +67 -1
- data/lib/relaton/itu/model/structured_identifier.rb +19 -0
- data/lib/relaton/itu/processor.rb +10 -4
- data/lib/relaton/itu/pubid.rb +27 -5
- data/lib/relaton/itu/recommendation_fields.rb +334 -0
- data/lib/relaton/itu/recommendation_parser.rb +18 -149
- data/lib/relaton/itu/scraper.rb +13 -3
- data/lib/relaton/itu.rb +2 -1
- data/lib/relaton/jcgm/bibdata.rb +8 -0
- data/lib/relaton/jcgm/bibitem.rb +8 -0
- data/lib/relaton/jcgm/bibliography.rb +97 -0
- data/lib/relaton/jcgm/data_fetcher.rb +81 -0
- data/lib/relaton/jcgm/docidentifier.rb +102 -0
- data/lib/relaton/jcgm/doctype.rb +12 -0
- data/lib/relaton/jcgm/ext.rb +23 -0
- data/lib/relaton/jcgm/item.rb +20 -0
- data/lib/relaton/jcgm/item_base.rb +18 -0
- data/lib/relaton/jcgm/item_data.rb +6 -0
- data/lib/relaton/jcgm/meetings_parser.rb +175 -0
- data/lib/relaton/jcgm/processor.rb +71 -0
- data/lib/relaton/jcgm/relation.rb +9 -0
- data/lib/relaton/jcgm/structured_identifier.rb +40 -0
- data/lib/relaton/jcgm/util.rb +8 -0
- data/lib/relaton/jcgm.rb +24 -0
- data/lib/relaton/jis/bibliography.rb +8 -10
- data/lib/relaton/jis/data_fetcher.rb +21 -19
- data/lib/relaton/jis/docidentifier.rb +104 -5
- data/lib/relaton/jis/hit.rb +18 -23
- data/lib/relaton/jis/hit_collection.rb +19 -18
- data/lib/relaton/jis/processor.rb +1 -1
- data/lib/relaton/jis.rb +2 -3
- data/lib/relaton/logger/channels/gh_issue.rb +78 -13
- data/lib/relaton/nist/data_fetcher.rb +63 -13
- data/lib/relaton/nist/docidentifier.rb +165 -0
- data/lib/relaton/nist/item.rb +2 -0
- data/lib/relaton/nist/item_base.rb +16 -0
- data/lib/relaton/nist/mods_parser.rb +38 -12
- data/lib/relaton/nist/processor.rb +2 -1
- data/lib/relaton/nist/relation.rb +3 -0
- data/lib/relaton/nist/scraper.rb +6 -3
- data/lib/relaton/oasis/bibliography.rb +147 -6
- data/lib/relaton/oasis/data_fetcher.rb +41 -5
- data/lib/relaton/oasis/data_parser_utils.rb +37 -3
- data/lib/relaton/oasis/docidentifier.rb +54 -0
- data/lib/relaton/oasis/item.rb +3 -0
- data/lib/relaton/oasis/processor.rb +7 -1
- data/lib/relaton/oasis.rb +14 -1
- data/lib/relaton/ogc/data_fetcher.rb +23 -2
- data/lib/relaton/ogc/docidentifier.rb +105 -0
- data/lib/relaton/ogc/hit_collection.rb +78 -3
- data/lib/relaton/ogc/processor.rb +2 -1
- data/lib/relaton/ogc.rb +5 -1
- data/lib/relaton/oiml/bibliography.rb +90 -15
- data/lib/relaton/oiml/docidentifier.rb +18 -3
- data/lib/relaton/omg/docidentifier.rb +67 -0
- data/lib/relaton/omg/item.rb +1 -0
- data/lib/relaton/omg/processor.rb +1 -0
- data/lib/relaton/omg/scraper.rb +61 -16
- data/lib/relaton/omg.rb +1 -0
- data/lib/relaton/plateau/bibliography.rb +10 -3
- data/lib/relaton/plateau/data_fetcher.rb +25 -2
- data/lib/relaton/plateau/handbook_parser.rb +8 -1
- data/lib/relaton/plateau/hit.rb +10 -2
- data/lib/relaton/plateau/hit_collection.rb +31 -11
- data/lib/relaton/plateau/processor.rb +3 -1
- data/lib/relaton/plateau/technical_report_parser.rb +8 -1
- data/lib/relaton/plateau.rb +2 -1
- data/lib/relaton/sdo/config.rb +34 -0
- data/lib/relaton/sdo/fetcher.rb +52 -0
- data/lib/relaton/sdo/logo.rb +95 -0
- data/lib/relaton/sdo/name.rb +26 -0
- data/lib/relaton/sdo/organization.rb +71 -0
- data/lib/relaton/sdo/store.rb +49 -0
- data/lib/relaton/sdo.rb +29 -0
- data/lib/relaton/version.rb +1 -1
- data/lib/relaton/w3c/bibliography.rb +132 -12
- data/lib/relaton/w3c/data_fetcher.rb +194 -16
- data/lib/relaton/w3c/data_parser.rb +3 -3
- data/lib/relaton/w3c/docidentifier.rb +48 -0
- data/lib/relaton/w3c/governor.rb +32 -0
- data/lib/relaton/w3c/item.rb +3 -0
- data/lib/relaton/w3c/pubid.rb +12 -0
- data/lib/relaton/w3c/safe_realize.rb +110 -21
- data/lib/relaton/w3c.rb +12 -1
- data/lib/relaton/xsf/bibliography.rb +61 -1
- data/lib/relaton/xsf/data_fetcher.rb +55 -5
- data/lib/relaton/xsf/docidentifier.rb +46 -0
- data/lib/relaton/xsf/hit_collection.rb +31 -3
- data/lib/relaton/xsf/item.rb +6 -0
- data/lib/relaton/xsf/processor.rb +1 -0
- data/lib/relaton/xsf.rb +5 -1
- data/lib/relaton.rb +42 -0
- metadata +135 -24
- data/lib/relaton/ieee/pub_id.rb +0 -161
- data/lib/relaton/index/id_number.rb +0 -30
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
require "etc"
|
|
2
2
|
require "parallel"
|
|
3
|
+
require "pubid"
|
|
4
|
+
require "pubid/ietf"
|
|
3
5
|
require "relaton/core"
|
|
4
6
|
require_relative "../ietf"
|
|
5
7
|
require_relative "bibxml_parser"
|
|
@@ -21,12 +23,30 @@ module Relaton
|
|
|
21
23
|
when "ietf-rfc-entries" then fetch_ieft_rfcs
|
|
22
24
|
end
|
|
23
25
|
index.save
|
|
26
|
+
report_unindexed
|
|
27
|
+
report_unparsed
|
|
28
|
+
report_collisions
|
|
24
29
|
end
|
|
25
30
|
|
|
26
31
|
private
|
|
27
32
|
|
|
33
|
+
# The published index is the pubid-structured `index-v2` (relaton#109).
|
|
34
|
+
# `pubid_class:` is not decoration: `FileIO#save` serialises an id to its
|
|
35
|
+
# `_type:` hash only when it is an instance of the configured class, so
|
|
36
|
+
# without it — or without parsing the id below — this writes a v1-shaped
|
|
37
|
+
# file under a v2 name.
|
|
38
|
+
# `url: nil` is load-bearing, not decoration. Scraper opens the same
|
|
39
|
+
# `:IETF` pool key with a `url:`, and `Type#actual?` skips the URL check
|
|
40
|
+
# when the caller omits it (`!args.key?(:url)`) — so in a process where a
|
|
41
|
+
# lookup ran first, omitting it here would hand the crawl the
|
|
42
|
+
# remote-backed Type and `save` would write to `~/.relaton/ietf/` instead
|
|
43
|
+
# of `./`, publishing no index at all. Passing it explicitly forces a
|
|
44
|
+
# local-file Type.
|
|
28
45
|
def index
|
|
29
|
-
@index ||= Relaton::Index.find_or_create
|
|
46
|
+
@index ||= Relaton::Index.find_or_create(
|
|
47
|
+
:IETF, url: nil, file: "#{INDEXFILE}.yaml",
|
|
48
|
+
pubid_class: ::Pubid::Ietf::Identifier
|
|
49
|
+
)
|
|
30
50
|
end
|
|
31
51
|
|
|
32
52
|
#
|
|
@@ -34,8 +54,16 @@ module Relaton
|
|
|
34
54
|
#
|
|
35
55
|
def fetch_ieft_rfcsubseries
|
|
36
56
|
idx = Rfc::Index.from_xml(rfc_index)
|
|
57
|
+
# Keyed by the normalised doc-id, built once for the whole crawl:
|
|
58
|
+
# `Entry` looks constituents up by `Entry.squish(ref)`, and doing it
|
|
59
|
+
# per-entry over ~9,800 RFCs would rebuild this table 367 times.
|
|
37
60
|
rfc_map = (idx.rfc_entries || []).each_with_object({}) do |entry, h|
|
|
38
|
-
|
|
61
|
+
key = Rfc::Entry.squish(entry.doc_id)
|
|
62
|
+
if h.key?(key)
|
|
63
|
+
Util.warn "Duplicate RFC doc-id `#{entry.doc_id}` after normalisation " \
|
|
64
|
+
"(`#{key}`); the later entry wins for constituent lookup"
|
|
65
|
+
end
|
|
66
|
+
h[key] = entry
|
|
39
67
|
end
|
|
40
68
|
idx.subseries_entries.each do |entry|
|
|
41
69
|
save_doc entry.to_item(rfc_map, wg_names: wg_names)
|
|
@@ -54,6 +82,10 @@ module Relaton
|
|
|
54
82
|
#
|
|
55
83
|
def fetch_ieft_internet_drafts
|
|
56
84
|
series_groups, singleton_paths = group_draft_paths
|
|
85
|
+
# Workers fork from here, so `unique_output_file`'s reservation cannot
|
|
86
|
+
# see a peer's claim and `write_unique` must never overwrite. Set
|
|
87
|
+
# before the first fork, so the children inherit it.
|
|
88
|
+
@cross_process = true
|
|
57
89
|
|
|
58
90
|
series_results = parallelize(series_groups.to_a) do |(series, paths_info)|
|
|
59
91
|
process_series(series, paths_info)
|
|
@@ -63,7 +95,13 @@ module Relaton
|
|
|
63
95
|
process_singleton(path)
|
|
64
96
|
end
|
|
65
97
|
|
|
66
|
-
(series_results + singleton_results).compact
|
|
98
|
+
entries, unparsed = (series_results + singleton_results).compact
|
|
99
|
+
.partition { |r| r[:unparsed].nil? }
|
|
100
|
+
# Tallied here, not in the worker: a counter incremented in a Parallel
|
|
101
|
+
# worker process is lost on the way back (see record_index_entry).
|
|
102
|
+
@unparsed = unparsed.map { |r| "#{r[:unparsed]} (#{r[:error]})" }
|
|
103
|
+
reconcile_output_files entries
|
|
104
|
+
entries.each { |r| record_index_entry(r) }
|
|
67
105
|
end
|
|
68
106
|
|
|
69
107
|
#
|
|
@@ -114,16 +152,22 @@ module Relaton
|
|
|
114
152
|
# for the parent.
|
|
115
153
|
#
|
|
116
154
|
def process_series(series, paths_info)
|
|
117
|
-
|
|
118
|
-
bib =
|
|
155
|
+
parsed = paths_info.sort_by { |p| p[:ver].to_i }.map do |p|
|
|
156
|
+
bib, marker = parse_bibxml(p[:path])
|
|
157
|
+
next marker unless bib
|
|
158
|
+
|
|
119
159
|
bib.version = [Bib::Version.new(draft: p[:ver])]
|
|
120
160
|
p.merge(bib: bib, source: bib.source)
|
|
121
161
|
end
|
|
162
|
+
# A file that failed to parse must not reach `sorted`:
|
|
163
|
+
# link_neighbor_relations and build_unversioned_doc both dereference
|
|
164
|
+
# `entry[:bib]`, and a dropped version is better than a nil one.
|
|
165
|
+
sorted, skipped = parsed.partition { |e| e[:unparsed].nil? }
|
|
122
166
|
link_neighbor_relations(sorted) if @format != "bibxml"
|
|
123
167
|
|
|
124
168
|
results = sorted.map { |entry| serialize_and_write(entry[:bib]) }
|
|
125
169
|
results << serialize_and_write(build_unversioned_doc(series, sorted)) if @format != "bibxml"
|
|
126
|
-
results.compact
|
|
170
|
+
results.compact + skipped
|
|
127
171
|
end
|
|
128
172
|
|
|
129
173
|
#
|
|
@@ -133,11 +177,82 @@ module Relaton
|
|
|
133
177
|
file = File.basename(path, ".xml")
|
|
134
178
|
is_draft = file.include?("D.draft-")
|
|
135
179
|
ver = is_draft ? file[/(\d+)$/, 1] : nil
|
|
136
|
-
bib =
|
|
180
|
+
bib, marker = parse_bibxml(path)
|
|
181
|
+
return marker unless bib
|
|
182
|
+
|
|
137
183
|
bib.version = [Bib::Version.new(draft: ver)] if ver
|
|
138
184
|
serialize_and_write(bib)
|
|
139
185
|
end
|
|
140
186
|
|
|
187
|
+
#
|
|
188
|
+
# Read and parse one bibxml file, or nil if it cannot be parsed.
|
|
189
|
+
#
|
|
190
|
+
# Rescues `StandardError` rather than a narrow list on purpose: lutaml
|
|
191
|
+
# raises `InvalidFormatError` on bad bytes, but the converter also runs
|
|
192
|
+
# regexes over parsed text (`parse_surname_initials` and friends), and
|
|
193
|
+
# those raise `ArgumentError: invalid byte sequence` on anything that slips
|
|
194
|
+
# through. One unparseable file must cost one document, never the crawl —
|
|
195
|
+
# the drafts path runs under Parallel.map, which discards every result from
|
|
196
|
+
# the pass when a worker raises.
|
|
197
|
+
#
|
|
198
|
+
# @param path [String]
|
|
199
|
+
# @return [Relaton::Ietf::ItemData, nil]
|
|
200
|
+
#
|
|
201
|
+
# @param path [String]
|
|
202
|
+
# @return [Array(Relaton::Ietf::ItemData, nil), Array(nil, Hash)]
|
|
203
|
+
# the record, or nil plus a marker carrying why
|
|
204
|
+
def parse_bibxml(path)
|
|
205
|
+
bib = BibXMLParser.parse(read_bibxml(path))
|
|
206
|
+
bib ? [bib, nil] : [nil, unparsed_marker(path, "parser returned no record")]
|
|
207
|
+
rescue StandardError => e
|
|
208
|
+
[nil, unparsed_marker(path, "#{e.class}: #{e.message.to_s.lines.first.to_s.strip}")]
|
|
209
|
+
end
|
|
210
|
+
|
|
211
|
+
# Marshal-friendly stand-in for a record, carried back to the parent so the
|
|
212
|
+
# skip can be counted where a tally survives. The reason rides along rather
|
|
213
|
+
# than sitting in an ivar, which a later file would overwrite.
|
|
214
|
+
def unparsed_marker(path, error)
|
|
215
|
+
{ unparsed: path, error: error }
|
|
216
|
+
end
|
|
217
|
+
|
|
218
|
+
#
|
|
219
|
+
# Read a bibxml file as UTF-8, recovering Windows-1252 bytes.
|
|
220
|
+
#
|
|
221
|
+
# These files declare `encoding='UTF-8'` but some carry CP1252 — smart
|
|
222
|
+
# quotes and accented Latin letters. `File.read(encoding: "UTF-8")` only
|
|
223
|
+
# *tags* the string, so those reach lutaml as invalid UTF-8 and it raises.
|
|
224
|
+
#
|
|
225
|
+
# `scrub` with a block transcodes each invalid *run* as CP1252 while
|
|
226
|
+
# leaving valid UTF-8 untouched. Both halves matter:
|
|
227
|
+
#
|
|
228
|
+
# * Not plain `scrub`, which substitutes U+FFFD: `client’s` would become
|
|
229
|
+
# `client\uFFFDs` and `Muñoz` `Mu\uFFFDoz` — author surnames included.
|
|
230
|
+
# The damage is length-preserving, so a length check will not catch it.
|
|
231
|
+
# * Not a whole-file CP1252 re-decode, which mangles a file that is
|
|
232
|
+
# genuinely UTF-8 apart from one stray byte: `Muñoz café ’` would come
|
|
233
|
+
# back as `Muñoz café ’`, silently, since the result is valid UTF-8 and
|
|
234
|
+
# nothing raises. No such file exists in today's corpus (0 of the 125
|
|
235
|
+
# affected contain valid multi-byte UTF-8) but it grows daily, and
|
|
236
|
+
# "decodes losslessly as CP1252" is weak evidence of correctness —
|
|
237
|
+
# CP1252 maps 251 of 256 byte values.
|
|
238
|
+
#
|
|
239
|
+
# `undef: :replace` covers the five bytes CP1252 leaves undefined
|
|
240
|
+
# (0x81 0x8D 0x8F 0x90 0x9D), so this returns valid UTF-8 rather than
|
|
241
|
+
# raising and costing the whole file.
|
|
242
|
+
#
|
|
243
|
+
# @param path [String]
|
|
244
|
+
# @return [String] UTF-8, valid encoding
|
|
245
|
+
#
|
|
246
|
+
def read_bibxml(path)
|
|
247
|
+
utf8 = File.binread(path).force_encoding(Encoding::UTF_8)
|
|
248
|
+
return utf8 if utf8.valid_encoding?
|
|
249
|
+
|
|
250
|
+
utf8.scrub do |bad|
|
|
251
|
+
bad.force_encoding(Encoding::WINDOWS_1252)
|
|
252
|
+
.encode(Encoding::UTF_8, undef: :replace)
|
|
253
|
+
end
|
|
254
|
+
end
|
|
255
|
+
|
|
141
256
|
#
|
|
142
257
|
# Append immediate-neighbor `updates` / `updatedBy` relations in memory.
|
|
143
258
|
# Single-version series get no relations (no neighbors).
|
|
@@ -160,6 +275,12 @@ module Relaton
|
|
|
160
275
|
# `includes` relations to every version. Uses the latest version's
|
|
161
276
|
# title/abstract from memory.
|
|
162
277
|
#
|
|
278
|
+
# The aggregator is *synthesised* — there is no upstream document for it,
|
|
279
|
+
# so `date`, `ext` (hence doctype) and `source` can only be inherited from
|
|
280
|
+
# its newest constituent, which `sorted` already holds in memory. Without
|
|
281
|
+
# that inheritance it publishes undated (and so unsorted on the Pages
|
|
282
|
+
# index, which sorts by date) and with no document type at all.
|
|
283
|
+
#
|
|
163
284
|
# @return [Relaton::Ietf::ItemData, nil]
|
|
164
285
|
#
|
|
165
286
|
def build_unversioned_doc(series, sorted)
|
|
@@ -173,7 +294,11 @@ module Relaton
|
|
|
173
294
|
rel = sorted.map { |e| version_relation({ ref: e[:ref], source: e[:source] }, "includes") }
|
|
174
295
|
ItemData.new(
|
|
175
296
|
title: last_v.title, abstract: last_v.abstract, formattedref: Bib::Formattedref.new(content: series),
|
|
176
|
-
docidentifier: [docid], relation: rel
|
|
297
|
+
docidentifier: [docid], relation: rel,
|
|
298
|
+
# dup'd, not shared: these are the newest version's own objects, and
|
|
299
|
+
# aliasing them would make any later edit to the aggregator mutate the
|
|
300
|
+
# `-NN` record too.
|
|
301
|
+
date: last_v.date&.dup, ext: last_v.ext&.dup, source: last_v.source&.dup
|
|
177
302
|
)
|
|
178
303
|
end
|
|
179
304
|
|
|
@@ -254,10 +379,37 @@ module Relaton
|
|
|
254
379
|
entry.docidentifier.detect { |i| i.type == "Internet-Draft" && i.primary }&.content
|
|
255
380
|
end
|
|
256
381
|
id ||= entry.docnumber || entry.formattedref.content
|
|
257
|
-
file =
|
|
258
|
-
File.write file, content, encoding: "UTF-8"
|
|
382
|
+
file = write_unique(id, content)
|
|
259
383
|
primary = entry.docidentifier.detect(&:primary) || entry.docidentifier.first
|
|
260
|
-
|
|
384
|
+
# `docid` is the id the file was written under and `plain_file` the name
|
|
385
|
+
# it would have taken uncontested; `reconcile_output_files` needs both.
|
|
386
|
+
# Neither is `index_id`: `id` falls back through docnumber and
|
|
387
|
+
# formattedref, so on the RFC path the two differ ("RFC0001" vs "RFC 1").
|
|
388
|
+
{ docnumber: entry.docnumber, docid: id, file: file,
|
|
389
|
+
plain_file: output_file(id), index_id: primary.content,
|
|
390
|
+
pubid: parse_pubid(primary.content) }
|
|
391
|
+
end
|
|
392
|
+
|
|
393
|
+
#
|
|
394
|
+
# Parse a record's primary docidentifier into the pubid the index stores.
|
|
395
|
+
#
|
|
396
|
+
# Deliberately here rather than in `record_index_entry`: this runs inside
|
|
397
|
+
# the `Parallel` workers, and a pubid identifier survives the Marshal round
|
|
398
|
+
# trip Parallel does on the return value. Parsing in the parent instead
|
|
399
|
+
# would put ~0.7 ms per record back on the serial path — some minutes over
|
|
400
|
+
# the 167k-draft crawl, all of it outside the parallelism this fetcher is
|
|
401
|
+
# built around.
|
|
402
|
+
#
|
|
403
|
+
# @param [String] content primary docidentifier content
|
|
404
|
+
# @return [Pubid::Ietf::Identifier, nil] nil when pubid rejects it
|
|
405
|
+
#
|
|
406
|
+
def parse_pubid(content)
|
|
407
|
+
::Pubid::Ietf::Identifier.parse content
|
|
408
|
+
rescue StandardError => e
|
|
409
|
+
# Full message: the tail is the part that says *what shape* pubid
|
|
410
|
+
# stopped accepting, which is the whole point of the warning.
|
|
411
|
+
Util.warn "Not indexing `#{content}`: #{e.message}"
|
|
412
|
+
nil
|
|
261
413
|
end
|
|
262
414
|
|
|
263
415
|
#
|
|
@@ -270,7 +422,138 @@ module Relaton
|
|
|
270
422
|
elsif check_duplicate
|
|
271
423
|
@files << result[:file]
|
|
272
424
|
end
|
|
273
|
-
|
|
425
|
+
# A record whose identifier pubid rejects is written but not indexed —
|
|
426
|
+
# never fatal. The index load is all-or-nothing (`deserialize_id` raises
|
|
427
|
+
# on the first bad id and `load_index` then rejects the *entire* index),
|
|
428
|
+
# so one malformed upstream record must cost one document, not every
|
|
429
|
+
# lookup. All 176,862 published ids parse today; this guards drift.
|
|
430
|
+
# Counted here, in the parent, because a worker's tally would be lost.
|
|
431
|
+
unless result[:pubid]
|
|
432
|
+
@unindexed = @unindexed.to_i + 1
|
|
433
|
+
return
|
|
434
|
+
end
|
|
435
|
+
|
|
436
|
+
index.add_or_update result[:pubid], result[:file]
|
|
437
|
+
end
|
|
438
|
+
|
|
439
|
+
#
|
|
440
|
+
# Settle, in the parent, which record keeps which filename.
|
|
441
|
+
#
|
|
442
|
+
# `output_file` is not injective, so distinct docids can want one path.
|
|
443
|
+
# `write_unique` refuses to clobber, but a forked worker cannot know
|
|
444
|
+
# WHICH of the clashing docids deserves the plain name — it only knows the
|
|
445
|
+
# path was taken. Left there, the winner would follow the race and the two
|
|
446
|
+
# filenames would swap between crawls, churning the data repo.
|
|
447
|
+
#
|
|
448
|
+
# So the parent decides once it can see every docid: within a group of
|
|
449
|
+
# records that wanted one path, the alphabetically first docid keeps it
|
|
450
|
+
# and the rest take their digest variant. Runs over EVERY group, not just
|
|
451
|
+
# clashing ones — a lone record that fell back to a digest path (a
|
|
452
|
+
# leftover file, any transient clash) must get its plain name back, or the
|
|
453
|
+
# published filename churns and the old file is orphaned.
|
|
454
|
+
#
|
|
455
|
+
# @param [Array<Hash>] results worker results, mutated in place so
|
|
456
|
+
# `record_index_entry` indexes the final path
|
|
457
|
+
#
|
|
458
|
+
def reconcile_output_files(results)
|
|
459
|
+
results.group_by { |r| r[:plain_file] }.each do |plain, group|
|
|
460
|
+
next if plain.nil? # hand-built results, and the bibxml format
|
|
461
|
+
|
|
462
|
+
targets = assign_output_files plain, group
|
|
463
|
+
record_collision targets
|
|
464
|
+
filled = Set.new
|
|
465
|
+
order(group, targets, plain).each do |result|
|
|
466
|
+
target = targets[result[:docid].to_s]
|
|
467
|
+
place_output_file result, target, filled.add?(target).nil?
|
|
468
|
+
end
|
|
469
|
+
end
|
|
470
|
+
end
|
|
471
|
+
|
|
472
|
+
# docid => the filename it should end up with. Sorting the *unique* docids
|
|
473
|
+
# leaves no tie to break, so the assignment cannot drift between crawls
|
|
474
|
+
# the way an unstable `sort_by` over the results would.
|
|
475
|
+
def assign_output_files(plain, group)
|
|
476
|
+
group.map { |r| r[:docid].to_s }.uniq.sort.each_with_index.to_h do |docid, i|
|
|
477
|
+
[docid, i.zero? ? plain : digest_output_file(docid)]
|
|
478
|
+
end
|
|
479
|
+
end
|
|
480
|
+
|
|
481
|
+
# Records already at their target first (they are no-ops), then the movers
|
|
482
|
+
# bound for a digest path, then the one bound for `plain`. That ordering is
|
|
483
|
+
# load-bearing: a loser may be sitting ON the plain path, and moving the
|
|
484
|
+
# winner there first would destroy it.
|
|
485
|
+
def order(group, targets, plain)
|
|
486
|
+
group.sort_by do |r|
|
|
487
|
+
target = targets[r[:docid].to_s]
|
|
488
|
+
[r[:file] == target ? 0 : 1, target == plain ? 1 : 0, r[:file].to_s]
|
|
489
|
+
end
|
|
490
|
+
end
|
|
491
|
+
|
|
492
|
+
#
|
|
493
|
+
# Move one record's file to the name the parent chose for it.
|
|
494
|
+
#
|
|
495
|
+
# `duplicate` means another result for the SAME docid already holds the
|
|
496
|
+
# target. Today that is one file and one warning, so drop the stray rather
|
|
497
|
+
# than publish the document twice under two names.
|
|
498
|
+
#
|
|
499
|
+
def place_output_file(result, target, duplicate)
|
|
500
|
+
file = result[:file]
|
|
501
|
+
return result[:file] = target if file.nil? || file == target
|
|
502
|
+
|
|
503
|
+
if duplicate
|
|
504
|
+
# The document is at the target whatever happens next, so the result
|
|
505
|
+
# follows it first. The stray may ALREADY be gone: two workers that
|
|
506
|
+
# both lost the race to the plain path land on one fallback name,
|
|
507
|
+
# because that name is keyed on the docid -- so the first of them
|
|
508
|
+
# renamed this very file onto the target. Leaving the result on its
|
|
509
|
+
# old name would put an index row on a path that no longer exists.
|
|
510
|
+
result[:file] = target
|
|
511
|
+
return unless File.exist?(file)
|
|
512
|
+
|
|
513
|
+
Util.warn "Duplicate document `#{result[:docid]}`: dropping #{file}, keeping #{target}"
|
|
514
|
+
return File.delete(file)
|
|
515
|
+
end
|
|
516
|
+
return unless File.exist?(file)
|
|
517
|
+
|
|
518
|
+
File.rename file, target
|
|
519
|
+
result[:file] = target
|
|
520
|
+
rescue SystemCallError => e
|
|
521
|
+
# A late failure must not throw away a multi-hour crawl. The record keeps
|
|
522
|
+
# the name it already has on disk.
|
|
523
|
+
Util.warn "Could not move #{file} to #{target}: #{e.message}"
|
|
524
|
+
end
|
|
525
|
+
|
|
526
|
+
def record_collision(targets)
|
|
527
|
+
return if targets.size < 2
|
|
528
|
+
|
|
529
|
+
# targets.keys is already the uniqued, sorted docid list.
|
|
530
|
+
(@collisions ||= []) << targets.keys
|
|
531
|
+
end
|
|
532
|
+
|
|
533
|
+
# One line for a crawl that writes ~177k records, not one per collision.
|
|
534
|
+
def report_collisions
|
|
535
|
+
return if @collisions.nil? || @collisions.empty?
|
|
536
|
+
|
|
537
|
+
Util.warn "#{@collisions.size} filename collision(s): distinct docids sanitize to " \
|
|
538
|
+
"one filename and were given separate files. " \
|
|
539
|
+
"First: #{@collisions.first(5).map { |g| g.join(' <-> ') }.join('; ')}"
|
|
540
|
+
end
|
|
541
|
+
|
|
542
|
+
# Skips are per-record warnings in a crawl that writes ~177k of them, so
|
|
543
|
+
# restate the total where it can actually be noticed.
|
|
544
|
+
# One line for a crawl that reads ~167k files, not one per skip.
|
|
545
|
+
def report_unparsed
|
|
546
|
+
return if @unparsed.nil? || @unparsed.empty?
|
|
547
|
+
|
|
548
|
+
Util.warn "#{@unparsed.size} file(s) skipped: could not be parsed. " \
|
|
549
|
+
"First: #{@unparsed.first(5).join(', ')}"
|
|
550
|
+
end
|
|
551
|
+
|
|
552
|
+
def report_unindexed
|
|
553
|
+
return unless @unindexed.to_i.positive?
|
|
554
|
+
|
|
555
|
+
Util.warn "#{@unindexed} document(s) written but not indexed: " \
|
|
556
|
+
"identifier not parseable by Pubid::Ietf"
|
|
274
557
|
end
|
|
275
558
|
|
|
276
559
|
end
|
|
@@ -59,9 +59,13 @@ module Relaton
|
|
|
59
59
|
#
|
|
60
60
|
def remove_index_file
|
|
61
61
|
require_relative "../ietf"
|
|
62
|
-
Relaton::Index.find_or_create(:
|
|
63
|
-
|
|
64
|
-
|
|
62
|
+
Relaton::Index.find_or_create(:IETF, url: true, file: "#{INDEXFILE}.yaml").remove_file
|
|
63
|
+
# Also clear the three per-type caches a previously released relaton
|
|
64
|
+
# left in ~/.relaton; nothing writes them now, so without this they
|
|
65
|
+
# would sit there forever, unreachable by `relaton clear`.
|
|
66
|
+
%i[RFC RSS IDS].each do |type|
|
|
67
|
+
Relaton::Index.find_or_create(type, url: true, file: "index-v1.yaml").remove_file
|
|
68
|
+
end
|
|
65
69
|
end
|
|
66
70
|
end
|
|
67
71
|
end
|
|
@@ -125,10 +125,29 @@ module Relaton
|
|
|
125
125
|
is_also&.doc_id&.any? || false
|
|
126
126
|
end
|
|
127
127
|
|
|
128
|
+
#
|
|
129
|
+
# Normalise a doc-id for constituent lookup: fold case and strip
|
|
130
|
+
# whitespace and dots.
|
|
131
|
+
#
|
|
132
|
+
# Relation targets and the docids they reference disagree on both across
|
|
133
|
+
# the published corpus (`dyndNS` vs `dyndns`), which leaves records
|
|
134
|
+
# undated — and, in `build_relations`, silently downgrades a full
|
|
135
|
+
# constituent bibitem to a minimal one — under a strict lookup. Both
|
|
136
|
+
# sides go through this: the caller keys its index by it, `Entry` looks
|
|
137
|
+
# up by it.
|
|
138
|
+
#
|
|
139
|
+
# @param id [String, nil]
|
|
140
|
+
# @return [String]
|
|
141
|
+
#
|
|
142
|
+
def self.squish(id)
|
|
143
|
+
id.to_s.gsub(/[\s.]/, "").downcase
|
|
144
|
+
end
|
|
145
|
+
|
|
128
146
|
#
|
|
129
147
|
# Convert to Relaton::Ietf::ItemData
|
|
130
148
|
#
|
|
131
|
-
# @param rfc_index [Hash{String => Entry}, nil] lookup of RFC entries
|
|
149
|
+
# @param rfc_index [Hash{String => Entry}, nil] lookup of RFC entries,
|
|
150
|
+
# keyed by `Entry.squish(doc_id)` — see that method for why
|
|
132
151
|
# @return [Relaton::Ietf::ItemData, nil]
|
|
133
152
|
#
|
|
134
153
|
def to_item(rfc_index = nil, wg_names: {})
|
|
@@ -174,6 +193,7 @@ module Relaton
|
|
|
174
193
|
script: ["Latn"],
|
|
175
194
|
source: build_link,
|
|
176
195
|
formattedref: build_formattedref,
|
|
196
|
+
date: build_subseries_date(rfc_index),
|
|
177
197
|
relation: build_relations(rfc_index, wg_names: wg_names),
|
|
178
198
|
series: build_series,
|
|
179
199
|
ext: Ext.new(doctype: Doctype.new(content: "rfc"), stream: stream, flavor: "ietf"),
|
|
@@ -211,12 +231,28 @@ module Relaton
|
|
|
211
231
|
return [] unless is_also&.doc_id
|
|
212
232
|
|
|
213
233
|
is_also.doc_id.map do |ref|
|
|
214
|
-
rfc_entry = rfc_index&.[](ref)
|
|
234
|
+
rfc_entry = rfc_index&.[](Entry.squish(ref))
|
|
215
235
|
bibitem = rfc_entry ? rfc_entry.to_rfc_item(wg_names: wg_names) : build_minimal_bibitem(ref)
|
|
216
236
|
Relaton::Ietf::Relation.new(type: "includes", bibitem: bibitem)
|
|
217
237
|
end.compact
|
|
218
238
|
end
|
|
219
239
|
|
|
240
|
+
# A sub-series entry in rfc-index.xml is a pointer and nothing more —
|
|
241
|
+
# `<doc-id>` plus `<is-also>`, with no date, title, author or status,
|
|
242
|
+
# because a sub-series has no metadata of its own. Rather than publish a
|
|
243
|
+
# dateless record, take the date of the newest RFC it includes.
|
|
244
|
+
#
|
|
245
|
+
# @param rfc_index [Hash{String => Entry}, nil]
|
|
246
|
+
# @return [Array<Bib::Date>] empty when no constituent resolves or none
|
|
247
|
+
# carries a date
|
|
248
|
+
def build_subseries_date(rfc_index)
|
|
249
|
+
newest = (is_also&.doc_id || [])
|
|
250
|
+
.filter_map { |ref| rfc_index&.[](Entry.squish(ref)) }
|
|
251
|
+
.flat_map(&:build_rfc_date)
|
|
252
|
+
.max_by { |d| d.at.to_s }
|
|
253
|
+
newest ? [newest] : []
|
|
254
|
+
end
|
|
255
|
+
|
|
220
256
|
def build_minimal_bibitem(ref)
|
|
221
257
|
id = ref.sub(/^([A-Z]+)0*(\d+)$/, '\1 \2')
|
|
222
258
|
docid = Bib::Docidentifier.new(type: "IETF", content: id, primary: true)
|
|
@@ -246,7 +282,7 @@ module Relaton
|
|
|
246
282
|
[Bib::Uri.new(type: "src", content: "https://www.rfc-editor.org/info/rfc#{shortnum}")]
|
|
247
283
|
end
|
|
248
284
|
|
|
249
|
-
def build_rfc_date
|
|
285
|
+
public def build_rfc_date
|
|
250
286
|
(date || []).map do |d|
|
|
251
287
|
month_num = ::Date::MONTHNAMES.index(d.month).to_s.rjust(2, "0")
|
|
252
288
|
date_str = "#{d.year}-#{month_num}"
|
data/lib/relaton/ietf/scraper.rb
CHANGED
|
@@ -1,62 +1,95 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
|
+
require "pubid"
|
|
4
|
+
require "pubid/ietf"
|
|
5
|
+
|
|
3
6
|
module Relaton
|
|
4
7
|
module Ietf
|
|
5
8
|
# Scraper module
|
|
6
9
|
module Scraper
|
|
7
10
|
extend Scraper
|
|
8
11
|
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
+
# The combined corpus — RFCs, the RFC sub-series and Internet-Drafts in one
|
|
13
|
+
# repo, with one pubid `index-v2` covering all ~177k records (relaton#109).
|
|
14
|
+
# It replaces the three per-type repos this flavor used to read; those keep
|
|
15
|
+
# publishing their `index-v1` for released relatons, untouched.
|
|
16
|
+
IETF = "https://raw.githubusercontent.com/relaton/relaton-data-ietf/main/"
|
|
12
17
|
|
|
13
18
|
# @param text [String]
|
|
14
|
-
# @return [
|
|
19
|
+
# @return [Relaton::Ietf::ItemData, nil]
|
|
15
20
|
def scrape_page(text)
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
21
|
+
id = parse_id text.sub(/\AIETF\s+/, "")
|
|
22
|
+
return unless id
|
|
23
|
+
|
|
24
|
+
fetch_doc id
|
|
20
25
|
rescue Timeout::Error, Errno::EINVAL, Errno::ECONNRESET, EOFError,
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
raise Relaton::RequestError, "No document found for #{
|
|
26
|
+
Net::HTTPBadResponse, Net::HTTPHeaderSyntaxError,
|
|
27
|
+
Net::ProtocolError, SocketError
|
|
28
|
+
raise Relaton::RequestError, "No document found for #{text} reference"
|
|
24
29
|
end
|
|
25
30
|
|
|
26
31
|
private
|
|
27
32
|
|
|
28
|
-
#
|
|
29
|
-
#
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
row = index.search(ref).first
|
|
43
|
-
get_page "#{RFC}#{row[:file]}" if row
|
|
33
|
+
# The index stores parsed pubids, so the query has to be one too.
|
|
34
|
+
# `Type#search_candidates` narrows only when the query is not a String
|
|
35
|
+
# (`@file_io.sorted && id && !id.is_a?(String)`); a String falls through to
|
|
36
|
+
# `match_item`'s `item[:id].to_s.include?(id)`, which renders every pubid in
|
|
37
|
+
# the index on every lookup — measured at ~40 s per reference against the
|
|
38
|
+
# 177k-row index, versus sub-millisecond for a parsed one.
|
|
39
|
+
# `exact:` keeps the match exact. `Type#search` without it takes
|
|
40
|
+
# pubid's subset match, where a component the query omits is a wildcard —
|
|
41
|
+
# and an Internet-Draft slug omits the version, so `draft-foo` would match
|
|
42
|
+
# every `draft-foo-NN` and `.first` would return an arbitrary one (20,513
|
|
43
|
+
# such pairs in the first 40k rows of the index fixture).
|
|
44
|
+
def fetch_doc(id)
|
|
45
|
+
row = index.search(id, exact: true).first
|
|
46
|
+
get_page "#{IETF}#{row[:file]}" if row
|
|
44
47
|
end
|
|
45
48
|
|
|
46
|
-
def
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
49
|
+
def index
|
|
50
|
+
Relaton::Index.find_or_create(
|
|
51
|
+
:IETF, url: "#{IETF}#{INDEXFILE}.zip", file: "#{INDEXFILE}.yaml",
|
|
52
|
+
pubid_class: ::Pubid::Ietf::Identifier
|
|
53
|
+
)
|
|
50
54
|
end
|
|
51
55
|
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
+
#
|
|
57
|
+
# Parse a reference into the identifier the index is keyed by.
|
|
58
|
+
#
|
|
59
|
+
# Normalises the two Internet-Draft spellings callers use — `I-D.<slug>`
|
|
60
|
+
# and `I-D <slug>` — onto the `draft-…` form pubid parses and the index
|
|
61
|
+
# stores. The bare `I-D.ietf-quic-transport` spelling (the bibxml anchor,
|
|
62
|
+
# and the `docnumber` IETF records carry) gains the `draft-` stem: the old
|
|
63
|
+
# plain-string index matched it by substring, and matching is exact now.
|
|
64
|
+
#
|
|
65
|
+
# The capture is `(\S.*)`, not `(.+)`: it must start non-space so it cannot
|
|
66
|
+
# overlap the preceding `\s*`. Behaviour is unchanged on every real
|
|
67
|
+
# reference — with `(.+)` the greedy `\s*` only ever yields ground on an
|
|
68
|
+
# all-whitespace remainder, which pubid rejects anyway — but the
|
|
69
|
+
# unambiguous form is about twice as fast on a pathological input and is
|
|
70
|
+
# what CodeQL's rb/polynomial-redos models. Measured before changing it:
|
|
71
|
+
# the old form was already linear (0.66 ms at 80k chars), so this is
|
|
72
|
+
# clarity, not a ReDoS fix.
|
|
73
|
+
#
|
|
74
|
+
# @param ref [String]
|
|
75
|
+
# @return [Pubid::Ietf::Identifier, nil] nil when pubid has no grammar for
|
|
76
|
+
# it, so an out-of-flavor reference logs "Not found." instead of raising
|
|
77
|
+
#
|
|
78
|
+
def parse_id(ref)
|
|
79
|
+
if (draft = ref[/\AI-D[.\s]\s*(\S.*)\z/m, 1])
|
|
80
|
+
ref = draft.start_with?("draft-") ? draft : "draft-#{draft}"
|
|
81
|
+
end
|
|
82
|
+
::Pubid::Ietf::Identifier.parse ref
|
|
83
|
+
rescue StandardError => e
|
|
84
|
+
# Logged, not swallowed: this repo git-pins pubid to a moving `main`,
|
|
85
|
+
# so a grammar regression would otherwise present as every IETF
|
|
86
|
+
# reference quietly reporting "Not found."
|
|
87
|
+
Util.debug "`#{ref}` is not an IETF identifier: #{e.message}"
|
|
88
|
+
nil
|
|
56
89
|
end
|
|
57
90
|
|
|
58
91
|
# @param uri [String]
|
|
59
|
-
# @return [
|
|
92
|
+
# @return [Relaton::Ietf::ItemData, nil] HTTP response body
|
|
60
93
|
def get_page(uri)
|
|
61
94
|
res = Net::HTTP.get_response(URI(uri))
|
|
62
95
|
return unless res.code == "200"
|
data/lib/relaton/ietf.rb
CHANGED
|
@@ -14,7 +14,10 @@ require_relative "ietf/bibliography"
|
|
|
14
14
|
|
|
15
15
|
module Relaton
|
|
16
16
|
module Ietf
|
|
17
|
-
|
|
17
|
+
# The pubid-structured index, written by DataFetcher and read by Scraper
|
|
18
|
+
# (relaton#109). One constant again now that both sides are on it — the
|
|
19
|
+
# second one existed only while producer and consumer were split.
|
|
20
|
+
INDEXFILE = "index-v2".freeze
|
|
18
21
|
# Returns hash of XML reammar
|
|
19
22
|
# @return [String]
|
|
20
23
|
def self.grammar_hash
|