relaton 3.0.0.pre.alpha.1 → 3.0.0.pre.alpha.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (270) hide show
  1. checksums.yaml +4 -4
  2. data/README.adoc +57 -1
  3. data/bin/console +0 -1
  4. data/lib/relaton/3gpp/bibliography.rb +82 -7
  5. data/lib/relaton/3gpp/data_fetcher.rb +51 -3
  6. data/lib/relaton/3gpp/docidentifier.rb +114 -0
  7. data/lib/relaton/3gpp/item.rb +6 -0
  8. data/lib/relaton/3gpp/parser.rb +1 -1
  9. data/lib/relaton/3gpp/processor.rb +4 -1
  10. data/lib/relaton/3gpp.rb +5 -1
  11. data/lib/relaton/adobe/bibdata.rb +8 -0
  12. data/lib/relaton/adobe/bibitem.rb +8 -0
  13. data/lib/relaton/adobe/bibliography.rb +92 -0
  14. data/lib/relaton/adobe/docidentifier.rb +49 -0
  15. data/lib/relaton/adobe/doctype.rb +14 -0
  16. data/lib/relaton/adobe/ext.rb +32 -0
  17. data/lib/relaton/adobe/item.rb +15 -0
  18. data/lib/relaton/adobe/item_base.rb +18 -0
  19. data/lib/relaton/adobe/item_data.rb +6 -0
  20. data/lib/relaton/adobe/processor.rb +45 -0
  21. data/lib/relaton/adobe/util.rb +8 -0
  22. data/lib/relaton/adobe.rb +37 -0
  23. data/lib/relaton/bib/model/address.rb +2 -2
  24. data/lib/relaton/bib/model/docidentifier.rb +24 -9
  25. data/lib/relaton/bib/model/localized_string.rb +1 -1
  26. data/lib/relaton/bib/model/structured_identifier.rb +10 -9
  27. data/lib/relaton/bib/sanitizer.rb +202 -6
  28. data/lib/relaton/bib.rb +0 -2
  29. data/lib/relaton/bipm/bibliography.rb +159 -10
  30. data/lib/relaton/bipm/data_fetcher.rb +26 -2
  31. data/lib/relaton/bipm/data_outcomes_parser.rb +3 -4
  32. data/lib/relaton/bipm/id_parser.rb +5 -4
  33. data/lib/relaton/bipm/model/structured_identifier.rb +21 -0
  34. data/lib/relaton/bipm/processor.rb +2 -2
  35. data/lib/relaton/bipm/rawdata_bipm_metrologia/fetcher.rb +2 -4
  36. data/lib/relaton/bipm/si_brochure_parser.rb +5 -3
  37. data/lib/relaton/bipm.rb +7 -1
  38. data/lib/relaton/bsi/bibliography.rb +115 -43
  39. data/lib/relaton/bsi/hit.rb +14 -0
  40. data/lib/relaton/bsi/hit_collection.rb +15 -16
  41. data/lib/relaton/bsi/model/docidentifier.rb +99 -1
  42. data/lib/relaton/bsi/processor.rb +1 -0
  43. data/lib/relaton/calconnect/bibliography.rb +12 -14
  44. data/lib/relaton/calconnect/data_fetcher.rb +77 -9
  45. data/lib/relaton/calconnect/docidentifier.rb +80 -0
  46. data/lib/relaton/calconnect/hit_collection.rb +65 -57
  47. data/lib/relaton/calconnect/model/item.rb +7 -0
  48. data/lib/relaton/calconnect/processor.rb +7 -1
  49. data/lib/relaton/calconnect.rb +11 -1
  50. data/lib/relaton/ccsds/data/fetcher.rb +17 -12
  51. data/lib/relaton/ccsds/data/parser.rb +1 -1
  52. data/lib/relaton/ccsds/hit_collection.rb +6 -1
  53. data/lib/relaton/ccsds/model/docidentifier.rb +121 -0
  54. data/lib/relaton/ccsds/model/item.rb +2 -0
  55. data/lib/relaton/cen/bibliography.rb +75 -47
  56. data/lib/relaton/cen/hit.rb +16 -1
  57. data/lib/relaton/cen/hit_collection.rb +59 -11
  58. data/lib/relaton/cen/model/docidentifier.rb +92 -1
  59. data/lib/relaton/cen/processor.rb +13 -8
  60. data/lib/relaton/cen/scraper.rb +13 -5
  61. data/lib/relaton/cen.rb +1 -0
  62. data/lib/relaton/cie/data_fetcher.rb +215 -30
  63. data/lib/relaton/cie/processor.rb +3 -1
  64. data/lib/relaton/cie/scrapper.rb +15 -2
  65. data/lib/relaton/cie.rb +2 -1
  66. data/lib/relaton/core/data_fetcher.rb +150 -3
  67. data/lib/relaton/core/governor.rb +320 -0
  68. data/lib/relaton/core/pacer.rb +134 -0
  69. data/lib/relaton/core/processor.rb +19 -0
  70. data/lib/relaton/core/request_error.rb +14 -0
  71. data/lib/relaton/core.rb +3 -0
  72. data/lib/relaton/db/registry.rb +41 -1
  73. data/lib/relaton/doi/crossref.rb +19 -2
  74. data/lib/relaton/doi/parser.rb +109 -15
  75. data/lib/relaton/easc/bibdata.rb +8 -0
  76. data/lib/relaton/easc/bibitem.rb +8 -0
  77. data/lib/relaton/easc/bibliography.rb +95 -0
  78. data/lib/relaton/easc/docidentifier.rb +100 -0
  79. data/lib/relaton/easc/doctype.rb +14 -0
  80. data/lib/relaton/easc/ext.rb +44 -0
  81. data/lib/relaton/easc/item.rb +13 -0
  82. data/lib/relaton/easc/item_base.rb +18 -0
  83. data/lib/relaton/easc/item_data.rb +6 -0
  84. data/lib/relaton/easc/processor.rb +46 -0
  85. data/lib/relaton/easc/util.rb +8 -0
  86. data/lib/relaton/easc.rb +35 -0
  87. data/lib/relaton/ecma/bibliography.rb +93 -25
  88. data/lib/relaton/ecma/data_fetcher.rb +71 -12
  89. data/lib/relaton/ecma/docidentifier.rb +124 -0
  90. data/lib/relaton/ecma/item.rb +2 -0
  91. data/lib/relaton/ecma/memento_parser.rb +1 -1
  92. data/lib/relaton/ecma/page_fetcher.rb +15 -3
  93. data/lib/relaton/ecma/parser_common.rb +2 -2
  94. data/lib/relaton/ecma/processor.rb +4 -1
  95. data/lib/relaton/ecma/standard_parser.rb +2 -2
  96. data/lib/relaton/ecma.rb +10 -1
  97. data/lib/relaton/etsi/bibliography.rb +67 -2
  98. data/lib/relaton/etsi/data_fetcher.rb +43 -4
  99. data/lib/relaton/etsi/processor.rb +3 -1
  100. data/lib/relaton/etsi.rb +2 -1
  101. data/lib/relaton/gb/bibliography.rb +55 -29
  102. data/lib/relaton/gb/docidentifier.rb +58 -9
  103. data/lib/relaton/gb/processor.rb +3 -0
  104. data/lib/relaton/gb/scraper.rb +27 -10
  105. data/lib/relaton/gost/bibdata.rb +8 -0
  106. data/lib/relaton/gost/bibitem.rb +8 -0
  107. data/lib/relaton/gost/bibliography.rb +107 -0
  108. data/lib/relaton/gost/docidentifier.rb +80 -0
  109. data/lib/relaton/gost/doctype.rb +16 -0
  110. data/lib/relaton/gost/ext.rb +46 -0
  111. data/lib/relaton/gost/item.rb +15 -0
  112. data/lib/relaton/gost/item_base.rb +18 -0
  113. data/lib/relaton/gost/item_data.rb +6 -0
  114. data/lib/relaton/gost/processor.rb +49 -0
  115. data/lib/relaton/gost/util.rb +8 -0
  116. data/lib/relaton/gost.rb +36 -0
  117. data/lib/relaton/iala/bibdata.rb +8 -0
  118. data/lib/relaton/iala/bibitem.rb +8 -0
  119. data/lib/relaton/iala/bibliography.rb +146 -0
  120. data/lib/relaton/iala/docidentifier.rb +89 -0
  121. data/lib/relaton/iala/doctype.rb +18 -0
  122. data/lib/relaton/iala/ext.rb +32 -0
  123. data/lib/relaton/iala/item.rb +21 -0
  124. data/lib/relaton/iala/item_base.rb +18 -0
  125. data/lib/relaton/iala/item_data.rb +6 -0
  126. data/lib/relaton/iala/processor.rb +43 -0
  127. data/lib/relaton/iala/relation.rb +7 -0
  128. data/lib/relaton/iala/util.rb +8 -0
  129. data/lib/relaton/iala.rb +35 -0
  130. data/lib/relaton/iana/bibliography.rb +67 -14
  131. data/lib/relaton/iana/data_fetcher.rb +35 -5
  132. data/lib/relaton/iana/processor.rb +3 -1
  133. data/lib/relaton/iana.rb +12 -1
  134. data/lib/relaton/iec/data_fetcher.rb +7 -1
  135. data/lib/relaton/iec/hit_collection.rb +1 -1
  136. data/lib/relaton/iec/model/docidentifier.rb +9 -5
  137. data/lib/relaton/iec/model/ext.rb +2 -2
  138. data/lib/relaton/iec/processor.rb +1 -0
  139. data/lib/relaton/ieee/bibliography.rb +25 -3
  140. data/lib/relaton/ieee/data_fetcher.rb +158 -17
  141. data/lib/relaton/ieee/idams_parser.rb +18 -11
  142. data/lib/relaton/ieee/processor.rb +4 -1
  143. data/lib/relaton/ieee/rawbib_id_parser.rb +291 -86
  144. data/lib/relaton/ieee.rb +2 -1
  145. data/lib/relaton/ietf/data_fetcher.rb +295 -12
  146. data/lib/relaton/ietf/processor.rb +7 -3
  147. data/lib/relaton/ietf/rfc/entry.rb +39 -3
  148. data/lib/relaton/ietf/scraper.rb +69 -36
  149. data/lib/relaton/ietf.rb +4 -1
  150. data/lib/relaton/iho/bibliography.rb +1 -1
  151. data/lib/relaton/iho/docidentifier.rb +1 -1
  152. data/lib/relaton/index/file_io.rb +11 -11
  153. data/lib/relaton/index/file_storage.rb +6 -1
  154. data/lib/relaton/index/pool.rb +6 -1
  155. data/lib/relaton/index/shard_source.rb +201 -0
  156. data/lib/relaton/index/type.rb +63 -12
  157. data/lib/relaton/index.rb +2 -1
  158. data/lib/relaton/iso/bibliography.rb +20 -15
  159. data/lib/relaton/iso/data_fetcher.rb +3 -3
  160. data/lib/relaton/iso/data_parser.rb +17 -3
  161. data/lib/relaton/iso/hit_collection.rb +27 -15
  162. data/lib/relaton/iso/item_data.rb +22 -0
  163. data/lib/relaton/iso/model/docidentifier.rb +24 -12
  164. data/lib/relaton/iso/processor.rb +1 -0
  165. data/lib/relaton/iso/scraper.rb +19 -3
  166. data/lib/relaton/itu/bibliography.rb +9 -4
  167. data/lib/relaton/itu/data_crawler_r.rb +664 -0
  168. data/lib/relaton/itu/data_fetcher.rb +496 -50
  169. data/lib/relaton/itu/data_merge_r.rb +149 -0
  170. data/lib/relaton/itu/data_parser_r.rb +163 -89
  171. data/lib/relaton/itu/data_parser_t.rb +228 -0
  172. data/lib/relaton/itu/family_cache.rb +177 -0
  173. data/lib/relaton/itu/governor.rb +56 -0
  174. data/lib/relaton/itu/hit.rb +9 -3
  175. data/lib/relaton/itu/hit_collection.rb +258 -86
  176. data/lib/relaton/itu/model/docidentifier.rb +67 -1
  177. data/lib/relaton/itu/model/structured_identifier.rb +19 -0
  178. data/lib/relaton/itu/processor.rb +10 -4
  179. data/lib/relaton/itu/pubid.rb +27 -5
  180. data/lib/relaton/itu/recommendation_fields.rb +334 -0
  181. data/lib/relaton/itu/recommendation_parser.rb +18 -149
  182. data/lib/relaton/itu/scraper.rb +13 -3
  183. data/lib/relaton/itu.rb +2 -1
  184. data/lib/relaton/jcgm/bibdata.rb +8 -0
  185. data/lib/relaton/jcgm/bibitem.rb +8 -0
  186. data/lib/relaton/jcgm/bibliography.rb +97 -0
  187. data/lib/relaton/jcgm/data_fetcher.rb +81 -0
  188. data/lib/relaton/jcgm/docidentifier.rb +102 -0
  189. data/lib/relaton/jcgm/doctype.rb +12 -0
  190. data/lib/relaton/jcgm/ext.rb +23 -0
  191. data/lib/relaton/jcgm/item.rb +20 -0
  192. data/lib/relaton/jcgm/item_base.rb +18 -0
  193. data/lib/relaton/jcgm/item_data.rb +6 -0
  194. data/lib/relaton/jcgm/meetings_parser.rb +175 -0
  195. data/lib/relaton/jcgm/processor.rb +71 -0
  196. data/lib/relaton/jcgm/relation.rb +9 -0
  197. data/lib/relaton/jcgm/structured_identifier.rb +40 -0
  198. data/lib/relaton/jcgm/util.rb +8 -0
  199. data/lib/relaton/jcgm.rb +24 -0
  200. data/lib/relaton/jis/bibliography.rb +8 -10
  201. data/lib/relaton/jis/data_fetcher.rb +21 -19
  202. data/lib/relaton/jis/docidentifier.rb +104 -5
  203. data/lib/relaton/jis/hit.rb +18 -23
  204. data/lib/relaton/jis/hit_collection.rb +19 -18
  205. data/lib/relaton/jis/processor.rb +1 -1
  206. data/lib/relaton/jis.rb +2 -3
  207. data/lib/relaton/logger/channels/gh_issue.rb +78 -13
  208. data/lib/relaton/nist/data_fetcher.rb +63 -13
  209. data/lib/relaton/nist/docidentifier.rb +165 -0
  210. data/lib/relaton/nist/item.rb +2 -0
  211. data/lib/relaton/nist/item_base.rb +16 -0
  212. data/lib/relaton/nist/mods_parser.rb +38 -12
  213. data/lib/relaton/nist/processor.rb +2 -1
  214. data/lib/relaton/nist/relation.rb +3 -0
  215. data/lib/relaton/nist/scraper.rb +6 -3
  216. data/lib/relaton/oasis/bibliography.rb +147 -6
  217. data/lib/relaton/oasis/data_fetcher.rb +41 -5
  218. data/lib/relaton/oasis/data_parser_utils.rb +37 -3
  219. data/lib/relaton/oasis/docidentifier.rb +54 -0
  220. data/lib/relaton/oasis/item.rb +3 -0
  221. data/lib/relaton/oasis/processor.rb +7 -1
  222. data/lib/relaton/oasis.rb +14 -1
  223. data/lib/relaton/ogc/data_fetcher.rb +23 -2
  224. data/lib/relaton/ogc/docidentifier.rb +105 -0
  225. data/lib/relaton/ogc/hit_collection.rb +78 -3
  226. data/lib/relaton/ogc/processor.rb +2 -1
  227. data/lib/relaton/ogc.rb +5 -1
  228. data/lib/relaton/oiml/bibliography.rb +90 -15
  229. data/lib/relaton/oiml/docidentifier.rb +18 -3
  230. data/lib/relaton/omg/docidentifier.rb +67 -0
  231. data/lib/relaton/omg/item.rb +1 -0
  232. data/lib/relaton/omg/processor.rb +1 -0
  233. data/lib/relaton/omg/scraper.rb +61 -16
  234. data/lib/relaton/omg.rb +1 -0
  235. data/lib/relaton/plateau/bibliography.rb +10 -3
  236. data/lib/relaton/plateau/data_fetcher.rb +25 -2
  237. data/lib/relaton/plateau/handbook_parser.rb +8 -1
  238. data/lib/relaton/plateau/hit.rb +10 -2
  239. data/lib/relaton/plateau/hit_collection.rb +31 -11
  240. data/lib/relaton/plateau/processor.rb +3 -1
  241. data/lib/relaton/plateau/technical_report_parser.rb +8 -1
  242. data/lib/relaton/plateau.rb +2 -1
  243. data/lib/relaton/sdo/config.rb +34 -0
  244. data/lib/relaton/sdo/fetcher.rb +52 -0
  245. data/lib/relaton/sdo/logo.rb +95 -0
  246. data/lib/relaton/sdo/name.rb +26 -0
  247. data/lib/relaton/sdo/organization.rb +71 -0
  248. data/lib/relaton/sdo/store.rb +49 -0
  249. data/lib/relaton/sdo.rb +29 -0
  250. data/lib/relaton/version.rb +1 -1
  251. data/lib/relaton/w3c/bibliography.rb +132 -12
  252. data/lib/relaton/w3c/data_fetcher.rb +194 -16
  253. data/lib/relaton/w3c/data_parser.rb +3 -3
  254. data/lib/relaton/w3c/docidentifier.rb +48 -0
  255. data/lib/relaton/w3c/governor.rb +32 -0
  256. data/lib/relaton/w3c/item.rb +3 -0
  257. data/lib/relaton/w3c/pubid.rb +12 -0
  258. data/lib/relaton/w3c/safe_realize.rb +110 -21
  259. data/lib/relaton/w3c.rb +12 -1
  260. data/lib/relaton/xsf/bibliography.rb +61 -1
  261. data/lib/relaton/xsf/data_fetcher.rb +55 -5
  262. data/lib/relaton/xsf/docidentifier.rb +46 -0
  263. data/lib/relaton/xsf/hit_collection.rb +31 -3
  264. data/lib/relaton/xsf/item.rb +6 -0
  265. data/lib/relaton/xsf/processor.rb +1 -0
  266. data/lib/relaton/xsf.rb +5 -1
  267. data/lib/relaton.rb +42 -0
  268. metadata +135 -24
  269. data/lib/relaton/ieee/pub_id.rb +0 -161
  270. data/lib/relaton/index/id_number.rb +0 -30
@@ -1,5 +1,7 @@
1
1
  require "etc"
2
2
  require "parallel"
3
+ require "pubid"
4
+ require "pubid/ietf"
3
5
  require "relaton/core"
4
6
  require_relative "../ietf"
5
7
  require_relative "bibxml_parser"
@@ -21,12 +23,30 @@ module Relaton
21
23
  when "ietf-rfc-entries" then fetch_ieft_rfcs
22
24
  end
23
25
  index.save
26
+ report_unindexed
27
+ report_unparsed
28
+ report_collisions
24
29
  end
25
30
 
26
31
  private
27
32
 
33
+ # The published index is the pubid-structured `index-v2` (relaton#109).
34
+ # `pubid_class:` is not decoration: `FileIO#save` serialises an id to its
35
+ # `_type:` hash only when it is an instance of the configured class, so
36
+ # without it — or without parsing the id below — this writes a v1-shaped
37
+ # file under a v2 name.
38
+ # `url: nil` is load-bearing, not decoration. Scraper opens the same
39
+ # `:IETF` pool key with a `url:`, and `Type#actual?` skips the URL check
40
+ # when the caller omits it (`!args.key?(:url)`) — so in a process where a
41
+ # lookup ran first, omitting it here would hand the crawl the
42
+ # remote-backed Type and `save` would write to `~/.relaton/ietf/` instead
43
+ # of `./`, publishing no index at all. Passing it explicitly forces a
44
+ # local-file Type.
28
45
  def index
29
- @index ||= Relaton::Index.find_or_create :IETF, file: "#{INDEXFILE}.yaml"
46
+ @index ||= Relaton::Index.find_or_create(
47
+ :IETF, url: nil, file: "#{INDEXFILE}.yaml",
48
+ pubid_class: ::Pubid::Ietf::Identifier
49
+ )
30
50
  end
31
51
 
32
52
  #
@@ -34,8 +54,16 @@ module Relaton
34
54
  #
35
55
  def fetch_ieft_rfcsubseries
36
56
  idx = Rfc::Index.from_xml(rfc_index)
57
+ # Keyed by the normalised doc-id, built once for the whole crawl:
58
+ # `Entry` looks constituents up by `Entry.squish(ref)`, and doing it
59
+ # per-entry over ~9,800 RFCs would rebuild this table 367 times.
37
60
  rfc_map = (idx.rfc_entries || []).each_with_object({}) do |entry, h|
38
- h[entry.doc_id] = entry
61
+ key = Rfc::Entry.squish(entry.doc_id)
62
+ if h.key?(key)
63
+ Util.warn "Duplicate RFC doc-id `#{entry.doc_id}` after normalisation " \
64
+ "(`#{key}`); the later entry wins for constituent lookup"
65
+ end
66
+ h[key] = entry
39
67
  end
40
68
  idx.subseries_entries.each do |entry|
41
69
  save_doc entry.to_item(rfc_map, wg_names: wg_names)
@@ -54,6 +82,10 @@ module Relaton
54
82
  #
55
83
  def fetch_ieft_internet_drafts
56
84
  series_groups, singleton_paths = group_draft_paths
85
+ # Workers fork from here, so `unique_output_file`'s reservation cannot
86
+ # see a peer's claim and `write_unique` must never overwrite. Set
87
+ # before the first fork, so the children inherit it.
88
+ @cross_process = true
57
89
 
58
90
  series_results = parallelize(series_groups.to_a) do |(series, paths_info)|
59
91
  process_series(series, paths_info)
@@ -63,7 +95,13 @@ module Relaton
63
95
  process_singleton(path)
64
96
  end
65
97
 
66
- (series_results + singleton_results).compact.each { |r| record_index_entry(r) }
98
+ entries, unparsed = (series_results + singleton_results).compact
99
+ .partition { |r| r[:unparsed].nil? }
100
+ # Tallied here, not in the worker: a counter incremented in a Parallel
101
+ # worker process is lost on the way back (see record_index_entry).
102
+ @unparsed = unparsed.map { |r| "#{r[:unparsed]} (#{r[:error]})" }
103
+ reconcile_output_files entries
104
+ entries.each { |r| record_index_entry(r) }
67
105
  end
68
106
 
69
107
  #
@@ -114,16 +152,22 @@ module Relaton
114
152
  # for the parent.
115
153
  #
116
154
  def process_series(series, paths_info)
117
- sorted = paths_info.sort_by { |p| p[:ver].to_i }.map do |p|
118
- bib = BibXMLParser.parse(File.read(p[:path], encoding: "UTF-8"))
155
+ parsed = paths_info.sort_by { |p| p[:ver].to_i }.map do |p|
156
+ bib, marker = parse_bibxml(p[:path])
157
+ next marker unless bib
158
+
119
159
  bib.version = [Bib::Version.new(draft: p[:ver])]
120
160
  p.merge(bib: bib, source: bib.source)
121
161
  end
162
+ # A file that failed to parse must not reach `sorted`:
163
+ # link_neighbor_relations and build_unversioned_doc both dereference
164
+ # `entry[:bib]`, and a dropped version is better than a nil one.
165
+ sorted, skipped = parsed.partition { |e| e[:unparsed].nil? }
122
166
  link_neighbor_relations(sorted) if @format != "bibxml"
123
167
 
124
168
  results = sorted.map { |entry| serialize_and_write(entry[:bib]) }
125
169
  results << serialize_and_write(build_unversioned_doc(series, sorted)) if @format != "bibxml"
126
- results.compact
170
+ results.compact + skipped
127
171
  end
128
172
 
129
173
  #
@@ -133,11 +177,82 @@ module Relaton
133
177
  file = File.basename(path, ".xml")
134
178
  is_draft = file.include?("D.draft-")
135
179
  ver = is_draft ? file[/(\d+)$/, 1] : nil
136
- bib = BibXMLParser.parse(File.read(path, encoding: "UTF-8"))
180
+ bib, marker = parse_bibxml(path)
181
+ return marker unless bib
182
+
137
183
  bib.version = [Bib::Version.new(draft: ver)] if ver
138
184
  serialize_and_write(bib)
139
185
  end
140
186
 
187
+ #
188
+ # Read and parse one bibxml file, or nil if it cannot be parsed.
189
+ #
190
+ # Rescues `StandardError` rather than a narrow list on purpose: lutaml
191
+ # raises `InvalidFormatError` on bad bytes, but the converter also runs
192
+ # regexes over parsed text (`parse_surname_initials` and friends), and
193
+ # those raise `ArgumentError: invalid byte sequence` on anything that slips
194
+ # through. One unparseable file must cost one document, never the crawl —
195
+ # the drafts path runs under Parallel.map, which discards every result from
196
+ # the pass when a worker raises.
197
+ #
198
+ # @param path [String]
199
+ # @return [Relaton::Ietf::ItemData, nil]
200
+ #
201
+ # @param path [String]
202
+ # @return [Array(Relaton::Ietf::ItemData, nil), Array(nil, Hash)]
203
+ # the record, or nil plus a marker carrying why
204
+ def parse_bibxml(path)
205
+ bib = BibXMLParser.parse(read_bibxml(path))
206
+ bib ? [bib, nil] : [nil, unparsed_marker(path, "parser returned no record")]
207
+ rescue StandardError => e
208
+ [nil, unparsed_marker(path, "#{e.class}: #{e.message.to_s.lines.first.to_s.strip}")]
209
+ end
210
+
211
+ # Marshal-friendly stand-in for a record, carried back to the parent so the
212
+ # skip can be counted where a tally survives. The reason rides along rather
213
+ # than sitting in an ivar, which a later file would overwrite.
214
+ def unparsed_marker(path, error)
215
+ { unparsed: path, error: error }
216
+ end
217
+
218
+ #
219
+ # Read a bibxml file as UTF-8, recovering Windows-1252 bytes.
220
+ #
221
+ # These files declare `encoding='UTF-8'` but some carry CP1252 — smart
222
+ # quotes and accented Latin letters. `File.read(encoding: "UTF-8")` only
223
+ # *tags* the string, so those reach lutaml as invalid UTF-8 and it raises.
224
+ #
225
+ # `scrub` with a block transcodes each invalid *run* as CP1252 while
226
+ # leaving valid UTF-8 untouched. Both halves matter:
227
+ #
228
+ # * Not plain `scrub`, which substitutes U+FFFD: `client’s` would become
229
+ # `client\uFFFDs` and `Muñoz` `Mu\uFFFDoz` — author surnames included.
230
+ # The damage is length-preserving, so a length check will not catch it.
231
+ # * Not a whole-file CP1252 re-decode, which mangles a file that is
232
+ # genuinely UTF-8 apart from one stray byte: `Muñoz café ’` would come
233
+ # back as `Muñoz café ’`, silently, since the result is valid UTF-8 and
234
+ # nothing raises. No such file exists in today's corpus (0 of the 125
235
+ # affected contain valid multi-byte UTF-8) but it grows daily, and
236
+ # "decodes losslessly as CP1252" is weak evidence of correctness —
237
+ # CP1252 maps 251 of 256 byte values.
238
+ #
239
+ # `undef: :replace` covers the five bytes CP1252 leaves undefined
240
+ # (0x81 0x8D 0x8F 0x90 0x9D), so this returns valid UTF-8 rather than
241
+ # raising and costing the whole file.
242
+ #
243
+ # @param path [String]
244
+ # @return [String] UTF-8, valid encoding
245
+ #
246
+ def read_bibxml(path)
247
+ utf8 = File.binread(path).force_encoding(Encoding::UTF_8)
248
+ return utf8 if utf8.valid_encoding?
249
+
250
+ utf8.scrub do |bad|
251
+ bad.force_encoding(Encoding::WINDOWS_1252)
252
+ .encode(Encoding::UTF_8, undef: :replace)
253
+ end
254
+ end
255
+
141
256
  #
142
257
  # Append immediate-neighbor `updates` / `updatedBy` relations in memory.
143
258
  # Single-version series get no relations (no neighbors).
@@ -160,6 +275,12 @@ module Relaton
160
275
  # `includes` relations to every version. Uses the latest version's
161
276
  # title/abstract from memory.
162
277
  #
278
+ # The aggregator is *synthesised* — there is no upstream document for it,
279
+ # so `date`, `ext` (hence doctype) and `source` can only be inherited from
280
+ # its newest constituent, which `sorted` already holds in memory. Without
281
+ # that inheritance it publishes undated (and so unsorted on the Pages
282
+ # index, which sorts by date) and with no document type at all.
283
+ #
163
284
  # @return [Relaton::Ietf::ItemData, nil]
164
285
  #
165
286
  def build_unversioned_doc(series, sorted)
@@ -173,7 +294,11 @@ module Relaton
173
294
  rel = sorted.map { |e| version_relation({ ref: e[:ref], source: e[:source] }, "includes") }
174
295
  ItemData.new(
175
296
  title: last_v.title, abstract: last_v.abstract, formattedref: Bib::Formattedref.new(content: series),
176
- docidentifier: [docid], relation: rel
297
+ docidentifier: [docid], relation: rel,
298
+ # dup'd, not shared: these are the newest version's own objects, and
299
+ # aliasing them would make any later edit to the aggregator mutate the
300
+ # `-NN` record too.
301
+ date: last_v.date&.dup, ext: last_v.ext&.dup, source: last_v.source&.dup
177
302
  )
178
303
  end
179
304
 
@@ -254,10 +379,37 @@ module Relaton
254
379
  entry.docidentifier.detect { |i| i.type == "Internet-Draft" && i.primary }&.content
255
380
  end
256
381
  id ||= entry.docnumber || entry.formattedref.content
257
- file = output_file(id)
258
- File.write file, content, encoding: "UTF-8"
382
+ file = write_unique(id, content)
259
383
  primary = entry.docidentifier.detect(&:primary) || entry.docidentifier.first
260
- { docnumber: entry.docnumber, file: file, index_id: primary.content }
384
+ # `docid` is the id the file was written under and `plain_file` the name
385
+ # it would have taken uncontested; `reconcile_output_files` needs both.
386
+ # Neither is `index_id`: `id` falls back through docnumber and
387
+ # formattedref, so on the RFC path the two differ ("RFC0001" vs "RFC 1").
388
+ { docnumber: entry.docnumber, docid: id, file: file,
389
+ plain_file: output_file(id), index_id: primary.content,
390
+ pubid: parse_pubid(primary.content) }
391
+ end
392
+
393
+ #
394
+ # Parse a record's primary docidentifier into the pubid the index stores.
395
+ #
396
+ # Deliberately here rather than in `record_index_entry`: this runs inside
397
+ # the `Parallel` workers, and a pubid identifier survives the Marshal round
398
+ # trip Parallel does on the return value. Parsing in the parent instead
399
+ # would put ~0.7 ms per record back on the serial path — some minutes over
400
+ # the 167k-draft crawl, all of it outside the parallelism this fetcher is
401
+ # built around.
402
+ #
403
+ # @param [String] content primary docidentifier content
404
+ # @return [Pubid::Ietf::Identifier, nil] nil when pubid rejects it
405
+ #
406
+ def parse_pubid(content)
407
+ ::Pubid::Ietf::Identifier.parse content
408
+ rescue StandardError => e
409
+ # Full message: the tail is the part that says *what shape* pubid
410
+ # stopped accepting, which is the whole point of the warning.
411
+ Util.warn "Not indexing `#{content}`: #{e.message}"
412
+ nil
261
413
  end
262
414
 
263
415
  #
@@ -270,7 +422,138 @@ module Relaton
270
422
  elsif check_duplicate
271
423
  @files << result[:file]
272
424
  end
273
- index.add_or_update result[:index_id], result[:file]
425
+ # A record whose identifier pubid rejects is written but not indexed —
426
+ # never fatal. The index load is all-or-nothing (`deserialize_id` raises
427
+ # on the first bad id and `load_index` then rejects the *entire* index),
428
+ # so one malformed upstream record must cost one document, not every
429
+ # lookup. All 176,862 published ids parse today; this guards drift.
430
+ # Counted here, in the parent, because a worker's tally would be lost.
431
+ unless result[:pubid]
432
+ @unindexed = @unindexed.to_i + 1
433
+ return
434
+ end
435
+
436
+ index.add_or_update result[:pubid], result[:file]
437
+ end
438
+
439
+ #
440
+ # Settle, in the parent, which record keeps which filename.
441
+ #
442
+ # `output_file` is not injective, so distinct docids can want one path.
443
+ # `write_unique` refuses to clobber, but a forked worker cannot know
444
+ # WHICH of the clashing docids deserves the plain name — it only knows the
445
+ # path was taken. Left there, the winner would follow the race and the two
446
+ # filenames would swap between crawls, churning the data repo.
447
+ #
448
+ # So the parent decides once it can see every docid: within a group of
449
+ # records that wanted one path, the alphabetically first docid keeps it
450
+ # and the rest take their digest variant. Runs over EVERY group, not just
451
+ # clashing ones — a lone record that fell back to a digest path (a
452
+ # leftover file, any transient clash) must get its plain name back, or the
453
+ # published filename churns and the old file is orphaned.
454
+ #
455
+ # @param [Array<Hash>] results worker results, mutated in place so
456
+ # `record_index_entry` indexes the final path
457
+ #
458
+ def reconcile_output_files(results)
459
+ results.group_by { |r| r[:plain_file] }.each do |plain, group|
460
+ next if plain.nil? # hand-built results, and the bibxml format
461
+
462
+ targets = assign_output_files plain, group
463
+ record_collision targets
464
+ filled = Set.new
465
+ order(group, targets, plain).each do |result|
466
+ target = targets[result[:docid].to_s]
467
+ place_output_file result, target, filled.add?(target).nil?
468
+ end
469
+ end
470
+ end
471
+
472
+ # docid => the filename it should end up with. Sorting the *unique* docids
473
+ # leaves no tie to break, so the assignment cannot drift between crawls
474
+ # the way an unstable `sort_by` over the results would.
475
+ def assign_output_files(plain, group)
476
+ group.map { |r| r[:docid].to_s }.uniq.sort.each_with_index.to_h do |docid, i|
477
+ [docid, i.zero? ? plain : digest_output_file(docid)]
478
+ end
479
+ end
480
+
481
+ # Records already at their target first (they are no-ops), then the movers
482
+ # bound for a digest path, then the one bound for `plain`. That ordering is
483
+ # load-bearing: a loser may be sitting ON the plain path, and moving the
484
+ # winner there first would destroy it.
485
+ def order(group, targets, plain)
486
+ group.sort_by do |r|
487
+ target = targets[r[:docid].to_s]
488
+ [r[:file] == target ? 0 : 1, target == plain ? 1 : 0, r[:file].to_s]
489
+ end
490
+ end
491
+
492
+ #
493
+ # Move one record's file to the name the parent chose for it.
494
+ #
495
+ # `duplicate` means another result for the SAME docid already holds the
496
+ # target. Today that is one file and one warning, so drop the stray rather
497
+ # than publish the document twice under two names.
498
+ #
499
+ def place_output_file(result, target, duplicate)
500
+ file = result[:file]
501
+ return result[:file] = target if file.nil? || file == target
502
+
503
+ if duplicate
504
+ # The document is at the target whatever happens next, so the result
505
+ # follows it first. The stray may ALREADY be gone: two workers that
506
+ # both lost the race to the plain path land on one fallback name,
507
+ # because that name is keyed on the docid -- so the first of them
508
+ # renamed this very file onto the target. Leaving the result on its
509
+ # old name would put an index row on a path that no longer exists.
510
+ result[:file] = target
511
+ return unless File.exist?(file)
512
+
513
+ Util.warn "Duplicate document `#{result[:docid]}`: dropping #{file}, keeping #{target}"
514
+ return File.delete(file)
515
+ end
516
+ return unless File.exist?(file)
517
+
518
+ File.rename file, target
519
+ result[:file] = target
520
+ rescue SystemCallError => e
521
+ # A late failure must not throw away a multi-hour crawl. The record keeps
522
+ # the name it already has on disk.
523
+ Util.warn "Could not move #{file} to #{target}: #{e.message}"
524
+ end
525
+
526
+ def record_collision(targets)
527
+ return if targets.size < 2
528
+
529
+ # targets.keys is already the uniqued, sorted docid list.
530
+ (@collisions ||= []) << targets.keys
531
+ end
532
+
533
+ # One line for a crawl that writes ~177k records, not one per collision.
534
+ def report_collisions
535
+ return if @collisions.nil? || @collisions.empty?
536
+
537
+ Util.warn "#{@collisions.size} filename collision(s): distinct docids sanitize to " \
538
+ "one filename and were given separate files. " \
539
+ "First: #{@collisions.first(5).map { |g| g.join(' <-> ') }.join('; ')}"
540
+ end
541
+
542
+ # Skips are per-record warnings in a crawl that writes ~177k of them, so
543
+ # restate the total where it can actually be noticed.
544
+ # One line for a crawl that reads ~167k files, not one per skip.
545
+ def report_unparsed
546
+ return if @unparsed.nil? || @unparsed.empty?
547
+
548
+ Util.warn "#{@unparsed.size} file(s) skipped: could not be parsed. " \
549
+ "First: #{@unparsed.first(5).join(', ')}"
550
+ end
551
+
552
+ def report_unindexed
553
+ return unless @unindexed.to_i.positive?
554
+
555
+ Util.warn "#{@unindexed} document(s) written but not indexed: " \
556
+ "identifier not parseable by Pubid::Ietf"
274
557
  end
275
558
 
276
559
  end
@@ -59,9 +59,13 @@ module Relaton
59
59
  #
60
60
  def remove_index_file
61
61
  require_relative "../ietf"
62
- Relaton::Index.find_or_create(:RFC, url: true, file: "#{INDEXFILE}.yaml").remove_file
63
- Relaton::Index.find_or_create(:RSS, url: true, file: "#{INDEXFILE}.yaml").remove_file
64
- Relaton::Index.find_or_create(:IDS, url: true, file: "#{INDEXFILE}.yaml").remove_file
62
+ Relaton::Index.find_or_create(:IETF, url: true, file: "#{INDEXFILE}.yaml").remove_file
63
+ # Also clear the three per-type caches a previously released relaton
64
+ # left in ~/.relaton; nothing writes them now, so without this they
65
+ # would sit there forever, unreachable by `relaton clear`.
66
+ %i[RFC RSS IDS].each do |type|
67
+ Relaton::Index.find_or_create(type, url: true, file: "index-v1.yaml").remove_file
68
+ end
65
69
  end
66
70
  end
67
71
  end
@@ -125,10 +125,29 @@ module Relaton
125
125
  is_also&.doc_id&.any? || false
126
126
  end
127
127
 
128
+ #
129
+ # Normalise a doc-id for constituent lookup: fold case and strip
130
+ # whitespace and dots.
131
+ #
132
+ # Relation targets and the docids they reference disagree on both across
133
+ # the published corpus (`dyndNS` vs `dyndns`), which leaves records
134
+ # undated — and, in `build_relations`, silently downgrades a full
135
+ # constituent bibitem to a minimal one — under a strict lookup. Both
136
+ # sides go through this: the caller keys its index by it, `Entry` looks
137
+ # up by it.
138
+ #
139
+ # @param id [String, nil]
140
+ # @return [String]
141
+ #
142
+ def self.squish(id)
143
+ id.to_s.gsub(/[\s.]/, "").downcase
144
+ end
145
+
128
146
  #
129
147
  # Convert to Relaton::Ietf::ItemData
130
148
  #
131
- # @param rfc_index [Hash{String => Entry}, nil] lookup of RFC entries by doc-id
149
+ # @param rfc_index [Hash{String => Entry}, nil] lookup of RFC entries,
150
+ # keyed by `Entry.squish(doc_id)` — see that method for why
132
151
  # @return [Relaton::Ietf::ItemData, nil]
133
152
  #
134
153
  def to_item(rfc_index = nil, wg_names: {})
@@ -174,6 +193,7 @@ module Relaton
174
193
  script: ["Latn"],
175
194
  source: build_link,
176
195
  formattedref: build_formattedref,
196
+ date: build_subseries_date(rfc_index),
177
197
  relation: build_relations(rfc_index, wg_names: wg_names),
178
198
  series: build_series,
179
199
  ext: Ext.new(doctype: Doctype.new(content: "rfc"), stream: stream, flavor: "ietf"),
@@ -211,12 +231,28 @@ module Relaton
211
231
  return [] unless is_also&.doc_id
212
232
 
213
233
  is_also.doc_id.map do |ref|
214
- rfc_entry = rfc_index&.[](ref)
234
+ rfc_entry = rfc_index&.[](Entry.squish(ref))
215
235
  bibitem = rfc_entry ? rfc_entry.to_rfc_item(wg_names: wg_names) : build_minimal_bibitem(ref)
216
236
  Relaton::Ietf::Relation.new(type: "includes", bibitem: bibitem)
217
237
  end.compact
218
238
  end
219
239
 
240
+ # A sub-series entry in rfc-index.xml is a pointer and nothing more —
241
+ # `<doc-id>` plus `<is-also>`, with no date, title, author or status,
242
+ # because a sub-series has no metadata of its own. Rather than publish a
243
+ # dateless record, take the date of the newest RFC it includes.
244
+ #
245
+ # @param rfc_index [Hash{String => Entry}, nil]
246
+ # @return [Array<Bib::Date>] empty when no constituent resolves or none
247
+ # carries a date
248
+ def build_subseries_date(rfc_index)
249
+ newest = (is_also&.doc_id || [])
250
+ .filter_map { |ref| rfc_index&.[](Entry.squish(ref)) }
251
+ .flat_map(&:build_rfc_date)
252
+ .max_by { |d| d.at.to_s }
253
+ newest ? [newest] : []
254
+ end
255
+
220
256
  def build_minimal_bibitem(ref)
221
257
  id = ref.sub(/^([A-Z]+)0*(\d+)$/, '\1 \2')
222
258
  docid = Bib::Docidentifier.new(type: "IETF", content: id, primary: true)
@@ -246,7 +282,7 @@ module Relaton
246
282
  [Bib::Uri.new(type: "src", content: "https://www.rfc-editor.org/info/rfc#{shortnum}")]
247
283
  end
248
284
 
249
- def build_rfc_date
285
+ public def build_rfc_date
250
286
  (date || []).map do |d|
251
287
  month_num = ::Date::MONTHNAMES.index(d.month).to_s.rjust(2, "0")
252
288
  date_str = "#{d.year}-#{month_num}"
@@ -1,62 +1,95 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require "pubid"
4
+ require "pubid/ietf"
5
+
3
6
  module Relaton
4
7
  module Ietf
5
8
  # Scraper module
6
9
  module Scraper
7
10
  extend Scraper
8
11
 
9
- IDS = "https://raw.githubusercontent.com/relaton/relaton-data-ids/refs/heads/v2/"
10
- RFC = "https://raw.githubusercontent.com/relaton/relaton-data-rfcs/refs/heads/v2/"
11
- RSS = "https://raw.githubusercontent.com/relaton/relaton-data-rfcsubseries/refs/heads/v2/"
12
+ # The combined corpus — RFCs, the RFC sub-series and Internet-Drafts in one
13
+ # repo, with one pubid `index-v2` covering all ~177k records (relaton#109).
14
+ # It replaces the three per-type repos this flavor used to read; those keep
15
+ # publishing their `index-v1` for released relatons, untouched.
16
+ IETF = "https://raw.githubusercontent.com/relaton/relaton-data-ietf/main/"
12
17
 
13
18
  # @param text [String]
14
- # @return [RelatonIetf::IetfBibliographicItem]
19
+ # @return [Relaton::Ietf::ItemData, nil]
15
20
  def scrape_page(text)
16
- # Remove initial "IETF " string if specified
17
- ref = text.gsub(/^IETF /, "")
18
- # ref.sub!(/(?<=^(?:RFC|BCP|FYI|STD))\s(\d+)/) { $1.rjust 4, "0" }
19
- rfc_item ref
21
+ id = parse_id text.sub(/\AIETF\s+/, "")
22
+ return unless id
23
+
24
+ fetch_doc id
20
25
  rescue Timeout::Error, Errno::EINVAL, Errno::ECONNRESET, EOFError,
21
- Net::HTTPBadResponse, Net::HTTPHeaderSyntaxError,
22
- Net::ProtocolError, SocketError
23
- raise Relaton::RequestError, "No document found for #{ref} reference"
26
+ Net::HTTPBadResponse, Net::HTTPHeaderSyntaxError,
27
+ Net::ProtocolError, SocketError
28
+ raise Relaton::RequestError, "No document found for #{text} reference"
24
29
  end
25
30
 
26
31
  private
27
32
 
28
- # @param ref [String]
29
- # @return [RelatonIetf::IetfBibliographicItem]
30
- def rfc_item(ref) # rubocop:disable Metrics/MethodLength
31
- case ref
32
- when /^RFC/ then get_rfcs ref
33
- when /^(?:BCP|FYI|STD)/ then get_rfcsubseries ref
34
- when /^I-D/
35
- ref.sub!(/^I-D[.\s]/, "")
36
- get_ids ref
37
- end
38
- end
39
-
40
- def get_rfcs(ref)
41
- index = Relaton::Index.find_or_create :RFC, url: "#{RFC}#{INDEXFILE}.zip", file: "#{INDEXFILE}.yaml"
42
- row = index.search(ref).first
43
- get_page "#{RFC}#{row[:file]}" if row
33
+ # The index stores parsed pubids, so the query has to be one too.
34
+ # `Type#search_candidates` narrows only when the query is not a String
35
+ # (`@file_io.sorted && id && !id.is_a?(String)`); a String falls through to
36
+ # `match_item`'s `item[:id].to_s.include?(id)`, which renders every pubid in
37
+ # the index on every lookup — measured at ~40 s per reference against the
38
+ # 177k-row index, versus sub-millisecond for a parsed one.
39
+ # `exact:` keeps the match exact. `Type#search` without it takes
40
+ # pubid's subset match, where a component the query omits is a wildcard —
41
+ # and an Internet-Draft slug omits the version, so `draft-foo` would match
42
+ # every `draft-foo-NN` and `.first` would return an arbitrary one (20,513
43
+ # such pairs in the first 40k rows of the index fixture).
44
+ def fetch_doc(id)
45
+ row = index.search(id, exact: true).first
46
+ get_page "#{IETF}#{row[:file]}" if row
44
47
  end
45
48
 
46
- def get_rfcsubseries(ref)
47
- index = Relaton::Index.find_or_create :RSS, url: "#{RSS}#{INDEXFILE}.zip", file: "#{INDEXFILE}.yaml"
48
- row = index.search(ref).first
49
- get_page "#{RSS}#{row[:file]}" if row
49
+ def index
50
+ Relaton::Index.find_or_create(
51
+ :IETF, url: "#{IETF}#{INDEXFILE}.zip", file: "#{INDEXFILE}.yaml",
52
+ pubid_class: ::Pubid::Ietf::Identifier
53
+ )
50
54
  end
51
55
 
52
- def get_ids(ref)
53
- index = Relaton::Index.find_or_create :IDS, url: "#{IDS}#{INDEXFILE}.zip", file: "#{INDEXFILE}.yaml"
54
- row = index.search(ref).first
55
- get_page "#{IDS}#{row[:file]}" if row
56
+ #
57
+ # Parse a reference into the identifier the index is keyed by.
58
+ #
59
+ # Normalises the two Internet-Draft spellings callers use — `I-D.<slug>`
60
+ # and `I-D <slug>` — onto the `draft-…` form pubid parses and the index
61
+ # stores. The bare `I-D.ietf-quic-transport` spelling (the bibxml anchor,
62
+ # and the `docnumber` IETF records carry) gains the `draft-` stem: the old
63
+ # plain-string index matched it by substring, and matching is exact now.
64
+ #
65
+ # The capture is `(\S.*)`, not `(.+)`: it must start non-space so it cannot
66
+ # overlap the preceding `\s*`. Behaviour is unchanged on every real
67
+ # reference — with `(.+)` the greedy `\s*` only ever yields ground on an
68
+ # all-whitespace remainder, which pubid rejects anyway — but the
69
+ # unambiguous form is about twice as fast on a pathological input and is
70
+ # what CodeQL's rb/polynomial-redos models. Measured before changing it:
71
+ # the old form was already linear (0.66 ms at 80k chars), so this is
72
+ # clarity, not a ReDoS fix.
73
+ #
74
+ # @param ref [String]
75
+ # @return [Pubid::Ietf::Identifier, nil] nil when pubid has no grammar for
76
+ # it, so an out-of-flavor reference logs "Not found." instead of raising
77
+ #
78
+ def parse_id(ref)
79
+ if (draft = ref[/\AI-D[.\s]\s*(\S.*)\z/m, 1])
80
+ ref = draft.start_with?("draft-") ? draft : "draft-#{draft}"
81
+ end
82
+ ::Pubid::Ietf::Identifier.parse ref
83
+ rescue StandardError => e
84
+ # Logged, not swallowed: this repo git-pins pubid to a moving `main`,
85
+ # so a grammar regression would otherwise present as every IETF
86
+ # reference quietly reporting "Not found."
87
+ Util.debug "`#{ref}` is not an IETF identifier: #{e.message}"
88
+ nil
56
89
  end
57
90
 
58
91
  # @param uri [String]
59
- # @return [RelatonIetf::IetfBibliographicItem, nil] HTTP response body
92
+ # @return [Relaton::Ietf::ItemData, nil] HTTP response body
60
93
  def get_page(uri)
61
94
  res = Net::HTTP.get_response(URI(uri))
62
95
  return unless res.code == "200"
data/lib/relaton/ietf.rb CHANGED
@@ -14,7 +14,10 @@ require_relative "ietf/bibliography"
14
14
 
15
15
  module Relaton
16
16
  module Ietf
17
- INDEXFILE = "index-v1".freeze
17
+ # The pubid-structured index, written by DataFetcher and read by Scraper
18
+ # (relaton#109). One constant again now that both sides are on it — the
19
+ # second one existed only while producer and consumer were split.
20
+ INDEXFILE = "index-v2".freeze
18
21
  # Returns hash of XML reammar
19
22
  # @return [String]
20
23
  def self.grammar_hash
@@ -89,7 +89,7 @@ module Relaton
89
89
  :iho,
90
90
  url: "#{ENDPOINT}#{INDEXFILE}.zip",
91
91
  file: "#{INDEXFILE}.yaml",
92
- pubid_class: ::Pubid::Iho::Identifiers::Base,
92
+ pubid_class: ::Pubid::Iho::Identifier,
93
93
  )
94
94
  end
95
95
 
@@ -24,7 +24,7 @@ module Relaton
24
24
  end
25
25
 
26
26
  def to_all_parts!
27
- @pubid&.all_parts = true
27
+ @pubid &&= @pubid.to_all_parts
28
28
  end
29
29
  end
30
30
  end