iev 0.4.7 → 0.4.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. checksums.yaml +4 -4
  2. data/.claude/scheduled_tasks.lock +1 -0
  3. data/.rubocop_todo.yml +5 -5
  4. data/Gemfile +1 -0
  5. data/TODO.reconcile/01-design.md +57 -0
  6. data/TODO.reconcile/02-change-models.md +44 -0
  7. data/TODO.reconcile/02-termbase-loader.md +48 -0
  8. data/TODO.reconcile/03-live-loader.md +28 -0
  9. data/TODO.reconcile/03-status-mapper.md +24 -0
  10. data/TODO.reconcile/04-concept-merger.md +45 -0
  11. data/TODO.reconcile/04-termbase-loader.md +28 -0
  12. data/TODO.reconcile/05-content-diff.md +45 -0
  13. data/TODO.reconcile/05-live-loader.md +20 -0
  14. data/TODO.reconcile/06-content-differ.md +32 -0
  15. data/TODO.reconcile/06-v3-serializer.md +42 -0
  16. data/TODO.reconcile/07-cli-and-run.md +27 -0
  17. data/TODO.reconcile/07-concept-merger.md +32 -0
  18. data/TODO.reconcile/08-report.md +53 -0
  19. data/TODO.reconcile/09-pipeline.md +23 -0
  20. data/TODO.reconcile/10-script-and-run.md +15 -0
  21. data/data/locales/glossarist_enums.yml +705 -0
  22. data/iev.gemspec +3 -1
  23. data/lib/iev/cli/command.rb +29 -0
  24. data/lib/iev/cli/command_helper.rb +1 -6
  25. data/lib/iev/config.rb +10 -1
  26. data/lib/iev/exporter.rb +4 -4
  27. data/lib/iev/multi_doc_yaml.rb +60 -0
  28. data/lib/iev/reconciler/change.rb +29 -0
  29. data/lib/iev/reconciler/change_set.rb +48 -0
  30. data/lib/iev/reconciler/concept_merger.rb +130 -0
  31. data/lib/iev/reconciler/content_differ.rb +157 -0
  32. data/lib/iev/reconciler/live_loader.rb +163 -0
  33. data/lib/iev/reconciler/pipeline.rb +104 -0
  34. data/lib/iev/reconciler/reconciled_concept.rb +14 -0
  35. data/lib/iev/reconciler/report.rb +118 -0
  36. data/lib/iev/reconciler/status_mapper.rb +29 -0
  37. data/lib/iev/reconciler/term_marker_parser.rb +196 -0
  38. data/lib/iev/reconciler/termbase_loader.rb +141 -0
  39. data/lib/iev/reconciler.rb +17 -0
  40. data/lib/iev/scraper/browser.rb +155 -50
  41. data/lib/iev/scraper/page_parser.rb +116 -24
  42. data/lib/iev/source_parser.rb +12 -1
  43. data/lib/iev/subject_area_concepts.rb +4 -2
  44. data/lib/iev/utilities.rb +7 -5
  45. data/lib/iev/version.rb +1 -1
  46. data/lib/iev.rb +2 -0
  47. data/scripts/find_unparseable.rb +22 -0
  48. metadata +50 -4
@@ -6,6 +6,10 @@ module Iev
6
6
  class Scraper
7
7
  # Shared headless browser utilities for fetching pages behind AWS WAF.
8
8
  module Browser
9
+ # Each profile is tagged with the host platform it can run on so
10
+ # that navigator.platform (set by Chrome from the real OS) agrees
11
+ # with the User-Agent and Sec-Ch-Ua-Platform headers. AWS WAF
12
+ # fingerprints this mismatch and refuses to clear its challenge.
9
13
  USER_AGENT_PROFILES = [
10
14
  {
11
15
  user_agent: "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) " \
@@ -13,6 +17,15 @@ module Iev
13
17
  "Chrome/131.0.0.0 Safari/537.36",
14
18
  platform: '"macOS"',
15
19
  chrome_version: "131",
20
+ host_platform: :mac,
21
+ },
22
+ {
23
+ user_agent: "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) " \
24
+ "AppleWebKit/537.36 (KHTML, like Gecko) " \
25
+ "Chrome/129.0.0.0 Safari/537.36",
26
+ platform: '"macOS"',
27
+ chrome_version: "129",
28
+ host_platform: :mac,
16
29
  },
17
30
  {
18
31
  user_agent: "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " \
@@ -20,85 +33,177 @@ module Iev
20
33
  "Chrome/130.0.0.0 Safari/537.36",
21
34
  platform: '"Windows"',
22
35
  chrome_version: "130",
36
+ host_platform: :windows,
23
37
  },
24
38
  {
25
- user_agent: "Mozilla/5.0 (X11; Linux x86_64) " \
39
+ user_agent: "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " \
26
40
  "AppleWebKit/537.36 (KHTML, like Gecko) " \
27
41
  "Chrome/131.0.0.0 Safari/537.36",
28
- platform: '"Linux"',
42
+ platform: '"Windows"',
29
43
  chrome_version: "131",
44
+ host_platform: :windows,
30
45
  },
31
46
  {
32
- user_agent: "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) " \
33
- "AppleWebKit/537.36 (KHTML, like Gecko) " \
34
- "Chrome/129.0.0.0 Safari/537.36",
35
- platform: '"macOS"',
36
- chrome_version: "129",
37
- },
38
- {
39
- user_agent: "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " \
47
+ user_agent: "Mozilla/5.0 (X11; Linux x86_64) " \
40
48
  "AppleWebKit/537.36 (KHTML, like Gecko) " \
41
49
  "Chrome/131.0.0.0 Safari/537.36",
42
- platform: '"Windows"',
50
+ platform: '"Linux"',
43
51
  chrome_version: "131",
52
+ host_platform: :linux,
44
53
  },
45
54
  ].freeze
46
55
 
47
- # Fetch a URL using headless Chrome, returning the page HTML.
48
- # Handles AWS WAF challenge pages by waiting for JS execution.
49
- def self.fetch(url, browser_opts: {})
50
- browser = Ferrum::Browser.new(
51
- headless: "new",
52
- timeout: 30,
53
- window_size: [1366, 768],
54
- browser_options: {
55
- "disable-blink-features" => "AutomationControlled",
56
- },
57
- **browser_opts,
58
- )
59
-
60
- browser.headers.set(random_headers)
61
- browser.go_to(url)
62
- browser.network.wait_for_idle(timeout: 15)
63
- html = browser.body
64
-
65
- if html.include?("403 ERROR") || html.include?("Request blocked")
66
- warn "IEV: AWS WAF blocked request for #{url}"
67
- return nil
68
- end
56
+ DEFAULT_LANG = "en-US,en"
57
+ DEFAULT_BROWSER_OPTIONS = {
58
+ "disable-blink-features" => "AutomationControlled",
59
+ "lang" => DEFAULT_LANG,
60
+ }.freeze
69
61
 
70
- html
71
- rescue Ferrum::Error, Ferrum::BrowserError => e
72
- warn "IEV: Browser error fetching #{url}: #{e.message}"
73
- nil
74
- ensure
75
- browser&.quit
62
+ # One-shot fetch. Each call spins up a fresh headless Chrome, fetches,
63
+ # and tears it down. Suitable for ad-hoc use; the WAF cookie does not
64
+ # survive between calls. Batch callers (Fetcher::Mirror) should use
65
+ # Session instead so the cookie set on the first successful challenge
66
+ # is reused across requests.
67
+ def self.fetch(url, **_browser_opts)
68
+ Session.new.fetch(url)
76
69
  end
77
70
 
71
+ # Returns request headers that match what real Chrome sends on a
72
+ # fresh address-bar navigation. AWS WAF fingerprints inconsistencies
73
+ # between these headers and the browser's runtime state, so we:
74
+ # - omit Sec-Fetch-* (Chrome computes those itself from the
75
+ # navigation context; setting them via Ferrum's Network domain
76
+ # overrides the real values and is detectable), and
77
+ # - keep Sec-Ch-Ua-Platform aligned with the host OS, which is
78
+ # what Chrome reports via navigator.platform.
78
79
  def self.random_headers
79
- profile = USER_AGENT_PROFILES.sample
80
- sec_ch_ua = "\"Google Chrome\";v=\"#{profile[:chrome_version]}\", " \
81
- "\"Chromium\";v=\"#{profile[:chrome_version]}\", " \
82
- "\"Not_A Brand\";v=\"24\""
80
+ profile = profile_for_host
81
+ static_headers.merge(headers_from_profile(profile))
82
+ end
83
83
 
84
+ def self.static_headers
84
85
  {
85
86
  "Accept" => "text/html,application/xhtml+xml,application/xml;q=0.9," \
86
87
  "image/avif,image/webp,image/apng,*/*;q=0.8," \
87
88
  "application/signed-exchange;v=b3;q=0.7",
88
- "Accept-Language" => "en-GB,en-US;q=0.9,en;q=0.8",
89
+ "Accept-Language" => "en-US,en;q=0.9",
89
90
  "Cache-Control" => "no-cache",
90
91
  "Pragma" => "no-cache",
91
- "Sec-Ch-Ua" => sec_ch_ua,
92
92
  "Sec-Ch-Ua-Mobile" => "?0",
93
- "Sec-Ch-Ua-Platform" => profile[:platform],
94
- "Sec-Fetch-Dest" => "document",
95
- "Sec-Fetch-Mode" => "navigate",
96
- "Sec-Fetch-Site" => "cross-site",
97
- "Sec-Fetch-User" => "?1",
98
93
  "Upgrade-Insecure-Requests" => "1",
94
+ }
95
+ end
96
+
97
+ def self.headers_from_profile(profile)
98
+ {
99
+ "Sec-Ch-Ua" => sec_ch_ua_for(profile),
100
+ "Sec-Ch-Ua-Platform" => profile[:platform],
99
101
  "User-Agent" => profile[:user_agent],
100
102
  }
101
103
  end
104
+
105
+ def self.profile_for_host
106
+ USER_AGENT_PROFILES.select do |profile|
107
+ profile[:host_platform] == host_platform
108
+ end.sample
109
+ end
110
+
111
+ def self.host_platform
112
+ case RUBY_PLATFORM
113
+ when /darwin/ then :mac
114
+ when /mswin|mingw|cygwin|bccwin|wince|emx/ then :windows
115
+ else :linux
116
+ end
117
+ end
118
+
119
+ def self.sec_ch_ua_for(profile)
120
+ "\"Google Chrome\";v=\"#{profile[:chrome_version]}\", " \
121
+ "\"Chromium\";v=\"#{profile[:chrome_version]}\", " \
122
+ "\"Not_A Brand\";v=\"24\""
123
+ end
124
+
125
+ # A long-lived headless Chrome session. Cookies persist across
126
+ # fetches, so once the AWS WAF challenge is cleared on the first
127
+ # request, subsequent requests reuse the token and succeed at
128
+ # near-100% rate. The Mirror creates one Session per run and
129
+ # shares it across all SequentialProbe iterations.
130
+ #
131
+ # Chrome leaks ~1MB per page load (mostly V8 heap that doesn't get
132
+ # GC'd between navigations). After ~1000 fetches the process is at
133
+ # ~1GB and the OOM risk climbs sharply. #restart quits the browser
134
+ # and starts a fresh one with a new cookie jar; the WAF challenge
135
+ # will need to be cleared again on the next fetch.
136
+ class Session
137
+ def initialize
138
+ @browser = Ferrum::Browser.new(
139
+ headless: "new",
140
+ timeout: 30,
141
+ window_size: [1366, 768],
142
+ browser_options: Browser::DEFAULT_BROWSER_OPTIONS,
143
+ )
144
+ @browser.headers.set(Browser.random_headers)
145
+ end
146
+
147
+ def fetch(url)
148
+ @browser.go_to(url)
149
+ @browser.network.wait_for_idle(timeout: 15)
150
+ reject_blocked(url, @browser.body)
151
+ rescue Ferrum::DeadBrowserError, Ferrum::NoSuchPageError,
152
+ Ferrum::NoSuchTargetError => e
153
+ # Chrome process or page/tab has crashed. Restart once, retry
154
+ # once. If the restart itself fails, surface as nil so the
155
+ # probe silently skips and the run continues.
156
+ if restart
157
+ retry
158
+ else
159
+ warn "IEV: Browser crashed, restart failed: #{e.message}"
160
+ nil
161
+ end
162
+ rescue Ferrum::BrowserError => e
163
+ if fetch_dead?(e.message) && restart
164
+ retry
165
+ else
166
+ warn "IEV: Browser error fetching #{url}: #{e.message}"
167
+ nil
168
+ end
169
+ rescue Ferrum::Error => e
170
+ warn "IEV: Browser error fetching #{url}: #{e.message}"
171
+ nil
172
+ end
173
+
174
+ def fetch_dead?(message)
175
+ message.include?("Browser is dead") ||
176
+ message.include?("given window is closed") ||
177
+ message.include?("Target closed")
178
+ end
179
+
180
+ def reject_blocked(url, html)
181
+ if html.include?("403 ERROR") || html.include?("Request blocked")
182
+ warn "IEV: AWS WAF blocked request for #{url}"
183
+ nil
184
+ else
185
+ html
186
+ end
187
+ end
188
+
189
+ # Quit the current browser and start a fresh one. The WAF cookie
190
+ # is lost, so the next fetch will go through the challenge cycle
191
+ # again. Call this after N fetches to bound Ferrum's memory growth,
192
+ # or rely on #fetch to call it automatically when Chrome dies.
193
+ # Returns true on success, false if the new browser fails to start.
194
+ def restart
195
+ quit
196
+ initialize
197
+ true
198
+ rescue StandardError => e
199
+ warn "IEV: Session restart failed: #{e.message}"
200
+ false
201
+ end
202
+
203
+ def quit
204
+ @browser&.quit
205
+ end
206
+ end
102
207
  end
103
208
  end
104
209
  end
@@ -4,13 +4,14 @@ module Iev
4
4
  class Scraper
5
5
  # Parses an Electropedia HTML page into a concept data hash.
6
6
  #
7
- # The Electropedia HTML structure is a table with rows for each language:
8
- # - Language row: <div align="center"><font color="#800080">en</font></div>
9
- # - Term cell: <b>term text</b> in the third <td>
10
- # - Definition row: next row's third <td> (if present)
11
- # - Empty/separator rows with <hr> or spacer images
7
+ # All extracted text (terms, definitions, notes, examples) is run
8
+ # through Iev::Utilities' semantic enrichment pipeline, converting
9
+ # HTML to AsciiDoc notation: <i>x</i> → stem:[x], <b>x</b> → *x*,
10
+ # <a href=IEV...> → {{urn:...}}, MathML → AsciiMath. This ensures
11
+ # the parsed output matches the format used in termbase.yaml.
12
12
  class PageParser
13
- # Map Electropedia HTML language codes to ISO 639-2/3 three-char codes.
13
+ include Utilities
14
+
14
15
  LANG_CODE_MAP = {
15
16
  "en" => "eng",
16
17
  "fr" => "fra",
@@ -54,28 +55,48 @@ module Iev
54
55
  private
55
56
 
56
57
  def find_iev_ref
57
- # Find the IEV reference cell to confirm the page is valid
58
58
  @doc.at_css("b:contains('#{@code}')") ||
59
59
  @doc.at_xpath("//td/b[contains(text(), '#{@code}')]")
60
60
  end
61
61
 
62
+ def term_domain
63
+ @code.rpartition("-").first
64
+ end
65
+
62
66
  def localized_concepts
63
67
  result = {}
64
- lang_sections.each do |lang, term_row, def_row|
68
+ lang_sections.each do |lang, area, term_row, def_row|
65
69
  term = extract_term(term_row)
66
70
  next unless term
67
71
 
68
- entry = { "term" => term }
72
+ entry = result[lang] || { "terms" => [] }
73
+ term_entry = { "designation" => term }
74
+ term_entry["geographical_area"] = area if area
75
+ entry["terms"] << term_entry
76
+
69
77
  definition = extract_definition(def_row)
70
- entry["definition"] = definition if definition
78
+ entry["definition"] = definition if definition && !entry.key?("definition")
71
79
 
72
80
  result[lang] = entry
73
81
  end
82
+
83
+ # Convert terms array to single "term" field for backward compat
84
+ # when there's only one term and no area
85
+ result.each_value do |entry|
86
+ terms = entry.delete("terms")
87
+ if terms.size == 1 && !terms.first["geographical_area"]
88
+ entry["term"] = terms.first["designation"]
89
+ else
90
+ entry["term"] = terms.map { |t| t["designation"] }.join("\n")
91
+ entry["term_areas"] = terms.each_with_object({}) do |t, h|
92
+ area = t["geographical_area"]
93
+ h[t["designation"]] = area if area
94
+ end
95
+ end
96
+ end
74
97
  result
75
98
  end
76
99
 
77
- # Finds all language sections in the table.
78
- # Returns array of [lang_code, term_row, definition_row] tuples.
79
100
  def lang_sections
80
101
  sections = []
81
102
  rows = content_rows
@@ -84,17 +105,30 @@ module Iev
84
105
  lang = extract_lang(row)
85
106
  next unless lang
86
107
 
87
- # The definition is in the next non-empty, non-separator row
108
+ area = extract_area(row)
88
109
  def_row = find_definition_row(rows, idx + 1)
89
- sections << [lang, row, def_row]
110
+ sections << [lang, area, row, def_row]
90
111
  end
91
112
 
92
113
  sections
93
114
  end
94
115
 
116
+ def extract_area(row)
117
+ tds = row.css("td")
118
+ return nil if tds.length < 2
119
+
120
+ area_cell = tds[1]
121
+ font = area_cell.at_css("font[color='#800080']")
122
+ return nil unless font
123
+
124
+ text = font.text.strip.downcase
125
+ return nil if text.empty?
126
+ return nil if text.match?(/\A(nl|be|nb|nn|br|pt|fr|de|us|gb|ch|at|ca|au|nz)\z/i) == false
127
+
128
+ text
129
+ end
130
+
95
131
  def content_rows
96
- # Find the main content table (the one with language data)
97
- # It's the largest table with IEV data
98
132
  tables = @doc.css("table")
99
133
  content_table = tables.max_by { |t| t.css("tr").length }
100
134
  content_table ? content_table.css("tr").to_a : []
@@ -109,15 +143,30 @@ module Iev
109
143
  end
110
144
 
111
145
  def extract_term(row)
112
- # Term is in the third <td> — may be in a <b> tag (en, fr) or plain text
113
146
  tds = row.css("td")
114
147
  return nil if tds.length < 3
115
148
 
116
149
  content_td = tds[2]
150
+ return nil if area_marker_cell?(content_td)
151
+
117
152
  bold = content_td.at_css("b")
118
153
 
119
- term = bold ? bold.text.strip : content_td.text.strip
120
- term.empty? ? nil : term
154
+ html = if bold
155
+ bold.inner_html.strip
156
+ else
157
+ content_td.inner_html.strip
158
+ end
159
+ return nil if html.empty?
160
+
161
+ enrich(html)
162
+ end
163
+
164
+ # True when the cell contains only a geographical-area marker font
165
+ # tag (e.g. <font color="#800080">nb</font>), not a real term.
166
+ def area_marker_cell?(cell)
167
+ font = cell.at_css("font[color='#800080']")
168
+ return false unless font
169
+ cell.at_css("b").nil? && cell.text.strip == font.text.strip && font.text.strip.length <= 3
121
170
  end
122
171
 
123
172
  def extract_definition(row)
@@ -127,15 +176,59 @@ module Iev
127
176
  return nil if tds.length < 3
128
177
 
129
178
  content_td = tds[2]
130
- # The definition is the text content, which may include MathML
131
179
  html = content_td.inner_html.strip
132
180
  return nil if html.empty? || html.match?(/\A<img.*ecblank/)
133
181
 
134
- html
182
+ enrich(html)
183
+ end
184
+
185
+ def enrich(html_text)
186
+ text = html_text.to_s
187
+ .gsub(/&lt;i&gt;/, "<i>")
188
+ .gsub(/&lt;\/i&gt;/, "</i>")
189
+ mathml_blocks, text = extract_mathml_blocks(text)
190
+ result = parse_anchor_tag(text, term_domain)
191
+ result = restore_and_convert_mathml(result, mathml_blocks)
192
+ result = replace_newlines(result)
193
+ result = convert_literal_italic(result)
194
+ result.strip
195
+ end
196
+
197
+ MATHML_PLACEHOLDER = "@@MATHML_%d@@"
198
+
199
+ def extract_mathml_blocks(text)
200
+ blocks = []
201
+ result = text.gsub(/<math[^>]*>.*?<\/math>/m) do |match|
202
+ blocks << match
203
+ format(MATHML_PLACEHOLDER, blocks.size - 1)
204
+ end
205
+ [blocks, result]
206
+ end
207
+
208
+ # Restore MathML blocks and convert each to AsciiMath individually.
209
+ # If the converter returns empty (unsupported expression like
210
+ # <mtable>), keep the original MathML so the formula is not lost.
211
+ def restore_and_convert_mathml(text, blocks)
212
+ return text if blocks.empty?
213
+
214
+ blocks.each_with_index do |mathml, i|
215
+ converted = Iev::Converter.mathml_to_asciimath(mathml).strip
216
+ replacement = converted.empty? ? mathml : converted
217
+ text = text.sub(format(MATHML_PLACEHOLDER, i), replacement)
218
+ end
219
+ text
220
+ end
221
+
222
+ # After Nokogiri processing, &lt;i&gt;entity-encoded tags appear as
223
+ # literal <i>...</i> text. Convert them to stem:[...] just like
224
+ # real <i> elements are converted by convert_italic.
225
+ def convert_literal_italic(text)
226
+ text.gsub(/<i>([^<]*)<\/i>/) do
227
+ inner = Regexp.last_match(1)
228
+ convert_italic(inner)
229
+ end
135
230
  end
136
231
 
137
- # Find the definition row following a language row.
138
- # Skip separator rows (empty, <hr>, or spacer images).
139
232
  def find_definition_row(rows, start_idx)
140
233
  return nil if start_idx >= rows.length
141
234
 
@@ -149,7 +242,6 @@ module Iev
149
242
  content = tds[2].inner_html.strip
150
243
  return nil if content.empty?
151
244
 
152
- # Skip rows that are only spacer images (unless they have <b> content)
153
245
  if content.match?(/\A<img.*ecblank/) && !content.include?("<b>")
154
246
  return nil
155
247
  end
@@ -406,7 +406,18 @@ module Iev
406
406
  return nil unless item
407
407
 
408
408
  item.source("src")
409
- rescue Relaton::RequestError, Socket::ResolutionError, SocketError => e
409
+ # Free-text refs like "ITU-R Rec. 431 MOD" are not parseable pubids.
410
+ # relaton 3 (pubid 2) raises Pubid::Errors::ParseError (a
411
+ # Parslet::ParseFailed) eagerly when building the search, whereas
412
+ # relaton 2 swallowed it. Link resolution is best-effort: warn and
413
+ # skip instead of failing the whole source parse.
414
+ # relaton 3.0.0.pre.alpha.6 (#205 strict routing) also raises
415
+ # UnknownReferenceError for spellings no flavor claims (the French
416
+ # "CEI", agency refs like "IAEA 4") where alpha.5's permissive
417
+ # parse answered nil. Same best-effort contract: warn and skip.
418
+ rescue Relaton::RequestError, Relaton::UnknownReferenceError,
419
+ Socket::ResolutionError, SocketError,
420
+ Parslet::ParseFailed => e
410
421
  warn e.message
411
422
  nil
412
423
  end
@@ -128,10 +128,12 @@ module Iev
128
128
 
129
129
  # --- RelatedConcept factory methods ---
130
130
 
131
+ # RelatedConcept#content is a lang-keyed hash since glossarist 2.13
132
+ # (L10N invariant: all relationship free-text is { "eng" => ... }).
131
133
  def broader_relation(target_uri)
132
134
  Glossarist::RelatedConcept.new(
133
135
  type: "broader",
134
- content: target_uri,
136
+ content: { "eng" => target_uri },
135
137
  ref: Glossarist::ConceptRef.new(source: "IEV", id: target_uri),
136
138
  )
137
139
  end
@@ -139,7 +141,7 @@ module Iev
139
141
  def narrower_relation(target_uri)
140
142
  Glossarist::RelatedConcept.new(
141
143
  type: "narrower",
142
- content: target_uri,
144
+ content: { "eng" => target_uri },
143
145
  ref: Glossarist::ConceptRef.new(source: "IEV", id: target_uri),
144
146
  )
145
147
  end
data/lib/iev/utilities.rb CHANGED
@@ -92,7 +92,9 @@ module Iev
92
92
  when "img"
93
93
  src = node["src"] || node.attributes.keys.first.to_s
94
94
  "#{IMAGE_PATH_PREFIX}/#{term_domain}/#{src}[]"
95
- when "p", "div", "span"
95
+ when "p"
96
+ "#{inner}\n"
97
+ when "div", "span"
96
98
  inner
97
99
  when "i"
98
100
  convert_italic(inner)
@@ -101,9 +103,9 @@ module Iev
101
103
  when "sup"
102
104
  inner.empty? ? "" : "^#{inner}^"
103
105
  when "ol"
104
- convert_list(node, ". ")
106
+ convert_list(node, ". ", term_domain)
105
107
  when "ul"
106
- convert_list(node, "* ")
108
+ convert_list(node, "* ", term_domain)
107
109
  when "li"
108
110
  inner
109
111
  when "font"
@@ -124,8 +126,8 @@ module Iev
124
126
  end
125
127
  end
126
128
 
127
- def convert_list(node, prefix)
128
- node.css("li").map { |li| "#{prefix}#{li.text}" }.join
129
+ def convert_list(node, prefix, term_domain)
130
+ node.css("li").map { |li| "#{prefix}#{nodes_to_adoc(li.children, term_domain)}" }.join("\n")
129
131
  end
130
132
 
131
133
  def convert_font(node, inner)
data/lib/iev/version.rb CHANGED
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Iev
4
- VERSION = "0.4.7"
4
+ VERSION = "0.4.8"
5
5
  end
data/lib/iev.rb CHANGED
@@ -32,7 +32,9 @@ module Iev
32
32
  autoload :FigureBuilder, "iev/figure_builder"
33
33
  autoload :IevCode, "iev/iev_code"
34
34
  autoload :Iso639Code, "iev/iso_639_code"
35
+ autoload :MultiDocYaml, "iev/multi_doc_yaml"
35
36
  autoload :Profiler, "iev/profiler"
37
+ autoload :Reconciler, "iev/reconciler"
36
38
  autoload :RelatonDb, "iev/relaton_db"
37
39
  autoload :Scraper, "iev/scraper"
38
40
  autoload :Section, "iev/section"
@@ -0,0 +1,22 @@
1
+ #!/usr/bin/env ruby
2
+ # Identify cached HTML pages that PageParser cannot turn into a concept.
3
+ # Run: bundle exec ruby -Ilib scripts/find_unparseable.rb
4
+ require "iev"
5
+
6
+ store = Iev::Fetcher::PageStore.new
7
+ ok = 0
8
+ failed = []
9
+ store.each_concept(scope: Iev::Fetcher::Scope.all).each do |code, html|
10
+ doc = Nokogiri::HTML(html)
11
+ if Iev::Scraper::PageParser.new(doc, code).parse
12
+ ok += 1
13
+ else
14
+ failed << code
15
+ end
16
+ end
17
+
18
+ puts "Parsed OK: #{ok}"
19
+ puts "Failed: #{failed.size}"
20
+ puts
21
+ puts "First 20 failed codes:"
22
+ failed.take(20).each { |c| puts " #{c}" }