iev 0.4.7 → 0.4.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.claude/scheduled_tasks.lock +1 -0
- data/.rubocop_todo.yml +5 -5
- data/Gemfile +1 -0
- data/TODO.reconcile/01-design.md +57 -0
- data/TODO.reconcile/02-change-models.md +44 -0
- data/TODO.reconcile/02-termbase-loader.md +48 -0
- data/TODO.reconcile/03-live-loader.md +28 -0
- data/TODO.reconcile/03-status-mapper.md +24 -0
- data/TODO.reconcile/04-concept-merger.md +45 -0
- data/TODO.reconcile/04-termbase-loader.md +28 -0
- data/TODO.reconcile/05-content-diff.md +45 -0
- data/TODO.reconcile/05-live-loader.md +20 -0
- data/TODO.reconcile/06-content-differ.md +32 -0
- data/TODO.reconcile/06-v3-serializer.md +42 -0
- data/TODO.reconcile/07-cli-and-run.md +27 -0
- data/TODO.reconcile/07-concept-merger.md +32 -0
- data/TODO.reconcile/08-report.md +53 -0
- data/TODO.reconcile/09-pipeline.md +23 -0
- data/TODO.reconcile/10-script-and-run.md +15 -0
- data/data/locales/glossarist_enums.yml +705 -0
- data/iev.gemspec +3 -1
- data/lib/iev/cli/command.rb +29 -0
- data/lib/iev/cli/command_helper.rb +1 -6
- data/lib/iev/config.rb +10 -1
- data/lib/iev/exporter.rb +4 -4
- data/lib/iev/multi_doc_yaml.rb +60 -0
- data/lib/iev/reconciler/change.rb +29 -0
- data/lib/iev/reconciler/change_set.rb +48 -0
- data/lib/iev/reconciler/concept_merger.rb +130 -0
- data/lib/iev/reconciler/content_differ.rb +157 -0
- data/lib/iev/reconciler/live_loader.rb +163 -0
- data/lib/iev/reconciler/pipeline.rb +104 -0
- data/lib/iev/reconciler/reconciled_concept.rb +14 -0
- data/lib/iev/reconciler/report.rb +118 -0
- data/lib/iev/reconciler/status_mapper.rb +29 -0
- data/lib/iev/reconciler/term_marker_parser.rb +196 -0
- data/lib/iev/reconciler/termbase_loader.rb +141 -0
- data/lib/iev/reconciler.rb +17 -0
- data/lib/iev/scraper/browser.rb +155 -50
- data/lib/iev/scraper/page_parser.rb +116 -24
- data/lib/iev/source_parser.rb +12 -1
- data/lib/iev/subject_area_concepts.rb +4 -2
- data/lib/iev/utilities.rb +7 -5
- data/lib/iev/version.rb +1 -1
- data/lib/iev.rb +2 -0
- data/scripts/find_unparseable.rb +22 -0
- metadata +50 -4
data/lib/iev/scraper/browser.rb
CHANGED
|
@@ -6,6 +6,10 @@ module Iev
|
|
|
6
6
|
class Scraper
|
|
7
7
|
# Shared headless browser utilities for fetching pages behind AWS WAF.
|
|
8
8
|
module Browser
|
|
9
|
+
# Each profile is tagged with the host platform it can run on so
|
|
10
|
+
# that navigator.platform (set by Chrome from the real OS) agrees
|
|
11
|
+
# with the User-Agent and Sec-Ch-Ua-Platform headers. AWS WAF
|
|
12
|
+
# fingerprints this mismatch and refuses to clear its challenge.
|
|
9
13
|
USER_AGENT_PROFILES = [
|
|
10
14
|
{
|
|
11
15
|
user_agent: "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) " \
|
|
@@ -13,6 +17,15 @@ module Iev
|
|
|
13
17
|
"Chrome/131.0.0.0 Safari/537.36",
|
|
14
18
|
platform: '"macOS"',
|
|
15
19
|
chrome_version: "131",
|
|
20
|
+
host_platform: :mac,
|
|
21
|
+
},
|
|
22
|
+
{
|
|
23
|
+
user_agent: "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) " \
|
|
24
|
+
"AppleWebKit/537.36 (KHTML, like Gecko) " \
|
|
25
|
+
"Chrome/129.0.0.0 Safari/537.36",
|
|
26
|
+
platform: '"macOS"',
|
|
27
|
+
chrome_version: "129",
|
|
28
|
+
host_platform: :mac,
|
|
16
29
|
},
|
|
17
30
|
{
|
|
18
31
|
user_agent: "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " \
|
|
@@ -20,85 +33,177 @@ module Iev
|
|
|
20
33
|
"Chrome/130.0.0.0 Safari/537.36",
|
|
21
34
|
platform: '"Windows"',
|
|
22
35
|
chrome_version: "130",
|
|
36
|
+
host_platform: :windows,
|
|
23
37
|
},
|
|
24
38
|
{
|
|
25
|
-
user_agent: "Mozilla/5.0 (
|
|
39
|
+
user_agent: "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " \
|
|
26
40
|
"AppleWebKit/537.36 (KHTML, like Gecko) " \
|
|
27
41
|
"Chrome/131.0.0.0 Safari/537.36",
|
|
28
|
-
platform: '"
|
|
42
|
+
platform: '"Windows"',
|
|
29
43
|
chrome_version: "131",
|
|
44
|
+
host_platform: :windows,
|
|
30
45
|
},
|
|
31
46
|
{
|
|
32
|
-
user_agent: "Mozilla/5.0 (
|
|
33
|
-
"AppleWebKit/537.36 (KHTML, like Gecko) " \
|
|
34
|
-
"Chrome/129.0.0.0 Safari/537.36",
|
|
35
|
-
platform: '"macOS"',
|
|
36
|
-
chrome_version: "129",
|
|
37
|
-
},
|
|
38
|
-
{
|
|
39
|
-
user_agent: "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " \
|
|
47
|
+
user_agent: "Mozilla/5.0 (X11; Linux x86_64) " \
|
|
40
48
|
"AppleWebKit/537.36 (KHTML, like Gecko) " \
|
|
41
49
|
"Chrome/131.0.0.0 Safari/537.36",
|
|
42
|
-
platform: '"
|
|
50
|
+
platform: '"Linux"',
|
|
43
51
|
chrome_version: "131",
|
|
52
|
+
host_platform: :linux,
|
|
44
53
|
},
|
|
45
54
|
].freeze
|
|
46
55
|
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
timeout: 30,
|
|
53
|
-
window_size: [1366, 768],
|
|
54
|
-
browser_options: {
|
|
55
|
-
"disable-blink-features" => "AutomationControlled",
|
|
56
|
-
},
|
|
57
|
-
**browser_opts,
|
|
58
|
-
)
|
|
59
|
-
|
|
60
|
-
browser.headers.set(random_headers)
|
|
61
|
-
browser.go_to(url)
|
|
62
|
-
browser.network.wait_for_idle(timeout: 15)
|
|
63
|
-
html = browser.body
|
|
64
|
-
|
|
65
|
-
if html.include?("403 ERROR") || html.include?("Request blocked")
|
|
66
|
-
warn "IEV: AWS WAF blocked request for #{url}"
|
|
67
|
-
return nil
|
|
68
|
-
end
|
|
56
|
+
DEFAULT_LANG = "en-US,en"
|
|
57
|
+
DEFAULT_BROWSER_OPTIONS = {
|
|
58
|
+
"disable-blink-features" => "AutomationControlled",
|
|
59
|
+
"lang" => DEFAULT_LANG,
|
|
60
|
+
}.freeze
|
|
69
61
|
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
62
|
+
# One-shot fetch. Each call spins up a fresh headless Chrome, fetches,
|
|
63
|
+
# and tears it down. Suitable for ad-hoc use; the WAF cookie does not
|
|
64
|
+
# survive between calls. Batch callers (Fetcher::Mirror) should use
|
|
65
|
+
# Session instead so the cookie set on the first successful challenge
|
|
66
|
+
# is reused across requests.
|
|
67
|
+
def self.fetch(url, **_browser_opts)
|
|
68
|
+
Session.new.fetch(url)
|
|
76
69
|
end
|
|
77
70
|
|
|
71
|
+
# Returns request headers that match what real Chrome sends on a
|
|
72
|
+
# fresh address-bar navigation. AWS WAF fingerprints inconsistencies
|
|
73
|
+
# between these headers and the browser's runtime state, so we:
|
|
74
|
+
# - omit Sec-Fetch-* (Chrome computes those itself from the
|
|
75
|
+
# navigation context; setting them via Ferrum's Network domain
|
|
76
|
+
# overrides the real values and is detectable), and
|
|
77
|
+
# - keep Sec-Ch-Ua-Platform aligned with the host OS, which is
|
|
78
|
+
# what Chrome reports via navigator.platform.
|
|
78
79
|
def self.random_headers
|
|
79
|
-
profile =
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
"\"Not_A Brand\";v=\"24\""
|
|
80
|
+
profile = profile_for_host
|
|
81
|
+
static_headers.merge(headers_from_profile(profile))
|
|
82
|
+
end
|
|
83
83
|
|
|
84
|
+
def self.static_headers
|
|
84
85
|
{
|
|
85
86
|
"Accept" => "text/html,application/xhtml+xml,application/xml;q=0.9," \
|
|
86
87
|
"image/avif,image/webp,image/apng,*/*;q=0.8," \
|
|
87
88
|
"application/signed-exchange;v=b3;q=0.7",
|
|
88
|
-
"Accept-Language" => "en-
|
|
89
|
+
"Accept-Language" => "en-US,en;q=0.9",
|
|
89
90
|
"Cache-Control" => "no-cache",
|
|
90
91
|
"Pragma" => "no-cache",
|
|
91
|
-
"Sec-Ch-Ua" => sec_ch_ua,
|
|
92
92
|
"Sec-Ch-Ua-Mobile" => "?0",
|
|
93
|
-
"Sec-Ch-Ua-Platform" => profile[:platform],
|
|
94
|
-
"Sec-Fetch-Dest" => "document",
|
|
95
|
-
"Sec-Fetch-Mode" => "navigate",
|
|
96
|
-
"Sec-Fetch-Site" => "cross-site",
|
|
97
|
-
"Sec-Fetch-User" => "?1",
|
|
98
93
|
"Upgrade-Insecure-Requests" => "1",
|
|
94
|
+
}
|
|
95
|
+
end
|
|
96
|
+
|
|
97
|
+
def self.headers_from_profile(profile)
|
|
98
|
+
{
|
|
99
|
+
"Sec-Ch-Ua" => sec_ch_ua_for(profile),
|
|
100
|
+
"Sec-Ch-Ua-Platform" => profile[:platform],
|
|
99
101
|
"User-Agent" => profile[:user_agent],
|
|
100
102
|
}
|
|
101
103
|
end
|
|
104
|
+
|
|
105
|
+
def self.profile_for_host
|
|
106
|
+
USER_AGENT_PROFILES.select do |profile|
|
|
107
|
+
profile[:host_platform] == host_platform
|
|
108
|
+
end.sample
|
|
109
|
+
end
|
|
110
|
+
|
|
111
|
+
def self.host_platform
|
|
112
|
+
case RUBY_PLATFORM
|
|
113
|
+
when /darwin/ then :mac
|
|
114
|
+
when /mswin|mingw|cygwin|bccwin|wince|emx/ then :windows
|
|
115
|
+
else :linux
|
|
116
|
+
end
|
|
117
|
+
end
|
|
118
|
+
|
|
119
|
+
def self.sec_ch_ua_for(profile)
|
|
120
|
+
"\"Google Chrome\";v=\"#{profile[:chrome_version]}\", " \
|
|
121
|
+
"\"Chromium\";v=\"#{profile[:chrome_version]}\", " \
|
|
122
|
+
"\"Not_A Brand\";v=\"24\""
|
|
123
|
+
end
|
|
124
|
+
|
|
125
|
+
# A long-lived headless Chrome session. Cookies persist across
|
|
126
|
+
# fetches, so once the AWS WAF challenge is cleared on the first
|
|
127
|
+
# request, subsequent requests reuse the token and succeed at
|
|
128
|
+
# near-100% rate. The Mirror creates one Session per run and
|
|
129
|
+
# shares it across all SequentialProbe iterations.
|
|
130
|
+
#
|
|
131
|
+
# Chrome leaks ~1MB per page load (mostly V8 heap that doesn't get
|
|
132
|
+
# GC'd between navigations). After ~1000 fetches the process is at
|
|
133
|
+
# ~1GB and the OOM risk climbs sharply. #restart quits the browser
|
|
134
|
+
# and starts a fresh one with a new cookie jar; the WAF challenge
|
|
135
|
+
# will need to be cleared again on the next fetch.
|
|
136
|
+
class Session
|
|
137
|
+
def initialize
|
|
138
|
+
@browser = Ferrum::Browser.new(
|
|
139
|
+
headless: "new",
|
|
140
|
+
timeout: 30,
|
|
141
|
+
window_size: [1366, 768],
|
|
142
|
+
browser_options: Browser::DEFAULT_BROWSER_OPTIONS,
|
|
143
|
+
)
|
|
144
|
+
@browser.headers.set(Browser.random_headers)
|
|
145
|
+
end
|
|
146
|
+
|
|
147
|
+
def fetch(url)
|
|
148
|
+
@browser.go_to(url)
|
|
149
|
+
@browser.network.wait_for_idle(timeout: 15)
|
|
150
|
+
reject_blocked(url, @browser.body)
|
|
151
|
+
rescue Ferrum::DeadBrowserError, Ferrum::NoSuchPageError,
|
|
152
|
+
Ferrum::NoSuchTargetError => e
|
|
153
|
+
# Chrome process or page/tab has crashed. Restart once, retry
|
|
154
|
+
# once. If the restart itself fails, surface as nil so the
|
|
155
|
+
# probe silently skips and the run continues.
|
|
156
|
+
if restart
|
|
157
|
+
retry
|
|
158
|
+
else
|
|
159
|
+
warn "IEV: Browser crashed, restart failed: #{e.message}"
|
|
160
|
+
nil
|
|
161
|
+
end
|
|
162
|
+
rescue Ferrum::BrowserError => e
|
|
163
|
+
if fetch_dead?(e.message) && restart
|
|
164
|
+
retry
|
|
165
|
+
else
|
|
166
|
+
warn "IEV: Browser error fetching #{url}: #{e.message}"
|
|
167
|
+
nil
|
|
168
|
+
end
|
|
169
|
+
rescue Ferrum::Error => e
|
|
170
|
+
warn "IEV: Browser error fetching #{url}: #{e.message}"
|
|
171
|
+
nil
|
|
172
|
+
end
|
|
173
|
+
|
|
174
|
+
def fetch_dead?(message)
|
|
175
|
+
message.include?("Browser is dead") ||
|
|
176
|
+
message.include?("given window is closed") ||
|
|
177
|
+
message.include?("Target closed")
|
|
178
|
+
end
|
|
179
|
+
|
|
180
|
+
def reject_blocked(url, html)
|
|
181
|
+
if html.include?("403 ERROR") || html.include?("Request blocked")
|
|
182
|
+
warn "IEV: AWS WAF blocked request for #{url}"
|
|
183
|
+
nil
|
|
184
|
+
else
|
|
185
|
+
html
|
|
186
|
+
end
|
|
187
|
+
end
|
|
188
|
+
|
|
189
|
+
# Quit the current browser and start a fresh one. The WAF cookie
|
|
190
|
+
# is lost, so the next fetch will go through the challenge cycle
|
|
191
|
+
# again. Call this after N fetches to bound Ferrum's memory growth,
|
|
192
|
+
# or rely on #fetch to call it automatically when Chrome dies.
|
|
193
|
+
# Returns true on success, false if the new browser fails to start.
|
|
194
|
+
def restart
|
|
195
|
+
quit
|
|
196
|
+
initialize
|
|
197
|
+
true
|
|
198
|
+
rescue StandardError => e
|
|
199
|
+
warn "IEV: Session restart failed: #{e.message}"
|
|
200
|
+
false
|
|
201
|
+
end
|
|
202
|
+
|
|
203
|
+
def quit
|
|
204
|
+
@browser&.quit
|
|
205
|
+
end
|
|
206
|
+
end
|
|
102
207
|
end
|
|
103
208
|
end
|
|
104
209
|
end
|
|
@@ -4,13 +4,14 @@ module Iev
|
|
|
4
4
|
class Scraper
|
|
5
5
|
# Parses an Electropedia HTML page into a concept data hash.
|
|
6
6
|
#
|
|
7
|
-
#
|
|
8
|
-
#
|
|
9
|
-
#
|
|
10
|
-
#
|
|
11
|
-
#
|
|
7
|
+
# All extracted text (terms, definitions, notes, examples) is run
|
|
8
|
+
# through Iev::Utilities' semantic enrichment pipeline, converting
|
|
9
|
+
# HTML to AsciiDoc notation: <i>x</i> → stem:[x], <b>x</b> → *x*,
|
|
10
|
+
# <a href=IEV...> → {{urn:...}}, MathML → AsciiMath. This ensures
|
|
11
|
+
# the parsed output matches the format used in termbase.yaml.
|
|
12
12
|
class PageParser
|
|
13
|
-
|
|
13
|
+
include Utilities
|
|
14
|
+
|
|
14
15
|
LANG_CODE_MAP = {
|
|
15
16
|
"en" => "eng",
|
|
16
17
|
"fr" => "fra",
|
|
@@ -54,28 +55,48 @@ module Iev
|
|
|
54
55
|
private
|
|
55
56
|
|
|
56
57
|
def find_iev_ref
|
|
57
|
-
# Find the IEV reference cell to confirm the page is valid
|
|
58
58
|
@doc.at_css("b:contains('#{@code}')") ||
|
|
59
59
|
@doc.at_xpath("//td/b[contains(text(), '#{@code}')]")
|
|
60
60
|
end
|
|
61
61
|
|
|
62
|
+
def term_domain
|
|
63
|
+
@code.rpartition("-").first
|
|
64
|
+
end
|
|
65
|
+
|
|
62
66
|
def localized_concepts
|
|
63
67
|
result = {}
|
|
64
|
-
lang_sections.each do |lang, term_row, def_row|
|
|
68
|
+
lang_sections.each do |lang, area, term_row, def_row|
|
|
65
69
|
term = extract_term(term_row)
|
|
66
70
|
next unless term
|
|
67
71
|
|
|
68
|
-
entry = { "
|
|
72
|
+
entry = result[lang] || { "terms" => [] }
|
|
73
|
+
term_entry = { "designation" => term }
|
|
74
|
+
term_entry["geographical_area"] = area if area
|
|
75
|
+
entry["terms"] << term_entry
|
|
76
|
+
|
|
69
77
|
definition = extract_definition(def_row)
|
|
70
|
-
entry["definition"] = definition if definition
|
|
78
|
+
entry["definition"] = definition if definition && !entry.key?("definition")
|
|
71
79
|
|
|
72
80
|
result[lang] = entry
|
|
73
81
|
end
|
|
82
|
+
|
|
83
|
+
# Convert terms array to single "term" field for backward compat
|
|
84
|
+
# when there's only one term and no area
|
|
85
|
+
result.each_value do |entry|
|
|
86
|
+
terms = entry.delete("terms")
|
|
87
|
+
if terms.size == 1 && !terms.first["geographical_area"]
|
|
88
|
+
entry["term"] = terms.first["designation"]
|
|
89
|
+
else
|
|
90
|
+
entry["term"] = terms.map { |t| t["designation"] }.join("\n")
|
|
91
|
+
entry["term_areas"] = terms.each_with_object({}) do |t, h|
|
|
92
|
+
area = t["geographical_area"]
|
|
93
|
+
h[t["designation"]] = area if area
|
|
94
|
+
end
|
|
95
|
+
end
|
|
96
|
+
end
|
|
74
97
|
result
|
|
75
98
|
end
|
|
76
99
|
|
|
77
|
-
# Finds all language sections in the table.
|
|
78
|
-
# Returns array of [lang_code, term_row, definition_row] tuples.
|
|
79
100
|
def lang_sections
|
|
80
101
|
sections = []
|
|
81
102
|
rows = content_rows
|
|
@@ -84,17 +105,30 @@ module Iev
|
|
|
84
105
|
lang = extract_lang(row)
|
|
85
106
|
next unless lang
|
|
86
107
|
|
|
87
|
-
|
|
108
|
+
area = extract_area(row)
|
|
88
109
|
def_row = find_definition_row(rows, idx + 1)
|
|
89
|
-
sections << [lang, row, def_row]
|
|
110
|
+
sections << [lang, area, row, def_row]
|
|
90
111
|
end
|
|
91
112
|
|
|
92
113
|
sections
|
|
93
114
|
end
|
|
94
115
|
|
|
116
|
+
def extract_area(row)
|
|
117
|
+
tds = row.css("td")
|
|
118
|
+
return nil if tds.length < 2
|
|
119
|
+
|
|
120
|
+
area_cell = tds[1]
|
|
121
|
+
font = area_cell.at_css("font[color='#800080']")
|
|
122
|
+
return nil unless font
|
|
123
|
+
|
|
124
|
+
text = font.text.strip.downcase
|
|
125
|
+
return nil if text.empty?
|
|
126
|
+
return nil if text.match?(/\A(nl|be|nb|nn|br|pt|fr|de|us|gb|ch|at|ca|au|nz)\z/i) == false
|
|
127
|
+
|
|
128
|
+
text
|
|
129
|
+
end
|
|
130
|
+
|
|
95
131
|
def content_rows
|
|
96
|
-
# Find the main content table (the one with language data)
|
|
97
|
-
# It's the largest table with IEV data
|
|
98
132
|
tables = @doc.css("table")
|
|
99
133
|
content_table = tables.max_by { |t| t.css("tr").length }
|
|
100
134
|
content_table ? content_table.css("tr").to_a : []
|
|
@@ -109,15 +143,30 @@ module Iev
|
|
|
109
143
|
end
|
|
110
144
|
|
|
111
145
|
def extract_term(row)
|
|
112
|
-
# Term is in the third <td> — may be in a <b> tag (en, fr) or plain text
|
|
113
146
|
tds = row.css("td")
|
|
114
147
|
return nil if tds.length < 3
|
|
115
148
|
|
|
116
149
|
content_td = tds[2]
|
|
150
|
+
return nil if area_marker_cell?(content_td)
|
|
151
|
+
|
|
117
152
|
bold = content_td.at_css("b")
|
|
118
153
|
|
|
119
|
-
|
|
120
|
-
|
|
154
|
+
html = if bold
|
|
155
|
+
bold.inner_html.strip
|
|
156
|
+
else
|
|
157
|
+
content_td.inner_html.strip
|
|
158
|
+
end
|
|
159
|
+
return nil if html.empty?
|
|
160
|
+
|
|
161
|
+
enrich(html)
|
|
162
|
+
end
|
|
163
|
+
|
|
164
|
+
# True when the cell contains only a geographical-area marker font
|
|
165
|
+
# tag (e.g. <font color="#800080">nb</font>), not a real term.
|
|
166
|
+
def area_marker_cell?(cell)
|
|
167
|
+
font = cell.at_css("font[color='#800080']")
|
|
168
|
+
return false unless font
|
|
169
|
+
cell.at_css("b").nil? && cell.text.strip == font.text.strip && font.text.strip.length <= 3
|
|
121
170
|
end
|
|
122
171
|
|
|
123
172
|
def extract_definition(row)
|
|
@@ -127,15 +176,59 @@ module Iev
|
|
|
127
176
|
return nil if tds.length < 3
|
|
128
177
|
|
|
129
178
|
content_td = tds[2]
|
|
130
|
-
# The definition is the text content, which may include MathML
|
|
131
179
|
html = content_td.inner_html.strip
|
|
132
180
|
return nil if html.empty? || html.match?(/\A<img.*ecblank/)
|
|
133
181
|
|
|
134
|
-
html
|
|
182
|
+
enrich(html)
|
|
183
|
+
end
|
|
184
|
+
|
|
185
|
+
def enrich(html_text)
|
|
186
|
+
text = html_text.to_s
|
|
187
|
+
.gsub(/<i>/, "<i>")
|
|
188
|
+
.gsub(/<\/i>/, "</i>")
|
|
189
|
+
mathml_blocks, text = extract_mathml_blocks(text)
|
|
190
|
+
result = parse_anchor_tag(text, term_domain)
|
|
191
|
+
result = restore_and_convert_mathml(result, mathml_blocks)
|
|
192
|
+
result = replace_newlines(result)
|
|
193
|
+
result = convert_literal_italic(result)
|
|
194
|
+
result.strip
|
|
195
|
+
end
|
|
196
|
+
|
|
197
|
+
MATHML_PLACEHOLDER = "@@MATHML_%d@@"
|
|
198
|
+
|
|
199
|
+
def extract_mathml_blocks(text)
|
|
200
|
+
blocks = []
|
|
201
|
+
result = text.gsub(/<math[^>]*>.*?<\/math>/m) do |match|
|
|
202
|
+
blocks << match
|
|
203
|
+
format(MATHML_PLACEHOLDER, blocks.size - 1)
|
|
204
|
+
end
|
|
205
|
+
[blocks, result]
|
|
206
|
+
end
|
|
207
|
+
|
|
208
|
+
# Restore MathML blocks and convert each to AsciiMath individually.
|
|
209
|
+
# If the converter returns empty (unsupported expression like
|
|
210
|
+
# <mtable>), keep the original MathML so the formula is not lost.
|
|
211
|
+
def restore_and_convert_mathml(text, blocks)
|
|
212
|
+
return text if blocks.empty?
|
|
213
|
+
|
|
214
|
+
blocks.each_with_index do |mathml, i|
|
|
215
|
+
converted = Iev::Converter.mathml_to_asciimath(mathml).strip
|
|
216
|
+
replacement = converted.empty? ? mathml : converted
|
|
217
|
+
text = text.sub(format(MATHML_PLACEHOLDER, i), replacement)
|
|
218
|
+
end
|
|
219
|
+
text
|
|
220
|
+
end
|
|
221
|
+
|
|
222
|
+
# After Nokogiri processing, <i>entity-encoded tags appear as
|
|
223
|
+
# literal <i>...</i> text. Convert them to stem:[...] just like
|
|
224
|
+
# real <i> elements are converted by convert_italic.
|
|
225
|
+
def convert_literal_italic(text)
|
|
226
|
+
text.gsub(/<i>([^<]*)<\/i>/) do
|
|
227
|
+
inner = Regexp.last_match(1)
|
|
228
|
+
convert_italic(inner)
|
|
229
|
+
end
|
|
135
230
|
end
|
|
136
231
|
|
|
137
|
-
# Find the definition row following a language row.
|
|
138
|
-
# Skip separator rows (empty, <hr>, or spacer images).
|
|
139
232
|
def find_definition_row(rows, start_idx)
|
|
140
233
|
return nil if start_idx >= rows.length
|
|
141
234
|
|
|
@@ -149,7 +242,6 @@ module Iev
|
|
|
149
242
|
content = tds[2].inner_html.strip
|
|
150
243
|
return nil if content.empty?
|
|
151
244
|
|
|
152
|
-
# Skip rows that are only spacer images (unless they have <b> content)
|
|
153
245
|
if content.match?(/\A<img.*ecblank/) && !content.include?("<b>")
|
|
154
246
|
return nil
|
|
155
247
|
end
|
data/lib/iev/source_parser.rb
CHANGED
|
@@ -406,7 +406,18 @@ module Iev
|
|
|
406
406
|
return nil unless item
|
|
407
407
|
|
|
408
408
|
item.source("src")
|
|
409
|
-
|
|
409
|
+
# Free-text refs like "ITU-R Rec. 431 MOD" are not parseable pubids.
|
|
410
|
+
# relaton 3 (pubid 2) raises Pubid::Errors::ParseError (a
|
|
411
|
+
# Parslet::ParseFailed) eagerly when building the search, whereas
|
|
412
|
+
# relaton 2 swallowed it. Link resolution is best-effort: warn and
|
|
413
|
+
# skip instead of failing the whole source parse.
|
|
414
|
+
# relaton 3.0.0.pre.alpha.6 (#205 strict routing) also raises
|
|
415
|
+
# UnknownReferenceError for spellings no flavor claims (the French
|
|
416
|
+
# "CEI", agency refs like "IAEA 4") where alpha.5's permissive
|
|
417
|
+
# parse answered nil. Same best-effort contract: warn and skip.
|
|
418
|
+
rescue Relaton::RequestError, Relaton::UnknownReferenceError,
|
|
419
|
+
Socket::ResolutionError, SocketError,
|
|
420
|
+
Parslet::ParseFailed => e
|
|
410
421
|
warn e.message
|
|
411
422
|
nil
|
|
412
423
|
end
|
|
@@ -128,10 +128,12 @@ module Iev
|
|
|
128
128
|
|
|
129
129
|
# --- RelatedConcept factory methods ---
|
|
130
130
|
|
|
131
|
+
# RelatedConcept#content is a lang-keyed hash since glossarist 2.13
|
|
132
|
+
# (L10N invariant: all relationship free-text is { "eng" => ... }).
|
|
131
133
|
def broader_relation(target_uri)
|
|
132
134
|
Glossarist::RelatedConcept.new(
|
|
133
135
|
type: "broader",
|
|
134
|
-
content: target_uri,
|
|
136
|
+
content: { "eng" => target_uri },
|
|
135
137
|
ref: Glossarist::ConceptRef.new(source: "IEV", id: target_uri),
|
|
136
138
|
)
|
|
137
139
|
end
|
|
@@ -139,7 +141,7 @@ module Iev
|
|
|
139
141
|
def narrower_relation(target_uri)
|
|
140
142
|
Glossarist::RelatedConcept.new(
|
|
141
143
|
type: "narrower",
|
|
142
|
-
content: target_uri,
|
|
144
|
+
content: { "eng" => target_uri },
|
|
143
145
|
ref: Glossarist::ConceptRef.new(source: "IEV", id: target_uri),
|
|
144
146
|
)
|
|
145
147
|
end
|
data/lib/iev/utilities.rb
CHANGED
|
@@ -92,7 +92,9 @@ module Iev
|
|
|
92
92
|
when "img"
|
|
93
93
|
src = node["src"] || node.attributes.keys.first.to_s
|
|
94
94
|
"#{IMAGE_PATH_PREFIX}/#{term_domain}/#{src}[]"
|
|
95
|
-
when "p"
|
|
95
|
+
when "p"
|
|
96
|
+
"#{inner}\n"
|
|
97
|
+
when "div", "span"
|
|
96
98
|
inner
|
|
97
99
|
when "i"
|
|
98
100
|
convert_italic(inner)
|
|
@@ -101,9 +103,9 @@ module Iev
|
|
|
101
103
|
when "sup"
|
|
102
104
|
inner.empty? ? "" : "^#{inner}^"
|
|
103
105
|
when "ol"
|
|
104
|
-
convert_list(node, ". ")
|
|
106
|
+
convert_list(node, ". ", term_domain)
|
|
105
107
|
when "ul"
|
|
106
|
-
convert_list(node, "* ")
|
|
108
|
+
convert_list(node, "* ", term_domain)
|
|
107
109
|
when "li"
|
|
108
110
|
inner
|
|
109
111
|
when "font"
|
|
@@ -124,8 +126,8 @@ module Iev
|
|
|
124
126
|
end
|
|
125
127
|
end
|
|
126
128
|
|
|
127
|
-
def convert_list(node, prefix)
|
|
128
|
-
node.css("li").map { |li| "#{prefix}#{li.
|
|
129
|
+
def convert_list(node, prefix, term_domain)
|
|
130
|
+
node.css("li").map { |li| "#{prefix}#{nodes_to_adoc(li.children, term_domain)}" }.join("\n")
|
|
129
131
|
end
|
|
130
132
|
|
|
131
133
|
def convert_font(node, inner)
|
data/lib/iev/version.rb
CHANGED
data/lib/iev.rb
CHANGED
|
@@ -32,7 +32,9 @@ module Iev
|
|
|
32
32
|
autoload :FigureBuilder, "iev/figure_builder"
|
|
33
33
|
autoload :IevCode, "iev/iev_code"
|
|
34
34
|
autoload :Iso639Code, "iev/iso_639_code"
|
|
35
|
+
autoload :MultiDocYaml, "iev/multi_doc_yaml"
|
|
35
36
|
autoload :Profiler, "iev/profiler"
|
|
37
|
+
autoload :Reconciler, "iev/reconciler"
|
|
36
38
|
autoload :RelatonDb, "iev/relaton_db"
|
|
37
39
|
autoload :Scraper, "iev/scraper"
|
|
38
40
|
autoload :Section, "iev/section"
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
#!/usr/bin/env ruby
|
|
2
|
+
# Identify cached HTML pages that PageParser cannot turn into a concept.
|
|
3
|
+
# Run: bundle exec ruby -Ilib scripts/find_unparseable.rb
|
|
4
|
+
require "iev"
|
|
5
|
+
|
|
6
|
+
store = Iev::Fetcher::PageStore.new
|
|
7
|
+
ok = 0
|
|
8
|
+
failed = []
|
|
9
|
+
store.each_concept(scope: Iev::Fetcher::Scope.all).each do |code, html|
|
|
10
|
+
doc = Nokogiri::HTML(html)
|
|
11
|
+
if Iev::Scraper::PageParser.new(doc, code).parse
|
|
12
|
+
ok += 1
|
|
13
|
+
else
|
|
14
|
+
failed << code
|
|
15
|
+
end
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
puts "Parsed OK: #{ok}"
|
|
19
|
+
puts "Failed: #{failed.size}"
|
|
20
|
+
puts
|
|
21
|
+
puts "First 20 failed codes:"
|
|
22
|
+
failed.take(20).each { |c| puts " #{c}" }
|