iev 0.4.6 → 0.4.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. checksums.yaml +4 -4
  2. data/.claude/scheduled_tasks.lock +1 -0
  3. data/.rubocop_todo.yml +5 -47
  4. data/CLAUDE.md +11 -1
  5. data/Gemfile +4 -2
  6. data/TODO.reconcile/01-design.md +57 -0
  7. data/TODO.reconcile/02-change-models.md +44 -0
  8. data/TODO.reconcile/02-termbase-loader.md +48 -0
  9. data/TODO.reconcile/03-live-loader.md +28 -0
  10. data/TODO.reconcile/03-status-mapper.md +24 -0
  11. data/TODO.reconcile/04-concept-merger.md +45 -0
  12. data/TODO.reconcile/04-termbase-loader.md +28 -0
  13. data/TODO.reconcile/05-content-diff.md +45 -0
  14. data/TODO.reconcile/05-live-loader.md +20 -0
  15. data/TODO.reconcile/06-content-differ.md +32 -0
  16. data/TODO.reconcile/06-v3-serializer.md +42 -0
  17. data/TODO.reconcile/07-cli-and-run.md +27 -0
  18. data/TODO.reconcile/07-concept-merger.md +32 -0
  19. data/TODO.reconcile/08-report.md +53 -0
  20. data/TODO.reconcile/09-pipeline.md +23 -0
  21. data/TODO.reconcile/10-script-and-run.md +15 -0
  22. data/data/locales/glossarist_enums.yml +705 -0
  23. data/iev.gemspec +4 -2
  24. data/lib/iev/bibliography_builder.rb +87 -0
  25. data/lib/iev/cli/command.rb +29 -0
  26. data/lib/iev/cli/command_helper.rb +1 -6
  27. data/lib/iev/config.rb +10 -1
  28. data/lib/iev/exporter.rb +37 -4
  29. data/lib/iev/figure_builder.rb +186 -0
  30. data/lib/iev/multi_doc_yaml.rb +60 -0
  31. data/lib/iev/reconciler/change.rb +29 -0
  32. data/lib/iev/reconciler/change_set.rb +48 -0
  33. data/lib/iev/reconciler/concept_merger.rb +130 -0
  34. data/lib/iev/reconciler/content_differ.rb +157 -0
  35. data/lib/iev/reconciler/live_loader.rb +163 -0
  36. data/lib/iev/reconciler/pipeline.rb +104 -0
  37. data/lib/iev/reconciler/reconciled_concept.rb +14 -0
  38. data/lib/iev/reconciler/report.rb +118 -0
  39. data/lib/iev/reconciler/status_mapper.rb +29 -0
  40. data/lib/iev/reconciler/term_marker_parser.rb +196 -0
  41. data/lib/iev/reconciler/termbase_loader.rb +141 -0
  42. data/lib/iev/reconciler.rb +17 -0
  43. data/lib/iev/scraper/browser.rb +155 -50
  44. data/lib/iev/scraper/page_parser.rb +116 -24
  45. data/lib/iev/source_parser.rb +12 -1
  46. data/lib/iev/subject_area_concepts.rb +4 -2
  47. data/lib/iev/utilities.rb +7 -5
  48. data/lib/iev/version.rb +1 -1
  49. data/lib/iev.rb +4 -0
  50. data/scripts/find_unparseable.rb +22 -0
  51. metadata +54 -6
@@ -0,0 +1,196 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Iev
4
+ module Reconciler
5
+ # Extracts grammatical markers, context qualifiers, and part-of-speech
6
+ # tags that are inline in Electropedia's live HTML term text, matching
7
+ # how the termbase stores them in separate fields.
8
+ #
9
+ # Two-pass approach:
10
+ # Pass 1: extract context qualifiers and IEV cross-references from
11
+ # angle brackets (<text>, <相关条目:IEV xxx>)
12
+ # Pass 2: extract gender/number/POS/prefix markers from remaining text
13
+ #
14
+ # Marker formats handled:
15
+ # Gender: , f , m , n (also multiple: , f, n)
16
+ # , ж , м , с (Serbian Cyrillic)
17
+ # , m/f (slash = both apply)
18
+ # Number: , pl , sg (also Serbian: јд, мн)
19
+ # , f pl (space-separated gender+number)
20
+ # Usage info: <text> → usage_info field
21
+ # Part of speech: , 名詞 , 명사 , noun , verb , adj. , agg , etc.
22
+ # Prefix: , Präfix , 접두사 , 接頭語 , etc. → prefix flag
23
+ class TermMarkerParser
24
+ GENDER_MAP = {
25
+ "f" => "feminine", "m" => "masculine", "n" => "neuter",
26
+ "ж" => "feminine", "м" => "masculine", "с" => "neuter",
27
+ }.freeze
28
+
29
+ NUMBER_MAP = {
30
+ "јд" => "singular", "мн" => "plural",
31
+ "sg" => "singular", "pl" => "plural",
32
+ }.freeze
33
+
34
+ PART_OF_SPEECH_MAP = {
35
+ "noun" => "noun",
36
+ "명사" => "noun", "名詞" => "noun", "اسم" => "noun",
37
+ "именица" => "noun",
38
+ "verb" => "verb", "verbo" => "verb",
39
+ "동사" => "verb", "動詞" => "verb", "فعل" => "verb",
40
+ "глагол" => "verb",
41
+ "adj" => "adj", "adj." => "adj", "adjective" => "adj",
42
+ "agg" => "adj", "agg." => "adj",
43
+ "형용사" => "adj", "形容詞" => "adj", "صفة" => "adj",
44
+ "adjektiv" => "adj", "придев" => "adj",
45
+ "adv" => "adv", "adv." => "adv", "adverb" => "adv",
46
+ "부사" => "adv", "副詞" => "adv",
47
+ "прилог" => "adv",
48
+ }.freeze
49
+
50
+ PREFIX_KEYWORDS = %w[
51
+ Präfix prefix préfixe 접두사 接頭語 接尾語
52
+ ].freeze
53
+
54
+ DOMAIN_RE = /<([^>]+)>/
55
+ USAGE_INFO_RE = /\(([^)]+)\)\s*(?=[,]?\s*(?:[fmnжмс]\b|[fmn]\/[fmn]|sg|pl|јд|мн|\z))/i
56
+ IEV_XREF_RE = /<[^>]*IEV\s*(\d{3}-\d{2}-\d{2,3})[^>]*>/i
57
+
58
+ SERBIAN_RE = /,\s*([жмс])\s+(јд|мн)\s*\z/
59
+ WESTERN_GENDER_NUMBER_RE = /,\s*([fmn])\s+(sg|pl)\.?\s*\z/i
60
+ WESTERN_NUMBER_RE = /,\s*(sg|pl)\.?\s*\z/i
61
+ SLASH_GENDER_RE = /,\s*([fmn])\/([fmn])\s*\z/i
62
+ WESTERN_RE = /[,]?\s+([fmn])\s*\z/
63
+ SPACE_GENDER_RE = /\s+([fmn])\s*\z/
64
+
65
+ POS_RE = /,\s*(#{PART_OF_SPEECH_MAP.keys.map { |k| Regexp.escape(k) }.join("|")})\s*\z/i
66
+ PREFIX_RE = /,\s*(#{PREFIX_KEYWORDS.map { |k| Regexp.escape(k) }.join("|")})\s*\z/
67
+
68
+ TRAILING_PUNCT_RE = /[,;\s]+$/
69
+
70
+ Result = Struct.new(:designation, :genders, :numbers, :domain,
71
+ :usage_info, :part_of_speech, :related_refs,
72
+ :is_prefix, keyword_init: true)
73
+
74
+ class << self
75
+ def parse(term)
76
+ return Result.new(designation: nil) unless term
77
+
78
+ text = decode_entities(term).strip.gsub(/\s+/, " ").gsub(/ /, " ")
79
+ related_refs, text = extract_iev_xrefs(text)
80
+ domain, text = extract_domain(text)
81
+ usage_info, text = extract_usage_info(text)
82
+ designation, genders, numbers, pos, is_prefix = extract_markers(text)
83
+
84
+ Result.new(
85
+ designation: designation,
86
+ genders: genders,
87
+ numbers: numbers,
88
+ domain: domain,
89
+ usage_info: usage_info,
90
+ part_of_speech: pos,
91
+ related_refs: related_refs,
92
+ is_prefix: is_prefix,
93
+ )
94
+ end
95
+
96
+ def parse_multiple(term_text)
97
+ return [] unless term_text
98
+
99
+ term_text
100
+ .split(/\n+/)
101
+ .map(&:strip)
102
+ .reject(&:empty?)
103
+ .map { |part| parse(part) }
104
+ end
105
+
106
+ private
107
+
108
+ def decode_entities(str)
109
+ str
110
+ .gsub(/&lt;/, "<")
111
+ .gsub(/&gt;/, ">")
112
+ .gsub(/&amp;/, "&")
113
+ .gsub(/&quot;/, '"')
114
+ .gsub(/&#39;/, "'")
115
+ end
116
+
117
+ # Extract domain qualifier from angle brackets: <text> -> domain
118
+ def extract_domain(text)
119
+ match = text.match(DOMAIN_RE)
120
+ return [nil, text] unless match
121
+
122
+ domain = match[1].strip
123
+ remaining = text.sub(match[0], "").gsub(TRAILING_PUNCT_RE, "").strip
124
+ remaining = remaining.sub(/^[,;\s]+/, "").strip
125
+ [domain, remaining]
126
+ end
127
+
128
+ def extract_iev_xrefs(text)
129
+ refs = []
130
+ remaining = text
131
+ while (match = remaining.match(IEV_XREF_RE))
132
+ refs << match[1]
133
+ remaining = remaining.sub(match[0], "").gsub(TRAILING_PUNCT_RE, "").strip
134
+ remaining = remaining.sub(/^[,;\s]+/, "").strip
135
+ end
136
+ [refs.empty? ? nil : refs, remaining]
137
+ end
138
+
139
+ # Extract usage_info from parentheses that precede trailing markers.
140
+ # E.g. "Orientierung (einer Kurve) f" -> usage_info="einer Kurve"
141
+ def extract_usage_info(text)
142
+ match = text.match(USAGE_INFO_RE)
143
+ return [nil, text] unless match
144
+
145
+ usage = match[1].strip
146
+ remaining = text.sub(match[0], "").gsub(/\s+/, " ").strip
147
+ [usage, remaining]
148
+ end
149
+
150
+ def extract_markers(text)
151
+ designation = text
152
+ genders = []
153
+ numbers = []
154
+ pos = nil
155
+ is_prefix = false
156
+
157
+ loop do
158
+ break if designation.nil? || designation.empty?
159
+
160
+ if (m = designation.match(SERBIAN_RE))
161
+ designation = designation[0, m.begin(0)].strip
162
+ genders << GENDER_MAP[m[1]] if GENDER_MAP[m[1]]
163
+ numbers << NUMBER_MAP[m[2]] if NUMBER_MAP[m[2]]
164
+ elsif (m = designation.match(SLASH_GENDER_RE))
165
+ designation = designation[0, m.begin(0)].strip
166
+ genders << GENDER_MAP[m[1]] if GENDER_MAP[m[1]]
167
+ genders << GENDER_MAP[m[2]] if GENDER_MAP[m[2]]
168
+ elsif (m = designation.match(WESTERN_GENDER_NUMBER_RE))
169
+ designation = designation[0, m.begin(0)].strip
170
+ genders << GENDER_MAP[m[1]] if GENDER_MAP[m[1]]
171
+ num_key = m[2].downcase.sub(/\.$/, "")
172
+ numbers << NUMBER_MAP[num_key] if NUMBER_MAP[num_key]
173
+ elsif (m = designation.match(WESTERN_NUMBER_RE))
174
+ designation = designation[0, m.begin(0)].strip
175
+ num_key = m[1].downcase.sub(/\.$/, "")
176
+ numbers << NUMBER_MAP[num_key] if NUMBER_MAP[num_key]
177
+ elsif (m = designation.match(POS_RE))
178
+ designation = designation[0, m.begin(0)].strip
179
+ pos = PART_OF_SPEECH_MAP[m[1].downcase]
180
+ elsif (m = designation.match(PREFIX_RE))
181
+ designation = designation[0, m.begin(0)].strip
182
+ is_prefix = true
183
+ elsif (m = designation.match(WESTERN_RE))
184
+ designation = designation[0, m.begin(0)].strip
185
+ genders << GENDER_MAP[m[1]] if GENDER_MAP[m[1]]
186
+ else
187
+ break
188
+ end
189
+ end
190
+
191
+ [designation, genders.uniq, numbers.uniq, pos, is_prefix]
192
+ end
193
+ end
194
+ end
195
+ end
196
+ end
@@ -0,0 +1,141 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "yaml"
4
+
5
+ module Iev
6
+ module Reconciler
7
+ # Loads termbase.yaml into a raw hash, then builds
8
+ # Glossarist::ManagedConcept objects on demand per code.
9
+ # This avoids the cost of constructing 22k model objects upfront.
10
+ class TermbaseLoader
11
+ # @param path [String, Pathname] path to termbase.yaml
12
+ def initialize(path)
13
+ @path = path.to_s
14
+ @data = nil
15
+ end
16
+
17
+ # @return [Hash<String, Hash>] raw termbase data keyed by code
18
+ def raw
19
+ @data ||= YAML.load_file(@path)
20
+ end
21
+
22
+ # @return [Array<String>] all codes in the termbase
23
+ def codes
24
+ raw.keys.map(&:to_s)
25
+ end
26
+
27
+ # Build a single ManagedConcept for the given code.
28
+ # @param code [String]
29
+ # @return [Glossarist::ManagedConcept, nil]
30
+ def get(code)
31
+ entry = raw[code] || raw[code.to_sym]
32
+ return nil unless entry.is_a?(Hash)
33
+
34
+ build_concept(code.to_s, entry)
35
+ end
36
+
37
+ private
38
+
39
+ def build_concept(code, entry)
40
+ concept = Glossarist::ManagedConcept.of_yaml(
41
+ "id" => code, "data" => { "id" => code },
42
+ )
43
+ concept.status = concept_status(entry)
44
+ concept.dates = concept_dates(entry)
45
+ concept.schema_version = "3"
46
+
47
+ build_localized(code, entry).each do |lc|
48
+ concept.add_l10n(lc)
49
+ end
50
+
51
+ concept
52
+ end
53
+
54
+ def concept_status(entry)
55
+ lang_data = first_language(entry)
56
+ StatusMapper.call(lang_data&.dig("entry_status"))
57
+ end
58
+
59
+ def concept_dates(entry)
60
+ lang_data = first_language(entry)
61
+ return [] unless lang_data
62
+
63
+ dates = []
64
+ if lang_data["date_accepted"]
65
+ dates << Glossarist::ConceptDate.new(
66
+ type: "accepted",
67
+ date: lang_data["date_accepted"],
68
+ )
69
+ end
70
+ if lang_data["date_amended"] && lang_data["date_amended"] != lang_data["date_accepted"]
71
+ dates << Glossarist::ConceptDate.new(
72
+ type: "amended",
73
+ date: lang_data["date_amended"],
74
+ )
75
+ end
76
+ dates
77
+ end
78
+
79
+ def first_language(entry)
80
+ entry.values.find { |v| v.is_a?(Hash) && v["terms"] }
81
+ end
82
+
83
+ def build_localized(code, entry)
84
+ result = []
85
+ entry.each do |lang, data|
86
+ next unless data.is_a?(Hash) && data["terms"]
87
+ lc = build_localized_concept(code, lang, data)
88
+ result << lc if lc
89
+ end
90
+ result
91
+ end
92
+
93
+ def build_localized_concept(code, lang, data)
94
+ cdata = Glossarist::ConceptData.new
95
+ cdata.id = code
96
+ cdata.language_code = lang
97
+ cdata.entry_status = StatusMapper.call(data["entry_status"])
98
+
99
+ cdata.terms = (data["terms"] || []).map do |t|
100
+ build_term_expression(t)
101
+ end
102
+
103
+ if data["definition"] && !data["definition"].empty?
104
+ cdata.definition = [Glossarist::DetailedDefinition.new(content: data["definition"])]
105
+ end
106
+
107
+ cdata.notes = Array(data["notes"]).map { |n| Glossarist::DetailedDefinition.new(content: n) }
108
+ cdata.examples = Array(data["examples"]).map { |e| Glossarist::DetailedDefinition.new(content: e) }
109
+
110
+ lc = Glossarist::LocalizedConcept.new
111
+ lc.id = code
112
+ lc.data = cdata
113
+ lc
114
+ end
115
+
116
+ def build_term_expression(term_data)
117
+ designation = term_data["designation"].to_s
118
+ parsed = TermMarkerParser.parse(designation)
119
+
120
+ expr = Glossarist::Designation::Expression.new(
121
+ designation: parsed.designation,
122
+ normative_status: term_data["normative_status"] || "preferred",
123
+ type: term_data["type"] || "expression",
124
+ )
125
+ expr.geographical_area = term_data["geographical_area"] if term_data["geographical_area"]
126
+ expr.field_of_application = term_data["usage_info"] || parsed.domain
127
+ expr.usage_info = parsed.usage_info if parsed.usage_info
128
+ expr.prefix = parsed.is_prefix if parsed.is_prefix
129
+
130
+ grammar = Glossarist::Designation::GrammarInfo.new
131
+ merged_genders = (Array(term_data["gender"]) + parsed.genders).flatten.compact.uniq
132
+ merged_numbers = (Array(term_data["plurality"]) + parsed.numbers).flatten.compact.uniq
133
+ grammar.gender = merged_genders if merged_genders.any?
134
+ grammar.number = merged_numbers if merged_numbers.any?
135
+ grammar.part_of_speech = parsed.part_of_speech if parsed.part_of_speech
136
+ expr.grammar_info = [grammar] if grammar.gender&.any? || grammar.number&.any? || grammar.part_of_speech
137
+ expr
138
+ end
139
+ end
140
+ end
141
+ end
@@ -0,0 +1,17 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Iev
4
+ module Reconciler
5
+ autoload :Change, "iev/reconciler/change"
6
+ autoload :ChangeSet, "iev/reconciler/change_set"
7
+ autoload :ConceptMerger, "iev/reconciler/concept_merger"
8
+ autoload :ContentDiffer, "iev/reconciler/content_differ"
9
+ autoload :LiveLoader, "iev/reconciler/live_loader"
10
+ autoload :Pipeline, "iev/reconciler/pipeline"
11
+ autoload :ReconciledConcept, "iev/reconciler/reconciled_concept"
12
+ autoload :Report, "iev/reconciler/report"
13
+ autoload :StatusMapper, "iev/reconciler/status_mapper"
14
+ autoload :TermMarkerParser, "iev/reconciler/term_marker_parser"
15
+ autoload :TermbaseLoader, "iev/reconciler/termbase_loader"
16
+ end
17
+ end
@@ -6,6 +6,10 @@ module Iev
6
6
  class Scraper
7
7
  # Shared headless browser utilities for fetching pages behind AWS WAF.
8
8
  module Browser
9
+ # Each profile is tagged with the host platform it can run on so
10
+ # that navigator.platform (set by Chrome from the real OS) agrees
11
+ # with the User-Agent and Sec-Ch-Ua-Platform headers. AWS WAF
12
+ # fingerprints this mismatch and refuses to clear its challenge.
9
13
  USER_AGENT_PROFILES = [
10
14
  {
11
15
  user_agent: "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) " \
@@ -13,6 +17,15 @@ module Iev
13
17
  "Chrome/131.0.0.0 Safari/537.36",
14
18
  platform: '"macOS"',
15
19
  chrome_version: "131",
20
+ host_platform: :mac,
21
+ },
22
+ {
23
+ user_agent: "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) " \
24
+ "AppleWebKit/537.36 (KHTML, like Gecko) " \
25
+ "Chrome/129.0.0.0 Safari/537.36",
26
+ platform: '"macOS"',
27
+ chrome_version: "129",
28
+ host_platform: :mac,
16
29
  },
17
30
  {
18
31
  user_agent: "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " \
@@ -20,85 +33,177 @@ module Iev
20
33
  "Chrome/130.0.0.0 Safari/537.36",
21
34
  platform: '"Windows"',
22
35
  chrome_version: "130",
36
+ host_platform: :windows,
23
37
  },
24
38
  {
25
- user_agent: "Mozilla/5.0 (X11; Linux x86_64) " \
39
+ user_agent: "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " \
26
40
  "AppleWebKit/537.36 (KHTML, like Gecko) " \
27
41
  "Chrome/131.0.0.0 Safari/537.36",
28
- platform: '"Linux"',
42
+ platform: '"Windows"',
29
43
  chrome_version: "131",
44
+ host_platform: :windows,
30
45
  },
31
46
  {
32
- user_agent: "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) " \
33
- "AppleWebKit/537.36 (KHTML, like Gecko) " \
34
- "Chrome/129.0.0.0 Safari/537.36",
35
- platform: '"macOS"',
36
- chrome_version: "129",
37
- },
38
- {
39
- user_agent: "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " \
47
+ user_agent: "Mozilla/5.0 (X11; Linux x86_64) " \
40
48
  "AppleWebKit/537.36 (KHTML, like Gecko) " \
41
49
  "Chrome/131.0.0.0 Safari/537.36",
42
- platform: '"Windows"',
50
+ platform: '"Linux"',
43
51
  chrome_version: "131",
52
+ host_platform: :linux,
44
53
  },
45
54
  ].freeze
46
55
 
47
- # Fetch a URL using headless Chrome, returning the page HTML.
48
- # Handles AWS WAF challenge pages by waiting for JS execution.
49
- def self.fetch(url, browser_opts: {})
50
- browser = Ferrum::Browser.new(
51
- headless: "new",
52
- timeout: 30,
53
- window_size: [1366, 768],
54
- browser_options: {
55
- "disable-blink-features" => "AutomationControlled",
56
- },
57
- **browser_opts,
58
- )
59
-
60
- browser.headers.set(random_headers)
61
- browser.go_to(url)
62
- browser.network.wait_for_idle(timeout: 15)
63
- html = browser.body
64
-
65
- if html.include?("403 ERROR") || html.include?("Request blocked")
66
- warn "IEV: AWS WAF blocked request for #{url}"
67
- return nil
68
- end
56
+ DEFAULT_LANG = "en-US,en"
57
+ DEFAULT_BROWSER_OPTIONS = {
58
+ "disable-blink-features" => "AutomationControlled",
59
+ "lang" => DEFAULT_LANG,
60
+ }.freeze
69
61
 
70
- html
71
- rescue Ferrum::Error, Ferrum::BrowserError => e
72
- warn "IEV: Browser error fetching #{url}: #{e.message}"
73
- nil
74
- ensure
75
- browser&.quit
62
+ # One-shot fetch. Each call spins up a fresh headless Chrome, fetches,
63
+ # and tears it down. Suitable for ad-hoc use; the WAF cookie does not
64
+ # survive between calls. Batch callers (Fetcher::Mirror) should use
65
+ # Session instead so the cookie set on the first successful challenge
66
+ # is reused across requests.
67
+ def self.fetch(url, **_browser_opts)
68
+ Session.new.fetch(url)
76
69
  end
77
70
 
71
+ # Returns request headers that match what real Chrome sends on a
72
+ # fresh address-bar navigation. AWS WAF fingerprints inconsistencies
73
+ # between these headers and the browser's runtime state, so we:
74
+ # - omit Sec-Fetch-* (Chrome computes those itself from the
75
+ # navigation context; setting them via Ferrum's Network domain
76
+ # overrides the real values and is detectable), and
77
+ # - keep Sec-Ch-Ua-Platform aligned with the host OS, which is
78
+ # what Chrome reports via navigator.platform.
78
79
  def self.random_headers
79
- profile = USER_AGENT_PROFILES.sample
80
- sec_ch_ua = "\"Google Chrome\";v=\"#{profile[:chrome_version]}\", " \
81
- "\"Chromium\";v=\"#{profile[:chrome_version]}\", " \
82
- "\"Not_A Brand\";v=\"24\""
80
+ profile = profile_for_host
81
+ static_headers.merge(headers_from_profile(profile))
82
+ end
83
83
 
84
+ def self.static_headers
84
85
  {
85
86
  "Accept" => "text/html,application/xhtml+xml,application/xml;q=0.9," \
86
87
  "image/avif,image/webp,image/apng,*/*;q=0.8," \
87
88
  "application/signed-exchange;v=b3;q=0.7",
88
- "Accept-Language" => "en-GB,en-US;q=0.9,en;q=0.8",
89
+ "Accept-Language" => "en-US,en;q=0.9",
89
90
  "Cache-Control" => "no-cache",
90
91
  "Pragma" => "no-cache",
91
- "Sec-Ch-Ua" => sec_ch_ua,
92
92
  "Sec-Ch-Ua-Mobile" => "?0",
93
- "Sec-Ch-Ua-Platform" => profile[:platform],
94
- "Sec-Fetch-Dest" => "document",
95
- "Sec-Fetch-Mode" => "navigate",
96
- "Sec-Fetch-Site" => "cross-site",
97
- "Sec-Fetch-User" => "?1",
98
93
  "Upgrade-Insecure-Requests" => "1",
94
+ }
95
+ end
96
+
97
+ def self.headers_from_profile(profile)
98
+ {
99
+ "Sec-Ch-Ua" => sec_ch_ua_for(profile),
100
+ "Sec-Ch-Ua-Platform" => profile[:platform],
99
101
  "User-Agent" => profile[:user_agent],
100
102
  }
101
103
  end
104
+
105
+ def self.profile_for_host
106
+ USER_AGENT_PROFILES.select do |profile|
107
+ profile[:host_platform] == host_platform
108
+ end.sample
109
+ end
110
+
111
+ def self.host_platform
112
+ case RUBY_PLATFORM
113
+ when /darwin/ then :mac
114
+ when /mswin|mingw|cygwin|bccwin|wince|emx/ then :windows
115
+ else :linux
116
+ end
117
+ end
118
+
119
+ def self.sec_ch_ua_for(profile)
120
+ "\"Google Chrome\";v=\"#{profile[:chrome_version]}\", " \
121
+ "\"Chromium\";v=\"#{profile[:chrome_version]}\", " \
122
+ "\"Not_A Brand\";v=\"24\""
123
+ end
124
+
125
+ # A long-lived headless Chrome session. Cookies persist across
126
+ # fetches, so once the AWS WAF challenge is cleared on the first
127
+ # request, subsequent requests reuse the token and succeed at
128
+ # near-100% rate. The Mirror creates one Session per run and
129
+ # shares it across all SequentialProbe iterations.
130
+ #
131
+ # Chrome leaks ~1MB per page load (mostly V8 heap that doesn't get
132
+ # GC'd between navigations). After ~1000 fetches the process is at
133
+ # ~1GB and the OOM risk climbs sharply. #restart quits the browser
134
+ # and starts a fresh one with a new cookie jar; the WAF challenge
135
+ # will need to be cleared again on the next fetch.
136
+ class Session
137
+ def initialize
138
+ @browser = Ferrum::Browser.new(
139
+ headless: "new",
140
+ timeout: 30,
141
+ window_size: [1366, 768],
142
+ browser_options: Browser::DEFAULT_BROWSER_OPTIONS,
143
+ )
144
+ @browser.headers.set(Browser.random_headers)
145
+ end
146
+
147
+ def fetch(url)
148
+ @browser.go_to(url)
149
+ @browser.network.wait_for_idle(timeout: 15)
150
+ reject_blocked(url, @browser.body)
151
+ rescue Ferrum::DeadBrowserError, Ferrum::NoSuchPageError,
152
+ Ferrum::NoSuchTargetError => e
153
+ # Chrome process or page/tab has crashed. Restart once, retry
154
+ # once. If the restart itself fails, surface as nil so the
155
+ # probe silently skips and the run continues.
156
+ if restart
157
+ retry
158
+ else
159
+ warn "IEV: Browser crashed, restart failed: #{e.message}"
160
+ nil
161
+ end
162
+ rescue Ferrum::BrowserError => e
163
+ if fetch_dead?(e.message) && restart
164
+ retry
165
+ else
166
+ warn "IEV: Browser error fetching #{url}: #{e.message}"
167
+ nil
168
+ end
169
+ rescue Ferrum::Error => e
170
+ warn "IEV: Browser error fetching #{url}: #{e.message}"
171
+ nil
172
+ end
173
+
174
+ def fetch_dead?(message)
175
+ message.include?("Browser is dead") ||
176
+ message.include?("given window is closed") ||
177
+ message.include?("Target closed")
178
+ end
179
+
180
+ def reject_blocked(url, html)
181
+ if html.include?("403 ERROR") || html.include?("Request blocked")
182
+ warn "IEV: AWS WAF blocked request for #{url}"
183
+ nil
184
+ else
185
+ html
186
+ end
187
+ end
188
+
189
+ # Quit the current browser and start a fresh one. The WAF cookie
190
+ # is lost, so the next fetch will go through the challenge cycle
191
+ # again. Call this after N fetches to bound Ferrum's memory growth,
192
+ # or rely on #fetch to call it automatically when Chrome dies.
193
+ # Returns true on success, false if the new browser fails to start.
194
+ def restart
195
+ quit
196
+ initialize
197
+ true
198
+ rescue StandardError => e
199
+ warn "IEV: Session restart failed: #{e.message}"
200
+ false
201
+ end
202
+
203
+ def quit
204
+ @browser&.quit
205
+ end
206
+ end
102
207
  end
103
208
  end
104
209
  end