iev 0.4.6 → 0.4.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.claude/scheduled_tasks.lock +1 -0
- data/.rubocop_todo.yml +5 -47
- data/CLAUDE.md +11 -1
- data/Gemfile +4 -2
- data/TODO.reconcile/01-design.md +57 -0
- data/TODO.reconcile/02-change-models.md +44 -0
- data/TODO.reconcile/02-termbase-loader.md +48 -0
- data/TODO.reconcile/03-live-loader.md +28 -0
- data/TODO.reconcile/03-status-mapper.md +24 -0
- data/TODO.reconcile/04-concept-merger.md +45 -0
- data/TODO.reconcile/04-termbase-loader.md +28 -0
- data/TODO.reconcile/05-content-diff.md +45 -0
- data/TODO.reconcile/05-live-loader.md +20 -0
- data/TODO.reconcile/06-content-differ.md +32 -0
- data/TODO.reconcile/06-v3-serializer.md +42 -0
- data/TODO.reconcile/07-cli-and-run.md +27 -0
- data/TODO.reconcile/07-concept-merger.md +32 -0
- data/TODO.reconcile/08-report.md +53 -0
- data/TODO.reconcile/09-pipeline.md +23 -0
- data/TODO.reconcile/10-script-and-run.md +15 -0
- data/data/locales/glossarist_enums.yml +705 -0
- data/iev.gemspec +4 -2
- data/lib/iev/bibliography_builder.rb +87 -0
- data/lib/iev/cli/command.rb +29 -0
- data/lib/iev/cli/command_helper.rb +1 -6
- data/lib/iev/config.rb +10 -1
- data/lib/iev/exporter.rb +37 -4
- data/lib/iev/figure_builder.rb +186 -0
- data/lib/iev/multi_doc_yaml.rb +60 -0
- data/lib/iev/reconciler/change.rb +29 -0
- data/lib/iev/reconciler/change_set.rb +48 -0
- data/lib/iev/reconciler/concept_merger.rb +130 -0
- data/lib/iev/reconciler/content_differ.rb +157 -0
- data/lib/iev/reconciler/live_loader.rb +163 -0
- data/lib/iev/reconciler/pipeline.rb +104 -0
- data/lib/iev/reconciler/reconciled_concept.rb +14 -0
- data/lib/iev/reconciler/report.rb +118 -0
- data/lib/iev/reconciler/status_mapper.rb +29 -0
- data/lib/iev/reconciler/term_marker_parser.rb +196 -0
- data/lib/iev/reconciler/termbase_loader.rb +141 -0
- data/lib/iev/reconciler.rb +17 -0
- data/lib/iev/scraper/browser.rb +155 -50
- data/lib/iev/scraper/page_parser.rb +116 -24
- data/lib/iev/source_parser.rb +12 -1
- data/lib/iev/subject_area_concepts.rb +4 -2
- data/lib/iev/utilities.rb +7 -5
- data/lib/iev/version.rb +1 -1
- data/lib/iev.rb +4 -0
- data/scripts/find_unparseable.rb +22 -0
- metadata +54 -6
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Iev
|
|
4
|
+
module Reconciler
|
|
5
|
+
# Extracts grammatical markers, context qualifiers, and part-of-speech
|
|
6
|
+
# tags that are inline in Electropedia's live HTML term text, matching
|
|
7
|
+
# how the termbase stores them in separate fields.
|
|
8
|
+
#
|
|
9
|
+
# Two-pass approach:
|
|
10
|
+
# Pass 1: extract context qualifiers and IEV cross-references from
|
|
11
|
+
# angle brackets (<text>, <相关条目:IEV xxx>)
|
|
12
|
+
# Pass 2: extract gender/number/POS/prefix markers from remaining text
|
|
13
|
+
#
|
|
14
|
+
# Marker formats handled:
|
|
15
|
+
# Gender: , f , m , n (also multiple: , f, n)
|
|
16
|
+
# , ж , м , с (Serbian Cyrillic)
|
|
17
|
+
# , m/f (slash = both apply)
|
|
18
|
+
# Number: , pl , sg (also Serbian: јд, мн)
|
|
19
|
+
# , f pl (space-separated gender+number)
|
|
20
|
+
# Usage info: <text> → usage_info field
|
|
21
|
+
# Part of speech: , 名詞 , 명사 , noun , verb , adj. , agg , etc.
|
|
22
|
+
# Prefix: , Präfix , 접두사 , 接頭語 , etc. → prefix flag
|
|
23
|
+
class TermMarkerParser
|
|
24
|
+
GENDER_MAP = {
|
|
25
|
+
"f" => "feminine", "m" => "masculine", "n" => "neuter",
|
|
26
|
+
"ж" => "feminine", "м" => "masculine", "с" => "neuter",
|
|
27
|
+
}.freeze
|
|
28
|
+
|
|
29
|
+
NUMBER_MAP = {
|
|
30
|
+
"јд" => "singular", "мн" => "plural",
|
|
31
|
+
"sg" => "singular", "pl" => "plural",
|
|
32
|
+
}.freeze
|
|
33
|
+
|
|
34
|
+
PART_OF_SPEECH_MAP = {
|
|
35
|
+
"noun" => "noun",
|
|
36
|
+
"명사" => "noun", "名詞" => "noun", "اسم" => "noun",
|
|
37
|
+
"именица" => "noun",
|
|
38
|
+
"verb" => "verb", "verbo" => "verb",
|
|
39
|
+
"동사" => "verb", "動詞" => "verb", "فعل" => "verb",
|
|
40
|
+
"глагол" => "verb",
|
|
41
|
+
"adj" => "adj", "adj." => "adj", "adjective" => "adj",
|
|
42
|
+
"agg" => "adj", "agg." => "adj",
|
|
43
|
+
"형용사" => "adj", "形容詞" => "adj", "صفة" => "adj",
|
|
44
|
+
"adjektiv" => "adj", "придев" => "adj",
|
|
45
|
+
"adv" => "adv", "adv." => "adv", "adverb" => "adv",
|
|
46
|
+
"부사" => "adv", "副詞" => "adv",
|
|
47
|
+
"прилог" => "adv",
|
|
48
|
+
}.freeze
|
|
49
|
+
|
|
50
|
+
PREFIX_KEYWORDS = %w[
|
|
51
|
+
Präfix prefix préfixe 접두사 接頭語 接尾語
|
|
52
|
+
].freeze
|
|
53
|
+
|
|
54
|
+
DOMAIN_RE = /<([^>]+)>/
|
|
55
|
+
USAGE_INFO_RE = /\(([^)]+)\)\s*(?=[,]?\s*(?:[fmnжмс]\b|[fmn]\/[fmn]|sg|pl|јд|мн|\z))/i
|
|
56
|
+
IEV_XREF_RE = /<[^>]*IEV\s*(\d{3}-\d{2}-\d{2,3})[^>]*>/i
|
|
57
|
+
|
|
58
|
+
SERBIAN_RE = /,\s*([жмс])\s+(јд|мн)\s*\z/
|
|
59
|
+
WESTERN_GENDER_NUMBER_RE = /,\s*([fmn])\s+(sg|pl)\.?\s*\z/i
|
|
60
|
+
WESTERN_NUMBER_RE = /,\s*(sg|pl)\.?\s*\z/i
|
|
61
|
+
SLASH_GENDER_RE = /,\s*([fmn])\/([fmn])\s*\z/i
|
|
62
|
+
WESTERN_RE = /[,]?\s+([fmn])\s*\z/
|
|
63
|
+
SPACE_GENDER_RE = /\s+([fmn])\s*\z/
|
|
64
|
+
|
|
65
|
+
POS_RE = /,\s*(#{PART_OF_SPEECH_MAP.keys.map { |k| Regexp.escape(k) }.join("|")})\s*\z/i
|
|
66
|
+
PREFIX_RE = /,\s*(#{PREFIX_KEYWORDS.map { |k| Regexp.escape(k) }.join("|")})\s*\z/
|
|
67
|
+
|
|
68
|
+
TRAILING_PUNCT_RE = /[,;\s]+$/
|
|
69
|
+
|
|
70
|
+
Result = Struct.new(:designation, :genders, :numbers, :domain,
|
|
71
|
+
:usage_info, :part_of_speech, :related_refs,
|
|
72
|
+
:is_prefix, keyword_init: true)
|
|
73
|
+
|
|
74
|
+
class << self
|
|
75
|
+
def parse(term)
|
|
76
|
+
return Result.new(designation: nil) unless term
|
|
77
|
+
|
|
78
|
+
text = decode_entities(term).strip.gsub(/\s+/, " ").gsub(/ /, " ")
|
|
79
|
+
related_refs, text = extract_iev_xrefs(text)
|
|
80
|
+
domain, text = extract_domain(text)
|
|
81
|
+
usage_info, text = extract_usage_info(text)
|
|
82
|
+
designation, genders, numbers, pos, is_prefix = extract_markers(text)
|
|
83
|
+
|
|
84
|
+
Result.new(
|
|
85
|
+
designation: designation,
|
|
86
|
+
genders: genders,
|
|
87
|
+
numbers: numbers,
|
|
88
|
+
domain: domain,
|
|
89
|
+
usage_info: usage_info,
|
|
90
|
+
part_of_speech: pos,
|
|
91
|
+
related_refs: related_refs,
|
|
92
|
+
is_prefix: is_prefix,
|
|
93
|
+
)
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
def parse_multiple(term_text)
|
|
97
|
+
return [] unless term_text
|
|
98
|
+
|
|
99
|
+
term_text
|
|
100
|
+
.split(/\n+/)
|
|
101
|
+
.map(&:strip)
|
|
102
|
+
.reject(&:empty?)
|
|
103
|
+
.map { |part| parse(part) }
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
private
|
|
107
|
+
|
|
108
|
+
def decode_entities(str)
|
|
109
|
+
str
|
|
110
|
+
.gsub(/</, "<")
|
|
111
|
+
.gsub(/>/, ">")
|
|
112
|
+
.gsub(/&/, "&")
|
|
113
|
+
.gsub(/"/, '"')
|
|
114
|
+
.gsub(/'/, "'")
|
|
115
|
+
end
|
|
116
|
+
|
|
117
|
+
# Extract domain qualifier from angle brackets: <text> -> domain
|
|
118
|
+
def extract_domain(text)
|
|
119
|
+
match = text.match(DOMAIN_RE)
|
|
120
|
+
return [nil, text] unless match
|
|
121
|
+
|
|
122
|
+
domain = match[1].strip
|
|
123
|
+
remaining = text.sub(match[0], "").gsub(TRAILING_PUNCT_RE, "").strip
|
|
124
|
+
remaining = remaining.sub(/^[,;\s]+/, "").strip
|
|
125
|
+
[domain, remaining]
|
|
126
|
+
end
|
|
127
|
+
|
|
128
|
+
def extract_iev_xrefs(text)
|
|
129
|
+
refs = []
|
|
130
|
+
remaining = text
|
|
131
|
+
while (match = remaining.match(IEV_XREF_RE))
|
|
132
|
+
refs << match[1]
|
|
133
|
+
remaining = remaining.sub(match[0], "").gsub(TRAILING_PUNCT_RE, "").strip
|
|
134
|
+
remaining = remaining.sub(/^[,;\s]+/, "").strip
|
|
135
|
+
end
|
|
136
|
+
[refs.empty? ? nil : refs, remaining]
|
|
137
|
+
end
|
|
138
|
+
|
|
139
|
+
# Extract usage_info from parentheses that precede trailing markers.
|
|
140
|
+
# E.g. "Orientierung (einer Kurve) f" -> usage_info="einer Kurve"
|
|
141
|
+
def extract_usage_info(text)
|
|
142
|
+
match = text.match(USAGE_INFO_RE)
|
|
143
|
+
return [nil, text] unless match
|
|
144
|
+
|
|
145
|
+
usage = match[1].strip
|
|
146
|
+
remaining = text.sub(match[0], "").gsub(/\s+/, " ").strip
|
|
147
|
+
[usage, remaining]
|
|
148
|
+
end
|
|
149
|
+
|
|
150
|
+
def extract_markers(text)
|
|
151
|
+
designation = text
|
|
152
|
+
genders = []
|
|
153
|
+
numbers = []
|
|
154
|
+
pos = nil
|
|
155
|
+
is_prefix = false
|
|
156
|
+
|
|
157
|
+
loop do
|
|
158
|
+
break if designation.nil? || designation.empty?
|
|
159
|
+
|
|
160
|
+
if (m = designation.match(SERBIAN_RE))
|
|
161
|
+
designation = designation[0, m.begin(0)].strip
|
|
162
|
+
genders << GENDER_MAP[m[1]] if GENDER_MAP[m[1]]
|
|
163
|
+
numbers << NUMBER_MAP[m[2]] if NUMBER_MAP[m[2]]
|
|
164
|
+
elsif (m = designation.match(SLASH_GENDER_RE))
|
|
165
|
+
designation = designation[0, m.begin(0)].strip
|
|
166
|
+
genders << GENDER_MAP[m[1]] if GENDER_MAP[m[1]]
|
|
167
|
+
genders << GENDER_MAP[m[2]] if GENDER_MAP[m[2]]
|
|
168
|
+
elsif (m = designation.match(WESTERN_GENDER_NUMBER_RE))
|
|
169
|
+
designation = designation[0, m.begin(0)].strip
|
|
170
|
+
genders << GENDER_MAP[m[1]] if GENDER_MAP[m[1]]
|
|
171
|
+
num_key = m[2].downcase.sub(/\.$/, "")
|
|
172
|
+
numbers << NUMBER_MAP[num_key] if NUMBER_MAP[num_key]
|
|
173
|
+
elsif (m = designation.match(WESTERN_NUMBER_RE))
|
|
174
|
+
designation = designation[0, m.begin(0)].strip
|
|
175
|
+
num_key = m[1].downcase.sub(/\.$/, "")
|
|
176
|
+
numbers << NUMBER_MAP[num_key] if NUMBER_MAP[num_key]
|
|
177
|
+
elsif (m = designation.match(POS_RE))
|
|
178
|
+
designation = designation[0, m.begin(0)].strip
|
|
179
|
+
pos = PART_OF_SPEECH_MAP[m[1].downcase]
|
|
180
|
+
elsif (m = designation.match(PREFIX_RE))
|
|
181
|
+
designation = designation[0, m.begin(0)].strip
|
|
182
|
+
is_prefix = true
|
|
183
|
+
elsif (m = designation.match(WESTERN_RE))
|
|
184
|
+
designation = designation[0, m.begin(0)].strip
|
|
185
|
+
genders << GENDER_MAP[m[1]] if GENDER_MAP[m[1]]
|
|
186
|
+
else
|
|
187
|
+
break
|
|
188
|
+
end
|
|
189
|
+
end
|
|
190
|
+
|
|
191
|
+
[designation, genders.uniq, numbers.uniq, pos, is_prefix]
|
|
192
|
+
end
|
|
193
|
+
end
|
|
194
|
+
end
|
|
195
|
+
end
|
|
196
|
+
end
|
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "yaml"
|
|
4
|
+
|
|
5
|
+
module Iev
|
|
6
|
+
module Reconciler
|
|
7
|
+
# Loads termbase.yaml into a raw hash, then builds
|
|
8
|
+
# Glossarist::ManagedConcept objects on demand per code.
|
|
9
|
+
# This avoids the cost of constructing 22k model objects upfront.
|
|
10
|
+
class TermbaseLoader
|
|
11
|
+
# @param path [String, Pathname] path to termbase.yaml
|
|
12
|
+
def initialize(path)
|
|
13
|
+
@path = path.to_s
|
|
14
|
+
@data = nil
|
|
15
|
+
end
|
|
16
|
+
|
|
17
|
+
# @return [Hash<String, Hash>] raw termbase data keyed by code
|
|
18
|
+
def raw
|
|
19
|
+
@data ||= YAML.load_file(@path)
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
# @return [Array<String>] all codes in the termbase
|
|
23
|
+
def codes
|
|
24
|
+
raw.keys.map(&:to_s)
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
# Build a single ManagedConcept for the given code.
|
|
28
|
+
# @param code [String]
|
|
29
|
+
# @return [Glossarist::ManagedConcept, nil]
|
|
30
|
+
def get(code)
|
|
31
|
+
entry = raw[code] || raw[code.to_sym]
|
|
32
|
+
return nil unless entry.is_a?(Hash)
|
|
33
|
+
|
|
34
|
+
build_concept(code.to_s, entry)
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
private
|
|
38
|
+
|
|
39
|
+
def build_concept(code, entry)
|
|
40
|
+
concept = Glossarist::ManagedConcept.of_yaml(
|
|
41
|
+
"id" => code, "data" => { "id" => code },
|
|
42
|
+
)
|
|
43
|
+
concept.status = concept_status(entry)
|
|
44
|
+
concept.dates = concept_dates(entry)
|
|
45
|
+
concept.schema_version = "3"
|
|
46
|
+
|
|
47
|
+
build_localized(code, entry).each do |lc|
|
|
48
|
+
concept.add_l10n(lc)
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
concept
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
def concept_status(entry)
|
|
55
|
+
lang_data = first_language(entry)
|
|
56
|
+
StatusMapper.call(lang_data&.dig("entry_status"))
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
def concept_dates(entry)
|
|
60
|
+
lang_data = first_language(entry)
|
|
61
|
+
return [] unless lang_data
|
|
62
|
+
|
|
63
|
+
dates = []
|
|
64
|
+
if lang_data["date_accepted"]
|
|
65
|
+
dates << Glossarist::ConceptDate.new(
|
|
66
|
+
type: "accepted",
|
|
67
|
+
date: lang_data["date_accepted"],
|
|
68
|
+
)
|
|
69
|
+
end
|
|
70
|
+
if lang_data["date_amended"] && lang_data["date_amended"] != lang_data["date_accepted"]
|
|
71
|
+
dates << Glossarist::ConceptDate.new(
|
|
72
|
+
type: "amended",
|
|
73
|
+
date: lang_data["date_amended"],
|
|
74
|
+
)
|
|
75
|
+
end
|
|
76
|
+
dates
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
def first_language(entry)
|
|
80
|
+
entry.values.find { |v| v.is_a?(Hash) && v["terms"] }
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
def build_localized(code, entry)
|
|
84
|
+
result = []
|
|
85
|
+
entry.each do |lang, data|
|
|
86
|
+
next unless data.is_a?(Hash) && data["terms"]
|
|
87
|
+
lc = build_localized_concept(code, lang, data)
|
|
88
|
+
result << lc if lc
|
|
89
|
+
end
|
|
90
|
+
result
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
def build_localized_concept(code, lang, data)
|
|
94
|
+
cdata = Glossarist::ConceptData.new
|
|
95
|
+
cdata.id = code
|
|
96
|
+
cdata.language_code = lang
|
|
97
|
+
cdata.entry_status = StatusMapper.call(data["entry_status"])
|
|
98
|
+
|
|
99
|
+
cdata.terms = (data["terms"] || []).map do |t|
|
|
100
|
+
build_term_expression(t)
|
|
101
|
+
end
|
|
102
|
+
|
|
103
|
+
if data["definition"] && !data["definition"].empty?
|
|
104
|
+
cdata.definition = [Glossarist::DetailedDefinition.new(content: data["definition"])]
|
|
105
|
+
end
|
|
106
|
+
|
|
107
|
+
cdata.notes = Array(data["notes"]).map { |n| Glossarist::DetailedDefinition.new(content: n) }
|
|
108
|
+
cdata.examples = Array(data["examples"]).map { |e| Glossarist::DetailedDefinition.new(content: e) }
|
|
109
|
+
|
|
110
|
+
lc = Glossarist::LocalizedConcept.new
|
|
111
|
+
lc.id = code
|
|
112
|
+
lc.data = cdata
|
|
113
|
+
lc
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
def build_term_expression(term_data)
|
|
117
|
+
designation = term_data["designation"].to_s
|
|
118
|
+
parsed = TermMarkerParser.parse(designation)
|
|
119
|
+
|
|
120
|
+
expr = Glossarist::Designation::Expression.new(
|
|
121
|
+
designation: parsed.designation,
|
|
122
|
+
normative_status: term_data["normative_status"] || "preferred",
|
|
123
|
+
type: term_data["type"] || "expression",
|
|
124
|
+
)
|
|
125
|
+
expr.geographical_area = term_data["geographical_area"] if term_data["geographical_area"]
|
|
126
|
+
expr.field_of_application = term_data["usage_info"] || parsed.domain
|
|
127
|
+
expr.usage_info = parsed.usage_info if parsed.usage_info
|
|
128
|
+
expr.prefix = parsed.is_prefix if parsed.is_prefix
|
|
129
|
+
|
|
130
|
+
grammar = Glossarist::Designation::GrammarInfo.new
|
|
131
|
+
merged_genders = (Array(term_data["gender"]) + parsed.genders).flatten.compact.uniq
|
|
132
|
+
merged_numbers = (Array(term_data["plurality"]) + parsed.numbers).flatten.compact.uniq
|
|
133
|
+
grammar.gender = merged_genders if merged_genders.any?
|
|
134
|
+
grammar.number = merged_numbers if merged_numbers.any?
|
|
135
|
+
grammar.part_of_speech = parsed.part_of_speech if parsed.part_of_speech
|
|
136
|
+
expr.grammar_info = [grammar] if grammar.gender&.any? || grammar.number&.any? || grammar.part_of_speech
|
|
137
|
+
expr
|
|
138
|
+
end
|
|
139
|
+
end
|
|
140
|
+
end
|
|
141
|
+
end
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Iev
|
|
4
|
+
module Reconciler
|
|
5
|
+
autoload :Change, "iev/reconciler/change"
|
|
6
|
+
autoload :ChangeSet, "iev/reconciler/change_set"
|
|
7
|
+
autoload :ConceptMerger, "iev/reconciler/concept_merger"
|
|
8
|
+
autoload :ContentDiffer, "iev/reconciler/content_differ"
|
|
9
|
+
autoload :LiveLoader, "iev/reconciler/live_loader"
|
|
10
|
+
autoload :Pipeline, "iev/reconciler/pipeline"
|
|
11
|
+
autoload :ReconciledConcept, "iev/reconciler/reconciled_concept"
|
|
12
|
+
autoload :Report, "iev/reconciler/report"
|
|
13
|
+
autoload :StatusMapper, "iev/reconciler/status_mapper"
|
|
14
|
+
autoload :TermMarkerParser, "iev/reconciler/term_marker_parser"
|
|
15
|
+
autoload :TermbaseLoader, "iev/reconciler/termbase_loader"
|
|
16
|
+
end
|
|
17
|
+
end
|
data/lib/iev/scraper/browser.rb
CHANGED
|
@@ -6,6 +6,10 @@ module Iev
|
|
|
6
6
|
class Scraper
|
|
7
7
|
# Shared headless browser utilities for fetching pages behind AWS WAF.
|
|
8
8
|
module Browser
|
|
9
|
+
# Each profile is tagged with the host platform it can run on so
|
|
10
|
+
# that navigator.platform (set by Chrome from the real OS) agrees
|
|
11
|
+
# with the User-Agent and Sec-Ch-Ua-Platform headers. AWS WAF
|
|
12
|
+
# fingerprints this mismatch and refuses to clear its challenge.
|
|
9
13
|
USER_AGENT_PROFILES = [
|
|
10
14
|
{
|
|
11
15
|
user_agent: "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) " \
|
|
@@ -13,6 +17,15 @@ module Iev
|
|
|
13
17
|
"Chrome/131.0.0.0 Safari/537.36",
|
|
14
18
|
platform: '"macOS"',
|
|
15
19
|
chrome_version: "131",
|
|
20
|
+
host_platform: :mac,
|
|
21
|
+
},
|
|
22
|
+
{
|
|
23
|
+
user_agent: "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) " \
|
|
24
|
+
"AppleWebKit/537.36 (KHTML, like Gecko) " \
|
|
25
|
+
"Chrome/129.0.0.0 Safari/537.36",
|
|
26
|
+
platform: '"macOS"',
|
|
27
|
+
chrome_version: "129",
|
|
28
|
+
host_platform: :mac,
|
|
16
29
|
},
|
|
17
30
|
{
|
|
18
31
|
user_agent: "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " \
|
|
@@ -20,85 +33,177 @@ module Iev
|
|
|
20
33
|
"Chrome/130.0.0.0 Safari/537.36",
|
|
21
34
|
platform: '"Windows"',
|
|
22
35
|
chrome_version: "130",
|
|
36
|
+
host_platform: :windows,
|
|
23
37
|
},
|
|
24
38
|
{
|
|
25
|
-
user_agent: "Mozilla/5.0 (
|
|
39
|
+
user_agent: "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " \
|
|
26
40
|
"AppleWebKit/537.36 (KHTML, like Gecko) " \
|
|
27
41
|
"Chrome/131.0.0.0 Safari/537.36",
|
|
28
|
-
platform: '"
|
|
42
|
+
platform: '"Windows"',
|
|
29
43
|
chrome_version: "131",
|
|
44
|
+
host_platform: :windows,
|
|
30
45
|
},
|
|
31
46
|
{
|
|
32
|
-
user_agent: "Mozilla/5.0 (
|
|
33
|
-
"AppleWebKit/537.36 (KHTML, like Gecko) " \
|
|
34
|
-
"Chrome/129.0.0.0 Safari/537.36",
|
|
35
|
-
platform: '"macOS"',
|
|
36
|
-
chrome_version: "129",
|
|
37
|
-
},
|
|
38
|
-
{
|
|
39
|
-
user_agent: "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " \
|
|
47
|
+
user_agent: "Mozilla/5.0 (X11; Linux x86_64) " \
|
|
40
48
|
"AppleWebKit/537.36 (KHTML, like Gecko) " \
|
|
41
49
|
"Chrome/131.0.0.0 Safari/537.36",
|
|
42
|
-
platform: '"
|
|
50
|
+
platform: '"Linux"',
|
|
43
51
|
chrome_version: "131",
|
|
52
|
+
host_platform: :linux,
|
|
44
53
|
},
|
|
45
54
|
].freeze
|
|
46
55
|
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
timeout: 30,
|
|
53
|
-
window_size: [1366, 768],
|
|
54
|
-
browser_options: {
|
|
55
|
-
"disable-blink-features" => "AutomationControlled",
|
|
56
|
-
},
|
|
57
|
-
**browser_opts,
|
|
58
|
-
)
|
|
59
|
-
|
|
60
|
-
browser.headers.set(random_headers)
|
|
61
|
-
browser.go_to(url)
|
|
62
|
-
browser.network.wait_for_idle(timeout: 15)
|
|
63
|
-
html = browser.body
|
|
64
|
-
|
|
65
|
-
if html.include?("403 ERROR") || html.include?("Request blocked")
|
|
66
|
-
warn "IEV: AWS WAF blocked request for #{url}"
|
|
67
|
-
return nil
|
|
68
|
-
end
|
|
56
|
+
DEFAULT_LANG = "en-US,en"
|
|
57
|
+
DEFAULT_BROWSER_OPTIONS = {
|
|
58
|
+
"disable-blink-features" => "AutomationControlled",
|
|
59
|
+
"lang" => DEFAULT_LANG,
|
|
60
|
+
}.freeze
|
|
69
61
|
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
62
|
+
# One-shot fetch. Each call spins up a fresh headless Chrome, fetches,
|
|
63
|
+
# and tears it down. Suitable for ad-hoc use; the WAF cookie does not
|
|
64
|
+
# survive between calls. Batch callers (Fetcher::Mirror) should use
|
|
65
|
+
# Session instead so the cookie set on the first successful challenge
|
|
66
|
+
# is reused across requests.
|
|
67
|
+
def self.fetch(url, **_browser_opts)
|
|
68
|
+
Session.new.fetch(url)
|
|
76
69
|
end
|
|
77
70
|
|
|
71
|
+
# Returns request headers that match what real Chrome sends on a
|
|
72
|
+
# fresh address-bar navigation. AWS WAF fingerprints inconsistencies
|
|
73
|
+
# between these headers and the browser's runtime state, so we:
|
|
74
|
+
# - omit Sec-Fetch-* (Chrome computes those itself from the
|
|
75
|
+
# navigation context; setting them via Ferrum's Network domain
|
|
76
|
+
# overrides the real values and is detectable), and
|
|
77
|
+
# - keep Sec-Ch-Ua-Platform aligned with the host OS, which is
|
|
78
|
+
# what Chrome reports via navigator.platform.
|
|
78
79
|
def self.random_headers
|
|
79
|
-
profile =
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
"\"Not_A Brand\";v=\"24\""
|
|
80
|
+
profile = profile_for_host
|
|
81
|
+
static_headers.merge(headers_from_profile(profile))
|
|
82
|
+
end
|
|
83
83
|
|
|
84
|
+
def self.static_headers
|
|
84
85
|
{
|
|
85
86
|
"Accept" => "text/html,application/xhtml+xml,application/xml;q=0.9," \
|
|
86
87
|
"image/avif,image/webp,image/apng,*/*;q=0.8," \
|
|
87
88
|
"application/signed-exchange;v=b3;q=0.7",
|
|
88
|
-
"Accept-Language" => "en-
|
|
89
|
+
"Accept-Language" => "en-US,en;q=0.9",
|
|
89
90
|
"Cache-Control" => "no-cache",
|
|
90
91
|
"Pragma" => "no-cache",
|
|
91
|
-
"Sec-Ch-Ua" => sec_ch_ua,
|
|
92
92
|
"Sec-Ch-Ua-Mobile" => "?0",
|
|
93
|
-
"Sec-Ch-Ua-Platform" => profile[:platform],
|
|
94
|
-
"Sec-Fetch-Dest" => "document",
|
|
95
|
-
"Sec-Fetch-Mode" => "navigate",
|
|
96
|
-
"Sec-Fetch-Site" => "cross-site",
|
|
97
|
-
"Sec-Fetch-User" => "?1",
|
|
98
93
|
"Upgrade-Insecure-Requests" => "1",
|
|
94
|
+
}
|
|
95
|
+
end
|
|
96
|
+
|
|
97
|
+
def self.headers_from_profile(profile)
|
|
98
|
+
{
|
|
99
|
+
"Sec-Ch-Ua" => sec_ch_ua_for(profile),
|
|
100
|
+
"Sec-Ch-Ua-Platform" => profile[:platform],
|
|
99
101
|
"User-Agent" => profile[:user_agent],
|
|
100
102
|
}
|
|
101
103
|
end
|
|
104
|
+
|
|
105
|
+
def self.profile_for_host
|
|
106
|
+
USER_AGENT_PROFILES.select do |profile|
|
|
107
|
+
profile[:host_platform] == host_platform
|
|
108
|
+
end.sample
|
|
109
|
+
end
|
|
110
|
+
|
|
111
|
+
def self.host_platform
|
|
112
|
+
case RUBY_PLATFORM
|
|
113
|
+
when /darwin/ then :mac
|
|
114
|
+
when /mswin|mingw|cygwin|bccwin|wince|emx/ then :windows
|
|
115
|
+
else :linux
|
|
116
|
+
end
|
|
117
|
+
end
|
|
118
|
+
|
|
119
|
+
def self.sec_ch_ua_for(profile)
|
|
120
|
+
"\"Google Chrome\";v=\"#{profile[:chrome_version]}\", " \
|
|
121
|
+
"\"Chromium\";v=\"#{profile[:chrome_version]}\", " \
|
|
122
|
+
"\"Not_A Brand\";v=\"24\""
|
|
123
|
+
end
|
|
124
|
+
|
|
125
|
+
# A long-lived headless Chrome session. Cookies persist across
|
|
126
|
+
# fetches, so once the AWS WAF challenge is cleared on the first
|
|
127
|
+
# request, subsequent requests reuse the token and succeed at
|
|
128
|
+
# near-100% rate. The Mirror creates one Session per run and
|
|
129
|
+
# shares it across all SequentialProbe iterations.
|
|
130
|
+
#
|
|
131
|
+
# Chrome leaks ~1MB per page load (mostly V8 heap that doesn't get
|
|
132
|
+
# GC'd between navigations). After ~1000 fetches the process is at
|
|
133
|
+
# ~1GB and the OOM risk climbs sharply. #restart quits the browser
|
|
134
|
+
# and starts a fresh one with a new cookie jar; the WAF challenge
|
|
135
|
+
# will need to be cleared again on the next fetch.
|
|
136
|
+
class Session
|
|
137
|
+
def initialize
|
|
138
|
+
@browser = Ferrum::Browser.new(
|
|
139
|
+
headless: "new",
|
|
140
|
+
timeout: 30,
|
|
141
|
+
window_size: [1366, 768],
|
|
142
|
+
browser_options: Browser::DEFAULT_BROWSER_OPTIONS,
|
|
143
|
+
)
|
|
144
|
+
@browser.headers.set(Browser.random_headers)
|
|
145
|
+
end
|
|
146
|
+
|
|
147
|
+
def fetch(url)
|
|
148
|
+
@browser.go_to(url)
|
|
149
|
+
@browser.network.wait_for_idle(timeout: 15)
|
|
150
|
+
reject_blocked(url, @browser.body)
|
|
151
|
+
rescue Ferrum::DeadBrowserError, Ferrum::NoSuchPageError,
|
|
152
|
+
Ferrum::NoSuchTargetError => e
|
|
153
|
+
# Chrome process or page/tab has crashed. Restart once, retry
|
|
154
|
+
# once. If the restart itself fails, surface as nil so the
|
|
155
|
+
# probe silently skips and the run continues.
|
|
156
|
+
if restart
|
|
157
|
+
retry
|
|
158
|
+
else
|
|
159
|
+
warn "IEV: Browser crashed, restart failed: #{e.message}"
|
|
160
|
+
nil
|
|
161
|
+
end
|
|
162
|
+
rescue Ferrum::BrowserError => e
|
|
163
|
+
if fetch_dead?(e.message) && restart
|
|
164
|
+
retry
|
|
165
|
+
else
|
|
166
|
+
warn "IEV: Browser error fetching #{url}: #{e.message}"
|
|
167
|
+
nil
|
|
168
|
+
end
|
|
169
|
+
rescue Ferrum::Error => e
|
|
170
|
+
warn "IEV: Browser error fetching #{url}: #{e.message}"
|
|
171
|
+
nil
|
|
172
|
+
end
|
|
173
|
+
|
|
174
|
+
def fetch_dead?(message)
|
|
175
|
+
message.include?("Browser is dead") ||
|
|
176
|
+
message.include?("given window is closed") ||
|
|
177
|
+
message.include?("Target closed")
|
|
178
|
+
end
|
|
179
|
+
|
|
180
|
+
def reject_blocked(url, html)
|
|
181
|
+
if html.include?("403 ERROR") || html.include?("Request blocked")
|
|
182
|
+
warn "IEV: AWS WAF blocked request for #{url}"
|
|
183
|
+
nil
|
|
184
|
+
else
|
|
185
|
+
html
|
|
186
|
+
end
|
|
187
|
+
end
|
|
188
|
+
|
|
189
|
+
# Quit the current browser and start a fresh one. The WAF cookie
|
|
190
|
+
# is lost, so the next fetch will go through the challenge cycle
|
|
191
|
+
# again. Call this after N fetches to bound Ferrum's memory growth,
|
|
192
|
+
# or rely on #fetch to call it automatically when Chrome dies.
|
|
193
|
+
# Returns true on success, false if the new browser fails to start.
|
|
194
|
+
def restart
|
|
195
|
+
quit
|
|
196
|
+
initialize
|
|
197
|
+
true
|
|
198
|
+
rescue StandardError => e
|
|
199
|
+
warn "IEV: Session restart failed: #{e.message}"
|
|
200
|
+
false
|
|
201
|
+
end
|
|
202
|
+
|
|
203
|
+
def quit
|
|
204
|
+
@browser&.quit
|
|
205
|
+
end
|
|
206
|
+
end
|
|
102
207
|
end
|
|
103
208
|
end
|
|
104
209
|
end
|