html2rss 0.24.0 → 0.25.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/lib/html2rss/auto_source/scraper/html.rb +68 -145
- data/lib/html2rss/auto_source/scraper/json_state/article_normalizer.rb +116 -0
- data/lib/html2rss/auto_source/scraper/json_state/candidate_detector.rb +63 -0
- data/lib/html2rss/auto_source/scraper/json_state/document_scanner.rb +179 -0
- data/lib/html2rss/auto_source/scraper/json_state/value_finder.rb +55 -0
- data/lib/html2rss/auto_source/scraper/json_state.rb +0 -379
- data/lib/html2rss/auto_source/scraper/semantic_html/entry_deduplicator.rb +35 -30
- data/lib/html2rss/auto_source/scraper/semantic_html.rb +40 -120
- data/lib/html2rss/auto_source/scraper/sitemap/parser.rb +161 -0
- data/lib/html2rss/auto_source/scraper/sitemap.rb +10 -7
- data/lib/html2rss/auto_source/scraper/wordpress_api.rb +6 -2
- data/lib/html2rss/auto_source/scraper.rb +81 -44
- data/lib/html2rss/auto_source/segment.rb +29 -0
- data/lib/html2rss/auto_source/segmenter/cluster.rb +163 -0
- data/lib/html2rss/auto_source/segmenter/list.rb +101 -0
- data/lib/html2rss/auto_source/segmenter/primary_link.rb +81 -0
- data/lib/html2rss/auto_source/segmenter/semantic.rb +104 -0
- data/lib/html2rss/auto_source/segmenter.rb +66 -0
- data/lib/html2rss/auto_source.rb +74 -21
- data/lib/html2rss/cli.rb +8 -13
- data/lib/html2rss/config/auto_source_contract.rb +2 -0
- data/lib/html2rss/config/request_controls.rb +33 -0
- data/lib/html2rss/config/validator.rb +15 -8
- data/lib/html2rss/config.rb +6 -2
- data/lib/html2rss/feed_builder/rss.rb +17 -5
- data/lib/html2rss/feed_pipeline.rb +4 -0
- data/lib/html2rss/feed_result.rb +1 -1
- data/lib/html2rss/html/article_extractor/category_extractor.rb +10 -36
- data/lib/html2rss/html/article_extractor/date_extractor.rb +2 -7
- data/lib/html2rss/html/article_extractor/enclosure_extractor.rb +5 -50
- data/lib/html2rss/html/article_extractor/image_extractor.rb +5 -16
- data/lib/html2rss/html/article_rules/category.rb +55 -0
- data/lib/html2rss/html/article_rules/date.rb +25 -0
- data/lib/html2rss/html/article_rules/enclosure.rb +72 -0
- data/lib/html2rss/html/article_rules/image.rb +109 -0
- data/lib/html2rss/html/article_rules.rb +10 -0
- data/lib/html2rss/html/rendering/audio_renderer.rb +2 -12
- data/lib/html2rss/html/rendering/escaped_attributes.rb +27 -0
- data/lib/html2rss/html/rendering/image_renderer.rb +2 -12
- data/lib/html2rss/html/rendering/pdf_renderer.rb +2 -8
- data/lib/html2rss/html/rendering/video_renderer.rb +2 -12
- data/lib/html2rss/html/sst_article_extractor.rb +279 -0
- data/lib/html2rss/link_destination/destination_facts.rb +40 -0
- data/lib/html2rss/link_destination/noise_policy.rb +79 -0
- data/lib/html2rss/link_destination/path_classifier.rb +205 -0
- data/lib/html2rss/link_destination/text_classifier.rb +64 -0
- data/lib/html2rss/link_destination.rb +8 -0
- data/lib/html2rss/request_service/faraday_strategy.rb +42 -1
- data/lib/html2rss/request_service/network_guard.rb +5 -3
- data/lib/html2rss/scoring/anchor_score.rb +34 -0
- data/lib/html2rss/scoring/cluster_scorer.rb +87 -0
- data/lib/html2rss/scoring/container_assessor.rb +83 -0
- data/lib/html2rss/scoring/engine.rb +101 -0
- data/lib/html2rss/scoring/link_resolver.rb +72 -0
- data/lib/html2rss/scoring/observation.rb +53 -0
- data/lib/html2rss/scoring/ranked_segment.rb +45 -0
- data/lib/html2rss/scoring/score.rb +30 -0
- data/lib/html2rss/scoring.rb +33 -0
- data/lib/html2rss/selectors/post_processors/sanitize_html.rb +10 -1
- data/lib/html2rss/selectors.rb +0 -20
- data/lib/html2rss/sst/attrs.rb +91 -0
- data/lib/html2rss/sst/document.rb +23 -0
- data/lib/html2rss/sst/index.rb +112 -0
- data/lib/html2rss/sst/node.rb +147 -0
- data/lib/html2rss/sst/normalizer.rb +171 -0
- data/lib/html2rss/sst/tags.rb +26 -0
- data/lib/html2rss/sst/text.rb +81 -0
- data/lib/html2rss/sst.rb +8 -0
- data/lib/html2rss/version.rb +1 -1
- data/lib/html2rss.rb +26 -26
- data/schema/html2rss-config.schema.json +20 -0
- metadata +42 -18
- data/lib/html2rss/auto_source/discovery/dom_clustering/group_scorer.rb +0 -80
- data/lib/html2rss/auto_source/discovery/dom_clustering/overlap_resolver.rb +0 -86
- data/lib/html2rss/auto_source/discovery/dom_clustering.rb +0 -119
- data/lib/html2rss/auto_source/discovery/list_candidates.rb +0 -94
- data/lib/html2rss/auto_source/discovery/semantic_anchor_candidates.rb +0 -219
- data/lib/html2rss/auto_source/discovery/semantic_containers.rb +0 -71
- data/lib/html2rss/auto_source/discovery/sitemap.rb +0 -159
- data/lib/html2rss/auto_source/discovery.rb +0 -14
- data/lib/html2rss/auto_source/link_heuristics/anchor_signals.rb +0 -28
- data/lib/html2rss/auto_source/link_heuristics/container_assessor.rb +0 -106
- data/lib/html2rss/auto_source/link_heuristics/container_signals.rb +0 -80
- data/lib/html2rss/auto_source/link_heuristics/destination_facts.rb +0 -42
- data/lib/html2rss/auto_source/link_heuristics/href_extractor.rb +0 -38
- data/lib/html2rss/auto_source/link_heuristics/path_classifier.rb +0 -221
- data/lib/html2rss/auto_source/link_heuristics/text_classifier.rb +0 -66
- data/lib/html2rss/auto_source/link_heuristics.rb +0 -139
|
@@ -1,66 +0,0 @@
|
|
|
1
|
-
# frozen_string_literal: true
|
|
2
|
-
|
|
3
|
-
module Html2rss
|
|
4
|
-
class AutoSource
|
|
5
|
-
class LinkHeuristics
|
|
6
|
-
# Classifies visible anchor text for utility and recommendation chrome.
|
|
7
|
-
class TextClassifier
|
|
8
|
-
# Prefix labels that usually identify navigation or subscription links.
|
|
9
|
-
UTILITY_PREFIX_PATTERN = /
|
|
10
|
-
\A\s*(
|
|
11
|
-
# English
|
|
12
|
-
view\s+all|see\s+all|all\s+news|subscribe|newsletter|comment\s+feed|comments\s+feed|join|premium|plus|
|
|
13
|
-
# German
|
|
14
|
-
alle\s+anzeigen|alle\s+news|abonnieren|newsletter|kommentar\s+feed|mitmachen|
|
|
15
|
-
# Spanish
|
|
16
|
-
ver\s+todos|ver\s+todo|todas\s+las\s+noticias|suscribirse|bolet(i|í)n|comentarios\s+feed|unirse|
|
|
17
|
-
# French
|
|
18
|
-
voir\s+tout|voir\s+tous|toutes\s+les\s+nouvelles|s['’]abonner|flux\s+de\s+commentaires|rejoindre
|
|
19
|
-
)\b
|
|
20
|
-
/ix
|
|
21
|
-
# Short labels that usually identify non-article navigation links.
|
|
22
|
-
UTILITY_PATTERN = /
|
|
23
|
-
\A\s*(
|
|
24
|
-
# English
|
|
25
|
-
about|contact|comments?|join|log\s+in|login|member(ship)?|
|
|
26
|
-
plus|premium|pricing|recommended(\s+for\s+you)?|
|
|
27
|
-
see\s+all|share|sign\s+up|signup|subscribe|view\s+all|
|
|
28
|
-
# German
|
|
29
|
-
(ue|ü)ber(\s+uns)?|kontakt|kommentare?|mitmachen|anmelden|login|
|
|
30
|
-
mitglied(schaft)?|empfohlen(\s+f(ue|ü)r\s+dich)?|alle\s+anzeigen|
|
|
31
|
-
teilen|registrieren|abonnieren|newsletter|
|
|
32
|
-
# Spanish
|
|
33
|
-
sobre(\s+nosotros)?|contacto|comentarios?|unirse|iniciar\s+sesion|
|
|
34
|
-
login|miembro|membres(i|í)a|recomendado(\s+para\s+ti)?|ver\s+todo|
|
|
35
|
-
compartir|registrarse|suscribirse|bolet(i|í)n|
|
|
36
|
-
# French
|
|
37
|
-
(a|à)\s+propos|(a|à)propos|contact|commentaires?|rejoindre|
|
|
38
|
-
se\s+connecter|login|membre|abonnement|recommand(e|é)(\s+pour\s+vous)?|
|
|
39
|
-
voir\s+tout|partager|s['’]inscrire|s['’]abonner|newsletter
|
|
40
|
-
)\b
|
|
41
|
-
/ix
|
|
42
|
-
# Labels for recommendation chrome rather than source articles.
|
|
43
|
-
RECOMMENDED_PATTERN = /
|
|
44
|
-
\A\s*(
|
|
45
|
-
recommended(\s+for\s+you)?|
|
|
46
|
-
empfohlen(\s+f(ue|ü)r\s+dich)?|
|
|
47
|
-
recomendado(\s+para\s+ti)?|
|
|
48
|
-
recommand(e|é)(\s+pour\s+vous)?
|
|
49
|
-
)\b
|
|
50
|
-
/ix
|
|
51
|
-
|
|
52
|
-
# @param text [String, #to_s] visible anchor text
|
|
53
|
-
# @return [Boolean] true when text matches a utility label
|
|
54
|
-
def utility?(text) = text.to_s.match?(UTILITY_PATTERN)
|
|
55
|
-
|
|
56
|
-
# @param text [String, #to_s] visible anchor text
|
|
57
|
-
# @return [Boolean] true when text begins with a utility label
|
|
58
|
-
def utility_prefix?(text) = text.to_s.match?(UTILITY_PREFIX_PATTERN)
|
|
59
|
-
|
|
60
|
-
# @param text [String, #to_s] visible anchor text
|
|
61
|
-
# @return [Boolean] true when text identifies recommendation chrome
|
|
62
|
-
def recommended?(text) = text.to_s.match?(RECOMMENDED_PATTERN)
|
|
63
|
-
end
|
|
64
|
-
end
|
|
65
|
-
end
|
|
66
|
-
end
|
|
@@ -1,139 +0,0 @@
|
|
|
1
|
-
# frozen_string_literal: true
|
|
2
|
-
|
|
3
|
-
module Html2rss
|
|
4
|
-
class AutoSource
|
|
5
|
-
##
|
|
6
|
-
# Shared link eligibility and scoring policy for AutoSource scrapers.
|
|
7
|
-
#
|
|
8
|
-
# Scrapers collect DOM observations; this module owns junk/noise rules and
|
|
9
|
-
# numeric weights so eligibility policy stays in one place.
|
|
10
|
-
class LinkHeuristics
|
|
11
|
-
# @param base_url [String, Html2rss::Url] page URL used to resolve relative hrefs
|
|
12
|
-
def initialize(base_url)
|
|
13
|
-
@base_url = base_url
|
|
14
|
-
@text_classifier = TextClassifier.new
|
|
15
|
-
@container_assessor = ContainerAssessor.new(text_classifier: @text_classifier)
|
|
16
|
-
end
|
|
17
|
-
|
|
18
|
-
# Builds normalized destination facts for an anchor element or href string.
|
|
19
|
-
#
|
|
20
|
-
# @param anchor_or_href [Nokogiri::XML::Element, String, #to_s] anchor element or href-like value
|
|
21
|
-
# @return [DestinationFacts, nil] normalized destination facts, or nil for blank/invalid URLs
|
|
22
|
-
def destination_facts(anchor_or_href)
|
|
23
|
-
return node_facts[anchor_or_href] if node_facts.key?(anchor_or_href)
|
|
24
|
-
|
|
25
|
-
href = HrefExtractor.call(anchor_or_href)
|
|
26
|
-
return unless href
|
|
27
|
-
|
|
28
|
-
res = memoized_destination_facts(href)
|
|
29
|
-
|
|
30
|
-
node_facts[anchor_or_href] = res if anchor_or_href.is_a?(Nokogiri::XML::Node)
|
|
31
|
-
res
|
|
32
|
-
rescue ArgumentError
|
|
33
|
-
nil
|
|
34
|
-
end
|
|
35
|
-
|
|
36
|
-
# @param text [String, #to_s] visible anchor text
|
|
37
|
-
# @return [Boolean] true when text matches a utility label
|
|
38
|
-
def utility_text?(text) = @text_classifier.utility?(text)
|
|
39
|
-
|
|
40
|
-
# @param text [String, #to_s] visible anchor text
|
|
41
|
-
# @return [Boolean] true when text begins with a utility label
|
|
42
|
-
def utility_prefix_text?(text) = @text_classifier.utility_prefix?(text)
|
|
43
|
-
|
|
44
|
-
# @param text [String, #to_s] visible anchor text
|
|
45
|
-
# @return [Boolean] true when text identifies recommendation chrome
|
|
46
|
-
def recommended_text?(text) = @text_classifier.recommended?(text)
|
|
47
|
-
|
|
48
|
-
##
|
|
49
|
-
# Whether an anchor is junk chrome rather than a content permalink.
|
|
50
|
-
#
|
|
51
|
-
# One eligibility home for Html and SemanticHtml: taxonomy/utility text
|
|
52
|
-
# rules plus optional DOM checks (icon-only, utility landmarks).
|
|
53
|
-
#
|
|
54
|
-
# @param text [String, #to_s] visible anchor text
|
|
55
|
-
# @param destination_facts [DestinationFacts, nil] route facts for the href
|
|
56
|
-
# @param anchor [Nokogiri::XML::Node, nil] anchor node for icon/landmark checks
|
|
57
|
-
# @param container [Nokogiri::XML::Node, nil] content container bounding landmark walks
|
|
58
|
-
# @param heading_anchor [Boolean] whether the anchor is the container heading link
|
|
59
|
-
# @return [Boolean] true when the anchor should be ignored
|
|
60
|
-
def noise_anchor?(text:, destination_facts:, anchor: nil, container: nil, heading_anchor: false) # rubocop:disable Metrics/CyclomaticComplexity, Metrics/PerceivedComplexity
|
|
61
|
-
return true unless destination_facts
|
|
62
|
-
|
|
63
|
-
destination_facts.taxonomy_path ||
|
|
64
|
-
short_utility_label?(text, destination_facts) ||
|
|
65
|
-
recommended_chrome?(text, destination_facts, heading_anchor:) ||
|
|
66
|
-
(utility_prefix_text?(text) && destination_facts.high_confidence_utility_destination) ||
|
|
67
|
-
(utility_text?(text) && destination_facts.vanity_path) ||
|
|
68
|
-
utility_text_chrome?(text, destination_facts, heading_anchor:) ||
|
|
69
|
-
icon_only_anchor?(anchor, text) ||
|
|
70
|
-
utility_landmark_ancestor?(anchor, container)
|
|
71
|
-
end
|
|
72
|
-
|
|
73
|
-
##
|
|
74
|
-
# Observes a container and builds ranking signals, including hard-junk.
|
|
75
|
-
#
|
|
76
|
-
# Delegates DOM observation to ContainerAssessor so SemanticHtml only
|
|
77
|
-
# orchestrates candidates and extraction, while ContainerSignals keeps
|
|
78
|
-
# scoring policy.
|
|
79
|
-
#
|
|
80
|
-
# @param container [Nokogiri::XML::Node] semantic container node
|
|
81
|
-
# @param selected_anchor [Nokogiri::XML::Node, nil] primary anchor for the container
|
|
82
|
-
# @param destination_facts [DestinationFacts, nil] route facts for the selected anchor
|
|
83
|
-
# @return [ContainerSignals] observation + scoring signals for the container
|
|
84
|
-
def assess_container(container, selected_anchor, destination_facts:)
|
|
85
|
-
@container_assessor.call(container, selected_anchor, destination_facts:)
|
|
86
|
-
end
|
|
87
|
-
|
|
88
|
-
private
|
|
89
|
-
|
|
90
|
-
def short_utility_label?(text, destination_facts)
|
|
91
|
-
destination_facts.utility_path &&
|
|
92
|
-
!destination_facts.content_path &&
|
|
93
|
-
!destination_facts.strong_post_suffix &&
|
|
94
|
-
text.to_s.scan(/\p{Alnum}+/).size <= 3
|
|
95
|
-
end
|
|
96
|
-
|
|
97
|
-
def recommended_chrome?(text, destination_facts, heading_anchor:)
|
|
98
|
-
# Heading-linked titles may start with "Recommended …" while still being
|
|
99
|
-
# real posts; container hard_junk? owns that call with publish markers.
|
|
100
|
-
!heading_anchor && recommended_text?(text) && destination_facts.shallow
|
|
101
|
-
end
|
|
102
|
-
|
|
103
|
-
def utility_text_chrome?(text, destination_facts, heading_anchor:)
|
|
104
|
-
return false if destination_facts.content_path
|
|
105
|
-
return false unless utility_text?(text)
|
|
106
|
-
|
|
107
|
-
!heading_anchor && !destination_facts.strong_post_suffix
|
|
108
|
-
end
|
|
109
|
-
|
|
110
|
-
def icon_only_anchor?(anchor, text)
|
|
111
|
-
return false unless anchor
|
|
112
|
-
|
|
113
|
-
!text.to_s.match?(/\p{Alnum}/) && !anchor.at_css('img, svg').nil?
|
|
114
|
-
end
|
|
115
|
-
|
|
116
|
-
def utility_landmark_ancestor?(anchor, container)
|
|
117
|
-
return false unless anchor && container
|
|
118
|
-
|
|
119
|
-
condition = lambda { |node|
|
|
120
|
-
node == container || Html2rss::Html::Navigator::UTILITY_LANDMARK_TAGS.include?(node.name)
|
|
121
|
-
}
|
|
122
|
-
landmark = Html2rss::Html::Navigator.parent_until_condition(anchor.parent, condition)
|
|
123
|
-
|
|
124
|
-
!landmark.nil? && landmark != container
|
|
125
|
-
end
|
|
126
|
-
|
|
127
|
-
def node_facts
|
|
128
|
-
@node_facts ||= {}.compare_by_identity
|
|
129
|
-
end
|
|
130
|
-
|
|
131
|
-
def memoized_destination_facts(href)
|
|
132
|
-
(@destination_facts ||= {})[href] ||= begin
|
|
133
|
-
url = Html2rss::Url.from_relative(href, @base_url)
|
|
134
|
-
DestinationFacts.build(url)
|
|
135
|
-
end
|
|
136
|
-
end
|
|
137
|
-
end
|
|
138
|
-
end
|
|
139
|
-
end
|