html2rss 0.24.0 → 0.25.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. checksums.yaml +4 -4
  2. data/lib/html2rss/auto_source/scraper/html.rb +68 -145
  3. data/lib/html2rss/auto_source/scraper/json_state/article_normalizer.rb +116 -0
  4. data/lib/html2rss/auto_source/scraper/json_state/candidate_detector.rb +63 -0
  5. data/lib/html2rss/auto_source/scraper/json_state/document_scanner.rb +179 -0
  6. data/lib/html2rss/auto_source/scraper/json_state/value_finder.rb +55 -0
  7. data/lib/html2rss/auto_source/scraper/json_state.rb +0 -379
  8. data/lib/html2rss/auto_source/scraper/semantic_html/entry_deduplicator.rb +35 -30
  9. data/lib/html2rss/auto_source/scraper/semantic_html.rb +40 -120
  10. data/lib/html2rss/auto_source/scraper/sitemap/parser.rb +161 -0
  11. data/lib/html2rss/auto_source/scraper/sitemap.rb +10 -7
  12. data/lib/html2rss/auto_source/scraper/wordpress_api.rb +6 -2
  13. data/lib/html2rss/auto_source/scraper.rb +81 -44
  14. data/lib/html2rss/auto_source/segment.rb +29 -0
  15. data/lib/html2rss/auto_source/segmenter/cluster.rb +163 -0
  16. data/lib/html2rss/auto_source/segmenter/list.rb +101 -0
  17. data/lib/html2rss/auto_source/segmenter/primary_link.rb +81 -0
  18. data/lib/html2rss/auto_source/segmenter/semantic.rb +104 -0
  19. data/lib/html2rss/auto_source/segmenter.rb +66 -0
  20. data/lib/html2rss/auto_source.rb +74 -21
  21. data/lib/html2rss/cli.rb +8 -13
  22. data/lib/html2rss/config/auto_source_contract.rb +2 -0
  23. data/lib/html2rss/config/request_controls.rb +33 -0
  24. data/lib/html2rss/config/validator.rb +15 -8
  25. data/lib/html2rss/config.rb +6 -2
  26. data/lib/html2rss/feed_builder/rss.rb +17 -5
  27. data/lib/html2rss/feed_pipeline.rb +4 -0
  28. data/lib/html2rss/feed_result.rb +1 -1
  29. data/lib/html2rss/html/article_extractor/category_extractor.rb +10 -36
  30. data/lib/html2rss/html/article_extractor/date_extractor.rb +2 -7
  31. data/lib/html2rss/html/article_extractor/enclosure_extractor.rb +5 -50
  32. data/lib/html2rss/html/article_extractor/image_extractor.rb +5 -16
  33. data/lib/html2rss/html/article_rules/category.rb +55 -0
  34. data/lib/html2rss/html/article_rules/date.rb +25 -0
  35. data/lib/html2rss/html/article_rules/enclosure.rb +72 -0
  36. data/lib/html2rss/html/article_rules/image.rb +109 -0
  37. data/lib/html2rss/html/article_rules.rb +10 -0
  38. data/lib/html2rss/html/rendering/audio_renderer.rb +2 -12
  39. data/lib/html2rss/html/rendering/escaped_attributes.rb +27 -0
  40. data/lib/html2rss/html/rendering/image_renderer.rb +2 -12
  41. data/lib/html2rss/html/rendering/pdf_renderer.rb +2 -8
  42. data/lib/html2rss/html/rendering/video_renderer.rb +2 -12
  43. data/lib/html2rss/html/sst_article_extractor.rb +279 -0
  44. data/lib/html2rss/link_destination/destination_facts.rb +40 -0
  45. data/lib/html2rss/link_destination/noise_policy.rb +79 -0
  46. data/lib/html2rss/link_destination/path_classifier.rb +205 -0
  47. data/lib/html2rss/link_destination/text_classifier.rb +64 -0
  48. data/lib/html2rss/link_destination.rb +8 -0
  49. data/lib/html2rss/request_service/faraday_strategy.rb +42 -1
  50. data/lib/html2rss/request_service/network_guard.rb +5 -3
  51. data/lib/html2rss/scoring/anchor_score.rb +34 -0
  52. data/lib/html2rss/scoring/cluster_scorer.rb +87 -0
  53. data/lib/html2rss/scoring/container_assessor.rb +83 -0
  54. data/lib/html2rss/scoring/engine.rb +101 -0
  55. data/lib/html2rss/scoring/link_resolver.rb +72 -0
  56. data/lib/html2rss/scoring/observation.rb +53 -0
  57. data/lib/html2rss/scoring/ranked_segment.rb +45 -0
  58. data/lib/html2rss/scoring/score.rb +30 -0
  59. data/lib/html2rss/scoring.rb +33 -0
  60. data/lib/html2rss/selectors/post_processors/sanitize_html.rb +10 -1
  61. data/lib/html2rss/selectors.rb +0 -20
  62. data/lib/html2rss/sst/attrs.rb +91 -0
  63. data/lib/html2rss/sst/document.rb +23 -0
  64. data/lib/html2rss/sst/index.rb +112 -0
  65. data/lib/html2rss/sst/node.rb +147 -0
  66. data/lib/html2rss/sst/normalizer.rb +171 -0
  67. data/lib/html2rss/sst/tags.rb +26 -0
  68. data/lib/html2rss/sst/text.rb +81 -0
  69. data/lib/html2rss/sst.rb +8 -0
  70. data/lib/html2rss/version.rb +1 -1
  71. data/lib/html2rss.rb +26 -26
  72. data/schema/html2rss-config.schema.json +20 -0
  73. metadata +42 -18
  74. data/lib/html2rss/auto_source/discovery/dom_clustering/group_scorer.rb +0 -80
  75. data/lib/html2rss/auto_source/discovery/dom_clustering/overlap_resolver.rb +0 -86
  76. data/lib/html2rss/auto_source/discovery/dom_clustering.rb +0 -119
  77. data/lib/html2rss/auto_source/discovery/list_candidates.rb +0 -94
  78. data/lib/html2rss/auto_source/discovery/semantic_anchor_candidates.rb +0 -219
  79. data/lib/html2rss/auto_source/discovery/semantic_containers.rb +0 -71
  80. data/lib/html2rss/auto_source/discovery/sitemap.rb +0 -159
  81. data/lib/html2rss/auto_source/discovery.rb +0 -14
  82. data/lib/html2rss/auto_source/link_heuristics/anchor_signals.rb +0 -28
  83. data/lib/html2rss/auto_source/link_heuristics/container_assessor.rb +0 -106
  84. data/lib/html2rss/auto_source/link_heuristics/container_signals.rb +0 -80
  85. data/lib/html2rss/auto_source/link_heuristics/destination_facts.rb +0 -42
  86. data/lib/html2rss/auto_source/link_heuristics/href_extractor.rb +0 -38
  87. data/lib/html2rss/auto_source/link_heuristics/path_classifier.rb +0 -221
  88. data/lib/html2rss/auto_source/link_heuristics/text_classifier.rb +0 -66
  89. data/lib/html2rss/auto_source/link_heuristics.rb +0 -139
@@ -1,159 +0,0 @@
1
- # frozen_string_literal: true
2
-
3
- require 'date'
4
- require 'nokogiri'
5
-
6
- module Html2rss
7
- class AutoSource
8
- module Discovery
9
- ##
10
- # Parses XML sitemap documents and extracts prioritized, recent article entries.
11
- class Sitemap
12
- # Default minimum priority threshold (0.0 to 1.0)
13
- DEFAULT_MIN_PRIORITY = 0.3
14
- # Default maximum age in days for sitemap entries
15
- DEFAULT_MAX_AGE_DAYS = 30
16
- # Google News sitemap XML namespace
17
- NEWS_NS = 'http://www.google.com/schemas/sitemap-news/0.9'
18
-
19
- # Result value object returned by .call
20
- # @!attribute entries [Array<SitemapEntry>] parsed entries (empty for sitemapindex docs)
21
- # @!attribute sub_sitemap_urls [Array<String>] child sitemap URLs (empty for urlset docs)
22
- Result = Data.define(:entries, :sub_sitemap_urls)
23
-
24
- # Immutable facts for one sitemap entry
25
- SitemapEntry = Data.define(:url, :title, :published_at, :priority, :changefreq)
26
-
27
- class << self
28
- ##
29
- # Parses sitemap XML body and returns a Result.
30
- #
31
- # @param xml_content [String] sitemap XML body string
32
- # @param min_priority [Float] minimum priority score (0.0..1.0)
33
- # @param max_age_days [Integer] maximum age in days for entries
34
- # @return [Result] result with entries and/or sub_sitemap_urls
35
- def call(xml_content, min_priority: DEFAULT_MIN_PRIORITY, max_age_days: DEFAULT_MAX_AGE_DAYS)
36
- new(xml_content, min_priority:, max_age_days:).call
37
- end
38
- end
39
-
40
- ##
41
- # @param xml_content [String]
42
- # @param min_priority [Float]
43
- # @param max_age_days [Integer]
44
- def initialize(xml_content, min_priority: DEFAULT_MIN_PRIORITY, max_age_days: DEFAULT_MAX_AGE_DAYS)
45
- @xml_content = xml_content
46
- @min_priority = min_priority
47
- @max_age_days = max_age_days
48
- end
49
-
50
- ##
51
- # @return [Result]
52
- def call
53
- document = Nokogiri::XML(xml_content)
54
- return Result.new(entries: [], sub_sitemap_urls: []) if invalid_xml?(document)
55
-
56
- if sitemapindex?(document)
57
- Result.new(entries: [], sub_sitemap_urls: extract_sub_sitemap_urls(document))
58
- else
59
- Result.new(entries: extract_entries(document), sub_sitemap_urls: [])
60
- end
61
- end
62
-
63
- private
64
-
65
- attr_reader :xml_content, :min_priority, :max_age_days
66
-
67
- def invalid_xml?(document)
68
- document.errors.any? && document.at_css('urlset, sitemapindex').nil?
69
- end
70
-
71
- def sitemapindex?(document)
72
- !document.at_xpath('//*[local-name()="sitemapindex"]').nil?
73
- end
74
-
75
- def extract_sub_sitemap_urls(document)
76
- document.xpath('//*[local-name()="sitemap"]/*[local-name()="loc"]').filter_map do |node|
77
- text = node.text.strip
78
- text unless text.empty?
79
- end
80
- end
81
-
82
- def extract_entries(document)
83
- document.xpath('//*[local-name()="url"]').filter_map { build_entry(_1) }
84
- end
85
-
86
- def build_entry(url_node)
87
- loc = parse_loc(url_node)
88
- return unless loc
89
- return if invalid_entry_signals?(url_node)
90
-
91
- published_at = parse_date(url_node)
92
- return if stale_date?(published_at)
93
-
94
- build_entry_facts(url_node, loc, published_at)
95
- end
96
-
97
- def build_entry_facts(url_node, loc, published_at)
98
- SitemapEntry.new(
99
- url: loc,
100
- title: parse_title(url_node),
101
- published_at:,
102
- priority: parse_priority(url_node),
103
- changefreq: parse_changefreq(url_node)
104
- )
105
- end
106
-
107
- def parse_loc(url_node)
108
- node = url_node.at_xpath('./*[local-name()="loc"]')
109
- text = node&.text&.strip
110
- text unless text.to_s.empty?
111
- end
112
-
113
- def parse_changefreq(url_node)
114
- node = url_node.at_xpath('./*[local-name()="changefreq"]')
115
- node ? node.text.strip.downcase : nil
116
- end
117
-
118
- def invalid_entry_signals?(url_node)
119
- parse_priority(url_node) < min_priority || parse_changefreq(url_node) == 'never'
120
- end
121
-
122
- def parse_priority(url_node)
123
- node = url_node.at_xpath('./*[local-name()="priority"]')
124
- text = node&.text&.strip
125
- text ? text.to_f : 0.5
126
- end
127
-
128
- def parse_date(url_node)
129
- raw_date = raw_date_from(url_node)
130
- return unless raw_date
131
-
132
- Time.parse(raw_date).utc.iso8601
133
- rescue ArgumentError
134
- nil
135
- end
136
-
137
- def raw_date_from(url_node)
138
- news_node = url_node.at_xpath('.//news:publication_date', 'news' => NEWS_NS)
139
- lastmod_node = url_node.at_xpath('./*[local-name()="lastmod"]')
140
- (news_node || lastmod_node)&.text&.strip
141
- end
142
-
143
- def parse_title(url_node)
144
- node = url_node.at_xpath('.//news:title', 'news' => NEWS_NS)
145
- node&.text&.strip
146
- end
147
-
148
- def stale_date?(published_at)
149
- return false unless published_at
150
-
151
- time = Time.parse(published_at)
152
- (Time.now.utc - time) > (max_age_days * 86_400)
153
- rescue ArgumentError
154
- false
155
- end
156
- end
157
- end
158
- end
159
- end
@@ -1,14 +0,0 @@
1
- # frozen_string_literal: true
2
-
3
- module Html2rss
4
- class AutoSource
5
- ##
6
- # Card/list discovery helpers used by AutoSource scrapers before field extraction.
7
- #
8
- # Html2rss::Html::ArticleExtractor fills article fields from a container; discovery finds those containers.
9
- #
10
- # DOM list discovery for anchorless or classless pages is owned by {Discovery::DomClustering}.
11
- module Discovery
12
- end
13
- end
14
- end
@@ -1,28 +0,0 @@
1
- # frozen_string_literal: true
2
-
3
- module Html2rss
4
- class AutoSource
5
- class LinkHeuristics
6
- # Score weights keyed by AnchorSignals member name.
7
- ANCHOR_SCORE_RULES = {
8
- heading_anchor: 100,
9
- heading_text_match: 20,
10
- meaningful_text: 10,
11
- content_like_destination: 10
12
- }.freeze
13
-
14
- # Anchor ranking signals used by semantic primary-link selection.
15
- AnchorSignals = Data.define(
16
- :heading_anchor,
17
- :heading_text_match,
18
- :meaningful_text,
19
- :content_like_destination
20
- ) do
21
- # @return [Integer] ranking score for one eligible anchor
22
- def score
23
- ANCHOR_SCORE_RULES.sum { |signal, weight| public_send(signal) ? weight : 0 }
24
- end
25
- end
26
- end
27
- end
28
- end
@@ -1,106 +0,0 @@
1
- # frozen_string_literal: true
2
-
3
- module Html2rss
4
- class AutoSource
5
- class LinkHeuristics
6
- ##
7
- # Turns container DOM observations into ContainerSignals inputs.
8
- #
9
- # Owns publish markers, headings, visible text, and class/id tokens so
10
- # LinkHeuristics can keep eligibility policy and ContainerSignals can
11
- # keep scoring without re-reading the DOM.
12
- class ContainerAssessor
13
- # Token pattern for content-like class/id markers on containers.
14
- CONTENT_TOKEN_REGEXP = begin
15
- words = PathClassifier::SEGMENT_SETS.fetch(:content)
16
- /(?:^|\s|[-_])(#{Regexp.union(words.to_a).source})(?:\s|[-_]|$)/i
17
- end.freeze
18
-
19
- # Token pattern for junk/utility class/id markers on containers.
20
- JUNK_TOKEN_REGEXP = begin
21
- words = PathClassifier::SEGMENT_SETS.fetch(:utility)
22
- /(?:^|\s|[-_])(#{Regexp.union(words.to_a).source})(?:\s|[-_]|$)/i
23
- end.freeze
24
-
25
- # @param text_classifier [TextClassifier] shared utility/recommended text policy
26
- def initialize(text_classifier:)
27
- @text_classifier = text_classifier
28
- end
29
-
30
- ##
31
- # Observes a container and builds ranking signals, including hard-junk.
32
- #
33
- # @param container [Nokogiri::XML::Node] semantic container node
34
- # @param selected_anchor [Nokogiri::XML::Node, nil] primary anchor for the container
35
- # @param destination_facts [DestinationFacts, nil] route facts for the selected anchor
36
- # @return [ContainerSignals] observation + scoring signals for the container
37
- # rubocop:disable Metrics/AbcSize, Metrics/MethodLength
38
- def call(container, selected_anchor, destination_facts:)
39
- title = entry_title(container, selected_anchor)
40
- tokens = container_tokens(container)
41
-
42
- ContainerSignals.new(
43
- title_word_count: word_count(title),
44
- path_length: destination_facts&.url&.path.to_s.length,
45
- content_path: destination_facts&.content_path,
46
- publish_marker: publish_marker?(container),
47
- descriptive_context: descriptive_context?(visible_text(container), title),
48
- article_container: container.name == 'article',
49
- content_tokens: content_tokens?(tokens),
50
- junk_tokens: junk_tokens?(tokens),
51
- utility_prefix_title: @text_classifier.utility_prefix?(title),
52
- recommended_title: @text_classifier.recommended?(title),
53
- utility_path: destination_facts&.utility_path,
54
- strong_post_suffix: destination_facts&.strong_post_suffix,
55
- shallow: destination_facts&.shallow,
56
- high_confidence_junk_path: destination_facts&.high_confidence_junk_path,
57
- high_confidence_utility_destination: destination_facts&.high_confidence_utility_destination,
58
- selected_anchor_present: !selected_anchor.nil?
59
- )
60
- end
61
- # rubocop:enable Metrics/AbcSize, Metrics/MethodLength
62
-
63
- private
64
-
65
- def publish_marker?(container)
66
- (@publish_markers ||= {}.compare_by_identity)[container] ||=
67
- !!container.at_css('time, [datetime], [itemprop="datePublished"], [itemprop="dateModified"]')
68
- end
69
-
70
- def descriptive_context?(container_text, title)
71
- snippet = container_text.to_s.sub(/\A#{Regexp.escape(title.to_s)}/i, '')
72
- snippet.length > 30 && word_count(snippet) >= 8
73
- end
74
-
75
- def heading_for(container)
76
- (@headings ||= {}.compare_by_identity)[container] ||=
77
- container.at_css(Html2rss::Html::Navigator::HEADING_TAGS.join(','))
78
- end
79
-
80
- def visible_text(node)
81
- return '' unless node
82
-
83
- (@visible_texts ||= {}.compare_by_identity)[node] ||= Html2rss::Html::Navigator.extract_visible_text(node).to_s.strip
84
- end
85
-
86
- def entry_title(container, selected_anchor) = visible_text(heading_for(container) || selected_anchor)
87
-
88
- def word_count(text)
89
- (@word_counts ||= {})[text] ||= begin
90
- count = 0
91
- text.to_s.scan(/\p{Alnum}+/) { count += 1 }
92
- count
93
- end
94
- end
95
-
96
- def container_tokens(container)
97
- (@container_tokens ||= {}.compare_by_identity)[container] ||= "#{container['class']} #{container['id']}"
98
- end
99
-
100
- def content_tokens?(tokens) = tokens.to_s.match?(CONTENT_TOKEN_REGEXP)
101
-
102
- def junk_tokens?(tokens) = tokens.to_s.match?(JUNK_TOKEN_REGEXP)
103
- end
104
- end
105
- end
106
- end
@@ -1,80 +0,0 @@
1
- # frozen_string_literal: true
2
-
3
- module Html2rss
4
- class AutoSource
5
- class LinkHeuristics
6
- # Container observations used to compute quality/junk scores.
7
- ContainerSignals = Data.define(
8
- :title_word_count,
9
- :path_length,
10
- :content_path,
11
- :publish_marker,
12
- :descriptive_context,
13
- :article_container,
14
- :content_tokens,
15
- :junk_tokens,
16
- :utility_prefix_title,
17
- :recommended_title,
18
- :utility_path,
19
- :strong_post_suffix,
20
- :shallow,
21
- :high_confidence_junk_path,
22
- :high_confidence_utility_destination,
23
- :selected_anchor_present
24
- ) do
25
- # @return [Integer] positive quality contribution for ranking
26
- def quality_score # rubocop:disable Metrics/CyclomaticComplexity, Metrics/PerceivedComplexity
27
- score = 0
28
- score += 40 if title_word_count >= 3
29
- score += 15 if title_word_count >= 7
30
- score += 20 if path_length > 6
31
- score += 15 if content_path
32
- score += 15 if publish_marker
33
- score += 10 if descriptive_context
34
- score += 10 if article_container
35
- score += 10 if content_tokens
36
- score
37
- end
38
-
39
- # @return [Integer] junk penalty subtracted from quality
40
- def junk_score # rubocop:disable Metrics/CyclomaticComplexity, Metrics/PerceivedComplexity
41
- score = 0
42
- score += 25 if non_content_utility_path?
43
- score += 15 if utility_prefix_title && title_word_count <= 6
44
- score += 10 if shallow
45
- score += 10 if weak_container?
46
- score += 10 if recommended_title && !content_path
47
- score += 5 if high_confidence_junk_path
48
- score += 15 if junk_tokens
49
- score
50
- end
51
-
52
- # @return [Integer] quality minus junk for stable ranking
53
- def final_score = quality_score - junk_score
54
-
55
- # @return [Boolean] true when the entry should be dropped before ranking
56
- def hard_junk? # rubocop:disable Metrics/CyclomaticComplexity, Metrics/PerceivedComplexity
57
- weak = weak_article_candidate?
58
-
59
- high_confidence_junk_path ||
60
- (selected_anchor_present && recommended_title && shallow && weak) ||
61
- (selected_anchor_present && utility_prefix_title &&
62
- high_confidence_utility_destination && weak)
63
- end
64
-
65
- private
66
-
67
- # @return [Boolean] true when article evidence is too weak to keep
68
- def weak_article_candidate?
69
- [article_container, publish_marker, descriptive_context, content_path].count(&:itself) < 2
70
- end
71
-
72
- def non_content_utility_path?
73
- utility_path && !content_path && !strong_post_suffix
74
- end
75
-
76
- def weak_container? = !publish_marker && !descriptive_context
77
- end
78
- end
79
- end
80
- end
@@ -1,42 +0,0 @@
1
- # frozen_string_literal: true
2
-
3
- module Html2rss
4
- class AutoSource
5
- class LinkHeuristics
6
- # Normalized URL plus reusable route-classification facts for one link.
7
- DestinationFacts = Data.define(
8
- :url,
9
- :destination,
10
- :segments,
11
- :content_path,
12
- :utility_path,
13
- :taxonomy_path,
14
- :vanity_path,
15
- :shallow,
16
- :strong_post_suffix,
17
- :high_confidence_junk_path,
18
- :high_confidence_utility_destination
19
- ) do
20
- # @param url [Html2rss::Url] normalized destination URL
21
- # @return [DestinationFacts] route facts for downstream link scoring
22
- def self.build(url) # rubocop:disable Metrics/MethodLength
23
- classifier = PathClassifier.new(url.path_segments)
24
-
25
- new(
26
- url:,
27
- destination: url.to_s,
28
- segments: classifier.segments,
29
- strong_post_suffix: classifier.strong_post_suffix?,
30
- content_path: classifier.content_path?,
31
- utility_path: classifier.utility_path?,
32
- taxonomy_path: classifier.taxonomy_path?,
33
- vanity_path: classifier.vanity_path?,
34
- shallow: classifier.shallow?,
35
- high_confidence_junk_path: classifier.junk_path?,
36
- high_confidence_utility_destination: classifier.utility_destination?
37
- )
38
- end
39
- end
40
- end
41
- end
42
- end
@@ -1,38 +0,0 @@
1
- # frozen_string_literal: true
2
-
3
- module Html2rss
4
- class AutoSource
5
- class LinkHeuristics
6
- # Extracts a normalized href from a Nokogiri anchor or raw href value.
7
- class HrefExtractor
8
- # Regexp to capture everything before the first '#'
9
- HREF_BASE_PATTERN = /\A([^#]*)/
10
-
11
- # @param anchor_or_href [Nokogiri::XML::Element, String, #to_s] anchor element or href-like value
12
- # @return [String, nil] href without fragment, or nil when blank
13
- def self.call(anchor_or_href) = new(anchor_or_href).call
14
-
15
- # @param anchor_or_href [Nokogiri::XML::Element, String, #to_s] anchor element or href-like value
16
- def initialize(anchor_or_href)
17
- @anchor_or_href = anchor_or_href
18
- end
19
-
20
- # @return [String, nil] href without fragment, or nil when blank
21
- def call
22
- href = case @anchor_or_href
23
- when Nokogiri::XML::Node
24
- @anchor_or_href['href']
25
- else
26
- @anchor_or_href
27
- end
28
-
29
- return unless href
30
-
31
- # Extract base part before # and strip whitespace
32
- base = href.to_s[HREF_BASE_PATTERN, 1].strip
33
- base unless base.empty?
34
- end
35
- end
36
- end
37
- end
38
- end
@@ -1,221 +0,0 @@
1
- # frozen_string_literal: true
2
-
3
- module Html2rss
4
- class AutoSource
5
- class LinkHeuristics
6
- # Classifies normalized destination path segments for scoring.
7
- class PathClassifier # rubocop:disable Metrics/ClassLength
8
- attr_reader :segments
9
-
10
- # Segment groups used to classify article, taxonomy, utility, and vanity routes.
11
- SEGMENT_SETS = {
12
- content: %w[
13
- article articles blog blogs changelog changelogs insight insights
14
- launch launches news post posts release releases story stories update updates
15
- artikel beitrag beitraege nachrichten neuigkeiten aktuelles
16
- articulo articulos noticia noticias entrada entradas publicacion publicaciones
17
- actualite actualites nouvelle nouvelles
18
- teaser teasers card cards
19
- ].to_set.freeze,
20
- utility: %w[
21
- about account archive archives author authors category categories comment comments
22
- contact feedback help login logout newsletter newsletters notification notifications
23
- preference preferences profile register search settings share signup subscribe
24
- tag tags topic topics
25
- feed feeds comment-feed comments-feed
26
- recommended
27
- for-you
28
- privacy terms cookie cookies
29
- join member members membership plus premium plans pricing user users
30
- kategorie kategorien schlagwort schlagworte thema themen autor autoren archiv
31
- ueber-uns ueber ueberuns profil kontakt impressum suche hilfe anmelden registrieren
32
- konto registrierung anmeldung abonnieren abo datenschutz nutzungsbedingungen agb
33
- categoria categorias etiqueta etiquetas tema temas autores archivos
34
- sobre-nosotros sobre quienes-somos buscar busqueda ayuda entrar ingresar
35
- registrarse registro cuenta suscribirse boletin privacidad condiciones
36
- categorie etiquette etiquettes sujet sujets theme themes auteur auteurs
37
- a-propos apropos recherche rechercher aide connexion s-inscrire
38
- sinscrire inscription compte s-abonner saboner lettre-information confidentialite mentions-legales cgu
39
- menu sidebar widget social modal popup banner promo ad ads
40
- related recommendation recommendations pagination pager
41
- ].to_set.freeze,
42
- high_confidence_junk: %w[
43
- about account archive archives author authors category categories comment comments
44
- contact cookie cookies feedback feed feeds help login logout notification notifications
45
- preference preferences privacy profile register search settings share signup subscribe
46
- tag tags terms topic topics comment-feed comments-feed user users
47
- kategorie kategorien schlagwort schlagworte thema themen autor autoren archiv
48
- ueber-uns ueber ueberuns profil kontakt impressum suche hilfe anmelden registrieren
49
- konto registrierung anmeldung abonnieren abo datenschutz nutzungsbedingungen agb
50
- categoria categorias etiqueta etiquetas tema temas autores archivos
51
- sobre-nosotros sobre quienes-somos buscar busqueda ayuda entrar ingresar
52
- registrarse registro cuenta suscribirse boletin privacidad condiciones
53
- categorie etiquette etiquettes sujet sujets theme themes auteur auteurs
54
- a-propos apropos recherche rechercher aide connexion s-inscrire
55
- sinscrire inscription compte s-abonner saboner lettre-information confidentialite mentions-legales cgu
56
- menu sidebar widget social modal popup banner promo ad ads
57
- related recommendation recommendations pagination pager
58
- ].to_set.freeze,
59
- taxonomy: %w[
60
- category categories tag tags topic topics
61
- kategorie kategorien schlagwort schlagworte thema themen
62
- categoria categorias etiqueta etiquetas tema temas
63
- categorie etiquette etiquettes sujet sujets theme themes
64
- ].to_set.freeze,
65
- vanity: %w[
66
- join membership plus premium pricing plans subscribe signup
67
- abonnieren abo
68
- suscribirse boletin
69
- s-abonner saboner
70
- ].to_set.freeze,
71
- deep_post_context: %w[
72
- press newsroom
73
- presse pressemitteilungen
74
- prensa
75
- ].to_set.freeze
76
- }.freeze
77
- # Path segment that begins with a year-like publishing marker.
78
- YEARISH_SEGMENT = /\A\d{4,}[\w-]*\z/
79
- # Hyphenated slug shape common to article permalinks.
80
- POST_SLUG_SEGMENT = /\A[a-z0-9]+(?:-[a-z0-9]+){2,}\z/i
81
-
82
- # @param segments [Array<String>] normalized URL path segments
83
- def initialize(segments)
84
- @segments = segments
85
- end
86
-
87
- # @return [Boolean] true when the route has article-like path evidence
88
- def content_path?
89
- @content_path ||= SEGMENT_SETS[:content].intersect?(segments) ||
90
- yearish_content_context?
91
- end
92
-
93
- # @return [Boolean] true when the route includes utility/navigation evidence
94
- def utility_path?
95
- @utility_path ||= SEGMENT_SETS[:utility].intersect?(segments)
96
- end
97
-
98
- # @return [Boolean] true when the route points at conversion or account chrome
99
- def vanity_path?
100
- @vanity_path ||= SEGMENT_SETS[:vanity].intersect?(segments)
101
- end
102
-
103
- # @return [Boolean] true when the route points at taxonomy/listing chrome
104
- def taxonomy_path?
105
- @taxonomy_path ||= SEGMENT_SETS[:taxonomy].intersect?(segments)
106
- end
107
-
108
- # @return [Boolean] true when the route is too shallow to strongly indicate an article
109
- def shallow?
110
- segment_count = segments.size
111
- junk_segments = SEGMENT_SETS.fetch(:high_confidence_junk)
112
-
113
- segment_count <= 1 || (segment_count == 2 && junk_segments.include?(segments.last))
114
- end
115
-
116
- # @return [Boolean] true when the final path segment looks like a post slug
117
- def strong_post_suffix?
118
- @strong_post_suffix ||= segments.any? &&
119
- included_last_segment? &&
120
- trusted_post_context?(segments.size - 1)
121
- end
122
-
123
- # @return [Boolean] true when every path segment is utility chrome
124
- def utility_only_route?
125
- junk_segments = SEGMENT_SETS.fetch(:high_confidence_junk)
126
-
127
- segments.all? { |segment| junk_segments.include?(segment) }
128
- end
129
-
130
- # @return [Boolean] true when the route is shallow and contains high-confidence noise
131
- def shallow_high_confidence_route?
132
- junk_segments = SEGMENT_SETS.fetch(:high_confidence_junk)
133
- vanity_segments = SEGMENT_SETS.fetch(:vanity)
134
-
135
- shallow? && segments.any? do |segment|
136
- junk_segments.include?(segment) || vanity_segments.include?(segment)
137
- end
138
- end
139
-
140
- # @return [Boolean] true when the leading segments are all utility chrome
141
- def deep_utility_context_route?
142
- all_junk?(segments.size - 1)
143
- end
144
-
145
- # @return [Boolean] true when the route is shallow and contains high-confidence noise
146
- def junk_path?
147
- return false if excluded_content_route?
148
-
149
- taxonomy_path? ||
150
- utility_only_route? ||
151
- deep_utility_context_route? ||
152
- shallow_high_confidence_route?
153
- end
154
-
155
- # @return [Boolean] true when the route points at conversion or account chrome
156
- def utility_destination?
157
- return false if excluded_content_route?
158
-
159
- vanity_path? || utility_route?
160
- end
161
-
162
- private
163
-
164
- def yearish_content_context?
165
- segments.any? { |segment| segment.match?(YEARISH_SEGMENT) } &&
166
- (strong_post_suffix? || trusted_post_context?(segments.size - 1))
167
- end
168
-
169
- def excluded_content_route?
170
- segments.empty? || content_path? || strong_post_suffix?
171
- end
172
-
173
- def utility_route?
174
- taxonomy_path? ||
175
- utility_only_route? ||
176
- deep_utility_context_route? ||
177
- shallow_utility_route?
178
- end
179
-
180
- def shallow_utility_route?
181
- shallow? && utility_path?
182
- end
183
-
184
- def all_junk?(limit)
185
- return false if limit <= 0
186
-
187
- junk_segments = SEGMENT_SETS.fetch(:high_confidence_junk)
188
- (0...limit).all? { |i| junk_segments.include?(segments[i]) }
189
- end
190
-
191
- def trusted_post_context?(limit)
192
- return false if limit <= 0
193
-
194
- content_segments = SEGMENT_SETS.fetch(:content)
195
- context_segments = SEGMENT_SETS.fetch(:deep_post_context)
196
-
197
- (0...limit).any? do |i|
198
- segment = segments[i]
199
- content_segments.include?(segment) ||
200
- segment.match?(PathClassifier::YEARISH_SEGMENT) ||
201
- context_segments.include?(segment)
202
- end
203
- end
204
-
205
- def included_last_segment?
206
- !excluded_last_segment? && slug_last_segment?
207
- end
208
-
209
- def excluded_last_segment?
210
- last = segments.last
211
- [SEGMENT_SETS[:high_confidence_junk], SEGMENT_SETS[:vanity]].any? { |set| set.include?(last) }
212
- end
213
-
214
- def slug_last_segment?
215
- last = segments.last
216
- last.match?(YEARISH_SEGMENT) || last.match?(POST_SLUG_SEGMENT)
217
- end
218
- end
219
- end
220
- end
221
- end