html2rss 0.24.0 → 0.25.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. checksums.yaml +4 -4
  2. data/lib/html2rss/auto_source/scraper/html.rb +68 -145
  3. data/lib/html2rss/auto_source/scraper/json_state/article_normalizer.rb +116 -0
  4. data/lib/html2rss/auto_source/scraper/json_state/candidate_detector.rb +63 -0
  5. data/lib/html2rss/auto_source/scraper/json_state/document_scanner.rb +179 -0
  6. data/lib/html2rss/auto_source/scraper/json_state/value_finder.rb +55 -0
  7. data/lib/html2rss/auto_source/scraper/json_state.rb +0 -379
  8. data/lib/html2rss/auto_source/scraper/semantic_html/entry_deduplicator.rb +35 -30
  9. data/lib/html2rss/auto_source/scraper/semantic_html.rb +40 -120
  10. data/lib/html2rss/auto_source/scraper/sitemap/parser.rb +161 -0
  11. data/lib/html2rss/auto_source/scraper/sitemap.rb +10 -7
  12. data/lib/html2rss/auto_source/scraper/wordpress_api.rb +6 -2
  13. data/lib/html2rss/auto_source/scraper.rb +81 -44
  14. data/lib/html2rss/auto_source/segment.rb +29 -0
  15. data/lib/html2rss/auto_source/segmenter/cluster.rb +163 -0
  16. data/lib/html2rss/auto_source/segmenter/list.rb +101 -0
  17. data/lib/html2rss/auto_source/segmenter/primary_link.rb +81 -0
  18. data/lib/html2rss/auto_source/segmenter/semantic.rb +104 -0
  19. data/lib/html2rss/auto_source/segmenter.rb +66 -0
  20. data/lib/html2rss/auto_source.rb +74 -21
  21. data/lib/html2rss/cli.rb +8 -13
  22. data/lib/html2rss/config/auto_source_contract.rb +2 -0
  23. data/lib/html2rss/config/request_controls.rb +33 -0
  24. data/lib/html2rss/config/validator.rb +15 -8
  25. data/lib/html2rss/config.rb +6 -2
  26. data/lib/html2rss/feed_builder/rss.rb +17 -5
  27. data/lib/html2rss/feed_pipeline.rb +4 -0
  28. data/lib/html2rss/feed_result.rb +1 -1
  29. data/lib/html2rss/html/article_extractor/category_extractor.rb +10 -36
  30. data/lib/html2rss/html/article_extractor/date_extractor.rb +2 -7
  31. data/lib/html2rss/html/article_extractor/enclosure_extractor.rb +5 -50
  32. data/lib/html2rss/html/article_extractor/image_extractor.rb +5 -16
  33. data/lib/html2rss/html/article_rules/category.rb +55 -0
  34. data/lib/html2rss/html/article_rules/date.rb +25 -0
  35. data/lib/html2rss/html/article_rules/enclosure.rb +72 -0
  36. data/lib/html2rss/html/article_rules/image.rb +109 -0
  37. data/lib/html2rss/html/article_rules.rb +10 -0
  38. data/lib/html2rss/html/rendering/audio_renderer.rb +2 -12
  39. data/lib/html2rss/html/rendering/escaped_attributes.rb +27 -0
  40. data/lib/html2rss/html/rendering/image_renderer.rb +2 -12
  41. data/lib/html2rss/html/rendering/pdf_renderer.rb +2 -8
  42. data/lib/html2rss/html/rendering/video_renderer.rb +2 -12
  43. data/lib/html2rss/html/sst_article_extractor.rb +279 -0
  44. data/lib/html2rss/link_destination/destination_facts.rb +40 -0
  45. data/lib/html2rss/link_destination/noise_policy.rb +79 -0
  46. data/lib/html2rss/link_destination/path_classifier.rb +205 -0
  47. data/lib/html2rss/link_destination/text_classifier.rb +64 -0
  48. data/lib/html2rss/link_destination.rb +8 -0
  49. data/lib/html2rss/request_service/faraday_strategy.rb +42 -1
  50. data/lib/html2rss/request_service/network_guard.rb +5 -3
  51. data/lib/html2rss/scoring/anchor_score.rb +34 -0
  52. data/lib/html2rss/scoring/cluster_scorer.rb +87 -0
  53. data/lib/html2rss/scoring/container_assessor.rb +83 -0
  54. data/lib/html2rss/scoring/engine.rb +101 -0
  55. data/lib/html2rss/scoring/link_resolver.rb +72 -0
  56. data/lib/html2rss/scoring/observation.rb +53 -0
  57. data/lib/html2rss/scoring/ranked_segment.rb +45 -0
  58. data/lib/html2rss/scoring/score.rb +30 -0
  59. data/lib/html2rss/scoring.rb +33 -0
  60. data/lib/html2rss/selectors/post_processors/sanitize_html.rb +10 -1
  61. data/lib/html2rss/selectors.rb +0 -20
  62. data/lib/html2rss/sst/attrs.rb +91 -0
  63. data/lib/html2rss/sst/document.rb +23 -0
  64. data/lib/html2rss/sst/index.rb +112 -0
  65. data/lib/html2rss/sst/node.rb +147 -0
  66. data/lib/html2rss/sst/normalizer.rb +171 -0
  67. data/lib/html2rss/sst/tags.rb +26 -0
  68. data/lib/html2rss/sst/text.rb +81 -0
  69. data/lib/html2rss/sst.rb +8 -0
  70. data/lib/html2rss/version.rb +1 -1
  71. data/lib/html2rss.rb +26 -26
  72. data/schema/html2rss-config.schema.json +20 -0
  73. metadata +42 -18
  74. data/lib/html2rss/auto_source/discovery/dom_clustering/group_scorer.rb +0 -80
  75. data/lib/html2rss/auto_source/discovery/dom_clustering/overlap_resolver.rb +0 -86
  76. data/lib/html2rss/auto_source/discovery/dom_clustering.rb +0 -119
  77. data/lib/html2rss/auto_source/discovery/list_candidates.rb +0 -94
  78. data/lib/html2rss/auto_source/discovery/semantic_anchor_candidates.rb +0 -219
  79. data/lib/html2rss/auto_source/discovery/semantic_containers.rb +0 -71
  80. data/lib/html2rss/auto_source/discovery/sitemap.rb +0 -159
  81. data/lib/html2rss/auto_source/discovery.rb +0 -14
  82. data/lib/html2rss/auto_source/link_heuristics/anchor_signals.rb +0 -28
  83. data/lib/html2rss/auto_source/link_heuristics/container_assessor.rb +0 -106
  84. data/lib/html2rss/auto_source/link_heuristics/container_signals.rb +0 -80
  85. data/lib/html2rss/auto_source/link_heuristics/destination_facts.rb +0 -42
  86. data/lib/html2rss/auto_source/link_heuristics/href_extractor.rb +0 -38
  87. data/lib/html2rss/auto_source/link_heuristics/path_classifier.rb +0 -221
  88. data/lib/html2rss/auto_source/link_heuristics/text_classifier.rb +0 -66
  89. data/lib/html2rss/auto_source/link_heuristics.rb +0 -139
@@ -1,66 +0,0 @@
1
- # frozen_string_literal: true
2
-
3
- module Html2rss
4
- class AutoSource
5
- class LinkHeuristics
6
- # Classifies visible anchor text for utility and recommendation chrome.
7
- class TextClassifier
8
- # Prefix labels that usually identify navigation or subscription links.
9
- UTILITY_PREFIX_PATTERN = /
10
- \A\s*(
11
- # English
12
- view\s+all|see\s+all|all\s+news|subscribe|newsletter|comment\s+feed|comments\s+feed|join|premium|plus|
13
- # German
14
- alle\s+anzeigen|alle\s+news|abonnieren|newsletter|kommentar\s+feed|mitmachen|
15
- # Spanish
16
- ver\s+todos|ver\s+todo|todas\s+las\s+noticias|suscribirse|bolet(i|í)n|comentarios\s+feed|unirse|
17
- # French
18
- voir\s+tout|voir\s+tous|toutes\s+les\s+nouvelles|s['’]abonner|flux\s+de\s+commentaires|rejoindre
19
- )\b
20
- /ix
21
- # Short labels that usually identify non-article navigation links.
22
- UTILITY_PATTERN = /
23
- \A\s*(
24
- # English
25
- about|contact|comments?|join|log\s+in|login|member(ship)?|
26
- plus|premium|pricing|recommended(\s+for\s+you)?|
27
- see\s+all|share|sign\s+up|signup|subscribe|view\s+all|
28
- # German
29
- (ue|ü)ber(\s+uns)?|kontakt|kommentare?|mitmachen|anmelden|login|
30
- mitglied(schaft)?|empfohlen(\s+f(ue|ü)r\s+dich)?|alle\s+anzeigen|
31
- teilen|registrieren|abonnieren|newsletter|
32
- # Spanish
33
- sobre(\s+nosotros)?|contacto|comentarios?|unirse|iniciar\s+sesion|
34
- login|miembro|membres(i|í)a|recomendado(\s+para\s+ti)?|ver\s+todo|
35
- compartir|registrarse|suscribirse|bolet(i|í)n|
36
- # French
37
- (a|à)\s+propos|(a|à)propos|contact|commentaires?|rejoindre|
38
- se\s+connecter|login|membre|abonnement|recommand(e|é)(\s+pour\s+vous)?|
39
- voir\s+tout|partager|s['’]inscrire|s['’]abonner|newsletter
40
- )\b
41
- /ix
42
- # Labels for recommendation chrome rather than source articles.
43
- RECOMMENDED_PATTERN = /
44
- \A\s*(
45
- recommended(\s+for\s+you)?|
46
- empfohlen(\s+f(ue|ü)r\s+dich)?|
47
- recomendado(\s+para\s+ti)?|
48
- recommand(e|é)(\s+pour\s+vous)?
49
- )\b
50
- /ix
51
-
52
- # @param text [String, #to_s] visible anchor text
53
- # @return [Boolean] true when text matches a utility label
54
- def utility?(text) = text.to_s.match?(UTILITY_PATTERN)
55
-
56
- # @param text [String, #to_s] visible anchor text
57
- # @return [Boolean] true when text begins with a utility label
58
- def utility_prefix?(text) = text.to_s.match?(UTILITY_PREFIX_PATTERN)
59
-
60
- # @param text [String, #to_s] visible anchor text
61
- # @return [Boolean] true when text identifies recommendation chrome
62
- def recommended?(text) = text.to_s.match?(RECOMMENDED_PATTERN)
63
- end
64
- end
65
- end
66
- end
@@ -1,139 +0,0 @@
1
- # frozen_string_literal: true
2
-
3
- module Html2rss
4
- class AutoSource
5
- ##
6
- # Shared link eligibility and scoring policy for AutoSource scrapers.
7
- #
8
- # Scrapers collect DOM observations; this module owns junk/noise rules and
9
- # numeric weights so eligibility policy stays in one place.
10
- class LinkHeuristics
11
- # @param base_url [String, Html2rss::Url] page URL used to resolve relative hrefs
12
- def initialize(base_url)
13
- @base_url = base_url
14
- @text_classifier = TextClassifier.new
15
- @container_assessor = ContainerAssessor.new(text_classifier: @text_classifier)
16
- end
17
-
18
- # Builds normalized destination facts for an anchor element or href string.
19
- #
20
- # @param anchor_or_href [Nokogiri::XML::Element, String, #to_s] anchor element or href-like value
21
- # @return [DestinationFacts, nil] normalized destination facts, or nil for blank/invalid URLs
22
- def destination_facts(anchor_or_href)
23
- return node_facts[anchor_or_href] if node_facts.key?(anchor_or_href)
24
-
25
- href = HrefExtractor.call(anchor_or_href)
26
- return unless href
27
-
28
- res = memoized_destination_facts(href)
29
-
30
- node_facts[anchor_or_href] = res if anchor_or_href.is_a?(Nokogiri::XML::Node)
31
- res
32
- rescue ArgumentError
33
- nil
34
- end
35
-
36
- # @param text [String, #to_s] visible anchor text
37
- # @return [Boolean] true when text matches a utility label
38
- def utility_text?(text) = @text_classifier.utility?(text)
39
-
40
- # @param text [String, #to_s] visible anchor text
41
- # @return [Boolean] true when text begins with a utility label
42
- def utility_prefix_text?(text) = @text_classifier.utility_prefix?(text)
43
-
44
- # @param text [String, #to_s] visible anchor text
45
- # @return [Boolean] true when text identifies recommendation chrome
46
- def recommended_text?(text) = @text_classifier.recommended?(text)
47
-
48
- ##
49
- # Whether an anchor is junk chrome rather than a content permalink.
50
- #
51
- # One eligibility home for Html and SemanticHtml: taxonomy/utility text
52
- # rules plus optional DOM checks (icon-only, utility landmarks).
53
- #
54
- # @param text [String, #to_s] visible anchor text
55
- # @param destination_facts [DestinationFacts, nil] route facts for the href
56
- # @param anchor [Nokogiri::XML::Node, nil] anchor node for icon/landmark checks
57
- # @param container [Nokogiri::XML::Node, nil] content container bounding landmark walks
58
- # @param heading_anchor [Boolean] whether the anchor is the container heading link
59
- # @return [Boolean] true when the anchor should be ignored
60
- def noise_anchor?(text:, destination_facts:, anchor: nil, container: nil, heading_anchor: false) # rubocop:disable Metrics/CyclomaticComplexity, Metrics/PerceivedComplexity
61
- return true unless destination_facts
62
-
63
- destination_facts.taxonomy_path ||
64
- short_utility_label?(text, destination_facts) ||
65
- recommended_chrome?(text, destination_facts, heading_anchor:) ||
66
- (utility_prefix_text?(text) && destination_facts.high_confidence_utility_destination) ||
67
- (utility_text?(text) && destination_facts.vanity_path) ||
68
- utility_text_chrome?(text, destination_facts, heading_anchor:) ||
69
- icon_only_anchor?(anchor, text) ||
70
- utility_landmark_ancestor?(anchor, container)
71
- end
72
-
73
- ##
74
- # Observes a container and builds ranking signals, including hard-junk.
75
- #
76
- # Delegates DOM observation to ContainerAssessor so SemanticHtml only
77
- # orchestrates candidates and extraction, while ContainerSignals keeps
78
- # scoring policy.
79
- #
80
- # @param container [Nokogiri::XML::Node] semantic container node
81
- # @param selected_anchor [Nokogiri::XML::Node, nil] primary anchor for the container
82
- # @param destination_facts [DestinationFacts, nil] route facts for the selected anchor
83
- # @return [ContainerSignals] observation + scoring signals for the container
84
- def assess_container(container, selected_anchor, destination_facts:)
85
- @container_assessor.call(container, selected_anchor, destination_facts:)
86
- end
87
-
88
- private
89
-
90
- def short_utility_label?(text, destination_facts)
91
- destination_facts.utility_path &&
92
- !destination_facts.content_path &&
93
- !destination_facts.strong_post_suffix &&
94
- text.to_s.scan(/\p{Alnum}+/).size <= 3
95
- end
96
-
97
- def recommended_chrome?(text, destination_facts, heading_anchor:)
98
- # Heading-linked titles may start with "Recommended …" while still being
99
- # real posts; container hard_junk? owns that call with publish markers.
100
- !heading_anchor && recommended_text?(text) && destination_facts.shallow
101
- end
102
-
103
- def utility_text_chrome?(text, destination_facts, heading_anchor:)
104
- return false if destination_facts.content_path
105
- return false unless utility_text?(text)
106
-
107
- !heading_anchor && !destination_facts.strong_post_suffix
108
- end
109
-
110
- def icon_only_anchor?(anchor, text)
111
- return false unless anchor
112
-
113
- !text.to_s.match?(/\p{Alnum}/) && !anchor.at_css('img, svg').nil?
114
- end
115
-
116
- def utility_landmark_ancestor?(anchor, container)
117
- return false unless anchor && container
118
-
119
- condition = lambda { |node|
120
- node == container || Html2rss::Html::Navigator::UTILITY_LANDMARK_TAGS.include?(node.name)
121
- }
122
- landmark = Html2rss::Html::Navigator.parent_until_condition(anchor.parent, condition)
123
-
124
- !landmark.nil? && landmark != container
125
- end
126
-
127
- def node_facts
128
- @node_facts ||= {}.compare_by_identity
129
- end
130
-
131
- def memoized_destination_facts(href)
132
- (@destination_facts ||= {})[href] ||= begin
133
- url = Html2rss::Url.from_relative(href, @base_url)
134
- DestinationFacts.build(url)
135
- end
136
- end
137
- end
138
- end
139
- end