gento 0.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
data/lib/gento/html.rb ADDED
@@ -0,0 +1,225 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "strscan"
4
+ require "cgi"
5
+ require "json"
6
+
7
+ module Gento
8
+ # A deliberately small HTML scanner.
9
+ #
10
+ # This is not a DOM parser and does not try to be one. It finds tags by name,
11
+ # reads their attributes, and grabs the raw text of non-void elements, which
12
+ # covers everything the adapters need (meta, link, script, img, title).
13
+ #
14
+ # Why not Nokogiri: it ships a native extension. Staying on pure Ruby keeps
15
+ # the gem usable from slim containers and from ruby.wasm, where compiling
16
+ # libxml is anywhere from painful to impossible.
17
+ class Html
18
+ VOID_ELEMENTS = %w[
19
+ area base br col embed hr img input link meta param source track wbr
20
+ ].freeze
21
+
22
+ attr_reader :source
23
+
24
+ def initialize(source)
25
+ # Defensive: adapters are the only caller in practice and Response has
26
+ # already normalized, but a caller handing us a raw body should not blow
27
+ # up mid-scan with an encoding error.
28
+ @source = Charset.normalize(source)
29
+ @tags = {}
30
+ @elements = {}
31
+ end
32
+
33
+ # Attribute hashes for every occurrence of `name`, in document order.
34
+ #
35
+ # Unlike #elements this never consumes element content, so it still finds
36
+ # tags nested inside another tag of the same name — which ordinary page
37
+ # markup is full of.
38
+ def tags(name)
39
+ @tags[name.to_s.downcase] ||= scan_tags(name.to_s)
40
+ end
41
+
42
+ # [attributes, inner_html] pairs for every occurrence of `name`.
43
+ #
44
+ # Only useful for elements that do not nest, such as <script> and <title>;
45
+ # for anything else use #tags.
46
+ def elements(name)
47
+ @elements[name.to_s.downcase] ||= scan_elements(name.to_s)
48
+ end
49
+
50
+ # Content of the first <meta> whose name/property/itemprop matches any key.
51
+ def meta(*keys)
52
+ meta_all(*keys).first
53
+ end
54
+
55
+ # Content of every matching <meta>, for properties that legitimately repeat.
56
+ def meta_all(*keys)
57
+ wanted = keys.flatten.map { |key| key.to_s.downcase }
58
+ tags("meta").filter_map { |attrs| meta_content(attrs, wanted) }
59
+ end
60
+
61
+ # href of the first <link> carrying the given rel token.
62
+ def link(rel)
63
+ wanted = rel.to_s.downcase
64
+ tags("link").each do |attrs|
65
+ tokens = attrs["rel"].to_s.downcase.split(/\s+/)
66
+ return attrs["href"] if tokens.include?(wanted)
67
+ end
68
+ nil
69
+ end
70
+
71
+ def title
72
+ inner = elements("title").first&.last
73
+ inner && decode_entities(inner.strip)
74
+ end
75
+
76
+ # Every parsed application/ld+json block, with @graph arrays flattened.
77
+ def json_ld
78
+ documents = elements("script").filter_map do |attrs, inner|
79
+ next unless attrs["type"].to_s.downcase.include?("ld+json")
80
+
81
+ parse_json(inner)
82
+ end
83
+ documents.flat_map { |doc| flatten_json_ld(doc) }
84
+ end
85
+
86
+ # Raw bodies of inline <script> tags, for sites that stash a JSON blob there.
87
+ def inline_scripts
88
+ elements("script").filter_map do |attrs, inner|
89
+ next if attrs.key?("src")
90
+
91
+ inner
92
+ end
93
+ end
94
+
95
+ # Any absolute URL appearing anywhere in the document, in document order,
96
+ # entity-decoded and deduplicated.
97
+ #
98
+ # Slide hosts put page images wherever suits them: Speaker Deck uses
99
+ # <a href>, Docswell a lazy-loading data attribute, Google Slides a CSS
100
+ # background inside a style attribute. Matching the URL shape across the
101
+ # raw document handles all three, and keeps working when a site moves its
102
+ # images from one element to another.
103
+ ABSOLUTE_URL = %r{https?://[^\s"'<>\\)]+}
104
+
105
+ def urls(pattern = //)
106
+ source.scan(ABSOLUTE_URL)
107
+ .map { |url| decode_entities(url) }
108
+ .grep(pattern)
109
+ .uniq
110
+ end
111
+
112
+ def decode_entities(text)
113
+ Entities.decode(text)
114
+ end
115
+
116
+ private
117
+
118
+ # A <meta> carries its key under any of three attributes depending on the
119
+ # vocabulary in use; an empty content is treated as absent.
120
+ def meta_content(attrs, wanted)
121
+ key = attrs["property"] || attrs["name"] || attrs["itemprop"]
122
+ return nil unless key && wanted.include?(key.downcase)
123
+
124
+ content = attrs["content"]
125
+ content unless content.nil? || content.empty?
126
+ end
127
+
128
+ def scan_tags(name)
129
+ results = []
130
+ scanner = StringScanner.new(@source)
131
+ opening = opening_pattern(name)
132
+
133
+ while scanner.skip_until(opening)
134
+ attributes, = scan_attributes(scanner)
135
+ results << attributes
136
+ end
137
+
138
+ results
139
+ end
140
+
141
+ def scan_elements(name)
142
+ results = []
143
+ scanner = StringScanner.new(@source)
144
+ opening = opening_pattern(name)
145
+ void = VOID_ELEMENTS.include?(name.downcase)
146
+
147
+ while scanner.skip_until(opening)
148
+ attributes, self_closing = scan_attributes(scanner)
149
+ inner = void || self_closing ? nil : scan_inner(scanner, name)
150
+ results << [attributes, inner]
151
+ end
152
+
153
+ results
154
+ end
155
+
156
+ def opening_pattern(name)
157
+ %r{<#{Regexp.escape(name)}(?=[\s>/])}i
158
+ end
159
+
160
+ # Consumes the attribute list and the closing `>` of an open tag.
161
+ def scan_attributes(scanner)
162
+ attributes = {}
163
+
164
+ loop do
165
+ scanner.skip(/\s+/)
166
+ return [attributes, true] if scanner.scan(%r{/\s*>})
167
+ return [attributes, false] if scanner.scan(">")
168
+ return [attributes, false] if scanner.eos?
169
+
170
+ name = scanner.scan(%r{[^\s=/>]+})
171
+ # Never stall: an unexpected byte here would otherwise loop forever.
172
+ next scanner.getch if name.nil?
173
+
174
+ scanner.skip(/\s*/)
175
+ attributes[name.downcase] = decode_entities(scan_attribute_value(scanner))
176
+ end
177
+ end
178
+
179
+ def scan_attribute_value(scanner)
180
+ return "" unless scanner.scan(/=\s*/)
181
+
182
+ if scanner.scan(/"([^"]*)"/) || scanner.scan(/'([^']*)'/)
183
+ scanner[1]
184
+ else
185
+ scanner.scan(/[^\s>]*/).to_s
186
+ end
187
+ end
188
+
189
+ # Returns the raw text up to the matching close tag.
190
+ #
191
+ # StringScanner#pos counts bytes while String#[] counts characters, so
192
+ # slicing has to go through byteslice or every multibyte document comes
193
+ # back mangled. Tag boundaries are ASCII, so byte slicing is safe here.
194
+ def scan_inner(scanner, name)
195
+ start = scanner.pos
196
+ closing = %r{</#{Regexp.escape(name)}\s*>}i
197
+
198
+ unless scanner.skip_until(closing)
199
+ scanner.terminate
200
+ return byteslice_from(start, @source.bytesize - start)
201
+ end
202
+
203
+ byteslice_from(start, scanner.pos - scanner.matched_size - start)
204
+ end
205
+
206
+ def byteslice_from(start, length)
207
+ slice = @source.byteslice(start, length).to_s
208
+ slice.valid_encoding? ? slice : slice.scrub
209
+ end
210
+
211
+ def parse_json(text)
212
+ JSON.parse(text.to_s)
213
+ rescue JSON::ParserError
214
+ nil
215
+ end
216
+
217
+ def flatten_json_ld(doc)
218
+ case doc
219
+ when Array then doc.flat_map { |entry| flatten_json_ld(entry) }
220
+ when Hash then doc.key?("@graph") ? flatten_json_ld(doc["@graph"]) + [doc] : [doc]
221
+ else []
222
+ end
223
+ end
224
+ end
225
+ end
@@ -0,0 +1,49 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Gento
4
+ # Maps a URL to the adapter that claims it.
5
+ #
6
+ # Adding support for a new slide host is one class plus one `register` call;
7
+ # nothing else in the library needs to know the site exists.
8
+ class Registry
9
+ class << self
10
+ def default
11
+ @default ||= new(
12
+ [
13
+ Adapters::SpeakerDeck,
14
+ Adapters::SlideShare,
15
+ Adapters::Docswell,
16
+ Adapters::GoogleSlides
17
+ ]
18
+ )
19
+ end
20
+
21
+ attr_writer :default
22
+ end
23
+
24
+ def initialize(adapters = [])
25
+ @adapters = adapters.dup
26
+ end
27
+
28
+ def adapters
29
+ @adapters.dup.freeze
30
+ end
31
+
32
+ def register(adapter)
33
+ @adapters.unshift(adapter)
34
+ self
35
+ end
36
+
37
+ def find(url)
38
+ @adapters.find { |adapter| adapter.handles?(url) }
39
+ end
40
+
41
+ def supports?(url)
42
+ !find(url).nil?
43
+ end
44
+
45
+ def providers
46
+ @adapters.map(&:provider)
47
+ end
48
+ end
49
+ end
@@ -0,0 +1,34 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Gento
4
+ # A minimal HTTP response, so adapters never touch a fetcher-specific object.
5
+ class Response
6
+ attr_reader :status, :headers, :body, :url
7
+
8
+ def initialize(status:, body:, url:, headers: {})
9
+ @status = Integer(status)
10
+ @url = url.to_s
11
+ @headers = headers.transform_keys { |key| key.to_s.downcase }.freeze
12
+ @body = Charset.normalize(body, content_type: @headers["content-type"])
13
+ freeze
14
+ end
15
+
16
+ def success?
17
+ status.between?(200, 299)
18
+ end
19
+
20
+ def content_type
21
+ headers["content-type"]
22
+ end
23
+
24
+ def json
25
+ JSON.parse(body)
26
+ rescue JSON::ParserError => e
27
+ raise ExtractionError, "expected JSON from #{url}: #{e.message}"
28
+ end
29
+
30
+ def html
31
+ Html.new(body)
32
+ end
33
+ end
34
+ end
@@ -0,0 +1,195 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Gento
4
+ # A parsed robots.txt, following RFC 9309.
5
+ #
6
+ # The one rule worth implementing carefully is precedence: **the longest
7
+ # matching pattern wins**, not the first one. Google's robots.txt ends with
8
+ #
9
+ # Allow: /presentation
10
+ # Disallow: /
11
+ #
12
+ # and a first-match reading would conclude that every Google Slides deck is
13
+ # off limits, which is the opposite of what that file says.
14
+ class Robots
15
+ # A single Allow or Disallow line.
16
+ class Rule
17
+ attr_reader :pattern
18
+
19
+ def initialize(pattern, allow:)
20
+ @pattern = pattern
21
+ @allow = allow
22
+ @regexp = self.class.compile(pattern)
23
+ freeze
24
+ end
25
+
26
+ def allow?
27
+ @allow
28
+ end
29
+
30
+ def covers?(path)
31
+ @regexp.match?(path)
32
+ end
33
+
34
+ # RFC 9309 orders rules by the length of the pattern, wildcards included.
35
+ def specificity
36
+ pattern.length
37
+ end
38
+
39
+ # `*` stands for any run of characters and `$` anchors the end; every
40
+ # other character is literal.
41
+ def self.compile(pattern)
42
+ anchored = pattern.end_with?("$")
43
+ body = anchored ? pattern[0..-2] : pattern
44
+ source = body.split("*", -1).map { |part| Regexp.escape(part) }.join(".*")
45
+ source += "\\z" if anchored
46
+
47
+ Regexp.new("\\A#{source}")
48
+ end
49
+ end
50
+
51
+ Group = Struct.new(:agents, :rules, :crawl_delay)
52
+
53
+ DIRECTIVES = %w[user-agent allow disallow crawl-delay].freeze
54
+ # A product token runs to the first slash or space: the "gento" of
55
+ # "gento/0.1.0 (+https://...)".
56
+ PRODUCT_TOKEN = %r{\A[^\s/]+}
57
+
58
+ attr_reader :groups, :reason
59
+
60
+ def initialize(groups = [], reason: nil)
61
+ @groups = groups.freeze
62
+ @reason = reason
63
+ freeze
64
+ end
65
+
66
+ class << self
67
+ def parse(body)
68
+ new(Parser.new(body).groups)
69
+ end
70
+
71
+ # No robots.txt to obey — the host said so with a 404, or answered with
72
+ # something that is not a robots.txt at all.
73
+ def allow_all
74
+ new
75
+ end
76
+
77
+ # RFC 9309 §2.3.1.4: a robots.txt that cannot be read is not permission
78
+ # to crawl. `reason` explains which failure it was.
79
+ def disallow_all(reason: nil)
80
+ new([Group.new(["*"], [Rule.new("/", allow: false)], nil)], reason: reason)
81
+ end
82
+
83
+ def product_token(user_agent)
84
+ user_agent.to_s[PRODUCT_TOKEN].to_s.downcase
85
+ end
86
+ end
87
+
88
+ def allowed?(path, agent: nil)
89
+ group = group_for(agent)
90
+ return true unless group
91
+
92
+ matching = group.rules.select { |rule| rule.covers?(path) }
93
+ return true if matching.empty?
94
+
95
+ # Longest pattern wins; an Allow and a Disallow of equal length are a
96
+ # tie the site did not resolve, and RFC 9309 gives those to Allow.
97
+ matching.max_by { |rule| [rule.specificity, rule.allow? ? 1 : 0] }.allow?
98
+ end
99
+
100
+ def crawl_delay(agent: nil)
101
+ group_for(agent)&.crawl_delay
102
+ end
103
+
104
+ def empty?
105
+ groups.empty?
106
+ end
107
+
108
+ private
109
+
110
+ # A crawler obeys exactly one group: the one naming it most specifically,
111
+ # falling back to `*`.
112
+ def group_for(agent)
113
+ token = self.class.product_token(agent)
114
+
115
+ groups.filter_map { |group| [group, specificity(group, token)] if applies?(group, token) }
116
+ .max_by(&:last)
117
+ &.first
118
+ end
119
+
120
+ def applies?(group, token)
121
+ group.agents.any? { |name| name == "*" || (!token.empty? && token.start_with?(name)) }
122
+ end
123
+
124
+ def specificity(group, token)
125
+ group.agents.filter_map { |name| name.length if name != "*" && token.start_with?(name) }
126
+ .max || 0
127
+ end
128
+
129
+ # robots.txt is a line-oriented `key: value` format. Anything unrecognised
130
+ # is skipped rather than treated as an error, which is also what keeps a
131
+ # page served in place of a robots.txt from parsing into rules.
132
+ class Parser
133
+ attr_reader :groups
134
+
135
+ def initialize(body)
136
+ @groups = []
137
+ @current = nil
138
+ @reading_agents = false
139
+
140
+ each_directive(body) { |key, value| apply(key, value) }
141
+ end
142
+
143
+ private
144
+
145
+ def apply(key, value)
146
+ case key
147
+ when "user-agent" then add_agent(value)
148
+ when "allow", "disallow" then add_rule(value, allow: key == "allow")
149
+ when "crawl-delay" then record_crawl_delay(value)
150
+ end
151
+ end
152
+
153
+ # Consecutive User-agent lines share one group; a rule line ends the
154
+ # list, so the next User-agent starts a new group.
155
+ def add_agent(value)
156
+ unless @reading_agents && @current
157
+ @current = Group.new([], [], nil)
158
+ @groups << @current
159
+ end
160
+
161
+ @current.agents << value.downcase
162
+ @reading_agents = true
163
+ end
164
+
165
+ def add_rule(value, allow:)
166
+ @reading_agents = false
167
+ # "Disallow:" with nothing after it lifts the restriction rather than
168
+ # imposing one, so there is no rule to record.
169
+ return if @current.nil? || value.empty?
170
+
171
+ @current.rules << Rule.new(value, allow: allow)
172
+ end
173
+
174
+ def record_crawl_delay(value)
175
+ @reading_agents = false
176
+ return if @current.nil?
177
+
178
+ @current.crawl_delay = Float(value)
179
+ rescue ArgumentError, TypeError
180
+ nil
181
+ end
182
+
183
+ def each_directive(body)
184
+ body.to_s.each_line do |line|
185
+ line = line.sub(/#.*/, "").strip
186
+ key, separator, value = line.partition(":")
187
+ next if separator.empty?
188
+
189
+ key = key.strip.downcase
190
+ yield key, value.strip if DIRECTIVES.include?(key)
191
+ end
192
+ end
193
+ end
194
+ end
195
+ end
@@ -0,0 +1,126 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "uri"
4
+
5
+ module Gento
6
+ # A Fetcher that consults robots.txt before letting a request through.
7
+ #
8
+ # It wraps another fetcher rather than living inside one, so it applies to
9
+ # every request an adapter makes — the deck page, the oEmbed endpoint, the
10
+ # embed view — and works just as well over an injected fetcher as over the
11
+ # net/http one.
12
+ #
13
+ # On by default. `Gento.fetch(url, robots: false)` turns it off; the
14
+ # caller then owns whatever that implies.
15
+ #
16
+ # One gap: the wrapped fetcher follows redirects itself, so a redirect into
17
+ # a disallowed path is not caught. Checking that needs a per-hop callback in
18
+ # the fetcher, which is not worth the interface it would add.
19
+ class RobotsFetcher < Fetcher
20
+ ROBOTS_PATH = "/robots.txt"
21
+ # Speaker Deck serves its robots.txt through the site's HTML layout unless
22
+ # this header asks for plain text, so it is load-bearing rather than
23
+ # politeness. The body is parsed whatever content type comes back: a page
24
+ # that is not a robots.txt yields no directives anyway, and refusing on
25
+ # content type alone would throw away real rules from a host that answers
26
+ # in HTML.
27
+ PLAIN_TEXT = { "accept" => "text/plain" }.freeze
28
+ DEFAULT_PORTS = { "http" => 80, "https" => 443 }.freeze
29
+
30
+ attr_reader :fetcher, :user_agent
31
+
32
+ def initialize(fetcher, user_agent: nil)
33
+ @fetcher = fetcher
34
+ @user_agent = user_agent || inherited_user_agent(fetcher)
35
+ @robots = {}
36
+ super()
37
+ end
38
+
39
+ def get(url, headers: {})
40
+ uri = coerce_uri(url)
41
+ guard!(uri) if uri
42
+
43
+ fetcher.get(url, headers: headers)
44
+ end
45
+
46
+ # The Crawl-delay the host asks for, in seconds, or nil.
47
+ #
48
+ # Not enforced here: one fetch is a handful of requests, and sleeping
49
+ # inside a fetcher would surprise a caller who already paces its own work.
50
+ # Exposed so a caller doing more than one deck can honour it.
51
+ def crawl_delay(url)
52
+ uri = coerce_uri(url)
53
+ return nil unless uri
54
+
55
+ robots_for(uri).crawl_delay(agent: user_agent)
56
+ end
57
+
58
+ def robots_for(uri)
59
+ @robots[origin(uri)] ||= load_robots(origin(uri))
60
+ end
61
+
62
+ private
63
+
64
+ def guard!(uri)
65
+ robots = robots_for(uri)
66
+ path = request_path(uri)
67
+ return if robots.allowed?(path, agent: user_agent)
68
+
69
+ raise RobotsDisallowedError.new(refusal(uri, path, robots), url: uri.to_s)
70
+ end
71
+
72
+ def refusal(uri, path, robots)
73
+ cause = robots.reason || "#{origin(uri)}#{ROBOTS_PATH} disallows #{path}"
74
+
75
+ "#{cause} for #{user_agent_label}. " \
76
+ "Pass robots: false to fetch it anyway, and take responsibility for doing so."
77
+ end
78
+
79
+ def user_agent_label
80
+ user_agent.to_s.empty? ? "this client" : user_agent.inspect
81
+ end
82
+
83
+ def load_robots(origin)
84
+ Robots.parse(fetcher.get("#{origin}#{ROBOTS_PATH}", headers: PLAIN_TEXT).body)
85
+ rescue ResponseError => e
86
+ absent?(e.status) ? Robots.allow_all : unreachable(origin, "responded #{e.status}")
87
+ rescue FetchError => e
88
+ unreachable(origin, e.message)
89
+ end
90
+
91
+ # 4xx is the host saying there is no robots.txt, which RFC 9309 reads as
92
+ # "help yourself". 429 is it saying to come back later, which is not the
93
+ # same thing.
94
+ def absent?(status)
95
+ status.to_i.between?(400, 499) && status.to_i != 429
96
+ end
97
+
98
+ def unreachable(origin, detail)
99
+ Robots.disallow_all(reason: "could not read #{origin}#{ROBOTS_PATH} (#{detail}), " \
100
+ "so there is no permission to rely on")
101
+ end
102
+
103
+ def origin(uri)
104
+ port = ":#{uri.port}" unless uri.port == DEFAULT_PORTS[uri.scheme]
105
+
106
+ "#{uri.scheme}://#{uri.host}#{port}"
107
+ end
108
+
109
+ # Rules match against the path and query together.
110
+ def request_path(uri)
111
+ path = uri.path.to_s.empty? ? "/" : uri.path
112
+ uri.query ? "#{path}?#{uri.query}" : path
113
+ end
114
+
115
+ def coerce_uri(url)
116
+ uri = url.is_a?(URI) ? url : URI.parse(url.to_s)
117
+ uri.is_a?(URI::HTTP) ? uri : nil
118
+ rescue URI::InvalidURIError
119
+ nil
120
+ end
121
+
122
+ def inherited_user_agent(fetcher)
123
+ fetcher.user_agent if fetcher.respond_to?(:user_agent)
124
+ end
125
+ end
126
+ end
@@ -0,0 +1,40 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Gento
4
+ # A single page of a deck.
5
+ #
6
+ # `url` always points at the original host. We deliberately never re-host or
7
+ # proxy slide images: this library reports where the images are, and leaves
8
+ # serving them to the site that owns them.
9
+ class Slide
10
+ attr_reader :number, :url, :width, :height, :thumbnail_url
11
+
12
+ def initialize(number:, url:, width: nil, height: nil, thumbnail_url: nil)
13
+ @number = Integer(number)
14
+ @url = url.to_s
15
+ @width = width && Integer(width)
16
+ @height = height && Integer(height)
17
+ @thumbnail_url = thumbnail_url&.to_s
18
+ freeze
19
+ end
20
+
21
+ def to_h
22
+ {
23
+ number: number,
24
+ url: url,
25
+ width: width,
26
+ height: height,
27
+ thumbnail_url: thumbnail_url
28
+ }.compact
29
+ end
30
+
31
+ def ==(other)
32
+ other.is_a?(Slide) && other.to_h == to_h
33
+ end
34
+ alias eql? ==
35
+
36
+ def hash
37
+ to_h.hash
38
+ end
39
+ end
40
+ end
@@ -0,0 +1,5 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Gento
4
+ VERSION = "0.1.1"
5
+ end